Push zunrplorkwkt #70

Merged
coissac merged 93 commits from push-zunrplorkwkt into main 2026-08-28 23:15:38 +00:00
61 changed files with 1034 additions and 6797 deletions
Showing only changes of commit 95fa0c93b2 - Show all commits
+5 -343
View File
@@ -2,15 +2,6 @@
# It is not intended for manual editing.
version = 4
[[package]]
name = "addr2line"
version = "0.25.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1b5d307320b3181d6d7954e663bd7c774a838b8220fe0593c86d9fb09f498b4b"
dependencies = [
"gimli",
]
[[package]]
name = "adler2"
version = "2.0.1"
@@ -39,15 +30,6 @@ dependencies = [
"memchr",
]
[[package]]
name = "aligned-vec"
version = "0.6.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dc890384c8602f339876ded803c97ad529f3842aba97f6392b3dba0dd171769b"
dependencies = [
"equator",
]
[[package]]
name = "allocator-api2"
version = "0.2.21"
@@ -143,21 +125,6 @@ dependencies = [
"cc",
]
[[package]]
name = "backtrace"
version = "0.3.76"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bb531853791a215d7c62a30daf0dde835f381ab5de4589cfe7c649d2cbe92bd6"
dependencies = [
"addr2line",
"cfg-if",
"libc",
"miniz_oxide",
"object",
"rustc-demangle",
"windows-link",
]
[[package]]
name = "base64"
version = "0.23.1"
@@ -215,15 +182,6 @@ dependencies = [
"wyz",
]
[[package]]
name = "block-buffer"
version = "0.10.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
dependencies = [
"generic-array",
]
[[package]]
name = "block-buffer"
version = "0.12.1"
@@ -329,7 +287,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d524456ba66e72eb8b115ff89e01e497f8e6d11d78b70b1aa13c0fbd97540a81"
dependencies = [
"cfg-if",
"cpufeatures 0.3.0",
"cpufeatures",
"rand_core 0.10.1",
]
@@ -492,24 +450,6 @@ dependencies = [
"unicode-segmentation",
]
[[package]]
name = "cpp_demangle"
version = "0.4.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f2bb79cb74d735044c972aae58ed0aaa9a837e85b01106a54c39e42e97f62253"
dependencies = [
"cfg-if",
]
[[package]]
name = "cpufeatures"
version = "0.2.17"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280"
dependencies = [
"libc",
]
[[package]]
name = "cpufeatures"
version = "0.3.0"
@@ -620,16 +560,6 @@ version = "0.2.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5"
[[package]]
name = "crypto-common"
version = "0.1.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
dependencies = [
"generic-array",
"typenum",
]
[[package]]
name = "crypto-common"
version = "0.2.2"
@@ -660,15 +590,6 @@ dependencies = [
"memchr",
]
[[package]]
name = "debugid"
version = "0.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bef552e6f588e446098f6ba40d89ac146c8c7b64aade83c051ee00bb5d2bc18d"
dependencies = [
"uuid",
]
[[package]]
name = "derive_more"
version = "2.1.1"
@@ -692,25 +613,15 @@ dependencies = [
"unicode-xid",
]
[[package]]
name = "digest"
version = "0.10.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
dependencies = [
"block-buffer 0.10.4",
"crypto-common 0.1.7",
]
[[package]]
name = "digest"
version = "0.11.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2"
dependencies = [
"block-buffer 0.12.1",
"block-buffer",
"const-oid",
"crypto-common 0.2.2",
"crypto-common",
]
[[package]]
@@ -782,26 +693,6 @@ dependencies = [
"syn 2.0.117",
]
[[package]]
name = "equator"
version = "0.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4711b213838dfee0117e3be6ac926007d7f433d7bbe33595975d4190cb07e6fc"
dependencies = [
"equator-macro",
]
[[package]]
name = "equator-macro"
version = "0.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "44f23cf4b44bfce11a86ace86f8a73ffdec849c9fd00a386a53d278bd9e81fb3"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.117",
]
[[package]]
name = "equivalent"
version = "1.0.2"
@@ -840,18 +731,6 @@ version = "0.1.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582"
[[package]]
name = "findshlibs"
version = "0.10.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "40b9e59cd0f7e0806cca4be089683ecb6434e602038df21fe6bf6711b2f07f64"
dependencies = [
"cc",
"lazy_static",
"libc",
"winapi",
]
[[package]]
name = "fixedbitset"
version = "0.4.2"
@@ -895,16 +774,6 @@ dependencies = [
"byteorder",
]
[[package]]
name = "generic-array"
version = "0.14.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
dependencies = [
"typenum",
"version_check",
]
[[package]]
name = "getrandom"
version = "0.2.17"
@@ -940,12 +809,6 @@ dependencies = [
"rand_core 0.10.1",
]
[[package]]
name = "gimli"
version = "0.32.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e629b9b98ef3dd8afe6ca2bd0f89306cec16d43d907889945bc5d6687f2f13c7"
[[package]]
name = "half"
version = "2.7.1"
@@ -1146,15 +1009,6 @@ dependencies = [
"either",
]
[[package]]
name = "itertools"
version = "0.12.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ba291022dbbd398a455acf126c1e341954079855bc60dfdda641363bd6922569"
dependencies = [
"either",
]
[[package]]
name = "itertools"
version = "0.14.0"
@@ -1197,7 +1051,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9e24a010dd405bd7ed803e5253182815b41bf2e6a80cc3bfc066658e03a198aa"
dependencies = [
"cfg-if",
"cpufeatures 0.3.0",
"cpufeatures",
]
[[package]]
@@ -1382,12 +1236,6 @@ dependencies = [
"windows 0.48.0",
]
[[package]]
name = "multimap"
version = "0.10.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1d87ecb2933e8aeadb3e3a02b828fed80a7528047e68b4f424523a0981a3a084"
[[package]]
name = "nanorand"
version = "0.6.1"
@@ -1633,40 +1481,6 @@ dependencies = [
[[package]]
name = "obikmer"
version = "1.2.2"
dependencies = [
"clap",
"csv",
"indicatif",
"kodama",
"obidebruinj",
"obifastwrite",
"obikalgorithm",
"obikfilter",
"obikindex",
"obikindexer",
"obikphylo",
"obikrope",
"obikseq",
"obikstats",
"obipipeline",
"obiread",
"obiskbuilder",
"obiskio",
"obisys",
"obitaxonomy",
"pprof",
"rayon",
"serde",
"serde_json",
"serde_yaml",
"speedytree",
"tracing",
"tracing-subscriber",
]
[[package]]
name = "obikmer2"
version = "1.2.2"
dependencies = [
"clap",
"csv",
@@ -1933,15 +1747,6 @@ dependencies = [
"objc2-foundation",
]
[[package]]
name = "object"
version = "0.37.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ff76201f031d8863c38aa7f905eca4f53abbfa15f609db4277d44cd8938f33fe"
dependencies = [
"memchr",
]
[[package]]
name = "once_cell"
version = "1.21.4"
@@ -2066,31 +1871,6 @@ dependencies = [
"portable-atomic",
]
[[package]]
name = "pprof"
version = "0.15.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "38a01da47675efa7673b032bf8efd8214f1917d89685e07e395ab125ea42b187"
dependencies = [
"aligned-vec",
"backtrace",
"cfg-if",
"findshlibs",
"libc",
"log",
"nix",
"once_cell",
"prost",
"prost-build",
"prost-derive",
"sha2",
"smallvec",
"spin",
"symbolic-demangle",
"tempfile",
"thiserror 2.0.18",
]
[[package]]
name = "ppv-lite86"
version = "0.2.21"
@@ -2100,16 +1880,6 @@ dependencies = [
"zerocopy",
]
[[package]]
name = "prettyplease"
version = "0.2.37"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b"
dependencies = [
"proc-macro2",
"syn 2.0.117",
]
[[package]]
name = "proc-macro-error-attr2"
version = "2.0.0"
@@ -2161,59 +1931,6 @@ dependencies = [
"unicode-ident",
]
[[package]]
name = "prost"
version = "0.12.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "deb1435c188b76130da55f17a466d252ff7b1418b2ad3e037d127b94e3411f29"
dependencies = [
"bytes",
"prost-derive",
]
[[package]]
name = "prost-build"
version = "0.12.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "22505a5c94da8e3b7c2996394d1c933236c4d743e81a410bcca4e6989fc066a4"
dependencies = [
"bytes",
"heck",
"itertools 0.12.1",
"log",
"multimap",
"once_cell",
"petgraph",
"prettyplease",
"prost",
"prost-types",
"regex",
"syn 2.0.117",
"tempfile",
]
[[package]]
name = "prost-derive"
version = "0.12.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "81bddcdb20abf9501610992b6759a4c888aef7d1a7247ef75e2404275ac24af1"
dependencies = [
"anyhow",
"itertools 0.12.1",
"proc-macro2",
"quote",
"syn 2.0.117",
]
[[package]]
name = "prost-types"
version = "0.12.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9091c90b0a32608e984ff2fa4091273cbdd755d54935c51d520887f4a1dbd5b0"
dependencies = [
"prost",
]
[[package]]
name = "ptr_hash"
version = "1.1.0"
@@ -2447,12 +2164,6 @@ dependencies = [
"windows-sys 0.52.0",
]
[[package]]
name = "rustc-demangle"
version = "0.1.27"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b50b8869d9fc858ce7266cce0194bd74df58b9d0e3f6df3a9fc8eb470d95c09d"
[[package]]
name = "rustc-hash"
version = "2.1.2"
@@ -2616,24 +2327,13 @@ dependencies = [
"unsafe-libyaml",
]
[[package]]
name = "sha2"
version = "0.10.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283"
dependencies = [
"cfg-if",
"cpufeatures 0.2.17",
"digest 0.10.7",
]
[[package]]
name = "sha3"
version = "0.11.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "be176f1a57ce4e3d31c1a166222d9768de5954f811601fb7ca06fc8203905ce1"
dependencies = [
"digest 0.11.3",
"digest",
"keccak",
]
@@ -2683,21 +2383,6 @@ dependencies = [
"rb_tree",
]
[[package]]
name = "spin"
version = "0.10.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3"
dependencies = [
"lock_api",
]
[[package]]
name = "stable_deref_trait"
version = "1.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596"
[[package]]
name = "strsim"
version = "0.11.1"
@@ -2741,29 +2426,6 @@ dependencies = [
"num-traits",
]
[[package]]
name = "symbolic-common"
version = "12.18.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "332615d90111d8eeaf86a84dc9bbe9f65d0d8c5cf11b4caccedc37754eb0dcfd"
dependencies = [
"debugid",
"memmap2",
"stable_deref_trait",
"uuid",
]
[[package]]
name = "symbolic-demangle"
version = "12.18.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "912017718eb4d21930546245af9a3475c9dccf15675a5c215664e76621afc471"
dependencies = [
"cpp_demangle",
"rustc-demangle",
"symbolic-common",
]
[[package]]
name = "syn"
version = "2.0.117"
+1 -1
View File
@@ -1,5 +1,5 @@
[workspace]
resolver = "3"
members = ["obikseq", "obiread", "obiskbuilder", "obifastwrite", "obikmer","obikmer2","obikrope","obipipeline", "obiskio","obidebruinj", "obicompactvec", "obisys", "obikindex", "obikindexer", "obikquery", "obikdump", "obikfilter", "obikselect", "obikrebuild", "obikmerge", "obikstats", "obikidxcache", "obitaxonomy", "obikentropy", "obikphylo", "obikalgorithm"]
members = ["obikseq", "obiread", "obiskbuilder", "obifastwrite", "obikmer","obikrope","obipipeline", "obiskio","obidebruinj", "obicompactvec", "obisys", "obikindex", "obikindexer", "obikquery", "obikdump", "obikfilter", "obikselect", "obikrebuild", "obikmerge", "obikstats", "obikidxcache", "obitaxonomy", "obikentropy", "obikphylo", "obikalgorithm"]
[profile.release]
debug = 1
+14 -17
View File
@@ -10,34 +10,31 @@ path = "src/main.rs"
[dependencies]
obikseq = { path = "../obikseq" }
obiread = { path = "../obiread" }
obiskbuilder = { path = "../obiskbuilder" }
obifastwrite = { path = "../obifastwrite" }
obidebruinj = { path = "../obidebruinj" }
obipipeline = { path = "../obipipeline" }
obikrope = { path = "../obikrope" }
obisys = { path = "../obisys" }
obiskio = { path = "../obiskio" }
obikindex = { path = "../obikindex", default-features = false }
obikindexer = { path = "../obikindexer" }
obikfilter = { path = "../obikfilter" }
obikstats = { path = "../obikstats" }
obikalgorithm = { path = "../obikalgorithm" }
obikmerge = { path = "../obikmerge" }
obikfilter = { path = "../obikfilter" }
obikselect = { path = "../obikselect" }
obikdump = { path = "../obikdump" }
obikrebuild = { path = "../obikrebuild" }
obikstats = { path = "../obikstats" }
obikquery = { path = "../obikquery" }
obikidxcache = { path = "../obikidxcache" }
obikphylo = { path = "../obikphylo" }
obitaxonomy = { path = "../obitaxonomy" }
obikrope = { path = "../obikrope" }
obifastwrite = { path = "../obifastwrite" }
obiskbuilder = { path = "../obiskbuilder" }
clap = { version = "4", features = ["derive"] }
serde = { version = "1", features = ["derive"] }
serde_json = "1"
serde_yaml = "0.9.33"
csv = "1"
kodama = "0.3.0"
speedytree = "0.1"
rayon = "1"
indicatif = "0.18"
ndarray = "0.17"
serde = { version = "1", features = ["derive"] }
serde_yaml = "0.9"
tracing = "0.1.44"
tracing-subscriber = { version = "0.3", features = ["fmt", "env-filter"] }
pprof = { version = "0.15", features = ["prost-codec"], optional = true }
[features]
default = ["numa"]
numa = ["obisys/numa"]
profiling = ["dep:pprof"]
+5 -2
View File
@@ -116,10 +116,13 @@ fn run_annotate(args: &AnnotateArgs) {
std::process::exit(1);
});
let headers = rdr.headers().unwrap_or_else(|e| {
let headers = rdr
.headers()
.unwrap_or_else(|e| {
eprintln!("error reading CSV headers: {e}");
std::process::exit(1);
}).clone();
})
.clone();
let id_col_idx = headers.iter().position(|h| h == args.id_col).unwrap_or_else(|| {
eprintln!("error: id column '{}' not found in CSV", args.id_col);
+11 -7
View File
@@ -1,12 +1,15 @@
use std::io::{self, BufWriter};
use std::path::PathBuf;
use std::sync::Arc;
use clap::Args;
use obikdump::IndexDump;
use obikfilter::KmerFilter;
use obikindex::KmerIndex;
use obisys::progress_bar;
use tracing::info;
use super::predicate::FilterArgs;
use super::predicate::GroupFilterArgs;
#[derive(Args)]
pub struct DumpArgs {
@@ -26,32 +29,33 @@ pub struct DumpArgs {
pub head: Option<usize>,
#[command(flatten)]
pub filter: FilterArgs,
pub group_filter: GroupFilterArgs,
}
pub fn run(args: DumpArgs) {
let idx = KmerIndex::open(&args.index).unwrap_or_else(|e| {
let idx = Arc::new(KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
});
}));
let n_genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
}).len();
info!(
"dumping {} partitions, {} genome(s)",
"dumping {} partition(s), {} genome(s)",
idx.n_partitions(),
n_genomes
);
let filters = args.filter.build_filters(&idx.meta());
let filters: Vec<Box<dyn KmerFilter>> = vec![Box::new(args.group_filter.build_filter(&idx.meta()))];
let pb = progress_bar("dump", idx.n_partitions() as u64, "partitions");
let stdout = io::stdout();
let mut out = BufWriter::new(stdout.lock());
idx.dump(&mut out, args.force_presence, args.debug, args.head, &filters, || pb.inc(1)).unwrap_or_else(|e| {
idx.dump(&mut out, args.force_presence, args.debug, args.head, &filters, || pb.inc(1))
.unwrap_or_else(|e| {
eprintln!("dump error: {e}");
std::process::exit(1);
});
+39 -40
View File
@@ -1,15 +1,17 @@
use std::path::PathBuf;
use std::sync::Arc;
use clap::Args;
use obikindex::{KmerIndex, MergeMode};
use obikindex::filter::{MaxTotalCount, MinComplexity, MinTotalCount};
use obisys::Reporter;
use obikalgorithm::Algorithm;
use obikfilter::{Filter, KmerFilter, MaxTotalCount, MinComplexity, MinTotalCount};
use obikindex::KmerIndex;
use obisys::{Progress, Reporter, Stage, progress_bar};
use tracing::info;
use super::predicate::FilterArgs as KmerFilterArgs;
use super::predicate::GroupFilterArgs;
#[derive(Args)]
pub struct FilterCmdArgs {
pub struct FilterArgs {
/// Source index directory
pub source: PathBuf,
@@ -18,7 +20,7 @@ pub struct FilterCmdArgs {
pub output: PathBuf,
#[command(flatten)]
pub filter: KmerFilterArgs,
pub group_filter: GroupFilterArgs,
/// Minimum total count across all genomes (count index only)
#[arg(long)]
@@ -28,15 +30,11 @@ pub struct FilterCmdArgs {
#[arg(long)]
pub max_total_count: Option<u32>,
/// Minimum normalized entropy (complexity) to keep a k-mer — same metric
/// as `obikmer index`'s --theta, applied here to k-mers already committed
/// to the source index (reconstructed from unitigs.bin). K-mers scoring
/// below this are removed.
/// Minimum normalized entropy (complexity) to keep a k-mer
#[arg(long)]
pub min_complexity: Option<f64>,
/// Maximum sub-word size for the complexity computation (see `obikmer
/// index`'s --level-max). Only used when --min-complexity is set.
/// Maximum sub-word size for the complexity computation (only used when --min-complexity is set)
#[arg(long, default_value_t = 6)]
pub complexity_level_max: usize,
@@ -44,34 +42,23 @@ pub struct FilterCmdArgs {
#[arg(long)]
pub presence: bool,
/// Pack the output's presence matrices in the dense format instead of the default sparse one
#[arg(long, default_value_t = false)]
pub dense: bool,
/// Overwrite existing output directory
#[arg(short, long)]
pub force: bool,
}
pub fn run(args: FilterCmdArgs) {
let src = KmerIndex::open(&args.source).unwrap_or_else(|e| {
pub fn run(args: FilterArgs) {
let src = Arc::new(KmerIndex::open(&args.source).unwrap_or_else(|e| {
eprintln!("error opening source index: {e}");
std::process::exit(1);
});
let mode = if args.presence || !src.meta().config.with_counts {
MergeMode::Presence
} else {
MergeMode::Count
};
let n_genomes = src.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
}).len();
info!(
"filter: {} genome(s), mode={:?}, source={}",
n_genomes, mode, args.source.display()
);
let mut filters = args.filter.build_filters(&src.meta());
}));
let mut filters: Vec<Box<dyn KmerFilter>> =
vec![Box::new(args.group_filter.build_filter(&src.meta()))];
if let Some(v) = args.min_total_count {
filters.push(Box::new(MinTotalCount { total: v }));
}
@@ -82,20 +69,32 @@ pub fn run(args: FilterCmdArgs) {
filters.push(Box::new(MinComplexity { level_max: args.complexity_level_max, theta }));
}
// Source is opened read-only above and needs no lock; only the
// destination is written.
let _lock = obisys::DirLock::acquire(&args.output).unwrap_or_else(|e| {
eprintln!("error locking output directory {}: {e}", args.output.display());
let n_genomes = src.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
});
}).len();
info!(
"filter: {} genome(s), source={}",
n_genomes, args.source.display()
);
let mut rep = Reporter::new();
KmerIndex::rebuild(&args.output, &src, &filters, mode, args.force, &mut rep)
.unwrap_or_else(|e| {
let t = Stage::start("filter");
let pb = progress_bar("filter", src.n_partitions() as u64, "partitions");
let mut alg = Filter::new(Arc::clone(&src), &args.output, &filters)
.presence(args.presence)
.force(args.force)
.sparse(!args.dense)
.on_progress(|_: Progress| pb.inc(1));
let dst = alg.run().unwrap_or_else(|e| {
eprintln!("error filtering index: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
info!("filtered index → {}", dst.dir().display());
alg.reporter().print();
rep.print();
info!("filtered index → {}", args.output.display());
}
-10
View File
@@ -168,16 +168,6 @@ pub fn run(args: IndexArgs) {
let output = args.output.clone();
let mut rep = Reporter::new();
// Locked for the whole build (including a possible --force removal +
// recreation below): a second `index` run resuming/overwriting the same
// output directory concurrently would otherwise corrupt it. Unlinking
// the lock file via --force's remove_dir_all is safe — the held file
// descriptor keeps the lock regardless of the directory entry.
let _lock = obisys::DirLock::acquire(&output).unwrap_or_else(|e| {
eprintln!("error locking output directory {}: {e}", output.display());
std::process::exit(1);
});
// ── Resolve evidence kind ────────────────────────────────────────────────
let (evidence, effective_kmer_size) = if args.approx {
let (z, b, fp) = resolve_approx_params(args.findere_z, args.evidence_bits, args.fp);
+47 -23
View File
@@ -1,8 +1,10 @@
use std::path::PathBuf;
use clap::Args;
use obikindex::{KmerIndex, MergeMode};
use obisys::Reporter;
use obikalgorithm::Algorithm;
use obikindex::KmerIndex;
use obikmerge::{Merge, MergeMode};
use obisys::{Progress, Reporter, Stage, progress_bar};
use tracing::info;
#[derive(Args)]
@@ -27,20 +29,23 @@ pub struct MergeArgs {
#[arg(long, default_value_t = false)]
pub rename_duplicates: bool,
/// Fraction of available RAM reserved as memory budget for parallel partition merging.
/// Reduce if OOM occurs despite the adaptive scheduler (e.g. --budget-fraction 0.3).
#[arg(long, default_value_t = 0.5)]
pub budget_fraction: f64,
/// Pack the output's presence matrices in the dense format instead of the default sparse one
#[arg(long, default_value_t = false)]
pub dense: bool,
}
pub fn run(args: MergeArgs) {
let sources: Vec<KmerIndex> = args.sources.iter().map(|p| {
let sources: Vec<KmerIndex> = args
.sources
.iter()
.map(|p| {
info!("opening source index: {}", p.display());
KmerIndex::open(p).unwrap_or_else(|e| {
eprintln!("error opening source index {}: {e}", p.display());
std::process::exit(1);
})
}).collect();
})
.collect();
// Auto-detect mode: count if all sources have count data, presence otherwise.
// --force-presence overrides to presence regardless.
@@ -52,34 +57,53 @@ pub fn run(args: MergeArgs) {
};
info!(
"merge mode: {}",
if mode == MergeMode::Count { "count" } else { "presence/absence" }
if mode == MergeMode::Count {
"count"
} else {
"presence/absence"
}
);
let source_refs: Vec<&KmerIndex> = sources.iter().collect();
let n_genomes: usize = sources.iter().map(|s| {
s.meta().genomes().unwrap_or_else(|e| {
let n_genomes: usize = sources
.iter()
.map(|s| {
s.meta()
.genomes()
.unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
}).len()
}).sum();
})
.len()
})
.sum();
info!(
"merging {} index(es), {} genome(s) total → {}",
sources.len(), n_genomes, args.output.display()
sources.len(),
n_genomes,
args.output.display()
);
// Only the destination is written; sources are opened read-only above
// and need no lock.
let _lock = obisys::DirLock::acquire(&args.output).unwrap_or_else(|e| {
eprintln!("error locking output directory {}: {e}", args.output.display());
std::process::exit(1);
});
let mut rep = Reporter::new();
KmerIndex::merge(&args.output, &source_refs, mode, args.force, args.rename_duplicates, args.budget_fraction, &mut rep).unwrap_or_else(|e| {
let t = Stage::start("merge");
let n_partitions = source_refs.first().map(|s| s.n_partitions()).unwrap_or(0);
let pb = progress_bar("merge", n_partitions as u64, "partitions");
let mut merge = Merge::new(&source_refs, &args.output, mode)
.force(args.force)
.rename_duplicates(args.rename_duplicates)
.sparse(!args.dense)
.on_progress(|_: Progress| pb.inc(1));
let dst = merge.run().unwrap_or_else(|e| {
eprintln!("error merging: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
info!("merge done — output at {}", dst.dir().display());
merge.reporter().print();
rep.print();
}
+7 -8
View File
@@ -1,16 +1,15 @@
pub mod annotate;
pub mod filter;
pub mod pack;
pub(crate) mod predicate;
pub mod select;
pub mod utils;
pub mod phylo;
pub mod convert;
pub mod dump;
pub mod estimate;
pub mod filter;
pub mod index;
pub mod merge;
pub mod nametree;
pub mod pack;
mod predicate;
pub mod phylo;
pub mod query;
pub mod reindex;
pub mod select;
pub mod superkmer;
pub mod unitig;
pub mod utils;
-146
View File
@@ -1,146 +0,0 @@
use std::path::{Path, PathBuf};
use clap::Args;
use tracing::info;
// ── Translate a numerically-labelled tree export back to real taxon names ──
//
// TNT/PhyG write bare numeric leaf labels (1-based, in the same order as the
// FASTA fed to them) — this reads that order back from the FASTA header
// line and emits a NEXUS `translate` table alongside the tree(s), unchanged
// otherwise. Readable directly by FigTree/PearTree/`ape` etc.
#[derive(Args)]
pub struct NameTreeArgs {
/// Tree file to translate: a TNT-style NEXUS export (`tree NAME = [&U]
/// ...;`, topology on the same or the next line) or a plain Newick file
/// (single `(...);` tree, no header)
pub tree: PathBuf,
/// FASTA file whose record order gives the numeric taxon labels
/// (1-based) — typically the `_sankoff.fasta`/`_snp.fasta` used to
/// produce `tree`
#[arg(long)]
pub fasta: PathBuf,
/// Output NEXUS file (taxa block + translate table + tree(s), topology
/// unchanged)
#[arg(short, long)]
pub output: PathBuf,
}
pub fn run(args: NameTreeArgs) {
let labels = read_fasta_labels(&args.fasta);
if labels.is_empty() {
eprintln!("error: no FASTA headers found in {}", args.fasta.display());
std::process::exit(1);
}
let content = std::fs::read_to_string(&args.tree).unwrap_or_else(|e| {
eprintln!("error reading {}: {e}", args.tree.display());
std::process::exit(1);
});
let trees = extract_trees(&content);
if trees.is_empty() {
eprintln!("error: no tree found in {}", args.tree.display());
std::process::exit(1);
}
write_named_nexus(&labels, &trees, &args.output);
}
fn read_fasta_labels(path: &Path) -> Vec<String> {
let content = std::fs::read_to_string(path).unwrap_or_else(|e| {
eprintln!("error reading {}: {e}", path.display());
std::process::exit(1);
});
content.lines()
.filter(|l| l.starts_with('>'))
.map(|l| {
let header = &l[1..];
// `obifastwrite::write_record` appends a ` {json}` annotation —
// not part of the taxon name.
match header.find(" {") {
Some(pos) => header[..pos].to_string(),
None => header.to_string(),
}
})
.collect()
}
/// Finds every `tree NAME = [&U] TOPOLOGY;` (rooting comment optional,
/// topology on the same line or the next non-empty one), or — if none of
/// that syntax is found — treats the whole file as one bare Newick tree.
fn extract_trees(content: &str) -> Vec<(String, String)> {
let lines: Vec<&str> = content.lines().collect();
let mut trees = Vec::new();
let mut i = 0;
while i < lines.len() {
let line = lines[i].trim();
if let Some(rest) = line.strip_prefix("tree ") {
if let Some(eq_pos) = rest.find('=') {
let name = rest[..eq_pos].trim().to_string();
let mut after_eq = rest[eq_pos + 1..].trim();
if after_eq.starts_with('[') {
if let Some(close) = after_eq.find(']') {
after_eq = after_eq[close + 1..].trim();
}
}
let topo = if after_eq.starts_with('(') {
after_eq.to_string()
} else {
i += 1;
while i < lines.len() && lines[i].trim().is_empty() {
i += 1;
}
lines.get(i).map(|s| s.trim().to_string()).unwrap_or_default()
};
if topo.starts_with('(') {
trees.push((name, topo));
}
}
}
i += 1;
}
if trees.is_empty() {
let trimmed = content.trim();
if trimmed.starts_with('(') && trimmed.ends_with(';') {
trees.push(("tree_1".to_string(), trimmed.to_string()));
}
}
trees
}
fn write_named_nexus(labels: &[String], trees: &[(String, String)], output: &Path) {
let mut out = String::new();
out.push_str("#NEXUS\n\n");
out.push_str("begin taxa;\n");
out.push_str(&format!(" dimensions ntax={};\n", labels.len()));
out.push_str(" taxlabels\n");
for lab in labels {
out.push_str(&format!(" {lab}\n"));
}
out.push_str(" ;\nend;\n\n");
out.push_str("begin trees;\n");
out.push_str(" translate\n");
let tr_lines: Vec<String> = labels.iter().enumerate()
.map(|(i, lab)| format!(" {} {lab}", i + 1))
.collect();
out.push_str(&tr_lines.join(",\n"));
out.push_str(";\n");
for (name, topo) in trees {
out.push_str(&format!(" tree {name} = [&U] {topo}\n"));
}
out.push_str("end;\n");
std::fs::write(output, &out).unwrap_or_else(|e| {
eprintln!("error writing {}: {e}", output.display());
std::process::exit(1);
});
info!(
"named tree(s) → {} ({} tree{}, {} taxa)",
output.display(), trees.len(), if trees.len() == 1 { "" } else { "s" }, labels.len(),
);
}
+30 -13
View File
@@ -1,8 +1,10 @@
use std::path::PathBuf;
use std::sync::Arc;
use clap::Args;
use obikindex::KmerIndex;
use obisys::{Reporter, Stage};
use obikrebuild::IndexCompact;
use obisys::{Reporter, Stage, progress_bar};
use tracing::info;
#[derive(Args)]
@@ -10,13 +12,17 @@ pub struct PackArgs {
/// Index directory to pack
pub index: PathBuf,
/// Pack presence matrices into the sparse, deduplicated on-disk format
/// instead of the dense one — see `DevDocMD/architecture/siblings.md`.
/// Smaller and faster for single-row access on real, sparse data;
/// column-oriented access (`--metric` distance matrices) is much
/// slower on the sparse format.
#[arg(long)]
pub sparse: bool,
/// Compact every partition's accumulated layers into one before packing
/// — undoes the multi-layer stopgap `merge` leaves behind.
#[arg(long, default_value_t = false)]
pub compact_layers: bool,
/// Pack presence and count matrices into the dense on-disk format instead
/// of the default sparse, deduplicated one. Dense is faster for
/// column-oriented access (`--metric` distance matrices); sparse is
/// smaller and faster for single-row access on real, sparse data.
#[arg(long, default_value_t = false)]
pub dense: bool,
}
pub fn run(args: PackArgs) {
@@ -27,10 +33,10 @@ pub fn run(args: PackArgs) {
std::process::exit(1);
});
let idx = KmerIndex::open(&args.index).unwrap_or_else(|e| {
let idx = Arc::new(KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
});
}));
let n_genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
@@ -43,13 +49,24 @@ pub fn run(args: PackArgs) {
);
let mut rep = Reporter::new();
let t = Stage::start("pack");
idx.pack_matrices(args.sparse).unwrap_or_else(|e| {
if args.compact_layers {
let t = Stage::start("compact layers");
let pb = progress_bar("compact", idx.n_partitions() as u64, "partitions");
idx.compact_layers(|| pb.inc(1)).unwrap_or_else(|e| {
eprintln!("compact error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
}
let t = Stage::start("pack");
idx.pack_matrices(!args.dense).unwrap_or_else(|e| {
eprintln!("pack error: {e}");
std::process::exit(1);
});
rep.push(t.stop());
rep.print();
}
+233 -218
View File
@@ -1,10 +1,16 @@
use std::path::PathBuf;
use clap::Args;
use obikindex::DistanceMetric;
use obikphylo::{DistanceMetric, SnpDistanceKind};
/// `--distance` value — either one of `obikphylo::DistanceMetric`'s
/// whole-index metrics (routed to `IndexCache::distance`) or one of
/// `obikphylo::SnpDistanceKind`'s `snp-*` corrections (routed to
/// `SiblingExt::snp_distance`, the sibling-annex pipeline) — two genuinely
/// different code paths behind one CLI vocabulary, see
/// `DevDocMD/theory/evolutionary_distances.md`, "`--distance` unification".
#[derive(clap::ValueEnum, Clone, Copy, Debug)]
pub enum MetricArg {
pub enum DistanceArg {
Jaccard,
Mash,
Hamming,
@@ -17,88 +23,105 @@ pub enum MetricArg {
Hellinger,
#[value(name = "hellinger-euclidean")]
HellingerEuclidean,
#[value(name = "snp-raw")]
SnpRaw,
#[value(name = "snp-jc")]
SnpJc,
#[value(name = "snp-k2p")]
SnpK2p,
#[value(name = "snp-k81")]
SnpK81,
#[value(name = "snp-f81")]
SnpF81,
#[value(name = "snp-t92")]
SnpT92,
#[value(name = "snp-tn93")]
SnpTn93,
#[value(name = "snp-tv")]
SnpTv,
}
impl From<MetricArg> for DistanceMetric {
fn from(m: MetricArg) -> Self {
match m {
MetricArg::Jaccard => DistanceMetric::Jaccard,
MetricArg::Mash => DistanceMetric::Mash,
MetricArg::Hamming => DistanceMetric::Hamming,
MetricArg::BrayCurtis => DistanceMetric::BrayCurtis,
MetricArg::RelfreqBrayCurtis => DistanceMetric::RelfreqBrayCurtis,
MetricArg::Euclidean => DistanceMetric::Euclidean,
MetricArg::RelfreqEuclidean => DistanceMetric::RelfreqEuclidean,
MetricArg::Hellinger => DistanceMetric::Hellinger,
MetricArg::HellingerEuclidean => DistanceMetric::HellingerEuclidean,
impl DistanceArg {
/// `Some` for the whole-index metrics, `None` for `snp-*` values.
pub fn as_classic(self) -> Option<DistanceMetric> {
Some(match self {
DistanceArg::Jaccard => DistanceMetric::Jaccard,
DistanceArg::Mash => DistanceMetric::Mash,
DistanceArg::Hamming => DistanceMetric::Hamming,
DistanceArg::BrayCurtis => DistanceMetric::BrayCurtis,
DistanceArg::RelfreqBrayCurtis => DistanceMetric::RelfreqBrayCurtis,
DistanceArg::Euclidean => DistanceMetric::Euclidean,
DistanceArg::RelfreqEuclidean => DistanceMetric::RelfreqEuclidean,
DistanceArg::Hellinger => DistanceMetric::Hellinger,
DistanceArg::HellingerEuclidean => DistanceMetric::HellingerEuclidean,
_ => return None,
})
}
/// `Some` for the `snp-*` values, `None` for the whole-index metrics.
pub fn as_snp(self) -> Option<SnpDistanceKind> {
Some(match self {
DistanceArg::SnpRaw => SnpDistanceKind::Raw,
DistanceArg::SnpJc => SnpDistanceKind::Jc,
DistanceArg::SnpK2p => SnpDistanceKind::K2p,
DistanceArg::SnpK81 => SnpDistanceKind::K81,
DistanceArg::SnpF81 => SnpDistanceKind::F81,
DistanceArg::SnpT92 => SnpDistanceKind::T92,
DistanceArg::SnpTn93 => SnpDistanceKind::Tn93,
DistanceArg::SnpTv => SnpDistanceKind::Tv,
_ => return None,
})
}
}
/// Genome-vs-genome distance computation: the whole-index `--distance` path
/// (classic metrics + `snp-*` corrections/NJ/UPGMA), annex construction
/// (`--sibling-annex`), annex diagnostics (`--sibling-stats`,
/// `--sibling-hist`), entropy reporting (`--shannon`), SNP pseudo-alignment
/// sampling (`--pseudo-alignment`, `--subsample`, `--free-loss`,
/// `--no-ambiguity`, `--entropy`/`--entropy-sd`), Sankoff cost-matrix
/// calibration (`--sankoff`, `--sankoff-ratio-ceiling`) and its TNT/PhyG/
/// IQ-TREE exports (`--tnt`, `--phyg`, `--iqtree`/`--iqtree-min-freq`,
/// `--sankoff-cost-scale`), and Family Overlap (`--family-overlap`,
/// `--min-shared-family`).
#[derive(Args)]
pub struct PhyloArgs {
/// Index directory
pub index: PathBuf,
/// Distance metric to compute
#[arg(long, value_enum, default_value = "jaccard")]
pub metric: MetricArg,
/// Minimum count to consider a kmer present when computing Jaccard on count indexes
#[arg(long, default_value = "1")]
pub presence_threshold: u32,
/// Also output the shared-kmer count matrix (CSV)
#[arg(long)]
pub shared_kmers: bool,
/// Compute and write a Neighbor-Joining tree (Newick)
#[arg(long)]
pub nj: bool,
/// Compute and write a UPGMA tree (Newick)
#[arg(long)]
pub upgma: bool,
/// Build the sibling-count/minorant annex on this (multi-genome) index
/// — see `DevDocMD/theory/evolutionary_distances.md`, Step 2b. Construction
/// only; does not by itself compute or write any statistics.
#[arg(long)]
pub sibling_annex: bool,
/// Exclude a genome (by its exact label) from every computation below
/// that reads the sibling annex — `--raw-snp-distance`/`--raw-snp-counts`,
/// `--snp`, and `--sankoff` (and everything `--sankoff` implies: the
/// cardinality/composition transition models, the exported
/// matrix/alignment, `--tnt`/`--phyg`/`--iqtree`). Repeatable. Does
/// *not* affect the plain `--metric` distance matrix/NJ/UPGMA path (a
/// different, unrelated computation). Applied by zeroing the excluded
/// genome's row/column after `raw_snp_distance` runs (a pair with zero
/// counts is already skipped by `base_pair_tally`/`cardinality_tally`,
/// so this needs no change to the underlying traversal) and by
/// dropping its row from `snp_pseudo_alignment`'s output — the annex
/// is still built/scanned for the excluded genome too, just not used
/// afterward. For a genome with almost no informative sites shared
/// with anything else (see `DevDocMD/theory/evolutionary_distances.md`,
/// the IQ-TREE/Mash rogue-taxon discussion), its presence can
/// otherwise silently bias the transition models.
/// Exclude a genome (by its exact label) — from `--pseudo-alignment`'s
/// sampling (a family whose only polymorphism lived in an excluded
/// genome is discarded during sampling, not filtered afterward — see
/// `obikphylo::siblings::extensions::SiblingExt::snp_pseudo_alignment`'s
/// own docs) and from the distance matrix / shared-kmer matrix CSV
/// output (row and column both dropped; the underlying computation
/// itself is unaffected). Repeatable.
#[arg(long = "exclude-genome", value_name = "LABEL")]
pub exclude_genome: Vec<String>,
/// Auto-exclude any genome whose mean shared-family count against every
/// other genome (same statistic as `--family-overlap`'s matrix, averaged
/// over each row excluding the diagonal) falls below this threshold —
/// same exclusion machinery as `--exclude-genome`, applied on top of it
/// rather than instead of it. Empirically, genomes below ~1000 shared
/// families on the 20-genome benchmark are exactly the ones that placed
/// themselves arbitrarily under `--tnt`/`--iqtree` (near-zero branch
/// lengths, grafted inside unrelated clades) — too little real
/// constraint on where they belong. See
/// `DevDocMD/theory/evolutionary_distances.md`, "Locus dropout under
/// incomplete coverage".
/// Auto-exclude any genome whose mean shared-variable-family count
/// against every other genome (`FamilyOverlap::mean_row` — the same
/// per-row statistic `--family-overlap`'s own matrix shows) falls below
/// this threshold — same exclusion machinery as `--exclude-genome`,
/// applied on top of it rather than instead of it. Applies to the
/// `snp-*` `--distance`/`--pseudo-alignment`/`--sankoff` computations
/// below (all sibling-annex-based); does *not* affect the whole-index
/// `--distance` metrics (jaccard, hamming, bray-curtis, ...) or their
/// matrix/NJ/UPGMA output — a genome with too little SNP-family
/// coverage to trust is a different concern from one whose plain k-mer
/// profile is simply divergent. Requires the Family Overlap annex
/// (built on demand if missing, same as every other annex here — see
/// `obikphylo::siblings::extensions::SiblingExt::family_overlap`'s own
/// docs).
#[arg(long, value_name = "N")]
pub min_shared_family: Option<f64>,
/// Build (or rebuild) the sibling-count/minorant annex — independent of
/// the distance metric below, meant to be run routinely, ahead of any
/// SNP-family distance computation that will later consume it.
#[arg(long)]
pub sibling_annex: bool,
/// Tally the sibling-count distribution (CSV) of an already-built annex
/// (run with `--sibling-annex` first, in this invocation or an earlier
/// one). A separate, occasional diagnostic pass — not run every time the
@@ -109,124 +132,83 @@ pub struct PhyloArgs {
/// Print just the global family-size histogram (1-4 members) of an
/// already-built annex — the `global` row `--sibling-stats` also
/// writes, but without the per-genome breakdown, so it skips
/// `--sibling-stats`'s cross-partition resolution entirely: reads only
/// the already-open annex mask and each layer's own `unitigs.bin`, cost
/// independent of the rest of the index. A quick sanity check that the
/// annex itself is sound, decoupled from `--sibling-stats`'s much
/// heavier per-genome pass.
/// `--sibling-stats`'s cross-partition resolution entirely (annex bits
/// only).
#[arg(long)]
pub sibling_hist: bool,
/// Compute the raw p-distance restricted to loci that are single-copy
/// in both genomes of each pair (an already-built sibling annex is
/// required — run with `--sibling-annex` first, in this invocation or
/// an earlier one). A quick way to test the central-position SNP
/// estimator against a real index; not the full `SnpTally` design.
#[arg(long)]
pub raw_snp_distance: bool,
/// Write the raw per-pair counts (`n_snp`, `n_shared`, `n_eligible`)
/// behind `--raw-snp-distance`'s ratio, one row per genome pair — a
/// diagnostic table, not a matrix. The ratio alone can't distinguish
/// "identical at every eligible locus" from "almost no eligible loci
/// at all" (e.g. `0.0` from 0/2 looks the same as `0.0` from 0/2000),
/// and that distinction matters a lot for genome pairs near the edge
/// of what central-position families can resolve (see
/// `DevDocMD/theory/evolutionary_distances.md`, "Run 3" and the
/// IQ-TREE/Mash comparison). Same annex requirement as
/// `--raw-snp-distance`.
#[arg(long)]
pub raw_snp_counts: bool,
/// Write a SNP-only pseudo-alignment (FASTA, IUPAC-coded) from an
/// already-built sibling annex — one row per genome, one column per
/// variable family (monomorphic families skipped), no flanking
/// sequence. See `DevDocMD/theory/evolutionary_distances.md`,
/// "Multi-genome framing: family as pseudo-alignment column".
#[arg(long)]
pub snp: bool,
/// Cap the number of variable families (non-monomorphic minorants,
/// `family_size() >= 2`) retained by `--snp`/`--sankoff` (and everything
/// `--sankoff` implies) and `--shannon`, to (approximately) this many —
/// sampled proportionally per layer, so the pseudo-alignment/entropy
/// report stays usable on an index far larger than the sample itself
/// (mandatory, not optional, once the index is large enough that a full
/// pseudo-alignment can't be materialized at all). See
/// `DevDocMD/architecture/siblings.md`, "`--subsample`/`--shannon`". If the
/// index has fewer non-monomorphic minorants than this, every one of
/// them is kept — no error, no under/over-shoot handling needed.
#[arg(long, value_name = "N")]
pub subsample: Option<usize>,
/// Enable entropy-biased selection: instead of a uniform draw among
/// eligible families, weight each candidate by an unnormalised Gaussian
/// kernel on its own entropy15 (`w = exp(-(entropy-mu)^2/(2*sigma^2))`,
/// `1` exactly at `entropy == mu`, decaying smoothly away from it — no
/// hard cutoff). Activates as soon as `--entropy` or `--entropy-sd` is
/// given; the other defaults to `1.0`/`0.5` if unset. Combines with
/// `--subsample N` (the joint accept probability is `p0 * w`, `p0`
/// chosen so the expected count is approximately `N`) or works alone
/// (a pure soft entropy filter over the whole index, no size target).
/// First use on an index pays a one-time cost building a per-layer
/// entropy annex (a full, unsampled scan); every later run reuses it.
/// See `DevDocMD/architecture/siblings.md`, "Entropy-biased selection".
#[arg(long, value_name = "MU")]
pub entropy: Option<f64>,
/// Standard deviation of `--entropy`'s Gaussian kernel. See `--entropy`.
#[arg(long, value_name = "SIGMA")]
pub entropy_sd: Option<f64>,
/// Write <prefix>_shannon.csv: per-family Shannon entropy (bits, over
/// the 15 non-empty subsets of `{A,C,G,T}`, `∅`/absent genomes excluded
/// from the denominator — see `DevDocMD/architecture/siblings.md`,
/// "Entropy definition") of every non-monomorphic minorant, one row per
/// family. Combine with `--subsample N` for a bounded diagnostic sample
/// instead of a full-index pass. Same annex requirement as `--snp`.
#[arg(long)]
pub shannon: bool,
/// Write an NxN CSV (`<prefix>_family_overlap.csv`) of, for each genome
/// pair, how many variable families (same set `--snp`'s pseudo-alignment
/// uses — `family_size() >= 2`) both genomes actually carry a call for
/// (neither is `∅`). A direct read of how much informative content two
/// genomes actually share at the family level — the diagnostic for why
/// a genome with little overlap with anything else (e.g. an
/// under-covered or very divergent one) ends up placed unstably by
/// `--tnt`/`--iqtree`: little-to-no shared, real data to constrain it.
/// Same annex requirement as `--snp`.
/// Write the Family Overlap matrix (CSV) — number of shared *variable*
/// families per genome pair, index-wide (`obikphylo::siblings::FamilyOverlap`).
/// Built on demand if missing (same as every other annex here); no
/// `--sibling-annex` prerequisite beyond that. Every genome is written,
/// unfiltered by `--exclude-genome`/`--min-shared-family` — a raw
/// coverage diagnostic, not a computation those exclusions are meant to
/// protect.
#[arg(long)]
pub family_overlap: bool,
/// Calibrate a 16-state Sankoff cost matrix and its matching
/// pseudo-alignment from an already-built sibling annex (run with
/// `--sibling-annex` first, in this invocation or an earlier one), for
/// use with TNT/PhyG. See `DevDocMD/theory/evolutionary_distances.md`,
/// "Sankoff parsimony as the resolution of the 16-state model problem".
/// Write a per-family Shannon entropy report (CSV) — requires an
/// already-built sibling annex (`--sibling-annex` first, in this
/// invocation or an earlier one). Always a full, unsampled scan of
/// every family (`--subsample`/`--entropy`/`--entropy-sd` below only
/// apply to `--pseudo-alignment`, not this).
#[arg(long)]
pub sankoff: bool,
pub shannon: bool,
/// Recode a family's non-detection (`∅`, no member observed in a
/// genome) as TNT/PhyG/IQ-TREE's own missing-data symbol (`?`) in
/// `--sankoff`'s FASTA and every export built from it (`--tnt`,
/// `--phyg`, `--iqtree`), instead of an ordinary, costed 16th alphabet
/// state (the default). For genome-skim/reduced-representation inputs
/// (coverage often < 1x), non-detection is dominated by sampling
/// failure, not true loss — scoring it as a real state risks grouping
/// genomes by shared undersampling rather than shared ancestry. `?`
/// (not `-`) because `-` still carries gap/indel semantics in these
/// tools; a non-detected family is not an observed deletion. See
/// `DevDocMD/theory/evolutionary_distances.md`, "Locus dropout under
/// incomplete coverage".
/// Write a SNP-only pseudo-alignment (FASTA) — requires
/// `--subsample <N>` and an already-built sibling annex
/// (`--sibling-annex` first, in this invocation or an earlier one).
#[arg(long)]
pub pseudo_alignment: bool,
/// Target number of variable sites to sample index-wide for
/// `--pseudo-alignment`/`--sankoff` (mandatory for both) and for a
/// `snp-*` `--distance` value (optional there: omitted means exhaustive
/// — every non-monomorphic minorant of the whole index, not an
/// approximation, see `obikphylo::siblings::SiblingExt::snp_distance`'s
/// own docs) — a target, not a guarantee when given (proportional
/// per-layer sampling; see
/// `obikphylo::siblings::SiblingExt::snp_pseudo_alignment`'s own docs).
#[arg(long)]
pub subsample: Option<usize>,
/// In `--pseudo-alignment`, treat a genome carrying none of a family's
/// observed members (`∅`) as missing data (`?`) rather than a real
/// character state.
#[arg(long)]
pub free_loss: bool,
/// In `--pseudo-alignment`, treat a genome carrying more than one
/// member of a family (ambiguous) as missing data (`?`) rather than an
/// IUPAC ambiguity code.
#[arg(long)]
pub no_ambiguity: bool,
/// Entropy-biased sampling target (Gaussian kernel mean) for
/// `--pseudo-alignment` — activates biasing as soon as this or
/// `--entropy-sd` is given; the other defaults to 1.0/0.5.
#[arg(long)]
pub entropy: Option<f64>,
/// Entropy-biased sampling kernel width (Gaussian standard deviation)
/// for `--pseudo-alignment` — see `--entropy`.
#[arg(long)]
pub entropy_sd: Option<f64>,
/// Calibrate a 16-state Sankoff cost matrix (and its matching
/// pseudo-alignment) from an already-built sibling annex — requires
/// `--subsample <N>`, and shares `--free-loss`/`--no-ambiguity`/
/// `--entropy`/`--entropy-sd` with `--pseudo-alignment` (one draw, same
/// selection feeds both the alignment and every calibration tally).
#[arg(long)]
pub sankoff: bool,
/// Exclude genome pairs whose raw SNP ratio exceeds this value from the
/// `p_hat` calibration pooled by `--sankoff` — a pair this close to
/// saturation carries no information about `p_hat` and would bias it
/// upward if pooled in (unlike a low eligible-loci count, which barely
/// moves the pooled estimate either way — see design doc).
/// base-pair (composition) calibration `--sankoff` pools — a pair this
/// close to substitution saturation carries no information about the
/// true substitution spectrum. Does *not* gate the cardinality
/// calibration (see `obikphylo::siblings::CardinalityTally`'s own
/// docs for why).
#[arg(long, default_value = "0.5")]
pub sankoff_ratio_ceiling: f64,
@@ -250,66 +232,99 @@ pub struct PhyloArgs {
pub phyg: bool,
/// Also write <prefix>_iqtree.model and <prefix>_iqtree.fasta, a
/// custom-model file and a matching
/// recoded alignment for genuine maximum-likelihood inference with
/// IQ-TREE (`iqtree3 -s ... --seqtype MORPH -m ...+ASC`) — real branch
/// lengths, unlike `--tnt`/`--phyg`'s parsimony step counts. The model
/// is the reversible `Q(i,j) = R(i,j)·π_j` construction: `R`
/// (exchangeability, symmetric) recovered from the same calibrated
/// cost matrix `--sankoff` computes, `π` the real empirical state
/// frequencies counted from the alignment (not IQ-TREE's `+FO`/`+F` —
/// neither applies to a custom-file model, see
/// `DevDocMD/theory/evolutionary_distances.md`). Only the states that
/// actually occur in this alignment are kept, compactly renumbered
/// (IQ-TREE infers its state count from the alignment itself, and a
/// gap in the numbering would silently misalign the model file).
/// Implies `--sankoff`.
/// custom-model file and a matching recoded alignment for genuine
/// maximum-likelihood inference with IQ-TREE (`iqtree3 -s ...
/// --seqtype MORPH -m ...+ASC`) — real branch lengths, unlike
/// `--tnt`/`--phyg`'s parsimony step counts. The model is the
/// reversible `Q(i,j) = R(i,j)·π_j` construction: `R` (exchangeability,
/// symmetric) recovered from the same calibrated cost matrix
/// `--sankoff` computes, `π` the real empirical state frequencies
/// counted from the alignment. Only the states that actually occur in
/// this alignment are kept, compactly renumbered (IQ-TREE infers its
/// state count from the alignment itself, and a gap in the numbering
/// would silently misalign the model file). Implies `--sankoff`.
#[arg(long)]
pub iqtree: bool,
/// Under `--iqtree --free-loss`, also recode to `?` (the same
/// missing-data treatment as `-`) any state whose empirical frequency
/// in the alignment falls below this threshold — not just genuinely
/// absent calls. States encoding 3 or 4 simultaneously-observed
/// central bases (IUPAC `V`/`H`/`K`.../`N` for 3, `N` for 4) are rare
/// by construction and, on real data, land in exactly this low-frequency
/// range — more likely assembly/detection noise than a genuine,
/// widely-preserved multi-way polymorphism, the same "sampling failure,
/// not true signal" reasoning `--free-loss` already applies to absence.
/// Confirmed on real data to matter: `iqtree3`'s own "Numerical
/// underflow for lh-derivative" warnings and the exact-zero
/// exchangeability rows this was meant to fix (see
/// `DevDocMD/theory/evolutionary_distances.md`) both trace back to
/// states this thin. No effect without `--free-loss` (there is no
/// missing-data symbol to recode to otherwise). `<prefix>_iqtree_states.csv`
/// reports the frequency actually used to decide.
/// absent calls. States encoding 3 or 4 simultaneously-observed central
/// bases (IUPAC `V`/`H`/`K`.../`N` for 3, `N` for 4) are rare by
/// construction and often land in exactly this low-frequency range —
/// more likely assembly/detection noise than a genuine, widely-preserved
/// multi-way polymorphism, the same "sampling failure, not true signal"
/// reasoning `--free-loss` already applies to absence. No effect
/// without `--free-loss` (there is no missing-data symbol to recode to
/// otherwise). `<prefix>_iqtree_states.csv` reports the frequency
/// actually used to decide.
#[arg(long, default_value = "0.001")]
pub iqtree_min_freq: f64,
/// Scale factor applied before rounding real-valued costs to the
/// integers both `--tnt`'s smatrix/cost commands and `--phyg`'s `tcm:`
/// matrix require. Keep this small: the total tree score is this scale
/// times the sum of per-character costs across every character (908k+
/// for a typical run here), and there are hints in TNT's own manual
/// that at least some of its internal accumulators are 32-bit — a large
/// scale risks a silent integer overflow (undetectable, not just a
/// crash) far more costly than the resolution a bigger factor would
/// buy. Shared between `--tnt` and `--phyg` rather than split into two
/// flags: both scale the same calibrated matrix for the same reason
/// (integer-only cost commands), and no PhyG-specific accumulator-width
/// constraint has actually been found to justify a different default.
/// times the sum of per-character costs across every character, and
/// there are hints in TNT's own manual that at least some of its
/// internal accumulators are 32-bit — a large scale risks a silent
/// integer overflow (undetectable, not just a crash) far more costly
/// than the resolution a bigger factor would buy. Shared between `--tnt`
/// and `--phyg` rather than split into two flags: both scale the same
/// calibrated matrix for the same reason (integer-only cost commands).
#[arg(long, default_value = "100")]
pub sankoff_cost_scale: f64,
/// Output prefix: <prefix>_dist.csv, <prefix>_shared.csv,
/// <prefix>_siblings.csv, <prefix>_sibling_hist.csv, <prefix>_rawsnp.csv,
/// <prefix>_rawsnp_counts.csv,
/// <prefix>_snp.fasta, <prefix>_family_overlap.csv, <prefix>_shannon.csv,
/// <prefix>_sankoff_matrix.csv, <prefix>_sankoff_params.yaml,
/// <prefix>_sankoff.fasta, <prefix>_sankoff.tnt, <prefix>_sankoff.tcm,
/// <prefix>_sankoff.pg, <prefix>_iqtree.model, <prefix>_iqtree.fasta,
/// <prefix>_nj.nwk, <prefix>_upgma.nwk.
/// If omitted, the distance matrix is written to stdout.
/// Distance to compute — either a whole-index metric (`jaccard`,
/// `mash`, `hamming`, `bray-curtis`, ...) or a `snp-*` correction over
/// the central-position SNP substitution spectrum (`snp-raw`, `snp-jc`,
/// `snp-k2p`, `snp-k81`, `snp-f81`, `snp-t92`, `snp-tn93`, `snp-tv`) —
/// the latter route to a different computation entirely
/// (`SiblingExt::snp_distance`, requires `--sibling-annex` first; see
/// `DevDocMD/theory/evolutionary_distances.md`, "`--distance`
/// unification" for the full catalog and why LogDet/Tajima-Nei/F84/
/// HKY85 aren't offered yet).
#[arg(long, value_enum, default_value = "jaccard")]
pub distance: DistanceArg,
/// Rate-heterogeneity correction (Jin-Nei gamma shape parameter `α`)
/// for `snp-*` `--distance` values that support it
/// (`obikphylo::SnpDistanceKind::supports_gamma`: every one except
/// `snp-raw`/`snp-tv`, which have nothing to correct/are deliberately
/// uncorrected). Has no effect on the whole-index metrics. Rejected at
/// runtime if given alongside an unsupported `--distance` value.
#[arg(long, value_name = "ALPHA")]
pub gamma_shape: Option<f64>,
/// Minimum count to consider a kmer present when computing Jaccard on count indexes
#[arg(long, default_value = "1")]
pub presence_threshold: u32,
/// Write the primary distance matrix as plain CSV instead of the
/// default relaxed-PHYLIP format (`n` on the first line, then one
/// `label<TAB>value...` row per genome — no 10-character label
/// truncation, unlike strict PHYLIP, not yet offered here). PHYLIP is
/// the default because it's what external NJ tools (PHYLIP `neighbor`,
/// FastME, T-REX, SplitsTree) actually read; CSV stays available for
/// scripting/inspection. Only affects the primary distance matrix —
/// `--shared-kmers` keeps its own CSV-only format regardless of this
/// flag.
#[arg(long)]
pub csv: bool,
/// Also output the shared-kmer count matrix (CSV)
#[arg(long)]
pub shared_kmers: bool,
/// Compute and write a Neighbor-Joining tree (Newick)
#[arg(long)]
pub nj: bool,
/// Compute and write a UPGMA tree (Newick)
#[arg(long)]
pub upgma: bool,
/// Output prefix: <prefix>_dist.csv, <prefix>_shared.csv, <prefix>_nj.nwk,
/// <prefix>_upgma.nwk. If omitted, the distance matrix is written to stdout.
#[arg(short, long)]
pub output: Option<PathBuf>,
}
@@ -1,98 +0,0 @@
use std::io::{BufWriter, Write};
use std::path::PathBuf;
use obikindex::KmerIndex;
use obikphylo::siblings::{SnpAlignment, SnpAlignmentExt};
use tracing::info;
// ── Family overlap: shared-family counts and the `--min-shared-family` /
// `--family-overlap` diagnostics built from them ────────────────────────────
//
// Same variable-family columns as `--snp`'s pseudo-alignment. Off-diagonal
// `[i][j]`: number of columns where both genome `i` and genome `j` carry a
// call (neither is `∅`) — how much informative family content two genomes
// actually share, the direct diagnostic for the rogue-taxon placement seen
// under `--free-loss` (a genome with little overlap with anything else has
// almost nothing left to constrain it). Diagonal `[i][i]` kept, deliberately
// not skipped: with `i == j` the condition "both non-`∅`" degenerates to
// "genome `i` non-`∅`", i.e. the total number of variable families genome
// `i` carries at all — a genome-level count worth having alongside the
// pairwise ones, not a separate computation.
/// `counts[i][j]` = number of variable-family columns where both genome `i`
/// and genome `j` carry a call (neither is `∅`). Shared between
/// `write_family_overlap_csv` and `--min-shared-family`'s auto-exclusion so
/// both read off the same definition of "shared family".
fn family_overlap_counts(alignment: &SnpAlignment) -> Vec<Vec<u64>> {
let n = alignment.sequences.len();
let mut counts = vec![vec![0u64; n]; n];
for i in 0..n {
for j in 0..n {
counts[i][j] = alignment.sequences[i].iter().zip(alignment.sequences[j].iter())
.filter(|&(&a, &b)| a != b'-' && b != b'-')
.count() as u64;
}
}
counts
}
/// Mean of row `i` in a `family_overlap_counts` matrix, excluding the
/// diagonal — how much informative content genome `i` shares with the
/// *average* other genome, the statistic `--min-shared-family` thresholds.
fn mean_offdiag(counts: &[Vec<u64>], i: usize) -> f64 {
let n = counts.len();
let sum: u64 = (0..n).filter(|&j| j != i).map(|j| counts[i][j]).sum();
sum as f64 / (n - 1) as f64
}
pub(super) fn write_family_overlap_csv(alignment: &SnpAlignment, labels: &[String], output: &Option<PathBuf>) {
let path = output.as_ref()
.map(|p| format!("{}_family_overlap.csv", p.display()))
.unwrap_or_else(|| "family_overlap.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
let n = labels.len();
let counts = family_overlap_counts(alignment);
write!(f, "genome").unwrap();
for g in labels { write!(f, ",{g}").unwrap(); }
writeln!(f).unwrap();
for (i, gi) in labels.iter().enumerate() {
write!(f, "{gi}").unwrap();
for j in 0..n {
write!(f, ",{}", counts[i][j]).unwrap();
}
writeln!(f).unwrap();
}
info!("family overlap matrix → {path}");
}
/// Sets `mask[i] = true` for every genome whose mean shared-family count
/// (`mean_offdiag`) falls below `threshold`, skipping genomes already
/// excluded (`mask[i]` already `true`, e.g. via `--exclude-genome`). Builds
/// its own `SnpAlignment` pass — same redundant-per-flag pattern already
/// used throughout `run()` (`--snp`/`--sankoff`/`--family-overlap` each call
/// `snp_pseudo_alignment` independently too).
pub(super) fn apply_min_shared_family_exclusion(
idx: &KmerIndex,
labels: &[String],
threshold: f64,
mask: &mut [bool],
) {
let alignment = idx.snp_pseudo_alignment(None, None).unwrap_or_else(|e| {
eprintln!("error computing SNP pseudo-alignment for --min-shared-family: {e}");
std::process::exit(1);
});
let counts = family_overlap_counts(&alignment);
for (i, label) in labels.iter().enumerate() {
if mask[i] {
continue; // already excluded via --exclude-genome
}
let mean = mean_offdiag(&counts, i);
if mean < threshold {
info!("--min-shared-family: excluding {label} (mean shared families = {mean:.1} < {threshold})");
mask[i] = true;
}
}
}
+50 -68
View File
@@ -15,38 +15,27 @@ use super::sankoff::{STATE_SYMBOL, state_index_table};
// performed on it here is different, hence no "sankoff" in these names,
// unlike `tnt::write_sankoff_tnt`/`phyg::write_sankoff_phyg`.
//
// Verified against the locally installed `iqtree3` binary/source (not just
// its docs — see `DevDocMD/theory/evolutionary_distances.md`, "Next
// direction: genuine ML branch lengths"), because the web documentation's
// `-mdef` NEXUS route turned out not to apply to a plain (non-mixture)
// custom morphology model. The real mechanism: pass a **file path**
// directly as `-m`, containing (as whitespace/newline-separated numbers)
// the lower-triangular exchangeability matrix `R` (`k(k-1)/2` values, PAML
// row-major order) immediately followed by the `k` state frequencies `π`
// on the same stream — `ModelMarkov::readRates`/`readStateFreq` read them
// in that exact order, no header, no separator required.
// The mechanism: pass a **file path** directly as `-m`, containing (as
// whitespace/newline-separated numbers) the lower-triangular exchangeability
// matrix `R` (`k(k-1)/2` values, PAML row-major order) immediately followed
// by the `k` state frequencies `π` on the same stream —
// `ModelMarkov::readRates`/`readStateFreq` read them in that exact order, no
// header, no separator required.
//
// `R` is recovered from the calibrated Sankoff cost matrix via
// `R(a,b) = exp(-cost(a,b))` (the cost is `-ln(rate)`, see "A concrete
// Sankoff cost matrix"), symmetric by construction (the underlying tally
// never captured direction). `π` is the real, empirical, non-uniform
// marginal frequency of each state across the whole alignment — precise
// enough at this sample size (908k+ sites) without spending IQ-TREE's own
// `+FO` ML degrees of freedom re-estimating it (checked: IQ-TREE's `+F`
// doesn't work as a shortcut here either, a custom-file model always
// requires the frequency line in the file itself). IQ-TREE reconstructs
// the (generally asymmetric) rate matrix internally as
// `Q(i,j) = R(i,j)·π_j` — reversible for *any* `π`, not just uniform,
// because `R` is symmetric.
// `R(a,b) = exp(-cost(a,b))` (the cost is `-ln(rate)`), symmetric by
// construction (the underlying tally never captured direction). `π` is the
// real, empirical, non-uniform marginal frequency of each state across the
// whole alignment. IQ-TREE reconstructs the (generally asymmetric) rate
// matrix internally as `Q(i,j) = R(i,j)·π_j` — reversible for *any* `π`, not
// just uniform, because `R` is symmetric.
//
// IQ-TREE infers its state count from the highest-ordinal symbol actually
// present in the alignment, not from a declared count (`--seqtype
// MORPH{N}` was tested and does not override this for real ML analysis,
// only for the `--alisim` simulator). So states that never occur anywhere
// in this particular alignment are dropped, and the survivors are
// renumbered compactly (`0..k-1`, order preserved) rather than leaving
// gaps that would silently misalign every value IQ-TREE reads. Both the
// model and the alignment must agree on this same renumbering, so it's
// present in the alignment, not from a declared count. So states that never
// occur anywhere in this particular alignment are dropped, and the
// survivors are renumbered compactly (`0..k-1`, order preserved) rather than
// leaving gaps that would silently misalign every value IQ-TREE reads. Both
// the model and the alignment must agree on this same renumbering, so it's
// computed once (`CompactAlphabet`) and shared between them.
const IQTREE_STATE_SYMBOL: [char; 16] = [
@@ -70,19 +59,16 @@ impl CompactAlphabet {
/// Under `--free-loss`, non-detection (`-`) becomes IQ-TREE's own missing
/// symbol (`?`) — ignored when IQ-TREE checks a site's constancy for
/// `+ASC`. A family kept as "variable" by `snp_pseudo_alignment`
/// (`family_size() >= 2`, a whole-annex property, oblivious to any one
/// column's actual calls) can still turn constant *among the genomes that
/// actually have data* once the non-detected ones are excluded from that
/// check — the same failure mode as the `--exclude-genome`/`drop_excluded`
/// fix in `mod.rs` (see `DevDocMD/theory/evolutionary_distances.md`, "Two
/// consistency bugs found and fixed post-implementation"), just triggered
/// by hiding cells instead of dropping whole rows. Same remedy: rescan
/// columns treating `-` as ignored, drop any where the remaining calls
/// agree on a single state. Parsimony (`--tnt`/`--phyg`) has no
/// no-invariant-site requirement, so this only runs on IQ-TREE's own copy
/// of the alignment, never mutating the one the caller also hands to those
/// two exports.
/// `+ASC`. A family kept as "variable" by the sampling (`family_size() >=
/// 2`, a whole-annex property, oblivious to any one column's actual calls)
/// can still turn constant *among the genomes that actually have data* once
/// the non-detected ones are excluded from that check — the same failure
/// mode `--exclude-genome` already had to account for, just triggered by
/// hiding cells instead of dropping whole rows. Same remedy: rescan columns
/// treating `-` as ignored, drop any where the remaining calls agree on a
/// single state. Parsimony (`--tnt`/`--phyg`) has no no-invariant-site
/// requirement, so this only runs on IQ-TREE's own copy of the alignment,
/// never mutating the one the caller also hands to those two exports.
fn drop_ascertainment_noninformative(alignment: &SnpAlignment) -> SnpAlignment {
let n_sites = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
let keep: Vec<bool> = (0..n_sites)
@@ -114,7 +100,7 @@ fn drop_ascertainment_noninformative(alignment: &SnpAlignment) -> SnpAlignment {
.collect()
})
.collect();
SnpAlignment { sequences }
SnpAlignment { sequences, genome_indices: alignment.genome_indices.clone() }
}
/// Recode every occurrence of a byte in `symbols` to `-` — the same
@@ -133,7 +119,7 @@ fn recode_symbols_as_absent(alignment: &SnpAlignment, symbols: &[u8]) -> SnpAlig
.collect()
})
.collect();
SnpAlignment { sequences }
SnpAlignment { sequences, genome_indices: alignment.genome_indices.clone() }
}
fn compact_alphabet(alignment: &SnpAlignment, free_loss: bool) -> CompactAlphabet {
@@ -212,7 +198,11 @@ fn write_iqtree_states_csv(alphabet: &CompactAlphabet, output: &Option<PathBuf>)
/// Write the `R` (exchangeability) + `π` (frequencies) model file IQ-TREE's
/// `-m <file>+ASC` reads. Returns the path, so the caller can print a
/// single combined "how to run this" message once the alignment is also
/// written.
/// written. The one bit of real computation this whole adapter does:
/// `R(a,b) = exp(-cost(a,b))`, recovering the exchangeability rate a
/// calibrated Sankoff parsimony cost implies for a continuous-time model —
/// a one-line inversion of the cost matrix's own `-ln(rate)` construction,
/// not a new estimate.
fn write_iqtree_model(
matrix: &[[f64; 16]; 16],
alphabet: &CompactAlphabet,
@@ -260,7 +250,7 @@ fn write_iqtree_model(
/// Write the pseudo-alignment recoded to the same compact `0..k-1` alphabet
/// as `write_iqtree_model`'s matrix — not `--sankoff`'s own IUPAC alphabet,
/// since IQ-TREE needs the symbol ordinal itself to match the surviving
/// state count (see this module's own doc comment on `MORPH{N}`).
/// state count (see this module's own doc comment on state inference).
fn write_iqtree_alignment(
alignment: &SnpAlignment,
labels: &[String],
@@ -279,7 +269,7 @@ fn write_iqtree_alignment(
std::process::exit(1);
}));
let n_sites = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
for (label, seq) in labels.iter().zip(alignment.sequences.iter()) {
for (&g, seq) in alignment.genome_indices.iter().zip(alignment.sequences.iter()) {
let recoded: Vec<u8> = seq
.iter()
.map(|&b| {
@@ -295,7 +285,7 @@ fn write_iqtree_alignment(
.collect();
write_record(
&recoded,
label,
&labels[g],
&[("n_sites", JsonVal::Num(n_sites as u64))],
&mut f,
)
@@ -335,12 +325,7 @@ pub(super) fn write_iqtree(
// `--iqtree-min-freq`: fold rare (likely-noisy) states into the same
// missing-data treatment `-` already gets under `--free-loss`, then
// recompute the alphabet on the further-filtered alignment — see
// `args.rs`'s docs on `--iqtree-min-freq` and
// `DevDocMD/theory/evolutionary_distances.md` for why (real data:
// states encoding 3-4 simultaneous central bases land in exactly this
// low-frequency range and correlate with `iqtree3`'s own numerical
// instability warnings).
// recompute the alphabet on the further-filtered alignment.
let refiltered;
let alignment = if free_loss {
let low_freq_symbols: Vec<u8> = alphabet
@@ -397,17 +382,18 @@ pub(super) fn write_iqtree(
mod tests {
use super::*;
fn alignment(sequences: Vec<Vec<u8>>) -> SnpAlignment {
let genome_indices = (0..sequences.len()).collect();
SnpAlignment { sequences, genome_indices }
}
#[test]
fn free_loss_excludes_absent_state_and_freq_sums_to_one() {
// 3 genomes, 2 sites. Site 0: g1='A', g2='C', g3='-' (absent).
// Site 1: g1='-', g2='-', g3='G'. Under free_loss, every '-' must
// be excluded from the frequency count entirely (not folded into
// state 0) — reproduces a user-reported suspicion that state 0
// ("absent") was still being counted despite being recoded to `?`
// (IQ-TREE's own missing symbol) in the alignment actually written.
let alignment = SnpAlignment {
sequences: vec![vec![b'A', b'-'], vec![b'C', b'-'], vec![b'-', b'G']],
};
// state 0).
let alignment = alignment(vec![vec![b'A', b'-'], vec![b'C', b'-'], vec![b'-', b'G']]);
let alphabet = compact_alphabet(&alignment, true);
@@ -427,9 +413,7 @@ mod tests {
#[test]
fn without_free_loss_absent_state_is_counted_normally() {
let alignment = SnpAlignment {
sequences: vec![vec![b'A', b'-'], vec![b'C', b'-'], vec![b'-', b'G']],
};
let alignment = alignment(vec![vec![b'A', b'-'], vec![b'C', b'-'], vec![b'-', b'G']]);
let alphabet = compact_alphabet(&alignment, false);
@@ -449,12 +433,10 @@ mod tests {
fn states_csv_maps_compact_symbols_back_to_canonical_ones() {
// 'A' (state 1) and 'G' (state 4) occur, '-' (state 0) excluded by
// --free-loss — compact index 0 -> 'A', compact index 1 -> 'G'.
let alignment = SnpAlignment {
sequences: vec![vec![b'A', b'-'], vec![b'-', b'G']],
};
let alignment = alignment(vec![vec![b'A', b'-'], vec![b'-', b'G']]);
let alphabet = compact_alphabet(&alignment, true);
let output = Some(
std::env::temp_dir().join(format!("obikmer_test_iqtree_states_{}", std::process::id())),
std::env::temp_dir().join(format!("obikmer2_test_iqtree_states_{}", std::process::id())),
);
let path = write_iqtree_states_csv(&alphabet, &output);
@@ -491,11 +473,11 @@ mod tests {
sequences[0].push(b'M');
sequences[1].push(b'A');
sequences[2].push(b'-');
let alignment = SnpAlignment { sequences };
let alignment = alignment(sequences);
let labels = vec!["g1".to_string(), "g2".to_string(), "g3".to_string()];
let matrix = [[0.0f64; 16]; 16];
let prefix = std::env::temp_dir().join(format!(
"obikmer_test_iqtree_minfreq_{}",
"obikmer2_test_iqtree_minfreq_{}",
std::process::id()
));
let output = Some(prefix.clone());
+315 -233
View File
@@ -1,70 +1,54 @@
mod args;
mod family_overlap;
mod iqtree;
mod outputs;
mod phyg;
mod phylip;
mod sankoff;
mod tnt;
use std::io::{self, BufWriter, Write};
use std::sync::Arc;
use kodama::{Method, linkage};
use obikidxcache::index_cache::IndexCache;
use obikindex::KmerIndex;
use obikphylo::{
cardinality_transition_probs, composition_transition_probs, pairwise_cost_matrix,
siblings::{
DistanceExt, EntropyBias, RawSnpDistanceOutput, SankoffBundleExt, ShannonEntropyExt,
SiblingExt, SiblingStatsExt, SnpAlignment, SnpAlignmentExt,
},
use obikphylo::siblings::{
EntropyBias, SiblingExt, cardinality_transition_probs, composition_transition_probs,
pairwise_cost_matrix,
};
use obikphylo::{Metrics, neighbor_joining, upgma};
use obisys::{Reporter, Stage};
use speedytree::{DistanceMatrix, Hybrid, NeighborJoiningSolver, to_newick};
use tracing::info;
pub use args::PhyloArgs;
use family_overlap::{apply_min_shared_family_exclusion, write_family_overlap_csv};
use iqtree::write_iqtree;
use outputs::{write_raw_snp_counts_csv, write_raw_snp_distance_csv, write_sibling_hist_csv, write_sibling_stats_csv, write_snp_fasta, upgma_to_newick};
use phyg::write_sankoff_phyg;
use phylip::write_phylip_relaxed;
use sankoff::{write_sankoff_alignment_fasta, write_sankoff_matrix_csv, write_sankoff_params};
use tnt::write_sankoff_tnt;
pub use args::PhyloArgs;
pub fn run(args: PhyloArgs) {
let idx = KmerIndex::open(&args.index).unwrap_or_else(|e| {
let idx = Arc::new(KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
});
}));
let labels: Vec<String> = idx.meta().genomes().unwrap_or_else(|e| {
let labels: Vec<String> = idx
.meta()
.genomes()
.unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
}).iter().map(|g| g.label.clone()).collect();
})
.iter()
.map(|g| g.label.clone())
.collect();
let n = labels.len();
let mut rep = Reporter::new();
// ── Entropy-biased selection (`--entropy`/`--entropy-sd`) ──────────────
// Activates as soon as either is given; the other defaults to 1.0/0.5.
// See `DevDocMD/architecture/siblings.md`, "Entropy-biased selection".
let entropy_bias = if args.entropy.is_some() || args.entropy_sd.is_some() {
Some(EntropyBias {
mu: args.entropy.unwrap_or(1.0),
sigma: args.entropy_sd.unwrap_or(0.5),
})
} else {
None
};
// ── Genome exclusion (`--exclude-genome`) ───────────────────────────────
// Applied by zeroing a `RawSnpDistanceOutput`'s excluded rows/columns
// (`zero_excluded_pairs`) — `base_pair_tally`/`cardinality_tally`
// already skip any pair with zero total counts, so this needs no
// change to `obikindex`'s traversal — and by dropping the excluded
// genome's row from a `SnpAlignment` plus the matching label
// (`drop_excluded`), since an all-`∅` row for an "excluded" genome
// would otherwise still
// reach TNT/PhyG/IQ-TREE as a real (empty) taxon.
// ── Genome exclusion (`--exclude-genome`) ───────────────────────────────────
// Resolved once, up front: `snp_pseudo_alignment` needs it baked into
// sampling itself (see its own docs), and the distance/shared-kmer CSV
// writers below just skip these rows/columns at write time — the
// underlying `cache.distance(...)` computation is unaffected either way.
let exclude_mask: Vec<bool> = {
let mut mask = vec![false; n];
for label in &args.exclude_genome {
@@ -76,259 +60,371 @@ pub fn run(args: PhyloArgs) {
}
}
}
if let Some(threshold) = args.min_shared_family {
apply_min_shared_family_exclusion(&idx, &labels, threshold, &mut mask);
}
mask
};
let zero_excluded_pairs = |result: &mut RawSnpDistanceOutput| {
for i in 0..n {
if !exclude_mask[i] {
continue;
}
for j in 0..n {
result.snp[[i, j]] = 0;
result.snp[[j, i]] = 0;
result.shared[[i, j]] = 0;
result.shared[[j, i]] = 0;
}
}
};
// `snp_pseudo_alignment`'s "variable family" criterion
// (`mask.family_size() >= 2`) is a property of the annex, computed
// over *every* genome in the index — unaffected by `--exclude-genome`.
// So dropping excluded rows alone can leave columns that are variable
// only thanks to an excluded genome now monomorphic among the
// survivors — silently wrong data for TNT/PhyG, and a hard failure
// for IQ-TREE's `+ASC` (verified: excluding 2 taxa on the 20-genome
// benchmark left 116,351 such columns). Re-check variability among the
// *kept* genomes only, after dropping rows, and drop those columns too.
let drop_excluded = |alignment: SnpAlignment| -> (SnpAlignment, Vec<String>) {
let mut sequences: Vec<Vec<u8>> = alignment.sequences.into_iter().enumerate()
.filter(|(i, _)| !exclude_mask[*i])
.map(|(_, seq)| seq)
.collect();
let kept_labels: Vec<String> = labels.iter().enumerate()
.filter(|(i, _)| !exclude_mask[*i])
.map(|(_, l)| l.clone())
.collect();
if exclude_mask.iter().any(|&excluded| excluded) && !sequences.is_empty() {
let n_sites = sequences[0].len();
let keep_col: Vec<bool> = (0..n_sites)
.map(|site| sequences.iter().any(|seq| seq[site] != sequences[0][site]))
.collect();
for seq in &mut sequences {
let mut kept = Vec::with_capacity(seq.len());
for (site, &b) in seq.iter().enumerate() {
if keep_col[site] {
kept.push(b);
}
}
*seq = kept;
}
}
let mut rep = Reporter::new();
(SnpAlignment { sequences }, kept_labels)
};
// Every partition/layer this needs is opened once, up front, and
// handed to `Metrics::distance`/`SiblingExt::build_sibling_annex`
// — see `obikquery`'s own use of `IndexCache` for the same reasoning
// (one open, many in-memory reads).
let cache = IndexCache::new(Arc::clone(&idx), None);
// ── Sibling-count/minorant annex (independent of the distance metric) ──
// Construction (`--sibling-annex`) and stats (`--sibling-stats`) are
// deliberately decoupled: the annex is meant to be (re)built routinely,
// the distribution only occasionally, on demand.
// Meant to be (re)built routinely, ahead of any SNP-family distance
// computation that will later consume it.
if args.sibling_annex {
// Writes into the index directory — hold an exclusive lock for the
// duration so a second, concurrent `--sibling-annex` run on the same
// index can't corrupt these writes (see obisys::DirLock).
// duration so a second, concurrent `--sibling-annex` run on the
// same index can't corrupt these writes (see obisys::DirLock).
let _lock = obisys::DirLock::acquire(&args.index).unwrap_or_else(|e| {
eprintln!("error locking index directory {}: {e}", args.index.display());
std::process::exit(1);
});
info!("building sibling-count/minorant annex");
let t = Stage::start("sibling_annex");
idx.build_sibling_annex().unwrap_or_else(|e| {
cache.build_sibling_annex().unwrap_or_else(|e| {
eprintln!("error building sibling annex: {e}");
std::process::exit(1);
});
rep.push(t.stop());
}
// ── Sibling-count distribution (`--sibling-stats`) ──────────────────────────
if args.sibling_stats {
let t = Stage::start("sibling_stats");
let stats = idx.sibling_annex_stats().unwrap_or_else(|e| {
let stats = cache.sibling_annex_stats().unwrap_or_else(|e| {
eprintln!("error computing sibling-annex stats: {e}");
std::process::exit(1);
});
rep.push(t.stop());
write_sibling_stats_csv(&stats, &labels, &args.output);
let path = args.output.as_ref()
.map(|p| format!("{}_siblings.csv", p.display()))
.unwrap_or_else(|| "siblings.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
// One row per genome (4 columns, family size 1-4: number of
// families of that size for which the genome carries at least one
// member), plus a `global` row — the actual deduplicated
// family-size histogram (`stats.counts`), NOT a sum of the
// per-genome columns (a family shared by several genomes would
// otherwise be counted once per genome it appears in).
writeln!(f, "genome,1,2,3,4").unwrap();
for (label, counts) in labels.iter().zip(stats.per_genome.iter()) {
writeln!(f, "{label},{},{},{},{}", counts[0], counts[1], counts[2], counts[3]).unwrap();
}
writeln!(
f, "global,{},{},{},{}",
stats.counts[0], stats.counts[1], stats.counts[2], stats.counts[3],
).unwrap();
info!("sibling-count distribution → {path}");
}
// ── Family-size histogram (`--sibling-hist`) ────────────────────────────────
if args.sibling_hist {
let t = Stage::start("sibling_hist");
let counts = idx.sibling_family_size_histogram().unwrap_or_else(|e| {
let counts = cache.sibling_family_size_histogram().unwrap_or_else(|e| {
eprintln!("error computing sibling family-size histogram: {e}");
std::process::exit(1);
});
rep.push(t.stop());
write_sibling_hist_csv(&counts, &args.output);
}
if args.raw_snp_distance {
let t = Stage::start("raw_snp_distance");
let mut result = idx.raw_snp_distance().unwrap_or_else(|e| {
eprintln!("error computing raw SNP distance: {e}");
std::process::exit(1);
});
rep.push(t.stop());
zero_excluded_pairs(&mut result);
write_raw_snp_distance_csv(&result, &labels, &args.output);
}
if args.raw_snp_counts {
let t = Stage::start("raw_snp_distance");
let mut result = idx.raw_snp_distance().unwrap_or_else(|e| {
eprintln!("error computing raw SNP distance: {e}");
std::process::exit(1);
});
rep.push(t.stop());
zero_excluded_pairs(&mut result);
write_raw_snp_counts_csv(&result, &labels, &args.output);
}
if args.snp {
let t = Stage::start("snp_pseudo_alignment");
let alignment = idx.snp_pseudo_alignment(args.subsample, entropy_bias).unwrap_or_else(|e| {
eprintln!("error computing SNP pseudo-alignment: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let (alignment, kept_labels) = drop_excluded(alignment);
write_snp_fasta(&alignment, &kept_labels, &args.output);
}
if args.family_overlap {
let t = Stage::start("snp_pseudo_alignment");
let alignment = idx.snp_pseudo_alignment(args.subsample, entropy_bias).unwrap_or_else(|e| {
eprintln!("error computing SNP pseudo-alignment: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let (alignment, kept_labels) = drop_excluded(alignment);
write_family_overlap_csv(&alignment, &kept_labels, &args.output);
}
if args.shannon {
let t = Stage::start("shannon_entropy");
let path = args.output.as_ref()
.map(|p| format!("{}_shannon.csv", p.display()))
.unwrap_or_else(|| "shannon.csv".into());
idx.shannon_entropy_csv(std::path::Path::new(&path), args.subsample, entropy_bias).unwrap_or_else(|e| {
.map(|p| format!("{}_sibling_hist.csv", p.display()))
.unwrap_or_else(|| "sibling_hist.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
writeln!(f, "size,count").unwrap();
for (size, count) in counts.iter().enumerate() {
writeln!(f, "{},{count}", size + 1).unwrap();
}
let total: u64 = counts.iter().sum();
info!(
"family-size histogram → {path} (total {total} famil{})",
if total == 1 { "y" } else { "ies" }
);
}
// ── Family Overlap matrix (`--family-overlap`) ──────────────────────────────
if args.family_overlap {
let t = Stage::start("family_overlap");
let overlap = cache.family_overlap().unwrap_or_else(|e| {
eprintln!("error computing family overlap: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let path = args.output.as_ref()
.map(|p| format!("{}_family_overlap.csv", p.display()))
.unwrap_or_else(|| "family_overlap.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
write!(f, "genome").unwrap();
for label in &labels { write!(f, ",{label}").unwrap(); }
writeln!(f).unwrap();
for (i, label) in labels.iter().enumerate() {
write!(f, "{label}").unwrap();
for j in 0..n { write!(f, ",{}", overlap.get(i, j)).unwrap(); }
writeln!(f).unwrap();
}
info!("family-overlap matrix → {path}");
}
// ── Shannon entropy report (`--shannon`) ────────────────────────────────────
if args.shannon {
let path = args.output.as_ref()
.map(|p| format!("{}_entropy.csv", p.display()))
.unwrap_or_else(|| "entropy.csv".into());
info!("computing per-family Shannon entropy");
let t = Stage::start("shannon_entropy");
cache.shannon_entropy_csv(std::path::Path::new(&path)).unwrap_or_else(|e| {
eprintln!("error computing Shannon entropy: {e}");
std::process::exit(1);
});
rep.push(t.stop());
info!("per-family Shannon entropy → {path}");
info!("entropy report → {path}");
}
if args.sankoff || args.tnt || args.phyg || args.iqtree {
// One shared, possibly-subsampled/entropy-biased selection, one
// pass to derive `included[i,j]`, one more fused pass for
// base_pair_tally + cardinality_tally + the pseudo-alignment — see
// `DevDocMD/architecture/siblings.md`, "`--free-loss`/`--tnt`
// pipeline" and "Wired into `pack`...".
let t = Stage::start("raw_snp_distance");
let bundle = idx.sankoff_bundle(args.subsample, entropy_bias, args.sankoff_ratio_ceiling, &exclude_mask).unwrap_or_else(|e| {
eprintln!("error computing sankoff inputs: {e}");
// ── `--min-shared-family` auto-exclusion ────────────────────────────────────
// Layered on top of `--exclude-genome`, not instead of it — a separate
// mask (not folded into `exclude_mask` itself) since it must reach
// `--pseudo-alignment`/`--sankoff`/`snp-*` `--distance` only, never the
// whole-index metrics' `kept`/`--shared-kmers` output (see
// `args::PhyloArgs::min_shared_family`'s own docs on why).
let snp_exclude_mask: Vec<bool> = match args.min_shared_family {
Some(threshold) => {
let overlap = cache.family_overlap().unwrap_or_else(|e| {
eprintln!("error computing family overlap: {e}");
std::process::exit(1);
});
let mut mask = exclude_mask.clone();
for (g, excluded) in mask.iter_mut().enumerate() {
if *excluded {
continue;
}
let mean = overlap.mean_row(g);
if mean < threshold {
info!(
"auto-excluding {} (--min-shared-family: mean shared-family count {mean:.1} < {threshold})",
labels[g]
);
*excluded = true;
}
}
mask
}
None => exclude_mask.clone(),
};
// Shared by `--pseudo-alignment` and `--sankoff` — same activation rule:
// either flag given activates entropy-biased sampling, the other
// defaults to 1.0/0.5.
let entropy_bias = if args.entropy.is_some() || args.entropy_sd.is_some() {
Some(EntropyBias {
mu: args.entropy.unwrap_or(1.0),
sigma: args.entropy_sd.unwrap_or(0.5),
})
} else {
None
};
// ── SNP pseudo-alignment (`--pseudo-alignment`) ─────────────────────────────
if args.pseudo_alignment {
let Some(subsample_n) = args.subsample else {
eprintln!("error: --pseudo-alignment requires --subsample <N>");
std::process::exit(1);
};
info!("sampling SNP pseudo-alignment (target {subsample_n} site(s))");
let t = Stage::start("pseudo_alignment");
let alignment = cache
.snp_pseudo_alignment(subsample_n, args.free_loss, args.no_ambiguity, &snp_exclude_mask, entropy_bias)
.unwrap_or_else(|e| {
eprintln!("error building pseudo-alignment: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let (base_tally, card_tally) = (bundle.base_pair_tally, bundle.cardinality_tally);
let p_card = cardinality_transition_probs(&card_tally);
let p_comp = composition_transition_probs(&base_tally);
let path = args.output.as_ref()
.map(|p| format!("{}_alignment.fasta", p.display()))
.unwrap_or_else(|| "alignment.fasta".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
let n_sites = alignment.sequences.first().map_or(0, Vec::len);
for (&g, seq) in alignment.genome_indices.iter().zip(alignment.sequences.iter()) {
obifastwrite::write_plain_record(seq, &labels[g], &mut f).unwrap();
}
info!("pseudo-alignment ({n_sites} site(s), {} genome(s)) → {path}", alignment.genome_indices.len());
}
// ── Sankoff cost-matrix calibration (`--sankoff`, `--tnt`, `--phyg`, `--iqtree`) ──
if args.sankoff || args.tnt || args.phyg || args.iqtree {
let Some(subsample_n) = args.subsample else {
eprintln!("error: --sankoff requires --subsample <N>");
std::process::exit(1);
};
info!("sampling Sankoff calibration bundle (target {subsample_n} site(s))");
let t = Stage::start("sankoff_bundle");
let bundle = cache
.sankoff_bundle(
subsample_n,
args.free_loss,
args.no_ambiguity,
&snp_exclude_mask,
entropy_bias,
args.sankoff_ratio_ceiling,
)
.unwrap_or_else(|e| {
eprintln!("error computing Sankoff calibration bundle: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let p_card = cardinality_transition_probs(&bundle.cardinality_tally);
let p_comp = composition_transition_probs(&bundle.base_pair_tally);
let matrix = pairwise_cost_matrix(&p_card, &p_comp, args.free_loss);
write_sankoff_matrix_csv(&matrix, &args.output);
write_sankoff_params(&card_tally, &p_card, &base_tally, &p_comp, args.sankoff_ratio_ceiling, &args.output);
let (alignment, kept_labels) = drop_excluded(bundle.alignment);
write_sankoff_alignment_fasta(&alignment, &kept_labels, &args.output, args.free_loss);
write_sankoff_matrix_csv(&matrix, &args.output);
write_sankoff_params(
&bundle.cardinality_tally,
&p_card,
&bundle.base_pair_tally,
&p_comp,
args.sankoff_ratio_ceiling,
&args.output,
);
write_sankoff_alignment_fasta(&bundle.alignment, &labels, &args.output, args.free_loss);
if args.tnt {
write_sankoff_tnt(&matrix, &alignment, &kept_labels, &args.output, args.sankoff_cost_scale, args.free_loss);
write_sankoff_tnt(
&matrix,
&bundle.alignment,
&labels,
&args.output,
args.sankoff_cost_scale,
args.free_loss,
);
}
if args.phyg {
write_sankoff_phyg(&matrix, &args.output, args.sankoff_cost_scale);
}
if args.iqtree {
write_iqtree(&matrix, &alignment, &kept_labels, &args.output, args.free_loss, args.iqtree_min_freq);
}
}
// `--sibling-annex`/`--sibling-stats`/`--raw-snp-distance`/`--snp`/
// `--sankoff`/`--tnt` are their own operation, not a modifier on top of
// a distance-metric computation — a metric was never requested by
// asking for any of them, so there is nothing for the rest of this
// function to compute. Not a historical accident to keep: stop here
// rather than always also running a Jaccard (or whichever `--metric`
// defaults to) pass and printing an unrequested matrix.
if args.sibling_annex
|| args.sibling_stats
|| args.sibling_hist
|| args.raw_snp_distance
|| args.raw_snp_counts
|| args.snp
|| args.family_overlap
|| args.shannon
|| args.sankoff
|| args.tnt
|| args.phyg
|| args.iqtree
{
rep.print();
return;
}
info!(
"computing {:?} distances for {} genome(s)",
args.metric, n
write_iqtree(
&matrix,
&bundle.alignment,
&labels,
&args.output,
args.free_loss,
args.iqtree_min_freq,
);
}
}
// ── Distance computation: classic whole-index metric vs. `snp-*` ───────────
// Two genuinely different code paths behind one `--distance` value — see
// `args::DistanceArg`'s own docs.
let (matrix, shared_kmers) = match args.distance.as_classic() {
Some(metric) => {
info!("computing {metric:?} distances for {n} genome(s)");
let need_shared = args.shared_kmers || args.nj || args.upgma;
let t = Stage::start("distance");
let result = idx
.distance(args.metric.into(), need_shared, args.presence_threshold)
let result = cache
.distance(metric, need_shared, args.presence_threshold)
.unwrap_or_else(|e| {
eprintln!("error computing distances: {e}");
std::process::exit(1);
});
rep.push(t.stop());
(result.matrix, result.shared_kmers)
}
None => {
if args.shared_kmers {
eprintln!("error: --shared-kmers has no meaning for a snp-* --distance value");
std::process::exit(1);
}
let kind = args.distance.as_snp().expect("DistanceArg is always classic or snp");
info!(
"computing {kind:?} SNP distance for {n} genome(s){}",
match args.subsample {
Some(n) => format!(" (subsampled, target {n} site(s))"),
None => " (exhaustive)".into(),
}
);
let t = Stage::start("snp_distance");
let matrix = cache
.snp_distance(
kind,
args.subsample,
args.free_loss,
args.no_ambiguity,
&snp_exclude_mask,
entropy_bias,
args.gamma_shape,
)
.unwrap_or_else(|e| {
eprintln!("error computing SNP distance: {e}");
std::process::exit(1);
});
rep.push(t.stop());
(matrix, None)
}
};
// ── Distance matrix → CSV ─────────────────────────────────────────────────
let write_dist_csv = |w: &mut dyn Write| {
// Rows/columns kept in every matrix output below — the computation
// above runs over every genome regardless; only the writers skip
// excluded ones.
let kept: Vec<usize> = (0..n).filter(|&i| !exclude_mask[i]).collect();
// ── Distance matrix → relaxed PHYLIP (default) or CSV (`--csv`) ────────────
let write_dist = |w: &mut dyn Write| {
if args.csv {
write!(w, "genome").unwrap();
for g in &labels { write!(w, ",{g}").unwrap(); }
for &j in &kept { write!(w, ",{}", labels[j]).unwrap(); }
writeln!(w).unwrap();
for (i, g) in labels.iter().enumerate() {
write!(w, "{g}").unwrap();
for j in 0..n {
write!(w, ",{:.6}", result.matrix[[i, j]]).unwrap();
for &i in &kept {
write!(w, "{}", labels[i]).unwrap();
for &j in &kept {
write!(w, ",{:.6}", matrix[[i, j]]).unwrap();
}
writeln!(w).unwrap();
}
} else {
write_phylip_relaxed(w, &labels, &kept, &matrix);
}
};
match &args.output {
Some(prefix) => {
let path = format!("{}_dist.csv", prefix.display());
let suffix = if args.csv { "_dist.csv" } else { "_dist.phy" };
let path = format!("{}{suffix}", prefix.display());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
write_dist_csv(&mut f);
write_dist(&mut f);
info!("distance matrix → {path}");
}
None => {
let stdout = io::stdout();
let mut out = BufWriter::new(stdout.lock());
write_dist_csv(&mut out);
write_dist(&mut out);
}
}
// ── Shared-kmer matrix → CSV ──────────────────────────────────────────────
if args.shared_kmers {
if let Some(shared) = &result.shared_kmers {
if let Some(shared) = &shared_kmers {
let path = args.output.as_ref()
.map(|p| format!("{}_shared.csv", p.display()))
.unwrap_or_else(|| "shared.csv".into());
@@ -337,31 +433,24 @@ pub fn run(args: PhyloArgs) {
std::process::exit(1);
}));
write!(f, "genome").unwrap();
for g in &labels { write!(f, ",{g}").unwrap(); }
for &j in &kept { write!(f, ",{}", labels[j]).unwrap(); }
writeln!(f).unwrap();
for (i, g) in labels.iter().enumerate() {
write!(f, "{g}").unwrap();
for j in 0..n { write!(f, ",{}", shared[[i, j]]).unwrap(); }
for &i in &kept {
write!(f, "{}", labels[i]).unwrap();
for &j in &kept { write!(f, ",{}", shared[[i, j]]).unwrap(); }
writeln!(f).unwrap();
}
info!("shared-kmer matrix → {path}");
}
}
// ── NJ tree via speedytree ────────────────────────────────────────────────
// ── NJ tree ────────────────────────────────────────────────────────────────
if args.nj {
let rows: Vec<Vec<f64>> = (0..n)
.map(|i| (0..n).map(|j| result.matrix[[i, j]]).collect())
.collect();
let dm = DistanceMatrix::build(rows, labels.clone()).unwrap_or_else(|e| {
eprintln!("error building distance matrix for NJ: {e}");
std::process::exit(1);
});
let tree = NeighborJoiningSolver::<Hybrid>::default(dm).solve().unwrap_or_else(|e| {
let tree = neighbor_joining(&matrix, &labels).unwrap_or_else(|e| {
eprintln!("error computing NJ tree: {e}");
std::process::exit(1);
});
let newick = to_newick(&tree);
let newick = tree.to_newick();
let path = args.output.as_ref()
.map(|p| format!("{}_nj.nwk", p.display()))
.unwrap_or_else(|| "nj.nwk".into());
@@ -372,16 +461,9 @@ pub fn run(args: PhyloArgs) {
info!("NJ tree → {path}");
}
// ── UPGMA tree via kodama ─────────────────────────────────────────────────
// ── UPGMA tree ───────────────────────────────────────────────────────────────
if args.upgma {
let mut condensed: Vec<f64> = Vec::with_capacity(n * (n - 1) / 2);
for i in 0..n {
for j in (i + 1)..n {
condensed.push(result.matrix[[i, j]]);
}
}
let dendro = linkage(&mut condensed, n, Method::Average);
let newick = upgma_to_newick(&dendro, &labels);
let newick = upgma(&matrix, &labels).to_newick();
let path = args.output.as_ref()
.map(|p| format!("{}_upgma.nwk", p.display()))
.unwrap_or_else(|| "upgma.nwk".into());
-187
View File
@@ -1,187 +0,0 @@
use std::io::{BufWriter, Write};
use std::path::PathBuf;
use obifastwrite::{JsonVal, write_record};
use obikphylo::siblings::{RawSnpDistanceOutput, SiblingAnnexStats, SnpAlignment};
use tracing::info;
// ── Family-size distribution → CSV ──────────────────────────────────────────
//
// Each row is a family (the up-to-4 k-mers sharing flanks, differing only at
// the centre), counted once — at its minorant — regardless of how many of
// its members are observed. Family size 1..4 (not "sibling count" 0..3):
// see `DevDocMD/theory/evolutionary_distances.md`, "Definitions".
pub(super) fn write_sibling_stats_csv(stats: &SiblingAnnexStats, labels: &[String], output: &Option<PathBuf>) {
// One row per genome (4 columns, family size 1-4: number of families of
// that size for which the genome carries at least one member), plus a
// `global` row — the actual deduplicated family-size histogram
// (`stats.counts`), NOT a sum of the per-genome columns (a family shared
// by several genomes would otherwise be counted once per genome it
// appears in, inflating the total beyond the real family count).
let path = output.as_ref()
.map(|p| format!("{}_siblings.csv", p.display()))
.unwrap_or_else(|| "siblings.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
writeln!(f, "genome,1,2,3,4").unwrap();
for (label, counts) in labels.iter().zip(stats.per_genome.iter()) {
writeln!(f, "{label},{},{},{},{}", counts[0], counts[1], counts[2], counts[3]).unwrap();
}
writeln!(
f, "global,{},{},{},{}",
stats.counts[0], stats.counts[1], stats.counts[2], stats.counts[3],
).unwrap();
let total: u64 = stats.counts.iter().sum();
info!("family-size distribution → {path} (total {total} famil{})",
if total == 1 { "y" } else { "ies" });
}
// ── Raw single-copy SNP distance → CSV ──────────────────────────────────────
//
// p_hat[i,j] = snp[i,j] / (snp[i,j] + shared[i,j]) over loci single-copy in
// both i and j — see `RawSnpDistanceOutput` / `KmerIndex::raw_snp_distance`.
// A single file: the distance matrix, with an eligible-loci count alongside
// each value so a 0/0 pair (no eligible locus at all) is distinguishable
// from a genuinely identical pair.
pub(super) fn write_raw_snp_distance_csv(result: &RawSnpDistanceOutput, labels: &[String], output: &Option<PathBuf>) {
let path = output.as_ref()
.map(|p| format!("{}_rawsnp.csv", p.display()))
.unwrap_or_else(|| "rawsnp.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
let n = labels.len();
write!(f, "genome").unwrap();
for g in labels { write!(f, ",{g}").unwrap(); }
writeln!(f).unwrap();
for (i, g) in labels.iter().enumerate() {
write!(f, "{g}").unwrap();
for j in 0..n {
let snp = result.snp[[i, j]];
let shared = result.shared[[i, j]];
let eligible = snp + shared;
if eligible == 0 {
write!(f, ",NA").unwrap();
} else {
write!(f, ",{:.6}", snp as f64 / eligible as f64).unwrap();
}
}
writeln!(f).unwrap();
}
info!("raw single-copy SNP distance matrix → {path}");
}
// ── Global family-size histogram (annex-only, no per-genome pass) → CSV ────
pub(super) fn write_sibling_hist_csv(counts: &[u64; 4], output: &Option<PathBuf>) {
let path = output.as_ref()
.map(|p| format!("{}_sibling_hist.csv", p.display()))
.unwrap_or_else(|| "sibling_hist.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
writeln!(f, "size,count").unwrap();
for (size, count) in counts.iter().enumerate() {
writeln!(f, "{},{count}", size + 1).unwrap();
}
let total: u64 = counts.iter().sum();
info!("family-size histogram → {path} (total {total} famil{})",
if total == 1 { "y" } else { "ies" });
}
// ── Raw single-copy SNP distance → per-pair diagnostic counts ──────────────
//
// A pair table (one row per unordered genome pair), not a matrix: the ratio
// alone can't distinguish "identical across every eligible locus" from
// "almost no eligible locus at all" — both can read `0.0`/`NA` in
// `--raw-snp-distance`'s output. Distinguishing them matters most exactly
// where it's easy to miss: genome pairs near the edge of what
// central-position families can resolve at all (deep cross-lineage splits,
// see `DevDocMD/theory/evolutionary_distances.md`, "Run 3" and the later
// IQ-TREE/Mash comparison — a `ratio=0.0` backed by 2 eligible loci is not
// the same claim as one backed by 2000).
pub(super) fn write_raw_snp_counts_csv(result: &RawSnpDistanceOutput, labels: &[String], output: &Option<PathBuf>) {
let path = output.as_ref()
.map(|p| format!("{}_rawsnp_counts.csv", p.display()))
.unwrap_or_else(|| "rawsnp_counts.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
let n = labels.len();
writeln!(f, "genome_a,genome_b,n_snp,n_shared,n_eligible,ratio").unwrap();
for i in 0..n {
for j in (i + 1)..n {
let snp = result.snp[[i, j]];
let shared = result.shared[[i, j]];
let eligible = snp + shared;
write!(f, "{},{},{snp},{shared},{eligible}", labels[i], labels[j]).unwrap();
if eligible == 0 {
writeln!(f, ",NA").unwrap();
} else {
writeln!(f, ",{:.6}", snp as f64 / eligible as f64).unwrap();
}
}
}
info!("raw single-copy SNP distance counts (diagnostic) → {path}");
}
// ── SNP-only pseudo-alignment → FASTA ───────────────────────────────────────
//
// One record per genome, IUPAC-coded, no flanking sequence — see
// `SnpAlignment` / `KmerIndex::snp_pseudo_alignment`. Uses the project's
// existing FASTA writer (`obifastwrite::write_record`) rather than
// hand-rolling one.
pub(super) fn write_snp_fasta(alignment: &SnpAlignment, labels: &[String], output: &Option<PathBuf>) {
let path = output.as_ref()
.map(|p| format!("{}_snp.fasta", p.display()))
.unwrap_or_else(|| "snp.fasta".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
let n_sites = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
for (label, seq) in labels.iter().zip(alignment.sequences.iter()) {
write_record(seq, label, &[("n_sites", JsonVal::Num(n_sites as u64))], &mut f).unwrap_or_else(|e| {
eprintln!("error writing {path}: {e}");
std::process::exit(1);
});
}
info!("SNP pseudo-alignment → {path} ({n_sites} site{})",
if n_sites == 1 { "" } else { "s" });
}
// ── UPGMA Newick from kodama dendrogram ───────────────────────────────────────
pub(super) fn upgma_to_newick(dendro: &kodama::Dendrogram<f64>, names: &[String]) -> String {
let n = names.len();
// node_labels[i]: Newick subtree string for node i (leaves 0..n, internals n..)
let mut labels: Vec<String> = names.to_vec();
// height of each node: leaves = 0, internal = dissimilarity/2
let mut heights: Vec<f64> = vec![0.0; 2 * n - 1];
for (k, step) in dendro.steps().iter().enumerate() {
let new_node = n + k;
let h = step.dissimilarity / 2.0;
heights[new_node] = h;
let c1 = step.cluster1;
let c2 = step.cluster2;
let bl1 = (h - heights[c1]).max(0.0);
let bl2 = (h - heights[c2]).max(0.0);
labels.push(format!(
"({label1}:{bl1:.6},{label2}:{bl2:.6})",
label1 = labels[c1],
label2 = labels[c2],
));
}
format!("{};", labels.last().unwrap())
}
+4 -5
View File
@@ -13,11 +13,10 @@ use super::sankoff::{STATE_SYMBOL, scaled_metric_matrix};
// as-is via `prefasta:`. PhyG auto-adds its own indel/gap state as an
// (n+1)-th row/column of the tcm — inert here since the alignment already
// encodes absence as an ordinary state (`0`), never as `-` (see
// `sankoff::write_sankoff_alignment_fasta`'s own comment on why, and the
// RAxML-era bug that motivated it). The gap row/column below reuses
// `matrix[i][0]`/`matrix[0][j]` (cost to/from `∅`) as the closest
// principled value for a state that, in practice, is never actually
// triggered.
// `sankoff::write_sankoff_alignment_fasta`'s own comment on why). The gap
// row/column below reuses `matrix[i][0]`/`matrix[0][j]` (cost to/from `∅`)
// as the closest principled value for a state that, in practice, is never
// actually triggered.
pub(super) fn write_sankoff_phyg(matrix: &[[f64; 16]; 16], output: &Option<PathBuf>, cost_scale: f64) {
let scaled_matrix = scaled_metric_matrix(matrix, cost_scale);
+42 -43
View File
@@ -1,3 +1,9 @@
//! Output writers for `--sankoff` — no calibration logic here, just
//! formatting: `obikphylo::siblings::SiblingExt::sankoff_bundle` and the
//! `cardinality_transition_probs`/`composition_transition_probs`/
//! `pairwise_cost_matrix` calibration functions do all the actual work in
//! `mod.rs`, this module only serialises their results.
use std::io::{BufWriter, Write};
use std::path::PathBuf;
@@ -7,19 +13,22 @@ use tracing::info;
// ── Sankoff pseudo-alignment → FASTA ────────────────────────────────────────
//
// Same data as `--snp`'s pseudo-alignment (`SnpAlignment`/
// Same data as `--pseudo-alignment`'s output (`SnpAlignment`/
// `snp_pseudo_alignment`), re-coded so its symbols match the accompanying
// `--sankoff-matrix` output exactly: `0` for the empty/absent state instead
// `--sankoff` matrix output exactly: `0` for the empty/absent state instead
// of `-`, which TNT/PhyG would otherwise read as their own gap character
// rather than our "family absent" state. Unless `free_loss` (`--free-loss`)
// is set, in which case `∅` is recoded to `?` instead — TNT/PhyG's own
// missing-data symbol, deliberately *not* `-` (still gap/indel semantics in
// both tools) — so non-detection costs nothing rather than being scored as
// an ordinary, calibrated state transition. See
// `DevDocMD/theory/evolutionary_distances.md`, "Locus dropout under incomplete
// coverage".
// an ordinary, calibrated state transition.
pub(super) fn write_sankoff_alignment_fasta(alignment: &SnpAlignment, labels: &[String], output: &Option<PathBuf>, free_loss: bool) {
pub(super) fn write_sankoff_alignment_fasta(
alignment: &SnpAlignment,
labels: &[String],
output: &Option<PathBuf>,
free_loss: bool,
) {
let path = output.as_ref()
.map(|p| format!("{}_sankoff.fasta", p.display()))
.unwrap_or_else(|| "sankoff.fasta".into());
@@ -28,33 +37,31 @@ pub(super) fn write_sankoff_alignment_fasta(alignment: &SnpAlignment, labels: &[
std::process::exit(1);
}));
let absent_symbol = if free_loss { b'?' } else { b'0' };
let n_sites = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
for (label, seq) in labels.iter().zip(alignment.sequences.iter()) {
let n_sites = alignment.sequences.first().map_or(0, Vec::len);
for (&g, seq) in alignment.genome_indices.iter().zip(alignment.sequences.iter()) {
let recoded: Vec<u8> = seq.iter().map(|&b| if b == b'-' { absent_symbol } else { b }).collect();
write_record(&recoded, label, &[("n_sites", JsonVal::Num(n_sites as u64))], &mut f).unwrap_or_else(|e| {
write_record(&recoded, &labels[g], &[("n_sites", JsonVal::Num(n_sites as u64))], &mut f)
.unwrap_or_else(|e| {
eprintln!("error writing {path}: {e}");
std::process::exit(1);
});
}
info!("Sankoff pseudo-alignment → {path} ({n_sites} site{})",
if n_sites == 1 { "" } else { "s" });
info!("Sankoff pseudo-alignment ({n_sites} site(s)) → {path}");
}
// ── Sankoff cost matrix → CSV ────────────────────────────────────────────────
//
// 16 states indexed by bitmask (bit 0=A, 1=C, 2=G, 3=T; state 0 is `∅`),
// matching the convention already used for `--snp`'s IUPAC-coded output and
// for the external TNT/PhyG scripts this feeds. Calibration report (p_hat,
// its variance, how many pairs/loci went into it, the resulting c_ctx) goes
// to the log, not the CSV, since it's a run-level fact, not per-cell data.
// matching the convention used for `--pseudo-alignment`'s IUPAC-coded output
// and for the external TNT/PhyG scripts this feeds.
// IUPAC ambiguity code per state (same mapping as `siblings::iupac_code`,
// already used for `--snp`'s pseudo-alignment — a biologist reads "R" as
// "A or G" without needing this file's convention explained), with `0`
// standing in for the empty state (`-` would collide with TNT/PhyG's own
// gap/range syntax). Bit order: 0=A, 1=C, 2=G, 3=T. This project's
// canonical alphabet — `tnt::write_sankoff_tnt` recodes it to TNT's own
// default alphabet at the adapter boundary, rather than using it here.
/// IUPAC ambiguity code per state (same mapping
/// `obikphylo::siblings::algorithms::masking::iupac_code` uses internally
/// for `--pseudo-alignment`), with `0` standing in for the empty state (`-`
/// would collide with TNT/PhyG's own gap/range syntax). Bit order: 0=A,
/// 1=C, 2=G, 3=T. This project's canonical alphabet for every Sankoff
/// export (`--tnt`/`--phyg`/`--iqtree` each recode it to their own alphabet
/// at their own adapter boundary, rather than using it directly).
pub(super) const STATE_SYMBOL: [char; 16] = [
'0', 'A', 'C', 'M', 'G', 'R', 'S', 'V', 'T', 'W', 'Y', 'H', 'K', 'D', 'B', 'N',
];
@@ -70,10 +77,7 @@ pub(super) fn state_index_table() -> [u8; 128] {
table
}
pub(super) fn write_sankoff_matrix_csv(
matrix: &[[f64; 16]; 16],
output: &Option<PathBuf>,
) {
pub(super) fn write_sankoff_matrix_csv(matrix: &[[f64; 16]; 16], output: &Option<PathBuf>) {
let path = output.as_ref()
.map(|p| format!("{}_sankoff_matrix.csv", p.display()))
.unwrap_or_else(|| "sankoff_matrix.csv".into());
@@ -95,14 +99,10 @@ pub(super) fn write_sankoff_matrix_csv(
// ── Sankoff calibration parameters → YAML report ────────────────────────────
//
// Everything `--sankoff` estimates from real data, in one durable,
// machine-readable file: `p_hat` and its variance (with how many pairs/loci
// went into it), the derived `c_ctx`, and the base-pair substitution tally
// (raw counts, not just the derived costs) — kept for the same reason raw
// counts are kept anywhere else in this project: costs are a modelling
// machine-readable file: the cardinality and base-pair transition tallies
// (raw counts, not just the derived probabilities) — costs are a modelling
// choice built *from* the counts, and reproducing/re-deriving them later
// needs the counts, not just their current derived value. Structured (YAML,
// not an ad hoc key=value text file) so R/Python/etc. can load it directly
// rather than re-parsing free text.
// needs the counts, not just their current derived value.
#[derive(serde::Serialize)]
struct CardinalityTransition {
@@ -183,16 +183,15 @@ pub(super) fn write_sankoff_params(
/// closure* of the result (Floyd-Warshall over the 16 states again, on the
/// now-integer values).
///
/// Unlike this project's earlier cost-matrix construction (a graph closed
/// by shortest path, guaranteeing a metric by construction), `matrix` here
/// comes from `obikindex::pairwise_cost_matrix`'s row-normalise-then-`-ln`
/// composition, which gives no such guarantee — so this closure isn't only
/// needed to correct integer-rounding artifacts (two real costs of `1.734`
/// each round to `173`, summing to `346`, while their own real sum `3.468`
/// rounds to `347` — TNT then reports "triangle inequality violated ...
/// Fixed" and silently substitutes its own corrected value), it may also
/// be the only thing making the *real-valued* matrix a metric in the first
/// place. Re-closing after rounding makes both corrections explicit and
/// `pairwise_cost_matrix`'s row-normalise-then-`-ln` construction gives no
/// guarantee of being a metric (unlike a cost graph closed by shortest path
/// by construction) — so this closure isn't only needed to correct
/// integer-rounding artifacts (two real costs of `1.734` each round to
/// `173`, summing to `346`, while their own real sum `3.468` rounds to
/// `347` — TNT then reports "triangle inequality violated ... Fixed" and
/// silently substitutes its own corrected value), it may also be the only
/// thing making the *real-valued* matrix a metric in the first place.
/// Re-closing after rounding makes both corrections explicit and
/// reproducible here instead, rather than left implicit and
/// tool-version-dependent.
pub(super) fn scaled_metric_matrix(matrix: &[[f64; 16]; 16], cost_scale: f64) -> [[i64; 16]; 16] {
+4 -3
View File
@@ -40,11 +40,12 @@ pub(super) fn write_sankoff_tnt(
}));
// IUPAC-ish symbol -> bitmask, to translate the alignment (which uses
// `STATE_SYMBOL`, `-` already normalised to `0` by `snp_pseudo_alignment`
// `STATE_SYMBOL`, `-` already normalised to `0` by `sankoff_bundle`
// callers) into TNT's alphabet without re-deriving state indices.
let iupac_to_state = state_index_table();
let n_sites = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
let kept_labels: Vec<&String> = alignment.genome_indices.iter().map(|&g| &labels[g]).collect();
writeln!(f, "xread").unwrap();
writeln!(f, "mxram 16000;").unwrap();
writeln!(f, "taxname =;").unwrap();
@@ -55,8 +56,8 @@ pub(super) fn write_sankoff_tnt(
"'obikmer central-position SNP families, calibrated Sankoff 16-state encoding'"
)
.unwrap();
writeln!(f, "{n_sites} {}", labels.len()).unwrap();
for (label, seq) in labels.iter().zip(alignment.sequences.iter()) {
writeln!(f, "{n_sites} {}", kept_labels.len()).unwrap();
for (label, seq) in kept_labels.iter().zip(alignment.sequences.iter()) {
write!(f, "{label} ").unwrap();
for &b in seq {
if free_loss && b == b'-' {
+20 -27
View File
@@ -1,10 +1,11 @@
use clap::Args;
use obikfilter::{GenomeSelector, GroupFilterParams, GroupQuorumFilter};
use obikindex::IndexMeta;
use obikfilter::{GroupFilterParams, KmerFilter, MetaPred};
/// CLI args for ingroup/outgroup filtering — embeddable in any command via `#[command(flatten)]`.
/// Ingroup/outgroup metadata-predicate quorum filtering — embeddable in any
/// command via `#[command(flatten)]` (`filter`, `dump`).
#[derive(Args)]
pub struct FilterArgs {
pub struct GroupFilterArgs {
/// Ingroup predicate (repeatable; AND). Forms: `key=v1|v2`, `key!=v`, `key~path`, `key!~path`, `*`/`all`
#[arg(long, value_name = "PRED")]
pub ingroup: Vec<String>,
@@ -23,12 +24,12 @@ pub struct FilterArgs {
#[arg(long, allow_hyphen_values = true)]
pub max_count: Option<isize>,
/// Minimum fraction of ingroup genomes containing the k-mer [0.01.0]
/// Minimum fraction of ingroup genomes containing the k-mer [0.0-1.0]
/// (default 1.0 when --ingroup is set, 0.0 otherwise)
#[arg(long)]
pub min_frac: Option<f64>,
/// Maximum fraction of ingroup genomes containing the k-mer [0.01.0]
/// Maximum fraction of ingroup genomes containing the k-mer [0.0-1.0]
#[arg(long)]
pub max_frac: Option<f64>,
@@ -43,11 +44,11 @@ pub struct FilterArgs {
#[arg(long, allow_hyphen_values = true)]
pub max_outgroup_count: Option<isize>,
/// Minimum fraction of outgroup genomes containing the k-mer [0.01.0]
/// Minimum fraction of outgroup genomes containing the k-mer [0.0-1.0]
#[arg(long)]
pub min_outgroup_frac: Option<f64>,
/// Maximum fraction of outgroup genomes containing the k-mer [0.01.0]
/// Maximum fraction of outgroup genomes containing the k-mer [0.0-1.0]
#[arg(long)]
pub max_outgroup_frac: Option<f64>,
@@ -56,24 +57,16 @@ pub struct FilterArgs {
pub presence_threshold: u32,
}
impl FilterArgs {
/// Parse predicates and build a filter list ready to pass to `iter_partition_kmers`.
pub fn build_filters(&self, meta: &IndexMeta) -> Vec<Box<dyn KmerFilter>> {
let ingroup_preds: Vec<MetaPred> = self.ingroup.iter()
.map(|s| MetaPred::parse(s).unwrap_or_else(|e| {
eprintln!("error in --ingroup: {e}");
impl GroupFilterArgs {
/// Parse `--ingroup`/`--outgroup` and build the quorum filter. Exits on error.
pub fn build_filter(&self, meta: &IndexMeta) -> GroupQuorumFilter {
let selector = GenomeSelector::parse(&self.ingroup, &self.outgroup).unwrap_or_else(|e| {
eprintln!("error in --ingroup/--outgroup: {e}");
std::process::exit(1);
}))
.collect();
let outgroup_preds: Vec<MetaPred> = self.outgroup.iter()
.map(|s| MetaPred::parse(s).unwrap_or_else(|e| {
eprintln!("error in --outgroup: {e}");
std::process::exit(1);
}))
.collect();
let filter = meta.build_group_filter(
&ingroup_preds,
&outgroup_preds,
});
selector
.build_group_filter(
meta,
GroupFilterParams {
threshold: self.presence_threshold,
min_count: self.min_count,
@@ -85,10 +78,10 @@ impl FilterArgs {
min_outgroup_frac: self.min_outgroup_frac,
max_outgroup_frac: self.max_outgroup_frac,
},
).unwrap_or_else(|e| {
)
.unwrap_or_else(|e| {
eprintln!("error in filter parameters: {e}");
std::process::exit(1);
});
vec![Box::new(filter)]
})
}
}
-74
View File
@@ -1,74 +0,0 @@
use std::collections::HashMap;
use obikindex::KmerDesc;
use obikseq::CanonicalKmer;
use obiread::record::SeqRecord;
use obiskbuilder::SuperKmerIter;
/// A batch of query sequences, with k-mers deduplicated directly (not just at
/// the superkmer level) and pre-split by partition.
///
/// Superkmer *construction* (`SuperKmerIter`) is still required — it's the
/// mechanism that computes minimizers and partition routing — but the dedup
/// key is the canonical k-mer, not the superkmer: two different superkmers
/// that happen to share a k-mer (read overlaps, repeats, a SNP splitting an
/// otherwise-identical run) are deduplicated too, not just identical whole
/// superkmers. This also means each unique k-mer triggers at most one MPHF
/// lookup, not one per occurrence.
pub struct QueryBatch {
/// Sequence ids in batch order.
pub ids: Vec<String>,
/// Raw sequence bytes (for output), in batch order.
pub seqs: Vec<Vec<u8>>,
/// Total kmer count per sequence (used for `--detail` coverage allocation).
pub n_kmers: Vec<u32>,
/// Deduplicated k-mer occurrences, one map per partition.
pub by_partition: Vec<HashMap<CanonicalKmer, Vec<KmerDesc>>>,
}
impl QueryBatch {
/// Build a batch from a vec of parsed sequence records, deduplicating
/// k-mers and routing them to partitions in the same pass.
pub fn from_records(
records: Vec<SeqRecord>,
k: usize,
level_max: usize,
theta: f64,
n_partitions: usize,
) -> Self {
let mut ids = Vec::with_capacity(records.len());
let mut seqs = Vec::with_capacity(records.len());
let mut n_kmers = Vec::with_capacity(records.len());
let mask = (n_partitions as u64) - 1;
let mut by_partition: Vec<HashMap<CanonicalKmer, Vec<KmerDesc>>> =
(0..n_partitions).map(|_| HashMap::new()).collect();
for (seq_idx, record) in records.into_iter().enumerate() {
let mut kmer_offset = 0u32;
for rsk in SuperKmerIter::new(&record.normalized, k, level_max, theta) {
let part_idx = (rsk.minimizer().seq_hash() & mask) as usize;
let map = &mut by_partition[part_idx];
for (j, kmer) in rsk.superkmer().iter_canonical_kmers().enumerate() {
map.entry(kmer).or_default().push(KmerDesc {
seq_idx: seq_idx as u32,
pos: kmer_offset + j as u32,
});
}
let n = (rsk.seql() - k + 1) as u32;
kmer_offset += n;
}
ids.push(record.id);
seqs.push(record.sequence);
n_kmers.push(kmer_offset);
}
Self {
ids,
seqs,
n_kmers,
by_partition,
}
}
}
-289
View File
@@ -1,289 +0,0 @@
use std::time::Instant;
use obikindex::KmerIndex;
use obikindex::{GenomeInfo, KmerDesc, QueryHit, QueryStats};
use obikrope::Rope;
use obikseq::CanonicalKmer;
use obiread::record::parse_chunk;
use tracing::debug;
use super::batch::QueryBatch;
use super::findere::{ConfirmedHit, sparse_findere_for_genome};
use super::output::emit_batch;
use super::smer_index::SmerIndex;
pub(super) struct SeqAcc {
pub(super) kmer_count: u32,
pub(super) kmer_missing: u32,
pub(super) genome_totals: Vec<u32>,
}
impl SeqAcc {
fn new(n_genomes: usize) -> Self {
Self {
kmer_count: 0,
kmer_missing: 0,
genome_totals: vec![0u32; n_genomes],
}
}
}
pub(super) fn process_chunk(
idx: &KmerIndex,
rope: Rope,
k: usize,
n_genomes: usize,
n_partitions: usize,
with_counts: bool,
effective_z: usize,
detail: bool,
count_missing: bool,
force_presence: bool,
presence_threshold: u32,
genomes: &[GenomeInfo],
) -> Vec<u8> {
let chunk_start = Instant::now();
let chunk_bytes = rope.len();
let records = parse_chunk(&rope, k);
if records.is_empty() {
return Vec::new();
}
let batch = QueryBatch::from_records(records, k, 6, 0.7, n_partitions);
let n_seqs = batch.ids.len();
// Estimate QueryBatch::by_partition's actual memory footprint: the
// k-mer-level dedup map (roadmap point 5) — one HashMap<CanonicalKmer,
// Vec<KmerDesc>> per partition, sized by *unique* k-mers, not shrunk by
// dedup. On real workloads with a low intra-chunk duplication rate this
// can dwarf every other per-chunk structure, including the sparse
// Findere ones logged further down — unlike those, chunk_bytes's formula
// (run()) does not account for this at all today. Measured by allocated
// capacity, not logical length, to reflect real memory pressure
// (HashMap/Vec growth slack) — `by_partition` is alive for the entire
// process_chunk call (never drained, only iterated by reference), so
// this is its footprint for the whole chunk lifetime, not a transient.
let hashmap_slot_bytes = (std::mem::size_of::<CanonicalKmer>()
+ std::mem::size_of::<Vec<KmerDesc>>()
+ 1) as u64; // +1 ≈ hashbrown control byte per slot
let by_partition_map_bytes: u64 = batch
.by_partition
.iter()
.map(|m| m.capacity() as u64 * hashmap_slot_bytes)
.sum();
let by_partition_desc_bytes: u64 = batch
.by_partition
.iter()
.flat_map(|m| m.values())
.map(|v| v.capacity() as u64 * std::mem::size_of::<KmerDesc>() as u64)
.sum();
let by_partition_bytes = by_partition_map_bytes + by_partition_desc_bytes;
debug!(
n_unique_kmers_total = batch.by_partition.iter().map(|m| m.len() as u64).sum::<u64>(),
by_partition_map_bytes,
by_partition_desc_bytes,
by_partition_bytes,
chunk_bytes,
"by_partition memory retained"
);
// Sparse bookkeeping for the whole chunk:
// - smer_index: O(total_smers) — is this s-mer in the index at all.
// - by_genome[g]: raw (seq_idx, pos_smer, value) hits for genome g, only
// ever containing nonzero entries (query_partition_with never emits a
// QueryHit::Value for a zero value) — empty for every genome this chunk
// never matched, which is the common case for unrelated queries.
let mut smer_index = SmerIndex::new(&batch.n_kmers);
let mut by_genome: Vec<Vec<(u32, u32, u32)>> = (0..n_genomes).map(|_| Vec::new()).collect();
// Dedup-ratio bookkeeping: occurrences (from batch.n_kmers, computed
// before dedup) vs. unique k-mers actually queried (query_stats) — the
// entire justification for k-mer-level dereplication (see query.md,
// Future work point 5). If this ratio stays close to 1.0 on real data,
// dereplication isn't paying for itself and that should show up here.
let n_occurrences: u64 = batch.n_kmers.iter().map(|&n| n as u64).sum();
let mut query_stats = QueryStats::default();
for (part_idx, kmers) in batch.by_partition.iter().enumerate() {
if kmers.is_empty() {
continue;
}
let stats = idx
.query_partition_with(
part_idx,
kmers,
n_genomes,
with_counts,
|event| match event {
QueryHit::Found(descs) => {
for desc in descs {
smer_index.mark_found(desc.seq_idx as usize, desc.pos as usize);
}
}
QueryHit::Value(descs, g, v) => {
for desc in descs {
by_genome[g].push((desc.seq_idx, desc.pos, v));
}
}
},
)
.unwrap_or_else(|e| {
eprintln!("query error on partition {part_idx}: {e}");
std::process::exit(1);
});
query_stats += stats;
}
debug!(
n_occurrences,
n_unique_kmers = query_stats.n_unique_kmers,
n_mphf_calls = query_stats.n_mphf_calls,
n_hits = query_stats.n_hits,
n_columns_scanned = query_stats.n_columns_scanned,
n_col_get_calls = query_stats.n_col_get_calls,
"k-mer dedup + column-major fetch"
);
// ── Sparse Findere: per-genome run detection + sliding-window minimum ────
//
// Confirmed z-windows, per genome, replace the dense win_min matrix:
// total retained memory is O(actual hits), not O(total_smers × n_genomes)
// — the whole point of this pass. See sparse_findere_for_genome's doc for
// why run detection is equivalent to the dense scan's semantics.
let presence = force_presence || !with_counts;
let threshold = presence_threshold;
let z = effective_z;
let n_kmers_out: Vec<usize> = batch
.n_kmers
.iter()
.map(|&n| {
let n = n as usize;
if n >= z { n - z + 1 } else { 0 }
})
.collect();
let mut out_offsets = Vec::with_capacity(n_seqs + 1);
{
let mut total = 0usize;
out_offsets.push(0);
for &n in &n_kmers_out {
total += n;
out_offsets.push(total);
}
}
let total_out = *out_offsets.last().unwrap_or(&0);
let n_dense_would_be = n_occurrences as u64 * n_genomes as u64;
let mut n_sparse_entries = 0u64;
let mut n_runs_total = 0usize;
let mut run_len_total = 0usize;
let mut confirmed_by_genome: Vec<Vec<ConfirmedHit>> = Vec::with_capacity(n_genomes);
for hits in &mut by_genome {
n_sparse_entries += hits.len() as u64;
let (confirmed, n_runs, run_len) = sparse_findere_for_genome(hits, z, presence, threshold);
n_runs_total += n_runs;
run_len_total += run_len;
confirmed_by_genome.push(confirmed);
}
debug!(
n_dense_would_be,
n_sparse_entries,
n_runs = n_runs_total,
avg_run_len = if n_runs_total > 0 { run_len_total as f64 / n_runs_total as f64 } else { 0.0 },
z,
"sparse Findere"
);
// Actual bytes retained by the sparse hit structures (by_genome +
// confirmed_by_genome, both alive simultaneously at this point — see the
// chunk-size formula's comment in `run()`), by allocated capacity rather
// than logical length so this reflects real memory pressure including
// Vec growth slack. `empirical_multiplier` is directly comparable to
// BYTES_PER_KMER_PER_GENOME (`run()`) — the ratio a cluster run's logs
// need to judge whether that constant is over- or under-conservative for
// real data, instead of guessing.
const HIT_ENTRY_BYTES: u64 = std::mem::size_of::<(u32, u32, u32)>() as u64;
let by_genome_bytes: u64 = by_genome.iter().map(|v| v.capacity() as u64 * HIT_ENTRY_BYTES).sum();
let confirmed_bytes: u64 = confirmed_by_genome.iter().map(|v| v.capacity() as u64 * HIT_ENTRY_BYTES).sum();
let retained_bytes = by_genome_bytes + confirmed_bytes;
debug!(
by_genome_bytes,
confirmed_bytes,
retained_bytes,
chunk_bytes,
empirical_multiplier = retained_bytes as f64 / chunk_bytes.max(1) as f64,
"sparse memory retained"
);
// ── Accumulate: genome totals (per genome, from confirmed hits) ──────────
let mut accs: Vec<SeqAcc> = (0..n_seqs).map(|_| SeqAcc::new(n_genomes)).collect();
let mut confirmed_any = vec![false; total_out];
for (g, hits) in confirmed_by_genome.iter().enumerate() {
for &(seq_idx, pos_out, c) in hits {
let abs_out = out_offsets[seq_idx as usize] + pos_out as usize;
confirmed_any[abs_out] = true;
accs[seq_idx as usize].genome_totals[g] += c;
}
}
// ── Accumulate: kmer_count / kmer_missing (per position, genome-independent) ─
for seq_idx in 0..n_seqs {
let out_n = n_kmers_out[seq_idx];
let acc = &mut accs[seq_idx];
for pos in 0..out_n {
let abs_out = out_offsets[seq_idx] + pos;
if confirmed_any[abs_out] {
acc.kmer_count += 1;
} else if !smer_index.is_in_index(seq_idx, pos) {
acc.kmer_missing += 1;
}
}
}
// ── Coverage (--detail): densify only when actually requested ────────────
let mut cov: Vec<Vec<Vec<u32>>> = if detail {
n_kmers_out.iter().map(|&n| vec![vec![0u32; n]; n_genomes]).collect()
} else {
Vec::new()
};
if detail {
for (g, hits) in confirmed_by_genome.iter().enumerate() {
for &(seq_idx, pos_out, c) in hits {
cov[seq_idx as usize][g][pos_out as usize] += c;
}
}
}
// Capacity estimate: actual sequence + ID bytes, plus JSON overhead per record.
// JSON per record ≈ 50 fixed chars + ~20 per genome (label + count value) + 100 (overhead).
let seq_bytes: usize = batch.seqs.iter().map(|s| s.len()).sum();
let id_bytes: usize = batch.ids.iter().map(|s| s.len()).sum();
let cap = seq_bytes + id_bytes + n_seqs * (4 + 50 + n_genomes * 20) + 100;
let mut buf = Vec::with_capacity(cap);
emit_batch(
&batch,
&accs,
genomes,
count_missing,
detail,
&cov,
&mut buf,
);
debug!(
chunk_bytes,
n_seqs,
n_smers = batch.n_kmers.iter().map(|&n| n as u64).sum::<u64>(),
wall_ms = chunk_start.elapsed().as_millis() as u64,
"process_chunk"
);
buf
}
-71
View File
@@ -1,71 +0,0 @@
use std::collections::VecDeque;
/// One confirmed z-window: genome `g`'s window ending at k-mer `pos` (the
/// *leftmost* s-mer of the window, i.e. the k_user-mer's output position) is
/// fully present and nonzero, with window-minimum `value`.
pub(super) type ConfirmedHit = (u32, u32, u32); // (seq_idx, pos_out, value)
/// Reduce one genome's raw sparse s-mer hits — `(seq_idx, pos_smer, raw_value)`,
/// unsorted, exactly as delivered by `QueryHit::Value` — into confirmed
/// z-windows, without ever visiting a position that had no hit at all.
///
/// A z-window is confirmed only when all z s-mers in it are present *and*
/// nonzero for this genome (matching the dense sliding-window's semantics,
/// where "not in index" or a zero value both contribute 0 to the window
/// minimum) — which can only happen inside a maximal run of consecutive
/// `pos_smer` values for the same sequence. `hits` is sorted in place by
/// `(seq_idx, pos_smer)` to expose those runs; the monotone-deque
/// window-minimum then runs per run, on run-relative indices, identical in
/// spirit to the dense version's whole-sequence scan.
///
/// Returns the confirmed hits plus `(n_runs, total_run_len)` for logging —
/// a low average run length relative to `z` means most hits fail to form a
/// complete window.
pub(super) fn sparse_findere_for_genome(
hits: &mut [(u32, u32, u32)],
z: usize,
presence: bool,
threshold: u32,
) -> (Vec<ConfirmedHit>, usize, usize) {
hits.sort_unstable_by_key(|&(seq, pos, _)| (seq, pos));
let mut confirmed = Vec::new();
let mut n_runs = 0usize;
let mut total_run_len = 0usize;
let mut dq: VecDeque<(usize, u32)> = VecDeque::new(); // (run-relative index, value)
let mut i = 0;
while i < hits.len() {
let seq = hits[i].0;
let mut j = i + 1;
while j < hits.len() && hits[j].0 == seq && hits[j].1 == hits[j - 1].1 + 1 {
j += 1;
}
let run = &hits[i..j];
n_runs += 1;
total_run_len += run.len();
dq.clear();
for (k, &(_, pos, val)) in run.iter().enumerate() {
while dq.back().map_or(false, |&(_, v)| v >= val) {
dq.pop_back();
}
dq.push_back((k, val));
while dq.front().map_or(false, |&(fk, _)| fk + z <= k) {
dq.pop_front();
}
if k + 1 >= z {
let win_min = dq.front().unwrap().1;
if win_min > 0 {
let pos_out = pos + 1 - z as u32;
let c = if presence { u32::from(win_min >= threshold) } else { win_min };
confirmed.push((seq, pos_out, c));
}
}
}
i = j;
}
(confirmed, n_runs, total_run_len)
}
+23 -27
View File
@@ -1,9 +1,3 @@
mod batch;
mod chunk;
mod findere;
mod output;
mod smer_index;
use std::io::{self, BufWriter, Write};
use std::path::PathBuf;
use std::sync::Arc;
@@ -11,16 +5,16 @@ use std::sync::atomic::{AtomicU32, AtomicU64, Ordering};
use std::time::Instant;
use clap::Args;
use obikidxcache::index_cache::IndexCache;
use obikindex::KmerIndex;
use obikrope::Rope;
use obikindex::layer::IndexMode;
use obikquery::process_chunk;
use obikrope::Rope;
use obipipeline::{Throttled, ThrottleGuard, throttle};
use obiread::chunk::read_sequence_chunks_sized;
use obisys::{Reporter, Stage, available_memory_bytes, spinner};
use tracing::{debug, info};
use chunk::process_chunk;
// ── Pipeline data ─────────────────────────────────────────────────────────────
enum QueryData {
@@ -139,24 +133,31 @@ pub fn run(args: QueryArgs) {
let with_counts = idx.meta().config.with_counts;
let n_workers = args.threads.max(1);
// Every partition/layer the query might touch is opened once, up front,
// and shared (via Arc) across every `obipipeline` worker — a query pass
// is then pure in-memory lookups, never a per-chunk disk open (the
// previous `obikindex`-based design's cost). `IndexCache` owns its
// `Arc<KmerIndex>`, so it's itself `'static`-capable, satisfying
// obipipeline's `Send + Sync + 'static` requirement on pipeline data —
// see `obikquery::query_layer`'s doc comment for why that rules out a
// borrow-based cache here.
let cache = Arc::new(IndexCache::new(Arc::clone(&idx), None));
// Chunk size: each chunk stays in memory for its entire processing lifetime.
//
// Per-chunk memory is no longer a dense n_genomes-wide buffer (removed in
// the sparse Findere rework, see process_chunk) — it now scales with
// Per-chunk memory is not a dense n_genomes-wide buffer — it scales with
// *actual hit count*, not with total_kmers_in_chunk × n_genomes
// unconditionally. BYTES_PER_KMER_PER_GENOME below is therefore a
// pathological-case bound, not a typical-case estimate: it protects
// against a fully-dense hit pattern (every k-mer of the query matching
// every genome — a degenerate case, e.g. low-complexity input theta-
// filtering should mostly reject, or an index of near-duplicate genomes),
// where by_genome and confirmed_by_genome (process_chunk) both end up
// holding one (seq_idx, pos, value) entry — 3 × u32 = 12 bytes, vs. 4
// bytes for the old dense encoding, where position was implicit in the
// array index — per (k-mer, genome) pair, and *coexist simultaneously*
// (by_genome isn't freed before confirmed_by_genome is built), for a
// worst case of ~24 bytes/pair before Vec growth slack. `cov` remains
// fully dense when --detail is set (unaffected by the sparse rework),
// still roughly doubling the n_genomes-scaled cost.
// where by_genome and confirmed_by_genome (obikquery::chunk::process_chunk)
// both end up holding one (seq_idx, pos, value) entry — 3 × u32 = 12
// bytes — per (k-mer, genome) pair, and *coexist simultaneously* (by_genome
// isn't freed before confirmed_by_genome is built), for a worst case of
// ~24 bytes/pair before Vec growth slack. `cov` remains fully dense when
// --detail is set, still roughly doubling the n_genomes-scaled cost.
//
// For realistic, sparse hit patterns actual memory is far below this
// bound — see the "sparse memory retained" debug log in process_chunk,
@@ -225,9 +226,7 @@ pub fn run(args: QueryArgs) {
// Throttled iterator over input file paths: at most `effective_max_open()`
// files are open at once. Opening + decompressing + chunking each file is
// now a Flat pipeline stage, executed across the `n_workers` pool — not
// serialised in the pipe's dedicated source thread (see steps::scatter /
// cmd::superkmer for the same pattern applied to indexing).
// a Flat pipeline stage, executed across the `n_workers` pool.
info!("query: chunk_size={}MiB, max_open_files={}", chunk_bytes / (1024 * 1024), args.effective_max_open());
let paths: Vec<PathBuf> = args.inputs.iter().map(PathBuf::from).collect();
@@ -282,7 +281,7 @@ pub fn run(args: QueryArgs) {
}
} : Path => Chunk,
| {
let idx = Arc::clone(&idx);
let cache = Arc::clone(&cache);
let genomes = Arc::clone(&genomes);
let total_bytes = Arc::clone(&total_bytes);
let chunks_active = Arc::clone(&chunks_active);
@@ -290,7 +289,7 @@ pub fn run(args: QueryArgs) {
chunks_active.fetch_add(1, Ordering::Relaxed);
let bytes = rope.len() as u64;
let out = process_chunk(
&idx, rope, k, n_genomes, n_partitions, with_counts,
&cache, rope, k, n_genomes, n_partitions, with_counts,
effective_z, detail, count_missing, force_presence, presence_threshold,
&genomes,
);
@@ -342,6 +341,3 @@ pub fn run(args: QueryArgs) {
rep.push(t.stop());
rep.print();
}
#[cfg(test)]
mod tests;
-52
View File
@@ -1,52 +0,0 @@
use std::io::Write;
use obikindex::GenomeInfo;
use super::batch::QueryBatch;
use super::chunk::SeqAcc;
pub(super) fn emit_batch(
batch: &QueryBatch,
accs: &[SeqAcc],
genomes: &[GenomeInfo],
count_missing: bool,
detail: bool,
cov: &[Vec<Vec<u32>>],
out: &mut impl Write,
) {
for (seq_idx, (id, seq)) in batch.ids.iter().zip(batch.seqs.iter()).enumerate() {
let acc = &accs[seq_idx];
let mut ann = serde_json::Map::new();
ann.insert("kmer_count".into(), acc.kmer_count.into());
if count_missing {
ann.insert("kmer_missing".into(), acc.kmer_missing.into());
}
let mut match_map = serde_json::Map::new();
for (g, genome) in genomes.iter().enumerate() {
if acc.genome_totals[g] != 0 {
match_map.insert(genome.label.clone(), acc.genome_totals[g].into());
}
}
ann.insert("kmer_strict_matches".into(), match_map.into());
if detail && !cov.is_empty() {
let mut cov_map = serde_json::Map::new();
for (g, genome) in genomes.iter().enumerate() {
let v: Vec<serde_json::Value> = cov[seq_idx][g].iter().map(|&x| x.into()).collect();
cov_map.insert(genome.label.clone(), v.into());
}
ann.insert("coverage".into(), cov_map.into());
}
// OBITools4 FASTA format: >id {"key":value,...}
let _ = out.write_all(b">");
let _ = out.write_all(id.as_bytes());
let _ = out.write_all(b" ");
let _ = serde_json::to_writer(&mut *out, &ann);
let _ = out.write_all(b"\n");
let _ = out.write_all(seq);
let _ = out.write_all(b"\n");
}
}
-41
View File
@@ -1,41 +0,0 @@
/// Tracks, per (sequence, s-mer position), whether the k-mer was found in the
/// index at all — independent of *which* genome(s) matched. Sized
/// `total_smers` (one `bool` per s-mer occurrence in the chunk), **not**
/// multiplied by `n_genomes`: this is the O(1)-per-position bookkeeping that
/// `kmer_missing` needs (the leftmost-s-mer-of-window membership test), kept
/// dense because it's already cheap — the `n_genomes`-scaled data lives in
/// the sparse per-genome hit lists built alongside it (see `process_chunk`).
pub(super) struct SmerIndex {
in_index: Vec<bool>, // total_smers
offsets: Vec<usize>, // offsets[i]..offsets[i+1] = s-mer range for sequence i
}
impl SmerIndex {
pub(super) fn new(n_kmers_per_seq: &[u32]) -> Self {
let mut offsets = Vec::with_capacity(n_kmers_per_seq.len() + 1);
let mut total = 0usize;
offsets.push(0);
for &n in n_kmers_per_seq {
total += n as usize;
offsets.push(total);
}
Self {
in_index: vec![false; total],
offsets,
}
}
/// Mark the k-mer at (seq, kmer) as found in the index — independent of
/// any particular genome's value. Called once per hit k-mer (stage 1 of
/// `query_partition_with`), regardless of how the column-major fetch
/// (stage 2) later reports per-genome values.
pub(super) fn mark_found(&mut self, seq: usize, kmer: usize) {
let abs = self.offsets[seq] + kmer;
self.in_index[abs] = true;
}
#[inline]
pub(super) fn is_in_index(&self, seq: usize, kmer: usize) -> bool {
self.in_index[self.offsets[seq] + kmer]
}
}
-233
View File
@@ -1,233 +0,0 @@
use obikrope::Rope;
use obikseq::CanonicalKmer;
use obiread::record::parse_chunk;
use super::batch::QueryBatch;
use super::findere::sparse_findere_for_genome;
const K: usize = 11;
const M: usize = 5;
/// Build a `QueryBatch` from raw FASTA text, going through the same
/// `Rope` + `parse_chunk` path `process_chunk` uses — avoids hand-building a
/// `normalized` `Rope`, which is an implementation detail of `obiread`.
///
/// `obikseq`'s global K/M params are thread-local under `test-utils` (see
/// `obikseq::params`), so setting them here is per-test-thread and does not
/// need coordination with other tests.
fn batch_from_fasta(fasta: &str, k: usize, n_partitions: usize) -> QueryBatch {
obikseq::set_k(k);
obikseq::set_m(M);
let mut rope = Rope::new(Some("text/fasta"));
rope.push(fasta.as_bytes().to_vec());
let records = parse_chunk(&rope, k);
QueryBatch::from_records(records, k, 6, 0.7, n_partitions)
}
fn total_occurrences(batch: &QueryBatch) -> u64 {
batch.n_kmers.iter().map(|&n| n as u64).sum()
}
fn total_unique_kmers(batch: &QueryBatch) -> u64 {
batch.by_partition.iter().map(|m| m.len() as u64).sum()
}
// A 60 bp sequence, arbitrary but fixed — no attempt is made to prove it is
// free of internal k=11 repeats; the tests below only rely on inequalities
// that hold regardless (see each test's comment).
const SEQ: &str = "CATTAGCGTACCTGATCAGGTTACAGCTTAGGCATCCAGTTGACCATGACTGGACTTAGC";
#[test]
fn single_sequence_yields_plausible_kmer_counts() {
// A single record can still contain internal repeats (SEQ isn't
// guaranteed repeat-free at k=11) — this only checks the batch is
// internally consistent, not a specific dedup ratio. The cross-record
// tests below make the actual, unconditional dedup claims.
let fasta = format!(">r1\n{SEQ}\n");
let batch = batch_from_fasta(&fasta, K, 1);
assert_eq!(batch.ids, vec!["r1".to_string()]);
let occurrences = total_occurrences(&batch);
let unique = total_unique_kmers(&batch);
assert!(occurrences > 0, "sequence should yield at least one k-mer");
assert!(unique > 0 && unique <= occurrences);
}
#[test]
fn duplicated_sequence_across_records_deduplicates() {
// Two records with byte-identical sequences: every k-mer in record 1
// exactly duplicates one in record 0, so unique kmers <= n_kmers[0],
// strictly less than the summed occurrences (2 * n_kmers[0]) as long as
// the sequence yields at least one k-mer. This holds regardless of
// whether SEQ has internal repeats.
let fasta = format!(">r1\n{SEQ}\n>r2\n{SEQ}\n");
let batch = batch_from_fasta(&fasta, K, 1);
assert_eq!(batch.ids.len(), 2);
let occurrences = total_occurrences(&batch);
let unique = total_unique_kmers(&batch);
assert!(batch.n_kmers[0] > 0);
assert_eq!(occurrences, batch.n_kmers[0] as u64 + batch.n_kmers[1] as u64);
assert!(
unique <= batch.n_kmers[0] as u64,
"identical sequences must not produce more unique k-mers than one copy has"
);
assert!(
unique < occurrences,
"k-mer-level dedup must collapse at least the cross-record duplication"
);
}
#[test]
fn duplicated_sequence_broadcasts_to_both_seq_indices() {
// Stronger than the ratio check above: pick any k-mer that hit in both
// records and confirm its occurrence list actually references both
// seq_idx 0 and seq_idx 1 — this is the specific new capability (dedup
// reaching across records/superkmers), not just a smaller unique count.
let fasta = format!(">r1\n{SEQ}\n>r2\n{SEQ}\n");
let batch = batch_from_fasta(&fasta, K, 1);
let shared = batch.by_partition[0]
.values()
.find(|descs| descs.iter().any(|d| d.seq_idx == 0) && descs.iter().any(|d| d.seq_idx == 1));
assert!(
shared.is_some(),
"expected at least one k-mer shared between the two identical records"
);
}
#[test]
fn empty_records_yield_empty_batch() {
let batch = batch_from_fasta("", K, 1);
assert!(batch.ids.is_empty());
assert_eq!(total_occurrences(&batch), 0);
assert_eq!(total_unique_kmers(&batch), 0);
}
#[test]
fn partition_routing_is_a_pure_function_of_the_kmer() {
// With n_partitions=4, every occurrence of a given k-mer must land in
// the same partition bucket as every other occurrence of that k-mer
// (partition routing is derived from the minimizer, shared by
// definition among instances of the same k-mer's containing superkmer
// in this test's single-sequence-pair setup).
let fasta = format!(">r1\n{SEQ}\n>r2\n{SEQ}\n");
let batch = batch_from_fasta(&fasta, K, 4);
let total_unique: u64 = batch.by_partition.iter().map(|m| m.len() as u64).sum();
assert!(total_unique > 0);
// No k-mer key appears in more than one partition's map.
let mut seen: std::collections::HashSet<CanonicalKmer> = std::collections::HashSet::new();
for map in &batch.by_partition {
for kmer in map.keys() {
assert!(seen.insert(*kmer), "k-mer routed to more than one partition");
}
}
}
// ── sparse_findere_for_genome vs. a dense reference implementation ──────────
//
// No property-testing crate (proptest/quickcheck) is a workspace dependency
// (checked before writing this — not adding one for a single test module,
// per this project's dependency-approval rule). A tiny deterministic xorshift
// PRNG, std-only, stands in for one.
/// Faithful reimplementation of the pre-phase-5 dense sliding-window scan —
/// the algorithm `sparse_findere_for_genome` replaced — used here only as a
/// correctness oracle, not in production code. Operates on one genome's
/// hits across possibly many sequences, exactly like the sparse version.
fn dense_reference_findere(
hits: &[(u32, u32, u32)],
seq_lens: &[usize],
z: usize,
presence: bool,
threshold: u32,
) -> Vec<(u32, u32, u32)> {
let mut by_seq: Vec<Vec<u32>> = seq_lens.iter().map(|&n| vec![0u32; n]).collect();
for &(seq, pos, val) in hits {
by_seq[seq as usize][pos as usize] = val;
}
let mut confirmed = Vec::new();
for (seq_idx, values) in by_seq.iter().enumerate() {
let n = values.len();
let mut dq: std::collections::VecDeque<(usize, u32)> = std::collections::VecDeque::new();
for i in 0..n {
let v_i = values[i];
while dq.front().map_or(false, |&(f, _)| f + z <= i) {
dq.pop_front();
}
while dq.back().map_or(false, |&(_, v)| v >= v_i) {
dq.pop_back();
}
dq.push_back((i, v_i));
if i + 1 >= z {
let win_min = dq.front().unwrap().1;
if win_min > 0 {
let pos_out = (i + 1 - z) as u32;
let c = if presence { u32::from(win_min >= threshold) } else { win_min };
confirmed.push((seq_idx as u32, pos_out, c));
}
}
}
}
confirmed
}
/// Minimal std-only xorshift64 PRNG — deterministic, seedable, no dependency.
struct Xorshift64(u64);
impl Xorshift64 {
fn next(&mut self) -> u64 {
self.0 ^= self.0 << 13;
self.0 ^= self.0 >> 7;
self.0 ^= self.0 << 17;
self.0
}
fn range(&mut self, n: u32) -> u32 {
(self.next() % n as u64) as u32
}
}
#[test]
fn sparse_findere_matches_dense_reference_on_random_inputs() {
let mut rng = Xorshift64(0x5eed_5eed_5eed_5eedu64);
for case in 0..200 {
let n_seqs = 1 + rng.range(4) as usize;
let seq_lens: Vec<usize> = (0..n_seqs).map(|_| 1 + rng.range(30) as usize).collect();
let z = 1 + rng.range(4) as usize;
let presence = rng.range(2) == 0;
let threshold = 1 + rng.range(3);
// Sparse density varies across cases, including edge cases (empty,
// fully dense) — deliberately not uniform, to stress both few-hits
// and many-overlapping-runs scenarios.
let density = rng.range(101);
let mut hits: Vec<(u32, u32, u32)> = Vec::new();
for (seq_idx, &len) in seq_lens.iter().enumerate() {
for pos in 0..len {
if rng.range(100) < density {
let val = 1 + rng.range(5); // never 0 — matches QueryHit::Value's invariant
hits.push((seq_idx as u32, pos as u32, val));
}
}
}
let mut sparse_input = hits.clone();
let (mut sparse_result, _, _) =
sparse_findere_for_genome(&mut sparse_input, z, presence, threshold);
let mut dense_result = dense_reference_findere(&hits, &seq_lens, z, presence, threshold);
sparse_result.sort_unstable();
dense_result.sort_unstable();
assert_eq!(
sparse_result, dense_result,
"case {case}: n_seqs={n_seqs} seq_lens={seq_lens:?} z={z} presence={presence} \
threshold={threshold} density={density} hits={hits:?}"
);
}
}
-75
View File
@@ -1,75 +0,0 @@
use std::path::PathBuf;
use clap::Args;
use obikindex::KmerIndex;
use obikindex::layer::IndexMode;
use obisys::Reporter;
use tracing::info;
use crate::cli::block_size_to_bits;
use super::index::resolve_approx_params;
#[derive(Args)]
pub struct ReindexArgs {
/// Index directory to convert (modified in-place)
pub index: PathBuf,
/// Convert to approximate evidence (default: convert to exact).
/// Requires --evidence-bits and/or -z and/or --fp.
#[arg(long, default_value_t = false)]
pub approx: bool,
/// Findere z parameter (≥1).
#[arg(short = 'z', long, default_value = None)]
pub findere_z: Option<u8>,
/// Fingerprint bits per slot (b).
#[arg(long, default_value = None)]
pub evidence_bits: Option<u8>,
/// Target false-positive rate per z-window.
#[arg(long, default_value = None)]
pub fp: Option<f64>,
/// Block size for exact evidence `.idx` (number of unitigs per block).
/// Ignored when converting to approximate evidence.
#[arg(long, default_value_t = 1)]
pub block_size: usize,
}
pub fn run(args: ReindexArgs) {
let target = if args.approx {
let (z, b, fp) = resolve_approx_params(args.findere_z, args.evidence_bits, args.fp);
info!("target: approximate evidence — b={b}, z={z}, fp={fp:.2e}");
IndexMode::Approx { b, z }
} else {
info!("target: exact evidence");
IndexMode::Exact
};
// Modifies the index in place; acquired before opening so a concurrent
// writer can't slip in between the open and the reindex below.
let _lock = obisys::DirLock::acquire(&args.index).unwrap_or_else(|e| {
eprintln!("error locking index directory {}: {e}", args.index.display());
std::process::exit(1);
});
let mut idx = KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
});
info!(
"current evidence: {:?}",
idx.meta().config.evidence,
);
let block_bits = block_size_to_bits(args.block_size);
let mut rep = Reporter::new();
idx.reindex(target, block_bits, &mut rep).unwrap_or_else(|e| {
eprintln!("reindex error: {e}");
std::process::exit(1);
});
rep.print();
}
+47 -174
View File
@@ -1,14 +1,12 @@
use std::collections::{BTreeMap, HashMap};
use std::path::PathBuf;
use clap::{Args, ValueEnum};
use obikindex::{IndexMeta, KmerIndex};
use obikindex::{AggOp, OutputCol};
use obisys::Reporter;
use obikalgorithm::Algorithm;
use obikindex::KmerIndex;
use obikselect::{AggOp, ColumnSpecParams, Select, build_output_cols};
use obisys::{Progress, Reporter, Stage, progress_bar};
use tracing::info;
// ── CLI types ─────────────────────────────────────────────────────────────────
#[derive(Debug, Clone, Copy, PartialEq, Eq, ValueEnum)]
pub enum AggOpArg {
Any,
@@ -37,13 +35,9 @@ pub struct SelectArgs {
/// Source index directory
pub source: PathBuf,
/// Output index directory (mutually exclusive with --in-place)
#[arg(long, conflicts_with = "in_place")]
pub output: Option<PathBuf>,
/// Rewrite the source index in-place (mutually exclusive with --output)
#[arg(long)]
pub in_place: bool,
/// Output index directory
#[arg(short, long)]
pub output: PathBuf,
/// Define a named group: `<name>:<pred>` (repeatable; mutually exclusive with --aggregate-by)
#[arg(long, value_name = "NAME:PRED", conflicts_with = "aggregate_by")]
@@ -69,173 +63,53 @@ pub struct SelectArgs {
#[arg(long, default_value = "0")]
pub presence_threshold: u32,
/// Pack the output's presence matrices in the dense format instead of the default sparse one
#[arg(long, default_value_t = false)]
pub dense: bool,
/// Overwrite existing output directory
#[arg(short, long)]
pub force: bool,
}
// ── Helpers ───────────────────────────────────────────────────────────────────
/// Split a repeatable `<name>:<value>` argument. Exits on malformed input.
fn parse_name_value(s: &str, flag: &str) -> (String, String) {
match s.find(':') {
Some(pos) => (s[..pos].trim().to_string(), s[pos + 1..].to_string()),
std::option::Option::None => {
None => {
eprintln!("error in {flag}: expected <name>:<value>, got: {s}");
std::process::exit(1);
}
}
}
fn parse_agg_op(s: &str) -> AggOp {
match s.to_lowercase().as_str() {
"any" => AggOp::Any,
"all" => AggOp::All,
"none" => AggOp::None,
"sum" => AggOp::Sum,
"min" => AggOp::Min,
"max" => AggOp::Max,
other => {
eprintln!("unknown aggregation operator: {other}; valid: any, all, none, sum, min, max");
std::process::exit(1);
}
}
}
fn default_op(src_is_count: bool) -> AggOp {
if src_is_count { AggOp::Sum } else { AggOp::Any }
}
// ── build_specs ───────────────────────────────────────────────────────────────
/// Resolve CLI arguments into an ordered list of `OutputCol`.
///
/// Returns `(specs, output_presence)`.
fn build_specs(
args: &SelectArgs,
meta: &IndexMeta,
src_is_count: bool,
) -> (Vec<OutputCol>, bool) {
let genomes = meta.genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
});
let genomes = &genomes;
// ── 1. Build group_indices: name → Vec<usize> ────────────────────────────
// Also keep insertion order for the default `--select *` case.
let mut group_order: Vec<String> = Vec::new();
let mut group_indices: HashMap<String, Vec<usize>> = HashMap::new();
if let Some(ref key) = args.aggregate_by {
// One group per unique value of `key`, in sorted order.
let mut value_to_indices: BTreeMap<String, Vec<usize>> = BTreeMap::new();
for (i, g) in genomes.iter().enumerate() {
if let Some(v) = g.meta.get(key) {
value_to_indices.entry(v.clone()).or_default().push(i);
}
}
for (v, idxs) in value_to_indices {
group_order.push(v.clone());
group_indices.insert(v, idxs);
}
} else {
for raw in &args.group {
let (name, pred) = parse_name_value(raw, "--group");
let idxs = meta.matching_genome_indices(&pred).unwrap_or_else(|e| {
eprintln!("error in --group {name}: {e}");
std::process::exit(1);
});
if !group_indices.contains_key(&name) {
group_order.push(name.clone());
}
group_indices.insert(name, idxs);
}
}
// ── 2. Build per-group ops ────────────────────────────────────────────────
let global_op = args.aggregate_op.map(AggOp::from);
let mut group_op: HashMap<String, AggOp> = HashMap::new();
for raw in &args.group_op {
let (name, op_str) = parse_name_value(raw, "--group-op");
if !group_indices.contains_key(&name) {
eprintln!("--group-op references undefined group: {name}");
std::process::exit(1);
}
group_op.insert(name, parse_agg_op(&op_str));
}
// ── 3. Genome label → index map for pass-through columns ─────────────────
let label_to_idx: HashMap<&str, usize> = genomes.iter().enumerate()
.map(|(i, g)| (g.label.as_str(), i))
.collect();
// ── 4. Determine output column names ─────────────────────────────────────
let col_names: Vec<String> = if let Some(ref sel) = args.select {
sel.clone()
} else if !group_order.is_empty() {
group_order.clone()
} else {
// Identity: all genomes in original order
genomes.iter().map(|g| g.label.clone()).collect()
};
// ── 5. Build OutputCol list ───────────────────────────────────────────────
let mut specs: Vec<OutputCol> = Vec::with_capacity(col_names.len());
for name in &col_names {
if let Some(idxs) = group_indices.get(name) {
let op = group_op.get(name)
.copied()
.or(global_op)
.unwrap_or_else(|| default_op(src_is_count));
specs.push(OutputCol { label: name.clone(), indices: idxs.clone(), op });
} else if let Some(&idx) = label_to_idx.get(name.as_str()) {
// Pass-through: single-element group with default op.
let op = default_op(src_is_count);
specs.push(OutputCol { label: name.clone(), indices: vec![idx], op });
} else {
eprintln!("--select: unknown column '{name}' (not a group name or genome label)");
std::process::exit(1);
}
}
if specs.is_empty() {
eprintln!("select: no output columns defined");
std::process::exit(1);
}
// ── 6. Determine output type ──────────────────────────────────────────────
let output_presence = !src_is_count
|| specs.iter().all(|s| s.op.is_logical());
(specs, output_presence)
}
// ── run ───────────────────────────────────────────────────────────────────────
pub fn run(args: SelectArgs) {
if !args.in_place && args.output.is_none() {
eprintln!("error: one of --output or --in-place must be specified");
std::process::exit(1);
}
// Lock whichever directory actually gets written: the source itself in
// --in-place mode, otherwise the (distinct) --output directory. Acquired
// before opening the source so a concurrent writer can't slip in between
// the open and the write below.
let lock_target = if args.in_place { &args.source } else { args.output.as_ref().unwrap() };
let _lock = obisys::DirLock::acquire(lock_target).unwrap_or_else(|e| {
eprintln!("error locking {}: {e}", lock_target.display());
std::process::exit(1);
});
let mut src = KmerIndex::open(&args.source).unwrap_or_else(|e| {
let src = KmerIndex::open(&args.source).unwrap_or_else(|e| {
eprintln!("error opening source index: {e}");
std::process::exit(1);
});
let group_preds: Vec<(String, String)> =
args.group.iter().map(|s| parse_name_value(s, "--group")).collect();
let group_ops: Vec<(String, String)> =
args.group_op.iter().map(|s| parse_name_value(s, "--group-op")).collect();
let src_is_count = src.meta().config.with_counts;
let (specs, output_presence) = build_specs(&args, &src.meta(), src_is_count);
let (specs, output_presence) = build_output_cols(
&src.meta(),
ColumnSpecParams {
group_preds: &group_preds,
aggregate_by: args.aggregate_by.as_deref(),
group_ops: &group_ops,
aggregate_op: args.aggregate_op.map(AggOp::from),
select: args.select.as_deref(),
src_is_count,
},
)
.unwrap_or_else(|e| {
eprintln!("error building output columns: {e}");
std::process::exit(1);
});
let n_genomes = src.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
@@ -249,23 +123,22 @@ pub fn run(args: SelectArgs) {
);
let mut rep = Reporter::new();
let t = Stage::start("select");
let pb = progress_bar("select", src.n_partitions() as u64, "partitions");
let mut alg = Select::new(&src, &args.output, &specs, output_presence)
.threshold(args.presence_threshold)
.force(args.force)
.sparse(!args.dense)
.on_progress(|_: Progress| pb.inc(1));
if args.in_place {
src.select_in_place(&specs, args.presence_threshold, output_presence, &mut rep)
.unwrap_or_else(|e| {
let dst = alg.run().unwrap_or_else(|e| {
eprintln!("select error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
info!("selected index → {}", dst.dir().display());
alg.reporter().print();
rep.print();
info!("selected in-place → {}", args.source.display());
} else {
let output = args.output.unwrap();
KmerIndex::select(&output, &src, &specs, args.presence_threshold, output_presence, args.force, &mut rep)
.unwrap_or_else(|e| {
eprintln!("select error: {e}");
std::process::exit(1);
});
rep.print();
info!("selected index → {}", output.display());
}
}
+20 -74
View File
@@ -1,17 +1,15 @@
use std::io::{self, BufWriter, Write};
use std::io::{self, BufWriter};
use std::path::PathBuf;
use std::sync::Mutex;
use std::sync::atomic::{AtomicUsize, Ordering};
use std::sync::Arc;
use clap::Args;
use obidebruinj::GraphDeBruijn;
use obifastwrite::write_unitig;
use obikdump::IndexUnitigs;
use obikfilter::KmerFilter;
use obikindex::KmerIndex;
use obisys::{Reporter, Stage, progress_bar, spinner};
use rayon::prelude::*;
use obisys::progress_bar;
use tracing::info;
use super::predicate::FilterArgs;
use super::predicate::GroupFilterArgs;
#[derive(Args)]
pub struct UnitigArgs {
@@ -19,83 +17,31 @@ pub struct UnitigArgs {
pub index: PathBuf,
#[command(flatten)]
pub filter: FilterArgs,
pub group_filter: GroupFilterArgs,
}
pub fn run(args: UnitigArgs) {
let idx = KmerIndex::open(&args.index).unwrap_or_else(|e| {
let idx = Arc::new(KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
});
}));
let k = idx.kmer_size();
let n = idx.n_partitions();
let n_genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
}).len().max(1);
let use_counts = idx.meta().config.with_counts;
info!(
"unitig: building de Bruijn graph from {} partition(s) (k={})",
idx.n_partitions(),
idx.kmer_size(),
);
info!("unitig: building de Bruijn graph from {n} partition(s) (k={k})");
let filters: Vec<Box<dyn KmerFilter>> = vec![Box::new(args.group_filter.build_filter(&idx.meta()))];
let pb = progress_bar("unitig", idx.n_partitions() as u64, "partitions");
let filters = args.filter.build_filters(&idx.meta());
let mut rep = Reporter::new();
let mut out = BufWriter::new(io::stdout());
// ── Phase 1 : collect filtered kmers in parallel ──────────────────────────
let pb = progress_bar("unitig", n as u64, "partitions");
let stage = Stage::start("build graph");
let g = (0..n)
.into_par_iter()
.fold(GraphDeBruijn::new, |mut local_g, i| {
idx
.iter_partition_kmers(i, use_counts, n_genomes, &filters, |kmer, _row| {
local_g.push(kmer);
true
})
.unwrap_or_else(|e| {
eprintln!("error reading partition {i}: {e}");
let n = idx.write_unitigs(&mut out, &filters, || pb.inc(1)).unwrap_or_else(|e| {
eprintln!("unitig error: {e}");
std::process::exit(1);
});
pb.inc(1);
local_g
})
.reduce(GraphDeBruijn::new, |mut a, b| {
a.merge(b);
a
});
pb.finish_and_clear();
rep.push(stage.stop());
info!("unitig: {} distinct k-mers", g.len());
// ── Phase 2 : compute degrees ─────────────────────────────────────────────
let pb = spinner("degrees");
let stage = Stage::start("compute degrees");
g.compute_degrees_and_mark_starts();
pb.finish_and_clear();
rep.push(stage.stop());
// ── Phase 3 : enumerate unitigs and write as FASTA ───────────────────────
let pb = spinner("unitig");
let out = Mutex::new(BufWriter::new(io::stdout()));
let j = AtomicUsize::new(0);
let stage = Stage::start("enumerate unitigs");
g.for_each_unitig(|nuc_iter| {
let unitig: obikseq::unitig::Unitig = nuc_iter.collect();
let idx = j.fetch_add(1, Ordering::Relaxed);
let mut w = out.lock().unwrap();
write_unitig(&unitig, k, 0, idx, &mut *w).unwrap_or_else(|e| {
eprintln!("write error: {e}");
std::process::exit(1);
});
if idx % 10_000 == 0 {
pb.set_message(format!("{idx} unitigs written"));
}
});
pb.finish_and_clear();
rep.push(stage.stop());
out.into_inner().unwrap().flush().expect("flush error");
rep.print();
info!("unitig: {n} unitig(s) written");
}
+16 -22
View File
@@ -1,22 +1,26 @@
use std::path::PathBuf;
use std::sync::Arc;
use obikalgorithm::Algorithm;
use obikindex::{GenomeInfo, KmerIndex};
use obikstats::IndexBitsPerKmer;
use obikstats::{BitsPerKmer, GenomeKmerCounts};
use tracing::info;
pub(super) fn run_stats(index_path: &PathBuf) {
let idx = KmerIndex::open(index_path).unwrap_or_else(|e| {
let idx = Arc::new(KmerIndex::open(index_path).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
});
let (total, per_genome) = idx.genome_kmer_counts().unwrap_or_else(|e| {
eprintln!("error computing stats: {e}");
std::process::exit(1);
});
}));
let genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
});
let (total, per_genome) = GenomeKmerCounts::new(Arc::clone(&idx)).run().unwrap_or_else(|e| {
eprintln!("error computing stats: {e}");
std::process::exit(1);
});
println!("genome,n_kmers");
for (g, &n) in genomes.iter().zip(per_genome.iter()) {
println!("{},{}", g.label, n);
@@ -25,14 +29,16 @@ pub(super) fn run_stats(index_path: &PathBuf) {
}
pub(super) fn run_bits_per_kmer(index_path: &PathBuf) {
let idx = KmerIndex::open(index_path).unwrap_or_else(|e| {
let idx = Arc::new(KmerIndex::open(index_path).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
});
let stats: IndexBitsPerKmer = idx.bits_per_kmer().unwrap_or_else(|e| {
}));
let stats = BitsPerKmer::new(idx).run().unwrap_or_else(|e| {
eprintln!("error computing bits/kmer: {e}");
std::process::exit(1);
});
println!("k-mers : {}", stats.n_kmers);
println!("genomes : {}", stats.n_genomes);
println!("mphf : {:6.2} bits/kmer", stats.mphf);
@@ -44,18 +50,6 @@ pub(super) fn run_bits_per_kmer(index_path: &PathBuf) {
println!("total : {:6.2} bits/kmer", stats.total);
}
pub(super) fn run_upgrade_index(index_path: &PathBuf) {
let idx = KmerIndex::open(index_path).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
});
idx.upgrade_layer_meta().unwrap_or_else(|e| {
eprintln!("upgrade error: {e}");
std::process::exit(1);
});
info!("upgrade-index: layer_meta.json written to all layers that were missing it");
}
pub(super) fn run_rename(index_path: &PathBuf, spec: &str) {
let (old_label, new_label) = parse_rename_spec(spec);
+2 -11
View File
@@ -5,7 +5,7 @@ use std::path::PathBuf;
use clap::Args;
use maintenance::{run_bits_per_kmer, run_stats, run_upgrade_index, run_rename};
use maintenance::{run_bits_per_kmer, run_stats, run_rename};
use partition_stats::run_partition_stats;
#[derive(Args)]
@@ -18,10 +18,6 @@ pub struct UtilsArgs {
#[arg(long, value_name = "NEW=OLD")]
pub new_label: Option<String>,
/// Add missing layer_meta.json files to each layer (single-index only)
#[arg(long)]
pub upgrade_index: bool,
/// Print bits-per-kmer statistics (single-index only)
#[arg(long)]
pub bits_per_kmer: bool,
@@ -47,11 +43,6 @@ pub fn run(args: UtilsArgs) {
run_rename(single_index(&args), spec);
}
if args.upgrade_index {
any = true;
run_upgrade_index(single_index(&args));
}
if args.bits_per_kmer {
any = true;
run_bits_per_kmer(single_index(&args));
@@ -70,7 +61,7 @@ pub fn run(args: UtilsArgs) {
if !any {
eprintln!(
"utils: no operation specified. \
Available: --new-label, --upgrade-index, --bits-per-kmer, --stats, --partition-stats"
Available: --new-label, --bits-per-kmer, --stats, --partition-stats"
);
std::process::exit(1);
}
+9 -5
View File
@@ -24,11 +24,15 @@ fn collect_rows(indexes: &[PathBuf]) -> Vec<PartRow> {
let n_parts = idx.n_partitions();
for i in 0..n_parts {
let mut bytes = 0u64;
for l in 0.. {
let p = idx.layer_unitigs_path(i, l);
if !p.exists() {
break;
}
let n_layers = idx.n_layers(i).unwrap_or_else(|e| {
eprintln!("error reading partition {i} of {}: {e}", path.display());
std::process::exit(1);
});
for l in 0..n_layers {
let p = idx.layer_unitigs_path(i, l).unwrap_or_else(|e| {
eprintln!("error reading layer {l} of partition {i} of {}: {e}", path.display());
std::process::exit(1);
});
if let Ok(m) = std::fs::metadata(&p) {
bytes += m.len();
}
+30 -58
View File
@@ -5,7 +5,7 @@ use clap::{Parser, Subcommand};
use tracing_subscriber::{EnvFilter, fmt};
#[derive(Parser)]
#[command(name = "obikmer", about = "DNA k-mer tools", version)]
#[command(name = "obikmer2", about = "DNA k-mer tools", version)]
struct Cli {
#[command(subcommand)]
command: Commands,
@@ -13,38 +13,35 @@ struct Cli {
#[derive(Subcommand)]
enum Commands {
/// Extract super-kmers from a sequence file and write to stdout
Superkmer(cmd::superkmer::SuperkmerArgs),
/// Build the complete genome index (scatter → dereplicate → count → layered MPHF)
Index(cmd::index::IndexArgs),
/// Merge multiple built indexes into one
/// Extract super-k-mers from input sequences and scatter them by partition
Superkmer(cmd::superkmer::SuperkmerArgs),
/// Merge multiple genome indexes into one
Merge(cmd::merge::MergeArgs),
/// Apply row-level selection (σ) to an index: retain only k-mers matching the predicates
Filter(cmd::filter::FilterCmdArgs),
/// Project and/or aggregate genome columns into a new or in-place index
/// Filter kmers out of an index by genome metadata / abundance / complexity
Filter(cmd::filter::FilterArgs),
/// Project/aggregate genome columns into a new index
Select(cmd::select::SelectArgs),
/// Query an index with sequences and annotate matches
Query(cmd::query::QueryArgs),
/// Dump all indexed kmers as CSV (kmer + per-genome counts or presence)
/// Dump an index's kmers as a CSV table
Dump(cmd::dump::DumpArgs),
/// Add or update genome metadata from a CSV file; or dump metadata as CSV
Annotate(cmd::annotate::AnnotateArgs),
/// Compute pairwise evolutionary-distance proxies between genomes (metric matrix, NJ/UPGMA
/// trees, SNP/Sankoff calibration, TNT/PhyG/IQ-TREE exports)
Phylo(cmd::phylo::PhyloArgs),
/// Translate a numerically-labelled tree export (TNT/PhyG) back to real taxon names, from
/// the FASTA that produced it
NameTree(cmd::nametree::NameTreeArgs),
/// Dump unitigs from a built index to stdout (debug)
/// Query sequences against an index, annotating each with per-genome matches
Query(cmd::query::QueryArgs),
/// Assemble an index's kmers into unitigs and write them as FASTA
Unitig(cmd::unitig::UnitigArgs),
/// Estimate approximate-index parameters (z, evidence bits, FP rates) before indexing
Estimate(cmd::estimate::EstimateArgs),
/// Convert an index's evidence in-place: exact ↔ approx
Reindex(cmd::reindex::ReindexArgs),
/// Miscellaneous index utilities (--rename, …)
Utils(cmd::utils::UtilsArgs),
/// Pack matrix column files into single-file format to reduce query I/O
/// Pack an index's matrices into single-file, sparse format by default (--dense to opt out), in place
Pack(cmd::pack::PackArgs),
/// Estimate approximate-evidence false-positive rates for given parameters
Estimate(cmd::estimate::EstimateArgs),
/// Read/write genome metadata (CSV) on an already-built index
Annotate(cmd::annotate::AnnotateArgs),
/// Maintenance/inspection operations on already-built indexes
Utils(cmd::utils::UtilsArgs),
/// Convert an index's evidence representation (exact/approximate/hybrid), in place
Convert(cmd::convert::ConvertArgs),
/// Genome-vs-genome distance matrix (+ optional NJ/UPGMA tree, sibling-annex-based
/// SNP corrections, Sankoff/TNT/PhyG/IQ-TREE exports)
Phylo(cmd::phylo::PhyloArgs),
}
fn main() {
@@ -55,46 +52,21 @@ fn main() {
.with_writer(std::io::stderr)
.init();
#[cfg(feature = "profiling")]
let _guard = {
let guard = pprof::ProfilerGuardBuilder::default()
.frequency(1000)
.build()
.expect("failed to start pprof profiler");
guard
};
let cli = Cli::parse();
match cli.command {
Commands::Superkmer(args) => cmd::superkmer::run(args),
Commands::Index(args) => cmd::index::run(args),
Commands::Superkmer(args) => cmd::superkmer::run(args),
Commands::Merge(args) => cmd::merge::run(args),
Commands::Dump(args) => cmd::dump::run(args),
Commands::Filter(args) => cmd::filter::run(args),
Commands::Select(args) => cmd::select::run(args),
Commands::Dump(args) => cmd::dump::run(args),
Commands::Query(args) => cmd::query::run(args),
Commands::Annotate(args) => cmd::annotate::run(args),
Commands::Phylo(args) => cmd::phylo::run(args),
Commands::NameTree(args) => cmd::nametree::run(args),
Commands::Unitig(args) => cmd::unitig::run(args),
Commands::Estimate(args) => cmd::estimate::run(args),
Commands::Reindex(args) => cmd::reindex::run(args),
Commands::Utils(args) => cmd::utils::run(args),
Commands::Pack(args) => cmd::pack::run(args),
}
#[cfg(feature = "profiling")]
{
use pprof::protos::Message;
if let Ok(report) = _guard.report().build() {
let mut bytes = Vec::new();
report
.pprof()
.expect("pprof encode failed")
.encode(&mut bytes)
.expect("pprof encode failed");
std::fs::write("profile.pb", &bytes).expect("cannot write profile.pb");
eprintln!("profile written to profile.pb");
}
Commands::Estimate(args) => cmd::estimate::run(args),
Commands::Annotate(args) => cmd::annotate::run(args),
Commands::Utils(args) => cmd::utils::run(args),
Commands::Convert(args) => cmd::convert::run(args),
Commands::Phylo(args) => cmd::phylo::run(args),
}
}
-40
View File
@@ -1,40 +0,0 @@
[package]
name = "obikmer2"
version = "1.2.2"
edition = "2024"
[[bin]]
name = "obikmer2"
path = "src/main.rs"
[dependencies]
obikseq = { path = "../obikseq" }
obiread = { path = "../obiread" }
obipipeline = { path = "../obipipeline" }
obisys = { path = "../obisys" }
obikindex = { path = "../obikindex", default-features = false }
obikindexer = { path = "../obikindexer" }
obikalgorithm = { path = "../obikalgorithm" }
obikmerge = { path = "../obikmerge" }
obikfilter = { path = "../obikfilter" }
obikselect = { path = "../obikselect" }
obikdump = { path = "../obikdump" }
obikrebuild = { path = "../obikrebuild" }
obikstats = { path = "../obikstats" }
obikquery = { path = "../obikquery" }
obikidxcache = { path = "../obikidxcache" }
obikphylo = { path = "../obikphylo" }
obikrope = { path = "../obikrope" }
obifastwrite = { path = "../obifastwrite" }
obiskbuilder = { path = "../obiskbuilder" }
clap = { version = "4", features = ["derive"] }
csv = "1"
ndarray = "0.17"
serde = { version = "1", features = ["derive"] }
serde_yaml = "0.9"
tracing = "0.1.44"
tracing-subscriber = { version = "0.3", features = ["fmt", "env-filter"] }
[features]
default = ["numa"]
numa = ["obisys/numa"]
-114
View File
@@ -1,114 +0,0 @@
use std::path::PathBuf;
use clap::Args;
use obiread::NucPage;
use obikseq::RoutableSuperKmer;
use obipipeline::Throttled;
// ── Shared arguments ──────────────────────────────────────────────────────────
#[derive(Args)]
pub struct CommonArgs {
/// Input files or directories (FASTA/FASTQ, optionally gzip-compressed).
/// If omitted, reads from stdin.
#[arg(num_args = 0..)]
pub inputs: Vec<String>,
/// k-mer size
#[arg(short, long, default_value_t = 31)]
pub kmer_size: usize,
/// Minimizer size
#[arg(short, long, default_value_t = 11)]
pub minimizer_size: usize,
/// Entropy threshold (k-mers with score ≤ theta are rejected)
#[arg(long, default_value_t = 0.7)]
pub theta: f64,
/// Maximum sub-word size for entropy computation
#[arg(long, default_value_t = 6)]
pub level_max: usize,
/// Number of partitions (rounded up to the next power of 2)
#[arg(short, long, default_value_t = 256)]
pub partitions: usize,
/// Number of worker threads
#[arg(
short = 'T',
long,
default_value_t = obisys::effective_parallelism()
)]
pub threads: usize,
/// Maximum number of input files open simultaneously.
/// Defaults to threads/4 (minimum 1). Keep below the number of workers
/// to ensure CPU workers are always available for the transform stage.
#[arg(long)]
pub max_open_files: Option<usize>,
}
/// Smallest `b` such that `2^b >= n` (i.e. `n.next_power_of_two().ilog2()`).
/// Minimum 1 (degenerate n=0 or n=1 → 1 partition).
pub fn partitions_to_bits(n: usize) -> usize {
n.max(1).next_power_of_two().trailing_zeros() as usize
}
/// Convert a block size (number of unitigs per block) to its `block_bits` exponent.
/// `block_size=1` → `block_bits=0` (one entry per unitig, O(1) random access).
pub fn block_size_to_bits(n: usize) -> u8 {
n.max(1).next_power_of_two().trailing_zeros() as u8
}
impl CommonArgs {
/// Validate k and m constraints. Exits on error.
pub fn validate(&self) {
let k = self.kmer_size;
let m = self.minimizer_size;
if k < 11 || k > 31 {
eprintln!("error: --kmer-size must be in [11, 31] (got {k})");
std::process::exit(1);
}
if k % 2 == 0 {
eprintln!("error: --kmer-size must be odd (got {k}); even k allows palindromic k-mers");
std::process::exit(1);
}
if m < 3 || m >= k {
eprintln!("error: --minimizer-size must be in [3, k−1] = [3, {}] (got {m})", k - 1);
std::process::exit(1);
}
if m % 2 == 0 {
eprintln!("error: --minimizer-size must be odd (got {m})");
std::process::exit(1);
}
}
pub fn effective_max_open(&self) -> usize {
self.max_open_files
.unwrap_or_else(|| (self.threads / 4).max(1))
.max(1)
}
pub fn seqfile_paths(&self) -> obiread::PathIter {
let paths: Vec<PathBuf> = if self.inputs.is_empty() {
vec![PathBuf::from("-")]
} else {
self.inputs.iter().map(PathBuf::from).collect()
};
obiread::PathIter::new(paths)
}
}
// ── Pipeline data carrier ─────────────────────────────────────────────────────
pub enum PipelineData {
Path(Throttled<PathBuf>),
NucPage(NucPage),
Batch(Vec<RoutableSuperKmer>),
}
unsafe impl Send for PipelineData {}
unsafe impl Sync for PipelineData {}
-184
View File
@@ -1,184 +0,0 @@
use std::collections::HashSet;
use std::io::{self, BufWriter, Write};
use std::path::PathBuf;
use clap::Args;
use obikindex::KmerIndex;
use tracing::info;
#[derive(Args)]
pub struct AnnotateArgs {
/// Index directory to annotate (modified in-place)
pub index: PathBuf,
/// CSV file with genome metadata (must contain an id column)
#[arg(long)]
pub csv: Option<PathBuf>,
/// CSV field separator
#[arg(long, default_value = ",")]
pub sep: char,
/// Name of the column that contains genome labels
#[arg(long, default_value = "id")]
pub id_col: String,
/// Value that means "delete / absent" (removes existing key if present)
#[arg(long, default_value = "NA")]
pub na_value: String,
/// Do not overwrite existing metadata keys
#[arg(long)]
pub no_overwrite: bool,
/// Dump all genome metadata as CSV (stdout) instead of reading a CSV
#[arg(long)]
pub dump: bool,
}
pub fn run(args: AnnotateArgs) {
if args.dump {
run_dump(&args);
} else {
run_annotate(&args);
}
}
fn run_dump(args: &AnnotateArgs) {
let idx = open_index(&args.index);
let genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
});
let genomes = &genomes;
// Collect all keys in stable order (sorted for determinism)
let mut key_set: HashSet<String> = HashSet::new();
for g in genomes {
for k in g.meta.keys() {
key_set.insert(k.clone());
}
}
let mut keys: Vec<String> = key_set.into_iter().collect();
keys.sort();
let stdout = io::stdout();
let mut out = BufWriter::new(stdout.lock());
// Header
write!(out, "id").unwrap();
for k in &keys {
write!(out, "{}{k}", args.sep).unwrap();
}
writeln!(out).unwrap();
// Rows
for g in genomes {
write!(out, "{}", g.label).unwrap();
for k in &keys {
let v = g.meta.get(k).map(|s| s.as_str()).unwrap_or("NA");
write!(out, "{}{v}", args.sep).unwrap();
}
writeln!(out).unwrap();
}
}
fn run_annotate(args: &AnnotateArgs) {
let csv_path = match &args.csv {
Some(p) => p.clone(),
None => {
eprintln!("error: --csv is required unless --dump is used");
std::process::exit(1);
}
};
let idx = open_index(&args.index);
let mut genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
});
// Build a label → genome index position map
let label_to_pos: std::collections::HashMap<String, usize> = genomes
.iter()
.enumerate()
.map(|(i, g)| (g.label.clone(), i))
.collect();
let sep = args.sep as u8;
let mut rdr = csv::ReaderBuilder::new()
.delimiter(sep)
.from_path(&csv_path)
.unwrap_or_else(|e| {
eprintln!("error opening {}: {e}", csv_path.display());
std::process::exit(1);
});
let headers = rdr
.headers()
.unwrap_or_else(|e| {
eprintln!("error reading CSV headers: {e}");
std::process::exit(1);
})
.clone();
let id_col_idx = headers.iter().position(|h| h == args.id_col).unwrap_or_else(|| {
eprintln!("error: id column '{}' not found in CSV", args.id_col);
std::process::exit(1);
});
let meta_cols: Vec<(usize, String)> = headers
.iter()
.enumerate()
.filter(|(i, _)| *i != id_col_idx)
.map(|(i, h)| (i, h.to_string()))
.collect();
let mut updated = 0usize;
let mut skipped = 0usize;
for result in rdr.records() {
let record = result.unwrap_or_else(|e| {
eprintln!("error reading CSV record: {e}");
std::process::exit(1);
});
let label = record.get(id_col_idx).unwrap_or("").to_string();
let pos = match label_to_pos.get(&label) {
Some(&p) => p,
None => {
skipped += 1;
continue;
}
};
let genome = &mut genomes[pos];
for (col_idx, key) in &meta_cols {
let val = record.get(*col_idx).unwrap_or("");
if val == args.na_value {
genome.meta.remove(key);
} else if args.no_overwrite && genome.meta.contains_key(key) {
// skip
} else {
genome.meta.insert(key.clone(), val.to_string());
}
}
updated += 1;
}
idx.meta().set_genomes(genomes).unwrap_or_else(|e| {
eprintln!("error writing index metadata: {e}");
std::process::exit(1);
});
info!("annotated {updated} genome(s), skipped {skipped} CSV row(s) with unknown label");
}
fn open_index(path: &PathBuf) -> KmerIndex {
KmerIndex::open(path).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
})
}
-63
View File
@@ -1,63 +0,0 @@
use std::io::{self, BufWriter};
use std::path::PathBuf;
use std::sync::Arc;
use clap::Args;
use obikdump::IndexDump;
use obikfilter::KmerFilter;
use obikindex::KmerIndex;
use obisys::progress_bar;
use tracing::info;
use super::predicate::GroupFilterArgs;
#[derive(Args)]
pub struct DumpArgs {
/// Index directory to dump
pub index: PathBuf,
/// Output presence/absence (0/1) even if the index stores counts
#[arg(long, default_value_t = false)]
pub force_presence: bool,
/// Prepend partition and layer columns to each row
#[arg(long, default_value_t = false)]
pub debug: bool,
/// Only output the first N kmers
#[arg(long)]
pub head: Option<usize>,
#[command(flatten)]
pub group_filter: GroupFilterArgs,
}
pub fn run(args: DumpArgs) {
let idx = Arc::new(KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
}));
let n_genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
}).len();
info!(
"dumping {} partition(s), {} genome(s)",
idx.n_partitions(),
n_genomes
);
let filters: Vec<Box<dyn KmerFilter>> = vec![Box::new(args.group_filter.build_filter(&idx.meta()))];
let pb = progress_bar("dump", idx.n_partitions() as u64, "partitions");
let stdout = io::stdout();
let mut out = BufWriter::new(stdout.lock());
idx.dump(&mut out, args.force_presence, args.debug, args.head, &filters, || pb.inc(1))
.unwrap_or_else(|e| {
eprintln!("dump error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
}
-38
View File
@@ -1,38 +0,0 @@
use clap::Args;
use super::index::resolve_approx_params;
#[derive(Args)]
pub struct EstimateArgs {
/// k-mer size used for querying (same as --kmer-size in index)
#[arg(short = 'k', long, default_value_t = 31)]
pub kmer_size: usize,
/// Findere z parameter: number of consecutive k-mers that must all match.
/// Effective indexed k-mer size is kmer_size - z + 1.
#[arg(short = 'z', long, default_value = None)]
pub findere_z: Option<u8>,
/// Fingerprint bits per slot (b). FP per z-window = 1/2^(b·z).
#[arg(long, default_value = None)]
pub evidence_bits: Option<u8>,
/// Target false-positive rate per z-window (e.g. 0.01).
#[arg(long, default_value = None)]
pub fp: Option<f64>,
}
pub fn run(args: EstimateArgs) {
let (z, b, fp_window) = resolve_approx_params(args.findere_z, args.evidence_bits, args.fp);
let k_query = args.kmer_size;
let k_index = k_query.saturating_sub(z as usize - 1);
let fp_kmer = 1.0_f64 / 2_f64.powi(b as i32);
println!("{:<22} {}", "k (query):", k_query);
println!("{:<22} {}", "k (indexed):", k_index);
println!("{:<22} {}", "z:", z);
println!("{:<22} {}", "evidence bits (b):", b);
println!("{:<22} {:.3e} (1/2^{})", "FP per k-mer:", fp_kmer, b);
println!("{:<22} {:.3e} (1/2^{})", "FP per z-window:", fp_window, b as u32 * z as u32);
}
-100
View File
@@ -1,100 +0,0 @@
use std::path::PathBuf;
use std::sync::Arc;
use clap::Args;
use obikalgorithm::Algorithm;
use obikfilter::{Filter, KmerFilter, MaxTotalCount, MinComplexity, MinTotalCount};
use obikindex::KmerIndex;
use obisys::{Progress, Reporter, Stage, progress_bar};
use tracing::info;
use super::predicate::GroupFilterArgs;
#[derive(Args)]
pub struct FilterArgs {
/// Source index directory
pub source: PathBuf,
/// Output index directory
#[arg(short, long)]
pub output: PathBuf,
#[command(flatten)]
pub group_filter: GroupFilterArgs,
/// Minimum total count across all genomes (count index only)
#[arg(long)]
pub min_total_count: Option<u32>,
/// Maximum total count across all genomes (count index only)
#[arg(long)]
pub max_total_count: Option<u32>,
/// Minimum normalized entropy (complexity) to keep a k-mer
#[arg(long)]
pub min_complexity: Option<f64>,
/// Maximum sub-word size for the complexity computation (only used when --min-complexity is set)
#[arg(long, default_value_t = 6)]
pub complexity_level_max: usize,
/// Output as presence/absence instead of counts
#[arg(long)]
pub presence: bool,
/// Pack the output's presence matrices in the dense format instead of the default sparse one
#[arg(long, default_value_t = false)]
pub dense: bool,
/// Overwrite existing output directory
#[arg(short, long)]
pub force: bool,
}
pub fn run(args: FilterArgs) {
let src = Arc::new(KmerIndex::open(&args.source).unwrap_or_else(|e| {
eprintln!("error opening source index: {e}");
std::process::exit(1);
}));
let mut filters: Vec<Box<dyn KmerFilter>> =
vec![Box::new(args.group_filter.build_filter(&src.meta()))];
if let Some(v) = args.min_total_count {
filters.push(Box::new(MinTotalCount { total: v }));
}
if let Some(v) = args.max_total_count {
filters.push(Box::new(MaxTotalCount { total: v }));
}
if let Some(theta) = args.min_complexity {
filters.push(Box::new(MinComplexity { level_max: args.complexity_level_max, theta }));
}
let n_genomes = src.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
}).len();
info!(
"filter: {} genome(s), source={}",
n_genomes, args.source.display()
);
let mut rep = Reporter::new();
let t = Stage::start("filter");
let pb = progress_bar("filter", src.n_partitions() as u64, "partitions");
let mut alg = Filter::new(Arc::clone(&src), &args.output, &filters)
.presence(args.presence)
.force(args.force)
.sparse(!args.dense)
.on_progress(|_: Progress| pb.inc(1));
let dst = alg.run().unwrap_or_else(|e| {
eprintln!("error filtering index: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
info!("filtered index → {}", dst.dir().display());
alg.reporter().print();
rep.print();
}
-353
View File
@@ -1,353 +0,0 @@
use std::path::PathBuf;
use std::time::Instant;
use clap::Args;
use obikalgorithm::Algorithm;
use obikindex::layer::IndexMode;
use obikindex::{GenomeInfo, IndexBuilder, IndexConfig, IndexState, KmerIndex};
use obikindexer::algorithms::counter::Counter;
use obikindexer::algorithms::dereplicator::Dereplicator;
use obikindexer::algorithms::layer_builder::LayerBuilder;
use obikindexer::algorithms::partitionner::PartitionRouter;
fn current_state(idx: &KmerIndex) -> IndexState {
idx.state().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
})
}
fn parse_key_value(s: &str) -> Result<(String, String), String> {
let pos = s
.find('=')
.ok_or_else(|| format!("invalid key=value: no '=' in '{s}'"))?;
Ok((s[..pos].to_string(), s[pos + 1..].to_string()))
}
use obisys::{Progress, Reporter, Stage, progress_bar, spinner};
use tracing::info;
use crate::cli::{CommonArgs, block_size_to_bits, partitions_to_bits};
#[derive(Args)]
pub struct IndexArgs {
/// Output index directory
#[arg(short, long)]
pub output: PathBuf,
/// Overwrite output directory if it already exists
#[arg(long, default_value_t = false)]
pub force: bool,
/// Genome label (default: input filename without path/extension)
#[arg(long)]
pub label: Option<String>,
/// Genome categorical metadata as key=value pairs (repeatable)
#[arg(long = "meta", value_parser = parse_key_value)]
pub meta: Vec<(String, String)>,
/// Minimum kmer abundance (inclusive)
#[arg(long, default_value_t = 1)]
pub min_abundance: u32,
/// Maximum kmer abundance (inclusive)
#[arg(long)]
pub max_abundance: Option<u32>,
/// Store kmer counts in the index (default: set membership only)
#[arg(long, default_value_t = false)]
pub with_counts: bool,
/// Keep intermediate build files (dereplicated superkmers, mphf1, counts1)
#[arg(long, default_value_t = false)]
pub keep_intermediate: bool,
/// Use approximate (fingerprint-based) evidence instead of exact evidence.
/// False-positive rate per z-window: 1/2^(b·z).
#[arg(long, default_value_t = false)]
pub approx: bool,
/// Findere z parameter: number of consecutive k-mers that must all match.
/// Effective indexed k-mer size is kmer_size - z + 1.
#[arg(short = 'z', long, default_value = None)]
pub findere_z: Option<u8>,
/// Fingerprint bits per slot (b). FP per z-window = 1/2^(b·z).
#[arg(long, default_value = None)]
pub evidence_bits: Option<u8>,
/// Target false-positive rate per z-window (e.g. 0.01).
/// Used to derive missing b or z.
#[arg(long, default_value = None)]
pub fp: Option<f64>,
/// Block size for exact evidence `.idx` (number of unitigs per block).
/// Must be a power of two; rounded up if not. Default 1 = O(1) random access.
#[arg(long, default_value_t = 1)]
pub block_size: usize,
#[command(flatten)]
pub common: CommonArgs,
}
/// Resolve the (z, b, fp) triplet from the user-supplied subset.
///
/// Model: FP = 1/2^(b·z) ⟹ b·z = ⌈-log₂(fp)⌉
///
/// Rules when one value is missing (conservative = ceiling):
/// given z, b → fp = 1/2^(b·z)
/// given z, fp → b = ⌈-log₂(fp) / z⌉
/// given b, fp → z = ⌈-log₂(fp) / b⌉
/// given z only → b = 8 (default), fp derived
/// given b only → z = 1 (default), fp derived
/// given fp only → b = 8 (default), z derived
/// none given → z = 1, b = 8, fp = 1/256
pub(crate) fn resolve_approx_params(
z_opt: Option<u8>,
b_opt: Option<u8>,
fp_opt: Option<f64>,
) -> (u8, u8, f64) {
const DEFAULT_B: u8 = 8;
const DEFAULT_Z: u8 = 1;
let bits_needed = |fp: f64| -> u8 { (-fp.log2()).ceil() as u8 };
match (z_opt, b_opt, fp_opt) {
// All three given: use b and z, recompute fp conservatively.
(Some(z), Some(b), Some(_fp)) => {
let fp = 1.0_f64 / (1u64 << (b as u32 * z as u32)) as f64;
(z, b, fp)
}
// Two given, derive third.
(Some(z), Some(b), None) => {
let fp = 1.0_f64 / (1u64 << (b as u32 * z as u32)) as f64;
(z, b, fp)
}
(Some(z), None, Some(fp)) => {
let bz = (-fp.log2()).ceil() as u32;
let b = ((bz + z as u32 - 1) / z as u32).max(1) as u8;
let actual_fp = 1.0_f64 / (1u64 << (b as u32 * z as u32)) as f64;
(z, b, actual_fp)
}
(None, Some(b), Some(fp)) => {
let bz = (-fp.log2()).ceil() as u32;
let z = ((bz + b as u32 - 1) / b as u32).max(1) as u8;
let actual_fp = 1.0_f64 / (1u64 << (b as u32 * z as u32)) as f64;
(z, b, actual_fp)
}
// One given, apply defaults for the other.
(Some(z), None, None) => {
let b = DEFAULT_B;
let fp = 1.0_f64 / (1u64 << (b as u32 * z as u32)) as f64;
(z, b, fp)
}
(None, Some(b), None) => {
let z = DEFAULT_Z;
let fp = 1.0_f64 / (1u64 << (b as u32 * z as u32)) as f64;
(z, b, fp)
}
(None, None, Some(fp)) => {
let b = DEFAULT_B;
let z = ((bits_needed(fp) as u32 + b as u32 - 1) / b as u32).max(1) as u8;
let actual_fp = 1.0_f64 / (1u64 << (b as u32 * z as u32)) as f64;
(z, b, actual_fp)
}
// None given: defaults.
(None, None, None) => {
let b = DEFAULT_B;
let z = DEFAULT_Z;
let fp = 1.0_f64 / (1u64 << (b as u32 * z as u32)) as f64;
(z, b, fp)
}
}
}
pub fn run(args: IndexArgs) {
args.common.validate();
let output = args.output.clone();
let mut rep = Reporter::new();
// ── Resolve evidence kind ────────────────────────────────────────────────
let (evidence, effective_kmer_size) = if args.approx {
let (z, b, fp) = resolve_approx_params(args.findere_z, args.evidence_bits, args.fp);
let k = args.common.kmer_size;
if z as usize >= k {
eprintln!(
"error: Findere z={z} must be < kmer-size={k} \
(effective kmer size kz+1 = {} 0)",
k as isize - z as isize + 1
);
std::process::exit(1);
}
let s = k - z as usize + 1;
info!("approximate evidence: b={b}, z={z}, fp={fp:.2e}, indexed kmer size={s}");
(IndexMode::Approx { b, z }, s)
} else {
(IndexMode::Exact, args.common.kmer_size)
};
// ── Open or create the index ─────────────────────────────────────────────
if KmerIndex::is_an_index(&output) {
if !args.force {
eprintln!(
"error: an index already exists at {} (use --force to overwrite it)",
output.display()
);
std::process::exit(1);
}
info!("--force: removing existing index at {}", output.display());
std::fs::remove_dir_all(&output).unwrap_or_else(|e| {
eprintln!("error removing existing index: {e}");
std::process::exit(1);
});
} else if output.exists() {
eprintln!(
"error: {} exists but is not an obikmer index, it cannot be deleted",
output.display()
);
std::process::exit(1);
}
let n_bits = partitions_to_bits(args.common.partitions);
let effective = 1usize << n_bits;
if effective != args.common.partitions {
info!(
"partitions: {} → {} (next power of 2)",
args.common.partitions, effective
);
}
let block_bits = block_size_to_bits(args.block_size);
let config = IndexConfig {
kmer_size: effective_kmer_size,
minimizer_size: args.common.minimizer_size,
n_bits,
with_counts: args.with_counts,
evidence: evidence.clone(),
block_bits,
};
let genome_info = args.label.as_ref().map(|label| {
GenomeInfo::validate_label(label).unwrap_or_else(|e| {
eprintln!("error: --label: {e}");
std::process::exit(1);
});
let mut info = GenomeInfo::new(label.clone());
for (k, v) in &args.meta {
info.meta.insert(k.clone(), v.clone());
}
info
});
let idx = KmerIndex::create(&output, config, genome_info).unwrap_or_else(|e| {
eprintln!("error creating index: {e}");
std::process::exit(1);
});
// ── Stage 1: scatter ─────────────────────────────────────────────────────
if current_state(&idx) < IndexState::Scattered {
let n_workers = args.common.threads.max(1);
let max_open = args.common.effective_max_open();
let t = Stage::start("scatter");
let pb = spinner("scatter");
let mut ema_rate: f64 = 0.0;
let mut last_t = Instant::now();
let mut last_bases: u64 = 0;
const ALPHA: f64 = 0.15;
let mut router = PartitionRouter::new(&idx)
.level_max(args.common.level_max)
.theta(args.common.theta)
.workers(n_workers)
.max_open(max_open)
.files(args.common.seqfile_paths())
.on_progress(|p: Progress| {
let now = Instant::now();
let dt = now.duration_since(last_t).as_secs_f64();
if dt > 0.0 {
let instant = (p.position - last_bases) as f64 / dt;
ema_rate = ALPHA * instant + (1.0 - ALPHA) * ema_rate;
}
last_t = now;
last_bases = p.position;
let bp = p.position as f64;
let (count_str, rate_str) = if bp >= 1e9 {
(
format!("{:.2} Gbp", bp / 1e9),
format!("{:.0} Mbp/s", ema_rate / 1e6),
)
} else {
(
format!("{:.0} Mbp", bp / 1e6),
format!("{:.0} Mbp/s", ema_rate / 1e6),
)
};
pb.set_message(format!("{count_str} {rate_str}"));
});
router.run().unwrap_or_else(|e| {
eprintln!("error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
drop(router); // ends the borrow of `idx` early — `PartitionRouter`'s `Drop` impl would otherwise extend it to the end of scope (`run()` already called `close()`, which marks scatter done, internally)
} else {
info!("scatter already done, skipping");
}
// ── Stage 2: dereplicate + count ─────────────────────────────────────────
if current_state(&idx) < IndexState::Counted {
let t = Stage::start("dereplicate");
let pb = progress_bar("dereplication", idx.n_partitions() as u64, "partitions");
Dereplicator::new(&idx)
.on_progress(|_: Progress| pb.inc(1))
.run()
.unwrap_or_else(|e| {
eprintln!("error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
let t = Stage::start("count_kmer");
let pb = progress_bar("counting", idx.n_partitions() as u64, "partitions");
// `Counter::run` writes `spectrums/{label}.json` and marks count
// done (`count.done`) internally once every partition succeeds.
Counter::new(&idx)
.keep_partial(args.keep_intermediate)
.on_progress(|_: Progress| pb.inc(1))
.run()
.unwrap_or_else(|e| {
eprintln!("error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
} else {
info!("dereplicate+count already done, skipping");
}
// ── Stage 3: build layered index ─────────────────────────────────────────
if current_state(&idx) < IndexState::Indexed {
let t = Stage::start("index");
let pb = progress_bar("index", idx.n_partitions() as u64, "partitions");
let total_kmers = LayerBuilder::new(&idx)
.min_abundance(args.min_abundance)
.max_abundance(args.max_abundance)
.keep_intermediate(args.keep_intermediate)
.on_progress(|_: Progress| pb.inc(1))
.run()
.unwrap_or_else(|e| {
eprintln!("error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
info!("done — {total_kmers} total kmers indexed");
rep.push(t.stop());
// `LayerBuilder::run` marks the index done (`index.done`) internally
// once every partition succeeds.
} else {
info!("index already built, skipping");
}
rep.print();
}
-109
View File
@@ -1,109 +0,0 @@
use std::path::PathBuf;
use clap::Args;
use obikalgorithm::Algorithm;
use obikindex::KmerIndex;
use obikmerge::{Merge, MergeMode};
use obisys::{Progress, Reporter, Stage, progress_bar};
use tracing::info;
#[derive(Args)]
pub struct MergeArgs {
/// Source index directories to merge
#[arg(required = true)]
pub sources: Vec<PathBuf>,
/// Output index directory
#[arg(short, long)]
pub output: PathBuf,
/// Overwrite output directory if it already exists
#[arg(long, default_value_t = false)]
pub force: bool,
/// Force presence/absence mode even if all sources have count data
#[arg(long, default_value_t = false)]
pub force_presence: bool,
/// Disambiguate duplicate genome labels by appending .1, .2, … instead of erroring
#[arg(long, default_value_t = false)]
pub rename_duplicates: bool,
/// Pack the output's presence matrices in the dense format instead of the default sparse one
#[arg(long, default_value_t = false)]
pub dense: bool,
}
pub fn run(args: MergeArgs) {
let sources: Vec<KmerIndex> = args
.sources
.iter()
.map(|p| {
info!("opening source index: {}", p.display());
KmerIndex::open(p).unwrap_or_else(|e| {
eprintln!("error opening source index {}: {e}", p.display());
std::process::exit(1);
})
})
.collect();
// Auto-detect mode: count if all sources have count data, presence otherwise.
// --force-presence overrides to presence regardless.
let all_have_counts = sources.iter().all(|s| s.meta().config.with_counts);
let mode = if !args.force_presence && all_have_counts {
MergeMode::Count
} else {
MergeMode::Presence
};
info!(
"merge mode: {}",
if mode == MergeMode::Count {
"count"
} else {
"presence/absence"
}
);
let source_refs: Vec<&KmerIndex> = sources.iter().collect();
let n_genomes: usize = sources
.iter()
.map(|s| {
s.meta()
.genomes()
.unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
})
.len()
})
.sum();
info!(
"merging {} index(es), {} genome(s) total → {}",
sources.len(),
n_genomes,
args.output.display()
);
let mut rep = Reporter::new();
let t = Stage::start("merge");
let n_partitions = source_refs.first().map(|s| s.n_partitions()).unwrap_or(0);
let pb = progress_bar("merge", n_partitions as u64, "partitions");
let mut merge = Merge::new(&source_refs, &args.output, mode)
.force(args.force)
.rename_duplicates(args.rename_duplicates)
.sparse(!args.dense)
.on_progress(|_: Progress| pb.inc(1));
let dst = merge.run().unwrap_or_else(|e| {
eprintln!("error merging: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
info!("merge done — output at {}", dst.dir().display());
merge.reporter().print();
rep.print();
}
-15
View File
@@ -1,15 +0,0 @@
pub mod annotate;
pub mod convert;
pub mod dump;
pub mod estimate;
pub mod filter;
pub mod index;
pub mod merge;
pub mod pack;
mod predicate;
pub mod phylo;
pub mod query;
pub mod select;
pub mod superkmer;
pub mod unitig;
pub mod utils;
-72
View File
@@ -1,72 +0,0 @@
use std::path::PathBuf;
use std::sync::Arc;
use clap::Args;
use obikindex::KmerIndex;
use obikrebuild::IndexCompact;
use obisys::{Reporter, Stage, progress_bar};
use tracing::info;
#[derive(Args)]
pub struct PackArgs {
/// Index directory to pack
pub index: PathBuf,
/// Compact every partition's accumulated layers into one before packing
/// — undoes the multi-layer stopgap `merge` leaves behind.
#[arg(long, default_value_t = false)]
pub compact_layers: bool,
/// Pack presence and count matrices into the dense on-disk format instead
/// of the default sparse, deduplicated one. Dense is faster for
/// column-oriented access (`--metric` distance matrices); sparse is
/// smaller and faster for single-row access on real, sparse data.
#[arg(long, default_value_t = false)]
pub dense: bool,
}
pub fn run(args: PackArgs) {
// Modifies the index in place; acquired before opening so a concurrent
// writer can't slip in between the open and the pack below.
let _lock = obisys::DirLock::acquire(&args.index).unwrap_or_else(|e| {
eprintln!("error locking index directory {}: {e}", args.index.display());
std::process::exit(1);
});
let idx = Arc::new(KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
}));
let n_genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
}).len();
info!(
"pack: {} partition(s), {} genome(s)",
idx.n_partitions(),
n_genomes,
);
let mut rep = Reporter::new();
if args.compact_layers {
let t = Stage::start("compact layers");
let pb = progress_bar("compact", idx.n_partitions() as u64, "partitions");
idx.compact_layers(|| pb.inc(1)).unwrap_or_else(|e| {
eprintln!("compact error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
}
let t = Stage::start("pack");
idx.pack_matrices(!args.dense).unwrap_or_else(|e| {
eprintln!("pack error: {e}");
std::process::exit(1);
});
rep.push(t.stop());
rep.print();
}
-332
View File
@@ -1,332 +0,0 @@
use std::path::PathBuf;
use clap::Args;
use obikphylo::{DistanceMetric, SnpDistanceKind};
/// `--distance` value — either one of `obikphylo::DistanceMetric`'s
/// whole-index metrics (routed to `IndexCache::distance`) or one of
/// `obikphylo::SnpDistanceKind`'s `snp-*` corrections (routed to
/// `SiblingExt::snp_distance`, the sibling-annex pipeline) — two genuinely
/// different code paths behind one CLI vocabulary, see
/// `DevDocMD/theory/evolutionary_distances.md`, "`--distance` unification".
#[derive(clap::ValueEnum, Clone, Copy, Debug)]
pub enum DistanceArg {
Jaccard,
Mash,
Hamming,
BrayCurtis,
#[value(name = "relfreq-bray-curtis")]
RelfreqBrayCurtis,
Euclidean,
#[value(name = "relfreq-euclidean")]
RelfreqEuclidean,
Hellinger,
#[value(name = "hellinger-euclidean")]
HellingerEuclidean,
#[value(name = "snp-raw")]
SnpRaw,
#[value(name = "snp-jc")]
SnpJc,
#[value(name = "snp-k2p")]
SnpK2p,
#[value(name = "snp-k81")]
SnpK81,
#[value(name = "snp-f81")]
SnpF81,
#[value(name = "snp-t92")]
SnpT92,
#[value(name = "snp-tn93")]
SnpTn93,
#[value(name = "snp-tv")]
SnpTv,
}
impl DistanceArg {
/// `Some` for the whole-index metrics, `None` for `snp-*` values.
pub fn as_classic(self) -> Option<DistanceMetric> {
Some(match self {
DistanceArg::Jaccard => DistanceMetric::Jaccard,
DistanceArg::Mash => DistanceMetric::Mash,
DistanceArg::Hamming => DistanceMetric::Hamming,
DistanceArg::BrayCurtis => DistanceMetric::BrayCurtis,
DistanceArg::RelfreqBrayCurtis => DistanceMetric::RelfreqBrayCurtis,
DistanceArg::Euclidean => DistanceMetric::Euclidean,
DistanceArg::RelfreqEuclidean => DistanceMetric::RelfreqEuclidean,
DistanceArg::Hellinger => DistanceMetric::Hellinger,
DistanceArg::HellingerEuclidean => DistanceMetric::HellingerEuclidean,
_ => return None,
})
}
/// `Some` for the `snp-*` values, `None` for the whole-index metrics.
pub fn as_snp(self) -> Option<SnpDistanceKind> {
Some(match self {
DistanceArg::SnpRaw => SnpDistanceKind::Raw,
DistanceArg::SnpJc => SnpDistanceKind::Jc,
DistanceArg::SnpK2p => SnpDistanceKind::K2p,
DistanceArg::SnpK81 => SnpDistanceKind::K81,
DistanceArg::SnpF81 => SnpDistanceKind::F81,
DistanceArg::SnpT92 => SnpDistanceKind::T92,
DistanceArg::SnpTn93 => SnpDistanceKind::Tn93,
DistanceArg::SnpTv => SnpDistanceKind::Tv,
_ => return None,
})
}
}
/// Partial transfer of `obikmer`'s `phylo` command: the whole-index
/// `--distance` path (classic metrics + `snp-*` corrections/NJ/UPGMA),
/// annex construction (`--sibling-annex`), annex diagnostics
/// (`--sibling-stats`, `--sibling-hist`), entropy reporting (`--shannon`),
/// SNP pseudo-alignment sampling (`--pseudo-alignment`, `--subsample`,
/// `--free-loss`, `--no-ambiguity`, `--entropy`/`--entropy-sd`), Sankoff
/// cost-matrix calibration (`--sankoff`, `--sankoff-ratio-ceiling`) and its
/// TNT/PhyG/IQ-TREE exports (`--tnt`, `--phyg`, `--iqtree`/
/// `--iqtree-min-freq`, `--sankoff-cost-scale`), and Family Overlap
/// (`--family-overlap`, `--min-shared-family`) — everything else
/// sibling-annex-based stays in `obikmer` until the rest of
/// `obikphylo::siblings` is reconnected (see the project memory on this).
#[derive(Args)]
pub struct PhyloArgs {
/// Index directory
pub index: PathBuf,
/// Exclude a genome (by its exact label) — from `--pseudo-alignment`'s
/// sampling (a family whose only polymorphism lived in an excluded
/// genome is discarded during sampling, not filtered afterward — see
/// `obikphylo::siblings::extensions::SiblingExt::snp_pseudo_alignment`'s
/// own docs) and from the distance matrix / shared-kmer matrix CSV
/// output (row and column both dropped; the underlying computation
/// itself is unaffected). Repeatable.
#[arg(long = "exclude-genome", value_name = "LABEL")]
pub exclude_genome: Vec<String>,
/// Auto-exclude any genome whose mean shared-variable-family count
/// against every other genome (`FamilyOverlap::mean_row` — the same
/// per-row statistic `--family-overlap`'s own matrix shows) falls below
/// this threshold — same exclusion machinery as `--exclude-genome`,
/// applied on top of it rather than instead of it. Applies to the
/// `snp-*` `--distance`/`--pseudo-alignment`/`--sankoff` computations
/// below (all sibling-annex-based); does *not* affect the whole-index
/// `--distance` metrics (jaccard, hamming, bray-curtis, ...) or their
/// matrix/NJ/UPGMA output — a genome with too little SNP-family
/// coverage to trust is a different concern from one whose plain k-mer
/// profile is simply divergent. Requires the Family Overlap annex
/// (built on demand if missing, same as every other annex here — see
/// `obikphylo::siblings::extensions::SiblingExt::family_overlap`'s own
/// docs).
#[arg(long, value_name = "N")]
pub min_shared_family: Option<f64>,
/// Build (or rebuild) the sibling-count/minorant annex — independent of
/// the distance metric below, meant to be run routinely, ahead of any
/// SNP-family distance computation that will later consume it.
#[arg(long)]
pub sibling_annex: bool,
/// Tally the sibling-count distribution (CSV) of an already-built annex
/// (run with `--sibling-annex` first, in this invocation or an earlier
/// one). A separate, occasional diagnostic pass — not run every time the
/// annex itself is (re)built.
#[arg(long)]
pub sibling_stats: bool,
/// Print just the global family-size histogram (1-4 members) of an
/// already-built annex — the `global` row `--sibling-stats` also
/// writes, but without the per-genome breakdown, so it skips
/// `--sibling-stats`'s cross-partition resolution entirely (annex bits
/// only).
#[arg(long)]
pub sibling_hist: bool,
/// Write the Family Overlap matrix (CSV) — number of shared *variable*
/// families per genome pair, index-wide (`obikphylo::siblings::FamilyOverlap`).
/// Built on demand if missing (same as every other annex here); no
/// `--sibling-annex` prerequisite beyond that. Every genome is written,
/// unfiltered by `--exclude-genome`/`--min-shared-family` — a raw
/// coverage diagnostic, not a computation those exclusions are meant to
/// protect.
#[arg(long)]
pub family_overlap: bool,
/// Write a per-family Shannon entropy report (CSV) — requires an
/// already-built sibling annex (`--sibling-annex` first, in this
/// invocation or an earlier one). Always a full, unsampled scan of
/// every family (`--subsample`/`--entropy`/`--entropy-sd` below only
/// apply to `--pseudo-alignment`, not this).
#[arg(long)]
pub shannon: bool,
/// Write a SNP-only pseudo-alignment (FASTA) — requires
/// `--subsample <N>` and an already-built sibling annex
/// (`--sibling-annex` first, in this invocation or an earlier one).
#[arg(long)]
pub pseudo_alignment: bool,
/// Target number of variable sites to sample index-wide for
/// `--pseudo-alignment`/`--sankoff` (mandatory for both) and for a
/// `snp-*` `--distance` value (optional there: omitted means exhaustive
/// — every non-monomorphic minorant of the whole index, not an
/// approximation, see `obikphylo::siblings::SiblingExt::snp_distance`'s
/// own docs) — a target, not a guarantee when given (proportional
/// per-layer sampling; see
/// `obikphylo::siblings::SiblingExt::snp_pseudo_alignment`'s own docs).
#[arg(long)]
pub subsample: Option<usize>,
/// In `--pseudo-alignment`, treat a genome carrying none of a family's
/// observed members (`∅`) as missing data (`?`) rather than a real
/// character state.
#[arg(long)]
pub free_loss: bool,
/// In `--pseudo-alignment`, treat a genome carrying more than one
/// member of a family (ambiguous) as missing data (`?`) rather than an
/// IUPAC ambiguity code.
#[arg(long)]
pub no_ambiguity: bool,
/// Entropy-biased sampling target (Gaussian kernel mean) for
/// `--pseudo-alignment` — activates biasing as soon as this or
/// `--entropy-sd` is given; the other defaults to 1.0/0.5.
#[arg(long)]
pub entropy: Option<f64>,
/// Entropy-biased sampling kernel width (Gaussian standard deviation)
/// for `--pseudo-alignment` — see `--entropy`.
#[arg(long)]
pub entropy_sd: Option<f64>,
/// Calibrate a 16-state Sankoff cost matrix (and its matching
/// pseudo-alignment) from an already-built sibling annex — requires
/// `--subsample <N>`, and shares `--free-loss`/`--no-ambiguity`/
/// `--entropy`/`--entropy-sd` with `--pseudo-alignment` (one draw, same
/// selection feeds both the alignment and every calibration tally).
#[arg(long)]
pub sankoff: bool,
/// Exclude genome pairs whose raw SNP ratio exceeds this value from the
/// base-pair (composition) calibration `--sankoff` pools — a pair this
/// close to substitution saturation carries no information about the
/// true substitution spectrum. Does *not* gate the cardinality
/// calibration (see `obikphylo::siblings::CardinalityTally`'s own
/// docs for why).
#[arg(long, default_value = "0.5")]
pub sankoff_ratio_ceiling: f64,
/// Also write <prefix>_sankoff.tnt, a ready-to-run TNT script (`proc
/// <file>;`) for the same matrix/alignment `--sankoff` computes —
/// recoded to TNT's default xread alphabet (0-9A-F only; TNT rejects
/// the wider IUPAC set `--sankoff`'s own output uses unless `nstates
/// dna` is set, which imposes TNT's own incompatible DNA encoding
/// instead) with integer-scaled costs (TNT's smatrix/cost commands
/// reject decimals). Implies `--sankoff`.
#[arg(long)]
pub tnt: bool,
/// Also write <prefix>_sankoff.tcm and <prefix>_sankoff.pg, a
/// custom-alphabet cost matrix and a ready-to-run PhyG script (`read`/
/// `search`/`report`) for the same matrix/alignment `--sankoff`
/// computes. Reuses `--sankoff`'s own `_sankoff.fasta` directly — PhyG's
/// `tcm:` alphabet is read from the matrix file itself, so the IUPAC+`0`
/// alphabet needs no recoding here, unlike `--tnt`. Implies `--sankoff`.
#[arg(long)]
pub phyg: bool,
/// Also write <prefix>_iqtree.model and <prefix>_iqtree.fasta, a
/// custom-model file and a matching recoded alignment for genuine
/// maximum-likelihood inference with IQ-TREE (`iqtree3 -s ...
/// --seqtype MORPH -m ...+ASC`) — real branch lengths, unlike
/// `--tnt`/`--phyg`'s parsimony step counts. The model is the
/// reversible `Q(i,j) = R(i,j)·π_j` construction: `R` (exchangeability,
/// symmetric) recovered from the same calibrated cost matrix
/// `--sankoff` computes, `π` the real empirical state frequencies
/// counted from the alignment. Only the states that actually occur in
/// this alignment are kept, compactly renumbered (IQ-TREE infers its
/// state count from the alignment itself, and a gap in the numbering
/// would silently misalign the model file). Implies `--sankoff`.
#[arg(long)]
pub iqtree: bool,
/// Under `--iqtree --free-loss`, also recode to `?` (the same
/// missing-data treatment as `-`) any state whose empirical frequency
/// in the alignment falls below this threshold — not just genuinely
/// absent calls. States encoding 3 or 4 simultaneously-observed central
/// bases (IUPAC `V`/`H`/`K`.../`N` for 3, `N` for 4) are rare by
/// construction and often land in exactly this low-frequency range —
/// more likely assembly/detection noise than a genuine, widely-preserved
/// multi-way polymorphism, the same "sampling failure, not true signal"
/// reasoning `--free-loss` already applies to absence. No effect
/// without `--free-loss` (there is no missing-data symbol to recode to
/// otherwise). `<prefix>_iqtree_states.csv` reports the frequency
/// actually used to decide.
#[arg(long, default_value = "0.001")]
pub iqtree_min_freq: f64,
/// Scale factor applied before rounding real-valued costs to the
/// integers both `--tnt`'s smatrix/cost commands and `--phyg`'s `tcm:`
/// matrix require. Keep this small: the total tree score is this scale
/// times the sum of per-character costs across every character, and
/// there are hints in TNT's own manual that at least some of its
/// internal accumulators are 32-bit — a large scale risks a silent
/// integer overflow (undetectable, not just a crash) far more costly
/// than the resolution a bigger factor would buy. Shared between `--tnt`
/// and `--phyg` rather than split into two flags: both scale the same
/// calibrated matrix for the same reason (integer-only cost commands).
#[arg(long, default_value = "100")]
pub sankoff_cost_scale: f64,
/// Distance to compute — either a whole-index metric (`jaccard`,
/// `mash`, `hamming`, `bray-curtis`, ...) or a `snp-*` correction over
/// the central-position SNP substitution spectrum (`snp-raw`, `snp-jc`,
/// `snp-k2p`, `snp-k81`, `snp-f81`, `snp-t92`, `snp-tn93`, `snp-tv`) —
/// the latter route to a different computation entirely
/// (`SiblingExt::snp_distance`, requires `--sibling-annex` first; see
/// `DevDocMD/theory/evolutionary_distances.md`, "`--distance`
/// unification" for the full catalog and why LogDet/Tajima-Nei/F84/
/// HKY85 aren't offered yet).
#[arg(long, value_enum, default_value = "jaccard")]
pub distance: DistanceArg,
/// Rate-heterogeneity correction (Jin-Nei gamma shape parameter `α`)
/// for `snp-*` `--distance` values that support it
/// (`obikphylo::SnpDistanceKind::supports_gamma`: every one except
/// `snp-raw`/`snp-tv`, which have nothing to correct/are deliberately
/// uncorrected). Has no effect on the whole-index metrics. Rejected at
/// runtime if given alongside an unsupported `--distance` value.
#[arg(long, value_name = "ALPHA")]
pub gamma_shape: Option<f64>,
/// Minimum count to consider a kmer present when computing Jaccard on count indexes
#[arg(long, default_value = "1")]
pub presence_threshold: u32,
/// Write the primary distance matrix as plain CSV instead of the
/// default relaxed-PHYLIP format (`n` on the first line, then one
/// `label<TAB>value...` row per genome — no 10-character label
/// truncation, unlike strict PHYLIP, not yet offered here). PHYLIP is
/// the default because it's what external NJ tools (PHYLIP `neighbor`,
/// FastME, T-REX, SplitsTree) actually read; CSV stays available for
/// scripting/inspection. Only affects the primary distance matrix —
/// `--shared-kmers` keeps its own CSV-only format regardless of this
/// flag.
#[arg(long)]
pub csv: bool,
/// Also output the shared-kmer count matrix (CSV)
#[arg(long)]
pub shared_kmers: bool,
/// Compute and write a Neighbor-Joining tree (Newick)
#[arg(long)]
pub nj: bool,
/// Compute and write a UPGMA tree (Newick)
#[arg(long)]
pub upgma: bool,
/// Output prefix: <prefix>_dist.csv, <prefix>_shared.csv, <prefix>_nj.nwk,
/// <prefix>_upgma.nwk. If omitted, the distance matrix is written to stdout.
#[arg(short, long)]
pub output: Option<PathBuf>,
}
-502
View File
@@ -1,502 +0,0 @@
use std::io::{BufWriter, Write};
use std::path::PathBuf;
use obifastwrite::{JsonVal, write_record};
use obikphylo::siblings::SnpAlignment;
use tracing::info;
use super::sankoff::{STATE_SYMBOL, state_index_table};
// ── Sankoff-calibrated data → IQ-TREE custom ML model + recoded alignment ──
//
// Not itself a Sankoff computation — IQ-TREE does maximum likelihood, not
// parsimony. Only the *source data* is shared with `--tnt`/`--phyg` (the
// calibrated 16-state cost matrix, the pseudo-alignment); the operation
// performed on it here is different, hence no "sankoff" in these names,
// unlike `tnt::write_sankoff_tnt`/`phyg::write_sankoff_phyg`.
//
// The mechanism: pass a **file path** directly as `-m`, containing (as
// whitespace/newline-separated numbers) the lower-triangular exchangeability
// matrix `R` (`k(k-1)/2` values, PAML row-major order) immediately followed
// by the `k` state frequencies `π` on the same stream —
// `ModelMarkov::readRates`/`readStateFreq` read them in that exact order, no
// header, no separator required.
//
// `R` is recovered from the calibrated Sankoff cost matrix via
// `R(a,b) = exp(-cost(a,b))` (the cost is `-ln(rate)`), symmetric by
// construction (the underlying tally never captured direction). `π` is the
// real, empirical, non-uniform marginal frequency of each state across the
// whole alignment. IQ-TREE reconstructs the (generally asymmetric) rate
// matrix internally as `Q(i,j) = R(i,j)·π_j` — reversible for *any* `π`, not
// just uniform, because `R` is symmetric.
//
// IQ-TREE infers its state count from the highest-ordinal symbol actually
// present in the alignment, not from a declared count. So states that never
// occur anywhere in this particular alignment are dropped, and the
// survivors are renumbered compactly (`0..k-1`, order preserved) rather than
// leaving gaps that would silently misalign every value IQ-TREE reads. Both
// the model and the alignment must agree on this same renumbering, so it's
// computed once (`CompactAlphabet`) and shared between them.
const IQTREE_STATE_SYMBOL: [char; 16] = [
'0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'A', 'B', 'C', 'D', 'E', 'F',
];
struct CompactAlphabet {
/// Canonical (0..16) state index -> compact index, for states that occur.
old_to_compact: [Option<u8>; 16],
/// Compact index -> canonical state index, order-preserving.
compact_to_old: Vec<u8>,
/// Empirical frequency of each compact-indexed state (sums to 1).
freq: Vec<f64>,
}
impl CompactAlphabet {
fn k(&self) -> usize {
self.compact_to_old.len()
}
}
/// Under `--free-loss`, non-detection (`-`) becomes IQ-TREE's own missing
/// symbol (`?`) — ignored when IQ-TREE checks a site's constancy for
/// `+ASC`. A family kept as "variable" by the sampling (`family_size() >=
/// 2`, a whole-annex property, oblivious to any one column's actual calls)
/// can still turn constant *among the genomes that actually have data* once
/// the non-detected ones are excluded from that check — the same failure
/// mode `--exclude-genome` already had to account for, just triggered by
/// hiding cells instead of dropping whole rows. Same remedy: rescan columns
/// treating `-` as ignored, drop any where the remaining calls agree on a
/// single state. Parsimony (`--tnt`/`--phyg`) has no no-invariant-site
/// requirement, so this only runs on IQ-TREE's own copy of the alignment,
/// never mutating the one the caller also hands to those two exports.
fn drop_ascertainment_noninformative(alignment: &SnpAlignment) -> SnpAlignment {
let n_sites = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
let keep: Vec<bool> = (0..n_sites)
.map(|site| {
let mut first: Option<u8> = None;
for seq in &alignment.sequences {
let b = seq[site];
if b == b'-' {
continue;
}
match first {
None => first = Some(b),
Some(f) if f != b => return true,
_ => {}
}
}
false // all calls missing, or all calls agree — non-informative
})
.collect();
let sequences = alignment
.sequences
.iter()
.map(|seq| {
seq.iter()
.zip(keep.iter())
.filter(|&(_, &k)| k)
.map(|(&b, _)| b)
.collect()
})
.collect();
SnpAlignment { sequences, genome_indices: alignment.genome_indices.clone() }
}
/// Recode every occurrence of a byte in `symbols` to `-` — the same
/// "absent" byte `drop_ascertainment_noninformative`/`compact_alphabet`
/// already treat specially under `--free-loss` (recoded to `?` further
/// downstream). Used by `--iqtree-min-freq` to fold rare, likely-noisy
/// states into the missing-data treatment before a second
/// `compact_alphabet` pass, without duplicating that treatment's logic.
fn recode_symbols_as_absent(alignment: &SnpAlignment, symbols: &[u8]) -> SnpAlignment {
let sequences = alignment
.sequences
.iter()
.map(|seq| {
seq.iter()
.map(|&b| if symbols.contains(&b) { b'-' } else { b })
.collect()
})
.collect();
SnpAlignment { sequences, genome_indices: alignment.genome_indices.clone() }
}
fn compact_alphabet(alignment: &SnpAlignment, free_loss: bool) -> CompactAlphabet {
let iupac_to_state = state_index_table();
let mut occurs = [false; 16];
let mut counts = [0u64; 16];
for seq in &alignment.sequences {
for &b in seq {
if free_loss && b == b'-' {
// `?`: IQ-TREE's own missing-data symbol for `--seqtype
// MORPH`, marginalised by Felsenstein pruning — not a
// numbered state, so excluded from `occurs`/`counts` and
// from the compact alphabet built below.
continue;
}
let b = if b == b'-' { b'0' } else { b };
let state = iupac_to_state[b as usize] as usize;
occurs[state] = true;
counts[state] += 1;
}
}
let mut old_to_compact: [Option<u8>; 16] = [None; 16];
let mut compact_to_old: Vec<u8> = Vec::new();
for old in 0..16 {
if occurs[old] {
old_to_compact[old] = Some(compact_to_old.len() as u8);
compact_to_old.push(old as u8);
}
}
let total: u64 = compact_to_old.iter().map(|&old| counts[old as usize]).sum();
let freq: Vec<f64> = compact_to_old
.iter()
.map(|&old| counts[old as usize] as f64 / total as f64)
.collect();
CompactAlphabet {
old_to_compact,
compact_to_old,
freq,
}
}
/// Write `<prefix>_iqtree_states.csv`: the mapping from IQ-TREE's own
/// compact state symbols (`0..9A-F`, what actually appears in
/// `_iqtree.fasta`/`_iqtree.model`) back to the canonical 16-state
/// alphabet (`STATE_SYMBOL` — the same one `_sankoff_matrix.csv` is
/// indexed by), plus each state's empirical frequency at full precision
/// (`_iqtree.model`'s own frequency line is truncated to 6 decimals).
/// Without this file, a compact index in `_iqtree.model`'s `R`/`π` output
/// (e.g. "state 0 has zero exchangeability with everything else") can't be
/// traced back to which real state that is.
fn write_iqtree_states_csv(alphabet: &CompactAlphabet, output: &Option<PathBuf>) -> String {
let path = output
.as_ref()
.map(|p| format!("{}_iqtree_states.csv", p.display()))
.unwrap_or_else(|| "iqtree_states.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
writeln!(f, "iqtree_symbol,canonical_symbol,frequency").unwrap();
for (compact, &old) in alphabet.compact_to_old.iter().enumerate() {
writeln!(
f,
"{},{},{}",
IQTREE_STATE_SYMBOL[compact], STATE_SYMBOL[old as usize], alphabet.freq[compact]
)
.unwrap();
}
path
}
/// Write the `R` (exchangeability) + `π` (frequencies) model file IQ-TREE's
/// `-m <file>+ASC` reads. Returns the path, so the caller can print a
/// single combined "how to run this" message once the alignment is also
/// written. The one bit of real computation this whole adapter does:
/// `R(a,b) = exp(-cost(a,b))`, recovering the exchangeability rate a
/// calibrated Sankoff parsimony cost implies for a continuous-time model —
/// a one-line inversion of the cost matrix's own `-ln(rate)` construction,
/// not a new estimate.
fn write_iqtree_model(
matrix: &[[f64; 16]; 16],
alphabet: &CompactAlphabet,
output: &Option<PathBuf>,
) -> String {
let rate = |old_i: u8, old_j: u8| (-matrix[old_i as usize][old_j as usize]).exp();
let model_path = output
.as_ref()
.map(|p| format!("{}_iqtree.model", p.display()))
.unwrap_or_else(|| "iqtree.model".into());
let mut f = BufWriter::new(std::fs::File::create(&model_path).unwrap_or_else(|e| {
eprintln!("error creating {model_path}: {e}");
std::process::exit(1);
}));
for i in 1..alphabet.k() {
let row: Vec<String> = (0..i)
.map(|j| {
format!(
"{:.6}",
rate(alphabet.compact_to_old[i], alphabet.compact_to_old[j])
)
})
.collect();
writeln!(f, "{}", row.join(" ")).unwrap();
}
writeln!(
f,
"{}",
alphabet
.freq
.iter()
.map(|p| format!("{p:.6}"))
.collect::<Vec<_>>()
.join(" ")
)
.unwrap();
info!(
"IQ-TREE model file → {model_path} ({} of 16 states present in the alignment)",
alphabet.k()
);
model_path
}
/// Write the pseudo-alignment recoded to the same compact `0..k-1` alphabet
/// as `write_iqtree_model`'s matrix — not `--sankoff`'s own IUPAC alphabet,
/// since IQ-TREE needs the symbol ordinal itself to match the surviving
/// state count (see this module's own doc comment on state inference).
fn write_iqtree_alignment(
alignment: &SnpAlignment,
labels: &[String],
alphabet: &CompactAlphabet,
output: &Option<PathBuf>,
free_loss: bool,
) -> (String, usize) {
let iupac_to_state = state_index_table();
let fasta_path = output
.as_ref()
.map(|p| format!("{}_iqtree.fasta", p.display()))
.unwrap_or_else(|| "iqtree.fasta".into());
let mut f = BufWriter::new(std::fs::File::create(&fasta_path).unwrap_or_else(|e| {
eprintln!("error creating {fasta_path}: {e}");
std::process::exit(1);
}));
let n_sites = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
for (&g, seq) in alignment.genome_indices.iter().zip(alignment.sequences.iter()) {
let recoded: Vec<u8> = seq
.iter()
.map(|&b| {
if free_loss && b == b'-' {
return b'?';
}
let b = if b == b'-' { b'0' } else { b };
let old = iupac_to_state[b as usize] as usize;
let compact = alphabet.old_to_compact[old]
.expect("state occurs in the alignment, so it must have a compact index");
IQTREE_STATE_SYMBOL[compact as usize] as u8
})
.collect();
write_record(
&recoded,
&labels[g],
&[("n_sites", JsonVal::Num(n_sites as u64))],
&mut f,
)
.unwrap_or_else(|e| {
eprintln!("error writing {fasta_path}: {e}");
std::process::exit(1);
});
}
(fasta_path, n_sites)
}
pub(super) fn write_iqtree(
matrix: &[[f64; 16]; 16],
alignment: &SnpAlignment,
labels: &[String],
output: &Option<PathBuf>,
free_loss: bool,
min_freq: f64,
) {
let filtered;
let alignment = if free_loss {
let before = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
filtered = drop_ascertainment_noninformative(alignment);
let after = filtered.sequences.first().map(|s| s.len()).unwrap_or(0);
if after != before {
info!(
"--free-loss: {before} → {after} sites (dropped columns non-informative once `-` \
is treated as missing required for +ASC)"
);
}
&filtered
} else {
alignment
};
let mut alphabet = compact_alphabet(alignment, free_loss);
// `--iqtree-min-freq`: fold rare (likely-noisy) states into the same
// missing-data treatment `-` already gets under `--free-loss`, then
// recompute the alphabet on the further-filtered alignment.
let refiltered;
let alignment = if free_loss {
let low_freq_symbols: Vec<u8> = alphabet
.compact_to_old
.iter()
.zip(alphabet.freq.iter())
.filter(|&(_, &f)| f < min_freq)
.map(|(&old, _)| STATE_SYMBOL[old as usize] as u8)
.collect();
if low_freq_symbols.is_empty() {
alignment
} else {
let recoded = recode_symbols_as_absent(alignment, &low_freq_symbols);
let before = recoded.sequences.first().map(|s| s.len()).unwrap_or(0);
refiltered = drop_ascertainment_noninformative(&recoded);
let after = refiltered.sequences.first().map(|s| s.len()).unwrap_or(0);
info!(
"--iqtree-min-freq {min_freq}: {} rare state(s) ({}) recoded as missing, {before} → {after} sites",
low_freq_symbols.len(),
low_freq_symbols
.iter()
.map(|&b| b as char)
.collect::<String>(),
);
alphabet = compact_alphabet(&refiltered, free_loss);
&refiltered
}
} else {
alignment
};
let states_path = write_iqtree_states_csv(&alphabet, output);
let model_path = write_iqtree_model(matrix, &alphabet, output);
let (fasta_path, n_sites) =
write_iqtree_alignment(alignment, labels, &alphabet, output, free_loss);
let prefix_name = output
.as_ref()
.and_then(|p| p.file_name())
.map(|n| format!("{}_iqtree", n.to_string_lossy()))
.unwrap_or_else(|| "iqtree".into());
info!(
"IQ-TREE alignment → {fasta_path} ({n_sites} sites, {} states)\n\
IQ-TREE state mapping {states_path}\n\
Run with:\n \
iqtree3 -s {fasta_path} --seqtype MORPH -m {model_path}+ASC --prefix {prefix_name} -T AUTO\n\
\n\
options -alrt 1000 -B 1000 can be added to evaluate robustness of the tree",
alphabet.k()
);
}
#[cfg(test)]
mod tests {
use super::*;
fn alignment(sequences: Vec<Vec<u8>>) -> SnpAlignment {
let genome_indices = (0..sequences.len()).collect();
SnpAlignment { sequences, genome_indices }
}
#[test]
fn free_loss_excludes_absent_state_and_freq_sums_to_one() {
// 3 genomes, 2 sites. Site 0: g1='A', g2='C', g3='-' (absent).
// Site 1: g1='-', g2='-', g3='G'. Under free_loss, every '-' must
// be excluded from the frequency count entirely (not folded into
// state 0).
let alignment = alignment(vec![vec![b'A', b'-'], vec![b'C', b'-'], vec![b'-', b'G']]);
let alphabet = compact_alphabet(&alignment, true);
assert!(
!alphabet.compact_to_old.contains(&0),
"state 0 (absent) must not appear in the compact alphabet under --free-loss, got {:?}",
alphabet.compact_to_old
);
let sum: f64 = alphabet.freq.iter().sum();
assert!(
(sum - 1.0).abs() < 1e-9,
"frequencies must sum to 1, got {sum} ({:?})",
alphabet.freq
);
assert_eq!(alphabet.k(), 3, "A, C, G — 3 real states, `-` excluded");
}
#[test]
fn without_free_loss_absent_state_is_counted_normally() {
let alignment = alignment(vec![vec![b'A', b'-'], vec![b'C', b'-'], vec![b'-', b'G']]);
let alphabet = compact_alphabet(&alignment, false);
assert!(
alphabet.compact_to_old.contains(&0),
"state 0 (absent, recoded from '-') must be counted when --free-loss is off"
);
let sum: f64 = alphabet.freq.iter().sum();
assert!(
(sum - 1.0).abs() < 1e-9,
"frequencies must sum to 1, got {sum} ({:?})",
alphabet.freq
);
}
#[test]
fn states_csv_maps_compact_symbols_back_to_canonical_ones() {
// 'A' (state 1) and 'G' (state 4) occur, '-' (state 0) excluded by
// --free-loss — compact index 0 -> 'A', compact index 1 -> 'G'.
let alignment = alignment(vec![vec![b'A', b'-'], vec![b'-', b'G']]);
let alphabet = compact_alphabet(&alignment, true);
let output = Some(
std::env::temp_dir().join(format!("obikmer2_test_iqtree_states_{}", std::process::id())),
);
let path = write_iqtree_states_csv(&alphabet, &output);
let csv = std::fs::read_to_string(&path).unwrap();
std::fs::remove_file(&path).ok();
let mut lines = csv.lines();
assert_eq!(
lines.next(),
Some("iqtree_symbol,canonical_symbol,frequency")
);
assert_eq!(lines.next(), Some("0,A,0.5"));
assert_eq!(lines.next(), Some("1,G,0.5"));
assert!(lines.next().is_none());
}
#[test]
fn iqtree_min_freq_folds_rare_states_into_missing() {
// 20 common A/C sites (60 calls total across 3 genomes) plus one
// site where genome 0 carries the rare ambiguity state `M` and
// genome 1 carries `A` (kept informative by the first
// ascertainment filter: two distinct non-`-` calls) — `M` ends up
// at 1/62, well below the 0.05 threshold used here.
let mut sequences: Vec<Vec<u8>> = vec![Vec::new(); 3];
for i in 0..20 {
let (a, b, c) = if i % 2 == 0 {
(b'A', b'C', b'A')
} else {
(b'C', b'A', b'C')
};
sequences[0].push(a);
sequences[1].push(b);
sequences[2].push(c);
}
sequences[0].push(b'M');
sequences[1].push(b'A');
sequences[2].push(b'-');
let alignment = alignment(sequences);
let labels = vec!["g1".to_string(), "g2".to_string(), "g3".to_string()];
let matrix = [[0.0f64; 16]; 16];
let prefix = std::env::temp_dir().join(format!(
"obikmer2_test_iqtree_minfreq_{}",
std::process::id()
));
let output = Some(prefix.clone());
write_iqtree(&matrix, &alignment, &labels, &output, true, 0.05);
let states_path = format!("{}_iqtree_states.csv", prefix.display());
let csv = std::fs::read_to_string(&states_path).unwrap();
assert!(
!csv.contains(",M,"),
"M (freq ~1/62) must be folded into missing under --iqtree-min-freq 0.05, got:\n{csv}"
);
assert!(
csv.contains(",A,") && csv.contains(",C,"),
"A/C must survive (well above threshold), got:\n{csv}"
);
for suffix in ["_iqtree_states.csv", "_iqtree.model", "_iqtree.fasta"] {
std::fs::remove_file(format!("{}{suffix}", prefix.display())).ok();
}
}
}
-478
View File
@@ -1,478 +0,0 @@
mod args;
mod iqtree;
mod phyg;
mod phylip;
mod sankoff;
mod tnt;
use std::io::{self, BufWriter, Write};
use std::sync::Arc;
use obikidxcache::index_cache::IndexCache;
use obikindex::KmerIndex;
use obikphylo::siblings::{
EntropyBias, SiblingExt, cardinality_transition_probs, composition_transition_probs,
pairwise_cost_matrix,
};
use obikphylo::{Metrics, neighbor_joining, upgma};
use obisys::{Reporter, Stage};
use tracing::info;
use iqtree::write_iqtree;
use phyg::write_sankoff_phyg;
use phylip::write_phylip_relaxed;
use sankoff::{write_sankoff_alignment_fasta, write_sankoff_matrix_csv, write_sankoff_params};
use tnt::write_sankoff_tnt;
pub use args::PhyloArgs;
pub fn run(args: PhyloArgs) {
let idx = Arc::new(KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
}));
let labels: Vec<String> = idx
.meta()
.genomes()
.unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
})
.iter()
.map(|g| g.label.clone())
.collect();
let n = labels.len();
// ── Genome exclusion (`--exclude-genome`) ───────────────────────────────────
// Resolved once, up front: `snp_pseudo_alignment` needs it baked into
// sampling itself (see its own docs), and the distance/shared-kmer CSV
// writers below just skip these rows/columns at write time — the
// underlying `cache.distance(...)` computation is unaffected either way.
let exclude_mask: Vec<bool> = {
let mut mask = vec![false; n];
for label in &args.exclude_genome {
match labels.iter().position(|l| l == label) {
Some(i) => mask[i] = true,
None => {
eprintln!("error: --exclude-genome {label:?} does not match any genome in this index");
std::process::exit(1);
}
}
}
mask
};
let mut rep = Reporter::new();
// Every partition/layer this needs is opened once, up front, and
// handed to `Metrics::distance`/`SiblingExt::build_sibling_annex`
// — see `obikquery`'s own use of `IndexCache` for the same reasoning
// (one open, many in-memory reads).
let cache = IndexCache::new(Arc::clone(&idx), None);
// ── Sibling-count/minorant annex (independent of the distance metric) ──
// Meant to be (re)built routinely, ahead of any SNP-family distance
// computation that will later consume it.
if args.sibling_annex {
// Writes into the index directory — hold an exclusive lock for the
// duration so a second, concurrent `--sibling-annex` run on the
// same index can't corrupt these writes (see obisys::DirLock).
let _lock = obisys::DirLock::acquire(&args.index).unwrap_or_else(|e| {
eprintln!("error locking index directory {}: {e}", args.index.display());
std::process::exit(1);
});
info!("building sibling-count/minorant annex");
let t = Stage::start("sibling_annex");
cache.build_sibling_annex().unwrap_or_else(|e| {
eprintln!("error building sibling annex: {e}");
std::process::exit(1);
});
rep.push(t.stop());
}
// ── Sibling-count distribution (`--sibling-stats`) ──────────────────────────
if args.sibling_stats {
let t = Stage::start("sibling_stats");
let stats = cache.sibling_annex_stats().unwrap_or_else(|e| {
eprintln!("error computing sibling-annex stats: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let path = args.output.as_ref()
.map(|p| format!("{}_siblings.csv", p.display()))
.unwrap_or_else(|| "siblings.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
// One row per genome (4 columns, family size 1-4: number of
// families of that size for which the genome carries at least one
// member), plus a `global` row — the actual deduplicated
// family-size histogram (`stats.counts`), NOT a sum of the
// per-genome columns (a family shared by several genomes would
// otherwise be counted once per genome it appears in).
writeln!(f, "genome,1,2,3,4").unwrap();
for (label, counts) in labels.iter().zip(stats.per_genome.iter()) {
writeln!(f, "{label},{},{},{},{}", counts[0], counts[1], counts[2], counts[3]).unwrap();
}
writeln!(
f, "global,{},{},{},{}",
stats.counts[0], stats.counts[1], stats.counts[2], stats.counts[3],
).unwrap();
info!("sibling-count distribution → {path}");
}
// ── Family-size histogram (`--sibling-hist`) ────────────────────────────────
if args.sibling_hist {
let t = Stage::start("sibling_hist");
let counts = cache.sibling_family_size_histogram().unwrap_or_else(|e| {
eprintln!("error computing sibling family-size histogram: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let path = args.output.as_ref()
.map(|p| format!("{}_sibling_hist.csv", p.display()))
.unwrap_or_else(|| "sibling_hist.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
writeln!(f, "size,count").unwrap();
for (size, count) in counts.iter().enumerate() {
writeln!(f, "{},{count}", size + 1).unwrap();
}
let total: u64 = counts.iter().sum();
info!(
"family-size histogram → {path} (total {total} famil{})",
if total == 1 { "y" } else { "ies" }
);
}
// ── Family Overlap matrix (`--family-overlap`) ──────────────────────────────
if args.family_overlap {
let t = Stage::start("family_overlap");
let overlap = cache.family_overlap().unwrap_or_else(|e| {
eprintln!("error computing family overlap: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let path = args.output.as_ref()
.map(|p| format!("{}_family_overlap.csv", p.display()))
.unwrap_or_else(|| "family_overlap.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
write!(f, "genome").unwrap();
for label in &labels { write!(f, ",{label}").unwrap(); }
writeln!(f).unwrap();
for (i, label) in labels.iter().enumerate() {
write!(f, "{label}").unwrap();
for j in 0..n { write!(f, ",{}", overlap.get(i, j)).unwrap(); }
writeln!(f).unwrap();
}
info!("family-overlap matrix → {path}");
}
// ── Shannon entropy report (`--shannon`) ────────────────────────────────────
if args.shannon {
let path = args.output.as_ref()
.map(|p| format!("{}_entropy.csv", p.display()))
.unwrap_or_else(|| "entropy.csv".into());
info!("computing per-family Shannon entropy");
let t = Stage::start("shannon_entropy");
cache.shannon_entropy_csv(std::path::Path::new(&path)).unwrap_or_else(|e| {
eprintln!("error computing Shannon entropy: {e}");
std::process::exit(1);
});
rep.push(t.stop());
info!("entropy report → {path}");
}
// ── `--min-shared-family` auto-exclusion ────────────────────────────────────
// Layered on top of `--exclude-genome`, not instead of it — a separate
// mask (not folded into `exclude_mask` itself) since it must reach
// `--pseudo-alignment`/`--sankoff`/`snp-*` `--distance` only, never the
// whole-index metrics' `kept`/`--shared-kmers` output (see
// `args::PhyloArgs::min_shared_family`'s own docs on why).
let snp_exclude_mask: Vec<bool> = match args.min_shared_family {
Some(threshold) => {
let overlap = cache.family_overlap().unwrap_or_else(|e| {
eprintln!("error computing family overlap: {e}");
std::process::exit(1);
});
let mut mask = exclude_mask.clone();
for (g, excluded) in mask.iter_mut().enumerate() {
if *excluded {
continue;
}
let mean = overlap.mean_row(g);
if mean < threshold {
info!(
"auto-excluding {} (--min-shared-family: mean shared-family count {mean:.1} < {threshold})",
labels[g]
);
*excluded = true;
}
}
mask
}
None => exclude_mask.clone(),
};
// Shared by `--pseudo-alignment` and `--sankoff` — same activation rule:
// either flag given activates entropy-biased sampling, the other
// defaults to 1.0/0.5.
let entropy_bias = if args.entropy.is_some() || args.entropy_sd.is_some() {
Some(EntropyBias {
mu: args.entropy.unwrap_or(1.0),
sigma: args.entropy_sd.unwrap_or(0.5),
})
} else {
None
};
// ── SNP pseudo-alignment (`--pseudo-alignment`) ─────────────────────────────
if args.pseudo_alignment {
let Some(subsample_n) = args.subsample else {
eprintln!("error: --pseudo-alignment requires --subsample <N>");
std::process::exit(1);
};
info!("sampling SNP pseudo-alignment (target {subsample_n} site(s))");
let t = Stage::start("pseudo_alignment");
let alignment = cache
.snp_pseudo_alignment(subsample_n, args.free_loss, args.no_ambiguity, &snp_exclude_mask, entropy_bias)
.unwrap_or_else(|e| {
eprintln!("error building pseudo-alignment: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let path = args.output.as_ref()
.map(|p| format!("{}_alignment.fasta", p.display()))
.unwrap_or_else(|| "alignment.fasta".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
let n_sites = alignment.sequences.first().map_or(0, Vec::len);
for (&g, seq) in alignment.genome_indices.iter().zip(alignment.sequences.iter()) {
obifastwrite::write_plain_record(seq, &labels[g], &mut f).unwrap();
}
info!("pseudo-alignment ({n_sites} site(s), {} genome(s)) → {path}", alignment.genome_indices.len());
}
// ── Sankoff cost-matrix calibration (`--sankoff`, `--tnt`, `--phyg`, `--iqtree`) ──
if args.sankoff || args.tnt || args.phyg || args.iqtree {
let Some(subsample_n) = args.subsample else {
eprintln!("error: --sankoff requires --subsample <N>");
std::process::exit(1);
};
info!("sampling Sankoff calibration bundle (target {subsample_n} site(s))");
let t = Stage::start("sankoff_bundle");
let bundle = cache
.sankoff_bundle(
subsample_n,
args.free_loss,
args.no_ambiguity,
&snp_exclude_mask,
entropy_bias,
args.sankoff_ratio_ceiling,
)
.unwrap_or_else(|e| {
eprintln!("error computing Sankoff calibration bundle: {e}");
std::process::exit(1);
});
rep.push(t.stop());
let p_card = cardinality_transition_probs(&bundle.cardinality_tally);
let p_comp = composition_transition_probs(&bundle.base_pair_tally);
let matrix = pairwise_cost_matrix(&p_card, &p_comp, args.free_loss);
write_sankoff_matrix_csv(&matrix, &args.output);
write_sankoff_params(
&bundle.cardinality_tally,
&p_card,
&bundle.base_pair_tally,
&p_comp,
args.sankoff_ratio_ceiling,
&args.output,
);
write_sankoff_alignment_fasta(&bundle.alignment, &labels, &args.output, args.free_loss);
if args.tnt {
write_sankoff_tnt(
&matrix,
&bundle.alignment,
&labels,
&args.output,
args.sankoff_cost_scale,
args.free_loss,
);
}
if args.phyg {
write_sankoff_phyg(&matrix, &args.output, args.sankoff_cost_scale);
}
if args.iqtree {
write_iqtree(
&matrix,
&bundle.alignment,
&labels,
&args.output,
args.free_loss,
args.iqtree_min_freq,
);
}
}
// ── Distance computation: classic whole-index metric vs. `snp-*` ───────────
// Two genuinely different code paths behind one `--distance` value — see
// `args::DistanceArg`'s own docs.
let (matrix, shared_kmers) = match args.distance.as_classic() {
Some(metric) => {
info!("computing {metric:?} distances for {n} genome(s)");
let need_shared = args.shared_kmers || args.nj || args.upgma;
let t = Stage::start("distance");
let result = cache
.distance(metric, need_shared, args.presence_threshold)
.unwrap_or_else(|e| {
eprintln!("error computing distances: {e}");
std::process::exit(1);
});
rep.push(t.stop());
(result.matrix, result.shared_kmers)
}
None => {
if args.shared_kmers {
eprintln!("error: --shared-kmers has no meaning for a snp-* --distance value");
std::process::exit(1);
}
let kind = args.distance.as_snp().expect("DistanceArg is always classic or snp");
info!(
"computing {kind:?} SNP distance for {n} genome(s){}",
match args.subsample {
Some(n) => format!(" (subsampled, target {n} site(s))"),
None => " (exhaustive)".into(),
}
);
let t = Stage::start("snp_distance");
let matrix = cache
.snp_distance(
kind,
args.subsample,
args.free_loss,
args.no_ambiguity,
&snp_exclude_mask,
entropy_bias,
args.gamma_shape,
)
.unwrap_or_else(|e| {
eprintln!("error computing SNP distance: {e}");
std::process::exit(1);
});
rep.push(t.stop());
(matrix, None)
}
};
// Rows/columns kept in every matrix output below — the computation
// above runs over every genome regardless; only the writers skip
// excluded ones.
let kept: Vec<usize> = (0..n).filter(|&i| !exclude_mask[i]).collect();
// ── Distance matrix → relaxed PHYLIP (default) or CSV (`--csv`) ────────────
let write_dist = |w: &mut dyn Write| {
if args.csv {
write!(w, "genome").unwrap();
for &j in &kept { write!(w, ",{}", labels[j]).unwrap(); }
writeln!(w).unwrap();
for &i in &kept {
write!(w, "{}", labels[i]).unwrap();
for &j in &kept {
write!(w, ",{:.6}", matrix[[i, j]]).unwrap();
}
writeln!(w).unwrap();
}
} else {
write_phylip_relaxed(w, &labels, &kept, &matrix);
}
};
match &args.output {
Some(prefix) => {
let suffix = if args.csv { "_dist.csv" } else { "_dist.phy" };
let path = format!("{}{suffix}", prefix.display());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
write_dist(&mut f);
info!("distance matrix → {path}");
}
None => {
let stdout = io::stdout();
let mut out = BufWriter::new(stdout.lock());
write_dist(&mut out);
}
}
// ── Shared-kmer matrix → CSV ──────────────────────────────────────────────
if args.shared_kmers {
if let Some(shared) = &shared_kmers {
let path = args.output.as_ref()
.map(|p| format!("{}_shared.csv", p.display()))
.unwrap_or_else(|| "shared.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
write!(f, "genome").unwrap();
for &j in &kept { write!(f, ",{}", labels[j]).unwrap(); }
writeln!(f).unwrap();
for &i in &kept {
write!(f, "{}", labels[i]).unwrap();
for &j in &kept { write!(f, ",{}", shared[[i, j]]).unwrap(); }
writeln!(f).unwrap();
}
info!("shared-kmer matrix → {path}");
}
}
// ── NJ tree ────────────────────────────────────────────────────────────────
if args.nj {
let tree = neighbor_joining(&matrix, &labels).unwrap_or_else(|e| {
eprintln!("error computing NJ tree: {e}");
std::process::exit(1);
});
let newick = tree.to_newick();
let path = args.output.as_ref()
.map(|p| format!("{}_nj.nwk", p.display()))
.unwrap_or_else(|| "nj.nwk".into());
std::fs::write(&path, &newick).unwrap_or_else(|e| {
eprintln!("error writing {path}: {e}");
std::process::exit(1);
});
info!("NJ tree → {path}");
}
// ── UPGMA tree ───────────────────────────────────────────────────────────────
if args.upgma {
let newick = upgma(&matrix, &labels).to_newick();
let path = args.output.as_ref()
.map(|p| format!("{}_upgma.nwk", p.display()))
.unwrap_or_else(|| "upgma.nwk".into());
std::fs::write(&path, &newick).unwrap_or_else(|e| {
eprintln!("error writing {path}: {e}");
std::process::exit(1);
});
info!("UPGMA tree → {path}");
}
rep.print();
}
-80
View File
@@ -1,80 +0,0 @@
use std::io::{BufWriter, Write};
use std::path::PathBuf;
use tracing::info;
use super::sankoff::{STATE_SYMBOL, scaled_metric_matrix};
// ── Sankoff cost matrix → PhyG custom-alphabet TCM + ready-to-run script ────
//
// PhyG's `tcm:STRING` format needs no alphabet recoding, unlike `--tnt`:
// its parser reads the alphabet straight from the tcm file's own first
// line, so `--sankoff`'s own `_sankoff.fasta` (already IUPAC+`0`) is reused
// as-is via `prefasta:`. PhyG auto-adds its own indel/gap state as an
// (n+1)-th row/column of the tcm — inert here since the alignment already
// encodes absence as an ordinary state (`0`), never as `-` (see
// `sankoff::write_sankoff_alignment_fasta`'s own comment on why). The gap
// row/column below reuses `matrix[i][0]`/`matrix[0][j]` (cost to/from `∅`)
// as the closest principled value for a state that, in practice, is never
// actually triggered.
pub(super) fn write_sankoff_phyg(matrix: &[[f64; 16]; 16], output: &Option<PathBuf>, cost_scale: f64) {
let scaled_matrix = scaled_metric_matrix(matrix, cost_scale);
let basename = |suffix: &str| -> String {
output.as_ref()
.and_then(|p| p.file_name())
.map(|n| format!("{}{suffix}", n.to_string_lossy()))
.unwrap_or_else(|| format!("sankoff{suffix}"))
};
let full_path = |suffix: &str| -> String {
output.as_ref()
.map(|p| format!("{}{suffix}", p.display()))
.unwrap_or_else(|| format!("sankoff{suffix}"))
};
let tcm_path = full_path("_sankoff.tcm");
let mut f = BufWriter::new(std::fs::File::create(&tcm_path).unwrap_or_else(|e| {
eprintln!("error creating {tcm_path}: {e}");
std::process::exit(1);
}));
let alphabet_line = STATE_SYMBOL.iter().map(|c| c.to_string()).collect::<Vec<_>>().join(" ");
writeln!(f, "{alphabet_line}").unwrap();
for i in 0..16 {
let mut row: Vec<i64> = (0..16).map(|j| scaled_matrix[i][j]).collect();
row.push(scaled_matrix[i][0]); // gap column: same cost as to/from ∅
writeln!(f, "{}", row.iter().map(|v| v.to_string()).collect::<Vec<_>>().join(" ")).unwrap();
}
let mut gap_row: Vec<i64> = (0..16).map(|j| scaled_matrix[0][j]).collect();
gap_row.push(0);
writeln!(f, "{}", gap_row.iter().map(|v| v.to_string()).collect::<Vec<_>>().join(" ")).unwrap();
info!("PhyG TCM → {tcm_path}");
let pg_path = full_path("_sankoff.pg");
let mut f = BufWriter::new(std::fs::File::create(&pg_path).unwrap_or_else(|e| {
eprintln!("error creating {pg_path}: {e}");
std::process::exit(1);
}));
let fasta_name = basename("_sankoff.fasta");
let tcm_name = basename("_sankoff.tcm");
let tre_name = basename("_sankoff.tre");
writeln!(f, "read(prefasta:\"{fasta_name}\", tcm:\"{tcm_name}\")").unwrap();
writeln!(f, "search(seconds:300, instances:4)").unwrap();
writeln!(f, "report(\"{tre_name}\", graphs, newick, overwrite)").unwrap();
let pg_dir = std::path::Path::new(&pg_path).parent()
.filter(|d| !d.as_os_str().is_empty())
.map(|d| d.display().to_string())
.unwrap_or_else(|| ".".into());
let pg_name = std::path::Path::new(&pg_path).file_name()
.map(|n| n.to_string_lossy().into_owned())
.unwrap_or_else(|| pg_path.clone());
info!(
"PhyG script → {pg_path} (costs scaled x{cost_scale:.0}, runs a default 300s/4-instance \
search and writes trees to {tre_name})\n\
Run it with:\n \
cd {pg_dir} && phyg {pg_name}\n\
(`phyg` must run from that directory `read()`/`report()` in the script use relative \
file names)"
);
}
-215
View File
@@ -1,215 +0,0 @@
//! Output writers for `--sankoff` — no calibration logic here, just
//! formatting: `obikphylo::siblings::SiblingExt::sankoff_bundle` and the
//! `cardinality_transition_probs`/`composition_transition_probs`/
//! `pairwise_cost_matrix` calibration functions do all the actual work in
//! `mod.rs`, this module only serialises their results.
use std::io::{BufWriter, Write};
use std::path::PathBuf;
use obifastwrite::{JsonVal, write_record};
use obikphylo::siblings::{BasePairTally, CardinalityTally, SnpAlignment};
use tracing::info;
// ── Sankoff pseudo-alignment → FASTA ────────────────────────────────────────
//
// Same data as `--pseudo-alignment`'s output (`SnpAlignment`/
// `snp_pseudo_alignment`), re-coded so its symbols match the accompanying
// `--sankoff` matrix output exactly: `0` for the empty/absent state instead
// of `-`, which TNT/PhyG would otherwise read as their own gap character
// rather than our "family absent" state. Unless `free_loss` (`--free-loss`)
// is set, in which case `∅` is recoded to `?` instead — TNT/PhyG's own
// missing-data symbol, deliberately *not* `-` (still gap/indel semantics in
// both tools) — so non-detection costs nothing rather than being scored as
// an ordinary, calibrated state transition.
pub(super) fn write_sankoff_alignment_fasta(
alignment: &SnpAlignment,
labels: &[String],
output: &Option<PathBuf>,
free_loss: bool,
) {
let path = output.as_ref()
.map(|p| format!("{}_sankoff.fasta", p.display()))
.unwrap_or_else(|| "sankoff.fasta".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
let absent_symbol = if free_loss { b'?' } else { b'0' };
let n_sites = alignment.sequences.first().map_or(0, Vec::len);
for (&g, seq) in alignment.genome_indices.iter().zip(alignment.sequences.iter()) {
let recoded: Vec<u8> = seq.iter().map(|&b| if b == b'-' { absent_symbol } else { b }).collect();
write_record(&recoded, &labels[g], &[("n_sites", JsonVal::Num(n_sites as u64))], &mut f)
.unwrap_or_else(|e| {
eprintln!("error writing {path}: {e}");
std::process::exit(1);
});
}
info!("Sankoff pseudo-alignment ({n_sites} site(s)) → {path}");
}
// ── Sankoff cost matrix → CSV ────────────────────────────────────────────────
//
// 16 states indexed by bitmask (bit 0=A, 1=C, 2=G, 3=T; state 0 is `∅`),
// matching the convention used for `--pseudo-alignment`'s IUPAC-coded output
// and for the external TNT/PhyG scripts this feeds.
/// IUPAC ambiguity code per state (same mapping
/// `obikphylo::siblings::algorithms::masking::iupac_code` uses internally
/// for `--pseudo-alignment`), with `0` standing in for the empty state (`-`
/// would collide with TNT/PhyG's own gap/range syntax). Bit order: 0=A,
/// 1=C, 2=G, 3=T. This project's canonical alphabet for every Sankoff
/// export (`--tnt`/`--phyg`/`--iqtree` each recode it to their own alphabet
/// at their own adapter boundary, rather than using it directly).
pub(super) const STATE_SYMBOL: [char; 16] = [
'0', 'A', 'C', 'M', 'G', 'R', 'S', 'V', 'T', 'W', 'Y', 'H', 'K', 'D', 'B', 'N',
];
/// `STATE_SYMBOL` byte -> state index (0..16), for adapters that need to
/// translate an alignment written in this alphabet into their own. Shared
/// rather than rebuilt per adapter (`tnt`, `iqtree`).
pub(super) fn state_index_table() -> [u8; 128] {
let mut table = [0u8; 128];
for (state, &sym) in STATE_SYMBOL.iter().enumerate() {
table[sym as usize] = state as u8;
}
table
}
pub(super) fn write_sankoff_matrix_csv(matrix: &[[f64; 16]; 16], output: &Option<PathBuf>) {
let path = output.as_ref()
.map(|p| format!("{}_sankoff_matrix.csv", p.display()))
.unwrap_or_else(|| "sankoff_matrix.csv".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
write!(f, "state").unwrap();
for sym in STATE_SYMBOL { write!(f, ",{sym}").unwrap(); }
writeln!(f).unwrap();
for (s, row) in matrix.iter().enumerate() {
write!(f, "{}", STATE_SYMBOL[s]).unwrap();
for cost in row { write!(f, ",{cost:.4}").unwrap(); }
writeln!(f).unwrap();
}
info!("Sankoff cost matrix → {path}");
}
// ── Sankoff calibration parameters → YAML report ────────────────────────────
//
// Everything `--sankoff` estimates from real data, in one durable,
// machine-readable file: the cardinality and base-pair transition tallies
// (raw counts, not just the derived probabilities) — costs are a modelling
// choice built *from* the counts, and reproducing/re-deriving them later
// needs the counts, not just their current derived value.
#[derive(serde::Serialize)]
struct CardinalityTransition {
from: usize,
to: usize,
count: u64,
probability: f64,
}
#[derive(serde::Serialize)]
struct CompositionTransition {
from: char,
to: char,
count: u64,
probability: f64,
}
#[derive(serde::Serialize)]
struct SankoffParamsReport {
ratio_ceiling: f64,
cardinality_transitions: Vec<CardinalityTransition>,
composition_transitions: Vec<CompositionTransition>,
}
pub(super) fn write_sankoff_params(
card_tally: &CardinalityTally,
p_card: &[[f64; 5]; 5],
base_tally: &BasePairTally,
p_comp: &[[f64; 4]; 4],
ratio_ceiling: f64,
output: &Option<PathBuf>,
) {
const BASE_LETTER: [char; 4] = ['A', 'C', 'G', 'T'];
let mut cardinality_transitions = Vec::with_capacity(25);
for a in 0..5 {
for b in 0..5 {
cardinality_transitions.push(CardinalityTransition {
from: a,
to: b,
count: card_tally.counts[a][b],
probability: p_card[a][b],
});
}
}
let mut composition_transitions = Vec::with_capacity(16);
for a in 0..4 {
for b in 0..4 {
let count = if a == b { base_tally.same[a] } else { base_tally.counts[a][b] };
composition_transitions.push(CompositionTransition {
from: BASE_LETTER[a],
to: BASE_LETTER[b],
count,
probability: p_comp[a][b],
});
}
}
let report = SankoffParamsReport { ratio_ceiling, cardinality_transitions, composition_transitions };
let path = output.as_ref()
.map(|p| format!("{}_sankoff_params.yaml", p.display()))
.unwrap_or_else(|| "sankoff_params.yaml".into());
let f = std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
});
serde_yaml::to_writer(f, &report).unwrap_or_else(|e| {
eprintln!("error writing {path}: {e}");
std::process::exit(1);
});
info!("Sankoff calibration parameters → {path}");
}
/// Scale `matrix` by `cost_scale` and round to integers (TNT's smatrix/cost
/// and PhyG's `tcm:` commands both reject decimals), then take the *metric
/// closure* of the result (Floyd-Warshall over the 16 states again, on the
/// now-integer values).
///
/// `pairwise_cost_matrix`'s row-normalise-then-`-ln` construction gives no
/// guarantee of being a metric (unlike a cost graph closed by shortest path
/// by construction) — so this closure isn't only needed to correct
/// integer-rounding artifacts (two real costs of `1.734` each round to
/// `173`, summing to `346`, while their own real sum `3.468` rounds to
/// `347` — TNT then reports "triangle inequality violated ... Fixed" and
/// silently substitutes its own corrected value), it may also be the only
/// thing making the *real-valued* matrix a metric in the first place.
/// Re-closing after rounding makes both corrections explicit and
/// reproducible here instead, rather than left implicit and
/// tool-version-dependent.
pub(super) fn scaled_metric_matrix(matrix: &[[f64; 16]; 16], cost_scale: f64) -> [[i64; 16]; 16] {
let mut m = [[0i64; 16]; 16];
for i in 0..16 {
for j in 0..16 {
m[i][j] = (matrix[i][j] * cost_scale).round() as i64;
}
}
for k in 0..16 {
for i in 0..16 {
for j in 0..16 {
let via = m[i][k] + m[k][j];
if via < m[i][j] {
m[i][j] = via;
}
}
}
}
m
}
-184
View File
@@ -1,184 +0,0 @@
use std::io::{BufWriter, Write};
use std::path::PathBuf;
use obikphylo::siblings::SnpAlignment;
use tracing::info;
use super::sankoff::{scaled_metric_matrix, state_index_table};
// ── Sankoff cost matrix + alignment → ready-to-run TNT script ──────────────
//
// TNT's *default* xread reader only accepts its own 0-9A-F alphabet (see
// its manual: "up to 16 states are allowed by xread, using symbols 0-9 ...
// and A-F") — the wider IUPAC set `STATE_SYMBOL` uses is rejected as an
// "alien symbol" unless `nstates dna` is set, which imposes TNT's own fixed
// DNA encoding instead, incompatible with a custom smatrix. And TNT's
// `smatrix`/`cost` commands reject decimal costs ("found symbol . when
// reading transformation costs"). So: recode to TNT's alphabet and
// integer-scale the costs here, at this adapter's boundary, rather than
// degrading the project's own canonical (IUPAC, real-valued) output.
const TNT_STATE_SYMBOL: [char; 16] = [
'0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'A', 'B', 'C', 'D', 'E', 'F',
];
pub(super) fn write_sankoff_tnt(
matrix: &[[f64; 16]; 16],
alignment: &SnpAlignment,
labels: &[String],
output: &Option<PathBuf>,
cost_scale: f64,
free_loss: bool,
) {
let path = output
.as_ref()
.map(|p| format!("{}_sankoff.tnt", p.display()))
.unwrap_or_else(|| "sankoff.tnt".into());
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
eprintln!("error creating {path}: {e}");
std::process::exit(1);
}));
// IUPAC-ish symbol -> bitmask, to translate the alignment (which uses
// `STATE_SYMBOL`, `-` already normalised to `0` by `sankoff_bundle`
// callers) into TNT's alphabet without re-deriving state indices.
let iupac_to_state = state_index_table();
let n_sites = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
let kept_labels: Vec<&String> = alignment.genome_indices.iter().map(|&g| &labels[g]).collect();
writeln!(f, "xread").unwrap();
writeln!(f, "mxram 16000;").unwrap();
writeln!(f, "taxname =;").unwrap();
writeln!(f, "taxname +50;").unwrap();
writeln!(
f,
"'obikmer central-position SNP families, calibrated Sankoff 16-state encoding'"
)
.unwrap();
writeln!(f, "{n_sites} {}", kept_labels.len()).unwrap();
for (label, seq) in kept_labels.iter().zip(alignment.sequences.iter()) {
write!(f, "{label} ").unwrap();
for &b in seq {
if free_loss && b == b'-' {
// `?`: TNT's own missing-data symbol, read directly, not
// routed through `TNT_STATE_SYMBOL` (there is no state for
// it) — see `write_sankoff_alignment_fasta`'s doc comment.
write!(f, "?").unwrap();
continue;
}
let b = if b == b'-' { b'0' } else { b };
let state = iupac_to_state[b as usize];
write!(f, "{}", TNT_STATE_SYMBOL[state as usize]).unwrap();
}
writeln!(f).unwrap();
}
writeln!(f, ";\n").unwrap();
let scaled_matrix = scaled_metric_matrix(matrix, cost_scale);
writeln!(f, "smatrix =0 (family16)").unwrap();
for i in 0..16 {
for j in (i + 1)..16 {
writeln!(
f,
"{}/{} {}",
TNT_STATE_SYMBOL[i], TNT_STATE_SYMBOL[j], scaled_matrix[i][j]
)
.unwrap();
}
}
writeln!(f, ";\n").unwrap();
writeln!(f, "ccode ( 0.{} ;", n_sites - 1).unwrap();
writeln!(f, "smatrix +0 0.{} ;", n_sites - 1).unwrap();
writeln!(f).unwrap();
// Basename only (not the full `path`/`output` prefix): TNT's natural
// workflow is to `cd` into the output directory before `proc`-ing the
// script, and an absolute path here would break if that directory is
// later moved or copied elsewhere.
let tre_name = output
.as_ref()
.and_then(|p| p.file_name())
.map(|n| format!("{}_sankoff.tre", n.to_string_lossy()))
.unwrap_or_else(|| "sankoff.tre".into());
// TNT's plain command parser has no comment syntax of its own — `/* */`
// and `[ ]` are only recognised inside the (separately-enabled) macro
// scripting language, and fail with "No command!" here otherwise
// (verified against this file with the local TNT binary). `quote` is
// the closest working equivalent: it prints free text and does not
// otherwise affect parsing, so it doubles as an explanation of the
// defaults below when the script is run. `;` ends a `quote` block like
// any other TNT command, so the text itself must avoid semicolons.
writeln!(f, "quote").unwrap();
writeln!(
f,
"Default search below (edit or delete this block to run your own strategy):"
)
.unwrap();
writeln!(
f,
" hold N : size of TNT's tree buffer (how many equally-parsimonious"
)
.unwrap();
writeln!(
f,
" trees it keeps in memory at once), 20 is a small, fast"
)
.unwrap();
writeln!(
f,
" default, raise it if mult reports it had to drop trees."
)
.unwrap();
writeln!(
f,
" mult : traditional search (random addition sequences followed by"
)
.unwrap();
writeln!(
f,
" TBR branch-swapping, TNT's own default replication count),"
)
.unwrap();
writeln!(
f,
" a reasonable first-pass strategy on this data's memory"
)
.unwrap();
writeln!(
f,
" footprint, xmult's ratchet/drift/tree-fusion buffers ran"
)
.unwrap();
writeln!(
f,
" this out of RAM at TNT's default mxram on this dataset."
)
.unwrap();
writeln!(
f,
" export - F : write the trees held in the buffer to file F, in"
)
.unwrap();
writeln!(
f,
" TNT/Hennig86 format ('-' means trees, as opposed to data)."
)
.unwrap();
writeln!(f, ";").unwrap();
writeln!(f, "hold 20;").unwrap();
writeln!(f, "mult;").unwrap();
writeln!(f, "export - {tre_name};").unwrap();
info!(
"TNT script → {path} (costs scaled x{cost_scale:.0}, runs a default `hold 20; mult;` \
search and writes trees to {tre_name} in TNT's working directory edit the trailing \
comment block in the script to change this)\n\
Run it with:\n \
printf 'proc {path};\\nquit;\\n' | tnt\n\
(or start `tnt` interactively and type `proc {path};`)"
);
}
-87
View File
@@ -1,87 +0,0 @@
use clap::Args;
use obikfilter::{GenomeSelector, GroupFilterParams, GroupQuorumFilter};
use obikindex::IndexMeta;
/// Ingroup/outgroup metadata-predicate quorum filtering — embeddable in any
/// command via `#[command(flatten)]` (`filter`, `dump`).
#[derive(Args)]
pub struct GroupFilterArgs {
/// Ingroup predicate (repeatable; AND). Forms: `key=v1|v2`, `key!=v`, `key~path`, `key!~path`, `*`/`all`
#[arg(long, value_name = "PRED")]
pub ingroup: Vec<String>,
/// Outgroup predicate (repeatable; OR). Forms: `key=v1|v2`, `key!=v`, `key~path`, `key!~path`, `*`/`all`
#[arg(long, value_name = "PRED")]
pub outgroup: Vec<String>,
/// Minimum number of ingroup genomes containing the k-mer
/// (negative: offset from group size, e.g. -1 = all but one)
#[arg(long, allow_hyphen_values = true)]
pub min_count: Option<isize>,
/// Maximum number of ingroup genomes containing the k-mer
/// (negative: offset from group size, e.g. -1 = all but one)
#[arg(long, allow_hyphen_values = true)]
pub max_count: Option<isize>,
/// Minimum fraction of ingroup genomes containing the k-mer [0.0-1.0]
/// (default 1.0 when --ingroup is set, 0.0 otherwise)
#[arg(long)]
pub min_frac: Option<f64>,
/// Maximum fraction of ingroup genomes containing the k-mer [0.0-1.0]
#[arg(long)]
pub max_frac: Option<f64>,
/// Minimum number of outgroup genomes containing the k-mer
/// (negative: offset from outgroup size, e.g. -1 = all but one)
#[arg(long, allow_hyphen_values = true)]
pub min_outgroup_count: Option<isize>,
/// Maximum number of outgroup genomes containing the k-mer
/// (default 0 when --outgroup is set, no constraint otherwise;
/// negative: offset from outgroup size, e.g. -1 = all but one)
#[arg(long, allow_hyphen_values = true)]
pub max_outgroup_count: Option<isize>,
/// Minimum fraction of outgroup genomes containing the k-mer [0.0-1.0]
#[arg(long)]
pub min_outgroup_frac: Option<f64>,
/// Maximum fraction of outgroup genomes containing the k-mer [0.0-1.0]
#[arg(long)]
pub max_outgroup_frac: Option<f64>,
/// Per-genome count threshold to consider a genome as "containing" the k-mer (default 0)
#[arg(long, default_value = "0")]
pub presence_threshold: u32,
}
impl GroupFilterArgs {
/// Parse `--ingroup`/`--outgroup` and build the quorum filter. Exits on error.
pub fn build_filter(&self, meta: &IndexMeta) -> GroupQuorumFilter {
let selector = GenomeSelector::parse(&self.ingroup, &self.outgroup).unwrap_or_else(|e| {
eprintln!("error in --ingroup/--outgroup: {e}");
std::process::exit(1);
});
selector
.build_group_filter(
meta,
GroupFilterParams {
threshold: self.presence_threshold,
min_count: self.min_count,
max_count: self.max_count,
min_frac: self.min_frac,
max_frac: self.max_frac,
min_outgroup_count: self.min_outgroup_count,
max_outgroup_count: self.max_outgroup_count,
min_outgroup_frac: self.min_outgroup_frac,
max_outgroup_frac: self.max_outgroup_frac,
},
)
.unwrap_or_else(|e| {
eprintln!("error in filter parameters: {e}");
std::process::exit(1);
})
}
}
-343
View File
@@ -1,343 +0,0 @@
use std::io::{self, BufWriter, Write};
use std::path::PathBuf;
use std::sync::Arc;
use std::sync::atomic::{AtomicU32, AtomicU64, Ordering};
use std::time::Instant;
use clap::Args;
use obikidxcache::index_cache::IndexCache;
use obikindex::KmerIndex;
use obikindex::layer::IndexMode;
use obikquery::process_chunk;
use obikrope::Rope;
use obipipeline::{Throttled, ThrottleGuard, throttle};
use obiread::chunk::read_sequence_chunks_sized;
use obisys::{Reporter, Stage, available_memory_bytes, spinner};
use tracing::{debug, info};
// ── Pipeline data ─────────────────────────────────────────────────────────────
enum QueryData {
Path(Throttled<PathBuf>),
Chunk(Rope),
Output(Vec<u8>),
}
// SAFETY: Rope contains Cell<u8> which is !Sync, but pipeline items are owned
// exclusively through channels — no item is ever shared across threads.
unsafe impl Send for QueryData {}
unsafe impl Sync for QueryData {}
// ── CLI ───────────────────────────────────────────────────────────────────────
#[derive(Args)]
pub struct QueryArgs {
/// Index directory
pub index: PathBuf,
/// Input sequences (FASTA/FASTQ, optionally gzip-compressed)
#[arg(num_args = 1..)]
pub inputs: Vec<String>,
/// Report per-position coverage vectors per genome (adds "coverage" to JSON)
#[arg(long)]
pub detail: bool,
/// Enable 1-mismatch approximate matching
#[arg(long)]
pub mismatch: bool,
/// Count k-mers absent from the index (adds kmer_missing annotation)
#[arg(long)]
pub count_missing: bool,
/// Report per-genome presence (0/1) instead of raw counts
#[arg(long)]
pub force_presence: bool,
/// Minimum accumulated match count to declare a genome present (implies --force-presence)
#[arg(long, default_value_t = 1)]
pub presence_threshold: u32,
/// Override the Findere z parameter from index metadata
#[arg(short = 'z', long)]
pub findere_z: Option<usize>,
/// Number of worker threads
#[arg(
short = 'T',
long,
default_value_t = obisys::effective_parallelism()
)]
pub threads: usize,
/// I/O chunk size in MiB (default: auto-sized from available RAM and thread count)
#[arg(long)]
pub chunk_size: Option<usize>,
/// Maximum number of input files open simultaneously.
/// Defaults to threads/4 (minimum 1). Keep below the number of workers
/// to ensure CPU workers are always available for the transform stage.
#[arg(long)]
pub max_open_files: Option<usize>,
}
impl QueryArgs {
pub fn effective_max_open(&self) -> usize {
self.max_open_files
.unwrap_or_else(|| (self.threads / 4).max(1))
.max(1)
}
}
// ── GuardedChunkIter — keeps the throttle slot guard alive until the file is exhausted ──
/// Wraps a per-file `Rope` chunk iterator together with its `ThrottleGuard`,
/// so the guard (and the throttle slot it holds) is only released once the
/// file has been fully read — never earlier, never held past that point.
struct GuardedChunkIter {
inner: Box<dyn Iterator<Item = Rope> + Send>,
_guard: ThrottleGuard,
files_open: Arc<AtomicU32>,
}
impl Iterator for GuardedChunkIter {
type Item = Rope;
fn next(&mut self) -> Option<Rope> {
self.inner.next()
}
}
impl Drop for GuardedChunkIter {
fn drop(&mut self) {
self.files_open.fetch_sub(1, Ordering::Relaxed);
}
}
// ── Entry point ───────────────────────────────────────────────────────────────
pub fn run(args: QueryArgs) {
let idx = Arc::new(KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
}));
let k = idx.kmer_size();
let genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
});
let n_genomes = genomes.len();
let genomes = Arc::new(genomes);
let n_partitions = idx.n_partitions();
let with_counts = idx.meta().config.with_counts;
let n_workers = args.threads.max(1);
// Every partition/layer the query might touch is opened once, up front,
// and shared (via Arc) across every `obipipeline` worker — a query pass
// is then pure in-memory lookups, never a per-chunk disk open (the
// previous `obikindex`-based design's cost). `IndexCache` owns its
// `Arc<KmerIndex>`, so it's itself `'static`-capable, satisfying
// obipipeline's `Send + Sync + 'static` requirement on pipeline data —
// see `obikquery::query_layer`'s doc comment for why that rules out a
// borrow-based cache here.
let cache = Arc::new(IndexCache::new(Arc::clone(&idx), None));
// Chunk size: each chunk stays in memory for its entire processing lifetime.
//
// Per-chunk memory is not a dense n_genomes-wide buffer — it scales with
// *actual hit count*, not with total_kmers_in_chunk × n_genomes
// unconditionally. BYTES_PER_KMER_PER_GENOME below is therefore a
// pathological-case bound, not a typical-case estimate: it protects
// against a fully-dense hit pattern (every k-mer of the query matching
// every genome — a degenerate case, e.g. low-complexity input theta-
// filtering should mostly reject, or an index of near-duplicate genomes),
// where by_genome and confirmed_by_genome (obikquery::chunk::process_chunk)
// both end up holding one (seq_idx, pos, value) entry — 3 × u32 = 12
// bytes — per (k-mer, genome) pair, and *coexist simultaneously* (by_genome
// isn't freed before confirmed_by_genome is built), for a worst case of
// ~24 bytes/pair before Vec growth slack. `cov` remains fully dense when
// --detail is set, still roughly doubling the n_genomes-scaled cost.
//
// For realistic, sparse hit patterns actual memory is far below this
// bound — see the "sparse memory retained" debug log in process_chunk,
// which reports the empirical bytes-per-raw-byte multiplier actually
// observed per chunk, directly comparable to BYTES_PER_KMER_PER_GENOME
// below. Tightening this constant for typical-case throughput (at the
// cost of pathological-case safety margin) is a deliberate tuning
// decision to make from that data, not something to guess at here.
//
// BASE_OVERHEAD approximates what scales with chunk_bytes alone,
// independent of n_genomes: the Rope itself, parsed SeqRecord sequence +
// normalised bytes, the superkmer dedup map, and the JSON output buffer.
// Like the n_genomes-scaled term, this is an estimate — validate against
// actual peak RSS (Stage::stop's `rss` in the summary table) on real
// workloads rather than trusting it blindly.
//
// We target ≤ 50 % of available RAM across all concurrent workers
// (SAFETY_FACTOR).
const BASE_OVERHEAD: u64 = 4;
const BYTES_PER_KMER_PER_GENOME: u64 = 8; // pathological-case bound — see comment above
const SAFETY_FACTOR: u64 = 2;
let detail_factor: u64 = if args.detail { 2 } else { 1 };
let overhead_multiplier =
BASE_OVERHEAD + n_genomes as u64 * BYTES_PER_KMER_PER_GENOME * detail_factor;
let chunk_bytes = args
.chunk_size
.map(|mb| mb * 1024 * 1024)
.unwrap_or_else(|| {
let avail = available_memory_bytes();
let computed = avail / (n_workers as u64 * overhead_multiplier * SAFETY_FACTOR);
computed.clamp(4 * 1024 * 1024, 256 * 1024 * 1024) as usize
});
debug!(
chunk_bytes,
n_genomes,
detail = args.detail,
overhead_multiplier,
estimated_peak_chunk_bytes = chunk_bytes as u64 * overhead_multiplier,
"chunk-size formula resolved"
);
let effective_z: usize = args
.findere_z
.unwrap_or_else(|| match idx.meta().config.evidence {
IndexMode::Approx { z, .. } | IndexMode::Hybrid { z, .. } => z as usize,
IndexMode::Exact => 1,
});
info!(
"query: k={k}, {} genome(s), with_counts={with_counts}, z={effective_z}, \
mismatch={}, detail={}",
n_genomes, args.mismatch, args.detail
);
if args.mismatch {
eprintln!("warning: --mismatch not yet implemented, ignored");
}
let detail = args.detail;
let count_missing = args.count_missing;
let force_presence = args.force_presence;
let presence_threshold = args.presence_threshold;
// Throttled iterator over input file paths: at most `effective_max_open()`
// files are open at once. Opening + decompressing + chunking each file is
// a Flat pipeline stage, executed across the `n_workers` pool.
info!("query: chunk_size={}MiB, max_open_files={}", chunk_bytes / (1024 * 1024), args.effective_max_open());
let paths: Vec<PathBuf> = args.inputs.iter().map(PathBuf::from).collect();
let path_source = throttle(paths.into_iter(), args.effective_max_open());
// Instrumentation: total bytes processed (for the EMA throughput readout),
// number of files currently open/being chunked, and number of chunks
// currently being processed by a worker — all read from the spinner loop
// below, updated from inside the pipe closures.
let total_bytes = Arc::new(AtomicU64::new(0));
let files_open = Arc::new(AtomicU32::new(0));
let chunks_active = Arc::new(AtomicU32::new(0));
let pipe = obipipeline::make_pipe! {
QueryData : Throttled<PathBuf> => Vec<u8>,
|| {
let files_open = Arc::clone(&files_open);
move |pw: Throttled<PathBuf>| -> GuardedChunkIter {
let path = pw.item;
let guard = pw.guard;
let path_str = path.to_str().unwrap_or("").to_owned();
files_open.fetch_add(1, Ordering::Relaxed);
let open_start = Instant::now();
// Hard-exit on file-open failure (mirrors the previous behaviour):
// propagating this as a pipeline Err would hit a known scheduler
// hang on early stage errors (obipipeline::scheduler::WorkerPool::run
// breaks its main loop without unblocking the still-running source
// thread, so the final `h.join()` never returns) — worth fixing in
// obipipeline itself, but out of scope here; sidestepping it like the
// original code already did is the safe choice for this change.
let iter = read_sequence_chunks_sized(&path_str, chunk_bytes).unwrap_or_else(|e| {
eprintln!("error opening {path_str}: {e}");
std::process::exit(1);
});
debug!(
path = %path_str,
open_ms = open_start.elapsed().as_millis() as u64,
"opened query input file"
);
let err_path = path_str.clone();
GuardedChunkIter {
inner: Box::new(iter.filter_map(move |r| match r {
Ok(rope) => Some(rope),
Err(e) => {
eprintln!("read error: {err_path}: {e}");
None
}
})),
_guard: guard,
files_open: Arc::clone(&files_open),
}
}
} : Path => Chunk,
| {
let cache = Arc::clone(&cache);
let genomes = Arc::clone(&genomes);
let total_bytes = Arc::clone(&total_bytes);
let chunks_active = Arc::clone(&chunks_active);
move |rope: Rope| {
chunks_active.fetch_add(1, Ordering::Relaxed);
let bytes = rope.len() as u64;
let out = process_chunk(
&cache, rope, k, n_genomes, n_partitions, with_counts,
effective_z, detail, count_missing, force_presence, presence_threshold,
&genomes,
);
total_bytes.fetch_add(bytes, Ordering::Relaxed);
chunks_active.fetch_sub(1, Ordering::Relaxed);
out
}
} : Chunk => Output,
};
let t = Stage::start("query");
let pb = spinner("query");
let mut ema_rate: f64 = 0.0;
let mut last_t = Instant::now();
let mut last_bytes: u64 = 0;
const ALPHA: f64 = 0.15;
let mut out = BufWriter::new(io::stdout());
for block in pipe.apply(path_source, n_workers, 2) {
if !block.is_empty() {
out.write_all(&block).expect("write error");
}
let now = Instant::now();
let dt = now.duration_since(last_t).as_secs_f64();
if dt > 0.1 {
let total = total_bytes.load(Ordering::Relaxed);
let instant = (total - last_bytes) as f64 / dt;
ema_rate = ALPHA * instant + (1.0 - ALPHA) * ema_rate;
last_t = now;
last_bytes = total;
let bp = total as f64;
let (count_str, rate_str) = if bp >= 1e9 {
(format!("{:.2} GB", bp / 1e9), format!("{:.0} MB/s", ema_rate / 1e6))
} else {
(format!("{:.0} MB", bp / 1e6), format!("{:.0} MB/s", ema_rate / 1e6))
};
let active = chunks_active.load(Ordering::Relaxed);
let open = files_open.load(Ordering::Relaxed);
pb.set_message(format!("{count_str} {rate_str} [files open: {open}, chunks in flight: {active}]"));
}
}
out.flush().expect("flush error");
pb.finish_and_clear();
let mut rep = Reporter::new();
rep.push(t.stop());
rep.print();
}
-144
View File
@@ -1,144 +0,0 @@
use std::path::PathBuf;
use clap::{Args, ValueEnum};
use obikalgorithm::Algorithm;
use obikindex::KmerIndex;
use obikselect::{AggOp, ColumnSpecParams, Select, build_output_cols};
use obisys::{Progress, Reporter, Stage, progress_bar};
use tracing::info;
#[derive(Debug, Clone, Copy, PartialEq, Eq, ValueEnum)]
pub enum AggOpArg {
Any,
All,
None,
Sum,
Min,
Max,
}
impl From<AggOpArg> for AggOp {
fn from(a: AggOpArg) -> Self {
match a {
AggOpArg::Any => AggOp::Any,
AggOpArg::All => AggOp::All,
AggOpArg::None => AggOp::None,
AggOpArg::Sum => AggOp::Sum,
AggOpArg::Min => AggOp::Min,
AggOpArg::Max => AggOp::Max,
}
}
}
#[derive(Args)]
pub struct SelectArgs {
/// Source index directory
pub source: PathBuf,
/// Output index directory
#[arg(short, long)]
pub output: PathBuf,
/// Define a named group: `<name>:<pred>` (repeatable; mutually exclusive with --aggregate-by)
#[arg(long, value_name = "NAME:PRED", conflicts_with = "aggregate_by")]
pub group: Vec<String>,
/// Per-group aggregation operator: `<name>:<op>` (repeatable)
#[arg(long, value_name = "NAME:OP")]
pub group_op: Vec<String>,
/// Auto-create one group per unique value of metadata key <KEY>
#[arg(long, value_name = "KEY", conflicts_with = "group")]
pub aggregate_by: Option<String>,
/// Aggregation operator for all auto-generated groups
#[arg(long, value_name = "OP")]
pub aggregate_op: Option<AggOpArg>,
/// Output columns in order: group names or genome labels, comma-separated
#[arg(long, value_name = "COL,...", value_delimiter = ',')]
pub select: Option<Vec<String>>,
/// Minimum count to consider a genome as "carrying" the k-mer (logical ops only)
#[arg(long, default_value = "0")]
pub presence_threshold: u32,
/// Pack the output's presence matrices in the dense format instead of the default sparse one
#[arg(long, default_value_t = false)]
pub dense: bool,
/// Overwrite existing output directory
#[arg(short, long)]
pub force: bool,
}
/// Split a repeatable `<name>:<value>` argument. Exits on malformed input.
fn parse_name_value(s: &str, flag: &str) -> (String, String) {
match s.find(':') {
Some(pos) => (s[..pos].trim().to_string(), s[pos + 1..].to_string()),
None => {
eprintln!("error in {flag}: expected <name>:<value>, got: {s}");
std::process::exit(1);
}
}
}
pub fn run(args: SelectArgs) {
let src = KmerIndex::open(&args.source).unwrap_or_else(|e| {
eprintln!("error opening source index: {e}");
std::process::exit(1);
});
let group_preds: Vec<(String, String)> =
args.group.iter().map(|s| parse_name_value(s, "--group")).collect();
let group_ops: Vec<(String, String)> =
args.group_op.iter().map(|s| parse_name_value(s, "--group-op")).collect();
let src_is_count = src.meta().config.with_counts;
let (specs, output_presence) = build_output_cols(
&src.meta(),
ColumnSpecParams {
group_preds: &group_preds,
aggregate_by: args.aggregate_by.as_deref(),
group_ops: &group_ops,
aggregate_op: args.aggregate_op.map(AggOp::from),
select: args.select.as_deref(),
src_is_count,
},
)
.unwrap_or_else(|e| {
eprintln!("error building output columns: {e}");
std::process::exit(1);
});
let n_genomes = src.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
}).len();
info!(
"select: {} genome(s) → {} output column(s), output={}",
n_genomes,
specs.len(),
if output_presence { "presence" } else { "count" },
);
let mut rep = Reporter::new();
let t = Stage::start("select");
let pb = progress_bar("select", src.n_partitions() as u64, "partitions");
let mut alg = Select::new(&src, &args.output, &specs, output_presence)
.threshold(args.presence_threshold)
.force(args.force)
.sparse(!args.dense)
.on_progress(|_: Progress| pb.inc(1));
let dst = alg.run().unwrap_or_else(|e| {
eprintln!("select error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
rep.push(t.stop());
info!("selected index → {}", dst.dir().display());
alg.reporter().print();
rep.print();
}
-72
View File
@@ -1,72 +0,0 @@
use std::io::{self, BufWriter, Write};
use std::path::PathBuf;
use clap::Args;
use obifastwrite::write_scatter;
use obikseq::{RoutableSuperKmer, set_k, set_m};
use obipipeline::{Throttled, throttle};
use crate::cli::{CommonArgs, PipelineData, partitions_to_bits};
#[derive(Args)]
pub struct SuperkmerArgs {
#[command(flatten)]
pub common: CommonArgs,
}
// ── Stage functions ───────────────────────────────────────────────────────────
fn write_batch(
batch: Vec<RoutableSuperKmer>,
out: &mut BufWriter<io::Stdout>,
partition_bits: usize,
k: usize,
m: usize,
) -> io::Result<()> {
let partition_mask = (1u64 << partition_bits) - 1;
for rsk in batch {
let minimizer = *rsk.minimizer();
let partition = (minimizer.seq_hash() & partition_mask) as usize;
write_scatter(rsk.superkmer(), out, k, m, partition, minimizer)?;
}
Ok(())
}
// ── Entry point ───────────────────────────────────────────────────────────────
pub fn run(args: SuperkmerArgs) {
args.common.validate();
let k = args.common.kmer_size;
let m = args.common.minimizer_size;
let theta = args.common.theta;
let level_max = args.common.level_max;
let partition_bits = partitions_to_bits(args.common.partitions);
let n_workers = args.common.threads.max(1);
let max_open = args.common.effective_max_open();
set_k(k);
set_m(m);
let path_source = throttle(args.common.seqfile_paths(), max_open);
let pipe = obipipeline::make_pipe! {
PipelineData : Throttled<PathBuf> => Vec<RoutableSuperKmer>,
||? {
let k = k;
move |pw: Throttled<PathBuf>| {
let path_str = pw.item.to_str().unwrap_or("").to_owned();
let _guard = pw.guard;
obiread::open_nuc_stream(&path_str, k)
}
} : Path => NucPage,
| { move |page| obiskbuilder::build_superkmers_page(page, k, level_max, theta) } : NucPage => Batch,
};
let mut out = BufWriter::new(io::stdout());
for batch in pipe.apply(path_source, n_workers, 1) {
write_batch(batch, &mut out, partition_bits, k, m).expect("write error");
}
out.flush().expect("flush error");
}
-47
View File
@@ -1,47 +0,0 @@
use std::io::{self, BufWriter};
use std::path::PathBuf;
use std::sync::Arc;
use clap::Args;
use obikdump::IndexUnitigs;
use obikfilter::KmerFilter;
use obikindex::KmerIndex;
use obisys::progress_bar;
use tracing::info;
use super::predicate::GroupFilterArgs;
#[derive(Args)]
pub struct UnitigArgs {
/// Index directory
pub index: PathBuf,
#[command(flatten)]
pub group_filter: GroupFilterArgs,
}
pub fn run(args: UnitigArgs) {
let idx = Arc::new(KmerIndex::open(&args.index).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
}));
info!(
"unitig: building de Bruijn graph from {} partition(s) (k={})",
idx.n_partitions(),
idx.kmer_size(),
);
let filters: Vec<Box<dyn KmerFilter>> = vec![Box::new(args.group_filter.build_filter(&idx.meta()))];
let pb = progress_bar("unitig", idx.n_partitions() as u64, "partitions");
let mut out = BufWriter::new(io::stdout());
let n = idx.write_unitigs(&mut out, &filters, || pb.inc(1)).unwrap_or_else(|e| {
eprintln!("unitig error: {e}");
std::process::exit(1);
});
pb.finish_and_clear();
info!("unitig: {n} unitig(s) written");
}
-113
View File
@@ -1,113 +0,0 @@
use std::path::PathBuf;
use std::sync::Arc;
use obikalgorithm::Algorithm;
use obikindex::{GenomeInfo, KmerIndex};
use obikstats::{BitsPerKmer, GenomeKmerCounts};
use tracing::info;
pub(super) fn run_stats(index_path: &PathBuf) {
let idx = Arc::new(KmerIndex::open(index_path).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
}));
let genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
});
let (total, per_genome) = GenomeKmerCounts::new(Arc::clone(&idx)).run().unwrap_or_else(|e| {
eprintln!("error computing stats: {e}");
std::process::exit(1);
});
println!("genome,n_kmers");
for (g, &n) in genomes.iter().zip(per_genome.iter()) {
println!("{},{}", g.label, n);
}
println!("total,{total}");
}
pub(super) fn run_bits_per_kmer(index_path: &PathBuf) {
let idx = Arc::new(KmerIndex::open(index_path).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
}));
let stats = BitsPerKmer::new(idx).run().unwrap_or_else(|e| {
eprintln!("error computing bits/kmer: {e}");
std::process::exit(1);
});
println!("k-mers : {}", stats.n_kmers);
println!("genomes : {}", stats.n_genomes);
println!("mphf : {:6.2} bits/kmer", stats.mphf);
println!("evidence : {:6.2} bits/kmer", stats.evidence);
println!(
"matrix : {:6.2} bits/kmer ({:.2} bits/kmer/genome)",
stats.matrix, stats.matrix_per_genome
);
println!("total : {:6.2} bits/kmer", stats.total);
}
pub(super) fn run_rename(index_path: &PathBuf, spec: &str) {
let (old_label, new_label) = parse_rename_spec(spec);
let idx = KmerIndex::open(index_path).unwrap_or_else(|e| {
eprintln!("error opening index: {e}");
std::process::exit(1);
});
let genomes = idx.meta().genomes().unwrap_or_else(|e| {
eprintln!("error reading index metadata: {e}");
std::process::exit(1);
});
let pos = genomes
.iter()
.position(|g| g.label == old_label)
.unwrap_or_else(|| {
eprintln!("error: genome '{old_label}' not found in index");
std::process::exit(1);
});
GenomeInfo::validate_label(&new_label).unwrap_or_else(|e| {
eprintln!("error: --new-label: {e}");
std::process::exit(1);
});
if genomes.iter().any(|g| g.label == new_label) {
eprintln!("error: label '{new_label}' already exists in index");
std::process::exit(1);
}
idx.meta().rename_genome(pos, new_label.clone()).unwrap_or_else(|e| {
eprintln!("error writing index metadata: {e}");
std::process::exit(1);
});
let spectrums_dir = index_path.join("spectrums");
let old_spectrum = spectrums_dir.join(format!("{old_label}.json"));
let new_spectrum = spectrums_dir.join(format!("{new_label}.json"));
if old_spectrum.exists() {
std::fs::rename(&old_spectrum, &new_spectrum).unwrap_or_else(|e| {
eprintln!("warning: could not rename spectrum file: {e}");
});
}
info!("renamed genome '{old_label}' → '{new_label}'");
}
fn parse_rename_spec(spec: &str) -> (String, String) {
let eq = spec.find('=').unwrap_or_else(|| {
eprintln!("error: --new-label expects NEW_LABEL=OLD_LABEL, got '{spec}'");
std::process::exit(1);
});
let new = spec[..eq].trim().to_string();
let old = spec[eq + 1..].trim().to_string();
if old.is_empty() || new.is_empty() {
eprintln!("error: --new-label: both old and new labels must be non-empty");
std::process::exit(1);
}
(old, new)
}
-76
View File
@@ -1,76 +0,0 @@
mod maintenance;
mod partition_stats;
use std::path::PathBuf;
use clap::Args;
use maintenance::{run_bits_per_kmer, run_stats, run_rename};
use partition_stats::run_partition_stats;
#[derive(Args)]
pub struct UtilsArgs {
/// Index directories to operate on (one or more)
#[arg(required = true, num_args = 1..)]
pub indexes: Vec<PathBuf>,
/// Set a new genome label: NEW_LABEL=OLD_LABEL (single-index only)
#[arg(long, value_name = "NEW=OLD")]
pub new_label: Option<String>,
/// Print bits-per-kmer statistics (single-index only)
#[arg(long)]
pub bits_per_kmer: bool,
/// Print per-genome k-mer counts as CSV (single-index only)
#[arg(long)]
pub stats: bool,
/// Print partition size distribution report (accepts multiple indexes)
#[arg(long)]
pub partition_stats: bool,
/// Write per-(partition, source) raw data as CSV to FILE (used with --partition-stats)
#[arg(long, value_name = "FILE")]
pub csv: Option<PathBuf>,
}
pub fn run(args: UtilsArgs) {
let mut any = false;
if let Some(spec) = &args.new_label {
any = true;
run_rename(single_index(&args), spec);
}
if args.bits_per_kmer {
any = true;
run_bits_per_kmer(single_index(&args));
}
if args.stats {
any = true;
run_stats(single_index(&args));
}
if args.partition_stats {
any = true;
run_partition_stats(&args.indexes, args.csv.as_deref());
}
if !any {
eprintln!(
"utils: no operation specified. \
Available: --new-label, --bits-per-kmer, --stats, --partition-stats"
);
std::process::exit(1);
}
}
fn single_index(args: &UtilsArgs) -> &PathBuf {
if args.indexes.len() > 1 {
eprintln!("utils: this option requires exactly one index (got {})", args.indexes.len());
std::process::exit(1);
}
&args.indexes[0]
}
@@ -1,206 +0,0 @@
use std::io::{self, Write};
use std::path::PathBuf;
use obikindex::KmerIndex;
/// Per-partition, per-source byte count of all unitigs.bin files summed across layers.
struct PartRow {
partition: usize,
source: String,
bytes: u64,
}
fn collect_rows(indexes: &[PathBuf]) -> Vec<PartRow> {
let mut rows = Vec::new();
for path in indexes {
let idx = KmerIndex::open(path).unwrap_or_else(|e| {
eprintln!("error opening index {}: {e}", path.display());
std::process::exit(1);
});
let name = path
.file_name()
.map(|n| n.to_string_lossy().into_owned())
.unwrap_or_else(|| path.display().to_string());
let n_parts = idx.n_partitions();
for i in 0..n_parts {
let mut bytes = 0u64;
let n_layers = idx.n_layers(i).unwrap_or_else(|e| {
eprintln!("error reading partition {i} of {}: {e}", path.display());
std::process::exit(1);
});
for l in 0..n_layers {
let p = idx.layer_unitigs_path(i, l).unwrap_or_else(|e| {
eprintln!("error reading layer {l} of partition {i} of {}: {e}", path.display());
std::process::exit(1);
});
if let Ok(m) = std::fs::metadata(&p) {
bytes += m.len();
}
}
rows.push(PartRow { partition: i, source: name.clone(), bytes });
}
}
rows
}
/// Sum bytes per partition across all sources.
fn partition_totals(rows: &[PartRow], n_parts: usize) -> Vec<u64> {
let mut totals = vec![0u64; n_parts];
for r in rows {
totals[r.partition] += r.bytes;
}
totals
}
fn stats_summary(totals: &[u64]) -> (u64, u64, f64, f64, u64, u64, u64) {
let mut sorted = totals.to_vec();
sorted.sort_unstable();
let n = sorted.len();
let min = sorted[0];
let max = sorted[n - 1];
let mean = sorted.iter().sum::<u64>() as f64 / n as f64;
let median = if n % 2 == 0 {
(sorted[n / 2 - 1] + sorted[n / 2]) as f64 / 2.0
} else {
sorted[n / 2] as f64
};
let p95 = sorted[(n as f64 * 0.95) as usize];
let p99 = sorted[(n as f64 * 0.99) as usize];
let variance = sorted
.iter()
.map(|&v| (v as f64 - mean).powi(2))
.sum::<f64>()
/ n as f64;
let std_dev = variance.sqrt();
(min, max, mean, median, p95, p99, std_dev as u64)
}
fn human_bytes(b: u64) -> String {
if b >= 1 << 30 {
format!("{:.1} GB", b as f64 / (1u64 << 30) as f64)
} else if b >= 1 << 20 {
format!("{:.1} MB", b as f64 / (1u64 << 20) as f64)
} else if b >= 1 << 10 {
format!("{:.1} KB", b as f64 / (1u64 << 10) as f64)
} else {
format!("{b} B")
}
}
fn ascii_histogram(totals: &[u64], n_buckets: usize, bar_width: usize) -> String {
let min = *totals.iter().min().unwrap();
let max = *totals.iter().max().unwrap();
if min == max {
return format!(" (all partitions identical: {})\n", human_bytes(min));
}
let bucket_size = (max - min).max(1) as f64 / n_buckets as f64;
let mut counts = vec![0usize; n_buckets];
for &v in totals {
let b = (((v - min) as f64 / bucket_size) as usize).min(n_buckets - 1);
counts[b] += 1;
}
let max_count = *counts.iter().max().unwrap();
let mut out = String::new();
for (i, &c) in counts.iter().enumerate() {
let lo = min + (i as f64 * bucket_size) as u64;
let hi = min + ((i + 1) as f64 * bucket_size) as u64;
let bar_len = if max_count > 0 { c * bar_width / max_count } else { 0 };
let bar = "".repeat(bar_len);
out.push_str(&format!(
" {:>8} – {:>8} │{:<width$} {}\n",
human_bytes(lo),
human_bytes(hi),
bar,
c,
width = bar_width
));
}
out
}
pub(super) fn run_partition_stats(indexes: &[PathBuf], csv_path: Option<&std::path::Path>) {
let rows = collect_rows(indexes);
if rows.is_empty() {
eprintln!("partition-stats: no data found");
std::process::exit(1);
}
let n_parts = rows.iter().map(|r| r.partition).max().unwrap() + 1;
let totals = partition_totals(&rows, n_parts);
let (min, max, mean, median, p95, p99, std_dev) = stats_summary(&totals);
// outliers: > median + 1.5 × IQR (approximate via > 1.5 × median as fallback)
let mut sorted_t = totals.clone();
sorted_t.sort_unstable();
let q1 = sorted_t[n_parts / 4] as f64;
let q3 = sorted_t[3 * n_parts / 4] as f64;
let iqr = q3 - q1;
let outlier_threshold = q3 + 1.5 * iqr;
let mut out = String::new();
out.push_str("# Partition size report\n\n");
out.push_str(&format!(
"Sources: {} \nPartitions: {} \n\n",
indexes.len(),
n_parts
));
out.push_str("## Summary statistics (total unitigs.bin bytes per partition, sum across sources)\n\n");
out.push_str("| Stat | Value |\n|---|---|\n");
out.push_str(&format!("| min | {} |\n", human_bytes(min)));
out.push_str(&format!("| max | {} |\n", human_bytes(max)));
out.push_str(&format!("| mean | {} |\n", human_bytes(mean as u64)));
out.push_str(&format!("| median | {} |\n", human_bytes(median as u64)));
out.push_str(&format!("| p95 | {} |\n", human_bytes(p95)));
out.push_str(&format!("| p99 | {} |\n", human_bytes(p99)));
out.push_str(&format!("| std | {} |\n", human_bytes(std_dev)));
out.push_str(&format!("| max/median ratio | {:.2}× |\n\n", max as f64 / median));
out.push_str("## Histogram\n\n```\n");
out.push_str(&ascii_histogram(&totals, 30, 40));
out.push_str("```\n\n");
let outliers: Vec<(usize, u64)> = totals
.iter()
.enumerate()
.filter(|(_, v)| **v as f64 > outlier_threshold)
.map(|(i, v)| (i, *v))
.collect();
if outliers.is_empty() {
out.push_str("## Outliers\n\nNone (threshold: Q3 + 1.5×IQR = ");
out.push_str(&human_bytes(outlier_threshold as u64));
out.push_str(").\n");
} else {
out.push_str(&format!(
"## Outliers (> Q3 + 1.5×IQR = {})\n\n| Partition | Total size | Ratio to median |\n|---|---|---|\n",
human_bytes(outlier_threshold as u64)
));
for (i, v) in &outliers {
out.push_str(&format!(
"| {} | {} | {:.2}× |\n",
i,
human_bytes(*v),
*v as f64 / median
));
}
out.push('\n');
}
print!("{out}");
if let Some(csv_out) = csv_path {
let file = std::fs::File::create(csv_out).unwrap_or_else(|e| {
eprintln!("error creating CSV file {}: {e}", csv_out.display());
std::process::exit(1);
});
let mut w = io::BufWriter::new(file);
writeln!(w, "partition,source,bytes").unwrap();
for r in &rows {
writeln!(w, "{},{},{}", r.partition, r.source, r.bytes).unwrap();
}
eprintln!("CSV written to {}", csv_out.display());
}
}
-72
View File
@@ -1,72 +0,0 @@
mod cli;
mod cmd;
use clap::{Parser, Subcommand};
use tracing_subscriber::{EnvFilter, fmt};
#[derive(Parser)]
#[command(name = "obikmer2", about = "DNA k-mer tools", version)]
struct Cli {
#[command(subcommand)]
command: Commands,
}
#[derive(Subcommand)]
enum Commands {
/// Build the complete genome index (scatter → dereplicate → count → layered MPHF)
Index(cmd::index::IndexArgs),
/// Extract super-k-mers from input sequences and scatter them by partition
Superkmer(cmd::superkmer::SuperkmerArgs),
/// Merge multiple genome indexes into one
Merge(cmd::merge::MergeArgs),
/// Filter kmers out of an index by genome metadata / abundance / complexity
Filter(cmd::filter::FilterArgs),
/// Project/aggregate genome columns into a new index
Select(cmd::select::SelectArgs),
/// Dump an index's kmers as a CSV table
Dump(cmd::dump::DumpArgs),
/// Query sequences against an index, annotating each with per-genome matches
Query(cmd::query::QueryArgs),
/// Assemble an index's kmers into unitigs and write them as FASTA
Unitig(cmd::unitig::UnitigArgs),
/// Pack an index's matrices into single-file, sparse format by default (--dense to opt out), in place
Pack(cmd::pack::PackArgs),
/// Estimate approximate-evidence false-positive rates for given parameters
Estimate(cmd::estimate::EstimateArgs),
/// Read/write genome metadata (CSV) on an already-built index
Annotate(cmd::annotate::AnnotateArgs),
/// Maintenance/inspection operations on already-built indexes
Utils(cmd::utils::UtilsArgs),
/// Convert an index's evidence representation (exact/approximate/hybrid), in place
Convert(cmd::convert::ConvertArgs),
/// Genome-vs-genome distance matrix (+ optional NJ/UPGMA tree) — partial transfer,
/// sibling-annex-based operations are not yet ported (see obikmer's own `phylo`)
Phylo(cmd::phylo::PhyloArgs),
}
fn main() {
fmt()
.with_env_filter(
EnvFilter::try_from_default_env().unwrap_or_else(|_| EnvFilter::new("info")),
)
.with_writer(std::io::stderr)
.init();
let cli = Cli::parse();
match cli.command {
Commands::Index(args) => cmd::index::run(args),
Commands::Superkmer(args) => cmd::superkmer::run(args),
Commands::Merge(args) => cmd::merge::run(args),
Commands::Filter(args) => cmd::filter::run(args),
Commands::Select(args) => cmd::select::run(args),
Commands::Dump(args) => cmd::dump::run(args),
Commands::Query(args) => cmd::query::run(args),
Commands::Unitig(args) => cmd::unitig::run(args),
Commands::Pack(args) => cmd::pack::run(args),
Commands::Estimate(args) => cmd::estimate::run(args),
Commands::Annotate(args) => cmd::annotate::run(args),
Commands::Utils(args) => cmd::utils::run(args),
Commands::Convert(args) => cmd::convert::run(args),
Commands::Phylo(args) => cmd::phylo::run(args),
}
}