large refactoring

This commit is contained in:
Eric Coissac
2026-08-17 09:28:53 +02:00
parent 49e66f16a2
commit 7bac0f3850
236 changed files with 77584 additions and 3733 deletions
+2 -2
View File
@@ -2,7 +2,7 @@
//! matrix — replaces `sankoff::build_cost_matrix`'s elementary-edge graph
//! and its Floyd-Warshall shortest-path closure, which double-counts
//! multi-hop transitions once IQ-TREE's own matrix exponential composes
//! them again. See `docmd/theory/evolutionary_distances.md`, "`R` via
//! them again. See `DevDocMD/theory/evolutionary_distances.md`, "`R` via
//! `exp(-cost)` is wrong for a CTMC; cardinality/composition decomposition".
//!
//! Two small, directly-estimated first-order Markov models (diagonals
@@ -133,7 +133,7 @@ fn best_pairing_cost(lost: &[u8], gained: &[u8], p_comp: &[[f64; 4]; 4]) -> f64
/// `free_loss`: drop the `P_cardinality(|A|→|B|)` factor entirely (never
/// added to `log_p`) — the same low/incomplete-coverage argument that
/// justifies recoding whole-family non-detection as `?` (see
/// `docmd/theory/evolutionary_distances.md`, "Locus dropout under
/// `DevDocMD/theory/evolutionary_distances.md`, "Locus dropout under
/// incomplete coverage") applies one level down: whether a genome shows 1
/// vs 2 (etc.) detected members of a *present* family is exactly as
/// vulnerable to sampling failure as whether the family was detected at
+1 -1
View File
@@ -1,7 +1,7 @@
//! Library-level phylogenetic functionality for `obikmer`, built as
//! extension traits over `obikindex::KmerIndex` and `obilayeredmap`'s
//! generic layer types — the `phylo` CLI command is a consumer of this
//! crate, not the owner of this logic (see `docmd/architecture/siblings.md`).
//! crate, not the owner of this logic (see `DevDocMD/architecture/siblings.md`).
//!
//! Starts with [`siblings`] (family presence-mask annex, SNP distance,
//! cardinality, pseudo-alignment); further phylo-domain functionality
+12 -8
View File
@@ -8,6 +8,7 @@ use obikindex::KmerIndex;
use super::cache::PartitionCache;
use super::family_scan::{scan_layer_families, Selection};
use super::subsample::EntropyBias;
/// IUPAC ambiguity code for a per-genome family presence mask (bit `b` set
/// iff the genome carries the member whose own central base is `b`):
@@ -16,7 +17,7 @@ use super::family_scan::{scan_layer_families, Selection};
/// `N`, the same convention used for diploid heterozygous VCF/FASTA sites);
/// all 4 bits -> `N`; no bits (genome carries none of the family's observed
/// members) -> `-` (no data at this locus for this genome).
fn iupac_code(mask: u8) -> u8 {
pub(super) fn iupac_code(mask: u8) -> u8 {
match mask & 0b1111 {
0b0000 => b'-',
0b0001 => b'A',
@@ -45,7 +46,7 @@ fn iupac_code(mask: u8) -> u8 {
/// deterministic sweep order as the annex build (partition, then layer, then
/// slot) — arbitrary but stable and identical across genomes, which is all a
/// pseudo-alignment needs (there is no natural genomic coordinate to sort by
/// once flanks are dropped). See `docmd/theory/evolutionary_distances.md`,
/// once flanks are dropped). See `DevDocMD/theory/evolutionary_distances.md`,
/// "Multi-genome framing: family as pseudo-alignment column".
pub struct SnpAlignment {
/// `sequences[g]` = genome `g`'s IUPAC-coded row, same length for every
@@ -55,7 +56,7 @@ pub struct SnpAlignment {
/// Adds [`snp_pseudo_alignment`](Self::snp_pseudo_alignment) to `KmerIndex` —
/// phylo-domain functionality, kept out of `obikindex` itself (see
/// `docmd/architecture/siblings.md`).
/// `DevDocMD/architecture/siblings.md`).
pub trait SnpAlignmentExt {
/// Build the SNP-only pseudo-alignment from an already-built sibling
/// annex (run [`build_sibling_annex`](super::build::SiblingAnnexBuildExt::build_sibling_annex) first).
@@ -63,14 +64,17 @@ pub trait SnpAlignmentExt {
/// `subsample`: `--subsample N` — cap the number of variable families
/// retained to (approximately) `N`, sampled proportionally per layer
/// among non-monomorphic minorants (see
/// `docmd/architecture/siblings.md`, "`--subsample`/`--shannon`").
/// `DevDocMD/architecture/siblings.md`, "`--subsample`/`--shannon`").
/// `None` keeps today's behaviour: every non-monomorphic minorant of
/// every layer.
fn snp_pseudo_alignment(&self, subsample: Option<usize>) -> OKIResult<SnpAlignment>;
/// every layer. `entropy_bias`: `--entropy`/`--entropy-sd` — weight
/// that sample (or, with `subsample = None`, filter the whole index)
/// by a Gaussian kernel on each family's entropy — see
/// `DevDocMD/architecture/siblings.md`, "Entropy-biased selection".
fn snp_pseudo_alignment(&self, subsample: Option<usize>, entropy_bias: Option<EntropyBias>) -> OKIResult<SnpAlignment>;
}
impl SnpAlignmentExt for KmerIndex {
fn snp_pseudo_alignment(&self, subsample: Option<usize>) -> OKIResult<SnpAlignment> {
fn snp_pseudo_alignment(&self, subsample: Option<usize>, entropy_bias: Option<EntropyBias>) -> OKIResult<SnpAlignment> {
let n_parts = self.n_partitions();
let n_genomes = self.meta().genomes.len();
let with_counts = self.meta().config.with_counts;
@@ -86,7 +90,7 @@ impl SnpAlignmentExt for KmerIndex {
.map_err(OKIError::Partition)?;
let cache = Arc::new(PartitionCache::build(&partition, n_parts, with_counts)?);
let layer_dirs = super::family_scan::sibling_layer_dirs(self)?;
let selections = super::subsample::compute_selections(&layer_dirs, subsample)?;
let selections = super::subsample::compute_selections(self, &layer_dirs, subsample, entropy_bias)?;
let pb = progress_bar("snp_pseudo_alignment", layer_dirs.len() as u64, "layers");
// One layer at a time, not `par_iter()` over layers — same
+2 -2
View File
@@ -29,7 +29,7 @@ use super::{olm_to_ok, FamilyMask, SiblingAnnexBuilder, ANNEX_FILE_NAME, INDEX_S
/// unlike the fan-out `Flat` stage this replaced, needs no `Arc` sharing.
///
/// The `usize` is this k-mer's position in `iter_kmers()`'s enumeration
/// order, **not** an MPHF slot — see `docmd/architecture/siblings.md`. It
/// order, **not** an MPHF slot — see `DevDocMD/architecture/siblings.md`. It
/// is the same index the annex file is written under.
struct SourceBatch {
items: Vec<(usize, CanonicalKmer)>,
@@ -144,7 +144,7 @@ fn build_layer_sibling_annex(
// `unitigs.bin`), never by MPHF slot — this layer's own k-mers are
// known members by construction, so no evidence check, no MPHF
// slot, and no slot -> k-mer reconstruction is needed or legitimate
// here (see `docmd/architecture/siblings.md`: the MPHF is not
// here (see `DevDocMD/architecture/siblings.md`: the MPHF is not
// invertible, and evidence answers membership, not identity). No
// separate in-memory accumulator: every concurrent write goes
// straight through `SiblingAnnexBuilder::atomic_slot` into the
+2 -2
View File
@@ -42,7 +42,7 @@ pub(super) enum Mat {
Presence(Layer<PersistentBitMatrix>),
/// Same role as `Presence`, over a layer packed by `pack --sparse`
/// (`PersistentSparseBitMatrix`) instead of the dense form — see
/// `docmd/architecture/siblings.md`'s sparse-matrix section. Every
/// `DevDocMD/architecture/siblings.md`'s sparse-matrix section. Every
/// `Mat` method below dispatches to the same `Layer<D>` generic code
/// (`D: LayerData`) as `Presence`, so this arm is purely a storage
/// choice, not a behavioural difference.
@@ -62,7 +62,7 @@ impl Mat {
/// callers that already know every kmer is a member of *this* layer
/// (e.g. it came from this layer's own `iter_minorants_batch`), so the
/// evidence check `find_slot`/`find` would perform is redundant work.
/// See `docmd/architecture/siblings.md`: iteration-pipeline kmers use
/// See `DevDocMD/architecture/siblings.md`: iteration-pipeline kmers use
/// `index`, never `find`.
pub(super) fn index_batch(&self, kmers: &[CanonicalKmer]) -> Vec<usize> {
match self {
+1 -1
View File
@@ -47,7 +47,7 @@ pub trait CardinalityExt {
/// Verified empirically: removing the filter measurably worsened a
/// real IQ-TREE run (log-likelihood dropped, `NNI search needs
/// unusual large number of steps to converge` warnings appeared) — see
/// `docmd/theory/evolutionary_distances.md` for the full account.
/// `DevDocMD/theory/evolutionary_distances.md` for the full account.
/// [`base_pair_tally`](Self::base_pair_tally)'s own diagonal (`same`)
/// gets the matching restriction via `scan_family_pairs`'s new
/// `variable` flag, rather than a `family_size()` check of its own (it
+2 -2
View File
@@ -13,7 +13,7 @@ use super::family_scan::{scan_layer_families, Selection};
/// Raw p-distance restricted to loci that are single-copy in **both**
/// genomes of a pair — the "stringent / paralogy-aware" locus eligibility
/// rule (`docmd/theory/evolutionary_distances.md`, "Locus eligibility"),
/// rule (`DevDocMD/theory/evolutionary_distances.md`, "Locus eligibility"),
/// without the JC/K2P/LogDet finalisation math: `snp[i,j]` + `shared[i,j]`
/// is the number of eligible loci, `snp[i,j] / (snp[i,j] + shared[i,j])` is
/// `p_hat`. A quick, self-contained way to sanity-check the estimator
@@ -197,7 +197,7 @@ impl DistanceExt for KmerIndex {
// (never varies anywhere) isn't a SNP-adjacent
// agreement, it's genome-wide background, and would
// otherwise swamp the diagonal (see
// `docmd/theory/evolutionary_distances.md`, the
// `DevDocMD/theory/evolutionary_distances.md`, the
// ascertainment-bias regression this was reverting).
same[bi as usize] += 1;
}
+73 -11
View File
@@ -1,5 +1,5 @@
//! `--shannon` — per-family Shannon entropy over the 15 non-empty subsets
//! of `{A,C,G,T}`. See `docmd/architecture/siblings.md`, "Entropy
//! of `{A,C,G,T}`. See `DevDocMD/architecture/siblings.md`, "Entropy
//! definition — settled 2026-08-15" and "`--subsample`/`--shannon`", for
//! the design discussion this implements: the project's calibrated 16-state
//! Sankoff cost matrix (`cardcomp::pairwise_cost_matrix`) treats every
@@ -10,7 +10,7 @@
//! reduction.
use std::io::{BufWriter, Write};
use std::path::Path;
use std::path::{Path, PathBuf};
use std::sync::Arc;
use obikpartitionner::KmerPartition;
@@ -20,7 +20,9 @@ use obikindex::{OKIError, OKIResult};
use obikindex::KmerIndex;
use super::cache::PartitionCache;
use super::entropy_annex::{EntropyAnnexBuilder, ENTROPY_ANNEX_FILE_NAME};
use super::family_scan::{Selection, scan_layer_families};
use super::subsample::EntropyBias;
/// Shannon entropy (bits) of one family's per-genome states, over the 15
/// non-empty subsets of `{A,C,G,T}` actually observed among the genomes
@@ -56,7 +58,7 @@ pub(super) fn family_entropy(genome_mask: &[u8]) -> Option<(f64, usize)> {
/// 4 plain nucleotide symbols instead of the 15-state space above — kept
/// alongside it (not instead of it) purely to measure how much the two
/// diverge on real data, per the discussion in
/// `docmd/architecture/siblings.md`, "`--subsample`/`--shannon`". A genome
/// `DevDocMD/architecture/siblings.md`, "`--subsample`/`--shannon`". A genome
/// carrying several bases at once (`genome_mask[g]` with more than one bit
/// set) contributes to *each* of those bases' counts — counted once per
/// base present, not split fractionally, and not folded into a combined
@@ -107,17 +109,22 @@ pub trait ShannonEntropyExt {
/// iteration-order index within that layer, the same numbering
/// `--subsample`'s selection is drawn from.
///
/// `subsample`: same meaning as
/// `subsample`/`entropy_bias`: same meaning as
/// [`SnpAlignmentExt::snp_pseudo_alignment`](super::SnpAlignmentExt::snp_pseudo_alignment)'s
/// — `None` streams every non-monomorphic minorant of the whole index
/// (a time cost, not a memory one: entropy is computed and written
/// per-family as soon as it resolves), `Some(n)` bounds the index's
/// cross-partition resolution work to ~`n` sampled families.
fn shannon_entropy_csv(&self, path: &Path, subsample: Option<usize>) -> OKIResult<()>;
/// — `subsample = None`, `entropy_bias = None` streams every
/// non-monomorphic minorant of the whole index (a time cost, not a
/// memory one: entropy is computed and written per-family as soon as it
/// resolves); `subsample = Some(n)` bounds the index's cross-partition
/// resolution work to ~`n` sampled families; `entropy_bias = Some(_)`
/// weights that sample by its own entropy (see
/// `DevDocMD/architecture/siblings.md`, "Entropy-biased selection") —
/// useful here specifically to see the biased sample's own entropy
/// distribution, not just to speed up a distance computation.
fn shannon_entropy_csv(&self, path: &Path, subsample: Option<usize>, entropy_bias: Option<EntropyBias>) -> OKIResult<()>;
}
impl ShannonEntropyExt for KmerIndex {
fn shannon_entropy_csv(&self, path: &Path, subsample: Option<usize>) -> OKIResult<()> {
fn shannon_entropy_csv(&self, path: &Path, subsample: Option<usize>, entropy_bias: Option<EntropyBias>) -> OKIResult<()> {
let n_parts = self.n_partitions();
let n_genomes = self.meta().genomes.len();
let with_counts = self.meta().config.with_counts;
@@ -133,7 +140,7 @@ impl ShannonEntropyExt for KmerIndex {
.map_err(OKIError::Partition)?;
let cache = Arc::new(PartitionCache::build(&partition, n_parts, with_counts)?);
let layer_dirs = super::family_scan::sibling_layer_dirs(self)?;
let selections = super::subsample::compute_selections(&layer_dirs, subsample)?;
let selections = super::subsample::compute_selections(self, &layer_dirs, subsample, entropy_bias)?;
let mut f = BufWriter::new(std::fs::File::create(path).map_err(OKIError::Io)?);
writeln!(f, "layer,family_idx,entropy15,entropy4,family_size,n_genomes_present").map_err(OKIError::Io)?;
@@ -168,3 +175,58 @@ impl ShannonEntropyExt for KmerIndex {
Ok(())
}
}
/// Builds the entropy annex (`entropy_annex.rs`) for every layer in
/// `layer_dirs` that doesn't already have one — a no-op, cheap existence
/// check per layer, once every layer's annex file exists. Called from
/// [`super::subsample::compute_selections`] whenever entropy-biased
/// selection is requested; see `DevDocMD/architecture/siblings.md`,
/// "Entropy-biased selection", for why this is the one place that pays a
/// full, unsampled scan (every non-monomorphic minorant's `genome_mask`
/// must be resolved to know its entropy) — every later read of the
/// resulting file is a plain positional mmap lookup, not a re-scan.
pub(super) fn ensure_entropy_annexes(index: &KmerIndex, layer_dirs: &[PathBuf]) -> OKIResult<()> {
let missing: Vec<usize> = layer_dirs.iter().enumerate()
.filter(|(_, dir)| !dir.join(ENTROPY_ANNEX_FILE_NAME).exists())
.map(|(i, _)| i)
.collect();
if missing.is_empty() {
return Ok(());
}
let n_parts = index.n_partitions();
let n_genomes = index.meta().genomes.len();
let with_counts = index.meta().config.with_counts;
let k = index.kmer_size();
let n_bits = n_parts.trailing_zeros() as usize;
let partition = KmerPartition::open_with_config(
index.root_path(),
index.kmer_size(),
index.minimizer_size(),
n_bits,
)
.map_err(OKIError::Partition)?;
let cache = Arc::new(PartitionCache::build(&partition, n_parts, with_counts)?);
let minorant_counts = super::subsample::minorant_counts(layer_dirs)?;
let pb = progress_bar("entropy_annex_build", missing.len() as u64, "layers");
for &li in &missing {
let layer_dir = &layer_dirs[li];
let mut builder = EntropyAnnexBuilder::new(minorant_counts[li] as usize, &layer_dir.join(ENTROPY_ANNEX_FILE_NAME))
.map_err(OKIError::Io)?;
scan_layer_families(layer_dir, n_parts, n_genomes, with_counts, k, &cache, &Selection::All, |family_idx, mask, genome_mask| {
if mask.family_size() < 2 {
return; // monomorphic minorant — leave the sentinel, never sampled
}
if let Some((h15, _)) = family_entropy(genome_mask) {
builder.set(family_idx, h15 as f32);
}
})?;
builder.close().map_err(OKIError::Io)?;
pb.inc(1);
}
pb.finish_and_clear();
Ok(())
}
+162
View File
@@ -0,0 +1,162 @@
//! Persisted, per-layer Shannon entropy (`entropy15`, see `entropy.rs`),
//! one value per **minorant family**, indexed by `family_idx` — the same
//! numbering `scan_layer_families`/`Selection`/`subsample::reservoir_sample_layer`
//! already use (every minorant of the layer, monomorphic ones included,
//! counted in iteration order; see `subsample.rs`'s module docs for why
//! `family_idx` and the raw annex slot are kept as two different counters).
//!
//! Exists so entropy-biased selection (`--entropy`/`--entropy-sd`) can
//! weight candidates without re-resolving each one's `genome_mask` (the
//! expensive cross-partition step) on every run — see
//! `DevDocMD/architecture/siblings.md`, "Entropy-biased selection": entropy is
//! only knowable once presence is resolved, so *building* this file still
//! costs a full, unsampled scan the first time (`ensure_entropy_annexes` in
//! `entropy.rs`), but every subsequent read is a plain positional mmap
//! lookup, restoring `Selection::Some`'s usual "skip resolving excluded
//! families entirely" speedup for every run after the first.
//!
//! Mirrors `SiblingAnnex`'s on-disk convention (magic + header + flat
//! per-entry data, read-only after build) but is not slot-indexed the way
//! `SiblingAnnex` is: `SiblingAnnex` is indexed by every k-mer of the layer
//! (needed because any slot might turn out to be a minorant), while this
//! file is built *from* an already-open `SiblingAnnex`, so it only ever
//! needs to be as long as the layer's minorant count — monomorphic
//! minorants still get an entry (a `-1.0` sentinel, entropy not meaningful
//! for them, never read), keeping the same family_idx numbering every
//! other consumer uses, but non-minorant k-mers (the vast majority of
//! slots) are never represented here at all.
use std::fs::{File, OpenOptions};
use std::io;
use std::path::Path;
use memmap2::{Mmap, MmapMut};
const MAGIC: [u8; 4] = *b"PENT";
// Header: magic(4) + _pad(4) + n(8) = 16 bytes. Data (4 bytes/entry) follows.
const HEADER_SIZE: usize = 16;
/// Sentinel: monomorphic minorant (no signal, never sampled) or otherwise
/// not applicable. Entropy is always `>= 0.0`, so any negative value is
/// unambiguous — chosen over `NaN` to avoid `NaN`-comparison footguns at
/// every read site.
const SENTINEL: f32 = -1.0;
pub(super) const ENTROPY_ANNEX_FILE_NAME: &str = "entropy.pent";
pub(super) struct EntropyAnnex {
mmap: Mmap,
n: usize,
}
impl EntropyAnnex {
pub(super) fn open(path: &Path) -> io::Result<Self> {
let mmap = unsafe { Mmap::map(&File::open(path)?)? };
if mmap.len() < HEADER_SIZE {
return Err(io::Error::new(io::ErrorKind::InvalidData, "PENT file too short"));
}
if mmap[0..4] != MAGIC {
return Err(io::Error::new(io::ErrorKind::InvalidData, "bad PENT magic"));
}
let n = u64::from_le_bytes(mmap[8..16].try_into().unwrap()) as usize;
if mmap.len() < HEADER_SIZE + n * 4 {
return Err(io::Error::new(io::ErrorKind::InvalidData, "PENT file truncated"));
}
Ok(Self { mmap, n })
}
pub(super) fn len(&self) -> usize { self.n }
/// `None` for the sentinel (monomorphic minorant, or not yet computed).
pub(super) fn get(&self, family_idx: usize) -> Option<f32> {
let off = HEADER_SIZE + family_idx * 4;
let v = f32::from_le_bytes(self.mmap[off..off + 4].try_into().unwrap());
(v >= 0.0).then_some(v)
}
}
pub(super) struct EntropyAnnexBuilder {
mmap: MmapMut,
}
impl EntropyAnnexBuilder {
/// Create a new annex of `n` entries (a layer's total minorant count) at
/// `path`, pre-initialised to the sentinel — entries never written
/// (monomorphic minorants) stay the sentinel, not an accidental `0.0`
/// (a real, achievable entropy value).
pub(super) fn new(n: usize, path: &Path) -> io::Result<Self> {
let file_size = HEADER_SIZE + n * 4;
let file = OpenOptions::new()
.read(true).write(true).create(true).truncate(true)
.open(path)?;
file.set_len(file_size as u64)?;
let mut mmap = unsafe { MmapMut::map_mut(&file)? };
mmap[0..4].copy_from_slice(&MAGIC);
mmap[4..8].copy_from_slice(&[0u8; 4]);
mmap[8..16].copy_from_slice(&(n as u64).to_le_bytes());
for i in 0..n {
let off = HEADER_SIZE + i * 4;
mmap[off..off + 4].copy_from_slice(&SENTINEL.to_le_bytes());
}
Ok(Self { mmap })
}
pub(super) fn set(&mut self, family_idx: usize, value: f32) {
debug_assert!(value >= 0.0, "entropy must be non-negative, got {value}");
let off = HEADER_SIZE + family_idx * 4;
self.mmap[off..off + 4].copy_from_slice(&value.to_le_bytes());
}
pub(super) fn close(self) -> io::Result<()> {
self.mmap.flush()
}
}
#[cfg(test)]
mod tests {
use super::*;
use tempfile::tempdir;
#[test]
fn sentinel_is_negative_and_unset_entries_read_as_none() {
let dir = tempdir().unwrap();
let path = dir.path().join("test.pent");
let builder = EntropyAnnexBuilder::new(4, &path).unwrap();
builder.close().unwrap();
let annex = EntropyAnnex::open(&path).unwrap();
for i in 0..4 {
assert_eq!(annex.get(i), None);
}
}
#[test]
fn roundtrip_including_zero_entropy() {
let dir = tempdir().unwrap();
let path = dir.path().join("test.pent");
let mut builder = EntropyAnnexBuilder::new(4, &path).unwrap();
let values = [0.0f32, 1.5, 3.9, 0.0001];
for (i, &v) in values.iter().enumerate() {
builder.set(i, v);
}
builder.close().unwrap();
let annex = EntropyAnnex::open(&path).unwrap();
for (i, &v) in values.iter().enumerate() {
assert_eq!(annex.get(i), Some(v), "entry {i}");
}
}
#[test]
fn mixed_sentinel_and_real_values() {
let dir = tempdir().unwrap();
let path = dir.path().join("test.pent");
let mut builder = EntropyAnnexBuilder::new(3, &path).unwrap();
builder.set(1, 2.0);
builder.close().unwrap();
let annex = EntropyAnnex::open(&path).unwrap();
assert_eq!(annex.get(0), None);
assert_eq!(annex.get(1), Some(2.0));
assert_eq!(annex.get(2), None);
assert_eq!(annex.len(), 3);
}
}
+3 -3
View File
@@ -76,7 +76,7 @@ const FAMILY_BATCH: usize = 65536;
/// Restricts [`scan_layer_families`] to a subset of a layer's families,
/// identified by their iteration-order index — see
/// `docmd/architecture/siblings.md`, "`--subsample`/`--shannon`", for why:
/// `DevDocMD/architecture/siblings.md`, "`--subsample`/`--shannon`", for why:
/// bounding which families actually pay the expensive cross-partition
/// resolution cost is what makes a sampled run cheap on an index far larger
/// than the sample itself. `All` (the default for every pre-existing
@@ -145,7 +145,7 @@ struct SourceBatch {
start_family_idx: usize,
/// One entry per minorant family in this batch, straight from
/// `iter_minorants_batch` — `order` (iteration-order index, not an MPHF
/// slot; see `docmd/architecture/siblings.md`), `kmer`, and `mask`
/// slot; see `DevDocMD/architecture/siblings.md`), `kmer`, and `mask`
/// already carried through, no second annex read.
entries: Vec<SiblingEntry>,
_permit: ThrottleGuard,
@@ -207,7 +207,7 @@ pub(super) fn scan_layer_families(
// Streamed straight from `iter_minorants_batch` (zips this layer's own
// `iter_kmers()` with the annex, both in iteration order — never an
// MPHF slot; see `docmd/architecture/siblings.md`) — never collected
// MPHF slot; see `DevDocMD/architecture/siblings.md`) — never collected
// into a `Vec` first: a layer can hold billions of k-mers, so
// materialising every minorant family up front is exactly the memory
// blowup an earlier version of this traversal was rewritten to avoid
+1 -1
View File
@@ -1,7 +1,7 @@
//! Phylo/sibling-domain iteration over a layer — an extension trait, not a
//! new field on `MphfLayer`/`Layer<D>`: "family"/"minorant" are phylo
//! concepts, `obilayeredmap` stays kmer/slot-mapping only (see
//! `docmd/architecture/siblings.md`).
//! `DevDocMD/architecture/siblings.md`).
//!
//! The sibling annex is persisted in the same order as `iter_kmers()`
//! (`build_sibling_annex`, see `build.rs`), so pairing them is a plain zip —
+6 -2
View File
@@ -1,6 +1,6 @@
//! Family presence-mask annex construction.
//!
//! See `docmd/theory/evolutionary_distances.md`, "Definitions: family, and
//! See `DevDocMD/theory/evolutionary_distances.md`, "Definitions: family, and
//! the canonical form of a family" and "Step 2b", for the full design
//! discussion this implements.
//!
@@ -32,7 +32,7 @@
//! most wall-clock time going into per-message channel send/notify
//! syscalls rather than the lookup itself, because a single k-mer's ≤3
//! variants is far too fine a granularity to amortise a pipeline's
//! synchronisation cost over. See `docmd/theory/evolutionary_distances.md`,
//! synchronisation cost over. See `DevDocMD/theory/evolutionary_distances.md`,
//! Step 2b, "Mechanism".
//!
//! Submodules, in the order data flows through them: [`cache`] (shared
@@ -48,9 +48,11 @@ mod cache;
mod cardinality;
mod distance;
mod entropy;
mod entropy_annex;
mod family_scan;
mod helpers;
mod iter;
mod sankoff_bundle;
mod siblingannex;
mod stats;
mod subsample;
@@ -64,8 +66,10 @@ pub use cardinality::{CardinalityExt, CardinalityTally};
pub use distance::{BasePairTally, DistanceExt, RawSnpDistanceOutput};
pub use entropy::ShannonEntropyExt;
pub use iter::{MinorantBatchIter, MinorantIter, SiblingBatchIter, SiblingEntry, SiblingIter, SiblingLayerExt};
pub use sankoff_bundle::{SankoffBundle, SankoffBundleExt};
pub(crate) use siblingannex::{FamilyMask, SiblingAnnex, SiblingAnnexBuilder};
pub use stats::{SiblingAnnexStats, SiblingStatsExt};
pub use subsample::EntropyBias;
use obilayeredmap::OLMError;
@@ -0,0 +1,237 @@
//! Fused entry point for the `--sankoff`/`--tnt`/`--phyg`/`--iqtree`
//! pipeline — see `DevDocMD/architecture/siblings.md`, "`--free-loss`/`--tnt`
//! pipeline: four independent scans, three of them unsampled" and "Wired
//! into `pack` and the sibling-annex build path" for the problem this
//! replaces: `raw_snp_distance`, `base_pair_tally`, `cardinality_tally` and
//! `snp_pseudo_alignment` used to each open their own `PartitionCache` and
//! independently re-scan the whole annex, three of the four ignoring
//! `--subsample` entirely.
//!
//! `base_pair_tally`/`cardinality_tally` both need `raw_snp_distance`'s
//! *complete* aggregate SNP/shared counts before they can derive
//! `included[i,j]` (the `ratio_ceiling` filter) — a genuine sequential
//! dependency, not an accidental one, so this is two passes, not one:
//!
//! - **Pass A**: aggregate SNP/shared counts per genome pair, over the
//! shared selection (computed once, [`super::subsample::compute_selections`]
//! — the *same* selection every other pass below uses, per the user's
//! own requirement: "il faut qu'on sélectionne tout le monde avec la
//! même sélection"). Excluded genomes (`--exclude-genome`) are zeroed
//! here, before `included` is derived from it — the ratio filter must
//! see the post-exclusion counts, not the raw ones.
//! - **Pass B**: base-pair tally, cardinality tally, and the pseudo-
//! alignment, fused into *one* scan over the same selection — the three
//! are mutually independent once `included` is known, so they are
//! accumulated together family-by-family instead of three more separate
//! scans.
use std::sync::Arc;
use ndarray::Array2;
use obikpartitionner::KmerPartition;
use obisys::progress_bar;
use obikindex::{OKIError, OKIResult};
use obikindex::KmerIndex;
use super::alignment::{iupac_code, SnpAlignment};
use super::cache::PartitionCache;
use super::cardinality::CardinalityTally;
use super::distance::{BasePairTally, RawSnpDistanceOutput};
use super::family_scan::{scan_layer_families, sibling_layer_dirs, Selection};
use super::subsample::{compute_selections, EntropyBias};
/// Every output the `--sankoff`/`--tnt`/`--phyg`/`--iqtree` pipeline needs,
/// computed together from one shared, possibly-subsampled/entropy-biased
/// selection — see the module docs.
pub struct SankoffBundle {
pub raw: RawSnpDistanceOutput,
pub base_pair_tally: BasePairTally,
pub cardinality_tally: CardinalityTally,
pub alignment: SnpAlignment,
}
/// Adds [`sankoff_bundle`](Self::sankoff_bundle) to `KmerIndex`.
pub trait SankoffBundleExt {
/// `subsample`/`entropy_bias`: see
/// [`SnpAlignmentExt::snp_pseudo_alignment`](super::SnpAlignmentExt::snp_pseudo_alignment).
/// `ratio_ceiling`: see
/// [`DistanceExt::base_pair_tally`](super::DistanceExt::base_pair_tally).
/// `exclude_mask[g]`: zero genome `g`'s row/column out of the aggregate
/// SNP/shared counts *before* `included[i,j]` is derived from them —
/// same semantics as `--exclude-genome` had on the old, separate
/// `raw_snp_distance` call.
fn sankoff_bundle(
&self,
subsample: Option<usize>,
entropy_bias: Option<EntropyBias>,
ratio_ceiling: f64,
exclude_mask: &[bool],
) -> OKIResult<SankoffBundle>;
}
impl SankoffBundleExt for KmerIndex {
fn sankoff_bundle(
&self,
subsample: Option<usize>,
entropy_bias: Option<EntropyBias>,
ratio_ceiling: f64,
exclude_mask: &[bool],
) -> OKIResult<SankoffBundle> {
let n_parts = self.n_partitions();
let n_genomes = self.meta().genomes.len();
let with_counts = self.meta().config.with_counts;
let k = self.kmer_size();
let n_bits = n_parts.trailing_zeros() as usize;
let partition = KmerPartition::open_with_config(
self.root_path(),
self.kmer_size(),
self.minimizer_size(),
n_bits,
)
.map_err(OKIError::Partition)?;
let cache = Arc::new(PartitionCache::build(&partition, n_parts, with_counts)?);
let layer_dirs = sibling_layer_dirs(self)?;
// Computed once — every pass below iterates the exact same
// families, in the exact same layers, per the shared-selection
// requirement this module exists to satisfy.
let selections = compute_selections(self, &layer_dirs, subsample, entropy_bias)?;
// ── Pass A: aggregate SNP/shared counts ────────────────────────────
let pb = progress_bar("raw_snp_distance", layer_dirs.len() as u64, "layers");
let mut snp = Array2::<u64>::zeros((n_genomes, n_genomes));
let mut shared = Array2::<u64>::zeros((n_genomes, n_genomes));
for (layer_dir, layer_selection) in layer_dirs.iter().zip(selections.iter()) {
let selection = match layer_selection {
None => Selection::All,
Some(set) => Selection::Some(set),
};
scan_layer_families(layer_dir, n_parts, n_genomes, with_counts, k, &cache, &selection, |_family_idx, _mask, genome_mask| {
let single_form = |g: usize| -> Option<u8> {
let m = genome_mask[g];
(m.count_ones() == 1).then(|| m.trailing_zeros() as u8)
};
for i in 0..n_genomes {
let Some(bi) = single_form(i) else { continue };
for j in (i + 1)..n_genomes {
let Some(bj) = single_form(j) else { continue };
if bi == bj {
shared[[i, j]] += 1;
shared[[j, i]] += 1;
} else {
snp[[i, j]] += 1;
snp[[j, i]] += 1;
}
}
}
})?;
pb.inc(1);
}
pb.finish_and_clear();
// `--exclude-genome`: zero excluded rows/columns before `included`
// is derived below — same effect the old standalone
// `zero_excluded_pairs` had, applied here since this is now the
// only place `included` gets computed for this pipeline.
for (i, &excluded) in exclude_mask.iter().enumerate() {
if !excluded {
continue;
}
for j in 0..n_genomes {
snp[[i, j]] = 0;
snp[[j, i]] = 0;
shared[[i, j]] = 0;
shared[[j, i]] = 0;
}
}
let raw = RawSnpDistanceOutput { snp, shared };
let included = Array2::from_shape_fn((n_genomes, n_genomes), |(i, j)| {
if i == j {
return false;
}
let s = raw.snp[[i, j]];
let total = s + raw.shared[[i, j]];
total > 0 && (s as f64 / total as f64) <= ratio_ceiling
});
// ── Pass B: base-pair tally + cardinality tally + alignment, fused ──
let pb = progress_bar("sankoff_pass_b", layer_dirs.len() as u64, "layers");
let mut bp_counts = [[0u64; 4]; 4];
let mut bp_same = [0u64; 4];
let mut card_counts = [[0u64; 5]; 5];
let mut sequences: Vec<Vec<u8>> = vec![Vec::new(); n_genomes];
for (layer_dir, layer_selection) in layer_dirs.iter().zip(selections.iter()) {
let selection = match layer_selection {
None => Selection::All,
Some(set) => Selection::Some(set),
};
scan_layer_families(layer_dir, n_parts, n_genomes, with_counts, k, &cache, &selection, |_family_idx, mask, genome_mask| {
let variable = mask.family_size() >= 2;
// base-pair tally — same pairwise single-form resolution as pass A.
let single_form = |g: usize| -> Option<u8> {
let m = genome_mask[g];
(m.count_ones() == 1).then(|| m.trailing_zeros() as u8)
};
for i in 0..n_genomes {
let Some(bi) = single_form(i) else { continue };
for j in (i + 1)..n_genomes {
let Some(bj) = single_form(j) else { continue };
if !included[[i, j]] {
continue;
}
if bi != bj {
bp_counts[bi as usize][bj as usize] += 1;
bp_counts[bj as usize][bi as usize] += 1;
} else if variable {
bp_same[bi as usize] += 1;
}
}
}
// cardinality tally — restricted to variable families, see
// `CardinalityExt::cardinality_tally`'s docs for why.
if variable {
for i in 0..n_genomes {
let card_i = genome_mask[i].count_ones() as usize;
for j in (i + 1)..n_genomes {
if !included[[i, j]] {
continue;
}
let card_j = genome_mask[j].count_ones() as usize;
card_counts[card_i][card_j] += 1;
if card_i != card_j {
card_counts[card_j][card_i] += 1;
}
}
}
}
// pseudo-alignment — same variable-family gate as
// `snp_pseudo_alignment`. Every family the shared selection
// actually visits already satisfies this when the
// selection is `Selection::Some` (only non-monomorphic
// minorants are ever selected), but the explicit check
// still matters under `Selection::All` (no `--subsample`),
// which visits monomorphic minorants too.
if variable {
for (g, &m) in genome_mask.iter().enumerate() {
sequences[g].push(iupac_code(m));
}
}
})?;
pb.inc(1);
}
pb.finish_and_clear();
Ok(SankoffBundle {
raw,
base_pair_tally: BasePairTally { counts: bp_counts, same: bp_same },
cardinality_tally: CardinalityTally { counts: card_counts },
alignment: SnpAlignment { sequences },
})
}
}
+1 -1
View File
@@ -1,6 +1,6 @@
//! Family presence-mask annex: a compact, read-only-after-build, per-slot
//! derived value used by the central-position SNP distance estimator (see
//! `docmd/theory/evolutionary_distances.md`, "Step 2b" and "Definitions:
//! `DevDocMD/theory/evolutionary_distances.md`, "Step 2b" and "Definitions:
//! family, and the canonical form of a family").
//!
//! Two bytes are stored per MPHF slot of a partition/layer, packed as four
+124 -8
View File
@@ -1,5 +1,5 @@
//! Proportional per-layer sampling of non-monomorphic minorants
//! (`family_size() >= 2`) — see `docmd/architecture/siblings.md`,
//! (`family_size() >= 2`) — see `DevDocMD/architecture/siblings.md`,
//! "`--subsample`/`--shannon`", for the full design discussion.
//!
//! Three steps, in order, each cheaper than the one before because it
@@ -25,10 +25,25 @@ use std::path::{Path, PathBuf};
use rand::Rng;
use rayon::prelude::*;
use obikindex::OKIResult;
use obikindex::{KmerIndex, OKIError, OKIResult};
use super::ANNEX_FILE_NAME;
use super::SiblingAnnex;
use super::entropy_annex::{EntropyAnnex, ENTROPY_ANNEX_FILE_NAME};
/// Gaussian-kernel entropy bias parameters — see
/// `DevDocMD/architecture/siblings.md`, "Entropy-biased selection". `mu`/
/// `sigma` are the `--entropy`/`--entropy-sd` CLI values (defaulted by the
/// caller, not here — this type carries only the resolved numbers). `pub`
/// (not `pub(super)`): part of the public signature of
/// `SnpAlignmentExt::snp_pseudo_alignment`,
/// `ShannonEntropyExt::shannon_entropy_csv`, and
/// `SankoffBundleExt::sankoff_bundle`, all consumed from `obikmer`.
#[derive(Clone, Copy)]
pub struct EntropyBias {
pub mu: f64,
pub sigma: f64,
}
/// A family is non-monomorphic (informative-eligible) iff at least one
/// sibling is registered alongside it — `family_size() >= 2`, i.e.
@@ -106,6 +121,70 @@ fn reservoir_sample_layer(layer_dir: &Path, n_layer: usize) -> OKIResult<HashSet
Ok(reservoir.into_iter().collect())
}
/// Total minorant count per layer (`is_minorant()`, monomorphic included)
/// — sizes the entropy annex (`entropy_annex.rs`), which needs exactly one
/// entry per minorant to keep its `family_idx` numbering aligned with
/// every other consumer (`scan_layer_families`, `Selection`,
/// `reservoir_sample_layer`'s own `family_idx` counter above).
pub(super) fn minorant_counts(layer_dirs: &[PathBuf]) -> OKIResult<Vec<u64>> {
layer_dirs
.par_iter()
.map(|layer_dir| -> OKIResult<u64> {
let annex = SiblingAnnex::open(&layer_dir.join(ANNEX_FILE_NAME))?;
let mut count = 0u64;
for slot in 0..annex.len() {
if annex.get(slot).is_some_and(|m| m.is_minorant()) {
count += 1;
}
}
Ok(count)
})
.collect()
}
/// Entropy-biased alternative to [`reservoir_sample_layer`]: an
/// independent accept/reject draw per non-monomorphic minorant, probability
/// `p0 * w(entropy)` where `w(entropy) = exp(-(entropy-mu)^2 / (2*sigma^2))`
/// (the unnormalised Gaussian kernel — `1` exactly at `entropy == mu`,
/// smoothly decaying away from it, no hard cutoff) and `p0` is the same
/// uniform rate the plain reservoir targets in expectation (`n / total_count`
/// index-wide, or `1.0` when there is no size target at all — `--entropy`
/// given without `--subsample`, a pure soft entropy filter over the whole
/// index). See `DevDocMD/architecture/siblings.md`, "Entropy-biased
/// selection", for why this is a single streaming pass with no
/// cross-partition resolution: entropy is read from the already-built
/// entropy annex (`ensure_entropy_annexes`), not recomputed here.
fn entropy_biased_sample_layer(layer_dir: &Path, p0: f64, bias: EntropyBias) -> OKIResult<HashSet<usize>> {
let annex = SiblingAnnex::open(&layer_dir.join(ANNEX_FILE_NAME))?;
let entropy_annex = EntropyAnnex::open(&layer_dir.join(ENTROPY_ANNEX_FILE_NAME)).map_err(OKIError::Io)?;
let mut selected = HashSet::new();
let mut family_idx: usize = 0;
let mut rng = rand::thread_rng();
for slot in 0..annex.len() {
let Some(mask) = annex.get(slot) else { continue };
if !mask.is_minorant() {
continue; // not a minorant at all — doesn't advance `family_idx` either
}
let this_family_idx = family_idx;
family_idx += 1;
debug_assert!(
this_family_idx < entropy_annex.len(),
"entropy annex ({}) shorter than this layer's minorant count — built against a different annex?",
entropy_annex.len(),
);
if mask.siblings() < 1 {
continue; // monomorphic minorant — no entropy, never eligible
}
let Some(entropy) = entropy_annex.get(this_family_idx) else { continue };
let d = (entropy as f64 - bias.mu) / bias.sigma;
let w = (-0.5 * d * d).exp();
if rng.r#gen::<f64>() < p0 * w {
selected.insert(this_family_idx);
}
}
Ok(selected)
}
/// Per-layer sampling outcome — `None` at a given layer means "keep every
/// non-monomorphic minorant of this layer" (either the `total_count <= n`
/// case, or a layer whose entire non-monomorphic set was already smaller
@@ -113,17 +192,54 @@ fn reservoir_sample_layer(layer_dir: &Path, n_layer: usize) -> OKIResult<HashSet
pub(super) type LayerSelection = Option<HashSet<usize>>;
/// Full entry point for every caller (`snp_pseudo_alignment`, the Shannon
/// entropy report): `n = None` means "no `--subsample`", i.e. keep every
/// entropy report, `sankoff_bundle`): `n = None` and `entropy_bias = None`
/// means "no `--subsample`, no `--entropy`/`--entropy-sd`", i.e. keep every
/// non-monomorphic minorant of every layer — skips steps 1 and 2 entirely,
/// since there is nothing to proportion.
pub(super) fn compute_selections(layer_dirs: &[PathBuf], n: Option<usize>) -> OKIResult<Vec<LayerSelection>> {
match n {
None => Ok(vec![None; layer_dirs.len()]),
///
/// `entropy_bias = Some(_)` switches selection from the plain uniform
/// reservoir to [`entropy_biased_sample_layer`] — see
/// `DevDocMD/architecture/siblings.md`, "Entropy-biased selection". `index` is
/// only used on that path, to build any missing per-layer entropy annex
/// first ([`super::entropy::ensure_entropy_annexes`]) — a one-time cost,
/// not paid again once every layer's annex file exists. The target size
/// `n` still proportions the *expected* rate across the whole index (a
/// single `p0 = n / total_count` applied uniformly at every layer already
/// yields each layer its proportional share, with no need to compute a
/// per-layer target the way [`sample_layers`] does for the uniform
/// reservoir) — but unlike the uniform reservoir, the result is
/// approximate, not exact (see [`entropy_biased_sample_layer`]'s docs).
pub(super) fn compute_selections(
index: &KmerIndex,
layer_dirs: &[PathBuf],
n: Option<usize>,
entropy_bias: Option<EntropyBias>,
) -> OKIResult<Vec<LayerSelection>> {
let Some(bias) = entropy_bias else {
return match n {
None => Ok(vec![None; layer_dirs.len()]),
Some(n) => {
let counts = non_monomorphic_counts(layer_dirs)?;
sample_layers(layer_dirs, &counts, n)
}
};
};
super::entropy::ensure_entropy_annexes(index, layer_dirs)?;
let p0 = match n {
None => 1.0,
Some(n) => {
let counts = non_monomorphic_counts(layer_dirs)?;
sample_layers(layer_dirs, &counts, n)
let total_count: u64 = counts.iter().sum();
if total_count == 0 { 0.0 } else { (n as f64 / total_count as f64).min(1.0) }
}
}
};
layer_dirs
.par_iter()
.map(|layer_dir| -> OKIResult<LayerSelection> {
Ok(Some(entropy_biased_sample_layer(layer_dir, p0, bias)?))
})
.collect()
}
/// Step 2 entry point: `n` is the total number of families requested across
+99 -6
View File
@@ -14,8 +14,11 @@ use super::build::SiblingAnnexBuildExt;
use super::cardinality::CardinalityExt;
use super::distance::DistanceExt;
use super::entropy::ShannonEntropyExt;
use super::entropy_annex::{EntropyAnnex, ENTROPY_ANNEX_FILE_NAME};
use super::helpers::is_minorant;
use super::sankoff_bundle::SankoffBundleExt;
use super::stats::SiblingStatsExt;
use super::subsample::EntropyBias;
use super::{FamilyMask, SiblingAnnex, ANNEX_FILE_NAME, INDEX_SUBDIR};
// k must be >= 11 (project constraint, "k ∈ [11,31]"); k=11, level_max=1,
@@ -271,7 +274,7 @@ fn family_scan_consumers_agree_on_one_sibling_each() {
// snp_pseudo_alignment: one variable family, one column — g1's row
// reads 'C' (its own member), g2's reads 'G'.
let alignment = merged.snp_pseudo_alignment(None).expect("snp_pseudo_alignment");
let alignment = merged.snp_pseudo_alignment(None, None).expect("snp_pseudo_alignment");
assert_eq!(alignment.sequences[i1], vec![b'C']);
assert_eq!(alignment.sequences[i2], vec![b'G']);
@@ -389,13 +392,13 @@ fn subsample_and_shannon_on_one_variable_family() {
// `family_scan_consumers_agree_on_one_sibling_each`) — total_count = 1.
// Asking for far more than that must fall back to "keep everything",
// identical to the unsampled (`None`) result.
let sampled = merged.snp_pseudo_alignment(Some(1000)).expect("snp_pseudo_alignment (subsample >> total)");
let sampled = merged.snp_pseudo_alignment(Some(1000), None).expect("snp_pseudo_alignment (subsample >> total)");
assert_eq!(sampled.sequences[i1], vec![b'C']);
assert_eq!(sampled.sequences[i2], vec![b'G']);
// Asking for 0 must keep nothing — an empty (but still one-row-per-genome)
// alignment, not an error.
let empty = merged.snp_pseudo_alignment(Some(0)).expect("snp_pseudo_alignment (subsample 0)");
let empty = merged.snp_pseudo_alignment(Some(0), None).expect("snp_pseudo_alignment (subsample 0)");
assert!(empty.sequences[i1].is_empty());
assert!(empty.sequences[i2].is_empty());
@@ -404,7 +407,7 @@ fn subsample_and_shannon_on_one_variable_family() {
// uniform split -> -2*(0.5*log2(0.5)) = 1 — the two definitions
// coincide here since no genome carries more than one base).
let csv_path = dir.path().join("shannon.csv");
merged.shannon_entropy_csv(&csv_path, Some(1000)).expect("shannon_entropy_csv");
merged.shannon_entropy_csv(&csv_path, Some(1000), None).expect("shannon_entropy_csv");
let csv = std::fs::read_to_string(&csv_path).unwrap();
let mut lines = csv.lines();
assert_eq!(lines.next(), Some("layer,family_idx,entropy15,entropy4,family_size,n_genomes_present"));
@@ -419,6 +422,96 @@ fn subsample_and_shannon_on_one_variable_family() {
assert_eq!(fields[5], "2", "n_genomes_present");
}
#[test]
fn sankoff_bundle_matches_old_separate_calls() {
// Same fixture as `sibling_annex_one_sibling_each`/
// `subsample_and_shannon_on_one_variable_family`: one non-monomorphic
// family, g1='C' (minorant), g2='G'. Regression check that fusing
// raw_snp_distance/base_pair_tally/cardinality_tally/snp_pseudo_alignment
// into `sankoff_bundle`'s two passes produces identical results to the
// old four independent calls on `Selection::All` (no `--subsample`,
// no `--entropy`) — see `DevDocMD/architecture/siblings.md`, "Wired into
// `pack`...".
let dir = tempdir().unwrap();
let g1 = build_single_genome_index(dir.path(), "g1", b"AACCGCTTAAG");
let g2 = build_single_genome_index(dir.path(), "g2", b"AACCGGTTAAG");
let merged = merge_two(dir.path(), &g1, &g2);
merged.build_sibling_annex().expect("build_sibling_annex");
let n_genomes = merged.meta().genomes.len();
let exclude_mask = vec![false; n_genomes];
let ratio_ceiling = 0.5;
let expected_raw = merged.raw_snp_distance().expect("raw_snp_distance");
let expected_base = merged.base_pair_tally(&expected_raw, ratio_ceiling).expect("base_pair_tally");
let expected_card = merged.cardinality_tally(&expected_raw, ratio_ceiling).expect("cardinality_tally");
let expected_alignment = merged.snp_pseudo_alignment(None, None).expect("snp_pseudo_alignment");
let bundle = merged.sankoff_bundle(None, None, ratio_ceiling, &exclude_mask).expect("sankoff_bundle");
assert_eq!(bundle.raw.snp, expected_raw.snp);
assert_eq!(bundle.raw.shared, expected_raw.shared);
assert_eq!(bundle.base_pair_tally.counts, expected_base.counts);
assert_eq!(bundle.base_pair_tally.same, expected_base.same);
assert_eq!(bundle.cardinality_tally.counts, expected_card.counts);
assert_eq!(bundle.alignment.sequences, expected_alignment.sequences);
}
#[test]
fn entropy_annex_builds_on_demand_and_biases_selection() {
// Same fixture, same single non-monomorphic family, known entropy15 =
// 1.0 bit exactly (see `subsample_and_shannon_on_one_variable_family`).
let dir = tempdir().unwrap();
let g1 = build_single_genome_index(dir.path(), "g1", b"AACCGCTTAAG");
let g2 = build_single_genome_index(dir.path(), "g2", b"AACCGGTTAAG");
let merged = merge_two(dir.path(), &g1, &g2);
merged.build_sibling_annex().expect("build_sibling_annex");
let layer_dirs = super::family_scan::sibling_layer_dirs(&merged).expect("layer dirs");
for layer_dir in &layer_dirs {
assert!(
!layer_dir.join(ENTROPY_ANNEX_FILE_NAME).exists(),
"entropy annex must not exist before any entropy-biased selection is requested"
);
}
// This fixture spans 2 layers (a single-genome bootstrap layer plus the
// merged one, see `merge_two`) — the one non-monomorphic family lives
// in whichever layer actually got a non-empty selection, not
// necessarily `layer_dirs[0]`.
let find_layer = |selections: &[super::subsample::LayerSelection]| {
selections.iter().position(|s| matches!(s, Some(set) if !set.is_empty()))
};
// mu == the family's real entropy, tiny sigma -> weight == 1 exactly,
// p0 == 1.0 (no `--subsample`) -> the family is selected with
// probability 1 (up to a measure-zero event on the uniform draw).
let bias_center = EntropyBias { mu: 1.0, sigma: 0.5 };
let selections = super::subsample::compute_selections(&merged, &layer_dirs, None, Some(bias_center))
.expect("compute_selections (entropy-biased)");
for layer_dir in &layer_dirs {
assert!(
layer_dir.join(ENTROPY_ANNEX_FILE_NAME).exists(),
"compute_selections must build the entropy annex for every layer on first use"
);
}
let li = find_layer(&selections).expect("mu at the true entropy must select the one non-monomorphic family, in some layer");
assert_eq!(selections[li], Some(std::collections::HashSet::from([0])));
// The annex now exists — read it back directly and check the stored
// value matches `--shannon`'s own entropy15 computation exactly.
let entropy_annex = EntropyAnnex::open(&layer_dirs[li].join(ENTROPY_ANNEX_FILE_NAME)).expect("open entropy annex");
let stored = entropy_annex.get(0).expect("family 0 has a real (non-sentinel) entropy value");
assert!((stored as f64 - 1.0).abs() < 1e-6, "stored entropy should be 1 bit, got {stored}");
// mu far from any achievable entropy, tiny sigma -> weight underflows
// to exactly 0.0 -> never selected, deterministically, in any layer.
let bias_far = EntropyBias { mu: 100.0, sigma: 0.001 };
let selections_far = super::subsample::compute_selections(&merged, &layer_dirs, None, Some(bias_far))
.expect("compute_selections (entropy-biased, far)");
assert!(find_layer(&selections_far).is_none(), "mu far from the true entropy must never select it, in any layer");
}
#[test]
#[ignore]
fn diag_real_index_layer_distribution() {
@@ -433,7 +526,7 @@ fn diag_real_index_layer_distribution() {
println!("n_layers={} total={} n_zero_layers={} max={} min_nonzero={}",
counts.len(), total, n_zero, max, min_nonzero);
let n = 100_000usize;
let selections = super::subsample::compute_selections(&layer_dirs, Some(n)).expect("selections");
let selections = super::subsample::compute_selections(&idx, &layer_dirs, Some(n), None).expect("selections");
let selected_total: usize = selections.iter().map(|s| match s {
None => 0, // shouldn't happen when total_count > n
Some(set) => set.len(),
@@ -803,7 +896,7 @@ fn bench_real_persistent_sparse_bit_matrix_on_disk_size() {
);
// ── Column-major access, "just for fun" ─────────────────────────────
// The whole point of `docmd/architecture/siblings.md`'s "Explicitly
// The whole point of `DevDocMD/architecture/siblings.md`'s "Explicitly
// deferred" section: dense is genome-major (native, contiguous column
// access), sparse is k-mer-major (no column method at all — reading
// one column means decoding every row and keeping one bit each time).