Introduce concurrency-bounded parallel processing for index cache layers

Implement an `obipipeline::Throttle` with an RAII guard to acquire and release concurrency slots. Expose new bounded parallel methods on `IndexCache` to process cached layers with a configurable cap. Refactor downstream aggregation logic to use single-pass parallel map-reduce instead of manual collect-map-reduce sequences, enforcing a maximum of 8 concurrent layer scans to bound memory usage.
This commit is contained in:
Eric Coissac committed 2026-08-28 20:20:12 +02:00
1 parent 9dee6dcd08
commit d084396aba
6 files changed
+211 -62

No files matched your search

+32 -22
View File
@@ -2,13 +2,16 @@ use std::fs;
use std::path::Path;
use std::sync::Arc;
use rayon::prelude::*;
use obikalgorithm::Algorithm;
use obikidxcache::index_cache::IndexCache;
use obikindex::KmerIndex;
use obikindex::layer::KmerLayer;
/// Layers scanned at once per `par_map_reduce` call — see
/// `obikidxcache::IndexCache::par_map_reduce`'s own docs for why this is a
/// separate knob from thread count.
const MAX_CONCURRENT_LAYERS: usize = 8;
/// Bits per kmer broken down by index component.
pub struct IndexBitsPerKmer {
/// Total distinct k-mers across all partitions and layers.
@@ -105,16 +108,16 @@ impl Algorithm for BitsPerKmer {
fn run(&mut self) -> obikalgorithm::Result<IndexBitsPerKmer> {
let n_genomes = self.index.meta().genomes()?.len().max(1);
let cache = IndexCache::new(Arc::clone(&self.index), None);
let layers: Vec<&KmerLayer> = cache.iter().collect();
let (n_kmers, mphf_b, evidence_b, matrix_b) = layers
.par_iter()
.map(|layer| layer_bytes(layer))
.map(|lb| (lb.n_kmers, lb.mphf, lb.evidence, lb.matrix))
.reduce(
|| (0usize, 0u64, 0u64, 0u64),
|a, b| (a.0 + b.0, a.1 + b.1, a.2 + b.2, a.3 + b.3),
);
let (n_kmers, mphf_b, evidence_b, matrix_b) = cache.par_map_reduce(
MAX_CONCURRENT_LAYERS,
|| (0usize, 0u64, 0u64, 0u64),
|layer| {
let lb = layer_bytes(layer);
(lb.n_kmers, lb.mphf, lb.evidence, lb.matrix)
},
|a, b| (a.0 + b.0, a.1 + b.1, a.2 + b.2, a.3 + b.3),
);
if n_kmers == 0 {
return Ok(IndexBitsPerKmer {
@@ -166,19 +169,26 @@ impl Algorithm for GenomeKmerCounts {
fn run(&mut self) -> obikalgorithm::Result<(usize, Vec<u64>)> {
let n_genomes = self.index.meta().genomes()?.len();
let cache = IndexCache::new(Arc::clone(&self.index), None);
let layers: Vec<&KmerLayer> = cache.iter().collect();
let total_kmers: usize = layers.iter().map(|l| l.n()).sum();
let mut total_counts = vec![0u64; n_genomes];
let per_layer_weights: Vec<_> = layers.par_iter().map(|l| l.col_weights()).collect();
for weights in per_layer_weights {
for (g, &v) in weights.iter().enumerate() {
if g < n_genomes {
total_counts[g] += v;
let (total_kmers, total_counts) = cache.par_map_reduce(
MAX_CONCURRENT_LAYERS,
|| (0usize, vec![0u64; n_genomes]),
|layer| {
let mut counts = vec![0u64; n_genomes];
for (g, &w) in layer.col_weights().iter().enumerate() {
if g < n_genomes {
counts[g] = w;
}
}
}
}
(layer.n(), counts)
},
|(ka, mut ca), (kb, cb)| {
for (x, y) in ca.iter_mut().zip(cb.iter()) {
*x += y;
}
(ka + kb, ca)
},
);
Ok((total_kmers, total_counts))
}