Introduce concurrency-bounded parallel processing for index cache layers
Implement an `obipipeline::Throttle` with an RAII guard to acquire and release concurrency slots. Expose new bounded parallel methods on `IndexCache` to process cached layers with a configurable cap. Refactor downstream aggregation logic to use single-pass parallel map-reduce instead of manual collect-map-reduce sequences, enforcing a maximum of 8 concurrent layer scans to bound memory usage.
This commit is contained in:
1 parent
9dee6dcd08
commit
d084396aba
6 files changed
+211
-62
No files matched your search
+32
-22
@@ -2,13 +2,16 @@ use std::fs;
|
||||
use std::path::Path;
|
||||
use std::sync::Arc;
|
||||
|
||||
use rayon::prelude::*;
|
||||
|
||||
use obikalgorithm::Algorithm;
|
||||
use obikidxcache::index_cache::IndexCache;
|
||||
use obikindex::KmerIndex;
|
||||
use obikindex::layer::KmerLayer;
|
||||
|
||||
/// Layers scanned at once per `par_map_reduce` call — see
|
||||
/// `obikidxcache::IndexCache::par_map_reduce`'s own docs for why this is a
|
||||
/// separate knob from thread count.
|
||||
const MAX_CONCURRENT_LAYERS: usize = 8;
|
||||
|
||||
/// Bits per kmer broken down by index component.
|
||||
pub struct IndexBitsPerKmer {
|
||||
/// Total distinct k-mers across all partitions and layers.
|
||||
@@ -105,16 +108,16 @@ impl Algorithm for BitsPerKmer {
|
||||
fn run(&mut self) -> obikalgorithm::Result<IndexBitsPerKmer> {
|
||||
let n_genomes = self.index.meta().genomes()?.len().max(1);
|
||||
let cache = IndexCache::new(Arc::clone(&self.index), None);
|
||||
let layers: Vec<&KmerLayer> = cache.iter().collect();
|
||||
|
||||
let (n_kmers, mphf_b, evidence_b, matrix_b) = layers
|
||||
.par_iter()
|
||||
.map(|layer| layer_bytes(layer))
|
||||
.map(|lb| (lb.n_kmers, lb.mphf, lb.evidence, lb.matrix))
|
||||
.reduce(
|
||||
|| (0usize, 0u64, 0u64, 0u64),
|
||||
|a, b| (a.0 + b.0, a.1 + b.1, a.2 + b.2, a.3 + b.3),
|
||||
);
|
||||
let (n_kmers, mphf_b, evidence_b, matrix_b) = cache.par_map_reduce(
|
||||
MAX_CONCURRENT_LAYERS,
|
||||
|| (0usize, 0u64, 0u64, 0u64),
|
||||
|layer| {
|
||||
let lb = layer_bytes(layer);
|
||||
(lb.n_kmers, lb.mphf, lb.evidence, lb.matrix)
|
||||
},
|
||||
|a, b| (a.0 + b.0, a.1 + b.1, a.2 + b.2, a.3 + b.3),
|
||||
);
|
||||
|
||||
if n_kmers == 0 {
|
||||
return Ok(IndexBitsPerKmer {
|
||||
@@ -166,19 +169,26 @@ impl Algorithm for GenomeKmerCounts {
|
||||
fn run(&mut self) -> obikalgorithm::Result<(usize, Vec<u64>)> {
|
||||
let n_genomes = self.index.meta().genomes()?.len();
|
||||
let cache = IndexCache::new(Arc::clone(&self.index), None);
|
||||
let layers: Vec<&KmerLayer> = cache.iter().collect();
|
||||
|
||||
let total_kmers: usize = layers.iter().map(|l| l.n()).sum();
|
||||
|
||||
let mut total_counts = vec![0u64; n_genomes];
|
||||
let per_layer_weights: Vec<_> = layers.par_iter().map(|l| l.col_weights()).collect();
|
||||
for weights in per_layer_weights {
|
||||
for (g, &v) in weights.iter().enumerate() {
|
||||
if g < n_genomes {
|
||||
total_counts[g] += v;
|
||||
let (total_kmers, total_counts) = cache.par_map_reduce(
|
||||
MAX_CONCURRENT_LAYERS,
|
||||
|| (0usize, vec![0u64; n_genomes]),
|
||||
|layer| {
|
||||
let mut counts = vec![0u64; n_genomes];
|
||||
for (g, &w) in layer.col_weights().iter().enumerate() {
|
||||
if g < n_genomes {
|
||||
counts[g] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
(layer.n(), counts)
|
||||
},
|
||||
|(ka, mut ca), (kb, cb)| {
|
||||
for (x, y) in ca.iter_mut().zip(cb.iter()) {
|
||||
*x += y;
|
||||
}
|
||||
(ka + kb, ca)
|
||||
},
|
||||
);
|
||||
|
||||
Ok((total_kmers, total_counts))
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user