Add directory-level locking and introduce layered index cache

Introduces directory-level locking to prevent concurrent index operations from corrupting shared directories, along with explicit APIs for acquiring, probing, and releasing locks. Restructures the index cache crate to use a layered store architecture that eagerly initializes metadata and provides fast hierarchical lookups. Updates dependent modules, test suites, and CLI commands to align with the refactored API surface, and adds an end-to-end smoke test for validation.
This commit is contained in:
Eric Coissac committed 2026-08-22 14:04:46 +02:00
1 parent 2419a6c21d
commit 7183e3adb4
15 files changed
+509 -77

No files matched your search

+80 -42
View File
@@ -23,13 +23,13 @@ use super::{DEFAULT_BLOCK_BITS, MAGIC, idx_path};
/// both modes. Without `.idx` they fall back to an O(i) sequential scan —
/// correct but slower.
pub struct UnitigFileReader {
mmap: Mmap,
mmap: Mmap,
block_offsets: Vec<u32>,
n_unitigs: usize,
n_kmers: usize,
k: usize,
block_bits: u8,
mask: usize, // (1 << block_bits) - 1
n_unitigs: usize,
n_kmers: usize,
k: usize,
block_bits: u8,
mask: usize, // (1 << block_bits) - 1
}
impl UnitigFileReader {
@@ -52,13 +52,13 @@ impl UnitigFileReader {
let mmap = unsafe { Mmap::map(&file).map_err(SKError::Io)? };
let k = obikseq::params::k();
let mut offset = 0usize;
let mut offset = 0usize;
let mut n_unitigs = 0usize;
let mut n_kmers = 0usize;
let mut n_kmers = 0usize;
while offset < mmap.len() {
let seql_minus_k = mmap[offset] as usize;
n_kmers += seql_minus_k + 1;
offset += 1 + (seql_minus_k + k + 3) / 4;
n_kmers += seql_minus_k + 1;
offset += 1 + (seql_minus_k + k + 3) / 4;
n_unitigs += 1;
}
@@ -94,11 +94,22 @@ impl UnitigFileReader {
})
}
pub fn len(&self) -> usize { self.n_unitigs }
pub fn is_empty(&self) -> bool { self.n_unitigs == 0 }
pub fn n_kmers(&self) -> usize { self.n_kmers }
pub fn block_bits(&self) -> u8 { self.block_bits }
pub fn has_direct_access(&self) -> bool { !self.block_offsets.is_empty() }
pub fn len(&self) -> usize {
self.n_unitigs
}
pub fn is_empty(&self) -> bool {
self.n_unitigs == 0
}
pub fn n_kmers(&self) -> usize {
self.n_kmers
}
pub fn block_bits(&self) -> u8 {
self.block_bits
}
#[inline]
pub fn has_direct_access(&self) -> bool {
!self.block_offsets.is_empty()
}
/// Byte offset of record `i` in the mmap.
///
@@ -106,12 +117,12 @@ impl UnitigFileReader {
/// sequential scan otherwise.
#[inline]
fn chunk_start(&self, i: usize) -> usize {
if !self.block_offsets.is_empty() {
if self.has_direct_access() {
if self.block_bits == 0 {
return self.block_offsets[i] as usize;
}
let block = i >> self.block_bits;
let rem = i & self.mask;
let rem = i & self.mask;
let mut offset = self.block_offsets[block] as usize;
for _ in 0..rem {
let seql_minus_k = self.mmap[offset] as usize;
@@ -136,10 +147,12 @@ impl UnitigFileReader {
/// Reconstruct chunk `i` as a [`Unitig`].
pub fn unitig(&self, i: usize) -> Unitig {
let offset = self.chunk_start(i);
let seql = self.mmap[offset] as usize + self.k;
let offset = self.chunk_start(i);
let seql = self.mmap[offset] as usize + self.k;
let byte_len = (seql + 3) / 4;
let bytes = self.mmap[offset + 1..offset + 1 + byte_len].to_vec().into_boxed_slice();
let bytes = self.mmap[offset + 1..offset + 1 + byte_len]
.to_vec()
.into_boxed_slice();
Unitig::new((seql % 4) as u8, bytes)
}
@@ -171,20 +184,26 @@ impl UnitigFileReader {
// ── Sequential iterators (O(n) running-offset cursor) ─────────────────────
pub(crate) fn iter_chunks_sequential(&self) -> impl Iterator<Item = (usize, Unitig)> + '_ {
let k = self.k;
let k = self.k;
let mmap = &*self.mmap;
let n = self.n_unitigs;
let n = self.n_unitigs;
let mut offset = 0usize;
(0..n).map(move |chunk_id| {
let seql = mmap[offset] as usize + k;
let seql = mmap[offset] as usize + k;
let byte_len = (seql + 3) / 4;
let bytes = mmap[offset + 1..offset + 1 + byte_len].to_vec().into_boxed_slice();
let bytes = mmap[offset + 1..offset + 1 + byte_len]
.to_vec()
.into_boxed_slice();
offset += 1 + byte_len;
(chunk_id, Unitig::new((seql % 4) as u8, bytes))
})
}
/// Iterate all unitigs sequentially. Works without `.idx` (sequential open).
/// Iterate all unitigs sequentially, paired with their `chunk_id`.
/// Goes through [`iter_chunks_sequential`](Self::iter_chunks_sequential) —
/// never touches `chunk_start`/`block_offsets`, so it works identically
/// whether opened via `open_sequential` or `open_direct_access` (`.idx`
/// loaded or not).
pub fn iter_unitigs(&self) -> impl Iterator<Item = (usize, Unitig)> + '_ {
self.iter_chunks_sequential()
}
@@ -194,15 +213,27 @@ impl UnitigFileReader {
.flat_map(|(_, u)| u.into_kmers())
}
pub fn iter_canonical_kmers(&self) -> impl Iterator<Item = CanonicalKmer> + '_ {
self.iter_chunks_sequential()
.flat_map(|(_, u)| u.into_canonical_kmers())
}
/// Sequential scan, `.idx` not required — despite the name, "indexed"
/// describes what's returned (each k-mer paired with its `(chunk_id,
/// rank)` position), not a dependency on the `.idx` sidecar. Works
/// identically whether or not `.idx` exists, since it goes through
/// [`iter_chunks_sequential`](Self::iter_chunks_sequential) — never
/// `chunk_start`/`block_offsets`. This is what `build_exact_evidence`/
/// `build()` use to fill `evidence.bin` from a freshly written
/// `unitigs.bin`, before `.idx` exists.
pub fn iter_indexed_canonical_kmers(
&self,
) -> impl Iterator<Item = (CanonicalKmer, usize, usize)> + '_ {
self.iter_chunks_sequential()
.flat_map(|(chunk_id, u)| {
u.into_canonical_kmers()
.enumerate()
.map(move |(rank, kmer)| (kmer, chunk_id, rank))
})
self.iter_chunks_sequential().flat_map(|(chunk_id, u)| {
u.into_canonical_kmers()
.enumerate()
.map(move |(rank, kmer)| (kmer, chunk_id, rank))
})
}
/// Same streamed sequence as [`iter_indexed_canonical_kmers`](Self::iter_indexed_canonical_kmers),
@@ -222,7 +253,9 @@ impl UnitigFileReader {
let mmap = &*this.mmap;
let seql = mmap[offset] as usize + k;
let byte_len = (seql + 3) / 4;
let bytes = mmap[offset + 1..offset + 1 + byte_len].to_vec().into_boxed_slice();
let bytes = mmap[offset + 1..offset + 1 + byte_len]
.to_vec()
.into_boxed_slice();
offset += 1 + byte_len;
(chunk_id, Unitig::new((seql % 4) as u8, bytes))
})
@@ -238,8 +271,9 @@ fn read_idx(path: &Path) -> SKResult<(usize, usize, u8, Vec<u32>)> {
let data = std::fs::read(path).map_err(SKError::Io)?;
let mut pos = 0;
let magic_bytes = data.get(pos..pos + 4)
.ok_or(SKError::Truncated { context: "unitig index: magic" })?;
let magic_bytes = data.get(pos..pos + 4).ok_or(SKError::Truncated {
context: "unitig index: magic",
})?;
if magic_bytes != &MAGIC {
return Err(SKError::BadMagic {
expected: "UIX3",
@@ -248,8 +282,9 @@ fn read_idx(path: &Path) -> SKResult<(usize, usize, u8, Vec<u32>)> {
}
pos += 4;
let bb_bytes = data.get(pos..pos + 4)
.ok_or(SKError::Truncated { context: "unitig index: block_bits" })?;
let bb_bytes = data.get(pos..pos + 4).ok_or(SKError::Truncated {
context: "unitig index: block_bits",
})?;
let block_bits_u32 = u32::from_le_bytes(bb_bytes.try_into().unwrap());
if block_bits_u32 > 31 {
return Err(SKError::InvalidData {
@@ -260,13 +295,15 @@ fn read_idx(path: &Path) -> SKResult<(usize, usize, u8, Vec<u32>)> {
let block_bits = block_bits_u32 as u8;
pos += 4;
let n_bytes = data.get(pos..pos + 4)
.ok_or(SKError::Truncated { context: "unitig index: n_unitigs" })?;
let n_bytes = data.get(pos..pos + 4).ok_or(SKError::Truncated {
context: "unitig index: n_unitigs",
})?;
let n_unitigs = u32::from_le_bytes(n_bytes.try_into().unwrap()) as usize;
pos += 4;
let nk_bytes = data.get(pos..pos + 8)
.ok_or(SKError::Truncated { context: "unitig index: n_kmers" })?;
let nk_bytes = data.get(pos..pos + 8).ok_or(SKError::Truncated {
context: "unitig index: n_kmers",
})?;
let n_kmers = u64::from_le_bytes(nk_bytes.try_into().unwrap()) as usize;
pos += 8;
@@ -275,8 +312,9 @@ fn read_idx(path: &Path) -> SKResult<(usize, usize, u8, Vec<u32>)> {
let n_offsets = n_blocks + 1;
let mut block_offsets = Vec::with_capacity(n_offsets);
for _ in 0..n_offsets {
let off_bytes = data.get(pos..pos + 4)
.ok_or(SKError::Truncated { context: "unitig index: block_offsets" })?;
let off_bytes = data.get(pos..pos + 4).ok_or(SKError::Truncated {
context: "unitig index: block_offsets",
})?;
block_offsets.push(u32::from_le_bytes(off_bytes.try_into().unwrap()));
pos += 4;
}