Compare commits
118
Commits
817b02cbc1
...
v1.1.42
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a63692b8c4 | ||
|
|
fa82989ea9 | ||
|
|
5f95e866f8 | ||
|
|
2e7cfc4368 | ||
|
|
f5e508ed33 | ||
|
|
49f329edd5 | ||
|
|
1a470eab9e | ||
|
|
ba990a48a0 | ||
|
|
ea914bb536 | ||
|
|
8bc6d533e5 | ||
|
|
45df9919e5 | ||
|
|
2610a4af79 | ||
|
|
dc3392865f | ||
|
|
fd2c23e7df | ||
|
|
912f788f7f | ||
|
|
e725523898 | ||
|
|
165982fb07 | ||
|
|
2740f52326 | ||
|
|
dff5d2f457 | ||
|
|
eea884f393 | ||
|
|
ae42a061bd | ||
|
|
040eff140c | ||
|
|
a348637f3b | ||
|
|
9d7ced4493 | ||
|
|
9d49929b0c | ||
|
|
61c390503d | ||
|
|
8f0ceec784 | ||
|
|
00b4b1fa51 | ||
|
|
e96ad38c8e | ||
|
|
4fc7860825 | ||
|
|
5bdc0f826a | ||
|
|
cd2f2f9417 | ||
|
|
7844239a8e | ||
|
|
2b37e8aac4 | ||
|
|
67b4e4da53 | ||
|
|
66ab4c6db1 | ||
|
|
f84dd539bf | ||
|
|
6378734e1c | ||
|
|
b3a617cce1 | ||
|
|
2080e5e8a9 | ||
|
|
45ed2bc9b8 | ||
|
|
aa126fd89d | ||
|
|
c612132763 | ||
|
|
19660f8cd0 | ||
|
|
7b07540a69 | ||
|
|
89c43e28f5 | ||
|
|
b9b2e42ad2 | ||
|
|
ca42fdff2f | ||
|
|
136cd89efb | ||
|
|
a4bbf607b7 | ||
|
|
9927100a1c | ||
|
|
527258f822 | ||
|
|
ef62f1947e | ||
|
|
d02316dcf6 | ||
|
|
c323b3eaef | ||
|
|
b77d8e9ca0 | ||
|
|
7c5bab3694 | ||
|
|
fab4e0d6de | ||
|
|
973a3f3d6e | ||
|
|
1a839a295a | ||
|
|
2ea58703c7 | ||
|
|
ac3ef106e7 | ||
|
|
469e53b6f5 | ||
|
|
9f1df96ea7 | ||
|
|
4e4cce2879 | ||
|
|
68b05b93c4 | ||
|
|
0a668cf8a6 | ||
|
|
e6d6942e2f | ||
|
|
bf9c9aeacb | ||
|
|
22a65857a1 | ||
|
|
d16a867640 | ||
|
|
616050075f | ||
|
|
e22afe9621 | ||
|
|
bdfac71e65 | ||
|
|
a00bb37478 | ||
|
|
d30a4efd9b | ||
|
|
6baf2e64ca | ||
|
|
c0a71a2d49 | ||
|
|
a609c1af95 | ||
|
|
3d32be8a83 | ||
|
|
c4c71dc892 | ||
|
|
4e625afaba | ||
|
|
a522c0907e | ||
|
|
c1d6f277ce | ||
|
|
9356be4ec0 | ||
|
|
c694e1f2b0 | ||
|
|
280ca1f5a3 | ||
|
|
9abb2db92f | ||
|
|
7c1efa9cbb | ||
|
|
4c4524766c | ||
|
|
7eea71fdcd | ||
|
|
f91c5a3f79 | ||
|
|
fb4962c4fe | ||
|
|
1d38d87ff9 | ||
|
|
93559c3294 | ||
|
|
1f0d77d5bf | ||
|
|
eeba43ac4f | ||
|
|
7ed7b26039 | ||
|
|
26de90f18d | ||
|
|
497d250d8a | ||
|
|
aa98e82875 | ||
|
|
5ff5b04d2d | ||
|
|
df7b400fda | ||
|
|
d1717688d2 | ||
|
|
cde6457eea | ||
|
|
b6fcbc545f | ||
|
|
9578f991f4 | ||
|
|
1cd7916e06 | ||
|
|
bc92dc4592 | ||
|
|
a9567ad023 | ||
|
|
4a64718fd1 | ||
|
|
7a87e911b6 | ||
|
|
313d73838a | ||
|
|
175ea5bbd0 | ||
|
|
c6ea0c53e3 | ||
|
|
ea767376bd | ||
|
|
f1d76f3203 | ||
|
|
c4071eb450 |
@@ -0,0 +1,41 @@
|
||||
pname: CI
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches: ['main']
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
working-directory: src
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
run: |
|
||||
curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y --default-toolchain stable
|
||||
echo "$HOME/.cargo/bin" >> $GITHUB_PATH
|
||||
|
||||
- name: Cache cargo registry
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
src/target
|
||||
# v2: bump this salt to force a clean cache when a stuck/killed job
|
||||
# may have saved a corrupted incremental-compilation `src/target`
|
||||
# (observed 2026-08-11: a stale test binary deadlocked in NUMA
|
||||
# worker startup; a from-scratch rebuild of the same source fixed
|
||||
# it instantly, pointing at cache corruption rather than a source
|
||||
# bug).
|
||||
key: ${{ runner.os }}-cargo-v2-${{ hashFiles('src/Cargo.lock') }}
|
||||
restore-keys: ${{ runner.os }}-cargo-v2-
|
||||
|
||||
- name: Build
|
||||
run: cargo build --release
|
||||
|
||||
- name: Test
|
||||
run: cargo test --release
|
||||
@@ -0,0 +1,127 @@
|
||||
name: Release
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "v*"
|
||||
|
||||
jobs:
|
||||
create-release:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
release_id: ${{ steps.create.outputs.release_id }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Create Gitea release
|
||||
id: create
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEATOKEN }}
|
||||
TAG: ${{ github.ref_name }}
|
||||
run: |
|
||||
sudo apt-get update -qq && sudo apt-get install -y -qq jq
|
||||
body=$(git for-each-ref --format='%(contents)' "refs/tags/$TAG")
|
||||
release_id=$(curl -s -X POST \
|
||||
"${{ github.server_url }}/api/v1/repos/${{ github.repository }}/releases" \
|
||||
-H "Authorization: token $GITEA_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "{\"tag_name\":\"$TAG\",\"name\":\"$TAG\",\"body\":$(echo "$body" | jq -Rs .)}" | jq -r '.id')
|
||||
echo "release_id=$release_id" >> $GITHUB_OUTPUT
|
||||
|
||||
build-linux-x86_64:
|
||||
needs: create-release
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
working-directory: src
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust + zigbuild
|
||||
run: |
|
||||
curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y --default-toolchain stable
|
||||
echo "$HOME/.cargo/bin" >> $GITHUB_PATH
|
||||
sudo apt-get update -qq && sudo apt-get install -y -qq jq
|
||||
pip install ziglang --quiet --break-system-packages
|
||||
$HOME/.cargo/bin/cargo install cargo-zigbuild
|
||||
$HOME/.cargo/bin/rustup target add x86_64-unknown-linux-musl
|
||||
|
||||
- name: Create musl C/C++ wrappers
|
||||
run: |
|
||||
ZIG=$(python3 -c "import ziglang, os; print(os.path.join(os.path.dirname(ziglang.__file__), 'zig'))")
|
||||
printf '#!/bin/sh\nexec "%s" cc -target x86_64-linux-musl "$@"\n' "$ZIG" | sudo tee /usr/local/bin/x86_64-linux-musl-gcc > /dev/null
|
||||
printf '#!/bin/sh\nexec "%s" c++ -target x86_64-linux-musl "$@"\n' "$ZIG" | sudo tee /usr/local/bin/x86_64-linux-musl-g++ > /dev/null
|
||||
sudo chmod +x /usr/local/bin/x86_64-linux-musl-gcc /usr/local/bin/x86_64-linux-musl-g++
|
||||
|
||||
- name: Cache cargo registry
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
src/target
|
||||
key: linux-musl-cargo-${{ hashFiles('src/Cargo.lock') }}
|
||||
restore-keys: linux-musl-cargo-
|
||||
|
||||
- name: Build static binary
|
||||
env:
|
||||
PKG_CONFIG_ALLOW_CROSS: "1"
|
||||
run: cargo zigbuild --release --target x86_64-unknown-linux-musl
|
||||
|
||||
- name: Prepare and upload artifact
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEATOKEN }}
|
||||
RELEASE_ID: ${{ needs.create-release.outputs.release_id }}
|
||||
run: |
|
||||
mkdir -p /tmp/dist
|
||||
cp target/x86_64-unknown-linux-musl/release/obikmer /tmp/dist/obikmer-linux-x86_64
|
||||
strip /tmp/dist/obikmer-linux-x86_64
|
||||
curl -s -X POST \
|
||||
"${{ github.server_url }}/api/v1/repos/${{ github.repository }}/releases/$RELEASE_ID/assets" \
|
||||
-H "Authorization: token $GITEA_TOKEN" \
|
||||
-F "attachment=@/tmp/dist/obikmer-linux-x86_64"
|
||||
|
||||
build-macos-arm64:
|
||||
needs: create-release
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Login to registry
|
||||
run: echo "${{ secrets.REGISTRYTOKEN }}" | docker login registry.metabarcoding.org -u ${{ secrets.REGISTRYUSER }} --password-stdin
|
||||
|
||||
- name: Cache cargo registry
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
src/target
|
||||
key: macos-arm64-cargo-${{ hashFiles('src/Cargo.lock') }}
|
||||
restore-keys: macos-arm64-cargo-
|
||||
|
||||
- name: Build macOS binary
|
||||
run: |
|
||||
CID=$(docker create \
|
||||
-w /src/src \
|
||||
registry.metabarcoding.org/cibuilder/rustcrossosx:latest \
|
||||
cargo build --release --target aarch64-apple-darwin --no-default-features)
|
||||
docker cp . "$CID:/src"
|
||||
docker start -a "$CID"
|
||||
STATUS=$(docker wait "$CID")
|
||||
mkdir -p /tmp/dist
|
||||
docker cp "$CID:/src/src/target/aarch64-apple-darwin/release/obikmer" /tmp/dist/obikmer-macos-arm64
|
||||
docker rm "$CID" > /dev/null
|
||||
[ "$STATUS" -eq 0 ]
|
||||
|
||||
- name: Prepare and upload artifact
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEATOKEN }}
|
||||
RELEASE_ID: ${{ needs.create-release.outputs.release_id }}
|
||||
run: |
|
||||
curl -s -X POST \
|
||||
"${{ github.server_url }}/api/v1/repos/${{ github.repository }}/releases/$RELEASE_ID/assets" \
|
||||
-H "Authorization: token $GITEA_TOKEN" \
|
||||
-F "attachment=@/tmp/dist/obikmer-macos-arm64"
|
||||
+15
@@ -8,4 +8,19 @@ data-stress
|
||||
*.pb
|
||||
./**/*.json
|
||||
*.bin
|
||||
*.log
|
||||
*.csv
|
||||
Betula_exilis--IGA-24-33
|
||||
benchmark/genomes
|
||||
benchmark/simulated_data
|
||||
benchmark/specimen_index_presence
|
||||
benchmark/specimen_index_count
|
||||
benchmark/global_index_presence
|
||||
benchmark/all_specific
|
||||
benchmark/global_index_count
|
||||
benchmark/stats
|
||||
benchmark/reference_index
|
||||
benchmark/reference_dist
|
||||
benchmark/obikmer_dist
|
||||
benchmark/specific_index_count
|
||||
benchmark/specific_index_presence
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
/cache
|
||||
/project.local.yml
|
||||
@@ -0,0 +1,133 @@
|
||||
# the name by which the project can be referenced within Serena
|
||||
project_name: "obikmer"
|
||||
|
||||
|
||||
# list of languages for which language servers are started; choose from:
|
||||
# al angular ansible bash clojure
|
||||
# cpp cpp_ccls crystal csharp csharp_omnisharp
|
||||
# dart elixir elm erlang fortran
|
||||
# fsharp go groovy haskell haxe
|
||||
# hlsl html java json julia
|
||||
# kotlin lean4 lua luau markdown
|
||||
# matlab msl nix ocaml pascal
|
||||
# perl php php_phpactor powershell python
|
||||
# python_jedi python_ty r rego ruby
|
||||
# ruby_solargraph rust scala scss solidity
|
||||
# svelte swift systemverilog terraform toml
|
||||
# typescript typescript_vts vue yaml zig
|
||||
# (This list may be outdated. For the current list, see values of Language enum here:
|
||||
# https://github.com/oraios/serena/blob/main/src/solidlsp/ls_config.py
|
||||
# For some languages, there are alternative language servers, e.g. csharp_omnisharp, ruby_solargraph.)
|
||||
# Note:
|
||||
# - For C, use cpp
|
||||
# - For JavaScript, use typescript
|
||||
# - For Angular projects, use angular (subsumes typescript+html; requires `npm install` in the project root)
|
||||
# - For Svelte projects, use svelte (subsumes typescript/javascript for .svelte projects; requires npm)
|
||||
# - For SCSS / Sass / plain CSS, use scss (some-sass-language-server handles all three)
|
||||
# - For Free Pascal/Lazarus, use pascal
|
||||
# Special requirements:
|
||||
# Some languages require additional setup/installations.
|
||||
# See here for details: https://oraios.github.io/serena/01-about/020_programming-languages.html#language-servers
|
||||
# When using multiple languages, the first language server that supports a given file will be used for that file.
|
||||
# The first language is the default language and the respective language server will be used as a fallback.
|
||||
# Note that when using the JetBrains backend, language servers are not used and this list is correspondingly ignored.
|
||||
languages:
|
||||
- rust
|
||||
|
||||
# the encoding used by text files in the project
|
||||
# For a list of possible encodings, see https://docs.python.org/3.11/library/codecs.html#standard-encodings
|
||||
encoding: "utf-8"
|
||||
|
||||
# line ending convention to use when writing source files.
|
||||
# Possible values: unset (use global setting), "lf", "crlf", or "native" (platform default)
|
||||
# This does not affect Serena's own files (e.g. memories and configuration files), which always use native line endings.
|
||||
line_ending:
|
||||
|
||||
# The language backend to use for this project.
|
||||
# If not set, the global setting from serena_config.yml is used.
|
||||
# Valid values: LSP, JetBrains
|
||||
# Note: the backend is fixed at startup. If a project with a different backend
|
||||
# is activated post-init, an error will be returned.
|
||||
language_backend:
|
||||
|
||||
# whether to use project's .gitignore files to ignore files
|
||||
ignore_all_files_in_gitignore: true
|
||||
|
||||
# advanced configuration option allowing to configure language server-specific options.
|
||||
# Maps the language key to the options.
|
||||
# Have a look at the docstring of the constructors of the LS implementations within solidlsp (e.g., for C# or PHP) to see which options are available.
|
||||
# No documentation on options means no options are available.
|
||||
ls_specific_settings: {}
|
||||
|
||||
# list of additional workspace folder paths for cross-package reference support (e.g. in monorepos).
|
||||
# Paths can be absolute or relative to the project root.
|
||||
# Each folder is registered as an LSP workspace folder, enabling language servers to discover
|
||||
# symbols and references across package boundaries.
|
||||
# Currently supported for: TypeScript.
|
||||
# Example:
|
||||
# additional_workspace_folders:
|
||||
# - ../sibling-package
|
||||
# - ../shared-lib
|
||||
additional_workspace_folders: []
|
||||
|
||||
# list of additional paths to ignore in this project.
|
||||
# Same syntax as gitignore, so you can use * and **.
|
||||
# Note: global ignored_paths from serena_config.yml are also applied additively.
|
||||
ignored_paths: []
|
||||
|
||||
# whether the project is in read-only mode
|
||||
# If set to true, all editing tools will be disabled and attempts to use them will result in an error
|
||||
# Added on 2025-04-18
|
||||
read_only: false
|
||||
|
||||
# list of tool names to exclude.
|
||||
# This extends the existing exclusions (e.g. from the global configuration)
|
||||
# Find the list of tools here: https://oraios.github.io/serena/01-about/035_tools.html
|
||||
excluded_tools: []
|
||||
|
||||
# list of tools to include that would otherwise be disabled (particularly optional tools that are disabled by default).
|
||||
# This extends the existing inclusions (e.g. from the global configuration).
|
||||
# Find the list of tools here: https://oraios.github.io/serena/01-about/035_tools.html
|
||||
included_optional_tools: []
|
||||
|
||||
# fixed set of tools to use as the base tool set (if non-empty), replacing Serena's default set of tools.
|
||||
# This cannot be combined with non-empty excluded_tools or included_optional_tools.
|
||||
# Find the list of tools here: https://oraios.github.io/serena/01-about/035_tools.html
|
||||
fixed_tools: []
|
||||
|
||||
# list of mode names that are to be activated by default, overriding the setting in the global configuration.
|
||||
# The full set of modes to be activated is base_modes (from global config) + default_modes + added_modes.
|
||||
# If the setting is undefined/empty, the default_modes from the global configuration (serena_config.yml) apply.
|
||||
# Otherwise, this overrides the setting from the global configuration (serena_config.yml).
|
||||
# Therefore, you can set this to [] if you do not want the default modes defined in the global config to apply
|
||||
# for this project.
|
||||
# This setting can, in turn, be overridden by CLI parameters (--mode).
|
||||
# See https://oraios.github.io/serena/02-usage/050_configuration.html#modes
|
||||
default_modes:
|
||||
|
||||
# list of mode names to be activated additionally for this project, e.g. ["query-projects"]
|
||||
# The full set of modes to be activated is base_modes (from global config) + default_modes + added_modes.
|
||||
# See https://oraios.github.io/serena/02-usage/050_configuration.html#modes
|
||||
added_modes:
|
||||
|
||||
# initial prompt for the project. It will always be given to the LLM upon activating the project
|
||||
# (contrary to the memories, which are loaded on demand).
|
||||
initial_prompt: ""
|
||||
|
||||
# time budget (seconds) per tool call for the retrieval of additional symbol information
|
||||
# such as docstrings or parameter information.
|
||||
# This overrides the corresponding setting in the global configuration; see the documentation there.
|
||||
# If null or missing, use the setting from the global configuration.
|
||||
symbol_info_budget:
|
||||
|
||||
# list of regex patterns which, when matched, mark a memory entry as read‑only.
|
||||
# Extends the list from the global configuration, merging the two lists.
|
||||
read_only_memory_patterns: []
|
||||
|
||||
# list of regex patterns for memories to completely ignore.
|
||||
# Matching memories will not appear in list_memories or activate_project output
|
||||
# and cannot be accessed via read_memory or write_memory.
|
||||
# To access ignored memory files, use the read_file tool on the raw file path.
|
||||
# Extends the list from the global configuration, merging the two lists.
|
||||
# Example: ["_archive/.*", "_episodes/.*"]
|
||||
ignored_memory_patterns: []
|
||||
@@ -73,3 +73,29 @@ Lors de l'ajout de nouveaux fichiers Markdown dans `docmd/`, mettre à jour la s
|
||||
---
|
||||
|
||||
Je continue à poser mes questions et à guider la discussion.
|
||||
|
||||
---
|
||||
|
||||
## MCP Tools
|
||||
|
||||
**Règle absolue : avant tout travail de code, appeler `mcp__serena__initial_instructions` pour charger les instructions Serena.**
|
||||
|
||||
### Hiérarchie des outils pour ce projet Rust
|
||||
|
||||
**Navigation et édition de code → serena en priorité**
|
||||
- Trouver un symbole, une déclaration, les implémentations d'un trait : `mcp__serena__find_symbol`, `mcp__serena__find_declaration`, `mcp__serena__find_implementations`
|
||||
- Trouver les usages d'un symbole : `mcp__serena__find_referencing_symbols`
|
||||
- Diagnostics LSP (erreurs de compilation) : `mcp__serena__get_diagnostics_for_file`
|
||||
- Vue d'ensemble d'un fichier : `mcp__serena__get_symbols_overview`
|
||||
- Modifier le corps d'une fonction/impl : `mcp__serena__replace_symbol_body`
|
||||
- Ne pas utiliser `cclsp` quand serena couvre le besoin
|
||||
|
||||
**Analyse architecturale → jcodemunch**
|
||||
- Hotspots, couplage, dead code, dépendances entre modules
|
||||
- Utiliser avant de refactorer une zone critique
|
||||
|
||||
**Raisonnement complexe → sequential-thinking**
|
||||
- Décisions d'architecture, choix d'algorithme, trade-offs non triviaux
|
||||
|
||||
**Documentation de crates → context7**
|
||||
- Toujours consulter avant d'utiliser une API de bibliothèque externe
|
||||
|
||||
@@ -22,6 +22,7 @@ $(MKDOCS): $(VENV)/bin/activate
|
||||
mkdocs mkdocs-material \
|
||||
mkdocs-mermaid2-plugin \
|
||||
mkdocs-bibtex
|
||||
$(PIP) install --quiet --upgrade InSilicoSeq
|
||||
|
||||
# ── obikmer binary ───────────────────────────────────────────────────────────
|
||||
|
||||
@@ -62,3 +63,36 @@ clean-doc:
|
||||
.PHONY: clean
|
||||
clean: clean-doc
|
||||
rm -rf $(VENV)
|
||||
|
||||
# ── release ───────────────────────────────────────────────────────────────────
|
||||
|
||||
CARGO_TOML := $(CARGO_DIR)/obikmer/Cargo.toml
|
||||
|
||||
.PHONY: bump-version
|
||||
bump-version:
|
||||
@current=$$(grep '^version = ' $(CARGO_TOML) | head -n 1 | sed 's/version = "\(.*\)"/\1/'); \
|
||||
if [ -n "$(RELEASE)" ]; then \
|
||||
new_version="$(RELEASE)"; \
|
||||
else \
|
||||
major=$$(echo $$current | cut -d. -f1); \
|
||||
minor=$$(echo $$current | cut -d. -f2); \
|
||||
patch=$$(echo $$current | cut -d. -f3); \
|
||||
new_patch=$$((patch + 1)); \
|
||||
new_version="$$major.$$minor.$$new_patch"; \
|
||||
fi; \
|
||||
echo "Version: $$current -> $$new_version"; \
|
||||
sed -i.bak "s/^version = \"$$current\"/version = \"$$new_version\"/" $(CARGO_TOML) && \
|
||||
rm $(CARGO_TOML).bak
|
||||
|
||||
.PHONY: release
|
||||
release: bump-version
|
||||
@jj auto-describe
|
||||
@jj git push --change @
|
||||
@new_version=$$(grep '^version = ' $(CARGO_TOML) | head -n 1 | sed 's/version = "\(.*\)"/\1/'); \
|
||||
git_hash=$$(jj log -r @ --no-graph -T 'commit_id'); \
|
||||
commits=$$(jj log -r 'latest(tags())..@' --no-graph -T 'description ++ "\n"' 2>/dev/null || \
|
||||
jj log --no-graph -T 'description ++ "\n"' --limit 30); \
|
||||
notes=$$(printf 'Write concise markdown release notes for obikmer (a Rust kmer genomics tool). Be technical and direct. Base them strictly on these commit messages:\n\n%s' "$$commits" | aichat 2>/dev/null); \
|
||||
tag_msg="$${notes:-Release v$$new_version}"; \
|
||||
git tag -a "v$$new_version" -m "$$tag_msg" "$$git_hash" && \
|
||||
git push origin "v$$new_version"
|
||||
|
||||
@@ -0,0 +1,230 @@
|
||||
# Requires GNU Make >= 4.3 (grouped targets &:) — use gmake on macOS
|
||||
BINARY := ../src/target/release/obikmer
|
||||
VENV_PY := ../.venv/bin/python3
|
||||
|
||||
GENOMES := $(wildcard genomes/*.fna.gz)
|
||||
|
||||
# SPECIMENS, SPECIES, and the full dependency graph are generated by
|
||||
# make_deps.py from the genome FASTA headers — like .d files in C.
|
||||
# Make rebuilds deps.mk whenever genomes/ changes and restarts.
|
||||
-include deps.mk
|
||||
|
||||
REF_NPZS := $(SPECIMENS:%=reference_index/%.npz)
|
||||
REF_DIST_CSVS := $(addprefix reference_dist/, \
|
||||
shared_kmers.csv hamming_dist.csv jaccard_dist.csv \
|
||||
bray_curtis_dist.csv relfreq_bray_curtis_dist.csv \
|
||||
euclidean_dist.csv relfreq_euclidean_dist.csv \
|
||||
hellinger_dist.csv hellinger_euclidean_dist.csv)
|
||||
OBIKMER_PRESENCE_DIST := $(addprefix obikmer_dist/presence/, \
|
||||
jaccard_dist.csv jaccard_shared.csv jaccard_nj.nwk \
|
||||
hamming_dist.csv hamming_nj.nwk)
|
||||
OBIKMER_COUNT_DIST := $(addprefix obikmer_dist/count/, \
|
||||
jaccard_dist.csv jaccard_shared.csv jaccard_nj.nwk \
|
||||
bray_curtis_dist.csv bray_curtis_nj.nwk \
|
||||
relfreq_bray_curtis_dist.csv relfreq_bray_curtis_nj.nwk \
|
||||
euclidean_dist.csv euclidean_nj.nwk \
|
||||
relfreq_euclidean_dist.csv relfreq_euclidean_nj.nwk \
|
||||
hellinger_dist.csv hellinger_nj.nwk \
|
||||
hellinger_euclidean_dist.csv hellinger_euclidean_nj.nwk)
|
||||
DIST_COMPARISON := stats/dist_comparison/summary.csv
|
||||
PRESENCE_DONE := $(SPECIMENS:%=specimen_index_presence/%/index.done)
|
||||
PRESENCE_STATS := $(SPECIMENS:%=stats/indexing_presence/%.stats)
|
||||
COUNT_DONE := $(SPECIMENS:%=specimen_index_count/%/index.done)
|
||||
COUNT_STATS := $(SPECIMENS:%=stats/indexing_count/%.stats)
|
||||
VERIFY_PRESENCE_STATS := $(SPECIMENS:%=stats/verify_presence/%.stats)
|
||||
VERIFY_COUNT_STATS := $(SPECIMENS:%=stats/verify_count/%.stats)
|
||||
SPECIFIC_PRESENCE_DONE := $(SPECIES:%=specific_index_presence/%/index.done)
|
||||
SPECIFIC_PRESENCE_STATS := $(SPECIES:%=stats/specific_kmer_presence/%.stats)
|
||||
SPECIFIC_COUNT_DONE := $(SPECIES:%=specific_index_count/%/index.done)
|
||||
SPECIFIC_COUNT_STATS := $(SPECIES:%=stats/specific_kmer_count/%.stats)
|
||||
SIMULATED_READS := $(foreach s,$(SPECIMENS),simulated_data/$(subst --,/,$s)/reads_R1.fastq.gz)
|
||||
|
||||
.NOTPARALLEL:
|
||||
|
||||
.PHONY: all simulate reference reference_dist \
|
||||
obikmer_dist obikmer_dist_presence obikmer_dist_count \
|
||||
dist_comparison \
|
||||
index_presence index_count \
|
||||
aggregate_index_presence aggregate_index_count \
|
||||
merge_presence merge_count \
|
||||
verify_presence verify_count \
|
||||
aggregate_verify_presence aggregate_verify_count \
|
||||
verify_merge_presence verify_merge_count \
|
||||
filter_presence filter_count \
|
||||
aggregate_filter_presence aggregate_filter_count
|
||||
|
||||
verify_merge_presence: stats/verify_merge_presence/current.csv
|
||||
verify_merge_count: stats/verify_merge_count/current.csv
|
||||
|
||||
all: aggregate_verify_presence aggregate_verify_count \
|
||||
verify_merge_presence verify_merge_count \
|
||||
aggregate_filter_presence aggregate_filter_count \
|
||||
dist_comparison
|
||||
|
||||
# ── dependency file ───────────────────────────────────────────────────────────
|
||||
|
||||
deps.mk: $(GENOMES)
|
||||
$(VENV_PY) make_deps.py $^ > $@
|
||||
|
||||
# ── simulation ────────────────────────────────────────────────────────────────
|
||||
# Prerequisites (genome → reads) are in deps.mk; $< is the genome file.
|
||||
|
||||
$(SIMULATED_READS):
|
||||
bash simulate_one.sh $< $(dir $@)
|
||||
|
||||
simulate: $(SIMULATED_READS)
|
||||
|
||||
# ── reference kmer sets ───────────────────────────────────────────────────────
|
||||
# Prerequisites (reads → npz) are in deps.mk.
|
||||
|
||||
reference_index/%.npz:
|
||||
bash build_reference.sh $*
|
||||
|
||||
reference: $(REF_NPZS)
|
||||
|
||||
# ── reference distance matrices ───────────────────────────────────────────────
|
||||
|
||||
$(REF_DIST_CSVS) &: $(REF_NPZS) build_reference_dist.py
|
||||
$(VENV_PY) build_reference_dist.py
|
||||
|
||||
reference_dist: $(REF_DIST_CSVS)
|
||||
|
||||
# ── obikmer distance (presence index) ────────────────────────────────────────
|
||||
|
||||
$(OBIKMER_PRESENCE_DIST) &: global_index_presence/index.done $(BINARY)
|
||||
mkdir -p obikmer_dist/presence
|
||||
$(BINARY) distance \
|
||||
--output obikmer_dist/presence/jaccard \
|
||||
--metric jaccard --shared-kmers --nj \
|
||||
global_index_presence
|
||||
$(BINARY) distance \
|
||||
--output obikmer_dist/presence/hamming \
|
||||
--metric hamming --nj \
|
||||
global_index_presence
|
||||
|
||||
obikmer_dist_presence: $(OBIKMER_PRESENCE_DIST)
|
||||
|
||||
# ── obikmer distance (count index) ───────────────────────────────────────────
|
||||
|
||||
$(OBIKMER_COUNT_DIST) &: global_index_count/index.done $(BINARY)
|
||||
mkdir -p obikmer_dist/count
|
||||
$(BINARY) distance \
|
||||
--output obikmer_dist/count/jaccard \
|
||||
--metric jaccard --shared-kmers --nj \
|
||||
global_index_count
|
||||
$(BINARY) distance \
|
||||
--output obikmer_dist/count/bray_curtis \
|
||||
--metric bray-curtis --nj \
|
||||
global_index_count
|
||||
$(BINARY) distance \
|
||||
--output obikmer_dist/count/relfreq_bray_curtis \
|
||||
--metric relfreq-bray-curtis --nj \
|
||||
global_index_count
|
||||
$(BINARY) distance \
|
||||
--output obikmer_dist/count/euclidean \
|
||||
--metric euclidean --nj \
|
||||
global_index_count
|
||||
$(BINARY) distance \
|
||||
--output obikmer_dist/count/relfreq_euclidean \
|
||||
--metric relfreq-euclidean --nj \
|
||||
global_index_count
|
||||
$(BINARY) distance \
|
||||
--output obikmer_dist/count/hellinger \
|
||||
--metric hellinger --nj \
|
||||
global_index_count
|
||||
$(BINARY) distance \
|
||||
--output obikmer_dist/count/hellinger_euclidean \
|
||||
--metric hellinger-euclidean --nj \
|
||||
global_index_count
|
||||
|
||||
obikmer_dist_count: $(OBIKMER_COUNT_DIST)
|
||||
|
||||
obikmer_dist: obikmer_dist_presence obikmer_dist_count
|
||||
|
||||
# ── distance comparison ───────────────────────────────────────────────────────
|
||||
|
||||
$(DIST_COMPARISON): $(REF_DIST_CSVS) $(OBIKMER_PRESENCE_DIST) $(OBIKMER_COUNT_DIST) compare_all_dist.py
|
||||
$(VENV_PY) compare_all_dist.py --out $(DIST_COMPARISON)
|
||||
|
||||
dist_comparison: $(DIST_COMPARISON)
|
||||
|
||||
# ── per-specimen indexing ─────────────────────────────────────────────────────
|
||||
# Prerequisites (reads → index.done + .stats) are in deps.mk.
|
||||
|
||||
specimen_index_presence/%/index.done \
|
||||
stats/indexing_presence/%.stats &: $(BINARY)
|
||||
bash index_one_presence.sh $*
|
||||
|
||||
specimen_index_count/%/index.done \
|
||||
stats/indexing_count/%.stats &: $(BINARY)
|
||||
bash index_one_count.sh $*
|
||||
|
||||
index_presence: $(PRESENCE_DONE)
|
||||
index_count: $(COUNT_DONE)
|
||||
|
||||
# ── indexing stats aggregation ────────────────────────────────────────────────
|
||||
|
||||
aggregate_index_presence: $(PRESENCE_STATS)
|
||||
bash aggregate_stats.sh indexing_presence
|
||||
|
||||
aggregate_index_count: $(COUNT_STATS)
|
||||
bash aggregate_stats.sh indexing_count
|
||||
|
||||
# ── global merge ──────────────────────────────────────────────────────────────
|
||||
|
||||
global_index_presence/index.done: $(PRESENCE_DONE) $(BINARY)
|
||||
bash merge_presence.sh
|
||||
|
||||
global_index_count/index.done: $(COUNT_DONE) $(BINARY)
|
||||
bash merge_count.sh
|
||||
|
||||
merge_presence: global_index_presence/index.done
|
||||
merge_count: global_index_count/index.done
|
||||
|
||||
# ── per-specimen verification ─────────────────────────────────────────────────
|
||||
# Prerequisites (index.done + npz → .stats) are in deps.mk.
|
||||
|
||||
stats/verify_presence/%.stats:
|
||||
bash verify_one_presence.sh $*
|
||||
|
||||
stats/verify_count/%.stats:
|
||||
bash verify_one_count.sh $*
|
||||
|
||||
verify_presence: $(VERIFY_PRESENCE_STATS)
|
||||
verify_count: $(VERIFY_COUNT_STATS)
|
||||
|
||||
# ── verification stats aggregation ───────────────────────────────────────────
|
||||
|
||||
aggregate_verify_presence: $(VERIFY_PRESENCE_STATS)
|
||||
bash aggregate_stats.sh verify_presence
|
||||
|
||||
aggregate_verify_count: $(VERIFY_COUNT_STATS)
|
||||
bash aggregate_stats.sh verify_count
|
||||
|
||||
# ── species-specific indexes ──────────────────────────────────────────────────
|
||||
# Prerequisites (global index → specific index) are in deps.mk.
|
||||
|
||||
specific_index_presence/%/index.done \
|
||||
stats/specific_kmer_presence/%.stats &: $(BINARY)
|
||||
bash filter_one_presence.sh $*
|
||||
|
||||
specific_index_count/%/index.done \
|
||||
stats/specific_kmer_count/%.stats &: $(BINARY)
|
||||
bash filter_one_count.sh $*
|
||||
|
||||
filter_presence: $(SPECIFIC_PRESENCE_DONE)
|
||||
filter_count: $(SPECIFIC_COUNT_DONE)
|
||||
|
||||
aggregate_filter_presence: $(SPECIFIC_PRESENCE_STATS)
|
||||
bash aggregate_stats.sh specific_kmer_presence
|
||||
|
||||
aggregate_filter_count: $(SPECIFIC_COUNT_STATS)
|
||||
bash aggregate_stats.sh specific_kmer_count
|
||||
|
||||
# ── merged index verification ─────────────────────────────────────────────────
|
||||
|
||||
stats/verify_merge_presence/current.csv: $(REF_NPZS) global_index_presence/index.done
|
||||
bash verify_merge_presence.sh
|
||||
|
||||
stats/verify_merge_count/current.csv: $(REF_NPZS) global_index_count/index.done
|
||||
bash verify_merge_count.sh
|
||||
@@ -0,0 +1,132 @@
|
||||
# Benchmark pipeline
|
||||
|
||||
Requires **GNU Make ≥ 4.3** (grouped targets `&:`). On macOS use `gmake`.
|
||||
|
||||
```
|
||||
gmake all # full pipeline
|
||||
gmake simulate # simulation only
|
||||
gmake reference # reference kmer sets only
|
||||
```
|
||||
|
||||
## Pipeline overview
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
GENOMES["genomes/*.fna.gz"]
|
||||
BIN["obikmer binary"]
|
||||
|
||||
GENOMES --> simulate
|
||||
simulate --> simdata[("simulated_data/")]
|
||||
|
||||
simdata --> reference
|
||||
reference --> refnpz[("reference_index/*.npz")]
|
||||
|
||||
subgraph presence ["Presence track"]
|
||||
simdata --> index_presence
|
||||
BIN --> index_presence
|
||||
index_presence --> pres_done[("specimen_index_presence/")]
|
||||
index_presence --> pres_istats[("stats/indexing_presence/")]
|
||||
pres_istats --> aggregate_index_presence
|
||||
|
||||
pres_done --> merge_presence
|
||||
BIN --> merge_presence
|
||||
merge_presence --> gpres[("global_index_presence/")]
|
||||
|
||||
refnpz --> verify_presence
|
||||
pres_done --> verify_presence
|
||||
verify_presence --> vpres_stats[("stats/verify_presence/")]
|
||||
vpres_stats --> aggregate_verify_presence
|
||||
|
||||
gpres --> filter_presence
|
||||
BIN --> filter_presence
|
||||
filter_presence --> spec_pres[("specific_index_presence/")]
|
||||
filter_presence --> spec_pres_stats[("stats/specific_kmer_presence/")]
|
||||
spec_pres_stats --> aggregate_filter_presence
|
||||
|
||||
refnpz --> verify_merge_presence
|
||||
gpres --> verify_merge_presence
|
||||
verify_merge_presence --> vmp[("stats/verify_merge_presence/")]
|
||||
end
|
||||
|
||||
subgraph count ["Count track"]
|
||||
simdata --> index_count
|
||||
BIN --> index_count
|
||||
index_count --> count_done[("specimen_index_count/")]
|
||||
index_count --> count_istats[("stats/indexing_count/")]
|
||||
count_istats --> aggregate_index_count
|
||||
|
||||
count_done --> merge_count
|
||||
BIN --> merge_count
|
||||
merge_count --> gcount[("global_index_count/")]
|
||||
|
||||
refnpz --> verify_count
|
||||
count_done --> verify_count
|
||||
verify_count --> vcount_stats[("stats/verify_count/")]
|
||||
vcount_stats --> aggregate_verify_count
|
||||
|
||||
gcount --> filter_count
|
||||
BIN --> filter_count
|
||||
filter_count --> spec_count[("specific_index_count/")]
|
||||
filter_count --> spec_count_stats[("stats/specific_kmer_count/")]
|
||||
spec_count_stats --> aggregate_filter_count
|
||||
|
||||
refnpz --> verify_merge_count
|
||||
gcount --> verify_merge_count
|
||||
verify_merge_count --> vmc[("stats/verify_merge_count/")]
|
||||
end
|
||||
|
||||
aggregate_verify_presence --> all
|
||||
aggregate_verify_count --> all
|
||||
vmp --> all
|
||||
vmc --> all
|
||||
all -. "$(MAKE) re-eval" .-> aggregate_filter_presence
|
||||
all -. "$(MAKE) re-eval" .-> aggregate_filter_count
|
||||
```
|
||||
|
||||
## Steps
|
||||
|
||||
| Target | Script | Description |
|
||||
|---|---|---|
|
||||
| `simulate` | `simulate.sh` | Simulate sequencing reads from the reference genomes |
|
||||
| `reference` | `build_reference.sh` | Build reference kmer sets (`.npz`) from simulation truth |
|
||||
| `index_presence` | `index_one_presence.sh` | Index each specimen (presence mode) |
|
||||
| `index_count` | `index_one_count.sh` | Index each specimen (count mode) |
|
||||
| `aggregate_index_presence` | `aggregate_stats.sh` | Aggregate per-specimen indexing stats (presence) |
|
||||
| `aggregate_index_count` | `aggregate_stats.sh` | Aggregate per-specimen indexing stats (count) |
|
||||
| `merge_presence` | `merge_presence.sh` | Merge all specimen presence indexes into a global index |
|
||||
| `merge_count` | `merge_count.sh` | Merge all specimen count indexes into a global index |
|
||||
| `verify_presence` | `verify_one_presence.sh` | Verify each specimen presence index against reference |
|
||||
| `verify_count` | `verify_one_count.sh` | Verify each specimen count index against reference |
|
||||
| `aggregate_verify_presence` | `aggregate_stats.sh` | Aggregate per-specimen verification stats (presence) |
|
||||
| `aggregate_verify_count` | `aggregate_stats.sh` | Aggregate per-specimen verification stats (count) |
|
||||
| `filter_presence` | `filter_one_presence.sh` | Extract species-specific presence indexes from global index |
|
||||
| `filter_count` | `filter_one_count.sh` | Extract species-specific count indexes from global index |
|
||||
| `aggregate_filter_presence` | `aggregate_stats.sh` | Aggregate species-specific kmer stats (presence) |
|
||||
| `aggregate_filter_count` | `aggregate_stats.sh` | Aggregate species-specific kmer stats (count) |
|
||||
| `verify_merge_presence` | `verify_merge_presence.sh` | Verify global presence index against all reference sets |
|
||||
| `verify_merge_count` | `verify_merge_count.sh` | Verify global count index against all reference sets |
|
||||
|
||||
## Directory layout
|
||||
|
||||
```
|
||||
benchmark/
|
||||
├── genomes/ # input reference genomes (.fna.gz)
|
||||
├── simulated_data/ # generated by simulate
|
||||
│ └── <species>/<specimen>/
|
||||
├── reference_index/ # reference kmer sets (.npz)
|
||||
├── specimen_index_presence/ # per-specimen presence indexes
|
||||
├── specimen_index_count/ # per-specimen count indexes
|
||||
├── global_index_presence/ # merged global presence index
|
||||
├── global_index_count/ # merged global count index
|
||||
├── specific_index_presence/ # species-specific presence indexes
|
||||
├── specific_index_count/ # species-specific count indexes
|
||||
└── stats/ # all benchmark statistics
|
||||
├── indexing_presence/
|
||||
├── indexing_count/
|
||||
├── verify_presence/
|
||||
├── verify_count/
|
||||
├── specific_kmer_presence/
|
||||
├── specific_kmer_count/
|
||||
├── verify_merge_presence/
|
||||
└── verify_merge_count/
|
||||
```
|
||||
Executable
+53
@@ -0,0 +1,53 @@
|
||||
#!/usr/bin/env bash
|
||||
# Usage: aggregate_stats.sh TYPE
|
||||
# TYPE = indexing_presence | indexing_count | verify_presence | verify_count
|
||||
#
|
||||
# Reads all stats/TYPE/*.stats files (one CSV data row each, no header).
|
||||
# Creates a new stats/TYPE/run_NNN.csv only if any .stats file is newer than
|
||||
# the most recent run CSV (idempotent when nothing changed).
|
||||
set -euo pipefail
|
||||
|
||||
TYPE="$1"
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/${TYPE}"
|
||||
|
||||
case "${TYPE}" in
|
||||
indexing_presence|indexing_count)
|
||||
HEADER="run,species,strain,scatter_wall_s,scatter_rss_b,dereplicate_wall_s,dereplicate_rss_b,count_kmer_wall_s,count_kmer_rss_b,index_wall_s,index_rss_b,total_wall_s,total_rss_b"
|
||||
;;
|
||||
verify_presence)
|
||||
HEADER="run,species,strain,ref_kmers,idx_kmers,false_neg,false_pos,fn_pct,fp_pct"
|
||||
;;
|
||||
verify_count)
|
||||
HEADER="run,species,strain,ref_kmers,idx_kmers,false_neg,false_pos,count_mismatch,fn_pct,fp_pct,cm_pct"
|
||||
;;
|
||||
specific_kmer_presence|specific_kmer_count)
|
||||
HEADER="run,species,rebuild_wall_s,rebuild_rss_b,pack_wall_s,pack_rss_b,filter_total_wall_s,filter_total_rss_b,select_wall_s,select_rss_b,select_total_wall_s,select_total_rss_b"
|
||||
;;
|
||||
*)
|
||||
echo "ERROR: unknown stats type '${TYPE}'" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
# Find most recent existing run CSV (empty string if none).
|
||||
latest_csv=$(find "${STATS_DIR}" -maxdepth 1 -name 'run_*.csv' 2>/dev/null | sort | tail -1)
|
||||
|
||||
# Check if any .stats file is newer than the latest run CSV.
|
||||
if [[ -n "${latest_csv}" ]] && \
|
||||
[[ -z "$(find "${STATS_DIR}" -maxdepth 1 -name '*.stats' -newer "${latest_csv}" 2>/dev/null)" ]]; then
|
||||
echo "[${TYPE}] stats up to date (${latest_csv})"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
run_n=$(printf '%03d' "$(find "${STATS_DIR}" -maxdepth 1 -name 'run_*.csv' 2>/dev/null | wc -l | tr -d ' ')")
|
||||
CSV="${STATS_DIR}/run_${run_n}.csv"
|
||||
|
||||
echo "${HEADER}" >"${CSV}"
|
||||
|
||||
# Sort .stats files by name for reproducible row order.
|
||||
while IFS= read -r stats_file; do
|
||||
sed "s/^/${run_n},/" "${stats_file}"
|
||||
done < <(find "${STATS_DIR}" -maxdepth 1 -name '*.stats' | sort) >>"${CSV}"
|
||||
|
||||
echo "[${TYPE}] run ${run_n} → ${CSV}"
|
||||
Executable
+137
@@ -0,0 +1,137 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Build a reference kmer index from paired-end FASTQ reads.
|
||||
|
||||
Extracts canonical kmers — min(kmer, revcomp(kmer)) encoded as uint64 —
|
||||
counts their abundances, and saves a sorted numpy pair (kmers, counts).
|
||||
|
||||
Output .npz arrays
|
||||
kmers : uint64, sorted ascending — canonical kmer integers
|
||||
counts : uint32, same order — raw read abundances
|
||||
"""
|
||||
import argparse
|
||||
import gzip
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
# ── encoding ────────────────────────────────────────────────────────────────
|
||||
|
||||
_ENCODE = {'A': 0, 'C': 1, 'G': 2, 'T': 3,
|
||||
'a': 0, 'c': 1, 'g': 2, 't': 3}
|
||||
|
||||
# Lookup table: revcomp of one byte (4 bases, 8 bits).
|
||||
# Precomputed once at import time.
|
||||
_REVCOMP8 = [0] * 256
|
||||
for _i in range(256):
|
||||
_rc, _x = 0, _i
|
||||
for _ in range(4):
|
||||
_rc = (_rc << 2) | (3 - (_x & 3))
|
||||
_x >>= 2
|
||||
_REVCOMP8[_i] = _rc
|
||||
del _i, _rc, _x
|
||||
|
||||
|
||||
def revcomp_int(kmer: int, k: int) -> int:
|
||||
"""Reverse-complement of a kmer encoded as an integer (2 bits/base).
|
||||
|
||||
Uses byte-level lookup (4 bases at a time) for speed.
|
||||
"""
|
||||
rc = 0
|
||||
bits_left = 2 * k
|
||||
while bits_left > 0:
|
||||
chunk = min(8, bits_left)
|
||||
rc_byte = _REVCOMP8[kmer & 0xFF] >> (8 - chunk)
|
||||
rc = (rc << chunk) | rc_byte
|
||||
kmer >>= chunk
|
||||
bits_left -= chunk
|
||||
return rc
|
||||
|
||||
|
||||
# ── FASTQ parsing ────────────────────────────────────────────────────────────
|
||||
|
||||
def iter_sequences(path: str):
|
||||
"""Yield raw sequences from a (gzipped) FASTQ file."""
|
||||
opener = gzip.open if path.endswith('.gz') else open
|
||||
with opener(path, 'rt') as fh:
|
||||
while True:
|
||||
if not fh.readline(): # '@' header
|
||||
break
|
||||
seq = fh.readline().rstrip('\n')
|
||||
fh.readline() # '+'
|
||||
fh.readline() # quality
|
||||
yield seq
|
||||
|
||||
|
||||
# ── kmer counting ────────────────────────────────────────────────────────────
|
||||
|
||||
def count_kmers(paths: list[str], k: int) -> dict[int, int]:
|
||||
mask = (1 << (2 * k)) - 1
|
||||
counts: dict[int, int] = defaultdict(int)
|
||||
n_reads = 0
|
||||
|
||||
for path in paths:
|
||||
for seq in iter_sequences(path):
|
||||
n_reads += 1
|
||||
kmer = 0
|
||||
run = 0 # consecutive valid bases
|
||||
|
||||
for c in seq:
|
||||
b = _ENCODE.get(c)
|
||||
if b is None: # N or unexpected character → reset
|
||||
kmer = 0
|
||||
run = 0
|
||||
continue
|
||||
kmer = ((kmer << 2) | b) & mask
|
||||
run += 1
|
||||
if run >= k:
|
||||
rc = revcomp_int(kmer, k)
|
||||
counts[kmer if kmer <= rc else rc] += 1
|
||||
|
||||
if n_reads % 100_000 == 0:
|
||||
print(f' {n_reads:,} reads processed, '
|
||||
f'{len(counts):,} distinct kmers so far',
|
||||
file=sys.stderr)
|
||||
|
||||
print(f' {n_reads:,} reads total, {len(counts):,} distinct kmers',
|
||||
file=sys.stderr)
|
||||
return counts
|
||||
|
||||
|
||||
# ── main ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('reads', nargs='+', metavar='FASTQ',
|
||||
help='Input reads (FASTQ, gzip OK)')
|
||||
ap.add_argument('-k', '--kmer-size', type=int, default=31,
|
||||
metavar='K')
|
||||
ap.add_argument('--min-abundance', type=int, default=1,
|
||||
metavar='N', help='Drop kmers with count < N (default 1)')
|
||||
ap.add_argument('-o', '--output', required=True,
|
||||
metavar='FILE', help='Output .npz path')
|
||||
args = ap.parse_args()
|
||||
|
||||
print(f'k={args.kmer_size} files={len(args.reads)}', file=sys.stderr)
|
||||
counts = count_kmers(args.reads, args.kmer_size)
|
||||
|
||||
if args.min_abundance > 1:
|
||||
before = len(counts)
|
||||
counts = {k: v for k, v in counts.items() if v >= args.min_abundance}
|
||||
print(f' min-abundance={args.min_abundance}: '
|
||||
f'{before - len(counts):,} kmers dropped, '
|
||||
f'{len(counts):,} retained',
|
||||
file=sys.stderr)
|
||||
|
||||
print(f'Sorting and saving → {args.output}', file=sys.stderr)
|
||||
kmers_arr = np.fromiter(sorted(counts), dtype=np.uint64, count=len(counts))
|
||||
counts_arr = np.array([counts[int(k)] for k in kmers_arr], dtype=np.uint32)
|
||||
|
||||
np.savez_compressed(args.output, kmers=kmers_arr, counts=counts_arr)
|
||||
print(f'Done {len(kmers_arr):,} kmers → {args.output}', file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Executable
+39
@@ -0,0 +1,39 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
SIMDATA_DIR="${SCRIPT_DIR}/simulated_data"
|
||||
REF_DIR="${SCRIPT_DIR}/reference_index"
|
||||
PYTHON="${SCRIPT_DIR}/../.venv/bin/python3"
|
||||
BUILD_PY="${SCRIPT_DIR}/build_reference.py"
|
||||
|
||||
KMER_SIZE="${KMER_SIZE:-31}"
|
||||
MIN_ABUNDANCE="${MIN_ABUNDANCE:-1}"
|
||||
|
||||
mkdir -p "${REF_DIR}"
|
||||
|
||||
for species_dir in "${SIMDATA_DIR}"/*/; do
|
||||
[[ -d "${species_dir}" ]] || continue
|
||||
species=$(basename "${species_dir}")
|
||||
|
||||
for strain_dir in "${species_dir}"*/; do
|
||||
[[ -d "${strain_dir}" ]] || continue
|
||||
strain=$(basename "${strain_dir}")
|
||||
|
||||
r1="${strain_dir}/reads_R1.fastq.gz"
|
||||
r2="${strain_dir}/reads_R2.fastq.gz"
|
||||
if [[ ! -f "${r1}" || ! -f "${r2}" ]]; then
|
||||
echo "SKIP ${species}--${strain}: reads not found" >&2
|
||||
continue
|
||||
fi
|
||||
|
||||
out="${REF_DIR}/${species}--${strain}.npz"
|
||||
echo "[${species}--${strain}] → ${out}"
|
||||
|
||||
"${PYTHON}" "${BUILD_PY}" \
|
||||
--kmer-size "${KMER_SIZE}" \
|
||||
--min-abundance "${MIN_ABUNDANCE}" \
|
||||
--output "${out}" \
|
||||
"${r1}" "${r2}"
|
||||
done
|
||||
done
|
||||
Executable
+226
@@ -0,0 +1,226 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Compute reference pairwise distance matrices from per-specimen .npz kmer indexes.
|
||||
|
||||
Reads all .npz files in reference_index/ (each containing sorted uint64 `kmers`
|
||||
and uint32 `counts`), computes all distance metrics supported by `obikmer distance`,
|
||||
and writes one CSV per metric to reference_dist/.
|
||||
|
||||
Output CSV format matches `obikmer distance --output`:
|
||||
- first row: "genome", then specimen names
|
||||
- subsequent rows: specimen name, then float or int values
|
||||
|
||||
Metrics written
|
||||
jaccard_dist.csv Jaccard distance (presence/absence)
|
||||
shared_kmers.csv Shared-kmer count matrix (intersection size)
|
||||
bray_curtis_dist.csv Bray-Curtis dissimilarity (raw counts)
|
||||
relfreq_bray_curtis_dist.csv Bray-Curtis on relative frequencies
|
||||
euclidean_dist.csv Euclidean distance (raw counts)
|
||||
relfreq_euclidean_dist.csv Euclidean distance on relative frequencies
|
||||
hellinger_dist.csv Hellinger distance
|
||||
hellinger_euclidean_dist.csv Euclidean distance in Hellinger space
|
||||
"""
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
# ── pairwise helpers ──────────────────────────────────────────────────────────
|
||||
|
||||
def shared_indices(a_kmers: np.ndarray, b_kmers: np.ndarray):
|
||||
"""Return index arrays (idx_a, idx_b) for kmers present in both sets.
|
||||
|
||||
Both arrays must be sorted uint64. Uses searchsorted: O(|B| log |A|).
|
||||
"""
|
||||
pos = np.searchsorted(a_kmers, b_kmers)
|
||||
pos = np.clip(pos, 0, len(a_kmers) - 1)
|
||||
mask = a_kmers[pos] == b_kmers
|
||||
idx_b = np.where(mask)[0]
|
||||
idx_a = pos[idx_b]
|
||||
return idx_a, idx_b
|
||||
|
||||
|
||||
def pairwise_stats(specimens: list[dict]) -> dict[str, np.ndarray]:
|
||||
"""Compute all pairwise distance matrices at once.
|
||||
|
||||
Returns a dict metric_name → ndarray (n×n float64 or int64).
|
||||
Each specimen dict has keys: name, kmers, counts.
|
||||
"""
|
||||
n = len(specimens)
|
||||
|
||||
# Pre-compute per-specimen scalars
|
||||
kmer_counts = np.array([len(s['kmers']) for s in specimens], dtype=np.uint64)
|
||||
count_sums = np.array([s['counts'].sum() for s in specimens], dtype=np.uint64)
|
||||
|
||||
# Per-specimen sum-of-squares (for Euclidean decomposition)
|
||||
sq_sums = np.array([(s['counts'].astype(np.float64) ** 2).sum() for s in specimens])
|
||||
|
||||
# Allocate output matrices
|
||||
shared_mat = np.zeros((n, n), dtype=np.uint64)
|
||||
hamming_mat = np.zeros((n, n), dtype=np.float64)
|
||||
jaccard_mat = np.zeros((n, n), dtype=np.float64)
|
||||
bray_mat = np.zeros((n, n), dtype=np.float64)
|
||||
relfreq_bray = np.zeros((n, n), dtype=np.float64)
|
||||
euclidean_mat = np.zeros((n, n), dtype=np.float64)
|
||||
relfreq_eucl = np.zeros((n, n), dtype=np.float64)
|
||||
hellinger_mat = np.zeros((n, n), dtype=np.float64)
|
||||
hell_eucl_mat = np.zeros((n, n), dtype=np.float64)
|
||||
|
||||
for i in range(n):
|
||||
a_km = specimens[i]['kmers']
|
||||
a_ct = specimens[i]['counts'].astype(np.float64)
|
||||
sa = float(count_sums[i])
|
||||
na = int(kmer_counts[i])
|
||||
|
||||
for j in range(i + 1, n):
|
||||
b_km = specimens[j]['kmers']
|
||||
b_ct = specimens[j]['counts'].astype(np.float64)
|
||||
sb = float(count_sums[j])
|
||||
nb = int(kmer_counts[j])
|
||||
|
||||
idx_a, idx_b = shared_indices(a_km, b_km)
|
||||
inter = len(idx_a)
|
||||
|
||||
ca_sh = a_ct[idx_a]
|
||||
cb_sh = b_ct[idx_b]
|
||||
|
||||
# ── Presence metrics ──────────────────────────────────────────────
|
||||
|
||||
union = na + nb - inter
|
||||
jac = (1.0 - inter / union) if union else 0.0
|
||||
hamming = float(na + nb - 2 * inter) # |A Δ B|
|
||||
|
||||
# ── Count metrics ─────────────────────────────────────────────────
|
||||
|
||||
# Bray-Curtis: 1 - 2*Σmin(a,b) / (Σa + Σb)
|
||||
sum_min = np.minimum(ca_sh, cb_sh).sum()
|
||||
denom_bc = sa + sb
|
||||
bc = (1.0 - 2.0 * sum_min / denom_bc) if denom_bc else 0.0
|
||||
|
||||
# RelfreqBray: 1 - Σmin(a/sa, b/sb) [only shared contribute]
|
||||
if sa and sb:
|
||||
rfb = 1.0 - np.minimum(ca_sh / sa, cb_sh / sb).sum()
|
||||
else:
|
||||
rfb = 0.0
|
||||
|
||||
# Euclidean: √(Σa² + Σb² - 2·Σ(a·b)_shared)
|
||||
cross = (ca_sh * cb_sh).sum()
|
||||
eucl_partial = sq_sums[i] + sq_sums[j] - 2.0 * cross
|
||||
eucl = np.sqrt(max(eucl_partial, 0.0))
|
||||
|
||||
# RelfreqEuclidean: √(Σ(a/sa - b/sb)²)
|
||||
# = √(Σa²/sa² + Σb²/sb² - 2·Σ(a·b)_shared/(sa·sb))
|
||||
if sa and sb:
|
||||
rf_cross = (ca_sh / sa * (cb_sh / sb)).sum()
|
||||
rfe_partial = (sq_sums[i] / sa**2
|
||||
+ sq_sums[j] / sb**2
|
||||
- 2.0 * rf_cross)
|
||||
rfe = np.sqrt(max(rfe_partial, 0.0))
|
||||
else:
|
||||
rfe = 0.0
|
||||
|
||||
# Hellinger partial: Σ(√(a/sa) - √(b/sb))² over global universe
|
||||
# = 2 - 2·Σ√(a·b)_shared / √(sa·sb)
|
||||
if sa and sb:
|
||||
bc_coeff = np.sqrt(ca_sh * cb_sh).sum() / np.sqrt(sa * sb)
|
||||
hell_partial = max(2.0 - 2.0 * bc_coeff, 0.0)
|
||||
else:
|
||||
hell_partial = 0.0
|
||||
|
||||
sq2 = np.sqrt(2.0)
|
||||
hell = np.sqrt(hell_partial) / sq2
|
||||
hell_euc = np.sqrt(hell_partial)
|
||||
|
||||
# ── Fill symmetric matrices ───────────────────────────────────────
|
||||
for mat, val in [
|
||||
(shared_mat, inter),
|
||||
(hamming_mat, hamming),
|
||||
(jaccard_mat, jac),
|
||||
(bray_mat, bc),
|
||||
(relfreq_bray, rfb),
|
||||
(euclidean_mat, eucl),
|
||||
(relfreq_eucl, rfe),
|
||||
(hellinger_mat, hell),
|
||||
(hell_eucl_mat, hell_euc),
|
||||
]:
|
||||
mat[i, j] = val
|
||||
mat[j, i] = val
|
||||
|
||||
return {
|
||||
'shared_kmers': shared_mat,
|
||||
'hamming_dist': hamming_mat,
|
||||
'jaccard_dist': jaccard_mat,
|
||||
'bray_curtis_dist': bray_mat,
|
||||
'relfreq_bray_curtis_dist': relfreq_bray,
|
||||
'euclidean_dist': euclidean_mat,
|
||||
'relfreq_euclidean_dist': relfreq_eucl,
|
||||
'hellinger_dist': hellinger_mat,
|
||||
'hellinger_euclidean_dist': hell_eucl_mat,
|
||||
}
|
||||
|
||||
|
||||
# ── I/O ───────────────────────────────────────────────────────────────────────
|
||||
|
||||
def write_csv(path: Path, labels: list[str], mat: np.ndarray, fmt: str) -> None:
|
||||
with path.open('w') as fh:
|
||||
fh.write('genome,' + ','.join(labels) + '\n')
|
||||
for i, label in enumerate(labels):
|
||||
row = ','.join(format(mat[i, j], fmt) for j in range(len(labels)))
|
||||
fh.write(f'{label},{row}\n')
|
||||
print(f' → {path}', file=sys.stderr)
|
||||
|
||||
|
||||
# ── main ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('--ref-dir', default='reference_index',
|
||||
help='Directory with per-specimen .npz files (default: reference_index)')
|
||||
ap.add_argument('--out-dir', default='reference_dist',
|
||||
help='Output directory for CSV files (default: reference_dist)')
|
||||
args = ap.parse_args()
|
||||
|
||||
ref_dir = Path(args.ref_dir)
|
||||
out_dir = Path(args.out_dir)
|
||||
out_dir.mkdir(exist_ok=True)
|
||||
|
||||
npz_files = sorted(ref_dir.glob('*.npz'))
|
||||
if not npz_files:
|
||||
print(f'ERROR: no .npz files found in {ref_dir}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
print(f'Loading {len(npz_files)} specimen(s) from {ref_dir}/', file=sys.stderr)
|
||||
specimens = []
|
||||
for f in npz_files:
|
||||
data = np.load(f)
|
||||
specimens.append({
|
||||
'name': f.stem,
|
||||
'kmers': data['kmers'],
|
||||
'counts': data['counts'],
|
||||
})
|
||||
print(f' {f.stem}: {len(data["kmers"]):,} kmers', file=sys.stderr)
|
||||
|
||||
labels = [s['name'] for s in specimens]
|
||||
n = len(labels)
|
||||
print(f'\nComputing pairwise distances for {n} specimens…', file=sys.stderr)
|
||||
|
||||
matrices = pairwise_stats(specimens)
|
||||
|
||||
print(f'\nWriting CSVs to {out_dir}/', file=sys.stderr)
|
||||
write_csv(out_dir / 'shared_kmers.csv', labels, matrices['shared_kmers'], 'd')
|
||||
write_csv(out_dir / 'hamming_dist.csv', labels, matrices['hamming_dist'], '.6f')
|
||||
write_csv(out_dir / 'jaccard_dist.csv', labels, matrices['jaccard_dist'], '.6f')
|
||||
write_csv(out_dir / 'bray_curtis_dist.csv', labels, matrices['bray_curtis_dist'], '.6f')
|
||||
write_csv(out_dir / 'relfreq_bray_curtis_dist.csv', labels, matrices['relfreq_bray_curtis_dist'], '.6f')
|
||||
write_csv(out_dir / 'euclidean_dist.csv', labels, matrices['euclidean_dist'], '.6f')
|
||||
write_csv(out_dir / 'relfreq_euclidean_dist.csv', labels, matrices['relfreq_euclidean_dist'], '.6f')
|
||||
write_csv(out_dir / 'hellinger_dist.csv', labels, matrices['hellinger_dist'], '.6f')
|
||||
write_csv(out_dir / 'hellinger_euclidean_dist.csv', labels, matrices['hellinger_euclidean_dist'], '.6f')
|
||||
|
||||
print('\nDone.', file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Executable
+182
@@ -0,0 +1,182 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Compare all reference distance matrices against obikmer distance outputs.
|
||||
|
||||
Reads from:
|
||||
reference_dist/ — ground-truth matrices computed by build_reference_dist.py
|
||||
obikmer_dist/ — matrices produced by `obikmer distance`
|
||||
|
||||
Handles label reordering: both matrices are sorted by genome label before
|
||||
element-wise comparison, so column/row order differences are irrelevant.
|
||||
|
||||
Output: stats/dist_comparison/summary.csv
|
||||
comparison,max_abs,mean_abs,rmse,n_pairs,status
|
||||
"""
|
||||
import csv
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
# ── CSV loading ───────────────────────────────────────────────────────────────
|
||||
|
||||
def load_matrix(path: Path) -> tuple[list[str], np.ndarray]:
|
||||
"""Load a distance-matrix CSV; return (sorted_labels, matrix_float64)."""
|
||||
with path.open() as fh:
|
||||
reader = csv.reader(fh)
|
||||
header = next(reader)[1:] # skip 'genome' column
|
||||
raw: dict[str, list[float]] = {}
|
||||
for row in reader:
|
||||
raw[row[0]] = [float(x) for x in row[1:]]
|
||||
|
||||
label_to_col = {h: i for i, h in enumerate(header)}
|
||||
labels = sorted(raw.keys())
|
||||
n = len(labels)
|
||||
mat = np.zeros((n, n), dtype=np.float64)
|
||||
for i, ri in enumerate(labels):
|
||||
for j, cj in enumerate(labels):
|
||||
mat[i, j] = raw[ri][label_to_col[cj]]
|
||||
return labels, mat
|
||||
|
||||
|
||||
# ── comparison ────────────────────────────────────────────────────────────────
|
||||
|
||||
def compare(label: str,
|
||||
ref_path: Path,
|
||||
obi_path: Path,
|
||||
tol: float = 1e-4) -> dict:
|
||||
if not ref_path.exists():
|
||||
return {'comparison': label, 'status': 'REF_MISSING',
|
||||
'max_abs': '', 'mean_abs': '', 'rmse': '', 'n_pairs': ''}
|
||||
if not obi_path.exists():
|
||||
return {'comparison': label, 'status': 'OBI_MISSING',
|
||||
'max_abs': '', 'mean_abs': '', 'rmse': '', 'n_pairs': ''}
|
||||
|
||||
ref_labels, ref_mat = load_matrix(ref_path)
|
||||
obi_labels, obi_mat = load_matrix(obi_path)
|
||||
|
||||
if ref_labels != obi_labels:
|
||||
only_ref = sorted(set(ref_labels) - set(obi_labels))
|
||||
only_obi = sorted(set(obi_labels) - set(ref_labels))
|
||||
print(f' [{label}] label mismatch — '
|
||||
f'only_ref={only_ref} only_obi={only_obi}', file=sys.stderr)
|
||||
return {'comparison': label, 'status': 'LABEL_MISMATCH',
|
||||
'max_abs': '', 'mean_abs': '', 'rmse': '', 'n_pairs': ''}
|
||||
|
||||
n = len(ref_labels)
|
||||
# Off-diagonal mask
|
||||
mask = ~np.eye(n, dtype=bool)
|
||||
diff = np.abs(ref_mat[mask] - obi_mat[mask])
|
||||
n_pairs = diff.size
|
||||
|
||||
max_abs = float(diff.max())
|
||||
mean_abs = float(diff.mean())
|
||||
rmse = float(np.sqrt((diff ** 2).mean()))
|
||||
status = 'PASS' if max_abs <= tol else 'FAIL'
|
||||
|
||||
print(f' [{label}] n={n_pairs} '
|
||||
f'max={max_abs:.3e} mean={mean_abs:.3e} rmse={rmse:.3e} {status}',
|
||||
file=sys.stderr)
|
||||
return {
|
||||
'comparison': label,
|
||||
'max_abs': f'{max_abs:.6e}',
|
||||
'mean_abs': f'{mean_abs:.6e}',
|
||||
'rmse': f'{rmse:.6e}',
|
||||
'n_pairs': str(n_pairs),
|
||||
'status': status,
|
||||
}
|
||||
|
||||
|
||||
# ── comparison table ──────────────────────────────────────────────────────────
|
||||
|
||||
# (label, ref_csv, obikmer_csv)
|
||||
# The reference jaccard/shared is presence-based, which should match both
|
||||
# presence/jaccard and count/jaccard (threshold=1).
|
||||
COMPARISONS = [
|
||||
# ── presence index ────────────────────────────────────────────────────────
|
||||
('presence/jaccard_dist',
|
||||
'reference_dist/jaccard_dist.csv',
|
||||
'obikmer_dist/presence/jaccard_dist.csv'),
|
||||
|
||||
('presence/jaccard_shared',
|
||||
'reference_dist/shared_kmers.csv',
|
||||
'obikmer_dist/presence/jaccard_shared.csv'),
|
||||
|
||||
('presence/hamming_dist',
|
||||
'reference_dist/hamming_dist.csv',
|
||||
'obikmer_dist/presence/hamming_dist.csv'),
|
||||
|
||||
# ── count index (jaccard cross-check) ─────────────────────────────────────
|
||||
('count/jaccard_dist',
|
||||
'reference_dist/jaccard_dist.csv',
|
||||
'obikmer_dist/count/jaccard_dist.csv'),
|
||||
|
||||
('count/jaccard_shared',
|
||||
'reference_dist/shared_kmers.csv',
|
||||
'obikmer_dist/count/jaccard_shared.csv'),
|
||||
|
||||
# ── count index (count-based metrics) ────────────────────────────────────
|
||||
('count/bray_curtis_dist',
|
||||
'reference_dist/bray_curtis_dist.csv',
|
||||
'obikmer_dist/count/bray_curtis_dist.csv'),
|
||||
|
||||
('count/relfreq_bray_curtis_dist',
|
||||
'reference_dist/relfreq_bray_curtis_dist.csv',
|
||||
'obikmer_dist/count/relfreq_bray_curtis_dist.csv'),
|
||||
|
||||
('count/euclidean_dist',
|
||||
'reference_dist/euclidean_dist.csv',
|
||||
'obikmer_dist/count/euclidean_dist.csv'),
|
||||
|
||||
('count/relfreq_euclidean_dist',
|
||||
'reference_dist/relfreq_euclidean_dist.csv',
|
||||
'obikmer_dist/count/relfreq_euclidean_dist.csv'),
|
||||
|
||||
('count/hellinger_dist',
|
||||
'reference_dist/hellinger_dist.csv',
|
||||
'obikmer_dist/count/hellinger_dist.csv'),
|
||||
|
||||
('count/hellinger_euclidean_dist',
|
||||
'reference_dist/hellinger_euclidean_dist.csv',
|
||||
'obikmer_dist/count/hellinger_euclidean_dist.csv'),
|
||||
]
|
||||
|
||||
|
||||
# ── main ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
def main() -> None:
|
||||
import argparse
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('--tol', type=float, default=1e-4,
|
||||
help='Max abs diff threshold for PASS/FAIL (default 1e-4)')
|
||||
ap.add_argument('--out', default='stats/dist_comparison/summary.csv',
|
||||
help='Output summary CSV path')
|
||||
args = ap.parse_args()
|
||||
|
||||
out_path = Path(args.out)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
print(f'Comparing {len(COMPARISONS)} matrix pairs…', file=sys.stderr)
|
||||
rows = []
|
||||
for label, ref, obi in COMPARISONS:
|
||||
rows.append(compare(label, Path(ref), Path(obi), tol=args.tol))
|
||||
|
||||
fields = ['comparison', 'max_abs', 'mean_abs', 'rmse', 'n_pairs', 'status']
|
||||
with out_path.open('w', newline='') as fh:
|
||||
w = csv.DictWriter(fh, fieldnames=fields)
|
||||
w.writeheader()
|
||||
w.writerows(rows)
|
||||
|
||||
print(f'\n→ {out_path}', file=sys.stderr)
|
||||
|
||||
n_fail = sum(1 for r in rows if r.get('status') == 'FAIL')
|
||||
n_pass = sum(1 for r in rows if r.get('status') == 'PASS')
|
||||
print(f'Summary: {n_pass} PASS {n_fail} FAIL '
|
||||
f'{len(rows) - n_pass - n_fail} SKIP', file=sys.stderr)
|
||||
if n_fail:
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,199 @@
|
||||
SPECIMENS := Escherichia_coli--K-12_MG1655 Escherichia_coli--EDL933 Salmonella_enterica--LT2 Escherichia_coli--CFT073 Bacillus_subtilis--168 Salmonella_enterica--P125109 Shouchella_clausii--KSM-K16 Escherichia_coli--K-12_W3110 Klebsiella_pneumoniae--MGH_78578 Opitutus_terrae--PB90-1 Saccharolobus_islandicus--M.16.4 Acidobacterium_capsulatum--ATCC_51196 Salmonella_enterica--AKU_12601 Proteus_mirabilis--HI4320 Salmonella_enterica--CT18 Klebsiella_pneumoniae--HS11286 Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1 Klebsiella_pneumoniae--ATCC_13883 Yersinia_ruckeri--YRB Candidozyma_auris--GCF_003013715.1_ASM301371v2
|
||||
SPECIES := Escherichia_coli Salmonella_enterica Bacillus_subtilis Shouchella_clausii Klebsiella_pneumoniae Opitutus_terrae Saccharolobus_islandicus Acidobacterium_capsulatum Proteus_mirabilis Wolbachia_endosymbiont Yersinia_ruckeri Candidozyma_auris
|
||||
|
||||
# Escherichia_coli--K-12_MG1655
|
||||
simulated_data/Escherichia_coli/K-12_MG1655/reads_R1.fastq.gz: genomes/GCF_000005845.2_ASM584v2_genomic.fna.gz
|
||||
reference_index/Escherichia_coli--K-12_MG1655.npz: simulated_data/Escherichia_coli/K-12_MG1655/reads_R1.fastq.gz
|
||||
specimen_index_presence/Escherichia_coli--K-12_MG1655/index.done stats/indexing_presence/Escherichia_coli--K-12_MG1655.stats: simulated_data/Escherichia_coli/K-12_MG1655/reads_R1.fastq.gz
|
||||
specimen_index_count/Escherichia_coli--K-12_MG1655/index.done stats/indexing_count/Escherichia_coli--K-12_MG1655.stats: simulated_data/Escherichia_coli/K-12_MG1655/reads_R1.fastq.gz
|
||||
stats/verify_presence/Escherichia_coli--K-12_MG1655.stats: reference_index/Escherichia_coli--K-12_MG1655.npz specimen_index_presence/Escherichia_coli--K-12_MG1655/index.done
|
||||
stats/verify_count/Escherichia_coli--K-12_MG1655.stats: reference_index/Escherichia_coli--K-12_MG1655.npz specimen_index_count/Escherichia_coli--K-12_MG1655/index.done
|
||||
|
||||
# Escherichia_coli--EDL933
|
||||
simulated_data/Escherichia_coli/EDL933/reads_R1.fastq.gz: genomes/GCF_000006665.1_ASM666v1_genomic.fna.gz
|
||||
reference_index/Escherichia_coli--EDL933.npz: simulated_data/Escherichia_coli/EDL933/reads_R1.fastq.gz
|
||||
specimen_index_presence/Escherichia_coli--EDL933/index.done stats/indexing_presence/Escherichia_coli--EDL933.stats: simulated_data/Escherichia_coli/EDL933/reads_R1.fastq.gz
|
||||
specimen_index_count/Escherichia_coli--EDL933/index.done stats/indexing_count/Escherichia_coli--EDL933.stats: simulated_data/Escherichia_coli/EDL933/reads_R1.fastq.gz
|
||||
stats/verify_presence/Escherichia_coli--EDL933.stats: reference_index/Escherichia_coli--EDL933.npz specimen_index_presence/Escherichia_coli--EDL933/index.done
|
||||
stats/verify_count/Escherichia_coli--EDL933.stats: reference_index/Escherichia_coli--EDL933.npz specimen_index_count/Escherichia_coli--EDL933/index.done
|
||||
|
||||
# Salmonella_enterica--LT2
|
||||
simulated_data/Salmonella_enterica/LT2/reads_R1.fastq.gz: genomes/GCF_000006945.2_ASM694v2_genomic.fna.gz
|
||||
reference_index/Salmonella_enterica--LT2.npz: simulated_data/Salmonella_enterica/LT2/reads_R1.fastq.gz
|
||||
specimen_index_presence/Salmonella_enterica--LT2/index.done stats/indexing_presence/Salmonella_enterica--LT2.stats: simulated_data/Salmonella_enterica/LT2/reads_R1.fastq.gz
|
||||
specimen_index_count/Salmonella_enterica--LT2/index.done stats/indexing_count/Salmonella_enterica--LT2.stats: simulated_data/Salmonella_enterica/LT2/reads_R1.fastq.gz
|
||||
stats/verify_presence/Salmonella_enterica--LT2.stats: reference_index/Salmonella_enterica--LT2.npz specimen_index_presence/Salmonella_enterica--LT2/index.done
|
||||
stats/verify_count/Salmonella_enterica--LT2.stats: reference_index/Salmonella_enterica--LT2.npz specimen_index_count/Salmonella_enterica--LT2/index.done
|
||||
|
||||
# Escherichia_coli--CFT073
|
||||
simulated_data/Escherichia_coli/CFT073/reads_R1.fastq.gz: genomes/GCF_000007445.1_ASM744v1_genomic.fna.gz
|
||||
reference_index/Escherichia_coli--CFT073.npz: simulated_data/Escherichia_coli/CFT073/reads_R1.fastq.gz
|
||||
specimen_index_presence/Escherichia_coli--CFT073/index.done stats/indexing_presence/Escherichia_coli--CFT073.stats: simulated_data/Escherichia_coli/CFT073/reads_R1.fastq.gz
|
||||
specimen_index_count/Escherichia_coli--CFT073/index.done stats/indexing_count/Escherichia_coli--CFT073.stats: simulated_data/Escherichia_coli/CFT073/reads_R1.fastq.gz
|
||||
stats/verify_presence/Escherichia_coli--CFT073.stats: reference_index/Escherichia_coli--CFT073.npz specimen_index_presence/Escherichia_coli--CFT073/index.done
|
||||
stats/verify_count/Escherichia_coli--CFT073.stats: reference_index/Escherichia_coli--CFT073.npz specimen_index_count/Escherichia_coli--CFT073/index.done
|
||||
|
||||
# Bacillus_subtilis--168
|
||||
simulated_data/Bacillus_subtilis/168/reads_R1.fastq.gz: genomes/GCF_000009045.1_ASM904v1_genomic.fna.gz
|
||||
reference_index/Bacillus_subtilis--168.npz: simulated_data/Bacillus_subtilis/168/reads_R1.fastq.gz
|
||||
specimen_index_presence/Bacillus_subtilis--168/index.done stats/indexing_presence/Bacillus_subtilis--168.stats: simulated_data/Bacillus_subtilis/168/reads_R1.fastq.gz
|
||||
specimen_index_count/Bacillus_subtilis--168/index.done stats/indexing_count/Bacillus_subtilis--168.stats: simulated_data/Bacillus_subtilis/168/reads_R1.fastq.gz
|
||||
stats/verify_presence/Bacillus_subtilis--168.stats: reference_index/Bacillus_subtilis--168.npz specimen_index_presence/Bacillus_subtilis--168/index.done
|
||||
stats/verify_count/Bacillus_subtilis--168.stats: reference_index/Bacillus_subtilis--168.npz specimen_index_count/Bacillus_subtilis--168/index.done
|
||||
|
||||
# Salmonella_enterica--P125109
|
||||
simulated_data/Salmonella_enterica/P125109/reads_R1.fastq.gz: genomes/GCF_000009505.1_ASM950v1_genomic.fna.gz
|
||||
reference_index/Salmonella_enterica--P125109.npz: simulated_data/Salmonella_enterica/P125109/reads_R1.fastq.gz
|
||||
specimen_index_presence/Salmonella_enterica--P125109/index.done stats/indexing_presence/Salmonella_enterica--P125109.stats: simulated_data/Salmonella_enterica/P125109/reads_R1.fastq.gz
|
||||
specimen_index_count/Salmonella_enterica--P125109/index.done stats/indexing_count/Salmonella_enterica--P125109.stats: simulated_data/Salmonella_enterica/P125109/reads_R1.fastq.gz
|
||||
stats/verify_presence/Salmonella_enterica--P125109.stats: reference_index/Salmonella_enterica--P125109.npz specimen_index_presence/Salmonella_enterica--P125109/index.done
|
||||
stats/verify_count/Salmonella_enterica--P125109.stats: reference_index/Salmonella_enterica--P125109.npz specimen_index_count/Salmonella_enterica--P125109/index.done
|
||||
|
||||
# Shouchella_clausii--KSM-K16
|
||||
simulated_data/Shouchella_clausii/KSM-K16/reads_R1.fastq.gz: genomes/GCF_000009825.1_ASM982v1_genomic.fna.gz
|
||||
reference_index/Shouchella_clausii--KSM-K16.npz: simulated_data/Shouchella_clausii/KSM-K16/reads_R1.fastq.gz
|
||||
specimen_index_presence/Shouchella_clausii--KSM-K16/index.done stats/indexing_presence/Shouchella_clausii--KSM-K16.stats: simulated_data/Shouchella_clausii/KSM-K16/reads_R1.fastq.gz
|
||||
specimen_index_count/Shouchella_clausii--KSM-K16/index.done stats/indexing_count/Shouchella_clausii--KSM-K16.stats: simulated_data/Shouchella_clausii/KSM-K16/reads_R1.fastq.gz
|
||||
stats/verify_presence/Shouchella_clausii--KSM-K16.stats: reference_index/Shouchella_clausii--KSM-K16.npz specimen_index_presence/Shouchella_clausii--KSM-K16/index.done
|
||||
stats/verify_count/Shouchella_clausii--KSM-K16.stats: reference_index/Shouchella_clausii--KSM-K16.npz specimen_index_count/Shouchella_clausii--KSM-K16/index.done
|
||||
|
||||
# Escherichia_coli--K-12_W3110
|
||||
simulated_data/Escherichia_coli/K-12_W3110/reads_R1.fastq.gz: genomes/GCF_000010245.2_ASM1024v1_genomic.fna.gz
|
||||
reference_index/Escherichia_coli--K-12_W3110.npz: simulated_data/Escherichia_coli/K-12_W3110/reads_R1.fastq.gz
|
||||
specimen_index_presence/Escherichia_coli--K-12_W3110/index.done stats/indexing_presence/Escherichia_coli--K-12_W3110.stats: simulated_data/Escherichia_coli/K-12_W3110/reads_R1.fastq.gz
|
||||
specimen_index_count/Escherichia_coli--K-12_W3110/index.done stats/indexing_count/Escherichia_coli--K-12_W3110.stats: simulated_data/Escherichia_coli/K-12_W3110/reads_R1.fastq.gz
|
||||
stats/verify_presence/Escherichia_coli--K-12_W3110.stats: reference_index/Escherichia_coli--K-12_W3110.npz specimen_index_presence/Escherichia_coli--K-12_W3110/index.done
|
||||
stats/verify_count/Escherichia_coli--K-12_W3110.stats: reference_index/Escherichia_coli--K-12_W3110.npz specimen_index_count/Escherichia_coli--K-12_W3110/index.done
|
||||
|
||||
# Klebsiella_pneumoniae--MGH_78578
|
||||
simulated_data/Klebsiella_pneumoniae/MGH_78578/reads_R1.fastq.gz: genomes/GCF_000016305.1_ASM1630v1_genomic.fna.gz
|
||||
reference_index/Klebsiella_pneumoniae--MGH_78578.npz: simulated_data/Klebsiella_pneumoniae/MGH_78578/reads_R1.fastq.gz
|
||||
specimen_index_presence/Klebsiella_pneumoniae--MGH_78578/index.done stats/indexing_presence/Klebsiella_pneumoniae--MGH_78578.stats: simulated_data/Klebsiella_pneumoniae/MGH_78578/reads_R1.fastq.gz
|
||||
specimen_index_count/Klebsiella_pneumoniae--MGH_78578/index.done stats/indexing_count/Klebsiella_pneumoniae--MGH_78578.stats: simulated_data/Klebsiella_pneumoniae/MGH_78578/reads_R1.fastq.gz
|
||||
stats/verify_presence/Klebsiella_pneumoniae--MGH_78578.stats: reference_index/Klebsiella_pneumoniae--MGH_78578.npz specimen_index_presence/Klebsiella_pneumoniae--MGH_78578/index.done
|
||||
stats/verify_count/Klebsiella_pneumoniae--MGH_78578.stats: reference_index/Klebsiella_pneumoniae--MGH_78578.npz specimen_index_count/Klebsiella_pneumoniae--MGH_78578/index.done
|
||||
|
||||
# Opitutus_terrae--PB90-1
|
||||
simulated_data/Opitutus_terrae/PB90-1/reads_R1.fastq.gz: genomes/GCF_000019965.1_ASM1996v1_genomic.fna.gz
|
||||
reference_index/Opitutus_terrae--PB90-1.npz: simulated_data/Opitutus_terrae/PB90-1/reads_R1.fastq.gz
|
||||
specimen_index_presence/Opitutus_terrae--PB90-1/index.done stats/indexing_presence/Opitutus_terrae--PB90-1.stats: simulated_data/Opitutus_terrae/PB90-1/reads_R1.fastq.gz
|
||||
specimen_index_count/Opitutus_terrae--PB90-1/index.done stats/indexing_count/Opitutus_terrae--PB90-1.stats: simulated_data/Opitutus_terrae/PB90-1/reads_R1.fastq.gz
|
||||
stats/verify_presence/Opitutus_terrae--PB90-1.stats: reference_index/Opitutus_terrae--PB90-1.npz specimen_index_presence/Opitutus_terrae--PB90-1/index.done
|
||||
stats/verify_count/Opitutus_terrae--PB90-1.stats: reference_index/Opitutus_terrae--PB90-1.npz specimen_index_count/Opitutus_terrae--PB90-1/index.done
|
||||
|
||||
# Saccharolobus_islandicus--M.16.4
|
||||
simulated_data/Saccharolobus_islandicus/M.16.4/reads_R1.fastq.gz: genomes/GCF_000022445.1_ASM2244v1_genomic.fna.gz
|
||||
reference_index/Saccharolobus_islandicus--M.16.4.npz: simulated_data/Saccharolobus_islandicus/M.16.4/reads_R1.fastq.gz
|
||||
specimen_index_presence/Saccharolobus_islandicus--M.16.4/index.done stats/indexing_presence/Saccharolobus_islandicus--M.16.4.stats: simulated_data/Saccharolobus_islandicus/M.16.4/reads_R1.fastq.gz
|
||||
specimen_index_count/Saccharolobus_islandicus--M.16.4/index.done stats/indexing_count/Saccharolobus_islandicus--M.16.4.stats: simulated_data/Saccharolobus_islandicus/M.16.4/reads_R1.fastq.gz
|
||||
stats/verify_presence/Saccharolobus_islandicus--M.16.4.stats: reference_index/Saccharolobus_islandicus--M.16.4.npz specimen_index_presence/Saccharolobus_islandicus--M.16.4/index.done
|
||||
stats/verify_count/Saccharolobus_islandicus--M.16.4.stats: reference_index/Saccharolobus_islandicus--M.16.4.npz specimen_index_count/Saccharolobus_islandicus--M.16.4/index.done
|
||||
|
||||
# Acidobacterium_capsulatum--ATCC_51196
|
||||
simulated_data/Acidobacterium_capsulatum/ATCC_51196/reads_R1.fastq.gz: genomes/GCF_000022565.1_ASM2256v1_genomic.fna.gz
|
||||
reference_index/Acidobacterium_capsulatum--ATCC_51196.npz: simulated_data/Acidobacterium_capsulatum/ATCC_51196/reads_R1.fastq.gz
|
||||
specimen_index_presence/Acidobacterium_capsulatum--ATCC_51196/index.done stats/indexing_presence/Acidobacterium_capsulatum--ATCC_51196.stats: simulated_data/Acidobacterium_capsulatum/ATCC_51196/reads_R1.fastq.gz
|
||||
specimen_index_count/Acidobacterium_capsulatum--ATCC_51196/index.done stats/indexing_count/Acidobacterium_capsulatum--ATCC_51196.stats: simulated_data/Acidobacterium_capsulatum/ATCC_51196/reads_R1.fastq.gz
|
||||
stats/verify_presence/Acidobacterium_capsulatum--ATCC_51196.stats: reference_index/Acidobacterium_capsulatum--ATCC_51196.npz specimen_index_presence/Acidobacterium_capsulatum--ATCC_51196/index.done
|
||||
stats/verify_count/Acidobacterium_capsulatum--ATCC_51196.stats: reference_index/Acidobacterium_capsulatum--ATCC_51196.npz specimen_index_count/Acidobacterium_capsulatum--ATCC_51196/index.done
|
||||
|
||||
# Salmonella_enterica--AKU_12601
|
||||
simulated_data/Salmonella_enterica/AKU_12601/reads_R1.fastq.gz: genomes/GCF_000026565.1_ASM2656v1_genomic.fna.gz
|
||||
reference_index/Salmonella_enterica--AKU_12601.npz: simulated_data/Salmonella_enterica/AKU_12601/reads_R1.fastq.gz
|
||||
specimen_index_presence/Salmonella_enterica--AKU_12601/index.done stats/indexing_presence/Salmonella_enterica--AKU_12601.stats: simulated_data/Salmonella_enterica/AKU_12601/reads_R1.fastq.gz
|
||||
specimen_index_count/Salmonella_enterica--AKU_12601/index.done stats/indexing_count/Salmonella_enterica--AKU_12601.stats: simulated_data/Salmonella_enterica/AKU_12601/reads_R1.fastq.gz
|
||||
stats/verify_presence/Salmonella_enterica--AKU_12601.stats: reference_index/Salmonella_enterica--AKU_12601.npz specimen_index_presence/Salmonella_enterica--AKU_12601/index.done
|
||||
stats/verify_count/Salmonella_enterica--AKU_12601.stats: reference_index/Salmonella_enterica--AKU_12601.npz specimen_index_count/Salmonella_enterica--AKU_12601/index.done
|
||||
|
||||
# Proteus_mirabilis--HI4320
|
||||
simulated_data/Proteus_mirabilis/HI4320/reads_R1.fastq.gz: genomes/GCF_000069965.1_ASM6996v1_genomic.fna.gz
|
||||
reference_index/Proteus_mirabilis--HI4320.npz: simulated_data/Proteus_mirabilis/HI4320/reads_R1.fastq.gz
|
||||
specimen_index_presence/Proteus_mirabilis--HI4320/index.done stats/indexing_presence/Proteus_mirabilis--HI4320.stats: simulated_data/Proteus_mirabilis/HI4320/reads_R1.fastq.gz
|
||||
specimen_index_count/Proteus_mirabilis--HI4320/index.done stats/indexing_count/Proteus_mirabilis--HI4320.stats: simulated_data/Proteus_mirabilis/HI4320/reads_R1.fastq.gz
|
||||
stats/verify_presence/Proteus_mirabilis--HI4320.stats: reference_index/Proteus_mirabilis--HI4320.npz specimen_index_presence/Proteus_mirabilis--HI4320/index.done
|
||||
stats/verify_count/Proteus_mirabilis--HI4320.stats: reference_index/Proteus_mirabilis--HI4320.npz specimen_index_count/Proteus_mirabilis--HI4320/index.done
|
||||
|
||||
# Salmonella_enterica--CT18
|
||||
simulated_data/Salmonella_enterica/CT18/reads_R1.fastq.gz: genomes/GCF_000195995.1_ASM19599v1_genomic.fna.gz
|
||||
reference_index/Salmonella_enterica--CT18.npz: simulated_data/Salmonella_enterica/CT18/reads_R1.fastq.gz
|
||||
specimen_index_presence/Salmonella_enterica--CT18/index.done stats/indexing_presence/Salmonella_enterica--CT18.stats: simulated_data/Salmonella_enterica/CT18/reads_R1.fastq.gz
|
||||
specimen_index_count/Salmonella_enterica--CT18/index.done stats/indexing_count/Salmonella_enterica--CT18.stats: simulated_data/Salmonella_enterica/CT18/reads_R1.fastq.gz
|
||||
stats/verify_presence/Salmonella_enterica--CT18.stats: reference_index/Salmonella_enterica--CT18.npz specimen_index_presence/Salmonella_enterica--CT18/index.done
|
||||
stats/verify_count/Salmonella_enterica--CT18.stats: reference_index/Salmonella_enterica--CT18.npz specimen_index_count/Salmonella_enterica--CT18/index.done
|
||||
|
||||
# Klebsiella_pneumoniae--HS11286
|
||||
simulated_data/Klebsiella_pneumoniae/HS11286/reads_R1.fastq.gz: genomes/GCF_000240185.1_ASM24018v2_genomic.fna.gz
|
||||
reference_index/Klebsiella_pneumoniae--HS11286.npz: simulated_data/Klebsiella_pneumoniae/HS11286/reads_R1.fastq.gz
|
||||
specimen_index_presence/Klebsiella_pneumoniae--HS11286/index.done stats/indexing_presence/Klebsiella_pneumoniae--HS11286.stats: simulated_data/Klebsiella_pneumoniae/HS11286/reads_R1.fastq.gz
|
||||
specimen_index_count/Klebsiella_pneumoniae--HS11286/index.done stats/indexing_count/Klebsiella_pneumoniae--HS11286.stats: simulated_data/Klebsiella_pneumoniae/HS11286/reads_R1.fastq.gz
|
||||
stats/verify_presence/Klebsiella_pneumoniae--HS11286.stats: reference_index/Klebsiella_pneumoniae--HS11286.npz specimen_index_presence/Klebsiella_pneumoniae--HS11286/index.done
|
||||
stats/verify_count/Klebsiella_pneumoniae--HS11286.stats: reference_index/Klebsiella_pneumoniae--HS11286.npz specimen_index_count/Klebsiella_pneumoniae--HS11286/index.done
|
||||
|
||||
# Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1
|
||||
simulated_data/Wolbachia_endosymbiont/GCF_000306885.1_ASM30688v1/reads_R1.fastq.gz: genomes/GCF_000306885.1_ASM30688v1_genomic.fna.gz
|
||||
reference_index/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1.npz: simulated_data/Wolbachia_endosymbiont/GCF_000306885.1_ASM30688v1/reads_R1.fastq.gz
|
||||
specimen_index_presence/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1/index.done stats/indexing_presence/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1.stats: simulated_data/Wolbachia_endosymbiont/GCF_000306885.1_ASM30688v1/reads_R1.fastq.gz
|
||||
specimen_index_count/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1/index.done stats/indexing_count/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1.stats: simulated_data/Wolbachia_endosymbiont/GCF_000306885.1_ASM30688v1/reads_R1.fastq.gz
|
||||
stats/verify_presence/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1.stats: reference_index/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1.npz specimen_index_presence/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1/index.done
|
||||
stats/verify_count/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1.stats: reference_index/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1.npz specimen_index_count/Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1/index.done
|
||||
|
||||
# Klebsiella_pneumoniae--ATCC_13883
|
||||
simulated_data/Klebsiella_pneumoniae/ATCC_13883/reads_R1.fastq.gz: genomes/GCF_000742135.1_ASM74213v1_genomic.fna.gz
|
||||
reference_index/Klebsiella_pneumoniae--ATCC_13883.npz: simulated_data/Klebsiella_pneumoniae/ATCC_13883/reads_R1.fastq.gz
|
||||
specimen_index_presence/Klebsiella_pneumoniae--ATCC_13883/index.done stats/indexing_presence/Klebsiella_pneumoniae--ATCC_13883.stats: simulated_data/Klebsiella_pneumoniae/ATCC_13883/reads_R1.fastq.gz
|
||||
specimen_index_count/Klebsiella_pneumoniae--ATCC_13883/index.done stats/indexing_count/Klebsiella_pneumoniae--ATCC_13883.stats: simulated_data/Klebsiella_pneumoniae/ATCC_13883/reads_R1.fastq.gz
|
||||
stats/verify_presence/Klebsiella_pneumoniae--ATCC_13883.stats: reference_index/Klebsiella_pneumoniae--ATCC_13883.npz specimen_index_presence/Klebsiella_pneumoniae--ATCC_13883/index.done
|
||||
stats/verify_count/Klebsiella_pneumoniae--ATCC_13883.stats: reference_index/Klebsiella_pneumoniae--ATCC_13883.npz specimen_index_count/Klebsiella_pneumoniae--ATCC_13883/index.done
|
||||
|
||||
# Yersinia_ruckeri--YRB
|
||||
simulated_data/Yersinia_ruckeri/YRB/reads_R1.fastq.gz: genomes/GCF_000834255.1_ASM83425v1_genomic.fna.gz
|
||||
reference_index/Yersinia_ruckeri--YRB.npz: simulated_data/Yersinia_ruckeri/YRB/reads_R1.fastq.gz
|
||||
specimen_index_presence/Yersinia_ruckeri--YRB/index.done stats/indexing_presence/Yersinia_ruckeri--YRB.stats: simulated_data/Yersinia_ruckeri/YRB/reads_R1.fastq.gz
|
||||
specimen_index_count/Yersinia_ruckeri--YRB/index.done stats/indexing_count/Yersinia_ruckeri--YRB.stats: simulated_data/Yersinia_ruckeri/YRB/reads_R1.fastq.gz
|
||||
stats/verify_presence/Yersinia_ruckeri--YRB.stats: reference_index/Yersinia_ruckeri--YRB.npz specimen_index_presence/Yersinia_ruckeri--YRB/index.done
|
||||
stats/verify_count/Yersinia_ruckeri--YRB.stats: reference_index/Yersinia_ruckeri--YRB.npz specimen_index_count/Yersinia_ruckeri--YRB/index.done
|
||||
|
||||
# Candidozyma_auris--GCF_003013715.1_ASM301371v2
|
||||
simulated_data/Candidozyma_auris/GCF_003013715.1_ASM301371v2/reads_R1.fastq.gz: genomes/GCF_003013715.1_ASM301371v2_genomic.fna.gz
|
||||
reference_index/Candidozyma_auris--GCF_003013715.1_ASM301371v2.npz: simulated_data/Candidozyma_auris/GCF_003013715.1_ASM301371v2/reads_R1.fastq.gz
|
||||
specimen_index_presence/Candidozyma_auris--GCF_003013715.1_ASM301371v2/index.done stats/indexing_presence/Candidozyma_auris--GCF_003013715.1_ASM301371v2.stats: simulated_data/Candidozyma_auris/GCF_003013715.1_ASM301371v2/reads_R1.fastq.gz
|
||||
specimen_index_count/Candidozyma_auris--GCF_003013715.1_ASM301371v2/index.done stats/indexing_count/Candidozyma_auris--GCF_003013715.1_ASM301371v2.stats: simulated_data/Candidozyma_auris/GCF_003013715.1_ASM301371v2/reads_R1.fastq.gz
|
||||
stats/verify_presence/Candidozyma_auris--GCF_003013715.1_ASM301371v2.stats: reference_index/Candidozyma_auris--GCF_003013715.1_ASM301371v2.npz specimen_index_presence/Candidozyma_auris--GCF_003013715.1_ASM301371v2/index.done
|
||||
stats/verify_count/Candidozyma_auris--GCF_003013715.1_ASM301371v2.stats: reference_index/Candidozyma_auris--GCF_003013715.1_ASM301371v2.npz specimen_index_count/Candidozyma_auris--GCF_003013715.1_ASM301371v2/index.done
|
||||
|
||||
# Escherichia_coli
|
||||
specific_index_presence/Escherichia_coli/index.done stats/specific_kmer_presence/Escherichia_coli.stats: global_index_presence/index.done
|
||||
specific_index_count/Escherichia_coli/index.done stats/specific_kmer_count/Escherichia_coli.stats: global_index_count/index.done
|
||||
# Salmonella_enterica
|
||||
specific_index_presence/Salmonella_enterica/index.done stats/specific_kmer_presence/Salmonella_enterica.stats: global_index_presence/index.done
|
||||
specific_index_count/Salmonella_enterica/index.done stats/specific_kmer_count/Salmonella_enterica.stats: global_index_count/index.done
|
||||
# Bacillus_subtilis
|
||||
specific_index_presence/Bacillus_subtilis/index.done stats/specific_kmer_presence/Bacillus_subtilis.stats: global_index_presence/index.done
|
||||
specific_index_count/Bacillus_subtilis/index.done stats/specific_kmer_count/Bacillus_subtilis.stats: global_index_count/index.done
|
||||
# Shouchella_clausii
|
||||
specific_index_presence/Shouchella_clausii/index.done stats/specific_kmer_presence/Shouchella_clausii.stats: global_index_presence/index.done
|
||||
specific_index_count/Shouchella_clausii/index.done stats/specific_kmer_count/Shouchella_clausii.stats: global_index_count/index.done
|
||||
# Klebsiella_pneumoniae
|
||||
specific_index_presence/Klebsiella_pneumoniae/index.done stats/specific_kmer_presence/Klebsiella_pneumoniae.stats: global_index_presence/index.done
|
||||
specific_index_count/Klebsiella_pneumoniae/index.done stats/specific_kmer_count/Klebsiella_pneumoniae.stats: global_index_count/index.done
|
||||
# Opitutus_terrae
|
||||
specific_index_presence/Opitutus_terrae/index.done stats/specific_kmer_presence/Opitutus_terrae.stats: global_index_presence/index.done
|
||||
specific_index_count/Opitutus_terrae/index.done stats/specific_kmer_count/Opitutus_terrae.stats: global_index_count/index.done
|
||||
# Saccharolobus_islandicus
|
||||
specific_index_presence/Saccharolobus_islandicus/index.done stats/specific_kmer_presence/Saccharolobus_islandicus.stats: global_index_presence/index.done
|
||||
specific_index_count/Saccharolobus_islandicus/index.done stats/specific_kmer_count/Saccharolobus_islandicus.stats: global_index_count/index.done
|
||||
# Acidobacterium_capsulatum
|
||||
specific_index_presence/Acidobacterium_capsulatum/index.done stats/specific_kmer_presence/Acidobacterium_capsulatum.stats: global_index_presence/index.done
|
||||
specific_index_count/Acidobacterium_capsulatum/index.done stats/specific_kmer_count/Acidobacterium_capsulatum.stats: global_index_count/index.done
|
||||
# Proteus_mirabilis
|
||||
specific_index_presence/Proteus_mirabilis/index.done stats/specific_kmer_presence/Proteus_mirabilis.stats: global_index_presence/index.done
|
||||
specific_index_count/Proteus_mirabilis/index.done stats/specific_kmer_count/Proteus_mirabilis.stats: global_index_count/index.done
|
||||
# Wolbachia_endosymbiont
|
||||
specific_index_presence/Wolbachia_endosymbiont/index.done stats/specific_kmer_presence/Wolbachia_endosymbiont.stats: global_index_presence/index.done
|
||||
specific_index_count/Wolbachia_endosymbiont/index.done stats/specific_kmer_count/Wolbachia_endosymbiont.stats: global_index_count/index.done
|
||||
# Yersinia_ruckeri
|
||||
specific_index_presence/Yersinia_ruckeri/index.done stats/specific_kmer_presence/Yersinia_ruckeri.stats: global_index_presence/index.done
|
||||
specific_index_count/Yersinia_ruckeri/index.done stats/specific_kmer_count/Yersinia_ruckeri.stats: global_index_count/index.done
|
||||
# Candidozyma_auris
|
||||
specific_index_presence/Candidozyma_auris/index.done stats/specific_kmer_presence/Candidozyma_auris.stats: global_index_presence/index.done
|
||||
specific_index_count/Candidozyma_auris/index.done stats/specific_kmer_count/Candidozyma_auris.stats: global_index_count/index.done
|
||||
Executable
+48
@@ -0,0 +1,48 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
assemblies=(
|
||||
GCF_000005845.2
|
||||
GCF_000010245.2
|
||||
GCF_000007445.1
|
||||
GCF_000006665.1
|
||||
|
||||
GCF_000006945.2
|
||||
GCF_000195995.1
|
||||
GCF_000009505.1
|
||||
GCF_000026565.1
|
||||
|
||||
GCF_000016305.1
|
||||
GCF_000019965.1
|
||||
GCF_000240185.1
|
||||
GCF_000742135.1
|
||||
|
||||
GCF_000069965.1
|
||||
GCF_000022565.1
|
||||
GCF_000306885.1
|
||||
GCF_003013715.1
|
||||
|
||||
GCF_000009045.1
|
||||
GCF_000009825.1
|
||||
GCF_000022445.1
|
||||
GCF_000834255.1
|
||||
)
|
||||
|
||||
mkdir -p genomes
|
||||
|
||||
for acc in "${assemblies[@]}"; do
|
||||
echo "Downloading ${acc}"
|
||||
|
||||
datasets download genome accession "${acc}" \
|
||||
--include genome \
|
||||
--filename "${acc}.zip"
|
||||
|
||||
unzip -q "${acc}.zip" -d "${acc}"
|
||||
find "${acc}" -name "*.fna" |
|
||||
while read file; do
|
||||
obiconvert -Z ${file} >genomes/$(basename ${file}).gz
|
||||
done
|
||||
|
||||
rm -rf "${acc}" "${acc}.zip"
|
||||
done
|
||||
Executable
+108
@@ -0,0 +1,108 @@
|
||||
#!/usr/bin/env bash
|
||||
# Usage: filter_one_count.sh SPECIES
|
||||
# Filters global_index_count to keep only kmers specific to SPECIES,
|
||||
# then selects the SPECIES column in-place.
|
||||
# Outputs:
|
||||
# specific_index_count/SPECIES/index.done (written by obikmer select)
|
||||
# stats/specific_kmer_count/SPECIES.stats (one CSV data row, no header)
|
||||
set -euo pipefail
|
||||
|
||||
SPECIES="$1"
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
|
||||
SOURCE="${SCRIPT_DIR}/global_index_count"
|
||||
OUTPUT="${SCRIPT_DIR}/specific_index_count/${SPECIES}"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/specific_kmer_count"
|
||||
STATS_FILE="${STATS_DIR}/${SPECIES}.stats"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
echo "[${SPECIES}] filter (count) → ${OUTPUT}"
|
||||
|
||||
LOG_FILTER=$(mktemp)
|
||||
LOG_SELECT=$(mktemp)
|
||||
trap 'rm -f "${LOG_FILTER}" "${LOG_SELECT}"' EXIT
|
||||
|
||||
"${BINARY}" filter \
|
||||
--output "${OUTPUT}" \
|
||||
--force \
|
||||
--ingroup "species=${SPECIES}" \
|
||||
--outgroup all \
|
||||
--min-frac 0.5 \
|
||||
--max-frac 1.0 \
|
||||
--max-outgroup-count 0 \
|
||||
"${SOURCE}" \
|
||||
2>"${LOG_FILTER}"
|
||||
|
||||
cat "${LOG_FILTER}" >&2
|
||||
|
||||
"${BINARY}" select \
|
||||
--in-place \
|
||||
--group "${SPECIES}:species=${SPECIES}" \
|
||||
--group-op "${SPECIES}:any" \
|
||||
--select "${SPECIES}" \
|
||||
"${OUTPUT}" \
|
||||
2>"${LOG_SELECT}"
|
||||
|
||||
cat "${LOG_SELECT}" >&2
|
||||
|
||||
python3 - "${SPECIES}" "${LOG_FILTER}" "${LOG_SELECT}" <<'PYEOF' >"${STATS_FILE}"
|
||||
import sys, re
|
||||
|
||||
species, log_filter, log_select = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
|
||||
def strip_ansi(s):
|
||||
return re.sub(r'\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]', '', s)
|
||||
|
||||
def parse_wall(s):
|
||||
s = s.strip()
|
||||
if s.endswith('ms'): return float(s[:-2]) / 1000.0
|
||||
if s.endswith('s'): return float(s[:-1])
|
||||
return 0.0
|
||||
|
||||
def parse_rss(s):
|
||||
m = re.match(r'([\d.]+)\s*(GB|MB|KB|B)', s.strip())
|
||||
if not m: return 0
|
||||
return int(float(m.group(1)) * {'GB': 1<<30, 'MB': 1<<20, 'KB': 1024, 'B': 1}[m.group(2)])
|
||||
|
||||
def is_sep(s):
|
||||
return bool(s) and not re.search(r'[A-Za-z0-9]', s)
|
||||
|
||||
def parse_reporter(logfile):
|
||||
stats = {}
|
||||
state = 'scan'
|
||||
with open(logfile, errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = strip_ansi(raw.rstrip('\n'))
|
||||
s = line.strip()
|
||||
if state == 'scan':
|
||||
if re.search(r'\bstage\b.*\bwall\b', line):
|
||||
state = 'in_header'
|
||||
elif state == 'in_header':
|
||||
if is_sep(s): state = 'rows'
|
||||
elif state == 'rows':
|
||||
if is_sep(s): state = 'total'
|
||||
elif s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 4:
|
||||
stats[parts[0]] = (parse_wall(parts[1]), parse_rss(parts[3]))
|
||||
elif state == 'total':
|
||||
if s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 3:
|
||||
stats['TOTAL'] = (parse_wall(parts[1]),
|
||||
parse_rss(parts[3]) if len(parts) > 3 else 0)
|
||||
break
|
||||
return stats
|
||||
|
||||
f = parse_reporter(log_filter)
|
||||
s = parse_reporter(log_select)
|
||||
|
||||
row = [species]
|
||||
for stage, d in [('rebuild', f), ('pack', f), ('filter_total', f), ('select', s), ('select_total', s)]:
|
||||
key = 'TOTAL' if stage.endswith('_total') else stage
|
||||
w, r = d.get(key, ('', ''))
|
||||
row += [f'{w:.3f}' if isinstance(w, float) else '', str(r)]
|
||||
print(','.join(row))
|
||||
PYEOF
|
||||
Executable
+108
@@ -0,0 +1,108 @@
|
||||
#!/usr/bin/env bash
|
||||
# Usage: filter_one_presence.sh SPECIES
|
||||
# Filters global_index_presence to keep only kmers specific to SPECIES,
|
||||
# then selects the SPECIES column in-place.
|
||||
# Outputs:
|
||||
# specific_index_presence/SPECIES/index.done (written by obikmer select)
|
||||
# stats/specific_kmer_presence/SPECIES.stats (one CSV data row, no header)
|
||||
set -euo pipefail
|
||||
|
||||
SPECIES="$1"
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
|
||||
SOURCE="${SCRIPT_DIR}/global_index_presence"
|
||||
OUTPUT="${SCRIPT_DIR}/specific_index_presence/${SPECIES}"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/specific_kmer_presence"
|
||||
STATS_FILE="${STATS_DIR}/${SPECIES}.stats"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
echo "[${SPECIES}] filter (presence) → ${OUTPUT}"
|
||||
|
||||
LOG_FILTER=$(mktemp)
|
||||
LOG_SELECT=$(mktemp)
|
||||
trap 'rm -f "${LOG_FILTER}" "${LOG_SELECT}"' EXIT
|
||||
|
||||
"${BINARY}" filter \
|
||||
--output "${OUTPUT}" \
|
||||
--force \
|
||||
--ingroup "species=${SPECIES}" \
|
||||
--outgroup all \
|
||||
--min-frac 0.5 \
|
||||
--max-frac 1.0 \
|
||||
--max-outgroup-count 0 \
|
||||
"${SOURCE}" \
|
||||
2>"${LOG_FILTER}"
|
||||
|
||||
cat "${LOG_FILTER}" >&2
|
||||
|
||||
"${BINARY}" select \
|
||||
--in-place \
|
||||
--group "${SPECIES}:species=${SPECIES}" \
|
||||
--group-op "${SPECIES}:any" \
|
||||
--select "${SPECIES}" \
|
||||
"${OUTPUT}" \
|
||||
2>"${LOG_SELECT}"
|
||||
|
||||
cat "${LOG_SELECT}" >&2
|
||||
|
||||
python3 - "${SPECIES}" "${LOG_FILTER}" "${LOG_SELECT}" <<'PYEOF' >"${STATS_FILE}"
|
||||
import sys, re
|
||||
|
||||
species, log_filter, log_select = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
|
||||
def strip_ansi(s):
|
||||
return re.sub(r'\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]', '', s)
|
||||
|
||||
def parse_wall(s):
|
||||
s = s.strip()
|
||||
if s.endswith('ms'): return float(s[:-2]) / 1000.0
|
||||
if s.endswith('s'): return float(s[:-1])
|
||||
return 0.0
|
||||
|
||||
def parse_rss(s):
|
||||
m = re.match(r'([\d.]+)\s*(GB|MB|KB|B)', s.strip())
|
||||
if not m: return 0
|
||||
return int(float(m.group(1)) * {'GB': 1<<30, 'MB': 1<<20, 'KB': 1024, 'B': 1}[m.group(2)])
|
||||
|
||||
def is_sep(s):
|
||||
return bool(s) and not re.search(r'[A-Za-z0-9]', s)
|
||||
|
||||
def parse_reporter(logfile):
|
||||
stats = {}
|
||||
state = 'scan'
|
||||
with open(logfile, errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = strip_ansi(raw.rstrip('\n'))
|
||||
s = line.strip()
|
||||
if state == 'scan':
|
||||
if re.search(r'\bstage\b.*\bwall\b', line):
|
||||
state = 'in_header'
|
||||
elif state == 'in_header':
|
||||
if is_sep(s): state = 'rows'
|
||||
elif state == 'rows':
|
||||
if is_sep(s): state = 'total'
|
||||
elif s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 4:
|
||||
stats[parts[0]] = (parse_wall(parts[1]), parse_rss(parts[3]))
|
||||
elif state == 'total':
|
||||
if s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 3:
|
||||
stats['TOTAL'] = (parse_wall(parts[1]),
|
||||
parse_rss(parts[3]) if len(parts) > 3 else 0)
|
||||
break
|
||||
return stats
|
||||
|
||||
f = parse_reporter(log_filter)
|
||||
s = parse_reporter(log_select)
|
||||
|
||||
row = [species]
|
||||
for stage, d in [('rebuild', f), ('pack', f), ('filter_total', f), ('select', s), ('select_total', s)]:
|
||||
key = 'TOTAL' if stage.endswith('_total') else stage
|
||||
w, r = d.get(key, ('', ''))
|
||||
row += [f'{w:.3f}' if isinstance(w, float) else '', str(r)]
|
||||
print(','.join(row))
|
||||
PYEOF
|
||||
Executable
+103
@@ -0,0 +1,103 @@
|
||||
#!/usr/bin/env bash
|
||||
# Usage: index_one_count.sh SPECIMEN
|
||||
# SPECIMEN = "species--strain" (Make pattern stem)
|
||||
# Outputs:
|
||||
# specimen_index_count/SPECIMEN/index.done (written by obikmer)
|
||||
# stats/indexing_count/SPECIMEN.stats (one CSV data row, no header)
|
||||
set -euo pipefail
|
||||
|
||||
SPECIMEN="$1"
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
|
||||
species="${SPECIMEN%%--*}"
|
||||
strain="${SPECIMEN#*--}"
|
||||
|
||||
READS_DIR="${SCRIPT_DIR}/simulated_data/${species}/${strain}"
|
||||
INDEX_PATH="${SCRIPT_DIR}/specimen_index_count/${SPECIMEN}"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/indexing_count"
|
||||
STATS_FILE="${STATS_DIR}/${SPECIMEN}.stats"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
r1="${READS_DIR}/reads_R1.fastq.gz"
|
||||
r2="${READS_DIR}/reads_R2.fastq.gz"
|
||||
if [[ ! -f "${r1}" || ! -f "${r2}" ]]; then
|
||||
echo "ERROR: reads not found in ${READS_DIR}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "[${SPECIMEN}] indexing (count) → ${INDEX_PATH}"
|
||||
|
||||
STDERR_LOG=$(mktemp)
|
||||
trap 'rm -f "${STDERR_LOG}"' EXIT
|
||||
|
||||
"${BINARY}" index \
|
||||
--output "${INDEX_PATH}" \
|
||||
--force \
|
||||
--theta 0 \
|
||||
--with-counts \
|
||||
--label "${SPECIMEN}" \
|
||||
--meta "species=${species}" \
|
||||
"${r1}" "${r2}" \
|
||||
2>"${STDERR_LOG}"
|
||||
|
||||
cat "${STDERR_LOG}" >&2
|
||||
|
||||
python3 - "${species}" "${strain}" "${STDERR_LOG}" <<'PYEOF' >"${STATS_FILE}"
|
||||
import sys, re
|
||||
|
||||
species, strain, logfile = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
|
||||
def strip_ansi(s):
|
||||
return re.sub(r'\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]', '', s)
|
||||
|
||||
def parse_wall(s):
|
||||
s = s.strip()
|
||||
if s.endswith('ms'): return float(s[:-2]) / 1000.0
|
||||
if s.endswith('s'): return float(s[:-1])
|
||||
return 0.0
|
||||
|
||||
def parse_rss(s):
|
||||
m = re.match(r'([\d.]+)\s*(GB|MB|KB|B)', s.strip())
|
||||
if not m: return 0
|
||||
return int(float(m.group(1)) * {'GB': 1<<30, 'MB': 1<<20, 'KB': 1024, 'B': 1}[m.group(2)])
|
||||
|
||||
def is_sep(s):
|
||||
return bool(s) and not re.search(r'[A-Za-z0-9]', s)
|
||||
|
||||
stats = {}
|
||||
state = 'scan'
|
||||
|
||||
with open(logfile, errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = strip_ansi(raw.rstrip('\n'))
|
||||
s = line.strip()
|
||||
if state == 'scan':
|
||||
if re.search(r'\bstage\b.*\bwall\b', line):
|
||||
state = 'in_header'
|
||||
elif state == 'in_header':
|
||||
if is_sep(s): state = 'rows'
|
||||
elif state == 'rows':
|
||||
if is_sep(s): state = 'total'
|
||||
elif s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 4:
|
||||
stats[parts[0]] = (parse_wall(parts[1]), parse_rss(parts[3]))
|
||||
elif state == 'total':
|
||||
if s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 3:
|
||||
stats[parts[0]] = (parse_wall(parts[1]),
|
||||
parse_rss(parts[3]) if len(parts) > 3 else 0)
|
||||
break
|
||||
|
||||
STAGE_ORDER = ['scatter', 'dereplicate', 'count_kmer', 'index']
|
||||
row = [species, strain]
|
||||
for stage in STAGE_ORDER:
|
||||
w, r = stats.get(stage, ('', ''))
|
||||
row += [f'{w:.3f}' if isinstance(w, float) else '', str(r)]
|
||||
tw, tr = stats.get('TOTAL', ('', ''))
|
||||
row += [f'{tw:.3f}' if isinstance(tw, float) else '', str(tr)]
|
||||
print(','.join(row))
|
||||
PYEOF
|
||||
Executable
+102
@@ -0,0 +1,102 @@
|
||||
#!/usr/bin/env bash
|
||||
# Usage: index_one_presence.sh SPECIMEN
|
||||
# SPECIMEN = "species--strain" (Make pattern stem)
|
||||
# Outputs:
|
||||
# specimen_index_presence/SPECIMEN/index.done (written by obikmer)
|
||||
# stats/indexing_presence/SPECIMEN.stats (one CSV data row, no header)
|
||||
set -euo pipefail
|
||||
|
||||
SPECIMEN="$1"
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
|
||||
species="${SPECIMEN%%--*}"
|
||||
strain="${SPECIMEN#*--}"
|
||||
|
||||
READS_DIR="${SCRIPT_DIR}/simulated_data/${species}/${strain}"
|
||||
INDEX_PATH="${SCRIPT_DIR}/specimen_index_presence/${SPECIMEN}"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/indexing_presence"
|
||||
STATS_FILE="${STATS_DIR}/${SPECIMEN}.stats"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
r1="${READS_DIR}/reads_R1.fastq.gz"
|
||||
r2="${READS_DIR}/reads_R2.fastq.gz"
|
||||
if [[ ! -f "${r1}" || ! -f "${r2}" ]]; then
|
||||
echo "ERROR: reads not found in ${READS_DIR}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "[${SPECIMEN}] indexing (presence) → ${INDEX_PATH}"
|
||||
|
||||
STDERR_LOG=$(mktemp)
|
||||
trap 'rm -f "${STDERR_LOG}"' EXIT
|
||||
|
||||
"${BINARY}" index \
|
||||
--output "${INDEX_PATH}" \
|
||||
--force \
|
||||
--theta 0 \
|
||||
--label "${SPECIMEN}" \
|
||||
--meta "species=${species}" \
|
||||
"${r1}" "${r2}" \
|
||||
2>"${STDERR_LOG}"
|
||||
|
||||
cat "${STDERR_LOG}" >&2
|
||||
|
||||
python3 - "${species}" "${strain}" "${STDERR_LOG}" <<'PYEOF' >"${STATS_FILE}"
|
||||
import sys, re
|
||||
|
||||
species, strain, logfile = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
|
||||
def strip_ansi(s):
|
||||
return re.sub(r'\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]', '', s)
|
||||
|
||||
def parse_wall(s):
|
||||
s = s.strip()
|
||||
if s.endswith('ms'): return float(s[:-2]) / 1000.0
|
||||
if s.endswith('s'): return float(s[:-1])
|
||||
return 0.0
|
||||
|
||||
def parse_rss(s):
|
||||
m = re.match(r'([\d.]+)\s*(GB|MB|KB|B)', s.strip())
|
||||
if not m: return 0
|
||||
return int(float(m.group(1)) * {'GB': 1<<30, 'MB': 1<<20, 'KB': 1024, 'B': 1}[m.group(2)])
|
||||
|
||||
def is_sep(s):
|
||||
return bool(s) and not re.search(r'[A-Za-z0-9]', s)
|
||||
|
||||
stats = {}
|
||||
state = 'scan'
|
||||
|
||||
with open(logfile, errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = strip_ansi(raw.rstrip('\n'))
|
||||
s = line.strip()
|
||||
if state == 'scan':
|
||||
if re.search(r'\bstage\b.*\bwall\b', line):
|
||||
state = 'in_header'
|
||||
elif state == 'in_header':
|
||||
if is_sep(s): state = 'rows'
|
||||
elif state == 'rows':
|
||||
if is_sep(s): state = 'total'
|
||||
elif s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 4:
|
||||
stats[parts[0]] = (parse_wall(parts[1]), parse_rss(parts[3]))
|
||||
elif state == 'total':
|
||||
if s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 3:
|
||||
stats[parts[0]] = (parse_wall(parts[1]),
|
||||
parse_rss(parts[3]) if len(parts) > 3 else 0)
|
||||
break
|
||||
|
||||
STAGE_ORDER = ['scatter', 'dereplicate', 'count_kmer', 'index']
|
||||
row = [species, strain]
|
||||
for stage in STAGE_ORDER:
|
||||
w, r = stats.get(stage, ('', ''))
|
||||
row += [f'{w:.3f}' if isinstance(w, float) else '', str(r)]
|
||||
tw, tr = stats.get('TOTAL', ('', ''))
|
||||
row += [f'{tw:.3f}' if isinstance(tw, float) else '', str(tr)]
|
||||
print(','.join(row))
|
||||
PYEOF
|
||||
@@ -0,0 +1,118 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate deps.mk — pure dependency declarations for the benchmark pipeline.
|
||||
|
||||
Like C .d files: only target: prerequisites lines, no recipes.
|
||||
Recipes stay in the Makefile as generic rules.
|
||||
"""
|
||||
import gzip
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
STOP_WORDS = {'complete', 'chromosome', 'whole', 'sequence', 'genome',
|
||||
'endosymbiont', 'of'}
|
||||
STOP_PREFIXES = ('scaffold', 'contig', 'plasmid')
|
||||
|
||||
|
||||
def is_stop(tok):
|
||||
t = tok.lower()
|
||||
return t in STOP_WORDS or any(t.startswith(p) for p in STOP_PREFIXES)
|
||||
|
||||
|
||||
def sanitize(s):
|
||||
return re.sub(r'[^A-Za-z0-9._-]', '_', s).strip('_')
|
||||
|
||||
|
||||
def collect_tokens(text):
|
||||
parts = []
|
||||
for tok in text.split():
|
||||
tok = tok.rstrip(',.')
|
||||
if is_stop(tok):
|
||||
break
|
||||
parts.append(sanitize(tok))
|
||||
return '_'.join(filter(None, parts))
|
||||
|
||||
|
||||
def parse_organism(defn, gcf_id):
|
||||
words = defn.split()
|
||||
species = sanitize(words[0] + '_' + words[1])
|
||||
|
||||
m = re.search(r'\bstr\.\s+(\S+)(?:\s+substr\.\s+(\S+))?', defn)
|
||||
if m:
|
||||
strain = sanitize(m.group(1))
|
||||
if m.group(2):
|
||||
strain += '_' + sanitize(m.group(2))
|
||||
return species, strain
|
||||
|
||||
m = re.search(r'\bstrain\b\s+(.*)', defn)
|
||||
if m:
|
||||
strain = collect_tokens(m.group(1))
|
||||
if strain:
|
||||
return species, strain
|
||||
|
||||
remainder = re.sub(r'^\S+ \S+\s*', '', defn)
|
||||
remainder = re.sub(r'^subsp\.\s+\S+\s*', '', remainder)
|
||||
remainder = re.sub(r'^serovar\s+\S+\s*', '', remainder)
|
||||
strain = collect_tokens(remainder)
|
||||
return species, strain if strain else gcf_id
|
||||
|
||||
|
||||
def first_definition(path):
|
||||
with gzip.open(path, 'rt') as fh:
|
||||
for line in fh:
|
||||
if line.startswith('>'):
|
||||
m = re.search(r'"definition":"([^"]*)"', line)
|
||||
return m.group(1) if m else line[1:].split()[0]
|
||||
return Path(path).stem
|
||||
|
||||
|
||||
def main():
|
||||
entries = [] # (specimen, species, sim_dir, genome_path)
|
||||
species_seen = []
|
||||
|
||||
for path in sorted(sys.argv[1:]):
|
||||
gcf_id = Path(path).name.replace('_genomic.fna.gz', '')
|
||||
defn = first_definition(path)
|
||||
sp, st = parse_organism(defn, gcf_id)
|
||||
specimen = f'{sp}--{st}'
|
||||
sim_dir = f'simulated_data/{sp}/{st}'
|
||||
entries.append((specimen, sp, sim_dir, path))
|
||||
if sp not in species_seen:
|
||||
species_seen.append(sp)
|
||||
|
||||
specimens = [e[0] for e in entries]
|
||||
print('SPECIMENS :=', ' '.join(specimens))
|
||||
print('SPECIES :=', ' '.join(species_seen))
|
||||
|
||||
for specimen, species, sim_dir, genome in entries:
|
||||
reads = f'{sim_dir}/reads_R1.fastq.gz'
|
||||
p_done = f'specimen_index_presence/{specimen}/index.done'
|
||||
p_stats = f'stats/indexing_presence/{specimen}.stats'
|
||||
c_done = f'specimen_index_count/{specimen}/index.done'
|
||||
c_stats = f'stats/indexing_count/{specimen}.stats'
|
||||
ref = f'reference_index/{specimen}.npz'
|
||||
vp = f'stats/verify_presence/{specimen}.stats'
|
||||
vc = f'stats/verify_count/{specimen}.stats'
|
||||
|
||||
print()
|
||||
print(f'# {specimen}')
|
||||
print(f'{reads}: {genome}')
|
||||
print(f'{ref}: {reads}')
|
||||
print(f'{p_done} {p_stats}: {reads}')
|
||||
print(f'{c_done} {c_stats}: {reads}')
|
||||
print(f'{vp}: {ref} {p_done}')
|
||||
print(f'{vc}: {ref} {c_done}')
|
||||
|
||||
print()
|
||||
for sp in species_seen:
|
||||
sp_done = f'specific_index_presence/{sp}/index.done'
|
||||
sp_stats = f'stats/specific_kmer_presence/{sp}.stats'
|
||||
sc_done = f'specific_index_count/{sp}/index.done'
|
||||
sc_stats = f'stats/specific_kmer_count/{sp}.stats'
|
||||
print(f'# {sp}')
|
||||
print(f'{sp_done} {sp_stats}: global_index_presence/index.done')
|
||||
print(f'{sc_done} {sc_stats}: global_index_count/index.done')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Executable
+103
@@ -0,0 +1,103 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
IDX_DIR="${SCRIPT_DIR}/specimen_index_count"
|
||||
OUTPUT="${SCRIPT_DIR}/global_index_count"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/merge_count"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
run_n=$(printf '%03d' "$(find "${STATS_DIR}" -maxdepth 1 -name 'run_*.csv' | wc -l | tr -d ' ')")
|
||||
CSV="${STATS_DIR}/run_${run_n}.csv"
|
||||
|
||||
printf 'run,n_sources,bootstrap_wall_s,bootstrap_rss_b,spectrums_wall_s,spectrums_rss_b,merge_partitions_wall_s,merge_partitions_rss_b,pack_wall_s,pack_rss_b,total_wall_s,total_rss_b\n' >"${CSV}"
|
||||
|
||||
parse_reporter() {
|
||||
local run="$1" n_sources="$2" logfile="$3"
|
||||
python3 - "$run" "$n_sources" "$logfile" <<'PYEOF'
|
||||
import sys, re
|
||||
|
||||
run, n_sources, logfile = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
|
||||
def strip_ansi(s):
|
||||
return re.sub(r'\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]', '', s)
|
||||
|
||||
def parse_wall(s):
|
||||
s = s.strip()
|
||||
if s.endswith('ms'): return float(s[:-2]) / 1000.0
|
||||
if s.endswith('s'): return float(s[:-1])
|
||||
return 0.0
|
||||
|
||||
def parse_rss(s):
|
||||
m = re.match(r'([\d.]+)\s*(GB|MB|KB|B)', s.strip())
|
||||
if not m: return 0
|
||||
return int(float(m.group(1)) * {'GB': 1<<30, 'MB': 1<<20, 'KB': 1024, 'B': 1}[m.group(2)])
|
||||
|
||||
def is_sep(s):
|
||||
return bool(s) and not re.search(r'[A-Za-z0-9]', s)
|
||||
|
||||
stats = {}
|
||||
state = 'scan'
|
||||
|
||||
with open(logfile, errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = strip_ansi(raw.rstrip('\n'))
|
||||
s = line.strip()
|
||||
|
||||
if state == 'scan':
|
||||
if re.search(r'\bstage\b.*\bwall\b', line):
|
||||
state = 'in_header'
|
||||
elif state == 'in_header':
|
||||
if is_sep(s):
|
||||
state = 'rows'
|
||||
elif state == 'rows':
|
||||
if is_sep(s):
|
||||
state = 'total'
|
||||
elif s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 4:
|
||||
stats[parts[0]] = (parse_wall(parts[1]), parse_rss(parts[3]))
|
||||
elif state == 'total':
|
||||
if s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 3:
|
||||
stats[parts[0]] = (parse_wall(parts[1]),
|
||||
parse_rss(parts[3]) if len(parts) > 3 else 0)
|
||||
break
|
||||
|
||||
STAGE_ORDER = ['bootstrap', 'spectrums', 'merge_partitions', 'pack']
|
||||
row = [run, n_sources]
|
||||
for stage in STAGE_ORDER:
|
||||
w, r = stats.get(stage, ('', ''))
|
||||
row += [f'{w:.3f}' if isinstance(w, float) else '', str(r)]
|
||||
tw, tr = stats.get('TOTAL', ('', ''))
|
||||
row += [f'{tw:.3f}' if isinstance(tw, float) else '', str(tr)]
|
||||
print(','.join(row))
|
||||
PYEOF
|
||||
}
|
||||
|
||||
mapfile -t sources < <(find "${IDX_DIR}" -mindepth 1 -maxdepth 1 -type d | sort)
|
||||
|
||||
if [[ ${#sources[@]} -eq 0 ]]; then
|
||||
echo "ERROR: no indexes found in ${IDX_DIR}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Merging ${#sources[@]} count indexes → ${OUTPUT}"
|
||||
printf ' %s\n' "${sources[@]}"
|
||||
|
||||
STDERR_LOG=$(mktemp)
|
||||
trap 'rm -f "${STDERR_LOG}"' EXIT
|
||||
|
||||
"${BINARY}" merge \
|
||||
--output "${OUTPUT}" \
|
||||
--force \
|
||||
"${sources[@]}" \
|
||||
2>"${STDERR_LOG}"
|
||||
|
||||
cat "${STDERR_LOG}" >&2
|
||||
parse_reporter "${run_n}" "${#sources[@]}" "${STDERR_LOG}" >>"${CSV}"
|
||||
|
||||
echo "Done. Run ${run_n} → ${CSV}"
|
||||
Executable
+104
@@ -0,0 +1,104 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
IDX_DIR="${SCRIPT_DIR}/specimen_index_presence"
|
||||
OUTPUT="${SCRIPT_DIR}/global_index_presence"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/merge_presence"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
run_n=$(printf '%03d' "$(find "${STATS_DIR}" -maxdepth 1 -name 'run_*.csv' | wc -l | tr -d ' ')")
|
||||
CSV="${STATS_DIR}/run_${run_n}.csv"
|
||||
|
||||
printf 'run,n_sources,bootstrap_wall_s,bootstrap_rss_b,spectrums_wall_s,spectrums_rss_b,merge_partitions_wall_s,merge_partitions_rss_b,pack_wall_s,pack_rss_b,total_wall_s,total_rss_b\n' >"${CSV}"
|
||||
|
||||
parse_reporter() {
|
||||
local run="$1" n_sources="$2" logfile="$3"
|
||||
python3 - "$run" "$n_sources" "$logfile" <<'PYEOF'
|
||||
import sys, re
|
||||
|
||||
run, n_sources, logfile = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
|
||||
def strip_ansi(s):
|
||||
return re.sub(r'\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]', '', s)
|
||||
|
||||
def parse_wall(s):
|
||||
s = s.strip()
|
||||
if s.endswith('ms'): return float(s[:-2]) / 1000.0
|
||||
if s.endswith('s'): return float(s[:-1])
|
||||
return 0.0
|
||||
|
||||
def parse_rss(s):
|
||||
m = re.match(r'([\d.]+)\s*(GB|MB|KB|B)', s.strip())
|
||||
if not m: return 0
|
||||
return int(float(m.group(1)) * {'GB': 1<<30, 'MB': 1<<20, 'KB': 1024, 'B': 1}[m.group(2)])
|
||||
|
||||
def is_sep(s):
|
||||
return bool(s) and not re.search(r'[A-Za-z0-9]', s)
|
||||
|
||||
stats = {}
|
||||
state = 'scan'
|
||||
|
||||
with open(logfile, errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = strip_ansi(raw.rstrip('\n'))
|
||||
s = line.strip()
|
||||
|
||||
if state == 'scan':
|
||||
if re.search(r'\bstage\b.*\bwall\b', line):
|
||||
state = 'in_header'
|
||||
elif state == 'in_header':
|
||||
if is_sep(s):
|
||||
state = 'rows'
|
||||
elif state == 'rows':
|
||||
if is_sep(s):
|
||||
state = 'total'
|
||||
elif s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 4:
|
||||
stats[parts[0]] = (parse_wall(parts[1]), parse_rss(parts[3]))
|
||||
elif state == 'total':
|
||||
if s:
|
||||
parts = re.split(r' +', s)
|
||||
if len(parts) >= 3:
|
||||
stats[parts[0]] = (parse_wall(parts[1]),
|
||||
parse_rss(parts[3]) if len(parts) > 3 else 0)
|
||||
break
|
||||
|
||||
STAGE_ORDER = ['bootstrap', 'spectrums', 'merge_partitions', 'pack']
|
||||
row = [run, n_sources]
|
||||
for stage in STAGE_ORDER:
|
||||
w, r = stats.get(stage, ('', ''))
|
||||
row += [f'{w:.3f}' if isinstance(w, float) else '', str(r)]
|
||||
tw, tr = stats.get('TOTAL', ('', ''))
|
||||
row += [f'{tw:.3f}' if isinstance(tw, float) else '', str(tr)]
|
||||
print(','.join(row))
|
||||
PYEOF
|
||||
}
|
||||
|
||||
mapfile -t sources < <(find "${IDX_DIR}" -mindepth 1 -maxdepth 1 -type d | sort)
|
||||
|
||||
if [[ ${#sources[@]} -eq 0 ]]; then
|
||||
echo "ERROR: no indexes found in ${IDX_DIR}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Merging ${#sources[@]} presence indexes → ${OUTPUT}"
|
||||
printf ' %s\n' "${sources[@]}"
|
||||
|
||||
STDERR_LOG=$(mktemp)
|
||||
trap 'rm -f "${STDERR_LOG}"' EXIT
|
||||
|
||||
"${BINARY}" merge \
|
||||
--output "${OUTPUT}" \
|
||||
--force \
|
||||
--force-presence \
|
||||
"${sources[@]}" \
|
||||
2>"${STDERR_LOG}"
|
||||
|
||||
cat "${STDERR_LOG}" >&2
|
||||
parse_reporter "${run_n}" "${#sources[@]}" "${STDERR_LOG}" >>"${CSV}"
|
||||
|
||||
echo "Done. Run ${run_n} → ${CSV}"
|
||||
Executable
+12
@@ -0,0 +1,12 @@
|
||||
#!/usr/bin/env bash
|
||||
# Simulate all genomes. Delegates to simulate_one.sh per genome.
|
||||
# Prefer running via `gmake simulate` which handles individual dependencies.
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
|
||||
for genome_file in "${SCRIPT_DIR}"/genomes/*.fna.gz; do
|
||||
out_dir=$("${SCRIPT_DIR}/../.venv/bin/python3" "${SCRIPT_DIR}/make_deps.py" \
|
||||
--dir-for "${genome_file}")
|
||||
bash "${SCRIPT_DIR}/simulate_one.sh" "${genome_file}" "${out_dir}"
|
||||
done
|
||||
@@ -0,0 +1,33 @@
|
||||
#!/usr/bin/env bash
|
||||
# Usage: simulate_one.sh genome.fna.gz output_dir
|
||||
# Simulates paired-end HiSeq reads for a single genome.
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
ISS="${SCRIPT_DIR}/../.venv/bin/iss"
|
||||
COVERAGE=15
|
||||
READ_LENGTH=150
|
||||
CPUS="${CPUS:-$(sysctl -n hw.logicalcpu 2>/dev/null || nproc 2>/dev/null || echo 2)}"
|
||||
|
||||
genome_file="$1"
|
||||
out_dir="$2"
|
||||
|
||||
mkdir -p "${out_dir}"
|
||||
|
||||
tmp_fasta=$(mktemp "${TMPDIR:-/tmp}/obikmer_XXXXXX.fna")
|
||||
trap 'rm -f "${tmp_fasta}"' EXIT
|
||||
|
||||
gzip -dc "${genome_file}" > "${tmp_fasta}"
|
||||
|
||||
genome_size=$(grep -v "^>" "${tmp_fasta}" | tr -d '[:space:]' | wc -c | tr -d ' ')
|
||||
n_reads=$(python3 -c "import math; print(math.ceil(${COVERAGE} * ${genome_size} / (2 * ${READ_LENGTH})))")
|
||||
|
||||
echo "[${out_dir}] genome=${genome_size} bp → ${n_reads} read pairs (${COVERAGE}x HiSeq)"
|
||||
|
||||
"${ISS}" generate \
|
||||
--genomes "${tmp_fasta}" \
|
||||
--model HiSeq \
|
||||
--n_reads "${n_reads}" \
|
||||
--cpus "${CPUS}" \
|
||||
--compress \
|
||||
--output "${out_dir}/reads"
|
||||
@@ -0,0 +1,21 @@
|
||||
genome,Candidozyma_auris--GCF_003013715.1_ASM301371v2,Acidobacterium_capsulatum--ATCC_51196,Bacillus_subtilis--168,Escherichia_coli--CFT073,Escherichia_coli--EDL933,Escherichia_coli--K-12_MG1655,Escherichia_coli--K-12_W3110,Klebsiella_pneumoniae--ATCC_13883,Klebsiella_pneumoniae--HS11286,Klebsiella_pneumoniae--MGH_78578,Opitutus_terrae--PB90-1,Proteus_mirabilis--HI4320,Saccharolobus_islandicus--M.16.4,Salmonella_enterica--AKU_12601,Salmonella_enterica--CT18,Salmonella_enterica--LT2,Salmonella_enterica--P125109,Shouchella_clausii--KSM-K16,Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1,Yersinia_ruckeri--YRB
|
||||
Candidozyma_auris--GCF_003013715.1_ASM301371v2,0.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000
|
||||
Acidobacterium_capsulatum--ATCC_51196,1.000000,0.000000,0.999981,0.999990,0.999989,0.999987,0.999987,0.999990,0.999988,0.999988,0.999994,0.999989,1.000000,0.999988,0.999987,0.999987,0.999988,0.999989,0.999991,0.999987
|
||||
Bacillus_subtilis--168,1.000000,0.999981,0.000000,0.999990,0.999989,0.999989,0.999989,0.999989,0.999988,0.999986,0.999995,0.999985,0.999999,0.999988,0.999987,0.999989,0.999988,0.999778,0.999993,0.999987
|
||||
Escherichia_coli--CFT073,1.000000,0.999990,0.999990,0.000000,0.825741,0.807495,0.807218,0.991156,0.996855,0.997849,0.999996,0.999633,1.000000,0.993885,0.996736,0.994148,0.993821,0.999991,0.999984,0.999291
|
||||
Escherichia_coli--EDL933,1.000000,0.999989,0.999989,0.825741,0.000000,0.735107,0.734775,0.996126,0.998058,0.997908,0.999997,0.999640,1.000000,0.993993,0.997126,0.994390,0.994059,0.999991,0.999986,0.999292
|
||||
Escherichia_coli--K-12_MG1655,1.000000,0.999987,0.999989,0.807495,0.735107,0.000000,0.382567,0.996190,0.997747,0.997455,0.999996,0.999604,1.000000,0.993444,0.996645,0.993773,0.993431,0.999989,0.999984,0.999174
|
||||
Escherichia_coli--K-12_W3110,1.000000,0.999987,0.999989,0.807218,0.734775,0.382567,0.000000,0.996220,0.997761,0.997467,0.999995,0.999604,1.000000,0.993445,0.996669,0.993769,0.993443,0.999990,0.999985,0.999165
|
||||
Klebsiella_pneumoniae--ATCC_13883,1.000000,0.999990,0.999989,0.991156,0.996126,0.996190,0.996220,0.000000,0.845220,0.840545,0.999997,0.999648,1.000000,0.996177,0.998128,0.996268,0.996052,0.999990,0.999987,0.999325
|
||||
Klebsiella_pneumoniae--HS11286,1.000000,0.999988,0.999988,0.996855,0.998058,0.997747,0.997761,0.845220,0.000000,0.906475,0.999996,0.999683,1.000000,0.997724,0.995697,0.997776,0.997769,0.999989,0.999979,0.999463
|
||||
Klebsiella_pneumoniae--MGH_78578,1.000000,0.999988,0.999986,0.997849,0.997908,0.997455,0.997467,0.840545,0.906475,0.000000,0.999996,0.999704,1.000000,0.997928,0.995054,0.997844,0.997868,0.999990,0.999980,0.999479
|
||||
Opitutus_terrae--PB90-1,1.000000,0.999994,0.999995,0.999996,0.999997,0.999996,0.999995,0.999997,0.999996,0.999996,0.000000,0.999997,0.999998,0.999996,0.999996,0.999996,0.999995,0.999997,0.999993,0.999996
|
||||
Proteus_mirabilis--HI4320,1.000000,0.999989,0.999985,0.999633,0.999640,0.999604,0.999604,0.999648,0.999683,0.999704,0.999997,0.000000,1.000000,0.999604,0.999699,0.999622,0.999613,0.999987,0.999983,0.999505
|
||||
Saccharolobus_islandicus--M.16.4,1.000000,1.000000,0.999999,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,0.999998,1.000000,0.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000,1.000000
|
||||
Salmonella_enterica--AKU_12601,1.000000,0.999988,0.999988,0.993885,0.993993,0.993444,0.993445,0.996177,0.997724,0.997928,0.999996,0.999604,1.000000,0.000000,0.869238,0.682277,0.663383,0.999990,0.999985,0.999260
|
||||
Salmonella_enterica--CT18,1.000000,0.999987,0.999987,0.996736,0.997126,0.996645,0.996669,0.998128,0.995697,0.995054,0.999996,0.999699,1.000000,0.869238,0.000000,0.890872,0.886148,0.999988,0.999976,0.999524
|
||||
Salmonella_enterica--LT2,1.000000,0.999987,0.999989,0.994148,0.994390,0.993773,0.993769,0.996268,0.997776,0.997844,0.999996,0.999622,1.000000,0.682277,0.890872,0.000000,0.622606,0.999989,0.999985,0.999296
|
||||
Salmonella_enterica--P125109,1.000000,0.999988,0.999988,0.993821,0.994059,0.993431,0.993443,0.996052,0.997769,0.997868,0.999995,0.999613,1.000000,0.663383,0.886148,0.622606,0.000000,0.999988,0.999983,0.999270
|
||||
Shouchella_clausii--KSM-K16,1.000000,0.999989,0.999778,0.999991,0.999991,0.999989,0.999990,0.999990,0.999989,0.999990,0.999997,0.999987,1.000000,0.999990,0.999988,0.999989,0.999988,0.000000,0.999991,0.999988
|
||||
Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1,1.000000,0.999991,0.999993,0.999984,0.999986,0.999984,0.999985,0.999987,0.999979,0.999980,0.999993,0.999983,1.000000,0.999985,0.999976,0.999985,0.999983,0.999991,0.000000,0.999983
|
||||
Yersinia_ruckeri--YRB,1.000000,0.999987,0.999987,0.999291,0.999292,0.999174,0.999165,0.999325,0.999463,0.999479,0.999996,0.999505,1.000000,0.999260,0.999524,0.999296,0.999270,0.999988,0.999983,0.000000
|
||||
|
@@ -0,0 +1 @@
|
||||
(((((((((((Candidozyma_auris--GCF_003013715.1_ASM301371v2:0.5000001881725941,Saccharolobus_islandicus--M.16.4:0.4999993211600824):0.0000023411501775538747,Opitutus_terrae--PB90-1:0.499997075187947):0.0000029791191795691675,(Acidobacterium_capsulatum--ATCC_51196:0.49999227771334689,(Bacillus_subtilis--168:0.49988797935621456,Shouchella_clausii--KSM-K16:0.49988984146059159):0.0001037210285571577):0.0000023959836053522034):0.0000034093646568700288,Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1:0.4999920159222422):0.000199555100890203,Proteus_mirabilis--HI4320:0.49979129185300427):0.00010103619067070024,Yersinia_ruckeri--YRB:0.4996806650749249):0.0013719139155004,(Klebsiella_pneumoniae--HS11286:0.43798845051648258,(Klebsiella_pneumoniae--ATCC_13883:0.41780293826821265,Klebsiella_pneumoniae--MGH_78578:0.42274184870836559):0.017586732339732737):0.0604124197073832):0.0006482538063555254,(Salmonella_enterica--CT18:0.43952894448143017,(Salmonella_enterica--AKU_12601:0.3357977326267918,(Salmonella_enterica--LT2:0.31203395843666389,Salmonella_enterica--P125109:0.31057217324861216):0.025729515856701136):0.10292985918524672):0.05825411485542886):0.08937928015651564,Escherichia_coli--CFT073:0.40806501650701029):0.0410131211869626,Escherichia_coli--EDL933:0.3681464750911808):0.1755112579711463,Escherichia_coli--K-12_MG1655:0.19129818036662728,Escherichia_coli--K-12_W3110:0.19126872019906239);
|
||||
@@ -0,0 +1,21 @@
|
||||
genome,Candidozyma_auris--GCF_003013715.1_ASM301371v2,Acidobacterium_capsulatum--ATCC_51196,Bacillus_subtilis--168,Escherichia_coli--CFT073,Escherichia_coli--EDL933,Escherichia_coli--K-12_MG1655,Escherichia_coli--K-12_W3110,Klebsiella_pneumoniae--ATCC_13883,Klebsiella_pneumoniae--HS11286,Klebsiella_pneumoniae--MGH_78578,Opitutus_terrae--PB90-1,Proteus_mirabilis--HI4320,Saccharolobus_islandicus--M.16.4,Salmonella_enterica--AKU_12601,Salmonella_enterica--CT18,Salmonella_enterica--LT2,Salmonella_enterica--P125109,Shouchella_clausii--KSM-K16,Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1,Yersinia_ruckeri--YRB
|
||||
Candidozyma_auris--GCF_003013715.1_ASM301371v2,0,0,0,0,0,0,0,0,0,0,0,0,8,0,1,0,0,0,0,3
|
||||
Acidobacterium_capsulatum--ATCC_51196,0,0,203,119,128,141,140,116,109,111,78,112,0,136,109,147,134,117,55,129
|
||||
Bacillus_subtilis--168,0,203,0,124,132,128,123,133,109,130,66,158,6,131,112,124,135,2393,46,124
|
||||
Escherichia_coli--CFT073,0,119,124,0,1966777,1998059,1999094,117743,32029,22312,63,4225,0,74946,31918,73311,76585,113,128,7854
|
||||
Escherichia_coli--EDL933,0,128,132,1966777,0,2627885,2628700,52488,20134,22064,48,4202,0,74655,28602,71244,74665,112,108,7963
|
||||
Escherichia_coli--K-12_MG1655,0,141,128,1998059,2627885,0,4452541,48302,21382,24602,47,4277,0,75729,30449,73622,76778,119,111,8566
|
||||
Escherichia_coli--K-12_W3110,0,140,123,1999094,2628700,4452541,0,47894,21226,24470,68,4278,0,75658,30207,73614,76583,112,108,8660
|
||||
Klebsiella_pneumoniae--ATCC_13883,0,116,133,117743,52488,48302,47894,0,1416091,1477759,42,4172,0,48296,18988,48144,50416,120,106,7712
|
||||
Klebsiella_pneumoniae--HS11286,0,109,109,32029,20134,21382,21226,1416091,0,644063,42,2738,0,21498,29758,21606,21376,99,102,4417
|
||||
Klebsiella_pneumoniae--MGH_78578,0,111,130,22312,22064,24602,24470,1477759,644063,0,42,2614,0,19948,35067,21330,20813,97,102,4374
|
||||
Opitutus_terrae--PB90-1,0,78,66,63,48,47,68,42,42,42,0,43,18,57,42,53,66,39,58,43
|
||||
Proteus_mirabilis--HI4320,0,112,158,4225,4202,4277,4278,4172,2738,2614,43,0,0,4254,2481,4166,4215,131,103,4704
|
||||
Saccharolobus_islandicus--M.16.4,8,0,6,0,0,0,0,0,0,0,18,0,0,0,0,0,0,0,0,0
|
||||
Salmonella_enterica--AKU_12601,0,136,131,74946,74655,75729,75658,48296,21498,19948,57,4254,0,0,1047731,2857146,2951421,117,108,7643
|
||||
Salmonella_enterica--CT18,1,109,112,31918,28602,30449,30207,18988,29758,35067,42,2481,0,1047731,0,917948,940297,106,106,3716
|
||||
Salmonella_enterica--LT2,0,147,124,73311,71244,73622,73614,48144,21606,21330,53,4166,0,2857146,917948,0,3284800,122,108,7460
|
||||
Salmonella_enterica--P125109,0,134,135,76585,74665,76778,76583,50416,21376,20813,66,4215,0,2951421,940297,3284800,0,134,124,7645
|
||||
Shouchella_clausii--KSM-K16,0,117,2393,113,112,119,112,120,99,97,39,131,0,117,106,122,134,0,58,124
|
||||
Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1,0,55,46,128,108,111,108,106,102,102,58,103,0,108,106,108,124,58,0,96
|
||||
Yersinia_ruckeri--YRB,3,129,124,7854,7963,8566,8660,7712,4417,4374,43,4704,0,7643,3716,7460,7645,124,96,0
|
||||
|
Executable
+181
@@ -0,0 +1,181 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Compare an obikmer count index against a reference kmer set (presence + counts).
|
||||
|
||||
Loads the reference .npz (sorted uint64 kmers + uint32 counts from build_reference.py),
|
||||
streams `obikmer dump` from a --with-counts index, then reports:
|
||||
- false negatives : kmers in reference absent from the index
|
||||
- false positives : kmers in the index absent from the reference
|
||||
- count mismatches: kmers present in both but with differing counts
|
||||
|
||||
Output to stdout: one CSV row
|
||||
species,strain,ref_kmers,idx_kmers,false_neg,false_pos,count_mismatch,
|
||||
fn_pct,fp_pct,cm_pct
|
||||
"""
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
# ── encoding ──────────────────────────────────────────────────────────────────
|
||||
|
||||
_ENCODE = {'A': 0, 'C': 1, 'G': 2, 'T': 3,
|
||||
'a': 0, 'c': 1, 'g': 2, 't': 3}
|
||||
|
||||
_DECODE = ['A', 'C', 'G', 'T']
|
||||
|
||||
|
||||
def encode_kmer(s: str) -> int:
|
||||
kmer = 0
|
||||
for c in s:
|
||||
kmer = (kmer << 2) | _ENCODE[c]
|
||||
return kmer
|
||||
|
||||
|
||||
def decode_kmer(val: int, k: int) -> str:
|
||||
bases = []
|
||||
for _ in range(k):
|
||||
bases.append(_DECODE[val & 3])
|
||||
val >>= 2
|
||||
return ''.join(reversed(bases))
|
||||
|
||||
|
||||
# ── dump parsing ──────────────────────────────────────────────────────────────
|
||||
|
||||
def load_index(obikmer_bin: str, index_dir: str) -> tuple[np.ndarray, np.ndarray]:
|
||||
"""Stream `obikmer dump` and return (kmers_sorted_uint64, counts_uint32)."""
|
||||
cmd = [obikmer_bin, 'dump', index_dir]
|
||||
proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL,
|
||||
text=True)
|
||||
kmers, counts = [], []
|
||||
header = True
|
||||
for line in proc.stdout:
|
||||
if header:
|
||||
header = False
|
||||
continue
|
||||
parts = line.rstrip('\n').split(',')
|
||||
kmers.append(encode_kmer(parts[0]))
|
||||
counts.append(int(parts[1]))
|
||||
proc.wait()
|
||||
if proc.returncode != 0:
|
||||
print(f'ERROR: obikmer dump exited {proc.returncode}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
order = np.argsort(np.array(kmers, dtype=np.uint64), kind='stable')
|
||||
return (np.array(kmers, dtype=np.uint64)[order],
|
||||
np.array(counts, dtype=np.uint32)[order])
|
||||
|
||||
|
||||
# ── comparison ────────────────────────────────────────────────────────────────
|
||||
|
||||
def compare(ref_kmers: np.ndarray, ref_counts: np.ndarray,
|
||||
idx_kmers: np.ndarray, idx_counts: np.ndarray,
|
||||
) -> tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray, np.ndarray]:
|
||||
"""Return (false_neg, false_pos, cm_ref_kmers, cm_ref_counts, cm_idx_counts).
|
||||
|
||||
All arrays sorted; cm_* cover kmers present in both arrays but with
|
||||
differing counts.
|
||||
"""
|
||||
false_neg = np.setdiff1d(ref_kmers, idx_kmers, assume_unique=True)
|
||||
false_pos = np.setdiff1d(idx_kmers, ref_kmers, assume_unique=True)
|
||||
|
||||
# Count mismatches among shared kmers.
|
||||
# Both arrays are sorted so we can use searchsorted.
|
||||
pos_in_idx = np.searchsorted(idx_kmers, ref_kmers)
|
||||
pos_in_idx = np.clip(pos_in_idx, 0, len(idx_kmers) - 1)
|
||||
shared_mask = idx_kmers[pos_in_idx] == ref_kmers
|
||||
|
||||
shared_ref_counts = ref_counts[shared_mask]
|
||||
shared_idx_counts = idx_counts[pos_in_idx[shared_mask]]
|
||||
mismatch_mask = shared_ref_counts != shared_idx_counts
|
||||
|
||||
cm_kmers = ref_kmers[shared_mask][mismatch_mask]
|
||||
cm_ref_counts = shared_ref_counts[mismatch_mask]
|
||||
cm_idx_counts = shared_idx_counts[mismatch_mask]
|
||||
|
||||
return false_neg, false_pos, cm_kmers, cm_ref_counts, cm_idx_counts
|
||||
|
||||
|
||||
# ── main ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('reference', metavar='REF_NPZ', nargs='?',
|
||||
help='Reference .npz file')
|
||||
ap.add_argument('index', metavar='INDEX_DIR', nargs='?',
|
||||
help='obikmer index directory (built with --with-counts)')
|
||||
ap.add_argument('--obikmer', default='obikmer',
|
||||
help='Path to obikmer binary')
|
||||
ap.add_argument('--species', default='')
|
||||
ap.add_argument('--strain', default='')
|
||||
ap.add_argument('--header', action='store_true',
|
||||
help='Print CSV header and exit')
|
||||
ap.add_argument('--save-fp', metavar='FILE',
|
||||
help='Save false-positive kmer strings to FILE')
|
||||
ap.add_argument('--save-fn', metavar='FILE',
|
||||
help='Save false-negative kmer strings to FILE')
|
||||
ap.add_argument('--save-cm', metavar='FILE',
|
||||
help='Save count-mismatch rows (kmer,ref_count,idx_count) to FILE')
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.header:
|
||||
print('species,strain,ref_kmers,idx_kmers,'
|
||||
'false_neg,false_pos,count_mismatch,'
|
||||
'fn_pct,fp_pct,cm_pct')
|
||||
return
|
||||
|
||||
# Detect k
|
||||
cmd1 = [args.obikmer, 'dump', '--head', '1', args.index]
|
||||
out1 = subprocess.check_output(cmd1, stderr=subprocess.DEVNULL, text=True)
|
||||
k = len(out1.splitlines()[1].split(',')[0])
|
||||
|
||||
# Load reference
|
||||
print(f'Loading reference: {args.reference}', file=sys.stderr)
|
||||
npz = np.load(args.reference)
|
||||
ref_kmers = npz['kmers'] # sorted uint64
|
||||
ref_counts = npz['counts'] # uint32
|
||||
|
||||
# Load index
|
||||
print(f'Streaming dump (k={k}): {args.index}', file=sys.stderr)
|
||||
idx_kmers, idx_counts = load_index(args.obikmer, args.index)
|
||||
|
||||
print(f'k={k} ref={len(ref_kmers):,} idx={len(idx_kmers):,}', file=sys.stderr)
|
||||
|
||||
false_neg, false_pos, cm_kmers, cm_ref, cm_idx = compare(
|
||||
ref_kmers, ref_counts, idx_kmers, idx_counts)
|
||||
|
||||
n_shared = len(ref_kmers) - len(false_neg)
|
||||
fn_pct = 100.0 * len(false_neg) / len(ref_kmers) if len(ref_kmers) else 0.0
|
||||
fp_pct = 100.0 * len(false_pos) / len(idx_kmers) if len(idx_kmers) else 0.0
|
||||
cm_pct = 100.0 * len(cm_kmers) / n_shared if n_shared else 0.0
|
||||
|
||||
print(f'false negatives : {len(false_neg):,} ({fn_pct:.4f}%)', file=sys.stderr)
|
||||
print(f'false positives : {len(false_pos):,} ({fp_pct:.4f}%)', file=sys.stderr)
|
||||
print(f'count mismatches: {len(cm_kmers):,} ({cm_pct:.4f}% of shared)',
|
||||
file=sys.stderr)
|
||||
|
||||
if args.save_fn and len(false_neg):
|
||||
with open(args.save_fn, 'w') as fh:
|
||||
for v in false_neg:
|
||||
fh.write(decode_kmer(int(v), k) + '\n')
|
||||
|
||||
if args.save_fp and len(false_pos):
|
||||
with open(args.save_fp, 'w') as fh:
|
||||
for v in false_pos:
|
||||
fh.write(decode_kmer(int(v), k) + '\n')
|
||||
|
||||
if args.save_cm and len(cm_kmers):
|
||||
with open(args.save_cm, 'w') as fh:
|
||||
fh.write('kmer,ref_count,idx_count\n')
|
||||
for v, rc, ic in zip(cm_kmers, cm_ref, cm_idx):
|
||||
fh.write(f'{decode_kmer(int(v), k)},{rc},{ic}\n')
|
||||
|
||||
print(f'{args.species},{args.strain},'
|
||||
f'{len(ref_kmers)},{len(idx_kmers)},'
|
||||
f'{len(false_neg)},{len(false_pos)},{len(cm_kmers)},'
|
||||
f'{fn_pct:.4f},{fp_pct:.4f},{cm_pct:.4f}')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Executable
+201
@@ -0,0 +1,201 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Verify the merged count index against all per-specimen reference sets.
|
||||
|
||||
Streams `obikmer dump` once on the merged index, accumulates per-specimen
|
||||
kmer+count pairs from each column, then compares each against its reference .npz.
|
||||
|
||||
Output to stdout: one CSV row per specimen (same columns as verify_count.py)
|
||||
species,strain,ref_kmers,idx_kmers,false_neg,false_pos,count_mismatch,
|
||||
fn_pct,fp_pct,cm_pct
|
||||
"""
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
# ── encoding ──────────────────────────────────────────────────────────────────
|
||||
|
||||
_ENCODE = {'A': 0, 'C': 1, 'G': 2, 'T': 3,
|
||||
'a': 0, 'c': 1, 'g': 2, 't': 3}
|
||||
|
||||
_DECODE = ['A', 'C', 'G', 'T']
|
||||
|
||||
|
||||
def encode_kmer(s: str) -> int:
|
||||
kmer = 0
|
||||
for c in s:
|
||||
kmer = (kmer << 2) | _ENCODE[c]
|
||||
return kmer
|
||||
|
||||
|
||||
def decode_kmer(val: int, k: int) -> str:
|
||||
bases = []
|
||||
for _ in range(k):
|
||||
bases.append(_DECODE[val & 3])
|
||||
val >>= 2
|
||||
return ''.join(reversed(bases))
|
||||
|
||||
|
||||
# ── single-pass dump ──────────────────────────────────────────────────────────
|
||||
|
||||
def stream_merged_dump(obikmer_bin: str, index_dir: str,
|
||||
) -> tuple[list[str], dict[str, tuple[list[int], list[int]]]]:
|
||||
"""Stream the merged dump once.
|
||||
|
||||
Returns:
|
||||
specimen_names : column labels in dump order
|
||||
per_specimen : mapping label → (kmer_ints, counts) for entries > 0
|
||||
"""
|
||||
cmd = [obikmer_bin, 'dump', index_dir]
|
||||
proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL,
|
||||
text=True)
|
||||
|
||||
header_line = proc.stdout.readline().rstrip('\n')
|
||||
cols = header_line.split(',')
|
||||
specimen_names = cols[1:]
|
||||
per_specimen: dict[str, tuple[list[int], list[int]]] = {
|
||||
name: ([], []) for name in specimen_names}
|
||||
|
||||
for line in proc.stdout:
|
||||
parts = line.rstrip('\n').split(',')
|
||||
kmer_int = encode_kmer(parts[0])
|
||||
for i, name in enumerate(specimen_names):
|
||||
count = int(parts[i + 1])
|
||||
if count > 0:
|
||||
per_specimen[name][0].append(kmer_int)
|
||||
per_specimen[name][1].append(count)
|
||||
|
||||
proc.wait()
|
||||
if proc.returncode != 0:
|
||||
print(f'ERROR: obikmer dump exited {proc.returncode}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
return specimen_names, per_specimen
|
||||
|
||||
|
||||
# ── per-specimen comparison ───────────────────────────────────────────────────
|
||||
|
||||
def compare_specimen(name: str,
|
||||
kmer_list: list[int],
|
||||
count_list: list[int],
|
||||
ref_dir: Path,
|
||||
k: int,
|
||||
save_fn: Path | None,
|
||||
save_fp: Path | None,
|
||||
save_cm: Path | None,
|
||||
) -> str:
|
||||
ref_path = ref_dir / f'{name}.npz'
|
||||
if not ref_path.exists():
|
||||
print(f' SKIP {name}: no reference at {ref_path}', file=sys.stderr)
|
||||
return ''
|
||||
|
||||
species = name.split('--')[0]
|
||||
strain = name[len(species) + 2:]
|
||||
|
||||
npz = np.load(ref_path)
|
||||
ref_kmers = npz['kmers'] # sorted uint64
|
||||
ref_counts = npz['counts'] # uint32
|
||||
|
||||
order = np.argsort(np.array(kmer_list, dtype=np.uint64), kind='stable')
|
||||
idx_kmers = np.array(kmer_list, dtype=np.uint64)[order]
|
||||
idx_counts = np.array(count_list, dtype=np.uint32)[order]
|
||||
|
||||
false_neg = np.setdiff1d(ref_kmers, idx_kmers, assume_unique=True)
|
||||
false_pos = np.setdiff1d(idx_kmers, ref_kmers, assume_unique=True)
|
||||
|
||||
# Count mismatches among shared kmers
|
||||
pos_in_idx = np.searchsorted(idx_kmers, ref_kmers)
|
||||
pos_in_idx = np.clip(pos_in_idx, 0, len(idx_kmers) - 1)
|
||||
shared_mask = idx_kmers[pos_in_idx] == ref_kmers
|
||||
mismatch_mask = ref_counts[shared_mask] != idx_counts[pos_in_idx[shared_mask]]
|
||||
cm_kmers = ref_kmers[shared_mask][mismatch_mask]
|
||||
cm_ref = ref_counts[shared_mask][mismatch_mask]
|
||||
cm_idx = idx_counts[pos_in_idx[shared_mask]][mismatch_mask]
|
||||
|
||||
n_shared = int(shared_mask.sum())
|
||||
fn_pct = 100.0 * len(false_neg) / len(ref_kmers) if len(ref_kmers) else 0.0
|
||||
fp_pct = 100.0 * len(false_pos) / len(idx_kmers) if len(idx_kmers) else 0.0
|
||||
cm_pct = 100.0 * len(cm_kmers) / n_shared if n_shared else 0.0
|
||||
|
||||
print(f' {name}: ref={len(ref_kmers):,} idx={len(idx_kmers):,} '
|
||||
f'fn={len(false_neg):,} ({fn_pct:.4f}%) '
|
||||
f'fp={len(false_pos):,} ({fp_pct:.4f}%) '
|
||||
f'cm={len(cm_kmers):,} ({cm_pct:.4f}%)',
|
||||
file=sys.stderr)
|
||||
|
||||
if save_fn and len(false_neg):
|
||||
fn_file = save_fn / f'{name}_fn.txt'
|
||||
fn_file.write_text('\n'.join(decode_kmer(int(v), k) for v in false_neg) + '\n')
|
||||
|
||||
if save_fp and len(false_pos):
|
||||
fp_file = save_fp / f'{name}_fp.txt'
|
||||
fp_file.write_text('\n'.join(decode_kmer(int(v), k) for v in false_pos) + '\n')
|
||||
|
||||
if save_cm and len(cm_kmers):
|
||||
cm_file = save_cm / f'{name}_cm.csv'
|
||||
lines = ['kmer,ref_count,idx_count']
|
||||
for v, rc, ic in zip(cm_kmers, cm_ref, cm_idx):
|
||||
lines.append(f'{decode_kmer(int(v), k)},{rc},{ic}')
|
||||
cm_file.write_text('\n'.join(lines) + '\n')
|
||||
|
||||
return (f'{species},{strain},'
|
||||
f'{len(ref_kmers)},{len(idx_kmers)},'
|
||||
f'{len(false_neg)},{len(false_pos)},{len(cm_kmers)},'
|
||||
f'{fn_pct:.4f},{fp_pct:.4f},{cm_pct:.4f}')
|
||||
|
||||
|
||||
# ── main ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('index', metavar='INDEX_DIR', nargs='?',
|
||||
help='Merged count index directory')
|
||||
ap.add_argument('ref_dir', metavar='REF_DIR', nargs='?',
|
||||
help='Directory containing per-specimen .npz reference files')
|
||||
ap.add_argument('--obikmer', default='obikmer')
|
||||
ap.add_argument('--header', action='store_true',
|
||||
help='Print CSV header and exit')
|
||||
ap.add_argument('--save-fn', metavar='DIR',
|
||||
help='Directory for false-negative kmer lists')
|
||||
ap.add_argument('--save-fp', metavar='DIR',
|
||||
help='Directory for false-positive kmer lists')
|
||||
ap.add_argument('--save-cm', metavar='DIR',
|
||||
help='Directory for count-mismatch CSV files')
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.header:
|
||||
print('species,strain,ref_kmers,idx_kmers,'
|
||||
'false_neg,false_pos,count_mismatch,'
|
||||
'fn_pct,fp_pct,cm_pct')
|
||||
return
|
||||
|
||||
ref_dir = Path(args.ref_dir)
|
||||
save_fn = Path(args.save_fn) if args.save_fn else None
|
||||
save_fp = Path(args.save_fp) if args.save_fp else None
|
||||
save_cm = Path(args.save_cm) if args.save_cm else None
|
||||
for d in (save_fn, save_fp, save_cm):
|
||||
if d: d.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
out1 = subprocess.check_output(
|
||||
[args.obikmer, 'dump', '--head', '1', args.index],
|
||||
stderr=subprocess.DEVNULL, text=True)
|
||||
k = len(out1.splitlines()[1].split(',')[0])
|
||||
|
||||
print(f'k={k} streaming merged dump: {args.index}', file=sys.stderr)
|
||||
specimen_names, per_specimen = stream_merged_dump(args.obikmer, args.index)
|
||||
print(f'{len(specimen_names)} specimen columns loaded', file=sys.stderr)
|
||||
|
||||
for name in specimen_names:
|
||||
kmers, counts = per_specimen[name]
|
||||
row = compare_specimen(name, kmers, counts, ref_dir, k,
|
||||
save_fn, save_fp, save_cm)
|
||||
if row:
|
||||
print(row)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Executable
+27
@@ -0,0 +1,27 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
INDEX="${SCRIPT_DIR}/global_index_count"
|
||||
REF_DIR="${SCRIPT_DIR}/reference_index"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/verify_merge_count"
|
||||
PYTHON="${SCRIPT_DIR}/../.venv/bin/python3"
|
||||
VERIFY_PY="${SCRIPT_DIR}/verify_merge_count.py"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
CURRENT="${STATS_DIR}/current.csv"
|
||||
|
||||
"${PYTHON}" "${VERIFY_PY}" --header >"${CURRENT}"
|
||||
|
||||
"${PYTHON}" "${VERIFY_PY}" \
|
||||
--obikmer "${BINARY}" \
|
||||
"${INDEX}" "${REF_DIR}" \
|
||||
>>"${CURRENT}"
|
||||
|
||||
run_n=$(printf '%03d' "$(find "${STATS_DIR}" -maxdepth 1 -name 'count_*.csv' | wc -l | tr -d ' ')")
|
||||
ARCHIVE="${STATS_DIR}/count_${run_n}.csv"
|
||||
cp "${CURRENT}" "${ARCHIVE}"
|
||||
|
||||
echo "Done. Results → ${ARCHIVE}"
|
||||
Executable
+170
@@ -0,0 +1,170 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Verify the merged presence index against all per-specimen reference sets.
|
||||
|
||||
Streams `obikmer dump` once on the merged index, accumulates per-specimen
|
||||
kmer sets from each column, then compares each against its reference .npz.
|
||||
|
||||
Output to stdout: one CSV row per specimen (same columns as verify_presence.py)
|
||||
species,strain,ref_kmers,idx_kmers,false_neg,false_pos,fn_pct,fp_pct
|
||||
"""
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
# ── encoding ──────────────────────────────────────────────────────────────────
|
||||
|
||||
_ENCODE = {'A': 0, 'C': 1, 'G': 2, 'T': 3,
|
||||
'a': 0, 'c': 1, 'g': 2, 't': 3}
|
||||
|
||||
_DECODE = ['A', 'C', 'G', 'T']
|
||||
|
||||
|
||||
def encode_kmer(s: str) -> int:
|
||||
kmer = 0
|
||||
for c in s:
|
||||
kmer = (kmer << 2) | _ENCODE[c]
|
||||
return kmer
|
||||
|
||||
|
||||
def decode_kmer(val: int, k: int) -> str:
|
||||
bases = []
|
||||
for _ in range(k):
|
||||
bases.append(_DECODE[val & 3])
|
||||
val >>= 2
|
||||
return ''.join(reversed(bases))
|
||||
|
||||
|
||||
# ── single-pass dump ──────────────────────────────────────────────────────────
|
||||
|
||||
def stream_merged_dump(obikmer_bin: str, index_dir: str,
|
||||
) -> tuple[list[str], dict[str, list[int]]]:
|
||||
"""Stream the merged dump once.
|
||||
|
||||
Returns:
|
||||
specimen_names : column labels in dump order (excluding 'kmer')
|
||||
per_specimen : mapping label → list of kmer ints where presence > 0
|
||||
"""
|
||||
cmd = [obikmer_bin, 'dump', index_dir]
|
||||
proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL,
|
||||
text=True)
|
||||
|
||||
header_line = proc.stdout.readline().rstrip('\n')
|
||||
cols = header_line.split(',')
|
||||
specimen_names = cols[1:] # first col is 'kmer'
|
||||
per_specimen: dict[str, list[int]] = {name: [] for name in specimen_names}
|
||||
|
||||
for line in proc.stdout:
|
||||
parts = line.rstrip('\n').split(',')
|
||||
kmer_int = encode_kmer(parts[0])
|
||||
for i, name in enumerate(specimen_names):
|
||||
if int(parts[i + 1]) > 0:
|
||||
per_specimen[name].append(kmer_int)
|
||||
|
||||
proc.wait()
|
||||
if proc.returncode != 0:
|
||||
print(f'ERROR: obikmer dump exited {proc.returncode}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
return specimen_names, per_specimen
|
||||
|
||||
|
||||
# ── per-specimen comparison ───────────────────────────────────────────────────
|
||||
|
||||
def compare_specimen(name: str,
|
||||
kmer_list: list[int],
|
||||
ref_dir: Path,
|
||||
k: int,
|
||||
save_fn: Path | None,
|
||||
save_fp: Path | None,
|
||||
) -> str:
|
||||
"""Compare one specimen column against its reference .npz.
|
||||
|
||||
Returns a CSV row string.
|
||||
"""
|
||||
ref_path = ref_dir / f'{name}.npz'
|
||||
if not ref_path.exists():
|
||||
print(f' SKIP {name}: no reference at {ref_path}', file=sys.stderr)
|
||||
return ''
|
||||
|
||||
species = name.split('--')[0]
|
||||
strain = name[len(species) + 2:]
|
||||
|
||||
ref_kmers = np.load(ref_path)['kmers'] # sorted uint64
|
||||
idx_kmers = np.array(sorted(kmer_list), dtype=np.uint64)
|
||||
|
||||
false_neg = np.setdiff1d(ref_kmers, idx_kmers, assume_unique=True)
|
||||
false_pos = np.setdiff1d(idx_kmers, ref_kmers, assume_unique=True)
|
||||
|
||||
fn_pct = 100.0 * len(false_neg) / len(ref_kmers) if len(ref_kmers) else 0.0
|
||||
fp_pct = 100.0 * len(false_pos) / len(idx_kmers) if len(idx_kmers) else 0.0
|
||||
|
||||
print(f' {name}: ref={len(ref_kmers):,} idx={len(idx_kmers):,} '
|
||||
f'fn={len(false_neg):,} ({fn_pct:.4f}%) '
|
||||
f'fp={len(false_pos):,} ({fp_pct:.4f}%)',
|
||||
file=sys.stderr)
|
||||
|
||||
if save_fn and len(false_neg):
|
||||
fn_file = save_fn / f'{name}_fn.txt'
|
||||
fn_file.write_text('\n'.join(decode_kmer(int(v), k) for v in false_neg) + '\n')
|
||||
|
||||
if save_fp and len(false_pos):
|
||||
fp_file = save_fp / f'{name}_fp.txt'
|
||||
fp_file.write_text('\n'.join(decode_kmer(int(v), k) for v in false_pos) + '\n')
|
||||
|
||||
return (f'{species},{strain},'
|
||||
f'{len(ref_kmers)},{len(idx_kmers)},'
|
||||
f'{len(false_neg)},{len(false_pos)},'
|
||||
f'{fn_pct:.4f},{fp_pct:.4f}')
|
||||
|
||||
|
||||
# ── main ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('index', metavar='INDEX_DIR', nargs='?',
|
||||
help='Merged presence index directory')
|
||||
ap.add_argument('ref_dir', metavar='REF_DIR', nargs='?',
|
||||
help='Directory containing per-specimen .npz reference files')
|
||||
ap.add_argument('--obikmer', default='obikmer')
|
||||
ap.add_argument('--header', action='store_true',
|
||||
help='Print CSV header and exit')
|
||||
ap.add_argument('--save-fn', metavar='DIR',
|
||||
help='Directory to save false-negative kmer lists')
|
||||
ap.add_argument('--save-fp', metavar='DIR',
|
||||
help='Directory to save false-positive kmer lists')
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.header:
|
||||
print('species,strain,ref_kmers,idx_kmers,'
|
||||
'false_neg,false_pos,fn_pct,fp_pct')
|
||||
return
|
||||
|
||||
ref_dir = Path(args.ref_dir)
|
||||
save_fn = Path(args.save_fn) if args.save_fn else None
|
||||
save_fp = Path(args.save_fp) if args.save_fp else None
|
||||
if save_fn: save_fn.mkdir(parents=True, exist_ok=True)
|
||||
if save_fp: save_fp.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Detect k
|
||||
out1 = subprocess.check_output(
|
||||
[args.obikmer, 'dump', '--head', '1', args.index],
|
||||
stderr=subprocess.DEVNULL, text=True)
|
||||
k = len(out1.splitlines()[1].split(',')[0])
|
||||
|
||||
print(f'k={k} streaming merged dump: {args.index}', file=sys.stderr)
|
||||
specimen_names, per_specimen = stream_merged_dump(args.obikmer, args.index)
|
||||
print(f'{len(specimen_names)} specimen columns loaded', file=sys.stderr)
|
||||
|
||||
for name in specimen_names:
|
||||
row = compare_specimen(name, per_specimen[name], ref_dir, k, save_fn, save_fp)
|
||||
if row:
|
||||
print(row)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Executable
+27
@@ -0,0 +1,27 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
INDEX="${SCRIPT_DIR}/global_index_presence"
|
||||
REF_DIR="${SCRIPT_DIR}/reference_index"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/verify_merge_presence"
|
||||
PYTHON="${SCRIPT_DIR}/../.venv/bin/python3"
|
||||
VERIFY_PY="${SCRIPT_DIR}/verify_merge_presence.py"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
CURRENT="${STATS_DIR}/current.csv"
|
||||
|
||||
"${PYTHON}" "${VERIFY_PY}" --header >"${CURRENT}"
|
||||
|
||||
"${PYTHON}" "${VERIFY_PY}" \
|
||||
--obikmer "${BINARY}" \
|
||||
"${INDEX}" "${REF_DIR}" \
|
||||
>>"${CURRENT}"
|
||||
|
||||
run_n=$(printf '%03d' "$(find "${STATS_DIR}" -maxdepth 1 -name 'presence_*.csv' | wc -l | tr -d ' ')")
|
||||
ARCHIVE="${STATS_DIR}/presence_${run_n}.csv"
|
||||
cp "${CURRENT}" "${ARCHIVE}"
|
||||
|
||||
echo "Done. Results → ${ARCHIVE}"
|
||||
Executable
+30
@@ -0,0 +1,30 @@
|
||||
#!/usr/bin/env bash
|
||||
# Usage: verify_one_count.sh SPECIMEN
|
||||
# SPECIMEN = "species--strain" (Make pattern stem)
|
||||
# Output: stats/verify_count/SPECIMEN.stats (one CSV data row, no header)
|
||||
set -euo pipefail
|
||||
|
||||
SPECIMEN="$1"
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
PYTHON="${SCRIPT_DIR}/../.venv/bin/python3"
|
||||
VERIFY_PY="${SCRIPT_DIR}/verify_count.py"
|
||||
|
||||
species="${SPECIMEN%%--*}"
|
||||
strain="${SPECIMEN#*--}"
|
||||
|
||||
REF_NPZ="${SCRIPT_DIR}/reference_index/${SPECIMEN}.npz"
|
||||
INDEX_DIR="${SCRIPT_DIR}/specimen_index_count/${SPECIMEN}"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/verify_count"
|
||||
STATS_FILE="${STATS_DIR}/${SPECIMEN}.stats"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
echo "[${SPECIMEN}] verifying count"
|
||||
|
||||
"${PYTHON}" "${VERIFY_PY}" \
|
||||
--obikmer "${BINARY}" \
|
||||
--species "${species}" \
|
||||
--strain "${strain}" \
|
||||
"${REF_NPZ}" "${INDEX_DIR}" \
|
||||
>"${STATS_FILE}"
|
||||
Executable
+30
@@ -0,0 +1,30 @@
|
||||
#!/usr/bin/env bash
|
||||
# Usage: verify_one_presence.sh SPECIMEN
|
||||
# SPECIMEN = "species--strain" (Make pattern stem)
|
||||
# Output: stats/verify_presence/SPECIMEN.stats (one CSV data row, no header)
|
||||
set -euo pipefail
|
||||
|
||||
SPECIMEN="$1"
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
BINARY="${SCRIPT_DIR}/../src/target/release/obikmer"
|
||||
PYTHON="${SCRIPT_DIR}/../.venv/bin/python3"
|
||||
VERIFY_PY="${SCRIPT_DIR}/verify_presence.py"
|
||||
|
||||
species="${SPECIMEN%%--*}"
|
||||
strain="${SPECIMEN#*--}"
|
||||
|
||||
REF_NPZ="${SCRIPT_DIR}/reference_index/${SPECIMEN}.npz"
|
||||
INDEX_DIR="${SCRIPT_DIR}/specimen_index_presence/${SPECIMEN}"
|
||||
STATS_DIR="${SCRIPT_DIR}/stats/verify_presence"
|
||||
STATS_FILE="${STATS_DIR}/${SPECIMEN}.stats"
|
||||
|
||||
mkdir -p "${STATS_DIR}"
|
||||
|
||||
echo "[${SPECIMEN}] verifying presence"
|
||||
|
||||
"${PYTHON}" "${VERIFY_PY}" \
|
||||
--obikmer "${BINARY}" \
|
||||
--species "${species}" \
|
||||
--strain "${strain}" \
|
||||
"${REF_NPZ}" "${INDEX_DIR}" \
|
||||
>"${STATS_FILE}"
|
||||
Executable
+139
@@ -0,0 +1,139 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Compare an obikmer index against a reference kmer set (presence/absence).
|
||||
|
||||
Loads the reference .npz (sorted uint64 kmers built by build_reference.py),
|
||||
streams the output of `obikmer dump`, encodes each kmer string to uint64,
|
||||
then reports false negatives and false positives using numpy set operations.
|
||||
|
||||
Output to stdout: one CSV row
|
||||
species, strain, ref_kmers, idx_kmers, false_neg, false_pos, fn_pct, fp_pct
|
||||
"""
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
# ── encoding ──────────────────────────────────────────────────────────────────
|
||||
|
||||
_ENCODE = {'A': 0, 'C': 1, 'G': 2, 'T': 3,
|
||||
'a': 0, 'c': 1, 'g': 2, 't': 3}
|
||||
|
||||
_DECODE = ['A', 'C', 'G', 'T']
|
||||
|
||||
|
||||
def encode_kmer(s: str) -> int:
|
||||
kmer = 0
|
||||
for c in s:
|
||||
kmer = (kmer << 2) | _ENCODE[c]
|
||||
return kmer
|
||||
|
||||
|
||||
def decode_kmer(val: int, k: int) -> str:
|
||||
bases = []
|
||||
for _ in range(k):
|
||||
bases.append(_DECODE[val & 3])
|
||||
val >>= 2
|
||||
return ''.join(reversed(bases))
|
||||
|
||||
|
||||
# ── dump parsing ──────────────────────────────────────────────────────────────
|
||||
|
||||
def load_index_kmers(obikmer_bin: str, index_dir: str) -> np.ndarray:
|
||||
"""Stream `obikmer dump` and return a sorted uint64 array of kmer integers."""
|
||||
cmd = [obikmer_bin, 'dump', index_dir]
|
||||
proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL,
|
||||
text=True)
|
||||
kmers = []
|
||||
header = True
|
||||
for line in proc.stdout:
|
||||
if header:
|
||||
header = False
|
||||
continue
|
||||
kmer_str = line.split(',', 1)[0]
|
||||
kmers.append(encode_kmer(kmer_str))
|
||||
proc.wait()
|
||||
if proc.returncode != 0:
|
||||
print(f'ERROR: obikmer dump exited {proc.returncode}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
arr = np.array(kmers, dtype=np.uint64)
|
||||
arr.sort()
|
||||
return arr
|
||||
|
||||
|
||||
# ── comparison ────────────────────────────────────────────────────────────────
|
||||
|
||||
def compare(ref: np.ndarray, idx: np.ndarray) -> tuple[np.ndarray, np.ndarray]:
|
||||
"""Return (false_negatives, false_positives) as uint64 arrays."""
|
||||
false_neg = np.setdiff1d(ref, idx, assume_unique=True)
|
||||
false_pos = np.setdiff1d(idx, ref, assume_unique=True)
|
||||
return false_neg, false_pos
|
||||
|
||||
|
||||
# ── main ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('reference', metavar='REF_NPZ', nargs='?', help='Reference .npz file')
|
||||
ap.add_argument('index', metavar='INDEX_DIR', nargs='?', help='obikmer index directory')
|
||||
ap.add_argument('--obikmer', default='obikmer', help='Path to obikmer binary')
|
||||
ap.add_argument('--species', default='', help='Species label for CSV row')
|
||||
ap.add_argument('--strain', default='', help='Strain label for CSV row')
|
||||
ap.add_argument('--header', action='store_true', help='Print CSV header and exit')
|
||||
ap.add_argument('--save-fp', metavar='FILE',
|
||||
help='Save false-positive kmer strings to FILE')
|
||||
ap.add_argument('--save-fn', metavar='FILE',
|
||||
help='Save false-negative kmer strings to FILE')
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.header:
|
||||
print('species,strain,ref_kmers,idx_kmers,'
|
||||
'false_neg,false_pos,fn_pct,fp_pct')
|
||||
return
|
||||
|
||||
# Detect k from the index (one cheap call before the full dump).
|
||||
cmd1 = [args.obikmer, 'dump', '--head', '1', args.index]
|
||||
out1 = subprocess.check_output(cmd1, stderr=subprocess.DEVNULL, text=True)
|
||||
k = len(out1.splitlines()[1].split(',')[0])
|
||||
|
||||
# Load reference
|
||||
print(f'Loading reference: {args.reference}', file=sys.stderr)
|
||||
npz = np.load(args.reference)
|
||||
ref_kmers = npz['kmers'] # already sorted uint64
|
||||
|
||||
# Load index
|
||||
print(f'Streaming dump (k={k}): {args.index}', file=sys.stderr)
|
||||
idx_kmers = load_index_kmers(args.obikmer, args.index)
|
||||
|
||||
print(f'k={k} ref={len(ref_kmers):,} idx={len(idx_kmers):,}', file=sys.stderr)
|
||||
|
||||
false_neg, false_pos = compare(ref_kmers, idx_kmers)
|
||||
|
||||
fn_pct = 100.0 * len(false_neg) / len(ref_kmers) if len(ref_kmers) else 0.0
|
||||
fp_pct = 100.0 * len(false_pos) / len(idx_kmers) if len(idx_kmers) else 0.0
|
||||
|
||||
print(f'false negatives: {len(false_neg):,} ({fn_pct:.4f}%)', file=sys.stderr)
|
||||
print(f'false positives: {len(false_pos):,} ({fp_pct:.4f}%)', file=sys.stderr)
|
||||
|
||||
if args.save_fn and len(false_neg):
|
||||
with open(args.save_fn, 'w') as fh:
|
||||
for v in false_neg:
|
||||
fh.write(decode_kmer(int(v), k) + '\n')
|
||||
print(f'False negatives saved → {args.save_fn}', file=sys.stderr)
|
||||
|
||||
if args.save_fp and len(false_pos):
|
||||
with open(args.save_fp, 'w') as fh:
|
||||
for v in false_pos:
|
||||
fh.write(decode_kmer(int(v), k) + '\n')
|
||||
print(f'False positives saved → {args.save_fp}', file=sys.stderr)
|
||||
|
||||
print(f'{args.species},{args.strain},'
|
||||
f'{len(ref_kmers)},{len(idx_kmers)},'
|
||||
f'{len(false_neg)},{len(false_pos)},'
|
||||
f'{fn_pct:.4f},{fp_pct:.4f}')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,323 @@
|
||||
# NUMA-aware partition runner
|
||||
|
||||
## Problem
|
||||
|
||||
All partition-level parallel loops in obikindex currently fall into two
|
||||
categories:
|
||||
|
||||
**Naive Rayon** — used in `build_layers`, `pack_matrices`, `dump`, `select`,
|
||||
`stats`, `rebuild`, `reindex`:
|
||||
|
||||
```rust
|
||||
(0..n).into_par_iter().for_each(|i| work(i));
|
||||
```
|
||||
|
||||
Threads come from the global Rayon pool with no NUMA awareness. On
|
||||
multi-socket machines this produces cross-socket memory traffic and degrades
|
||||
performance super-linearly (see [NUMA-aware worker pools](numa_worker_pools.md)).
|
||||
|
||||
**Ad-hoc adaptive pool** — used in `merge`:
|
||||
|
||||
A bespoke implementation with pre-spawned workers, channel-based dispatch, and
|
||||
activation control. It handles NUMA correctly but is not reusable.
|
||||
|
||||
Both cases should be replaced by a single generic mechanism.
|
||||
|
||||
## Unified model
|
||||
|
||||
The key insight is that **UMA is just the NUMA case with a single node**. The
|
||||
runner always works the same way: one controller thread per node, each
|
||||
independently managing its own workers with the same adaptive logic. The only
|
||||
difference between UMA and NUMA is the number of nodes and whether workers are
|
||||
pinned.
|
||||
|
||||
```
|
||||
NUMA (k nodes) UMA (1 node)
|
||||
|
||||
controller-0 controller-1 … controller-0
|
||||
│ │ │
|
||||
workers[0] workers[1] workers[0]
|
||||
(pinned) (pinned) (global pool)
|
||||
└───────────────┴──────────────────┘
|
||||
shared work queue
|
||||
```
|
||||
|
||||
On each node, the Rayon `ThreadPool` is pinned to that node's CPUs.
|
||||
`pool.install()` ensures all internal Rayon calls (inside the work function)
|
||||
use the node-local pool. Linux first-touch then places heap allocations in
|
||||
local DRAM automatically.
|
||||
|
||||
On UMA the global Rayon pool is used directly — no pinning, no overhead.
|
||||
|
||||
## Adaptive mechanism
|
||||
|
||||
Each controller follows the same logic regardless of node count:
|
||||
|
||||
1. Pre-spawn `workers_per_node` dormant worker threads (blocked on `activate_rx`).
|
||||
2. Activate the first worker immediately.
|
||||
3. Loop on result channel with a `SPAWN_POLL` timeout:
|
||||
- On result: call `on_done`; check whether to activate the next worker.
|
||||
- On timeout: same check.
|
||||
- Activation criterion: `should_spawn_worker(active, global_efficiency, prev_efficiency)`.
|
||||
4. Drop `activate_tx` when done — dormant workers exit cleanly.
|
||||
|
||||
**Global CPU efficiency** (`CpuSample`, reads `/proc/stat` on Linux) is used by
|
||||
all controllers — no per-node measurement needed. The signal is coarser than
|
||||
per-node efficiency but correct in practice: if any node saturates memory
|
||||
bandwidth, the global efficiency drops and all controllers stop activating
|
||||
workers. Using a standard portable primitive avoids platform-specific CPU
|
||||
accounting and keeps the implementation clean.
|
||||
|
||||
## Proposed API
|
||||
|
||||
```rust
|
||||
pub struct PartitionRunner {
|
||||
// One entry per NUMA node; one entry total on UMA.
|
||||
nodes: Vec<NodeConfig>,
|
||||
}
|
||||
|
||||
struct NodeConfig {
|
||||
pool: Option<Arc<rayon::ThreadPool>>, // None = global Rayon pool (UMA)
|
||||
cpu_ids: Vec<usize>, // empty = no pinning (UMA)
|
||||
max_workers: usize,
|
||||
}
|
||||
|
||||
impl PartitionRunner {
|
||||
/// Detect topology and build the runner.
|
||||
/// Returns a single-node runner on UMA / macOS / hwloc failure.
|
||||
pub fn new() -> Self;
|
||||
|
||||
/// Run `f(i)` for every index in `order`, collecting results.
|
||||
///
|
||||
/// `on_done(i, result, elapsed)` is called under an internal mutex as
|
||||
/// each partition completes — use it for progress bars and aggregation.
|
||||
/// The runner serialises all calls to `on_done` via an internal
|
||||
/// `Arc<Mutex<C>>`, so no `Sync` bound is required on the callback.
|
||||
/// `Send` is required because the Arc clone crosses thread boundaries.
|
||||
///
|
||||
/// Serialisation is free in practice: a partition takes seconds to
|
||||
/// minutes; the callback takes microseconds. Contention is negligible.
|
||||
///
|
||||
/// Returns the first error from `f`, if any.
|
||||
pub fn run<F, R, E, C>(
|
||||
&self,
|
||||
order: &[usize],
|
||||
f: F,
|
||||
on_done: C,
|
||||
) -> Result<(), E>
|
||||
where
|
||||
F: Fn(usize) -> Result<R, E> + Send + Sync,
|
||||
R: Send,
|
||||
E: Send,
|
||||
C: FnMut(usize, R, Duration) + Send; // Send required, Sync is not
|
||||
}
|
||||
```
|
||||
|
||||
`order` is caller-supplied so each command chooses its scheduling strategy:
|
||||
largest-first for `merge`, sequential for `build_layers`, etc.
|
||||
|
||||
## Migration examples
|
||||
|
||||
### merge.rs (before: ~180 lines of bespoke machinery)
|
||||
|
||||
```rust
|
||||
let runner = PartitionRunner::new();
|
||||
runner.run(
|
||||
&order,
|
||||
|i| dst_partition.merge_partition(i, srcs, mode, n_dst_genomes, block_bits, evidence)
|
||||
.map_err(OKIError::Partition),
|
||||
|i, g_len, dur| {
|
||||
pb.inc(1);
|
||||
debug!("partition {i}: done in {:.1}s — {g_len} new kmers", dur.as_secs_f64());
|
||||
part_stats.push(PartStat { id: i, unitig_bytes: partition_sizes[i], g_len });
|
||||
},
|
||||
)?;
|
||||
```
|
||||
|
||||
### index.rs build_layers (before: naive into_par_iter)
|
||||
|
||||
```rust
|
||||
let order: Vec<usize> = (0..n).collect();
|
||||
let runner = PartitionRunner::new();
|
||||
runner.run(
|
||||
&order,
|
||||
|i| self.partition.build_index_layer(i, min_ab, max_ab, with_counts, &evidence, block_bits)
|
||||
.map_err(OKIError::Partition),
|
||||
|_, n_kmers, _| {
|
||||
total_kmers.fetch_add(n_kmers, Ordering::Relaxed);
|
||||
pb.inc(1);
|
||||
},
|
||||
)?;
|
||||
```
|
||||
|
||||
All other sites (`pack_matrices`, `dump`, `select`, etc.) follow the same
|
||||
pattern.
|
||||
|
||||
## Placement
|
||||
|
||||
`PartitionRunner` lives in `obikindex/src/numa.rs` alongside `NumaSetup`.
|
||||
It depends only on standard library primitives and Rayon — no new dependencies.
|
||||
|
||||
A single `PartitionRunner` instance can be built once per command invocation
|
||||
and reused across multiple `run()` calls (e.g. `merge` runs
|
||||
`merge_partitions` then `pack_matrices`).
|
||||
|
||||
## Known issue: CPU-only activation signal stalls on I/O-bound stages
|
||||
|
||||
Observed on a real `filter` run (109 genomes, 256 partitions, 8×24-core NUMA):
|
||||
`rebuild` (CPU-bound — k-mer construction) scales cleanly from 9 to 43 active
|
||||
workers as `CpuSample::do_i_activate` (`obisys::lib.rs`) sees efficiency climb.
|
||||
`pack_matrices` (I/O-bound — reopens and recomposes per-genome column files
|
||||
into `.pbmx`/`.pcmx`) activates one extra worker then flatlines at 10/192 for
|
||||
the rest of the stage, even though 256 partitions keep completing over several
|
||||
minutes. This matches the documented intent (§ Adaptive mechanism — "avoids
|
||||
over-provisioning ... I/O-bound ... workloads") but conflates two different
|
||||
things: *"CPU is not the bottleneck"* and *"more workers would not help"*. On
|
||||
storage with real queue depth (NVMe, RAID, parallel FS) the second stage could
|
||||
still benefit from more concurrent workers even with flat CPU usage — a signal
|
||||
the current mechanism cannot see.
|
||||
|
||||
A one-off artefact was also found in the same log: right after a stage
|
||||
transition, `do_i_activate` produced a physically impossible spike (efficiency
|
||||
~94 cores on a 192-core box) because it has no minimum-window guard — unlike
|
||||
its sibling `cpu_efficiency`, which returns `0.0` if `wall < 0.1s`
|
||||
(`obisys::lib.rs:260`). `do_i_activate` unconditionally overwrites
|
||||
`self.wall`/`self.user_secs`/`self.sys_secs` even when the elapsed window is
|
||||
too short to be meaningful, so a burst of rapid completions right after
|
||||
activating a worker can divide a real CPU delta by a near-zero wall delta.
|
||||
|
||||
### Implemented: I/O signal + shared debounce guard
|
||||
|
||||
`IoSample` (`obisys::lib.rs`, alongside `CpuSample`) is fed by
|
||||
`read_bytes`/`write_bytes` from `/proc/self/io` on Linux (actual bytes
|
||||
submitted to the block layer — not `rchar`/`wchar`, which also count
|
||||
page-cache hits, and not `ru_inblock`/`ru_oublock`, unreliable on macOS), with
|
||||
a `proc_pid_rusage(RUSAGE_INFO_V4)` fallback on macOS
|
||||
(`ri_diskio_bytesread`/`ri_diskio_byteswritten`, FFI only via `libc`, no new
|
||||
dependency — same pattern as the existing `getrusage` bindings). Any other
|
||||
target degrades gracefully to a signal that never triggers (falls back to
|
||||
CPU-only activation), same pattern as `cgroup_v2_available`.
|
||||
|
||||
`maybe_activate` (`numa.rs`) activates a worker if *either* signal still shows
|
||||
headroom, making `PartitionRunner` adapt to whichever resource is actually the
|
||||
bottleneck without per-call configuration. Both samplers are called
|
||||
unconditionally — no `||` short-circuit — so neither window starves behind
|
||||
whichever signal fires first:
|
||||
|
||||
```rust
|
||||
let cpu_threshold = CPU_SPAWN_THRESHOLD * activation.last_step() as f64;
|
||||
let cpu_wants_more = cpu_sample.do_i_activate(cpu_threshold);
|
||||
let io_wants_more = io_sample.do_i_activate(IO_SPAWN_THRESHOLD);
|
||||
if cpu_wants_more || io_wants_more {
|
||||
activation.grow(GROWTH_DIVISOR, n_total);
|
||||
}
|
||||
```
|
||||
|
||||
The CPU threshold is *not* the flat absolute delta it started as: it scales
|
||||
with `activation.last_step()` — the number of workers activated in the last
|
||||
growth step, tracked by `NodeActivation` (`numa.rs`) and updated every time
|
||||
`grow()` actually grows something. Growing by 8 workers should add ~8 cores of
|
||||
efficiency if the workload is truly CPU-bound; requiring only
|
||||
`CPU_SPAWN_THRESHOLD` (20 %) of that expected gain confirms the growth was
|
||||
useful without demanding perfect linear scaling. Scaling by the *last step's
|
||||
size* rather than the cumulative total keeps the bar equally meaningful
|
||||
whether it's the 2nd growth step or the 20th — a flat absolute threshold
|
||||
(0.2 core) is a strong signal at 8 active workers but pure noise at 150; a
|
||||
threshold scaled by the *cumulative* total instead (considered and rejected)
|
||||
would have made the bar essentially impossible to clear late in the ramp,
|
||||
strangling exactly the CPU-bound saturation the mechanism exists to allow.
|
||||
|
||||
Unlike the CPU signal (an absolute delta in cores — a bounded, portable unit),
|
||||
raw I/O throughput has no natural scale across devices, so `IoSample` uses a
|
||||
**relative** growth threshold instead of an absolute one:
|
||||
|
||||
```rust
|
||||
pub fn do_i_activate(&mut self, threshold: f64) -> bool {
|
||||
let elapsed = self.wall.elapsed().as_secs_f64();
|
||||
if elapsed < 0.1 { return false; } // state untouched — window keeps accumulating
|
||||
|
||||
let n = Self::read_bytes();
|
||||
let rate = n.saturating_sub(self.bytes) as f64 / elapsed;
|
||||
let activate = if self.previous_rate == 0.0 {
|
||||
rate > 0.0 // bootstrap: any measured throughput is signal
|
||||
} else {
|
||||
(rate - self.previous_rate) / self.previous_rate >= threshold
|
||||
};
|
||||
|
||||
self.bytes = n;
|
||||
self.wall = Instant::now(); // reset only on a real sample
|
||||
activate
|
||||
}
|
||||
```
|
||||
|
||||
The `elapsed < 0.1s → return false without mutating state` guard was also
|
||||
back-ported into `CpuSample::do_i_activate` (previously missing — source of
|
||||
the ~94-core artefact above) — one fix for both problems, and it removes the
|
||||
need for any arbitrary I/O-rate floor: a short/noisy window is rejected
|
||||
outright rather than papered over with a hardware-dependent constant.
|
||||
|
||||
Both spawn thresholds (`CPU_SPAWN_THRESHOLD`, `IO_SPAWN_THRESHOLD`, module-level
|
||||
`const` in `numa.rs`, both `0.2`) are a starting point, not a derived value:
|
||||
`0.2` (20 % relative growth) for `IoSample` was chosen to match the CPU
|
||||
threshold's *implicit* relative sensitivity (in the observed log, an 8→9
|
||||
worker step raised efficiency by ~12 %) — but I/O throughput is lumpier than
|
||||
CPU time (buffered writes flush in bursts), so it needs empirical validation
|
||||
against a real `pack` run before being considered final.
|
||||
|
||||
## Known issue: ramp-up too slow, and confused with node count
|
||||
|
||||
The original design started `n_nodes` workers (one per node) and grew one
|
||||
worker at a time. On a real `filter` run this took ~10 minutes to climb from
|
||||
9 to ~40 active workers even on the CPU-bound `rebuild` stage — most of a
|
||||
35-minute stage spent under-provisioned while waiting for evidence to
|
||||
accumulate one worker at a time. There is no scale-down mechanism (`n_active`
|
||||
only grows), so the original caution was deliberate — but a quarter of
|
||||
available cores is still far from saturation, and the real risk zone (over-provisioning
|
||||
a memory-bandwidth-bound stage) only shows up much later in the ramp, near
|
||||
full occupancy — not at 25 %.
|
||||
|
||||
The fix decouples ramp speed from node *count*: both the initial size and the
|
||||
growth step are a fraction of `workers_per_node` (node *size*), applied
|
||||
identically on every node. A single-NUMA-node (UMA) machine ramps exactly as
|
||||
fast as an 8-node one — growing by `n_nodes` per step, as first considered,
|
||||
would have degenerated to "grow by 1" on UMA, reproducing the original
|
||||
problem for exactly the machines that need the fix most.
|
||||
|
||||
```rust
|
||||
// NodeActivation::grow — called both at startup (activate_initial) and on
|
||||
// every CPU/IO-triggered growth step, with a different divisor each time.
|
||||
let wanted = (self.caps[idx] / divisor).max(1); // INITIAL_DIVISOR=4 at startup, GROWTH_DIVISOR=8 per step
|
||||
let room = self.caps[idx].saturating_sub(self.active[idx]);
|
||||
let grow = wanted.min(room).min(n_total.saturating_sub(self.total));
|
||||
```
|
||||
|
||||
This also fixed a latent correctness gap: the original single shared
|
||||
`activate_tx`/`activate_rx` pair had *no* per-node addressing — sending one
|
||||
activation signal woke up whichever dormant worker (from any node) happened
|
||||
to win the race on that channel. `crossbeam_channel` gives no fairness
|
||||
guarantee across competing receivers, so "round-robin across nodes" was an
|
||||
assumption the code never actually enforced. `PartitionRunner::run` now opens
|
||||
one activation channel per node (`activate_txs`/`activate_rxs`, one pair per
|
||||
`NodeConfig`); `NodeActivation` (`numa.rs`) tracks how many of each node's
|
||||
dormant workers have been woken and grows every node by the same amount per
|
||||
step, capped by that node's remaining dormant workers and by the run's total
|
||||
budget (`n_total`) — balance across nodes is now guaranteed by construction,
|
||||
not incidental to channel implementation details.
|
||||
|
||||
## Open questions
|
||||
|
||||
- **Error handling**: `run` currently returns the first error; remaining errors
|
||||
are dropped. A `Vec<E>` return would give complete diagnostics.
|
||||
|
||||
- **`INITIAL_DIVISOR` / `GROWTH_DIVISOR` tuning**: currently `4` and `8`
|
||||
(start at 1/4 of a node's cores, grow by 1/8 per step), chosen to fix an
|
||||
observed too-slow ramp — not yet validated against a real `pack` (I/O-bound)
|
||||
run, where over-provisioning risk is different from the CPU-bound `rebuild`
|
||||
case this was tuned against.
|
||||
|
||||
- **`on_done` ordering**: the runner serialises calls to `on_done` via an
|
||||
internal `Arc<Mutex<C>>`. `Send` is required (the Arc clone crosses thread
|
||||
boundaries); `Sync` is not (only one thread holds the lock at a time).
|
||||
Contention is negligible because a partition takes seconds while the callback
|
||||
takes microseconds. The callback is therefore simple to write (plain
|
||||
`Vec::push`, plain `FnMut`) with no measurable performance cost.
|
||||
@@ -0,0 +1,97 @@
|
||||
# NUMA-aware worker pools for merge
|
||||
|
||||
## Problem
|
||||
|
||||
The merge command's bottleneck is `compute_degrees` in `obidebruinj`: a random pointer-chase over 20–70 M node hash maps that saturates DRAM bandwidth. When multiple partition workers run concurrently, they contend for the shared memory bus, causing super-linear slowdown (measured: 0.016 µs/node solo → 0.95 µs/node with 4–5 concurrent workers, ×60 degradation).
|
||||
|
||||
Modern HPC nodes are multi-socket NUMA machines (observed: 2 sockets × 4 NUMA nodes × 24 cores = 192 cores). Cross-NUMA memory traffic compounds the contention:
|
||||
|
||||
- Full 192-core run: ~15 min/partition (×10 worse than M3 Mac)
|
||||
- `taskset` restricted to 4 NUMA nodes (96 cores): ~90 s/partition
|
||||
- OAR job on 1 NUMA node (24 cores): ~80 s/partition, same throughput as 96 cores
|
||||
|
||||
**Conclusion**: the bottleneck is memory bandwidth per NUMA node, not core count. 24 cores on one NUMA node achieve the same throughput as 96 cores across four.
|
||||
|
||||
## Strategy
|
||||
|
||||
Run N worker groups in parallel, one per NUMA node, each with its own Rayon thread pool whose threads are pinned to the NUMA node's CPUs. Linux's first-touch policy then places graph allocations on local DRAM automatically — no explicit NUMA allocator needed.
|
||||
|
||||
Expected throughput: N × single-NUMA throughput. On the 8-NUMA-node HPC: 8 × ~80 s = 9–10 min total instead of >60 min with the current single-pool approach.
|
||||
|
||||
## Rayon thread pool isolation
|
||||
|
||||
Rayon provides `ThreadPool::install(|| { ... })`: any Rayon call (`par_iter`, `current_num_threads`, etc.) inside the closure uses *that* pool exclusively. Wrapping `merge_partition` in `pool.install()` redirects all downstream Rayon calls — including those in `debruijn.rs` and `partition.rs` — without touching those crates.
|
||||
|
||||
```rust
|
||||
// worker thread, assigned to NUMA pool `pool`
|
||||
pool.install(|| {
|
||||
dst_partition.merge_partition(i, srcs, mode, n_dst_genomes, block_bits, evidence)
|
||||
})
|
||||
```
|
||||
|
||||
`rayon::current_num_threads()` inside `merge_partition` will return the pool size (e.g. 24), not the global thread count — which is the right value for buffer sizing.
|
||||
|
||||
## Thread pinning
|
||||
|
||||
`ThreadPoolBuilder::spawn_handler` provides a hook executed for each thread at creation. Inside, `libc::sched_setaffinity` pins the thread to a CPU set:
|
||||
|
||||
```rust
|
||||
let cpus: Vec<usize> = numa_node_cpus(node); // from /sys/devices/system/node/nodeN/cpulist
|
||||
rayon::ThreadPoolBuilder::new()
|
||||
.num_threads(cpus.len())
|
||||
.spawn_handler(move |thread| {
|
||||
let mut b = std::thread::Builder::new();
|
||||
std::thread::Builder::new().spawn(move || {
|
||||
pin_to_cpus(&cpus); // sched_setaffinity via libc
|
||||
thread.run()
|
||||
})?;
|
||||
Ok(())
|
||||
})
|
||||
.build()?
|
||||
```
|
||||
|
||||
NUMA topology is read from `/sys/devices/system/node/node*/cpulist` — no `libnuma` dependency required. If the `numa` crate is linked, `numa_available()` / `numa_run_on_node()` are an alternative.
|
||||
|
||||
## Memory locality
|
||||
|
||||
Linux allocates pages on the NUMA node of the thread that first writes them (first-touch policy). Once Rayon threads are pinned to node N, all graph data built by those threads lands on node N's DRAM. No changes to the allocator, no explicit `numa_alloc_onnode` calls.
|
||||
|
||||
## Adaptive spawn criterion
|
||||
|
||||
The current criterion uses `std::thread::available_parallelism()` (returns total cores = 192) and `max_workers = n_cores / 2`. With NUMA pools:
|
||||
|
||||
- `n_cores` per pool = cores per NUMA node (e.g. 24)
|
||||
- `max_workers` per pool = pool size / 2 (e.g. 12)
|
||||
- CPU efficiency is measured per pool, not globally
|
||||
|
||||
Each NUMA group runs its own independent adaptive pool. Workers are distributed across NUMA groups round-robin or by workload (partition assignment can be pre-split by NUMA group index).
|
||||
|
||||
## Required changes
|
||||
|
||||
| File | Change |
|
||||
|------|--------|
|
||||
| `obikindex/src/merge.rs` | Detect NUMA topology; build N `ThreadPool`s with pinned threads; assign each pre-spawned worker to a pool; wrap `merge_partition` in `pool.install()` |
|
||||
| `obikindex/src/merge.rs` | Replace `available_parallelism()` with per-NUMA core count for spawn criterion |
|
||||
| `obikpartitionner/src/merge_layer.rs` | No change — `merge_partition` already works inside any Rayon context |
|
||||
| `obidebruinj/src/debruijn.rs` | No change — `par_iter` and `current_num_threads` are pool-context-aware |
|
||||
| `obikpartitionner/src/partition.rs` | No change — same reason |
|
||||
|
||||
## Platform guard
|
||||
|
||||
NUMA pinning is Linux-only. The fallback is the current single global pool:
|
||||
|
||||
```rust
|
||||
#[cfg(target_os = "linux")]
|
||||
fn build_numa_pools() -> Option<Vec<rayon::ThreadPool>> { ... }
|
||||
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
fn build_numa_pools() -> Option<Vec<rayon::ThreadPool>> { None }
|
||||
```
|
||||
|
||||
When `build_numa_pools()` returns `None` (macOS, UMA, or single-socket), `merge.rs` uses the existing code path unchanged.
|
||||
|
||||
## Open questions
|
||||
|
||||
- **Partition assignment**: split partitions by NUMA group up-front (static) or use a shared queue with per-group workers stealing from a common pool? Static split is simpler; stealing is better for load balance when partitions vary widely in size.
|
||||
- **Intra-NUMA adaptive criterion**: with 24 cores and ~3–5 effective workers per NUMA node, the current marginal-gain criterion needs re-tuning or can be left as-is with per-pool `n_cores = 24`.
|
||||
- **I/O**: partition data (unitig files) is on a shared filesystem. With 8 concurrent NUMA groups, I/O concurrency increases 8× — need to verify the filesystem (Lustre or local SSD) can absorb it without becoming the new bottleneck.
|
||||
+255
-38
@@ -16,27 +16,43 @@ Given a set of query sequences, determine for each sequence how many of its k-me
|
||||
|
||||
## Algorithm
|
||||
|
||||
The query follows the same superkmer-based partitioning strategy used at indexing time.
|
||||
The query follows the same superkmer-based partitioning strategy used at indexing time. Everything below happens inside `process_chunk` (`query.rs`); there is no separate per-stage function, but the internal data flow is staged: k-mer-level dereplication, a two-part MPHF/column-major matrix lookup (`obikpartitionner::query_partition_with`), and a sparse Findere pass, each producing sparse intermediate structures rather than one dense allocation for the whole chunk.
|
||||
|
||||
```
|
||||
for each chunk of sequences (parallel workers via obipipeline):
|
||||
build QueryBatch: decompose all sequences into s-mers via superkmers, deduplicate
|
||||
allocate seq_results[seq_idx][smer_pos] = None ← per-sequence s-mer result vectors
|
||||
split superkmers by partition via minimiser hash
|
||||
for each chunk of sequences (parallel workers via obipipeline, one call to process_chunk):
|
||||
build QueryBatch (QueryBatch::from_records):
|
||||
decompose all sequences into superkmers (SuperKmerIter) — construction only,
|
||||
not the dedup key
|
||||
deduplicate at k-mer granularity, split by partition in the same pass:
|
||||
by_partition: Vec<HashMap<CanonicalKmer, Vec<KmerDesc>>> ← KmerDesc = (seq_idx, pos)
|
||||
allocate SmerIndex (SmerIndex::new): in_index: Vec<bool>, sized total_smers —
|
||||
NOT multiplied by n_genomes
|
||||
allocate by_genome: Vec<Vec<(seq_idx, pos, value)>>, one empty Vec per genome —
|
||||
stays empty (zero cost) for every genome this chunk never matches
|
||||
for each partition p:
|
||||
query_partition(p, superkmers_routed_to_p)
|
||||
→ load QueryLayer(s) for p
|
||||
→ for each s-mer in each superkmer: MphfLayer::find(smer)
|
||||
fill seq_results[seq_idx][kmer_offset + j] from partition results
|
||||
for each sequence:
|
||||
apply_findere(seq_results[seq_idx], effective_z) ← per full sequence
|
||||
accumulate confirmed k-mer results into acc and cov
|
||||
emit annotated sequences
|
||||
query_partition_with(p, kmers_for_p, on_event):
|
||||
stage 1 (MPHF-only): for each unique k-mer, try each layer's MphfLayer::find
|
||||
in turn, stop at the first hit; bucket confirmed hits by (layer, slot);
|
||||
emit QueryHit::Found(descs) once per hit k-mer
|
||||
stage 2 (column-major fetch): for each layer with ≥1 hit, for each genome
|
||||
column g in 0..layer.n_cols(): scan that layer's bucketed slots, look up
|
||||
col_value(g, slot); emit QueryHit::Value(descs, g, value) on nonzero
|
||||
on_event dispatches: Found → SmerIndex::mark_found for every desc;
|
||||
Value → push (seq_idx, pos, value) into by_genome[g]
|
||||
for each genome g with ≥1 hit (sparse_findere_for_genome):
|
||||
sort by_genome[g] by (seq_idx, pos); detect maximal runs of consecutive pos
|
||||
within one seq_idx; monotone-deque window-minimum scoped to each run →
|
||||
confirmed_by_genome[g]: Vec<(seq_idx, pos_out, value)>
|
||||
accumulate genome_totals per sequence from confirmed_by_genome (per genome, direct)
|
||||
accumulate kmer_count / kmer_missing per (sequence, output position), O(1) each,
|
||||
using only the confirmed-any bitmap and SmerIndex — independent of n_genomes
|
||||
if --detail: densify confirmed_by_genome into per-(seq, genome) coverage arrays
|
||||
emit annotated sequences (emit_batch)
|
||||
```
|
||||
|
||||
Superkmers that appear more than once in the batch (same sequence or across sequences) are deduplicated: each unique `RoutableSuperKmer` is queried once per partition, and the result is broadcast to every `SKDesc` entry that references it.
|
||||
Superkmers that appear more than once in the batch (same sequence or across sequences), or different superkmers that happen to share a k-mer (read overlaps, repeats, a SNP splitting an otherwise-identical run), are deduplicated at k-mer granularity: each unique `CanonicalKmer` triggers at most one MPHF lookup and, on hit, one matrix fetch, broadcast to every `KmerDesc` occurrence referencing it.
|
||||
|
||||
**Findere requires full-sequence aggregation.** `apply_findere` is applied once per sequence on the complete s-mer result vector, after all partitions have contributed. Applying it per superkmer would produce false negatives at superkmer boundaries, where the z-window spans two superkmers.
|
||||
**Findere requires full-sequence aggregation.** The sliding window (now per-run, not per-sequence — see [Findere z-window filter](#findere-z-window-filter)) only ever runs after all partitions have contributed their hits to `by_genome`. Applying it per superkmer would produce false negatives at superkmer boundaries, where the z-window spans two superkmers.
|
||||
|
||||
Batches are processed in parallel via `obipipeline` workers; the `--threads` flag controls the number of worker threads.
|
||||
|
||||
@@ -44,25 +60,46 @@ Batches are processed in parallel via `obipipeline` workers; the `--threads` fla
|
||||
|
||||
## Findere z-window filter
|
||||
|
||||
For approximate index modes, the index physically stores s-mers of size `s = k_user − z + 1`. At query time, `set_k(s)` is in effect, so queries naturally produce s-mer results. `apply_findere` then aggregates z consecutive s-mer results into one k_user-mer answer:
|
||||
For approximate index modes, the index physically stores s-mers of size `s = k_user − z + 1`; `idx.kmer_size()` (bound to `k` in `process_chunk`) is this physically-indexed s-mer size, so decomposing the query at `k` naturally produces s-mer results.
|
||||
|
||||
```rust
|
||||
fn apply_findere(
|
||||
results: &[Option<Box<[u32]>>], // N s-mer results
|
||||
z: usize,
|
||||
n_genomes: usize,
|
||||
) -> Vec<Option<Box<[u32]>>> // N − z + 1 k_user-mer results
|
||||
The z-window aggregation is **sparse**, per genome, implemented in `sparse_findere_for_genome` (`query.rs`) — a run-detection pass followed by a monotone-deque sliding-window minimum scoped to each run, not a dense scan over every s-mer position of every sequence:
|
||||
|
||||
```
|
||||
sparse_findere_for_genome(hits, z, presence, threshold):
|
||||
// hits: raw (seq_idx, pos_smer, value) triples for this genome, as delivered
|
||||
// by query_partition_with's QueryHit::Value — only ever nonzero entries;
|
||||
// a position with no hit for this genome simply has no entry at all.
|
||||
sort hits by (seq_idx, pos_smer)
|
||||
|
||||
for each maximal run of consecutive pos_smer values within the same seq_idx:
|
||||
dq: VecDeque<(run-relative index, value)>
|
||||
for k, (_, pos, value) in enumerate(run):
|
||||
maintain dq monotone non-decreasing (pop back while back.value >= value)
|
||||
push (k, value)
|
||||
evict dq entries with run-relative index <= k - z
|
||||
if k + 1 >= z:
|
||||
win_min = dq.front().value
|
||||
if win_min > 0:
|
||||
pos_out = pos + 1 - z
|
||||
confirmed.push((seq_idx, pos_out, adjust(win_min)))
|
||||
return confirmed
|
||||
```
|
||||
|
||||
Input length N (s-mers), output length N − z + 1 (k_user-mers).
|
||||
A window can only be confirmed (`win_min > 0`) when all `z` s-mers in it are present *and* nonzero for this genome — which, by construction, can only happen strictly inside one contiguous run of hits (any gap — an absent or zero-valued s-mer — forces `win_min = 0` for every window spanning it, exactly matching the old dense scan's "not in index counts as 0" rule, just never materialising the zero). The deque logic is otherwise identical to the pre-sparsification version; it's scoped to run-relative indices instead of the whole sequence.
|
||||
|
||||
For each genome g independently, a sliding window of size z scans the input. Output position i is confirmed for genome g iff all z values `results[i..i+z][g]` are nonzero (`None` counts as zero for all genomes). The scan is O(n) per genome.
|
||||
This runs once per genome that has at least one hit in the chunk (`process_chunk` iterates `by_genome`, one `Vec<(seq_idx, pos_smer, value)>` per genome, built from `QueryHit::Value` during the partition loop — genomes with zero hits in this chunk have an empty `Vec` and cost nothing beyond the iteration itself). Total work is `O(hits log hits)` per genome (the sort) rather than `O(n_smers)` per genome regardless of hit count — a genuine complexity win on top of the memory one, for the common case where most `(chunk, genome)` pairs have no or few hits.
|
||||
|
||||
Output values come from `results[i]` (leftmost s-mer of each window); genomes not confirmed are zeroed. If all genomes are zero, the position is returned as `None`.
|
||||
Output position `pos_out` is confirmed for genome `g` iff its run produced a nonzero `win_min` — equivalent to "all `z` consecutive s-mer values in the window are nonzero for `g`", same semantics as before.
|
||||
|
||||
**Short sequences**: when the s-mer count is less than z, no complete window can form — `apply_findere` returns an empty vector. K-mers from sequences shorter than k_user are not emitted.
|
||||
**The value reported per confirmed position is the window minimum, not the leftmost s-mer's raw value** — unchanged from the dense version. For presence indexes (0/1 values) this is equivalent to a logical AND either way. For count indexes it is not: the accumulated count for genome `g` at position `pos_out` is the minimum across the window, the weakest link — not the leftmost s-mer's own count. The presence/count adjustment (`u32::from(win_min >= threshold)` vs. raw `win_min`) is applied once, inside `sparse_findere_for_genome`, rather than later during accumulation.
|
||||
|
||||
**Exact indexes**: `z = 1`, `apply_findere` is a passthrough (output length = input length).
|
||||
**`kmer_missing` bookkeeping is independent of the per-genome sparse structures**, by design (see roadmap point 9): a lightweight dense `SmerIndex` (`in_index: Vec<bool>`, sized `total_smers` — **not** multiplied by `n_genomes`) is populated from `QueryHit::Found` during the partition loop, one entry per hit k-mer regardless of which genome(s) it matched. A position with no genome confirmed counts as `kmer_missing` iff the leftmost s-mer of that window is absent from `SmerIndex` entirely (see [`kmer_missing` semantics](#kmer_missing-semantics)).
|
||||
|
||||
**Coverage (`--detail`)** is built by re-scanning each genome's confirmed-hit list (already computed, no extra pass over raw data) and densifying into the `[u32; n_kmers_out]` arrays the JSON output format requires — but only when `--detail` is actually requested; the sparse structures cost nothing extra when it isn't.
|
||||
|
||||
**Short sequences**: when a sequence's s-mer count is less than `z`, its run(s) — if any hits exist at all — can never reach length `z`, so no window is ever confirmed for it; no k_user-mer is emitted, same outcome as the dense version's `n_kmers_out == 0` early-skip, reached here as a natural consequence rather than a separate check.
|
||||
|
||||
**Exact indexes**: `z = 1`, every single-hit "run" of length 1 immediately satisfies `k + 1 >= z`, so every hit is its own confirmed window with `win_min` equal to its own value — a passthrough, as before.
|
||||
|
||||
### Effective z at query time
|
||||
|
||||
@@ -85,14 +122,17 @@ The `-z` CLI option overrides the index metadata value. A higher z increases str
|
||||
|
||||
### `QueryLayer` variant selection
|
||||
|
||||
`QueryLayer::open` in `query_layer.rs` selects the data matrix to pair with `MphfLayer`:
|
||||
`QueryLayer::open` (`obikpartitionner/src/query_layer.rs:28-45`) only ever returns two variants — `Presence` or `Count`, checked in this order:
|
||||
|
||||
| Condition | Variant | Data returned per k-mer |
|
||||
|---|---|---|
|
||||
| `with_counts=true` and `counts/` exists | `Count` | raw count per genome |
|
||||
| `presence/` exists | `Presence` | 0/1 per genome (bit matrix) |
|
||||
| only `counts/` exists | `Count` | counts used as-is |
|
||||
| neither exists | `SetOnly` | 1 for every genome |
|
||||
| Order | Condition | Variant | Data returned per k-mer |
|
||||
|---|---|---|---|
|
||||
| 1 | `with_counts=true` and `counts/` exists | `Count` | raw count per genome |
|
||||
| 2 | (else) `presence/` exists, or `counts/` doesn't exist at all | `Presence` | see below |
|
||||
| 3 | (else — `counts/` exists, `presence/` doesn't, `with_counts=false`) | `Count` | counts used as-is |
|
||||
|
||||
There is no `QueryLayer::SetOnly` variant. The "no on-disk matrix at all" case is handled one level down: `Presence` wraps `PersistentBitMatrix`, whose own `open()` (`obicompactvec/src/bitmatrix.rs:260-288`) auto-detects among **three** internal representations — `Packed` (`presence/matrix.pbmx`), `Columnar` (`presence/meta.json`), or `Implicit { n_rows, n_cols }` when neither file exists (built from `layer_meta.json`, `fill_row` returning all-`1`s without touching disk). This is where "1 for every genome" actually happens — not at the `QueryLayer` level.
|
||||
|
||||
**Worth double-checking, not confirmed as a bug**: `PersistentBitMatrix::open`'s `Implicit` branch constructs `Implicit { n_rows: meta.n, n_cols: 1 }` — `n_cols` is hardcoded to `1`, not to the layer's actual `n_genomes`. `fill_row` for `Implicit` only writes `buf[..1]`, leaving the rest of a longer `n_genomes`-sized buffer untouched (zeroed by the caller beforehand). If this path is ever reached for a layer covering more than one genome, only genome index 0 would read as present. Whether that's reachable in practice (layers might always be single-genome when they fall back to `Implicit`) wasn't verified here — flagging for follow-up, not fixing.
|
||||
|
||||
---
|
||||
|
||||
@@ -123,7 +163,7 @@ Coverage reflects confirmed k_user-mers only. The vectors are emitted in the JSO
|
||||
|
||||
## `kmer_missing` semantics
|
||||
|
||||
`kmer_missing` counts k_user-mer positions where the first s-mer (`seq_results[seq_idx][pos]`) is `None` — i.e. absent from the index entirely. K-mers where the z-window fails because a later s-mer is absent or zero are not counted as missing (the first s-mer being present is used as proxy for index membership).
|
||||
`kmer_missing` counts k_user-mer positions where the leftmost s-mer of the window (`smer_index.is_in_index(seq_idx, pos)`, `SmerIndex`) is `false` — i.e. absent from the index entirely. K-mers where the z-window fails because a later s-mer is absent or zero (but the leftmost one is present) are not counted as missing — the leftmost s-mer being present is used as proxy for index membership.
|
||||
|
||||
---
|
||||
|
||||
@@ -152,8 +192,8 @@ Genome keys follow the iteration order of `meta.genomes`.
|
||||
| Key | Type | Condition | Semantics |
|
||||
|---|---|---|---|
|
||||
| `kmer_count` | int | always | k-mers confirmed (post-Findere) with at least one genome match |
|
||||
| `kmer_missing` | int | `--count-missing` | k-mers absent from the index entirely (pre-Findere None) |
|
||||
| `kmer_strict_matches` | object | always | per-genome accumulated value (label → count or 0/1) |
|
||||
| `kmer_missing` | int | `--count-missing` | k-mers absent from the index entirely (leftmost s-mer of the window not found) |
|
||||
| `kmer_strict_matches` | object | always | per-genome accumulated value, non-zero entries only (label → count or 0/1) |
|
||||
| `coverage` | object | `--detail` | per-genome array of per-position contributions (label → [u32]) |
|
||||
|
||||
`kmer_count + kmer_missing` ≤ total k_user-mers in the sequence. The gap corresponds to k_user-mers whose z-window was not fully confirmed (at least one s-mer absent or zero for all genomes) but whose first s-mer was present in the index.
|
||||
@@ -165,7 +205,7 @@ Genome keys follow the iteration order of `meta.genomes`.
|
||||
```
|
||||
obikmer query <index> [--detail] [--mismatch] [--count-missing]
|
||||
[--force-presence] [--presence-threshold <n>]
|
||||
[-z <z>] [-T <threads>]
|
||||
[-z <z>] [-T <threads>] [--chunk-size <MiB>]
|
||||
<query.fa> [<query2.fa> ...]
|
||||
```
|
||||
|
||||
@@ -177,6 +217,7 @@ obikmer query <index> [--detail] [--mismatch] [--count-missing]
|
||||
| `--force-presence` | off | Report 0/1 per genome regardless of index counts |
|
||||
| `--presence-threshold` | 1 | Minimum count to declare genome present |
|
||||
| `-T` / `--threads` | all CPUs | Worker threads |
|
||||
| `--chunk-size` | auto (from available RAM and thread count) | I/O chunk size in MiB — see [Future work, point 3](#throughput--parallelism--identified-potential-not-yet-implemented) for why the auto-sizing formula currently under-estimates memory on indexes with many genomes |
|
||||
|
||||
`--mismatch` is accepted but currently ignored with a warning on stderr.
|
||||
|
||||
@@ -187,3 +228,179 @@ obikmer query <index> [--detail] [--mismatch] [--count-missing]
|
||||
- **`--mismatch`**: 1-mismatch approximate matching — generate `3·k` single-substitution variants per k-mer, look each up independently.
|
||||
- **Read classification** (`--classify`): assign each read to the genome with the highest match score.
|
||||
- **Whitelist / blacklist filtering**: threshold-based accept/reject on per-genome match scores.
|
||||
|
||||
### Throughput & parallelism — identified potential (not yet implemented)
|
||||
|
||||
Observed on a 192-core (8×24 NUMA) machine: `query` uses ~10 cores or fewer, and the default chunk size gets the process OOM-killed. Root causes and candidate fixes, in dependency order:
|
||||
|
||||
**1. Single-threaded I/O source (main core-utilization bottleneck).**
|
||||
`run()` builds `all_chunks` via `paths.into_iter().flat_map(read_sequence_chunks_sized(...))` and passes it directly as the `input` iterator to `pipe.apply()`. In `obipipeline::Pipe::apply` (`scheduler.rs`), `input.next()` is called exclusively from the dedicated source thread — so file opening, decompression, and FASTA/FASTQ chunk-boundary parsing for *all* input files run serially in one thread, regardless of `--threads`. Compare with `steps::scatter` (used by `index`) and `cmd/superkmer.rs`: there, file opening + streaming is itself a `Flat` pipeline stage (`||?`), executed across the `n_workers` pool, with `obipipeline::throttle(paths, max_open)` bounding concurrently-open files in the source thread. That pattern parallelises I/O across files (and NUMA nodes); `query.rs` cannot.
|
||||
Fix direction: restructure `query`'s pipe with an initial `Flat` stage analogous to `scatter`'s, opening/chunking files across workers instead of in `flat_map`.
|
||||
|
||||
**2. Gzip decompression is inherently single-threaded per file.**
|
||||
`niffler`/`flate2` (used by `xopen`) do standard DEFLATE, which has no parallel-decodable structure for an arbitrary stream. Fix (1) parallelises *across* files but not *within* one large gzip file. Parking a possible fix (`rapidgzip-rs`) is tracked in [chunkreader.md](../implementation/chunkreader.md#future-work--parallel-gzip-decompression-in-xopen).
|
||||
|
||||
**3. Chunk-size memory formula ignores `n_genomes`.**
|
||||
`chunk_bytes = available_memory_bytes() / (n_workers * 16)` (`query.rs:407-414`) assumes a fixed ~8–16× overhead per raw input byte. But `KmerResults::new` (`query.rs:165-179`) allocates `data: Vec<u32>` sized `total_kmers_in_chunk × n_genomes` — dense, **for every k-mer position in the chunk, hit or not** — plus `win_min` and (with `--detail`) `cov`, same scaling. Real per-chunk memory is `O(n_genomes)`, not constant; the formula doesn't know `n_genomes` at all. This is the direct cause of the OOM kill on indexes with many reference genomes.
|
||||
|
||||
**4. MPHF lookup and matrix-row fetch are fused, not staged.**
|
||||
`QueryLayer::find_into` (`obikpartitionner/src/query_layer.rs:48-67`) does the MPHF `find` *and* the `fill_row` matrix read in one call per k-mer, inside a single-threaded loop (`query_partition_with`). There is no separation between "is this k-mer indexed" (cheap, `O(1)`, independent of `n_genomes`) and "what are its per-genome values" (the expensive, `n_genomes`-scaling part).
|
||||
|
||||
**5. Dereplication should happen at k-mer granularity, directly — not via an intermediate superkmer-level dedup.**
|
||||
`QueryBatch::from_records` currently dereplicates at the *superkmer* level (`HashMap<RoutableSuperKmer, Vec<SKDesc>>`, `query.rs:112`). This misses redundancy between k-mers shared by *different* superkmers (read overlaps, repeats, a SNP splitting an otherwise-identical run). Superkmer *construction* (`SuperKmerIter`) stays mandatory — it is the mechanism that computes minimizers/partition routing, not an optional dedup layer — but the dedup structure built on top of it should key directly on `CanonicalKmer`, in the same pass: `HashMap<CanonicalKmer, Vec<(seq_idx, pos)>>`. This also means the MPHF `find` itself runs once per **distinct** k-mer instead of once per occurrence — a win independent of the matrix-fetch cost below.
|
||||
|
||||
**6. Stage 1 output: bucket confirmed hits by layer, keyed by MPHF slot.**
|
||||
For each unique canonical k-mer, MPHF lookup across a partition's layers stops at the first match (`query_partition_with:105-111`) — a k-mer belongs to at most one layer. So stage 1's output can be reshaped directly into:
|
||||
```
|
||||
HashMap<layer_idx, HashMap<slot, Vec<(seq_idx, pos)>>>
|
||||
```
|
||||
replacing the `CanonicalKmer` key by the resolved `slot` (compact integer, and exactly what stage 2 needs to address the matrix). K-mers matching no layer simply have no entry here (they still count toward `in_index`/`kmer_missing` bookkeeping, which stays `O(1)` per position, independent of `n_genomes`).
|
||||
|
||||
**7. Partition-level parallelism is currently absent — and a NUMA-aware mechanism for exactly this already exists, unused, in `obikindex`.**
|
||||
`process_chunk`'s partition loop (`query.rs:250-278`, `for (part_idx, part_sks) in by_part.iter().enumerate()`) processes every partition of a chunk sequentially on the single worker thread that owns that chunk. This is a parallelism axis on its own, independent of the column question below.
|
||||
More importantly: `docmd/architecture/numa_partition_runner.md` and `numa_worker_pools.md` document `PartitionRunner` (`obikindex/src/numa.rs`), **already implemented** and already used by `merge.rs`, `index.rs` (`build_layers`), `select.rs`, `reindex.rs`, `rebuild.rs` — one controller thread per NUMA node, a Rayon pool pinned to that node's CPUs (`hwlocality`, `numa` feature, default-on in `obikindex/Cargo.toml`), adaptive worker activation driven by *both* a CPU-efficiency signal and an I/O-throughput signal (`CpuSample`/`IoSample`, `/proc/self/io` on Linux). It exists precisely because a naive `into_par_iter()` on the global Rayon pool measurably degrades ×60 on this codebase's own 192-core/8-NUMA reference machine (`numa_worker_pools.md`, § Problem) once workers contend for cross-socket memory bandwidth on shared mmap'd/hashed structures — exactly the shape of the matrix-column scan in point 8 below.
|
||||
`obikmer` already depends on `obikindex` (`obikmer/Cargo.toml`, for `KmerIndex`), so `PartitionRunner` is directly reachable from `cmd/query.rs` — no new dependency. Both the partition-level loop and (see point 8) the genome-column scan should be driven through it rather than through ad-hoc `rayon::into_par_iter()`, to avoid reproducing the already-measured-and-fixed contention problem. Also relevant: the "CPU-only signal stalls on I/O-bound stages" issue documented for `pack_matrices` (mmap-heavy, page-fault-bound) applies just as much to a column-major mmap scan over persistent matrices — reuse the existing dual CPU/IO activation signal rather than re-deriving one.
|
||||
>
|
||||
> **Correction from implementation (Phase 4 below)**: this turned out not to be viable as described. `PartitionRunner::run()`'s actual body spawns roughly one OS thread per worker slot across every NUMA node **on every call** (confirmed by reading `numa.rs`, not just its doc comments) — fine for the one-call-per-command-invocation batch usage in `merge`/`build_layers`, but `query_partition_with` runs once per `(chunk, partition)`, far too frequently to absorb that spawn cost. Partition-level parallelism via `PartitionRunner` is deferred, not implemented. See Phase 4's "What did not ship, and why" for the detail.
|
||||
|
||||
**8. Stage 2: column-major matrix fetch, parallel across genome columns — via `PartitionRunner`, not naive `rayon`.**
|
||||
Both persistent matrix formats are column-oriented on disk: `ColumnarCompactIntMatrix`/`ColumnarBitMatrix` (`obicompactvec/src/{intmatrix,bitmatrix}.rs`) mmap one file per genome column; `PackedCompactIntMatrix`/`PackedBitMatrix` mmap one region-offset per column in a single file. `fill_row(slot, buf)` as used today (`query.rs:262-272` via `on_hit`) reads **one slot across all `n_genomes` columns** per hit — the worst possible access pattern for this layout (up to `n_genomes` scattered mmap regions touched per single k-mer).
|
||||
Better: for each layer, walk the matrix **column by column** (genome by genome): for each genome, scan the `slot` keys collected in step 6 for that layer and call `col.get(slot)`, keeping only nonzero results, and broadcast to the associated `(seq_idx, pos)` list. Total `get()` calls are unchanged (`n_hits × n_genomes` in the worst case) — the win is locality (sequential access within one mmap'd column at a time, not scattered across all columns per hit), not fewer operations.
|
||||
Columns are independent (read-only, disjoint mmap regions) → embarrassingly parallel across genomes, *but* — per point 7 — `obicompactvec`'s existing `into_par_iter()` over `0..n_cols` (`sum()`, `count_nonzero()`, pairwise distance matrices) is the **naive, unpinned** pattern the rest of the codebase is actively migrating away from, not a model to copy here. Route this through `PartitionRunner` (or the same NUMA-pool machinery) instead. Two things to settle when this is designed: how the partition axis (point 7), the column axis, and the existing chunk-level `n_workers` `obipipeline` pool compose without oversubscribing the machine (three different concurrency mechanisms — raw-thread pipe workers, `PartitionRunner`'s pinned Rayon pools, and whatever drives the column scan — need a single reconciled thread budget, not three independent ones); and the threshold below which per-column dispatch overhead outweighs the gain (small `n_genomes` or small per-layer hit counts) — to be measured, not assumed.
|
||||
>
|
||||
> **Correction from implementation (Phase 4 below)**: column-major fetch is implemented — but as a plain sequential loop, not parallelised via `PartitionRunner`. Same reason as point 7's correction above. The column-major *locality* win (the actual claim of this point) does not depend on adding parallelism on top of it, and is validated independently. Column-level parallelism is deferred pending a mechanism that fits this call frequency (candidates noted in Phase 4).
|
||||
|
||||
**9. Sparse per-genome representation, fed directly to Findere.**
|
||||
Stage 2's output should be `HashMap<genome_idx, Vec<(seq_idx, position, count)>>`, **sorted by `(seq_idx, position)`** once collected, instead of a dense `KmerResults`-style matrix — the key must carry `seq_idx`, not just `genome_idx`, because a chunk batches many sequences and `position` is only meaningful within one; a plain `Vec<(position, count)>` per genome would silently mix positions from different sequences and corrupt the sliding-window scan. This bounds retained memory by actual nonzero hits on both axes (position sparsity from non-matching k-mers, genome sparsity from a matched k-mer typically belonging to only a handful of genomes out of possibly many). The Findere sliding-window (`process_chunk`, the `win_min`/deque loop) would need reworking to run per `(sequence, genome)` over its sparse, sorted `(position, count)` list — detect runs of ≥`z` consecutive positions, window-min within each run — instead of today's dense `O(total_kmers × n_genomes)` scan. This is also a genuine complexity win (`O(hits log hits)` per genome vs. dense scan), not just memory.
|
||||
**Not covered by this sparsification**: `--detail`'s `cov` accumulator (`query.rs:304-308`) has the identical `n_genomes`-dense scaling problem and wasn't folded into points above. It doesn't need to be retained densely throughout processing, though — only the final JSON serialization (`emit_batch`) requires a dense `[u32]` per `(seq, genome)`, and only for the sequences actually being output with `--detail`. Densification can stay a late, output-time-only step, reconstructed from the sparse per-genome lists.
|
||||
|
||||
**Secondary patterns available from `scatter.rs`/`superkmer.rs`, not yet in `query.rs`:**
|
||||
- `throttle()` + `CommonArgs::effective_max_open()` to bound concurrently-open input files (query.rs defines its own `QueryArgs`, doesn't reuse this).
|
||||
- Progress bar with EMA throughput + live active-worker gauges (`obisys::spinner`, `flat_active`/`transform_active` counters) — diagnostic value for locating the bottleneck.
|
||||
- `obisys::Reporter`/`Stage::start`/`stop` timing per phase (used by `index`, `filter`; absent from `query`).
|
||||
|
||||
None of this is implemented yet — parked here as a coherent roadmap while the design is discussed further. Suggested dependency order: (1) I/O parallelism → (3) genome-aware chunk sizing → (4)–(9) staged/k-mer-deduped/NUMA-aware-partition-and-column-major/sparse query engine (larger refactor, biggest structural payoff — reuses `PartitionRunner` rather than inventing a new parallelism mechanism) → (2) parallel gzip (separate, orthogonal, tracked in chunkreader.md) → secondary diagnostics patterns.
|
||||
|
||||
---
|
||||
|
||||
## Implementation plan
|
||||
|
||||
Concrete, phased translation of the roadmap above. Phases 0–2 are small, independent, low-risk, and each individually testable against current `query` output — land them first, in order, and measure on the reference 192-core/8-NUMA machine before deciding whether phases 3–5 (the staged/sparse engine, the larger structural payoff) are still worth their cost. Phases 3–5 are one coordinated change spanning `obikmer`, `obikpartitionner`, and `obicompactvec` — they should not be split across releases mid-way, because the intermediate state (e.g. k-mer-level dedup feeding the old dense `KmerResults`) has no correctness or performance benefit on its own. Phase 6 is unrelated to phases 0–5 and can happen any time, independently, if `rapidgzip-rs` is validated (see [chunkreader.md](../implementation/chunkreader.md#future-work--parallel-gzip-decompression-in-xopen)).
|
||||
|
||||
Instrumentation is deliberately sequenced *before* the I/O fix (reordering the roadmap's own listed order), because every later phase's justification rests on a measurement ("to be measured, not assumed" appears throughout the roadmap above) — without it, phases 3–5 would be undertaken on faith.
|
||||
|
||||
Performance measurement on the reference 192-core/8-NUMA machine is done by the project owner, not from this development environment (macOS, 16 cores — `PartitionRunner`'s NUMA pinning is Linux-only, so even phase 4's mechanism can't be functionally exercised for its actual purpose here). Each phase below is therefore written to be *self-measuring*: the debug-level logging it adds must be enough, on its own, to judge whether that phase's algorithmic choice paid off from a cluster run's logs, without needing to attach a profiler.
|
||||
|
||||
### Conventions applied to every phase below
|
||||
|
||||
**Debug logging.** Every phase that changes an algorithmic choice (not phase 0, which *is* the logging) adds `tracing::debug!`/`trace!` at points that let a cluster run's logs answer "did this help": counts, ratios, and timings that quantify the specific claim that phase makes — e.g. phase 3 must log how many MPHF `find` calls were saved by k-mer-level dedup (the whole justification for that phase), phase 4 must log per-column scan timings, phase 5 must log actual retained-memory / sparsity ratios achieved. Prefer one structured `debug!` per chunk (fields, not prose) over free-text — the cluster logs will be the only evidence available for judging these choices, so they need to be grep/awk-able, not just readable.
|
||||
|
||||
**Unit tests.** This project's convention (`obiread`, `obikseq`, `obidebruinj`, `obicompactvec`, `obilayeredmap`, `obiskio`, `obifastwrite`) is `#[cfg(test)] #[path = "tests/<name>.rs"] mod tests;` at the bottom of the source file, with the actual test code in a sibling `src/tests/<name>.rs`. Neither `obikmer` nor `obikpartitionner` (the two crates phases 3 and 5 touch most) currently have a `src/tests/` directory at all — this needs creating, following the existing pattern exactly, not inventing a new one.
|
||||
|
||||
**Workflow (`jj`).** Work happens in a fresh `jj` commit, easy to abandon. `jj new` between phases is reasonable where it helps isolate a phase for review, but only when the working copy compiles at that point (project convention) — phase 3's internal sub-steps (batch dedup change, then `query_layer.rs` split, then the new return shape) will likely not each compile independently since they're one coupled change, so treat "commit boundary" and "plan phase boundary" as related but not forced to match 1:1; use judgement per phase rather than mechanically splitting on every bullet.
|
||||
|
||||
### Phase 0 — Instrumentation (prerequisite for measuring every later phase)
|
||||
|
||||
**Goal**: make core utilization, throughput, and per-stage timing visible on a real run, so phases 1–5 can be justified with numbers instead of assumption.
|
||||
|
||||
- `obikmer/src/cmd/query.rs`: wrap `run()`'s main loop with `obisys::Reporter`/`Stage::start("query")`/`.stop()`, printed at the end via `rep.print()` — same pattern as `index.rs`/`filter.rs`.
|
||||
- Add an `obisys::spinner("query")` progress bar around the `pipe.apply(...)` loop, with an EMA throughput readout (bases/s or k-mers/s, mirroring `steps::scatter`'s `ema_rate` computation, `scatter.rs:88-118`) and live gauges for "chunks in flight" / "workers busy" — reuse the `AtomicU32` counter pattern from `scatter.rs` (`flat_active`, `transform_active`) rather than inventing a new one.
|
||||
- Add `max_open_files: Option<usize>` to `QueryArgs` and a `effective_max_open()` method mirroring `CommonArgs::effective_max_open()` (`obikmer/src/cli.rs:90-94`) — needed by phase 1's `throttle()` call. (`QueryArgs` can't just embed `CommonArgs` — it doesn't take `kmer_size`/`minimizer_size`/`partitions`/`level_max`/`theta` from the CLI, those come from the index metadata — so this is a small standalone addition, not a flatten.)
|
||||
- Add one structured `debug!` per `process_chunk` call: chunk byte size, sequence count, s-mer count, wall time, and (once later phases exist) the fields they add — this single log line is the baseline every later phase's own logging gets compared against.
|
||||
- **Validation**: none needed beyond "the numbers appear and look sane" — this phase changes no query logic or output.
|
||||
- **Deliverable used by every phase below**: a before/after throughput and core-utilization measurement on the reference machine.
|
||||
|
||||
### Phase 1 — Parallel per-file I/O (fixes root cause of low core utilization)
|
||||
|
||||
**Goal**: file opening, decompression, and chunk-boundary parsing run across the `n_workers` pool instead of serially in the pipe's dedicated source thread.
|
||||
|
||||
- `obikmer/src/cmd/query.rs`:
|
||||
- Replace the `paths.into_iter().flat_map(read_sequence_chunks_sized(...))` construction (current `run()`, building `all_chunks`) with `obipipeline::throttle(paths.into_iter(), args.effective_max_open())`, passed as the pipe's `input`.
|
||||
- Add a new `QueryData::Path(PathBuf)` variant (alongside `Chunk`/`Output`) to carry the throttled path through the pipe's type-erasure mechanism.
|
||||
- Add a new **first** pipe stage, `Flat`/fallible (`||?`), modeled on `scatter.rs:60-86` and `superkmer.rs:54-65`: given a `Throttled<PathBuf>`, call `read_sequence_chunks_sized(path, chunk_bytes)` and yield each `Rope` chunk, keeping `pw.guard` alive until the file's iterator is exhausted (reuse or adapt `scatter.rs`'s `GuardedIter` wrapper — same lifetime problem, same fix).
|
||||
- The existing `process_chunk` transform stage becomes the pipe's **second** stage, unchanged in its own logic — it still receives one `Rope` chunk at a time, just no longer all coming from one serial source.
|
||||
- `make_pipe!` invocation grows from one stage (`Chunk => Output`) to two (`Path => Chunk => Output`).
|
||||
- Log, per file: time spent waiting on the `throttle()` slot (queueing due to `max_open`), and time spent opening/decompressing/producing the first chunk — this is what directly proves (or disproves) that I/O is now spread across workers instead of serialized.
|
||||
- **Validation**: run `query` on a small multi-file input, diff output against the pre-change version — content must be identical; **record order across files is not guaranteed to be preserved** even before this change (chunk-level dispatch across `n_workers` already reorders completions), so the diff must be order-insensitive (sort by read id, or compare as sets) if it wasn't already.
|
||||
- **Measure**: core utilization on the reference machine with several large input files, compare against phase 0's baseline.
|
||||
|
||||
### Phase 2 — Genome-aware chunk-size formula (fixes OOM)
|
||||
|
||||
**Goal**: `chunk_bytes` reflects actual per-chunk memory (`O(n_genomes)`), not a fixed multiplier.
|
||||
|
||||
- `obikmer/src/cmd/query.rs`, `run()`: `n_genomes` and `args.detail` are already computed above the `chunk_bytes` calculation (`n_genomes` at the top of `run()`, before line 407 in the current file) — reorder if needed, then replace:
|
||||
```rust
|
||||
let computed = avail / (n_workers as u64 * 16);
|
||||
```
|
||||
with a formula that scales the divisor by `n_genomes` (and roughly doubles it when `--detail` is set, since `cov` duplicates the per-genome accumulation): e.g. `per_chunk_multiplier = base_overhead + n_genomes as u64 * BYTES_PER_KMER_PER_GENOME * if detail { 2 } else { 1 }`, replacing the flat `16`. `BYTES_PER_KMER_PER_GENOME` should be derived from `KmerResults`'s actual layout (`4` bytes per `u32` entry in `data`, plus the `bool` in `in_index`, plus `win_min`'s equal-sized buffer) rather than guessed.
|
||||
- `args.chunk_size` (manual `--chunk-size` override) keeps taking priority, unchanged.
|
||||
- Log the resolved `chunk_bytes`, `n_genomes`, and the estimated peak per-chunk memory (`chunk_bytes` × the same multiplier used to derive it) once at startup — lets a cluster run confirm the estimate was actually respected, not just that the process didn't get OOM-killed (which could also happen to be true for the wrong reason).
|
||||
- **Validation**: build a test index with a large `n_genomes` (e.g. hundreds), run `query` with default chunk sizing under a memory limit (`ulimit -v` or a cgroup), confirm it no longer gets OOM-killed and that memory scales as predicted when `n_genomes` grows.
|
||||
- **Note**: this phase is superseded once phase 5 lands (sparse retained memory no longer scales with `n_genomes × total_kmers` at all) — but it's needed immediately regardless, since phases 3–5 are a bigger, riskier change and users need a working `query` in the meantime.
|
||||
|
||||
### Phase 3 — K-mer-level dereplication, staged MPHF/matrix lookup
|
||||
|
||||
**Goal**: replace superkmer-level dedup with k-mer-level dedup (roadmap point 5), and split the fused MPHF-find/matrix-fetch (point 4) so stage 1's output is bucketed by layer and MPHF slot (point 6).
|
||||
|
||||
- `obikmer/src/cmd/query.rs`:
|
||||
- Replace `QueryBatch::from_records`'s dedup map (`HashMap<RoutableSuperKmer, Vec<SKDesc>>`, current `query.rs:112`) with a per-partition `HashMap<CanonicalKmer, Vec<(seq_idx: u32, pos: u32)>>`, built in the same `SuperKmerIter` pass: superkmer construction and partition routing (`part_idx` from the superkmer's minimizer hash) are unchanged, only the granularity of what gets deduplicated changes — each `CanonicalKmer` within a superkmer is inserted individually instead of the whole superkmer being the dedup key.
|
||||
- **Verified**: `CanonicalKmer` (`obikseq/src/kmer.rs:390`, `pub type CanonicalKmer = CanonicalKmerOf<KLen>`) — the underlying `CanonicalKmerOf<L>` derives `Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash` (`kmer.rs:269`). Usable as a `HashMap`/`HashSet` key as-is, no change needed.
|
||||
- `obikpartitionner/src/query_layer.rs`:
|
||||
- Split `QueryLayer::find_into` (`query_layer.rs:48-67`) into two methods: `find_slot(&self, kmer: CanonicalKmer) -> Option<usize>` (MPHF only, no matrix touch) and keep `fill_row` as-is for phase 4 to call later.
|
||||
- Replace `query_partition_with`'s inner loop (`query_layer.rs:103-113`) with a version that, for each unique `CanonicalKmer`, calls `find_slot` across the partition's layers (stopping at first hit, same as today), and instead of immediately filling a row, records `(layer_idx, slot)`.
|
||||
- New return shape for the partition-level query, replacing today's `on_hit(sk_idx, kmer_idx, row)` callback: `HashMap<layer_idx, HashMap<slot, Vec<(seq_idx, pos)>>>` (roadmap point 6) — built directly from the k-mer dedup map's `Vec<(seq_idx,pos)>` values, keyed by the resolved slot instead of the k-mer.
|
||||
- **This phase alone has no throughput benefit yet** (matrix fetch still happens, just deferred) beyond the k-mer-level dedup itself (fewer MPHF calls when queries have overlapping/repeated k-mers) — its purpose is to produce the input phase 4 needs. Land phase 3+4 together, not phase 3 alone, per the "don't split 3–5 across releases" note above.
|
||||
- Log, per chunk: total k-mer occurrences vs. unique `CanonicalKmer` count (the dedup ratio — the entire justification for this phase) and the resulting MPHF `find` call count. If the dedup ratio is close to `1.0` on real query data (little redundancy), that's the cluster run telling us this phase wasn't worth it — the logging needs to be able to say that, not just confirm the happy path.
|
||||
- **Unit tests**: create `obikmer/src/cmd/tests/query.rs` (new `src/tests/` dir for this crate, following the project's `#[cfg(test)] #[path = "tests/query.rs"] mod tests;` convention) and `obikpartitionner/src/tests/query_layer.rs` (likewise new for this crate). Cover: the k-mer-level dedup map construction on synthetic sequences with known repeated/overlapping k-mers (assert unique-kmer count and occurrence lists); the `find_slot`/bucket-by-layer-and-slot construction against a small hand-built `QueryLayer` fixture, asserting the `(layer_idx, slot, seq_idx, pos)` tuples match what the old per-occurrence loop would have produced.
|
||||
|
||||
### Phase 4 — Column-major matrix fetch (roadmap points 7–8) — implemented, NUMA parallelism deferred
|
||||
|
||||
**Goal (revised during implementation)**: replace `fill_row`-per-hit (row-major, worst-case mmap locality) with a column-major scan. `PartitionRunner` turned out to be the wrong mechanism for this at this call granularity — see below; the column-major fetch itself is implemented and validated, without it.
|
||||
|
||||
**What shipped:**
|
||||
- `obicompactvec`: the per-column accessors this phase needed **already existed** — `PersistentCompactIntMatrix::col_view(c)` and `PersistentBitMatrix::col_view(c)` are public, and `IntSliceView::get(slot)`/`BitSliceView::get(slot)` are public — the original plan underestimated how much of this plumbing the pairwise-distance code (`dump`/`select`/`stats`) had already required. The one real gap: `PersistentBitMatrix::col_view()` panics on the `Implicit` variant (the documented mono-genome fast path, `bitmatrix.rs`). Added `PersistentBitMatrix::get(c, slot) -> u32` (`bitmatrix.rs`), a non-panicking column-major point lookup that returns `1` for `Implicit` regardless of `c` — the smallest surface needed, not a new `col_get` API from scratch.
|
||||
- `obikpartitionner/src/query_layer.rs`: `query_partition_with` is now two explicit stages, matching roadmap points 6–8: **stage 1** (MPHF-only, per unique k-mer, bucket hits by `(layer_idx, slot)`, emits `QueryHit::Found`) then **stage 2** (per layer with ≥1 hit, column-major: for each genome column `g` in `0..layer.n_cols().min(n_genomes)`, scan that layer's bucketed slots and call `col_value(g, slot)`, emitting `QueryHit::Value(descs, g, value)` on nonzero). `QueryHit` is a single enum delivered through one `FnMut(QueryHit)` callback — an earlier two-closure design (`on_found` + `on_value`) didn't borrow-check, since the caller's single mutable accumulator (`KmerResults`) can't be captured by two separate `FnMut` closures passed to the same call.
|
||||
- `obikmer/src/cmd/query.rs`: `KmerResults::set` (row-major, whole-row-at-once) replaced by `mark_found` (stage 1: flag a position as indexed, independent of any genome's value) and `set_one` (stage 2: write one genome's value at one position). `QueryStats` extended with `n_columns_scanned`/`n_col_get_calls`, logged per chunk.
|
||||
- Total `get()`-equivalent calls are unchanged from the row-major version (`n_hits × n_cols` in the worst case, confirmed by `n_col_get_calls` in the debug log) — the win is locality (sequential access within one layer's column at a time, across `mmap`'d regions, instead of jumping across all columns per hit), exactly as predicted.
|
||||
|
||||
**What did not ship, and why — `PartitionRunner` is architecturally the wrong tool here:**
|
||||
Reading `obikindex/src/numa.rs`'s actual `run()` body (not just its doc comments) shows every call spawns a timer thread **plus one OS thread per worker slot on every NUMA node** (`std::thread::scope` + one `s.spawn()` per node per `max_workers`) — on the 192-core/8-NUMA reference machine, that's on the order of 190+ fresh OS threads spawned **per call**. This is fine for its actual, established usage in this codebase (`merge.rs`, `index.rs`'s `build_layers`): one `PartitionRunner::new()` + one `run()` call per command invocation, amortised over a batch of ~256 long-running partitions. It is not fine for `query`'s call pattern: `query_partition_with` runs once per `(chunk, partition)`, potentially thousands of times per second — spawning ~190 OS threads that often to scan a handful of genome columns would very likely cost far more than the row-major approach it's meant to replace. This is exactly the "resolve empirically, don't assume" composition risk the roadmap flagged, just resolved by reading the mechanism's actual cost before wiring it in, rather than by measuring a regression on the cluster after the fact.
|
||||
The column-major loop in stage 2 is therefore a **plain sequential loop** for now — it captures the whole, provable locality win (roadmap point 8's actual claim) without adding any parallelism mechanism. Genome-column-level parallelism (point 8's "bonus" axis) and partition-level parallelism (point 7) are both deferred — not abandoned. Candidates for a follow-up, once there's a concrete profiling need: (a) `rayon`'s already-warm global pool (`into_par_iter()`) for the column axis specifically — cheap to invoke repeatedly since it doesn't spawn threads per call, though it's the same "naive rayon" pattern `numa_worker_pools.md` warns about for a *different* workload (random pointer-chasing over large hash maps); a column scan's access pattern (sequential reads within one `mmap`'d region) has a different contention profile and hasn't been shown to have the same problem — needs its own measurement, not an assumption either way; (b) restructuring so `PartitionRunner` is invoked once per whole `query` run (or per large batch of chunks) rather than per `(chunk, partition)`, amortising its spawn cost the way `merge`/`build_layers` do — a bigger structural change than this phase's scope.
|
||||
- Log (implemented): `QueryStats::n_columns_scanned`/`n_col_get_calls`, folded into the existing per-chunk `debug!("k-mer dedup + column-major fetch", ...)` line (`query.rs`) alongside phase 3's dedup counters.
|
||||
- **Unit tests**: extended `obikpartitionner/src/tests/query_layer.rs` (phase 3's file) — `query_partition_with`'s empty/missing-index paths updated for the new `QueryStats` fields and single-callback signature.
|
||||
- **Validation performed**: full workspace build + `cargo test --workspace`, zero failures. Functional validation against real indexes: (1) a single-genome index — output byte-identical to pre-phase-4 (same `kmer_count`/`kmer_strict_matches` on every record); (2) the existing 20-genome `benchmark/global_index_presence` index — runs correctly, `n_hits=0` for an unrelated query (expected: no shared k-mers between a plant read and a bacterial reference set), no panics, confirming the `Implicit`/multi-column bounds logic doesn't crash on a real multi-genome, mixed-format index; (3) **the critical correctness case**: built two single-sequence-pair test genomes, merged into one 2-genome index, queried with reads from both — reads from `genomeA` matched **only** `genomeA` (`kmer_count` identical to the pre-dedup occurrence count, zero leakage into `genomeB`'s column) and vice versa. This is the test that would have caught a column-index mixup, an off-by-one in `n_cols`, or cross-genome bleed from the stage-1/stage-2 split — it passed cleanly.
|
||||
- **Not yet done**: the microbenchmark comparing column-major vs. the old row-major access pattern's wall time / page-fault counters on a large-`n_genomes` layer — needs a realistically large multi-genome index and, for the page-fault counters specifically, Linux (not available from this development environment). Left for cluster validation alongside phases 1–3's own pending measurements.
|
||||
|
||||
### Phase 5 — Sparse Findere rework (roadmap point 9)
|
||||
|
||||
**Goal**: replace the dense `KmerResults`/`win_min` sliding-window scan with one operating on phase 4's sparse per-genome output.
|
||||
|
||||
- `obikmer/src/cmd/query.rs`, `process_chunk`:
|
||||
- Remove `KmerResults` (`query.rs:157-202`) and the dense `win_min` allocation (`query.rs:290-291`, sized `max_n_kmers × n_genomes`).
|
||||
- Keep a lightweight dense `in_index: Vec<bool>` per chunk (sized `total_kmers`, independent of `n_genomes`) from phase 3's stage 1 — still needed for `kmer_missing` bookkeeping (leftmost-s-mer-of-window membership test), which phase 4's sparse structure doesn't carry (a k-mer with no genome hit has no entry there at all).
|
||||
- New per-`(seq_idx, genome)` scan: for each genome's `Vec<(seq_idx, pos, count)>` (sorted, per phase 4), group by `seq_idx` (contiguous after sort), then within each sequence's positions detect runs of `pos, pos+1, pos+2, ...` of length ≥ `z`; within each run, the existing monotone-deque window-minimum logic (`query.rs`'s current `dq` loop, conceptually unchanged) applies — but the deque now only scans real entries in the run, never zero-filled gaps.
|
||||
- Update `SeqAcc` accumulation and `emit_batch` to consume this per-genome sparse iteration instead of `results.val`/`results.is_in_index`.
|
||||
- `--detail`/`cov`: build sparsely during the same scan (only positions with a confirmed contribution get an entry), densify into the `[u32]` JSON array only in `emit_batch`, only for genomes/sequences actually being serialized (per roadmap point 9's note, `query.rs:304-308`'s current dense allocation goes away).
|
||||
- Log, per chunk: total sparse entries retained vs. what the old dense `KmerResults` would have allocated (`total_smers × n_genomes`) — the sparsity ratio is this phase's entire reason for existing, so it must be directly visible in the logs, not inferred from process RSS. Also log the run-detection stats (number of runs found, average run length) — a low average run length relative to `z` would mean most positions still fail to form a full window, worth knowing.
|
||||
- **Unit tests**: `obikmer/src/cmd/tests/query.rs` (extended from phase 3) — the property test described below is the primary deliverable here, not an afterthought; write it as an actual `#[test]` (or a small internal fuzz/property-style loop over randomized fixtures if a property-testing crate isn't already a dependency — check before adding one, per this project's dependency-approval rule) rather than a one-off manual comparison.
|
||||
- **Validation — this is the correctness-critical phase**: property-test comparing old (dense, pre-phase-3) and new (sparse) implementations on the same randomized input/index fixtures, asserting identical `kmer_count`, `kmer_missing`, `kmer_strict_matches`, and (with `--detail`) `coverage` for every sequence. Keep both implementations compiled side by side (behind a debug-only flag or a temporary parallel code path) only for the duration of this validation; delete the dense path once parity is confirmed — per this project's own convention, superseded code is not kept "just in case."
|
||||
- **Update `docmd/architecture/query.md` itself**: once this phase lands, the "Findere z-window filter" section (which currently — correctly — describes the dense deque-over-`0..n_smers` scan) needs another pass to describe the sparse run-detection algorithm instead, as already flagged when this phase was discussed.
|
||||
|
||||
**Implemented as planned, no deviations discovered this time.** What shipped:
|
||||
- `KmerResults` removed entirely, replaced by `SmerIndex` (`in_index: Vec<bool>` + `offsets`, unchanged size/purpose, renamed since it's no longer "results" — just the O(1)-per-position "was this k-mer found at all" bookkeeping) and `by_genome: Vec<Vec<(seq_idx, pos, value)>>` (one empty `Vec` per genome until a hit arrives — genomes with zero hits in a chunk cost nothing beyond the outer `Vec`'s own allocation).
|
||||
- New `sparse_findere_for_genome(hits, z, presence, threshold) -> (Vec<ConfirmedHit>, n_runs, total_run_len)` (`query.rs`): sorts one genome's raw hits by `(seq_idx, pos)`, detects maximal runs of consecutive `pos` within one sequence, runs the same monotone-deque window-minimum as before but scoped to each run (run-relative indices for eviction, absolute `pos` for computing `pos_out`). Presence/count adjustment (`u32::from(win_min >= threshold)` vs. raw) is applied inside this function, once per confirmed hit, rather than later during accumulation.
|
||||
- `process_chunk` restructured into three passes after the partition loop: (1) run `sparse_findere_for_genome` per genome, collecting `confirmed_by_genome` and run-detection stats; (2) accumulate `genome_totals` directly from `confirmed_by_genome` and mark a `confirmed_any: Vec<bool>` (sized `total_kmers_out`, not `× n_genomes`); (3) a position-only pass (`O(total_kmers_out)`, no genome factor) computing `kmer_count`/`kmer_missing` from `confirmed_any` + `SmerIndex`. `cov` (`--detail`) is populated by re-scanning `confirmed_by_genome` — only when `--detail` is actually set, otherwise skipped entirely.
|
||||
- Debug log added (`"sparse Findere"`): `n_dense_would_be` (`n_occurrences × n_genomes` — what the deleted dense path would have allocated), `n_sparse_entries` (what's actually retained), `n_runs`/`avg_run_len` (per the plan's ask, to see whether hits mostly fail to form complete windows).
|
||||
- **Unit tests**: `sparse_findere_matches_dense_reference_on_random_inputs` (`obikmer/src/cmd/tests/query.rs`) — 200 randomized cases (sequence count/length, `z`, presence/count mode, threshold, hit density from sparse to fully-dense) comparing `sparse_findere_for_genome` against `dense_reference_findere`, a faithful reimplementation of the deleted dense algorithm kept only as a test-local correctness oracle (no property-testing crate added — checked first, none was a workspace dependency; a small `std`-only xorshift64 PRNG stands in for one, deterministic and dependency-free). All 200 cases pass.
|
||||
- **Functional validation performed**: full workspace build + `cargo test --workspace`, zero failures. End-to-end against real indexes: baseline output (no flags) unchanged from pre-phase-5 recorded values on the same fixtures; `--count-missing` correct (`kmer_missing: 0` on a self-match); `--detail` correct — coverage array length matches `kmer_count`, and critically, re-ran the two-genome cross-contamination check from phase 4 with `--detail --count-missing`: `genomeA` reads show coverage sum `106` for `genomeA` and `0` for `genomeB` (and vice versa) — confirms the sparse-to-dense `cov` reconstruction doesn't leak across genomes either, not just the scalar `kmer_strict_matches` path.
|
||||
- This phase's roadmap item ("update the Findere z-window filter section") — done, see above; the "Algorithm" section's pseudocode was also updated, since it still named `KmerResults`/`SKDesc` from before phases 3–4.
|
||||
|
||||
### Phase 6 — Parallel gzip decompression (independent, optional)
|
||||
|
||||
Tracked separately in [chunkreader.md](../implementation/chunkreader.md#future-work--parallel-gzip-decompression-in-xopen); parked pending validation of `rapidgzip-rs` on real data. Not a dependency of, or a dependency for, phases 0–5 — `xopen` is shared infrastructure (`obiread`), phase 1 benefits from it but doesn't require it (phase 1 parallelises *across* files; this phase would additionally parallelise *within* one large file).
|
||||
|
||||
### Cross-cutting risks
|
||||
|
||||
- **Thread-budget oversubscription** (phase 4): the single biggest unresolved design question in this whole plan — see phase 4's composition note. Should be settled with real measurements early in phase 4, not assumed from the design alone.
|
||||
- **`obicompactvec` API surface growth** (phase 4): new public per-column accessors are additive (existing `fill_row`/`row` stay for other callers — `dump`, `select`, distance computations) — no breaking change expected, but worth checking `obicompactvec`'s other callers aren't already relying on `fill_row` being the only/cheapest access path in a way that would make maintaining two access patterns (row-major and column-major) a real maintenance cost rather than a one-off addition.
|
||||
- **`PersistentBitMatrix::Implicit`'s hardcoded `n_cols: 1` — resolved, not a bug.** `LayerMeta`'s own doc comment (`obicompactvec/src/layer_meta.rs:1-9`) states it is written "alongside `mphf.bin`" and read by `PersistentBitMatrix::open` "to determine `n_rows` for **the implicit (mono-genome presence/absence) case**" — i.e. `Implicit` is a documented single-genome fast path (no presence matrix needed when there is trivially one genome), not a generic "no matrix built yet" fallback. `n_cols: 1` is correct by design for the case it's meant to handle. Phase 4's column loop is safe as planned — this was worth checking once, doesn't need further action.
|
||||
|
||||
@@ -0,0 +1,105 @@
|
||||
# Rebuild / filter — column-first design
|
||||
|
||||
## Problem with the current two-pass design
|
||||
|
||||
`rebuild_partition` currently makes **two full passes** over source data:
|
||||
|
||||
**Pass 1** — read unitigs → MPHF lookup (source) → read row (108 values) → apply filter → push kmer into `GraphDeBruijn`, **discard row**.
|
||||
|
||||
**Pass 2** — read unitigs again → MPHF lookup again → read row again → for each passing kmer, look up slot in new MPHF → fill column builders.
|
||||
|
||||
Both passes do random access into the source matrix: for each kmer, the MPHF returns a slot, then we read 108 values scattered across 108 column positions. This is cache-hostile even with a packed matrix (`.pbmx`), because the matrix is column-major: consecutive row reads jump across the file.
|
||||
|
||||
## Memory budget
|
||||
|
||||
The `keep` bitvector costs **1 bit per slot**. With 256 partitions and realistic kmer counts, each partition holds at most a few tens of millions of slots → a few MB per bitvector. Even in the absolute worst case (800 M slots), it stays under 100 MB. This is negligible.
|
||||
|
||||
The `slot_map` option (Option B, 8–16 bytes per slot) is heavier but still bounded: at 15 M slots and 8 bytes, that is 120 MB per partition, acceptable for a single worker.
|
||||
|
||||
## Key observation
|
||||
|
||||
**The filter operates on column values, not on kmers.** A filter like `--max-outgroup-count 0` only needs to know, for each slot, whether any outgroup column is non-zero. It does not need to know which kmer occupies that slot.
|
||||
|
||||
This means filtering can be done as a **sequential column scan** that produces a `keep: BitVec[n_slots]` — no MPHF lookups, no kmer knowledge, perfectly cache-friendly.
|
||||
|
||||
## Proposed single-scan design
|
||||
|
||||
### Step 1 — column scan → `keep` bitvector
|
||||
|
||||
```
|
||||
for each column c in source matrix:
|
||||
read column c sequentially (one mmap range)
|
||||
update keep[slot] according to filter contribution of column c
|
||||
```
|
||||
|
||||
For `GroupQuorumFilter` with ingroup/outgroup:
|
||||
- ingroup columns: count presence per slot → `ingroup_count[slot]`
|
||||
- outgroup columns: `keep[slot] &= (value[slot] == 0)` (early-exit possible)
|
||||
|
||||
Result: `keep: BitVec` of size `n_slots`, computed with purely sequential IO.
|
||||
|
||||
### Step 2 — unitig scan → kept kmers + new MPHF
|
||||
|
||||
```
|
||||
for each kmer in unitig files:
|
||||
old_slot = old_MPHF(kmer)
|
||||
if keep[old_slot]:
|
||||
push kmer into new GraphDeBruijn
|
||||
record (old_slot, kmer) ← or just old_slot in order
|
||||
```
|
||||
|
||||
Build new MPHF from `GraphDeBruijn` via `materialize_layer`.
|
||||
|
||||
### Step 3 — fill new matrix
|
||||
|
||||
Two sub-options:
|
||||
|
||||
**Option A — from recorded (old_slot, kmer) pairs:**
|
||||
|
||||
```
|
||||
for each (old_slot, kmer) in recorded list:
|
||||
new_slot = new_MPHF(kmer)
|
||||
for each column c:
|
||||
new_matrix[new_slot, c] = old_matrix[old_slot, c]
|
||||
```
|
||||
|
||||
Memory cost: `n_kept × (8 + 8)` bytes for `(old_slot: usize, kmer: CanonicalKmer)`.
|
||||
For species-specific filters, `n_kept` is small. For unfiltered rebuild, `n_kept = n_slots`.
|
||||
|
||||
**Option B — column-by-column copy using old→new slot mapping:**
|
||||
|
||||
Precompute `slot_map: Vec<Option<usize>>` of size `n_slots`:
|
||||
- For each kmer in unitig file: `slot_map[old_MPHF(kmer)] = Some(new_MPHF(kmer))`
|
||||
|
||||
Then for each source column:
|
||||
```
|
||||
read source column sequentially
|
||||
for each slot where slot_map[slot] = Some(new_slot):
|
||||
write value to new column at new_slot
|
||||
```
|
||||
|
||||
Memory cost: `n_slots × sizeof(usize)` for the slot map (one usize per source slot).
|
||||
IO pattern: sequential read of each source column → random write into new column builders.
|
||||
|
||||
Option B avoids storing kmer values and works uniformly regardless of filter selectivity.
|
||||
|
||||
## Comparison
|
||||
|
||||
| | Current | Proposed |
|
||||
|---|---|---|
|
||||
| Disk reads | 2× unitigs + 2× random matrix | 1× columns (sequential) + 1× unitigs |
|
||||
| MPHF lookups (source) | 2× N_kmers | 1× N_kept (step 2) or 0 (option B, col scan only) |
|
||||
| Cache behavior | poor (random row access) | good (sequential column scan) |
|
||||
| Extra memory | none | slot_map (option B) or (old_slot, kmer) list (option A) |
|
||||
|
||||
## Files to modify
|
||||
|
||||
- `src/obikpartitionner/src/rebuild_layer.rs` — `rebuild_partition` and `iter_src_layers`
|
||||
- Possibly `src/obicompactvec/` — add column iterator API if not already present
|
||||
- `src/obilayeredmap/` — check if per-column sequential access is exposed on `SrcLayerData`
|
||||
|
||||
## Open questions
|
||||
|
||||
- Does `SrcLayerData` expose per-column sequential iteration, or only `lookup(kmer, n_genomes)` random access?
|
||||
- For option B: are new column builders writable in random-slot order (i.e. `set_val(slot, value)` without sequential constraint)?
|
||||
- For `GroupQuorumFilter` specifically: can the filter be decomposed into independent per-column contributions, or does it need the full row?
|
||||
@@ -107,3 +107,19 @@ stateDiagram-v2
|
||||
`restart` is updated each time a `+` is found. When any state fails its expected input, the scan jumps back to `restart` and continues from there — guaranteeing that a `@` in a quality line cannot be accepted as a record start, because the `\n+\n` structure immediately following it (going backward) will not be found.
|
||||
|
||||
Returns the byte offset of the `@` that starts the last complete record.
|
||||
|
||||
---
|
||||
|
||||
## Future work — parallel gzip decompression in `xopen`
|
||||
|
||||
`obiread::xopen` (`xopen.rs`) decompresses gzip via `niffler` → `flate2`, which is single-threaded (standard DEFLATE has no parallel-decodable structure). For large local gzip inputs this single-threaded decompression can become the throughput bottleneck feeding the `query`/`index`/`superkmer` pipelines, since chunk/page production for a given file is serialized ahead of the worker pool.
|
||||
|
||||
Candidate: special-case local, on-disk, gzip-magic-detected paths in `open_raw`/`xopen` to use [`rapidgzip-rs`](https://github.com/alekseizarubin/rapidgzip-rs) (`ReaderBuilder::new().parallelism(n).open(path)`, implements `Read + Seek`) instead of `niffler`, keeping `niffler` for every other case: `stdin` (`-`), HTTP(S) sources, and all non-gzip formats (bzip2, xz, zstd — less used in practice here).
|
||||
|
||||
Constraints identified so far (not yet validated against real data):
|
||||
- Branch point must move earlier than the current `decompress()` call in `open_raw` — rapidgzip's fast path needs the file **path**, not an already-opened generic `Read`, so the gzip/local-file detection has to happen before the generic `File::open` + `niffler::send::get_reader` path is taken.
|
||||
- `stdin` and HTTP sources are not seekable — they stay on `niffler` regardless; the gain only applies to local on-disk `.gz` files.
|
||||
- `rapidgzip-sys` vendors a native C++ engine: requires CMake ≥ 3.17, a C++17 compiler, and `nasm` on x86 targets — a real build-toolchain addition, not just a pure-Rust crate.
|
||||
- Low maturity of the Rust binding at review time (2 GitHub stars, ~15 commits, April 2026 latest release) — the underlying C++ engine is validated (HPDC 2023 paper), but the binding itself has limited production track record.
|
||||
|
||||
Decision: parked for now. Before adopting, validate on real data: throughput vs. `niffler` on representative large `.gz` inputs, and byte-for-byte correctness of decompressed output.
|
||||
|
||||
@@ -29,16 +29,23 @@ Multiple values separated by `|` are always OR-ed within the predicate.
|
||||
|
||||
### Path matching (`~` and `!~`)
|
||||
|
||||
Metadata values can represent hierarchical taxonomic paths such as
|
||||
Metadata values can represent hierarchical concept paths such as
|
||||
`/Eukaryota/Viridiplantae/Streptophyta/Betulaceae/Betula/nana`.
|
||||
|
||||
- **Absolute pattern** (starts with `/`): the value must start with the pattern
|
||||
at a segment boundary.
|
||||
`taxon~/Betulaceae/Betula` matches `/Betulaceae/Betula/nana` and
|
||||
`/Betulaceae/Betula` but not `/Betulaceae/Betuloides/…`.
|
||||
- **Bare segment** (no leading `/`): the value must contain the pattern as an
|
||||
exact path component anywhere.
|
||||
`taxon~Betula` matches any path that has `Betula` as one of its segments.
|
||||
Stored taxonomy values always start with `/` (the root of the path).
|
||||
Query patterns do **not** need to start with `/` — a leading `/` is an optional
|
||||
start anchor, not a requirement.
|
||||
|
||||
| Pattern form | Semantics |
|
||||
|---|---|
|
||||
| `A/B` | contiguous sub-path A then B, anywhere in the value |
|
||||
| `/A/B` | value starts with A then B |
|
||||
| `A/B$` | value ends with A then B |
|
||||
| `/A/B$` | value is exactly A then B |
|
||||
| `A@x/B` | A with class `x` followed by B with any class |
|
||||
|
||||
- `taxon~/Betulaceae/Betula` matches any path that starts with `Betulaceae` then `Betula`.
|
||||
- `taxon~Betula` matches any path containing `Betula` as a segment, anywhere.
|
||||
|
||||
### Missing metadata key → NA
|
||||
|
||||
@@ -85,18 +92,48 @@ For each genome:
|
||||
|
||||
| Flag | Applies to | Meaning |
|
||||
|------|-----------|---------|
|
||||
| `--min-count N` | ingroup | k-mer present in at least N ingroup genomes |
|
||||
| `--max-count N` | ingroup | k-mer present in at most N ingroup genomes |
|
||||
| `--min-count N` | ingroup | k-mer present in at least N ingroup genomes (N may be negative, see below) |
|
||||
| `--max-count N` | ingroup | k-mer present in at most N ingroup genomes (N may be negative, see below) |
|
||||
| `--min-frac F` | ingroup | k-mer present in at least fraction F of ingroup genomes |
|
||||
| `--max-frac F` | ingroup | k-mer present in at most fraction F of ingroup genomes |
|
||||
| `--min-outgroup-count N` | outgroup | k-mer present in at least N outgroup genomes |
|
||||
| `--max-outgroup-count N` | outgroup | k-mer present in at most N outgroup genomes |
|
||||
| `--min-outgroup-count N` | outgroup | k-mer present in at least N outgroup genomes (N may be negative, see below) |
|
||||
| `--max-outgroup-count N` | outgroup | k-mer present in at most N outgroup genomes (N may be negative, see below) |
|
||||
| `--min-outgroup-frac F` | outgroup | k-mer present in at least fraction F of outgroup genomes |
|
||||
| `--max-outgroup-frac F` | outgroup | k-mer present in at most fraction F of outgroup genomes |
|
||||
| `--min-total-count N` | all genomes | sum of per-genome counts ≥ N (`filter` only) |
|
||||
| `--max-total-count N` | all genomes | sum of per-genome counts ≤ N (`filter` only) |
|
||||
| `--presence-threshold N` | all | per-genome count > N to be considered "present" (default 0) |
|
||||
|
||||
### Negative counts — offset from group size
|
||||
|
||||
The four integer count flags (`--min-count`, `--max-count`, `--min-outgroup-count`,
|
||||
`--max-outgroup-count`) accept **negative** values, interpreted as an offset counted
|
||||
down from the group size `n`, resolved at run time once `n` is known:
|
||||
|
||||
| Value | Effective threshold |
|
||||
|-------|---------------------|
|
||||
| `N ≥ 0` | literal absolute count `N` |
|
||||
| `-x` (x > 0) | `max(1, n − x)` — "all but x" |
|
||||
|
||||
`-1` literally means *all but one*, `-2` *all but two*, and so on. This expresses
|
||||
a quorum relative to the group size that a plain fraction cannot state exactly
|
||||
(e.g. "present in every genome except at most one" is `n−1`, which is `0.9` for
|
||||
`n = 10` but `0.857…` for `n = 7`).
|
||||
|
||||
The threshold is **floored at 1**, never 0: the negative form always keeps
|
||||
constraining the group. Without the floor, `--min-count -1` on a singleton
|
||||
ingroup (`n = 1`) would resolve to `0` ("at least 0") and silently drop the
|
||||
constraint; the floor makes it `1` ("present in that one genome") instead.
|
||||
|
||||
To express a count of `0` (e.g. "absent from the ingroup"), use the literal `0`,
|
||||
not a negative — `0` and `-0` are indistinguishable, so the offset form starts at
|
||||
`-1`.
|
||||
|
||||
> **Edge case** — on an *empty* group (`n = 0`, e.g. a predicate matching no
|
||||
> genome), a negative count still resolves to `1`, an impossible constraint that
|
||||
> rejects every k-mer. This is consistent with an empty group letting nothing
|
||||
> through, but differs from the "no constraint" behaviour of the fraction flags.
|
||||
|
||||
**Conditional defaults** — the defaults for `--min-frac` and `--max-outgroup-count` depend on two conditions:
|
||||
whether the corresponding group was declared, **and** whether any quorum flag for that group was explicitly set.
|
||||
|
||||
@@ -208,6 +245,17 @@ obikmer filter src --output dst \
|
||||
--max-outgroup-count 0
|
||||
```
|
||||
|
||||
Noise-tolerant core — keep k-mers present in *all but one* ingroup genome
|
||||
(`-1` = `n−1`) and absent from *all but one* of the outgroup:
|
||||
|
||||
```sh
|
||||
obikmer filter src --output dst \
|
||||
--ingroup "genus=Betula" \
|
||||
--outgroup "*" \
|
||||
--min-count -1 \
|
||||
--max-outgroup-count -1
|
||||
```
|
||||
|
||||
To dump only k-mers specific to *Betula nana*:
|
||||
|
||||
```sh
|
||||
|
||||
@@ -0,0 +1,533 @@
|
||||
# obicompactvec — Complete Reference
|
||||
|
||||
## Module structure
|
||||
|
||||
```
|
||||
src/obicompactvec/src/
|
||||
lib.rs public re-exports
|
||||
views.rs BitSliceView<'a>, IntSliceView<'a> — zero-copy read views
|
||||
traits.rs ColumnWeights, CountPartials, BitPartials (matrix aggregation)
|
||||
bitvec.rs PersistentBitVec, PersistentBitVecBuilder, BitIter
|
||||
reader.rs PersistentCompactIntVec (read-only)
|
||||
builder.rs PersistentCompactIntVecBuilder (read-write)
|
||||
tempintvec.rs TempCompactIntVec, TempCompactIntVecBuilder (temp-file-backed)
|
||||
tempbitvec.rs TempBitVec, TempBitVecBuilder (temp-file-backed)
|
||||
bitmatrix.rs PersistentBitMatrix, PersistentBitMatrixBuilder
|
||||
intmatrix.rs PersistentCompactIntMatrix, PersistentCompactIntMatrixBuilder
|
||||
colgroup.rs ColGroup, MatrixGroupOps trait
|
||||
format.rs file format constants, encode/decode helpers
|
||||
layer_meta.rs LayerMeta (column metadata)
|
||||
meta.rs matrix metadata
|
||||
```
|
||||
|
||||
```mermaid
|
||||
graph TD
|
||||
views --> bitvec
|
||||
views --> builder
|
||||
views --> tempbitvec
|
||||
views --> tempintvec
|
||||
views --> bitmatrix
|
||||
views --> intmatrix
|
||||
format --> reader
|
||||
format --> builder
|
||||
reader --> intmatrix
|
||||
reader --> tempintvec
|
||||
builder --> intmatrix
|
||||
builder --> tempintvec
|
||||
bitvec --> tempbitvec
|
||||
bitvec --> bitmatrix
|
||||
tempintvec --> intmatrix
|
||||
tempintvec --> bitmatrix
|
||||
tempbitvec --> intmatrix
|
||||
tempbitvec --> bitmatrix
|
||||
colgroup --> intmatrix
|
||||
colgroup --> bitmatrix
|
||||
layer_meta --> bitmatrix
|
||||
layer_meta --> intmatrix
|
||||
meta --> bitmatrix
|
||||
meta --> intmatrix
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Compact int encoding
|
||||
|
||||
All integer vectors use the same two-tier encoding regardless of storage backend.
|
||||
|
||||
**Primary array** — one `u8` per slot:
|
||||
|
||||
- Values **0–254** are stored directly. No overhead.
|
||||
- Value **255 is a sentinel**: the slot's actual value is ≥ 255 and lives in the overflow store.
|
||||
|
||||
**Overflow store** — maps slot index to a `u32` value ≥ 255:
|
||||
|
||||
- In `PersistentCompactIntVecBuilder`: a `HashMap<usize, u32>` in RAM.
|
||||
- In `PersistentCompactIntVec` (reader): a sorted `[(slot: u64, value: u32)]` array in the mmap, with a sparse L1-resident index for binary search.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
slot --> P["primary[slot]: u8"]
|
||||
P -->|"< 255"| V["value = byte (0–254)"]
|
||||
P -->|"= 255 sentinel"| OV["overflow store"]
|
||||
OV -->|"Builder"| HM["HashMap<usize, u32>\nin RAM"]
|
||||
OV -->|"PersistentCompactIntVec"| SA["sorted [(slot,value)] in mmap\n+ sparse L1 index"]
|
||||
```
|
||||
|
||||
**Key property — sentinel 255 = +∞ on `u8`:**
|
||||
|
||||
- `min(a, 255) = a` for all `a ≤ 254` → correct when only one side is overflow
|
||||
- `max(a, 255) = 255` → correct sentinel when either side is overflow
|
||||
- Only the **both-overflow** case requires reading actual values from the overflow store.
|
||||
|
||||
In practice, k (overflow count) ≪ n (total slots). Observed genomic data: ~0.07% of kmer slots are in overflow.
|
||||
|
||||
---
|
||||
|
||||
## View types
|
||||
|
||||
The previous trait hierarchy (`BitSlice`, `BitSliceMut`, `IntSlice`, `IntSliceMut`) has been replaced by two concrete zero-copy view structs with inherent methods. Views are **`Copy`** — passing them is free. All read operations live on these two types.
|
||||
|
||||
### `BitSliceView<'a>`
|
||||
|
||||
```rust
|
||||
#[derive(Clone, Copy)]
|
||||
pub struct BitSliceView<'a> { pub(crate) words: &'a [u64], pub(crate) n: usize }
|
||||
```
|
||||
|
||||
Bit `i` is at `words[i >> 6]` bit `i & 63` (LSB-first). Padding bits in the last word are zero.
|
||||
|
||||
| Method | Cost |
|
||||
|---|---|
|
||||
| `len()`, `is_empty()` | O(1) |
|
||||
| `get(slot)` | O(1) |
|
||||
| `count_ones()` | POPCNT per word, O(n/64) |
|
||||
| `count_zeros()` | `n − count_ones()`, O(n/64) |
|
||||
| `iter() -> BitSliceIter<'a>` | O(1) setup, O(n) iteration |
|
||||
| `partial_jaccard_dist(other: BitSliceView)` | `(a&b).popcount`, `(a\|b).popcount` per word, O(n/64) |
|
||||
| `jaccard_dist(other: BitSliceView)` | from partial, O(n/64) |
|
||||
| `hamming_dist(other: BitSliceView)` | `(a^b).popcount` per word, O(n/64) |
|
||||
|
||||
`BitSliceIter<'a>`: word-level scan; one word per 64 iterations.
|
||||
|
||||
### `IntSliceView<'a>`
|
||||
|
||||
```rust
|
||||
#[derive(Clone, Copy)]
|
||||
pub struct IntSliceView<'a> {
|
||||
pub(crate) primary: &'a [u8],
|
||||
pub(crate) overflow_raw: &'a [u8], // sorted [(slot:u64, value:u32)] entries
|
||||
pub(crate) n_overflow: usize,
|
||||
pub(crate) n: usize,
|
||||
}
|
||||
```
|
||||
|
||||
`overflow_raw` contains `n_overflow` entries of `OVERFLOW_ENTRY_SIZE` bytes each, sorted by slot. The sort invariant is established at `close()`/`freeze()` time.
|
||||
|
||||
| Method | Cost |
|
||||
|---|---|
|
||||
| `len()`, `is_empty()` | O(1) |
|
||||
| `primary_bytes()` | O(1) |
|
||||
| `overflow_entries() -> impl Iterator<(usize,u32)>` | O(n_overflow) iteration |
|
||||
| `get(slot)` | O(1) primary; binary search O(log k) for overflow slots |
|
||||
| `iter() -> IntSliceViewIter<'a>` | merge scan, O(n + k) |
|
||||
| `sum()` | byte scan + overflow, O(n + k) |
|
||||
| `count_nonzero()` | byte scan, O(n) |
|
||||
| Distance methods (`bray_dist`, `euclidean_dist`, `jaccard_dist`, …) | O(n + k) |
|
||||
|
||||
`IntSliceViewIter<'a>`: merge scan using `overflow_pos` index. Requires sorted overflow — guaranteed by the construction lifecycle.
|
||||
|
||||
**Builder `view()` vs reader `view()`:** `PersistentCompactIntVecBuilder` stores overflow as an unsorted `HashMap`, not raw bytes. Its `view()` returns an `IntSliceView` with `overflow_raw = &[]` and `n_overflow = 0`. This is intentional — the view is primarily useful after `freeze()`. During building, callers that need overflow use `overflow_entries()` directly.
|
||||
|
||||
---
|
||||
|
||||
## Concrete types
|
||||
|
||||
```mermaid
|
||||
classDiagram
|
||||
class BitSliceView {
|
||||
+words: &[u64]
|
||||
+n: usize
|
||||
+get(slot) bool
|
||||
+count_ones() u64
|
||||
+iter() BitSliceIter
|
||||
+jaccard_dist/hamming_dist(other: BitSliceView)
|
||||
}
|
||||
class IntSliceView {
|
||||
+primary: &[u8]
|
||||
+overflow_raw: &[u8]
|
||||
+n_overflow: usize
|
||||
+n: usize
|
||||
+get(slot) u32
|
||||
+iter() IntSliceViewIter
|
||||
+overflow_entries() Iterator
|
||||
+bray_dist/euclidean_dist/…(other: IntSliceView)
|
||||
}
|
||||
class PersistentBitVec {
|
||||
-mmap: Mmap
|
||||
-n: usize
|
||||
+view() BitSliceView
|
||||
+get(slot) bool
|
||||
+count_ones/zeros() u64
|
||||
+iter() BitIter
|
||||
+partial_jaccard_dist(&Self) (u64,u64)
|
||||
+jaccard_dist/hamming_dist(&Self) …
|
||||
}
|
||||
class PersistentBitVecBuilder {
|
||||
-mmap: MmapMut
|
||||
-n: usize
|
||||
+view() BitSliceView
|
||||
+set(slot, bool)
|
||||
+or/and/xor/not(BitSliceView)
|
||||
+copy_from(BitSliceView)
|
||||
+close() / finish() → PersistentBitVec
|
||||
}
|
||||
class PersistentCompactIntVec {
|
||||
-mmap: Mmap
|
||||
-n: usize
|
||||
-n_overflow: usize
|
||||
-step: usize
|
||||
-index: Vec~(usize,usize)~
|
||||
+view() IntSliceView
|
||||
+get(slot) u32
|
||||
+iter() Iter
|
||||
+sum/count_nonzero() u64
|
||||
+bray_dist/euclidean_dist/… (&Self)
|
||||
}
|
||||
class PersistentCompactIntVecBuilder {
|
||||
-mmap: MmapMut
|
||||
-n: usize
|
||||
-overflow: HashMap~usize,u32~
|
||||
+view() IntSliceView
|
||||
+set(slot, u32) / get(slot) u32
|
||||
+inc / inc_present / inc_present_fast
|
||||
+inc_predicate / inc_predicate_fast
|
||||
+add/min/max/diff/mask_with(…View)
|
||||
+primary_bytes/primary_bytes_mut()
|
||||
+close() / finish() → PersistentCompactIntVec
|
||||
}
|
||||
|
||||
PersistentBitVec --> BitSliceView : view()
|
||||
PersistentBitVecBuilder --> BitSliceView : view()
|
||||
PersistentCompactIntVec --> IntSliceView : view()
|
||||
PersistentCompactIntVecBuilder --> IntSliceView : view() (primary only)
|
||||
PersistentBitVecBuilder --> PersistentBitVec : close() then open()
|
||||
PersistentCompactIntVecBuilder --> PersistentCompactIntVec : close() then open()
|
||||
```
|
||||
|
||||
### `PersistentBitVec` / `PersistentBitVecBuilder`
|
||||
|
||||
`PersistentBitVec` is the read-only type. `view()` returns a `BitSliceView<'_>` over the mmap word array. Direct inherent methods delegate to the view: `count_ones()`, `count_zeros()`, `partial_jaccard_dist(&Self)`, `jaccard_dist(&Self)`, `hamming_dist(&Self)`.
|
||||
|
||||
`BitIter<'a>` — exported iterator for `PersistentBitVec::iter()`:
|
||||
|
||||
```rust
|
||||
pub struct BitIter<'a> { pub(crate) words: &'a [u64], pub(crate) slot: usize, pub(crate) n: usize }
|
||||
```
|
||||
|
||||
`PersistentBitVecBuilder` is the read-write type. Mutation operations accept `BitSliceView<'_>`:
|
||||
|
||||
| Method | Cost |
|
||||
|---|---|
|
||||
| `set(slot, bool)` | O(1) |
|
||||
| `view() -> BitSliceView<'_>` | O(1) |
|
||||
| `or/and/xor(BitSliceView)` | word-level, O(n/64), SIMD-friendly |
|
||||
| `not()` | `w ^= u64::MAX` per word, re-masks last word | O(n/64) |
|
||||
| `copy_from(BitSliceView)` | `copy_from_slice` | O(n/64) |
|
||||
|
||||
### `PersistentCompactIntVec` / `PersistentCompactIntVecBuilder`
|
||||
|
||||
`PersistentCompactIntVec` is the read-only type. `view()` returns an `IntSliceView<'_>` over the mmap primary and overflow arrays. Inherent `iter()` is a merge scan (`Iter` struct). Inherent `sum()` and `count_nonzero()` use fast byte-scan helpers.
|
||||
|
||||
`PersistentCompactIntVecBuilder` is the read-write type. Mutation methods on the builder fall into two categories:
|
||||
|
||||
**Point mutations:**
|
||||
|
||||
| Method | Note |
|
||||
|---|---|
|
||||
| `set(slot, u32)` | writes primary[slot] or 255+overflow |
|
||||
| `get(slot) -> u32` | reads primary byte or HashMap |
|
||||
| `inc(slot)` | `get` + `set`, O(1) |
|
||||
|
||||
**Bulk computation methods** — accept view arguments:
|
||||
|
||||
| Method | Semantics | Overflow |
|
||||
|---|---|---|
|
||||
| `inc_present(BitSliceView)` | `+= 1` at each 1-bit | via `inc`, safe for any group size |
|
||||
| `inc_present_fast(BitSliceView)` | same, raw u8 `+= 1` | `debug_assert` no 255 reached |
|
||||
| `inc_predicate(IntSliceView, pred)` | `+= 1` where `pred(col[s])` | two-pass, safe |
|
||||
| `inc_predicate_fast(IntSliceView, pred)` | same, raw u8 | `debug_assert` no 255 reached |
|
||||
| `add(IntSliceView)` | `self[s] += other[s]` | primary fast path + overflow fallback |
|
||||
| `min(IntSliceView)` | byte min + both-overflow fixup | see algorithm below |
|
||||
| `max(IntSliceView)` | pre-pass + byte max | see algorithm below |
|
||||
| `diff(IntSliceView)` | saturating sub | self<255 hot path |
|
||||
| `mask_with(BitSliceView)` | zeros slots where mask bit = 0 | O(n_zeros) |
|
||||
|
||||
**`inc_present_fast` / `inc_predicate_fast` invariant:** caller guarantees no counter reaches 255 during the operation (group size < 255 for `inc_present_fast`, or chunk size < 255 for `inc_predicate_fast`). Violation is caught by `debug_assert` in dev builds.
|
||||
|
||||
**`min` algorithm:**
|
||||
|
||||
Exploits 255 = +∞: byte-level min is correct unless both sides are overflow.
|
||||
|
||||
```
|
||||
snapshot self_ov: Vec<(slot,val)>
|
||||
snapshot other_ov: HashMap<slot,val>
|
||||
clear_overflow()
|
||||
Pass 1 — byte min, SIMD-vectorizable, O(n)
|
||||
Pass 2 — both-overflow fixup, O(k_self):
|
||||
for (slot, self_val) in self_ov:
|
||||
if slot ∈ other_ov: set(slot, min(self_val, other_ov[slot]))
|
||||
```
|
||||
|
||||
**`max` algorithm:**
|
||||
|
||||
Cannot do byte max first — `max(255, b<255)=255` overwrites self's original overflow value. Pre-pass reads self's value at other's overflow slots before the byte pass.
|
||||
|
||||
```
|
||||
Pre-pass O(k_other): for (slot, other_val) in other.overflow_entries():
|
||||
set(slot, max(self.get(slot), other_val))
|
||||
Pass 1 — byte max, SIMD-vectorizable, O(n)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Matrix types
|
||||
|
||||
Four matrix types, two encodings × two formats:
|
||||
|
||||
| | Columnar format | Packed format |
|
||||
|---|---|---|
|
||||
| **Bit** | `PersistentBitMatrix` (Columnar variant) | `PersistentBitMatrix` (Packed variant) |
|
||||
| **Int** | `PersistentCompactIntMatrix` (Columnar variant) | `PersistentCompactIntMatrix` (Packed variant) |
|
||||
|
||||
Both matrix types are enums (`Columnar` / `Packed` / `Implicit` for bit) behind a transparent API. `col_view(c)` returns the appropriate view directly:
|
||||
|
||||
```rust
|
||||
// PersistentBitMatrix
|
||||
pub fn col_view(&self, c: usize) -> BitSliceView<'_>
|
||||
|
||||
// PersistentCompactIntMatrix
|
||||
pub fn col_view(&self, c: usize) -> IntSliceView<'_>
|
||||
```
|
||||
|
||||
No wrapper enums (`BitColView`, `IntColView`): the caller receives a `Copy` view struct immediately usable with any view method or bulk builder method.
|
||||
|
||||
`pack_compact_int_matrix` and `pack_bit_matrix` convert columnar → packed format.
|
||||
|
||||
---
|
||||
|
||||
## Aggregation traits (matrix level)
|
||||
|
||||
### ColumnWeights
|
||||
|
||||
```rust
|
||||
trait ColumnWeights: Send + Sync {
|
||||
fn col_weights(&self) -> Array1<u64>; // sum per column
|
||||
fn partial_kmer_counts(&self) -> Array1<u64>; // default = col_weights()
|
||||
}
|
||||
```
|
||||
|
||||
`partial_kmer_counts` is overridden for count matrices to return `count_nonzero` per column (distinct kmers) rather than total count.
|
||||
|
||||
### CountPartials
|
||||
|
||||
Abstract required methods: `partial_bray`, `partial_euclidean`, `partial_threshold_jaccard`, `partial_relfreq_bray`, `partial_relfreq_euclidean`, `partial_hellinger`.
|
||||
|
||||
**Additivity rule:** self-contained partials (`partial_bray`, `partial_euclidean`, `partial_threshold_jaccard`) can be element-wise summed across all `(partition, layer)` pairs. Normalised partials (`partial_relfreq_*`, `partial_hellinger`) require the **global** `col_weights` (accumulated across all layers and all partitions) as parameter.
|
||||
|
||||
**`partial_threshold_jaccard` returns `(inter, union)`** because `union[i,j]` depends on both columns simultaneously.
|
||||
|
||||
Provided finalisations:
|
||||
|
||||
| Finalisation | Formula |
|
||||
|---|---|
|
||||
| `bray_dist_matrix()` | `1 − 2·partial_bray[i,j] / (w[i] + w[j])` |
|
||||
| `euclidean_dist_matrix()` | `√partial_euclidean[i,j]` |
|
||||
| `threshold_jaccard_dist_matrix(t)` | `1 − inter[i,j] / union[i,j]` |
|
||||
| `relfreq_bray_dist_matrix()` | `1 − partial_relfreq_bray[i,j]` |
|
||||
| `relfreq_euclidean_dist_matrix()` | `√partial_relfreq_euclidean[i,j]` |
|
||||
| `hellinger_dist_matrix()` | `√partial_hellinger[i,j] / √2` |
|
||||
| `hellinger_euclidean_dist_matrix()` | `√partial_hellinger[i,j]` |
|
||||
| `threshold_mash_dist_matrix(k, t)` | Mash distance, derived from `threshold_jaccard_dist_matrix(t)` — no separate partial |
|
||||
|
||||
### BitPartials
|
||||
|
||||
Required: `partial_jaccard() -> (Array2<u64>, Array2<u64>)`, `partial_hamming() -> Array2<u64>`. Both additive across layers and partitions.
|
||||
|
||||
Provided finalisations also include `jaccard_dist_matrix()`, `hamming_dist_matrix()`, and `mash_dist_matrix(k)`.
|
||||
|
||||
### Mash distance
|
||||
|
||||
`mash_dist_matrix`/`threshold_mash_dist_matrix` add no new additive primitive: both are a pointwise transform of the existing Jaccard distance matrix, per the Mash mutation-rate estimator [@Mash-distances-doc; @Fan2015-mash-formula]:
|
||||
|
||||
```
|
||||
D = -1/k · ln(2J / (1+J)), J = 1 - d_jaccard
|
||||
```
|
||||
|
||||
`J ≤ 0` (i.e. `d_jaccard ≥ 1`, no shared k-mers) maps to `D = 1` (maximal distance) rather than the `ln` singularity at `J = 0`.
|
||||
|
||||
---
|
||||
|
||||
## Temp-file-backed types
|
||||
|
||||
**All inter-function results use temp-file-backed types** so the OS can page them out under memory pressure. This matters in practice: processing dozens of layers × hundreds of partitions in parallel would otherwise accumulate gigabytes of live anonymous memory.
|
||||
|
||||
### Lifecycle
|
||||
|
||||
```
|
||||
TempCompactIntVecBuilder::new(n) → writable mmap in TempDir
|
||||
↓ (inc_present_fast / inc_predicate_fast / add / mask_with / …)
|
||||
.freeze() → TempCompactIntVec (read-only mmap + TempDir)
|
||||
↓ (optional)
|
||||
.make_persistent(path) → PersistentCompactIntVec (permanent file)
|
||||
```
|
||||
|
||||
Same pattern for `TempBitVecBuilder` → `TempBitVec` → `PersistentBitVec`.
|
||||
|
||||
**Drop order**: `TempCompactIntVec { vec: PersistentCompactIntVec, _temp: TempDir }` — Rust drops fields in declaration order. `vec` (mmap) released before `_temp` (directory deleted). No explicit `drop()` needed.
|
||||
|
||||
### TempCompactIntVec / TempCompactIntVecBuilder
|
||||
|
||||
```rust
|
||||
pub struct TempCompactIntVec {
|
||||
vec: PersistentCompactIntVec,
|
||||
_temp: TempDir, // dropped after vec
|
||||
}
|
||||
|
||||
pub(crate) struct TempCompactIntVecBuilder {
|
||||
builder: PersistentCompactIntVecBuilder,
|
||||
temp: TempDir,
|
||||
}
|
||||
```
|
||||
|
||||
`TempCompactIntVec`: read access via `get(slot)`, `sum()`, `iter()`, `view() -> IntSliceView<'_>`.
|
||||
|
||||
`TempCompactIntVecBuilder`: full delegation to inner `PersistentCompactIntVecBuilder` — all bulk computation methods (`inc_present_fast`, `inc_predicate_fast`, `add`, `min`, `max`, `diff`, `mask_with`) are exposed as `pub(crate)`.
|
||||
|
||||
### TempBitVec / TempBitVecBuilder
|
||||
|
||||
```rust
|
||||
pub struct TempBitVec {
|
||||
vec: PersistentBitVec,
|
||||
_temp: TempDir,
|
||||
}
|
||||
|
||||
pub(crate) struct TempBitVecBuilder {
|
||||
builder: PersistentBitVecBuilder,
|
||||
temp: TempDir,
|
||||
}
|
||||
```
|
||||
|
||||
`TempBitVec`: read access via `get(slot)`, `count_ones()`, `view() -> BitSliceView<'_>`, `iter()`.
|
||||
|
||||
`TempBitVecBuilder`: exposes `set(slot, bool)`, `or(BitSliceView)`, and:
|
||||
|
||||
```rust
|
||||
pub(crate) fn or_where(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool)
|
||||
```
|
||||
|
||||
`or_where` — two passes, no intermediate allocation:
|
||||
|
||||
```
|
||||
Pass 1 — primary bytes, O(n):
|
||||
for slot in 0..n:
|
||||
b = col.primary_bytes()[slot]
|
||||
if b < 255 AND pred(b as u32): self.set(slot, true)
|
||||
|
||||
Pass 2 — overflow, O(k):
|
||||
for (slot, val) in col.overflow_entries():
|
||||
if pred(val): self.set(slot, true)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Filter / Select API
|
||||
|
||||
### ColGroup
|
||||
|
||||
```rust
|
||||
pub struct ColGroup { pub name: String, pub indices: Vec<usize> }
|
||||
```
|
||||
|
||||
Defined **once at the index level** from column metadata. Valid in all matrices of all layers and partitions — column structure is identical across the entire hierarchy; only rows (kmer slots) are partitioned.
|
||||
|
||||
### Composition axis
|
||||
|
||||
- **Across partitions**: kmer space is partitioned → partial results **concatenated** (disjoint kmer ranges).
|
||||
- **Across layers**: same kmer space, different counts → partial results **aggregated** (add, OR, etc.).
|
||||
|
||||
### MatrixGroupOps
|
||||
|
||||
Five required primitives + two default methods derived from them. All return temp-file-backed types.
|
||||
|
||||
```rust
|
||||
pub trait MatrixGroupOps {
|
||||
// required
|
||||
fn partial_group_presence_count(&self, g: &ColGroup, threshold: u32)
|
||||
-> io::Result<TempCompactIntVec>;
|
||||
fn partial_group_sum(&self, g: &ColGroup)
|
||||
-> io::Result<TempCompactIntVec>;
|
||||
fn partial_group_any(&self, g: &ColGroup, threshold: u32)
|
||||
-> io::Result<TempBitVec>;
|
||||
fn partial_group_min(&self, g: &ColGroup)
|
||||
-> io::Result<TempCompactIntVec>;
|
||||
fn partial_group_max(&self, g: &ColGroup)
|
||||
-> io::Result<TempCompactIntVec>;
|
||||
|
||||
// defaults derived from partial_group_presence_count
|
||||
fn partial_group_all(&self, g: &ColGroup, threshold: u32)
|
||||
-> io::Result<TempBitVec>; // slot=1 iff count == g.indices.len()
|
||||
fn partial_group_none(&self, g: &ColGroup, threshold: u32)
|
||||
-> io::Result<TempBitVec>; // slot=1 iff count == 0
|
||||
}
|
||||
```
|
||||
|
||||
Implemented for both `PersistentCompactIntMatrix` and `PersistentBitMatrix`.
|
||||
|
||||
For **bit matrices**: values are 0/1, so `partial_group_sum` = `partial_group_presence_count(g, 1)`; `partial_group_min` is AND (set first column then mask-with remaining); `partial_group_max` is OR via `partial_group_any` + `inc_present`.
|
||||
|
||||
**`partial_group_presence_count` — chunking for large groups:**
|
||||
|
||||
When `g.indices.len() < 255`: per-slot counts stay within `u8` range. Use `inc_present_fast` (bit) or `inc_predicate_fast(col_view(c), |v| v >= threshold)` (int) — raw u8 increment, no overflow entry written.
|
||||
|
||||
When `g.indices.len() ≥ 255`: process in chunks of 254 columns, accumulate via `.add(chunk_frozen.view())`.
|
||||
|
||||
**`partial_group_min` (int matrix)**: copy first column via `.add(col_view(first))` (start from 0 ⇒ copy), then `.min(col_view(c))` for remaining.
|
||||
|
||||
**`partial_group_max` (int matrix)**: `.max(col_view(c))` for all columns (start from 0 ⇒ first column acts as copy).
|
||||
|
||||
**`partial_group_any`** uses `or_where` on `TempBitVecBuilder` (two-pass: primary bytes then overflow entries).
|
||||
|
||||
**`partial_group_all` / `partial_group_none`** (default): call `partial_group_presence_count`, then iterate slots to produce the bit result. O(n) extra pass, not chunked.
|
||||
|
||||
### add_col_from — matrix builder integration
|
||||
|
||||
Both matrix builders accept temp-file results directly:
|
||||
|
||||
```rust
|
||||
// PersistentBitMatrixBuilder
|
||||
fn add_col_from(&mut self, src: &TempBitVec) -> io::Result<()>
|
||||
fn add_col_from_int(&mut self, src: &TempCompactIntVec) -> io::Result<()> // nonzero → 1
|
||||
|
||||
// PersistentCompactIntMatrixBuilder
|
||||
fn add_col_from(&mut self, src: &TempCompactIntVec) -> io::Result<()>
|
||||
fn add_col_from_bit(&mut self, src: &TempBitVec) -> io::Result<()> // bit → 0/1 u32
|
||||
```
|
||||
|
||||
`add_col_from` copies the temp file to the matrix directory and increments `n_cols`; `close()` writes `meta.json` with the final column count. No separate `write_meta` step needed.
|
||||
|
||||
### mask_with
|
||||
|
||||
Direct method on `PersistentCompactIntVecBuilder` (and delegation via `TempCompactIntVecBuilder`). Zeros every slot where the corresponding mask bit is 0. Iterates only zero bits — O(n_zeros), O(1) when mask is all-ones.
|
||||
|
||||
```
|
||||
for (w_idx, word) in mask.words():
|
||||
if word == u64::MAX: continue // skip all-ones words
|
||||
zeros = !word
|
||||
while zeros != 0:
|
||||
bit = trailing_zeros(zeros)
|
||||
s = w_idx * 64 + bit
|
||||
if primary[s] != 0: set(s, 0) // clears overflow entry too
|
||||
zeros &= zeros − 1
|
||||
```
|
||||
|
||||
Terminal operation for Filter (retain only selected kmer slots in a count vector) and Select (positional selection without MPHF).
|
||||
@@ -0,0 +1,143 @@
|
||||
# `obitaxonomy` — taxonomy concept paths
|
||||
|
||||
`obitaxonomy` is a dependency-free crate that defines a typed representation
|
||||
of hierarchical concept paths (taxonomic or otherwise) stored in genome metadata.
|
||||
|
||||
---
|
||||
|
||||
## Concept path syntax
|
||||
|
||||
A concept path is stored as a metadata value with the prefix `taxonomy:/`:
|
||||
|
||||
```
|
||||
taxonomy:/enterobacteriaceae@family/Escherichia@genus/Escherichia coli@species
|
||||
```
|
||||
|
||||
Structure:
|
||||
|
||||
- The `taxonomy:/` prefix is the type discriminator. Any metadata value starting
|
||||
with it is parsed as a `TaxPath`; all others remain plain strings.
|
||||
- The remainder is one or more `/`-separated segments.
|
||||
- Each segment is `name` or `name@rank`, where `rank` is a label for the
|
||||
taxonomic level (e.g. `family`, `genus`, `species`).
|
||||
- Rank annotations are **optional per segment** and can be mixed freely.
|
||||
- Spaces are allowed in both names and ranks.
|
||||
|
||||
### Reserved character
|
||||
|
||||
`@` is reserved throughout the taxonomy system and may **not** appear in:
|
||||
|
||||
| Context | Constraint |
|
||||
|---------|------------|
|
||||
| Segment name | forbidden |
|
||||
| Rank/class label | forbidden |
|
||||
| Metadata key names | forbidden (used as `key@rank` in predicate syntax) |
|
||||
|
||||
`@` is freely allowed in plain-text metadata values (non-taxonomy).
|
||||
|
||||
### Parse errors
|
||||
|
||||
| Condition | Error |
|
||||
|-----------|-------|
|
||||
| Value does not start with `taxonomy:/` | `MissingPrefix` |
|
||||
| No segments after the prefix | `EmptyPath` |
|
||||
| Segment with empty name (consecutive `/`) | `EmptySegmentName` |
|
||||
| Segment with trailing `@` and no rank (`name@`) | `EmptyRankName` |
|
||||
| Segment with more than one `@` | `AmbiguousRank` |
|
||||
|
||||
---
|
||||
|
||||
## Public API
|
||||
|
||||
### `TaxSegment`
|
||||
|
||||
A single node: a name and an optional rank.
|
||||
|
||||
```rust
|
||||
seg.name() // &str
|
||||
seg.rank() // Option<&str>
|
||||
seg.to_string() // "name" or "name@rank"
|
||||
TaxSegment::parse(s) // Result<TaxSegment, TaxError>
|
||||
```
|
||||
|
||||
### `TaxPath`
|
||||
|
||||
```rust
|
||||
TaxPath::parse(s) // Result<TaxPath, TaxError>
|
||||
path.segments() // &[TaxSegment]
|
||||
path.depth() // usize — number of segments
|
||||
path.is_ancestor_of(&other) // bool — prefix match by name, ranks ignored
|
||||
path.name_at_rank("genus") // Option<&str>
|
||||
path.to_string() // reconstructs "taxonomy:/…"
|
||||
```
|
||||
|
||||
`is_ancestor_of` compares segment **names** only — rank annotations are
|
||||
informational and do not affect the ancestry relation.
|
||||
|
||||
```rust
|
||||
let a: TaxPath = "taxonomy:/Enterobacteriaceae@family/Escherichia@genus".parse()?;
|
||||
let b: TaxPath = "taxonomy:/Enterobacteriaceae@family/Escherichia@genus/Escherichia coli@species".parse()?;
|
||||
|
||||
assert!(a.is_ancestor_of(&b)); // true
|
||||
assert!(b.is_ancestor_of(&a)); // false
|
||||
assert!(a.is_ancestor_of(&a)); // true (equal ⇒ ancestor)
|
||||
|
||||
assert_eq!(b.name_at_rank("species"), Some("Escherichia coli"));
|
||||
assert_eq!(b.name_at_rank("genus"), Some("Escherichia"));
|
||||
assert_eq!(b.name_at_rank("order"), None);
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Integration with `GenomeInfo`
|
||||
|
||||
At index load time, every metadata value is inspected once:
|
||||
|
||||
- Starts with `taxonomy:/` → parsed into `TaxPath`, stored in `genome.taxonomy`.
|
||||
- Otherwise → kept as-is in `genome.meta`.
|
||||
|
||||
```rust
|
||||
struct GenomeInfo {
|
||||
label: String,
|
||||
meta: HashMap<String, String>, // plain text metadata
|
||||
taxonomy: HashMap<String, TaxPath>, // parsed taxonomy metadata
|
||||
}
|
||||
```
|
||||
|
||||
The raw string is not duplicated. `TaxPath::to_string()` reconstructs the
|
||||
original value losslessly for serialisation.
|
||||
|
||||
---
|
||||
|
||||
## Predicate operators (in `filter` / `select`)
|
||||
|
||||
Path predicates use the `~` / `!~` operators. The **stored value** always starts
|
||||
with `/` (rooted path); the **query pattern** does not need to.
|
||||
|
||||
### Path pattern syntax
|
||||
|
||||
| Pattern | Semantics |
|
||||
|---------|-----------|
|
||||
| `A/B` | contiguous sub-path A then B, anywhere in the value |
|
||||
| `/A/B` | value starts with A then B (start-anchored) |
|
||||
| `A/B$` | value ends with A then B (end-anchored) |
|
||||
| `/A/B$` | value is exactly A then B (fully anchored) |
|
||||
| `A@x/B` | A with class `x` followed by B with any class |
|
||||
| `A@x/B@y` | A with class `x` followed by B with class `y` |
|
||||
|
||||
A segment pattern without `@` matches the segment name regardless of its stored class.
|
||||
|
||||
### Rank-aware queries
|
||||
|
||||
```
|
||||
key@rank=value
|
||||
```
|
||||
|
||||
| Predicate form | Semantics |
|
||||
|----------------|-----------|
|
||||
| `key@rank=value` | genome's `key` has `value` at rank `rank` |
|
||||
| `key@rank!=value` | does not |
|
||||
| `key@rank=v1\|v2` | value at `rank` is `v1` or `v2` |
|
||||
|
||||
`~` combined with `@rank` on the key (e.g. `key@genus~pattern`) is not defined
|
||||
and is rejected at parse time.
|
||||
+1
-1
@@ -13,7 +13,7 @@
|
||||
| `query` | Query an index with sequences and annotate matches |
|
||||
| `dump` | Dump all indexed k-mers as CSV (kmer + per-genome counts or presence); supports the shared [kmer filtering](implementation/filtering.md) system; `--head N` limits output to the first N k-mers |
|
||||
| `annotate` | Add or update genome metadata from a CSV file; or dump metadata as CSV |
|
||||
| `distance` | Compute pairwise distance matrix between genomes; optionally build NJ/UPGMA trees; `--presence-threshold N` sets the minimum count to consider a k-mer present when computing Jaccard on count indexes (default 1) |
|
||||
| `distance` | Compute pairwise distance matrix between genomes (`--metric jaccard\|mash\|hamming\|bray-curtis\|relfreq-bray-curtis\|euclidean\|relfreq-euclidean\|hellinger\|hellinger-euclidean`); optionally build NJ/UPGMA trees; `--presence-threshold N` sets the minimum count to consider a k-mer present when computing Jaccard/Mash on count indexes (default 1) |
|
||||
| `unitig` | Build a global de Bruijn graph across all partitions and enumerate its unitigs as FASTA; supports the shared [kmer filtering](implementation/filtering.md) system |
|
||||
| `select` | Project and/or aggregate genome columns into a new or in-place index; the column-axis counterpart of `filter` (see [select](implementation/select.md)) |
|
||||
| `estimate` | Estimate approximate-index parameters (z, evidence bits, FP rates) before indexing |
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
# Installation
|
||||
|
||||
## Prerequisites
|
||||
|
||||
### Rust toolchain
|
||||
|
||||
`obikmer` requires **Rust 1.85 or later** (edition 2024). Install or update via [rustup](https://rustup.rs):
|
||||
|
||||
```bash
|
||||
curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh
|
||||
rustup update stable
|
||||
```
|
||||
|
||||
### C build environment (required for hwloc)
|
||||
|
||||
`obikmer` embeds [hwloc](https://www.open-mpi.org/projects/hwloc/) (Hardware Locality) for NUMA-aware thread placement on multi-socket machines. hwloc is built from source at compile time via the `vendored` feature of the `hwlocality` crate. This requires a standard C build environment.
|
||||
|
||||
#### Linux (Debian/Ubuntu)
|
||||
|
||||
```bash
|
||||
apt install build-essential automake libtool autoconf pkg-config
|
||||
```
|
||||
|
||||
#### Linux (RHEL/Rocky/AlmaLinux)
|
||||
|
||||
```bash
|
||||
dnf install gcc make automake libtool autoconf pkgconfig
|
||||
```
|
||||
|
||||
#### HPC clusters
|
||||
|
||||
Most HPC clusters provide these tools via the module system:
|
||||
|
||||
```bash
|
||||
module load gcc automake libtool autoconf
|
||||
```
|
||||
|
||||
If in doubt, check whether `autoreconf --version` and `libtool --version` return successfully.
|
||||
|
||||
#### macOS
|
||||
|
||||
```bash
|
||||
brew install automake libtool autoconf pkg-config
|
||||
```
|
||||
|
||||
## Building
|
||||
|
||||
```bash
|
||||
git clone <repository-url>
|
||||
cd obikmer/src
|
||||
cargo build --release
|
||||
```
|
||||
|
||||
The compiled binary is at `target/release/obikmer`.
|
||||
|
||||
### Building on HPC clusters (network filesystems)
|
||||
|
||||
HPC home directories are typically on a network filesystem (Lustre, NFS) optimised for large sequential reads — not for the thousands of small file operations that Cargo generates during compilation. Building directly on such a filesystem can be extremely slow (0.1% CPU utilisation, tens of minutes for what should take seconds).
|
||||
|
||||
**Always redirect the build directory to a local scratch disk:**
|
||||
|
||||
```bash
|
||||
CARGO_TARGET_DIR=/scratch/$USER/cargo-target cargo build --release
|
||||
```
|
||||
|
||||
Adapt the path to the local scratch available on your cluster (`/var/tmp`, `/tmp`, `/scratch/local`, etc.). Once built, copy the binary to a permanent location:
|
||||
|
||||
```bash
|
||||
cp /scratch/$USER/cargo-target/release/obikmer ~/bin/
|
||||
```
|
||||
|
||||
## NUMA support
|
||||
|
||||
NUMA-aware thread placement is active automatically on multi-socket Linux machines (detected at runtime via hwloc). No special build flag is required — the detection is built in and falls back gracefully to the single-pool adaptive strategy on:
|
||||
|
||||
- macOS (Apple Silicon, unified memory)
|
||||
- single-socket Linux machines
|
||||
- any system where hwloc reports only one NUMA node
|
||||
|
||||
## Verifying the installation
|
||||
|
||||
```bash
|
||||
obikmer --help
|
||||
```
|
||||
@@ -241,3 +241,21 @@
|
||||
volume = 33,
|
||||
year = 2017,
|
||||
bdsk-url-1 = {http://dx.doi.org/10.1093/bioinformatics/btw832}}
|
||||
|
||||
@misc{Mash-distances-doc,
|
||||
author = {{Marbl Lab}},
|
||||
howpublished = {Mash documentation},
|
||||
title = {Mash Distance},
|
||||
url = {https://mash.readthedocs.io/en/latest/distances.html},
|
||||
urldate = {2026-07-09},
|
||||
year = 2026}
|
||||
|
||||
@article{Fan2015-mash-formula,
|
||||
author = {Fan, Huan and Ives, Anthony R and Surget-Groba, Yann and Cannon, Charles H},
|
||||
doi = {10.1186/s12864-015-1647-5},
|
||||
journal = {BMC Genomics},
|
||||
number = 1,
|
||||
title = {An assembly and alignment-free method of phylogeny reconstruction from next-generation sequencing data},
|
||||
url = {https://doi.org/10.1186/s12864-015-1647-5},
|
||||
volume = 16,
|
||||
year = 2015}
|
||||
|
||||
+32
-16
@@ -1,6 +1,6 @@
|
||||
# Kmer entropy filter
|
||||
|
||||
Low-complexity kmers (polyA, polyT, tandem repeats) are detected and excluded during phase 1. The filter computes a **normalized Shannon entropy** over sub-words of multiple sizes, corrected for two sources of bias: the small number of observations within a single kmer, and the unequal sizes of circular equivalence classes.
|
||||
Low-complexity kmers (polyA, polyT, tandem repeats) are detected and excluded during phase 1. The filter computes a **normalized Shannon entropy** over sub-words of multiple sizes, corrected for one source of bias: the small number of observations within a single kmer relative to the number of possible sub-words.
|
||||
|
||||
## Sub-word frequencies
|
||||
|
||||
@@ -8,17 +8,15 @@ For a kmer of length k and a sub-word size ws (1 ≤ ws ≤ ws_max, typically ws
|
||||
|
||||
$$w_i = \text{kmer}[i \mathinner{..} i+ws-1], \quad i = 0, \ldots, n_{\text{words}}-1$$
|
||||
|
||||
Each sub-word is mapped to its **circular canonical form**: the lexicographic minimum among all cyclic rotations of the word **and all cyclic rotations of its reverse complement**. This extended equivalence relation ensures that entropy(K) = entropy(revcomp(K)) — the filter is strand-symmetric. Let $s_j$ be the size of equivalence class $j$ (number of distinct raw words mapping to canonical form $j$), and $f_j$ the count of canonical form $j$ among the $n_{\text{words}}$ sub-words ($\sum_j f_j = n_{\text{words}}$).
|
||||
Each sub-word is tallied under its own raw 2-bit-packed value — **no canonicalization**. Let $f_j$ be the count of raw word $j$ among the $n_{\text{words}}$ sub-words ($\sum_j f_j = n_{\text{words}}$), over the $4^{ws}$ possible raw words.
|
||||
|
||||
An earlier version of this filter first folded each sub-word into a circular+reverse-complement equivalence class, then "unfolded" the observed class frequency back onto its members to correct for unequal class sizes. That machinery bought nothing it was claimed for — see *Why no equivalence classes* below — while measurably weakening detection of the very sequences the filter exists to catch, so it was removed.
|
||||
|
||||
## Corrected Shannon entropy
|
||||
|
||||
The circular equivalence classes have unequal sizes: under a uniform distribution over all $4^{ws}$ raw words, class $j$ is visited with probability $s_j / 4^{ws}$, not $1/n_a$. Computing entropy directly over canonical classes therefore underestimates the entropy of a random sequence.
|
||||
$$H_{\text{corr}} = \log(n_{\text{words}}) - \frac{1}{n_{\text{words}}} \sum_j f_j \log f_j$$
|
||||
|
||||
The correction "unfolds" each canonical class back to its member raw words, redistributing each observation of class $j$ equally among its $s_j$ members:
|
||||
|
||||
$$H_{\text{corr}} = \log(n_{\text{words}}) - \frac{1}{n_{\text{words}}} \sum_j f_j \log f_j + \frac{1}{n_{\text{words}}} \sum_j f_j \log s_j$$
|
||||
|
||||
The last term is the correction for unequal class sizes. For a uniformly random sequence ($f_j \approx n_{\text{words}} \cdot s_j / 4^{ws}$), this gives $H_{\text{corr}} \approx \log(4^{ws}) = 2 \cdot ws \cdot \log 2$, the maximum entropy over raw words.
|
||||
This is a plain Shannon entropy over the observed raw-word frequencies.
|
||||
|
||||
## Maximum entropy correction for small samples
|
||||
|
||||
@@ -42,27 +40,45 @@ $$\text{entropy}(kmer) = \min_{ws=1}^{ws_{\max}} \hat{H}(ws)$$
|
||||
|
||||
A value near 0 indicates low complexity (e.g. AAAA…); near 1 indicates high complexity. A kmer is rejected if $\text{entropy}(kmer) < \theta$, where $\theta$ is a collection parameter (default 0.7). The minimum across word sizes ensures that any scale of repetition is detected independently: polyA is caught at ws=1, dinucleotide repeats at ws=2, etc.
|
||||
|
||||
## Why no equivalence classes
|
||||
|
||||
A prior design folded each sub-word into the canonical form of its circular-rotation + reverse-complement equivalence class before tallying, on the reasoning that (a) it guarantees $\text{entropy}(K) = \text{entropy}(\text{revcomp}(K))$, and (b) collapsing phase-shifted repeats (e.g. `ATG` ≡ `TGA` ≡ `GAT`) into one class better reflects that they are "the same" low-complexity pattern.
|
||||
|
||||
Both properties already hold for the raw, unfolded entropy above, without any class machinery:
|
||||
|
||||
- **Reverse complement**: for any K of length n, window $j$ of $\text{revcomp}(K)$ equals $\text{revcomp}$ of window $(n{-}ws{-}j)$ of K. This is a bijection between the window sets under which each window maps to its own revcomp — and revcomp is itself a bijection (involution) on the space of raw ws-mers. So the multiset of raw-word frequencies for $\text{revcomp}(K)$ is exactly a relabeling of the multiset for K, and Shannon entropy — a function of the frequency multiset alone — is exactly invariant. No folding required, for any K.
|
||||
- **Tandem repeats**: a period-p repeat sampled by a stride-1 sliding window naturally cycles through its own rotations as raw tokens (e.g. `ATGATGATG…` yields the raw words `ATG`, `TGA`, `GAT` in rotation as the window slides). The low diversity this represents (few distinct raw words out of $4^{ws}$ possible) is already visible in the raw frequency distribution — no folding needed to detect it.
|
||||
|
||||
What the fold-then-unfold step actually did was credit each observed class with the frequency of equivalence-class members that were **never observed on the read strand**, inflating $H_{\text{corr}}$ for genuine repeats. Worked example: k=31, ws=3, kmer = `ATG` repeated ($n_{\text{words}}=29$, all 29 windows fall into one class of size 6 under the old scheme — 3 rotations × forward/revcomp):
|
||||
|
||||
| | $H_{\text{corr}}$ | normalized |
|
||||
|---|---|---|
|
||||
| old (folded, class size 6) | $\log 6 \approx 1.79$ | $\approx 0.53$ |
|
||||
| current (raw, unfolded) | $\log 3 \approx 1.10$ | $\approx 0.33$ |
|
||||
|
||||
The gap is not a rounding artifact: per sub-word order, the folded score for this same repeat swings from 0.53 (ws=3, aligned with the period) up to **1.03** (ws=5, misaligned with the period) — i.e. a period-3 repeat could score *above* the theoretical maximum for a random sequence, depending on which ws happens to divide the repeat's period. The raw formula stays flat at ≈0.33–0.40 across ws=2..6 regardless of alignment, which is the robustness the "minimum across ws" design was meant to provide in the first place.
|
||||
|
||||
## Interpretation as an effective number of classes
|
||||
|
||||
$H_{\text{corr}}$ is a standard Shannon entropy over raw words (after unfolding the equivalence classes), so the classical perplexity interpretation holds directly: $N_{\text{eff}} = e^{H_{\text{corr}}}$ is the number of equiprobable classes that would yield the same entropy.
|
||||
$H_{\text{corr}}$ is a standard Shannon entropy over raw words, so the classical perplexity interpretation holds directly: $N_{\text{eff}} = e^{H_{\text{corr}}}$ is the number of equiprobable raw words that would yield the same entropy.
|
||||
|
||||
For the normalised score $\hat{H}$, dividing by $H_{\text{max}}$ changes the logarithm base:
|
||||
For the normalised score $\hat{H}$, dividing by $H_{\max}$ changes the logarithm base:
|
||||
|
||||
$$\hat{H} = \frac{\log N_{\text{eff}}}{\log N_{\text{max}}} = \log_{N_{\text{max}}} N_{\text{eff}} \quad \Longleftrightarrow \quad N_{\text{eff}} = N_{\text{max}}^{\,\hat{H}}$$
|
||||
$$\hat{H} = \frac{\log N_{\text{eff}}}{\log N_{\max}} = \log_{N_{\max}} N_{\text{eff}} \quad \Longleftrightarrow \quad N_{\text{eff}} = N_{\max}^{\,\hat{H}}$$
|
||||
|
||||
The property is preserved: $\hat{H}$ is the logarithm (in base $N_{\text{max}}$) of the effective number of equi-represented classes.
|
||||
The property is preserved: $\hat{H}$ is the logarithm (in base $N_{\max}$) of the effective number of equi-represented raw words.
|
||||
|
||||
In the large-sample limit ($n_{\text{words}} \gg 4^{ws}$), $N_{\text{max}} \approx 4^{ws}$, giving:
|
||||
In the large-sample limit ($n_{\text{words}} \gg 4^{ws}$), $N_{\max} \approx 4^{ws}$, giving:
|
||||
|
||||
$$N_{\text{eff}} \approx 4^{ws \cdot \hat{H}}$$
|
||||
|
||||
This has a clean interpretation: $ws \cdot \hat{H}$ is the **effective word length** (in bases) of a perfectly uniform distribution that would produce the same entropy. At $\hat{H} = 1$ the full space of $4^{ws}$ words is used; at $\hat{H} = 0.5$ with ws=2, only $4^1 = 4$ effective classes out of 16 are occupied.
|
||||
This has a clean interpretation: $ws \cdot \hat{H}$ is the **effective word length** (in bases) of a perfectly uniform distribution that would produce the same entropy. At $\hat{H} = 1$ the full space of $4^{ws}$ words is used; at $\hat{H} = 0.5$ with ws=2, only $4^1 = 4$ effective words out of 16 are occupied.
|
||||
|
||||
In our actual regime, $n_{\text{words}}$ is small and $4^{ws}$ can exceed $n_{\text{words}}$, so $H_{\text{max}} < \log(4^{ws})$ due to the small-sample correction. The exact effective count is $N_{\text{max}}^{\hat{H}}$, not $4^{ws \cdot \hat{H}}$.
|
||||
In our actual regime, $n_{\text{words}}$ is small and $4^{ws}$ can exceed $n_{\text{words}}$, so $H_{\max} < \log(4^{ws})$ due to the small-sample correction. The exact effective count is $N_{\max}^{\hat{H}}$, not $4^{ws \cdot \hat{H}}$.
|
||||
|
||||
## Properties
|
||||
|
||||
The entropy score is a function of the kmer sequence alone — it does not depend on the surrounding context or on the position within any genome. Two consequences:
|
||||
|
||||
- **Orientation invariance**: $\text{entropy}(K) = \text{entropy}(\text{revcomp}(K))$, guaranteed by the strand-symmetric canonical form.
|
||||
- **Orientation invariance**: $\text{entropy}(K) = \text{entropy}(\text{revcomp}(K))$ — see *Why no equivalence classes* above for why this holds without any explicit strand-folding step.
|
||||
- **Context independence**: the same kmer is always rejected or always kept, regardless of which genome it occurs in, where in that genome it appears, or which strand is considered. The filter defines a fixed partition of the kmer space into low-complexity and valid kmers.
|
||||
|
||||
@@ -3,10 +3,14 @@
|
||||
|
||||
## Code couvert
|
||||
|
||||
- `obiskbuilder/src/entropy_table.rs` — filtre Shannon sur les kmers à basse complexité
|
||||
- `obiskbuilder/src/lib.rs` — application du filtre lors du scatter (phase 1)
|
||||
- `obikentropy/src/table.rs`, `obikentropy/src/tracker.rs` — formule d'entropie et tables de correction petits effectifs
|
||||
- `obikentropy/src/kmer_entropy.rs` — entropie d'un kmer isolé (`KmerEntropy`)
|
||||
- `obiskbuilder/src/rolling_stat.rs` — composition de `obikentropy::EntropyTracker` dans le suivi streaming (sélection de minimiseur + entropie)
|
||||
- `obiskbuilder/src/iter.rs`, `obiskbuilder/src/stream_iter.rs` — application du filtre lors du scatter (phase 1)
|
||||
|
||||
## Notes
|
||||
|
||||
Document théorique stable. Vérifier que les paramètres `theta` et `level_max` dans le CLI
|
||||
Le repli en classes d'équivalence circulaires + brin inverse (décrit dans une version antérieure de ce document) a été supprimé : voir la section « Why no equivalence classes » de `entropy.md` pour la justification théorique et numérique.
|
||||
|
||||
Vérifier que les paramètres `theta` et `level_max` dans le CLI
|
||||
(`obikmer/src/cli.rs` → `CommonArgs`) correspondent bien à ce qui est décrit.
|
||||
|
||||
@@ -0,0 +1,994 @@
|
||||
# Central-position SNP distance (discussion)
|
||||
|
||||
Not implemented. Design discussion for a substitution-rate estimator that
|
||||
observes SNPs directly from paired-genome k-mer comparison, as an alternative
|
||||
to Mash's Poisson-Jaccard inference (see [obicompactvec](../implementation/obicompactvec.md)
|
||||
for the implemented Jaccard/Mash distances).
|
||||
|
||||
## Motivation
|
||||
|
||||
**Primary intent: restrict the comparison to what is actually comparable.**
|
||||
Mash's Jaccard is computed over the **union** of both genomes' k-mer content:
|
||||
anything not identically shared is folded into a single undifferentiated
|
||||
mass, whether the cause is a point substitution, a genuinely absent
|
||||
homologous region (lineage-specific content, gene-family expansion, HGT,
|
||||
genome-size asymmetry), or a diverged paralogous copy. The model then
|
||||
back-infers a single mutation rate from that mass, silently attributing
|
||||
non-homology to mutation. The central-SNP approach instead conditions every
|
||||
comparison on local, positive evidence of homology: a locus only enters the
|
||||
statistic if its `2m` flanking bases (`m = (k-1)/2`) are found intact in
|
||||
*both* genomes — genuinely absent or non-homologous content is excluded from
|
||||
the comparison entirely (neither numerator nor denominator), rather than
|
||||
silently counted as divergence. This is a conditioning on comparability, not
|
||||
just a richer summary statistic — see "Statistic and correspondence with
|
||||
`shared`" below for how it plays out against genome-size asymmetry and
|
||||
diverged gene families, and "Heterozygosity, ploidy, and consensus-assembly
|
||||
inputs" for the corresponding paralogy/heterozygosity filter.
|
||||
|
||||
**Secondary benefit: access to the substitution's nature.** Because the
|
||||
central base of an odd-k window is directly observable once the flanks are
|
||||
confirmed conserved, this also yields more than a rate — the
|
||||
transition/transversion split — enabling classical corrected distances
|
||||
(Jukes-Cantor, Kimura 2-parameter, LogDet) that a single Jaccard scalar
|
||||
cannot support.
|
||||
|
||||
## Statistic and correspondence with `shared`
|
||||
|
||||
A genomic position `p` is covered by `k` overlapping k-mer windows. Requiring
|
||||
the substitution to sit at the window's **center** makes exactly one window
|
||||
per SNP eligible — a 1:1 correspondence between SNP and center-neighbor k-mer
|
||||
pair, avoiding the ~k-fold overcount of an any-position neighbor search.
|
||||
|
||||
A locus with a fully conserved `k`-window (flanks **and** center) is an
|
||||
exact-shared k-mer at that locus; a locus with conserved flanks but a
|
||||
substituted center is a "central SNP". Both count each locus exactly once, in
|
||||
matching units:
|
||||
|
||||
```
|
||||
p_hat[i,j] = SNP[i,j] / (SNP[i,j] + shared[i,j])
|
||||
```
|
||||
|
||||
`p_hat` is `P(center substituted | 2m flanks conserved)`. `shared[i,j]` here
|
||||
is **not** the general-purpose `shared_kmers` matrix used by Jaccard/Mash
|
||||
(`--shared-kmers`, `BitPartials::partial_jaccard` /
|
||||
`CountPartials::partial_threshold_jaccard`) — that matrix counts raw k-mer
|
||||
identity with no per-genome copy-number constraint, whereas `p_hat`'s
|
||||
denominator applies the eligibility rule defined below (raw or
|
||||
paralogy-filtered). Both `SNP` and `shared` are accumulated by the same
|
||||
sweep, from the same per-locus candidate set (source k-mer + 3 variants),
|
||||
under the same eligibility rule — see "Locus eligibility" below, and
|
||||
"Heterozygosity, ploidy, and consensus-assembly inputs" for why the
|
||||
copy-number constraint matters and what it costs.
|
||||
|
||||
**Canonical invariance**: for odd k, the central position maps to itself under
|
||||
reverse-complement (`m -> k-1-m = m`, base complemented). A transition maps to
|
||||
a transition, a transversion to a transversion — the transition/transversion
|
||||
split is well-defined in canonical space.
|
||||
|
||||
### Definitions: family, and the canonical form of a family
|
||||
|
||||
**Family.** The family of a k-mer `x` is the set of (up to) 4 k-mers sharing
|
||||
`x`'s `2m` flanking bases, differing only at the central base `m`. Membership
|
||||
is a property of the flank pattern, not of `x` itself: any of the 4 possible
|
||||
central substitutions belongs to the same family.
|
||||
|
||||
**`central_canonical_neighbors()`** (`obikseq`, `CanonicalKmerOf::central_canonical_neighbors`)
|
||||
generates all 4 members from any one of them (observed or not), each
|
||||
independently canonicalised (`.canonical()`, i.e. `min(kmer, revcomp(kmer))`).
|
||||
This independent canonicalisation is necessary because a central substitution
|
||||
can flip which orientation is lexicographically smaller — two members of the
|
||||
same family can end up canonicalised in *different* orientations. Despite
|
||||
that, the **set** of 4 resulting canonical k-mers is invariant: calling
|
||||
`central_canonical_neighbors()` on any member of a family — present in the
|
||||
index or not — yields the same 4 values. This is relied upon throughout the
|
||||
rest of this document.
|
||||
|
||||
**Canonical form of a family.** Because orientation can differ member to
|
||||
member, "which of the 4 is the reference" cannot be defined relative to
|
||||
*whichever member happened to be visited first*, nor relative to the
|
||||
minorant (see below) — both are data-dependent (they depend on what is
|
||||
actually observed), so using either as the reference would make the
|
||||
reference itself vary depending on what happens to be present in a given
|
||||
index. Instead: **the canonical form of a family is, by definition, the
|
||||
member whose own central base — read in its own already-canonical
|
||||
orientation — is `A`.** This is well-defined for every family, computed
|
||||
purely from the flank pattern, whether or not that specific member (or any
|
||||
member at all) is actually observed anywhere in the index. Concretely: call
|
||||
`central_canonical_neighbors()` on any member (observed or not) to get the
|
||||
family's 4 canonical forms; the one among them whose own centre nucleotide is
|
||||
`A` is the family's canonical form. The other 3 (`C`, `G`, `T`) are labelled
|
||||
relative to *that* fixed reference, not relative to the calling member's own
|
||||
orientation.
|
||||
|
||||
**Consequence for the minorant.** With this fixed A-referenced labelling,
|
||||
`minorant` (the smallest raw encoding among the family's *observed* members,
|
||||
introduced further below) becomes directly computable rather than needing to
|
||||
be tracked as extra state: regenerate the family's 4 canonical forms from
|
||||
any member's own k-mer (cheap, no lookup), compare the raw encodings of
|
||||
whichever are marked present, and take the smallest. No separate stored bit
|
||||
is required — see Step 2b below, where this replaces the earlier
|
||||
minorant-bit design.
|
||||
|
||||
## Locus eligibility: raw definition vs. paralogy filter
|
||||
|
||||
For each k-mer `x` observed in genome A (source, one MPHF slot; the 3
|
||||
central-position variants generated as in the sweep below): check whether
|
||||
A's locus (flanks fixed) is resolvable in genome B under one of the 4
|
||||
central forms.
|
||||
|
||||
**Raw / no model.** The locus counts in the denominator iff at least one of
|
||||
the 4 forms is present in B; it counts in the numerator iff the form found in
|
||||
B differs from A's own. No constraint on A's or B's own copy number at this
|
||||
locus. Open question, not resolved: what if **more than one** of the 4 forms
|
||||
is present in B simultaneously (ambiguous target — count once arbitrarily,
|
||||
count all, or drop)? The stringent filter below sidesteps the question by
|
||||
construction rather than answering it.
|
||||
|
||||
**Stringent / paralogy-aware.** The locus counts only if exactly one of the
|
||||
4 forms is present in A **and** exactly one is present in B (`count == 1` at
|
||||
that slot too, when a count index is available, to also exclude same-allele
|
||||
duplicates that presence alone cannot see). This drops the raw definition's
|
||||
ambiguous-B case automatically, at the cost of also dropping heterozygous
|
||||
sites indiscriminately alongside true duplications (see "Heterozygosity,
|
||||
ploidy, and consensus-assembly inputs" below).
|
||||
|
||||
**Rejected: parsimony-based multiset pairing for multiplicity > 1.** Rather
|
||||
than dropping ambiguous loci, pair identical alleles between A and B first
|
||||
(0-mutation explanation preferred), then take `min(unmatched_A, unmatched_B)`
|
||||
as inferred SNP pairs. Rejected on two grounds: (1) circularity — selecting
|
||||
pairs by minimal apparent divergence, then measuring divergence on those same
|
||||
pairs, deflates the estimate by construction, not a neutral heuristic; (2)
|
||||
the discriminating signal is a single base among 4 possible values, and the
|
||||
flanks are *already* guaranteed identical for every candidate by
|
||||
construction (that is how the locus was selected) — no information remains
|
||||
in a k-mer window to tell which copy in B truly corresponds to which copy in
|
||||
A once multiplicity > 1 on either side. Any pairing rule invents a
|
||||
correspondence the data cannot support. Multiplicity > 1 is treated as
|
||||
non-identifiable, not as a puzzle to solve with a heuristic.
|
||||
|
||||
## Multi-genome framing: family as pseudo-alignment column
|
||||
|
||||
**Idea.** Instead of resolving locus eligibility and correspondence one
|
||||
genome pair at a time, treat a family as a column of a pseudo multiple
|
||||
alignment across *all* genomes simultaneously: for each family, each genome
|
||||
has either a net single-copy state (`A`/`C`/`G`/`T`, when the genome carries
|
||||
exactly one of the 4 forms) or "missing" (`?`, multi-copy or absent). Flank
|
||||
conservation (the `2m` bases fixed by construction) supplies positional
|
||||
homology for free — the same role a real MSA would play, without alignment
|
||||
software, gap penalties, or progressive-alignment approximations. Stacking
|
||||
one such column per family, genomes as rows, produces a genuine SNP
|
||||
pseudo-alignment matrix, not just a bag of pairwise distances.
|
||||
|
||||
**Precedent.** This is the same principle behind reference-free
|
||||
k-mer-based phylogenomics tools — SKA (Split K-mer Analysis, Harris 2018) and
|
||||
kSNP: split the k-mer around a variable center, use flank identity to call
|
||||
homologous columns across arbitrarily many genomes with no reference and no
|
||||
MSA step, then feed the resulting pseudo-alignment to standard phylogenetic
|
||||
tools. Landing on the same design independently is a good sign, not a
|
||||
coincidence.
|
||||
|
||||
**Resolves the pairwise-correspondence problem, properly.** The "Rejected:
|
||||
parsimony-based multiset pairing" case above failed because, with only two
|
||||
genomes' cardinalities to look at, there is no external constraint to justify
|
||||
picking one correspondence between leftover alleles over another — `min(a,b)`
|
||||
is a lower bound dressed up as a point estimate (see the follow-up discussion
|
||||
on Felsenstein-style parsimony inconsistency: minimum-event explanations are
|
||||
systematically biased low whenever homoplasy/multiplicity is real, not
|
||||
noise-cancelling). With `N` genomes and many families jointly, the same
|
||||
question can be answered the way real phylogenetics answers it: ancestral
|
||||
state reconstruction / ML mapping over a tree estimated from the whole
|
||||
column set. The tree supplies the missing constraint that two isolated
|
||||
columns cannot — this is the principled way out, not a heuristic replacement
|
||||
for one.
|
||||
|
||||
**Relation to what's already implemented.** `KmerIndex::raw_snp_distance`
|
||||
already computes, internally, per family, exactly this row — `single_form:
|
||||
Vec<Option<u8>>`, one entry per genome, `None` where ambiguous/absent —
|
||||
before immediately collapsing it into pairwise `snp[i,j]`/`shared[i,j]`
|
||||
tallies. The pivot this section proposes is small at the implementation
|
||||
level: stop collapsing early, and surface the per-family row as a first-class
|
||||
artifact (a `families x genomes` matrix). Pairwise raw p-distance becomes one
|
||||
projection of that matrix (what's computed today), not the primary object;
|
||||
downstream, the matrix itself could feed real phylogenetic tools (parsimony/
|
||||
ML, e.g. RAxML/IQ-TREE-style) instead of only NJ/UPGMA on a homemade
|
||||
pairwise-distance matrix.
|
||||
|
||||
**Caveat: column completeness shrinks with `N`.** The probability that a
|
||||
family's flanks stay intact simultaneously across all `N` genomes decays with
|
||||
`N` (same ascertainment-bias mechanism as Bias 1 above, compounded over more
|
||||
genomes) — fully-resolved columns (no `?` anywhere) become rare as more
|
||||
genomes are added. Same missing-data situation any real multi-species
|
||||
alignment faces, and phylogenetic tools already handle it well; the practical
|
||||
implication is that columns should be allowed partial coverage (>=2 resolved
|
||||
genomes, not unanimous) rather than requiring every genome to be net
|
||||
single-copy at that locus.
|
||||
|
||||
## Context, detectability, and a 3-way ordinal distance per pair
|
||||
|
||||
Empirical follow-up to the pseudo-alignment idea above: `obikmer distance
|
||||
--snp` was run on a real 20-genome benchmark index and the resulting FASTA
|
||||
fed to `raxml-ng`. Two problems surfaced, both traced back to conflating
|
||||
distinct notions under one symbol.
|
||||
|
||||
**"Context", precisely.** Sharing a central base between two genomes is not
|
||||
just sharing a nucleotide — it is sharing a **context**: the `2m` flanking
|
||||
bases, identical, which is a homology claim about that flanked window
|
||||
(guaranteed non-coincidental by k-specificity, Bias 4 above), *not* a claim
|
||||
about orthology or paralogy of the copy each genome carries. `A` opposite `C`
|
||||
= same context, divergent centre. `A` opposite nothing = **this context is
|
||||
not observed in one of the two genomes** — informative, not neutral.
|
||||
|
||||
**Why the IUPAC/DNA encoding used for the first `--snp` test was wrong.**
|
||||
Feeding IUPAC-coded ambiguity into a standard DNA model (`raxml-ng --model
|
||||
GTR+G`) is a semantic mismatch: Felsenstein-pruning ML treats an ambiguous
|
||||
tip as "exactly one true state, unknown which" (a uniform partial-likelihood
|
||||
vector over compatible bases), not "these states are simultaneously
|
||||
present". The two encodings look identical (same IUPAC letters) but the
|
||||
software reads them backwards from what was intended — this invalidates the
|
||||
literal branch lengths from that first experiment (topology-level groupings
|
||||
by genus were still informative, see the worked example further down).
|
||||
|
||||
**Why `-` (absence) must not be scored as similarity, but also must not be
|
||||
scored as a shared character between two absences.** Two genomes both
|
||||
lacking a context are not observed to resemble each other at that locus —
|
||||
neither is observed to differ from the other either. It is a symmetric
|
||||
non-observation, uninformative for that pair, and should contribute nothing
|
||||
(not a small positive nor a small negative signal) to their distance. A
|
||||
genome carrying a state (`A`) against one carrying none is a different case
|
||||
entirely: informative, and should not be scored as neutral "missing data"
|
||||
the way a generic DNA/ML pipeline would.
|
||||
|
||||
**Detectability vs existence — a deliberate simplification, accepted.**
|
||||
"Context not observed" conflates two different biological events: (1) true
|
||||
loss of the locus, (2) the locus still exists but a mutation/indel *outside*
|
||||
the centre, anywhere in the `2m` flanks, broke k-mer recognition. The design
|
||||
adopts a rigorist stance on purpose: any flank-breaking mutation counts as
|
||||
"this context no longer exists", full stop — because both causes (1) and (2)
|
||||
independently require *at least* one more mutational event than a lone
|
||||
central substitution would. This licenses treating "context absent in one of
|
||||
the two genomes" as a **lower bound** on distance strictly greater than a
|
||||
plain central SNP, without needing to know which of the two causes applies.
|
||||
Coarser than a true event count, and accepted as such (fine substitution-type
|
||||
modelling, e.g. transition/transversion weighting, is a secondary
|
||||
refinement, not required for this to be useful).
|
||||
|
||||
**Resulting ordinal distance between two genomes at one family/context:**
|
||||
|
||||
| Comparison | Distance | Meaning |
|
||||
|---|---|---|
|
||||
| same centre (`A`/`A`) | `0` | identical |
|
||||
| different centre, both single-copy (`A`/`C`) | `1` | plain central SNP |
|
||||
| one genome has a state, the other has none | `>1` (lower bound) | context undetectable in one genome — at least one extra mutational event, of unknown type |
|
||||
| neither genome has any state (`∅`/`∅`) | excluded | symmetric non-observation, not comparable, contributes nothing |
|
||||
|
||||
This is a direct extension of `KmerIndex::raw_snp_distance` (`obikindex/src/siblings.rs`),
|
||||
which today only implements the `0`/`1` rows and silently drops everything
|
||||
else (including the informative `>1` row) rather than scoring it.
|
||||
|
||||
**Open, not yet resolved:**
|
||||
- Calibrating `>1` to a real number for tools expecting continuous distances
|
||||
(NJ/UPGMA, ML branch lengths), rather than an arbitrary placeholder.
|
||||
Natural route: estimate `p_hat` from the resolved (`0`/`1`) sites first,
|
||||
then use the already-derived ascertainment formula (`P(usable window
|
||||
showing a central SNP) = p * (1-p)^(2m)`, Bias 1 above) to derive a
|
||||
model-consistent value for the `>1` bucket instead of guessing a constant.
|
||||
- Where multi-copy/ambiguous states (the IUPAC case: a genome carrying more
|
||||
than one form) fit into this ordinal scheme — plausibly also `>1` by the
|
||||
same "at least one extra event" argument (a second form appearing is a
|
||||
gain, itself an event), but not yet worked out.
|
||||
|
||||
**Practical alternative validated for the pseudo-alignment output itself**
|
||||
(orthogonal to the ordinal-distance question above, useful regardless of how
|
||||
`>1` ends up calibrated): re-encode each family as 4 independent binary
|
||||
presence/absence characters (`A`,`C`,`G`,`T` columns) instead of one IUPAC
|
||||
column, feed to a `BIN`-type model instead of `DNA`. `∅` becomes an explicit
|
||||
`0000` state (identity with another `0000`, not missing data) rather than a
|
||||
gap — removes the semantic mismatch above by construction. Known cost,
|
||||
accepted for now: a plain central substitution (`A` -> `C`) becomes 2 binary
|
||||
flips (`1000` -> `0100`), overweighting substitutions relative to true
|
||||
gain/loss events, and the 4 sub-characters of one family are not
|
||||
statistically independent the way a generic `BIN` model assumes. A proper
|
||||
fix (single 16-state alphabet, i.e. the powerset of `{A,C,G,T}`, with a
|
||||
substitution-rate structure that respects the subset lattice rather than a
|
||||
fully general 16x16 GTR-analogue) is very likely not expressible in
|
||||
`raxml-ng`'s `MULTI` datatype as-is (Mk or fully-general rates only) and a
|
||||
fully general 16-state rate matrix is almost certainly unidentifiable here
|
||||
(states of cardinality >=3 are ~2% of sites in the benchmark run). Treated as
|
||||
a longer-term research question, not a near-term implementation target.
|
||||
|
||||
## Sankoff parsimony as the resolution of the 16-state model problem
|
||||
|
||||
The "longer-term research question" just above (a 16-state alphabet — the
|
||||
powerset of `{A,C,G,T}` — with a substitution structure that respects the
|
||||
subset lattice) turns out to have a near-term answer, once the *unification*
|
||||
question below is worked through.
|
||||
|
||||
**Distance methods (NJ/UPGMA/ME) vs. character methods (parsimony/ML): not a
|
||||
deep philosophical divide, but a real practical distinction for this
|
||||
project.** Historically "phenetic" (characters -> distances -> tree) and
|
||||
"cladistic" (characters -> tree directly) approaches were presented as
|
||||
opposed schools; the modern view is a mathematical continuity, not a
|
||||
dichotomy — Minimum Evolution (ME: find the tree minimizing total branch
|
||||
length from a distance matrix) and Maximum Parsimony (MP: find the tree
|
||||
minimizing total character-state changes) are both instances of "minimise a
|
||||
global explanatory cost", and coincide under simple encodings (see Farris
|
||||
1983, "The Logical Basis of Phylogenetic Analysis"; the MP/ME connection is
|
||||
developed in the Minimum Evolution / Balanced Minimum Evolution literature,
|
||||
e.g. Nei and colleagues — citations not independently re-verified here, flag
|
||||
before quoting further). NJ's own agglomeration step already uses the whole
|
||||
distance matrix jointly (the Q-matrix), not just the pair being merged — an
|
||||
earlier claim in this discussion that distance methods are "blind" to
|
||||
cross-taxon structure at every stage was too strong.
|
||||
|
||||
What *does* remain a real, structural distinction for this project: in a
|
||||
character method, a given character's cost is **re-evaluated per candidate
|
||||
topology** during tree search (the same family can cost 1 change under one
|
||||
topology, 2 under another). In a pairwise-distance pipeline
|
||||
(`raw_snp_distance` as it exists today), each family's contribution to
|
||||
`d(i,j)` is computed **once**, independent of any candidate topology, before
|
||||
NJ/UPGMA ever runs — so a question like "does this shared `∅` look like a
|
||||
synapomorphy under topology T" can never be posed in that pipeline, for any
|
||||
T. That question is only answerable by a method that tests candidate
|
||||
topologies and re-scores characters under each — i.e. a character method.
|
||||
|
||||
**Sankoff parsimony directly resolves the `∅`/gain-loss/substitution
|
||||
question, without the identifiability problem of a fitted 16-state model.**
|
||||
Sankoff's algorithm generalises Fitch parsimony to an arbitrary
|
||||
user-supplied cost matrix between states (`obikseq`/`obikindex` would treat
|
||||
each family as a `2^{4}`-state character, state = subset of `{A,C,G,T}`
|
||||
observed in that genome, `∅` included as a real state, not a gap). The
|
||||
previous 16-state idea failed specifically because *fitting* a full 16x16
|
||||
rate matrix by ML is unidentifiable at this data volume; Sankoff sidesteps
|
||||
that because the cost matrix is **fixed a priori from domain knowledge**, not
|
||||
estimated — e.g. `c({A},{C}) = 1` (a substitution), `c({A},{A,C}) = 1` (a
|
||||
gain), `c({A,C},{A}) = 1` (a loss), `c({A,C},{G,T}) = 2` (two changes) — no
|
||||
estimation, no overparameterisation. This reframes gain/loss and central
|
||||
substitution as two cost categories with independently chosen weights,
|
||||
exactly the "two families of parameters" (`mu_substitution`, `mu_gain/loss`)
|
||||
floated earlier in this discussion, now with an actual algorithmic home.
|
||||
|
||||
**Caveat, not blocking for this project's scope.** Sankoff is still
|
||||
parsimony: in principle exposed to Felsenstein's statistical-inconsistency
|
||||
result under long-branch attraction (already invoked earlier against
|
||||
`D_F = min(a,b)`) — parsimony and ML only provably coincide in the
|
||||
short-branch regime. This is not a practical concern here because it is
|
||||
exactly this estimator's declared target (closely related genomes, short
|
||||
branches) — the regime where parsimony's known failure mode does not apply —
|
||||
but worth stating explicitly as a scope guard rather than leaving it
|
||||
implicit.
|
||||
|
||||
**Cheapest next experiment: don't write a Sankoff tree-search engine, use
|
||||
one that exists.** The hard part of a from-scratch implementation is not the
|
||||
Sankoff DP itself (a straightforward dynamic program over a *fixed* tree) but
|
||||
the topology search (SPR/NNI with incremental re-scoring) that comes for
|
||||
free with `raxml-ng` on the ML side. **TNT** (Tree analysis using New
|
||||
Technology, free, standard in morphological cladistics) already implements
|
||||
Sankoff parsimony with a custom cost matrix plus topology search — the
|
||||
family-state matrix (already close to what `--snp` produces, minus the
|
||||
IUPAC/DNA-model mismatch) could be fed there directly, no new code required,
|
||||
before considering a bespoke engine.
|
||||
|
||||
**The concrete comparison this unlocks:** run both pipelines on the same
|
||||
family data —
|
||||
`k-mer families -> D_ij -> NJ/ME/UPGMA` (phenetic, what exists today) vs.
|
||||
`k-mer families -> characters -> argmin_T Sankoff-cost(T)` (cladistic, via
|
||||
TNT) — and compare the resulting topologies. Agreement would validate that
|
||||
the pairwise-distance projection preserves the phylogenetic signal;
|
||||
disagreement would pinpoint exactly what the projection to a single number
|
||||
per pair loses. Not yet run.
|
||||
|
||||
## Heterozygosity, ploidy, and consensus-assembly inputs
|
||||
|
||||
A within-genome multiplicity signal (more than one of the 4 central forms
|
||||
present at a locus) is produced identically by two distinct causes:
|
||||
paralogous duplication and diploid/polyploid heterozygosity. K-mer data alone
|
||||
cannot distinguish them. The one real discriminator is sequencing depth
|
||||
(heterozygous site: total depth of the present forms ~= the genome's
|
||||
single-copy average; duplication: ~2x or more) — but that signal only exists
|
||||
if genome "counts" are raw-read depth (FASTQ input), not occurrence counts in
|
||||
an assembled FASTA, where per-locus depth is not preserved.
|
||||
|
||||
**Magnitude is taxon- and mating-system-dependent, not universal.**
|
||||
Heterozygosity density: mammals ~1 site / 1-1.5 kb (~0.1%); highly
|
||||
outcrossing plants (maize, poplar) reported an order of magnitude higher
|
||||
(~1%); self-fertilising plants (*Arabidopsis thaliana*) near zero — but with
|
||||
a documented failure mode where segmental duplication masquerades as
|
||||
"pseudo-heterozygosity"; fungi split between haploid vegetative stages
|
||||
(non-issue) and dikaryotic Basidiomycetes, where two long-diverged haploid
|
||||
nuclei coexist without fusing. The estimator's target use case (closely
|
||||
related genomes, k=31) is exactly where the stringent filter above costs the
|
||||
least for low-heterozygosity taxa and the most for outcrossing/dikaryotic
|
||||
ones — no universal threshold; this is a scope caveat to document, not a
|
||||
problem to solve generically.
|
||||
|
||||
**Why assembled-consensus inputs don't make measured distances wrong.**
|
||||
Phylogenetic inputs are near-universally assemblies, not raw reads, and
|
||||
assemblers collapse heterozygous sites to one consensus allele per
|
||||
position — effectively an arbitrary, largely uncorrelated-between-assemblies
|
||||
choice at each het site. This does not inject unbounded noise: standard
|
||||
population genetics gives `d_xy = d_a + (pi_A + pi_B)/2` — the expected
|
||||
pairwise difference between a random allele of population A and a random
|
||||
allele of population B equals the net (fixed) divergence `d_a` plus the
|
||||
average of the two populations' own within-population diversity `pi`.
|
||||
Consensus flattening realises exactly this random-allele draw, so the
|
||||
measured genome-to-genome distance is a `d_xy`-like quantity, not `d_a` —
|
||||
inflated by heterozygosity by a well-characterised additive term, not
|
||||
distorted unpredictably. The term is negligible when `pi << d_xy` (the common
|
||||
case for cross-species comparisons), and becomes material precisely in the
|
||||
two cases already flagged above: very closely related genomes (this
|
||||
estimator's explicit target) and highly heterozygous outcrossing organisms,
|
||||
where `pi` and `d_xy` are the same order of magnitude.
|
||||
|
||||
Caveat: this assumes the flattening is uncorrelated with the phylogenetic
|
||||
signal — plausible for de novo assembly, not guaranteed for reference-guided
|
||||
assembly biased toward one allele (e.g. the reference's) at each het site,
|
||||
which would turn the noise term into a systematic bias toward the reference
|
||||
lineage. Not evaluated here.
|
||||
|
||||
**Forward-looking implication, not part of the current design.** The
|
||||
multiplicity > 1 signal discarded by the stringent filter is a crude
|
||||
per-genome proxy for `pi` (under low background paralogy). If a `pi_hat` per
|
||||
genome were tallied alongside `SnpTally`, a `d_a` correction
|
||||
(`p_hat - mean(pi_hat_i, pi_hat_j)/2`, roughly) could recover an estimate
|
||||
closer to net divergence instead of `d_xy` — a possible extension, not
|
||||
scoped here.
|
||||
|
||||
## Sufficient statistic: 4x4 base-pair tally
|
||||
|
||||
Tabulating the joint distribution of `(center_i, center_j)` over conserved-flank
|
||||
loci, per genome pair, is sufficient for every downstream correction:
|
||||
|
||||
| Estimator | Input | Formula |
|
||||
|---|---|---|
|
||||
| Raw p-distance | total off-diagonal / total | `p = SNP / (SNP + shared)` |
|
||||
| Jukes-Cantor | p | `d = -3/4 * ln(1 - 4p/3)` |
|
||||
| Kimura 2-parameter | transition rate P, transversion rate Q | `d = 1/2 ln(1/(1-2P-Q)) + 1/4 ln(1/(1-2Q))` |
|
||||
| LogDet/paralinear | full 4x4 + base-composition margins | `d ~= -1/4 ln det(F)`, robust to non-stationary base composition |
|
||||
|
||||
JC/K2P need only the total and the transition/transversion split (the
|
||||
diagonal collapses to a single "shared" total). LogDet needs the full 4x4,
|
||||
already populated at no extra cost (see Step 1/2 below).
|
||||
|
||||
Memory for the 4x4 tally: `n^2 * 16` counters. Trivial for the project's
|
||||
genome-scale use case (tens to hundreds of genomes); ~13 GB at n=10^4 — outside
|
||||
scope but worth flagging if n grows.
|
||||
|
||||
## Biases (properties of the estimator, not defects)
|
||||
|
||||
1. **Conserved-flank ascertainment bias.** Only SNPs with intact `2m`-base
|
||||
flanks are visible; window-intact probability decays as `(1-p)^{2m}`. For
|
||||
k=31 (2m=30): 0.74 at p=1%, 0.21 at p=5%, 0.04 at p=10%. This estimator
|
||||
targets **closely related genomes**. Under rate heterogeneity across sites
|
||||
(universal in practice), conserved flanks correlate with slow centers, so
|
||||
`p_hat` underestimates the genome-wide average rate — it specifically
|
||||
estimates the substitution rate of **conserved regions**.
|
||||
Two distinct factors are at play here, not one: `P(centre of a given
|
||||
window is a SNP) = p` exactly, **independent of k** — a direct restatement
|
||||
of the raw per-site rate via the bijective window<->centre-position
|
||||
correspondence (Statistic section above), not a k-dependent quantity.
|
||||
`(1-p)^{2m}` is the *separate*, genuinely k-dependent ascertainment factor
|
||||
(are the flanks also intact). The two multiply:
|
||||
`P(usable window showing a central SNP) = p * (1-p)^{2m}` — e.g. at
|
||||
p=1/31 (~3.2%), k=31: `p * (1-p)^30 ~= 0.0323 * 0.374 ~= 1.2%`, i.e. about
|
||||
1 window in 83, not 1 in 31 (which is only the centre-mutated fraction,
|
||||
before requiring intact flanks).
|
||||
2. **Bias toward isolated SNPs.** Two SNPs within k of each other disqualify
|
||||
each other's flanks. Hypervariable regions are invisible by construction.
|
||||
3. **Indels are invisible.** A frameshift destroys k-mer matches in a block;
|
||||
this channel captures substitutions only. Indel divergence shows up as lost
|
||||
shared k-mers (lower Jaccard/Mash), not as SNP signal.
|
||||
4. **k-dependent specificity.** "A k-mer match implies common ancestry" is
|
||||
quantitative. For a 3 Gbp genome, expected random flank-30 collisions
|
||||
(k=31): `(3e9)^2 / 4^30 ~= 8` — negligible. At k=21: `(3e9)^2 / 4^20 ~= 2e6`
|
||||
— no longer negligible. k=31 is safe; k<=21 is marginal to unreliable for
|
||||
large genomes. The large k that guarantees homology is the same k that
|
||||
shrinks the detectable-divergence window — an inherent tension.
|
||||
|
||||
## Implementation: avoid materializing a de Bruijn graph
|
||||
|
||||
A central-SNP pair is topologically a simple bubble in the colored de Bruijn
|
||||
graph (source/sink k-mer shared, two length-k branches differing only at the
|
||||
midpoint). Classical bubble-calling (Cortex/discoSNP-style) finds these, but
|
||||
requires the graph — nodes plus adjacency for ~10^9 colored k-mers — resident
|
||||
in memory. **Rejected**: prohibitive RAM for this project's scale.
|
||||
|
||||
A naive per-pair generalisation of variant lookup across n genomes (query each
|
||||
non-shared k-mer's 3 central variants against every counterpart genome's
|
||||
index) costs `O(n^2 . N . 3)` random lookups, with the same k-mer's 3 variants
|
||||
regenerated and requeried once per counterpart genome — pure redundant work.
|
||||
**Rejected** as the basis for an n-genome design.
|
||||
|
||||
## Implementation: sequential per-partition sweep (no scratch, no graph)
|
||||
|
||||
`KmerIndex::distance()` already opens every partition's `presence_store`/
|
||||
`count_store` simultaneously, memory-mapped, into one `LayeredStore`
|
||||
(`distance.rs:73-77`). "Querying another partition" is therefore not a new
|
||||
I/O pattern to design — it is the same O(1) MPHF+evidence lookup the `query`
|
||||
command already performs at scale. This lets the SNP tally be computed with
|
||||
**no scratch files and no auxiliary graph**, by sweeping partitions once each
|
||||
as a source:
|
||||
|
||||
1. For source partition `p`, enumerate its **distinct** k-mers (one per MPHF
|
||||
slot; each already carries its full multi-genome presence/count vector —
|
||||
no need to explode per (k-mer, genome) occurrence).
|
||||
2. For each, generate the 3 central-substitution variants and **canonicalise
|
||||
each independently** (`min(kmer, revcomp)`, exactly as any normal query) —
|
||||
this avoids the orientation edge case a masked-flank grouping would have
|
||||
(a substitution that flips canonical orientation is handled correctly
|
||||
because each variant is canonicalised on its own, not inferred from a
|
||||
fixed-orientation flank key).
|
||||
3. Compute each variant's target partition `q` via its minimizer; batch/sort
|
||||
the partition's outgoing variant queries by `q` for locality.
|
||||
4. Look up each variant in `q`'s already-mmap'd MPHF+evidence; on a hit,
|
||||
combine the source's presence vector (base `a`) with the variant's
|
||||
presence vector (base `b`): for every `i` carrying `a` and `j` carrying
|
||||
`b`, `tally[i,j][a,b] += 1`.
|
||||
|
||||
**Deduplication needs no persisted state.** Sweeping partitions in a fixed
|
||||
order `p = 0, 1, ..., P-1` and only acting on a variant when its target
|
||||
partition `q >= p` guarantees each unordered SNP pair is counted exactly
|
||||
once: a pair with `q < p` was already resolved earlier, when `q` was itself
|
||||
the source partition and `p` (being `>= q`) was a valid forward target. No
|
||||
cross-partition flag array is needed — the sweep order *is* the
|
||||
deduplication rule. Within the same partition (`q == p`), a lightweight
|
||||
transient tie-break suffices: either a `#slots(p)`-bit scratch flag reset per
|
||||
partition, or simply comparing the two k-mers' raw `u64` encodings and only
|
||||
counting when `kmer_source < kmer_variant` — no storage at all.
|
||||
|
||||
This is a distinct computation stage, not a `partial_*` in the existing
|
||||
additive-by-partition sense: step 3-4 read across partition boundaries by
|
||||
construction, unlike the row-local `partial_jaccard`/`partial_threshold_jaccard`
|
||||
primitives. But it requires no new index files, permanent or scratch:
|
||||
`unitigs.bin`, `mphf.bin`, `evidence.bin`, and the presence/count columns are
|
||||
read as-is, and the only extra memory is the current partition's small
|
||||
outgoing-query batch (`#kmers(p) * 3`, released once `p` is done) plus the
|
||||
persistent `tally` accumulator (`n^2 * 16` counters, see above).
|
||||
|
||||
**Outer loop (over source partitions `p`) must stay sequential.** Two
|
||||
independent reasons, not just one: (a) memory — the bounded-footprint claim
|
||||
above only holds with one partition's outgoing-query batch in flight; running
|
||||
`T` source partitions concurrently multiplies that batch by `T`, exactly the
|
||||
blowup the design avoids; (b) correctness — the `q >= p` deduplication rule
|
||||
requires partitions to be claimed as sources in a fixed order; running `p1 <
|
||||
p2` concurrently gives no guarantee `p1` has finished claiming its `q >= p1`
|
||||
targets before `p2` starts claiming its own, breaking the "counted exactly
|
||||
once" property.
|
||||
|
||||
**Inner loop (target-partition lookups for a fixed `p`) parallelises safely.**
|
||||
Each lookup is an O(1) read against an already-mmap'd structure, independent
|
||||
of the others, with no growing allocation — no memory blowup, no ordering
|
||||
dependency between different `q`. The only shared mutable state is `tally`;
|
||||
give each worker thread a **thread-local partial tally** (fixed `n^2 * 16`
|
||||
size, independent of partition size) and merge into the global `tally` once
|
||||
`p`'s inner loop completes — the same reduce-then-merge pattern Rayon already
|
||||
uses elsewhere in this codebase to open partitions in parallel. Extra memory:
|
||||
`#threads * n^2 * 16`, negligible (~512 MB at n~1000, 32 threads) and
|
||||
unrelated to partition size.
|
||||
|
||||
**Cost**: `3 * N_distinct` MPHF lookups total across the whole index (each
|
||||
partition swept once as source) — the same order of magnitude and the same
|
||||
operation as running `query` over the index's entire k-mer content against
|
||||
itself, three times. This is the tool's already-optimized regime, not a new
|
||||
I/O profile to validate.
|
||||
|
||||
## Cheaper: subsampling
|
||||
|
||||
Since the target is a ratio, restricting the source-partition sweep to a
|
||||
bottom-`s` hash sketch (only enumerate k-mers with `hash < threshold` as
|
||||
sources) divides the lookup count by the sampling factor without biasing
|
||||
`p_hat`. Mash-like tradeoff: rate estimated from a sample, not the full
|
||||
k-mer set.
|
||||
|
||||
## Recommendation
|
||||
|
||||
Sequential per-partition sweep (Route D): reuse the already-mmap'd
|
||||
per-partition MPHF/evidence/presence structures for O(1) variant lookups,
|
||||
dedup via the fixed sweep-order rule (`q >= p`, plus an in-partition
|
||||
tie-break), no scratch files, no graph materialisation. Both the SNP
|
||||
(off-diagonal) and shared (diagonal) counts are accumulated by this same
|
||||
sweep, under the locus-eligibility rule chosen (raw or paralogy-filtered) —
|
||||
not reused from the general-purpose `shared_kmers` matrix, whose raw-identity
|
||||
definition does not apply the same copy-number constraint. Distances (p, JC,
|
||||
K2P, LogDet) as finalisations of the resulting 4x4 tally, mirroring the
|
||||
`partial_* -> *_dist_matrix` pattern used for Jaccard/Mash/Bray-Curtis/etc.
|
||||
|
||||
## Detailed implementation plan
|
||||
|
||||
Grounded in the current codebase. File/type references are anchors, not
|
||||
prescriptions; adjust to reality when implementing.
|
||||
|
||||
### Step 0 — new low-level primitives (`obikseq`)
|
||||
|
||||
Two helpers do not yet exist and are prerequisites:
|
||||
|
||||
1. **Central neighbours.** `CanonicalKmerOf<L>` already exposes
|
||||
`left_canonical_neighbors()` / `right_canonical_neighbors()`
|
||||
(`obikseq/src/kmer.rs`), each returning the 4 canonicalised neighbours at
|
||||
an end position. Add `central_canonical_neighbors()` returning the 4
|
||||
variants at position `m = (k-1)/2` (each independently canonicalised via
|
||||
`.canonical()`). The 3 that differ from the source are the query variants;
|
||||
skip the identity. Building on `nucleotide(i)` / the raw 2-bit layout keeps
|
||||
it O(1).
|
||||
2. **Lone-k-mer minimiser.** Routing a *synthetic* variant to its partition
|
||||
needs its minimiser, but `RollingStat` (`obiskbuilder/src/rolling_stat.rs`)
|
||||
only computes minimisers incrementally along a sequence. Add a standalone
|
||||
`minimizer(kmer) -> Minimizer` that scans the `k-m+1` m-mer windows
|
||||
(`PackedSeq::mmer`, `obikseq/src/packed_seq.rs`), canonicalises each, and
|
||||
takes the min by `seq_hash()` — the same selection `RollingStat` performs,
|
||||
evaluated once. Partition index is then
|
||||
`(minimizer.seq_hash() & (n_partitions - 1)) as usize`, exactly as
|
||||
`QueryBatch::from_records` (`obikmer/src/cmd/query.rs:142`); `n_partitions`
|
||||
is a power of two so the mask is valid.
|
||||
|
||||
### Step 1 — the tally accumulator (`obikindex`)
|
||||
|
||||
A `SnpTally` holding, per genome pair, the 4x4 joint count of central bases:
|
||||
`n * n * 4 * 4` `u64` (or a packed lower-triangular form since it is
|
||||
symmetric). Provide `merge(&mut self, other: &SnpTally)` for the thread-local
|
||||
reduce, and accessors yielding, per pair `(i,j)`: total off-diagonal (SNP),
|
||||
diagonal (shared, i.e. `p_hat`'s denominator minus SNP), transition count
|
||||
`P`, transversion count `Q`. The diagonal is always populated — it is not an
|
||||
optional LogDet-only extra, since `p_hat`'s denominator is no longer sourced
|
||||
from the external `shared_kmers` matrix (see "Locus eligibility" and
|
||||
"Statistic and correspondence with `shared`" above): the source k-mer's own
|
||||
presence/count vector, already in hand when it is enumerated, supplies the
|
||||
diagonal entry directly, at no extra lookup cost.
|
||||
|
||||
### Step 2 — the sweep (`obikindex`, new `snp.rs`)
|
||||
|
||||
Mirror `distance.rs`: open the presence or count store per partition. But
|
||||
instead of a per-partition `partial_*`, run the sequential source sweep:
|
||||
|
||||
```text
|
||||
for p in 0..n_partitions: # OUTER — sequential
|
||||
open source partition p's layers (QueryLayer-style, obikpartitionner)
|
||||
enumerate distinct canonical k-mers of p (one per MPHF slot) with their
|
||||
presence/count vectors # column-major, as query stage 2
|
||||
par_iter over these source k-mers: # INNER — rayon, thread-local tally
|
||||
apply eligibility rule to the source's own vector (raw: none; # diagonal
|
||||
stringent: exactly one of the 4 forms present in each genome) # gate
|
||||
for i in genomes eligible with source base a:
|
||||
for j in genomes eligible with source base a:
|
||||
thread_tally[i,j][a,a] += 1 # diagonal — no extra lookup
|
||||
for each of the 3 central variants:
|
||||
q = partition_of(variant)
|
||||
if q < p: continue # dedup: forward targets only
|
||||
if q == p and variant <= source.raw(): continue # in-partition tie-break
|
||||
slot = layers[q].find_slot(variant) # MphfLayer::find, mmap'd
|
||||
if hit:
|
||||
vb = variant presence/count vector
|
||||
apply eligibility rule to vb (as above)
|
||||
for i in eligible genomes with source base a:
|
||||
for j in eligible genomes with variant base b:
|
||||
thread_tally[i,j][a,b] += 1
|
||||
merge thread-local tallies into global SnpTally
|
||||
```
|
||||
|
||||
The inner lookup is precisely `QueryLayer::find_slot` +
|
||||
`col_value(g, slot)` (`obikpartitionner/src/query_layer.rs`) — reuse or factor
|
||||
out that path rather than reimplementing MPHF access. Enumerating "all distinct
|
||||
k-mers of a partition with their vectors" is the `dump`/`query` stage-2
|
||||
column-major scan already implemented in `dump_layer.rs` /
|
||||
`query_partition_with`; factor a reusable iterator if none fits.
|
||||
|
||||
`presence_threshold` applies exactly as elsewhere: a genome "carries base b"
|
||||
iff its count at that slot is `>= presence_threshold` (trivially `>= 1` for
|
||||
presence indexes).
|
||||
|
||||
### Open problem (unresolved, session end — not yet fully convinced)
|
||||
|
||||
The `q >= p` / tie-break dedup rule in Step 2's pseudocode above is **flawed**
|
||||
for the stringent (paralogy-filtered) eligibility rule: it only ever brings
|
||||
two family members into view at once (the source and one looked-up variant),
|
||||
never all four simultaneously, and which subset gets compared depends on
|
||||
partition sweep order. "Exactly one of the 4 forms present in genome A" is a
|
||||
whole-family property and cannot be decided correctly from a sequence of
|
||||
pairwise, order-dependent glimpses — the pseudocode above needs revision, not
|
||||
just the eligibility gate bolted onto it as written.
|
||||
|
||||
Direction discussed, **not yet settled**:
|
||||
|
||||
1. **Every distinct source k-mer looks up all 3 variants unconditionally**
|
||||
(drop the `q < p` skip entirely) so that every observed family member
|
||||
independently gathers all 4 vectors (its own + whichever of the 3
|
||||
variants exist) at once — a whole-family, order-independent view, computed
|
||||
redundantly once per observed member. Same total lookup order of
|
||||
magnitude as already budgeted (`3 * N_distinct`), just organised
|
||||
differently (no lookup actually skipped, versus the original rule which
|
||||
skipped roughly half).
|
||||
2. **Tie-break after gathering, not before**: only the member whose own
|
||||
canonical encoding is the smallest *among the members actually observed*
|
||||
(now known, since all were just looked up) writes to `SnpTally`; the
|
||||
others silently discard their redundant computation. Deterministic,
|
||||
order-independent — as a side effect this also removes the "outer loop
|
||||
must stay sequential" constraint from the cost/parallelism discussion
|
||||
above, since no step depends on partition processing order any more.
|
||||
3. **Proposed optimisation**: precompute, once at index build time, a
|
||||
compact global (not per-genome) annex per MPHF slot — the count of
|
||||
*other* family members observed anywhere in the dataset (0-3). Slots with
|
||||
count 0 (majority under low divergence and few genomes, but see the
|
||||
scaling caveat below) need no cross-lookup at all: eligibility reduces to
|
||||
a local `count == 1` check at that single slot, and only slots with count
|
||||
>= 1 enter the 3-lookup sweep machinery above. Revised (see Step 2b
|
||||
below): minorant status *is* stored alongside the count after all, on 3
|
||||
bits rather than 2 — it comes for free from the same lookups needed to
|
||||
count siblings, and storing it lets the sweep discard non-minorant slots
|
||||
without re-fetching anything.
|
||||
|
||||
**Minorant/sibling-count relationship, worked out precisely.** "Minorant" is
|
||||
a one-way implication from sibling count, not an equivalence: `0 siblings
|
||||
=> minorant` (trivially — with no other observed member, the k-mer is by
|
||||
definition the smallest of the observed set, itself alone), and its
|
||||
contrapositive `not minorant => >= 1 sibling`. The converse does not hold:
|
||||
being the minorant says nothing about sibling count — a minorant can have 0,
|
||||
1, 2 or 3 siblings, all with larger encodings than itself. Consequence: this
|
||||
confirms, as a logical necessity rather than a heuristic, that a 0-sibling
|
||||
slot can always write its diagonal contribution with zero ambiguity and no
|
||||
lookup (it is unconditionally its own minorant) — but it gives no shortcut
|
||||
for the >= 1-sibling case, where minorant status still requires the actual
|
||||
comparison of gathered encodings; sibling count alone never determines it.
|
||||
|
||||
**When to compute the annex, and cache invalidation.** Sibling count is a
|
||||
property of the whole set of columns (genomes/groups) currently in the
|
||||
index, not of any single genome — it cannot be computed correctly at
|
||||
mono-genome build time (a family may gain siblings, or its minorant may
|
||||
change, once more genomes are merged in later). Computing it eagerly at
|
||||
every `merge` would also waste work on intermediate merged states nobody
|
||||
ever queries. Instead: compute it lazily, on first `distance` call against a
|
||||
given index, and persist the result alongside that index for subsequent
|
||||
calls — the same lazy-derived-cache pattern `PersistentBitMatrix` already
|
||||
uses for `Columnar` -> `Packed`. This requires no explicit invalidation for
|
||||
`merge` or `filter` (`obikindex/src/merge.rs`, `obikmer/src/cmd/filter.rs`):
|
||||
both only ever write to a fresh `--output` directory, never mutate an input
|
||||
index in place, so a re-merged/re-filtered index is simply a new state with
|
||||
no annex yet. `select --in-place` (`select_layer.rs:139-235`) is the
|
||||
exception: it aggregates genome columns into groups (Any/All/None/Sum/Min/
|
||||
Max) by mutating the existing index's files without changing its location.
|
||||
It does not remove k-mer rows, but it can still change eligibility and
|
||||
sibling counts derived from those rows (e.g. a `Sum` over several
|
||||
single-copy genomes can read as multi-copy at the group level). Because it
|
||||
mutates in place, **`select --in-place` must explicitly invalidate (delete
|
||||
or mark stale) any cached sibling-count annex for that index** — the one
|
||||
operation in the current pipeline where this doesn't happen for free.
|
||||
|
||||
Not yet convinced this is the right shape, and Step 2's pseudocode above has
|
||||
not been rewritten to match — flagged for the next pass rather than resolved
|
||||
here.
|
||||
|
||||
### Step 2b — sibling-count / minorant annex (consolidated plan)
|
||||
|
||||
Scope: only the precursor annex — not the SNP tally itself, whose Step 2
|
||||
sweep remains unresolved above. This piece is simpler than the sweep,
|
||||
because it writes to an independent per-slot value, not a shared
|
||||
cross-k-mer accumulator, so it needs no dedup/ownership logic at all at this
|
||||
stage.
|
||||
|
||||
**Revised annex encoding — 4-bit presence mask, not 3-bit (minorant +
|
||||
count).** Superseded after settling the "canonical form of a family"
|
||||
definition above. The 3-bit design (1 minorant bit + 2-bit sibling count,
|
||||
§ below, kept for the historical record) had two problems: it discards
|
||||
*which* variants are present (only how many), so any future consumer
|
||||
(the SNP sweep, or a stats pass — see below) that needs to know which bases
|
||||
exist still has to regenerate and blindly re-query all 3 candidates; and
|
||||
the minorant bit's meaning was tied to whichever member was visited, not to
|
||||
a fixed reference. Storing instead a **4-bit mask** — one bit per base
|
||||
(A/C/G/T), set iff that member of the family (labelled relative to the
|
||||
family's fixed canonical form, i.e. the member with `A` at the centre — see
|
||||
above) is observed anywhere in the index — fixes both:
|
||||
- **Sibling count is derived, not stored**: `siblings = popcount(mask) - 1`.
|
||||
- **Minorant is derived, not stored**: regenerate the family's 4 canonical
|
||||
forms from the slot's own k-mer (cheap, no lookup — see above), compare
|
||||
the raw encodings of whichever bits are set in the mask, take the
|
||||
smallest.
|
||||
- **A future consumer knows exactly which variants to (re-)query** —
|
||||
`popcount(mask) - 1` lookups instead of always 3, and it knows *which*
|
||||
3 (or fewer) to issue, not just how many hits to expect.
|
||||
- The all-zero value (no base present at all) is still logically
|
||||
unreachable as a real result — the slot's *own* base is always present in
|
||||
its own family — so it remains available as a free "not yet computed"
|
||||
sentinel, exactly as before.
|
||||
|
||||
1. **Primitive.** Reuse `central_canonical_neighbors()` from Step 0
|
||||
unchanged — the 3 canonicalised central-substitution variants of a k-mer
|
||||
(plus the identity, i.e. all 4 members of the family — see "Definitions"
|
||||
above).
|
||||
2. **New annex type** (`obicompactvec`, alongside `bitmatrix.rs`): a 4-bit-
|
||||
per-slot packed array (the presence mask above), one per partition — same
|
||||
on-disk shape family as `PersistentBitMatrix`'s `Packed` variant, but
|
||||
simpler (no per-genome columns, a single derived read-only value per
|
||||
slot).
|
||||
<details><summary>Superseded 3-bit design (historical)</summary>
|
||||
3 bits, storing minorant status alongside sibling count directly, since
|
||||
it came for free from the same lookups (point 3 below) — 5 real states
|
||||
(not-minorant; minorant with 0/1/2/3 siblings) fit in 3 bits (8 states,
|
||||
3 unused). This let the future SNP sweep discard a non-minorant slot
|
||||
instantly, with no lookup at all. The otherwise-unreachable combination
|
||||
"not-minorant + 0 siblings" (0 siblings always implies minorant) doubled
|
||||
as the "not yet computed" sentinel. Replaced by the 4-bit mask above,
|
||||
which subsumes this benefit (minorant still derivable, now for free at
|
||||
read time rather than stored) while also fixing the "which variant"
|
||||
blindness.
|
||||
</details>
|
||||
3. **Computation pass** (`obikindex`, new `siblings.rs`): **one
|
||||
`obipipeline` run per layer, iterated sequentially over the index's
|
||||
layers** — settled after two false starts, worth recording both.
|
||||
- *False start 1*: "fully parallel over every partition/slot at once,
|
||||
no ordering at all". Correctness is fine with this (sibling count and
|
||||
minorant are order-independent, unlike the old `q >= p` dedup they
|
||||
replace), but it reintroduces, at a larger scale, exactly the
|
||||
memory-blowup the original Step 2 sweep's sequential-outer-loop
|
||||
constraint existed to prevent: scattering every source partition at
|
||||
once multiplies the in-flight outgoing-query volume by the number of
|
||||
partitions.
|
||||
- *False start 2*: push the layer loop itself into the pipeline (source
|
||||
= the index's layers, a first `Flat` stage expands each layer into
|
||||
its k-mers). `obipipeline`'s scheduler already bounds memory on its
|
||||
own — it dispatches every item through a **shared** worker pool at
|
||||
each stage boundary (`scheduler.rs:217-372`, `dispatch()` into a
|
||||
common `worker_tx` queue, any free worker picks up any pending item;
|
||||
not "one worker owns a chunk end to end"), with a biased `Select`
|
||||
that prioritises draining items already advanced in the chain over
|
||||
admitting new source items (`scheduler.rs:271-282`: stage results
|
||||
outrank the source, "vider le pipeline en priorité" / "dernier
|
||||
recours" for new data) — so bounded channel `capacity` plus this
|
||||
drain-first bias already caps in-flight work without any external
|
||||
sequential discipline. Correct, but it means k-mers from several
|
||||
layers can be completing concurrently, so the sink would need to
|
||||
track several open per-layer annex-file writers at once — real,
|
||||
avoidable complexity.
|
||||
- **Settled design**: keep the layer loop external and sequential —
|
||||
not for memory (the pipeline's own `capacity`/priority mechanism
|
||||
already provides that, for free, regardless), but so each pipeline
|
||||
run's sink targets exactly one layer's annex file, no concurrent
|
||||
multi-writer bookkeeping. Per layer: source = that layer's distinct
|
||||
k-mers; a `Flat` (1->N) stage generates the 3 central variants of a
|
||||
k-mer, each tagged with its origin (local slot); a transform stage
|
||||
routes each variant to its target partition (unchanged per-k-mer
|
||||
minimiser); a transform stage performs the lookup (existence-only —
|
||||
`find_slot` hit/miss, cheaper than the SNP sweep's full column
|
||||
fetch); a final stage/sink folds each answer into its origin's
|
||||
running state (below) and, once a layer's k-mers are all resolved,
|
||||
flushes the completed array to that layer's annex file. Many small,
|
||||
single-purpose stages on purpose, to let the scheduler interleave
|
||||
them finely across many in-flight items — this deliberately does
|
||||
**not** mirror how `obipipeline` is used elsewhere today: `query.rs`'s
|
||||
`process_chunk` lumps parse+route+query+serialise into one closure
|
||||
(`query.rs:325,743-758`), and `scatter.rs` only pipelines file-
|
||||
reading/superkmer construction, routing partitions afterwards in a
|
||||
plain sequential loop (`KmerPartition::write_batch`,
|
||||
`partition.rs:140`) — both under-use the fine-grained scheduling the
|
||||
mechanism offers, so they are not precedents to copy, only existing
|
||||
(and arguably improvable, out of scope here) usages. Cross-partition
|
||||
lookups (querying another layer's MPHF for a variant) remain
|
||||
necessary as before — only the *output* side is kept single-layer.
|
||||
- **Reconciliation**: processed at the granularity of one *answer batch
|
||||
per destination partition*, not one source k-mer at a time — this is a
|
||||
proper shuffle, not a per-k-mer wait. Each source partition `p` holds a
|
||||
small array of running states `(minorant = true, siblings = 0)`, one
|
||||
per local slot, initialised at scatter time and **persisting across
|
||||
however many destination-partition batches answer it** (up to 3, one
|
||||
per variant, not necessarily all from the same `q`). Every scattered
|
||||
query carries an origin tag (source partition + local slot) so its
|
||||
answer can be routed back. When target partition `q` returns its batch
|
||||
(all answers for every query that named `q`, regardless of which source
|
||||
k-mer or which source partition they came from), that batch is walked
|
||||
once, locally, and each answer updates — via its origin tag — the
|
||||
matching entry in *its* source partition's array: a miss changes
|
||||
nothing; a hit does `siblings += 1`, and if the found sibling's own
|
||||
encoding is smaller than the source's, `minorant = false`. A given
|
||||
source k-mer's state is final only once every destination batch
|
||||
concerning it has been folded in; its partition's array is flushed to
|
||||
the persistent annex once complete. Commutative per entry, so the order
|
||||
in which destination batches arrive and get folded in doesn't matter.
|
||||
|
||||
**Open optimisation, not adopted yet — real tradeoff, not a free win.**
|
||||
Since looking up sibling `y` from `x`'s visit already yields everything
|
||||
needed to fill `y`'s own annex entry too, one visit per *family* could in
|
||||
principle replace one visit per *observed family member* — cutting this
|
||||
pass's cost roughly by the average family size instead of paying
|
||||
`3 * N_distinct` regardless. But it means threads processing different
|
||||
source k-mers can end up writing the *same* sibling's slot concurrently —
|
||||
the fully independent, ownership-free parallelism of the plan above is
|
||||
deliberately traded away for this gain. It stays safe only because the
|
||||
computed value for a given slot is deterministic regardless of who
|
||||
computes it, so redundant concurrent writes converge to the same
|
||||
value — correct as long as each write is atomic, no locking needed — but
|
||||
it is a real design complexity increase over "every member redoes its
|
||||
own 3 lookups independently," not a strict improvement to adopt by
|
||||
default.
|
||||
4. **Trigger and caching** (`obikindex::KmerIndex`/`distance.rs`): compute
|
||||
lazily on first `distance` call for an SNP-family metric against a given
|
||||
index; check for an existing annex file first (mirrors
|
||||
`PersistentBitMatrix::open()`'s auto-detect-and-fall-back,
|
||||
`bitmatrix.rs:264-287`); if absent, run step 3 and persist; if present,
|
||||
mmap and reuse.
|
||||
5. **Invalidation.** `merge` and `filter` always write to a fresh `--output`
|
||||
directory (`obikindex/src/merge.rs`, `obikmer/src/cmd/filter.rs`) so a
|
||||
re-merged/re-filtered index simply has no annex yet — nothing to
|
||||
invalidate. `select --in-place` (`select_layer.rs:139-235`) mutates
|
||||
columns of an existing index without changing its location, which can
|
||||
change sibling counts without removing rows — it must explicitly delete
|
||||
any cached annex for that index as part of its in-place rewrite.
|
||||
6. **Testing**: hand-built tiny indexes with known sibling counts (0-3);
|
||||
order-independence (recompute twice on a static index, identical
|
||||
result, given the fully-parallel no-ownership design); invalidation
|
||||
(annex absent/correctly recomputed after `select --in-place`); once
|
||||
Step 2's sweep is fixed, a regression check that sibling_count == 0
|
||||
slots are never looked up cross-partition during the sweep.
|
||||
|
||||
Cost: `3 * N_distinct` existence-only lookups, computed once per index
|
||||
state and amortised over every subsequent `distance` call that reuses the
|
||||
cached annex — cheaper per-lookup than the sweep itself (hit/miss only, no
|
||||
column fetch).
|
||||
|
||||
### Step 3 — finalisation (`obikindex`)
|
||||
|
||||
From the global `SnpTally` alone (diagonal and off-diagonal both populated by
|
||||
the sweep, see Step 1/2 — no dependency on the external `shared_kmers`
|
||||
matrix), derive n x n distance matrices, each a pure function of the
|
||||
accumulated counts (same shape as `jaccard_to_mash`):
|
||||
|
||||
- `p_hat[i,j] = SNP / (SNP + shared)`
|
||||
- Jukes-Cantor, Kimura-2P (from `P`, `Q`), optionally LogDet (needs the
|
||||
diagonal + base-composition margins).
|
||||
|
||||
Guard the singularities (`p >= 3/4` for JC, `1-2P-Q <= 0` or `1-2Q <= 0` for
|
||||
K2P) by clamping to a max distance, as `jaccard_to_mash` clamps `J <= 0`.
|
||||
|
||||
### Step 4 — surfacing (`obikindex` + `obikmer` CLI)
|
||||
|
||||
These metrics do not fit `DistanceMetric`'s current `LayeredStore`-partial
|
||||
dispatch (they need the cross-partition sweep and produce a different
|
||||
intermediate). Two options, to decide:
|
||||
|
||||
- **(a)** New `DistanceMetric` variants (`Pdistance`, `JukesCantor`,
|
||||
`Kimura2P`, `LogDet`) whose `KmerIndex::distance` arm calls the sweep
|
||||
(`snp.rs`) instead of the partial path, still returning `DistanceOutput`.
|
||||
Keeps one CLI surface (`--metric jukes-cantor`), at the cost of a branch in
|
||||
`distance()` that ignores the `LayeredStore` it built.
|
||||
- **(b)** A dedicated pathway (`KmerIndex::snp_distance`) and a distinct CLI
|
||||
entry, if mixing a cross-partition sweep into the partition-local `distance`
|
||||
command is judged architecturally muddy.
|
||||
|
||||
Recommendation: (a) for user ergonomics (all pairwise distances under
|
||||
`distance`, all feeding NJ/UPGMA/`--shared-kmers` unchanged), but compute the
|
||||
sweep lazily only when an SNP-family metric is requested, so the existing
|
||||
metrics keep their partition-local fast path untouched.
|
||||
|
||||
### Step 5 — subsampling flag
|
||||
|
||||
Add `--snp-sample <fraction>` (or a bottom-`s` hash threshold): restrict the
|
||||
source-k-mer enumeration in Step 2 to `seq_hash(kmer) < threshold`. Divides
|
||||
lookups proportionally; `p_hat` is unbiased. Off by default (exact).
|
||||
|
||||
### Testing
|
||||
|
||||
- **Primitive unit tests**: `central_canonical_neighbors` on hand-checked
|
||||
k-mers incl. palindrome-boundary cases; lone-k-mer `minimizer` against
|
||||
`RollingStat`'s incremental result on the same k-mer.
|
||||
- **End-to-end tiny index**: two 1-genome indexes differing by a handful of
|
||||
known isolated SNPs (transitions and transversions placed by hand), assert
|
||||
exact `SNP`, `P`, `Q` counts and the resulting JC/K2P values.
|
||||
- **Dedup invariant**: assert the tally is identical regardless of genome/
|
||||
partition order and that no pair is double-counted (compare against a
|
||||
brute-force all-pairs reference on a small index).
|
||||
- **Subsampling**: `p_hat` within sampling error of the exact run.
|
||||
|
||||
### Suggested phasing
|
||||
|
||||
1. Step 0 primitives + their unit tests (self-contained, no distance wiring).
|
||||
This also unblocks the long-declared-but-unimplemented `query --mismatch`
|
||||
(`obikmer/src/cmd/query.rs:676`, currently a warning), which needs the same
|
||||
neighbour + routing machinery.
|
||||
2. `SnpTally` + finalisation math with a brute-force (non-swept) reference
|
||||
backend, validated on a tiny index.
|
||||
3. The real per-partition sweep (Step 2) behind the same finalisation; assert
|
||||
it matches the brute-force backend.
|
||||
4. CLI surfacing (Step 4a) and NJ/UPGMA integration (already generic over the
|
||||
matrix).
|
||||
5. Subsampling (Step 5).
|
||||
|
||||
## References
|
||||
|
||||
The Mash mutation-rate model this discussion contrasts with:
|
||||
[@Mash-distances-doc; @Fan2015-mash-formula].
|
||||
@@ -29,12 +29,14 @@ extra_javascript:
|
||||
|
||||
nav:
|
||||
- Home: index.md
|
||||
- Installation: installation.md
|
||||
- Theory:
|
||||
- Kmers and super-kmers: kmers.md
|
||||
- DNA encoding: theory/encoding.md
|
||||
- Entropy filter: theory/entropy.md
|
||||
- Minimizer selection: theory/minimizer.md
|
||||
- Partitioning architecture: theory/indexing.md
|
||||
- Central-position SNP distance (discussion): theory/evolutionary_distances.md
|
||||
- Implementation:
|
||||
- SuperKmer: implementation/superkmer.md
|
||||
- Kmer: implementation/kmer.md
|
||||
@@ -52,9 +54,12 @@ nav:
|
||||
- Merge parallelism & memory: implementation/merge_parallelism.md
|
||||
- Kmer filtering: implementation/filtering.md
|
||||
- Select command: implementation/select.md
|
||||
- obitaxonomy crate: implementation/obitaxonomy.md
|
||||
- Architecture:
|
||||
- Sequences: architecture/sequences/invariant.md
|
||||
- Kmer index: architecture/index_architecture.md
|
||||
- NUMA-aware worker pools: architecture/numa_worker_pools.md
|
||||
- NUMA-aware partition runner: architecture/numa_partition_runner.md
|
||||
|
||||
watch:
|
||||
- docmd
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
# La crate obicompactvector
|
||||
|
||||
Le code actuelle est ce qu'il est. Ce n'est pad la vrérité absolue, c'est un premier effort d'implémentation rien de plus. Ci-dessous je vais décrire les objectif et la structure qui devrait être. LA VERITE A ATTEINDRE.
|
||||
|
||||
La crate fournie des représentations les plus compact possible en mémoire de matrice de comptage ou de présence de k-mer dans des génomes. Chaque colonne représente un génome chaque ligne un kmer. une matrice est une collection de vecteur ou chacun des vecteur est un colonne de la matrice.
|
||||
|
||||
Les matrices comme les colonnes ont vocation à être persistante. Les données sont stockées dans des fichiers binaires. Les données sont mappées en mémoire via `mmap`
|
||||
|
||||
Les structure sont par essence immutables. Il existe des représentations mutables des colonnes qui permettent leur construction. À la fin de leur construction, les colonnes sont fermée ce qui les rends immutable.
|
||||
|
||||
Les matrices peuvent êtres représenté de deux façons:
|
||||
- via un répertoire contenant une collection de fichier colonnes
|
||||
- via un fichier matrix qui est la concatenation de plusieurs fichiers colonnes.
|
||||
|
||||
|
||||
## Les matrices de comptage
|
||||
|
||||
Ce sont des matrice d'entiers positif la plus part du temps de petites valeurs (inferieurs à 255). On assume que toutes les valeurs sont représentables sur un `u32`
|
||||
|
||||
## Les matrices de presence
|
||||
|
||||
Ce sont des matrices de boolean représenté comme des champs de bits
|
||||
|
||||
Il existe une forme implicite des vecteur de présence, qui n'est représenté par aucun fichier pour lequel toutes les valeurs sont vraies
|
||||
|
||||
## représentation légère des colonnes
|
||||
|
||||
Les colonnes qu'elles soient de unitiaire (fichier colonne) ou partie d'un fichier composite matrice peuvent être représenté par un objet léger donnant acces à ces valeurs ainsi qu'à la longeur du vecteurs. Toutes les méthodes de calcules doivent uniquement travailler à partir de ces représentations légère unifiées des colonnes.
|
||||
|
||||
### Représentation légère d'un vecteur de présence
|
||||
|
||||
Le vecteur est représenté par
|
||||
- un champs de bits encodé comme un [u64]
|
||||
- un usize encodant la longeur du champs de bits
|
||||
|
||||
### Représentation légère d'un vecteur de présence
|
||||
|
||||
Le vecteur est représenté par
|
||||
- un vecteur [u8] encodant directement les valeur faibe du vecteur [0,255[
|
||||
La valeur 255 est une valeur sentinelle indiquant que la valeure vraie est >=255
|
||||
et se trouvent dans une structure d'overflow
|
||||
- un iterateur de (usize,u32) listant les valeurs d'overflow coorespondant aux valeurs
|
||||
sentinels (255) du [u8]
|
||||
- un usize encodant la longeur du champs de bits
|
||||
Generated
+319
-6
@@ -128,6 +128,12 @@ version = "0.4.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2ad8689a486416c401ea15715a4694de30054248ec627edbf31f49cb64ee4086"
|
||||
|
||||
[[package]]
|
||||
name = "arrayvec"
|
||||
version = "0.7.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7c02d123df017efcdfbd739ef81735b36c5ba83ec3c59c80a9d7ecc718f92e50"
|
||||
|
||||
[[package]]
|
||||
name = "as-slice"
|
||||
version = "0.2.1"
|
||||
@@ -143,6 +149,15 @@ version = "1.5.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c08606f8c3cbf4ce6ec8e28fb0014a2c086708fe954eaa885384a6165172e7e8"
|
||||
|
||||
[[package]]
|
||||
name = "autotools"
|
||||
version = "0.2.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ef941527c41b0fc0dd48511a8154cd5fc7e29200a0ff8b7203c5d777dbc795cf"
|
||||
dependencies = [
|
||||
"cc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "backtrace"
|
||||
version = "0.3.76"
|
||||
@@ -224,6 +239,15 @@ dependencies = [
|
||||
"generic-array",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "block-buffer"
|
||||
version = "0.12.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d2f6c7dbe95a6ed67ad9f18e57daf93a2f034c524b99fd2b76d18fdfeb6660aa"
|
||||
dependencies = [
|
||||
"hybrid-array",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "block-pseudorand"
|
||||
version = "0.1.2"
|
||||
@@ -415,6 +439,15 @@ version = "1.1.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c8d4a3bb8b1e0c1050499d1815f5ab16d04f0959b233085fb31653fbfc9d98f9"
|
||||
|
||||
[[package]]
|
||||
name = "cmake"
|
||||
version = "0.1.58"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c0f78a02292a74a88ac736019ab962ece0bc380e3f977bf72e376c5d78ff0678"
|
||||
dependencies = [
|
||||
"cc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "colorchoice"
|
||||
version = "1.0.5"
|
||||
@@ -464,6 +497,21 @@ dependencies = [
|
||||
"windows-sys 0.59.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "const-oid"
|
||||
version = "0.10.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a6ef517f0926dd24a1582492c791b6a4818a4d94e789a334894aa15b0d12f55c"
|
||||
|
||||
[[package]]
|
||||
name = "convert_case"
|
||||
version = "0.10.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "633458d4ef8c78b72454de2d54fd6ab2e60f9e02be22f3c6104cdc8a4e0fceb9"
|
||||
dependencies = [
|
||||
"unicode-segmentation",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "core-foundation-sys"
|
||||
version = "0.8.7"
|
||||
@@ -488,6 +536,15 @@ dependencies = [
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cpufeatures"
|
||||
version = "0.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8b2a41393f66f16b0823bb79094d54ac5fbd34ab292ddafb9a0456ac9f87d201"
|
||||
dependencies = [
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crc32fast"
|
||||
version = "1.5.0"
|
||||
@@ -601,6 +658,15 @@ dependencies = [
|
||||
"typenum",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crypto-common"
|
||||
version = "0.2.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ce6e4c961d6cd6c9a86db418387425e8bdeaf05b3c8bc1411e6dca4c252f1453"
|
||||
dependencies = [
|
||||
"hybrid-array",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "csv"
|
||||
version = "1.4.0"
|
||||
@@ -640,14 +706,48 @@ dependencies = [
|
||||
"uuid",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "derive_more"
|
||||
version = "2.1.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d751e9e49156b02b44f9c1815bcb94b984cdcc4396ecc32521c739452808b134"
|
||||
dependencies = [
|
||||
"derive_more-impl",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "derive_more-impl"
|
||||
version = "2.1.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "799a97264921d8623a957f6c3b9011f3b5492f557bbb7a5a19b7fa6d06ba8dcb"
|
||||
dependencies = [
|
||||
"convert_case",
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"rustc_version",
|
||||
"syn",
|
||||
"unicode-xid",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "digest"
|
||||
version = "0.10.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
|
||||
dependencies = [
|
||||
"block-buffer",
|
||||
"crypto-common",
|
||||
"block-buffer 0.10.4",
|
||||
"crypto-common 0.1.7",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "digest"
|
||||
version = "0.11.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2"
|
||||
dependencies = [
|
||||
"block-buffer 0.12.1",
|
||||
"const-oid",
|
||||
"crypto-common 0.2.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -742,6 +842,16 @@ version = "2.4.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6"
|
||||
|
||||
[[package]]
|
||||
name = "filetime"
|
||||
version = "0.2.29"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5c287a33c7f0a620c38e641e7f60827713987b3c0f26e8ddc9462cc69cf75759"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "find-msvc-tools"
|
||||
version = "0.1.9"
|
||||
@@ -916,6 +1026,65 @@ version = "0.5.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c"
|
||||
|
||||
[[package]]
|
||||
name = "http"
|
||||
version = "1.4.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6970f50e31d6fc17d3fa27329444bfa74e196cf62e95052a3f6fee181dba6425"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"itoa",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "httparse"
|
||||
version = "1.10.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87"
|
||||
|
||||
[[package]]
|
||||
name = "hwlocality"
|
||||
version = "1.0.0-alpha.12"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4c2e65a48d3b300843ac84a2fe8e166bb5a5b00f30054593bcee8157e4b465fd"
|
||||
dependencies = [
|
||||
"arrayvec",
|
||||
"bitflags 2.11.1",
|
||||
"derive_more",
|
||||
"errno",
|
||||
"hwlocality-sys",
|
||||
"libc",
|
||||
"strum",
|
||||
"thiserror 2.0.18",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "hwlocality-sys"
|
||||
version = "0.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "10a83c43a772c1f774b806deb44891c2a9578eb33cec48aad513482e0da3d4d4"
|
||||
dependencies = [
|
||||
"autotools",
|
||||
"cmake",
|
||||
"flate2",
|
||||
"libc",
|
||||
"pkg-config",
|
||||
"sha3",
|
||||
"tar",
|
||||
"ureq 3.3.0",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "hybrid-array"
|
||||
version = "0.4.12"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9155a582abd142abc056962c29e3ce5ff2ad5469f4246b537ed42c5deba857da"
|
||||
dependencies = [
|
||||
"typenum",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "icu_collections"
|
||||
version = "2.2.0"
|
||||
@@ -1145,6 +1314,16 @@ dependencies = [
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "keccak"
|
||||
version = "0.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9e24a010dd405bd7ed803e5253182815b41bf2e6a80cc3bfc066658e03a198aa"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"cpufeatures 0.3.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "kodama"
|
||||
version = "0.2.3"
|
||||
@@ -1503,28 +1682,40 @@ dependencies = [
|
||||
"xxhash-rust",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "obikentropy"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"obikseq",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "obikindex"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"crossbeam-channel",
|
||||
"hwlocality",
|
||||
"indicatif",
|
||||
"ndarray",
|
||||
"obicompactvec",
|
||||
"obikpartitionner",
|
||||
"obikseq",
|
||||
"obilayeredmap",
|
||||
"obipipeline",
|
||||
"obiread",
|
||||
"obiskbuilder",
|
||||
"obiskio",
|
||||
"obisys",
|
||||
"rayon",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"tempfile",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "obikmer"
|
||||
version = "0.1.0"
|
||||
version = "1.1.42"
|
||||
dependencies = [
|
||||
"clap",
|
||||
"csv",
|
||||
@@ -1542,6 +1733,7 @@ dependencies = [
|
||||
"obiskbuilder",
|
||||
"obiskio",
|
||||
"obisys",
|
||||
"obitaxonomy",
|
||||
"pprof",
|
||||
"rayon",
|
||||
"serde_json",
|
||||
@@ -1561,6 +1753,7 @@ dependencies = [
|
||||
"niffler 3.0.0",
|
||||
"obicompactvec",
|
||||
"obidebruinj",
|
||||
"obikentropy",
|
||||
"obikrope",
|
||||
"obikseq",
|
||||
"obilayeredmap",
|
||||
@@ -1636,14 +1829,16 @@ dependencies = [
|
||||
"regex",
|
||||
"tracing",
|
||||
"tracing-subscriber",
|
||||
"ureq",
|
||||
"ureq 2.12.1",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "obiskbuilder"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"criterion2",
|
||||
"lazy_static",
|
||||
"obikentropy",
|
||||
"obikrope",
|
||||
"obikseq",
|
||||
"obiread",
|
||||
@@ -1673,6 +1868,10 @@ dependencies = [
|
||||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "obitaxonomy"
|
||||
version = "0.1.0"
|
||||
|
||||
[[package]]
|
||||
name = "object"
|
||||
version = "0.37.3"
|
||||
@@ -2177,6 +2376,15 @@ version = "2.1.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "94300abf3f1ae2e2b8ffb7b58043de3d399c73fa6f4b73826402a5c457614dbe"
|
||||
|
||||
[[package]]
|
||||
name = "rustc_version"
|
||||
version = "0.4.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cfcb3a22ef46e85b45de6ee7e79d063319ebb6594faafcf1c225ea92ab6e9b92"
|
||||
dependencies = [
|
||||
"semver",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustix"
|
||||
version = "1.1.4"
|
||||
@@ -2263,6 +2471,12 @@ dependencies = [
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "semver"
|
||||
version = "1.0.28"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd"
|
||||
|
||||
[[package]]
|
||||
name = "serde"
|
||||
version = "1.0.228"
|
||||
@@ -2313,8 +2527,18 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"cpufeatures",
|
||||
"digest",
|
||||
"cpufeatures 0.2.17",
|
||||
"digest 0.10.7",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "sha3"
|
||||
version = "0.11.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "be176f1a57ce4e3d31c1a166222d9768de5954f811601fb7ca06fc8203905ce1"
|
||||
dependencies = [
|
||||
"digest 0.11.3",
|
||||
"keccak",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2375,6 +2599,27 @@ version = "0.11.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f"
|
||||
|
||||
[[package]]
|
||||
name = "strum"
|
||||
version = "0.28.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9628de9b8791db39ceda2b119bbe13134770b56c138ec1d3af810d045c04f9bd"
|
||||
dependencies = [
|
||||
"strum_macros",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "strum_macros"
|
||||
version = "0.28.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ab85eea0270ee17587ed4156089e10b9e6880ee688791d45a905f5b1ca36f664"
|
||||
dependencies = [
|
||||
"heck",
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "subtle"
|
||||
version = "2.6.1"
|
||||
@@ -2470,6 +2715,17 @@ version = "1.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "55937e1799185b12863d447f42597ed69d9928686b8d88a1df17376a097d8369"
|
||||
|
||||
[[package]]
|
||||
name = "tar"
|
||||
version = "0.4.46"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3f6221d9a6003c78398e3b239969f352578258df48c8eb051caadae0015bc840"
|
||||
dependencies = [
|
||||
"filetime",
|
||||
"libc",
|
||||
"xattr",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tempfile"
|
||||
version = "3.27.0"
|
||||
@@ -2645,12 +2901,24 @@ version = "1.0.24"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-segmentation"
|
||||
version = "1.13.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c6f5d3c3b1bf09027a88a6bc961fc00497d651009560b5463668dc81b0fa87a8"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-width"
|
||||
version = "0.2.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b4ac048d71ede7ee76d585517add45da530660ef4390e49b098733c6e897f254"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-xid"
|
||||
version = "0.2.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ebc1c04c71510c7f702b52b7c350734c9ff1295c464a03335b00bb84fc54f853"
|
||||
|
||||
[[package]]
|
||||
name = "untrusted"
|
||||
version = "0.9.0"
|
||||
@@ -2673,6 +2941,35 @@ dependencies = [
|
||||
"webpki-roots 0.26.11",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ureq"
|
||||
version = "3.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dea7109cdcd5864d4eeb1b58a1648dc9bf520360d7af16ec26d0a9354bafcfc0"
|
||||
dependencies = [
|
||||
"base64",
|
||||
"flate2",
|
||||
"log",
|
||||
"percent-encoding",
|
||||
"rustls",
|
||||
"rustls-pki-types",
|
||||
"ureq-proto",
|
||||
"utf8-zero",
|
||||
"webpki-roots 1.0.7",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ureq-proto"
|
||||
version = "0.6.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e994ba84b0bd1b1b0cf92878b7ef898a5c1760108fe7b6010327e274917a808c"
|
||||
dependencies = [
|
||||
"base64",
|
||||
"http",
|
||||
"httparse",
|
||||
"log",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "url"
|
||||
version = "2.5.8"
|
||||
@@ -2685,6 +2982,12 @@ dependencies = [
|
||||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "utf8-zero"
|
||||
version = "0.8.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b8c0a043c9540bae7c578c88f91dda8bd82e59ae27c21baca69c8b191aaf5a6e"
|
||||
|
||||
[[package]]
|
||||
name = "utf8_iter"
|
||||
version = "1.0.4"
|
||||
@@ -3110,6 +3413,16 @@ dependencies = [
|
||||
"tap",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "xattr"
|
||||
version = "1.6.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32e45ad4206f6d2479085147f02bc2ef834ac85886624a23575ae137c8aa8156"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"rustix",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "xxhash-rust"
|
||||
version = "0.8.15"
|
||||
|
||||
+1
-1
@@ -1,5 +1,5 @@
|
||||
[workspace]
|
||||
resolver = "3"
|
||||
members = ["obikseq", "obiread", "obiskbuilder", "obifastwrite", "obikmer","obikrope","obipipeline", "obikpartitionner","obiskio","obidebruinj","obilayeredmap", "obicompactvec", "obisys", "obikindex"]
|
||||
members = ["obikseq", "obiread", "obiskbuilder", "obifastwrite", "obikmer","obikrope","obipipeline", "obikpartitionner","obiskio","obidebruinj","obilayeredmap", "obicompactvec", "obisys", "obikindex", "obitaxonomy", "obikentropy"]
|
||||
[profile.release]
|
||||
debug = 1
|
||||
|
||||
Binary file not shown.
@@ -7,6 +7,6 @@ edition = "2024"
|
||||
memmap2 = "0.9"
|
||||
ndarray = "0.16"
|
||||
rayon = "1"
|
||||
tempfile = "3"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
|
||||
+243
-106
@@ -1,5 +1,5 @@
|
||||
use std::fs::{self, File};
|
||||
use std::io::{self, Write as _};
|
||||
use std::io::{self, BufWriter, Read as _, Write as _};
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use memmap2::Mmap;
|
||||
@@ -7,8 +7,12 @@ use ndarray::{Array1, Array2};
|
||||
use rayon::prelude::*;
|
||||
|
||||
use crate::bitvec::{PersistentBitVec, PersistentBitVecBuilder};
|
||||
use crate::colgroup::{ColGroup, MatrixGroupOps};
|
||||
use crate::layer_meta::LayerMeta;
|
||||
use crate::meta::MatrixMeta;
|
||||
use crate::tempbitvec::{TempBitVec, TempBitVecBuilder};
|
||||
use crate::tempintvec::{TempCompactIntVec, TempCompactIntVecBuilder};
|
||||
use crate::views::BitSliceView;
|
||||
|
||||
fn col_path(dir: &Path, col: usize) -> PathBuf {
|
||||
dir.join(format!("col_{col:06}.pbiv"))
|
||||
@@ -54,34 +58,11 @@ impl ColumnarBitMatrix {
|
||||
}
|
||||
|
||||
pub(crate) fn partial_jaccard_dist_matrix(&self) -> (Array2<u64>, Array2<u64>) {
|
||||
let n = self.n_cols();
|
||||
let results: Vec<(usize, usize, u64, u64)> = upper_pairs(n)
|
||||
.into_par_iter()
|
||||
.map(|(i, j)| {
|
||||
let (inter, union) = self.col(i).partial_jaccard_dist(self.col(j));
|
||||
(i, j, inter, union)
|
||||
})
|
||||
.collect();
|
||||
let mut inter_m = Array2::zeros((n, n));
|
||||
let mut union_m = Array2::zeros((n, n));
|
||||
for (i, j, inter, union) in results {
|
||||
inter_m[[i, j]] = inter; inter_m[[j, i]] = inter;
|
||||
union_m[[i, j]] = union; union_m[[j, i]] = union;
|
||||
}
|
||||
(inter_m, union_m)
|
||||
pairwise2_matrix(self.n_cols(), |i, j| self.col(i).partial_jaccard_dist(self.col(j)))
|
||||
}
|
||||
|
||||
pub(crate) fn partial_hamming_dist_matrix(&self) -> Array2<u64> {
|
||||
self.pairwise_u64(|i, j| self.col(i).hamming_dist(self.col(j)))
|
||||
}
|
||||
|
||||
fn pairwise_u64(&self, f: impl Fn(usize, usize) -> u64 + Sync) -> Array2<u64> {
|
||||
let n = self.n_cols();
|
||||
let results: Vec<(usize, usize, u64)> = upper_pairs(n)
|
||||
.into_par_iter()
|
||||
.map(|(i, j)| (i, j, f(i, j)))
|
||||
.collect();
|
||||
fill_symmetric(n, results.into_iter().map(|(i, j, v)| (i, j, v, v)))
|
||||
pairwise_matrix(self.n_cols(), |i, j| self.col(i).hamming_dist(self.col(j)))
|
||||
}
|
||||
|
||||
pub(crate) fn append_column(dir: &Path, value_of: impl Fn(usize) -> bool) -> io::Result<()> {
|
||||
@@ -147,113 +128,116 @@ impl PackedBitMatrix {
|
||||
}).collect()
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn col_bytes(&self, c: usize) -> &[u8] {
|
||||
let start = self.data_offsets[c];
|
||||
let len = (self.n_rows + 7) / 8;
|
||||
&self.mmap[start..start + len]
|
||||
&self.mmap[start..start + self.n_rows.div_ceil(8)]
|
||||
}
|
||||
|
||||
fn count_ones_col(&self, c: usize) -> u64 {
|
||||
let bytes = self.col_bytes(c);
|
||||
let full = self.n_rows / 8;
|
||||
let rem = self.n_rows % 8;
|
||||
let mut n: u64 = bytes[..full].iter().map(|b| b.count_ones() as u64).sum();
|
||||
if rem > 0 { n += (bytes[full] & ((1u8 << rem) - 1)).count_ones() as u64; }
|
||||
n
|
||||
fn col_words(&self, c: usize) -> &[u64] {
|
||||
let nw = self.n_rows.div_ceil(64);
|
||||
// SAFETY: data_offsets[c] is always 8-byte aligned.
|
||||
// PBMX header = 24 + n_cols×8 (multiple of 8); each PBIV blob =
|
||||
// 16 + nwords×8 (multiple of 8); mmap base is page-aligned.
|
||||
let ptr = self.mmap[self.data_offsets[c]..].as_ptr() as *const u64;
|
||||
unsafe { std::slice::from_raw_parts(ptr, nw) }
|
||||
}
|
||||
|
||||
fn pair_op(&self, i: usize, j: usize, and_or: bool) -> u64 {
|
||||
let ai = self.col_bytes(i);
|
||||
let aj = self.col_bytes(j);
|
||||
let full = self.n_rows / 8;
|
||||
let rem = self.n_rows % 8;
|
||||
let mut n: u64 = ai[..full].iter().zip(aj[..full].iter())
|
||||
.map(|(a, b)| if and_or { a & b } else { a ^ b }.count_ones() as u64)
|
||||
.sum();
|
||||
if rem > 0 {
|
||||
let mask = (1u8 << rem) - 1;
|
||||
let last = if and_or { ai[full] & aj[full] } else { ai[full] ^ aj[full] };
|
||||
n += (last & mask).count_ones() as u64;
|
||||
}
|
||||
n
|
||||
pub(crate) fn col_slice(&self, c: usize) -> BitSliceView<'_> {
|
||||
BitSliceView::new(self.col_words(c), self.n_rows)
|
||||
}
|
||||
|
||||
fn partial_jaccard_col(&self, i: usize, j: usize) -> (u64, u64) {
|
||||
let ai = self.col_bytes(i);
|
||||
let aj = self.col_bytes(j);
|
||||
let full = self.n_rows / 8;
|
||||
let rem = self.n_rows % 8;
|
||||
let (mut inter, mut union) = ai[..full].iter().zip(aj[..full].iter())
|
||||
.fold((0u64, 0u64), |(inter, union), (a, b)| {
|
||||
(inter + (a & b).count_ones() as u64,
|
||||
union + (a | b).count_ones() as u64)
|
||||
});
|
||||
if rem > 0 {
|
||||
let mask = (1u8 << rem) - 1;
|
||||
inter += ((ai[full] & aj[full]) & mask).count_ones() as u64;
|
||||
union += ((ai[full] | aj[full]) & mask).count_ones() as u64;
|
||||
}
|
||||
(inter, union)
|
||||
pub(crate) fn col_persist(&self, c: usize, path: &Path) -> io::Result<PersistentBitVecBuilder> {
|
||||
PersistentBitVecBuilder::from_raw_bytes(self.col_bytes(c), self.n_rows, path)
|
||||
}
|
||||
|
||||
pub(crate) fn count_ones(&self) -> Array1<u64> {
|
||||
Array1::from_vec(
|
||||
(0..self.n_cols).into_par_iter().map(|c| self.count_ones_col(c)).collect()
|
||||
(0..self.n_cols).into_par_iter()
|
||||
.map(|c| self.col_slice(c).count_ones())
|
||||
.collect()
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn partial_jaccard_dist_matrix(&self) -> (Array2<u64>, Array2<u64>) {
|
||||
let n = self.n_cols;
|
||||
let results: Vec<(usize, usize, u64, u64)> = upper_pairs(n)
|
||||
.into_par_iter()
|
||||
.map(|(i, j)| { let (inter, union) = self.partial_jaccard_col(i, j); (i, j, inter, union) })
|
||||
.collect();
|
||||
let mut inter_m = Array2::zeros((n, n));
|
||||
let mut union_m = Array2::zeros((n, n));
|
||||
for (i, j, inter, union) in results {
|
||||
inter_m[[i, j]] = inter; inter_m[[j, i]] = inter;
|
||||
union_m[[i, j]] = union; union_m[[j, i]] = union;
|
||||
}
|
||||
(inter_m, union_m)
|
||||
pairwise2_matrix(self.n_cols, |i, j| {
|
||||
self.col_slice(i).partial_jaccard_dist(self.col_slice(j))
|
||||
})
|
||||
}
|
||||
|
||||
pub(crate) fn partial_hamming_dist_matrix(&self) -> Array2<u64> {
|
||||
let n = self.n_cols;
|
||||
let results: Vec<(usize, usize, u64)> = upper_pairs(n)
|
||||
.into_par_iter()
|
||||
.map(|(i, j)| (i, j, self.pair_op(i, j, false)))
|
||||
.collect();
|
||||
fill_symmetric(n, results.into_iter().map(|(i, j, v)| (i, j, v, v)))
|
||||
pairwise_matrix(self.n_cols, |i, j| {
|
||||
self.col_slice(i).hamming_dist(self.col_slice(j))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Reads just the `n_cols` field from an existing packed matrix's header,
|
||||
/// without mapping the file. Used by `pack_bit_matrix` to tell a genuinely
|
||||
/// complete pack from a stale one that predates a later column-widening.
|
||||
fn packed_bit_matrix_n_cols(path: &Path) -> io::Result<usize> {
|
||||
let mut f = File::open(path)?;
|
||||
let mut header = [0u8; PBMX_HEADER];
|
||||
f.read_exact(&mut header)?;
|
||||
Ok(u64::from_le_bytes(header[16..24].try_into().unwrap()) as usize)
|
||||
}
|
||||
|
||||
/// Build `presence/matrix.pbmx` from existing `col_*.pbiv` files.
|
||||
pub fn pack_bit_matrix(dir: &Path) -> io::Result<()> {
|
||||
let meta = MatrixMeta::load(dir)?;
|
||||
let n_cols = meta.n_cols;
|
||||
let packed_path = dir.join("matrix.pbmx");
|
||||
|
||||
let col_files: Vec<Vec<u8>> = (0..n_cols)
|
||||
.map(|c| fs::read(col_path(dir, c)))
|
||||
.collect::<io::Result<_>>()?;
|
||||
let meta = match MatrixMeta::load(dir) {
|
||||
Ok(meta) => meta,
|
||||
Err(e) => {
|
||||
// No columnar data pending: either this layer was already
|
||||
// packed and cleaned up (matrix.pbmx complete, nothing left to
|
||||
// do), or genuinely nothing was ever written here.
|
||||
return if packed_path.exists() { Ok(()) } else { Err(e) };
|
||||
}
|
||||
};
|
||||
|
||||
let header_size = PBMX_HEADER + n_cols * 8;
|
||||
let mut col_offset = header_size;
|
||||
let mut offsets = Vec::with_capacity(n_cols);
|
||||
for data in &col_files {
|
||||
offsets.push(col_offset as u64);
|
||||
col_offset += data.len();
|
||||
// A `matrix.pbmx` can already exist here even though columnar data is
|
||||
// still pending — e.g. copied verbatim from a merge's base source
|
||||
// before this layer was widened with more genome columns (see
|
||||
// `obikpartitionner::merge_partition`). Only skip (re-)packing if the
|
||||
// existing file already reflects the current column count; otherwise
|
||||
// the columnar files are newer and must be (re-)packed, overwriting the
|
||||
// stale one — never silently discarded as "leftover cleanup".
|
||||
if packed_bit_matrix_n_cols(&packed_path).ok() == Some(meta.n_cols) {
|
||||
for c in 0..meta.n_cols { let _ = fs::remove_file(col_path(dir, c)); }
|
||||
let _ = fs::remove_file(dir.join("meta.json"));
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let packed_path = dir.join("matrix.pbmx");
|
||||
let mut file = File::create(&packed_path)?;
|
||||
file.write_all(&PBMX_MAGIC)?;
|
||||
file.write_all(&[0u8; 4])?;
|
||||
file.write_all(&(meta.n as u64).to_le_bytes())?;
|
||||
file.write_all(&(n_cols as u64).to_le_bytes())?;
|
||||
for &off in &offsets { file.write_all(&off.to_le_bytes())?; }
|
||||
for data in &col_files { file.write_all(data)?; }
|
||||
drop(file);
|
||||
let n_cols = meta.n_cols;
|
||||
|
||||
// Compute offsets from file sizes — no column data loaded into RAM.
|
||||
let col_sizes: Vec<u64> = (0..n_cols)
|
||||
.map(|c| fs::metadata(col_path(dir, c)).map(|m| m.len()))
|
||||
.collect::<io::Result<_>>()?;
|
||||
|
||||
let header_size = (PBMX_HEADER + n_cols * 8) as u64;
|
||||
let mut col_offset = header_size;
|
||||
let mut offsets = Vec::with_capacity(n_cols);
|
||||
for &size in &col_sizes {
|
||||
offsets.push(col_offset);
|
||||
col_offset += size;
|
||||
}
|
||||
|
||||
// Write to a temp file; rename atomically so a killed process never leaves
|
||||
// a truncated matrix.pbmx that would be mistaken for a complete file.
|
||||
let tmp_path = dir.join("matrix.pbmx.tmp");
|
||||
let mut out = BufWriter::new(File::create(&tmp_path)?);
|
||||
out.write_all(&PBMX_MAGIC)?;
|
||||
out.write_all(&[0u8; 4])?;
|
||||
out.write_all(&(meta.n as u64).to_le_bytes())?;
|
||||
out.write_all(&(n_cols as u64).to_le_bytes())?;
|
||||
for &off in &offsets { out.write_all(&off.to_le_bytes())?; }
|
||||
for c in 0..n_cols {
|
||||
io::copy(&mut File::open(col_path(dir, c))?, &mut out)?;
|
||||
}
|
||||
out.flush()?;
|
||||
drop(out);
|
||||
fs::rename(&tmp_path, &packed_path)?;
|
||||
|
||||
for c in 0..n_cols { fs::remove_file(col_path(dir, c))?; }
|
||||
fs::remove_file(dir.join("meta.json"))?;
|
||||
@@ -326,6 +310,37 @@ impl PersistentBitMatrix {
|
||||
}
|
||||
}
|
||||
|
||||
pub fn col_view(&self, c: usize) -> BitSliceView<'_> {
|
||||
match self {
|
||||
Self::Columnar(m) => m.col(c).view(),
|
||||
Self::Packed(m) => m.col_slice(c),
|
||||
Self::Implicit { .. } => panic!("col_view() not available on Implicit PersistentBitMatrix"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Column-major point lookup: value at column `c`, slot `slot`, as 0/1.
|
||||
///
|
||||
/// Unlike [`col_view`](Self::col_view), this never panics on `Implicit`
|
||||
/// (every column reads as present, per the mono-genome fast path) — safe
|
||||
/// to call for any `c < self.n_cols()`.
|
||||
pub fn get(&self, c: usize, slot: usize) -> u32 {
|
||||
match self {
|
||||
Self::Columnar(m) => m.col(c).get(slot) as u32,
|
||||
Self::Packed(m) => m.col_slice(c).get(slot) as u32,
|
||||
Self::Implicit { .. } => 1,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn col_persist(&self, c: usize, path: &Path) -> io::Result<PersistentBitVecBuilder> {
|
||||
match self {
|
||||
Self::Columnar(m) => PersistentBitVecBuilder::build_from(m.col(c), path),
|
||||
Self::Packed(m) => m.col_persist(c, path),
|
||||
Self::Implicit { n_rows, .. } => {
|
||||
PersistentBitVecBuilder::new_ones(*n_rows, path)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn row(&self, slot: usize) -> Box<[bool]> {
|
||||
match self {
|
||||
Self::Columnar(m) => m.row(slot),
|
||||
@@ -422,12 +437,93 @@ impl PersistentBitMatrixBuilder {
|
||||
PersistentBitVecBuilder::new(self.n, &path)
|
||||
}
|
||||
|
||||
pub fn add_col_ones(&mut self) -> io::Result<PersistentBitVecBuilder> {
|
||||
let path = col_path(&self.dir, self.n_cols);
|
||||
self.n_cols += 1;
|
||||
PersistentBitVecBuilder::new_ones(self.n, &path)
|
||||
}
|
||||
|
||||
pub fn add_col_from(&mut self, src: &TempBitVec) -> io::Result<()> {
|
||||
src.make_persistent(&col_path(&self.dir, self.n_cols))?;
|
||||
self.n_cols += 1;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn add_col_from_int(&mut self, src: &TempCompactIntVec) -> io::Result<()> {
|
||||
let path = col_path(&self.dir, self.n_cols);
|
||||
self.n_cols += 1;
|
||||
let mut b = PersistentBitVecBuilder::new(self.n, &path)?;
|
||||
b.or_where(src.view(), |v| v > 0);
|
||||
b.close()
|
||||
}
|
||||
|
||||
pub fn close(self) -> io::Result<()> {
|
||||
MatrixMeta { n: self.n, n_cols: self.n_cols }.save(&self.dir)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Helpers ───────────────────────────────────────────────────────────────────
|
||||
// ── MatrixGroupOps ────────────────────────────────────────────────────────────
|
||||
|
||||
impl MatrixGroupOps for PersistentBitMatrix {
|
||||
fn partial_group_presence_count(&self, g: &ColGroup, _threshold: u32) -> io::Result<TempCompactIntVec> {
|
||||
// Bit matrices store 0/1 — threshold is structurally always 1.
|
||||
let n = self.n();
|
||||
if g.indices.len() < 255 {
|
||||
let mut builder = TempCompactIntVecBuilder::new(n)?;
|
||||
for &c in &g.indices {
|
||||
builder.inc_present_fast(self.col_view(c));
|
||||
}
|
||||
builder.freeze()
|
||||
} else {
|
||||
let mut result = TempCompactIntVecBuilder::new(n)?;
|
||||
for chunk in g.indices.chunks(254) {
|
||||
let mut chunk_b = TempCompactIntVecBuilder::new(n)?;
|
||||
for &c in chunk {
|
||||
chunk_b.inc_present_fast(self.col_view(c));
|
||||
}
|
||||
let frozen = chunk_b.freeze()?;
|
||||
result.add(frozen.view());
|
||||
}
|
||||
result.freeze()
|
||||
}
|
||||
}
|
||||
|
||||
fn partial_group_sum(&self, g: &ColGroup) -> io::Result<TempCompactIntVec> {
|
||||
// For bit matrices, sum = count of 1-bits — identical to presence_count.
|
||||
self.partial_group_presence_count(g, 1)
|
||||
}
|
||||
|
||||
fn partial_group_any(&self, g: &ColGroup, _threshold: u32) -> io::Result<TempBitVec> {
|
||||
let n = self.n();
|
||||
let mut result = TempBitVecBuilder::new(n)?;
|
||||
for &c in &g.indices {
|
||||
result.or(self.col_view(c));
|
||||
}
|
||||
result.freeze()
|
||||
}
|
||||
|
||||
fn partial_group_min(&self, g: &ColGroup) -> io::Result<TempCompactIntVec> {
|
||||
// min of 0/1 values = AND: 1 only if ALL columns are 1
|
||||
let n = self.n();
|
||||
let mut result = TempCompactIntVecBuilder::new(n)?;
|
||||
if let Some((&first, rest)) = g.indices.split_first() {
|
||||
result.inc_present_fast(self.col_view(first));
|
||||
for &c in rest { result.mask_with(self.col_view(c)); }
|
||||
}
|
||||
result.freeze()
|
||||
}
|
||||
|
||||
fn partial_group_max(&self, g: &ColGroup) -> io::Result<TempCompactIntVec> {
|
||||
// max of 0/1 values = OR: 1 if any column is 1
|
||||
let any = self.partial_group_any(g, 1)?;
|
||||
let n = any.len();
|
||||
let mut result = TempCompactIntVecBuilder::new(n)?;
|
||||
result.inc_present(any.view());
|
||||
result.freeze()
|
||||
}
|
||||
}
|
||||
|
||||
// ── Shared matrix helpers (also used by intmatrix.rs) ─────────────────────────
|
||||
|
||||
fn upper_pairs(n: usize) -> Vec<(usize, usize)> {
|
||||
(0..n).flat_map(|i| (i + 1..n).map(move |j| (i, j))).collect()
|
||||
@@ -439,3 +535,44 @@ where T: Clone + Default {
|
||||
for (i, j, vij, vji) in vals { m[[i, j]] = vij; m[[j, i]] = vji; }
|
||||
m
|
||||
}
|
||||
|
||||
/// Compute a symmetric `n×n` matrix in parallel by evaluating `f(i,j)` for
|
||||
/// all upper-triangle pairs, plus `f(i,i)` for the diagonal. `T: Copy` avoids
|
||||
/// the `.clone()` needed for the lower-triangle mirror.
|
||||
///
|
||||
/// The diagonal is *not* generally `T::default()`: for a self-comparison,
|
||||
/// `f(i,i)` is often the column's own weight (e.g. intersection-with-self —
|
||||
/// see `pairwise2_matrix`), not zero. Distance finalisations that need a
|
||||
/// zero diagonal (self-distance) already overwrite it explicitly.
|
||||
pub(crate) fn pairwise_matrix<T>(n: usize, f: impl Fn(usize, usize) -> T + Sync) -> Array2<T>
|
||||
where T: Copy + Default + Send {
|
||||
let results: Vec<(usize, usize, T)> = upper_pairs(n)
|
||||
.into_par_iter().map(|(i, j)| (i, j, f(i, j))).collect();
|
||||
let mut m = fill_symmetric(n, results.into_iter().map(|(i, j, v)| (i, j, v, v)));
|
||||
for i in 0..n { m[[i, i]] = f(i, i); }
|
||||
m
|
||||
}
|
||||
|
||||
/// Same as `pairwise_matrix` but `f` returns two values that fill two
|
||||
/// symmetric matrices simultaneously (e.g. intersection + union for Jaccard).
|
||||
/// The diagonal is `f(i,i)` (e.g. a genome's kmer count intersected with
|
||||
/// itself), not `T::default()` — see `pairwise_matrix` for why that matters.
|
||||
pub(crate) fn pairwise2_matrix<T>(n: usize, f: impl Fn(usize, usize) -> (T, T) + Sync) -> (Array2<T>, Array2<T>)
|
||||
where T: Copy + Default + Send {
|
||||
let results: Vec<(usize, usize, T, T)> = upper_pairs(n)
|
||||
.into_par_iter()
|
||||
.map(|(i, j)| { let (a, b) = f(i, j); (i, j, a, b) })
|
||||
.collect();
|
||||
let mut m0 = Array2::from_elem((n, n), T::default());
|
||||
let mut m1 = Array2::from_elem((n, n), T::default());
|
||||
for (i, j, a, b) in results {
|
||||
m0[[i, j]] = a; m0[[j, i]] = a;
|
||||
m1[[i, j]] = b; m1[[j, i]] = b;
|
||||
}
|
||||
for i in 0..n {
|
||||
let (a, b) = f(i, i);
|
||||
m0[[i, i]] = a;
|
||||
m1[[i, i]] = b;
|
||||
}
|
||||
(m0, m1)
|
||||
}
|
||||
|
||||
+221
-179
@@ -5,29 +5,25 @@ use std::path::{Path, PathBuf};
|
||||
use memmap2::{Mmap, MmapMut};
|
||||
|
||||
use crate::reader::PersistentCompactIntVec;
|
||||
use crate::views::{BitSliceIter, BitSliceView, IntSliceView};
|
||||
|
||||
const MAGIC: [u8; 4] = *b"PBIV";
|
||||
|
||||
// Header: magic(4) + _pad(4) + n(8) = 16 bytes.
|
||||
// Data starts at offset 16, which is divisible by 8 → u64-aligned
|
||||
// (mmap base is page-aligned, 16 % 8 == 0).
|
||||
// Data starts at offset 16, u64-aligned (mmap base is page-aligned, 16 % 8 == 0).
|
||||
const HEADER_SIZE: usize = 16;
|
||||
|
||||
#[inline]
|
||||
fn n_words(n: usize) -> usize {
|
||||
n.div_ceil(64)
|
||||
}
|
||||
pub(crate) fn n_words(n: usize) -> usize { n.div_ceil(64) }
|
||||
|
||||
#[inline]
|
||||
fn n_bytes_for_words(n: usize) -> usize {
|
||||
n_words(n) * 8
|
||||
}
|
||||
fn n_bytes_for_words(n: usize) -> usize { n_words(n) * 8 }
|
||||
|
||||
// ── Reader ────────────────────────────────────────────────────────────────────
|
||||
// ── PersistentBitVec ──────────────────────────────────────────────────────────
|
||||
|
||||
pub struct PersistentBitVec {
|
||||
mmap: Mmap,
|
||||
n: usize,
|
||||
n: usize,
|
||||
path: PathBuf,
|
||||
}
|
||||
|
||||
@@ -35,157 +31,145 @@ impl PersistentBitVec {
|
||||
pub fn open(path: &Path) -> io::Result<Self> {
|
||||
let mmap = unsafe { Mmap::map(&File::open(path)?)? };
|
||||
if mmap.len() < HEADER_SIZE {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidData,
|
||||
"PBIV file too short",
|
||||
));
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidData, "PBIV file too short"));
|
||||
}
|
||||
if &mmap[0..4] != &MAGIC {
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidData, "bad PBIV magic"));
|
||||
}
|
||||
let n = u64::from_le_bytes(mmap[8..16].try_into().unwrap()) as usize;
|
||||
Ok(Self {
|
||||
mmap,
|
||||
n,
|
||||
path: path.to_path_buf(),
|
||||
})
|
||||
Ok(Self { mmap, n, path: path.to_path_buf() })
|
||||
}
|
||||
|
||||
pub fn path(&self) -> &Path {
|
||||
&self.path
|
||||
}
|
||||
pub fn len(&self) -> usize {
|
||||
self.n
|
||||
}
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.n == 0
|
||||
}
|
||||
pub fn path(&self) -> &Path { &self.path }
|
||||
pub fn len(&self) -> usize { self.n }
|
||||
pub fn is_empty(&self) -> bool { self.n == 0 }
|
||||
|
||||
pub fn get(&self, slot: usize) -> bool {
|
||||
(self.mmap[HEADER_SIZE + (slot >> 3)] >> (slot & 7)) & 1 != 0
|
||||
}
|
||||
|
||||
// Used by iter() and get(): exact byte window, no padding.
|
||||
fn data_bytes(&self) -> &[u8] {
|
||||
&self.mmap[HEADER_SIZE..HEADER_SIZE + self.n.div_ceil(8)]
|
||||
}
|
||||
|
||||
// Bulk word view. SAFETY: mmap is page-aligned, HEADER_SIZE=16 is divisible by 8,
|
||||
// so &mmap[HEADER_SIZE] is u64-aligned. Slice length is n_words * 8 bytes.
|
||||
// SAFETY: mmap is page-aligned, HEADER_SIZE=16 divisible by 8 → u64-aligned.
|
||||
fn data_words(&self) -> &[u64] {
|
||||
let nw = n_words(self.n);
|
||||
let nw = n_words(self.n);
|
||||
let ptr = self.mmap[HEADER_SIZE..].as_ptr() as *const u64;
|
||||
unsafe { std::slice::from_raw_parts(ptr, nw) }
|
||||
}
|
||||
|
||||
pub fn count_ones(&self) -> u64 {
|
||||
// Padding bits in the last word are 0, so no masking needed.
|
||||
self.data_words()
|
||||
.iter()
|
||||
.map(|w| w.count_ones() as u64)
|
||||
.sum()
|
||||
pub fn view(&self) -> BitSliceView<'_> {
|
||||
BitSliceView::new(self.data_words(), self.n)
|
||||
}
|
||||
|
||||
pub fn count_zeros(&self) -> u64 {
|
||||
self.n as u64 - self.count_ones()
|
||||
}
|
||||
pub fn words(&self) -> &[u64] { self.data_words() }
|
||||
|
||||
pub fn jaccard_dist(&self, other: &PersistentBitVec) -> f64 {
|
||||
let (inter, union) = self.partial_jaccard_dist(other);
|
||||
if union == 0 {
|
||||
return 0.0;
|
||||
}
|
||||
1.0 - inter as f64 / union as f64
|
||||
}
|
||||
pub fn count_ones(&self) -> u64 { self.view().count_ones() }
|
||||
pub fn count_zeros(&self) -> u64 { self.view().count_zeros() }
|
||||
|
||||
pub fn partial_jaccard_dist(&self, other: &PersistentBitVec) -> (u64, u64) {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
self.data_words()
|
||||
.iter()
|
||||
.zip(other.data_words())
|
||||
.fold((0u64, 0u64), |(i, u), (&a, &b)| {
|
||||
(
|
||||
i + (a & b).count_ones() as u64,
|
||||
u + (a | b).count_ones() as u64,
|
||||
)
|
||||
})
|
||||
self.view().partial_jaccard_dist(other.view())
|
||||
}
|
||||
pub fn jaccard_dist(&self, other: &PersistentBitVec) -> f64 {
|
||||
self.view().jaccard_dist(other.view())
|
||||
}
|
||||
|
||||
pub fn hamming_dist(&self, other: &PersistentBitVec) -> u64 {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
self.data_words()
|
||||
.iter()
|
||||
.zip(other.data_words())
|
||||
.map(|(&a, &b)| (a ^ b).count_ones() as u64)
|
||||
.sum()
|
||||
self.view().hamming_dist(other.view())
|
||||
}
|
||||
|
||||
pub fn iter(&self) -> BitIter<'_> {
|
||||
BitIter {
|
||||
bytes: self.data_bytes(),
|
||||
slot: 0,
|
||||
n: self.n,
|
||||
}
|
||||
BitIter { words: self.data_words(), slot: 0, n: self.n }
|
||||
}
|
||||
}
|
||||
|
||||
impl<'a> IntoIterator for &'a PersistentBitVec {
|
||||
type Item = bool;
|
||||
type IntoIter = BitIter<'a>;
|
||||
fn into_iter(self) -> BitIter<'a> {
|
||||
self.iter()
|
||||
}
|
||||
fn into_iter(self) -> BitIter<'a> { self.iter() }
|
||||
}
|
||||
|
||||
// ── BitIter ───────────────────────────────────────────────────────────────────
|
||||
|
||||
pub struct BitIter<'a> {
|
||||
bytes: &'a [u8],
|
||||
slot: usize,
|
||||
n: usize,
|
||||
words: &'a [u64],
|
||||
slot: usize,
|
||||
n: usize,
|
||||
}
|
||||
|
||||
impl ExactSizeIterator for BitIter<'_> {}
|
||||
|
||||
impl Iterator for BitIter<'_> {
|
||||
type Item = bool;
|
||||
|
||||
fn next(&mut self) -> Option<bool> {
|
||||
if self.slot >= self.n {
|
||||
return None;
|
||||
}
|
||||
let v = (self.bytes[self.slot >> 3] >> (self.slot & 7)) & 1 != 0;
|
||||
if self.slot >= self.n { return None; }
|
||||
let v = (self.words[self.slot >> 6] >> (self.slot & 63)) & 1 != 0;
|
||||
self.slot += 1;
|
||||
Some(v)
|
||||
}
|
||||
|
||||
fn size_hint(&self) -> (usize, Option<usize>) {
|
||||
let rem = self.n - self.slot;
|
||||
(rem, Some(rem))
|
||||
}
|
||||
}
|
||||
|
||||
// ── Builder ───────────────────────────────────────────────────────────────────
|
||||
// ── PersistentBitVecBuilder ───────────────────────────────────────────────────
|
||||
|
||||
pub struct PersistentBitVecBuilder {
|
||||
mmap: MmapMut,
|
||||
n: usize,
|
||||
n: usize,
|
||||
path: PathBuf,
|
||||
}
|
||||
|
||||
impl PersistentBitVecBuilder {
|
||||
pub fn new(n: usize, path: &Path) -> io::Result<Self> {
|
||||
let file_size = HEADER_SIZE + n_bytes_for_words(n);
|
||||
let mut file = OpenOptions::new()
|
||||
.read(true)
|
||||
.write(true)
|
||||
.create(true)
|
||||
.truncate(true)
|
||||
.read(true).write(true).create(true).truncate(true)
|
||||
.open(path)?;
|
||||
file.write_all(&MAGIC)?;
|
||||
file.write_all(&[0u8; 4])?; // padding
|
||||
file.write_all(&[0u8; 4])?;
|
||||
file.write_all(&(n as u64).to_le_bytes())?;
|
||||
file.seek(SeekFrom::Start(0))?;
|
||||
file.set_len(file_size as u64)?;
|
||||
let mmap = unsafe { MmapMut::map_mut(&file)? };
|
||||
Ok(Self { mmap, n })
|
||||
Ok(Self { mmap, n, path: path.to_path_buf() })
|
||||
}
|
||||
|
||||
pub fn from_raw_bytes(bytes: &[u8], n: usize, path: &Path) -> io::Result<Self> {
|
||||
let file_size = HEADER_SIZE + n_bytes_for_words(n);
|
||||
let file = OpenOptions::new()
|
||||
.read(true).write(true).create(true).truncate(true)
|
||||
.open(path)?;
|
||||
file.set_len(file_size as u64)?;
|
||||
let mut mmap = unsafe { MmapMut::map_mut(&file)? };
|
||||
mmap[0..4].copy_from_slice(&MAGIC);
|
||||
mmap[8..16].copy_from_slice(&(n as u64).to_le_bytes());
|
||||
mmap[HEADER_SIZE..HEADER_SIZE + bytes.len()].copy_from_slice(bytes);
|
||||
Ok(Self { mmap, n, path: path.to_path_buf() })
|
||||
}
|
||||
|
||||
/// Create an all-ones bit vector of length `n` at `path`.
|
||||
///
|
||||
/// More efficient than `new(n, path)` + `not()`: the data is written as
|
||||
/// 0xFF bytes in a single sequential pass, with no intermediate all-zeros state.
|
||||
pub fn new_ones(n: usize, path: &Path) -> io::Result<Self> {
|
||||
let nw = n_words(n);
|
||||
let file_size = HEADER_SIZE + nw * 8;
|
||||
let mut file = OpenOptions::new()
|
||||
.read(true).write(true).create(true).truncate(true)
|
||||
.open(path)?;
|
||||
file.write_all(&MAGIC)?;
|
||||
file.write_all(&[0u8; 4])?;
|
||||
file.write_all(&(n as u64).to_le_bytes())?;
|
||||
file.write_all(&vec![0xFFu8; nw * 8])?;
|
||||
file.seek(SeekFrom::Start(0))?;
|
||||
file.set_len(file_size as u64)?;
|
||||
let mut mmap = unsafe { MmapMut::map_mut(&file)? };
|
||||
// Clear padding bits in the last word so trailing bits are always 0.
|
||||
let rem = n % 64;
|
||||
if rem != 0 {
|
||||
let ptr = mmap[HEADER_SIZE..].as_mut_ptr() as *mut u64;
|
||||
let words = unsafe { std::slice::from_raw_parts_mut(ptr, nw) };
|
||||
words[nw - 1] &= (1u64 << rem) - 1;
|
||||
}
|
||||
Ok(Self { mmap, n, path: path.to_path_buf() })
|
||||
}
|
||||
|
||||
pub fn build_from(source: &PersistentBitVec, path: &Path) -> io::Result<Self> {
|
||||
@@ -193,86 +177,14 @@ impl PersistentBitVecBuilder {
|
||||
let file = OpenOptions::new().read(true).write(true).open(path)?;
|
||||
let mmap = unsafe { MmapMut::map_mut(&file)? };
|
||||
let n = source.len();
|
||||
Ok(Self { mmap, n })
|
||||
Ok(Self { mmap, n, path: path.to_path_buf() })
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize {
|
||||
self.n
|
||||
}
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.n == 0
|
||||
}
|
||||
|
||||
pub fn get(&self, slot: usize) -> bool {
|
||||
(self.mmap[HEADER_SIZE + (slot >> 3)] >> (slot & 7)) & 1 != 0
|
||||
}
|
||||
|
||||
pub fn set(&mut self, slot: usize, value: bool) {
|
||||
let byte = HEADER_SIZE + (slot >> 3);
|
||||
let bit = 1u8 << (slot & 7);
|
||||
if value {
|
||||
self.mmap[byte] |= bit;
|
||||
} else {
|
||||
self.mmap[byte] &= !bit;
|
||||
}
|
||||
}
|
||||
|
||||
// SAFETY: same alignment argument as PersistentBitVec::data_words.
|
||||
fn data_words_mut(&mut self) -> &mut [u64] {
|
||||
let nw = n_words(self.n);
|
||||
let ptr = self.mmap[HEADER_SIZE..].as_mut_ptr() as *mut u64;
|
||||
unsafe { std::slice::from_raw_parts_mut(ptr, nw) }
|
||||
}
|
||||
|
||||
pub fn and(&mut self, other: &PersistentBitVec) {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
for (sw, &ow) in self.data_words_mut().iter_mut().zip(other.data_words()) {
|
||||
*sw &= ow;
|
||||
}
|
||||
}
|
||||
|
||||
pub fn or(&mut self, other: &PersistentBitVec) {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
for (sw, &ow) in self.data_words_mut().iter_mut().zip(other.data_words()) {
|
||||
*sw |= ow;
|
||||
}
|
||||
}
|
||||
|
||||
pub fn xor(&mut self, other: &PersistentBitVec) {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
for (sw, &ow) in self.data_words_mut().iter_mut().zip(other.data_words()) {
|
||||
*sw ^= ow;
|
||||
}
|
||||
}
|
||||
|
||||
pub fn not(&mut self) {
|
||||
let rem = self.n % 64;
|
||||
let words = self.data_words_mut();
|
||||
for w in words.iter_mut() {
|
||||
*w ^= u64::MAX;
|
||||
}
|
||||
// Zero padding bits in the last word so count_ones / jaccard remain correct.
|
||||
if rem != 0 {
|
||||
if let Some(last) = words.last_mut() {
|
||||
*last &= (1u64 << rem) - 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert a count vector to a bit vector: bit set iff count >= threshold.
|
||||
/// Fills u64 words directly from the count iterator — O(n), no bit-level set() overhead.
|
||||
pub fn build_from_counts(
|
||||
source: &PersistentCompactIntVec,
|
||||
threshold: u32,
|
||||
path: &Path,
|
||||
) -> io::Result<Self> {
|
||||
pub fn build_from_counts(source: &PersistentCompactIntVec, threshold: u32, path: &Path) -> io::Result<Self> {
|
||||
let n = source.len();
|
||||
let file_size = HEADER_SIZE + n_bytes_for_words(n);
|
||||
let mut file = OpenOptions::new()
|
||||
.read(true)
|
||||
.write(true)
|
||||
.create(true)
|
||||
.truncate(true)
|
||||
.read(true).write(true).create(true).truncate(true)
|
||||
.open(path)?;
|
||||
file.write_all(&MAGIC)?;
|
||||
file.write_all(&[0u8; 4])?;
|
||||
@@ -280,27 +192,157 @@ impl PersistentBitVecBuilder {
|
||||
file.seek(SeekFrom::Start(0))?;
|
||||
file.set_len(file_size as u64)?;
|
||||
let mut mmap = unsafe { MmapMut::map_mut(&file)? };
|
||||
|
||||
{
|
||||
let nw = n_words(n);
|
||||
let nw = n_words(n);
|
||||
let ptr = mmap[HEADER_SIZE..].as_mut_ptr() as *mut u64;
|
||||
let words = unsafe { std::slice::from_raw_parts_mut(ptr, nw) };
|
||||
for (slot, count) in source.iter().enumerate() {
|
||||
if count >= threshold {
|
||||
words[slot >> 6] |= 1u64 << (slot & 63);
|
||||
}
|
||||
if count >= threshold { words[slot >> 6] |= 1u64 << (slot & 63); }
|
||||
}
|
||||
}
|
||||
|
||||
Ok(Self { mmap, n })
|
||||
Ok(Self { mmap, n, path: path.to_path_buf() })
|
||||
}
|
||||
|
||||
/// Convert a count vector to a presence/absence bit vector (threshold = 1).
|
||||
pub fn build_from_presence(source: &PersistentCompactIntVec, path: &Path) -> io::Result<Self> {
|
||||
Self::build_from_counts(source, 1, path)
|
||||
}
|
||||
|
||||
pub fn close(self) -> io::Result<()> {
|
||||
self.mmap.flush()
|
||||
pub fn len(&self) -> usize { self.n }
|
||||
pub fn is_empty(&self) -> bool { self.n == 0 }
|
||||
|
||||
pub fn get(&self, slot: usize) -> bool {
|
||||
(self.mmap[HEADER_SIZE + (slot >> 3)] >> (slot & 7)) & 1 != 0
|
||||
}
|
||||
|
||||
pub fn set(&mut self, slot: usize, value: bool) {
|
||||
let bit = 1u64 << (slot & 63);
|
||||
if value { self.data_words_mut()[slot >> 6] |= bit; }
|
||||
else { self.data_words_mut()[slot >> 6] &= !bit; }
|
||||
}
|
||||
|
||||
fn data_words(&self) -> &[u64] {
|
||||
let nw = n_words(self.n);
|
||||
let ptr = self.mmap[HEADER_SIZE..].as_ptr() as *const u64;
|
||||
unsafe { std::slice::from_raw_parts(ptr, nw) }
|
||||
}
|
||||
|
||||
// SAFETY: same alignment argument as PersistentBitVec::data_words.
|
||||
fn data_words_mut(&mut self) -> &mut [u64] {
|
||||
let nw = n_words(self.n);
|
||||
let ptr = self.mmap[HEADER_SIZE..].as_mut_ptr() as *mut u64;
|
||||
unsafe { std::slice::from_raw_parts_mut(ptr, nw) }
|
||||
}
|
||||
|
||||
pub fn view(&self) -> BitSliceView<'_> {
|
||||
BitSliceView::new(self.data_words(), self.n)
|
||||
}
|
||||
|
||||
pub fn words(&self) -> &[u64] { self.data_words() }
|
||||
|
||||
pub fn copy_from(&mut self, src: BitSliceView<'_>) {
|
||||
assert_eq!(self.n, src.len(), "BitSliceView length mismatch");
|
||||
self.data_words_mut().copy_from_slice(src.words());
|
||||
}
|
||||
|
||||
pub fn and(&mut self, other: BitSliceView<'_>) {
|
||||
assert_eq!(self.n, other.len(), "BitSliceView length mismatch");
|
||||
for (w, &o) in self.data_words_mut().iter_mut().zip(other.words()) { *w &= o; }
|
||||
}
|
||||
|
||||
pub fn or(&mut self, other: BitSliceView<'_>) {
|
||||
assert_eq!(self.n, other.len(), "BitSliceView length mismatch");
|
||||
for (w, &o) in self.data_words_mut().iter_mut().zip(other.words()) { *w |= o; }
|
||||
}
|
||||
|
||||
pub fn xor(&mut self, other: BitSliceView<'_>) {
|
||||
assert_eq!(self.n, other.len(), "BitSliceView length mismatch");
|
||||
for (w, &o) in self.data_words_mut().iter_mut().zip(other.words()) { *w ^= o; }
|
||||
}
|
||||
|
||||
pub fn not(&mut self) {
|
||||
let rem = self.n % 64;
|
||||
let words = self.data_words_mut();
|
||||
for w in words.iter_mut() { *w ^= u64::MAX; }
|
||||
if rem != 0 {
|
||||
if let Some(last) = words.last_mut() { *last &= (1u64 << rem) - 1; }
|
||||
}
|
||||
}
|
||||
|
||||
/// OR in bits at slots where `pred(col[slot])` is true.
|
||||
pub fn or_where(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
assert_eq!(self.n, col.len(), "IntSliceView length mismatch");
|
||||
let n = self.n;
|
||||
let primary = col.primary_bytes();
|
||||
let words = self.data_words_mut();
|
||||
let nw = n_words(n);
|
||||
for wi in 0..nw {
|
||||
let base = wi * 64;
|
||||
let limit = (base + 64).min(n);
|
||||
let mut mask = 0u64;
|
||||
for bit in 0..(limit - base) {
|
||||
let b = primary[base + bit];
|
||||
if b < 255 && pred(b as u32) { mask |= 1u64 << bit; }
|
||||
}
|
||||
words[wi] |= mask;
|
||||
}
|
||||
for (slot, val) in col.overflow_entries() {
|
||||
if pred(val) { words[slot >> 6] |= 1u64 << (slot & 63); }
|
||||
}
|
||||
}
|
||||
|
||||
/// Clear bits at slots where `pred(col[slot])` is false.
|
||||
pub fn and_where(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
assert_eq!(self.n, col.len(), "IntSliceView length mismatch");
|
||||
let n = self.n;
|
||||
let primary = col.primary_bytes();
|
||||
let words = self.data_words_mut();
|
||||
let nw = n_words(n);
|
||||
for wi in 0..nw {
|
||||
let base = wi * 64;
|
||||
let limit = (base + 64).min(n);
|
||||
let mut mask = 0u64;
|
||||
for bit in 0..(limit - base) {
|
||||
let b = primary[base + bit];
|
||||
if b < 255 && !pred(b as u32) { mask |= 1u64 << bit; }
|
||||
}
|
||||
words[wi] &= !mask;
|
||||
}
|
||||
for (slot, val) in col.overflow_entries() {
|
||||
if !pred(val) { words[slot >> 6] &= !(1u64 << (slot & 63)); }
|
||||
}
|
||||
}
|
||||
|
||||
/// Toggle bits at slots where `pred(col[slot])` is true.
|
||||
pub fn xor_where(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
assert_eq!(self.n, col.len(), "IntSliceView length mismatch");
|
||||
let n = self.n;
|
||||
let primary = col.primary_bytes();
|
||||
let words = self.data_words_mut();
|
||||
let nw = n_words(n);
|
||||
for wi in 0..nw {
|
||||
let base = wi * 64;
|
||||
let limit = (base + 64).min(n);
|
||||
let mut mask = 0u64;
|
||||
for bit in 0..(limit - base) {
|
||||
let b = primary[base + bit];
|
||||
if b < 255 && pred(b as u32) { mask |= 1u64 << bit; }
|
||||
}
|
||||
words[wi] ^= mask;
|
||||
}
|
||||
for (slot, val) in col.overflow_entries() {
|
||||
if pred(val) { words[slot >> 6] ^= 1u64 << (slot & 63); }
|
||||
}
|
||||
}
|
||||
|
||||
pub fn iter(&self) -> BitSliceIter<'_> {
|
||||
self.view().iter()
|
||||
}
|
||||
|
||||
pub fn close(self) -> io::Result<()> { self.mmap.flush() }
|
||||
|
||||
pub fn finish(self) -> io::Result<PersistentBitVec> {
|
||||
let path = self.path.clone();
|
||||
self.close()?;
|
||||
PersistentBitVec::open(&path)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,71 +5,57 @@ use std::path::{Path, PathBuf};
|
||||
|
||||
use memmap2::MmapMut;
|
||||
|
||||
use crate::format::{HEADER_SIZE, OVERFLOW_ENTRY_SIZE, finalize_pciv};
|
||||
use crate::format::{byte_count_nonzero, byte_sum, HEADER_SIZE, finalize_pciv, parse_overflow_entry};
|
||||
use crate::reader::PersistentCompactIntVec;
|
||||
use crate::views::{BitSliceView, IntSliceView};
|
||||
|
||||
pub struct PersistentCompactIntVecBuilder {
|
||||
path: PathBuf,
|
||||
mmap: MmapMut,
|
||||
n: usize,
|
||||
path: PathBuf,
|
||||
mmap: MmapMut,
|
||||
n: usize,
|
||||
overflow: HashMap<usize, u32>,
|
||||
}
|
||||
|
||||
impl PersistentCompactIntVecBuilder {
|
||||
/// Create a new, zero-filled PCIV at `path`. Primary is mmapped immediately.
|
||||
pub fn new(n: usize, path: &Path) -> io::Result<Self> {
|
||||
let file = OpenOptions::new()
|
||||
.read(true)
|
||||
.write(true)
|
||||
.create(true)
|
||||
.truncate(true)
|
||||
.read(true).write(true).create(true).truncate(true)
|
||||
.open(path)?;
|
||||
file.set_len((HEADER_SIZE + n) as u64)?;
|
||||
let mmap = unsafe { MmapMut::map_mut(&file)? };
|
||||
Ok(Self {
|
||||
path: path.to_path_buf(),
|
||||
mmap,
|
||||
n,
|
||||
overflow: HashMap::new(),
|
||||
})
|
||||
Ok(Self { path: path.to_path_buf(), mmap, n, overflow: HashMap::new() })
|
||||
}
|
||||
|
||||
pub fn from_raw_primary(primary: &[u8], overflow: HashMap<usize, u32>, path: &Path) -> io::Result<Self> {
|
||||
let n = primary.len();
|
||||
let file = OpenOptions::new()
|
||||
.read(true).write(true).create(true).truncate(true)
|
||||
.open(path)?;
|
||||
file.set_len((HEADER_SIZE + n) as u64)?;
|
||||
let mut mmap = unsafe { MmapMut::map_mut(&file)? };
|
||||
mmap[HEADER_SIZE..HEADER_SIZE + n].copy_from_slice(primary);
|
||||
Ok(Self { path: path.to_path_buf(), mmap, n, overflow })
|
||||
}
|
||||
|
||||
/// Copy `source`'s file to `path`, mmap the primary section, load overflow into RAM.
|
||||
/// Avoids iterating all n slots: the file copy is OS-level, overflow loading is O(n_overflow).
|
||||
pub fn build_from(source: &PersistentCompactIntVec, path: &Path) -> io::Result<Self> {
|
||||
fs::copy(source.path(), path)?;
|
||||
|
||||
let file = OpenOptions::new().read(true).write(true).open(path)?;
|
||||
let mmap = unsafe { MmapMut::map_mut(&file)? };
|
||||
|
||||
let n = source.len();
|
||||
let n = source.len();
|
||||
let n_overflow = u64::from_le_bytes(mmap[16..24].try_into().unwrap()) as usize;
|
||||
let data_offset = HEADER_SIZE + n;
|
||||
|
||||
let mut overflow = HashMap::with_capacity(n_overflow);
|
||||
for i in 0..n_overflow {
|
||||
let off = data_offset + i * OVERFLOW_ENTRY_SIZE;
|
||||
let slot = u64::from_le_bytes(mmap[off..off + 8].try_into().unwrap()) as usize;
|
||||
let value = u32::from_le_bytes(mmap[off + 8..off + 12].try_into().unwrap());
|
||||
let (slot, value) = parse_overflow_entry(&mmap, data_offset, i);
|
||||
overflow.insert(slot, value);
|
||||
}
|
||||
|
||||
Ok(Self {
|
||||
path: path.to_path_buf(),
|
||||
mmap,
|
||||
n,
|
||||
overflow,
|
||||
})
|
||||
Ok(Self { path: path.to_path_buf(), mmap, n, overflow })
|
||||
}
|
||||
|
||||
/// Get the value at the given slot, handling overflow if necessary.
|
||||
pub fn get(&self, slot: usize) -> u32 {
|
||||
match self.mmap[HEADER_SIZE + slot] {
|
||||
255 => *self
|
||||
.overflow
|
||||
.get(&slot)
|
||||
.expect("sentinel without overflow entry"),
|
||||
v => v as u32,
|
||||
255 => *self.overflow.get(&slot).expect("sentinel without overflow entry"),
|
||||
v => v as u32,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -83,61 +69,201 @@ impl PersistentCompactIntVecBuilder {
|
||||
}
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize {
|
||||
self.n
|
||||
pub fn len(&self) -> usize { self.n }
|
||||
pub fn is_empty(&self) -> bool { self.n == 0 }
|
||||
|
||||
pub fn primary_bytes(&self) -> &[u8] { &self.mmap[HEADER_SIZE..HEADER_SIZE + self.n] }
|
||||
pub fn primary_bytes_mut(&mut self) -> &mut [u8] { &mut self.mmap[HEADER_SIZE..HEADER_SIZE + self.n] }
|
||||
pub fn clear_overflow(&mut self) { self.overflow.clear(); }
|
||||
|
||||
pub fn sum(&self) -> u64 {
|
||||
byte_sum(&self.mmap[HEADER_SIZE..HEADER_SIZE + self.n], self.overflow.values().copied())
|
||||
}
|
||||
pub fn count_nonzero(&self) -> u64 {
|
||||
byte_count_nonzero(&self.mmap[HEADER_SIZE..HEADER_SIZE + self.n])
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.n == 0
|
||||
pub fn view(&self) -> IntSliceView<'_> {
|
||||
// Builder overflow is a HashMap, not sorted raw bytes — convert on the fly
|
||||
// by collecting into a sorted vec and storing in a thread-local buffer.
|
||||
// For read-back during building, just call get(slot) directly.
|
||||
// view() is primarily useful AFTER freeze (on PersistentCompactIntVec).
|
||||
// Here we expose it via a zero-alloc path: primary only, no overflow raw.
|
||||
// Callers that need overflow_entries during building use overflow_entries().
|
||||
let primary = &self.mmap[HEADER_SIZE..HEADER_SIZE + self.n];
|
||||
IntSliceView::new(primary, &[], 0, self.n)
|
||||
}
|
||||
|
||||
pub fn min(&mut self, other: &PersistentCompactIntVec) {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
for (slot, other_val) in other.iter().enumerate() {
|
||||
if other_val < self.get(slot) {
|
||||
self.set(slot, other_val);
|
||||
pub fn overflow_entries(&self) -> impl Iterator<Item = (usize, u32)> + '_ {
|
||||
self.overflow.iter().map(|(&k, &v)| (k, v))
|
||||
}
|
||||
|
||||
pub fn inc(&mut self, slot: usize) {
|
||||
let v = self.get(slot);
|
||||
self.set(slot, v.saturating_add(1));
|
||||
}
|
||||
|
||||
// ── Computation methods ───────────────────────────────────────────────────
|
||||
|
||||
/// Increment one counter per 1-bit of `col`. Safe for any group size.
|
||||
pub fn inc_present(&mut self, col: BitSliceView<'_>) {
|
||||
let n = self.n;
|
||||
for (wi, &word) in col.words().iter().enumerate() {
|
||||
if word == 0 { continue; }
|
||||
let mut w = word;
|
||||
while w != 0 {
|
||||
let bit = w.trailing_zeros() as usize;
|
||||
let slot = wi * 64 + bit;
|
||||
if slot < n { self.inc(slot); }
|
||||
w &= w - 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn max(&mut self, other: &PersistentCompactIntVec) {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
for (slot, other_val) in other.iter().enumerate() {
|
||||
if other_val > self.get(slot) {
|
||||
self.set(slot, other_val);
|
||||
/// Increment one counter per 1-bit of `col`, using raw u8 arithmetic.
|
||||
/// Caller guarantees no counter will reach 255 (group size < 255).
|
||||
pub fn inc_present_fast(&mut self, col: BitSliceView<'_>) {
|
||||
{
|
||||
let primary = self.primary_bytes_mut();
|
||||
let n = primary.len();
|
||||
for (wi, &word) in col.words().iter().enumerate() {
|
||||
if word == 0 { continue; }
|
||||
let mut w = word;
|
||||
while w != 0 {
|
||||
let bit = w.trailing_zeros() as usize;
|
||||
let s = wi * 64 + bit;
|
||||
if s < n { primary[s] += 1; }
|
||||
w &= w - 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
debug_assert!(
|
||||
!self.primary_bytes().contains(&255),
|
||||
"sentinel 255 reached in inc_present_fast — group size must be < 255"
|
||||
);
|
||||
}
|
||||
|
||||
/// Two-pass: primary bytes then overflow. Increments `self[slot]` for each
|
||||
/// slot where `pred(col[slot])` is true. Safe for any group size.
|
||||
pub fn inc_predicate(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
let n = col.len();
|
||||
for slot in 0..n {
|
||||
let b = col.primary_bytes()[slot];
|
||||
if b < 255 && pred(b as u32) {
|
||||
self.inc(slot);
|
||||
}
|
||||
}
|
||||
for (slot, val) in col.overflow_entries() {
|
||||
if pred(val) { self.inc(slot); }
|
||||
}
|
||||
}
|
||||
|
||||
/// Fast two-pass: raw u8 arithmetic. Caller guarantees no counter reaches 255.
|
||||
pub fn inc_predicate_fast(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
let n = col.len();
|
||||
{
|
||||
let primary = self.primary_bytes_mut();
|
||||
for slot in 0..n {
|
||||
let b = col.primary_bytes()[slot];
|
||||
if b < 255 && pred(b as u32) {
|
||||
primary[slot] += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (slot, val) in col.overflow_entries() {
|
||||
if pred(val) { self.primary_bytes_mut()[slot] += 1; }
|
||||
}
|
||||
debug_assert!(
|
||||
!self.primary_bytes().contains(&255),
|
||||
"sentinel 255 reached in inc_predicate_fast — group size must be < 255"
|
||||
);
|
||||
}
|
||||
|
||||
pub fn add(&mut self, other: IntSliceView<'_>) {
|
||||
let n = self.n;
|
||||
for s in 0..n {
|
||||
let sb = self.primary_bytes()[s];
|
||||
let ob = other.primary_bytes()[s];
|
||||
if sb < 255 && ob < 255 {
|
||||
let sum = sb as u32 + ob as u32;
|
||||
if sum < 255 { self.primary_bytes_mut()[s] = sum as u8; }
|
||||
else { self.set(s, sum); }
|
||||
} else {
|
||||
let sv = self.get(s);
|
||||
let ov = other.get(s);
|
||||
self.set(s, sv + ov);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn add(&mut self, other: &PersistentCompactIntVec) {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
for (slot, other_val) in other.iter().enumerate() {
|
||||
let cur = self.get(slot);
|
||||
self.set(slot, cur.checked_add(other_val).expect("u32 overflow in add"));
|
||||
pub fn min(&mut self, other: IntSliceView<'_>) {
|
||||
let self_ov: Vec<(usize, u32)> = self.overflow_entries().collect();
|
||||
let other_ov: HashMap<usize, u32> = other.overflow_entries().collect();
|
||||
self.clear_overflow();
|
||||
for (a, &b) in self.primary_bytes_mut().iter_mut().zip(other.primary_bytes()) {
|
||||
if b < *a { *a = b; }
|
||||
}
|
||||
for (slot, self_val) in self_ov {
|
||||
if let Some(&other_val) = other_ov.get(&slot) {
|
||||
self.set(slot, self_val.min(other_val));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn diff(&mut self, other: &PersistentCompactIntVec) {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
for (slot, other_val) in other.iter().enumerate() {
|
||||
self.set(slot, self.get(slot).saturating_sub(other_val));
|
||||
pub fn max(&mut self, other: IntSliceView<'_>) {
|
||||
for (slot, other_val) in other.overflow_entries() {
|
||||
let sv = self.get(slot);
|
||||
self.set(slot, sv.max(other_val));
|
||||
}
|
||||
for (a, &b) in self.primary_bytes_mut().iter_mut().zip(other.primary_bytes()) {
|
||||
if b > *a { *a = b; }
|
||||
}
|
||||
}
|
||||
|
||||
pub fn diff(&mut self, other: IntSliceView<'_>) {
|
||||
let n = self.n;
|
||||
for s in 0..n {
|
||||
let sb = self.primary_bytes()[s];
|
||||
let ob = other.primary_bytes()[s];
|
||||
if sb < 255 {
|
||||
self.primary_bytes_mut()[s] = if ob < 255 { sb.saturating_sub(ob) } else { 0 };
|
||||
} else {
|
||||
let sv = self.get(s);
|
||||
let ov = if ob < 255 { ob as u32 } else { other.get(s) };
|
||||
self.set(s, sv.saturating_sub(ov));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn mask_with(&mut self, mask: BitSliceView<'_>) {
|
||||
let n = self.n;
|
||||
for (wi, &word) in mask.words().iter().enumerate() {
|
||||
if word == u64::MAX { continue; }
|
||||
let mut zeros = !word;
|
||||
while zeros != 0 {
|
||||
let bit = zeros.trailing_zeros() as usize;
|
||||
let s = wi * 64 + bit;
|
||||
if s < n {
|
||||
let b = self.primary_bytes()[s];
|
||||
if b != 0 { self.set(s, 0); }
|
||||
}
|
||||
zeros &= zeros - 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Flush the primary mmap, then write sorted overflow data + index and fix the header.
|
||||
pub fn close(self) -> io::Result<()> {
|
||||
self.mmap.flush()?;
|
||||
let Self {
|
||||
path,
|
||||
mmap,
|
||||
n,
|
||||
overflow,
|
||||
} = self;
|
||||
let Self { path, mmap, n, overflow } = self;
|
||||
drop(mmap);
|
||||
|
||||
let mut entries: Vec<(usize, u32)> = overflow.into_iter().collect();
|
||||
entries.sort_unstable_by_key(|&(slot, _)| slot);
|
||||
|
||||
finalize_pciv(&path, n, &entries)
|
||||
}
|
||||
|
||||
pub fn finish(self) -> io::Result<PersistentCompactIntVec> {
|
||||
let path = self.path.clone();
|
||||
self.close()?;
|
||||
PersistentCompactIntVec::open(&path)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,137 @@
|
||||
use std::io;
|
||||
|
||||
use crate::tempbitvec::{TempBitVec, TempBitVecBuilder};
|
||||
use crate::tempintvec::TempCompactIntVec;
|
||||
|
||||
// ── ColGroup ──────────────────────────────────────────────────────────────────
|
||||
|
||||
/// A named subset of columns, identified by their indices within the matrix.
|
||||
///
|
||||
/// Defined once at the index level; the same indices are valid across all
|
||||
/// partitions and layers because the column structure (samples / genomes) is
|
||||
/// identical everywhere — only the row space (kmer slots) is partitioned.
|
||||
pub struct ColGroup {
|
||||
pub name: String,
|
||||
pub indices: Vec<usize>,
|
||||
}
|
||||
|
||||
impl ColGroup {
|
||||
pub fn new(name: impl Into<String>, indices: Vec<usize>) -> Self {
|
||||
Self { name: name.into(), indices }
|
||||
}
|
||||
}
|
||||
|
||||
// ── MatrixGroupOps ────────────────────────────────────────────────────────────
|
||||
|
||||
/// Per-matrix group aggregations.
|
||||
///
|
||||
/// `partial_group_presence_count`, `partial_group_sum`, `partial_group_any`,
|
||||
/// `partial_group_min`, `partial_group_max` are the primitives; each impl must
|
||||
/// provide all five.
|
||||
///
|
||||
/// `partial_group_all` and `partial_group_none` have default implementations
|
||||
/// derived from `partial_group_presence_count` and should rarely need overriding.
|
||||
pub trait MatrixGroupOps {
|
||||
/// Per-slot count of group columns whose value ≥ `threshold`.
|
||||
fn partial_group_presence_count(&self, g: &ColGroup, threshold: u32) -> io::Result<TempCompactIntVec>;
|
||||
|
||||
/// Per-slot sum of values across all group columns.
|
||||
fn partial_group_sum(&self, g: &ColGroup) -> io::Result<TempCompactIntVec>;
|
||||
|
||||
/// Per-slot OR: 1 if any group column has value ≥ `threshold`.
|
||||
fn partial_group_any(&self, g: &ColGroup, threshold: u32) -> io::Result<TempBitVec>;
|
||||
|
||||
/// Per-slot min value across all group columns (0 if group is empty).
|
||||
fn partial_group_min(&self, g: &ColGroup) -> io::Result<TempCompactIntVec>;
|
||||
|
||||
/// Per-slot max value across all group columns (0 if group is empty).
|
||||
fn partial_group_max(&self, g: &ColGroup) -> io::Result<TempCompactIntVec>;
|
||||
|
||||
/// Per-slot AND: 1 if ALL group columns have value ≥ `threshold`.
|
||||
fn partial_group_all(&self, g: &ColGroup, threshold: u32) -> io::Result<TempBitVec> {
|
||||
let counts = self.partial_group_presence_count(g, threshold)?;
|
||||
let n = counts.len();
|
||||
let n_required = g.indices.len() as u32;
|
||||
let mut b = TempBitVecBuilder::new(n)?;
|
||||
b.or_where(counts.view(), |v| v >= n_required);
|
||||
b.freeze()
|
||||
}
|
||||
|
||||
/// Per-slot NOR: 1 if NO group column has value ≥ `threshold`.
|
||||
fn partial_group_none(&self, g: &ColGroup, threshold: u32) -> io::Result<TempBitVec> {
|
||||
let counts = self.partial_group_presence_count(g, threshold)?;
|
||||
let n = counts.len();
|
||||
let mut b = TempBitVecBuilder::new(n)?;
|
||||
b.or_where(counts.view(), |v| v == 0);
|
||||
b.freeze()
|
||||
}
|
||||
}
|
||||
|
||||
// ── FilterMask — expression tree for column-based slot filters ────────────────
|
||||
|
||||
/// A composable filter expression that can be evaluated against a matrix
|
||||
/// using only column operations (no MPHF lookup per kmer).
|
||||
///
|
||||
/// `threshold` semantics follow [`MatrixGroupOps::partial_group_presence_count`]:
|
||||
/// a slot contributes to the count when its value is **≥ threshold**.
|
||||
/// To match the row-level filter (`value > t`), callers should pass `t + 1`.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum FilterMask {
|
||||
/// Slot passes if count of columns in `indices` with value ≥ `threshold` is ≥ `min_count`.
|
||||
PresenceGeq { indices: Vec<usize>, threshold: u32, min_count: usize },
|
||||
/// Slot passes if count of columns in `indices` with value ≥ `threshold` is ≤ `max_count`.
|
||||
PresenceLeq { indices: Vec<usize>, threshold: u32, max_count: usize },
|
||||
/// Slot passes if sum of values across `indices` columns is ≥ `min_sum`.
|
||||
SumGeq { indices: Vec<usize>, min_sum: u32 },
|
||||
/// Slot passes if sum of values across `indices` columns is ≤ `max_sum`.
|
||||
SumLeq { indices: Vec<usize>, max_sum: u32 },
|
||||
/// Slot passes if it passes all sub-expressions. Empty `And` is always true.
|
||||
And(Vec<FilterMask>),
|
||||
}
|
||||
|
||||
/// Evaluate a [`FilterMask`] against `mat`, returning a per-slot `TempBitVec`
|
||||
/// where bit=1 means the slot passes the filter.
|
||||
pub fn eval_filter_mask(expr: &FilterMask, mat: &dyn MatrixGroupOps, n: usize) -> io::Result<TempBitVec> {
|
||||
match expr {
|
||||
FilterMask::PresenceGeq { indices, threshold, min_count } => {
|
||||
let g = ColGroup::new("", indices.clone());
|
||||
let counts = mat.partial_group_presence_count(&g, *threshold)?;
|
||||
let mut b = TempBitVecBuilder::new(n)?;
|
||||
let mc = *min_count as u32;
|
||||
b.or_where(counts.view(), |v| v >= mc);
|
||||
b.freeze()
|
||||
}
|
||||
FilterMask::PresenceLeq { indices, threshold, max_count } => {
|
||||
let g = ColGroup::new("", indices.clone());
|
||||
let counts = mat.partial_group_presence_count(&g, *threshold)?;
|
||||
let mut b = TempBitVecBuilder::new(n)?;
|
||||
let mc = *max_count as u32;
|
||||
b.or_where(counts.view(), |v| v <= mc);
|
||||
b.freeze()
|
||||
}
|
||||
FilterMask::SumGeq { indices, min_sum } => {
|
||||
let g = ColGroup::new("", indices.clone());
|
||||
let sums = mat.partial_group_sum(&g)?;
|
||||
let mut b = TempBitVecBuilder::new(n)?;
|
||||
let ms = *min_sum;
|
||||
b.or_where(sums.view(), |v| v >= ms);
|
||||
b.freeze()
|
||||
}
|
||||
FilterMask::SumLeq { indices, max_sum } => {
|
||||
let g = ColGroup::new("", indices.clone());
|
||||
let sums = mat.partial_group_sum(&g)?;
|
||||
let mut b = TempBitVecBuilder::new(n)?;
|
||||
let ms = *max_sum;
|
||||
b.or_where(sums.view(), |v| v <= ms);
|
||||
b.freeze()
|
||||
}
|
||||
FilterMask::And(parts) => {
|
||||
let mut b = TempBitVecBuilder::new_ones(n)?;
|
||||
for part in parts {
|
||||
let m = eval_filter_mask(part, mat, n)?;
|
||||
b.and(m.view());
|
||||
}
|
||||
b.freeze()
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -13,6 +13,44 @@ pub const OVERFLOW_ENTRY_SIZE: usize = 12;
|
||||
// Index entry: slot(u64) + pos(u64) = 16 bytes.
|
||||
pub const INDEX_ENTRY_SIZE: usize = 16;
|
||||
|
||||
/// Sum all values in a compact-int primary byte slice, correcting for overflow sentinels.
|
||||
///
|
||||
/// `primary` is the raw `&[u8]` where 255 is a sentinel for large values.
|
||||
/// `overflow` yields the true values (≥ 255) for each sentinel, in any order.
|
||||
#[inline]
|
||||
pub(crate) fn byte_sum(primary: &[u8], overflow: impl Iterator<Item = u32>) -> u64 {
|
||||
let raw: u64 = primary.iter().map(|&b| b as u64).sum();
|
||||
let (n, ov) = overflow.fold((0u64, 0u64), |(n, s), v| (n + 1, s + v as u64));
|
||||
raw - 255 * n + ov
|
||||
}
|
||||
|
||||
/// Count non-zero values in a compact-int primary byte slice.
|
||||
///
|
||||
/// Overflow sentinels (255) are always non-zero by construction, so a single
|
||||
/// `b != 0` test is sufficient — no overflow map lookup needed.
|
||||
#[inline]
|
||||
pub(crate) fn byte_count_nonzero(primary: &[u8]) -> u64 {
|
||||
primary.iter().filter(|&&b| b != 0).count() as u64
|
||||
}
|
||||
|
||||
/// Parse a single overflow entry `(slot, value)` from a byte slice.
|
||||
#[inline]
|
||||
pub fn parse_overflow_entry(data: &[u8], base: usize, i: usize) -> (usize, u32) {
|
||||
let off = base + i * OVERFLOW_ENTRY_SIZE;
|
||||
let slot = u64::from_le_bytes(data[off..off+8].try_into().unwrap()) as usize;
|
||||
let value = u32::from_le_bytes(data[off+8..off+12].try_into().unwrap());
|
||||
(slot, value)
|
||||
}
|
||||
|
||||
/// Parse a single sparse-index entry `(slot, pos)` from a byte slice.
|
||||
#[inline]
|
||||
pub fn parse_index_entry(data: &[u8], base: usize, i: usize) -> (usize, usize) {
|
||||
let off = base + i * INDEX_ENTRY_SIZE;
|
||||
let slot = u64::from_le_bytes(data[off..off+8].try_into().unwrap()) as usize;
|
||||
let pos = u64::from_le_bytes(data[off+8..off+16].try_into().unwrap()) as usize;
|
||||
(slot, pos)
|
||||
}
|
||||
|
||||
// Sparse index target: ≤ 32 KB in L1 cache (16 B per entry → 2048 entries).
|
||||
pub const L1_INDEX_ENTRIES: usize = 2048;
|
||||
|
||||
|
||||
+192
-224
@@ -1,16 +1,20 @@
|
||||
use std::cmp::Ordering;
|
||||
use std::fs::{self, File};
|
||||
use std::io::{self, Write as _};
|
||||
use std::io::{self, BufWriter, Read as _, Write as _};
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use memmap2::Mmap;
|
||||
use ndarray::{Array1, Array2};
|
||||
use rayon::prelude::*;
|
||||
|
||||
use crate::bitmatrix::{pairwise_matrix, pairwise2_matrix};
|
||||
use crate::builder::PersistentCompactIntVecBuilder;
|
||||
use crate::format::{HEADER_SIZE, INDEX_ENTRY_SIZE, OVERFLOW_ENTRY_SIZE};
|
||||
use crate::colgroup::{ColGroup, MatrixGroupOps};
|
||||
use crate::format::{HEADER_SIZE, OVERFLOW_ENTRY_SIZE};
|
||||
use crate::meta::MatrixMeta;
|
||||
use crate::reader::PersistentCompactIntVec;
|
||||
use crate::tempbitvec::{TempBitVec, TempBitVecBuilder};
|
||||
use crate::tempintvec::{TempCompactIntVec, TempCompactIntVecBuilder};
|
||||
use crate::views::IntSliceView;
|
||||
|
||||
fn col_path(dir: &Path, col: usize) -> PathBuf {
|
||||
dir.join(format!("col_{col:06}.pciv"))
|
||||
@@ -41,9 +45,7 @@ impl ColumnarCompactIntMatrix {
|
||||
}
|
||||
|
||||
pub(crate) fn fill_row(&self, slot: usize, buf: &mut [u32]) {
|
||||
for (c, col) in self.cols.iter().enumerate() {
|
||||
buf[c] = col.get(slot);
|
||||
}
|
||||
for (c, col) in self.cols.iter().enumerate() { buf[c] = col.get(slot); }
|
||||
}
|
||||
|
||||
pub(crate) fn sum(&self) -> Array1<u64> {
|
||||
@@ -63,49 +65,26 @@ impl ColumnarCompactIntMatrix {
|
||||
}
|
||||
|
||||
pub(crate) fn partial_bray_dist_matrix(&self) -> Array2<u64> {
|
||||
self.pairwise_u64(|i, j| self.col(i).partial_bray_dist(self.col(j)))
|
||||
pairwise_matrix(self.n_cols(), |i, j| self.col(i).partial_bray_dist(self.col(j)))
|
||||
}
|
||||
|
||||
pub(crate) fn partial_euclidean_dist_matrix(&self) -> Array2<f64> {
|
||||
self.pairwise(|i, j| self.col(i).partial_euclidean_dist(self.col(j)))
|
||||
pairwise_matrix(self.n_cols(), |i, j| self.col(i).partial_euclidean_dist(self.col(j)))
|
||||
}
|
||||
|
||||
pub(crate) fn partial_threshold_jaccard_dist_matrix(
|
||||
&self, threshold: u32,
|
||||
) -> (Array2<u64>, Array2<u64>) {
|
||||
let n = self.n_cols();
|
||||
let pairs = upper_pairs(n);
|
||||
let results: Vec<(usize, usize, u64, u64)> = pairs
|
||||
.into_par_iter()
|
||||
.map(|(i, j)| {
|
||||
let (inter, union) =
|
||||
self.col(i).partial_threshold_jaccard_dist(self.col(j), threshold);
|
||||
(i, j, inter, union)
|
||||
})
|
||||
.collect();
|
||||
let mut inter_m = Array2::zeros((n, n));
|
||||
let mut union_m = Array2::zeros((n, n));
|
||||
for (i, j, inter, union) in results {
|
||||
inter_m[[i, j]] = inter; inter_m[[j, i]] = inter;
|
||||
union_m[[i, j]] = union; union_m[[j, i]] = union;
|
||||
}
|
||||
(inter_m, union_m)
|
||||
pub(crate) fn partial_threshold_jaccard_dist_matrix(&self, threshold: u32) -> (Array2<u64>, Array2<u64>) {
|
||||
pairwise2_matrix(self.n_cols(), |i, j| self.col(i).partial_threshold_jaccard_dist(self.col(j), threshold))
|
||||
}
|
||||
|
||||
pub(crate) fn partial_relfreq_bray_dist_matrix(&self, col_sums: &Array1<u64>) -> Array2<f64> {
|
||||
self.pairwise(|i, j| {
|
||||
pairwise_matrix(self.n_cols(), |i, j| {
|
||||
self.col(i).partial_relfreq_bray_dist(self.col(j), col_sums[i] as f64, col_sums[j] as f64)
|
||||
})
|
||||
}
|
||||
|
||||
pub(crate) fn partial_relfreq_euclidean_dist_matrix(&self, col_sums: &Array1<u64>) -> Array2<f64> {
|
||||
self.pairwise(|i, j| {
|
||||
pairwise_matrix(self.n_cols(), |i, j| {
|
||||
self.col(i).partial_relfreq_euclidean_dist(self.col(j), col_sums[i] as f64, col_sums[j] as f64)
|
||||
})
|
||||
}
|
||||
|
||||
pub(crate) fn partial_hellinger_euclidean_dist_matrix(&self, col_sums: &Array1<u64>) -> Array2<f64> {
|
||||
self.pairwise(|i, j| {
|
||||
pairwise_matrix(self.n_cols(), |i, j| {
|
||||
self.col(i).partial_hellinger_euclidean_dist(self.col(j), col_sums[i] as f64, col_sums[j] as f64)
|
||||
})
|
||||
}
|
||||
@@ -118,20 +97,6 @@ impl ColumnarCompactIntMatrix {
|
||||
meta.n_cols += 1;
|
||||
meta.save(dir)
|
||||
}
|
||||
|
||||
fn pairwise(&self, f: impl Fn(usize, usize) -> f64 + Sync) -> Array2<f64> {
|
||||
let n = self.n_cols();
|
||||
let results: Vec<(usize, usize, f64)> = upper_pairs(n)
|
||||
.into_par_iter().map(|(i, j)| (i, j, f(i, j))).collect();
|
||||
fill_symmetric(n, results.into_iter().map(|(i, j, v)| (i, j, v, v)))
|
||||
}
|
||||
|
||||
fn pairwise_u64(&self, f: impl Fn(usize, usize) -> u64 + Sync) -> Array2<u64> {
|
||||
let n = self.n_cols();
|
||||
let results: Vec<(usize, usize, u64)> = upper_pairs(n)
|
||||
.into_par_iter().map(|(i, j)| (i, j, f(i, j))).collect();
|
||||
fill_symmetric(n, results.into_iter().map(|(i, j, v)| (i, j, v, v)))
|
||||
}
|
||||
}
|
||||
|
||||
// ── PackedCompactIntMatrix ────────────────────────────────────────────────────
|
||||
@@ -139,13 +104,10 @@ impl ColumnarCompactIntMatrix {
|
||||
const PCMX_MAGIC: [u8; 4] = *b"PCMX";
|
||||
const PCMX_HEADER: usize = 24; // magic(4) + pad(4) + n_rows(8) + n_cols(8)
|
||||
|
||||
/// Per-column metadata pre-parsed from the embedded PCIV header.
|
||||
struct ColInfo {
|
||||
primary_start: usize, // absolute mmap offset to primary array
|
||||
data_offset: usize, // absolute mmap offset to overflow array
|
||||
primary_start: usize,
|
||||
data_offset: usize,
|
||||
n_overflow: usize,
|
||||
step: usize,
|
||||
index: Vec<(usize, usize)>,
|
||||
}
|
||||
|
||||
pub struct PackedCompactIntMatrix {
|
||||
@@ -171,61 +133,31 @@ impl PackedCompactIntMatrix {
|
||||
for c in 0..n_cols {
|
||||
let off_pos = PCMX_HEADER + c * 8;
|
||||
let col_base = u64::from_le_bytes(mmap[off_pos..off_pos+8].try_into().unwrap()) as usize;
|
||||
// Parse embedded PCIV header at col_base
|
||||
let n_ov = u64::from_le_bytes(mmap[col_base+16..col_base+24].try_into().unwrap()) as usize;
|
||||
let n_idx = u64::from_le_bytes(mmap[col_base+24..col_base+32].try_into().unwrap()) as usize;
|
||||
let step = u64::from_le_bytes(mmap[col_base+32..col_base+40].try_into().unwrap()) as usize;
|
||||
let n_pciv = u64::from_le_bytes(mmap[col_base+8..col_base+16].try_into().unwrap()) as usize;
|
||||
|
||||
let n_ov = u64::from_le_bytes(mmap[col_base+16..col_base+24].try_into().unwrap()) as usize;
|
||||
let n_pciv = u64::from_le_bytes(mmap[col_base+8..col_base+16].try_into().unwrap()) as usize;
|
||||
let primary_start = col_base + HEADER_SIZE;
|
||||
let data_offset = primary_start + n_pciv;
|
||||
let index_offset = data_offset + n_ov * OVERFLOW_ENTRY_SIZE;
|
||||
|
||||
let mut index = Vec::with_capacity(n_idx);
|
||||
for i in 0..n_idx {
|
||||
let ioff = index_offset + i * INDEX_ENTRY_SIZE;
|
||||
let slot = u64::from_le_bytes(mmap[ioff..ioff+8].try_into().unwrap()) as usize;
|
||||
let pos = u64::from_le_bytes(mmap[ioff+8..ioff+16].try_into().unwrap()) as usize;
|
||||
index.push((slot, pos));
|
||||
}
|
||||
columns.push(ColInfo { primary_start, data_offset, n_overflow: n_ov, step, index });
|
||||
columns.push(ColInfo { primary_start, data_offset, n_overflow: n_ov });
|
||||
}
|
||||
|
||||
Ok(Self { mmap, n_rows, n_cols, columns })
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn get(&self, col: usize, slot: usize) -> u32 {
|
||||
let ci = &self.columns[col];
|
||||
let v = self.mmap[ci.primary_start + slot];
|
||||
if v < 255 { return v as u32; }
|
||||
self.overflow_get(ci, slot)
|
||||
pub(crate) fn col_view(&self, c: usize) -> IntSliceView<'_> {
|
||||
let ci = &self.columns[c];
|
||||
let primary = &self.mmap[ci.primary_start..ci.primary_start + self.n_rows];
|
||||
let overflow_raw = &self.mmap[ci.data_offset..ci.data_offset + ci.n_overflow * OVERFLOW_ENTRY_SIZE];
|
||||
IntSliceView::new(primary, overflow_raw, ci.n_overflow, self.n_rows)
|
||||
}
|
||||
|
||||
fn overflow_get(&self, ci: &ColInfo, slot: usize) -> u32 {
|
||||
let (pos_start, pos_end) = if ci.step == 0 {
|
||||
(0, ci.n_overflow)
|
||||
} else {
|
||||
let i = ci.index.partition_point(|&(s, _)| s <= slot).saturating_sub(1);
|
||||
let start = ci.index[i].1;
|
||||
let end = if i + 1 < ci.index.len() { ci.index[i+1].1 } else { ci.n_overflow };
|
||||
(start, end)
|
||||
};
|
||||
let mut lo = pos_start;
|
||||
let mut hi = pos_end;
|
||||
while lo < hi {
|
||||
let mid = lo + (hi - lo) / 2;
|
||||
let off = ci.data_offset + mid * OVERFLOW_ENTRY_SIZE;
|
||||
let stored = u64::from_le_bytes(self.mmap[off..off+8].try_into().unwrap()) as usize;
|
||||
match stored.cmp(&slot) {
|
||||
Ordering::Equal => return u32::from_le_bytes(self.mmap[off+8..off+12].try_into().unwrap()),
|
||||
Ordering::Less => lo = mid + 1,
|
||||
Ordering::Greater => hi = mid,
|
||||
}
|
||||
}
|
||||
panic!("slot {slot} marked overflow but not found")
|
||||
pub(crate) fn col_persist(&self, c: usize, path: &Path) -> io::Result<PersistentCompactIntVecBuilder> {
|
||||
let view = self.col_view(c);
|
||||
let overflow: std::collections::HashMap<usize, u32> = view.overflow_entries().collect();
|
||||
PersistentCompactIntVecBuilder::from_raw_primary(view.primary_bytes(), overflow, path)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn get(&self, col: usize, slot: usize) -> u32 { self.col_view(col).get(slot) }
|
||||
|
||||
pub(crate) fn fill_row(&self, slot: usize, buf: &mut [u32]) {
|
||||
for c in 0..self.n_cols { buf[c] = self.get(c, slot); }
|
||||
}
|
||||
@@ -236,149 +168,123 @@ impl PackedCompactIntMatrix {
|
||||
|
||||
pub(crate) fn sum(&self) -> Array1<u64> {
|
||||
Array1::from_vec(
|
||||
(0..self.n_cols).into_par_iter()
|
||||
.map(|c| (0..self.n_rows).map(|s| self.get(c, s) as u64).sum())
|
||||
.collect()
|
||||
(0..self.n_cols).into_par_iter().map(|c| self.col_view(c).sum()).collect()
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn count_nonzero(&self) -> Array1<u64> {
|
||||
Array1::from_vec(
|
||||
(0..self.n_cols).into_par_iter()
|
||||
.map(|c| (0..self.n_rows).filter(|&s| self.get(c, s) > 0).count() as u64)
|
||||
.collect()
|
||||
(0..self.n_cols).into_par_iter().map(|c| self.col_view(c).count_nonzero()).collect()
|
||||
)
|
||||
}
|
||||
|
||||
// ── Pair primitives ───────────────────────────────────────────────────────
|
||||
|
||||
fn pair_partial_bray(&self, i: usize, j: usize) -> u64 {
|
||||
(0..self.n_rows).map(|s| self.get(i, s).min(self.get(j, s)) as u64).sum()
|
||||
self.col_view(i).iter().zip(self.col_view(j).iter()).map(|(a, b)| a.min(b) as u64).sum()
|
||||
}
|
||||
|
||||
fn pair_partial_euclidean(&self, i: usize, j: usize) -> f64 {
|
||||
(0..self.n_rows).map(|s| {
|
||||
let d = self.get(i, s) as f64 - self.get(j, s) as f64;
|
||||
d * d
|
||||
}).sum()
|
||||
self.col_view(i).iter().zip(self.col_view(j).iter())
|
||||
.map(|(a, b)| { let d = a as f64 - b as f64; d * d }).sum()
|
||||
}
|
||||
|
||||
fn pair_partial_threshold_jaccard(&self, i: usize, j: usize, t: u32) -> (u64, u64) {
|
||||
let (mut inter, mut union) = (0u64, 0u64);
|
||||
for s in 0..self.n_rows {
|
||||
let a = self.get(i, s) >= t;
|
||||
let b = self.get(j, s) >= t;
|
||||
if a && b { inter += 1; }
|
||||
if a || b { union += 1; }
|
||||
}
|
||||
(inter, union)
|
||||
self.col_view(i).iter().zip(self.col_view(j).iter())
|
||||
.fold((0u64, 0u64), |(inter, uni), (a, b)| {
|
||||
let ap = a >= t; let bp = b >= t;
|
||||
(inter + (ap & bp) as u64, uni + (ap | bp) as u64)
|
||||
})
|
||||
}
|
||||
|
||||
fn pair_partial_relfreq_bray(&self, i: usize, j: usize, si: f64, sj: f64) -> f64 {
|
||||
if si == 0.0 || sj == 0.0 { return 0.0; }
|
||||
(0..self.n_rows).map(|s| {
|
||||
(self.get(i, s) as f64 / si).min(self.get(j, s) as f64 / sj)
|
||||
}).sum()
|
||||
self.col_view(i).iter().zip(self.col_view(j).iter())
|
||||
.map(|(a, b)| (a as f64 / si).min(b as f64 / sj)).sum()
|
||||
}
|
||||
|
||||
fn pair_partial_relfreq_euclidean(&self, i: usize, j: usize, si: f64, sj: f64) -> f64 {
|
||||
if si == 0.0 || sj == 0.0 { return 0.0; }
|
||||
(0..self.n_rows).map(|s| {
|
||||
let d = self.get(i, s) as f64 / si - self.get(j, s) as f64 / sj;
|
||||
d * d
|
||||
}).sum()
|
||||
self.col_view(i).iter().zip(self.col_view(j).iter())
|
||||
.map(|(a, b)| { let d = a as f64 / si - b as f64 / sj; d * d }).sum()
|
||||
}
|
||||
|
||||
fn pair_partial_hellinger(&self, i: usize, j: usize, si: f64, sj: f64) -> f64 {
|
||||
if si == 0.0 || sj == 0.0 { return 0.0; }
|
||||
(0..self.n_rows).map(|s| {
|
||||
let d = (self.get(i, s) as f64 / si).sqrt() - (self.get(j, s) as f64 / sj).sqrt();
|
||||
d * d
|
||||
}).sum()
|
||||
}
|
||||
|
||||
// ── Matrix methods ────────────────────────────────────────────────────────
|
||||
|
||||
fn pairwise<T>(&self, f: impl Fn(usize, usize) -> T + Sync) -> Array2<T>
|
||||
where T: Clone + Default + Send {
|
||||
let n = self.n_cols;
|
||||
let results: Vec<(usize, usize, T)> = upper_pairs(n)
|
||||
.into_par_iter().map(|(i, j)| (i, j, f(i, j))).collect();
|
||||
fill_symmetric(n, results.into_iter().map(|(i, j, v)| { let w = v.clone(); (i, j, v, w) }))
|
||||
}
|
||||
|
||||
fn pairwise_u64(&self, f: impl Fn(usize, usize) -> u64 + Sync) -> Array2<u64> {
|
||||
let n = self.n_cols;
|
||||
let results: Vec<(usize, usize, u64)> = upper_pairs(n)
|
||||
.into_par_iter().map(|(i, j)| (i, j, f(i, j))).collect();
|
||||
fill_symmetric(n, results.into_iter().map(|(i, j, v)| (i, j, v, v)))
|
||||
self.col_view(i).iter().zip(self.col_view(j).iter())
|
||||
.map(|(a, b)| { let d = (a as f64 / si).sqrt() - (b as f64 / sj).sqrt(); d * d }).sum()
|
||||
}
|
||||
|
||||
pub(crate) fn partial_bray_dist_matrix(&self) -> Array2<u64> {
|
||||
self.pairwise_u64(|i, j| self.pair_partial_bray(i, j))
|
||||
pairwise_matrix(self.n_cols, |i, j| self.pair_partial_bray(i, j))
|
||||
}
|
||||
|
||||
|
||||
pub(crate) fn partial_euclidean_dist_matrix(&self) -> Array2<f64> {
|
||||
self.pairwise(|i, j| self.pair_partial_euclidean(i, j))
|
||||
pairwise_matrix(self.n_cols, |i, j| self.pair_partial_euclidean(i, j))
|
||||
}
|
||||
|
||||
pub(crate) fn partial_threshold_jaccard_dist_matrix(&self, t: u32) -> (Array2<u64>, Array2<u64>) {
|
||||
let n = self.n_cols;
|
||||
let results: Vec<(usize, usize, u64, u64)> = upper_pairs(n)
|
||||
.into_par_iter()
|
||||
.map(|(i, j)| { let (inter, union) = self.pair_partial_threshold_jaccard(i, j, t); (i, j, inter, union) })
|
||||
.collect();
|
||||
let mut inter_m = Array2::zeros((n, n));
|
||||
let mut union_m = Array2::zeros((n, n));
|
||||
for (i, j, inter, union) in results {
|
||||
inter_m[[i, j]] = inter; inter_m[[j, i]] = inter;
|
||||
union_m[[i, j]] = union; union_m[[j, i]] = union;
|
||||
}
|
||||
(inter_m, union_m)
|
||||
pairwise2_matrix(self.n_cols, |i, j| self.pair_partial_threshold_jaccard(i, j, t))
|
||||
}
|
||||
|
||||
pub(crate) fn partial_relfreq_bray_dist_matrix(&self, col_sums: &Array1<u64>) -> Array2<f64> {
|
||||
self.pairwise(|i, j| self.pair_partial_relfreq_bray(i, j, col_sums[i] as f64, col_sums[j] as f64))
|
||||
pairwise_matrix(self.n_cols, |i, j| self.pair_partial_relfreq_bray(i, j, col_sums[i] as f64, col_sums[j] as f64))
|
||||
}
|
||||
|
||||
pub(crate) fn partial_relfreq_euclidean_dist_matrix(&self, col_sums: &Array1<u64>) -> Array2<f64> {
|
||||
self.pairwise(|i, j| self.pair_partial_relfreq_euclidean(i, j, col_sums[i] as f64, col_sums[j] as f64))
|
||||
pairwise_matrix(self.n_cols, |i, j| self.pair_partial_relfreq_euclidean(i, j, col_sums[i] as f64, col_sums[j] as f64))
|
||||
}
|
||||
|
||||
pub(crate) fn partial_hellinger_euclidean_dist_matrix(&self, col_sums: &Array1<u64>) -> Array2<f64> {
|
||||
self.pairwise(|i, j| self.pair_partial_hellinger(i, j, col_sums[i] as f64, col_sums[j] as f64))
|
||||
pairwise_matrix(self.n_cols, |i, j| self.pair_partial_hellinger(i, j, col_sums[i] as f64, col_sums[j] as f64))
|
||||
}
|
||||
}
|
||||
|
||||
/// Reads just the `n_cols` field from an existing packed matrix's header,
|
||||
/// without mapping the file. Used by `pack_compact_int_matrix` to tell a
|
||||
/// genuinely complete pack from a stale one that predates a later
|
||||
/// column-widening.
|
||||
fn packed_int_matrix_n_cols(path: &Path) -> io::Result<usize> {
|
||||
let mut f = File::open(path)?;
|
||||
let mut header = [0u8; PCMX_HEADER];
|
||||
f.read_exact(&mut header)?;
|
||||
Ok(u64::from_le_bytes(header[16..24].try_into().unwrap()) as usize)
|
||||
}
|
||||
|
||||
/// Build `counts/matrix.pcmx` from existing `col_*.pciv` files.
|
||||
pub fn pack_compact_int_matrix(dir: &Path) -> io::Result<()> {
|
||||
let meta = MatrixMeta::load(dir)?;
|
||||
let n_cols = meta.n_cols;
|
||||
let packed_path = dir.join("matrix.pcmx");
|
||||
|
||||
let col_files: Vec<Vec<u8>> = (0..n_cols)
|
||||
.map(|c| fs::read(col_path(dir, c)))
|
||||
.collect::<io::Result<_>>()?;
|
||||
let meta = match MatrixMeta::load(dir) {
|
||||
Ok(meta) => meta,
|
||||
Err(e) => {
|
||||
// No columnar data pending: either this layer was already
|
||||
// packed and cleaned up (matrix.pcmx complete, nothing left to
|
||||
// do), or genuinely nothing was ever written here.
|
||||
return if packed_path.exists() { Ok(()) } else { Err(e) };
|
||||
}
|
||||
};
|
||||
|
||||
let header_size = PCMX_HEADER + n_cols * 8;
|
||||
let mut col_offset = header_size;
|
||||
let mut offsets = Vec::with_capacity(n_cols);
|
||||
for data in &col_files {
|
||||
offsets.push(col_offset as u64);
|
||||
col_offset += data.len();
|
||||
// A `matrix.pcmx` can already exist here even though columnar data is
|
||||
// still pending — e.g. copied verbatim from a merge's base source
|
||||
// before this layer was widened with more genome columns (see
|
||||
// `obikpartitionner::merge_partition`). Only skip (re-)packing if the
|
||||
// existing file already reflects the current column count; otherwise
|
||||
// the columnar files are newer and must be (re-)packed, overwriting the
|
||||
// stale one — never silently discarded as "leftover cleanup".
|
||||
if packed_int_matrix_n_cols(&packed_path).ok() == Some(meta.n_cols) {
|
||||
for c in 0..meta.n_cols { let _ = fs::remove_file(col_path(dir, c)); }
|
||||
let _ = fs::remove_file(dir.join("meta.json"));
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let packed_path = dir.join("matrix.pcmx");
|
||||
let mut file = File::create(&packed_path)?;
|
||||
file.write_all(&PCMX_MAGIC)?;
|
||||
file.write_all(&[0u8; 4])?;
|
||||
file.write_all(&(meta.n as u64).to_le_bytes())?;
|
||||
file.write_all(&(n_cols as u64).to_le_bytes())?;
|
||||
for &off in &offsets { file.write_all(&off.to_le_bytes())?; }
|
||||
for data in &col_files { file.write_all(data)?; }
|
||||
drop(file);
|
||||
|
||||
let n_cols = meta.n_cols;
|
||||
let col_sizes: Vec<u64> = (0..n_cols)
|
||||
.map(|c| fs::metadata(col_path(dir, c)).map(|m| m.len()))
|
||||
.collect::<io::Result<_>>()?;
|
||||
let header_size = (PCMX_HEADER + n_cols * 8) as u64;
|
||||
let mut col_offset = header_size;
|
||||
let mut offsets = Vec::with_capacity(n_cols);
|
||||
for &size in &col_sizes { offsets.push(col_offset); col_offset += size; }
|
||||
let tmp_path = dir.join("matrix.pcmx.tmp");
|
||||
let mut out = BufWriter::new(File::create(&tmp_path)?);
|
||||
out.write_all(&PCMX_MAGIC)?;
|
||||
out.write_all(&[0u8; 4])?;
|
||||
out.write_all(&(meta.n as u64).to_le_bytes())?;
|
||||
out.write_all(&(n_cols as u64).to_le_bytes())?;
|
||||
for &off in &offsets { out.write_all(&off.to_le_bytes())?; }
|
||||
for c in 0..n_cols { io::copy(&mut File::open(col_path(dir, c))?, &mut out)?; }
|
||||
out.flush()?;
|
||||
drop(out);
|
||||
fs::rename(&tmp_path, &packed_path)?;
|
||||
for c in 0..n_cols { fs::remove_file(col_path(dir, c))?; }
|
||||
fs::remove_file(dir.join("meta.json"))?;
|
||||
Ok(())
|
||||
@@ -392,18 +298,14 @@ pub enum PersistentCompactIntMatrix {
|
||||
}
|
||||
|
||||
impl PersistentCompactIntMatrix {
|
||||
/// Open from `layer_dir`, auto-detecting Packed or Columnar.
|
||||
pub fn open(layer_dir: &Path) -> io::Result<Self> {
|
||||
let counts_dir = layer_dir.join("counts");
|
||||
|
||||
if counts_dir.join("matrix.pcmx").exists() {
|
||||
return Ok(Self::Packed(PackedCompactIntMatrix::open(&counts_dir.join("matrix.pcmx"))?));
|
||||
}
|
||||
|
||||
if MatrixMeta::load(&counts_dir).is_ok() {
|
||||
return Ok(Self::Columnar(ColumnarCompactIntMatrix::open(&counts_dir)?));
|
||||
}
|
||||
|
||||
Err(io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("no count matrix found in {} — run 'obikmer upgrade'", layer_dir.display()),
|
||||
@@ -413,7 +315,6 @@ impl PersistentCompactIntMatrix {
|
||||
pub fn n(&self) -> usize {
|
||||
match self { Self::Columnar(m) => m.n(), Self::Packed(m) => m.n_rows }
|
||||
}
|
||||
|
||||
pub fn n_cols(&self) -> usize {
|
||||
match self { Self::Columnar(m) => m.n_cols(), Self::Packed(m) => m.n_cols }
|
||||
}
|
||||
@@ -425,22 +326,32 @@ impl PersistentCompactIntMatrix {
|
||||
}
|
||||
}
|
||||
|
||||
pub fn col_view(&self, c: usize) -> IntSliceView<'_> {
|
||||
match self {
|
||||
Self::Columnar(m) => m.col(c).view(),
|
||||
Self::Packed(m) => m.col_view(c),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn col_persist(&self, c: usize, path: &Path) -> io::Result<PersistentCompactIntVecBuilder> {
|
||||
match self {
|
||||
Self::Columnar(m) => PersistentCompactIntVecBuilder::build_from(m.col(c), path),
|
||||
Self::Packed(m) => m.col_persist(c, path),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn row(&self, slot: usize) -> Box<[u32]> {
|
||||
match self { Self::Columnar(m) => m.row(slot), Self::Packed(m) => m.row(slot) }
|
||||
}
|
||||
|
||||
pub fn fill_row(&self, slot: usize, buf: &mut [u32]) {
|
||||
match self { Self::Columnar(m) => m.fill_row(slot, buf), Self::Packed(m) => m.fill_row(slot, buf) }
|
||||
}
|
||||
|
||||
pub fn sum(&self) -> Array1<u64> {
|
||||
match self { Self::Columnar(m) => m.sum(), Self::Packed(m) => m.sum() }
|
||||
}
|
||||
|
||||
pub fn count_nonzero(&self) -> Array1<u64> {
|
||||
match self { Self::Columnar(m) => m.count_nonzero(), Self::Packed(m) => m.count_nonzero() }
|
||||
}
|
||||
|
||||
pub fn partial_bray_dist_matrix(&self) -> Array2<u64> {
|
||||
match self { Self::Columnar(m) => m.partial_bray_dist_matrix(), Self::Packed(m) => m.partial_bray_dist_matrix() }
|
||||
}
|
||||
@@ -459,7 +370,6 @@ impl PersistentCompactIntMatrix {
|
||||
pub fn partial_hellinger_euclidean_dist_matrix(&self, col_sums: &Array1<u64>) -> Array2<f64> {
|
||||
match self { Self::Columnar(m) => m.partial_hellinger_euclidean_dist_matrix(col_sums), Self::Packed(m) => m.partial_hellinger_euclidean_dist_matrix(col_sums) }
|
||||
}
|
||||
|
||||
pub fn append_column(dir: &Path, value_of: impl Fn(usize) -> u32) -> io::Result<()> {
|
||||
ColumnarCompactIntMatrix::append_column(dir, value_of)
|
||||
}
|
||||
@@ -475,12 +385,12 @@ impl ColumnWeights for PersistentCompactIntMatrix {
|
||||
}
|
||||
|
||||
impl CountPartials for PersistentCompactIntMatrix {
|
||||
fn partial_bray(&self) -> Array2<u64> { self.partial_bray_dist_matrix() }
|
||||
fn partial_euclidean(&self) -> Array2<f64> { self.partial_euclidean_dist_matrix() }
|
||||
fn partial_bray(&self) -> Array2<u64> { self.partial_bray_dist_matrix() }
|
||||
fn partial_euclidean(&self) -> Array2<f64> { self.partial_euclidean_dist_matrix() }
|
||||
fn partial_threshold_jaccard(&self, t: u32) -> (Array2<u64>, Array2<u64>) { self.partial_threshold_jaccard_dist_matrix(t) }
|
||||
fn partial_relfreq_bray(&self, g: &Array1<u64>) -> Array2<f64> { self.partial_relfreq_bray_dist_matrix(g) }
|
||||
fn partial_relfreq_euclidean(&self, g: &Array1<u64>) -> Array2<f64> { self.partial_relfreq_euclidean_dist_matrix(g) }
|
||||
fn partial_hellinger(&self, g: &Array1<u64>) -> Array2<f64> { self.partial_hellinger_euclidean_dist_matrix(g) }
|
||||
fn partial_relfreq_bray(&self, g: &Array1<u64>) -> Array2<f64> { self.partial_relfreq_bray_dist_matrix(g) }
|
||||
fn partial_relfreq_euclidean(&self, g: &Array1<u64>) -> Array2<f64> { self.partial_relfreq_euclidean_dist_matrix(g) }
|
||||
fn partial_hellinger(&self, g: &Array1<u64>) -> Array2<f64> { self.partial_hellinger_euclidean_dist_matrix(g) }
|
||||
}
|
||||
|
||||
// ── Builder ───────────────────────────────────────────────────────────────────
|
||||
@@ -496,30 +406,88 @@ impl PersistentCompactIntMatrixBuilder {
|
||||
fs::create_dir_all(dir)?;
|
||||
Ok(Self { dir: dir.to_path_buf(), n, n_cols: 0 })
|
||||
}
|
||||
|
||||
pub fn n(&self) -> usize { self.n }
|
||||
pub fn n_cols(&self) -> usize { self.n_cols }
|
||||
|
||||
pub fn add_col(&mut self) -> io::Result<PersistentCompactIntVecBuilder> {
|
||||
let path = col_path(&self.dir, self.n_cols);
|
||||
self.n_cols += 1;
|
||||
PersistentCompactIntVecBuilder::new(self.n, &path)
|
||||
}
|
||||
|
||||
pub fn add_col_from(&mut self, src: &TempCompactIntVec) -> io::Result<()> {
|
||||
src.make_persistent(&col_path(&self.dir, self.n_cols))?;
|
||||
self.n_cols += 1;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn add_col_from_bit(&mut self, src: &TempBitVec) -> io::Result<()> {
|
||||
let path = col_path(&self.dir, self.n_cols);
|
||||
self.n_cols += 1;
|
||||
let mut b = PersistentCompactIntVecBuilder::new(self.n, &path)?;
|
||||
b.inc_present(src.view());
|
||||
b.close()
|
||||
}
|
||||
|
||||
pub fn close(self) -> io::Result<()> {
|
||||
MatrixMeta { n: self.n, n_cols: self.n_cols }.save(&self.dir)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Helpers ───────────────────────────────────────────────────────────────────
|
||||
// ── MatrixGroupOps ────────────────────────────────────────────────────────────
|
||||
|
||||
fn upper_pairs(n: usize) -> Vec<(usize, usize)> {
|
||||
(0..n).flat_map(|i| (i + 1..n).map(move |j| (i, j))).collect()
|
||||
}
|
||||
impl MatrixGroupOps for PersistentCompactIntMatrix {
|
||||
fn partial_group_presence_count(&self, g: &ColGroup, threshold: u32) -> io::Result<TempCompactIntVec> {
|
||||
let n = self.n();
|
||||
if g.indices.len() < 255 {
|
||||
let mut builder = TempCompactIntVecBuilder::new(n)?;
|
||||
for &c in &g.indices {
|
||||
builder.inc_predicate_fast(self.col_view(c), |v| v >= threshold);
|
||||
}
|
||||
builder.freeze()
|
||||
} else {
|
||||
let mut result = TempCompactIntVecBuilder::new(n)?;
|
||||
for chunk in g.indices.chunks(254) {
|
||||
let mut chunk_b = TempCompactIntVecBuilder::new(n)?;
|
||||
for &c in chunk {
|
||||
chunk_b.inc_predicate_fast(self.col_view(c), |v| v >= threshold);
|
||||
}
|
||||
let frozen = chunk_b.freeze()?;
|
||||
result.add(frozen.view());
|
||||
}
|
||||
result.freeze()
|
||||
}
|
||||
}
|
||||
|
||||
fn fill_symmetric<T>(n: usize, vals: impl Iterator<Item = (usize, usize, T, T)>) -> Array2<T>
|
||||
where T: Clone + Default {
|
||||
let mut m = Array2::from_elem((n, n), T::default());
|
||||
for (i, j, vij, vji) in vals { m[[i, j]] = vij; m[[j, i]] = vji; }
|
||||
m
|
||||
fn partial_group_sum(&self, g: &ColGroup) -> io::Result<TempCompactIntVec> {
|
||||
let n = self.n();
|
||||
let mut result = TempCompactIntVecBuilder::new(n)?;
|
||||
for &c in &g.indices { result.add(self.col_view(c)); }
|
||||
result.freeze()
|
||||
}
|
||||
|
||||
fn partial_group_any(&self, g: &ColGroup, threshold: u32) -> io::Result<TempBitVec> {
|
||||
let n = self.n();
|
||||
let mut result = TempBitVecBuilder::new(n)?;
|
||||
for &c in &g.indices {
|
||||
result.or_where(self.col_view(c), |v| v >= threshold);
|
||||
}
|
||||
result.freeze()
|
||||
}
|
||||
|
||||
fn partial_group_min(&self, g: &ColGroup) -> io::Result<TempCompactIntVec> {
|
||||
let n = self.n();
|
||||
let mut result = TempCompactIntVecBuilder::new(n)?;
|
||||
if let Some((&first, rest)) = g.indices.split_first() {
|
||||
result.add(self.col_view(first));
|
||||
for &c in rest { result.min(self.col_view(c)); }
|
||||
}
|
||||
result.freeze()
|
||||
}
|
||||
|
||||
fn partial_group_max(&self, g: &ColGroup) -> io::Result<TempCompactIntVec> {
|
||||
let n = self.n();
|
||||
let mut result = TempCompactIntVecBuilder::new(n)?;
|
||||
for &c in &g.indices { result.max(self.col_view(c)); }
|
||||
result.freeze()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -23,11 +23,6 @@ impl LayerMeta {
|
||||
}
|
||||
|
||||
fn parse(s: &str) -> Option<Self> {
|
||||
let key = "\"n\":";
|
||||
let pos = s.find(key)? + key.len();
|
||||
let rest = s[pos..].trim_start();
|
||||
let end = rest.find(|c: char| !c.is_ascii_digit()).unwrap_or(rest.len());
|
||||
let n = rest[..end].parse().ok()?;
|
||||
Some(Self { n })
|
||||
Some(Self { n: crate::meta::field(s, "n")? })
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,20 +1,30 @@
|
||||
mod bitvec;
|
||||
mod bitmatrix;
|
||||
mod builder;
|
||||
mod colgroup;
|
||||
mod format;
|
||||
mod intmatrix;
|
||||
mod layer_meta;
|
||||
mod meta;
|
||||
mod reader;
|
||||
mod siblingannex;
|
||||
mod tempbitvec;
|
||||
mod tempintvec;
|
||||
mod views;
|
||||
pub mod traits;
|
||||
|
||||
pub use bitvec::{BitIter, PersistentBitVec, PersistentBitVecBuilder};
|
||||
pub use bitmatrix::{PersistentBitMatrix, PersistentBitMatrixBuilder, pack_bit_matrix};
|
||||
pub use builder::PersistentCompactIntVecBuilder;
|
||||
pub use colgroup::{ColGroup, FilterMask, MatrixGroupOps, eval_filter_mask};
|
||||
pub use intmatrix::{PersistentCompactIntMatrix, PersistentCompactIntMatrixBuilder, pack_compact_int_matrix};
|
||||
pub use layer_meta::LayerMeta;
|
||||
pub use reader::PersistentCompactIntVec;
|
||||
pub use siblingannex::{FamilyMask, SiblingAnnex, SiblingAnnexBuilder};
|
||||
pub use reader::{PersistentCompactIntVec, Iter as CompactIntVecIter};
|
||||
pub use tempbitvec::{TempBitVec, TempBitVecBuilder};
|
||||
pub use tempintvec::{TempCompactIntVec, TempCompactIntVecBuilder};
|
||||
pub use traits::{BitPartials, ColumnWeights, CountPartials};
|
||||
pub use views::{BitSliceView, BitSliceIter, IntSliceView, IntSliceViewIter};
|
||||
|
||||
#[cfg(test)]
|
||||
#[path = "tests/mod.rs"]
|
||||
|
||||
@@ -23,7 +23,7 @@ fn parse(s: &str) -> Option<MatrixMeta> {
|
||||
Some(MatrixMeta { n: field(s, "n")?, n_cols: field(s, "n_cols")? })
|
||||
}
|
||||
|
||||
fn field(s: &str, name: &str) -> Option<usize> {
|
||||
pub(crate) fn field(s: &str, name: &str) -> Option<usize> {
|
||||
let key = format!("\"{}\":", name);
|
||||
let pos = s.find(&key)? + key.len();
|
||||
let rest = s[pos..].trim_start();
|
||||
|
||||
+70
-211
@@ -4,7 +4,8 @@ use std::path::{Path, PathBuf};
|
||||
|
||||
use memmap2::Mmap;
|
||||
|
||||
use crate::format::{HEADER_SIZE, INDEX_ENTRY_SIZE, MAGIC, OVERFLOW_ENTRY_SIZE};
|
||||
use crate::format::{byte_count_nonzero, byte_sum, HEADER_SIZE, MAGIC, OVERFLOW_ENTRY_SIZE, parse_index_entry};
|
||||
use crate::views::IntSliceView;
|
||||
|
||||
pub struct PersistentCompactIntVec {
|
||||
mmap: Mmap,
|
||||
@@ -18,100 +19,60 @@ pub struct PersistentCompactIntVec {
|
||||
}
|
||||
|
||||
impl PersistentCompactIntVec {
|
||||
/// Opens a persistent compact int vector from the given path.
|
||||
pub fn open(path: &Path) -> io::Result<Self> {
|
||||
let mmap = unsafe { Mmap::map(&File::open(path)?)? };
|
||||
|
||||
if mmap.len() < HEADER_SIZE {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidData,
|
||||
"PCIV file too short",
|
||||
));
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidData, "PCIV file too short"));
|
||||
}
|
||||
if &mmap[0..4] != &MAGIC {
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidData, "bad PCIV magic"));
|
||||
}
|
||||
|
||||
let n = u64::from_le_bytes(mmap[8..16].try_into().unwrap()) as usize;
|
||||
let n = u64::from_le_bytes(mmap[8..16].try_into().unwrap()) as usize;
|
||||
let n_overflow = u64::from_le_bytes(mmap[16..24].try_into().unwrap()) as usize;
|
||||
let n_index = u64::from_le_bytes(mmap[24..32].try_into().unwrap()) as usize;
|
||||
let step = u64::from_le_bytes(mmap[32..40].try_into().unwrap()) as usize;
|
||||
let n_index = u64::from_le_bytes(mmap[24..32].try_into().unwrap()) as usize;
|
||||
let step = u64::from_le_bytes(mmap[32..40].try_into().unwrap()) as usize;
|
||||
|
||||
let primary_offset = HEADER_SIZE;
|
||||
let data_offset = primary_offset + n;
|
||||
let index_offset = data_offset + n_overflow * OVERFLOW_ENTRY_SIZE;
|
||||
let data_offset = primary_offset + n;
|
||||
let index_offset = data_offset + n_overflow * OVERFLOW_ENTRY_SIZE;
|
||||
|
||||
let mut index = Vec::with_capacity(n_index);
|
||||
for i in 0..n_index {
|
||||
let off = index_offset + i * INDEX_ENTRY_SIZE;
|
||||
let slot = u64::from_le_bytes(mmap[off..off + 8].try_into().unwrap()) as usize;
|
||||
let pos = u64::from_le_bytes(mmap[off + 8..off + 16].try_into().unwrap()) as usize;
|
||||
index.push((slot, pos));
|
||||
index.push(parse_index_entry(&mmap, index_offset, i));
|
||||
}
|
||||
|
||||
Ok(Self {
|
||||
mmap,
|
||||
n,
|
||||
n_overflow,
|
||||
step,
|
||||
index,
|
||||
primary_offset,
|
||||
data_offset,
|
||||
path: path.to_path_buf(),
|
||||
})
|
||||
Ok(Self { mmap, n, n_overflow, step, index, primary_offset, data_offset, path: path.to_path_buf() })
|
||||
}
|
||||
|
||||
/// Returns the path of the compact int vector file.
|
||||
pub fn path(&self) -> &Path {
|
||||
&self.path
|
||||
}
|
||||
pub fn path(&self) -> &Path { &self.path }
|
||||
pub fn len(&self) -> usize { self.n }
|
||||
pub fn is_empty(&self) -> bool { self.n == 0 }
|
||||
|
||||
/// Returns the length of the compact int vector.
|
||||
pub fn len(&self) -> usize {
|
||||
self.n
|
||||
}
|
||||
|
||||
/// Returns whether the compact int vector is empty.
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.n == 0
|
||||
}
|
||||
|
||||
/// Returns the value at the given slot.
|
||||
pub fn get(&self, slot: usize) -> u32 {
|
||||
match self.mmap[self.primary_offset + slot] {
|
||||
255 => self.overflow_get(slot),
|
||||
v => v as u32,
|
||||
v => v as u32,
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns the value at the given slot from the overflow region.
|
||||
fn overflow_get(&self, slot: usize) -> u32 {
|
||||
let pos_start;
|
||||
let pos_end;
|
||||
|
||||
if self.step == 0 {
|
||||
pos_start = 0;
|
||||
pos_end = self.n_overflow;
|
||||
let (pos_start, pos_end) = if self.step == 0 {
|
||||
(0, self.n_overflow)
|
||||
} else {
|
||||
let i = self
|
||||
.index
|
||||
.partition_point(|&(s, _)| s <= slot)
|
||||
.saturating_sub(1);
|
||||
pos_start = self.index[i].1;
|
||||
pos_end = if i + 1 < self.index.len() {
|
||||
self.index[i + 1].1
|
||||
} else {
|
||||
self.n_overflow
|
||||
};
|
||||
}
|
||||
|
||||
let i = self.index.partition_point(|&(s, _)| s <= slot).saturating_sub(1);
|
||||
let start = self.index[i].1;
|
||||
let end = if i + 1 < self.index.len() { self.index[i + 1].1 } else { self.n_overflow };
|
||||
(start, end)
|
||||
};
|
||||
let mut lo = pos_start;
|
||||
let mut hi = pos_end;
|
||||
while lo < hi {
|
||||
let mid = lo + (hi - lo) / 2;
|
||||
match self.data_slot(mid).cmp(&slot) {
|
||||
std::cmp::Ordering::Equal => return self.data_value(mid),
|
||||
std::cmp::Ordering::Less => lo = mid + 1,
|
||||
std::cmp::Ordering::Equal => return self.data_value(mid),
|
||||
std::cmp::Ordering::Less => lo = mid + 1,
|
||||
std::cmp::Ordering::Greater => hi = mid,
|
||||
}
|
||||
}
|
||||
@@ -119,144 +80,91 @@ impl PersistentCompactIntVec {
|
||||
}
|
||||
|
||||
#[inline]
|
||||
/// Returns the slot at the given index in the overflow region.
|
||||
fn data_slot(&self, i: usize) -> usize {
|
||||
let off = self.data_offset + i * OVERFLOW_ENTRY_SIZE;
|
||||
u64::from_le_bytes(self.mmap[off..off + 8].try_into().unwrap()) as usize
|
||||
}
|
||||
|
||||
#[inline]
|
||||
/// Returns the value at the given index in the overflow region.
|
||||
fn data_value(&self, i: usize) -> u32 {
|
||||
let off = self.data_offset + i * OVERFLOW_ENTRY_SIZE + 8;
|
||||
u32::from_le_bytes(self.mmap[off..off + 4].try_into().unwrap())
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn sum(&self) -> u64 {
|
||||
self.iter().map(|v| v as u64).sum()
|
||||
let primary = &self.mmap[self.primary_offset..self.primary_offset + self.n];
|
||||
byte_sum(primary, (0..self.n_overflow).map(|i| self.data_value(i)))
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn count_nonzero(&self) -> u64 {
|
||||
self.iter().filter(|&v| v > 0).count() as u64
|
||||
let primary = &self.mmap[self.primary_offset..self.primary_offset + self.n];
|
||||
byte_count_nonzero(primary)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
/// Returns the Bray-Curtis distance between two compact int vectors.
|
||||
/// Lightweight zero-copy view — primary and overflow point into the mmap.
|
||||
pub fn view(&self) -> IntSliceView<'_> {
|
||||
let primary = &self.mmap[self.primary_offset..self.primary_offset + self.n];
|
||||
let overflow_raw = &self.mmap[self.data_offset..self.data_offset + self.n_overflow * OVERFLOW_ENTRY_SIZE];
|
||||
IntSliceView::new(primary, overflow_raw, self.n_overflow, self.n)
|
||||
}
|
||||
|
||||
pub fn iter(&self) -> Iter<'_> {
|
||||
Iter { pciv: self, slot: 0, overflow_pos: 0 }
|
||||
}
|
||||
|
||||
// ── Distance methods ──────────────────────────────────────────────────────
|
||||
|
||||
pub fn bray_dist(&self, other: &PersistentCompactIntVec) -> f64 {
|
||||
let sum_min = self.partial_bray_dist(other);
|
||||
let denom = self.sum() + other.sum();
|
||||
if denom == 0 {
|
||||
return 0.0;
|
||||
}
|
||||
1.0 - 2.0 * sum_min as f64 / denom as f64
|
||||
if denom == 0 { 0.0 } else { 1.0 - 2.0 * sum_min as f64 / denom as f64 }
|
||||
}
|
||||
|
||||
/// Returns `Σ_slot min(self[slot], other[slot])` — the additive numerator of Bray-Curtis.
|
||||
/// The denominator `sum_a + sum_b` is obtained from `self.sum() + other.sum()`.
|
||||
pub fn partial_bray_dist(&self, other: &PersistentCompactIntVec) -> u64 {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
self.iter()
|
||||
.zip(other.iter())
|
||||
.map(|(a, b)| a.min(b) as u64)
|
||||
.sum()
|
||||
self.iter().zip(other.iter()).map(|(a, b)| a.min(b) as u64).sum()
|
||||
}
|
||||
|
||||
/// Returns the relative frequency Bray-Curtis distance between two compact int vectors.
|
||||
///
|
||||
/// This is a variant of [`bray_dist`] that uses relative frequencies instead of raw counts.
|
||||
pub fn relfreq_bray_dist(&self, other: &PersistentCompactIntVec) -> f64 {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
let sum_a = self.sum() as f64;
|
||||
let sum_b = other.sum() as f64;
|
||||
if sum_a == 0.0 && sum_b == 0.0 {
|
||||
return 0.0;
|
||||
}
|
||||
let sum_min = self.partial_relfreq_bray_dist(other, sum_a, sum_b);
|
||||
1.0 - sum_min
|
||||
let sa = self.sum() as f64;
|
||||
let sb = other.sum() as f64;
|
||||
if sa == 0.0 && sb == 0.0 { return 0.0; }
|
||||
1.0 - self.partial_relfreq_bray_dist(other, sa, sb)
|
||||
}
|
||||
|
||||
/// Returns the partial relative frequency Bray-Curtis distance between two compact int vectors.
|
||||
///
|
||||
/// This is used internally by [`relfreq_bray_dist`] and to easily compute the relative frequency
|
||||
/// Bray-Curtis distance over a set of vector pairs.
|
||||
///
|
||||
/// Arguments:
|
||||
/// - `other`: the other compact int vector to compare with
|
||||
/// - `sum_a`: the sum of the first vector's counts
|
||||
/// - `sum_b`: the sum of the second vector's counts
|
||||
///
|
||||
/// Returns the sum of the minimum relative frequencies at each index.
|
||||
pub fn partial_relfreq_bray_dist(
|
||||
&self,
|
||||
other: &PersistentCompactIntVec,
|
||||
sum_a: f64,
|
||||
sum_b: f64,
|
||||
) -> f64 {
|
||||
pub fn partial_relfreq_bray_dist(&self, other: &PersistentCompactIntVec, sum_a: f64, sum_b: f64) -> f64 {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
let sum_min: f64 = self
|
||||
.iter()
|
||||
.zip(other.iter())
|
||||
self.iter().zip(other.iter())
|
||||
.map(|(a, b)| {
|
||||
let pa = if sum_a > 0.0 { a as f64 / sum_a } else { 0.0 };
|
||||
let pb = if sum_b > 0.0 { b as f64 / sum_b } else { 0.0 };
|
||||
pa.min(pb)
|
||||
})
|
||||
.sum();
|
||||
sum_min
|
||||
.sum()
|
||||
}
|
||||
|
||||
/// Returns the euclidean distance between two compact int vectors.
|
||||
pub fn euclidean_dist(&self, other: &PersistentCompactIntVec) -> f64 {
|
||||
self.partial_euclidean_dist(other).sqrt()
|
||||
}
|
||||
|
||||
/// Returns the partial euclidean distance between two compact int vectors.
|
||||
///
|
||||
/// This is used internally by [`euclidean_dist`] and to easily compute the euclidean distance
|
||||
/// over a set of vector pairs.
|
||||
///
|
||||
/// The result is the sum of the squared differences between corresponding elements of the two
|
||||
/// vectors.
|
||||
pub fn partial_euclidean_dist(&self, other: &PersistentCompactIntVec) -> f64 {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
self.iter()
|
||||
.zip(other.iter())
|
||||
.map(|(a, b)| {
|
||||
let d = a as f64 - b as f64;
|
||||
d * d
|
||||
})
|
||||
self.iter().zip(other.iter())
|
||||
.map(|(a, b)| { let d = a as f64 - b as f64; d * d })
|
||||
.sum()
|
||||
}
|
||||
|
||||
/// Returns the relative frequency euclidean distance between two compact int vectors.
|
||||
///
|
||||
/// This is a variant of [`euclidean_dist`] that uses relative frequencies instead of raw counts.
|
||||
pub fn relfreq_euclidean_dist(&self, other: &PersistentCompactIntVec) -> f64 {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
let sum_a = self.sum() as f64;
|
||||
let sum_b = other.sum() as f64;
|
||||
if sum_a == 0.0 && sum_b == 0.0 {
|
||||
return 0.0;
|
||||
}
|
||||
self.partial_relfreq_euclidean_dist(other, sum_a, sum_b)
|
||||
.sqrt()
|
||||
let sa = self.sum() as f64;
|
||||
let sb = other.sum() as f64;
|
||||
if sa == 0.0 && sb == 0.0 { return 0.0; }
|
||||
self.partial_relfreq_euclidean_dist(other, sa, sb).sqrt()
|
||||
}
|
||||
|
||||
/// Returns the partial relative frequency euclidean distance between two compact int vectors.
|
||||
///
|
||||
/// This is used internally by [`relfreq_euclidean_dist`] and to easily compute the relative frequency
|
||||
/// euclidean distance over a set of vector pairs.
|
||||
pub fn partial_relfreq_euclidean_dist(
|
||||
&self,
|
||||
other: &PersistentCompactIntVec,
|
||||
sum_a: f64,
|
||||
sum_b: f64,
|
||||
) -> f64 {
|
||||
pub fn partial_relfreq_euclidean_dist(&self, other: &PersistentCompactIntVec, sum_a: f64, sum_b: f64) -> f64 {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
self.iter()
|
||||
.zip(other.iter())
|
||||
self.iter().zip(other.iter())
|
||||
.map(|(a, b)| {
|
||||
let pa = if sum_a > 0.0 { a as f64 / sum_a } else { 0.0 };
|
||||
let pb = if sum_b > 0.0 { b as f64 / sum_b } else { 0.0 };
|
||||
@@ -266,46 +174,19 @@ impl PersistentCompactIntVec {
|
||||
.sum()
|
||||
}
|
||||
|
||||
/// Returns the Euclidean distance between two compact int vectors using the Hellinger transform.
|
||||
///
|
||||
/// The Hellinger transform is applied to the raw counts of each vector, and the result is
|
||||
/// the Euclidean distance between the transformed vectors. The Hellinger transform is defined
|
||||
/// as the square root of the relative frequencies.
|
||||
pub fn hellinger_euclidean_dist(&self, other: &PersistentCompactIntVec) -> f64 {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
let sum_a = self.sum() as f64;
|
||||
let sum_b = other.sum() as f64;
|
||||
if sum_a == 0.0 && sum_b == 0.0 {
|
||||
return 0.0;
|
||||
}
|
||||
self.partial_hellinger_euclidean_dist(other, sum_a, sum_b)
|
||||
.sqrt()
|
||||
let sa = self.sum() as f64;
|
||||
let sb = other.sum() as f64;
|
||||
if sa == 0.0 && sb == 0.0 { return 0.0; }
|
||||
self.partial_hellinger_euclidean_dist(other, sa, sb).sqrt()
|
||||
}
|
||||
|
||||
/// Returns the partial Hellinger Euclidean distance between two compact int vectors.
|
||||
///
|
||||
/// This is used internally by [`hellinger_euclidean_dist`] and to easily compute the Hellinger
|
||||
/// Euclidean distance over a set of vector pairs.
|
||||
pub fn partial_hellinger_euclidean_dist(
|
||||
&self,
|
||||
other: &PersistentCompactIntVec,
|
||||
sum_a: f64,
|
||||
sum_b: f64,
|
||||
) -> f64 {
|
||||
pub fn partial_hellinger_euclidean_dist(&self, other: &PersistentCompactIntVec, sum_a: f64, sum_b: f64) -> f64 {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
self.iter()
|
||||
.zip(other.iter())
|
||||
self.iter().zip(other.iter())
|
||||
.map(|(a, b)| {
|
||||
let pa = if sum_a > 0.0 {
|
||||
(a as f64 / sum_a).sqrt()
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
let pb = if sum_b > 0.0 {
|
||||
(b as f64 / sum_b).sqrt()
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
let pa = if sum_a > 0.0 { (a as f64 / sum_a).sqrt() } else { 0.0 };
|
||||
let pb = if sum_b > 0.0 { (b as f64 / sum_b).sqrt() } else { 0.0 };
|
||||
let d = pa - pb;
|
||||
d * d
|
||||
})
|
||||
@@ -317,22 +198,13 @@ impl PersistentCompactIntVec {
|
||||
}
|
||||
|
||||
pub fn threshold_jaccard_dist(&self, other: &PersistentCompactIntVec, threshold: u32) -> f64 {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
let (intersection, union) = self.partial_threshold_jaccard_dist(other, threshold);
|
||||
if union == 0 {
|
||||
return 0.0;
|
||||
}
|
||||
1.0 - intersection as f64 / union as f64
|
||||
if union == 0 { 0.0 } else { 1.0 - intersection as f64 / union as f64 }
|
||||
}
|
||||
|
||||
pub fn partial_threshold_jaccard_dist(
|
||||
&self,
|
||||
other: &PersistentCompactIntVec,
|
||||
threshold: u32,
|
||||
) -> (u64, u64) {
|
||||
pub fn partial_threshold_jaccard_dist(&self, other: &PersistentCompactIntVec, threshold: u32) -> (u64, u64) {
|
||||
assert_eq!(self.n, other.len(), "length mismatch");
|
||||
self.iter()
|
||||
.zip(other.iter())
|
||||
self.iter().zip(other.iter())
|
||||
.fold((0u64, 0u64), |(inter, uni), (a, b)| {
|
||||
let ap = a >= threshold;
|
||||
let bp = b >= threshold;
|
||||
@@ -343,23 +215,12 @@ impl PersistentCompactIntVec {
|
||||
pub fn jaccard_dist(&self, other: &PersistentCompactIntVec) -> f64 {
|
||||
self.threshold_jaccard_dist(other, 1)
|
||||
}
|
||||
|
||||
pub fn iter(&self) -> Iter<'_> {
|
||||
Iter {
|
||||
pciv: self,
|
||||
slot: 0,
|
||||
overflow_pos: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<'a> IntoIterator for &'a PersistentCompactIntVec {
|
||||
type Item = u32;
|
||||
type IntoIter = Iter<'a>;
|
||||
|
||||
fn into_iter(self) -> Iter<'a> {
|
||||
self.iter()
|
||||
}
|
||||
fn into_iter(self) -> Iter<'a> { self.iter() }
|
||||
}
|
||||
|
||||
pub struct Iter<'a> {
|
||||
@@ -374,9 +235,7 @@ impl Iterator for Iter<'_> {
|
||||
type Item = u32;
|
||||
|
||||
fn next(&mut self) -> Option<u32> {
|
||||
if self.slot >= self.pciv.n {
|
||||
return None;
|
||||
}
|
||||
if self.slot >= self.pciv.n { return None; }
|
||||
let v = self.pciv.mmap[self.pciv.primary_offset + self.slot];
|
||||
self.slot += 1;
|
||||
if v < 255 {
|
||||
|
||||
@@ -0,0 +1,245 @@
|
||||
//! Family presence-mask annex: a compact, read-only-after-build, per-slot
|
||||
//! derived value used by the central-position SNP distance estimator (see
|
||||
//! `docmd/theory/evolutionary_distances.md`, "Step 2b" and "Definitions:
|
||||
//! family, and the canonical form of a family").
|
||||
//!
|
||||
//! One byte is stored per MPHF slot of a partition/layer, its low 4 bits
|
||||
//! encoding a **presence mask** for the slot's k-mer's "family" (the up to 4
|
||||
//! k-mers sharing the same flanks, differing only at the central base):
|
||||
//! bit `b` (`b` = 0..3, in the fixed A/C/G/T = 0/1/2/3 encoding already used
|
||||
//! for a single nucleotide) is set iff the family member whose *own* central
|
||||
//! base — in its own canonical orientation — is `b`, is observed anywhere in
|
||||
//! the current multi-genome index. This is a property of the whole index,
|
||||
//! not of any one genome.
|
||||
//!
|
||||
//! Both facts the earlier (superseded) 3-bit design stored explicitly are
|
||||
//! derived from the mask instead, not stored:
|
||||
//! - sibling count = `popcount(mask) - 1`;
|
||||
//! - minorant = regenerate the family's 4 canonical forms from the slot's
|
||||
//! own k-mer (`CanonicalKmerOf::central_canonical_neighbors`, cheap, no
|
||||
//! lookup), compare the raw encodings of whichever are set in the mask,
|
||||
//! take the smallest — see `obikindex::siblings`.
|
||||
//!
|
||||
//! Mask value 0 is logically unreachable as a real result (a slot's own base
|
||||
//! is always present in its own family) and is reused as the "not yet
|
||||
//! computed" sentinel: annex files are pre-initialised to all-zero, and a
|
||||
//! real value is only ever written once, by the computation pass.
|
||||
//!
|
||||
//! Deliberately simpler than a true 4-bit pack (1 byte/slot instead of 4
|
||||
//! bits/slot): correctness and simplicity first, for a first implementation.
|
||||
//! Packing to 4 bits/slot is a pure storage-density follow-up, not a
|
||||
//! behavioural change, left for later.
|
||||
|
||||
use std::fs::{File, OpenOptions};
|
||||
use std::io;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use memmap2::{Mmap, MmapMut};
|
||||
|
||||
const MAGIC: [u8; 4] = *b"PSIB";
|
||||
|
||||
// Header: magic(4) + _pad(4) + n(8) = 16 bytes. Data (1 byte/slot) follows.
|
||||
const HEADER_SIZE: usize = 16;
|
||||
|
||||
/// A family presence mask: bit `b` set iff the member whose own canonical
|
||||
/// central base is `b` (0=A, 1=C, 2=G, 3=T) is observed in the index.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct FamilyMask(u8);
|
||||
|
||||
impl FamilyMask {
|
||||
/// The empty mask — never a valid *computed* result (a slot's own base
|
||||
/// is always present in its own family) — used only to build up a mask
|
||||
/// via repeated [`with`](Self::with) calls before storing it.
|
||||
pub const EMPTY: FamilyMask = FamilyMask(0);
|
||||
|
||||
/// Set bit `base` (0=A, 1=C, 2=G, 3=T).
|
||||
#[inline]
|
||||
pub fn with(self, base: u8) -> Self {
|
||||
debug_assert!(base < 4, "base out of range: {base}");
|
||||
FamilyMask(self.0 | (1 << base))
|
||||
}
|
||||
|
||||
/// Is the member with central base `base` (0..3) present?
|
||||
#[inline]
|
||||
pub fn has(self, base: u8) -> bool {
|
||||
debug_assert!(base < 4, "base out of range: {base}");
|
||||
self.0 & (1 << base) != 0
|
||||
}
|
||||
|
||||
/// Number of family members observed anywhere in the index (1..=4).
|
||||
#[inline]
|
||||
pub fn family_size(self) -> u32 {
|
||||
self.0.count_ones()
|
||||
}
|
||||
|
||||
/// Number of *other* members observed (0..=3) — `family_size() - 1`.
|
||||
#[inline]
|
||||
pub fn siblings(self) -> u32 {
|
||||
self.family_size() - 1
|
||||
}
|
||||
|
||||
/// Raw bitmask (bit `b` = base `b` present) — for callers that build up
|
||||
/// a mask via their own bit operations (e.g. concurrently, via an
|
||||
/// `AtomicU8`) and only need the `FamilyMask` wrapper at the end.
|
||||
#[inline]
|
||||
pub fn bits(self) -> u8 {
|
||||
self.0
|
||||
}
|
||||
|
||||
/// Construct from a raw bitmask (only the low 4 bits are kept).
|
||||
#[inline]
|
||||
pub fn from_bits(bits: u8) -> Self {
|
||||
FamilyMask(bits & 0b1111)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn encode(self) -> u8 {
|
||||
self.0
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn decode(byte: u8) -> Option<Self> {
|
||||
if byte == 0 {
|
||||
// Unreachable for a real result — reserved as the "not yet
|
||||
// computed" sentinel.
|
||||
return None;
|
||||
}
|
||||
Some(FamilyMask(byte & 0b1111))
|
||||
}
|
||||
}
|
||||
|
||||
// ── SiblingAnnex (reader) ───────────────────────────────────────────────────
|
||||
|
||||
pub struct SiblingAnnex {
|
||||
mmap: Mmap,
|
||||
n: usize,
|
||||
path: PathBuf,
|
||||
}
|
||||
|
||||
impl SiblingAnnex {
|
||||
pub fn open(path: &Path) -> io::Result<Self> {
|
||||
let mmap = unsafe { Mmap::map(&File::open(path)?)? };
|
||||
if mmap.len() < HEADER_SIZE {
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidData, "PSIB file too short"));
|
||||
}
|
||||
if mmap[0..4] != MAGIC {
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidData, "bad PSIB magic"));
|
||||
}
|
||||
let n = u64::from_le_bytes(mmap[8..16].try_into().unwrap()) as usize;
|
||||
if mmap.len() < HEADER_SIZE + n {
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidData, "PSIB file truncated"));
|
||||
}
|
||||
Ok(Self { mmap, n, path: path.to_path_buf() })
|
||||
}
|
||||
|
||||
pub fn path(&self) -> &Path { &self.path }
|
||||
pub fn len(&self) -> usize { self.n }
|
||||
pub fn is_empty(&self) -> bool { self.n == 0 }
|
||||
|
||||
/// `None` means the slot has not (yet) been computed — see module docs.
|
||||
pub fn get(&self, slot: usize) -> Option<FamilyMask> {
|
||||
FamilyMask::decode(self.mmap[HEADER_SIZE + slot])
|
||||
}
|
||||
}
|
||||
|
||||
// ── SiblingAnnexBuilder (writer) ────────────────────────────────────────────
|
||||
|
||||
pub struct SiblingAnnexBuilder {
|
||||
mmap: MmapMut,
|
||||
n: usize,
|
||||
path: PathBuf,
|
||||
}
|
||||
|
||||
impl SiblingAnnexBuilder {
|
||||
/// Create a new annex of `n` slots at `path`, pre-initialised to the
|
||||
/// "not yet computed" sentinel (all-zero).
|
||||
pub fn new(n: usize, path: &Path) -> io::Result<Self> {
|
||||
let file_size = HEADER_SIZE + n;
|
||||
let file = OpenOptions::new()
|
||||
.read(true).write(true).create(true).truncate(true)
|
||||
.open(path)?;
|
||||
file.set_len(file_size as u64)?;
|
||||
let mut mmap = unsafe { MmapMut::map_mut(&file)? };
|
||||
mmap[0..4].copy_from_slice(&MAGIC);
|
||||
mmap[4..8].copy_from_slice(&[0u8; 4]);
|
||||
mmap[8..16].copy_from_slice(&(n as u64).to_le_bytes());
|
||||
// Data region left at 0 by `set_len`/mmap — the sentinel value.
|
||||
Ok(Self { mmap, n, path: path.to_path_buf() })
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize { self.n }
|
||||
pub fn is_empty(&self) -> bool { self.n == 0 }
|
||||
|
||||
pub fn get(&self, slot: usize) -> Option<FamilyMask> {
|
||||
FamilyMask::decode(self.mmap[HEADER_SIZE + slot])
|
||||
}
|
||||
|
||||
pub fn set(&mut self, slot: usize, mask: FamilyMask) {
|
||||
// Redundant concurrent writes from independent recomputation paths
|
||||
// converge to the same encoded byte for a given slot, so a plain
|
||||
// store here is safe even without external synchronisation, as long
|
||||
// as the byte write itself is atomic (true for a single aligned
|
||||
// byte on every platform this project targets).
|
||||
self.mmap[HEADER_SIZE + slot] = mask.encode();
|
||||
}
|
||||
|
||||
pub fn close(self) -> io::Result<()> { self.mmap.flush() }
|
||||
|
||||
pub fn finish(self) -> io::Result<SiblingAnnex> {
|
||||
let path = self.path.clone();
|
||||
self.close()?;
|
||||
SiblingAnnex::open(&path)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use tempfile::tempdir;
|
||||
|
||||
#[test]
|
||||
fn sentinel_is_zero_and_unset_slots_read_as_uncomputed() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("test.psib");
|
||||
let builder = SiblingAnnexBuilder::new(4, &path).unwrap();
|
||||
for slot in 0..4 {
|
||||
assert_eq!(builder.get(slot), None);
|
||||
}
|
||||
builder.close().unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn roundtrip_all_valid_masks() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("test.psib");
|
||||
let mut builder = SiblingAnnexBuilder::new(4, &path).unwrap();
|
||||
|
||||
let masks = [
|
||||
FamilyMask::EMPTY.with(0), // just A: family size 1
|
||||
FamilyMask::EMPTY.with(0).with(3), // A + T: size 2
|
||||
FamilyMask::EMPTY.with(1).with(2).with(3), // C+G+T: size 3
|
||||
FamilyMask::EMPTY.with(0).with(1).with(2).with(3), // all 4
|
||||
];
|
||||
for (slot, mask) in masks.iter().enumerate() {
|
||||
builder.set(slot, *mask);
|
||||
}
|
||||
let annex = builder.finish().unwrap();
|
||||
for (slot, mask) in masks.iter().enumerate() {
|
||||
assert_eq!(annex.get(slot), Some(*mask));
|
||||
}
|
||||
assert_eq!(annex.get(0).unwrap().siblings(), 0);
|
||||
assert_eq!(annex.get(1).unwrap().siblings(), 1);
|
||||
assert_eq!(annex.get(2).unwrap().siblings(), 2);
|
||||
assert_eq!(annex.get(3).unwrap().siblings(), 3);
|
||||
assert_eq!(annex.get(3).unwrap().family_size(), 4);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn has_reflects_individual_bits() {
|
||||
let mask = FamilyMask::EMPTY.with(0).with(2);
|
||||
assert!(mask.has(0));
|
||||
assert!(!mask.has(1));
|
||||
assert!(mask.has(2));
|
||||
assert!(!mask.has(3));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,111 @@
|
||||
use std::io;
|
||||
use std::path::Path;
|
||||
|
||||
use tempfile::TempDir;
|
||||
|
||||
use crate::bitvec::{PersistentBitVec, PersistentBitVecBuilder};
|
||||
use crate::views::{BitSliceIter, BitSliceView, IntSliceView};
|
||||
|
||||
// ── TempBitVec — frozen read-only, auto-deleted on drop ──────────────────────
|
||||
|
||||
pub struct TempBitVec {
|
||||
vec: PersistentBitVec,
|
||||
// Dropped after `vec` (field order), so the mmap is released before the
|
||||
// temp directory is deleted.
|
||||
_temp: TempDir,
|
||||
}
|
||||
|
||||
impl TempBitVec {
|
||||
pub fn make_persistent(&self, path: &Path) -> io::Result<PersistentBitVec> {
|
||||
std::fs::copy(self.vec.path(), path)?;
|
||||
PersistentBitVec::open(path)
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize {
|
||||
self.vec.len()
|
||||
}
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.vec.is_empty()
|
||||
}
|
||||
pub fn get(&self, slot: usize) -> bool {
|
||||
self.vec.get(slot)
|
||||
}
|
||||
pub fn count_ones(&self) -> u64 {
|
||||
self.vec.count_ones()
|
||||
}
|
||||
pub fn view(&self) -> BitSliceView<'_> {
|
||||
self.vec.view()
|
||||
}
|
||||
pub fn iter(&self) -> BitSliceIter<'_> {
|
||||
self.view().iter()
|
||||
}
|
||||
}
|
||||
|
||||
// ── TempBitVecBuilder — mutable, becomes TempBitVec on freeze ────────────────
|
||||
|
||||
pub struct TempBitVecBuilder {
|
||||
builder: PersistentBitVecBuilder,
|
||||
temp: TempDir,
|
||||
}
|
||||
|
||||
impl TempBitVecBuilder {
|
||||
pub fn new(n: usize) -> io::Result<Self> {
|
||||
let temp = TempDir::new()?;
|
||||
let path = temp.path().join("data.pbiv");
|
||||
let builder = PersistentBitVecBuilder::new(n, &path)?;
|
||||
Ok(Self { builder, temp })
|
||||
}
|
||||
|
||||
pub fn new_ones(n: usize) -> io::Result<Self> {
|
||||
let temp = TempDir::new()?;
|
||||
let path = temp.path().join("data.pbiv");
|
||||
let builder = PersistentBitVecBuilder::new_ones(n, &path)?;
|
||||
Ok(Self { builder, temp })
|
||||
}
|
||||
|
||||
pub fn freeze(self) -> io::Result<TempBitVec> {
|
||||
let Self { builder, temp } = self;
|
||||
let vec = builder.finish()?;
|
||||
Ok(TempBitVec { vec, _temp: temp })
|
||||
}
|
||||
|
||||
pub fn set(&mut self, slot: usize, value: bool) {
|
||||
self.builder.set(slot, value);
|
||||
}
|
||||
|
||||
pub fn view(&self) -> BitSliceView<'_> {
|
||||
self.builder.view()
|
||||
}
|
||||
|
||||
pub fn or(&mut self, other: BitSliceView<'_>) {
|
||||
self.builder.or(other);
|
||||
}
|
||||
|
||||
pub fn and(&mut self, other: BitSliceView<'_>) {
|
||||
self.builder.and(other);
|
||||
}
|
||||
|
||||
pub fn xor(&mut self, other: BitSliceView<'_>) {
|
||||
self.builder.xor(other);
|
||||
}
|
||||
|
||||
pub fn not(&mut self) {
|
||||
self.builder.not();
|
||||
}
|
||||
|
||||
pub fn copy_from(&mut self, src: BitSliceView<'_>) {
|
||||
self.builder.copy_from(src);
|
||||
}
|
||||
|
||||
pub fn or_where(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
self.builder.or_where(col, pred);
|
||||
}
|
||||
|
||||
pub fn and_where(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
self.builder.and_where(col, pred);
|
||||
}
|
||||
|
||||
pub fn xor_where(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
self.builder.xor_where(col, pred);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
use std::io;
|
||||
use std::path::Path;
|
||||
|
||||
use tempfile::TempDir;
|
||||
|
||||
use crate::builder::PersistentCompactIntVecBuilder;
|
||||
use crate::reader::PersistentCompactIntVec;
|
||||
use crate::views::{BitSliceView, IntSliceView};
|
||||
|
||||
// ── TempCompactIntVec — frozen read-only, auto-deleted on drop ────────────────
|
||||
|
||||
pub struct TempCompactIntVec {
|
||||
vec: PersistentCompactIntVec,
|
||||
// Dropped after `vec` (field order), so the mmap is released before the
|
||||
// temp directory is deleted.
|
||||
_temp: TempDir,
|
||||
}
|
||||
|
||||
impl TempCompactIntVec {
|
||||
pub fn make_persistent(&self, path: &Path) -> io::Result<PersistentCompactIntVec> {
|
||||
std::fs::copy(self.vec.path(), path)?;
|
||||
PersistentCompactIntVec::open(path)
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize { self.vec.len() }
|
||||
pub fn is_empty(&self) -> bool { self.vec.is_empty() }
|
||||
pub fn get(&self, slot: usize) -> u32 { self.vec.get(slot) }
|
||||
pub fn sum(&self) -> u64 { self.vec.sum() }
|
||||
pub fn view(&self) -> IntSliceView<'_> { self.vec.view() }
|
||||
pub fn iter(&self) -> crate::reader::Iter<'_> { self.vec.iter() }
|
||||
}
|
||||
|
||||
// ── TempCompactIntVecBuilder — mutable, becomes TempCompactIntVec on freeze ──
|
||||
|
||||
pub struct TempCompactIntVecBuilder {
|
||||
builder: PersistentCompactIntVecBuilder,
|
||||
temp: TempDir,
|
||||
}
|
||||
|
||||
impl TempCompactIntVecBuilder {
|
||||
pub fn new(n: usize) -> io::Result<Self> {
|
||||
let temp = TempDir::new()?;
|
||||
let path = temp.path().join("data.pciv");
|
||||
let builder = PersistentCompactIntVecBuilder::new(n, &path)?;
|
||||
Ok(Self { builder, temp })
|
||||
}
|
||||
|
||||
pub fn freeze(self) -> io::Result<TempCompactIntVec> {
|
||||
let Self { builder, temp } = self;
|
||||
let vec = builder.finish()?;
|
||||
Ok(TempCompactIntVec { vec, _temp: temp })
|
||||
}
|
||||
|
||||
pub fn n(&self) -> usize { self.builder.len() }
|
||||
|
||||
pub fn set(&mut self, slot: usize, value: u32) { self.builder.set(slot, value); }
|
||||
pub fn get(&self, slot: usize) -> u32 { self.builder.get(slot) }
|
||||
|
||||
pub fn primary_bytes(&self) -> &[u8] { self.builder.primary_bytes() }
|
||||
pub fn primary_bytes_mut(&mut self) -> &mut [u8] { self.builder.primary_bytes_mut() }
|
||||
|
||||
pub fn inc_present(&mut self, col: BitSliceView<'_>) {
|
||||
self.builder.inc_present(col);
|
||||
}
|
||||
|
||||
pub fn inc_present_fast(&mut self, col: BitSliceView<'_>) {
|
||||
self.builder.inc_present_fast(col);
|
||||
}
|
||||
|
||||
pub fn inc_predicate(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
self.builder.inc_predicate(col, pred);
|
||||
}
|
||||
|
||||
pub fn inc_predicate_fast(&mut self, col: IntSliceView<'_>, pred: impl Fn(u32) -> bool) {
|
||||
self.builder.inc_predicate_fast(col, pred);
|
||||
}
|
||||
|
||||
pub fn add(&mut self, other: IntSliceView<'_>) {
|
||||
self.builder.add(other);
|
||||
}
|
||||
|
||||
pub fn mask_with(&mut self, mask: BitSliceView<'_>) {
|
||||
self.builder.mask_with(mask);
|
||||
}
|
||||
|
||||
pub fn min(&mut self, other: IntSliceView<'_>) { self.builder.min(other); }
|
||||
pub fn max(&mut self, other: IntSliceView<'_>) { self.builder.max(other); }
|
||||
pub fn diff(&mut self, other: IntSliceView<'_>) { self.builder.diff(other); }
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
use tempfile::tempdir;
|
||||
|
||||
use crate::{PersistentBitMatrix, PersistentBitMatrixBuilder};
|
||||
use crate::{pack_bit_matrix, PersistentBitMatrix, PersistentBitMatrixBuilder};
|
||||
use crate::traits::BitPartials;
|
||||
|
||||
fn make_matrix(cols: &[&[bool]]) -> (tempfile::TempDir, PersistentBitMatrix) {
|
||||
@@ -203,3 +203,57 @@ fn partial_hamming_matches_hamming() {
|
||||
let full = m.hamming_dist_matrix();
|
||||
assert_eq!(partial, full);
|
||||
}
|
||||
|
||||
// ── col_view on Packed ────────────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn col_view_packed_values() {
|
||||
let (dir, _) = make_matrix(&[
|
||||
&[true, false, true, true],
|
||||
&[false, true, false, true],
|
||||
]);
|
||||
pack_bit_matrix(&dir.path().join("presence")).unwrap();
|
||||
let m = PersistentBitMatrix::open(dir.path()).unwrap();
|
||||
|
||||
// col 0: [T, F, T, T]
|
||||
let v0 = m.col_view(0);
|
||||
assert_eq!(v0.len(), 4);
|
||||
assert_eq!(v0.get(0), true);
|
||||
assert_eq!(v0.get(1), false);
|
||||
assert_eq!(v0.get(2), true);
|
||||
assert_eq!(v0.get(3), true);
|
||||
assert_eq!(v0.count_ones(), 3);
|
||||
|
||||
// col 1: [F, T, F, T]
|
||||
let v1 = m.col_view(1);
|
||||
assert_eq!(v1.get(0), false);
|
||||
assert_eq!(v1.get(1), true);
|
||||
assert_eq!(v1.get(2), false);
|
||||
assert_eq!(v1.get(3), true);
|
||||
assert_eq!(v1.count_ones(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn col_view_packed_matches_columnar() {
|
||||
let data: &[&[bool]] = &[
|
||||
&[true, false, true, false, true, true, false, true],
|
||||
&[false, false, true, true, false, true, true, false],
|
||||
&[true, true, true, false, false, false, true, true],
|
||||
];
|
||||
let (dir_col, m_col) = make_matrix(data);
|
||||
let (dir_pack, _) = make_matrix(data);
|
||||
pack_bit_matrix(&dir_pack.path().join("presence")).unwrap();
|
||||
let m_pack = PersistentBitMatrix::open(dir_pack.path()).unwrap();
|
||||
|
||||
for c in 0..data.len() {
|
||||
let col_ref = m_col.col(c);
|
||||
let col_view = m_pack.col_view(c);
|
||||
assert_eq!(col_view.len(), col_ref.len(), "col={c} len");
|
||||
for s in 0..col_ref.len() {
|
||||
assert_eq!(col_view.get(s), col_ref.get(s), "col={c} slot={s}");
|
||||
}
|
||||
assert_eq!(col_view.count_ones(), col_ref.count_ones(), "col={c} count_ones");
|
||||
assert_eq!(col_view.words(), col_ref.words(), "col={c} words");
|
||||
}
|
||||
drop(dir_col);
|
||||
}
|
||||
|
||||
@@ -77,7 +77,7 @@ fn op_and() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("out.pbiv");
|
||||
let mut b = PersistentBitVecBuilder::build_from(&ra, &path).unwrap();
|
||||
b.and(&rb);
|
||||
b.and(rb.view());
|
||||
b.close().unwrap();
|
||||
let r = PersistentBitVec::open(&path).unwrap();
|
||||
assert_eq!(r.iter().collect::<Vec<_>>(), vec![true, false, false, false]);
|
||||
@@ -90,7 +90,7 @@ fn op_or() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("out.pbiv");
|
||||
let mut b = PersistentBitVecBuilder::build_from(&ra, &path).unwrap();
|
||||
b.or(&rb);
|
||||
b.or(rb.view());
|
||||
b.close().unwrap();
|
||||
let r = PersistentBitVec::open(&path).unwrap();
|
||||
assert_eq!(r.iter().collect::<Vec<_>>(), vec![true, true, true, false]);
|
||||
@@ -103,7 +103,7 @@ fn op_xor() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("out.pbiv");
|
||||
let mut b = PersistentBitVecBuilder::build_from(&ra, &path).unwrap();
|
||||
b.xor(&rb);
|
||||
b.xor(rb.view());
|
||||
b.close().unwrap();
|
||||
let r = PersistentBitVec::open(&path).unwrap();
|
||||
assert_eq!(r.iter().collect::<Vec<_>>(), vec![false, true, true, false]);
|
||||
|
||||
@@ -0,0 +1,223 @@
|
||||
use tempfile::tempdir;
|
||||
|
||||
use crate::{
|
||||
ColGroup, MatrixGroupOps,
|
||||
PersistentBitMatrix, PersistentBitMatrixBuilder,
|
||||
PersistentCompactIntMatrix, PersistentCompactIntMatrixBuilder,
|
||||
};
|
||||
use crate::{PersistentBitVecBuilder, PersistentCompactIntVec, PersistentCompactIntVecBuilder};
|
||||
|
||||
// ── helpers ───────────────────────────────────────────────────────────────────
|
||||
|
||||
fn make_int_matrix(cols: &[&[u32]]) -> (tempfile::TempDir, PersistentCompactIntMatrix) {
|
||||
let n = cols.first().map_or(0, |c| c.len());
|
||||
let dir = tempdir().unwrap();
|
||||
let mut b = PersistentCompactIntMatrixBuilder::new(n, &dir.path().join("counts")).unwrap();
|
||||
for &col in cols {
|
||||
let mut cb = b.add_col().unwrap();
|
||||
for (slot, &v) in col.iter().enumerate() { cb.set(slot, v); }
|
||||
cb.close().unwrap();
|
||||
}
|
||||
b.close().unwrap();
|
||||
let m = PersistentCompactIntMatrix::open(dir.path()).unwrap();
|
||||
(dir, m)
|
||||
}
|
||||
|
||||
fn make_bit_matrix(cols: &[&[bool]]) -> (tempfile::TempDir, PersistentBitMatrix) {
|
||||
let n = cols.first().map_or(0, |c| c.len());
|
||||
let dir = tempdir().unwrap();
|
||||
let presence = dir.path().join("presence");
|
||||
let mut b = PersistentBitMatrixBuilder::new(n, &presence).unwrap();
|
||||
for &col in cols {
|
||||
let mut cb = b.add_col().unwrap();
|
||||
for (slot, &v) in col.iter().enumerate() { cb.set(slot, v); }
|
||||
cb.close().unwrap();
|
||||
}
|
||||
b.close().unwrap();
|
||||
let m = PersistentBitMatrix::open(dir.path()).unwrap();
|
||||
(dir, m)
|
||||
}
|
||||
|
||||
// ── IntMatrix: partial_group_sum ──────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn int_partial_group_sum_basic() {
|
||||
// col0=[1,2,3], col1=[10,20,30], col2=[100,0,5]
|
||||
// group {0,2}: sum = [101, 2, 8]
|
||||
let (_d, m) = make_int_matrix(&[&[1, 2, 3], &[10, 20, 30], &[100, 0, 5]]);
|
||||
let g = ColGroup::new("g", vec![0, 2]);
|
||||
let result = m.partial_group_sum(&g).unwrap();
|
||||
assert_eq!(result.get(0), 101);
|
||||
assert_eq!(result.get(1), 2);
|
||||
assert_eq!(result.get(2), 8);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn int_partial_group_sum_with_overflow() {
|
||||
// col0=[300,0], col1=[200,400]: group {0,1}: sum=[500, 400]
|
||||
let (_d, m) = make_int_matrix(&[&[300, 0], &[200, 400]]);
|
||||
let g = ColGroup::new("g", vec![0, 1]);
|
||||
let result = m.partial_group_sum(&g).unwrap();
|
||||
assert_eq!(result.get(0), 500);
|
||||
assert_eq!(result.get(1), 400);
|
||||
assert_eq!(result.sum(), 900);
|
||||
}
|
||||
|
||||
// ── IntMatrix: partial_group_presence_count ───────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn int_partial_group_presence_count() {
|
||||
// col0=[5,1,0,3], col1=[2,0,4,3], col2=[0,3,1,0]
|
||||
// threshold=2: col0: [T,F,F,T], col1: [T,F,T,T], col2: [F,T,F,F]
|
||||
// group {0,1,2}: counts = [2, 1, 1, 2]
|
||||
let (_d, m) = make_int_matrix(&[&[5, 1, 0, 3], &[2, 0, 4, 3], &[0, 3, 1, 0]]);
|
||||
let g = ColGroup::new("g", vec![0, 1, 2]);
|
||||
let result = m.partial_group_presence_count(&g, 2).unwrap();
|
||||
assert_eq!(result.get(0), 2);
|
||||
assert_eq!(result.get(1), 1);
|
||||
assert_eq!(result.get(2), 1);
|
||||
assert_eq!(result.get(3), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn int_partial_group_presence_count_with_overflow() {
|
||||
// col0=[300,0,10], col1=[0,400,10], col2=[1,1,10]
|
||||
// threshold=5: col0: [T,F,T], col1: [F,T,T], col2: [F,F,T]
|
||||
// group {0,1,2}: counts = [1, 1, 3]
|
||||
let (_d, m) = make_int_matrix(&[&[300, 0, 10], &[0, 400, 10], &[1, 1, 10]]);
|
||||
let g = ColGroup::new("g", vec![0, 1, 2]);
|
||||
let result = m.partial_group_presence_count(&g, 5).unwrap();
|
||||
assert_eq!(result.get(0), 1);
|
||||
assert_eq!(result.get(1), 1);
|
||||
assert_eq!(result.get(2), 3);
|
||||
}
|
||||
|
||||
// ── IntMatrix: partial_group_any ──────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn int_partial_group_any() {
|
||||
// col0=[0,3,0,1], col1=[2,0,0,0], col2=[0,0,5,0]
|
||||
// threshold=2: col0: [F,T,F,F], col1: [T,F,F,F], col2: [F,F,T,F]
|
||||
// group {0,1,2}: any = [T, T, T, F]
|
||||
let (_d, m) = make_int_matrix(&[&[0, 3, 0, 1], &[2, 0, 0, 0], &[0, 0, 5, 0]]);
|
||||
let g = ColGroup::new("g", vec![0, 1, 2]);
|
||||
let result = m.partial_group_any(&g, 2).unwrap();
|
||||
assert_eq!(result.get(0), true);
|
||||
assert_eq!(result.get(1), true);
|
||||
assert_eq!(result.get(2), true);
|
||||
assert_eq!(result.get(3), false);
|
||||
}
|
||||
|
||||
// ── IntMatrix: mask_with ──────────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn mask_with_zeros_selected_slots() {
|
||||
// count vec [10, 20, 30, 40], mask [T, F, T, F] → [10, 0, 30, 0]
|
||||
let dir = tempdir().unwrap();
|
||||
let mut v = PersistentCompactIntVecBuilder::new(4, &dir.path().join("v.pciv")).unwrap();
|
||||
v.set(0, 10); v.set(1, 20); v.set(2, 30); v.set(3, 40);
|
||||
let mut mask = PersistentBitVecBuilder::new(4, &dir.path().join("m.pbiv")).unwrap();
|
||||
mask.set(0, true); mask.set(2, true);
|
||||
v.mask_with(mask.view());
|
||||
v.close().unwrap();
|
||||
let r = PersistentCompactIntVec::open(&dir.path().join("v.pciv")).unwrap();
|
||||
assert_eq!(r.get(0), 10);
|
||||
assert_eq!(r.get(1), 0);
|
||||
assert_eq!(r.get(2), 30);
|
||||
assert_eq!(r.get(3), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mask_with_overflow_slot_zeroed() {
|
||||
// overflow slot (value 500) masked out → removed from overflow, primary=0
|
||||
let dir = tempdir().unwrap();
|
||||
let mut v = PersistentCompactIntVecBuilder::new(3, &dir.path().join("v.pciv")).unwrap();
|
||||
v.set(0, 10); v.set(1, 500); v.set(2, 5);
|
||||
let mut mask = PersistentBitVecBuilder::new(3, &dir.path().join("m.pbiv")).unwrap();
|
||||
mask.set(0, true); mask.set(2, true); // slot 1 masked out
|
||||
v.mask_with(mask.view());
|
||||
v.close().unwrap();
|
||||
let r = PersistentCompactIntVec::open(&dir.path().join("v.pciv")).unwrap();
|
||||
assert_eq!(r.get(0), 10);
|
||||
assert_eq!(r.get(1), 0);
|
||||
assert_eq!(r.get(2), 5);
|
||||
let ov: Vec<_> = r.view().overflow_entries().collect();
|
||||
assert!(ov.is_empty(), "overflow entry for masked-out slot should be gone");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mask_with_all_ones_is_noop() {
|
||||
let dir = tempdir().unwrap();
|
||||
let mut v = PersistentCompactIntVecBuilder::new(4, &dir.path().join("v.pciv")).unwrap();
|
||||
v.set(0, 300); v.set(1, 1); v.set(2, 0); v.set(3, 42);
|
||||
let mask = PersistentBitVecBuilder::new_ones(4, &dir.path().join("m.pbiv")).unwrap();
|
||||
v.mask_with(mask.view());
|
||||
v.close().unwrap();
|
||||
let r = PersistentCompactIntVec::open(&dir.path().join("v.pciv")).unwrap();
|
||||
assert_eq!(r.get(0), 300);
|
||||
assert_eq!(r.get(1), 1);
|
||||
assert_eq!(r.get(2), 0);
|
||||
assert_eq!(r.get(3), 42);
|
||||
}
|
||||
|
||||
// ── BitMatrix: partial_group_presence_count ───────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn bit_partial_group_presence_count() {
|
||||
// col0=[T,F,T,F], col1=[T,T,F,F], col2=[F,T,T,F]
|
||||
// group {0,1,2}: counts = [2, 2, 2, 0]
|
||||
let (_d, m) = make_bit_matrix(&[
|
||||
&[true, false, true, false],
|
||||
&[true, true, false, false],
|
||||
&[false,true, true, false],
|
||||
]);
|
||||
let g = ColGroup::new("g", vec![0, 1, 2]);
|
||||
let result = m.partial_group_presence_count(&g, 1).unwrap();
|
||||
assert_eq!(result.get(0), 2);
|
||||
assert_eq!(result.get(1), 2);
|
||||
assert_eq!(result.get(2), 2);
|
||||
assert_eq!(result.get(3), 0);
|
||||
}
|
||||
|
||||
// ── BitMatrix: partial_group_any ──────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn bit_partial_group_any() {
|
||||
// col0=[T,F,F], col1=[F,F,T], group {0,1}: any = [T, F, T]
|
||||
let (_d, m) = make_bit_matrix(&[
|
||||
&[true, false, false],
|
||||
&[false, false, true],
|
||||
]);
|
||||
let g = ColGroup::new("g", vec![0, 1]);
|
||||
let result = m.partial_group_any(&g, 1).unwrap();
|
||||
assert_eq!(result.get(0), true);
|
||||
assert_eq!(result.get(1), false);
|
||||
assert_eq!(result.get(2), true);
|
||||
}
|
||||
|
||||
// ── Composition: partial results are additive ─────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn int_presence_count_additive_across_split() {
|
||||
// Simulate two partitions (different kmer ranges) whose counts should add.
|
||||
// Global data for col0: [5,1,0,3,2], col1: [2,0,4,3,1] — threshold=2
|
||||
// Split: partition A = slots 0..2, partition B = slots 2..5
|
||||
let data_a: &[&[u32]] = &[&[5, 1], &[2, 0]];
|
||||
let data_b: &[&[u32]] = &[&[0, 3, 2], &[4, 3, 1]];
|
||||
let (_da, ma) = make_int_matrix(data_a);
|
||||
let (_db, mb) = make_int_matrix(data_b);
|
||||
let g = ColGroup::new("g", vec![0, 1]);
|
||||
|
||||
let pa = ma.partial_group_presence_count(&g, 2).unwrap();
|
||||
let pb = mb.partial_group_presence_count(&g, 2).unwrap();
|
||||
|
||||
// Concatenate by adding (disjoint kmer ranges — here we just verify
|
||||
// individual results match the expected per-partition counts).
|
||||
// partition A: col0=[5≥2,1<2]=[T,F], col1=[2≥2,0<2]=[T,F] → [2, 0]
|
||||
assert_eq!(pa.get(0), 2);
|
||||
assert_eq!(pa.get(1), 0);
|
||||
// partition B: col0=[0<2,3≥2,2≥2]=[F,T,T], col1=[4≥2,3≥2,1<2]=[T,T,F] → [1, 2, 1]
|
||||
assert_eq!(pb.get(0), 1);
|
||||
assert_eq!(pb.get(1), 2);
|
||||
assert_eq!(pb.get(2), 1);
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
use tempfile::tempdir;
|
||||
|
||||
use crate::{PersistentCompactIntMatrix, PersistentCompactIntMatrixBuilder};
|
||||
use crate::{pack_compact_int_matrix, PersistentCompactIntMatrix, PersistentCompactIntMatrixBuilder};
|
||||
use crate::traits::CountPartials;
|
||||
|
||||
fn make_matrix(cols: &[&[u32]]) -> (tempfile::TempDir, PersistentCompactIntMatrix) {
|
||||
@@ -243,6 +243,61 @@ fn partial_hellinger_matches_full() {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn col_view_packed_values() {
|
||||
// Build Columnar with overflow values (≥ 255), pack, reopen as Packed, exercise col_view().
|
||||
let (dir, _col) = make_matrix(&[&[10, 300, 500], &[200, 50, 1000]]);
|
||||
pack_compact_int_matrix(&dir.path().join("counts")).unwrap();
|
||||
let m = PersistentCompactIntMatrix::open(dir.path()).unwrap();
|
||||
|
||||
// col 0: [10, 300, 500] — two overflow slots
|
||||
let v0 = m.col_view(0);
|
||||
assert_eq!(v0.get(0), 10);
|
||||
assert_eq!(v0.get(1), 300);
|
||||
assert_eq!(v0.get(2), 500);
|
||||
assert_eq!(v0.sum(), 810);
|
||||
assert_eq!(v0.count_nonzero(), 3);
|
||||
let mut ov0: Vec<(usize, u32)> = v0.overflow_entries().collect();
|
||||
ov0.sort_unstable_by_key(|&(s, _)| s);
|
||||
assert_eq!(ov0, vec![(1, 300), (2, 500)]);
|
||||
|
||||
// col 1: [200, 50, 1000] — one overflow slot
|
||||
let v1 = m.col_view(1);
|
||||
assert_eq!(v1.get(0), 200);
|
||||
assert_eq!(v1.get(1), 50);
|
||||
assert_eq!(v1.get(2), 1000);
|
||||
let mut ov1: Vec<(usize, u32)> = v1.overflow_entries().collect();
|
||||
ov1.sort_unstable_by_key(|&(s, _)| s);
|
||||
assert_eq!(ov1, vec![(2, 1000)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn col_view_packed_matches_columnar() {
|
||||
// Same data, compare col_view() on Packed against col() on Columnar slot-by-slot.
|
||||
let data: &[&[u32]] = &[&[0, 255, 1, 300, 128], &[500, 3, 0, 700, 42]];
|
||||
let (dir_col, m_col) = make_matrix(data);
|
||||
// Re-build in a separate dir so we can pack without touching m_col's files.
|
||||
let (dir_pack, _) = make_matrix(data);
|
||||
pack_compact_int_matrix(&dir_pack.path().join("counts")).unwrap();
|
||||
let m_pack = PersistentCompactIntMatrix::open(dir_pack.path()).unwrap();
|
||||
|
||||
for c in 0..data.len() {
|
||||
let col_ref = m_col.col(c);
|
||||
let col_view = m_pack.col_view(c);
|
||||
assert_eq!(col_view.len(), col_ref.len());
|
||||
for s in 0..col_ref.len() {
|
||||
assert_eq!(col_view.get(s), col_ref.get(s), "col={c} slot={s}");
|
||||
}
|
||||
assert_eq!(col_view.sum(), col_ref.sum(), "col={c} sum");
|
||||
let mut ov_view: Vec<(usize, u32)> = col_view.overflow_entries().collect();
|
||||
let mut ov_ref: Vec<(usize, u32)> = col_ref.view().overflow_entries().collect();
|
||||
ov_view.sort_unstable_by_key(|&(s, _)| s);
|
||||
ov_ref.sort_unstable_by_key(|&(s, _)| s);
|
||||
assert_eq!(ov_view, ov_ref, "col={c} overflow_entries");
|
||||
}
|
||||
drop(dir_col);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn partial_relfreq_bray_additive_across_split() {
|
||||
// Split rows [1,2,3,4,5] between two matrices; partial sums should add up.
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
mod bitmatrix;
|
||||
mod bitvec;
|
||||
mod colgroup;
|
||||
mod intmatrix;
|
||||
|
||||
use tempfile::tempdir;
|
||||
@@ -169,7 +170,7 @@ fn combine_min() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("out.pciv");
|
||||
let mut b = PersistentCompactIntVecBuilder::build_from(&ra, &path).unwrap();
|
||||
b.min(&rb);
|
||||
b.min(rb.view());
|
||||
b.close().unwrap();
|
||||
let r = PersistentCompactIntVec::open(&path).unwrap();
|
||||
assert_eq!(r.iter().collect::<Vec<_>>(), vec![10, 100, 0, 800]);
|
||||
@@ -182,7 +183,7 @@ fn combine_max() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("out.pciv");
|
||||
let mut b = PersistentCompactIntVecBuilder::build_from(&ra, &path).unwrap();
|
||||
b.max(&rb);
|
||||
b.max(rb.view());
|
||||
b.close().unwrap();
|
||||
let r = PersistentCompactIntVec::open(&path).unwrap();
|
||||
assert_eq!(r.iter().collect::<Vec<_>>(), vec![20, 300, 500, 1000]);
|
||||
@@ -195,7 +196,7 @@ fn combine_add() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("out.pciv");
|
||||
let mut b = PersistentCompactIntVecBuilder::build_from(&ra, &path).unwrap();
|
||||
b.add(&rb);
|
||||
b.add(rb.view());
|
||||
b.close().unwrap();
|
||||
let r = PersistentCompactIntVec::open(&path).unwrap();
|
||||
assert_eq!(r.iter().collect::<Vec<_>>(), vec![30, 300, 5, 101]);
|
||||
@@ -220,7 +221,7 @@ fn combine_diff() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("out.pciv");
|
||||
let mut b = PersistentCompactIntVecBuilder::build_from(&ra, &path).unwrap();
|
||||
b.diff(&rb);
|
||||
b.diff(rb.view());
|
||||
b.close().unwrap();
|
||||
let r = PersistentCompactIntVec::open(&path).unwrap();
|
||||
assert_eq!(r.iter().collect::<Vec<_>>(), vec![10, 700, 0, 0]);
|
||||
|
||||
@@ -1,6 +1,17 @@
|
||||
use ndarray::{Array1, Array2};
|
||||
|
||||
/// Column-level weight statistic — total count or presence count per column.
|
||||
/// Convert a Jaccard distance matrix (`1 - J`) into a Mash distance matrix, per
|
||||
/// https://mash.readthedocs.io/en/latest/distances.html:
|
||||
/// `D = -1/k * ln(2J / (1+J))`.
|
||||
fn jaccard_to_mash(d_jaccard: &Array2<f64>, k: usize) -> Array2<f64> {
|
||||
d_jaccard.mapv(|d| {
|
||||
let j = 1.0 - d;
|
||||
if j <= 0.0 { 1.0 }
|
||||
else { -1.0 / k as f64 * (2.0 * j / (1.0 + j)).ln() }
|
||||
})
|
||||
}
|
||||
|
||||
// ── Column-level weight statistic — total count or presence count per column.
|
||||
/// Additive across layers and partitions; used as denominator in normalised distances.
|
||||
///
|
||||
/// `partial_kmer_counts` returns the number of **distinct k-mers** present per
|
||||
@@ -74,6 +85,12 @@ pub trait CountPartials: ColumnWeights {
|
||||
m
|
||||
}
|
||||
|
||||
/// Mash distance (https://mash.readthedocs.io/en/latest/distances.html), derived
|
||||
/// from the presence-threshold Jaccard distance.
|
||||
fn threshold_mash_dist_matrix(&self, k: usize, threshold: u32) -> Array2<f64> {
|
||||
jaccard_to_mash(&self.threshold_jaccard_dist_matrix(threshold), k)
|
||||
}
|
||||
|
||||
fn relfreq_bray_dist_matrix(&self) -> Array2<f64> {
|
||||
let global = self.col_weights();
|
||||
let mut m = self.partial_relfreq_bray(&global).mapv(|v| 1.0 - v);
|
||||
@@ -126,6 +143,12 @@ pub trait BitPartials: ColumnWeights {
|
||||
m
|
||||
}
|
||||
|
||||
/// Mash distance (https://mash.readthedocs.io/en/latest/distances.html), derived
|
||||
/// from the Jaccard distance.
|
||||
fn mash_dist_matrix(&self, k: usize) -> Array2<f64> {
|
||||
jaccard_to_mash(&self.jaccard_dist_matrix(), k)
|
||||
}
|
||||
|
||||
fn hamming_dist_matrix(&self) -> Array2<u64> {
|
||||
self.partial_hamming()
|
||||
}
|
||||
|
||||
@@ -0,0 +1,278 @@
|
||||
use crate::format::{byte_count_nonzero, byte_sum, parse_overflow_entry};
|
||||
|
||||
// ── BitSliceView ──────────────────────────────────────────────────────────────
|
||||
|
||||
/// Lightweight, copy-able read-only view over a u64 word array.
|
||||
/// Bit `i` is in `words[i >> 6]` at position `i & 63`. Padding bits are zero.
|
||||
#[derive(Clone, Copy)]
|
||||
pub struct BitSliceView<'a> {
|
||||
pub(crate) words: &'a [u64],
|
||||
pub(crate) n: usize,
|
||||
}
|
||||
|
||||
impl<'a> BitSliceView<'a> {
|
||||
#[inline]
|
||||
pub fn new(words: &'a [u64], n: usize) -> Self { Self { words, n } }
|
||||
|
||||
pub fn len(&self) -> usize { self.n }
|
||||
pub fn is_empty(&self) -> bool { self.n == 0 }
|
||||
pub fn words(&self) -> &'a [u64] { self.words }
|
||||
|
||||
#[inline]
|
||||
pub fn get(&self, slot: usize) -> bool {
|
||||
(self.words[slot >> 6] >> (slot & 63)) & 1 != 0
|
||||
}
|
||||
|
||||
pub fn count_ones(&self) -> u64 {
|
||||
self.words.iter().map(|w| w.count_ones() as u64).sum()
|
||||
}
|
||||
pub fn count_zeros(&self) -> u64 { self.n as u64 - self.count_ones() }
|
||||
|
||||
pub fn iter(&self) -> BitSliceIter<'a> {
|
||||
BitSliceIter { words: self.words, slot: 0, n: self.n }
|
||||
}
|
||||
|
||||
pub fn partial_jaccard_dist(self, other: BitSliceView<'_>) -> (u64, u64) {
|
||||
assert_eq!(self.n, other.n, "BitSliceView length mismatch");
|
||||
self.words.iter().zip(other.words)
|
||||
.fold((0u64, 0u64), |(i, u), (&a, &b)| {
|
||||
(i + (a & b).count_ones() as u64, u + (a | b).count_ones() as u64)
|
||||
})
|
||||
}
|
||||
|
||||
pub fn jaccard_dist(self, other: BitSliceView<'_>) -> f64 {
|
||||
let (inter, union) = self.partial_jaccard_dist(other);
|
||||
if union == 0 { 0.0 } else { 1.0 - inter as f64 / union as f64 }
|
||||
}
|
||||
|
||||
pub fn hamming_dist(self, other: BitSliceView<'_>) -> u64 {
|
||||
assert_eq!(self.n, other.n, "BitSliceView length mismatch");
|
||||
self.words.iter().zip(other.words)
|
||||
.map(|(&a, &b)| (a ^ b).count_ones() as u64)
|
||||
.sum()
|
||||
}
|
||||
}
|
||||
|
||||
// ── BitSliceIter ──────────────────────────────────────────────────────────────
|
||||
|
||||
pub struct BitSliceIter<'a> {
|
||||
words: &'a [u64],
|
||||
slot: usize,
|
||||
n: usize,
|
||||
}
|
||||
|
||||
impl Iterator for BitSliceIter<'_> {
|
||||
type Item = bool;
|
||||
fn next(&mut self) -> Option<bool> {
|
||||
if self.slot >= self.n { return None; }
|
||||
let v = (self.words[self.slot >> 6] >> (self.slot & 63)) & 1 != 0;
|
||||
self.slot += 1;
|
||||
Some(v)
|
||||
}
|
||||
fn size_hint(&self) -> (usize, Option<usize>) {
|
||||
let rem = self.n - self.slot;
|
||||
(rem, Some(rem))
|
||||
}
|
||||
}
|
||||
impl ExactSizeIterator for BitSliceIter<'_> {}
|
||||
|
||||
// ── IntSliceView ──────────────────────────────────────────────────────────────
|
||||
|
||||
/// Lightweight, copy-able read-only view over a compact-int primary array plus
|
||||
/// its sorted raw overflow bytes. Zero-copy: all data lives in the caller's mmap.
|
||||
#[derive(Clone, Copy)]
|
||||
pub struct IntSliceView<'a> {
|
||||
pub(crate) primary: &'a [u8],
|
||||
pub(crate) overflow_raw: &'a [u8], // n_overflow × OVERFLOW_ENTRY_SIZE bytes, sorted by slot
|
||||
pub(crate) n_overflow: usize,
|
||||
pub(crate) n: usize,
|
||||
}
|
||||
|
||||
impl<'a> IntSliceView<'a> {
|
||||
#[inline]
|
||||
pub fn new(primary: &'a [u8], overflow_raw: &'a [u8], n_overflow: usize, n: usize) -> Self {
|
||||
Self { primary, overflow_raw, n_overflow, n }
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize { self.n }
|
||||
pub fn is_empty(&self) -> bool { self.n == 0 }
|
||||
pub fn primary_bytes(&self) -> &'a [u8] { self.primary }
|
||||
pub fn n_overflow(&self) -> usize { self.n_overflow }
|
||||
|
||||
pub fn overflow_entries(&self) -> impl Iterator<Item = (usize, u32)> + 'a {
|
||||
let raw = self.overflow_raw;
|
||||
let n_ov = self.n_overflow;
|
||||
(0..n_ov).map(move |i| parse_overflow_entry(raw, 0, i))
|
||||
}
|
||||
|
||||
/// O(log n_overflow) via binary search (overflow is always sorted by slot).
|
||||
pub fn get(&self, slot: usize) -> u32 {
|
||||
let b = self.primary[slot];
|
||||
if b < 255 { return b as u32; }
|
||||
let mut lo = 0usize;
|
||||
let mut hi = self.n_overflow;
|
||||
while lo < hi {
|
||||
let mid = lo + (hi - lo) / 2;
|
||||
let (s, v) = parse_overflow_entry(self.overflow_raw, 0, mid);
|
||||
match s.cmp(&slot) {
|
||||
std::cmp::Ordering::Equal => return v,
|
||||
std::cmp::Ordering::Less => lo = mid + 1,
|
||||
std::cmp::Ordering::Greater => hi = mid,
|
||||
}
|
||||
}
|
||||
panic!("slot {slot} marked overflow but not found")
|
||||
}
|
||||
|
||||
/// Sequential merge scan: yields all n values in slot order.
|
||||
pub fn iter(&self) -> IntSliceViewIter<'a> {
|
||||
IntSliceViewIter {
|
||||
primary: self.primary,
|
||||
overflow_raw: self.overflow_raw,
|
||||
slot: 0,
|
||||
overflow_pos: 0,
|
||||
n: self.n,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn sum(&self) -> u64 {
|
||||
byte_sum(self.primary, self.overflow_entries().map(|(_, v)| v))
|
||||
}
|
||||
|
||||
pub fn count_nonzero(&self) -> u64 {
|
||||
byte_count_nonzero(self.primary)
|
||||
}
|
||||
|
||||
// ── Distance methods ──────────────────────────────────────────────────────
|
||||
|
||||
pub fn partial_bray_dist(self, other: IntSliceView<'_>) -> u64 {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
self.iter().zip(other.iter()).map(|(a, b)| a.min(b) as u64).sum()
|
||||
}
|
||||
|
||||
pub fn bray_dist(self, other: IntSliceView<'_>) -> f64 {
|
||||
let sum_min = self.partial_bray_dist(other);
|
||||
let denom = self.sum() + other.sum();
|
||||
if denom == 0 { 0.0 } else { 1.0 - 2.0 * sum_min as f64 / denom as f64 }
|
||||
}
|
||||
|
||||
pub fn partial_relfreq_bray_dist(self, other: IntSliceView<'_>, sa: f64, sb: f64) -> f64 {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
self.iter().zip(other.iter())
|
||||
.map(|(a, b)| {
|
||||
let pa = if sa > 0.0 { a as f64 / sa } else { 0.0 };
|
||||
let pb = if sb > 0.0 { b as f64 / sb } else { 0.0 };
|
||||
pa.min(pb)
|
||||
})
|
||||
.sum()
|
||||
}
|
||||
|
||||
pub fn relfreq_bray_dist(self, other: IntSliceView<'_>) -> f64 {
|
||||
let sa = self.sum() as f64;
|
||||
let sb = other.sum() as f64;
|
||||
if sa == 0.0 && sb == 0.0 { return 0.0; }
|
||||
1.0 - self.partial_relfreq_bray_dist(other, sa, sb)
|
||||
}
|
||||
|
||||
pub fn partial_euclidean_dist(self, other: IntSliceView<'_>) -> f64 {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
self.iter().zip(other.iter())
|
||||
.map(|(a, b)| { let d = a as f64 - b as f64; d * d })
|
||||
.sum()
|
||||
}
|
||||
|
||||
pub fn euclidean_dist(self, other: IntSliceView<'_>) -> f64 {
|
||||
self.partial_euclidean_dist(other).sqrt()
|
||||
}
|
||||
|
||||
pub fn partial_relfreq_euclidean_dist(self, other: IntSliceView<'_>, sa: f64, sb: f64) -> f64 {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
self.iter().zip(other.iter())
|
||||
.map(|(a, b)| {
|
||||
let pa = if sa > 0.0 { a as f64 / sa } else { 0.0 };
|
||||
let pb = if sb > 0.0 { b as f64 / sb } else { 0.0 };
|
||||
let d = pa - pb;
|
||||
d * d
|
||||
})
|
||||
.sum()
|
||||
}
|
||||
|
||||
pub fn relfreq_euclidean_dist(self, other: IntSliceView<'_>) -> f64 {
|
||||
let sa = self.sum() as f64;
|
||||
let sb = other.sum() as f64;
|
||||
if sa == 0.0 && sb == 0.0 { return 0.0; }
|
||||
self.partial_relfreq_euclidean_dist(other, sa, sb).sqrt()
|
||||
}
|
||||
|
||||
pub fn partial_hellinger_euclidean_dist(self, other: IntSliceView<'_>, sa: f64, sb: f64) -> f64 {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
self.iter().zip(other.iter())
|
||||
.map(|(a, b)| {
|
||||
let pa = if sa > 0.0 { (a as f64 / sa).sqrt() } else { 0.0 };
|
||||
let pb = if sb > 0.0 { (b as f64 / sb).sqrt() } else { 0.0 };
|
||||
let d = pa - pb;
|
||||
d * d
|
||||
})
|
||||
.sum()
|
||||
}
|
||||
|
||||
pub fn hellinger_euclidean_dist(self, other: IntSliceView<'_>) -> f64 {
|
||||
let sa = self.sum() as f64;
|
||||
let sb = other.sum() as f64;
|
||||
if sa == 0.0 && sb == 0.0 { return 0.0; }
|
||||
self.partial_hellinger_euclidean_dist(other, sa, sb).sqrt()
|
||||
}
|
||||
|
||||
pub fn hellinger_dist(self, other: IntSliceView<'_>) -> f64 {
|
||||
self.hellinger_euclidean_dist(other) / std::f64::consts::SQRT_2
|
||||
}
|
||||
|
||||
pub fn partial_threshold_jaccard_dist(self, other: IntSliceView<'_>, threshold: u32) -> (u64, u64) {
|
||||
assert_eq!(self.n, other.n, "length mismatch");
|
||||
self.iter().zip(other.iter())
|
||||
.fold((0u64, 0u64), |(inter, uni), (a, b)| {
|
||||
let ap = a >= threshold;
|
||||
let bp = b >= threshold;
|
||||
(inter + (ap & bp) as u64, uni + (ap | bp) as u64)
|
||||
})
|
||||
}
|
||||
|
||||
pub fn threshold_jaccard_dist(self, other: IntSliceView<'_>, threshold: u32) -> f64 {
|
||||
let (inter, union) = self.partial_threshold_jaccard_dist(other, threshold);
|
||||
if union == 0 { 0.0 } else { 1.0 - inter as f64 / union as f64 }
|
||||
}
|
||||
|
||||
pub fn jaccard_dist(self, other: IntSliceView<'_>) -> f64 {
|
||||
self.threshold_jaccard_dist(other, 1)
|
||||
}
|
||||
}
|
||||
|
||||
// ── IntSliceViewIter ──────────────────────────────────────────────────────────
|
||||
|
||||
pub struct IntSliceViewIter<'a> {
|
||||
primary: &'a [u8],
|
||||
overflow_raw: &'a [u8],
|
||||
slot: usize,
|
||||
overflow_pos: usize,
|
||||
n: usize,
|
||||
}
|
||||
|
||||
impl Iterator for IntSliceViewIter<'_> {
|
||||
type Item = u32;
|
||||
fn next(&mut self) -> Option<u32> {
|
||||
if self.slot >= self.n { return None; }
|
||||
let v = self.primary[self.slot];
|
||||
self.slot += 1;
|
||||
if v < 255 {
|
||||
Some(v as u32)
|
||||
} else {
|
||||
let (_, val) = parse_overflow_entry(self.overflow_raw, 0, self.overflow_pos);
|
||||
self.overflow_pos += 1;
|
||||
Some(val)
|
||||
}
|
||||
}
|
||||
fn size_hint(&self) -> (usize, Option<usize>) {
|
||||
let rem = self.n - self.slot;
|
||||
(rem, Some(rem))
|
||||
}
|
||||
}
|
||||
impl ExactSizeIterator for IntSliceViewIter<'_> {}
|
||||
@@ -3,6 +3,7 @@ use crossbeam_channel;
|
||||
use hashbrown::HashMap;
|
||||
use obikseq::k;
|
||||
use obikseq::{CanonicalKmer, Sequence, Unitig};
|
||||
#[cfg(not(any(test, feature = "test-utils")))]
|
||||
use rayon::iter::{IntoParallelRefIterator, ParallelIterator};
|
||||
use std::cell::RefCell;
|
||||
use std::fmt;
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
[package]
|
||||
name = "obikentropy"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
obikseq = { path = "../obikseq" }
|
||||
|
||||
[dev-dependencies]
|
||||
obikseq = { path = "../obikseq", features = ["test-utils"] }
|
||||
@@ -4,57 +4,6 @@ use std::path::PathBuf;
|
||||
const K_MAX: usize = 32;
|
||||
const WS_MAX: usize = 6;
|
||||
|
||||
fn normalize_circular(kmer: u64, ws: usize) -> u64 {
|
||||
let mask = (1u64 << (ws * 2)) - 1;
|
||||
let mut canonical = kmer & mask;
|
||||
let mut current = canonical;
|
||||
for _ in 0..ws - 1 {
|
||||
let top = (current >> ((ws - 1) * 2)) & 3;
|
||||
current = ((current << 2) | top) & mask;
|
||||
if current < canonical {
|
||||
canonical = current;
|
||||
}
|
||||
}
|
||||
canonical
|
||||
}
|
||||
|
||||
fn revcomp_raw(x: u64, k: usize) -> u64 {
|
||||
let x = !x;
|
||||
let x = x.swap_bytes();
|
||||
let x = ((x >> 4) & 0x0F0F0F0F0F0F0F0F) | ((x & 0x0F0F0F0F0F0F0F0F) << 4);
|
||||
let x = ((x >> 2) & 0x3333333333333333) | ((x & 0x3333333333333333) << 2);
|
||||
x << (64 - 2 * k)
|
||||
}
|
||||
|
||||
fn build_normalized_kmer(k: usize) -> Vec<u64> {
|
||||
let n = 1usize << (k * 2);
|
||||
let shift = 64 - k * 2;
|
||||
let mut result = vec![0u64; n];
|
||||
for i in 0..n {
|
||||
let la = (i as u64) << shift;
|
||||
let ra = i as u64;
|
||||
let rc_ra = revcomp_raw(la, k) >> shift;
|
||||
let circ = normalize_circular(ra, k);
|
||||
let circ_rc = normalize_circular(rc_ra, k);
|
||||
result[i] = if circ < circ_rc { circ } else { circ_rc };
|
||||
}
|
||||
result
|
||||
}
|
||||
|
||||
fn build_ln_class(norm: &[u64]) -> Vec<f64> {
|
||||
let n = norm.len();
|
||||
let mut sizes = vec![0u32; n];
|
||||
for &c in norm {
|
||||
sizes[c as usize] += 1;
|
||||
}
|
||||
norm.iter()
|
||||
.map(|&c| {
|
||||
let s = sizes[c as usize];
|
||||
if s > 0 { (s as f64).ln() } else { 0.0 }
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn build_n_log_n() -> [f64; K_MAX + 1] {
|
||||
let mut t = [0.0f64; K_MAX + 1];
|
||||
for n in 1..=K_MAX {
|
||||
@@ -63,6 +12,9 @@ fn build_n_log_n() -> [f64; K_MAX + 1] {
|
||||
t
|
||||
}
|
||||
|
||||
/// Max achievable entropy over `4^ws` raw sub-words given only `nwords`
|
||||
/// observations (most-uniform integer partition), per
|
||||
/// `docmd/theory/entropy.md`.
|
||||
fn build_emax() -> [[f64; WS_MAX + 1]; K_MAX + 1] {
|
||||
let mut t = [[0.0f64; WS_MAX + 1]; K_MAX + 1];
|
||||
for k in 2..=K_MAX {
|
||||
@@ -125,13 +77,6 @@ fn main() {
|
||||
let out_dir = PathBuf::from(std::env::var("OUT_DIR").unwrap());
|
||||
let mut out = String::new();
|
||||
|
||||
for k in 1..=6usize {
|
||||
let n = 1usize << (k * 2);
|
||||
let norm = build_normalized_kmer(k);
|
||||
let ln_class = build_ln_class(&norm);
|
||||
emit_f64_1d(&mut out, &format!("LN_CLASS{k}"), n, &ln_class);
|
||||
}
|
||||
|
||||
let n_log_n = build_n_log_n();
|
||||
emit_f64_1d(&mut out, "N_LOG_N", K_MAX + 1, &n_log_n);
|
||||
|
||||
@@ -141,5 +86,5 @@ fn main() {
|
||||
let log_nwords = build_log_nwords();
|
||||
emit_f64_2d(&mut out, "LOG_NWORDS", K_MAX + 1, WS_MAX + 1, &log_nwords);
|
||||
|
||||
fs::write(out_dir.join("ln_class_tables.rs"), out).unwrap();
|
||||
fs::write(out_dir.join("entropy_tables.rs"), out).unwrap();
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
//! Normalized entropy of an isolated, already-built k-mer (e.g. one
|
||||
//! reconstructed from an index's `unitigs.bin`, with no surrounding
|
||||
//! sequence) — drives the window through [`EntropyTracker`] one base at a
|
||||
//! time, exactly like the streaming path, so a `theta` threshold means the
|
||||
//! same thing whether applied during index construction or after the fact
|
||||
//! (e.g. `obikmer filter`).
|
||||
|
||||
use obikseq::CanonicalKmer;
|
||||
|
||||
use crate::tracker::EntropyTracker;
|
||||
|
||||
/// Extension trait: compute the normalized entropy of a single canonical
|
||||
/// k-mer, independent of any surrounding sequence.
|
||||
pub trait KmerEntropy {
|
||||
/// Normalized entropy across sub-word orders `1..=level_max` (the
|
||||
/// minimum is taken across orders). Lower means less complex; `theta`
|
||||
/// in `index`/`filter` rejects k-mers with a score `< theta`.
|
||||
fn entropy(&self, level_max: usize) -> f64;
|
||||
}
|
||||
|
||||
impl KmerEntropy for CanonicalKmer {
|
||||
fn entropy(&self, level_max: usize) -> f64 {
|
||||
let raw = self.raw(); // left-aligned, 2 bits/base, MSB-first
|
||||
let k = obikseq::params::k();
|
||||
let mask = (!0u64) >> (64 - k * 2);
|
||||
|
||||
let mut tracker = EntropyTracker::new(k);
|
||||
let mut rolling: u64 = 0;
|
||||
for i in 0..k {
|
||||
let shift = 64 - 2 * (i + 1);
|
||||
let base = (raw >> shift) & 3;
|
||||
rolling = ((rolling << 2) | base) & mask;
|
||||
tracker.push(i + 1, rolling);
|
||||
}
|
||||
tracker.normalized_entropy(level_max)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
#[path = "tests/kmer_entropy.rs"]
|
||||
mod tests;
|
||||
@@ -0,0 +1,17 @@
|
||||
//! Normalized k-mer entropy: formulas, tables, and a streaming tracker.
|
||||
//!
|
||||
//! This crate holds every piece of the entropy computation described in
|
||||
//! `docmd/theory/entropy.md`: the compile-time tables ([`table`], private),
|
||||
//! the incremental accumulator ([`EntropyTracker`]) that callers compose
|
||||
//! into their own streaming state, and the [`KmerEntropy`] convenience trait
|
||||
//! for scoring a single, already-built k-mer.
|
||||
|
||||
#![deny(missing_docs)]
|
||||
|
||||
mod kmer_entropy;
|
||||
mod ring;
|
||||
mod table;
|
||||
mod tracker;
|
||||
|
||||
pub use kmer_entropy::KmerEntropy;
|
||||
pub use tracker::EntropyTracker;
|
||||
@@ -0,0 +1,40 @@
|
||||
//! Stack-allocated ring buffer backing the sliding sub-word windows.
|
||||
|
||||
/// Fixed-capacity ring buffer backed by a stack array.
|
||||
/// N must be a power of two; operations are branchless via `% N`.
|
||||
pub(crate) struct Ring<T: Copy + Default, const N: usize> {
|
||||
buf: [T; N],
|
||||
head: usize,
|
||||
len: usize,
|
||||
}
|
||||
|
||||
impl<T: Copy + Default, const N: usize> Ring<T, N> {
|
||||
#[inline]
|
||||
pub(crate) fn new() -> Self {
|
||||
Self {
|
||||
buf: [T::default(); N],
|
||||
head: 0,
|
||||
len: 0,
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn clear(&mut self) {
|
||||
self.len = 0;
|
||||
self.head = 0;
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn push_back(&mut self, val: T) {
|
||||
self.buf[(self.head + self.len) % N] = val;
|
||||
self.len += 1;
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn pop_front(&mut self) -> T {
|
||||
let val = self.buf[self.head];
|
||||
self.head = (self.head + 1) % N;
|
||||
self.len -= 1;
|
||||
val
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
//! Compile-time tables backing the normalized k-mer entropy formula: the
|
||||
//! max-entropy correction for small samples. See `docmd/theory/entropy.md`.
|
||||
//!
|
||||
//! Entropy is computed directly on raw (non-canonicalized) sub-words — no
|
||||
//! equivalence-class folding. Empirically (see the discussion that produced
|
||||
//! this crate's history), folding sub-words into circular/revcomp classes
|
||||
//! before unfolding them back buys nothing for the invariances it was meant
|
||||
//! to guarantee (both hold for raw sub-word entropy already, by a direct
|
||||
//! bijection argument for revcomp and by the sliding window's own dynamics
|
||||
//! for tandem repeats), while it measurably *weakens* detection of the
|
||||
//! low-complexity sequences the filter exists to catch.
|
||||
|
||||
include!(concat!(env!("OUT_DIR"), "/entropy_tables.rs"));
|
||||
|
||||
pub(crate) const WS_MAX: usize = 6;
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) const fn n_log_n(n: usize) -> f64 {
|
||||
N_LOG_N[n]
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) const fn emax(k: usize, ws: usize) -> f64 {
|
||||
EMAX[k][ws]
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) const fn log_nwords(k: usize, ws: usize) -> f64 {
|
||||
LOG_NWORDS[k][ws]
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
use super::*;
|
||||
use obikseq::Sequence;
|
||||
use obikseq::kmer::Kmer;
|
||||
|
||||
const K: usize = 21;
|
||||
const LEVEL_MAX: usize = 6;
|
||||
|
||||
fn kmer_from_ascii(seq: &[u8]) -> CanonicalKmer {
|
||||
obikseq::set_k(K);
|
||||
Kmer::from_ascii(seq).expect("valid k-mer sequence").canonical()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn homopolymer_scores_lower_than_diverse_sequence() {
|
||||
let homopolymer = kmer_from_ascii(b"AAAAAAAAAAAAAAAAAAAAA"); // 21 bases
|
||||
let diverse = kmer_from_ascii(b"CATTAGCGTACCTGATCAGGT"); // 21 bases, same as used elsewhere in this workspace's tests
|
||||
|
||||
let e_homopolymer = homopolymer.entropy(LEVEL_MAX);
|
||||
let e_diverse = diverse.entropy(LEVEL_MAX);
|
||||
|
||||
assert!(
|
||||
e_homopolymer < e_diverse,
|
||||
"homopolymer ({e_homopolymer}) should score lower than a diverse sequence ({e_diverse})"
|
||||
);
|
||||
// A pure homopolymer is the most degenerate case representable — its
|
||||
// score should sit near the bottom of the range, not just "somewhat lower".
|
||||
assert!(e_homopolymer < 0.3, "homopolymer entropy unexpectedly high: {e_homopolymer}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn entropy_is_deterministic_for_the_same_kmer() {
|
||||
let a = kmer_from_ascii(b"CATTAGCGTACCTGATCAGGT");
|
||||
let b = kmer_from_ascii(b"CATTAGCGTACCTGATCAGGT");
|
||||
assert_eq!(a.entropy(LEVEL_MAX), b.entropy(LEVEL_MAX));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn entropy_is_within_zero_one_range() {
|
||||
let mut repeat = "AT".repeat(K / 2 + 1);
|
||||
repeat.truncate(K);
|
||||
|
||||
for seq in [
|
||||
"AAAAAAAAAAAAAAAAAAAAA".to_string(),
|
||||
repeat,
|
||||
"CATTAGCGTACCTGATCAGGT".to_string(),
|
||||
] {
|
||||
assert_eq!(seq.len(), K, "test sequence must be exactly K bases: {seq:?}");
|
||||
let kmer = kmer_from_ascii(seq.as_bytes());
|
||||
let e = kmer.entropy(LEVEL_MAX);
|
||||
assert!((0.0..=1.0).contains(&e), "entropy {e} out of [0,1] for {seq:?}");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,255 @@
|
||||
//! Incremental (streaming) normalized k-mer entropy.
|
||||
//!
|
||||
//! [`EntropyTracker`] maintains, over a sliding window of the last `k` bases,
|
||||
//! the per-sub-word-size raw-word frequency statistics needed to evaluate
|
||||
//! the corrected Shannon entropy described in `docmd/theory/entropy.md`,
|
||||
//! updated in O(1) per base rather than recomputed from scratch. No
|
||||
//! canonicalization is applied — each sub-word is tallied under its own raw
|
||||
//! 2-bit-packed value; only the small-sample max-entropy correction departs
|
||||
//! from a textbook Shannon entropy.
|
||||
//!
|
||||
//! It carries no notion of minimizers or superkmer segmentation — callers
|
||||
//! that need both (e.g. `obiskbuilder::RollingStat`) compose an
|
||||
//! `EntropyTracker` as a plain field alongside their own state, so the two
|
||||
//! concerns update in the same streaming pass without being conflated in one
|
||||
//! struct.
|
||||
|
||||
use crate::ring::Ring;
|
||||
use crate::table::{WS_MAX, emax, log_nwords, n_log_n};
|
||||
|
||||
/// Incremental normalized-entropy accumulator over a sliding window of `k`
|
||||
/// bases. Composed as a plain field by callers that also need other
|
||||
/// per-base state (e.g. minimizer selection) in the same streaming pass.
|
||||
pub struct EntropyTracker {
|
||||
k: usize,
|
||||
steady: bool,
|
||||
|
||||
// Sliding-window queues over the last `k` raw sub-words, one per word
|
||||
// size — stack-allocated, capacity ≤ k ≤ 31.
|
||||
k1q: Ring<u64, 32>,
|
||||
k2q: Ring<u64, 32>,
|
||||
k3q: Ring<u64, 32>,
|
||||
k4q: Ring<u64, 32>,
|
||||
k5q: Ring<u64, 32>,
|
||||
k6q: Ring<u64, 32>,
|
||||
|
||||
// Frequency count arrays, indexed by the raw sub-word value (2 bits per
|
||||
// base). Max count per cell ≤ k ≤ 31 → u8 is sufficient.
|
||||
k1c: [u8; 4],
|
||||
k2c: [u8; 16],
|
||||
k3c: [u8; 64],
|
||||
k4c: [u8; 256],
|
||||
k5c: [u8; 1024],
|
||||
k6c: [u8; 4096],
|
||||
|
||||
sum_f_log_f: [f64; WS_MAX + 1],
|
||||
}
|
||||
|
||||
impl EntropyTracker {
|
||||
/// New tracker for a window of `k` bases (1..=31).
|
||||
pub fn new(k: usize) -> Self {
|
||||
Self {
|
||||
k,
|
||||
steady: false,
|
||||
k1q: Ring::new(),
|
||||
k2q: Ring::new(),
|
||||
k3q: Ring::new(),
|
||||
k4q: Ring::new(),
|
||||
k5q: Ring::new(),
|
||||
k6q: Ring::new(),
|
||||
k1c: [0; 4],
|
||||
k2c: [0; 16],
|
||||
k3c: [0; 64],
|
||||
k4c: [0; 256],
|
||||
k5c: [0; 1024],
|
||||
k6c: [0; 4096],
|
||||
sum_f_log_f: [0.0; WS_MAX + 1],
|
||||
}
|
||||
}
|
||||
|
||||
/// Clear all accumulated state, ready to track a new window from
|
||||
/// scratch (`k` is unchanged).
|
||||
pub fn reset(&mut self) {
|
||||
self.steady = false;
|
||||
|
||||
self.k1c.fill(0);
|
||||
self.k2c.fill(0);
|
||||
self.k3c.fill(0);
|
||||
self.k4c.fill(0);
|
||||
self.k5c.fill(0);
|
||||
self.k6c.fill(0);
|
||||
|
||||
self.k1q.clear();
|
||||
self.k2q.clear();
|
||||
self.k3q.clear();
|
||||
self.k4q.clear();
|
||||
self.k5q.clear();
|
||||
self.k6q.clear();
|
||||
|
||||
self.sum_f_log_f = [0.0; WS_MAX + 1];
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn update_sums_decrement<const K: usize>(sum_f_log_f: &mut [f64; WS_MAX + 1], f: usize) {
|
||||
sum_f_log_f[K] += n_log_n(f - 1) - n_log_n(f);
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn update_sums_increment<const K: usize>(sum_f_log_f: &mut [f64; WS_MAX + 1], g: usize) {
|
||||
sum_f_log_f[K] += n_log_n(g + 1) - n_log_n(g);
|
||||
}
|
||||
|
||||
/// Advance the window by one base. `received` is the caller's running
|
||||
/// count of bases pushed so far (1-based, i.e. after this base);
|
||||
/// `rolling_kmer` is the current right-aligned, 2-bit-packed k-mer
|
||||
/// window (same convention as `obiskbuilder::RollingStat::rolling_k`).
|
||||
pub fn push(&mut self, received: usize, rolling_kmer: u64) {
|
||||
let raw1 = rolling_kmer & 3;
|
||||
let raw2 = rolling_kmer & 15;
|
||||
let raw3 = rolling_kmer & 63;
|
||||
let raw4 = rolling_kmer & 255;
|
||||
let raw5 = rolling_kmer & 1023;
|
||||
let raw6 = rolling_kmer & 4095;
|
||||
|
||||
if received > self.k {
|
||||
let old1 = self.k1q.pop_front();
|
||||
let f1 = self.k1c[old1 as usize] as usize;
|
||||
Self::update_sums_decrement::<1>(&mut self.sum_f_log_f, f1);
|
||||
self.k1c[old1 as usize] -= 1;
|
||||
|
||||
let old2 = self.k2q.pop_front();
|
||||
let f2 = self.k2c[old2 as usize] as usize;
|
||||
Self::update_sums_decrement::<2>(&mut self.sum_f_log_f, f2);
|
||||
self.k2c[old2 as usize] -= 1;
|
||||
|
||||
let old3 = self.k3q.pop_front();
|
||||
let f3 = self.k3c[old3 as usize] as usize;
|
||||
Self::update_sums_decrement::<3>(&mut self.sum_f_log_f, f3);
|
||||
self.k3c[old3 as usize] -= 1;
|
||||
|
||||
let old4 = self.k4q.pop_front();
|
||||
let f4 = self.k4c[old4 as usize] as usize;
|
||||
Self::update_sums_decrement::<4>(&mut self.sum_f_log_f, f4);
|
||||
self.k4c[old4 as usize] -= 1;
|
||||
|
||||
let old5 = self.k5q.pop_front();
|
||||
let f5 = self.k5c[old5 as usize] as usize;
|
||||
Self::update_sums_decrement::<5>(&mut self.sum_f_log_f, f5);
|
||||
self.k5c[old5 as usize] -= 1;
|
||||
|
||||
let old6 = self.k6q.pop_front();
|
||||
let f6 = self.k6c[old6 as usize] as usize;
|
||||
Self::update_sums_decrement::<6>(&mut self.sum_f_log_f, f6);
|
||||
self.k6c[old6 as usize] -= 1;
|
||||
}
|
||||
|
||||
if self.steady {
|
||||
let g1 = self.k1c[raw1 as usize] as usize;
|
||||
Self::update_sums_increment::<1>(&mut self.sum_f_log_f, g1);
|
||||
self.k1c[raw1 as usize] += 1;
|
||||
self.k1q.push_back(raw1);
|
||||
|
||||
let g2 = self.k2c[raw2 as usize] as usize;
|
||||
Self::update_sums_increment::<2>(&mut self.sum_f_log_f, g2);
|
||||
self.k2c[raw2 as usize] += 1;
|
||||
self.k2q.push_back(raw2);
|
||||
|
||||
let g3 = self.k3c[raw3 as usize] as usize;
|
||||
Self::update_sums_increment::<3>(&mut self.sum_f_log_f, g3);
|
||||
self.k3c[raw3 as usize] += 1;
|
||||
self.k3q.push_back(raw3);
|
||||
|
||||
let g4 = self.k4c[raw4 as usize] as usize;
|
||||
Self::update_sums_increment::<4>(&mut self.sum_f_log_f, g4);
|
||||
self.k4c[raw4 as usize] += 1;
|
||||
self.k4q.push_back(raw4);
|
||||
|
||||
let g5 = self.k5c[raw5 as usize] as usize;
|
||||
Self::update_sums_increment::<5>(&mut self.sum_f_log_f, g5);
|
||||
self.k5c[raw5 as usize] += 1;
|
||||
self.k5q.push_back(raw5);
|
||||
|
||||
let g6 = self.k6c[raw6 as usize] as usize;
|
||||
Self::update_sums_increment::<6>(&mut self.sum_f_log_f, g6);
|
||||
self.k6c[raw6 as usize] += 1;
|
||||
self.k6q.push_back(raw6);
|
||||
} else {
|
||||
self.push_warmup_increments(received, raw1, raw2, raw3, raw4, raw5, raw6);
|
||||
}
|
||||
}
|
||||
|
||||
#[cold]
|
||||
#[inline(never)]
|
||||
fn push_warmup_increments(
|
||||
&mut self,
|
||||
received: usize,
|
||||
raw1: u64, raw2: u64, raw3: u64,
|
||||
raw4: u64, raw5: u64, raw6: u64,
|
||||
) {
|
||||
let g1 = self.k1c[raw1 as usize] as usize;
|
||||
Self::update_sums_increment::<1>(&mut self.sum_f_log_f, g1);
|
||||
self.k1c[raw1 as usize] += 1;
|
||||
self.k1q.push_back(raw1);
|
||||
|
||||
if received >= 2 {
|
||||
let g2 = self.k2c[raw2 as usize] as usize;
|
||||
Self::update_sums_increment::<2>(&mut self.sum_f_log_f, g2);
|
||||
self.k2c[raw2 as usize] += 1;
|
||||
self.k2q.push_back(raw2);
|
||||
|
||||
if received >= 3 {
|
||||
let g3 = self.k3c[raw3 as usize] as usize;
|
||||
Self::update_sums_increment::<3>(&mut self.sum_f_log_f, g3);
|
||||
self.k3c[raw3 as usize] += 1;
|
||||
self.k3q.push_back(raw3);
|
||||
|
||||
if received >= 4 {
|
||||
let g4 = self.k4c[raw4 as usize] as usize;
|
||||
Self::update_sums_increment::<4>(&mut self.sum_f_log_f, g4);
|
||||
self.k4c[raw4 as usize] += 1;
|
||||
self.k4q.push_back(raw4);
|
||||
|
||||
if received >= 5 {
|
||||
let g5 = self.k5c[raw5 as usize] as usize;
|
||||
Self::update_sums_increment::<5>(&mut self.sum_f_log_f, g5);
|
||||
self.k5c[raw5 as usize] += 1;
|
||||
self.k5q.push_back(raw5);
|
||||
|
||||
if received >= 6 {
|
||||
let g6 = self.k6c[raw6 as usize] as usize;
|
||||
Self::update_sums_increment::<6>(&mut self.sum_f_log_f, g6);
|
||||
self.k6c[raw6 as usize] += 1;
|
||||
self.k6q.push_back(raw6);
|
||||
self.steady = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Normalized entropy at sub-word size `order` (1..=6). The caller is
|
||||
/// responsible for not calling this before the window is full (`k`
|
||||
/// bases pushed) — an empty/partial window yields a meaningless value.
|
||||
pub fn entropy(&self, order: usize) -> f64 {
|
||||
let k = self.k;
|
||||
let em = emax(k, order);
|
||||
if em <= 0.0 {
|
||||
return 1.0;
|
||||
}
|
||||
let nwords = k - order + 1;
|
||||
let log_nw = log_nwords(k, order);
|
||||
let nw_f = nwords as f64;
|
||||
let h_corr = log_nw - self.sum_f_log_f[order] / nw_f;
|
||||
(h_corr / em).max(0.0)
|
||||
}
|
||||
|
||||
/// Minimum of [`Self::entropy`] over sub-word sizes `1..=order_max`, same
|
||||
/// caller responsibility re: window readiness as `entropy`.
|
||||
pub fn normalized_entropy(&self, order_max: usize) -> f64 {
|
||||
let min_e = (1..=order_max)
|
||||
.map(|ws| self.entropy(ws))
|
||||
.fold(f64::MAX, f64::min);
|
||||
if min_e == f64::MAX { 1.0 } else { min_e }
|
||||
}
|
||||
}
|
||||
@@ -10,6 +10,8 @@ obiskio = { path = "../obiskio" }
|
||||
obisys = { path = "../obisys" }
|
||||
obicompactvec = { path = "../obicompactvec" }
|
||||
obilayeredmap = { path = "../obilayeredmap" }
|
||||
obiskbuilder = { path = "../obiskbuilder" }
|
||||
obipipeline = { path = "../obipipeline" }
|
||||
ndarray = "0.16"
|
||||
rayon = "1"
|
||||
crossbeam-channel = "0.5"
|
||||
@@ -17,3 +19,12 @@ serde = { version = "1", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
indicatif = "0.17"
|
||||
tracing = "0.1.44"
|
||||
hwlocality = { version = "1.0.0-alpha.11", features = ["vendored"], optional = true }
|
||||
|
||||
[dev-dependencies]
|
||||
obiread = { path = "../obiread" }
|
||||
tempfile = "3"
|
||||
|
||||
[features]
|
||||
default = ["numa"]
|
||||
numa = ["hwlocality"]
|
||||
|
||||
@@ -14,6 +14,8 @@ pub enum DistanceMetric {
|
||||
Jaccard,
|
||||
/// Hamming distance (number of differing kmer positions) on presence/absence data.
|
||||
Hamming,
|
||||
/// Mash distance on presence/absence data (Jaccard-derived mutation-rate estimate).
|
||||
Mash,
|
||||
/// Bray-Curtis dissimilarity on raw counts.
|
||||
BrayCurtis,
|
||||
/// Bray-Curtis dissimilarity normalised by per-genome total counts.
|
||||
@@ -84,6 +86,7 @@ impl KmerIndex {
|
||||
DistanceMetric::Hellinger => CountPartials::hellinger_dist_matrix(&global),
|
||||
DistanceMetric::HellingerEuclidean => CountPartials::hellinger_euclidean_dist_matrix(&global),
|
||||
DistanceMetric::Jaccard => CountPartials::threshold_jaccard_dist_matrix(&global, presence_threshold),
|
||||
DistanceMetric::Mash => CountPartials::threshold_mash_dist_matrix(&global, self.kmer_size(), presence_threshold),
|
||||
DistanceMetric::Hamming => {
|
||||
return Err(OKIError::InvalidInput(
|
||||
"Hamming is only available for presence/absence indexes".into(),
|
||||
@@ -108,6 +111,7 @@ impl KmerIndex {
|
||||
|
||||
let matrix = match metric {
|
||||
DistanceMetric::Jaccard => BitPartials::jaccard_dist_matrix(&global),
|
||||
DistanceMetric::Mash => BitPartials::mash_dist_matrix(&global, self.kmer_size()),
|
||||
DistanceMetric::Hamming => {
|
||||
BitPartials::hamming_dist_matrix(&global).mapv(|v| v as f64)
|
||||
}
|
||||
|
||||
+29
-45
@@ -1,8 +1,6 @@
|
||||
use std::collections::BTreeMap;
|
||||
use std::fs;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use obikpartitionner::{KmerPartition, KmerSpectrum};
|
||||
use obilayeredmap;
|
||||
@@ -152,31 +150,25 @@ impl KmerIndex {
|
||||
let with_counts = self.meta.config.with_counts;
|
||||
let evidence = self.meta.config.evidence.clone();
|
||||
let block_bits = self.meta.config.block_bits;
|
||||
let total_kmers = AtomicUsize::new(0);
|
||||
let mut total_kmers: usize = 0;
|
||||
let pb = progress_bar("index", n as u64, "partitions");
|
||||
|
||||
let pb = Arc::new(Mutex::new(progress_bar("index", n as u64, "partitions")));
|
||||
|
||||
(0..n).into_par_iter().for_each(|i| {
|
||||
match self.partition.build_index_layer(i, min_ab, max_ab, with_counts, &evidence, block_bits) {
|
||||
Ok(0) => {}
|
||||
Ok(n_kmers) => {
|
||||
total_kmers.fetch_add(n_kmers, Ordering::Relaxed);
|
||||
let pb = pb.lock().unwrap();
|
||||
let order: Vec<usize> = (0..n).collect();
|
||||
let runner = crate::numa::PartitionRunner::new();
|
||||
runner.run(
|
||||
&order,
|
||||
|i| self.partition.build_index_layer(i, min_ab, max_ab, with_counts, &evidence, block_bits),
|
||||
|i, n_kmers, _| {
|
||||
if n_kmers > 0 {
|
||||
total_kmers += n_kmers;
|
||||
pb.inc(1);
|
||||
pb.set_message(format!("{i}: {n_kmers} kmers"));
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!("error building layer for partition {i}: {e}");
|
||||
std::process::exit(1);
|
||||
}
|
||||
}
|
||||
});
|
||||
},
|
||||
).map_err(OKIError::Partition)?;
|
||||
|
||||
pb.lock().unwrap().finish_and_clear();
|
||||
info!(
|
||||
"done — {} total kmers indexed",
|
||||
total_kmers.load(Ordering::Relaxed)
|
||||
);
|
||||
pb.finish_and_clear();
|
||||
info!("done — {} total kmers indexed", total_kmers);
|
||||
|
||||
if !keep_intermediate {
|
||||
for i in 0..n {
|
||||
@@ -211,35 +203,27 @@ impl KmerIndex {
|
||||
use obilayeredmap::meta::PartitionMeta;
|
||||
|
||||
let n = self.n_partitions();
|
||||
let errors: Vec<_> = (0..n)
|
||||
.into_par_iter()
|
||||
.filter_map(|i| {
|
||||
let order: Vec<usize> = (0..n).collect();
|
||||
let pb = progress_bar("pack", n as u64, "partitions");
|
||||
crate::numa::PartitionRunner::new().run(
|
||||
&order,
|
||||
|i| -> OKIResult<()> {
|
||||
let index_dir = self.partition.part_dir(i).join("index");
|
||||
if !index_dir.exists() { return None; }
|
||||
let meta = match PartitionMeta::load(&index_dir) {
|
||||
Ok(m) => m,
|
||||
Err(e) => return Some(OKIError::Io(std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))),
|
||||
};
|
||||
if !index_dir.exists() { return Ok(()); }
|
||||
let meta = PartitionMeta::load(&index_dir)
|
||||
.map_err(|e| OKIError::Io(std::io::Error::new(std::io::ErrorKind::Other, e.to_string())))?;
|
||||
for l in 0..meta.n_layers {
|
||||
let layer_dir = index_dir.join(format!("layer_{l}"));
|
||||
let presence_dir = layer_dir.join("presence");
|
||||
let counts_dir = layer_dir.join("counts");
|
||||
if presence_dir.exists() {
|
||||
if let Err(e) = pack_bit_matrix(&presence_dir) {
|
||||
return Some(OKIError::Io(e));
|
||||
}
|
||||
}
|
||||
if counts_dir.exists() {
|
||||
if let Err(e) = pack_compact_int_matrix(&counts_dir) {
|
||||
return Some(OKIError::Io(e));
|
||||
}
|
||||
}
|
||||
if presence_dir.exists() { pack_bit_matrix(&presence_dir).map_err(OKIError::Io)?; }
|
||||
if counts_dir.exists() { pack_compact_int_matrix(&counts_dir).map_err(OKIError::Io)?; }
|
||||
}
|
||||
None
|
||||
})
|
||||
.collect();
|
||||
|
||||
if let Some(e) = errors.into_iter().next() { return Err(e); }
|
||||
Ok(())
|
||||
},
|
||||
|_, _, _| { pb.inc(1); },
|
||||
)?;
|
||||
pb.finish_and_clear();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
|
||||
@@ -5,9 +5,11 @@ mod distance;
|
||||
mod dump;
|
||||
mod index;
|
||||
mod merge;
|
||||
mod numa;
|
||||
mod rebuild;
|
||||
mod reindex;
|
||||
mod select;
|
||||
mod siblings;
|
||||
mod stats;
|
||||
|
||||
pub use error::{OKIError, OKIResult};
|
||||
@@ -17,3 +19,4 @@ pub use merge::MergeMode;
|
||||
pub use meta::{validate_label, GenomeInfo, IndexConfig, IndexMeta, META_FILENAME};
|
||||
pub use state::{IndexState, SENTINEL_COUNTED, SENTINEL_INDEXED, SENTINEL_SCATTERED};
|
||||
pub use stats::IndexBitsPerKmer;
|
||||
pub use siblings::{RawSnpDistanceOutput, SiblingAnnexStats, SnpAlignment};
|
||||
|
||||
+17
-140
@@ -2,10 +2,8 @@ use std::collections::HashMap;
|
||||
use std::fs;
|
||||
use std::io;
|
||||
use std::path::Path;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use crossbeam_channel::unbounded;
|
||||
use obisys::{CpuSample, Reporter, Stage, progress_bar, spinner};
|
||||
use obisys::{Reporter, Stage, progress_bar, spinner};
|
||||
use tracing::{debug, info};
|
||||
|
||||
use obilayeredmap::IndexMode;
|
||||
@@ -13,7 +11,7 @@ use obilayeredmap::IndexMode;
|
||||
use crate::error::{OKIError, OKIResult};
|
||||
use crate::index::KmerIndex;
|
||||
use crate::meta::{GenomeInfo, IndexMeta};
|
||||
use crate::state::IndexState;
|
||||
use crate::state::{IndexState, SENTINEL_INDEXED};
|
||||
|
||||
pub use obikpartitionner::MergeMode;
|
||||
|
||||
@@ -223,156 +221,33 @@ impl KmerIndex {
|
||||
let mut order: Vec<usize> = (0..n_partitions).collect();
|
||||
order.sort_unstable_by_key(|&i| std::cmp::Reverse(partition_sizes[i]));
|
||||
|
||||
// ── Adaptive worker pool ──────────────────────────────────────────
|
||||
// Start with 1 worker thread. After each completed partition,
|
||||
// measure CPU efficiency (via getrusage delta). If efficiency is
|
||||
// below the spawn threshold and more partitions remain, spawn one
|
||||
// additional worker. Workers share a crossbeam channel of partition
|
||||
// IDs; each reports (id, g_len, duration) on a result channel.
|
||||
const SPAWN_THRESHOLD: f64 = 0.95; // spawn when >5% capacity idle
|
||||
let n_cores = std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(1);
|
||||
let max_workers = (n_cores / 2).max(1);
|
||||
let _ = budget_fraction; // kept in signature for CLI compatibility
|
||||
|
||||
let (part_tx, part_rx) = unbounded::<usize>();
|
||||
let (result_tx, result_rx) =
|
||||
unbounded::<(usize, Result<usize, obiskio::SKError>, Duration)>();
|
||||
// activate_tx: controller sends () to wake the next dormant worker.
|
||||
// Dropping activate_tx closes the channel; dormant workers exit.
|
||||
let (activate_tx, activate_rx) = unbounded::<()>();
|
||||
|
||||
for &i in &order {
|
||||
part_tx.send(i).ok();
|
||||
}
|
||||
drop(part_tx);
|
||||
|
||||
let mut part_stats: Vec<PartStat> = Vec::with_capacity(n_partitions);
|
||||
let mut n_workers = 0usize;
|
||||
let mut cpu_sample = CpuSample::now();
|
||||
// Efficiency measured just before each spawn, used to assess
|
||||
// whether the previous worker delivered its expected marginal gain.
|
||||
let mut efficiency_at_last_spawn = 0.0f64;
|
||||
|
||||
// Shadow as references so closures can capture them by copy.
|
||||
let srcs = &srcs;
|
||||
let evidence = &evidence;
|
||||
|
||||
std::thread::scope(|s| -> OKIResult<()> {
|
||||
// Pre-spawn max_workers threads; each waits for an activation
|
||||
// signal before consuming from part_rx.
|
||||
for _ in 0..max_workers {
|
||||
let prx = part_rx.clone();
|
||||
let rtx = result_tx.clone();
|
||||
let arx = activate_rx.clone();
|
||||
s.spawn(move || {
|
||||
if arx.recv().is_ok() {
|
||||
for i in &prx {
|
||||
let t = Instant::now();
|
||||
let r = dst_partition.merge_partition(
|
||||
i,
|
||||
srcs,
|
||||
mode,
|
||||
n_dst_genomes,
|
||||
block_bits,
|
||||
evidence,
|
||||
);
|
||||
rtx.send((i, r, t.elapsed())).ok();
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
drop(result_tx);
|
||||
let runner = crate::numa::PartitionRunner::new();
|
||||
let mut part_stats: Vec<PartStat> = Vec::with_capacity(n_partitions);
|
||||
|
||||
// Activate first worker immediately.
|
||||
activate_tx.send(()).ok();
|
||||
n_workers = 1;
|
||||
|
||||
const SPAWN_POLL: Duration = Duration::from_secs(10);
|
||||
|
||||
let mut completed = 0usize;
|
||||
while completed < n_partitions {
|
||||
let result = result_rx.recv_timeout(SPAWN_POLL);
|
||||
|
||||
// On timeout: no partition finished yet, just check efficiency.
|
||||
let (i, r, dur) = match result {
|
||||
Ok(v) => v,
|
||||
Err(crossbeam_channel::RecvTimeoutError::Timeout) => {
|
||||
if n_workers < max_workers {
|
||||
let eff = cpu_sample.cpu_efficiency(n_cores);
|
||||
if eff < SPAWN_THRESHOLD {
|
||||
debug!(
|
||||
"activated worker {} (poll) — efficiency {:.0}%",
|
||||
n_workers + 1,
|
||||
eff * 100.0,
|
||||
);
|
||||
efficiency_at_last_spawn = eff;
|
||||
activate_tx.send(()).ok();
|
||||
n_workers += 1;
|
||||
cpu_sample = CpuSample::now();
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
Err(crossbeam_channel::RecvTimeoutError::Disconnected) => {
|
||||
return Err(OKIError::Io(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
"worker channel closed",
|
||||
)));
|
||||
}
|
||||
};
|
||||
let g_len = r.map_err(OKIError::Partition)?;
|
||||
runner.run(
|
||||
&order,
|
||||
|i| dst_partition.merge_partition(i, srcs, mode, n_dst_genomes, block_bits, evidence),
|
||||
|i, g_len, dur| {
|
||||
pb.inc(1);
|
||||
debug!(
|
||||
"partition {i}: done in {:.1}s — {} new kmers",
|
||||
dur.as_secs_f64(),
|
||||
g_len
|
||||
);
|
||||
part_stats.push(PartStat {
|
||||
id: i,
|
||||
unitig_bytes: partition_sizes[i],
|
||||
g_len,
|
||||
});
|
||||
completed += 1;
|
||||
|
||||
if n_workers < max_workers && completed < n_partitions {
|
||||
let eff = cpu_sample.cpu_efficiency(n_cores);
|
||||
// For the first spawn use SPAWN_THRESHOLD.
|
||||
// For subsequent spawns: the previous worker should
|
||||
// have raised efficiency by at least a quarter of the expected
|
||||
// marginal gain (1/n_workers). If not, adding another
|
||||
// worker won't help.
|
||||
let should_spawn = if n_workers == 1 {
|
||||
eff < SPAWN_THRESHOLD
|
||||
} else {
|
||||
let gain = eff - efficiency_at_last_spawn;
|
||||
let expected = 1.0 / n_workers as f64;
|
||||
gain >= expected * 0.25
|
||||
};
|
||||
if should_spawn {
|
||||
debug!(
|
||||
"activated worker {} — efficiency {:.0}%, gain vs prev {:.0}%",
|
||||
n_workers + 1,
|
||||
eff * 100.0,
|
||||
(eff - efficiency_at_last_spawn) * 100.0,
|
||||
);
|
||||
efficiency_at_last_spawn = eff;
|
||||
activate_tx.send(()).ok();
|
||||
n_workers += 1;
|
||||
cpu_sample = CpuSample::now();
|
||||
}
|
||||
}
|
||||
}
|
||||
// Close activate_tx: dormant workers exit cleanly.
|
||||
drop(activate_tx);
|
||||
Ok(())
|
||||
})?;
|
||||
);
|
||||
part_stats.push(PartStat { id: i, unitig_bytes: partition_sizes[i], g_len });
|
||||
},
|
||||
).map_err(OKIError::Partition)?;
|
||||
|
||||
pb.finish_and_clear();
|
||||
|
||||
// ── Diagnostic report ─────────────────────────────────────────────
|
||||
print_merge_partition_report(&part_stats, n_workers, max_workers);
|
||||
print_merge_partition_report(&part_stats, runner.max_workers());
|
||||
|
||||
rep.push(t.stop());
|
||||
}
|
||||
@@ -388,13 +263,15 @@ impl KmerIndex {
|
||||
rep.push(t.stop());
|
||||
}
|
||||
|
||||
fs::File::create(output.join(SENTINEL_INDEXED)).map_err(OKIError::Io)?;
|
||||
|
||||
KmerIndex::open(output)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Diagnostic report ─────────────────────────────────────────────────────────
|
||||
|
||||
fn print_merge_partition_report(stats: &[PartStat], n_workers: usize, max_workers: usize) {
|
||||
fn print_merge_partition_report(stats: &[PartStat], max_workers: usize) {
|
||||
let total_new: usize = stats.iter().map(|s| s.g_len).sum();
|
||||
let non_empty = stats.iter().filter(|s| s.unitig_bytes > 0).count();
|
||||
|
||||
@@ -408,7 +285,7 @@ fn print_merge_partition_report(stats: &[PartStat], n_workers: usize, max_worker
|
||||
" {} partition(s) processed, {} total new kmers",
|
||||
non_empty, total_new,
|
||||
);
|
||||
info!(" workers spawned: {n_workers} / {max_workers} (max)",);
|
||||
info!(" max workers: {max_workers}");
|
||||
|
||||
// Top 8 partitions by new-kmer count
|
||||
let mut by_new: Vec<&PartStat> = stats.iter().filter(|s| s.g_len > 0).collect();
|
||||
|
||||
@@ -0,0 +1,493 @@
|
||||
// NUMA-aware partition runner via hwlocality.
|
||||
//
|
||||
// Detects NUMA topology using hwloc (cross-platform: Linux, macOS, etc.) and
|
||||
// builds one Rayon ThreadPool per NUMA node with threads pinned to that node's
|
||||
// CPUs. Linux first-touch policy then places graph allocations in local DRAM
|
||||
// automatically — no explicit memory binding needed.
|
||||
//
|
||||
// UMA systems (single socket, Apple Silicon, etc.) are the degenerate case:
|
||||
// one synthetic node containing all cores, no pool, no pinning.
|
||||
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use crossbeam_channel::unbounded;
|
||||
#[cfg(feature = "numa")]
|
||||
use hwlocality::Topology;
|
||||
#[cfg(feature = "numa")]
|
||||
use hwlocality::cpu::binding::CpuBindingFlags;
|
||||
#[cfg(feature = "numa")]
|
||||
use hwlocality::cpu::cpuset::CpuSet;
|
||||
#[cfg(feature = "numa")]
|
||||
use hwlocality::object::types::ObjectType;
|
||||
use obisys::{CpuSample, IoSample};
|
||||
use tracing::debug;
|
||||
|
||||
// ── Public interface ──────────────────────────────────────────────────────────
|
||||
|
||||
pub struct NumaSetup {
|
||||
/// One entry per NUMA node. `None` on UMA systems (no pool, no pinning).
|
||||
pub pools: Vec<Option<Arc<rayon::ThreadPool>>>,
|
||||
/// CPU indices for each NUMA node, in node order.
|
||||
pub cpus_per_node: Vec<Vec<usize>>,
|
||||
}
|
||||
|
||||
impl NumaSetup {
|
||||
/// Maximum worker slots per node (one per physical core in the node).
|
||||
pub fn workers_per_node(&self) -> usize {
|
||||
self.cpus_per_node
|
||||
.first()
|
||||
.map(|c| c.len().max(1))
|
||||
.unwrap_or(1)
|
||||
}
|
||||
}
|
||||
|
||||
/// Detect NUMA topology and build per-node Rayon pools.
|
||||
/// Always succeeds: falls back to a single synthetic UMA node on failure.
|
||||
#[cfg(feature = "numa")]
|
||||
pub fn build() -> NumaSetup {
|
||||
if let Ok(topology) = Topology::new() {
|
||||
let nodes: Vec<Vec<usize>> = topology
|
||||
.objects_with_type(ObjectType::NUMANode)
|
||||
.filter_map(|obj| obj.cpuset())
|
||||
.map(|cpuset| {
|
||||
cpuset
|
||||
.iter_set()
|
||||
.map(|idx| usize::from(idx))
|
||||
.collect::<Vec<_>>()
|
||||
})
|
||||
.filter(|v| !v.is_empty())
|
||||
.collect();
|
||||
|
||||
if nodes.len() > 1 {
|
||||
if let Some(pools) = nodes
|
||||
.iter()
|
||||
.map(|cpus| build_pool(cpus).map(|p| Some(Arc::new(p))))
|
||||
.collect::<Option<Vec<_>>>()
|
||||
{
|
||||
debug!(
|
||||
"NUMA topology: {} node(s), {} core(s)/node",
|
||||
nodes.len(),
|
||||
nodes.first().map_or(0, |v| v.len()),
|
||||
);
|
||||
return NumaSetup {
|
||||
pools,
|
||||
cpus_per_node: nodes,
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// UMA fallback: single synthetic node, all cores, no pool, no pinning.
|
||||
let n_cores = obisys::effective_parallelism();
|
||||
debug!("UMA: single synthetic node, {} core(s)", n_cores);
|
||||
NumaSetup {
|
||||
pools: vec![None],
|
||||
cpus_per_node: vec![(0..n_cores).collect()],
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "numa"))]
|
||||
pub fn build() -> NumaSetup {
|
||||
let n_cores = obisys::effective_parallelism();
|
||||
debug!("UMA: single synthetic node, {} core(s)", n_cores);
|
||||
NumaSetup {
|
||||
pools: vec![None],
|
||||
cpus_per_node: vec![(0..n_cores).collect()],
|
||||
}
|
||||
}
|
||||
|
||||
/// Bind the calling thread to `cpu_indices` using hwloc.
|
||||
/// Silently returns on any error so the thread still runs, just unbound.
|
||||
#[cfg(feature = "numa")]
|
||||
pub fn pin_current_thread(cpu_indices: &[usize]) {
|
||||
let Ok(topology) = Topology::new() else {
|
||||
return;
|
||||
};
|
||||
let mut cpuset = CpuSet::new();
|
||||
for &idx in cpu_indices {
|
||||
cpuset.set(idx);
|
||||
}
|
||||
let _ = topology.bind_cpu(&cpuset, CpuBindingFlags::THREAD);
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "numa"))]
|
||||
pub fn pin_current_thread(_cpu_indices: &[usize]) {}
|
||||
|
||||
// ── Internal helpers ──────────────────────────────────────────────────────────
|
||||
|
||||
#[cfg(feature = "numa")]
|
||||
fn build_pool(cpus: &[usize]) -> Option<rayon::ThreadPool> {
|
||||
let cpus = cpus.to_vec();
|
||||
rayon::ThreadPoolBuilder::new()
|
||||
.num_threads(cpus.len())
|
||||
.spawn_handler(move |thread| {
|
||||
let cpus = cpus.clone();
|
||||
std::thread::Builder::new().spawn(move || {
|
||||
pin_current_thread(&cpus);
|
||||
thread.run();
|
||||
})?;
|
||||
Ok(())
|
||||
})
|
||||
.build()
|
||||
.ok()
|
||||
}
|
||||
|
||||
// ── PartitionRunner ─────────────────────────────────────────────────────────
|
||||
|
||||
/// Growth step (fraction of a node's worker capacity added per activation
|
||||
/// event, see [`NodeActivation::grow`]).
|
||||
const GROWTH_DIVISOR: usize = 8;
|
||||
/// Minimum CPU efficiency growth to activate more workers, as a fraction of
|
||||
/// the size of the *last growth step* (e.g. `0.2` after adding 8 workers
|
||||
/// requires the next check to show at least +1.6 cores of growth — 20 % of
|
||||
/// the ~8 cores those 8 workers should contribute if the workload is truly
|
||||
/// CPU-bound). Scaling by the last step's size — not the cumulative total —
|
||||
/// keeps the bar meaningful regardless of how many workers are already
|
||||
/// active, instead of demanding an ever-larger absolute jump as the pool
|
||||
/// grows.
|
||||
const CPU_SPAWN_THRESHOLD: f64 = 0.2;
|
||||
/// Minimum I/O throughput growth (relative) to activate more workers.
|
||||
const IO_SPAWN_THRESHOLD: f64 = 0.2;
|
||||
|
||||
struct NodeConfig {
|
||||
pool: Option<Arc<rayon::ThreadPool>>,
|
||||
cpu_ids: Vec<usize>,
|
||||
max_workers: usize,
|
||||
}
|
||||
|
||||
/// Generic NUMA-aware runner for partition-level parallel work.
|
||||
///
|
||||
/// Workers are distributed evenly across NUMA nodes and pinned to their
|
||||
/// node's CPUs. UMA is the degenerate case: one node, no pinning.
|
||||
///
|
||||
/// Workers are pre-spawned dormant, one activation channel per node so
|
||||
/// growth always targets a specific node rather than whichever dormant
|
||||
/// worker happens to wake up first on a shared channel. Growth (both the
|
||||
/// initial count and each subsequent step) is expressed as a fraction of
|
||||
/// `workers_per_node`, applied identically to every node, so the pace of
|
||||
/// ramp-up depends on node size rather than node count — a single-NUMA-node
|
||||
/// (UMA) machine ramps just as fast as an 8-node one.
|
||||
///
|
||||
/// # Termination
|
||||
///
|
||||
/// ```text
|
||||
/// drop(part_tx) → part_rx drains → workers exit → drop their result_tx
|
||||
/// drop(result_tx) → result_rx closes → controller loop exits
|
||||
/// drop(activate_txs) → dormant workers exit cleanly
|
||||
/// ```
|
||||
pub struct PartitionRunner {
|
||||
nodes: Vec<NodeConfig>,
|
||||
}
|
||||
|
||||
impl PartitionRunner {
|
||||
/// Total worker slots across all nodes.
|
||||
pub fn max_workers(&self) -> usize {
|
||||
self.nodes.iter().map(|n| n.max_workers).sum()
|
||||
}
|
||||
|
||||
/// Detect topology and build. Always succeeds.
|
||||
pub fn new() -> Self {
|
||||
let ns = build();
|
||||
let wpn = ns.workers_per_node();
|
||||
debug!(
|
||||
"PartitionRunner: {} node(s) × {} worker(s)/node max",
|
||||
ns.pools.len(),
|
||||
wpn,
|
||||
);
|
||||
let nodes = ns
|
||||
.pools
|
||||
.into_iter()
|
||||
.zip(ns.cpus_per_node)
|
||||
.map(|(pool, cpu_ids)| NodeConfig {
|
||||
pool,
|
||||
cpu_ids,
|
||||
max_workers: wpn,
|
||||
})
|
||||
.collect();
|
||||
Self { nodes }
|
||||
}
|
||||
|
||||
/// Run `f(i)` for every index in `order`.
|
||||
///
|
||||
/// Workers are pre-spawned dormant and activated adaptively, per node:
|
||||
/// `(workers_per_node / INITIAL_DIVISOR).max(1)` are woken immediately on
|
||||
/// every node, then `(workers_per_node / GROWTH_DIVISOR).max(1)` more per
|
||||
/// node each time the check below fires. A timer thread fires that check
|
||||
/// every `TIMER_SECS` seconds; each completed partition resets that timer
|
||||
/// (forcing an immediate check) and also triggers its own inline check. A
|
||||
/// growth step happens whenever CPU efficiency grows by at least
|
||||
/// `CPU_SPAWN_THRESHOLD` of what the last growth step should have
|
||||
/// contributed, or I/O throughput grows by at least `IO_SPAWN_THRESHOLD`
|
||||
/// (relative) since the last check — whichever resource is the actual
|
||||
/// bottleneck still shows headroom.
|
||||
///
|
||||
/// `on_done(i, result, elapsed)` is called from the controller thread as
|
||||
/// each partition completes — suitable for progress bars and result
|
||||
/// aggregation.
|
||||
///
|
||||
/// Returns the first error produced by `f`, if any.
|
||||
pub fn run<F, R, E, C>(&self, order: &[usize], f: F, mut on_done: C) -> Result<(), E>
|
||||
where
|
||||
F: Fn(usize) -> Result<R, E> + Send + Sync,
|
||||
R: Send,
|
||||
E: Send,
|
||||
C: FnMut(usize, R, Duration) + Send,
|
||||
{
|
||||
let n_total = order.len();
|
||||
if n_total == 0 {
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
const TIMER_SECS: u64 = 30;
|
||||
const INITIAL_DIVISOR: usize = 4;
|
||||
|
||||
// ── Channels ──────────────────────────────────────────────────────────
|
||||
let (part_tx, part_rx) = unbounded::<usize>();
|
||||
// reset_tx: controller → timer ("reset the 30 s window")
|
||||
let (reset_tx, reset_rx) = unbounded::<()>();
|
||||
// event_tx: workers + timer → controller (unified event stream)
|
||||
let (event_tx, event_rx) = unbounded::<WorkerEvent<R, E>>();
|
||||
// One activation channel per node: growth always targets a specific
|
||||
// node, rather than whichever dormant worker happens to win the race
|
||||
// on a channel shared across all nodes.
|
||||
let (activate_txs, activate_rxs): (Vec<_>, Vec<_>) =
|
||||
(0..self.nodes.len()).map(|_| unbounded::<()>()).unzip();
|
||||
|
||||
for &i in order {
|
||||
part_tx.send(i).ok();
|
||||
}
|
||||
drop(part_tx);
|
||||
|
||||
let max_workers = self.max_workers();
|
||||
let node_caps: Vec<usize> = self.nodes.iter().map(|n| n.max_workers).collect();
|
||||
let f = &f;
|
||||
|
||||
let mut first_err: Option<E> = None;
|
||||
|
||||
std::thread::scope(|s| {
|
||||
// ── Timer thread ──────────────────────────────────────────────────
|
||||
// Sends TimerTick every TIMER_SECS seconds. Resets its window each
|
||||
// time reset_rx receives a message (i.e. on partition completion).
|
||||
let timer_tx = event_tx.clone();
|
||||
s.spawn(move || {
|
||||
let period = Duration::from_secs(TIMER_SECS);
|
||||
loop {
|
||||
crossbeam_channel::select! {
|
||||
recv(reset_rx) -> r => {
|
||||
if r.is_err() { break; } // reset_tx dropped → exit
|
||||
}
|
||||
default(period) => {
|
||||
if timer_tx.send(WorkerEvent::TimerTick).is_err() { break; }
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
// ── Pre-spawn workers dormant, grouped by node ────────────────────
|
||||
// Each worker listens on its own node's activation channel only.
|
||||
for (node, arx) in self.nodes.iter().zip(activate_rxs.iter()) {
|
||||
let cpu_ids = &node.cpu_ids;
|
||||
for _ in 0..node.max_workers {
|
||||
let prx = part_rx.clone();
|
||||
let etx = event_tx.clone();
|
||||
let arx = arx.clone();
|
||||
let pool = node.pool.clone();
|
||||
|
||||
s.spawn(move || {
|
||||
if arx.recv().is_err() {
|
||||
return;
|
||||
}
|
||||
if !cpu_ids.is_empty() {
|
||||
pin_current_thread(cpu_ids);
|
||||
}
|
||||
for i in &prx {
|
||||
let t = Instant::now();
|
||||
let r = match &pool {
|
||||
Some(p) => p.install(|| f(i)),
|
||||
None => f(i),
|
||||
};
|
||||
etx.send(WorkerEvent::Completed(i, r, t.elapsed())).ok();
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
// Drop controller's event_tx: event_rx closes when all workers +
|
||||
// timer have exited.
|
||||
drop(event_tx);
|
||||
|
||||
// ── Controller ────────────────────────────────────────────────────
|
||||
let mut activation = NodeActivation::new(&activate_txs, &node_caps, max_workers);
|
||||
activation.activate_initial(INITIAL_DIVISOR, n_total);
|
||||
|
||||
let mut cpu_sample = CpuSample::now();
|
||||
let mut io_sample = IoSample::now();
|
||||
let mut completed = 0usize;
|
||||
|
||||
while completed < n_total {
|
||||
let Ok(event) = event_rx.recv() else { break };
|
||||
match event {
|
||||
WorkerEvent::Completed(i, r, dur) => {
|
||||
match r {
|
||||
Ok(v) => on_done(i, v, dur),
|
||||
Err(e) => {
|
||||
if first_err.is_none() {
|
||||
first_err = Some(e);
|
||||
}
|
||||
}
|
||||
}
|
||||
completed += 1;
|
||||
// Reset the 30 s timer.
|
||||
reset_tx.send(()).ok();
|
||||
// Inline check: same logic as a timer tick.
|
||||
maybe_activate(
|
||||
&mut activation,
|
||||
&mut cpu_sample,
|
||||
&mut io_sample,
|
||||
completed,
|
||||
n_total,
|
||||
);
|
||||
}
|
||||
WorkerEvent::TimerTick => {
|
||||
maybe_activate(
|
||||
&mut activation,
|
||||
&mut cpu_sample,
|
||||
&mut io_sample,
|
||||
completed,
|
||||
n_total,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Dormant workers exit once every sender for their node's channel
|
||||
// is dropped — `activate_txs` holds the only ones.
|
||||
drop(activate_txs);
|
||||
// Timer thread exits when reset_tx closes.
|
||||
drop(reset_tx);
|
||||
});
|
||||
|
||||
match first_err {
|
||||
Some(e) => Err(e),
|
||||
None => Ok(()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── Internal event type ───────────────────────────────────────────────────────
|
||||
|
||||
enum WorkerEvent<R, E> {
|
||||
Completed(usize, Result<R, E>, Duration),
|
||||
TimerTick,
|
||||
}
|
||||
|
||||
/// Tracks how many of each node's dormant workers have been woken, and
|
||||
/// grows every node by the same amount at each step (capped by that node's
|
||||
/// remaining dormant workers and by the run's total budget) so load stays
|
||||
/// balanced across nodes at every point in time — never just "one more
|
||||
/// worker somewhere". Also remembers the size of the last real growth step
|
||||
/// (`last_step`), used to scale the CPU activation threshold to what that
|
||||
/// step could plausibly have contributed (see `maybe_activate`).
|
||||
struct NodeActivation<'a> {
|
||||
txs: &'a [crossbeam_channel::Sender<()>],
|
||||
caps: &'a [usize],
|
||||
active: Vec<usize>,
|
||||
total: usize,
|
||||
max: usize,
|
||||
last_step: usize,
|
||||
}
|
||||
|
||||
impl<'a> NodeActivation<'a> {
|
||||
fn new(txs: &'a [crossbeam_channel::Sender<()>], caps: &'a [usize], max: usize) -> Self {
|
||||
Self {
|
||||
txs,
|
||||
caps,
|
||||
active: vec![0; txs.len()],
|
||||
total: 0,
|
||||
max,
|
||||
last_step: 0,
|
||||
}
|
||||
}
|
||||
|
||||
fn total(&self) -> usize {
|
||||
self.total
|
||||
}
|
||||
fn last_step(&self) -> usize {
|
||||
self.last_step
|
||||
}
|
||||
fn max(&self) -> usize {
|
||||
self.max
|
||||
}
|
||||
fn is_full(&self) -> bool {
|
||||
self.total >= self.max
|
||||
}
|
||||
|
||||
/// Wake up to `(node_cap / divisor).max(1)` dormant workers on every
|
||||
/// node, capped by `n_total`. Called once at startup, unconditionally.
|
||||
fn activate_initial(&mut self, divisor: usize, n_total: usize) {
|
||||
self.grow(divisor, n_total);
|
||||
}
|
||||
|
||||
/// Same per-node sizing as [`activate_initial`](Self::activate_initial),
|
||||
/// applied as a growth step. Returns the number of workers actually
|
||||
/// activated (may be less than requested once a node or the total
|
||||
/// budget is exhausted). Updates `last_step` when it actually grew.
|
||||
fn grow(&mut self, divisor: usize, n_total: usize) -> usize {
|
||||
let before = self.total;
|
||||
for idx in 0..self.txs.len() {
|
||||
let wanted = (self.caps[idx] / divisor).max(1);
|
||||
let room = self.caps[idx].saturating_sub(self.active[idx]);
|
||||
let grow = wanted.min(room).min(n_total.saturating_sub(self.total));
|
||||
for _ in 0..grow {
|
||||
self.txs[idx].send(()).ok();
|
||||
}
|
||||
self.active[idx] += grow;
|
||||
self.total += grow;
|
||||
}
|
||||
let grew = self.total - before;
|
||||
if grew > 0 {
|
||||
self.last_step = grew;
|
||||
}
|
||||
grew
|
||||
}
|
||||
}
|
||||
|
||||
fn maybe_activate(
|
||||
activation: &mut NodeActivation,
|
||||
cpu_sample: &mut CpuSample,
|
||||
io_sample: &mut IoSample,
|
||||
completed: usize,
|
||||
n_total: usize,
|
||||
) {
|
||||
if activation.is_full() || completed >= n_total {
|
||||
return;
|
||||
}
|
||||
|
||||
// Expect roughly 1 core of extra efficiency per worker activated in the
|
||||
// last growth step (CPU-bound case); require at least CPU_SPAWN_THRESHOLD
|
||||
// (20 %) of that expected gain before growing again. Scaling by the last
|
||||
// step's size — not the cumulative total — keeps the bar meaningful
|
||||
// regardless of how many workers are already active: growing by 8 should
|
||||
// always take ~+1.6 cores to confirm, whether that's the 2nd growth step
|
||||
// or the 20th.
|
||||
let cpu_threshold = CPU_SPAWN_THRESHOLD * activation.last_step() as f64;
|
||||
|
||||
// Call both unconditionally (no `||` short-circuit): each sampler must
|
||||
// advance its own window every tick, regardless of what the other one
|
||||
// reports, or it would starve behind whichever signal fires first.
|
||||
let cpu_wants_more = cpu_sample.do_i_activate(cpu_threshold);
|
||||
let io_wants_more = io_sample.do_i_activate(IO_SPAWN_THRESHOLD * activation.last_step() as f64);
|
||||
if !(cpu_wants_more || io_wants_more) {
|
||||
return;
|
||||
}
|
||||
|
||||
let grew = activation.grow(GROWTH_DIVISOR, n_total);
|
||||
if grew > 0 {
|
||||
debug!(
|
||||
"activated {} worker(s) — {}/{} active",
|
||||
grew,
|
||||
activation.total(),
|
||||
activation.max()
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -4,7 +4,6 @@ use std::path::Path;
|
||||
|
||||
use obikpartitionner::{KmerFilter, KmerPartition, MergeMode};
|
||||
use obisys::{Reporter, Stage, progress_bar};
|
||||
use rayon::prelude::*;
|
||||
use tracing::info;
|
||||
|
||||
use crate::error::{OKIError, OKIResult};
|
||||
@@ -83,30 +82,25 @@ impl KmerIndex {
|
||||
let src_partition = &src.partition;
|
||||
let block_bits = meta.config.block_bits;
|
||||
|
||||
let errors: Vec<obiskio::SKError> = (0..n_partitions)
|
||||
.into_par_iter()
|
||||
.filter_map(|i| {
|
||||
let result = dst_partition
|
||||
.rebuild_partition(src_partition, i, filters, mode, n_genomes, block_bits)
|
||||
.err();
|
||||
pb.inc(1);
|
||||
result
|
||||
})
|
||||
.collect();
|
||||
let order: Vec<usize> = (0..n_partitions).collect();
|
||||
let runner = crate::numa::PartitionRunner::new();
|
||||
runner.run(
|
||||
&order,
|
||||
|i| dst_partition.rebuild_partition(src_partition, i, filters, mode, n_genomes, block_bits),
|
||||
|_, _, _| { pb.inc(1); },
|
||||
).map_err(OKIError::Partition)?;
|
||||
|
||||
pb.finish_and_clear();
|
||||
|
||||
if let Some(e) = errors.into_iter().next() {
|
||||
return Err(OKIError::Partition(e));
|
||||
}
|
||||
|
||||
rep.push(t.stop());
|
||||
|
||||
// Write SENTINEL_INDEXED — output is ready to use.
|
||||
fs::File::create(output.join(SENTINEL_INDEXED))?;
|
||||
|
||||
let idx = KmerIndex::open(output)?;
|
||||
let t_pack = Stage::start("pack");
|
||||
idx.pack_matrices()?;
|
||||
rep.push(t_pack.stop());
|
||||
Ok(idx)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,7 +3,6 @@ use std::path::Path;
|
||||
use obilayeredmap::{IndexMode, layer::Layer};
|
||||
use obilayeredmap::meta::PartitionMeta;
|
||||
use obisys::{Reporter, Stage, progress_bar};
|
||||
use rayon::prelude::*;
|
||||
use tracing::info;
|
||||
|
||||
use crate::error::{OKIError, OKIResult};
|
||||
@@ -45,25 +44,17 @@ impl KmerIndex {
|
||||
let t = Stage::start("reindex");
|
||||
let pb = progress_bar("reindex", n as u64, "partitions");
|
||||
|
||||
let errors: Vec<String> = (0..n)
|
||||
.into_par_iter()
|
||||
.filter_map(|i| {
|
||||
let res = reindex_partition(
|
||||
&self.partition.part_dir(i).join("index"),
|
||||
&target,
|
||||
block_bits,
|
||||
);
|
||||
pb.inc(1);
|
||||
res.err().map(|e| format!("partition {i}: {e}"))
|
||||
})
|
||||
.collect();
|
||||
let order: Vec<usize> = (0..n).collect();
|
||||
let runner = crate::numa::PartitionRunner::new();
|
||||
runner.run(
|
||||
&order,
|
||||
|i| reindex_partition(&self.partition.part_dir(i).join("index"), &target, block_bits)
|
||||
.map_err(|e| OKIError::InvalidInput(format!("partition {i}: {e}"))),
|
||||
|_, _, _| { pb.inc(1); },
|
||||
)?;
|
||||
|
||||
pb.finish_and_clear();
|
||||
|
||||
if let Some(e) = errors.into_iter().next() {
|
||||
return Err(OKIError::InvalidInput(e));
|
||||
}
|
||||
|
||||
self.meta.config.evidence = target;
|
||||
if matches!(self.meta.config.evidence, IndexMode::Exact) {
|
||||
self.meta.config.block_bits = block_bits;
|
||||
|
||||
+24
-40
@@ -3,8 +3,7 @@ use std::io;
|
||||
use std::path::Path;
|
||||
|
||||
use obikpartitionner::{KmerPartition, OutputCol, PARTITIONS_SUBDIR};
|
||||
use obisys::{Stage, progress_bar};
|
||||
use rayon::prelude::*;
|
||||
use obisys::{Reporter, Stage, progress_bar};
|
||||
use tracing::info;
|
||||
|
||||
use crate::error::{OKIError, OKIResult};
|
||||
@@ -26,6 +25,7 @@ impl KmerIndex {
|
||||
threshold: u32,
|
||||
output_presence: bool,
|
||||
force: bool,
|
||||
rep: &mut Reporter,
|
||||
) -> OKIResult<Self> {
|
||||
let output = output.as_ref();
|
||||
|
||||
@@ -72,31 +72,23 @@ impl KmerIndex {
|
||||
let pb = progress_bar("select", n_partitions as u64, "partitions");
|
||||
let src_partition = &src.partition;
|
||||
|
||||
let errors: Vec<obiskio::SKError> = (0..n_partitions)
|
||||
.into_par_iter()
|
||||
.filter_map(|i| {
|
||||
let result = dst_partition.select_partition(
|
||||
src_partition, i, specs,
|
||||
n_src_genomes, threshold, output_presence,
|
||||
false,
|
||||
);
|
||||
pb.inc(1);
|
||||
result.err()
|
||||
})
|
||||
.collect();
|
||||
let order: Vec<usize> = (0..n_partitions).collect();
|
||||
let runner = crate::numa::PartitionRunner::new();
|
||||
runner.run(
|
||||
&order,
|
||||
|i| dst_partition.select_partition(src_partition, i, specs, n_src_genomes, threshold, output_presence, false),
|
||||
|_, _, _| { pb.inc(1); },
|
||||
).map_err(OKIError::Partition)?;
|
||||
|
||||
pb.finish_and_clear();
|
||||
|
||||
if let Some(e) = errors.into_iter().next() {
|
||||
return Err(OKIError::Partition(e));
|
||||
}
|
||||
|
||||
let _ = t.stop();
|
||||
rep.push(t.stop());
|
||||
|
||||
fs::File::create(output.join(SENTINEL_INDEXED))?;
|
||||
|
||||
let idx = KmerIndex::open(output)?;
|
||||
let t_pack = Stage::start("pack");
|
||||
idx.pack_matrices()?;
|
||||
rep.push(t_pack.stop());
|
||||
Ok(idx)
|
||||
}
|
||||
|
||||
@@ -108,6 +100,7 @@ impl KmerIndex {
|
||||
specs: &[OutputCol],
|
||||
threshold: u32,
|
||||
output_presence: bool,
|
||||
rep: &mut Reporter,
|
||||
) -> OKIResult<()> {
|
||||
if self.state() != IndexState::Indexed {
|
||||
return Err(OKIError::NotIndexed(self.root_path.clone()));
|
||||
@@ -116,7 +109,6 @@ impl KmerIndex {
|
||||
let n_src_genomes = self.meta.genomes.len();
|
||||
let n_partitions = self.partition.n_partitions();
|
||||
|
||||
// Open a second handle to the same path so we can borrow src and dst simultaneously.
|
||||
let src_partition = KmerPartition::open_with_config(
|
||||
&self.root_path,
|
||||
self.meta.config.kmer_size,
|
||||
@@ -132,35 +124,27 @@ impl KmerIndex {
|
||||
let t = Stage::start("select");
|
||||
let pb = progress_bar("select", n_partitions as u64, "partitions");
|
||||
|
||||
let errors: Vec<obiskio::SKError> = (0..n_partitions)
|
||||
.into_par_iter()
|
||||
.filter_map(|i| {
|
||||
let result = self.partition.select_partition(
|
||||
&src_partition, i, specs,
|
||||
n_src_genomes, threshold, output_presence,
|
||||
true,
|
||||
);
|
||||
pb.inc(1);
|
||||
result.err()
|
||||
})
|
||||
.collect();
|
||||
let partition = &self.partition;
|
||||
let order: Vec<usize> = (0..n_partitions).collect();
|
||||
let runner = crate::numa::PartitionRunner::new();
|
||||
runner.run(
|
||||
&order,
|
||||
|i| partition.select_partition(&src_partition, i, specs, n_src_genomes, threshold, output_presence, true),
|
||||
|_, _, _| { pb.inc(1); },
|
||||
).map_err(OKIError::Partition)?;
|
||||
|
||||
pb.finish_and_clear();
|
||||
rep.push(t.stop());
|
||||
|
||||
if let Some(e) = errors.into_iter().next() {
|
||||
return Err(OKIError::Partition(e));
|
||||
}
|
||||
|
||||
let _ = t.stop();
|
||||
|
||||
// Update index.meta with new genome list and with_counts flag.
|
||||
self.meta.config.with_counts = !output_presence;
|
||||
self.meta.genomes = specs.iter()
|
||||
.map(|s| GenomeInfo::new(s.label.clone()))
|
||||
.collect();
|
||||
self.meta.write(&self.root_path)?;
|
||||
|
||||
let t_pack = Stage::start("pack");
|
||||
self.pack_matrices()?;
|
||||
rep.push(t_pack.stop());
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "obikmer"
|
||||
version = "0.1.0"
|
||||
version = "1.1.42"
|
||||
edition = "2024"
|
||||
|
||||
[[bin]]
|
||||
@@ -18,7 +18,8 @@ obikrope = { path = "../obikrope" }
|
||||
obikpartitionner = { path = "../obikpartitionner" }
|
||||
obisys = { path = "../obisys" }
|
||||
obiskio = { path = "../obiskio" }
|
||||
obikindex = { path = "../obikindex" }
|
||||
obikindex = { path = "../obikindex", default-features = false }
|
||||
obitaxonomy = { path = "../obitaxonomy" }
|
||||
obilayeredmap = { path = "../obilayeredmap" }
|
||||
clap = { version = "4", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
@@ -32,4 +33,6 @@ tracing-subscriber = { version = "0.3", features = ["fmt", "env-filter"] }
|
||||
pprof = { version = "0.13", features = ["prost-codec"], optional = true }
|
||||
|
||||
[features]
|
||||
default = ["numa"]
|
||||
numa = ["obikindex/numa"]
|
||||
profiling = ["dep:pprof"]
|
||||
|
||||
+3
-66
@@ -1,9 +1,9 @@
|
||||
use std::path::PathBuf;
|
||||
use std::sync::{Arc, Condvar, Mutex};
|
||||
|
||||
use clap::Args;
|
||||
use obiread::NucPage;
|
||||
use obikseq::RoutableSuperKmer;
|
||||
use obipipeline::Throttled;
|
||||
|
||||
// ── Shared arguments ──────────────────────────────────────────────────────────
|
||||
|
||||
@@ -38,9 +38,7 @@ pub struct CommonArgs {
|
||||
#[arg(
|
||||
short = 'T',
|
||||
long,
|
||||
default_value_t = std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(1)
|
||||
default_value_t = obisys::effective_parallelism()
|
||||
)]
|
||||
pub threads: usize,
|
||||
|
||||
@@ -103,54 +101,10 @@ impl CommonArgs {
|
||||
}
|
||||
}
|
||||
|
||||
// ── Open-file throttling ──────────────────────────────────────────────────────
|
||||
|
||||
struct FileSlots {
|
||||
count: Mutex<usize>,
|
||||
condvar: Condvar,
|
||||
max: usize,
|
||||
}
|
||||
|
||||
impl FileSlots {
|
||||
fn new(max: usize) -> Self {
|
||||
Self { count: Mutex::new(0), condvar: Condvar::new(), max }
|
||||
}
|
||||
|
||||
fn acquire(&self) {
|
||||
let mut count = self.count.lock().unwrap();
|
||||
while *count >= self.max {
|
||||
count = self.condvar.wait(count).unwrap();
|
||||
}
|
||||
*count += 1;
|
||||
}
|
||||
|
||||
fn release(&self) {
|
||||
let mut count = self.count.lock().unwrap();
|
||||
*count -= 1;
|
||||
self.condvar.notify_one();
|
||||
}
|
||||
}
|
||||
|
||||
struct SlotsGuard(Arc<FileSlots>);
|
||||
|
||||
impl Drop for SlotsGuard {
|
||||
fn drop(&mut self) {
|
||||
self.0.release();
|
||||
}
|
||||
}
|
||||
|
||||
// ── Pipeline data carrier ─────────────────────────────────────────────────────
|
||||
|
||||
/// A path bundled with an opaque guard token.
|
||||
/// The guard is acquired in the source thread and dropped by the flat worker
|
||||
/// once the file is fully read, releasing the open-file slot.
|
||||
pub struct PathWithSlot {
|
||||
pub path: PathBuf,
|
||||
pub _guard: Box<dyn Send + 'static>,
|
||||
}
|
||||
|
||||
pub enum PipelineData {
|
||||
Path(PathWithSlot),
|
||||
Path(Throttled<PathBuf>),
|
||||
NucPage(NucPage),
|
||||
Batch(Vec<RoutableSuperKmer>),
|
||||
}
|
||||
@@ -158,20 +112,3 @@ pub enum PipelineData {
|
||||
unsafe impl Send for PipelineData {}
|
||||
unsafe impl Sync for PipelineData {}
|
||||
|
||||
/// Wrap a path iterator so that at most `max_open` files are open simultaneously.
|
||||
/// Acquisition happens in the caller's thread (the pipeline source thread),
|
||||
/// never inside a worker, preventing deadlocks.
|
||||
pub fn throttle_paths(
|
||||
source: impl Iterator<Item = PathBuf> + Send + 'static,
|
||||
max_open: usize,
|
||||
) -> impl Iterator<Item = PathWithSlot> + Send + 'static {
|
||||
let slots = Arc::new(FileSlots::new(max_open));
|
||||
source.map(move |path| {
|
||||
slots.acquire();
|
||||
PathWithSlot {
|
||||
path,
|
||||
_guard: Box::new(SlotsGuard(Arc::clone(&slots))),
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -3,13 +3,15 @@ use std::path::PathBuf;
|
||||
|
||||
use clap::Args;
|
||||
use kodama::{Method, linkage};
|
||||
use obikindex::{DistanceMetric, KmerIndex};
|
||||
use obifastwrite::{JsonVal, write_record};
|
||||
use obikindex::{DistanceMetric, KmerIndex, RawSnpDistanceOutput, SiblingAnnexStats, SnpAlignment};
|
||||
use speedytree::{DistanceMatrix, Hybrid, NeighborJoiningSolver, to_newick};
|
||||
use tracing::info;
|
||||
|
||||
#[derive(clap::ValueEnum, Clone, Copy, Debug)]
|
||||
pub enum MetricArg {
|
||||
Jaccard,
|
||||
Mash,
|
||||
Hamming,
|
||||
BrayCurtis,
|
||||
#[value(name = "relfreq-bray-curtis")]
|
||||
@@ -26,6 +28,7 @@ impl From<MetricArg> for DistanceMetric {
|
||||
fn from(m: MetricArg) -> Self {
|
||||
match m {
|
||||
MetricArg::Jaccard => DistanceMetric::Jaccard,
|
||||
MetricArg::Mash => DistanceMetric::Mash,
|
||||
MetricArg::Hamming => DistanceMetric::Hamming,
|
||||
MetricArg::BrayCurtis => DistanceMetric::BrayCurtis,
|
||||
MetricArg::RelfreqBrayCurtis => DistanceMetric::RelfreqBrayCurtis,
|
||||
@@ -62,7 +65,37 @@ pub struct DistanceArgs {
|
||||
#[arg(long)]
|
||||
pub upgma: bool,
|
||||
|
||||
/// Build the sibling-count/minorant annex on this (multi-genome) index
|
||||
/// — see `docmd/theory/evolutionary_distances.md`, Step 2b. Construction
|
||||
/// only; does not by itself compute or write any statistics.
|
||||
#[arg(long)]
|
||||
pub sibling_annex: bool,
|
||||
|
||||
/// Tally the sibling-count distribution (CSV) of an already-built annex
|
||||
/// (run with `--sibling-annex` first, in this invocation or an earlier
|
||||
/// one). A separate, occasional diagnostic pass — not run every time the
|
||||
/// annex itself is (re)built.
|
||||
#[arg(long)]
|
||||
pub sibling_stats: bool,
|
||||
|
||||
/// Compute the raw p-distance restricted to loci that are single-copy
|
||||
/// in both genomes of each pair (an already-built sibling annex is
|
||||
/// required — run with `--sibling-annex` first, in this invocation or
|
||||
/// an earlier one). A quick way to test the central-position SNP
|
||||
/// estimator against a real index; not the full `SnpTally` design.
|
||||
#[arg(long)]
|
||||
pub raw_snp_distance: bool,
|
||||
|
||||
/// Write a SNP-only pseudo-alignment (FASTA, IUPAC-coded) from an
|
||||
/// already-built sibling annex — one row per genome, one column per
|
||||
/// variable family (monomorphic families skipped), no flanking
|
||||
/// sequence. See `docmd/theory/evolutionary_distances.md`,
|
||||
/// "Multi-genome framing: family as pseudo-alignment column".
|
||||
#[arg(long)]
|
||||
pub snp: bool,
|
||||
|
||||
/// Output prefix: <prefix>_dist.csv, <prefix>_shared.csv,
|
||||
/// <prefix>_siblings.csv, <prefix>_rawsnp.csv, <prefix>_snp.fasta,
|
||||
/// <prefix>_nj.nwk, <prefix>_upgma.nwk.
|
||||
/// If omitted, the distance matrix is written to stdout.
|
||||
#[arg(short, long)]
|
||||
@@ -78,6 +111,51 @@ pub fn run(args: DistanceArgs) {
|
||||
|
||||
let labels: Vec<String> = idx.meta().genomes.iter().map(|g| g.label.clone()).collect();
|
||||
let n = labels.len();
|
||||
|
||||
// ── Sibling-count/minorant annex (independent of the distance metric) ──
|
||||
// Construction (`--sibling-annex`) and stats (`--sibling-stats`) are
|
||||
// deliberately decoupled: the annex is meant to be (re)built routinely,
|
||||
// the distribution only occasionally, on demand.
|
||||
if args.sibling_annex {
|
||||
info!("building sibling-count/minorant annex");
|
||||
idx.build_sibling_annex().unwrap_or_else(|e| {
|
||||
eprintln!("error building sibling annex: {e}");
|
||||
std::process::exit(1);
|
||||
});
|
||||
}
|
||||
if args.sibling_stats {
|
||||
let stats = idx.sibling_annex_stats().unwrap_or_else(|e| {
|
||||
eprintln!("error computing sibling-annex stats: {e}");
|
||||
std::process::exit(1);
|
||||
});
|
||||
write_sibling_stats_csv(&stats, &labels, &args.output);
|
||||
}
|
||||
if args.raw_snp_distance {
|
||||
let result = idx.raw_snp_distance().unwrap_or_else(|e| {
|
||||
eprintln!("error computing raw SNP distance: {e}");
|
||||
std::process::exit(1);
|
||||
});
|
||||
write_raw_snp_distance_csv(&result, &labels, &args.output);
|
||||
}
|
||||
if args.snp {
|
||||
let alignment = idx.snp_pseudo_alignment().unwrap_or_else(|e| {
|
||||
eprintln!("error computing SNP pseudo-alignment: {e}");
|
||||
std::process::exit(1);
|
||||
});
|
||||
write_snp_fasta(&alignment, &labels, &args.output);
|
||||
}
|
||||
|
||||
// `--sibling-annex`/`--sibling-stats`/`--raw-snp-distance`/`--snp` are
|
||||
// their own operation, not a modifier on top of a distance-metric
|
||||
// computation — a metric was never requested by asking for any of them,
|
||||
// so there is nothing for the rest of this function to compute. Not a
|
||||
// historical accident to keep: stop here rather than always also
|
||||
// running a Jaccard (or whichever `--metric` defaults to) pass and
|
||||
// printing an unrequested matrix.
|
||||
if args.sibling_annex || args.sibling_stats || args.raw_snp_distance || args.snp {
|
||||
return;
|
||||
}
|
||||
|
||||
info!(
|
||||
"computing {:?} distances for {} genome(s)",
|
||||
args.metric, n
|
||||
@@ -189,6 +267,103 @@ pub fn run(args: DistanceArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
// ── Family-size distribution → CSV ──────────────────────────────────────────
|
||||
//
|
||||
// Each row is a family (the up-to-4 k-mers sharing flanks, differing only at
|
||||
// the centre), counted once — at its minorant — regardless of how many of
|
||||
// its members are observed. Family size 1..4 (not "sibling count" 0..3):
|
||||
// see `docmd/theory/evolutionary_distances.md`, "Definitions".
|
||||
|
||||
fn write_sibling_stats_csv(stats: &SiblingAnnexStats, labels: &[String], output: &Option<PathBuf>) {
|
||||
// One row per genome (4 columns, family size 1-4: number of families of
|
||||
// that size for which the genome carries at least one member), plus a
|
||||
// `global` row — the actual deduplicated family-size histogram
|
||||
// (`stats.counts`), NOT a sum of the per-genome columns (a family shared
|
||||
// by several genomes would otherwise be counted once per genome it
|
||||
// appears in, inflating the total beyond the real family count).
|
||||
let path = output.as_ref()
|
||||
.map(|p| format!("{}_siblings.csv", p.display()))
|
||||
.unwrap_or_else(|| "siblings.csv".into());
|
||||
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
|
||||
eprintln!("error creating {path}: {e}");
|
||||
std::process::exit(1);
|
||||
}));
|
||||
writeln!(f, "genome,1,2,3,4").unwrap();
|
||||
for (label, counts) in labels.iter().zip(stats.per_genome.iter()) {
|
||||
writeln!(f, "{label},{},{},{},{}", counts[0], counts[1], counts[2], counts[3]).unwrap();
|
||||
}
|
||||
writeln!(
|
||||
f, "global,{},{},{},{}",
|
||||
stats.counts[0], stats.counts[1], stats.counts[2], stats.counts[3],
|
||||
).unwrap();
|
||||
let total: u64 = stats.counts.iter().sum();
|
||||
info!("family-size distribution → {path} (total {total} famil{})",
|
||||
if total == 1 { "y" } else { "ies" });
|
||||
}
|
||||
|
||||
// ── Raw single-copy SNP distance → CSV ──────────────────────────────────────
|
||||
//
|
||||
// p_hat[i,j] = snp[i,j] / (snp[i,j] + shared[i,j]) over loci single-copy in
|
||||
// both i and j — see `RawSnpDistanceOutput` / `KmerIndex::raw_snp_distance`.
|
||||
// A single file: the distance matrix, with an eligible-loci count alongside
|
||||
// each value so a 0/0 pair (no eligible locus at all) is distinguishable
|
||||
// from a genuinely identical pair.
|
||||
|
||||
fn write_raw_snp_distance_csv(result: &RawSnpDistanceOutput, labels: &[String], output: &Option<PathBuf>) {
|
||||
let path = output.as_ref()
|
||||
.map(|p| format!("{}_rawsnp.csv", p.display()))
|
||||
.unwrap_or_else(|| "rawsnp.csv".into());
|
||||
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
|
||||
eprintln!("error creating {path}: {e}");
|
||||
std::process::exit(1);
|
||||
}));
|
||||
let n = labels.len();
|
||||
write!(f, "genome").unwrap();
|
||||
for g in labels { write!(f, ",{g}").unwrap(); }
|
||||
writeln!(f).unwrap();
|
||||
for (i, g) in labels.iter().enumerate() {
|
||||
write!(f, "{g}").unwrap();
|
||||
for j in 0..n {
|
||||
let snp = result.snp[[i, j]];
|
||||
let shared = result.shared[[i, j]];
|
||||
let eligible = snp + shared;
|
||||
if eligible == 0 {
|
||||
write!(f, ",NA").unwrap();
|
||||
} else {
|
||||
write!(f, ",{:.6}", snp as f64 / eligible as f64).unwrap();
|
||||
}
|
||||
}
|
||||
writeln!(f).unwrap();
|
||||
}
|
||||
info!("raw single-copy SNP distance matrix → {path}");
|
||||
}
|
||||
|
||||
// ── SNP-only pseudo-alignment → FASTA ───────────────────────────────────────
|
||||
//
|
||||
// One record per genome, IUPAC-coded, no flanking sequence — see
|
||||
// `SnpAlignment` / `KmerIndex::snp_pseudo_alignment`. Uses the project's
|
||||
// existing FASTA writer (`obifastwrite::write_record`) rather than
|
||||
// hand-rolling one.
|
||||
|
||||
fn write_snp_fasta(alignment: &SnpAlignment, labels: &[String], output: &Option<PathBuf>) {
|
||||
let path = output.as_ref()
|
||||
.map(|p| format!("{}_snp.fasta", p.display()))
|
||||
.unwrap_or_else(|| "snp.fasta".into());
|
||||
let mut f = BufWriter::new(std::fs::File::create(&path).unwrap_or_else(|e| {
|
||||
eprintln!("error creating {path}: {e}");
|
||||
std::process::exit(1);
|
||||
}));
|
||||
let n_sites = alignment.sequences.first().map(|s| s.len()).unwrap_or(0);
|
||||
for (label, seq) in labels.iter().zip(alignment.sequences.iter()) {
|
||||
write_record(seq, label, &[("n_sites", JsonVal::Num(n_sites as u64))], &mut f).unwrap_or_else(|e| {
|
||||
eprintln!("error writing {path}: {e}");
|
||||
std::process::exit(1);
|
||||
});
|
||||
}
|
||||
info!("SNP pseudo-alignment → {path} ({n_sites} site{})",
|
||||
if n_sites == 1 { "" } else { "s" });
|
||||
}
|
||||
|
||||
// ── UPGMA Newick from kodama dendrogram ───────────────────────────────────────
|
||||
|
||||
fn upgma_to_newick(dendro: &kodama::Dendrogram<f64>, names: &[String]) -> String {
|
||||
|
||||
@@ -2,14 +2,14 @@ use std::path::PathBuf;
|
||||
|
||||
use clap::Args;
|
||||
use obikindex::{KmerIndex, MergeMode};
|
||||
use obikpartitionner::filter::{MaxTotalCount, MinTotalCount};
|
||||
use obikpartitionner::filter::{MaxTotalCount, MinComplexity, MinTotalCount};
|
||||
use obisys::Reporter;
|
||||
use tracing::info;
|
||||
|
||||
use super::predicate::FilterArgs as KmerFilterArgs;
|
||||
|
||||
#[derive(Args)]
|
||||
pub struct FilterArgs {
|
||||
pub struct FilterCmdArgs {
|
||||
/// Source index directory
|
||||
pub source: PathBuf,
|
||||
|
||||
@@ -28,6 +28,18 @@ pub struct FilterArgs {
|
||||
#[arg(long)]
|
||||
pub max_total_count: Option<u32>,
|
||||
|
||||
/// Minimum normalized entropy (complexity) to keep a k-mer — same metric
|
||||
/// as `obikmer index`'s --theta, applied here to k-mers already committed
|
||||
/// to the source index (reconstructed from unitigs.bin). K-mers scoring
|
||||
/// below this are removed.
|
||||
#[arg(long)]
|
||||
pub min_complexity: Option<f64>,
|
||||
|
||||
/// Maximum sub-word size for the complexity computation (see `obikmer
|
||||
/// index`'s --level-max). Only used when --min-complexity is set.
|
||||
#[arg(long, default_value_t = 6)]
|
||||
pub complexity_level_max: usize,
|
||||
|
||||
/// Output as presence/absence instead of counts
|
||||
#[arg(long)]
|
||||
pub presence: bool,
|
||||
@@ -37,7 +49,7 @@ pub struct FilterArgs {
|
||||
pub force: bool,
|
||||
}
|
||||
|
||||
pub fn run(args: FilterArgs) {
|
||||
pub fn run(args: FilterCmdArgs) {
|
||||
let src = KmerIndex::open(&args.source).unwrap_or_else(|e| {
|
||||
eprintln!("error opening source index: {e}");
|
||||
std::process::exit(1);
|
||||
@@ -62,6 +74,9 @@ pub fn run(args: FilterArgs) {
|
||||
if let Some(v) = args.max_total_count {
|
||||
filters.push(Box::new(MaxTotalCount { total: v }));
|
||||
}
|
||||
if let Some(theta) = args.min_complexity {
|
||||
filters.push(Box::new(MinComplexity { level_max: args.complexity_level_max, theta }));
|
||||
}
|
||||
|
||||
let mut rep = Reporter::new();
|
||||
KmerIndex::rebuild(&args.output, &src, &filters, mode, args.force, &mut rep)
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user