Remove precomputed data assets, configurations, and documentation

This change removes precomputed model parameters, phylogenetic tree datasets, k-mer spectrum data, and compressed profile archives. It also deletes runtime logs and detailed pipeline documentation. The repository ignore list is updated to exclude the sandbox directory.
This commit is contained in:
Eric Coissac
2026-08-20 13:15:56 +02:00
parent 69747dcb53
commit 3b65319529
13 changed files with 1 additions and 5971 deletions
+1
View File
@@ -5,6 +5,7 @@
.serena/
.zed/
memory/
sandbox/
src/target
data-stress
*.fasta
-167
View File
@@ -1,167 +0,0 @@
ratio_ceiling: 0.5
cardinality_transitions:
- from: 0
to: 0
count: 122014779
probability: 0.870049812090934
- from: 0
to: 1
count: 17966272
probability: 0.12811195254940888
- from: 0
to: 2
count: 233683
probability: 0.001666321505518981
- from: 0
to: 3
count: 24109
probability: 0.00017191385413811494
- from: 0
to: 4
count: 0
probability: 0.0
- from: 1
to: 0
count: 17966272
probability: 0.967023782579572
- from: 1
to: 1
count: 602798
probability: 0.03244523972983381
- from: 1
to: 2
count: 9562
probability: 0.0005146688978673966
- from: 1
to: 3
count: 303
probability: 0.00001630879272681669
- from: 1
to: 4
count: 0
probability: 0.0
- from: 2
to: 0
count: 233683
probability: 0.9585500516842502
- from: 2
to: 1
count: 9562
probability: 0.039222603245442765
- from: 2
to: 2
count: 438
probability: 0.0017966429848885097
- from: 2
to: 3
count: 105
probability: 0.00043070208541847837
- from: 2
to: 4
count: 0
probability: 0.0
- from: 3
to: 0
count: 24109
probability: 0.9827171564831044
- from: 3
to: 1
count: 303
probability: 0.012350711286838137
- from: 3
to: 2
count: 105
probability: 0.004279949455834998
- from: 3
to: 3
count: 16
probability: 0.0006521827742224758
- from: 3
to: 4
count: 0
probability: 0.0
- from: 4
to: 0
count: 0
probability: 0.0
- from: 4
to: 1
count: 0
probability: 0.0
- from: 4
to: 2
count: 0
probability: 0.0
- from: 4
to: 3
count: 0
probability: 0.0
- from: 4
to: 4
count: 0
probability: 0.0
composition_transitions:
- from: 'A'
to: 'A'
count: 0
probability: 0.0
- from: 'A'
to: 'C'
count: 0
probability: 0.0
- from: 'A'
to: 'G'
count: 0
probability: 0.0
- from: 'A'
to: 'T'
count: 0
probability: 0.0
- from: 'C'
to: 'A'
count: 0
probability: 0.0
- from: 'C'
to: 'C'
count: 96389
probability: 0.9663832688335907
- from: 'C'
to: 'G'
count: 1290
probability: 0.012933368089671353
- from: 'C'
to: 'T'
count: 2063
probability: 0.020683363076737984
- from: 'G'
to: 'A'
count: 0
probability: 0.0
- from: 'G'
to: 'C'
count: 1290
probability: 0.005696948820201646
- from: 'G'
to: 'G'
count: 222720
probability: 0.9835848381669073
- from: 'G'
to: 'T'
count: 2427
probability: 0.010718213012891003
- from: 'T'
to: 'A'
count: 0
probability: 0.0
- from: 'T'
to: 'C'
count: 2063
probability: 0.007305266661709142
- from: 'T'
to: 'G'
count: 2427
probability: 0.008594223067362136
- from: 'T'
to: 'T'
count: 277909
probability: 0.9841005102709287
-167
View File
@@ -1,167 +0,0 @@
ratio_ceiling: 0.5
cardinality_transitions:
- from: 0
to: 0
count: 13019012
probability: 0.8682962128540935
- from: 0
to: 1
count: 1945560
probability: 0.12975810913150784
- from: 0
to: 2
count: 26381
probability: 0.0017594670310852958
- from: 0
to: 3
count: 2792
probability: 0.000186210983313375
- from: 0
to: 4
count: 0
probability: 0.0
- from: 1
to: 0
count: 1945560
probability: 0.9666005228573729
- from: 1
to: 1
count: 66078
probability: 0.03282912341401421
- from: 1
to: 2
count: 1123
probability: 0.0005579331334776772
- from: 1
to: 3
count: 25
probability: 0.000012420595135300028
- from: 1
to: 4
count: 0
probability: 0.0
- from: 2
to: 0
count: 26381
probability: 0.9567346050627402
- from: 2
to: 1
count: 1123
probability: 0.04072677159643142
- from: 2
to: 2
count: 51
probability: 0.0018495684340320592
- from: 2
to: 3
count: 19
probability: 0.0006890549067962574
- from: 2
to: 4
count: 0
probability: 0.0
- from: 3
to: 0
count: 2792
probability: 0.9841381741275996
- from: 3
to: 1
count: 25
probability: 0.008812125484666901
- from: 3
to: 2
count: 19
probability: 0.0066972153683468455
- from: 3
to: 3
count: 1
probability: 0.00035248501938667606
- from: 3
to: 4
count: 0
probability: 0.0
- from: 4
to: 0
count: 0
probability: 0.0
- from: 4
to: 1
count: 0
probability: 0.0
- from: 4
to: 2
count: 0
probability: 0.0
- from: 4
to: 3
count: 0
probability: 0.0
- from: 4
to: 4
count: 0
probability: 0.0
composition_transitions:
- from: 'A'
to: 'A'
count: 0
probability: 0.0
- from: 'A'
to: 'C'
count: 0
probability: 0.0
- from: 'A'
to: 'G'
count: 0
probability: 0.0
- from: 'A'
to: 'T'
count: 0
probability: 0.0
- from: 'C'
to: 'A'
count: 0
probability: 0.0
- from: 'C'
to: 'C'
count: 10771
probability: 0.9619540948468339
- from: 'C'
to: 'G'
count: 163
probability: 0.014557470751094042
- from: 'C'
to: 'T'
count: 263
probability: 0.023488434402071982
- from: 'G'
to: 'A'
count: 0
probability: 0.0
- from: 'G'
to: 'C'
count: 163
probability: 0.006662851536952256
- from: 'G'
to: 'G'
count: 24073
probability: 0.9840173315892741
- from: 'G'
to: 'T'
count: 228
probability: 0.009319816873773708
- from: 'T'
to: 'A'
count: 0
probability: 0.0
- from: 'T'
to: 'C'
count: 263
probability: 0.008464484567603231
- from: 'T'
to: 'G'
count: 228
probability: 0.007338032248720672
- from: 'T'
to: 'T'
count: 30580
probability: 0.9841974831836761
-275
View File
@@ -1,275 +0,0 @@
2026-04-27T20:36:04.208955Z  INFO obikmer::cmd::partition: dereplicating...
2026-04-27T20:36:11.471489Z  INFO obikpartitionner::partition: counting kmers in partition 0/256
2026-04-27T20:36:11.471521Z  INFO obikpartitionner::partition: counting kmers in partition 128/256
2026-04-27T20:36:11.471530Z  INFO obikpartitionner::partition: counting kmers in partition 32/256
2026-04-27T20:36:11.471535Z  INFO obikpartitionner::partition: counting kmers in partition 12/256
2026-04-27T20:36:11.471551Z  INFO obikpartitionner::partition: counting kmers in partition 76/256
2026-04-27T20:36:11.471522Z  INFO obikpartitionner::partition: counting kmers in partition 2/256
2026-04-27T20:36:11.471525Z  INFO obikpartitionner::partition: counting kmers in partition 4/256
2026-04-27T20:36:11.471535Z  INFO obikpartitionner::partition: counting kmers in partition 96/256
2026-04-27T20:36:11.471535Z  INFO obikpartitionner::partition: counting kmers in partition 80/256
2026-04-27T20:36:11.471535Z  INFO obikpartitionner::partition: counting kmers in partition 10/256
2026-04-27T20:36:11.471538Z  INFO obikpartitionner::partition: counting kmers in partition 16/256
2026-04-27T20:36:11.471543Z  INFO obikpartitionner::partition: counting kmers in partition 224/256
2026-04-27T20:36:11.471537Z  INFO obikpartitionner::partition: counting kmers in partition 192/256
2026-04-27T20:36:11.471536Z  INFO obikpartitionner::partition: counting kmers in partition 64/256
2026-04-27T20:36:11.471537Z  INFO obikpartitionner::partition: counting kmers in partition 72/256
2026-04-27T20:36:11.471530Z  INFO obikpartitionner::partition: counting kmers in partition 8/256
2026-04-27T20:36:12.831739Z  INFO obikpartitionner::partition: counting kmers in partition 129/256
2026-04-27T20:36:13.555104Z  INFO obikpartitionner::partition: counting kmers in partition 9/256
2026-04-27T20:36:14.077082Z  INFO obikpartitionner::partition: counting kmers in partition 17/256
2026-04-27T20:36:14.203310Z  INFO obikpartitionner::partition: counting kmers in partition 77/256
2026-04-27T20:36:14.316403Z  INFO obikpartitionner::partition: counting kmers in partition 3/256
2026-04-27T20:36:14.415177Z  INFO obikpartitionner::partition: counting kmers in partition 33/256
2026-04-27T20:36:14.467689Z  INFO obikpartitionner::partition: counting kmers in partition 97/256
2026-04-27T20:36:14.559650Z  INFO obikpartitionner::partition: counting kmers in partition 5/256
2026-04-27T20:36:14.591431Z  INFO obikpartitionner::partition: counting kmers in partition 81/256
2026-04-27T20:36:14.606508Z  INFO obikpartitionner::partition: counting kmers in partition 193/256
2026-04-27T20:36:14.611406Z  INFO obikpartitionner::partition: counting kmers in partition 73/256
2026-04-27T20:36:14.623813Z  INFO obikpartitionner::partition: counting kmers in partition 11/256
2026-04-27T20:36:14.665058Z  INFO obikpartitionner::partition: counting kmers in partition 65/256
2026-04-27T20:36:14.666877Z  INFO obikpartitionner::partition: counting kmers in partition 1/256
2026-04-27T20:36:14.667569Z  INFO obikpartitionner::partition: counting kmers in partition 13/256
2026-04-27T20:36:14.697486Z  INFO obikpartitionner::partition: counting kmers in partition 225/256
2026-04-27T20:36:14.698441Z  INFO obikpartitionner::partition: counting kmers in partition 130/256
2026-04-27T20:36:14.998749Z  INFO obikpartitionner::partition: counting kmers in partition 14/256
2026-04-27T20:36:15.485303Z  INFO obikpartitionner::partition: counting kmers in partition 18/256
2026-04-27T20:36:15.637613Z  INFO obikpartitionner::partition: counting kmers in partition 78/256
2026-04-27T20:36:15.722825Z  INFO obikpartitionner::partition: counting kmers in partition 98/256
2026-04-27T20:36:15.831703Z  INFO obikpartitionner::partition: counting kmers in partition 112/256
2026-04-27T20:36:16.161408Z  INFO obikpartitionner::partition: counting kmers in partition 66/256
2026-04-27T20:36:16.340113Z  INFO obikpartitionner::partition: counting kmers in partition 226/256
2026-04-27T20:36:16.506407Z  INFO obikpartitionner::partition: counting kmers in partition 82/256
2026-04-27T20:36:16.542430Z  INFO obikpartitionner::partition: counting kmers in partition 74/256
2026-04-27T20:36:16.550654Z  INFO obikpartitionner::partition: counting kmers in partition 34/256
2026-04-27T20:36:16.634719Z  INFO obikpartitionner::partition: counting kmers in partition 6/256
2026-04-27T20:36:16.763833Z  INFO obikpartitionner::partition: counting kmers in partition 194/256
2026-04-27T20:36:16.813285Z  INFO obikpartitionner::partition: counting kmers in partition 131/256
2026-04-27T20:36:16.913975Z  INFO obikpartitionner::partition: counting kmers in partition 208/256
2026-04-27T20:36:16.993526Z  INFO obikpartitionner::partition: counting kmers in partition 15/256
2026-04-27T20:36:17.092998Z  INFO obikpartitionner::partition: counting kmers in partition 160/256
2026-04-27T20:36:17.114964Z  INFO obikpartitionner::partition: counting kmers in partition 88/256
2026-04-27T20:36:17.148685Z  INFO obikpartitionner::partition: counting kmers in partition 79/256
2026-04-27T20:36:17.161989Z  INFO obikpartitionner::partition: counting kmers in partition 19/256
2026-04-27T20:36:17.180022Z  INFO obikpartitionner::partition: counting kmers in partition 113/256
2026-04-27T20:36:17.397037Z  INFO obikpartitionner::partition: counting kmers in partition 67/256
2026-04-27T20:36:17.484519Z  INFO obikpartitionner::partition: counting kmers in partition 99/256
2026-04-27T20:36:17.906842Z  INFO obikpartitionner::partition: counting kmers in partition 35/256
2026-04-27T20:36:18.146262Z  INFO obikpartitionner::partition: counting kmers in partition 75/256
2026-04-27T20:36:18.247130Z  INFO obikpartitionner::partition: counting kmers in partition 7/256
2026-04-27T20:36:18.303872Z  INFO obikpartitionner::partition: counting kmers in partition 227/256
2026-04-27T20:36:18.374448Z  INFO obikpartitionner::partition: counting kmers in partition 195/256
2026-04-27T20:36:18.382279Z  INFO obikpartitionner::partition: counting kmers in partition 83/256
2026-04-27T20:36:18.428092Z  INFO obikpartitionner::partition: counting kmers in partition 104/256
2026-04-27T20:36:18.442213Z  INFO obikpartitionner::partition: counting kmers in partition 132/256
2026-04-27T20:36:18.621883Z  INFO obikpartitionner::partition: counting kmers in partition 209/256
2026-04-27T20:36:18.784307Z  INFO obikpartitionner::partition: counting kmers in partition 161/256
2026-04-27T20:36:18.796465Z  INFO obikpartitionner::partition: counting kmers in partition 114/256
2026-04-27T20:36:18.877587Z  INFO obikpartitionner::partition: counting kmers in partition 120/256
2026-04-27T20:36:18.888423Z  INFO obikpartitionner::partition: counting kmers in partition 89/256
2026-04-27T20:36:18.918340Z  INFO obikpartitionner::partition: counting kmers in partition 20/256
2026-04-27T20:36:18.985341Z  INFO obikpartitionner::partition: counting kmers in partition 100/256
2026-04-27T20:36:19.080092Z  INFO obikpartitionner::partition: counting kmers in partition 68/256
2026-04-27T20:36:19.282760Z  INFO obikpartitionner::partition: counting kmers in partition 36/256
2026-04-27T20:36:19.485079Z  INFO obikpartitionner::partition: counting kmers in partition 92/256
2026-04-27T20:36:19.692354Z  INFO obikpartitionner::partition: counting kmers in partition 24/256
2026-04-27T20:36:19.746458Z  INFO obikpartitionner::partition: counting kmers in partition 228/256
2026-04-27T20:36:20.004054Z  INFO obikpartitionner::partition: counting kmers in partition 84/256
2026-04-27T20:36:20.164510Z  INFO obikpartitionner::partition: counting kmers in partition 196/256
2026-04-27T20:36:20.277288Z  INFO obikpartitionner::partition: counting kmers in partition 105/256
2026-04-27T20:36:20.349391Z  INFO obikpartitionner::partition: counting kmers in partition 133/256
2026-04-27T20:36:20.511981Z  INFO obikpartitionner::partition: counting kmers in partition 210/256
2026-04-27T20:36:20.709583Z  INFO obikpartitionner::partition: counting kmers in partition 101/256
2026-04-27T20:36:20.715870Z  INFO obikpartitionner::partition: counting kmers in partition 69/256
2026-04-27T20:36:20.728029Z  INFO obikpartitionner::partition: counting kmers in partition 115/256
2026-04-27T20:36:20.822257Z  INFO obikpartitionner::partition: counting kmers in partition 162/256
2026-04-27T20:36:20.982131Z  INFO obikpartitionner::partition: counting kmers in partition 90/256
2026-04-27T20:36:20.990122Z  INFO obikpartitionner::partition: counting kmers in partition 21/256
2026-04-27T20:36:21.030814Z  INFO obikpartitionner::partition: counting kmers in partition 121/256
2026-04-27T20:36:21.088309Z  INFO obikpartitionner::partition: counting kmers in partition 93/256
2026-04-27T20:36:21.134904Z  INFO obikpartitionner::partition: counting kmers in partition 37/256
2026-04-27T20:36:21.196814Z  INFO obikpartitionner::partition: counting kmers in partition 25/256
2026-04-27T20:36:21.444384Z  INFO obikpartitionner::partition: counting kmers in partition 85/256
2026-04-27T20:36:21.605343Z  INFO obikpartitionner::partition: counting kmers in partition 197/256
2026-04-27T20:36:21.954010Z  INFO obikpartitionner::partition: counting kmers in partition 106/256
2026-04-27T20:36:22.026167Z  INFO obikpartitionner::partition: counting kmers in partition 229/256
2026-04-27T20:36:22.194676Z  INFO obikpartitionner::partition: counting kmers in partition 134/256
2026-04-27T20:36:22.209911Z  INFO obikpartitionner::partition: counting kmers in partition 211/256
2026-04-27T20:36:22.281984Z  INFO obikpartitionner::partition: counting kmers in partition 70/256
2026-04-27T20:36:22.440637Z  INFO obikpartitionner::partition: counting kmers in partition 163/256
2026-04-27T20:36:22.481750Z  INFO obikpartitionner::partition: counting kmers in partition 91/256
2026-04-27T20:36:22.522458Z  INFO obikpartitionner::partition: counting kmers in partition 116/256
2026-04-27T20:36:22.775669Z  INFO obikpartitionner::partition: counting kmers in partition 22/256
2026-04-27T20:36:23.095766Z  INFO obikpartitionner::partition: counting kmers in partition 102/256
2026-04-27T20:36:23.119802Z  INFO obikpartitionner::partition: counting kmers in partition 94/256
2026-04-27T20:36:23.183138Z  INFO obikpartitionner::partition: counting kmers in partition 122/256
2026-04-27T20:36:23.186904Z  INFO obikpartitionner::partition: counting kmers in partition 38/256
2026-04-27T20:36:23.196250Z  INFO obikpartitionner::partition: counting kmers in partition 86/256
2026-04-27T20:36:23.206118Z  INFO obikpartitionner::partition: counting kmers in partition 26/256
2026-04-27T20:36:23.314520Z  INFO obikpartitionner::partition: counting kmers in partition 198/256
2026-04-27T20:36:23.355959Z  INFO obikpartitionner::partition: counting kmers in partition 107/256
2026-04-27T20:36:23.526146Z  INFO obikpartitionner::partition: counting kmers in partition 230/256
2026-04-27T20:36:23.723572Z  INFO obikpartitionner::partition: counting kmers in partition 212/256
2026-04-27T20:36:23.856569Z  INFO obikpartitionner::partition: counting kmers in partition 71/256
2026-04-27T20:36:23.910613Z  INFO obikpartitionner::partition: counting kmers in partition 164/256
2026-04-27T20:36:23.945560Z  INFO obikpartitionner::partition: counting kmers in partition 135/256
2026-04-27T20:36:24.085073Z  INFO obikpartitionner::partition: counting kmers in partition 117/256
2026-04-27T20:36:24.111144Z  INFO obikpartitionner::partition: counting kmers in partition 144/256
2026-04-27T20:36:24.188547Z  INFO obikpartitionner::partition: counting kmers in partition 23/256
2026-04-27T20:36:24.573867Z  INFO obikpartitionner::partition: counting kmers in partition 103/256
2026-04-27T20:36:24.604461Z  INFO obikpartitionner::partition: counting kmers in partition 27/256
2026-04-27T20:36:24.670189Z  INFO obikpartitionner::partition: counting kmers in partition 95/256
2026-04-27T20:36:24.935767Z  INFO obikpartitionner::partition: counting kmers in partition 123/256
2026-04-27T20:36:25.031507Z  INFO obikpartitionner::partition: counting kmers in partition 39/256
2026-04-27T20:36:25.039809Z  INFO obikpartitionner::partition: counting kmers in partition 108/256
2026-04-27T20:36:25.067950Z  INFO obikpartitionner::partition: counting kmers in partition 199/256
2026-04-27T20:36:25.149100Z  INFO obikpartitionner::partition: counting kmers in partition 231/256
2026-04-27T20:36:25.272346Z  INFO obikpartitionner::partition: counting kmers in partition 213/256
2026-04-27T20:36:25.400111Z  INFO obikpartitionner::partition: counting kmers in partition 165/256
2026-04-27T20:36:25.607381Z  INFO obikpartitionner::partition: counting kmers in partition 136/256
2026-04-27T20:36:25.902415Z  INFO obikpartitionner::partition: counting kmers in partition 152/256
2026-04-27T20:36:26.055400Z  INFO obikpartitionner::partition: counting kmers in partition 118/256
2026-04-27T20:36:26.138877Z  INFO obikpartitionner::partition: counting kmers in partition 119/256
2026-04-27T20:36:26.255235Z  INFO obikpartitionner::partition: counting kmers in partition 87/256
2026-04-27T20:36:26.350364Z  INFO obikpartitionner::partition: counting kmers in partition 148/256
2026-04-27T20:36:26.465089Z  INFO obikpartitionner::partition: counting kmers in partition 145/256
2026-04-27T20:36:26.682375Z  INFO obikpartitionner::partition: counting kmers in partition 40/256
2026-04-27T20:36:26.694729Z  INFO obikpartitionner::partition: counting kmers in partition 216/256
2026-04-27T20:36:26.777197Z  INFO obikpartitionner::partition: counting kmers in partition 28/256
2026-04-27T20:36:26.802996Z  INFO obikpartitionner::partition: counting kmers in partition 124/256
2026-04-27T20:36:26.972099Z  INFO obikpartitionner::partition: counting kmers in partition 200/256
2026-04-27T20:36:27.066713Z  INFO obikpartitionner::partition: counting kmers in partition 232/256
2026-04-27T20:36:27.112569Z  INFO obikpartitionner::partition: counting kmers in partition 166/256
2026-04-27T20:36:27.156124Z  INFO obikpartitionner::partition: counting kmers in partition 214/256
2026-04-27T20:36:27.231214Z  INFO obikpartitionner::partition: counting kmers in partition 137/256
2026-04-27T20:36:27.359784Z  INFO obikpartitionner::partition: counting kmers in partition 109/256
2026-04-27T20:36:27.369743Z  INFO obikpartitionner::partition: counting kmers in partition 153/256
2026-04-27T20:36:27.668409Z  INFO obikpartitionner::partition: counting kmers in partition 204/256
2026-04-27T20:36:27.701887Z  INFO obikpartitionner::partition: counting kmers in partition 202/256
2026-04-27T20:36:27.794402Z  INFO obikpartitionner::partition: counting kmers in partition 215/256
2026-04-27T20:36:28.058391Z  INFO obikpartitionner::partition: counting kmers in partition 149/256
2026-04-27T20:36:28.192195Z  INFO obikpartitionner::partition: counting kmers in partition 217/256
2026-04-27T20:36:28.369717Z  INFO obikpartitionner::partition: counting kmers in partition 125/256
2026-04-27T20:36:28.412339Z  INFO obikpartitionner::partition: counting kmers in partition 41/256
2026-04-27T20:36:28.496068Z  INFO obikpartitionner::partition: counting kmers in partition 29/256
2026-04-27T20:36:28.693510Z  INFO obikpartitionner::partition: counting kmers in partition 167/256
2026-04-27T20:36:28.808495Z  INFO obikpartitionner::partition: counting kmers in partition 201/256
2026-04-27T20:36:28.843564Z  INFO obikpartitionner::partition: counting kmers in partition 233/256
2026-04-27T20:36:28.917920Z  INFO obikpartitionner::partition: counting kmers in partition 140/256
2026-04-27T20:36:28.940416Z  INFO obikpartitionner::partition: counting kmers in partition 138/256
2026-04-27T20:36:29.125956Z  INFO obikpartitionner::partition: counting kmers in partition 110/256
2026-04-27T20:36:29.197246Z  INFO obikpartitionner::partition: counting kmers in partition 154/256
2026-04-27T20:36:29.376894Z  INFO obikpartitionner::partition: counting kmers in partition 203/256
2026-04-27T20:36:29.491887Z  INFO obikpartitionner::partition: counting kmers in partition 176/256
2026-04-27T20:36:29.501763Z  INFO obikpartitionner::partition: counting kmers in partition 205/256
2026-04-27T20:36:29.538499Z  INFO obikpartitionner::partition: counting kmers in partition 150/256
2026-04-27T20:36:29.685646Z  INFO obikpartitionner::partition: counting kmers in partition 218/256
2026-04-27T20:36:29.874786Z  INFO obikpartitionner::partition: counting kmers in partition 126/256
2026-04-27T20:36:30.021059Z  INFO obikpartitionner::partition: counting kmers in partition 42/256
2026-04-27T20:36:30.255780Z  INFO obikpartitionner::partition: counting kmers in partition 240/256
2026-04-27T20:36:30.602468Z  INFO obikpartitionner::partition: counting kmers in partition 141/256
2026-04-27T20:36:30.696718Z  INFO obikpartitionner::partition: counting kmers in partition 234/256
2026-04-27T20:36:30.784522Z  INFO obikpartitionner::partition: counting kmers in partition 139/256
2026-04-27T20:36:31.027016Z  INFO obikpartitionner::partition: counting kmers in partition 155/256
2026-04-27T20:36:31.073326Z  INFO obikpartitionner::partition: counting kmers in partition 30/256
2026-04-27T20:36:31.137264Z  INFO obikpartitionner::partition: counting kmers in partition 111/256
2026-04-27T20:36:31.143413Z  INFO obikpartitionner::partition: counting kmers in partition 168/256
2026-04-27T20:36:31.237329Z  INFO obikpartitionner::partition: counting kmers in partition 206/256
2026-04-27T20:36:31.252908Z  INFO obikpartitionner::partition: counting kmers in partition 236/256
2026-04-27T20:36:31.306110Z  INFO obikpartitionner::partition: counting kmers in partition 177/256
2026-04-27T20:36:31.431064Z  INFO obikpartitionner::partition: counting kmers in partition 151/256
2026-04-27T20:36:31.490837Z  INFO obikpartitionner::partition: counting kmers in partition 127/256
2026-04-27T20:36:31.528815Z  INFO obikpartitionner::partition: counting kmers in partition 43/256
2026-04-27T20:36:31.790858Z  INFO obikpartitionner::partition: counting kmers in partition 219/256
2026-04-27T20:36:31.803934Z  INFO obikpartitionner::partition: counting kmers in partition 241/256
2026-04-27T20:36:32.131051Z  INFO obikpartitionner::partition: counting kmers in partition 156/256
2026-04-27T20:36:32.433342Z  INFO obikpartitionner::partition: counting kmers in partition 235/256
2026-04-27T20:36:32.459699Z  INFO obikpartitionner::partition: counting kmers in partition 142/256
2026-04-27T20:36:32.532717Z  INFO obikpartitionner::partition: counting kmers in partition 31/256
2026-04-27T20:36:32.814909Z  INFO obikpartitionner::partition: counting kmers in partition 184/256
2026-04-27T20:36:32.967090Z  INFO obikpartitionner::partition: counting kmers in partition 169/256
2026-04-27T20:36:33.062660Z  INFO obikpartitionner::partition: counting kmers in partition 178/256
2026-04-27T20:36:33.076014Z  INFO obikpartitionner::partition: counting kmers in partition 237/256
2026-04-27T20:36:33.128961Z  INFO obikpartitionner::partition: counting kmers in partition 207/256
2026-04-27T20:36:33.228734Z  INFO obikpartitionner::partition: counting kmers in partition 180/256
2026-04-27T20:36:33.228905Z  INFO obikpartitionner::partition: counting kmers in partition 143/256
2026-04-27T20:36:33.266931Z  INFO obikpartitionner::partition: counting kmers in partition 44/256
2026-04-27T20:36:33.280333Z  INFO obikpartitionner::partition: counting kmers in partition 158/256
2026-04-27T20:36:33.368740Z  INFO obikpartitionner::partition: counting kmers in partition 220/256
2026-04-27T20:36:33.373469Z  INFO obikpartitionner::partition: counting kmers in partition 179/256
2026-04-27T20:36:33.409455Z  INFO obikpartitionner::partition: counting kmers in partition 157/256
2026-04-27T20:36:33.685036Z  INFO obikpartitionner::partition: counting kmers in partition 242/256
2026-04-27T20:36:34.174714Z  INFO obikpartitionner::partition: counting kmers in partition 182/256
2026-04-27T20:36:34.255362Z  INFO obikpartitionner::partition: counting kmers in partition 146/256
2026-04-27T20:36:34.284053Z  INFO obikpartitionner::partition: counting kmers in partition 170/256
2026-04-27T20:36:34.340602Z  INFO obikpartitionner::partition: counting kmers in partition 185/256
2026-04-27T20:36:34.748830Z  INFO obikpartitionner::partition: counting kmers in partition 45/256
2026-04-27T20:36:34.894515Z  INFO obikpartitionner::partition: counting kmers in partition 172/256
2026-04-27T20:36:35.342815Z  INFO obikpartitionner::partition: counting kmers in partition 238/256
2026-04-27T20:36:35.549647Z  INFO obikpartitionner::partition: counting kmers in partition 159/256
2026-04-27T20:36:35.556431Z  INFO obikpartitionner::partition: counting kmers in partition 181/256
2026-04-27T20:36:35.579877Z  INFO obikpartitionner::partition: counting kmers in partition 188/256
2026-04-27T20:36:35.619438Z  INFO obikpartitionner::partition: counting kmers in partition 221/256
2026-04-27T20:36:35.704344Z  INFO obikpartitionner::partition: counting kmers in partition 174/256
2026-04-27T20:36:35.891634Z  INFO obikpartitionner::partition: counting kmers in partition 190/256
2026-04-27T20:36:35.893065Z  INFO obikpartitionner::partition: counting kmers in partition 239/256
2026-04-27T20:36:35.978053Z  INFO obikpartitionner::partition: counting kmers in partition 183/256
2026-04-27T20:36:35.980835Z  INFO obikpartitionner::partition: counting kmers in partition 186/256
2026-04-27T20:36:36.010364Z  INFO obikpartitionner::partition: counting kmers in partition 243/256
2026-04-27T20:36:36.084972Z  INFO obikpartitionner::partition: counting kmers in partition 171/256
2026-04-27T20:36:36.085806Z  INFO obikpartitionner::partition: counting kmers in partition 147/256
2026-04-27T20:36:36.129893Z  INFO obikpartitionner::partition: counting kmers in partition 48/256
2026-04-27T20:36:36.188339Z  INFO obikpartitionner::partition: counting kmers in partition 46/256
2026-04-27T20:36:36.281975Z  INFO obikpartitionner::partition: counting kmers in partition 173/256
2026-04-27T20:36:37.004395Z  INFO obikpartitionner::partition: counting kmers in partition 187/256
2026-04-27T20:36:37.080479Z  INFO obikpartitionner::partition: counting kmers in partition 222/256
2026-04-27T20:36:37.100573Z  INFO obikpartitionner::partition: counting kmers in partition 189/256
2026-04-27T20:36:37.422152Z  INFO obikpartitionner::partition: counting kmers in partition 248/256
2026-04-27T20:36:37.446786Z  INFO obikpartitionner::partition: counting kmers in partition 175/256
2026-04-27T20:36:37.951518Z  INFO obikpartitionner::partition: counting kmers in partition 244/256
2026-04-27T20:36:38.235613Z  INFO obikpartitionner::partition: counting kmers in partition 191/256
2026-04-27T20:36:38.259288Z  INFO obikpartitionner::partition: counting kmers in partition 47/256
2026-04-27T20:36:38.379966Z  INFO obikpartitionner::partition: counting kmers in partition 252/256
2026-04-27T20:36:38.380690Z  INFO obikpartitionner::partition: counting kmers in partition 223/256
2026-04-27T20:36:38.520398Z  INFO obikpartitionner::partition: counting kmers in partition 56/256
2026-04-27T20:36:38.598256Z  INFO obikpartitionner::partition: counting kmers in partition 49/256
2026-04-27T20:36:38.680704Z  INFO obikpartitionner::partition: counting kmers in partition 254/256
2026-04-27T20:36:38.681038Z  INFO obikpartitionner::partition: counting kmers in partition 246/256
2026-04-27T20:36:38.813533Z  INFO obikpartitionner::partition: counting kmers in partition 247/256
2026-04-27T20:36:38.820199Z  INFO obikpartitionner::partition: counting kmers in partition 60/256
2026-04-27T20:36:38.961240Z  INFO obikpartitionner::partition: counting kmers in partition 52/256
2026-04-27T20:36:39.001585Z  INFO obikpartitionner::partition: counting kmers in partition 58/256
2026-04-27T20:36:39.050047Z  INFO obikpartitionner::partition: counting kmers in partition 249/256
2026-04-27T20:36:39.076449Z  INFO obikpartitionner::partition: counting kmers in partition 57/256
2026-04-27T20:36:39.078254Z  INFO obikpartitionner::partition: counting kmers in partition 250/256
2026-04-27T20:36:39.082390Z  INFO obikpartitionner::partition: counting kmers in partition 251/256
2026-04-27T20:36:39.082697Z  INFO obikpartitionner::partition: counting kmers in partition 245/256
2026-04-27T20:36:39.250465Z  INFO obikpartitionner::partition: counting kmers in partition 50/256
2026-04-27T20:36:39.448799Z  INFO obikpartitionner::partition: counting kmers in partition 54/256
2026-04-27T20:36:39.449166Z  INFO obikpartitionner::partition: counting kmers in partition 62/256
2026-04-27T20:36:39.450954Z  INFO obikpartitionner::partition: counting kmers in partition 63/256
2026-04-27T20:36:39.766480Z  INFO obikpartitionner::partition: counting kmers in partition 253/256
2026-04-27T20:36:39.838346Z  INFO obikpartitionner::partition: counting kmers in partition 59/256
2026-04-27T20:36:39.875987Z  INFO obikpartitionner::partition: counting kmers in partition 255/256
2026-04-27T20:36:39.876798Z  INFO obikpartitionner::partition: counting kmers in partition 55/256
2026-04-27T20:36:39.877292Z  INFO obikpartitionner::partition: counting kmers in partition 51/256
2026-04-27T20:36:40.244598Z  INFO obikpartitionner::partition: counting kmers in partition 61/256
2026-04-27T20:36:40.245589Z  INFO obikpartitionner::partition: counting kmers in partition 53/256
266.84 real 3776.35 user 115.80 sys
2423504896 maximum resident set size
0 average shared memory size
0 average unshared data size
0 average unshared stack size
292303 page reclaims
132311 page faults
0 swaps
0 block input operations
0 block output operations
0 messages sent
0 messages received
0 signals received
4462 voluntary context switches
1991864 involuntary context switches
54933227607056 instructions retired
11271230983971 cycles elapsed
2210777320 peak memory footprint
-1
View File
@@ -1 +0,0 @@
(Escherichia_coli--CFT073:0.7973338454,(Escherichia_coli--EDL933:0.5132866305,(Escherichia_coli--K-12_MG1655:0.2990687226,Escherichia_coli--K-12_W3110:0.3006288477)100:0.1655629834)69:0.0000007189,(((Klebsiella_pneumoniae--ATCC_13883:0.9661675519,(Klebsiella_pneumoniae--MGH_78578:0.3148953385,Yersinia_ruckeri--YRB:0.3576917943)84:0.4568464394)83:0.0727808943,(Klebsiella_pneumoniae--HS11286:0.1397401144,Proteus_mirabilis--HI4320:0.2192824480)77:0.3686486712)76:0.1162295711,((Salmonella_enterica--AKU_12601:0.6158730439,(Salmonella_enterica--LT2:0.4371443730,Salmonella_enterica--P125109:0.5815549929)100:0.1012287522)100:0.0714169372,Salmonella_enterica--CT18:0.2058143021)76:0.3641489914)100:0.2900565429);
-62
View File
@@ -1,62 +0,0 @@
#nexus
BEGIN Taxa;
DIMENSIONS ntax=13;
TAXLABELS
[1] 'Escherichia_coli--CFT073'
[2] 'Escherichia_coli--EDL933'
[3] 'Escherichia_coli--K-12_MG1655'
[4] 'Escherichia_coli--K-12_W3110'
[5] 'Klebsiella_pneumoniae--ATCC_13883'
[6] 'Klebsiella_pneumoniae--HS11286'
[7] 'Klebsiella_pneumoniae--MGH_78578'
[8] 'Proteus_mirabilis--HI4320'
[9] 'Salmonella_enterica--AKU_12601'
[10] 'Salmonella_enterica--CT18'
[11] 'Salmonella_enterica--LT2'
[12] 'Salmonella_enterica--P125109'
[13] 'Yersinia_ruckeri--YRB'
;
END; [Taxa]
BEGIN Splits;
DIMENSIONS ntax=13 nsplits=34;
FORMAT labels=no weights=yes confidences=no intervals=no;
MATRIX
100 2,
100 3,
100 4,
100 3 4,
69 2 3 4,
100 5,
100 7,
16 5 7,
100 6,
100 13,
16 6 13,
23 5 6 7 13,
100 8,
100 10,
23 8 10,
100 9,
100 11,
100 12,
100 11 12,
100 9 11 12,
23 8 9 10 11 12,
100 1 2 3 4,
100 1,
84 7 13,
83 5 7 13,
77 6 8,
76 5 6 7 8 13,
76 9 10 11 12,
31 1 2,
1 5 6 7 8,
1 10 13,
1 9 10 11 12 13,
1 5 6 13,
1 5 6 8,
;
END; [Splits]
-1000
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
-1
View File
@@ -1 +0,0 @@
(((((((((((((Salmonella_enterica--P125109:0.009029074806027838,Salmonella_enterica--LT2:0.010374535686866677):0.00299392644685007,Salmonella_enterica--AKU_12601:0.010129090864056669):0.008889984754373365,Salmonella_enterica--CT18:0.029563522158784489):0.05178212639099835,(Escherichia_coli--CFT073:0.018265758526676266,(Escherichia_coli--EDL933:0.01586939152374907,(Escherichia_coli--K-12_MG1655:0.0043321525684933419,Escherichia_coli--K-12_W3110:0.004373581717010194):0.007835004272817602):0.005522415854494232):0.05395833543383322):0.007199364670842333,(Klebsiella_pneumoniae--HS11286:0.027306681424389149,(Klebsiella_pneumoniae--ATCC_13883:0.015176538442299506,Klebsiella_pneumoniae--MGH_78578:0.026462018414958045):0.0015885400482290244):0.060738462006352359):0.029481404183583916,Yersinia_ruckeri--YRB:0.10269826524749153):0.004316815340436764,Proteus_mirabilis--HI4320:0.11745645944715785):0.04439101156727093,Wolbachia_endosymbiont--GCF_000306885.1_ASM30688v1:0.17151356406519997):0.011344480816758762,Acidobacterium_capsulatum--ATCC_51196:0.17121217909071463):0.01869709592355495,Shouchella_clausii--KSM-K16:0.1554729916092201):0.04397682726757515,Bacillus_subtilis--168:0.10403127799132669):0.08828869671555731,Opitutus_terrae--PB90-1:0.06380512639846864):0.41655625366738377,Candidozyma_auris--GCF_003013715.1_ASM301371v2:0.22735026269091164,Saccharolobus_islandicus--M.16.4:0.21891918239548398);
Binary file not shown.
BIN
View File
Binary file not shown.
-167
View File
@@ -1,167 +0,0 @@
ratio_ceiling: 0.5
cardinality_transitions:
- from: 0
to: 0
count: 54999793
probability: 0.7934021725308564
- from: 0
to: 1
count: 8873859
probability: 0.12801028195383377
- from: 0
to: 2
count: 5305545
probability: 0.07653539585976665
- from: 0
to: 3
count: 126109
probability: 0.001819191475424167
- from: 0
to: 4
count: 16149
probability: 0.00023295818011898336
- from: 1
to: 0
count: 8873859
probability: 0.8510749571674425
- from: 1
to: 1
count: 1041044
probability: 0.09984455215137214
- from: 1
to: 2
count: 507272
probability: 0.048651493749477304
- from: 1
to: 3
count: 4248
probability: 0.00040741760918753563
- from: 1
to: 4
count: 225
probability: 0.00002157932252052625
- from: 2
to: 0
count: 5305545
probability: 0.9117891245815953
- from: 2
to: 1
count: 507272
probability: 0.08717767784549091
- from: 2
to: 2
count: 5227
probability: 0.0008982907041949506
- from: 2
to: 3
count: 689
probability: 0.00011840870388182915
- from: 2
to: 4
count: 96
probability: 0.000016498164836945715
- from: 3
to: 0
count: 126109
probability: 0.9614164824273843
- from: 3
to: 1
count: 4248
probability: 0.03238545399100404
- from: 3
to: 2
count: 689
probability: 0.005252725470763132
- from: 3
to: 3
count: 98
probability: 0.0007471220553480217
- from: 3
to: 4
count: 26
probability: 0.00019821605550049553
- from: 4
to: 0
count: 16149
probability: 0.9781344639612356
- from: 4
to: 1
count: 225
probability: 0.013628104179285281
- from: 4
to: 2
count: 96
probability: 0.00581465778316172
- from: 4
to: 3
count: 26
probability: 0.0015748031496062992
- from: 4
to: 4
count: 14
probability: 0.0008479709267110841
composition_transitions:
- from: 'A'
to: 'A'
count: 161774
probability: 0.4701762136303263
- from: 'A'
to: 'C'
count: 27788
probability: 0.0807624007835592
- from: 'A'
to: 'G'
count: 130019
probability: 0.3778842157577942
- from: 'A'
to: 'T'
count: 24490
probability: 0.07117716982832031
- from: 'C'
to: 'A'
count: 27788
probability: 0.0780165140757087
- from: 'C'
to: 'C'
count: 184047
probability: 0.5167232390273484
- from: 'C'
to: 'G'
count: 19622
probability: 0.05508996830263265
- from: 'C'
to: 'T'
count: 124724
probability: 0.3501702785943102
- from: 'G'
to: 'A'
count: 130019
probability: 0.3610094570655886
- from: 'G'
to: 'C'
count: 19622
probability: 0.05448224926003876
- from: 'G'
to: 'G'
count: 183951
probability: 0.5107565097152884
- from: 'G'
to: 'T'
count: 26562
probability: 0.07375178395908417
- from: 'T'
to: 'A'
count: 24490
probability: 0.07335783586895636
- from: 'T'
to: 'C'
count: 124724
probability: 0.37360076443118473
- from: 'T'
to: 'G'
count: 26562
probability: 0.07956434611479049
- from: 'T'
to: 'T'
count: 158067
probability: 0.4734770535850684
-660
View File
@@ -1,660 +0,0 @@
<h1 id="obikmer">obikmer</h1>
<p><code>obikmer</code> is a Rust tool for manipulation, counting,
indexing, and set operations on DNA sequences represented as kmer
sets.</p>
<h2 id="constraints">Constraints</h2>
<ul>
<li>Target scale: metagenomic data, tens of Gbases, billions of
kmers</li>
<li>Maximum efficiency in computation, memory, and disk usage</li>
<li>k is odd, k ∈ [11, 31], fixed at runtime</li>
<li>Input formats: FASTA, FASTQ, gzip, streaming stdin</li>
</ul>
<h2 id="input-processing">Input processing</h2>
<p>For sequence files (FASTA/FASTQ):</p>
<ul>
<li>Cut sequence on non-ACGT nucleotides.</li>
<li>Discard fragments &lt; k.</li>
<li>For sequences &gt; 256 nt: cut at 256, start new with k-1 overlap
(preserves all kmers, no duplicates).</li>
</ul>
<h2 id="super-kmers">Super-kmers</h2>
<p>The fundamental stored unit is a <strong>canonical
super-kmer</strong>: a maximal run of consecutive kmers sharing the same
<strong>non-canonical</strong> (raw forward) minimizer, canonized at the
end.</p>
<h3 id="construction">Construction</h3>
<p>Super-kmers are built using the <strong>non-canonical
minimizer</strong> (minimum forward m-mer, no revcomp comparison). Two
consecutive kmers belong to the same super-kmer if and only if their raw
forward minimizer is identical.</p>
<p>At the end of each super-kmer, the minimizer is canonized:
<code>canonical(M) = min(M, revcomp(M))</code>.</p>
<ul>
<li>If <code>canonical(M) == M</code>: the super-kmer is already in
canonical orientation — keep as-is.</li>
<li>If <code>canonical(M) == revcomp(M)</code>: reverse-complement the
entire super-kmer (and its minimizer).</li>
</ul>
<p>The reverse-complement step does not cause any additional splitting —
it only flips the orientation.</p>
<h3 id="why-non-canonical-minimizer">Why non-canonical minimizer?</h3>
<p>Using the raw forward minimizer (rather than the canonical minimizer)
means fewer candidates are considered at each position → the minimizer
changes more often → super-kmers are <strong>shorter or equal</strong>,
never longer, compared to the canonical-minimizer definition.</p>
<p>The key benefit is at dereplication: a forward read and its
reverse-complement covering the same genomic region both produce the
<strong>same canonical super-kmer</strong> after the canonization step.
Under the canonical-minimizer definition, they would share the same
canonical minimizer but have different sequences (one being the revcomp
of the other) — they would not dereplicate. Here they do, giving a
compression factor of ~2× on paired or double-stranded data.</p>
<h3 id="length">Length</h3>
<p>For a random minimizer of length m over k-mers of length k, the
density of minimizer positions is approximately 2/(k−m+2) <span
class="citation" data-cites="Zheng2020-ji Golan2025-xf">[@Zheng2020-ji;
@Golan2025-xf]</span>, giving an expected super-kmer length of (k+m)/2
nucleotides. For k=31, m=13: expected ≈ 22 nt. In practice super-kmers
rarely exceed a few dozen nucleotides — the 256 nt cap is almost never
reached. When it is, the super-kmer is split at 256 nt with a k−1
overlap, preserving all kmers without duplication.</p>
<h2 id="binary-storage-format">Binary storage format</h2>
<p>Memory-mappable binary format for sequence collections: store
super-kmers end-to-end with u64-aligned headers (add padding if needed
for 8-byte alignment). Include index blocks every B super-kmers
(B=1000-10000) for fast jumping to position J (O(N/B + B) scan
vs. separate index O(1) direct access), where N is total super-kmers in
file. Represent header as union/struct for field access.</p>
<h2 id="partitioning-and-indexing-architecture">Partitioning and
indexing architecture</h2>
<p>The canonical minimizer of a super-kmer is hashed to produce a
<strong>p-bit routing value</strong> (p is a collection-level
parameter):</p>
<pre><code>canonical minimizer → hash(minimizer) → p-bit value → PART → partition directory</code></pre>
<p>PART is computed once at phase 1 to open the correct partition file,
then discarded. It is recomputed on the fly at query time. It is never
stored in the super-kmer header.</p>
<p>Each partition holds one BBHash instance (phase 6) that indexes kmers
as plain u64 values — the minimizer plays no role inside the
partition.</p>
<p>Properties: - <strong>2^p partitions</strong>: p freely tunable at
collection creation; not constrained to powers of 4 - <strong>Uniform
distribution</strong>: hashing the canonical minimizer corrects the
lexicographic bias inherent in minimizer selection <span
class="citation"
data-cites="Zheng2020-ji Zheng2021-cc Pan2024-hb Kille2023-px Golan2025-xf">[@Zheng2020-ji;
@Zheng2021-cc; @Pan2024-hb; @Kille2023-px; @Golan2025-xf]</span> - The
odd-length constraint on m only ensures canonical form correctness
during minimizer computation — it does not affect the partition
layout</p>
<p><strong>Entropy constraint:</strong> the canonical minimizer has 2m
bits of entropy. Requesting p bits from the hash requires 2m ≥ p,
i.e. <strong>p ≤ 2m</strong>:</p>
<table>
<thead>
<tr>
<th>m</th>
<th>p max</th>
<th>Partitions max</th>
</tr>
</thead>
<tbody>
<tr>
<td>9</td>
<td>2</td>
<td>4</td>
</tr>
<tr>
<td>11</td>
<td>6</td>
<td>64</td>
</tr>
<tr>
<td>13</td>
<td>10</td>
<td>1 024</td>
</tr>
<tr>
<td>15</td>
<td>14</td>
<td>16 384</td>
</tr>
</tbody>
</table>
<p>For the typical configuration k=31, m=13: up to 1 024 partitions.
Choosing p &gt; 2m−16 is possible but produces artificially colliding
PART values with no gain in routing granularity.</p>
<h2 id="construction-pipeline">Construction pipeline</h2>
<p>All phases after scatter are embarrassingly parallel across
partitions.</p>
<h3 id="phase-1-scatter">Phase 1 — Scatter</h3>
<p>Single streaming pass over raw input files (FASTA/FASTQ, gzip). FASTQ
quality scores are ignored. For each read:</p>
<ol type="1">
<li><strong>Ambiguous base filter</strong>: cut at any non-ACGT base;
discard fragments shorter than k.</li>
<li><strong>Entropy filter</strong>: scan each fragment with a sliding
window of size k. When the kmer <span class="math inline">$K_i = S[i
\mathinner{..} i+k-1]$</span> ended by nucleotide <span
class="math inline"><em>S</em>[<em>j</em>]</span> (with <span
class="math inline"><em>j</em> = <em>i</em> + <em>k</em> − 1</span>) has
entropy below threshold <span class="math inline"><em>θ</em></span>,
emit the current segment and start a new one (see algorithm below).
<span class="math inline"><em>K</em><sub><em>i</em></sub></span> belongs
to neither segment, and no valid kmer is lost.</li>
<li><strong>Super-kmer extraction</strong>: for each clean segment,
slide a minimizer window and group consecutive kmers sharing the same
raw minimizer; canonise at the end.</li>
<li><strong>Partition routing</strong>:
<code>hash(canonical_minimizer) → PART</code> → append super-kmer to
<code>partition/superkmers.bin.gz</code>.</li>
</ol>
<p><strong>Entropy filter algorithm:</strong></p>
<p>When <span class="math inline"><em>K</em><sub><em>i</em></sub></span>
(ended by <span class="math inline"><em>S</em>[<em>j</em>]</span>, <span
class="math inline"><em>j</em> = <em>i</em> + <em>k</em> − 1</span>)
fails the entropy threshold:</p>
<ul>
<li>Current segment <span class="math inline">$S[\textit{seg_start}
\mathinner{..} j-1]$</span> is emitted (last valid kmer = <span
class="math inline"><em>K</em><sub><em>i</em> − 1</sub></span>)</li>
<li>New segment starts at <span
class="math inline"><em>S</em>[<em>i</em> + 1]</span> (first new kmer =
<span
class="math inline"><em>K</em><sub><em>i</em> + 1</sub></span>)</li>
<li><span class="math inline"><em>K</em><sub><em>i</em></sub></span> is
excluded: current segment lacks <span
class="math inline"><em>S</em>[<em>j</em>]</span>, new segment lacks
<span class="math inline"><em>S</em>[<em>i</em>]</span></li>
<li>Overlap = <span class="math inline">$S[i+1 \mathinner{..}
j-1]$</span> = <span class="math inline"><em>k</em> − 2</span>
nucleotides</li>
</ul>
<pre class="pseudocode"><code>\begin{algorithm}
\caption{Entropy filter — sliding window segmentation}
\begin{algorithmic}
\PROCEDURE{EntropyFilter}{S, N, k, theta}
\STATE seg_start &lt;- 0
\STATE window &lt;- []
\FOR{j &lt;- 0 \TO N-1}
\STATE window.push(S[j])
\IF{|window| &lt; k}
\STATE continue
\ENDIF
\STATE i &lt;- j - k + 1
\IF{entropy(window) &lt;= theta}
\STATE emit S[seg_start .. j-1]
\STATE seg_start &lt;- i + 1
\STATE window &lt;- S[i+1 .. j]
\ELSE
\STATE window.pop_front()
\ENDIF
\ENDFOR
\STATE emit S[seg_start .. N-1]
\ENDPROCEDURE
\end{algorithmic}
\end{algorithm}</code></pre>
<p>Writes are sequential and append-only — IO-friendly. Gzip applied at
write time. Data volume ≈ raw genome size (2 bits/nt compaction offsets
header overhead).</p>
<h3 id="phase-2-dereplication">Phase 2 — Dereplication</h3>
<p>Performed independently per partition. Identical super-kmers are
consolidated and their COUNT accumulated — analogous to amplicon
dereplication in metabarcoding. Uses external bucket sort to stay within
RAM bounds:</p>
<p><strong>Pass 1</strong> (streaming): hash the nucleotide payload of
each super-kmer, route to one of B bucket files:</p>
<pre><code>hash(sequence) % B → bucket_i.bin</code></pre>
<p>B ≈ 100 is tunable; RAM needed ≈ partition_size / B.</p>
<p><strong>Pass 2</strong>: for each bucket, load into an in-memory
<code>HashMap&lt;sequence, COUNT&gt;</code>, dereplicate by summing
COUNT values, write consolidated super-kmers.</p>
<p>After dereplication: at Nx coverage the partition shrinks by ~x
(errors aside). The COUNT field in each super-kmer header = number of
times that exact super-kmer sequence was observed across all input
reads.</p>
<p><strong>Important:</strong> super-kmer COUNT ≠ individual kmer count.
A kmer can appear in multiple distinct super-kmers (same partition,
different flanking context); its true count = sum of COUNT of all
super-kmers containing it. A super-kmer with COUNT=1 may contain only
high-abundance kmers, each appearing in many other super-kmers.
Abundance filtering therefore cannot be applied at this phase.</p>
<h3 id="phase-3-per-kmer-count-aggregation-and-quorum-filtering">Phase 3
— Per-kmer count aggregation and quorum filtering</h3>
<p>For each dereplicated super-kmer, enumerate its kmers and accumulate
counts:</p>
<pre><code>for each super-kmer (sequence, COUNT):
for each kmer in sequence:
kmer_counts[canonical(kmer)] += COUNT</code></pre>
<p>Implemented as an external sort or a temporary HashMap, depending on
partition size. At the end of this phase, each distinct canonical kmer
has its exact total count.</p>
<p>Abundance filter applied here: kmers with
<code>total_count &lt; q</code> are discarded. <code>q</code> is a
collection parameter (0 = keep all, including singletons for ≤1x
data).</p>
<p>No pre-filter on super-kmer COUNT is possible at phase 2: a
super-kmer with COUNT=1 may contain only high-abundance kmers, each
present in many other super-kmers across the partition.</p>
<h3 id="phase-4-super-kmer-compaction">Phase 4 — Super-kmer
compaction</h3>
<p>The valid kmer set from phase 3 is used as a mask to rewrite the
super-kmer files:</p>
<pre><code>for each dereplicated super-kmer:
scan kmer by kmer
kmer not in valid set → break point (terminates current super-kmer)
kmer in valid set → extend current super-kmer</code></pre>
<p>Three cases per super-kmer: - <strong>All kmers valid</strong>
copied as-is - <strong>No kmer valid</strong> → discarded -
<strong>Mixed</strong> → split into sub-super-kmers at invalid
boundaries; each sub-super-kmer inherits the original COUNT</p>
<p>After splitting, re-apply dereplication (bucket sort, phase 2 method)
— splitting can produce new identical super-kmers. This re-dereplication
is cheap: the volume is already greatly reduced.</p>
<p>Output: a clean super-kmer file where every kmer passes quorum. This
file feeds phase 5.</p>
<h3 id="phase-5-local-de-bruijn-graph-and-unitig-construction">Phase 5 —
Local de Bruijn graph and unitig construction</h3>
<p>Within each partition, build a <strong>local de Bruijn graph</strong>
from the valid kmer set and compute its unitigs. All operations are
local to the partition — no cross-partition communication.</p>
<pre><code>valid kmers → HashSet&lt;u64&gt;
for each kmer K:
out_degree = |{K[1:]+b | b ∈ {A,C,G,T}} ∩ HashSet|
in_degree = |{b+K[:-1] | b ∈ {A,C,G,T}} ∩ HashSet|
internal node ↔ in_degree=1 AND out_degree=1
branching / dead-end → unitig start or end</code></pre>
<p>Traverse non-branching paths to assemble unitigs. Kmers whose
neighbours fall in other partitions appear as dead ends locally — they
terminate the unitig. The result: <strong>each kmer appears in exactly
one unitig</strong> within the partition.</p>
<p>The partition size (controlled by p) must be calibrated so that the
HashSet fits in RAM during this phase.</p>
<p>Output: <code>unitigs.bin</code> — the permanent evidence structure
for the partition. Each kmer in the partition appears at exactly one
(unitig_id, offset) location.</p>
<p><strong>Scope of local unitigs:</strong> these are unitigs of the
partition’s local de Bruijn graph, not global unitigs. A kmer whose k-1
successor or predecessor falls in another partition appears as a dead
end locally and terminates the unitig. This does not affect correctness
of verification but means partition-local unitigs cannot be directly
reused for global assembly.</p>
<h3 id="phase-6-bbhash-construction-and-index-finalisation">Phase 6 —
BBHash construction and index finalisation</h3>
<p>Built once on the definitive kmer set (all kmers in all unitigs of
the partition):</p>
<pre><code>kmers from unitigs → BBHash → bbhash.bin
→ counts.bin : packed n-bit array (or 1-bit for presence mode)
→ refs.bin : (unitig_id: u32, kmer_offset: u8) per kmer</code></pre>
<p>BBHash is built once — no rebuild. The n-bit width for
<code>counts.bin</code> is chosen from the observed count distribution
(n=5 covers ~97% of kmers at 15x; n=1 for presence mode). Counts
exceeding 2ⁿ−1 go into <code>overflow.bin</code> as sorted
<code>(bbhash_index: u32, count: u32)</code> pairs.</p>
<p><strong>Exact verification via unitig evidence:</strong></p>
<p><code>unitigs.bin</code> serves as the evidence structure: for any
query kmer, the stored unitig provides the ground truth to confirm or
deny its presence. BBHash maps every input to [0, N) including absent
kmers — the unitig read-back is the only way to guarantee exactness.</p>
<pre><code>query kmer q
→ minimizer(q) → hash → PART, KEY → partition
→ BBHash(q) → index i
→ refs[i] = (unitig_id, offset)
→ read unitig unitig_id from unitigs.bin
→ extract kmer at offset → compare with q
→ match : return counts[i] ← exact hit
→ no match: kmer absent ← BBHash collision on absent kmer</code></pre>
<p>One random disk access into <code>unitigs.bin</code> per query; the
unitig is the minimal, non-redundant evidence (each kmer stored
once).</p>
<h2 id="on-disk-collection-structure">On-disk collection structure</h2>
<p>Collections are too large to hold in RAM (hundreds of genomes,
billions of kmers). The collection lives on disk as a directory of
memory-mapped files:</p>
<pre><code>collection/
metadata.toml — collection parameters (see below)
superkmers.bin — super-kmer sequences, end-to-end, memory-mappable
superkmers.idx — index blocks every B super-kmers for fast seeking
part_XXXX/
bbhash.bin — BBHash function for this partition
counts.bin — packed n-bit count array (or 1-bit presence array)
refs.bin — back-references for exact verification</code></pre>
<p><strong>Collection parameters</strong> (stored in
<code>metadata.toml</code>):</p>
<table>
<thead>
<tr>
<th>Parameter</th>
<th>Role</th>
</tr>
</thead>
<tbody>
<tr>
<td>k</td>
<td>kmer length</td>
</tr>
<tr>
<td>m</td>
<td>minimizer length (odd, &lt; k)</td>
</tr>
<tr>
<td>p</td>
<td>partition bits (0 ≤ p ≤ min(14, 2m−16))</td>
</tr>
<tr>
<td>mode</td>
<td><code>presence</code> (1 bit/kmer) or <code>count</code> (n
bits/kmer)</td>
</tr>
<tr>
<td>n</td>
<td>bits per kmer in count mode (chosen at construction)</td>
</tr>
<tr>
<td>min_count</td>
<td>singleton filtering threshold (0 = keep all)</td>
</tr>
<tr>
<td>hash_fn</td>
<td>hash function identifier</td>
</tr>
<tr>
<td>hash_seed</td>
<td>seed for the hash function</td>
</tr>
</tbody>
</table>
<p><strong>Exact verification via back-reference:</strong></p>
<p>BBHash maps any input to [0, N) — including absent kmers. To avoid
false positives, <code>refs.bin</code> stores at each BBHash index a
back-reference <code>(superkmer_id: u32, kmer_offset: u8)</code> = 5
bytes. Query protocol:</p>
<pre><code>query kmer q
→ minimizer(q) → hash → PART, KEY → partition
→ BBHash(q) → index i
→ refs[i] = (sk_id, offset)
→ read super-kmer sk_id from superkmers.bin
→ extract kmer at offset → compare with q
→ if match: return counts[i] (exact)
→ if no match: kmer absent</code></pre>
<p><strong>Count storage — two modes:</strong></p>
<p><em>Presence mode</em> (coverage ≤ 1x, or when only presence/absence
matters): - <code>counts.bin</code> is a packed 1-bit array — all bits
set to 1 for indexed kmers - Singletons are the signal, not filtered</p>
<p><em>Count mode</em> (coverage &gt; 1x): - <code>counts.bin</code> is
a packed n-bit array; n chosen at construction from the observed
distribution - Value 0: absent sentinel; values 1..2ⁿ−2: direct count;
value 2ⁿ−1: overflow - Overflow counts stored in a separate
<code>overflow.bin</code> as sorted
<code>(index: u32, count: u32)</code> pairs - Empirically (k=31, 15x
coverage): n=5 covers 97% of real kmers, n=6 covers 99% - min_count
threshold filters low-frequency kmers (errors) before indexing; for ≤1x,
min_count=0</p>
<p><strong>BBHash:</strong> operates on kmer u64 values directly within
each partition. The minimizer plays no role inside the partition —
BBHash receives the kmer and builds the perfect hash naturally.</p>
<h2 id="kmer-entropy-filter">Kmer entropy filter</h2>
<p>Low-complexity kmers (polyA, polyT, tandem repeats) are detected and
excluded during phase 1. The filter computes a <strong>normalized
Shannon entropy</strong> over sub-words of multiple sizes, corrected for
two sources of bias: the small number of observations within a single
kmer, and the unequal sizes of circular equivalence classes.</p>
<h3 id="sub-word-frequencies">Sub-word frequencies</h3>
<p>For a kmer of length k and a sub-word size ws (1 ≤ ws ≤ ws_max,
typically ws_max = 6), extract the <span
class="math inline"><em>n</em><sub>words</sub> = <em>k</em> − <em>w</em><em>s</em> + 1</span>
overlapping sub-words by sliding a window of length ws:</p>
<p><span class="math display">$$w_i = \text{kmer}[i \mathinner{..}
i+ws-1], \quad i = 0, \ldots, n_{\text{words}}-1$$</span></p>
<p>Each sub-word is mapped to its <strong>circular canonical
form</strong>: the lexicographic minimum among all cyclic rotations of
the word. Let <span
class="math inline"><em>s</em><sub><em>j</em></sub></span> be the size
of equivalence class <span class="math inline"><em>j</em></span> (number
of distinct raw words mapping to canonical form <span
class="math inline"><em>j</em></span>), and <span
class="math inline"><em>f</em><sub><em>j</em></sub></span> the count of
canonical form <span class="math inline"><em>j</em></span> among the
<span class="math inline"><em>n</em><sub>words</sub></span> sub-words
(<span
class="math inline"><sub><em>j</em></sub><em>f</em><sub><em>j</em></sub> = <em>n</em><sub>words</sub></span>).</p>
<h3 id="corrected-shannon-entropy">Corrected Shannon entropy</h3>
<p>The circular equivalence classes have unequal sizes: under a uniform
distribution over all <span
class="math inline">4<sup><em>w</em><em>s</em></sup></span> raw words,
class <span class="math inline"><em>j</em></span> is visited with
probability <span
class="math inline"><em>s</em><sub><em>j</em></sub>/4<sup><em>w</em><em>s</em></sup></span>,
not <span class="math inline">1/<em>n</em><sub><em>a</em></sub></span>.
Computing entropy directly over canonical classes therefore
underestimates the entropy of a random sequence.</p>
<p>The correction “unfolds” each canonical class back to its member raw
words, redistributing each observation of class <span
class="math inline"><em>j</em></span> equally among its <span
class="math inline"><em>s</em><sub><em>j</em></sub></span> members:</p>
<p><span class="math display">$$H_{\text{corr}} = \log(n_{\text{words}})
- \frac{1}{n_{\text{words}}} \sum_j f_j \log f_j +
\frac{1}{n_{\text{words}}} \sum_j f_j \log s_j$$</span></p>
<p>The last term is the correction for unequal class sizes. For a
uniformly random sequence (<span
class="math inline"><em>f</em><sub><em>j</em></sub> ≈ <em>n</em><sub>words</sub> ⋅ <em>s</em><sub><em>j</em></sub>/4<sup><em>w</em><em>s</em></sup></span>),
this gives <span
class="math inline"><em>H</em><sub>corr</sub> ≈ log (4<sup><em>w</em><em>s</em></sup>) = 2 ⋅ <em>w</em><em>s</em> ⋅ log 2</span>,
the maximum entropy over raw words.</p>
<h3 id="maximum-entropy-correction-for-small-samples">Maximum entropy
correction for small samples</h3>
<p>With only <span class="math inline"><em>n</em><sub>words</sub></span>
observations over <span
class="math inline">4<sup><em>w</em><em>s</em></sup></span> possible raw
words, the achievable maximum entropy is bounded by the most uniform
integer distribution over <span
class="math inline">4<sup><em>w</em><em>s</em></sup></span>
categories.</p>
<p>Let <span
class="math inline"><em>c</em> = ⌊<em>n</em><sub>words</sub>/4<sup><em>w</em><em>s</em></sup></span>
and <span
class="math inline"><em>r</em> = <em>n</em><sub>words</sub> mod  4<sup><em>w</em><em>s</em></sup></span>.
The most uniform integer distribution assigns frequency <span
class="math inline"><em>c</em> + 1</span> to <span
class="math inline"><em>r</em></span> categories and <span
class="math inline"><em>c</em></span> to the remaining <span
class="math inline">4<sup><em>w</em><em>s</em></sup> − <em>r</em></span>,
with the convention <span class="math inline">0log 0 = 0</span>:</p>
<p><span class="math display">$$H_{\max} = -\left[(4^{ws} -
r)\,\frac{c}{n_{\text{words}}}\log\frac{c}{n_{\text{words}}} +
r\,\frac{c+1}{n_{\text{words}}}\log\frac{c+1}{n_{\text{words}}}\right]$$</span></p>
<p>When <span
class="math inline"><em>n</em><sub>words</sub>&lt; 4<sup><em>w</em><em>s</em></sup></span>:
<span class="math inline"><em>c</em> = 0</span>, <span
class="math inline"><em>r</em> = <em>n</em><sub>words</sub></span>, and
the formula reduces to <span
class="math inline"><em>H</em><sub>max</sub> = log (<em>n</em><sub>words</sub>)</span>
— a single unified expression covers both regimes. A truly random
sequence achieves <span
class="math inline"><em>H</em><sub>corr</sub> ≈ <em>H</em><sub>max</sub></span>.</p>
<h3 id="normalized-entropy">Normalized entropy</h3>
<p><span class="math display">$$\hat{H}(ws) =
\frac{H_{\text{corr}}}{H_{\max}} \in [0, 1]$$</span></p>
<h3 id="final-score">Final score</h3>
<p>The filter computes <span
class="math inline"><em></em>(<em>w</em><em>s</em>)</span> for each
word size ws from 1 to ws_max and returns the
<strong>minimum</strong>:</p>
<p><span class="math display">$$\text{entropy}(kmer) =
\min_{ws=1}^{ws_{\max}} \hat{H}(ws)$$</span></p>
<p>A value near 0 indicates low complexity (e.g. AAAA…); near 1
indicates high complexity. A kmer is rejected if <span
class="math inline">entropy(<em>k</em><em>m</em><em>e</em><em>r</em>) ≤ <em>θ</em></span>,
where <span class="math inline"><em>θ</em></span> is a collection
parameter. The minimum across word sizes ensures that any scale of
repetition is detected independently: polyA is caught at ws=1,
dinucleotide repeats at ws=2, etc.</p>
<h2 id="priority-operations">Priority operations</h2>
<ul>
<li>Kmer counting (frequencies)</li>
<li>Fast search / query</li>
<li>Set operations: union, intersection, difference</li>
</ul>
<h2 id="data-types">Data types</h2>
<p>Two types of DNA are represented:</p>
<ul>
<li><strong>Super-kmers</strong>: stored as a nucleotide sequence (2
bits/base, byte-aligned, no padding). Max length: 256 nucleotides (512
bits, 64 bytes). Each super-kmer has a <strong>32-bit
header</strong>:</li>
</ul>
<table>
<thead>
<tr>
<th>Field</th>
<th>Bits</th>
<th>Role</th>
</tr>
</thead>
<tbody>
<tr>
<td>COUNT</td>
<td>24</td>
<td>Occurrence count (≤ 16 M)</td>
</tr>
<tr>
<td>SEQL</td>
<td>8</td>
<td>Sequence length in nucleotides (≤ 256)</td>
</tr>
</tbody>
</table>
<p>Bit layout (MSB to LSB):</p>
<pre><code>[31:8] COUNT [7:0] SEQL</code></pre>
<p>Getters:</p>
<div class="sourceCode" id="cb12"><pre
class="sourceCode rust"><code class="sourceCode rust"><span id="cb12-1"><a href="#cb12-1" aria-hidden="true" tabindex="-1"></a><span class="kw">fn</span> seql(<span class="op">&amp;</span><span class="kw">self</span>) <span class="op">-&gt;</span> <span class="dt">u8</span> <span class="op">{</span> <span class="kw">self</span><span class="op">.</span><span class="dv">0</span> <span class="kw">as</span> <span class="dt">u8</span> <span class="op">}</span></span>
<span id="cb12-2"><a href="#cb12-2" aria-hidden="true" tabindex="-1"></a><span class="kw">fn</span> count(<span class="op">&amp;</span><span class="kw">self</span>) <span class="op">-&gt;</span> <span class="dt">u32</span> <span class="op">{</span> <span class="kw">self</span><span class="op">.</span><span class="dv">0</span> <span class="op">&gt;&gt;</span> <span class="dv">8</span> <span class="op">}</span></span>
<span id="cb12-3"><a href="#cb12-3" aria-hidden="true" tabindex="-1"></a><span class="kw">fn</span> increment(<span class="op">&amp;</span><span class="kw">mut</span> <span class="kw">self</span>) <span class="op">{</span> <span class="kw">self</span><span class="op">.</span><span class="dv">0</span> <span class="op">+=</span> <span class="dv">1</span> <span class="op">&lt;&lt;</span> <span class="dv">8</span><span class="op">;</span> <span class="op">}</span></span></code></pre></div>
<p>PART and KEY were considered as routing fields derived from
<code>hash(minimizer)</code> but are single-use at phase 1 only —
neither is stored. The nucleotide sequence is always byte-aligned (no
offset): after reverse-complementing a super-kmer, a single bit-shift
realigns the sequence so that nucleotide 0 always starts at bit 0 of
word[0]. This shift is paid once at canonisation (phase 1, at most half
of super-kmers) and eliminates any need for an alignment offset field.
The direct consequence for dereplication: the byte array can be hashed
directly without any offset adjustment.</p>
<p>Note: minimizer selection (window minimum) biases the raw minimizer
distribution toward lexicographically small values <span
class="citation"
data-cites="Zheng2020-ji Zheng2021-cc Pan2024-hb Kille2023-px Golan2025-xf">[@Zheng2020-ji;
@Zheng2021-cc; @Pan2024-hb; @Kille2023-px; @Golan2025-xf]</span>;
hashing corrects this before partition routing.</p>
<ul>
<li><strong>Kmers</strong>: short DNA subsequences (k bases),
represented in <code>u64</code>.</li>
</ul>
<h2 id="kmer-representation">Kmer representation</h2>
<p>Each base is encoded on 2 bits, so any kmer with k ≤ 31 fits in a
<code>u64</code>.</p>
<h3 id="bit-layout">Bit layout</h3>
<p>Nucleotide <code>i</code> occupies bits <code>64-i</code> and
<code>63-i</code>. For a kmer of length k, 2k bits are used; the
low-order 64 - 2k bits of the u64 are kept at 0 and have no meaning.</p>
<ul>
<li>Extraction of nuc[i]:
<code>(word &gt;&gt; (63 - 2*i)) &amp; 0b11</code>.</li>
</ul>
<p>For kmers (k ≤ 31), the whole sequence fits in a single
<code>u64</code>. For longer sequences up to 256 nucleotides (512 bits),
the representation extends across multiple words.</p>
<h3 id="complement-and-reverse-complement">Complement and reverse
complement</h3>
<p>The encoding is chosen so that the Watson-Crick complement of any
base is its bitwise NOT (on 2 bits):</p>
<table>
<thead>
<tr>
<th>Base</th>
<th>Encoding</th>
<th>Complement</th>
<th>NOT</th>
</tr>
</thead>
<tbody>
<tr>
<td>A</td>
<td><code>00</code></td>
<td>T</td>
<td><code>11</code></td>
</tr>
<tr>
<td>C</td>
<td><code>01</code></td>
<td>G</td>
<td><code>10</code></td>
</tr>
<tr>
<td>G</td>
<td><code>10</code></td>
<td>C</td>
<td><code>01</code></td>
</tr>
<tr>
<td>T</td>
<td><code>11</code></td>
<td>A</td>
<td><code>00</code></td>
</tr>
</tbody>
</table>
<p>This means <code>complement(base) = ~base &amp; 0b11</code>, and the
reverse complement of a kmer reduces to two cheap operations:</p>
<pre><code>revcomp(kmer) = reverse_bases(~kmer &amp; mask_2k)</code></pre>
<p>where <code>mask_2k = (1 &lt;&lt; 2k) - 1</code> zeros the unused
high bits, and <code>reverse_bases</code> reorders the 2-bit chunks from
LSB to MSB. No lookup table needed.</p>
<p>DNA being double-stranded, each kmer and its reverse complement are
equivalent. The canonical form is defined as:</p>
<pre><code>canonical(kmer) = min(kmer, revcomp(kmer))</code></pre>
<p>This halves the kmer space and ensures strand-independent
counting.</p>
<h2 id="sequence-operations">Sequence operations</h2>
<h3 id="reverse-complement">Reverse complement</h3>
<p>Reverse complement of a super-kmer sequence (used at most once per
super-kmer, during phase 1 canonisation):</p>
<ol type="1">
<li>Reverse byte order</li>
<li>Apply 256-entry lookup table (precomputed reverse-complemented
bytes) to each byte</li>
<li>Bit-shift the result to realign nucleotide 0 to bit 0 of
word[0]</li>
</ol>
<p>The shift amount depends only on the sequence length:
<code>shift = (len % 4) * 2</code> bits. After the shift, the sequence
is byte-aligned with no offset — the same invariant as the original.</p>
<div class="sourceCode" id="cb15"><pre
class="sourceCode rust"><code class="sourceCode rust"><span id="cb15-1"><a href="#cb15-1" aria-hidden="true" tabindex="-1"></a><span class="kw">fn</span> reverse_complement(words<span class="op">:</span> <span class="op">&amp;</span>[<span class="dt">u64</span>]<span class="op">,</span> len<span class="op">:</span> <span class="dt">usize</span>) <span class="op">-&gt;</span> <span class="dt">Vec</span><span class="op">&lt;</span><span class="dt">u64</span><span class="op">&gt;</span> <span class="op">{</span></span>
<span id="cb15-2"><a href="#cb15-2" aria-hidden="true" tabindex="-1"></a> <span class="kw">let</span> bytes <span class="op">=</span> words_to_bytes(words)<span class="op">;</span></span>
<span id="cb15-3"><a href="#cb15-3" aria-hidden="true" tabindex="-1"></a> <span class="kw">let</span> <span class="kw">mut</span> rev<span class="op">:</span> <span class="dt">Vec</span><span class="op">&lt;</span><span class="dt">u8</span><span class="op">&gt;</span> <span class="op">=</span> bytes<span class="op">.</span>iter()<span class="op">.</span>rev()<span class="op">.</span>map(<span class="op">|&amp;</span>b<span class="op">|</span> LOOKUP_TABLE[b <span class="kw">as</span> <span class="dt">usize</span>])<span class="op">.</span>collect()<span class="op">;</span></span>
<span id="cb15-4"><a href="#cb15-4" aria-hidden="true" tabindex="-1"></a> <span class="kw">let</span> shift <span class="op">=</span> (len <span class="op">%</span> <span class="dv">4</span>) <span class="op">*</span> <span class="dv">2</span><span class="op">;</span></span>
<span id="cb15-5"><a href="#cb15-5" aria-hidden="true" tabindex="-1"></a> <span class="cf">if</span> shift <span class="op">&gt;</span> <span class="dv">0</span> <span class="op">{</span> bit_shift_left(<span class="op">&amp;</span><span class="kw">mut</span> rev<span class="op">,</span> shift)<span class="op">;</span> <span class="op">}</span></span>
<span id="cb15-6"><a href="#cb15-6" aria-hidden="true" tabindex="-1"></a> bytes_to_words(rev)</span>
<span id="cb15-7"><a href="#cb15-7" aria-hidden="true" tabindex="-1"></a><span class="op">}</span></span></code></pre></div>
<h2 id="multithreading">Multithreading</h2>
<p>For HPC with 100+ cores: use Rayon for data parallelism (parallel
iterators), std::thread for custom threading, or async with Tokio for
I/O-bound tasks. Equivalent to Go’s goroutines but with
ownership/borrowing safety.</p>
<p>Rayon: Rust crate for automatic data parallelism. Provides parallel
iterators (par_iter) that distribute work across CPU cores, e.g.,
vec.par_iter().map(|x| x*2).collect(). Handles thread
spawning/synchronization transparently. Widely used and maintained by
the Rust project. Orthogonal to Tokio (async runtime): Rayon for CPU
parallelism, Tokio for async I/O.</p>
<p>Data pipelines: use Crossbeam channels (MPMC) with threads/async,
equivalent to Go’s channels + goroutines.</p>
<p>Channels: std::sync::mpsc (built-in, MPSC, simple); crossbeam (crate,
MPMC, faster, more features like select). Use Crossbeam for
multi-producer/multi-consumer scenarios.</p>
<h1 id="bibliographie">Bibliographie</h1>
<p>\bibliography</p>