diff --git a/doc/Probabilitymetrics.png b/doc/Probabilitymetrics.png
new file mode 100644
index 0000000..2a8897b
Binary files /dev/null and b/doc/Probabilitymetrics.png differ
diff --git a/doc/_book/Probabilitymetrics.png b/doc/_book/Probabilitymetrics.png
new file mode 100644
index 0000000..2a8897b
Binary files /dev/null and b/doc/_book/Probabilitymetrics.png differ
diff --git a/doc/_book/annexes.html b/doc/_book/annexes.html
index cb4a3ef..52bfbd9 100644
--- a/doc/_book/annexes.html
+++ b/doc/_book/annexes.html
@@ -7,7 +7,7 @@
 <meta name="viewport" content="width=device-width, initial-scale=1.0, user-scalable=yes">
 
 
-<title>OBITools V4 - 5&nbsp; Annexes</title>
+<title>OBITools V4 - 6&nbsp; Annexes</title>
 <style>
 code{white-space: pre-wrap;}
 span.smallcaps{font-variant: small-caps;}
@@ -70,7 +70,7 @@ ul.task-list li input[type="checkbox"] {
   <header id="quarto-header" class="headroom fixed-top">
   <nav class="quarto-secondary-nav" data-bs-toggle="collapse" data-bs-target="#quarto-sidebar" aria-controls="quarto-sidebar" aria-expanded="false" aria-label="Toggle sidebar navigation" onclick="if (window.quartoToggleHeadroom) { window.quartoToggleHeadroom(); }">
     <div class="container-fluid d-flex justify-content-between">
-      <h1 class="quarto-secondary-nav-title"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></h1>
+      <h1 class="quarto-secondary-nav-title"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></h1>
       <button type="button" class="quarto-btn-toggle btn" aria-label="Show secondary navigation">
         <i class="bi bi-chevron-right"></i>
       </button>
@@ -105,22 +105,27 @@ ul.task-list li input[type="checkbox"] {
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
+  <a href="./formats.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
+  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./annexes.html" class="sidebar-item-text sidebar-link active"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></a>
+  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./annexes.html" class="sidebar-item-text sidebar-link active"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></a>
   </div>
 </li>
         <li class="sidebar-item">
@@ -137,7 +142,7 @@ ul.task-list li input[type="checkbox"] {
     <h2 id="toc-title">Table of contents</h2>
    
   <ul>
-  <li><a href="#sequence-attributes" id="toc-sequence-attributes" class="nav-link active" data-scroll-target="#sequence-attributes"><span class="toc-section-number">5.0.1</span>  Sequence attributes</a></li>
+  <li><a href="#sequence-attributes" id="toc-sequence-attributes" class="nav-link active" data-scroll-target="#sequence-attributes"><span class="toc-section-number">6.0.1</span>  Sequence attributes</a></li>
   </ul>
 </nav>
     </div>
@@ -146,7 +151,7 @@ ul.task-list li input[type="checkbox"] {
 
 <header id="title-block-header" class="quarto-title-block default">
 <div class="quarto-title">
-<h1 class="title d-none d-lg-block"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></h1>
+<h1 class="title d-none d-lg-block"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></h1>
 </div>
 
 
@@ -161,79 +166,79 @@ ul.task-list li input[type="checkbox"] {
 
 </header>
 
-<section id="sequence-attributes" class="level3" data-number="5.0.1">
-<h3 data-number="5.0.1" class="anchored" data-anchor-id="sequence-attributes"><span class="header-section-number">5.0.1</span> Sequence attributes</h3>
-<section id="reserved-sequence-attributes" class="level4" data-number="5.0.1.1">
-<h4 data-number="5.0.1.1" class="anchored" data-anchor-id="reserved-sequence-attributes"><span class="header-section-number">5.0.1.1</span> Reserved sequence attributes</h4>
-<section id="ali_dir" class="level5" data-number="5.0.1.1.1">
-<h5 data-number="5.0.1.1.1" class="anchored" data-anchor-id="ali_dir"><span class="header-section-number">5.0.1.1.1</span> <code>ali_dir</code></h5>
-<section id="type-string" class="level6" data-number="5.0.1.1.1.1">
-<h6 data-number="5.0.1.1.1.1" class="anchored" data-anchor-id="type-string"><span class="header-section-number">5.0.1.1.1.1</span> Type : <code>string</code></h6>
+<section id="sequence-attributes" class="level3" data-number="6.0.1">
+<h3 data-number="6.0.1" class="anchored" data-anchor-id="sequence-attributes"><span class="header-section-number">6.0.1</span> Sequence attributes</h3>
+<section id="reserved-sequence-attributes" class="level4" data-number="6.0.1.1">
+<h4 data-number="6.0.1.1" class="anchored" data-anchor-id="reserved-sequence-attributes"><span class="header-section-number">6.0.1.1</span> Reserved sequence attributes</h4>
+<section id="ali_dir" class="level5" data-number="6.0.1.1.1">
+<h5 data-number="6.0.1.1.1" class="anchored" data-anchor-id="ali_dir"><span class="header-section-number">6.0.1.1.1</span> <code>ali_dir</code></h5>
+<section id="type-string" class="level6" data-number="6.0.1.1.1.1">
+<h6 data-number="6.0.1.1.1.1" class="anchored" data-anchor-id="type-string"><span class="header-section-number">6.0.1.1.1.1</span> Type : <code>string</code></h6>
 <p>The attribute can contain 2 string values <code>"left"</code> or <code>"right".</code></p>
 </section>
-<section id="set-by-the-obipairing-tool" class="level6" data-number="5.0.1.1.1.2">
-<h6 data-number="5.0.1.1.1.2" class="anchored" data-anchor-id="set-by-the-obipairing-tool"><span class="header-section-number">5.0.1.1.1.2</span> Set by the <em>obipairing</em> tool</h6>
+<section id="set-by-the-obipairing-tool" class="level6" data-number="6.0.1.1.1.2">
+<h6 data-number="6.0.1.1.1.2" class="anchored" data-anchor-id="set-by-the-obipairing-tool"><span class="header-section-number">6.0.1.1.1.2</span> Set by the <em>obipairing</em> tool</h6>
 <p>The alignment generated by <em>obipairing</em> is a 3’-end gap free algorithm. Two cases can occur when aligning the forward and reverse reads. If the barcode is long enough, both the reads overlap only on their 3’ ends. In such case, the alignment direction <code>ali_dir</code> is set to <em>left</em>. If the barcode is shorter than the read length, the paired reads overlap by their 5’ ends, and the complete barcode is sequenced by both the reads. In that later case, <code>ali_dir</code> is set to <em>right</em>.</p>
 </section>
 </section>
-<section id="ali_length" class="level5" data-number="5.0.1.1.2">
-<h5 data-number="5.0.1.1.2" class="anchored" data-anchor-id="ali_length"><span class="header-section-number">5.0.1.1.2</span> <code>ali_length</code></h5>
-<section id="set-by-the-obipairing-tool-1" class="level6" data-number="5.0.1.1.2.1">
-<h6 data-number="5.0.1.1.2.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-1"><span class="header-section-number">5.0.1.1.2.1</span> Set by the <em>obipairing</em> tool</h6>
+<section id="ali_length" class="level5" data-number="6.0.1.1.2">
+<h5 data-number="6.0.1.1.2" class="anchored" data-anchor-id="ali_length"><span class="header-section-number">6.0.1.1.2</span> <code>ali_length</code></h5>
+<section id="set-by-the-obipairing-tool-1" class="level6" data-number="6.0.1.1.2.1">
+<h6 data-number="6.0.1.1.2.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-1"><span class="header-section-number">6.0.1.1.2.1</span> Set by the <em>obipairing</em> tool</h6>
 <p>Length of the aligned parts when merging forward and reverse reads</p>
 </section>
 </section>
-<section id="count-the-number-of-sequence-occurrences" class="level5" data-number="5.0.1.1.3">
-<h5 data-number="5.0.1.1.3" class="anchored" data-anchor-id="count-the-number-of-sequence-occurrences"><span class="header-section-number">5.0.1.1.3</span> <code>count</code> : the number of sequence occurrences</h5>
-<section id="set-by-the-obiuniq-tool" class="level6" data-number="5.0.1.1.3.1">
-<h6 data-number="5.0.1.1.3.1" class="anchored" data-anchor-id="set-by-the-obiuniq-tool"><span class="header-section-number">5.0.1.1.3.1</span> Set by the <em>obiuniq</em> tool</h6>
+<section id="count-the-number-of-sequence-occurrences" class="level5" data-number="6.0.1.1.3">
+<h5 data-number="6.0.1.1.3" class="anchored" data-anchor-id="count-the-number-of-sequence-occurrences"><span class="header-section-number">6.0.1.1.3</span> <code>count</code> : the number of sequence occurrences</h5>
+<section id="set-by-the-obiuniq-tool" class="level6" data-number="6.0.1.1.3.1">
+<h6 data-number="6.0.1.1.3.1" class="anchored" data-anchor-id="set-by-the-obiuniq-tool"><span class="header-section-number">6.0.1.1.3.1</span> Set by the <em>obiuniq</em> tool</h6>
 <p>The <code>count</code> attribute indicates how-many strictly identical sequences have been merged in a single record. It contains an integer value. If it is absent this means that the sequence record represents a single occurrence of the sequence.</p>
 </section>
-<section id="getter-method-count" class="level6" data-number="5.0.1.1.3.2">
-<h6 data-number="5.0.1.1.3.2" class="anchored" data-anchor-id="getter-method-count"><span class="header-section-number">5.0.1.1.3.2</span> Getter : method <code>Count()</code></h6>
+<section id="getter-method-count" class="level6" data-number="6.0.1.1.3.2">
+<h6 data-number="6.0.1.1.3.2" class="anchored" data-anchor-id="getter-method-count"><span class="header-section-number">6.0.1.1.3.2</span> Getter : method <code>Count()</code></h6>
 <p>The <code>Count()</code> method allows to access to the count attribute as an integer value. If the <code>count</code> attribute is not defined for the given sequence, the value <em>1</em> is returned</p>
 </section>
 </section>
-<section id="merged_" class="level5" data-number="5.0.1.1.4">
-<h5 data-number="5.0.1.1.4" class="anchored" data-anchor-id="merged_"><span class="header-section-number">5.0.1.1.4</span> <code>merged_*</code></h5>
-<section id="type-mapstringint" class="level6" data-number="5.0.1.1.4.1">
-<h6 data-number="5.0.1.1.4.1" class="anchored" data-anchor-id="type-mapstringint"><span class="header-section-number">5.0.1.1.4.1</span> Type : <code>map[string]int</code></h6>
+<section id="merged_" class="level5" data-number="6.0.1.1.4">
+<h5 data-number="6.0.1.1.4" class="anchored" data-anchor-id="merged_"><span class="header-section-number">6.0.1.1.4</span> <code>merged_*</code></h5>
+<section id="type-mapstringint" class="level6" data-number="6.0.1.1.4.1">
+<h6 data-number="6.0.1.1.4.1" class="anchored" data-anchor-id="type-mapstringint"><span class="header-section-number">6.0.1.1.4.1</span> Type : <code>map[string]int</code></h6>
 </section>
-<section id="set-by-the-obiuniq-tool-1" class="level6" data-number="5.0.1.1.4.2">
-<h6 data-number="5.0.1.1.4.2" class="anchored" data-anchor-id="set-by-the-obiuniq-tool-1"><span class="header-section-number">5.0.1.1.4.2</span> Set by the <em>obiuniq</em> tool</h6>
+<section id="set-by-the-obiuniq-tool-1" class="level6" data-number="6.0.1.1.4.2">
+<h6 data-number="6.0.1.1.4.2" class="anchored" data-anchor-id="set-by-the-obiuniq-tool-1"><span class="header-section-number">6.0.1.1.4.2</span> Set by the <em>obiuniq</em> tool</h6>
 <p>The <code>-m</code> option of the <em>obiuniq</em> tools allows for keeping track of the distribution of the values stored in given attribute of interest. Often this option is used to summarise distribution of a sequence variant accross samples when <em>obiuniq</em> is run after running <em>obimultiplex</em>. The actual name of the attribute depends on the name of the monitored attribute. If <code>-m</code> option is used with the attribute <em>sample</em>, then this attribute names <em>merged_sample</em>.</p>
 </section>
 </section>
-<section id="mode" class="level5" data-number="5.0.1.1.5">
-<h5 data-number="5.0.1.1.5" class="anchored" data-anchor-id="mode"><span class="header-section-number">5.0.1.1.5</span> <code>mode</code></h5>
-<section id="set-by-the-obipairing-tool-2" class="level6" data-number="5.0.1.1.5.1">
-<h6 data-number="5.0.1.1.5.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-2"><span class="header-section-number">5.0.1.1.5.1</span> Set by the <em>obipairing</em> tool</h6>
+<section id="mode" class="level5" data-number="6.0.1.1.5">
+<h5 data-number="6.0.1.1.5" class="anchored" data-anchor-id="mode"><span class="header-section-number">6.0.1.1.5</span> <code>mode</code></h5>
+<section id="set-by-the-obipairing-tool-2" class="level6" data-number="6.0.1.1.5.1">
+<h6 data-number="6.0.1.1.5.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-2"><span class="header-section-number">6.0.1.1.5.1</span> Set by the <em>obipairing</em> tool</h6>
 <p><strong><code>obitag_ref_index</code></strong></p>
 </section>
-<section id="set-by-the-obirefidx-tool." class="level6" data-number="5.0.1.1.5.2">
-<h6 data-number="5.0.1.1.5.2" class="anchored" data-anchor-id="set-by-the-obirefidx-tool."><span class="header-section-number">5.0.1.1.5.2</span> Set by the <em>obirefidx</em> tool.</h6>
+<section id="set-by-the-obirefidx-tool." class="level6" data-number="6.0.1.1.5.2">
+<h6 data-number="6.0.1.1.5.2" class="anchored" data-anchor-id="set-by-the-obirefidx-tool."><span class="header-section-number">6.0.1.1.5.2</span> Set by the <em>obirefidx</em> tool.</h6>
 <p>It resumes to which taxonomic annotation a match to that sequence must lead according to the number of differences existing between the query sequence and the reference sequence having that tag.</p>
 </section>
-<section id="getter-method-count-1" class="level6" data-number="5.0.1.1.5.3">
-<h6 data-number="5.0.1.1.5.3" class="anchored" data-anchor-id="getter-method-count-1"><span class="header-section-number">5.0.1.1.5.3</span> Getter : method <code>Count()</code></h6>
+<section id="getter-method-count-1" class="level6" data-number="6.0.1.1.5.3">
+<h6 data-number="6.0.1.1.5.3" class="anchored" data-anchor-id="getter-method-count-1"><span class="header-section-number">6.0.1.1.5.3</span> Getter : method <code>Count()</code></h6>
 </section>
 </section>
-<section id="pairing_mismatches" class="level5" data-number="5.0.1.1.6">
-<h5 data-number="5.0.1.1.6" class="anchored" data-anchor-id="pairing_mismatches"><span class="header-section-number">5.0.1.1.6</span> <code>pairing_mismatches</code></h5>
-<section id="set-by-the-obipairing-tool-3" class="level6" data-number="5.0.1.1.6.1">
-<h6 data-number="5.0.1.1.6.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-3"><span class="header-section-number">5.0.1.1.6.1</span> Set by the <em>obipairing</em> tool</h6>
+<section id="pairing_mismatches" class="level5" data-number="6.0.1.1.6">
+<h5 data-number="6.0.1.1.6" class="anchored" data-anchor-id="pairing_mismatches"><span class="header-section-number">6.0.1.1.6</span> <code>pairing_mismatches</code></h5>
+<section id="set-by-the-obipairing-tool-3" class="level6" data-number="6.0.1.1.6.1">
+<h6 data-number="6.0.1.1.6.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-3"><span class="header-section-number">6.0.1.1.6.1</span> Set by the <em>obipairing</em> tool</h6>
 </section>
 </section>
-<section id="score" class="level5" data-number="5.0.1.1.7">
-<h5 data-number="5.0.1.1.7" class="anchored" data-anchor-id="score"><span class="header-section-number">5.0.1.1.7</span> <code>score</code></h5>
-<section id="set-by-the-obipairing-tool-4" class="level6" data-number="5.0.1.1.7.1">
-<h6 data-number="5.0.1.1.7.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-4"><span class="header-section-number">5.0.1.1.7.1</span> Set by the <em>obipairing</em> tool</h6>
+<section id="score" class="level5" data-number="6.0.1.1.7">
+<h5 data-number="6.0.1.1.7" class="anchored" data-anchor-id="score"><span class="header-section-number">6.0.1.1.7</span> <code>score</code></h5>
+<section id="set-by-the-obipairing-tool-4" class="level6" data-number="6.0.1.1.7.1">
+<h6 data-number="6.0.1.1.7.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-4"><span class="header-section-number">6.0.1.1.7.1</span> Set by the <em>obipairing</em> tool</h6>
 </section>
 </section>
-<section id="score_norm" class="level5" data-number="5.0.1.1.8">
-<h5 data-number="5.0.1.1.8" class="anchored" data-anchor-id="score_norm"><span class="header-section-number">5.0.1.1.8</span> <code>score_norm</code></h5>
-<section id="set-by-the-obipairing-tool-5" class="level6" data-number="5.0.1.1.8.1">
-<h6 data-number="5.0.1.1.8.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-5"><span class="header-section-number">5.0.1.1.8.1</span> Set by the <em>obipairing</em> tool</h6>
+<section id="score_norm" class="level5" data-number="6.0.1.1.8">
+<h5 data-number="6.0.1.1.8" class="anchored" data-anchor-id="score_norm"><span class="header-section-number">6.0.1.1.8</span> <code>score_norm</code></h5>
+<section id="set-by-the-obipairing-tool-5" class="level6" data-number="6.0.1.1.8.1">
+<h6 data-number="6.0.1.1.8.1" class="anchored" data-anchor-id="set-by-the-obipairing-tool-5"><span class="header-section-number">6.0.1.1.8.1</span> Set by the <em>obipairing</em> tool</h6>
 
 
 </section>
@@ -378,7 +383,7 @@ window.document.addEventListener("DOMContentLoaded", function (event) {
 <nav class="page-navigation">
   <div class="nav-page nav-page-previous">
       <a href="./library.html" class="pagination-link">
-        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></span>
+        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></span>
       </a>          
   </div>
   <div class="nav-page nav-page-next">
diff --git a/doc/_book/commands.html b/doc/_book/commands.html
index bbd7053..2d0eaa3 100644
--- a/doc/_book/commands.html
+++ b/doc/_book/commands.html
@@ -7,7 +7,7 @@
 <meta name="viewport" content="width=device-width, initial-scale=1.0, user-scalable=yes">
 
 
-<title>OBITools V4 - 3&nbsp; The OBITools V4 commands</title>
+<title>OBITools V4 - 4&nbsp; The OBITools V4 commands</title>
 <style>
 code{white-space: pre-wrap;}
 span.smallcaps{font-variant: small-caps;}
@@ -90,7 +90,7 @@ div.csl-indent {
   <header id="quarto-header" class="headroom fixed-top">
   <nav class="quarto-secondary-nav" data-bs-toggle="collapse" data-bs-target="#quarto-sidebar" aria-controls="quarto-sidebar" aria-expanded="false" aria-label="Toggle sidebar navigation" onclick="if (window.quartoToggleHeadroom) { window.quartoToggleHeadroom(); }">
     <div class="container-fluid d-flex justify-content-between">
-      <h1 class="quarto-secondary-nav-title"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></h1>
+      <h1 class="quarto-secondary-nav-title"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></h1>
       <button type="button" class="quarto-btn-toggle btn" aria-label="Show secondary navigation">
         <i class="bi bi-chevron-right"></i>
       </button>
@@ -125,22 +125,27 @@ div.csl-indent {
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
+  <a href="./formats.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./commands.html" class="sidebar-item-text sidebar-link active"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
+  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  <a href="./commands.html" class="sidebar-item-text sidebar-link active"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></a>
+  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></a>
   </div>
 </li>
         <li class="sidebar-item">
@@ -157,29 +162,29 @@ div.csl-indent {
     <h2 id="toc-title">Table of contents</h2>
    
   <ul>
-  <li><a href="#specifying-the-input-files-to-obitools-commands" id="toc-specifying-the-input-files-to-obitools-commands" class="nav-link active" data-scroll-target="#specifying-the-input-files-to-obitools-commands"><span class="toc-section-number">3.1</span>  Specifying the input files to <em>OBITools</em> commands</a></li>
-  <li><a href="#options-common-to-most-of-the-obitools-commands" id="toc-options-common-to-most-of-the-obitools-commands" class="nav-link" data-scroll-target="#options-common-to-most-of-the-obitools-commands"><span class="toc-section-number">3.2</span>  Options common to most of the <em>OBITools</em> commands</a>
+  <li><a href="#specifying-the-input-files-to-obitools-commands" id="toc-specifying-the-input-files-to-obitools-commands" class="nav-link active" data-scroll-target="#specifying-the-input-files-to-obitools-commands"><span class="toc-section-number">4.1</span>  Specifying the input files to <em>OBITools</em> commands</a></li>
+  <li><a href="#options-common-to-most-of-the-obitools-commands" id="toc-options-common-to-most-of-the-obitools-commands" class="nav-link" data-scroll-target="#options-common-to-most-of-the-obitools-commands"><span class="toc-section-number">4.2</span>  Options common to most of the <em>OBITools</em> commands</a>
   <ul class="collapse">
-  <li><a href="#specifying-input-format" id="toc-specifying-input-format" class="nav-link" data-scroll-target="#specifying-input-format"><span class="toc-section-number">3.2.1</span>  Specifying input format</a></li>
-  <li><a href="#specifying-output-format" id="toc-specifying-output-format" class="nav-link" data-scroll-target="#specifying-output-format"><span class="toc-section-number">3.2.2</span>  Specifying output format</a></li>
-  <li><a href="#format-of-the-annotations-in-fasta-and-fastq-files" id="toc-format-of-the-annotations-in-fasta-and-fastq-files" class="nav-link" data-scroll-target="#format-of-the-annotations-in-fasta-and-fastq-files"><span class="toc-section-number">3.2.3</span>  Format of the annotations in Fasta and Fastq files</a></li>
+  <li><a href="#specifying-input-format" id="toc-specifying-input-format" class="nav-link" data-scroll-target="#specifying-input-format"><span class="toc-section-number">4.2.1</span>  Specifying input format</a></li>
+  <li><a href="#specifying-output-format" id="toc-specifying-output-format" class="nav-link" data-scroll-target="#specifying-output-format"><span class="toc-section-number">4.2.2</span>  Specifying output format</a></li>
+  <li><a href="#the-fasta-and-fastq-annotations-format" id="toc-the-fasta-and-fastq-annotations-format" class="nav-link" data-scroll-target="#the-fasta-and-fastq-annotations-format"><span class="toc-section-number">4.2.3</span>  The Fasta and Fastq annotations format</a></li>
   </ul></li>
-  <li><a href="#obitools-expression-language" id="toc-obitools-expression-language" class="nav-link" data-scroll-target="#obitools-expression-language"><span class="toc-section-number">3.3</span>  OBITools expression language</a>
+  <li><a href="#obitools-expression-language" id="toc-obitools-expression-language" class="nav-link" data-scroll-target="#obitools-expression-language"><span class="toc-section-number">4.3</span>  OBITools expression language</a>
   <ul class="collapse">
-  <li><a href="#variables-usable-in-the-expression" id="toc-variables-usable-in-the-expression" class="nav-link" data-scroll-target="#variables-usable-in-the-expression"><span class="toc-section-number">3.3.1</span>  Variables usable in the expression</a></li>
-  <li><a href="#function-defined-in-the-language" id="toc-function-defined-in-the-language" class="nav-link" data-scroll-target="#function-defined-in-the-language"><span class="toc-section-number">3.3.2</span>  Function defined in the language</a></li>
-  <li><a href="#accessing-to-the-sequence-annotations" id="toc-accessing-to-the-sequence-annotations" class="nav-link" data-scroll-target="#accessing-to-the-sequence-annotations"><span class="toc-section-number">3.3.3</span>  Accessing to the sequence annotations</a></li>
+  <li><a href="#variables-usable-in-the-expression" id="toc-variables-usable-in-the-expression" class="nav-link" data-scroll-target="#variables-usable-in-the-expression"><span class="toc-section-number">4.3.1</span>  Variables usable in the expression</a></li>
+  <li><a href="#function-defined-in-the-language" id="toc-function-defined-in-the-language" class="nav-link" data-scroll-target="#function-defined-in-the-language"><span class="toc-section-number">4.3.2</span>  Function defined in the language</a></li>
+  <li><a href="#accessing-to-the-sequence-annotations" id="toc-accessing-to-the-sequence-annotations" class="nav-link" data-scroll-target="#accessing-to-the-sequence-annotations"><span class="toc-section-number">4.3.3</span>  Accessing to the sequence annotations</a></li>
   </ul></li>
-  <li><a href="#metabarcode-design-and-quality-assessment" id="toc-metabarcode-design-and-quality-assessment" class="nav-link" data-scroll-target="#metabarcode-design-and-quality-assessment"><span class="toc-section-number">3.4</span>  Metabarcode design and quality assessment</a></li>
-  <li><a href="#file-format-conversions" id="toc-file-format-conversions" class="nav-link" data-scroll-target="#file-format-conversions"><span class="toc-section-number">3.5</span>  File format conversions</a></li>
-  <li><a href="#sequence-annotations" id="toc-sequence-annotations" class="nav-link" data-scroll-target="#sequence-annotations"><span class="toc-section-number">3.6</span>  Sequence annotations</a></li>
-  <li><a href="#computations-on-sequences" id="toc-computations-on-sequences" class="nav-link" data-scroll-target="#computations-on-sequences"><span class="toc-section-number">3.7</span>  Computations on sequences</a>
+  <li><a href="#metabarcode-design-and-quality-assessment" id="toc-metabarcode-design-and-quality-assessment" class="nav-link" data-scroll-target="#metabarcode-design-and-quality-assessment"><span class="toc-section-number">4.4</span>  Metabarcode design and quality assessment</a></li>
+  <li><a href="#file-format-conversions" id="toc-file-format-conversions" class="nav-link" data-scroll-target="#file-format-conversions"><span class="toc-section-number">4.5</span>  File format conversions</a></li>
+  <li><a href="#sequence-annotations" id="toc-sequence-annotations" class="nav-link" data-scroll-target="#sequence-annotations"><span class="toc-section-number">4.6</span>  Sequence annotations</a></li>
+  <li><a href="#computations-on-sequences" id="toc-computations-on-sequences" class="nav-link" data-scroll-target="#computations-on-sequences"><span class="toc-section-number">4.7</span>  Computations on sequences</a>
   <ul class="collapse">
-  <li><a href="#obipairing" id="toc-obipairing" class="nav-link" data-scroll-target="#obipairing"><span class="toc-section-number">3.7.1</span>  <code>obipairing</code></a></li>
+  <li><a href="#obipairing" id="toc-obipairing" class="nav-link" data-scroll-target="#obipairing"><span class="toc-section-number">4.7.1</span>  <code>obipairing</code></a></li>
   </ul></li>
-  <li><a href="#sequence-sampling-and-filtering" id="toc-sequence-sampling-and-filtering" class="nav-link" data-scroll-target="#sequence-sampling-and-filtering"><span class="toc-section-number">3.8</span>  Sequence sampling and filtering</a>
+  <li><a href="#sequence-sampling-and-filtering" id="toc-sequence-sampling-and-filtering" class="nav-link" data-scroll-target="#sequence-sampling-and-filtering"><span class="toc-section-number">4.8</span>  Sequence sampling and filtering</a>
   <ul class="collapse">
-  <li><a href="#utilities" id="toc-utilities" class="nav-link" data-scroll-target="#utilities"><span class="toc-section-number">3.8.1</span>  Utilities</a></li>
+  <li><a href="#utilities" id="toc-utilities" class="nav-link" data-scroll-target="#utilities"><span class="toc-section-number">4.8.1</span>  Utilities</a></li>
   </ul></li>
   </ul>
 </nav>
@@ -189,7 +194,7 @@ div.csl-indent {
 
 <header id="title-block-header" class="quarto-title-block default">
 <div class="quarto-title">
-<h1 class="title d-none d-lg-block"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></h1>
+<h1 class="title d-none d-lg-block"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></h1>
 </div>
 
 
@@ -204,26 +209,26 @@ div.csl-indent {
 
 </header>
 
-<section id="specifying-the-input-files-to-obitools-commands" class="level2" data-number="3.1">
-<h2 data-number="3.1" class="anchored" data-anchor-id="specifying-the-input-files-to-obitools-commands"><span class="header-section-number">3.1</span> Specifying the input files to <em>OBITools</em> commands</h2>
+<section id="specifying-the-input-files-to-obitools-commands" class="level2" data-number="4.1">
+<h2 data-number="4.1" class="anchored" data-anchor-id="specifying-the-input-files-to-obitools-commands"><span class="header-section-number">4.1</span> Specifying the input files to <em>OBITools</em> commands</h2>
 </section>
-<section id="options-common-to-most-of-the-obitools-commands" class="level2" data-number="3.2">
-<h2 data-number="3.2" class="anchored" data-anchor-id="options-common-to-most-of-the-obitools-commands"><span class="header-section-number">3.2</span> Options common to most of the <em>OBITools</em> commands</h2>
-<section id="specifying-input-format" class="level3" data-number="3.2.1">
-<h3 data-number="3.2.1" class="anchored" data-anchor-id="specifying-input-format"><span class="header-section-number">3.2.1</span> Specifying input format</h3>
-<p>Five sequence formats are accepted for input files. <a href="#fasta-classical" title="Fasta format description">Fasta</a> and <a href="#fastq-classical" title="Fastq format description">Fastq</a> are the main ones, EMBL and Genbank allow the use of flat files produced by these two international databases. The last one, ecoPCR, is maintained for compatibility with previous <em>OBITools</em> and allows to read <em>ecoPCR</em> outputs as sequence files.</p>
+<section id="options-common-to-most-of-the-obitools-commands" class="level2" data-number="4.2">
+<h2 data-number="4.2" class="anchored" data-anchor-id="options-common-to-most-of-the-obitools-commands"><span class="header-section-number">4.2</span> Options common to most of the <em>OBITools</em> commands</h2>
+<section id="specifying-input-format" class="level3" data-number="4.2.1">
+<h3 data-number="4.2.1" class="anchored" data-anchor-id="specifying-input-format"><span class="header-section-number">4.2.1</span> Specifying input format</h3>
+<p>Five sequence formats are accepted for input files. <em>Fasta</em> (<a href="formats.html#sec-fasta"><span>Section&nbsp;2.1.2</span></a>) and <em>Fastq</em> (<a href="formats.html#sec-fastq"><span>Section&nbsp;2.1.3</span></a>) are the main ones, EMBL and Genbank allow the use of flat files produced by these two international databases. The last one, ecoPCR, is maintained for compatibility with previous <em>OBITools</em> and allows to read <em>ecoPCR</em> outputs as sequence files.</p>
 <ul>
 <li><code>--ecopcr</code> : Read data following the <em>ecoPCR</em> output format.</li>
 <li><code>--embl</code> Read data following the <em>EMBL</em> flatfile format.</li>
 <li><code>--genbank</code> Read data following the <em>Genbank</em> flatfile format.</li>
 </ul>
-<p>Several encoding schemes have been proposed for quality scores in <a href="#fastq-classical" title="Fastq format description">Fastq</a> format. Currently, <em>OBITools</em> considers Sanger encoding as the standard. For reasons of compatibility with older datasets produced with <em>Solexa</em> sequencers, it is possible, by using the following option, to force the use of the corresponding quality encoding scheme when reading these older files.</p>
+<p>Several encoding schemes have been proposed for quality scores in <em>Fastq</em> format. Currently, <em>OBITools</em> considers Sanger encoding as the standard. For reasons of compatibility with older datasets produced with <em>Solexa</em> sequencers, it is possible, by using the following option, to force the use of the corresponding quality encoding scheme when reading these older files.</p>
 <ul>
 <li><code>--solexa</code> Decodes quality string according to the Solexa specification. (default: false)</li>
 </ul>
 </section>
-<section id="specifying-output-format" class="level3" data-number="3.2.2">
-<h3 data-number="3.2.2" class="anchored" data-anchor-id="specifying-output-format"><span class="header-section-number">3.2.2</span> Specifying output format</h3>
+<section id="specifying-output-format" class="level3" data-number="4.2.2">
+<h3 data-number="4.2.2" class="anchored" data-anchor-id="specifying-output-format"><span class="header-section-number">4.2.2</span> Specifying output format</h3>
 <p>Only two output sequence formats are supported by OBITools, Fasta and Fastq. Fastq is used when output sequences are associated with quality information. Otherwise, Fasta is the default format. However, it is possible to force the output format by using one of the following two options. Forcing the use of Fasta results in the loss of quality information. Conversely, when the Fastq format is forced with sequences that have no quality data, dummy qualities set to 40 for each nucleotide are added.</p>
 <ul>
 <li><code>--fasta-output</code> Read data following the ecoPCR output format.</li>
@@ -231,12 +236,12 @@ div.csl-indent {
 </ul>
 <p>OBITools allows multiple input files to be specified for a single command.</p>
 <ul>
-<li><code>--no-order</code> When several input files are provided, indicates that there is no order among them. (default: false)</li>
+<li><code>--no-order</code> When several input files are provided, indicates that there is no order among them. (default: false). Using such option can increase a lot the processing of the data.</li>
 </ul>
 </section>
-<section id="format-of-the-annotations-in-fasta-and-fastq-files" class="level3" data-number="3.2.3">
-<h3 data-number="3.2.3" class="anchored" data-anchor-id="format-of-the-annotations-in-fasta-and-fastq-files"><span class="header-section-number">3.2.3</span> Format of the annotations in Fasta and Fastq files</h3>
-<p>OBITools extend the <a href="#fasta-classical" title="Fasta format description">Fasta</a> and <a href="#fastq-classical" title="Fastq format description">Fastq</a> formats by introducing a format for the title lines of these formats allowing to annotate every sequence. While the previous version of OBITools used an <em>ad-hoc</em> format for these annotation, this new version introduce the usage of the standard JSON format to store them.</p>
+<section id="the-fasta-and-fastq-annotations-format" class="level3" data-number="4.2.3">
+<h3 data-number="4.2.3" class="anchored" data-anchor-id="the-fasta-and-fastq-annotations-format"><span class="header-section-number">4.2.3</span> The Fasta and Fastq annotations format</h3>
+<p>OBITools extend the <a href="#the-fasta-sequence-format">Fasta</a> and <a href="#the-fastq-sequence-format">Fastq</a> formats by introducing a format for the title lines of these formats allowing to annotate every sequence. While the previous version of OBITools used an <em>ad-hoc</em> format for these annotation, this new version introduce the usage of the standard JSON format to store them.</p>
 <p>On input, OBITools automatically recognize the format of the annotations, but two options allows to force the parsing following one of them. You should normally not need to use these options.</p>
 <ul>
 <li><p><code>--input-OBI-header</code> FASTA/FASTQ title line annotations follow OBI format. (default: false)</p></li>
@@ -247,8 +252,8 @@ div.csl-indent {
 <li><p><code>--output-OBI-header|-O</code> output FASTA/FASTQ title line annotations follow OBI format. (default: false)</p></li>
 <li><p><code>--output-json-header</code> output FASTA/FASTQ title line annotations follow json format. (default: false)</p></li>
 </ul>
-<section id="system-related-options" class="level4" data-number="3.2.3.1">
-<h4 data-number="3.2.3.1" class="anchored" data-anchor-id="system-related-options"><span class="header-section-number">3.2.3.1</span> System related options</h4>
+<section id="system-related-options" class="level4" data-number="4.2.3.1">
+<h4 data-number="4.2.3.1" class="anchored" data-anchor-id="system-related-options"><span class="header-section-number">4.2.3.1</span> System related options</h4>
 <ul>
 <li><code>--debug</code> (default: false)</li>
 <li><code>--help\|-h\|-?</code> (default: false)</li>
@@ -258,78 +263,78 @@ div.csl-indent {
 </section>
 </section>
 </section>
-<section id="obitools-expression-language" class="level2" data-number="3.3">
-<h2 data-number="3.3" class="anchored" data-anchor-id="obitools-expression-language"><span class="header-section-number">3.3</span> OBITools expression language</h2>
+<section id="obitools-expression-language" class="level2" data-number="4.3">
+<h2 data-number="4.3" class="anchored" data-anchor-id="obitools-expression-language"><span class="header-section-number">4.3</span> OBITools expression language</h2>
 <p>Several OBITools (<em>e.g.</em> obigrep, obiannotate) allow the user to specify some simple expressions to compute values or define predicates. This expressions are parsed and evaluated using the <a href="https://pkg.go.dev/github.com/PaesslerAG/gval" title="Gval (Go eVALuate) for evaluating arbitrary expressions Go-like expressions.">gval</a> go package, which allows for evaluating go-Like expression.</p>
-<section id="variables-usable-in-the-expression" class="level3" data-number="3.3.1">
-<h3 data-number="3.3.1" class="anchored" data-anchor-id="variables-usable-in-the-expression"><span class="header-section-number">3.3.1</span> Variables usable in the expression</h3>
-<section id="sequence" class="level4" data-number="3.3.1.1">
-<h4 data-number="3.3.1.1" class="anchored" data-anchor-id="sequence"><span class="header-section-number">3.3.1.1</span> sequence</h4>
+<section id="variables-usable-in-the-expression" class="level3" data-number="4.3.1">
+<h3 data-number="4.3.1" class="anchored" data-anchor-id="variables-usable-in-the-expression"><span class="header-section-number">4.3.1</span> Variables usable in the expression</h3>
+<section id="sequence" class="level4" data-number="4.3.1.1">
+<h4 data-number="4.3.1.1" class="anchored" data-anchor-id="sequence"><span class="header-section-number">4.3.1.1</span> sequence</h4>
 <p>sequence is the sequence object on which the expression is evaluated</p>
 </section>
-<section id="annotation" class="level4" data-number="3.3.1.2">
-<h4 data-number="3.3.1.2" class="anchored" data-anchor-id="annotation"><span class="header-section-number">3.3.1.2</span> annotation</h4>
+<section id="annotation" class="level4" data-number="4.3.1.2">
+<h4 data-number="4.3.1.2" class="anchored" data-anchor-id="annotation"><span class="header-section-number">4.3.1.2</span> annotation</h4>
 </section>
 </section>
-<section id="function-defined-in-the-language" class="level3" data-number="3.3.2">
-<h3 data-number="3.3.2" class="anchored" data-anchor-id="function-defined-in-the-language"><span class="header-section-number">3.3.2</span> Function defined in the language</h3>
-<section id="len" class="level4" data-number="3.3.2.1">
-<h4 data-number="3.3.2.1" class="anchored" data-anchor-id="len"><span class="header-section-number">3.3.2.1</span> len</h4>
+<section id="function-defined-in-the-language" class="level3" data-number="4.3.2">
+<h3 data-number="4.3.2" class="anchored" data-anchor-id="function-defined-in-the-language"><span class="header-section-number">4.3.2</span> Function defined in the language</h3>
+<section id="len" class="level4" data-number="4.3.2.1">
+<h4 data-number="4.3.2.1" class="anchored" data-anchor-id="len"><span class="header-section-number">4.3.2.1</span> len</h4>
 </section>
-<section id="ismap" class="level4" data-number="3.3.2.2">
-<h4 data-number="3.3.2.2" class="anchored" data-anchor-id="ismap"><span class="header-section-number">3.3.2.2</span> ismap</h4>
+<section id="ismap" class="level4" data-number="4.3.2.2">
+<h4 data-number="4.3.2.2" class="anchored" data-anchor-id="ismap"><span class="header-section-number">4.3.2.2</span> ismap</h4>
 </section>
-<section id="hasattribute" class="level4" data-number="3.3.2.3">
-<h4 data-number="3.3.2.3" class="anchored" data-anchor-id="hasattribute"><span class="header-section-number">3.3.2.3</span> hasattribute</h4>
+<section id="hasattribute" class="level4" data-number="4.3.2.3">
+<h4 data-number="4.3.2.3" class="anchored" data-anchor-id="hasattribute"><span class="header-section-number">4.3.2.3</span> hasattribute</h4>
 </section>
-<section id="min" class="level4" data-number="3.3.2.4">
-<h4 data-number="3.3.2.4" class="anchored" data-anchor-id="min"><span class="header-section-number">3.3.2.4</span> min</h4>
+<section id="min" class="level4" data-number="4.3.2.4">
+<h4 data-number="4.3.2.4" class="anchored" data-anchor-id="min"><span class="header-section-number">4.3.2.4</span> min</h4>
 </section>
-<section id="max" class="level4" data-number="3.3.2.5">
-<h4 data-number="3.3.2.5" class="anchored" data-anchor-id="max"><span class="header-section-number">3.3.2.5</span> max</h4>
+<section id="max" class="level4" data-number="4.3.2.5">
+<h4 data-number="4.3.2.5" class="anchored" data-anchor-id="max"><span class="header-section-number">4.3.2.5</span> max</h4>
 </section>
 </section>
-<section id="accessing-to-the-sequence-annotations" class="level3" data-number="3.3.3">
-<h3 data-number="3.3.3" class="anchored" data-anchor-id="accessing-to-the-sequence-annotations"><span class="header-section-number">3.3.3</span> Accessing to the sequence annotations</h3>
+<section id="accessing-to-the-sequence-annotations" class="level3" data-number="4.3.3">
+<h3 data-number="4.3.3" class="anchored" data-anchor-id="accessing-to-the-sequence-annotations"><span class="header-section-number">4.3.3</span> Accessing to the sequence annotations</h3>
 </section>
 </section>
-<section id="metabarcode-design-and-quality-assessment" class="level2" data-number="3.4">
-<h2 data-number="3.4" class="anchored" data-anchor-id="metabarcode-design-and-quality-assessment"><span class="header-section-number">3.4</span> Metabarcode design and quality assessment</h2>
-<section id="obipcr" class="level4" data-number="3.4.0.1">
-<h4 data-number="3.4.0.1" class="anchored" data-anchor-id="obipcr"><span class="header-section-number">3.4.0.1</span> <code>obipcr</code></h4>
+<section id="metabarcode-design-and-quality-assessment" class="level2" data-number="4.4">
+<h2 data-number="4.4" class="anchored" data-anchor-id="metabarcode-design-and-quality-assessment"><span class="header-section-number">4.4</span> Metabarcode design and quality assessment</h2>
+<section id="obipcr" class="level4" data-number="4.4.0.1">
+<h4 data-number="4.4.0.1" class="anchored" data-anchor-id="obipcr"><span class="header-section-number">4.4.0.1</span> <code>obipcr</code></h4>
 <blockquote class="blockquote">
 <p>Replace the <code>ecoPCR</code> original <em>OBITools</em></p>
 </blockquote>
 </section>
 </section>
-<section id="file-format-conversions" class="level2" data-number="3.5">
-<h2 data-number="3.5" class="anchored" data-anchor-id="file-format-conversions"><span class="header-section-number">3.5</span> File format conversions</h2>
-<section id="obiconvert" class="level4" data-number="3.5.0.1">
-<h4 data-number="3.5.0.1" class="anchored" data-anchor-id="obiconvert"><span class="header-section-number">3.5.0.1</span> <code>obiconvert</code></h4>
+<section id="file-format-conversions" class="level2" data-number="4.5">
+<h2 data-number="4.5" class="anchored" data-anchor-id="file-format-conversions"><span class="header-section-number">4.5</span> File format conversions</h2>
+<section id="obiconvert" class="level4" data-number="4.5.0.1">
+<h4 data-number="4.5.0.1" class="anchored" data-anchor-id="obiconvert"><span class="header-section-number">4.5.0.1</span> <code>obiconvert</code></h4>
 </section>
 </section>
-<section id="sequence-annotations" class="level2" data-number="3.6">
-<h2 data-number="3.6" class="anchored" data-anchor-id="sequence-annotations"><span class="header-section-number">3.6</span> Sequence annotations</h2>
-<section id="obitag" class="level4" data-number="3.6.0.1">
-<h4 data-number="3.6.0.1" class="anchored" data-anchor-id="obitag"><span class="header-section-number">3.6.0.1</span> <code>obitag</code></h4>
+<section id="sequence-annotations" class="level2" data-number="4.6">
+<h2 data-number="4.6" class="anchored" data-anchor-id="sequence-annotations"><span class="header-section-number">4.6</span> Sequence annotations</h2>
+<section id="obitag" class="level4" data-number="4.6.0.1">
+<h4 data-number="4.6.0.1" class="anchored" data-anchor-id="obitag"><span class="header-section-number">4.6.0.1</span> <code>obitag</code></h4>
 </section>
 </section>
-<section id="computations-on-sequences" class="level2" data-number="3.7">
-<h2 data-number="3.7" class="anchored" data-anchor-id="computations-on-sequences"><span class="header-section-number">3.7</span> Computations on sequences</h2>
-<section id="obipairing" class="level3" data-number="3.7.1">
-<h3 data-number="3.7.1" class="anchored" data-anchor-id="obipairing"><span class="header-section-number">3.7.1</span> <code>obipairing</code></h3>
+<section id="computations-on-sequences" class="level2" data-number="4.7">
+<h2 data-number="4.7" class="anchored" data-anchor-id="computations-on-sequences"><span class="header-section-number">4.7</span> Computations on sequences</h2>
+<section id="obipairing" class="level3" data-number="4.7.1">
+<h3 data-number="4.7.1" class="anchored" data-anchor-id="obipairing"><span class="header-section-number">4.7.1</span> <code>obipairing</code></h3>
 <blockquote class="blockquote">
 <p>Replace the <code>illuminapairedends</code> original <em>OBITools</em></p>
 </blockquote>
-<section id="alignment-procedure" class="level4" data-number="3.7.1.1">
-<h4 data-number="3.7.1.1" class="anchored" data-anchor-id="alignment-procedure"><span class="header-section-number">3.7.1.1</span> Alignment procedure</h4>
+<section id="alignment-procedure" class="level4" data-number="4.7.1.1">
+<h4 data-number="4.7.1.1" class="anchored" data-anchor-id="alignment-procedure"><span class="header-section-number">4.7.1.1</span> Alignment procedure</h4>
 <p><code>obipairing</code> is introducing a new alignment algorithm compared to the <code>illuminapairedend</code> command of the <code>OBITools V2</code>. Nethertheless this new algorithm has been design to produce the same results than the previous, except in very few cases.</p>
 <p>The new algorithm is a two-step procedure. First, a FASTN-type algorithm <span class="citation" data-cites="Lipman1985-hw">(<a href="references.html#ref-Lipman1985-hw" role="doc-biblioref">Lipman and Pearson 1985</a>)</span> identifies the best offset between the two matched readings. This identifies the region of overlap.</p>
 <p>In the second step, the matching regions of the two reads are extracted along with a flanking sequence of <span class="math inline">\(\Delta\)</span> base pairs. The two subsequences are then aligned using a “one side free end-gap” dynamic programming algorithm. This latter step is only called if at least one mismatch is detected by the FASTP step.</p>
 <p>Unless the similarity between the two reads at their overlap region is very low, the addition of the flanking regions in the second step of the alignment ensures the same alignment as if the dynamic programming alignment was performed on the full reads.</p>
 </section>
-<section id="the-scoring-system" class="level4" data-number="3.7.1.2">
-<h4 data-number="3.7.1.2" class="anchored" data-anchor-id="the-scoring-system"><span class="header-section-number">3.7.1.2</span> The scoring system</h4>
+<section id="the-scoring-system" class="level4" data-number="4.7.1.2">
+<h4 data-number="4.7.1.2" class="anchored" data-anchor-id="the-scoring-system"><span class="header-section-number">4.7.1.2</span> The scoring system</h4>
 <p>In the dynamic programming step, the match and mismatch scores take into account the quality scores of the two aligned nucleotides. By taking these into account, the probability of a true match can be calculated for each aligned base pair.</p>
 <p>If we consider a nucleotide read with a quality score <span class="math inline">\(Q\)</span>, the probability of misreading this base (<span class="math inline">\(P_E\)</span>) is : <span class="math display">\[
 P_E = 10^{-\frac{Q}{10}}
@@ -394,38 +399,38 @@ P(MATCH | X_1 \neq X_2) =  (1-P_{E1})\frac{P_{E2}}{3} +  (1-P_{E2})\frac{P_{E1}}
 </div>
 </div>
 </section>
-<section id="obimultiplex" class="level4" data-number="3.7.1.3">
-<h4 data-number="3.7.1.3" class="anchored" data-anchor-id="obimultiplex"><span class="header-section-number">3.7.1.3</span> <code>obimultiplex</code></h4>
+<section id="obimultiplex" class="level4" data-number="4.7.1.3">
+<h4 data-number="4.7.1.3" class="anchored" data-anchor-id="obimultiplex"><span class="header-section-number">4.7.1.3</span> <code>obimultiplex</code></h4>
 <blockquote class="blockquote">
 <p>Replace the <code>ngsfilter</code> original <em>OBITools</em></p>
 </blockquote>
 </section>
-<section id="obicomplement" class="level4" data-number="3.7.1.4">
-<h4 data-number="3.7.1.4" class="anchored" data-anchor-id="obicomplement"><span class="header-section-number">3.7.1.4</span> <code>obicomplement</code></h4>
+<section id="obicomplement" class="level4" data-number="4.7.1.4">
+<h4 data-number="4.7.1.4" class="anchored" data-anchor-id="obicomplement"><span class="header-section-number">4.7.1.4</span> <code>obicomplement</code></h4>
 </section>
-<section id="obiclean" class="level4" data-number="3.7.1.5">
-<h4 data-number="3.7.1.5" class="anchored" data-anchor-id="obiclean"><span class="header-section-number">3.7.1.5</span> <code>obiclean</code></h4>
+<section id="obiclean" class="level4" data-number="4.7.1.5">
+<h4 data-number="4.7.1.5" class="anchored" data-anchor-id="obiclean"><span class="header-section-number">4.7.1.5</span> <code>obiclean</code></h4>
 </section>
-<section id="obiuniq" class="level4" data-number="3.7.1.6">
-<h4 data-number="3.7.1.6" class="anchored" data-anchor-id="obiuniq"><span class="header-section-number">3.7.1.6</span> <code>obiuniq</code></h4>
+<section id="obiuniq" class="level4" data-number="4.7.1.6">
+<h4 data-number="4.7.1.6" class="anchored" data-anchor-id="obiuniq"><span class="header-section-number">4.7.1.6</span> <code>obiuniq</code></h4>
 </section>
 </section>
 </section>
-<section id="sequence-sampling-and-filtering" class="level2" data-number="3.8">
-<h2 data-number="3.8" class="anchored" data-anchor-id="sequence-sampling-and-filtering"><span class="header-section-number">3.8</span> Sequence sampling and filtering</h2>
-<section id="obigrep" class="level4" data-number="3.8.0.1">
-<h4 data-number="3.8.0.1" class="anchored" data-anchor-id="obigrep"><span class="header-section-number">3.8.0.1</span> <code>obigrep</code></h4>
+<section id="sequence-sampling-and-filtering" class="level2" data-number="4.8">
+<h2 data-number="4.8" class="anchored" data-anchor-id="sequence-sampling-and-filtering"><span class="header-section-number">4.8</span> Sequence sampling and filtering</h2>
+<section id="obigrep" class="level4" data-number="4.8.0.1">
+<h4 data-number="4.8.0.1" class="anchored" data-anchor-id="obigrep"><span class="header-section-number">4.8.0.1</span> <code>obigrep</code></h4>
 </section>
-<section id="utilities" class="level3" data-number="3.8.1">
-<h3 data-number="3.8.1" class="anchored" data-anchor-id="utilities"><span class="header-section-number">3.8.1</span> Utilities</h3>
-<section id="obicount" class="level4" data-number="3.8.1.1">
-<h4 data-number="3.8.1.1" class="anchored" data-anchor-id="obicount"><span class="header-section-number">3.8.1.1</span> <code>obicount</code></h4>
+<section id="utilities" class="level3" data-number="4.8.1">
+<h3 data-number="4.8.1" class="anchored" data-anchor-id="utilities"><span class="header-section-number">4.8.1</span> Utilities</h3>
+<section id="obicount" class="level4" data-number="4.8.1.1">
+<h4 data-number="4.8.1.1" class="anchored" data-anchor-id="obicount"><span class="header-section-number">4.8.1.1</span> <code>obicount</code></h4>
 </section>
-<section id="obidistribute" class="level4" data-number="3.8.1.2">
-<h4 data-number="3.8.1.2" class="anchored" data-anchor-id="obidistribute"><span class="header-section-number">3.8.1.2</span> <code>obidistribute</code></h4>
+<section id="obidistribute" class="level4" data-number="4.8.1.2">
+<h4 data-number="4.8.1.2" class="anchored" data-anchor-id="obidistribute"><span class="header-section-number">4.8.1.2</span> <code>obidistribute</code></h4>
 </section>
-<section id="obifind" class="level4" data-number="3.8.1.3">
-<h4 data-number="3.8.1.3" class="anchored" data-anchor-id="obifind"><span class="header-section-number">3.8.1.3</span> <code>obifind</code></h4>
+<section id="obifind" class="level4" data-number="4.8.1.3">
+<h4 data-number="4.8.1.3" class="anchored" data-anchor-id="obifind"><span class="header-section-number">4.8.1.3</span> <code>obifind</code></h4>
 <blockquote class="blockquote">
 <p>Replace the <code>ecofind</code> original <em>OBITools.</em></p>
 </blockquote>
@@ -577,12 +582,12 @@ window.document.addEventListener("DOMContentLoaded", function (event) {
 <nav class="page-navigation">
   <div class="nav-page nav-page-previous">
       <a href="./tutorial.html" class="pagination-link">
-        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></span>
+        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></span>
       </a>          
   </div>
   <div class="nav-page nav-page-next">
       <a href="./library.html" class="pagination-link">
-        <span class="nav-page-text"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></span> <i class="bi bi-arrow-right-short"></i>
+        <span class="nav-page-text"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></span> <i class="bi bi-arrow-right-short"></i>
       </a>
   </div>
 </nav>
diff --git a/doc/_book/formats.html b/doc/_book/formats.html
new file mode 100644
index 0000000..24ac11e
--- /dev/null
+++ b/doc/_book/formats.html
@@ -0,0 +1,533 @@
+<!DOCTYPE html>
+<html xmlns="http://www.w3.org/1999/xhtml" lang="en" xml:lang="en"><head>
+
+<meta charset="utf-8">
+<meta name="generator" content="quarto-1.2.256">
+
+<meta name="viewport" content="width=device-width, initial-scale=1.0, user-scalable=yes">
+
+
+<title>OBITools V4 - 2&nbsp; File formats usable with OBITools</title>
+<style>
+code{white-space: pre-wrap;}
+span.smallcaps{font-variant: small-caps;}
+div.columns{display: flex; gap: min(4vw, 1.5em);}
+div.column{flex: auto; overflow-x: auto;}
+div.hanging-indent{margin-left: 1.5em; text-indent: -1.5em;}
+ul.task-list{list-style: none;}
+ul.task-list li input[type="checkbox"] {
+  width: 0.8em;
+  margin: 0 0.8em 0.2em -1.6em;
+  vertical-align: middle;
+}
+div.csl-bib-body { }
+div.csl-entry {
+  clear: both;
+}
+.hanging div.csl-entry {
+  margin-left:2em;
+  text-indent:-2em;
+}
+div.csl-left-margin {
+  min-width:2em;
+  float:left;
+}
+div.csl-right-inline {
+  margin-left:2em;
+  padding-left:1em;
+}
+div.csl-indent {
+  margin-left: 2em;
+}
+</style>
+
+
+<script src="site_libs/quarto-nav/quarto-nav.js"></script>
+<script src="site_libs/quarto-nav/headroom.min.js"></script>
+<script src="site_libs/clipboard/clipboard.min.js"></script>
+<script src="site_libs/quarto-search/autocomplete.umd.js"></script>
+<script src="site_libs/quarto-search/fuse.min.js"></script>
+<script src="site_libs/quarto-search/quarto-search.js"></script>
+<meta name="quarto:offset" content="./">
+<link href="./tutorial.html" rel="next">
+<link href="./intro.html" rel="prev">
+<script src="site_libs/quarto-html/quarto.js"></script>
+<script src="site_libs/quarto-html/popper.min.js"></script>
+<script src="site_libs/quarto-html/tippy.umd.min.js"></script>
+<script src="site_libs/quarto-html/anchor.min.js"></script>
+<link href="site_libs/quarto-html/tippy.css" rel="stylesheet">
+<link href="site_libs/quarto-html/quarto-syntax-highlighting.css" rel="stylesheet" id="quarto-text-highlighting-styles">
+<script src="site_libs/bootstrap/bootstrap.min.js"></script>
+<link href="site_libs/bootstrap/bootstrap-icons.css" rel="stylesheet">
+<link href="site_libs/bootstrap/bootstrap.min.css" rel="stylesheet" id="quarto-bootstrap" data-mode="light">
+<script id="quarto-search-options" type="application/json">{
+  "location": "sidebar",
+  "copy-button": false,
+  "collapse-after": 3,
+  "panel-placement": "start",
+  "type": "textbox",
+  "limit": 20,
+  "language": {
+    "search-no-results-text": "No results",
+    "search-matching-documents-text": "matching documents",
+    "search-copy-link-title": "Copy link to search",
+    "search-hide-matches-text": "Hide additional matches",
+    "search-more-match-text": "more match in this document",
+    "search-more-matches-text": "more matches in this document",
+    "search-clear-button-title": "Clear",
+    "search-detached-cancel-button-title": "Cancel",
+    "search-submit-button-title": "Submit"
+  }
+}</script>
+
+  <script src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-chtml-full.js" type="text/javascript"></script>
+
+</head>
+
+<body class="nav-sidebar floating">
+
+<div id="quarto-search-results"></div>
+  <header id="quarto-header" class="headroom fixed-top">
+  <nav class="quarto-secondary-nav" data-bs-toggle="collapse" data-bs-target="#quarto-sidebar" aria-controls="quarto-sidebar" aria-expanded="false" aria-label="Toggle sidebar navigation" onclick="if (window.quartoToggleHeadroom) { window.quartoToggleHeadroom(); }">
+    <div class="container-fluid d-flex justify-content-between">
+      <h1 class="quarto-secondary-nav-title"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></h1>
+      <button type="button" class="quarto-btn-toggle btn" aria-label="Show secondary navigation">
+        <i class="bi bi-chevron-right"></i>
+      </button>
+    </div>
+  </nav>
+</header>
+<!-- content -->
+<div id="quarto-content" class="quarto-container page-columns page-rows-contents page-layout-article">
+<!-- sidebar -->
+  <nav id="quarto-sidebar" class="sidebar collapse sidebar-navigation floating overflow-auto">
+    <div class="pt-lg-2 mt-2 text-left sidebar-header">
+    <div class="sidebar-title mb-0 py-0">
+      <a href="./">OBITools V4</a> 
+    </div>
+      </div>
+      <div class="mt-2 flex-shrink-0 align-items-center">
+        <div class="sidebar-search">
+        <div id="quarto-search" class="" title="Search"></div>
+        </div>
+      </div>
+    <div class="sidebar-menu-container"> 
+    <ul class="list-unstyled mt-1">
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./index.html" class="sidebar-item-text sidebar-link">Preface</a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./intro.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">1</span>&nbsp; <span class="chapter-title">The OBITools</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./formats.html" class="sidebar-item-text sidebar-link active"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./references.html" class="sidebar-item-text sidebar-link">References</a>
+  </div>
+</li>
+    </ul>
+    </div>
+</nav>
+<!-- margin-sidebar -->
+    <div id="quarto-margin-sidebar" class="sidebar margin-sidebar">
+        <nav id="TOC" role="doc-toc" class="toc-active">
+    <h2 id="toc-title">Table of contents</h2>
+   
+  <ul>
+  <li><a href="#the-dna-sequence-data" id="toc-the-dna-sequence-data" class="nav-link active" data-scroll-target="#the-dna-sequence-data"><span class="toc-section-number">2.1</span>  The DNA sequence data</a>
+  <ul class="collapse">
+  <li><a href="#the-iupac-code" id="toc-the-iupac-code" class="nav-link" data-scroll-target="#the-iupac-code"><span class="toc-section-number">2.1.1</span>  The IUPAC Code</a></li>
+  <li><a href="#sec-fasta" id="toc-sec-fasta" class="nav-link" data-scroll-target="#sec-fasta"><span class="toc-section-number">2.1.2</span>  The <em>fasta</em> sequence format</a></li>
+  <li><a href="#sec-fastq" id="toc-sec-fastq" class="nav-link" data-scroll-target="#sec-fastq"><span class="toc-section-number">2.1.3</span>  The <em>fastq</em> sequence format</a></li>
+  <li><a href="#file-extension" id="toc-file-extension" class="nav-link" data-scroll-target="#file-extension"><span class="toc-section-number">2.1.4</span>  File extension</a></li>
+  </ul></li>
+  </ul>
+</nav>
+    </div>
+<!-- main -->
+<main class="content" id="quarto-document-content">
+
+<header id="title-block-header" class="quarto-title-block default">
+<div class="quarto-title">
+<h1 class="title d-none d-lg-block"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></h1>
+</div>
+
+
+
+<div class="quarto-title-meta">
+
+    
+  
+    
+  </div>
+  
+
+</header>
+
+<p>OBITools manipulate have to manipulate DNA sequence data and taxonomical data. They can use some supplentary metadata describing the experiment and produce some stats about the processed DNA data. All the manipulated data are stored in text files, following standard data format.</p>
+<section id="the-dna-sequence-data" class="level2" data-number="2.1">
+<h2 data-number="2.1" class="anchored" data-anchor-id="the-dna-sequence-data"><span class="header-section-number">2.1</span> The DNA sequence data</h2>
+<p>Sequences can be stored following various format. OBITools knows some of them. The central formats for sequence files manipulated by OBITools scripts are the <a href="#the-fasta-sequence-format"><code>fasta</code></a> and <a href="#the-fastq-sequence-format"><code>fastq</code></a> format. OBITools extends the both these formats by specifying a syntax to include in the definition line data qualifying the sequence. All file formats use the <code>IUPAC</code> code for encoding nucleotides.</p>
+<p>Moreover these two formats that can be used as input and output formats, <strong>OBITools4</strong> can read the following format :</p>
+<ul>
+<li><a href="https://ena-docs.readthedocs.io/en/latest/submit/fileprep/flat-file-example.html">EBML flat file</a> format (use by ENA)</li>
+<li><a href="https://www.ncbi.nlm.nih.gov/Sitemap/samplerecord.html">Genbank flat file format</a></li>
+<li><a href="https://pythonhosted.org/OBITools/scripts/ecoPCR.html">ecoPCR output files</a></li>
+</ul>
+<section id="the-iupac-code" class="level3" data-number="2.1.1">
+<h3 data-number="2.1.1" class="anchored" data-anchor-id="the-iupac-code"><span class="header-section-number">2.1.1</span> The IUPAC Code</h3>
+<p>The International Union of Pure and Applied Chemistry (IUPAC_) defined the standard code for representing protein or DNA sequences.</p>
+<table class="table">
+<thead>
+<tr class="header">
+<th><strong>Code</strong></th>
+<th><strong>Nucleotide</strong></th>
+</tr>
+</thead>
+<tbody>
+<tr class="odd">
+<td>A</td>
+<td>Adenine</td>
+</tr>
+<tr class="even">
+<td>C</td>
+<td>Cytosine</td>
+</tr>
+<tr class="odd">
+<td>G</td>
+<td>Guanine</td>
+</tr>
+<tr class="even">
+<td>T</td>
+<td>Thymine</td>
+</tr>
+<tr class="odd">
+<td>U</td>
+<td>Uracil</td>
+</tr>
+<tr class="even">
+<td>R</td>
+<td>Purine (A or G)</td>
+</tr>
+<tr class="odd">
+<td>Y</td>
+<td>Pyrimidine (C, T, or U)</td>
+</tr>
+<tr class="even">
+<td>M</td>
+<td>C or A</td>
+</tr>
+<tr class="odd">
+<td>K</td>
+<td>T, U, or G</td>
+</tr>
+<tr class="even">
+<td>W</td>
+<td>T, U, or A</td>
+</tr>
+<tr class="odd">
+<td>S</td>
+<td>C or G</td>
+</tr>
+<tr class="even">
+<td>B</td>
+<td>C, T, U, or G (not A)</td>
+</tr>
+<tr class="odd">
+<td>D</td>
+<td>A, T, U, or G (not C)</td>
+</tr>
+<tr class="even">
+<td>H</td>
+<td>A, T, U, or C (not G)</td>
+</tr>
+<tr class="odd">
+<td>V</td>
+<td>A, C, or G (not T, not U)</td>
+</tr>
+<tr class="even">
+<td>N</td>
+<td>Any base (A, C, G, T, or U)</td>
+</tr>
+</tbody>
+</table>
+</section>
+<section id="sec-fasta" class="level3" data-number="2.1.2">
+<h3 data-number="2.1.2" class="anchored" data-anchor-id="sec-fasta"><span class="header-section-number">2.1.2</span> The <em>fasta</em> sequence format</h3>
+<p>The <strong>fasta format</strong> is certainly the most widely used sequence file format. This is certainly due to its great simplicity. It was originally created for the Lipman and Pearson <a href="http://www.ncbi.nlm.nih.gov/pubmed/3162770?dopt=Citation">FASTA program</a>. OBITools use in more of the classical <code>fasta</code> format an <code>extended version</code> of this format where structured data are included in the title line.</p>
+<p>In <em>fasta</em> format a sequence is represented by a title line beginning with a <strong><code>&gt;</code></strong> character and the sequences by itself following the :doc:<code>iupac</code> code. The sequence is usually split other severals lines of the same length (expect for the last one)</p>
+<pre><code>&gt;my_sequence this is my pretty sequence
+ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
+GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
+AACGACGTTGCAGTACGTTGCAGT</code></pre>
+<p>This is no special format for the title line excepting that this line should be unique. Usually the first word following the <strong>&gt;</strong> character is considered as the sequence identifier. The end of the title line corresponding to a description of the sequence. Several sequences can be concatenated in a same file. The description of the next sequence is just pasted at the end of the record of the previous one</p>
+<pre><code>&gt;sequence_A this is my first pretty sequence
+ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
+GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
+AACGACGTTGCAGTACGTTGCAGT
+&gt;sequence_B this is my second pretty sequence
+ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
+GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
+AACGACGTTGCAGTACGTTGCAGT
+&gt;sequence_C this is my third pretty sequence
+ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
+GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
+AACGACGTTGCAGTACGTTGCAGT</code></pre>
+</section>
+<section id="sec-fastq" class="level3" data-number="2.1.3">
+<h3 data-number="2.1.3" class="anchored" data-anchor-id="sec-fastq"><span class="header-section-number">2.1.3</span> The <em>fastq</em> sequence format<a href="#fn1" class="footnote-ref" id="fnref1" role="doc-noteref"><sup>1</sup></a></h3>
+<p>The <strong>FASTQ</strong> format is a text file format for storing both biological sequences (only nucleic acid sequences) and the associated quality scores. The sequence and score are each encoded by a single ASCII character. This format was originally developed by the Wellcome Trust Sanger Institute to link a <a href="#the-fasta-sequence-format">FASTA</a> sequence file to the corresponding quality data, but has recently become the de facto standard for storing results from high-throughput sequencers <span class="citation" data-cites="cock2010sanger">(<a href="references.html#ref-cock2010sanger" role="doc-biblioref">Cock et al. 2010</a>)</span>.</p>
+<p>A fastq file normally uses four lines per sequence.</p>
+<ul>
+<li>Line 1 begins with a ‘@’ character and is followed by a sequence identifier and an <em>optional</em> description (like a :ref:<code>fasta</code> title line).</li>
+<li>Line 2 is the raw sequence letters.</li>
+<li>Line 3 begins with a ‘+’ character and is <em>optionally</em> followed by the same sequence identifier (and any description) again.</li>
+<li>Line 4 encodes the quality values for the sequence in Line 2, and must contain the same number of symbols as letters in the sequence.</li>
+</ul>
+<p>A fastq file containing a single sequence might look like this:</p>
+<pre><code>@SEQ_ID
+GATTTGGGGTTCAAAGCAGTATCGATCAAATAGTAAATCCATTTGTTCAACTCACAGTTT
++
+!''*((((***+))%%%++)(%%%%).1***-+*''))**55CCF&gt;&gt;&gt;&gt;&gt;&gt;CCCCCCC65</code></pre>
+<p>The character ‘!’ represents the lowest quality while ‘~’ is the highest. Here are the quality value characters in left-to-right increasing order of quality (<code>ASCII</code>):</p>
+<pre><code>!"#$%&amp;'()*+,-./0123456789:;&lt;=&gt;?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\]^_`abcdefghijklmnopqrstuvwxyz{|}~</code></pre>
+<p>The original Sanger FASTQ files also allowed the sequence and quality strings to be wrapped (split over multiple lines), but this is generally discouraged as it can make parsing complicated due to the unfortunate choice of “@” and “+” as markers (these characters can also occur in the quality string).</p>
+<section id="sequence-quality-scores" class="level4 unnumbered">
+<h4 class="unnumbered anchored" data-anchor-id="sequence-quality-scores">Sequence quality scores</h4>
+<p>The Phred quality value <em>Q</em> is an integer mapping of <em>p</em> (i.e., the probability that the corresponding base call is incorrect). Two different equations have been in use. The first is the standard Sanger variant to assess reliability of a base call, otherwise known as Phred quality score:</p>
+<p><span class="math display">\[
+Q_\text{sanger} = -10 \, \log_{10} p
+\]</span></p>
+<p>The Solexa pipeline (i.e., the software delivered with the Illumina Genome Analyzer) earlier used a different mapping, encoding the odds <span class="math inline">\(\mathbf{p}/(1-\mathbf{p})\)</span> instead of the probability <span class="math inline">\(\mathbf{p}\)</span>:</p>
+<p><span class="math display">\[
+Q_\text{solexa-prior to v.1.3} = -10 \; \log_{10} \frac{p}{1-p}
+\]</span></p>
+<p>Although both mappings are asymptotically identical at higher quality values, they differ at lower quality levels (i.e., approximately <span class="math inline">\(\mathbf{p} &gt; 0.05\)</span>, or equivalently, <span class="math inline">\(\mathbf{Q} &lt; 13\)</span>).</p>
+<div id="fig-Probabilitymetrics" class="quarto-figure quarto-figure-center anchored">
+<figure class="figure">
+<p><img src="Probabilitymetrics.png" class="img-fluid figure-img"></p>
+<p></p><figcaption class="figure-caption">Figure&nbsp;2.1: Relationship between <em>Q</em> and <em>p</em> using the Sanger (red) and Solexa (black) equations (described above). The vertical dotted line indicates <span class="math inline">\(\mathbf{p}= 0.05\)</span>, or equivalently, <span class="math inline">\(Q = 13\)</span>.</figcaption><p></p>
+</figure>
+</div>
+<section id="encoding" class="level5 unnumbered">
+<h5 class="unnumbered anchored" data-anchor-id="encoding">Encoding</h5>
+<p>The <em>fastq</em> format had differente way of encoding the Phred quality score along the time. Here a breif history of these changes is presented.</p>
+<ul>
+<li>Sanger format can encode a Phred quality score from 0 to 93 using ASCII 33 to 126 (although in raw read data the Phred quality score rarely exceeds 60, higher scores are possible in assemblies or read maps).</li>
+<li>Solexa/Illumina 1.0 format can encode a Solexa/Illumina quality score from -5 to 62 using ASCII 59 to 126 (although in raw read data Solexa scores from -5 to 40 only are expected)</li>
+<li>Starting with Illumina 1.3 and before Illumina 1.8, the format encoded a Phred quality score from 0 to 62 using ASCII 64 to 126 (although in raw read data Phred scores from 0 to 40 only are expected).</li>
+<li>Starting in Illumina 1.5 and before Illumina 1.8, the Phred scores 0 to 2 have a slightly different meaning. The values 0 and 1 are no longer used and the value 2, encoded by ASCII 66 “B”.</li>
+</ul>
+<blockquote class="blockquote">
+<p>Sequencing Control Software, Version 2.6, (Catalog # SY-960-2601, Part # 15009921 Rev.&nbsp;A, November 2009, page 30) states the following: <em>If a read ends with a segment of mostly low quality (Q15 or below), then all of the quality values in the segment are replaced with a value of 2 (encoded as the letter B in Illumina’s text-based encoding of quality scores)… This Q2 indicator does not predict a specific error rate, but rather indicates that a specific final portion of the read should not be used in further analyses.</em> Also, the quality score encoded as “B” letter may occur internally within reads at least as late as pipeline version 1.6, as shown in the following example:</p>
+</blockquote>
+<pre><code>@HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1
+TTAATTGGTAAATAAATCTCCTAATAGCTTAGATNTTACCTTNNNNNNNNNNTAGTTTCTTGAGATTTGTTGGGGGAGACATTTTTGTGATTGCCTTGAT
++HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1
+efcfffffcfeefffcffffffddf`feed]`]_Ba_^__[YBBBBBBBBBBRTT\]][]dddd`ddd^dddadd^BBBBBBBBBBBBBBBBBBBBBBBB</code></pre>
+<p>An alternative interpretation of this ASCII encoding has been proposed. Also, in Illumina runs using PhiX controls, the character ‘B’ was observed to represent an “unknown quality score”. The error rate of ‘B’ reads was roughly 3 phred scores lower the mean observed score of a given run.</p>
+<ul>
+<li>Starting in Illumina 1.8, the quality scores have basically returned to the use of the Sanger format (Phred+33).</li>
+</ul>
+<p>OBItools support the Sanger format. It is nevertheless to read files encoded following the Solexa/Illumina format, that are still possible to find in old files, by applying a shift of 62.</p>
+</section>
+</section>
+</section>
+<section id="file-extension" class="level3" data-number="2.1.4">
+<h3 data-number="2.1.4" class="anchored" data-anchor-id="file-extension"><span class="header-section-number">2.1.4</span> File extension</h3>
+<p>There is no standard file extension for a FASTQ file, but .fq and .fastq, are commonly used.</p>
+
+
+<div id="refs" class="references csl-bib-body hanging-indent" role="doc-bibliography" style="display: none">
+<div id="ref-cock2010sanger" class="csl-entry" role="doc-biblioentry">
+Cock, Peter JA, Christopher J Fields, Naohisa Goto, Michael L Heuer, and Peter M Rice. 2010. <span>“The Sanger FASTQ File Format for Sequences with Quality Scores, and the Solexa/Illumina FASTQ Variants.”</span> <em>Nucleic Acids Research</em> 38 (6): 1767–71.
+</div>
+</div>
+</section>
+</section>
+<section id="footnotes" class="footnotes footnotes-end-of-document" role="doc-endnotes">
+<hr>
+<ol>
+<li id="fn1"><p>This article uses material from the Wikipedia article <a href="http://en.wikipedia.org/wiki/FASTQ_format"><code>FASTQ format</code></a> which is released under the <code>Creative Commons Attribution-Share-Alike License 3.0</code><a href="#fnref1" class="footnote-back" role="doc-backlink">↩︎</a></p></li>
+</ol>
+</section>
+
+</main> <!-- /main -->
+<script id="quarto-html-after-body" type="application/javascript">
+window.document.addEventListener("DOMContentLoaded", function (event) {
+  const toggleBodyColorMode = (bsSheetEl) => {
+    const mode = bsSheetEl.getAttribute("data-mode");
+    const bodyEl = window.document.querySelector("body");
+    if (mode === "dark") {
+      bodyEl.classList.add("quarto-dark");
+      bodyEl.classList.remove("quarto-light");
+    } else {
+      bodyEl.classList.add("quarto-light");
+      bodyEl.classList.remove("quarto-dark");
+    }
+  }
+  const toggleBodyColorPrimary = () => {
+    const bsSheetEl = window.document.querySelector("link#quarto-bootstrap");
+    if (bsSheetEl) {
+      toggleBodyColorMode(bsSheetEl);
+    }
+  }
+  toggleBodyColorPrimary();  
+  const icon = "";
+  const anchorJS = new window.AnchorJS();
+  anchorJS.options = {
+    placement: 'right',
+    icon: icon
+  };
+  anchorJS.add('.anchored');
+  const clipboard = new window.ClipboardJS('.code-copy-button', {
+    target: function(trigger) {
+      return trigger.previousElementSibling;
+    }
+  });
+  clipboard.on('success', function(e) {
+    // button target
+    const button = e.trigger;
+    // don't keep focus
+    button.blur();
+    // flash "checked"
+    button.classList.add('code-copy-button-checked');
+    var currentTitle = button.getAttribute("title");
+    button.setAttribute("title", "Copied!");
+    let tooltip;
+    if (window.bootstrap) {
+      button.setAttribute("data-bs-toggle", "tooltip");
+      button.setAttribute("data-bs-placement", "left");
+      button.setAttribute("data-bs-title", "Copied!");
+      tooltip = new bootstrap.Tooltip(button, 
+        { trigger: "manual", 
+          customClass: "code-copy-button-tooltip",
+          offset: [0, -8]});
+      tooltip.show();    
+    }
+    setTimeout(function() {
+      if (tooltip) {
+        tooltip.hide();
+        button.removeAttribute("data-bs-title");
+        button.removeAttribute("data-bs-toggle");
+        button.removeAttribute("data-bs-placement");
+      }
+      button.setAttribute("title", currentTitle);
+      button.classList.remove('code-copy-button-checked');
+    }, 1000);
+    // clear code selection
+    e.clearSelection();
+  });
+  function tippyHover(el, contentFn) {
+    const config = {
+      allowHTML: true,
+      content: contentFn,
+      maxWidth: 500,
+      delay: 100,
+      arrow: false,
+      appendTo: function(el) {
+          return el.parentElement;
+      },
+      interactive: true,
+      interactiveBorder: 10,
+      theme: 'quarto',
+      placement: 'bottom-start'
+    };
+    window.tippy(el, config); 
+  }
+  const noterefs = window.document.querySelectorAll('a[role="doc-noteref"]');
+  for (var i=0; i<noterefs.length; i++) {
+    const ref = noterefs[i];
+    tippyHover(ref, function() {
+      // use id or data attribute instead here
+      let href = ref.getAttribute('data-footnote-href') || ref.getAttribute('href');
+      try { href = new URL(href).hash; } catch {}
+      const id = href.replace(/^#\/?/, "");
+      const note = window.document.getElementById(id);
+      return note.innerHTML;
+    });
+  }
+  const findCites = (el) => {
+    const parentEl = el.parentElement;
+    if (parentEl) {
+      const cites = parentEl.dataset.cites;
+      if (cites) {
+        return {
+          el,
+          cites: cites.split(' ')
+        };
+      } else {
+        return findCites(el.parentElement)
+      }
+    } else {
+      return undefined;
+    }
+  };
+  var bibliorefs = window.document.querySelectorAll('a[role="doc-biblioref"]');
+  for (var i=0; i<bibliorefs.length; i++) {
+    const ref = bibliorefs[i];
+    const citeInfo = findCites(ref);
+    if (citeInfo) {
+      tippyHover(citeInfo.el, function() {
+        var popup = window.document.createElement('div');
+        citeInfo.cites.forEach(function(cite) {
+          var citeDiv = window.document.createElement('div');
+          citeDiv.classList.add('hanging-indent');
+          citeDiv.classList.add('csl-entry');
+          var biblioDiv = window.document.getElementById('ref-' + cite);
+          if (biblioDiv) {
+            citeDiv.innerHTML = biblioDiv.innerHTML;
+          }
+          popup.appendChild(citeDiv);
+        });
+        return popup.innerHTML;
+      });
+    }
+  }
+});
+</script>
+<nav class="page-navigation">
+  <div class="nav-page nav-page-previous">
+      <a href="./intro.html" class="pagination-link">
+        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">1</span>&nbsp; <span class="chapter-title">The OBITools</span></span>
+      </a>          
+  </div>
+  <div class="nav-page nav-page-next">
+      <a href="./tutorial.html" class="pagination-link">
+        <span class="nav-page-text"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></span> <i class="bi bi-arrow-right-short"></i>
+      </a>
+  </div>
+</nav>
+</div> <!-- /content -->
+
+
+
+</body></html>
\ No newline at end of file
diff --git a/doc/_book/index.html b/doc/_book/index.html
index 665ada0..9b9b303 100644
--- a/doc/_book/index.html
+++ b/doc/_book/index.html
@@ -125,22 +125,27 @@ div.csl-indent {
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
+  <a href="./formats.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
+  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></a>
+  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></a>
   </div>
 </li>
         <li class="sidebar-item">
diff --git a/doc/_book/intro.html b/doc/_book/intro.html
index cd2e22f..bf3b4d4 100644
--- a/doc/_book/intro.html
+++ b/doc/_book/intro.html
@@ -49,7 +49,7 @@ div.csl-indent {
 <script src="site_libs/quarto-search/fuse.min.js"></script>
 <script src="site_libs/quarto-search/quarto-search.js"></script>
 <meta name="quarto:offset" content="./">
-<link href="./tutorial.html" rel="next">
+<link href="./formats.html" rel="next">
 <link href="./index.html" rel="prev">
 <script src="site_libs/quarto-html/quarto.js"></script>
 <script src="site_libs/quarto-html/popper.min.js"></script>
@@ -125,22 +125,27 @@ div.csl-indent {
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
+  <a href="./formats.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
+  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></a>
+  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></a>
   </div>
 </li>
         <li class="sidebar-item">
@@ -158,16 +163,11 @@ div.csl-indent {
    
   <ul>
   <li><a href="#aims-of-obitools" id="toc-aims-of-obitools" class="nav-link active" data-scroll-target="#aims-of-obitools"><span class="toc-section-number">1.1</span>  Aims of <em>OBITools</em></a></li>
-  <li><a href="#file-formats-usable-with-obitools" id="toc-file-formats-usable-with-obitools" class="nav-link" data-scroll-target="#file-formats-usable-with-obitools"><span class="toc-section-number">1.2</span>  File formats usable with <em>OBITools</em></a>
+  <li><a href="#installation-of-the-obitools" id="toc-installation-of-the-obitools" class="nav-link" data-scroll-target="#installation-of-the-obitools"><span class="toc-section-number">1.2</span>  Installation of the obitools</a>
   <ul class="collapse">
-  <li><a href="#the-sequence-files" id="toc-the-sequence-files" class="nav-link" data-scroll-target="#the-sequence-files"><span class="toc-section-number">1.2.1</span>  The sequence files</a></li>
-  <li><a href="#the-iupac-code" id="toc-the-iupac-code" class="nav-link" data-scroll-target="#the-iupac-code"><span class="toc-section-number">1.2.2</span>  The IUPAC Code</a></li>
-  <li><a href="#classical-fasta" id="toc-classical-fasta" class="nav-link" data-scroll-target="#classical-fasta"><span class="toc-section-number">1.2.3</span>  The <em>fasta</em> format</a></li>
-  <li><a href="#classical-fastq" id="toc-classical-fastq" class="nav-link" data-scroll-target="#classical-fastq"><span class="toc-section-number">1.2.4</span>  The <em>fastq</em> sequence format</a></li>
+  <li><a href="#availability-of-the-obitools" id="toc-availability-of-the-obitools" class="nav-link" data-scroll-target="#availability-of-the-obitools"><span class="toc-section-number">1.2.1</span>  Availability of the OBITools</a></li>
+  <li><a href="#prerequisites" id="toc-prerequisites" class="nav-link" data-scroll-target="#prerequisites"><span class="toc-section-number">1.2.2</span>  Prerequisites</a></li>
   </ul></li>
-  <li><a href="#file-extension" id="toc-file-extension" class="nav-link" data-scroll-target="#file-extension"><span class="toc-section-number">1.3</span>  File extension</a></li>
-  <li><a href="#see-also" id="toc-see-also" class="nav-link" data-scroll-target="#see-also"><span class="toc-section-number">1.4</span>  See also</a></li>
-  <li><a href="#references" id="toc-references" class="nav-link" data-scroll-target="#references"><span class="toc-section-number">1.5</span>  References</a></li>
   </ul>
 </nav>
     </div>
@@ -191,201 +191,73 @@ div.csl-indent {
 
 </header>
 
+<p>The <em>OBITools4</em> are programs specifically designed for analyzing NGS data in a DNA metabarcoding context, taking into account taxonomic information. It is distributed as an open source software available on the following website: http://metabarcoding.org/obitools4.</p>
 <section id="aims-of-obitools" class="level2" data-number="1.1">
 <h2 data-number="1.1" class="anchored" data-anchor-id="aims-of-obitools"><span class="header-section-number">1.1</span> Aims of <em>OBITools</em></h2>
+<p>DNA metabarcoding is an efficient approach for biodiversity studies <span class="citation" data-cites="Taberlet2012-pf">(<a href="references.html#ref-Taberlet2012-pf" role="doc-biblioref">Taberlet et al. 2012</a>)</span>. Originally mainly developed by microbiologists <span class="citation" data-cites="Sogin2006-ab">(<em>e.g.</em> <a href="references.html#ref-Sogin2006-ab" role="doc-biblioref">Sogin et al. 2006</a>)</span>, it is now widely used for plants <span class="citation" data-cites="Sonstebo2010-vv Yoccoz2012-ix Parducci2012-rn">(<em>e.g.</em> <a href="references.html#ref-Sonstebo2010-vv" role="doc-biblioref">Sønstebø et al. 2010</a>; <a href="references.html#ref-Yoccoz2012-ix" role="doc-biblioref">Yoccoz et al. 2012</a>; <a href="references.html#ref-Parducci2012-rn" role="doc-biblioref">Parducci et al. 2012</a>)</span> and animals from meiofauna <span class="citation" data-cites="Chariton2010-cz Baldwin2013-yc">(<em>e.g.</em> <a href="references.html#ref-Chariton2010-cz" role="doc-biblioref">Chariton et al. 2010</a>; <a href="references.html#ref-Baldwin2013-yc" role="doc-biblioref">Baldwin et al. 2013</a>)</span> to larger organisms <span class="citation" data-cites="Andersen2012-gj Thomsen2012-au">(<em>e.g.</em> <a href="references.html#ref-Andersen2012-gj" role="doc-biblioref">Andersen et al. 2012</a>; <a href="references.html#ref-Thomsen2012-au" role="doc-biblioref">Thomsen et al. 2012</a>)</span>. Interestingly, this method is not limited to <em>sensu stricto</em> biodiversity surveys, but it can also be implemented in other ecological contexts such as for herbivore <span class="citation" data-cites="Valentini2009-ay Kowalczyk2011-kg">(e.g. <a href="references.html#ref-Valentini2009-ay" role="doc-biblioref">Valentini et al. 2009</a>; <a href="references.html#ref-Kowalczyk2011-kg" role="doc-biblioref">Kowalczyk et al. 2011</a>)</span> or carnivore <span class="citation" data-cites="Deagle2009-yh Shehzad2012-pn">(e.g. <a href="references.html#ref-Deagle2009-yh" role="doc-biblioref">Deagle, Kirkwood, and Jarman 2009</a>; <a href="references.html#ref-Shehzad2012-pn" role="doc-biblioref">Shehzad et al. 2012</a>)</span> diet analyses.</p>
+<p>Whatever the biological question under consideration, the DNA metabarcoding methodology relies heavily on next-generation sequencing (NGS), and generates considerable numbers of DNA sequence reads (typically million of reads). Manipulation of such large datasets requires dedicated programs usually running on a Unix system. Unix is an operating system, whose first version was created during the sixties. Since its early stages, it is dedicated to scientific computing and includes a large set of simple tools to efficiently process text files. Most of those programs can be viewed as filters extracting information from a text file to create a new text file. These programs process text files as streams, line per line, therefore allowing computation on a huge dataset without requiring a large memory. Unix programs usually print their results to their standard output (<em>stdout</em>), which by default is the terminal, so the results can be examined on screen. The main philosophy of the Unix environment is to allow easy redirection of the <em>stdout</em> either to a file, for saving the results, or to the standard input (<em>stdin</em>) of a second program thus allowing to easily create complex processing from simple base commands. Access to Unix computers is increasingly easier for scientists nowadays. Indeed, the Linux operating system, an open source version of Unix, can be freely installed on every PC machine and the MacOS operating system, running on Apple computers, is also a Unix system. The <em>OBITools</em> programs imitate Unix standard programs because they usually act as filters, reading their data from text files or the <em>stdin</em> and writing their results to the <em>stdout</em>. The main difference with classical Unix programs is that text files are not analyzed line per line but sequence record per sequence record (see below for a detailed description of a sequence record). Compared to packages for similar purposes like mothur <span class="citation" data-cites="Schloss2009-qy">(<a href="references.html#ref-Schloss2009-qy" role="doc-biblioref">Schloss et al. 2009</a>)</span> or QIIME <span class="citation" data-cites="Caporaso2010-ii">(<a href="references.html#ref-Caporaso2010-ii" role="doc-biblioref">Caporaso et al. 2010</a>)</span>, the <em>OBITools</em> mainly rely on filtering and sorting algorithms. This allows users to set up versatile data analysis pipelines (Figure 1), adjustable to the broad range of DNA metabarcoding applications. The innovation of the <em>OBITools</em> is their ability to take into account the taxonomic annotations, ultimately allowing sorting and filtering of sequence records based on the taxonomy.</p>
 </section>
-<section id="file-formats-usable-with-obitools" class="level2" data-number="1.2">
-<h2 data-number="1.2" class="anchored" data-anchor-id="file-formats-usable-with-obitools"><span class="header-section-number">1.2</span> File formats usable with <em>OBITools</em></h2>
-<section id="the-sequence-files" class="level3" data-number="1.2.1">
-<h3 data-number="1.2.1" class="anchored" data-anchor-id="the-sequence-files"><span class="header-section-number">1.2.1</span> The sequence files</h3>
-<p>Sequences can be stored following various format. OBITools knows some of them. The central formats for sequence files manipulated by OBITools scripts are the <code>fasta</code> and fastq format. OBITools extends the both these formats by specifying a syntax to include in the definition line data qualifying the sequence. All file formats use the <code>IUPAC</code> code for encoding nucleotides.</p>
+<section id="installation-of-the-obitools" class="level2" data-number="1.2">
+<h2 data-number="1.2" class="anchored" data-anchor-id="installation-of-the-obitools"><span class="header-section-number">1.2</span> Installation of the obitools</h2>
+<section id="availability-of-the-obitools" class="level3" data-number="1.2.1">
+<h3 data-number="1.2.1" class="anchored" data-anchor-id="availability-of-the-obitools"><span class="header-section-number">1.2.1</span> Availability of the OBITools</h3>
+<p>The <em>OBITools</em> are open source and protected by the <a href="http://www.cecill.info/licences/Licence_CeCILL_V2.1-en.html">CeCILL 2.1 license</a>.</p>
+<p>All the sources of the <a href="http://metabarcoding.org/obitools4"><em>OBITools4</em></a> can be downloaded from the metabarcoding git server (https://git.metabarcoding.org).</p>
 </section>
-<section id="the-iupac-code" class="level3" data-number="1.2.2">
-<h3 data-number="1.2.2" class="anchored" data-anchor-id="the-iupac-code"><span class="header-section-number">1.2.2</span> The IUPAC Code</h3>
-<p>The International Union of Pure and Applied Chemistry (IUPAC_) defined the standard code for representing protein or DNA sequences.</p>
-<section id="DNA-IUPAC" class="level4" data-number="1.2.2.1">
-<h4 data-number="1.2.2.1" class="anchored" data-anchor-id="DNA-IUPAC"><span class="header-section-number">1.2.2.1</span> Nucleic IUPAC Code</h4>
-<table class="table">
-<thead>
-<tr class="header">
-<th><strong>Code</strong></th>
-<th><strong>Nucleotide</strong></th>
-</tr>
-</thead>
-<tbody>
-<tr class="odd">
-<td>A</td>
-<td>Adenine</td>
-</tr>
-<tr class="even">
-<td>C</td>
-<td>Cytosine</td>
-</tr>
-<tr class="odd">
-<td>G</td>
-<td>Guanine</td>
-</tr>
-<tr class="even">
-<td>T</td>
-<td>Thymine</td>
-</tr>
-<tr class="odd">
-<td>U</td>
-<td>Uracil</td>
-</tr>
-<tr class="even">
-<td>R</td>
-<td>Purine (A or G)</td>
-</tr>
-<tr class="odd">
-<td>Y</td>
-<td>Pyrimidine (C, T, or U)</td>
-</tr>
-<tr class="even">
-<td>M</td>
-<td>C or A</td>
-</tr>
-<tr class="odd">
-<td>K</td>
-<td>T, U, or G</td>
-</tr>
-<tr class="even">
-<td>W</td>
-<td>T, U, or A</td>
-</tr>
-<tr class="odd">
-<td>S</td>
-<td>C or G</td>
-</tr>
-<tr class="even">
-<td>B</td>
-<td>C, T, U, or G (not A)</td>
-</tr>
-<tr class="odd">
-<td>D</td>
-<td>A, T, U, or G (not C)</td>
-</tr>
-<tr class="even">
-<td>H</td>
-<td>A, T, U, or C (not G)</td>
-</tr>
-<tr class="odd">
-<td>V</td>
-<td>A, C, or G (not T, not U)</td>
-</tr>
-<tr class="even">
-<td>N</td>
-<td>Any base (A, C, G, T, or U)</td>
-</tr>
-</tbody>
-</table>
-</section>
-</section>
-<section id="classical-fasta" class="level3" data-number="1.2.3">
-<h3 data-number="1.2.3" class="anchored" data-anchor-id="classical-fasta"><span class="header-section-number">1.2.3</span> The <em>fasta</em> format</h3>
-<p>The <strong>fasta format</strong> is certainly the most widely used sequence file format. This is certainly due to its great simplicity. It was originally created for the Lipman and Pearson <a href="http://www.ncbi.nlm.nih.gov/pubmed/3162770?dopt=Citation">FASTA program</a>. OBITools use in more of the classical :ref:<code>fasta</code> format an :ref:<code>extended version</code> of this format where structured data are included in the title line.</p>
-<p>In <em>fasta</em> format a sequence is represented by a title line beginning with a <strong><code>&gt;</code></strong> character and the sequences by itself following the :doc:<code>iupac</code> code. The sequence is usually split other severals lines of the same length (expect for the last one)</p>
-<pre><code>&gt;my_sequence this is my pretty sequence
-ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
-GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
-AACGACGTTGCAGTACGTTGCAGT</code></pre>
-<p>This is no special format for the title line excepting that this line should be unique. Usually the first word following the <strong>&gt;</strong> character is considered as the sequence identifier. The end of the title line corresponding to a description of the sequence. Several sequences can be concatenated in a same file. The description of the next sequence is just pasted at the end of the record of the previous one</p>
-<pre><code>&gt;sequence_A this is my first pretty sequence
-ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
-GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
-AACGACGTTGCAGTACGTTGCAGT
-&gt;sequence_B this is my second pretty sequence
-ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
-GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
-AACGACGTTGCAGTACGTTGCAGT
-&gt;sequence_C this is my third pretty sequence
-ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
-GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
-AACGACGTTGCAGTACGTTGCAGT</code></pre>
-</section>
-<section id="classical-fastq" class="level3" data-number="1.2.4">
-<h3 data-number="1.2.4" class="anchored" data-anchor-id="classical-fastq"><span class="header-section-number">1.2.4</span> The <em>fastq</em> sequence format<a href="#fn1" class="footnote-ref" id="fnref1" role="doc-noteref"><sup>1</sup></a></h3>
-<p><strong>fastq format</strong> is a text-based format for storing both a biological sequence (usually nucleotide sequence) and its corresponding quality scores. Both the sequence letter and quality score are encoded with a single ASCII character for brevity. It was originally developed at the <code>Wellcome Trust Sanger Institute</code> to bundle a <a href="#classical-fasta">fasta</a> sequence and its quality data, but has recently become the <em>de facto</em> standard for storing the output of high throughput sequencing instruments such as the Illumina Genome Analyzer Illumina <span class="citation" data-cites="cock2010sanger">(<a href="references.html#ref-cock2010sanger" role="doc-biblioref">Cock et al. 2010</a>)</span> .</p>
-<p>A fastq file normally uses four lines per sequence.</p>
-<ul>
-<li>Line 1 begins with a ‘@’ character and is followed by a sequence identifier and an <em>optional</em> description (like a :ref:<code>fasta</code> title line).</li>
-<li>Line 2 is the raw sequence letters.</li>
-<li>Line 3 begins with a ‘+’ character and is <em>optionally</em> followed by the same sequence identifier (and any description) again.</li>
-<li>Line 4 encodes the quality values for the sequence in Line 2, and must contain the same number of symbols as letters in the sequence.</li>
-</ul>
-<p>A fastq file containing a single sequence might look like this:</p>
-<pre><code>@SEQ_ID
-GATTTGGGGTTCAAAGCAGTATCGATCAAATAGTAAATCCATTTGTTCAACTCACAGTTT
-+
-!''*((((***+))%%%++)(%%%%).1***-+*''))**55CCF&gt;&gt;&gt;&gt;&gt;&gt;CCCCCCC65</code></pre>
-<p>The character ‘!’ represents the lowest quality while ‘~’ is the highest. Here are the quality value characters in left-to-right increasing order of quality (<code>ASCII</code>):</p>
-<pre><code>!"#$%&amp;'()*+,-./0123456789:;&lt;=&gt;?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\]^_`abcdefghijklmnopqrstuvwxyz{|}~</code></pre>
-<p>The original Sanger FASTQ files also allowed the sequence and quality strings to be wrapped (split over multiple lines), but this is generally discouraged as it can make parsing complicated due to the unfortunate choice of “@” and “+” as markers (these characters can also occur in the quality string).</p>
-<section id="variations" class="level4" data-number="1.2.4.1">
-<h4 data-number="1.2.4.1" class="anchored" data-anchor-id="variations"><span class="header-section-number">1.2.4.1</span> Variations</h4>
-<section id="quality" class="level5" data-number="1.2.4.1.1">
-<h5 data-number="1.2.4.1.1" class="anchored" data-anchor-id="quality"><span class="header-section-number">1.2.4.1.1</span> Quality</h5>
-<p>A quality value <em>Q</em> is an integer mapping of <em>p</em> (i.e., the probability that the corresponding base call is incorrect). Two different equations have been in use. The first is the standard Sanger variant to assess reliability of a base call, otherwise known as Phred quality score:</p>
-<p><span class="math display">\[
-Q_\text{sanger} = -10 \, \log_{10} p
-\]</span></p>
-<p>The Solexa pipeline (i.e., the software delivered with the Illumina Genome Analyzer) earlier used a different mapping, encoding the odds <span class="math inline">\(\mathbf{p}/(1-\mathbf{p})\)</span> instead of the probability <span class="math inline">\(\mathbf{p}\)</span>:</p>
-<p><span class="math display">\[
-Q_\text{solexa-prior to v.1.3} = -10 \, \log_{10} \frac{p}{1-p}
-\]</span></p>
-<p>Although both mappings are asymptotically identical at higher quality values, they differ at lower quality levels (i.e., approximately <span class="math inline">\(\mathbf{p} &gt; 0.05\)</span>, or equivalently, <span class="math inline">\(\mathbf{Q} &lt; 13\)</span>).</p>
-<p>|Relationship between <em>Q</em> and <em>p</em> using the Sanger (red) and Solexa (black) equations (described above). The vertical dotted line indicates <span class="math inline">\(\mathbf{p}= 0.05\)</span>, or equivalently, <span class="math inline">\(Q = 13\)</span>.|</p>
-</section>
-</section>
-<section id="encoding" class="level4" data-number="1.2.4.2">
-<h4 data-number="1.2.4.2" class="anchored" data-anchor-id="encoding"><span class="header-section-number">1.2.4.2</span> Encoding</h4>
-<ul>
-<li>Sanger format can encode a Phred quality score from 0 to 93 using ASCII 33 to 126 (although in raw read data the Phred quality score rarely exceeds 60, higher scores are possible in assemblies or read maps).</li>
-<li>Solexa/Illumina 1.0 format can encode a Solexa/Illumina quality score from -5 to 62 using ASCII 59 to 126 (although in raw read data Solexa scores from -5 to 40 only are expected)</li>
-<li>Starting with Illumina 1.3 and before Illumina 1.8, the format encoded a Phred quality score from 0 to 62 using ASCII 64 to 126 (although in raw read data Phred scores from 0 to 40 only are expected).</li>
-<li>Starting in Illumina 1.5 and before Illumina 1.8, the Phred scores 0 to 2 have a slightly different meaning. The values 0 and 1 are no longer used and the value 2, encoded by ASCII 66 “B”.</li>
-</ul>
-<p>Sequencing Control Software, Version 2.6, Catalog # SY-960-2601, Part # 15009921 Rev.&nbsp;A, November 2009]&nbsp;<a href="[http://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf](http://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf){.uri}" class="uri">[http://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf\\](http://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf){.uri}</a> (page 30) states the following: <em>If a read ends with a segment of mostly low quality (Q15 or below), then all of the quality values in the segment are replaced with a value of 2 (encoded as the letter B in Illumina’s text-based encoding of quality scores)… This Q2 indicator does not predict a specific error rate, but rather indicates that a specific final portion of the read should not be used in further analyses.</em> Also, the quality score encoded as “B” letter may occur internally within reads at least as late as pipeline version 1.6, as shown in the following example:</p>
-<pre><code>@HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1
-TTAATTGGTAAATAAATCTCCTAATAGCTTAGATNTTACCTTNNNNNNNNNNTAGTTTCTTGAGATTTGTTGGGGGAGACATTTTTGTGATTGCCTTGAT
-+HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1
-efcfffffcfeefffcffffffddf`feed]`]_Ba_^__[YBBBBBBBBBBRTT\]][]dddd`ddd^dddadd^BBBBBBBBBBBBBBBBBBBBBBBB</code></pre>
-<p>An alternative interpretation of this ASCII encoding has been proposed. Also, in Illumina runs using PhiX controls, the character ‘B’ was observed to represent an “unknown quality score”. The error rate of ‘B’ reads was roughly 3 phred scores lower the mean observed score of a given run.</p>
-<ul>
-<li>Starting in Illumina 1.8, the quality scores have basically returned to the use of the Sanger format (Phred+33).</li>
-</ul>
-</section>
-</section>
-</section>
-<section id="file-extension" class="level2" data-number="1.3">
-<h2 data-number="1.3" class="anchored" data-anchor-id="file-extension"><span class="header-section-number">1.3</span> File extension</h2>
-<p>There is no standard file extension for a FASTQ file, but .fq and .fastq, are commonly used.</p>
-</section>
-<section id="see-also" class="level2" data-number="1.4">
-<h2 data-number="1.4" class="anchored" data-anchor-id="see-also"><span class="header-section-number">1.4</span> See also</h2>
-<ul>
-<li>:ref:<code>fasta</code></li>
-</ul>
-</section>
-<section id="references" class="level2" data-number="1.5">
-<h2 data-number="1.5" class="anchored" data-anchor-id="references"><span class="header-section-number">1.5</span> References</h2>
-<p>.. [1] Cock et al (2009) The Sanger FASTQ file format for sequences with quality scores, and the Solexa/Illumina FASTQ variants. Nucleic Acids Research,</p>
-<p>.. [2] Illumina Quality Scores, Tobias Mann, Bioinformatics, San Diego, Illumina <code>1</code>__</p>
-<p>.. |Relationship between <em>Q</em> and <em>p</em> using the Sanger (red) and Solexa (black) equations (described above). The vertical dotted line indicates <em>p</em> = 0.05, or equivalently, <em>Q</em> Å 13.| image:: Probability metrics.png</p>
-<p>See <a href="http://en.wikipedia.org/wiki/FASTQ_format" class="uri">http://en.wikipedia.org/wiki/FASTQ_format</a></p>
+<section id="prerequisites" class="level3" data-number="1.2.2">
+<h3 data-number="1.2.2" class="anchored" data-anchor-id="prerequisites"><span class="header-section-number">1.2.2</span> Prerequisites</h3>
+<p>The <em>OBITools4</em> are developped using the <a href="https://go.dev/">GO programming language</a>, we stick to the latest version of the language, today the <span class="math inline">\(1.19.5\)</span>. If you want to download and compile the sources yourself, you first need to install the corresponding compiler on your system. Some parts of the soft are also written in C, therefore a recent C compiler is also requested, GCC on Linux or Windows, the Developer Tools on Mac.</p>
+<p>Whatever the installation you decide for, you will have to ensure that a C compiler is available on your system.</p>
 
 
 <div id="refs" class="references csl-bib-body hanging-indent" role="doc-bibliography" style="display: none">
-<div id="ref-cock2010sanger" class="csl-entry" role="doc-biblioentry">
-Cock, Peter JA, Christopher J Fields, Naohisa Goto, Michael L Heuer, and Peter M Rice. 2010. <span>“The Sanger FASTQ File Format for Sequences with Quality Scores, and the Solexa/Illumina FASTQ Variants.”</span> <em>Nucleic Acids Research</em> 38 (6): 1767–71.
+<div id="ref-Andersen2012-gj" class="csl-entry" role="doc-biblioentry">
+Andersen, Kenneth, Karen Lise Bird, Morten Rasmussen, James Haile, Henrik Breuning-Madsen, Kurt H Kjaer, Ludovic Orlando, M Thomas P Gilbert, and Eske Willerslev. 2012. <span>“<span class="nocase">Meta-barcoding of <span class="nocase">ë</span>dirt<span class="nocase">ı́</span>DNA from soil reflects vertebrate biodiversity</span>.”</span> <em>Molecular Ecology</em> 21 (8): 1966–79.
+</div>
+<div id="ref-Baldwin2013-yc" class="csl-entry" role="doc-biblioentry">
+Baldwin, Darren S, Matthew J Colloff, Gavin N Rees, Anthony A Chariton, Garth O Watson, Leon N Court, Diana M Hartley, et al. 2013. <span>“<span class="nocase">Impacts of inundation and drought on eukaryote biodiversity in semi-arid floodplain soils</span>.”</span> <em>Molecular Ecology</em> 22 (6): 1746–58. <a href="https://doi.org/10.1111/mec.12190">https://doi.org/10.1111/mec.12190</a>.
+</div>
+<div id="ref-Caporaso2010-ii" class="csl-entry" role="doc-biblioentry">
+Caporaso, J Gregory, Justin Kuczynski, Jesse Stombaugh, Kyle Bittinger, Frederic D Bushman, Elizabeth K Costello, Noah Fierer, et al. 2010. <span>“<span class="nocase">QIIME allows analysis of high-throughput community sequencing data</span>.”</span> <em>Nature Methods</em> 7 (5): 335–36. <a href="https://doi.org/10.1038/nmeth.f.303">https://doi.org/10.1038/nmeth.f.303</a>.
+</div>
+<div id="ref-Chariton2010-cz" class="csl-entry" role="doc-biblioentry">
+Chariton, Anthony A, Anthony C Roach, Stuart L Simpson, and Graeme E Batley. 2010. <span>“<span class="nocase">Influence of the choice of physical and chemistry variables on interpreting patterns of sediment contaminants and their relationships with estuarine macrobenthic communities</span>.”</span> <em>Marine and Freshwater Research</em>. <a href="https://doi.org/10.1071/mf09263">https://doi.org/10.1071/mf09263</a>.
+</div>
+<div id="ref-Deagle2009-yh" class="csl-entry" role="doc-biblioentry">
+Deagle, Bruce E, Roger Kirkwood, and Simon N Jarman. 2009. <span>“<span class="nocase">Analysis of Australian fur seal diet by pyrosequencing prey DNA in faeces</span>.”</span> <em>Molecular Ecology</em> 18 (9): 2022–38. <a href="https://doi.org/10.1111/j.1365-294X.2009.04158.x">https://doi.org/10.1111/j.1365-294X.2009.04158.x</a>.
+</div>
+<div id="ref-Kowalczyk2011-kg" class="csl-entry" role="doc-biblioentry">
+Kowalczyk, Rafał, Pierre Taberlet, Eric Coissac, Alice Valentini, Christian Miquel, Tomasz Kamiński, and Jan M Wójcik. 2011. <span>“<span class="nocase">Influence of management practices on large herbivore diet—Case of European bison in Bia<span class="nocase">ł</span>owie<span class="nocase">ż</span>a Primeval Forest (Poland)</span>.”</span> <em>Forest Ecology and Management</em> 261 (4): 821–28. <a href="https://doi.org/10.1016/j.foreco.2010.11.026">https://doi.org/10.1016/j.foreco.2010.11.026</a>.
+</div>
+<div id="ref-Parducci2012-rn" class="csl-entry" role="doc-biblioentry">
+Parducci, Laura, Tina Jørgensen, Mari Mette Tollefsrud, Ellen Elverland, Torbjørn Alm, Sonia L Fontana, K D Bennett, et al. 2012. <span>“<span class="nocase">Glacial survival of boreal trees in northern Scandinavia</span>.”</span> <em>Science</em> 335 (6072): 1083–86. <a href="https://doi.org/10.1126/science.1216043">https://doi.org/10.1126/science.1216043</a>.
+</div>
+<div id="ref-Schloss2009-qy" class="csl-entry" role="doc-biblioentry">
+Schloss, Patrick D, Sarah L Westcott, Thomas Ryabin, Justine R Hall, Martin Hartmann, Emily B Hollister, Ryan A Lesniewski, et al. 2009. <span>“<span class="nocase">Introducing mothur: open-source, platform-independent, community-supported software for describing and comparing microbial communities</span>.”</span> <em>Applied and Environmental Microbiology</em> 75 (23): 7537–41. <a href="https://doi.org/10.1128/AEM.01541-09">https://doi.org/10.1128/AEM.01541-09</a>.
+</div>
+<div id="ref-Shehzad2012-pn" class="csl-entry" role="doc-biblioentry">
+Shehzad, Wasim, Tiayyba Riaz, Muhammad A Nawaz, Christian Miquel, Carole Poillot, Safdar A Shah, Francois Pompanon, Eric Coissac, and Pierre Taberlet. 2012. <span>“<span class="nocase">Carnivore diet analysis based on next-generation sequencing: Application to the leopard cat (Prionailurus bengalensis) in Pakistan</span>.”</span> <em>Molecular Ecology</em> 21 (8): 1951–65. <a href="https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x">https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x</a>.
+</div>
+<div id="ref-Sogin2006-ab" class="csl-entry" role="doc-biblioentry">
+Sogin, Mitchell L, Hilary G Morrison, Julie A Huber, David Mark Welch, Susan M Huse, Phillip R Neal, Jesus M Arrieta, and Gerhard J Herndl. 2006. <span>“<span class="nocase">Microbial diversity in the deep sea and the underexplored "rare biosphere"</span>.”</span> <em>Proceedings of the National Academy of Sciences of the United States of America</em> 103 (32): 12115–20. <a href="https://doi.org/10.1073/pnas.0605127103">https://doi.org/10.1073/pnas.0605127103</a>.
+</div>
+<div id="ref-Sonstebo2010-vv" class="csl-entry" role="doc-biblioentry">
+Sønstebø, J H, L Gielly, A K Brysting, R Elven, M Edwards, J Haile, E Willerslev, et al. 2010. <span>“<span class="nocase">Using next-generation sequencing for molecular reconstruction of past Arctic vegetation and climate</span>.”</span> <em>Molecular Ecology Resources</em> 10 (6): 1009–18. <a href="https://doi.org/10.1111/j.1755-0998.2010.02855.x">https://doi.org/10.1111/j.1755-0998.2010.02855.x</a>.
+</div>
+<div id="ref-Taberlet2012-pf" class="csl-entry" role="doc-biblioentry">
+Taberlet, Pierre, Eric Coissac, Mehrdad Hajibabaei, and Loren H Rieseberg. 2012. <span>“<span>Environmental DNA</span>.”</span> <em>Molecular Ecology</em> 21 (8): 1789–93. <a href="https://doi.org/10.1111/j.1365-294X.2012.05542.x">https://doi.org/10.1111/j.1365-294X.2012.05542.x</a>.
+</div>
+<div id="ref-Thomsen2012-au" class="csl-entry" role="doc-biblioentry">
+Thomsen, Philip Francis, Jos Kielgast, Lars L Iversen, Carsten Wiuf, Morten Rasmussen, M Thomas P Gilbert, Ludovic Orlando, and Eske Willerslev. 2012. <span>“<span class="nocase">Monitoring endangered freshwater biodiversity using environmental DNA</span>.”</span> <em>Molecular Ecology</em> 21 (11): 2565–73. <a href="https://doi.org/10.1111/j.1365-294X.2011.05418.x">https://doi.org/10.1111/j.1365-294X.2011.05418.x</a>.
+</div>
+<div id="ref-Valentini2009-ay" class="csl-entry" role="doc-biblioentry">
+Valentini, Alice, Christian Miquel, Muhammad Ali Nawaz, Eva Bellemain, Eric Coissac, François Pompanon, Ludovic Gielly, et al. 2009. <span>“<span class="nocase">New perspectives in diet analysis based on DNA barcoding and parallel pyrosequencing: the trnL approach</span>.”</span> <em>Molecular Ecology Resources</em> 9 (1): 51–60. <a href="https://doi.org/10.1111/j.1755-0998.2008.02352.x">https://doi.org/10.1111/j.1755-0998.2008.02352.x</a>.
+</div>
+<div id="ref-Yoccoz2012-ix" class="csl-entry" role="doc-biblioentry">
+Yoccoz, N G, K A Bråthen, L Gielly, J Haile, M E Edwards, T Goslar, H Von Stedingk, et al. 2012. <span>“<span class="nocase">DNA from soil mirrors plant taxonomic and growth form diversity</span>.”</span> <em>Molecular Ecology</em> 21 (15): 3647–55. <a href="https://doi.org/10.1111/j.1365-294X.2012.05545.x">https://doi.org/10.1111/j.1365-294X.2012.05545.x</a>.
 </div>
 </div>
 </section>
-<section id="footnotes" class="footnotes footnotes-end-of-document" role="doc-endnotes">
-<hr>
-<ol>
-<li id="fn1"><p>This article uses material from the Wikipedia article <a href="http://en.wikipedia.org/wiki/FASTQ_format"><code>FASTQ format</code></a> which is released under the <code>Creative Commons Attribution-Share-Alike License 3.0</code><a href="#fnref1" class="footnote-back" role="doc-backlink">↩︎</a></p></li>
-</ol>
 </section>
 
 </main> <!-- /main -->
@@ -529,8 +401,8 @@ window.document.addEventListener("DOMContentLoaded", function (event) {
       </a>          
   </div>
   <div class="nav-page nav-page-next">
-      <a href="./tutorial.html" class="pagination-link">
-        <span class="nav-page-text"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></span> <i class="bi bi-arrow-right-short"></i>
+      <a href="./formats.html" class="pagination-link">
+        <span class="nav-page-text"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></span> <i class="bi bi-arrow-right-short"></i>
       </a>
   </div>
 </nav>
diff --git a/doc/_book/library.html b/doc/_book/library.html
index 35c7898..15f5a8e 100644
--- a/doc/_book/library.html
+++ b/doc/_book/library.html
@@ -7,7 +7,7 @@
 <meta name="viewport" content="width=device-width, initial-scale=1.0, user-scalable=yes">
 
 
-<title>OBITools V4 - 4&nbsp; The GO OBITools library</title>
+<title>OBITools V4 - 5&nbsp; The GO OBITools library</title>
 <style>
 code{white-space: pre-wrap;}
 span.smallcaps{font-variant: small-caps;}
@@ -133,7 +133,7 @@ code span.wa { color: #60a0b0; font-weight: bold; font-style: italic; } /* Warni
   <header id="quarto-header" class="headroom fixed-top">
   <nav class="quarto-secondary-nav" data-bs-toggle="collapse" data-bs-target="#quarto-sidebar" aria-controls="quarto-sidebar" aria-expanded="false" aria-label="Toggle sidebar navigation" onclick="if (window.quartoToggleHeadroom) { window.quartoToggleHeadroom(); }">
     <div class="container-fluid d-flex justify-content-between">
-      <h1 class="quarto-secondary-nav-title"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></h1>
+      <h1 class="quarto-secondary-nav-title"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></h1>
       <button type="button" class="quarto-btn-toggle btn" aria-label="Show secondary navigation">
         <i class="bi bi-chevron-right"></i>
       </button>
@@ -168,22 +168,27 @@ code span.wa { color: #60a0b0; font-weight: bold; font-style: italic; } /* Warni
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
+  <a href="./formats.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
+  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./library.html" class="sidebar-item-text sidebar-link active"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></a>
+  <a href="./library.html" class="sidebar-item-text sidebar-link active"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></a>
   </div>
 </li>
         <li class="sidebar-item">
@@ -200,18 +205,18 @@ code span.wa { color: #60a0b0; font-weight: bold; font-style: italic; } /* Warni
     <h2 id="toc-title">Table of contents</h2>
    
   <ul>
-  <li><a href="#biosequence" id="toc-biosequence" class="nav-link active" data-scroll-target="#biosequence"><span class="toc-section-number">4.1</span>  BioSequence</a>
+  <li><a href="#biosequence" id="toc-biosequence" class="nav-link active" data-scroll-target="#biosequence"><span class="toc-section-number">5.1</span>  BioSequence</a>
   <ul class="collapse">
-  <li><a href="#creating-new-instances" id="toc-creating-new-instances" class="nav-link" data-scroll-target="#creating-new-instances"><span class="toc-section-number">4.1.1</span>  Creating new instances</a></li>
-  <li><a href="#end-of-life-of-a-biosequence-instance" id="toc-end-of-life-of-a-biosequence-instance" class="nav-link" data-scroll-target="#end-of-life-of-a-biosequence-instance"><span class="toc-section-number">4.1.2</span>  End of life of a <code>BioSequence</code> instance</a></li>
-  <li><a href="#accessing-to-the-elements-of-a-sequence" id="toc-accessing-to-the-elements-of-a-sequence" class="nav-link" data-scroll-target="#accessing-to-the-elements-of-a-sequence"><span class="toc-section-number">4.1.3</span>  Accessing to the elements of a sequence</a></li>
-  <li><a href="#the-annotations-of-a-sequence" id="toc-the-annotations-of-a-sequence" class="nav-link" data-scroll-target="#the-annotations-of-a-sequence"><span class="toc-section-number">4.1.4</span>  The annotations of a sequence</a></li>
+  <li><a href="#creating-new-instances" id="toc-creating-new-instances" class="nav-link" data-scroll-target="#creating-new-instances"><span class="toc-section-number">5.1.1</span>  Creating new instances</a></li>
+  <li><a href="#end-of-life-of-a-biosequence-instance" id="toc-end-of-life-of-a-biosequence-instance" class="nav-link" data-scroll-target="#end-of-life-of-a-biosequence-instance"><span class="toc-section-number">5.1.2</span>  End of life of a <code>BioSequence</code> instance</a></li>
+  <li><a href="#accessing-to-the-elements-of-a-sequence" id="toc-accessing-to-the-elements-of-a-sequence" class="nav-link" data-scroll-target="#accessing-to-the-elements-of-a-sequence"><span class="toc-section-number">5.1.3</span>  Accessing to the elements of a sequence</a></li>
+  <li><a href="#the-annotations-of-a-sequence" id="toc-the-annotations-of-a-sequence" class="nav-link" data-scroll-target="#the-annotations-of-a-sequence"><span class="toc-section-number">5.1.4</span>  The annotations of a sequence</a></li>
   </ul></li>
-  <li><a href="#the-sequence-iterator" id="toc-the-sequence-iterator" class="nav-link" data-scroll-target="#the-sequence-iterator"><span class="toc-section-number">4.2</span>  The sequence iterator</a>
+  <li><a href="#the-sequence-iterator" id="toc-the-sequence-iterator" class="nav-link" data-scroll-target="#the-sequence-iterator"><span class="toc-section-number">5.2</span>  The sequence iterator</a>
   <ul class="collapse">
-  <li><a href="#basic-usage-of-a-sequence-iterator" id="toc-basic-usage-of-a-sequence-iterator" class="nav-link" data-scroll-target="#basic-usage-of-a-sequence-iterator"><span class="toc-section-number">4.2.1</span>  Basic usage of a sequence iterator</a></li>
-  <li><a href="#the-pipable-functions" id="toc-the-pipable-functions" class="nav-link" data-scroll-target="#the-pipable-functions"><span class="toc-section-number">4.2.2</span>  The <code>Pipable</code> functions</a></li>
-  <li><a href="#the-teeable-functions" id="toc-the-teeable-functions" class="nav-link" data-scroll-target="#the-teeable-functions"><span class="toc-section-number">4.2.3</span>  The <code>Teeable</code> functions</a></li>
+  <li><a href="#basic-usage-of-a-sequence-iterator" id="toc-basic-usage-of-a-sequence-iterator" class="nav-link" data-scroll-target="#basic-usage-of-a-sequence-iterator"><span class="toc-section-number">5.2.1</span>  Basic usage of a sequence iterator</a></li>
+  <li><a href="#the-pipable-functions" id="toc-the-pipable-functions" class="nav-link" data-scroll-target="#the-pipable-functions"><span class="toc-section-number">5.2.2</span>  The <code>Pipable</code> functions</a></li>
+  <li><a href="#the-teeable-functions" id="toc-the-teeable-functions" class="nav-link" data-scroll-target="#the-teeable-functions"><span class="toc-section-number">5.2.3</span>  The <code>Teeable</code> functions</a></li>
   </ul></li>
   </ul>
 </nav>
@@ -221,7 +226,7 @@ code span.wa { color: #60a0b0; font-weight: bold; font-style: italic; } /* Warni
 
 <header id="title-block-header" class="quarto-title-block default">
 <div class="quarto-title">
-<h1 class="title d-none d-lg-block"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></h1>
+<h1 class="title d-none d-lg-block"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></h1>
 </div>
 
 
@@ -236,15 +241,15 @@ code span.wa { color: #60a0b0; font-weight: bold; font-style: italic; } /* Warni
 
 </header>
 
-<section id="biosequence" class="level2" data-number="4.1">
-<h2 data-number="4.1" class="anchored" data-anchor-id="biosequence"><span class="header-section-number">4.1</span> BioSequence</h2>
+<section id="biosequence" class="level2" data-number="5.1">
+<h2 data-number="5.1" class="anchored" data-anchor-id="biosequence"><span class="header-section-number">5.1</span> BioSequence</h2>
 <p>The <code>BioSequence</code> class is used to represent biological sequences. It allows for storing : - the sequence itself as a <code>[]byte</code> - the sequencing quality score as a <code>[]byte</code> if needed - an identifier as a <code>string</code> - a definition as a <code>string</code> - a set of <em>(key, value)</em> pairs in a <code>map[sting]interface{}</code></p>
 <p>BioSequence is defined in the obiseq module and is included using the code</p>
 <div class="sourceCode" id="cb1"><pre class="sourceCode go code-with-copy"><code class="sourceCode go"><span id="cb1-1"><a href="#cb1-1" aria-hidden="true" tabindex="-1"></a><span class="kw">import</span> <span class="op">(</span></span>
 <span id="cb1-2"><a href="#cb1-2" aria-hidden="true" tabindex="-1"></a>    <span class="st">"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiseq"</span></span>
 <span id="cb1-3"><a href="#cb1-3" aria-hidden="true" tabindex="-1"></a><span class="op">)</span></span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
-<section id="creating-new-instances" class="level3" data-number="4.1.1">
-<h3 data-number="4.1.1" class="anchored" data-anchor-id="creating-new-instances"><span class="header-section-number">4.1.1</span> Creating new instances</h3>
+<section id="creating-new-instances" class="level3" data-number="5.1.1">
+<h3 data-number="5.1.1" class="anchored" data-anchor-id="creating-new-instances"><span class="header-section-number">5.1.1</span> Creating new instances</h3>
 <p>To create new instance, use</p>
 <ul>
 <li><code>MakeBioSequence(id string, sequence []byte, definition string) obiseq.BioSequence</code></li>
@@ -271,12 +276,12 @@ code span.wa { color: #60a0b0; font-weight: bold; font-style: italic; } /* Warni
 <pre><code>&gt;id definition containing potentially several words
 sequence</code></pre>
 </section>
-<section id="end-of-life-of-a-biosequence-instance" class="level3" data-number="4.1.2">
-<h3 data-number="4.1.2" class="anchored" data-anchor-id="end-of-life-of-a-biosequence-instance"><span class="header-section-number">4.1.2</span> End of life of a <code>BioSequence</code> instance</h3>
+<section id="end-of-life-of-a-biosequence-instance" class="level3" data-number="5.1.2">
+<h3 data-number="5.1.2" class="anchored" data-anchor-id="end-of-life-of-a-biosequence-instance"><span class="header-section-number">5.1.2</span> End of life of a <code>BioSequence</code> instance</h3>
 <p>When an instance of <code>BioSequence</code> is no longer in use, it is normally taken over by the GO garbage collector. If you know that an instance will never be used again, you can, if you wish, call the <code>Recycle</code> method on it to store the allocated memory elements in a <code>pool</code> to limit the allocation effort when many sequences are being handled. Once the recycle method has been called on an instance, you must ensure that no other method is called on it.</p>
 </section>
-<section id="accessing-to-the-elements-of-a-sequence" class="level3" data-number="4.1.3">
-<h3 data-number="4.1.3" class="anchored" data-anchor-id="accessing-to-the-elements-of-a-sequence"><span class="header-section-number">4.1.3</span> Accessing to the elements of a sequence</h3>
+<section id="accessing-to-the-elements-of-a-sequence" class="level3" data-number="5.1.3">
+<h3 data-number="5.1.3" class="anchored" data-anchor-id="accessing-to-the-elements-of-a-sequence"><span class="header-section-number">5.1.3</span> Accessing to the elements of a sequence</h3>
 <p>The different elements of an <code>obiseq.BioSequence</code> must be accessed using a set of methods. For the three main elements provided during the creation of a new instance methodes are :</p>
 <ul>
 <li><code>Id() string</code></li>
@@ -305,8 +310,8 @@ sequence</code></pre>
 <span id="cb4-14"><a href="#cb4-14" aria-hidden="true" tabindex="-1"></a>    myseq<span class="op">.</span>SetId<span class="op">(</span><span class="st">"SPE01_0001"</span><span class="op">)</span></span>
 <span id="cb4-15"><a href="#cb4-15" aria-hidden="true" tabindex="-1"></a>    fmt<span class="op">.</span>Println<span class="op">(</span>myseq<span class="op">.</span>Id<span class="op">())</span></span>
 <span id="cb4-16"><a href="#cb4-16" aria-hidden="true" tabindex="-1"></a><span class="op">}</span></span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
-<section id="different-ways-for-accessing-an-editing-the-sequence" class="level4" data-number="4.1.3.1">
-<h4 data-number="4.1.3.1" class="anchored" data-anchor-id="different-ways-for-accessing-an-editing-the-sequence"><span class="header-section-number">4.1.3.1</span> Different ways for accessing an editing the sequence</h4>
+<section id="different-ways-for-accessing-an-editing-the-sequence" class="level4" data-number="5.1.3.1">
+<h4 data-number="5.1.3.1" class="anchored" data-anchor-id="different-ways-for-accessing-an-editing-the-sequence"><span class="header-section-number">5.1.3.1</span> Different ways for accessing an editing the sequence</h4>
 <p>If <code>Sequence()</code>and <code>SetSequence(sequence []byte)</code> methods are the basic ones, several other methods exist.</p>
 <ul>
 <li><code>String() string</code> return the sequence directly converted to a <code>string</code> instance.</li>
@@ -331,8 +336,8 @@ sequence</code></pre>
 <span id="cb5-11"><a href="#cb5-11" aria-hidden="true" tabindex="-1"></a>    fmt<span class="op">.</span>Println<span class="op">(</span>myseq<span class="op">.</span>String<span class="op">())</span></span>
 <span id="cb5-12"><a href="#cb5-12" aria-hidden="true" tabindex="-1"></a><span class="op">}</span></span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
 </section>
-<section id="sequence-quality-scores" class="level4" data-number="4.1.3.2">
-<h4 data-number="4.1.3.2" class="anchored" data-anchor-id="sequence-quality-scores"><span class="header-section-number">4.1.3.2</span> Sequence quality scores</h4>
+<section id="sequence-quality-scores" class="level4" data-number="5.1.3.2">
+<h4 data-number="5.1.3.2" class="anchored" data-anchor-id="sequence-quality-scores"><span class="header-section-number">5.1.3.2</span> Sequence quality scores</h4>
 <p>Sequence quality scores cannot be initialized at the time of instance creation. You must use dedicated methods to add quality scores to a sequence.</p>
 <p>To be coherent the length of both the DNA sequence and que quality score sequence must be equal. But assessment of this constraint is realized. It is of the programmer responsability to check that invariant.</p>
 <p>While accessing to the quality scores relies on the method <code>Quality() []byte</code>, setting the quality need to call one of the following method. They run similarly to their sequence dedicated conterpart.</p>
@@ -344,8 +349,8 @@ sequence</code></pre>
 <p>In a way analogous to the <code>Clear</code> method, <code>ClearQualities()</code> empties the sequence of quality scores.</p>
 </section>
 </section>
-<section id="the-annotations-of-a-sequence" class="level3" data-number="4.1.4">
-<h3 data-number="4.1.4" class="anchored" data-anchor-id="the-annotations-of-a-sequence"><span class="header-section-number">4.1.4</span> The annotations of a sequence</h3>
+<section id="the-annotations-of-a-sequence" class="level3" data-number="5.1.4">
+<h3 data-number="5.1.4" class="anchored" data-anchor-id="the-annotations-of-a-sequence"><span class="header-section-number">5.1.4</span> The annotations of a sequence</h3>
 <p>A sequence can be annotated with attributes. Each attribute is associated with a value. An attribute is identified by its name. The name of an attribute consists of a character string containing no spaces or blank characters. Values can be of several types.</p>
 <ul>
 <li>Scalar types:
@@ -368,11 +373,11 @@ sequence</code></pre>
 </ul>
 </section>
 </section>
-<section id="the-sequence-iterator" class="level2" data-number="4.2">
-<h2 data-number="4.2" class="anchored" data-anchor-id="the-sequence-iterator"><span class="header-section-number">4.2</span> The sequence iterator</h2>
+<section id="the-sequence-iterator" class="level2" data-number="5.2">
+<h2 data-number="5.2" class="anchored" data-anchor-id="the-sequence-iterator"><span class="header-section-number">5.2</span> The sequence iterator</h2>
 <p>The pakage <em>obiter</em> provides an iterator mecanism for manipulating sequences. The main class provided by this package is <code>obiiter.IBioSequence</code>. An <code>IBioSequence</code> iterator provides batch of sequences.</p>
-<section id="basic-usage-of-a-sequence-iterator" class="level3" data-number="4.2.1">
-<h3 data-number="4.2.1" class="anchored" data-anchor-id="basic-usage-of-a-sequence-iterator"><span class="header-section-number">4.2.1</span> Basic usage of a sequence iterator</h3>
+<section id="basic-usage-of-a-sequence-iterator" class="level3" data-number="5.2.1">
+<h3 data-number="5.2.1" class="anchored" data-anchor-id="basic-usage-of-a-sequence-iterator"><span class="header-section-number">5.2.1</span> Basic usage of a sequence iterator</h3>
 <p>Many functions, among them functions reading sequences from a text file, return a <code>IBioSequence</code> iterator. The iterator class provides two main methods:</p>
 <ul>
 <li><code>Next() bool</code></li>
@@ -395,12 +400,12 @@ sequence</code></pre>
 <span id="cb6-14"><a href="#cb6-14" aria-hidden="true" tabindex="-1"></a><span class="op">}</span></span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
 <p>An <code>obiseq.BioSequenceBatch</code> instance is a set of sequences stored in an <code>obiseq.BioSequenceSlice</code> and a sequence number. The number of sequences in a batch is not defined. A batch can even contain zero sequences, if for example all sequences initially included in the batch have been filtered out at some stage of their processing.</p>
 </section>
-<section id="the-pipable-functions" class="level3" data-number="4.2.2">
-<h3 data-number="4.2.2" class="anchored" data-anchor-id="the-pipable-functions"><span class="header-section-number">4.2.2</span> The <code>Pipable</code> functions</h3>
+<section id="the-pipable-functions" class="level3" data-number="5.2.2">
+<h3 data-number="5.2.2" class="anchored" data-anchor-id="the-pipable-functions"><span class="header-section-number">5.2.2</span> The <code>Pipable</code> functions</h3>
 <p>A function consuming a <code>obiiter.IBioSequence</code> and returning a <code>obiiter.IBioSequence</code> is of class <code>obiiter.Pipable</code>.</p>
 </section>
-<section id="the-teeable-functions" class="level3" data-number="4.2.3">
-<h3 data-number="4.2.3" class="anchored" data-anchor-id="the-teeable-functions"><span class="header-section-number">4.2.3</span> The <code>Teeable</code> functions</h3>
+<section id="the-teeable-functions" class="level3" data-number="5.2.3">
+<h3 data-number="5.2.3" class="anchored" data-anchor-id="the-teeable-functions"><span class="header-section-number">5.2.3</span> The <code>Teeable</code> functions</h3>
 <p>A function consuming a <code>obiiter.IBioSequence</code> and returning two <code>obiiter.IBioSequence</code> instance is of class <code>obiiter.Teeable</code>.</p>
 
 
@@ -544,12 +549,12 @@ window.document.addEventListener("DOMContentLoaded", function (event) {
 <nav class="page-navigation">
   <div class="nav-page nav-page-previous">
       <a href="./commands.html" class="pagination-link">
-        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></span>
+        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></span>
       </a>          
   </div>
   <div class="nav-page nav-page-next">
       <a href="./annexes.html" class="pagination-link">
-        <span class="nav-page-text"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></span> <i class="bi bi-arrow-right-short"></i>
+        <span class="nav-page-text"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></span> <i class="bi bi-arrow-right-short"></i>
       </a>
   </div>
 </nav>
diff --git a/doc/_book/references.html b/doc/_book/references.html
index 921bc26..f93de28 100644
--- a/doc/_book/references.html
+++ b/doc/_book/references.html
@@ -123,22 +123,27 @@ div.csl-indent {
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
+  <a href="./formats.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
+  <a href="./tutorial.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></a>
+  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></a>
   </div>
 </li>
         <li class="sidebar-item">
@@ -174,29 +179,79 @@ div.csl-indent {
 </header>
 
 <div id="refs" class="references csl-bib-body hanging-indent" role="doc-bibliography">
+<div id="ref-Andersen2012-gj" class="csl-entry" role="doc-biblioentry">
+Andersen, Kenneth, Karen Lise Bird, Morten Rasmussen, James Haile,
+Henrik Breuning-Madsen, Kurt H Kjaer, Ludovic Orlando, M Thomas P
+Gilbert, and Eske Willerslev. 2012. <span>“<span class="nocase">Meta-barcoding of <span class="nocase">ë</span>dirt<span class="nocase">ı́</span>DNA from soil reflects vertebrate
+biodiversity</span>.”</span> <em>Molecular Ecology</em> 21 (8): 1966–79.
+</div>
+<div id="ref-Baldwin2013-yc" class="csl-entry" role="doc-biblioentry">
+Baldwin, Darren S, Matthew J Colloff, Gavin N Rees, Anthony A Chariton,
+Garth O Watson, Leon N Court, Diana M Hartley, et al. 2013. <span>“<span class="nocase">Impacts of inundation and drought on eukaryote
+biodiversity in semi-arid floodplain soils</span>.”</span> <em>Molecular
+Ecology</em> 22 (6): 1746–58. <a href="https://doi.org/10.1111/mec.12190">https://doi.org/10.1111/mec.12190</a>.
+</div>
 <div id="ref-Boyer2016-gq" class="csl-entry" role="doc-biblioentry">
 Boyer, Frédéric, Céline Mercier, Aurélie Bonin, Yvan Le Bras, Pierre
 Taberlet, and Eric Coissac. 2016. <span>“<span class="nocase">obitools:
 a unix-inspired software package for DNA metabarcoding</span>.”</span>
 <em>Molecular Ecology Resources</em> 16 (1): 176–82. <a href="https://doi.org/10.1111/1755-0998.12428">https://doi.org/10.1111/1755-0998.12428</a>.
 </div>
+<div id="ref-Caporaso2010-ii" class="csl-entry" role="doc-biblioentry">
+Caporaso, J Gregory, Justin Kuczynski, Jesse Stombaugh, Kyle Bittinger,
+Frederic D Bushman, Elizabeth K Costello, Noah Fierer, et al. 2010.
+<span>“<span class="nocase">QIIME allows analysis of high-throughput
+community sequencing data</span>.”</span> <em>Nature Methods</em> 7 (5):
+335–36. <a href="https://doi.org/10.1038/nmeth.f.303">https://doi.org/10.1038/nmeth.f.303</a>.
+</div>
+<div id="ref-Chariton2010-cz" class="csl-entry" role="doc-biblioentry">
+Chariton, Anthony A, Anthony C Roach, Stuart L Simpson, and Graeme E
+Batley. 2010. <span>“<span class="nocase">Influence of the choice of
+physical and chemistry variables on interpreting patterns of sediment
+contaminants and their relationships with estuarine macrobenthic
+communities</span>.”</span> <em>Marine and Freshwater Research</em>. <a href="https://doi.org/10.1071/mf09263">https://doi.org/10.1071/mf09263</a>.
+</div>
 <div id="ref-cock2010sanger" class="csl-entry" role="doc-biblioentry">
 Cock, Peter JA, Christopher J Fields, Naohisa Goto, Michael L Heuer, and
 Peter M Rice. 2010. <span>“The Sanger FASTQ File Format for Sequences
 with Quality Scores, and the Solexa/Illumina FASTQ Variants.”</span>
 <em>Nucleic Acids Research</em> 38 (6): 1767–71.
 </div>
+<div id="ref-Deagle2009-yh" class="csl-entry" role="doc-biblioentry">
+Deagle, Bruce E, Roger Kirkwood, and Simon N Jarman. 2009. <span>“<span class="nocase">Analysis of Australian fur seal diet by pyrosequencing
+prey DNA in faeces</span>.”</span> <em>Molecular Ecology</em> 18 (9):
+2022–38. <a href="https://doi.org/10.1111/j.1365-294X.2009.04158.x">https://doi.org/10.1111/j.1365-294X.2009.04158.x</a>.
+</div>
+<div id="ref-Kowalczyk2011-kg" class="csl-entry" role="doc-biblioentry">
+Kowalczyk, Rafał, Pierre Taberlet, Eric Coissac, Alice Valentini,
+Christian Miquel, Tomasz Kamiński, and Jan M Wójcik. 2011. <span>“<span class="nocase">Influence of management practices on large herbivore
+diet—Case of European bison in Bia<span class="nocase">ł</span>owie<span class="nocase">ż</span>a Primeval Forest (Poland)</span>.”</span>
+<em>Forest Ecology and Management</em> 261 (4): 821–28. <a href="https://doi.org/10.1016/j.foreco.2010.11.026">https://doi.org/10.1016/j.foreco.2010.11.026</a>.
+</div>
 <div id="ref-Lipman1985-hw" class="csl-entry" role="doc-biblioentry">
 Lipman, D J, and W R Pearson. 1985. <span>“<span class="nocase">Rapid
 and sensitive protein similarity searches</span>.”</span>
 <em>Science</em> 227 (4693): 1435–41. <a href="http://www.ncbi.nlm.nih.gov/pubmed/2983426">http://www.ncbi.nlm.nih.gov/pubmed/2983426</a>.
 </div>
+<div id="ref-Parducci2012-rn" class="csl-entry" role="doc-biblioentry">
+Parducci, Laura, Tina Jørgensen, Mari Mette Tollefsrud, Ellen Elverland,
+Torbjørn Alm, Sonia L Fontana, K D Bennett, et al. 2012. <span>“<span class="nocase">Glacial survival of boreal trees in northern
+Scandinavia</span>.”</span> <em>Science</em> 335 (6072): 1083–86. <a href="https://doi.org/10.1126/science.1216043">https://doi.org/10.1126/science.1216043</a>.
+</div>
 <div id="ref-Riaz2011-gn" class="csl-entry" role="doc-biblioentry">
 Riaz, Tiayyba, Wasim Shehzad, Alain Viari, François Pompanon, Pierre
 Taberlet, and Eric Coissac. 2011. <span>“<span class="nocase">ecoPrimers: inference of new DNA barcode markers from
 whole genome sequence analysis</span>.”</span> <em>Nucleic Acids
 Research</em> 39 (21): e145. <a href="https://doi.org/10.1093/nar/gkr732">https://doi.org/10.1093/nar/gkr732</a>.
 </div>
+<div id="ref-Schloss2009-qy" class="csl-entry" role="doc-biblioentry">
+Schloss, Patrick D, Sarah L Westcott, Thomas Ryabin, Justine R Hall,
+Martin Hartmann, Emily B Hollister, Ryan A Lesniewski, et al. 2009.
+<span>“<span class="nocase">Introducing mothur: open-source,
+platform-independent, community-supported software for describing and
+comparing microbial communities</span>.”</span> <em>Applied and
+Environmental Microbiology</em> 75 (23): 7537–41. <a href="https://doi.org/10.1128/AEM.01541-09">https://doi.org/10.1128/AEM.01541-09</a>.
+</div>
 <div id="ref-Seguritan2001-tg" class="csl-entry" role="doc-biblioentry">
 Seguritan, V, and F Rohwer. 2001. <span>“<span class="nocase">FastGroup:
 a program to dereplicate libraries of 16S rDNA sequences</span>.”</span>
@@ -210,6 +265,47 @@ based on next-generation sequencing: Application to the leopard cat
 (Prionailurus bengalensis) in Pakistan</span>.”</span> <em>Molecular
 Ecology</em> 21 (8): 1951–65. <a href="https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x">https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x</a>.
 </div>
+<div id="ref-Sogin2006-ab" class="csl-entry" role="doc-biblioentry">
+Sogin, Mitchell L, Hilary G Morrison, Julie A Huber, David Mark Welch,
+Susan M Huse, Phillip R Neal, Jesus M Arrieta, and Gerhard J Herndl.
+2006. <span>“<span class="nocase">Microbial diversity in the deep sea
+and the underexplored "rare biosphere"</span>.”</span> <em>Proceedings
+of the National Academy of Sciences of the United States of America</em>
+103 (32): 12115–20. <a href="https://doi.org/10.1073/pnas.0605127103">https://doi.org/10.1073/pnas.0605127103</a>.
+</div>
+<div id="ref-Sonstebo2010-vv" class="csl-entry" role="doc-biblioentry">
+Sønstebø, J H, L Gielly, A K Brysting, R Elven, M Edwards, J Haile, E
+Willerslev, et al. 2010. <span>“<span class="nocase">Using
+next-generation sequencing for molecular reconstruction of past Arctic
+vegetation and climate</span>.”</span> <em>Molecular Ecology
+Resources</em> 10 (6): 1009–18. <a href="https://doi.org/10.1111/j.1755-0998.2010.02855.x">https://doi.org/10.1111/j.1755-0998.2010.02855.x</a>.
+</div>
+<div id="ref-Taberlet2012-pf" class="csl-entry" role="doc-biblioentry">
+Taberlet, Pierre, Eric Coissac, Mehrdad Hajibabaei, and Loren H
+Rieseberg. 2012. <span>“<span>Environmental DNA</span>.”</span>
+<em>Molecular Ecology</em> 21 (8): 1789–93. <a href="https://doi.org/10.1111/j.1365-294X.2012.05542.x">https://doi.org/10.1111/j.1365-294X.2012.05542.x</a>.
+</div>
+<div id="ref-Thomsen2012-au" class="csl-entry" role="doc-biblioentry">
+Thomsen, Philip Francis, Jos Kielgast, Lars L Iversen, Carsten Wiuf,
+Morten Rasmussen, M Thomas P Gilbert, Ludovic Orlando, and Eske
+Willerslev. 2012. <span>“<span class="nocase">Monitoring endangered
+freshwater biodiversity using environmental DNA</span>.”</span>
+<em>Molecular Ecology</em> 21 (11): 2565–73. <a href="https://doi.org/10.1111/j.1365-294X.2011.05418.x">https://doi.org/10.1111/j.1365-294X.2011.05418.x</a>.
+</div>
+<div id="ref-Valentini2009-ay" class="csl-entry" role="doc-biblioentry">
+Valentini, Alice, Christian Miquel, Muhammad Ali Nawaz, Eva Bellemain,
+Eric Coissac, François Pompanon, Ludovic Gielly, et al. 2009.
+<span>“<span class="nocase">New perspectives in diet analysis based on
+DNA barcoding and parallel pyrosequencing: the trnL
+approach</span>.”</span> <em>Molecular Ecology Resources</em> 9 (1):
+51–60. <a href="https://doi.org/10.1111/j.1755-0998.2008.02352.x">https://doi.org/10.1111/j.1755-0998.2008.02352.x</a>.
+</div>
+<div id="ref-Yoccoz2012-ix" class="csl-entry" role="doc-biblioentry">
+Yoccoz, N G, K A Bråthen, L Gielly, J Haile, M E Edwards, T Goslar, H
+Von Stedingk, et al. 2012. <span>“<span class="nocase">DNA from soil
+mirrors plant taxonomic and growth form diversity</span>.”</span>
+<em>Molecular Ecology</em> 21 (15): 3647–55. <a href="https://doi.org/10.1111/j.1365-294X.2012.05545.x">https://doi.org/10.1111/j.1365-294X.2012.05545.x</a>.
+</div>
 </div>
 
 
@@ -351,7 +447,7 @@ window.document.addEventListener("DOMContentLoaded", function (event) {
 <nav class="page-navigation">
   <div class="nav-page nav-page-previous">
       <a href="./annexes.html" class="pagination-link">
-        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></span>
+        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></span>
       </a>          
   </div>
   <div class="nav-page nav-page-next">
diff --git a/doc/_book/search.json b/doc/_book/search.json
index 38a2247..52d8af1 100644
--- a/doc/_book/search.json
+++ b/doc/_book/search.json
@@ -11,132 +11,118 @@
     "href": "intro.html#aims-of-obitools",
     "title": "1  The OBITools",
     "section": "1.1 Aims of OBITools",
-    "text": "1.1 Aims of OBITools"
+    "text": "1.1 Aims of OBITools\nDNA metabarcoding is an efficient approach for biodiversity studies (Taberlet et al. 2012). Originally mainly developed by microbiologists (e.g. Sogin et al. 2006), it is now widely used for plants (e.g. Sønstebø et al. 2010; Yoccoz et al. 2012; Parducci et al. 2012) and animals from meiofauna (e.g. Chariton et al. 2010; Baldwin et al. 2013) to larger organisms (e.g. Andersen et al. 2012; Thomsen et al. 2012). Interestingly, this method is not limited to sensu stricto biodiversity surveys, but it can also be implemented in other ecological contexts such as for herbivore (e.g. Valentini et al. 2009; Kowalczyk et al. 2011) or carnivore (e.g. Deagle, Kirkwood, and Jarman 2009; Shehzad et al. 2012) diet analyses.\nWhatever the biological question under consideration, the DNA metabarcoding methodology relies heavily on next-generation sequencing (NGS), and generates considerable numbers of DNA sequence reads (typically million of reads). Manipulation of such large datasets requires dedicated programs usually running on a Unix system. Unix is an operating system, whose first version was created during the sixties. Since its early stages, it is dedicated to scientific computing and includes a large set of simple tools to efficiently process text files. Most of those programs can be viewed as filters extracting information from a text file to create a new text file. These programs process text files as streams, line per line, therefore allowing computation on a huge dataset without requiring a large memory. Unix programs usually print their results to their standard output (stdout), which by default is the terminal, so the results can be examined on screen. The main philosophy of the Unix environment is to allow easy redirection of the stdout either to a file, for saving the results, or to the standard input (stdin) of a second program thus allowing to easily create complex processing from simple base commands. Access to Unix computers is increasingly easier for scientists nowadays. Indeed, the Linux operating system, an open source version of Unix, can be freely installed on every PC machine and the MacOS operating system, running on Apple computers, is also a Unix system. The OBITools programs imitate Unix standard programs because they usually act as filters, reading their data from text files or the stdin and writing their results to the stdout. The main difference with classical Unix programs is that text files are not analyzed line per line but sequence record per sequence record (see below for a detailed description of a sequence record). Compared to packages for similar purposes like mothur (Schloss et al. 2009) or QIIME (Caporaso et al. 2010), the OBITools mainly rely on filtering and sorting algorithms. This allows users to set up versatile data analysis pipelines (Figure 1), adjustable to the broad range of DNA metabarcoding applications. The innovation of the OBITools is their ability to take into account the taxonomic annotations, ultimately allowing sorting and filtering of sequence records based on the taxonomy."
   },
   {
-    "objectID": "intro.html#file-formats-usable-with-obitools",
-    "href": "intro.html#file-formats-usable-with-obitools",
+    "objectID": "intro.html#installation-of-the-obitools",
+    "href": "intro.html#installation-of-the-obitools",
     "title": "1  The OBITools",
-    "section": "1.2 File formats usable with OBITools",
-    "text": "1.2 File formats usable with OBITools\n\n1.2.1 The sequence files\nSequences can be stored following various format. OBITools knows some of them. The central formats for sequence files manipulated by OBITools scripts are the fasta and fastq format. OBITools extends the both these formats by specifying a syntax to include in the definition line data qualifying the sequence. All file formats use the IUPAC code for encoding nucleotides.\n\n\n1.2.2 The IUPAC Code\nThe International Union of Pure and Applied Chemistry (IUPAC_) defined the standard code for representing protein or DNA sequences.\n\n1.2.2.1 Nucleic IUPAC Code\n\n\n\nCode\nNucleotide\n\n\n\n\nA\nAdenine\n\n\nC\nCytosine\n\n\nG\nGuanine\n\n\nT\nThymine\n\n\nU\nUracil\n\n\nR\nPurine (A or G)\n\n\nY\nPyrimidine (C, T, or U)\n\n\nM\nC or A\n\n\nK\nT, U, or G\n\n\nW\nT, U, or A\n\n\nS\nC or G\n\n\nB\nC, T, U, or G (not A)\n\n\nD\nA, T, U, or G (not C)\n\n\nH\nA, T, U, or C (not G)\n\n\nV\nA, C, or G (not T, not U)\n\n\nN\nAny base (A, C, G, T, or U)\n\n\n\n\n\n\n1.2.3 The fasta format\nThe fasta format is certainly the most widely used sequence file format. This is certainly due to its great simplicity. It was originally created for the Lipman and Pearson FASTA program. OBITools use in more of the classical :ref:fasta format an :ref:extended version of this format where structured data are included in the title line.\nIn fasta format a sequence is represented by a title line beginning with a > character and the sequences by itself following the :doc:iupac code. The sequence is usually split other severals lines of the same length (expect for the last one)\n>my_sequence this is my pretty sequence\nACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT\nGTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT\nAACGACGTTGCAGTACGTTGCAGT\nThis is no special format for the title line excepting that this line should be unique. Usually the first word following the > character is considered as the sequence identifier. The end of the title line corresponding to a description of the sequence. Several sequences can be concatenated in a same file. The description of the next sequence is just pasted at the end of the record of the previous one\n>sequence_A this is my first pretty sequence\nACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT\nGTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT\nAACGACGTTGCAGTACGTTGCAGT\n>sequence_B this is my second pretty sequence\nACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT\nGTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT\nAACGACGTTGCAGTACGTTGCAGT\n>sequence_C this is my third pretty sequence\nACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT\nGTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT\nAACGACGTTGCAGTACGTTGCAGT\n\n\n1.2.4 The fastq sequence format1\nfastq format is a text-based format for storing both a biological sequence (usually nucleotide sequence) and its corresponding quality scores. Both the sequence letter and quality score are encoded with a single ASCII character for brevity. It was originally developed at the Wellcome Trust Sanger Institute to bundle a fasta sequence and its quality data, but has recently become the de facto standard for storing the output of high throughput sequencing instruments such as the Illumina Genome Analyzer Illumina (Cock et al. 2010) .\nA fastq file normally uses four lines per sequence.\n\nLine 1 begins with a ‘@’ character and is followed by a sequence identifier and an optional description (like a :ref:fasta title line).\nLine 2 is the raw sequence letters.\nLine 3 begins with a ‘+’ character and is optionally followed by the same sequence identifier (and any description) again.\nLine 4 encodes the quality values for the sequence in Line 2, and must contain the same number of symbols as letters in the sequence.\n\nA fastq file containing a single sequence might look like this:\n@SEQ_ID\nGATTTGGGGTTCAAAGCAGTATCGATCAAATAGTAAATCCATTTGTTCAACTCACAGTTT\n+\n!''*((((***+))%%%++)(%%%%).1***-+*''))**55CCF>>>>>>CCCCCCC65\nThe character ‘!’ represents the lowest quality while ‘~’ is the highest. Here are the quality value characters in left-to-right increasing order of quality (ASCII):\n!\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~\nThe original Sanger FASTQ files also allowed the sequence and quality strings to be wrapped (split over multiple lines), but this is generally discouraged as it can make parsing complicated due to the unfortunate choice of “@” and “+” as markers (these characters can also occur in the quality string).\n\n1.2.4.1 Variations\n\n1.2.4.1.1 Quality\nA quality value Q is an integer mapping of p (i.e., the probability that the corresponding base call is incorrect). Two different equations have been in use. The first is the standard Sanger variant to assess reliability of a base call, otherwise known as Phred quality score:\n\\[\nQ_\\text{sanger} = -10 \\, \\log_{10} p\n\\]\nThe Solexa pipeline (i.e., the software delivered with the Illumina Genome Analyzer) earlier used a different mapping, encoding the odds \\(\\mathbf{p}/(1-\\mathbf{p})\\) instead of the probability \\(\\mathbf{p}\\):\n\\[\nQ_\\text{solexa-prior to v.1.3} = -10 \\, \\log_{10} \\frac{p}{1-p}\n\\]\nAlthough both mappings are asymptotically identical at higher quality values, they differ at lower quality levels (i.e., approximately \\(\\mathbf{p} > 0.05\\), or equivalently, \\(\\mathbf{Q} < 13\\)).\n|Relationship between Q and p using the Sanger (red) and Solexa (black) equations (described above). The vertical dotted line indicates \\(\\mathbf{p}= 0.05\\), or equivalently, \\(Q = 13\\).|\n\n\n\n1.2.4.2 Encoding\n\nSanger format can encode a Phred quality score from 0 to 93 using ASCII 33 to 126 (although in raw read data the Phred quality score rarely exceeds 60, higher scores are possible in assemblies or read maps).\nSolexa/Illumina 1.0 format can encode a Solexa/Illumina quality score from -5 to 62 using ASCII 59 to 126 (although in raw read data Solexa scores from -5 to 40 only are expected)\nStarting with Illumina 1.3 and before Illumina 1.8, the format encoded a Phred quality score from 0 to 62 using ASCII 64 to 126 (although in raw read data Phred scores from 0 to 40 only are expected).\nStarting in Illumina 1.5 and before Illumina 1.8, the Phred scores 0 to 2 have a slightly different meaning. The values 0 and 1 are no longer used and the value 2, encoded by ASCII 66 “B”.\n\nSequencing Control Software, Version 2.6, Catalog # SY-960-2601, Part # 15009921 Rev. A, November 2009] [http://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf\\\\](http://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf){.uri} (page 30) states the following: If a read ends with a segment of mostly low quality (Q15 or below), then all of the quality values in the segment are replaced with a value of 2 (encoded as the letter B in Illumina’s text-based encoding of quality scores)… This Q2 indicator does not predict a specific error rate, but rather indicates that a specific final portion of the read should not be used in further analyses. Also, the quality score encoded as “B” letter may occur internally within reads at least as late as pipeline version 1.6, as shown in the following example:\n@HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1\nTTAATTGGTAAATAAATCTCCTAATAGCTTAGATNTTACCTTNNNNNNNNNNTAGTTTCTTGAGATTTGTTGGGGGAGACATTTTTGTGATTGCCTTGAT\n+HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1\nefcfffffcfeefffcffffffddf`feed]`]_Ba_^__[YBBBBBBBBBBRTT\\]][]dddd`ddd^dddadd^BBBBBBBBBBBBBBBBBBBBBBBB\nAn alternative interpretation of this ASCII encoding has been proposed. Also, in Illumina runs using PhiX controls, the character ‘B’ was observed to represent an “unknown quality score”. The error rate of ‘B’ reads was roughly 3 phred scores lower the mean observed score of a given run.\n\nStarting in Illumina 1.8, the quality scores have basically returned to the use of the Sanger format (Phred+33)."
+    "section": "1.2 Installation of the obitools",
+    "text": "1.2 Installation of the obitools\n\n1.2.1 Availability of the OBITools\nThe OBITools are open source and protected by the CeCILL 2.1 license.\nAll the sources of the OBITools4 can be downloaded from the metabarcoding git server (https://git.metabarcoding.org).\n\n\n1.2.2 Prerequisites\nThe OBITools4 are developped using the GO programming language, we stick to the latest version of the language, today the \\(1.19.5\\). If you want to download and compile the sources yourself, you first need to install the corresponding compiler on your system. Some parts of the soft are also written in C, therefore a recent C compiler is also requested, GCC on Linux or Windows, the Developer Tools on Mac.\nWhatever the installation you decide for, you will have to ensure that a C compiler is available on your system.\n\n\n\n\nAndersen, Kenneth, Karen Lise Bird, Morten Rasmussen, James Haile, Henrik Breuning-Madsen, Kurt H Kjaer, Ludovic Orlando, M Thomas P Gilbert, and Eske Willerslev. 2012. “Meta-barcoding of ëdirtı́DNA from soil reflects vertebrate biodiversity.” Molecular Ecology 21 (8): 1966–79.\n\n\nBaldwin, Darren S, Matthew J Colloff, Gavin N Rees, Anthony A Chariton, Garth O Watson, Leon N Court, Diana M Hartley, et al. 2013. “Impacts of inundation and drought on eukaryote biodiversity in semi-arid floodplain soils.” Molecular Ecology 22 (6): 1746–58. https://doi.org/10.1111/mec.12190.\n\n\nCaporaso, J Gregory, Justin Kuczynski, Jesse Stombaugh, Kyle Bittinger, Frederic D Bushman, Elizabeth K Costello, Noah Fierer, et al. 2010. “QIIME allows analysis of high-throughput community sequencing data.” Nature Methods 7 (5): 335–36. https://doi.org/10.1038/nmeth.f.303.\n\n\nChariton, Anthony A, Anthony C Roach, Stuart L Simpson, and Graeme E Batley. 2010. “Influence of the choice of physical and chemistry variables on interpreting patterns of sediment contaminants and their relationships with estuarine macrobenthic communities.” Marine and Freshwater Research. https://doi.org/10.1071/mf09263.\n\n\nDeagle, Bruce E, Roger Kirkwood, and Simon N Jarman. 2009. “Analysis of Australian fur seal diet by pyrosequencing prey DNA in faeces.” Molecular Ecology 18 (9): 2022–38. https://doi.org/10.1111/j.1365-294X.2009.04158.x.\n\n\nKowalczyk, Rafał, Pierre Taberlet, Eric Coissac, Alice Valentini, Christian Miquel, Tomasz Kamiński, and Jan M Wójcik. 2011. “Influence of management practices on large herbivore diet—Case of European bison in Białowieża Primeval Forest (Poland).” Forest Ecology and Management 261 (4): 821–28. https://doi.org/10.1016/j.foreco.2010.11.026.\n\n\nParducci, Laura, Tina Jørgensen, Mari Mette Tollefsrud, Ellen Elverland, Torbjørn Alm, Sonia L Fontana, K D Bennett, et al. 2012. “Glacial survival of boreal trees in northern Scandinavia.” Science 335 (6072): 1083–86. https://doi.org/10.1126/science.1216043.\n\n\nSchloss, Patrick D, Sarah L Westcott, Thomas Ryabin, Justine R Hall, Martin Hartmann, Emily B Hollister, Ryan A Lesniewski, et al. 2009. “Introducing mothur: open-source, platform-independent, community-supported software for describing and comparing microbial communities.” Applied and Environmental Microbiology 75 (23): 7537–41. https://doi.org/10.1128/AEM.01541-09.\n\n\nShehzad, Wasim, Tiayyba Riaz, Muhammad A Nawaz, Christian Miquel, Carole Poillot, Safdar A Shah, Francois Pompanon, Eric Coissac, and Pierre Taberlet. 2012. “Carnivore diet analysis based on next-generation sequencing: Application to the leopard cat (Prionailurus bengalensis) in Pakistan.” Molecular Ecology 21 (8): 1951–65. https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x.\n\n\nSogin, Mitchell L, Hilary G Morrison, Julie A Huber, David Mark Welch, Susan M Huse, Phillip R Neal, Jesus M Arrieta, and Gerhard J Herndl. 2006. “Microbial diversity in the deep sea and the underexplored \"rare biosphere\".” Proceedings of the National Academy of Sciences of the United States of America 103 (32): 12115–20. https://doi.org/10.1073/pnas.0605127103.\n\n\nSønstebø, J H, L Gielly, A K Brysting, R Elven, M Edwards, J Haile, E Willerslev, et al. 2010. “Using next-generation sequencing for molecular reconstruction of past Arctic vegetation and climate.” Molecular Ecology Resources 10 (6): 1009–18. https://doi.org/10.1111/j.1755-0998.2010.02855.x.\n\n\nTaberlet, Pierre, Eric Coissac, Mehrdad Hajibabaei, and Loren H Rieseberg. 2012. “Environmental DNA.” Molecular Ecology 21 (8): 1789–93. https://doi.org/10.1111/j.1365-294X.2012.05542.x.\n\n\nThomsen, Philip Francis, Jos Kielgast, Lars L Iversen, Carsten Wiuf, Morten Rasmussen, M Thomas P Gilbert, Ludovic Orlando, and Eske Willerslev. 2012. “Monitoring endangered freshwater biodiversity using environmental DNA.” Molecular Ecology 21 (11): 2565–73. https://doi.org/10.1111/j.1365-294X.2011.05418.x.\n\n\nValentini, Alice, Christian Miquel, Muhammad Ali Nawaz, Eva Bellemain, Eric Coissac, François Pompanon, Ludovic Gielly, et al. 2009. “New perspectives in diet analysis based on DNA barcoding and parallel pyrosequencing: the trnL approach.” Molecular Ecology Resources 9 (1): 51–60. https://doi.org/10.1111/j.1755-0998.2008.02352.x.\n\n\nYoccoz, N G, K A Bråthen, L Gielly, J Haile, M E Edwards, T Goslar, H Von Stedingk, et al. 2012. “DNA from soil mirrors plant taxonomic and growth form diversity.” Molecular Ecology 21 (15): 3647–55. https://doi.org/10.1111/j.1365-294X.2012.05545.x."
   },
   {
-    "objectID": "intro.html#file-extension",
-    "href": "intro.html#file-extension",
-    "title": "1  The OBITools",
-    "section": "1.3 File extension",
-    "text": "1.3 File extension\nThere is no standard file extension for a FASTQ file, but .fq and .fastq, are commonly used."
-  },
-  {
-    "objectID": "intro.html#see-also",
-    "href": "intro.html#see-also",
-    "title": "1  The OBITools",
-    "section": "1.4 See also",
-    "text": "1.4 See also\n\n:ref:fasta"
-  },
-  {
-    "objectID": "intro.html#references",
-    "href": "intro.html#references",
-    "title": "1  The OBITools",
-    "section": "1.5 References",
-    "text": "1.5 References\n.. [1] Cock et al (2009) The Sanger FASTQ file format for sequences with quality scores, and the Solexa/Illumina FASTQ variants. Nucleic Acids Research,\n.. [2] Illumina Quality Scores, Tobias Mann, Bioinformatics, San Diego, Illumina 1__\n.. |Relationship between Q and p using the Sanger (red) and Solexa (black) equations (described above). The vertical dotted line indicates p = 0.05, or equivalently, Q Å 13.| image:: Probability metrics.png\nSee http://en.wikipedia.org/wiki/FASTQ_format\n\n\n\n\nCock, Peter JA, Christopher J Fields, Naohisa Goto, Michael L Heuer, and Peter M Rice. 2010. “The Sanger FASTQ File Format for Sequences with Quality Scores, and the Solexa/Illumina FASTQ Variants.” Nucleic Acids Research 38 (6): 1767–71."
+    "objectID": "formats.html#the-dna-sequence-data",
+    "href": "formats.html#the-dna-sequence-data",
+    "title": "2  File formats usable with OBITools",
+    "section": "2.1 The DNA sequence data",
+    "text": "2.1 The DNA sequence data\nSequences can be stored following various format. OBITools knows some of them. The central formats for sequence files manipulated by OBITools scripts are the fasta and fastq format. OBITools extends the both these formats by specifying a syntax to include in the definition line data qualifying the sequence. All file formats use the IUPAC code for encoding nucleotides.\nMoreover these two formats that can be used as input and output formats, OBITools4 can read the following format :\n\nEBML flat file format (use by ENA)\nGenbank flat file format\necoPCR output files\n\n\n2.1.1 The IUPAC Code\nThe International Union of Pure and Applied Chemistry (IUPAC_) defined the standard code for representing protein or DNA sequences.\n\n\n\nCode\nNucleotide\n\n\n\n\nA\nAdenine\n\n\nC\nCytosine\n\n\nG\nGuanine\n\n\nT\nThymine\n\n\nU\nUracil\n\n\nR\nPurine (A or G)\n\n\nY\nPyrimidine (C, T, or U)\n\n\nM\nC or A\n\n\nK\nT, U, or G\n\n\nW\nT, U, or A\n\n\nS\nC or G\n\n\nB\nC, T, U, or G (not A)\n\n\nD\nA, T, U, or G (not C)\n\n\nH\nA, T, U, or C (not G)\n\n\nV\nA, C, or G (not T, not U)\n\n\nN\nAny base (A, C, G, T, or U)\n\n\n\n\n\n2.1.2 The fasta sequence format\nThe fasta format is certainly the most widely used sequence file format. This is certainly due to its great simplicity. It was originally created for the Lipman and Pearson FASTA program. OBITools use in more of the classical fasta format an extended version of this format where structured data are included in the title line.\nIn fasta format a sequence is represented by a title line beginning with a > character and the sequences by itself following the :doc:iupac code. The sequence is usually split other severals lines of the same length (expect for the last one)\n>my_sequence this is my pretty sequence\nACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT\nGTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT\nAACGACGTTGCAGTACGTTGCAGT\nThis is no special format for the title line excepting that this line should be unique. Usually the first word following the > character is considered as the sequence identifier. The end of the title line corresponding to a description of the sequence. Several sequences can be concatenated in a same file. The description of the next sequence is just pasted at the end of the record of the previous one\n>sequence_A this is my first pretty sequence\nACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT\nGTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT\nAACGACGTTGCAGTACGTTGCAGT\n>sequence_B this is my second pretty sequence\nACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT\nGTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT\nAACGACGTTGCAGTACGTTGCAGT\n>sequence_C this is my third pretty sequence\nACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT\nGTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT\nAACGACGTTGCAGTACGTTGCAGT\n\n\n2.1.3 The fastq sequence format1\nThe FASTQ format is a text file format for storing both biological sequences (only nucleic acid sequences) and the associated quality scores. The sequence and score are each encoded by a single ASCII character. This format was originally developed by the Wellcome Trust Sanger Institute to link a FASTA sequence file to the corresponding quality data, but has recently become the de facto standard for storing results from high-throughput sequencers (Cock et al. 2010).\nA fastq file normally uses four lines per sequence.\n\nLine 1 begins with a ‘@’ character and is followed by a sequence identifier and an optional description (like a :ref:fasta title line).\nLine 2 is the raw sequence letters.\nLine 3 begins with a ‘+’ character and is optionally followed by the same sequence identifier (and any description) again.\nLine 4 encodes the quality values for the sequence in Line 2, and must contain the same number of symbols as letters in the sequence.\n\nA fastq file containing a single sequence might look like this:\n@SEQ_ID\nGATTTGGGGTTCAAAGCAGTATCGATCAAATAGTAAATCCATTTGTTCAACTCACAGTTT\n+\n!''*((((***+))%%%++)(%%%%).1***-+*''))**55CCF>>>>>>CCCCCCC65\nThe character ‘!’ represents the lowest quality while ‘~’ is the highest. Here are the quality value characters in left-to-right increasing order of quality (ASCII):\n!\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~\nThe original Sanger FASTQ files also allowed the sequence and quality strings to be wrapped (split over multiple lines), but this is generally discouraged as it can make parsing complicated due to the unfortunate choice of “@” and “+” as markers (these characters can also occur in the quality string).\n\nSequence quality scores\nThe Phred quality value Q is an integer mapping of p (i.e., the probability that the corresponding base call is incorrect). Two different equations have been in use. The first is the standard Sanger variant to assess reliability of a base call, otherwise known as Phred quality score:\n\\[\nQ_\\text{sanger} = -10 \\, \\log_{10} p\n\\]\nThe Solexa pipeline (i.e., the software delivered with the Illumina Genome Analyzer) earlier used a different mapping, encoding the odds \\(\\mathbf{p}/(1-\\mathbf{p})\\) instead of the probability \\(\\mathbf{p}\\):\n\\[\nQ_\\text{solexa-prior to v.1.3} = -10 \\; \\log_{10} \\frac{p}{1-p}\n\\]\nAlthough both mappings are asymptotically identical at higher quality values, they differ at lower quality levels (i.e., approximately \\(\\mathbf{p} > 0.05\\), or equivalently, \\(\\mathbf{Q} < 13\\)).\n\n\n\nFigure 2.1: Relationship between Q and p using the Sanger (red) and Solexa (black) equations (described above). The vertical dotted line indicates \\(\\mathbf{p}= 0.05\\), or equivalently, \\(Q = 13\\).\n\n\n\nEncoding\nThe fastq format had differente way of encoding the Phred quality score along the time. Here a breif history of these changes is presented.\n\nSanger format can encode a Phred quality score from 0 to 93 using ASCII 33 to 126 (although in raw read data the Phred quality score rarely exceeds 60, higher scores are possible in assemblies or read maps).\nSolexa/Illumina 1.0 format can encode a Solexa/Illumina quality score from -5 to 62 using ASCII 59 to 126 (although in raw read data Solexa scores from -5 to 40 only are expected)\nStarting with Illumina 1.3 and before Illumina 1.8, the format encoded a Phred quality score from 0 to 62 using ASCII 64 to 126 (although in raw read data Phred scores from 0 to 40 only are expected).\nStarting in Illumina 1.5 and before Illumina 1.8, the Phred scores 0 to 2 have a slightly different meaning. The values 0 and 1 are no longer used and the value 2, encoded by ASCII 66 “B”.\n\n\nSequencing Control Software, Version 2.6, (Catalog # SY-960-2601, Part # 15009921 Rev. A, November 2009, page 30) states the following: If a read ends with a segment of mostly low quality (Q15 or below), then all of the quality values in the segment are replaced with a value of 2 (encoded as the letter B in Illumina’s text-based encoding of quality scores)… This Q2 indicator does not predict a specific error rate, but rather indicates that a specific final portion of the read should not be used in further analyses. Also, the quality score encoded as “B” letter may occur internally within reads at least as late as pipeline version 1.6, as shown in the following example:\n\n@HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1\nTTAATTGGTAAATAAATCTCCTAATAGCTTAGATNTTACCTTNNNNNNNNNNTAGTTTCTTGAGATTTGTTGGGGGAGACATTTTTGTGATTGCCTTGAT\n+HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1\nefcfffffcfeefffcffffffddf`feed]`]_Ba_^__[YBBBBBBBBBBRTT\\]][]dddd`ddd^dddadd^BBBBBBBBBBBBBBBBBBBBBBBB\nAn alternative interpretation of this ASCII encoding has been proposed. Also, in Illumina runs using PhiX controls, the character ‘B’ was observed to represent an “unknown quality score”. The error rate of ‘B’ reads was roughly 3 phred scores lower the mean observed score of a given run.\n\nStarting in Illumina 1.8, the quality scores have basically returned to the use of the Sanger format (Phred+33).\n\nOBItools support the Sanger format. It is nevertheless to read files encoded following the Solexa/Illumina format, that are still possible to find in old files, by applying a shift of 62.\n\n\n\n\n2.1.4 File extension\nThere is no standard file extension for a FASTQ file, but .fq and .fastq, are commonly used.\n\n\n\n\nCock, Peter JA, Christopher J Fields, Naohisa Goto, Michael L Heuer, and Peter M Rice. 2010. “The Sanger FASTQ File Format for Sequences with Quality Scores, and the Solexa/Illumina FASTQ Variants.” Nucleic Acids Research 38 (6): 1767–71."
   },
   {
     "objectID": "tutorial.html#wolves-diet-based-on-dna-metabarcoding",
     "href": "tutorial.html#wolves-diet-based-on-dna-metabarcoding",
-    "title": "2  OBITools V4 Tutorial",
-    "section": "2.1 Wolves’ diet based on DNA metabarcoding",
-    "text": "2.1 Wolves’ diet based on DNA metabarcoding\nThe data used in this tutorial correspond to the analysis of four wolf scats, using the protocol published in Shehzad et al. (2012) for assessing carnivore diet. After extracting DNA from the faeces, the DNA amplifications were carried out using the primers TTAGATACCCCACTATGC and TAGAACAGGCTCCTCTAG amplifiying the 12S-V5 region (Riaz et al. 2011), together with a wolf blocking oligonucleotide.\nThe complete data set can be downloaded here: the tutorial dataset\nOnce the data file is downloaded, using a UNIX terminal unarchive the data from the tgz file.\n\ntar zxvf wolf_diet.tgz\n\nThat command create a new directory named wolf_data containing every required data files:\n\nfastq <fastq> files resulting of aGA IIx (Illumina) paired-end (2 x 108 bp) sequencing assay of DNA extracted and amplified from four wolf faeces:\n\nwolf_F.fastq\nwolf_R.fastq\n\nthe file describing the primers and tags used for all samples sequenced:\n\nwolf_diet_ngsfilter.txt The tags correspond to short and specific sequences added on the 5' end of each primer to distinguish the different samples\n\nthe file containing the reference database in a fasta format:\n\ndb_v05_r117.fasta This reference database has been extracted from the release 117 of EMBL using obipcr\n\n\n\n\n\nTo not mix raw data and processed data a new directory called results is created.\n\nmkdir results"
+    "title": "3  OBITools V4 Tutorial",
+    "section": "3.1 Wolves’ diet based on DNA metabarcoding",
+    "text": "3.1 Wolves’ diet based on DNA metabarcoding\nThe data used in this tutorial correspond to the analysis of four wolf scats, using the protocol published in Shehzad et al. (2012) for assessing carnivore diet. After extracting DNA from the faeces, the DNA amplifications were carried out using the primers TTAGATACCCCACTATGC and TAGAACAGGCTCCTCTAG amplifiying the 12S-V5 region (Riaz et al. 2011), together with a wolf blocking oligonucleotide.\nThe complete data set can be downloaded here: the tutorial dataset\nOnce the data file is downloaded, using a UNIX terminal unarchive the data from the tgz file.\n\ntar zxvf wolf_diet.tgz\n\nThat command create a new directory named wolf_data containing every required data files:\n\nfastq <fastq> files resulting of aGA IIx (Illumina) paired-end (2 x 108 bp) sequencing assay of DNA extracted and amplified from four wolf faeces:\n\nwolf_F.fastq\nwolf_R.fastq\n\nthe file describing the primers and tags used for all samples sequenced:\n\nwolf_diet_ngsfilter.txt The tags correspond to short and specific sequences added on the 5' end of each primer to distinguish the different samples\n\nthe file containing the reference database in a fasta format:\n\ndb_v05_r117.fasta This reference database has been extracted from the release 117 of EMBL using obipcr\n\n\n\n\n\nTo not mix raw data and processed data a new directory called results is created.\n\nmkdir results"
   },
   {
     "objectID": "tutorial.html#step-by-step-analysis",
     "href": "tutorial.html#step-by-step-analysis",
-    "title": "2  OBITools V4 Tutorial",
-    "section": "2.2 Step by step analysis",
-    "text": "2.2 Step by step analysis\n\n2.2.1 Recover full sequence reads from forward and reverse partial reads\nWhen using the result of a paired-end sequencing assay with supposedly overlapping forward and reverse reads, the first step is to recover the assembled sequence.\nThe forward and reverse reads of the same fragment are at the same line position in the two fastq files obtained after sequencing. Based on these two files, the assembly of the forward and reverse reads is done with the obipairing utility that aligns the two reads and returns the reconstructed sequence.\nIn our case, the command is:\n\nobipairing --min-identity=0.8 \\\n           --min-overlap=10 \\\n           -F wolf_data/wolf_F.fastq \\\n           -R wolf_data/wolf_R.fastq \\\n           > results/wolf.fastq \n\nThe --min-identity and --min-overlap options allow discarding sequences with low alignment quality. If after the aligment, the overlaping parts of the reads is shorter than 10 base pairs or the similarity over this aligned region is below 80% of identity, in the output file, the forward and reverse reads are not aligned but concatenated, and the value of the mode attribute in the sequence header is set to joined instead of alignment.\n\n\n2.2.2 Remove unaligned sequence records\nUnaligned sequences (:pymode=joined) cannot be used. The following command allows removing them from the dataset:\n\nobigrep -p 'annotations.mode != \"join\"' \\\n        results/wolf.fastq > results/wolf.ali.fastq\n\nThe -p requires a go like expression. annotations.mode != \"join\" means that if the value of the mode annotation of a sequence is different from join, the corresponding sequence record will be kept.\nThe first sequence record of wolf.ali.fastq can be obtained using the following command line:\n\nhead -n 4 results/wolf.ali.fastq\n\nThe folling piece of code appears on thew window of tour terminal.\n@HELIUM_000100422_612GNAAXX:7:108:5640:3823#0/1 {\"ali_dir\":\"left\",\"ali_length\":62,\"mode\":\"alignment\",\"pairing_mismatches\":{\"(T:26)->(G:13)\":62,\"(T:34)->(G:18)\":48},\"score\":484,\"score_norm\":0.968,\"seq_a_single\":46,\"seq_ab_match\":60,\"seq_b_single\":46}\nccgcctcctttagataccccactatgcttagccctaaacacaagtaattaatataacaaaattgttcgccagagtactaccggcaatagcttaaaactcaaaggacttggcggtgctttatacccttctagaggagcctgttctaaggaggcgg\n+\nCCCCCCCBCCCCCCCCCCCCCCCCCCCCCCBCCCCCBCCCCCCC<CcCccbe[`F`accXV<TA\\RYU\\\\ee_e[XZ[XEEEEEEEEEE?EEEEEEEEEEDEEEEEEECCCCCCCCCCCCCCCCCCCCCCCACCCCCACCCCCCCCCCCCCCCC\n\n\n2.2.3 Assign each sequence record to the corresponding sample/marker combination\nEach sequence record is assigned to its corresponding sample and marker using the data provided in a text file (here wolf_diet_ngsfilter.txt). This text file contains one line per sample, with the name of the experiment (several experiments can be included in the same file), the name of the tags (for example: aattaac if the same tag has been used on each extremity of the PCR products, or aattaac:gaagtag if the tags were different), the sequence of the forward primer, the sequence of the reverse primer, the letter T or F for sample identification using the forward primer and tag only or using both primers and both tags, respectively (see obimultiplex for details).\n\nobimultiplex -t wolf_data/wolf_diet_ngsfilter.txt \\\n             -u results/unidentified.fastq \\\n             results/wolf.ali.fastq \\\n             > results/wolf.ali.assigned.fastq\n\nThis command creates two files:\n\nunidentified.fastq containing all the sequence records that were not assigned to a sample/marker combination\nwolf.ali.assigned.fastq containing all the sequence records that were properly assigned to a sample/marker combination\n\nNote that each sequence record of the wolf.ali.assigned.fastq file contains only the barcode sequence as the sequences of primers and tags are removed by the obimultiplex program. Information concerning the experiment, sample, primers and tags is added as attributes in the sequence header.\nFor instance, the first sequence record of wolf.ali.assigned.fastq is:\n@HELIUM_000100422_612GNAAXX:7:108:5640:3823#0/1_sub[28..127] {\"ali_dir\":\"left\",\"ali_length\":62,\"direction\":\"direct\",\"experiment\":\"wolf_diet\",\"forward_match\":\"ttagataccccactatgc\",\"forward_mismatches\":0,\"forward_primer\":\"ttagataccccactatgc\",\"forward_tag\":\"gcctcct\",\"mode\":\"alignment\",\"pairing_mismatches\":{\"(T:26)->(G:13)\":35,\"(T:34)->(G:18)\":21},\"reverse_match\":\"tagaacaggctcctctag\",\"reverse_mismatches\":0,\"reverse_primer\":\"tagaacaggctcctctag\",\"reverse_tag\":\"gcctcct\",\"sample\":\"29a_F260619\",\"score\":484,\"score_norm\":0.968,\"seq_a_single\":46,\"seq_ab_match\":60,\"seq_b_single\":46}\nttagccctaaacacaagtaattaatataacaaaattgttcgccagagtactaccggcaatagcttaaaactcaaaggacttggcggtgctttataccctt\n+\nCCCBCCCCCBCCCCCCC<CcCccbe[`F`accXV<TA\\RYU\\\\ee_e[XZ[XEEEEEEEEEE?EEEEEEEEEEDEEEEEEECCCCCCCCCCCCCCCCCCC\n\n\n2.2.4 Dereplicate reads into uniq sequences\nThe same DNA molecule can be sequenced several times. In order to reduce both file size and computations time, and to get easier interpretable results, it is convenient to work with unique sequences instead of reads. To dereplicate such reads into unique sequences, we use the obiuniq command.\n\n\n\n\n\n\nDefinition: Dereplicate reads into unique sequences\n\n\n\ncompare all the reads in a data set to each other\ngroup strictly identical reads together\noutput the sequence for each group and its count in the original dataset (in this way, all duplicated reads are removed)\n\nDefinition adapted from Seguritan and Rohwer (2001)\n\n\n\nFor dereplication, we use the obiuniq command with the -m sample. The -m sample option is used to keep the information of the samples of origin for each uniquesequence.\n\nobiuniq -m sample \\\n        results/wolf.ali.assigned.fastq \\\n        > results/wolf.ali.assigned.uniq.fasta\n\nNote that obiuniq returns a fasta file.\nThe first sequence record of wolf.ali.assigned.uniq.fasta is:\n>HELIUM_000100422_612GNAAXX:7:93:6991:1942#0/1_sub[28..126] {\"ali_dir\":\"left\",\"ali_length\":63,\"count\":1,\"direction\":\"reverse\",\"experiment\":\"wolf_diet\",\"forward_match\":\"ttagataccccactatgc\",\"forward_mismatches\":0,\"forward_primer\":\"ttagataccccactatgc\",\"forward_tag\":\"gaatatc\",\"merged_sample\":{\"26a_F040644\":1},\"mode\":\"alignment\",\"pairing_mismatches\":{\"(A:10)->(G:34)\":76,\"(C:06)->(A:34)\":58},\"reverse_match\":\"tagaacaggctcctctag\",\"reverse_mismatches\":0,\"reverse_primer\":\"tagaacaggctcctctag\",\"reverse_tag\":\"gaatatc\",\"score\":730,\"score_norm\":0.968,\"seq_a_single\":45,\"seq_ab_match\":61,\"seq_b_single\":45}\nttagccctaaacataaacattcaataaacaagaatgttcgccagagaactactagcaaca\ngcctgaaactcaaaggacttggcggtgctttatatccct\nThe run of obiuniq has added two key=values entries in the header of the fasta sequence:\n\n\"merged_sample\":{\"29a_F260619\":1}: this sequence have been found once in a single sample called 29a_F260619\n\"count\":1 : the total count for this sequence is \\(1\\)\n\nTo keep only these two attributes, we can use the obiannotate command:\n\nobiannotate -k count -k merged_sample \\\n  results/wolf.ali.assigned.uniq.fasta \\\n  > results/wolf.ali.assigned.simple.fasta\n\nThe first five sequence records of wolf.ali.assigned.simple.fasta become:\n>HELIUM_000100422_612GNAAXX:7:26:18930:11105#0/1_sub[28..127] {\"count\":1,\"merged_sample\":{\"29a_F260619\":1}}\nttagccctaaacacaagtaattaatataacaaaatwattcgcyagagtactacmggcaat\nagctyaaarctcamagrwcttggcggtgctttataccctt\n>HELIUM_000100422_612GNAAXX:7:58:5711:11399#0/1_sub[28..127] {\"count\":1,\"merged_sample\":{\"29a_F260619\":1}}\nttagccctaaacacaagtaattaatataacaaaattattcgccagagtwctaccgssaat\nagcttaaaactcaaaggactgggcggtgctttataccctt\n>HELIUM_000100422_612GNAAXX:7:100:15836:9304#0/1_sub[28..127] {\"count\":1,\"merged_sample\":{\"29a_F260619\":1}}\nttagccctaaacatagataattacacaaacaaaattgttcaccagagtactagcggcaac\nagcttaaaactcaaaggacttggcggtgctttataccctt\n>HELIUM_000100422_612GNAAXX:7:55:13242:9085#0/1_sub[28..126] {\"count\":4,\"merged_sample\":{\"26a_F040644\":4}}\nttagccctaaacataaacattcaataaacaagagtgttcgccagagtactactagcaaca\ngcctgaaactcaaaggacttggcggtgctttacatccct\n>HELIUM_000100422_612GNAAXX:7:86:8429:13723#0/1_sub[28..127] {\"count\":7,\"merged_sample\":{\"15a_F730814\":5,\"29a_F260619\":2}}\nttagccctaaacacaagtaattaatataacaaaattattcgccagagtactaccggcaat\nagcttaaaactcaaaggactcggcggtgctttataccctt\n\n\n2.2.5 Denoise the sequence dataset\nTo have a set of sequences assigned to their corresponding samples does not mean that all sequences are biologically meaningful i.e. some of these sequences can contains PCR and/or sequencing errors, or chimeras.\n\nTag the sequences for PCR errors (sequence variants)\nThe obiclean program tags sequence variants as potential error generated during PCR amplification. We ask it to keep the head sequences (-H option) that are sequences which are not variants of another sequence with a count greater than 5% of their own count (-r 0.05 option).\n\nobiclean -s sample -r 0.05 -H \\\n  results/wolf.ali.assigned.simple.fasta \\\n      > results/wolf.ali.assigned.simple.clean.fasta \n\nOne of the sequence records of wolf.ali.assigned.simple.clean.fasta is:\n>HELIUM_000100422_612GNAAXX:7:66:4039:8016#0/1_sub[28..127] {\"count\":17,\"merged_sample\":{\"13a_F730603\":17},\"obiclean_head\":true,\"obiclean_headcount\":1,\"obiclean_internalcount\":0,\"obi\nclean_samplecount\":1,\"obiclean_singletoncount\":0,\"obiclean_status\":{\"13a_F730603\":\"h\"},\"obiclean_weight\":{\"13a_F730603\":25}}\nctagccttaaacacaaatagttatgcaaacaaaactattcgccagagtactaccggcaac\nagcccaaaactcaaaggacttggcggtgcttcacaccctt\nTo remove such sequences as much as possible, we first discard rare sequences and then rsequence variants that likely correspond to artifacts.\n\n\nGet some statistics about sequence counts\n\nobicount results/wolf.ali.assigned.simple.clean.fasta\n\ntime=\"2023-02-02T23:07:30+01:00\" level=info msg=\"Appending results/wolf.ali.assigned.simple.clean.fasta file\\n\"\n 2749 36409 273387\n\n\nThe dataset contains \\(4313\\) sequences variant corresponding to 42452 sequence reads. Most of the variants occur only a single time in the complete dataset and are usualy named singletons\n\nobigrep -p 'sequence.Count() == 1' results/wolf.ali.assigned.simple.clean.fasta \\\n    | obicount\n\ntime=\"2023-02-02T23:07:30+01:00\" level=info msg=\"Reading sequences from stdin in guessed\\n\"\ntime=\"2023-02-02T23:07:30+01:00\" level=info msg=\"Appending results/wolf.ali.assigned.simple.clean.fasta file\\n\"\ntime=\"2023-02-02T23:07:30+01:00\" level=info msg=\"On output use JSON headers\"\n 2309 2309 229912\n\n\nIn that dataset sigletons corresponds to \\(3511\\) variants.\nUsing R and the ROBIFastread package able to read headers of the fasta files produced by OBITools, we can get more complete statistics on the distribution of occurrencies.\n\nlibrary(ROBIFastread)\nlibrary(ggplot2)\n\nseqs <- read_obifasta(\"results/wolf.ali.assigned.simple.clean.fasta\",keys=\"count\")\n\nggplot(data = seqs,  mapping=aes(x = count)) +\n  geom_histogram(bins=100) +\n  scale_y_sqrt() +\n  scale_x_sqrt() +\n  geom_vline(xintercept = 10, col=\"red\", lty=2) +\n  xlab(\"number of occurrencies of a variant\") \n\n\n\n\nIn a similar way it is also possible to plot the distribution of the sequence length.\n\nggplot(data = seqs,  mapping=aes(x = nchar(sequence))) +\n  geom_histogram() +\n  scale_y_log10() +\n  geom_vline(xintercept = 80, col=\"red\", lty=2) +\n  xlab(\"sequence lengths in base pair\")\n\n\n\n\n\n\nKeep only the sequences having a count greater or equal to 10 and a length shorter than 80 bp\nBased on the previous observation, we set the cut-off for keeping sequences for further analysis to a count of 10. To do this, we use the obigrep <scripts/obigrep> command. The -p 'count>=10' option means that the python expression :pycount>=10 must be evaluated to :pyTrue for each sequence to be kept. Based on previous knowledge we also remove sequences with a length shorter than 80 bp (option -l) as we know that the amplified 12S-V5 barcode for vertebrates must have a length around 100bp.\n\nobigrep -l 80 -p 'sequence.Count() >= 10' results/wolf.ali.assigned.simple.clean.fasta \\\n    > results/wolf.ali.assigned.simple.clean.c10.l80.fasta\n\nThe first sequence record of results/wolf.ali.assigned.simple.clean.c10.l80.fasta is:\n>HELIUM_000100422_612GNAAXX:7:22:2603:18023#0/1_sub[28..127] {\"count\":12182,\"merged_sample\":{\"15a_F730814\":7559,\"29a_F260619\":4623},\"obiclean_head\":true,\"obiclean_headcount\":2,\"obiclean_internalcount\":0,\"obiclean_samplecount\":2,\"obiclean_singletoncount\":0,\"obiclean_status\":{\"15a_F730814\":\"h\",\"29a_F260619\":\"h\"},\"obiclean_weight\":{\"15a_F730814\":9165,\"29a_F260619\":6275}}\nttagccctaaacacaagtaattaatataacaaaattattcgccagagtactaccggcaat\nagcttaaaactcaaaggacttggcggtgctttataccctt\nAt that time in the data cleanning we have conserved :\n\nobicount results/wolf.ali.assigned.simple.clean.c10.l80.fasta\n\ntime=\"2023-02-02T23:07:31+01:00\" level=info msg=\"Appending results/wolf.ali.assigned.simple.clean.c10.l80.fasta file\\n\"\n 26 31337 2585\n\n\n\n\n\n2.2.6 Taxonomic assignment of sequences\nOnce denoising has been done, the next step in diet analysis is to assign the barcodes to the corresponding species in order to get the complete list of species associated to each sample.\nTaxonomic assignment of sequences requires a reference database compiling all possible species to be identified in the sample. Assignment is then done based on sequence comparison between sample sequences and reference sequences.\n\nDownload the taxonomy\nIt is always possible to download the complete taxonomy from NCBI using the following commands.\n\nmkdir TAXO\ncd TAXO\ncurl http://ftp.ncbi.nih.gov/pub/taxonomy/taxdump.tar.gz \\\n   | tar -zxvf -\ncd ..\n\nFor people have a low speed internet connection, a copy of the taxdump.tar.gz file is provided in the wolf_data directory. The NCBI taxonomy is dayly updated, but the one provided here is ok for running this tutorial.\nTo build the TAXO directory from the provided taxdump.tar.gz, you need to execute the following commands\n\nmkdir TAXO\ncd TAXO\ntar zxvf wolf_data/taxdump.tar.gz \ncd ..\n\n\n\nBuild a reference database\nOne way to build the reference database is to use the obipcr program to simulate a PCR and extract all sequences from a general purpose DNA database such as genbank or EMBL that can be amplified in silico by the two primers (here TTAGATACCCCACTATGC and TAGAACAGGCTCCTCTAG) used for PCR amplification.\nThe two steps to build this reference database would then be\n\nToday, the easiest database to download is Genbank. But this will take you more than a day and occupy more than half a terabyte on your hard drive. In the wolf_data directory, a shell script called download_gb.sh is provided to perform this task. It requires that the programs wget2 and curl are available on your computer.\nUse obipcr to simulate amplification and build a reference database based on the putatively amplified barcodes and their recorded taxonomic information.\n\nAs these steps can take a long time (about a day for the download and an hour for the PCR), we already provide the reference database produced by the following commands so you can skip its construction. Note that as the Genbank and taxonomic database evolve frequently, if you run the following commands you may get different results.\n\nDownload the sequences\n\nmkdir genbank\ncd genbank\n../wolf_data/install_gb.sh\ncd ..\n\nDO NOT RUN THIS COMMAND EXCEPT IF YOU ARE REALLY CONSIENT OF THE TIME AND DISK SPACE REQUIRED.\n\n\nUse obipcr to simulate an in silico` PCR\n\nobipcr -t TAXO -e 3 -l 50 -L 150 \\ \n       --forward TTAGATACCCCACTATGC \\\n       --reverse TAGAACAGGCTCCTCTAG \\\n       --no-order \\\n       genbank/Release-251/gb*.seq.gz\n       > results/v05.pcr.fasta\n\nNote that the primers must be in the same order both in wolf_diet_ngsfilter.txt and in the obipcr command. The part of the path indicating the Genbank release can change. Please check in your genbank directory the exact name of your release.\n\n\nClean the database\n\nfilter sequences so that they have a good taxonomic description at the species, genus, and family levels (obigrep command command below).\nremove redundant sequences (obiuniq command below).\nensure that the dereplicated sequences have a taxid at the family level (obigrep command below).\nensure that sequences each have a unique identification (obiannotate command below)\n\n\nobigrep -t TAXO \\\n          --require-rank species \\\n          --require-rank genus \\\n          --require-rank family \\\n          results/v05.ecopcr > results/v05_clean.fasta\n\nobiuniq -c taxid \\\n        results/v05_clean.fasta \\\n        > results/v05_clean_uniq.fasta\n\nobirefidx -t TAXO results/v05_clean_uniq.fasta \\\n        > results/v05_clean_uniq.indexed.fasta\n\n\n\nWarning\n\nFrom now on, for the sake of clarity, the following commands will use the filenames of the files provided with the tutorial. If you decided to run the last steps and use the files you have produced, you'll have to use results/v05_clean_uniq.indexed.fasta instead of wolf_data/db_v05_r117.indexed.fasta.\n\n\n\n\n\n2.2.7 Assign each sequence to a taxon\nOnce the reference database is built, taxonomic assignment can be carried out using the obitag command.\n\nobitag -t TAXO -R wolf_data/db_v05_r117.indexed.fasta \\\n       results/wolf.ali.assigned.simple.clean.c10.l80.fasta \\\n       > results/wolf.ali.assigned.simple.clean.c10.l80.taxo.fasta\n\nThe obitag adds several attributes in the sequence record header, among them:\n\nobitag_bestmatch=ACCESSION where ACCESSION is the id of hte sequence in the reference database that best aligns to the query sequence;\nobitag_bestid=FLOAT where FLOAT*100 is the percentage of identity between the best match sequence and the query sequence;\ntaxid=TAXID where TAXID is the final assignation of the sequence by obitag\nscientific_name=NAME where NAME is the scientific name of the assigned taxid.\n\nThe first sequence record of wolf.ali.assigned.simple.clean.c10.l80.taxo.fasta is:\n>HELIUM_000100422_612GNAAXX:7:81:18704:12346#0/1_sub[28..126] {\"count\":88,\"merged_sample\":{\"26a_F040644\":88},\"obiclean_head\":true,\"obiclean_headcount\":1,\"obiclean_internalcount\":0,\"obiclean_samplecount\":1,\"obiclean_singletoncount\":0,\"obiclean_status\":{\"26a_F040644\":\"h\"},\"obiclean_weight\":{\"26a_F040644\":208},\"obitag_bestid\":0.9207920792079208,\"obitag_bestmatch\":\"AY769263\",\"obitag_difference\":8,\"obitag_match_count\":1,\"obitag_rank\":\"clade\",\"scientific_name\":\"Boreoeutheria\",\"taxid\":1437010}\nttagccctaaacataaacattcaataaacaagaatgttcgccagaggactactagcaata\ngcttaaaactcaaaggacttggcggtgctttatatccct\n\n\n2.2.8 Generate the final result table\nSome unuseful attributes can be removed at this stage.\n\nobiclean_head\nobiclean_headcount\nobiclean_internalcount\nobiclean_samplecount\nobiclean_singletoncount\n\n\nobiannotate  --delete-tag=obiclean_head \\\n             --delete-tag=obiclean_headcount \\\n             --delete-tag=obiclean_internalcount \\\n             --delete-tag=obiclean_samplecount \\\n             --delete-tag=obiclean_singletoncount \\\n  results/wolf.ali.assigned.simple.clean.c10.l80.taxo.fasta \\\n  > results/wolf.ali.assigned.simple.clean.c10.l80.taxo.ann.fasta\n\nThe first sequence record of wolf.ali.assigned.simple.c10.l80.clean.taxo.ann.fasta is then:\n>HELIUM_000100422_612GNAAXX:7:84:16335:5083#0/1_sub[28..126] {\"count\":96,\"merged_sample\":{\"26a_F040644\":11,\"29a_F260619\":85},\"obiclean_status\":{\"26a_F040644\":\"s\",\"29a_F260619\":\"h\"},\"obiclean_weight\":{\"26a_F040644\":14,\"29a_F260619\":110},\"obitag_bestid\":0.9595959595959596,\"obitag_bestmatch\":\"AC187326\",\"obitag_difference\":4,\"obitag_match_count\":1,\"obitag_rank\":\"subspecies\",\"scientific_name\":\"Canis lupus familiaris\",\"taxid\":9615}\nttagccctaaacataagctattccataacaaaataattcgccagagaactactagcaaca\ngattaaacctcaaaggacttggcagtgctttatacccct\n\n\n2.2.9 Looking at the data in R\n\nlibrary(ROBIFastread)\nlibrary(vegan)\n\nLe chargement a nécessité le package : permute\n\n\nLe chargement a nécessité le package : lattice\n\n\nThis is vegan 2.6-4\n\nlibrary(magrittr)\n \n\ndiet_data <- read_obifasta(\"results/wolf.ali.assigned.simple.clean.c10.l80.taxo.fasta\") \ndiet_data %<>% extract_features(\"obitag_bestmatch\",\"obitag_rank\",\"scientific_name\",'taxid')\n\ndiet_tab <- extract_readcount(diet_data,key=\"obiclean_weight\")\ndiet_tab\n\n4 x 26 sparse Matrix of class \"dgCMatrix\"\n\n\n  [[ suppressing 26 column names 'HELIUM_000100422_612GNAAXX:7:30:17945:19531#0/1_sub[28..126]', 'HELIUM_000100422_612GNAAXX:7:94:16908:11285#0/1_sub[28..127]', 'HELIUM_000100422_612GNAAXX:7:100:4828:3492#0/1_sub[28..127]' ... ]]\n\n\n                                                                            \n26a_F040644 43    .  .  .   . 88   . 52 208 15 31  .    . 14 481 72 17  .  .\n13a_F730603  . 8409 22  1   .  .   .  .   .  .  . 20    .  .  19  .  . 15  .\n29a_F260619  .    .  . 13 353  . 391  .   .  .  .  . 6275  .   1  .  .  . 44\n15a_F730814  .    .  .  .   .  .   .  .   .  .  .  . 9165  .   5  .  .  .  .\n                                   \n26a_F040644 12830  14  . . 18  .  .\n13a_F730603     .   .  . 9  .  . 25\n29a_F260619     . 110 16 .  . 25  .\n15a_F730814     .   .  . 4  .  .  .\n\n\n\nThis file contains 26 sequences. You can deduce the diet of each sample:\n\n\n13a_F730603: Cervus elaphus\n15a_F730814: Capreolus capreolus\n26a_F040644: Marmota sp. (according to the location, it is Marmota marmota)\n29a_F260619: Capreolus capreolus\n\n\n\nNote that we also obtained a few wolf sequences although a wolf-blocking oligonucleotide was used.\n\n\n\n\nRiaz, Tiayyba, Wasim Shehzad, Alain Viari, François Pompanon, Pierre Taberlet, and Eric Coissac. 2011. “ecoPrimers: inference of new DNA barcode markers from whole genome sequence analysis.” Nucleic Acids Research 39 (21): e145. https://doi.org/10.1093/nar/gkr732.\n\n\nSeguritan, V, and F Rohwer. 2001. “FastGroup: a program to dereplicate libraries of 16S rDNA sequences.” BMC Bioinformatics 2 (October): 9. https://doi.org/10.1186/1471-2105-2-9.\n\n\nShehzad, Wasim, Tiayyba Riaz, Muhammad A Nawaz, Christian Miquel, Carole Poillot, Safdar A Shah, Francois Pompanon, Eric Coissac, and Pierre Taberlet. 2012. “Carnivore diet analysis based on next-generation sequencing: Application to the leopard cat (Prionailurus bengalensis) in Pakistan.” Molecular Ecology 21 (8): 1951–65. https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x."
+    "title": "3  OBITools V4 Tutorial",
+    "section": "3.2 Step by step analysis",
+    "text": "3.2 Step by step analysis\n\n3.2.1 Recover full sequence reads from forward and reverse partial reads\nWhen using the result of a paired-end sequencing assay with supposedly overlapping forward and reverse reads, the first step is to recover the assembled sequence.\nThe forward and reverse reads of the same fragment are at the same line position in the two fastq files obtained after sequencing. Based on these two files, the assembly of the forward and reverse reads is done with the obipairing utility that aligns the two reads and returns the reconstructed sequence.\nIn our case, the command is:\n\nobipairing --min-identity=0.8 \\\n           --min-overlap=10 \\\n           -F wolf_data/wolf_F.fastq \\\n           -R wolf_data/wolf_R.fastq \\\n           > results/wolf.fastq \n\nThe --min-identity and --min-overlap options allow discarding sequences with low alignment quality. If after the aligment, the overlaping parts of the reads is shorter than 10 base pairs or the similarity over this aligned region is below 80% of identity, in the output file, the forward and reverse reads are not aligned but concatenated, and the value of the mode attribute in the sequence header is set to joined instead of alignment.\n\n\n3.2.2 Remove unaligned sequence records\nUnaligned sequences (:pymode=joined) cannot be used. The following command allows removing them from the dataset:\n\nobigrep -p 'annotations.mode != \"join\"' \\\n        results/wolf.fastq > results/wolf.ali.fastq\n\nThe -p requires a go like expression. annotations.mode != \"join\" means that if the value of the mode annotation of a sequence is different from join, the corresponding sequence record will be kept.\nThe first sequence record of wolf.ali.fastq can be obtained using the following command line:\n\nhead -n 4 results/wolf.ali.fastq\n\nThe folling piece of code appears on thew window of tour terminal.\n@HELIUM_000100422_612GNAAXX:7:108:5640:3823#0/1 {\"ali_dir\":\"left\",\"ali_length\":62,\"mode\":\"alignment\",\"pairing_mismatches\":{\"(T:26)->(G:13)\":62,\"(T:34)->(G:18)\":48},\"score\":484,\"score_norm\":0.968,\"seq_a_single\":46,\"seq_ab_match\":60,\"seq_b_single\":46}\nccgcctcctttagataccccactatgcttagccctaaacacaagtaattaatataacaaaattgttcgccagagtactaccggcaatagcttaaaactcaaaggacttggcggtgctttatacccttctagaggagcctgttctaaggaggcgg\n+\nCCCCCCCBCCCCCCCCCCCCCCCCCCCCCCBCCCCCBCCCCCCC<CcCccbe[`F`accXV<TA\\RYU\\\\ee_e[XZ[XEEEEEEEEEE?EEEEEEEEEEDEEEEEEECCCCCCCCCCCCCCCCCCCCCCCACCCCCACCCCCCCCCCCCCCCC\n\n\n3.2.3 Assign each sequence record to the corresponding sample/marker combination\nEach sequence record is assigned to its corresponding sample and marker using the data provided in a text file (here wolf_diet_ngsfilter.txt). This text file contains one line per sample, with the name of the experiment (several experiments can be included in the same file), the name of the tags (for example: aattaac if the same tag has been used on each extremity of the PCR products, or aattaac:gaagtag if the tags were different), the sequence of the forward primer, the sequence of the reverse primer, the letter T or F for sample identification using the forward primer and tag only or using both primers and both tags, respectively (see obimultiplex for details).\n\nobimultiplex -t wolf_data/wolf_diet_ngsfilter.txt \\\n             -u results/unidentified.fastq \\\n             results/wolf.ali.fastq \\\n             > results/wolf.ali.assigned.fastq\n\nThis command creates two files:\n\nunidentified.fastq containing all the sequence records that were not assigned to a sample/marker combination\nwolf.ali.assigned.fastq containing all the sequence records that were properly assigned to a sample/marker combination\n\nNote that each sequence record of the wolf.ali.assigned.fastq file contains only the barcode sequence as the sequences of primers and tags are removed by the obimultiplex program. Information concerning the experiment, sample, primers and tags is added as attributes in the sequence header.\nFor instance, the first sequence record of wolf.ali.assigned.fastq is:\n@HELIUM_000100422_612GNAAXX:7:108:5640:3823#0/1_sub[28..127] {\"ali_dir\":\"left\",\"ali_length\":62,\"direction\":\"direct\",\"experiment\":\"wolf_diet\",\"forward_match\":\"ttagataccccactatgc\",\"forward_mismatches\":0,\"forward_primer\":\"ttagataccccactatgc\",\"forward_tag\":\"gcctcct\",\"mode\":\"alignment\",\"pairing_mismatches\":{\"(T:26)->(G:13)\":35,\"(T:34)->(G:18)\":21},\"reverse_match\":\"tagaacaggctcctctag\",\"reverse_mismatches\":0,\"reverse_primer\":\"tagaacaggctcctctag\",\"reverse_tag\":\"gcctcct\",\"sample\":\"29a_F260619\",\"score\":484,\"score_norm\":0.968,\"seq_a_single\":46,\"seq_ab_match\":60,\"seq_b_single\":46}\nttagccctaaacacaagtaattaatataacaaaattgttcgccagagtactaccggcaatagcttaaaactcaaaggacttggcggtgctttataccctt\n+\nCCCBCCCCCBCCCCCCC<CcCccbe[`F`accXV<TA\\RYU\\\\ee_e[XZ[XEEEEEEEEEE?EEEEEEEEEEDEEEEEEECCCCCCCCCCCCCCCCCCC\n\n\n3.2.4 Dereplicate reads into uniq sequences\nThe same DNA molecule can be sequenced several times. In order to reduce both file size and computations time, and to get easier interpretable results, it is convenient to work with unique sequences instead of reads. To dereplicate such reads into unique sequences, we use the obiuniq command.\n\n\n\n\n\n\nDefinition: Dereplicate reads into unique sequences\n\n\n\ncompare all the reads in a data set to each other\ngroup strictly identical reads together\noutput the sequence for each group and its count in the original dataset (in this way, all duplicated reads are removed)\n\nDefinition adapted from Seguritan and Rohwer (2001)\n\n\n\nFor dereplication, we use the obiuniq command with the -m sample. The -m sample option is used to keep the information of the samples of origin for each uniquesequence.\n\nobiuniq -m sample \\\n        results/wolf.ali.assigned.fastq \\\n        > results/wolf.ali.assigned.uniq.fasta\n\nNote that obiuniq returns a fasta file.\nThe first sequence record of wolf.ali.assigned.uniq.fasta is:\n>HELIUM_000100422_612GNAAXX:7:93:6991:1942#0/1_sub[28..126] {\"ali_dir\":\"left\",\"ali_length\":63,\"count\":1,\"direction\":\"reverse\",\"experiment\":\"wolf_diet\",\"forward_match\":\"ttagataccccactatgc\",\"forward_mismatches\":0,\"forward_primer\":\"ttagataccccactatgc\",\"forward_tag\":\"gaatatc\",\"merged_sample\":{\"26a_F040644\":1},\"mode\":\"alignment\",\"pairing_mismatches\":{\"(A:10)->(G:34)\":76,\"(C:06)->(A:34)\":58},\"reverse_match\":\"tagaacaggctcctctag\",\"reverse_mismatches\":0,\"reverse_primer\":\"tagaacaggctcctctag\",\"reverse_tag\":\"gaatatc\",\"score\":730,\"score_norm\":0.968,\"seq_a_single\":45,\"seq_ab_match\":61,\"seq_b_single\":45}\nttagccctaaacataaacattcaataaacaagaatgttcgccagagaactactagcaaca\ngcctgaaactcaaaggacttggcggtgctttatatccct\nThe run of obiuniq has added two key=values entries in the header of the fasta sequence:\n\n\"merged_sample\":{\"29a_F260619\":1}: this sequence have been found once in a single sample called 29a_F260619\n\"count\":1 : the total count for this sequence is \\(1\\)\n\nTo keep only these two attributes, we can use the obiannotate command:\n\nobiannotate -k count -k merged_sample \\\n  results/wolf.ali.assigned.uniq.fasta \\\n  > results/wolf.ali.assigned.simple.fasta\n\nThe first five sequence records of wolf.ali.assigned.simple.fasta become:\n>HELIUM_000100422_612GNAAXX:7:26:18930:11105#0/1_sub[28..127] {\"count\":1,\"merged_sample\":{\"29a_F260619\":1}}\nttagccctaaacacaagtaattaatataacaaaatwattcgcyagagtactacmggcaat\nagctyaaarctcamagrwcttggcggtgctttataccctt\n>HELIUM_000100422_612GNAAXX:7:58:5711:11399#0/1_sub[28..127] {\"count\":1,\"merged_sample\":{\"29a_F260619\":1}}\nttagccctaaacacaagtaattaatataacaaaattattcgccagagtwctaccgssaat\nagcttaaaactcaaaggactgggcggtgctttataccctt\n>HELIUM_000100422_612GNAAXX:7:100:15836:9304#0/1_sub[28..127] {\"count\":1,\"merged_sample\":{\"29a_F260619\":1}}\nttagccctaaacatagataattacacaaacaaaattgttcaccagagtactagcggcaac\nagcttaaaactcaaaggacttggcggtgctttataccctt\n>HELIUM_000100422_612GNAAXX:7:55:13242:9085#0/1_sub[28..126] {\"count\":4,\"merged_sample\":{\"26a_F040644\":4}}\nttagccctaaacataaacattcaataaacaagagtgttcgccagagtactactagcaaca\ngcctgaaactcaaaggacttggcggtgctttacatccct\n>HELIUM_000100422_612GNAAXX:7:86:8429:13723#0/1_sub[28..127] {\"count\":7,\"merged_sample\":{\"15a_F730814\":5,\"29a_F260619\":2}}\nttagccctaaacacaagtaattaatataacaaaattattcgccagagtactaccggcaat\nagcttaaaactcaaaggactcggcggtgctttataccctt\n\n\n3.2.5 Denoise the sequence dataset\nTo have a set of sequences assigned to their corresponding samples does not mean that all sequences are biologically meaningful i.e. some of these sequences can contains PCR and/or sequencing errors, or chimeras.\n\nTag the sequences for PCR errors (sequence variants)\nThe obiclean program tags sequence variants as potential error generated during PCR amplification. We ask it to keep the head sequences (-H option) that are sequences which are not variants of another sequence with a count greater than 5% of their own count (-r 0.05 option).\n\nobiclean -s sample -r 0.05 -H \\\n  results/wolf.ali.assigned.simple.fasta \\\n      > results/wolf.ali.assigned.simple.clean.fasta \n\nOne of the sequence records of wolf.ali.assigned.simple.clean.fasta is:\n>HELIUM_000100422_612GNAAXX:7:66:4039:8016#0/1_sub[28..127] {\"count\":17,\"merged_sample\":{\"13a_F730603\":17},\"obiclean_head\":true,\"obiclean_headcount\":1,\"obiclean_internalcount\":0,\"obi\nclean_samplecount\":1,\"obiclean_singletoncount\":0,\"obiclean_status\":{\"13a_F730603\":\"h\"},\"obiclean_weight\":{\"13a_F730603\":25}}\nctagccttaaacacaaatagttatgcaaacaaaactattcgccagagtactaccggcaac\nagcccaaaactcaaaggacttggcggtgcttcacaccctt\nTo remove such sequences as much as possible, we first discard rare sequences and then rsequence variants that likely correspond to artifacts.\n\n\nGet some statistics about sequence counts\n\nobicount results/wolf.ali.assigned.simple.clean.fasta\n\ntime=\"2023-02-03T11:50:36+01:00\" level=info msg=\"Appending results/wolf.ali.assigned.simple.clean.fasta file\\n\"\n 2749 36409 273387\n\n\nThe dataset contains \\(4313\\) sequences variant corresponding to 42452 sequence reads. Most of the variants occur only a single time in the complete dataset and are usualy named singletons\n\nobigrep -p 'sequence.Count() == 1' results/wolf.ali.assigned.simple.clean.fasta \\\n    | obicount\n\ntime=\"2023-02-03T11:50:36+01:00\" level=info msg=\"Reading sequences from stdin in guessed\\n\"\ntime=\"2023-02-03T11:50:36+01:00\" level=info msg=\"Appending results/wolf.ali.assigned.simple.clean.fasta file\\n\"\ntime=\"2023-02-03T11:50:36+01:00\" level=info msg=\"On output use JSON headers\"\n 2229 2229 221982\n\n\nIn that dataset sigletons corresponds to \\(3511\\) variants.\nUsing R and the ROBIFastread package able to read headers of the fasta files produced by OBITools, we can get more complete statistics on the distribution of occurrencies.\n\nlibrary(ROBIFastread)\nlibrary(ggplot2)\n\nseqs <- read_obifasta(\"results/wolf.ali.assigned.simple.clean.fasta\",keys=\"count\")\n\nggplot(data = seqs,  mapping=aes(x = count)) +\n  geom_histogram(bins=100) +\n  scale_y_sqrt() +\n  scale_x_sqrt() +\n  geom_vline(xintercept = 10, col=\"red\", lty=2) +\n  xlab(\"number of occurrencies of a variant\") \n\n\n\n\nIn a similar way it is also possible to plot the distribution of the sequence length.\n\nggplot(data = seqs,  mapping=aes(x = nchar(sequence))) +\n  geom_histogram() +\n  scale_y_log10() +\n  geom_vline(xintercept = 80, col=\"red\", lty=2) +\n  xlab(\"sequence lengths in base pair\")\n\n\n\n\n\n\nKeep only the sequences having a count greater or equal to 10 and a length shorter than 80 bp\nBased on the previous observation, we set the cut-off for keeping sequences for further analysis to a count of 10. To do this, we use the obigrep <scripts/obigrep> command. The -p 'count>=10' option means that the python expression :pycount>=10 must be evaluated to :pyTrue for each sequence to be kept. Based on previous knowledge we also remove sequences with a length shorter than 80 bp (option -l) as we know that the amplified 12S-V5 barcode for vertebrates must have a length around 100bp.\n\nobigrep -l 80 -p 'sequence.Count() >= 10' results/wolf.ali.assigned.simple.clean.fasta \\\n    > results/wolf.ali.assigned.simple.clean.c10.l80.fasta\n\nThe first sequence record of results/wolf.ali.assigned.simple.clean.c10.l80.fasta is:\n>HELIUM_000100422_612GNAAXX:7:22:2603:18023#0/1_sub[28..127] {\"count\":12182,\"merged_sample\":{\"15a_F730814\":7559,\"29a_F260619\":4623},\"obiclean_head\":true,\"obiclean_headcount\":2,\"obiclean_internalcount\":0,\"obiclean_samplecount\":2,\"obiclean_singletoncount\":0,\"obiclean_status\":{\"15a_F730814\":\"h\",\"29a_F260619\":\"h\"},\"obiclean_weight\":{\"15a_F730814\":9165,\"29a_F260619\":6275}}\nttagccctaaacacaagtaattaatataacaaaattattcgccagagtactaccggcaat\nagcttaaaactcaaaggacttggcggtgctttataccctt\nAt that time in the data cleanning we have conserved :\n\nobicount results/wolf.ali.assigned.simple.clean.c10.l80.fasta\n\ntime=\"2023-02-03T11:50:38+01:00\" level=info msg=\"Appending results/wolf.ali.assigned.simple.clean.c10.l80.fasta file\\n\"\n 26 31337 2585\n\n\n\n\n\n3.2.6 Taxonomic assignment of sequences\nOnce denoising has been done, the next step in diet analysis is to assign the barcodes to the corresponding species in order to get the complete list of species associated to each sample.\nTaxonomic assignment of sequences requires a reference database compiling all possible species to be identified in the sample. Assignment is then done based on sequence comparison between sample sequences and reference sequences.\n\nDownload the taxonomy\nIt is always possible to download the complete taxonomy from NCBI using the following commands.\n\nmkdir TAXO\ncd TAXO\ncurl http://ftp.ncbi.nih.gov/pub/taxonomy/taxdump.tar.gz \\\n   | tar -zxvf -\ncd ..\n\nFor people have a low speed internet connection, a copy of the taxdump.tar.gz file is provided in the wolf_data directory. The NCBI taxonomy is dayly updated, but the one provided here is ok for running this tutorial.\nTo build the TAXO directory from the provided taxdump.tar.gz, you need to execute the following commands\n\nmkdir TAXO\ncd TAXO\ntar zxvf wolf_data/taxdump.tar.gz \ncd ..\n\n\n\nBuild a reference database\nOne way to build the reference database is to use the obipcr program to simulate a PCR and extract all sequences from a general purpose DNA database such as genbank or EMBL that can be amplified in silico by the two primers (here TTAGATACCCCACTATGC and TAGAACAGGCTCCTCTAG) used for PCR amplification.\nThe two steps to build this reference database would then be\n\nToday, the easiest database to download is Genbank. But this will take you more than a day and occupy more than half a terabyte on your hard drive. In the wolf_data directory, a shell script called download_gb.sh is provided to perform this task. It requires that the programs wget2 and curl are available on your computer.\nUse obipcr to simulate amplification and build a reference database based on the putatively amplified barcodes and their recorded taxonomic information.\n\nAs these steps can take a long time (about a day for the download and an hour for the PCR), we already provide the reference database produced by the following commands so you can skip its construction. Note that as the Genbank and taxonomic database evolve frequently, if you run the following commands you may get different results.\n\nDownload the sequences\n\nmkdir genbank\ncd genbank\n../wolf_data/install_gb.sh\ncd ..\n\nDO NOT RUN THIS COMMAND EXCEPT IF YOU ARE REALLY CONSIENT OF THE TIME AND DISK SPACE REQUIRED.\n\n\nUse obipcr to simulate an in silico` PCR\n\nobipcr -t TAXO -e 3 -l 50 -L 150 \\ \n       --forward TTAGATACCCCACTATGC \\\n       --reverse TAGAACAGGCTCCTCTAG \\\n       --no-order \\\n       genbank/Release-251/gb*.seq.gz\n       > results/v05.pcr.fasta\n\nNote that the primers must be in the same order both in wolf_diet_ngsfilter.txt and in the obipcr command. The part of the path indicating the Genbank release can change. Please check in your genbank directory the exact name of your release.\n\n\nClean the database\n\nfilter sequences so that they have a good taxonomic description at the species, genus, and family levels (obigrep command command below).\nremove redundant sequences (obiuniq command below).\nensure that the dereplicated sequences have a taxid at the family level (obigrep command below).\nensure that sequences each have a unique identification (obiannotate command below)\n\n\nobigrep -t TAXO \\\n          --require-rank species \\\n          --require-rank genus \\\n          --require-rank family \\\n          results/v05.ecopcr > results/v05_clean.fasta\n\nobiuniq -c taxid \\\n        results/v05_clean.fasta \\\n        > results/v05_clean_uniq.fasta\n\nobirefidx -t TAXO results/v05_clean_uniq.fasta \\\n        > results/v05_clean_uniq.indexed.fasta\n\n\n\nWarning\n\nFrom now on, for the sake of clarity, the following commands will use the filenames of the files provided with the tutorial. If you decided to run the last steps and use the files you have produced, you'll have to use results/v05_clean_uniq.indexed.fasta instead of wolf_data/db_v05_r117.indexed.fasta.\n\n\n\n\n\n3.2.7 Assign each sequence to a taxon\nOnce the reference database is built, taxonomic assignment can be carried out using the obitag command.\n\nobitag -t TAXO -R wolf_data/db_v05_r117.indexed.fasta \\\n       results/wolf.ali.assigned.simple.clean.c10.l80.fasta \\\n       > results/wolf.ali.assigned.simple.clean.c10.l80.taxo.fasta\n\nThe obitag adds several attributes in the sequence record header, among them:\n\nobitag_bestmatch=ACCESSION where ACCESSION is the id of hte sequence in the reference database that best aligns to the query sequence;\nobitag_bestid=FLOAT where FLOAT*100 is the percentage of identity between the best match sequence and the query sequence;\ntaxid=TAXID where TAXID is the final assignation of the sequence by obitag\nscientific_name=NAME where NAME is the scientific name of the assigned taxid.\n\nThe first sequence record of wolf.ali.assigned.simple.clean.c10.l80.taxo.fasta is:\n>HELIUM_000100422_612GNAAXX:7:81:18704:12346#0/1_sub[28..126] {\"count\":88,\"merged_sample\":{\"26a_F040644\":88},\"obiclean_head\":true,\"obiclean_headcount\":1,\"obiclean_internalcount\":0,\"obiclean_samplecount\":1,\"obiclean_singletoncount\":0,\"obiclean_status\":{\"26a_F040644\":\"h\"},\"obiclean_weight\":{\"26a_F040644\":208},\"obitag_bestid\":0.9207920792079208,\"obitag_bestmatch\":\"AY769263\",\"obitag_difference\":8,\"obitag_match_count\":1,\"obitag_rank\":\"clade\",\"scientific_name\":\"Boreoeutheria\",\"taxid\":1437010}\nttagccctaaacataaacattcaataaacaagaatgttcgccagaggactactagcaata\ngcttaaaactcaaaggacttggcggtgctttatatccct\n\n\n3.2.8 Generate the final result table\nSome unuseful attributes can be removed at this stage.\n\nobiclean_head\nobiclean_headcount\nobiclean_internalcount\nobiclean_samplecount\nobiclean_singletoncount\n\n\nobiannotate  --delete-tag=obiclean_head \\\n             --delete-tag=obiclean_headcount \\\n             --delete-tag=obiclean_internalcount \\\n             --delete-tag=obiclean_samplecount \\\n             --delete-tag=obiclean_singletoncount \\\n  results/wolf.ali.assigned.simple.clean.c10.l80.taxo.fasta \\\n  > results/wolf.ali.assigned.simple.clean.c10.l80.taxo.ann.fasta\n\nThe first sequence record of wolf.ali.assigned.simple.c10.l80.clean.taxo.ann.fasta is then:\n>HELIUM_000100422_612GNAAXX:7:84:16335:5083#0/1_sub[28..126] {\"count\":96,\"merged_sample\":{\"26a_F040644\":11,\"29a_F260619\":85},\"obiclean_status\":{\"26a_F040644\":\"s\",\"29a_F260619\":\"h\"},\"obiclean_weight\":{\"26a_F040644\":14,\"29a_F260619\":110},\"obitag_bestid\":0.9595959595959596,\"obitag_bestmatch\":\"AC187326\",\"obitag_difference\":4,\"obitag_match_count\":1,\"obitag_rank\":\"subspecies\",\"scientific_name\":\"Canis lupus familiaris\",\"taxid\":9615}\nttagccctaaacataagctattccataacaaaataattcgccagagaactactagcaaca\ngattaaacctcaaaggacttggcagtgctttatacccct\n\n\n3.2.9 Looking at the data in R\n\nlibrary(ROBIFastread)\nlibrary(vegan)\n\nLe chargement a nécessité le package : permute\n\n\nLe chargement a nécessité le package : lattice\n\n\nThis is vegan 2.6-4\n\nlibrary(magrittr)\n \n\ndiet_data <- read_obifasta(\"results/wolf.ali.assigned.simple.clean.c10.l80.taxo.fasta\") \ndiet_data %<>% extract_features(\"obitag_bestmatch\",\"obitag_rank\",\"scientific_name\",'taxid')\n\ndiet_tab <- extract_readcount(diet_data,key=\"obiclean_weight\")\ndiet_tab\n\n4 x 26 sparse Matrix of class \"dgCMatrix\"\n\n\n  [[ suppressing 26 column names 'HELIUM_000100422_612GNAAXX:7:5:15939:5437#0/1_sub[28..126]', 'HELIUM_000100422_612GNAAXX:7:100:4828:3492#0/1_sub[28..127]', 'HELIUM_000100422_612GNAAXX:7:113:17236:15166#0/1_sub[28..126]' ... ]]\n\n\n                                                                             \n26a_F040644 12787  . 72   .  .  . 31 . 208 18 14 88 15  14   . 52 481    .  .\n13a_F730603     . 22  .   .  .  .  . 9   .  .  .  .  .   .   .  .  19    .  .\n29a_F260619     .  .  . 353 25 16  . .   .  .  .  .  . 110 391  .   1 6246 44\n15a_F730814     .  .  .   .  .  .  . 4   .  .  .  .  .   .   .  .   5 9165  .\n                                  \n26a_F040644  .  . 17  .  .    . 43\n13a_F730603  1 25  . 15 20 8409  .\n29a_F260619 13  .  .  .  .    .  .\n15a_F730814  .  .  .  .  .    .  .\n\n\n\nThis file contains 26 sequences. You can deduce the diet of each sample:\n\n\n13a_F730603: Cervus elaphus\n15a_F730814: Capreolus capreolus\n26a_F040644: Marmota sp. (according to the location, it is Marmota marmota)\n29a_F260619: Capreolus capreolus\n\n\n\nNote that we also obtained a few wolf sequences although a wolf-blocking oligonucleotide was used.\n\n\n\n\nRiaz, Tiayyba, Wasim Shehzad, Alain Viari, François Pompanon, Pierre Taberlet, and Eric Coissac. 2011. “ecoPrimers: inference of new DNA barcode markers from whole genome sequence analysis.” Nucleic Acids Research 39 (21): e145. https://doi.org/10.1093/nar/gkr732.\n\n\nSeguritan, V, and F Rohwer. 2001. “FastGroup: a program to dereplicate libraries of 16S rDNA sequences.” BMC Bioinformatics 2 (October): 9. https://doi.org/10.1186/1471-2105-2-9.\n\n\nShehzad, Wasim, Tiayyba Riaz, Muhammad A Nawaz, Christian Miquel, Carole Poillot, Safdar A Shah, Francois Pompanon, Eric Coissac, and Pierre Taberlet. 2012. “Carnivore diet analysis based on next-generation sequencing: Application to the leopard cat (Prionailurus bengalensis) in Pakistan.” Molecular Ecology 21 (8): 1951–65. https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x."
   },
   {
     "objectID": "commands.html#specifying-the-input-files-to-obitools-commands",
     "href": "commands.html#specifying-the-input-files-to-obitools-commands",
-    "title": "3  The OBITools V4 commands",
-    "section": "3.1 Specifying the input files to OBITools commands",
-    "text": "3.1 Specifying the input files to OBITools commands"
+    "title": "4  The OBITools V4 commands",
+    "section": "4.1 Specifying the input files to OBITools commands",
+    "text": "4.1 Specifying the input files to OBITools commands"
   },
   {
     "objectID": "commands.html#options-common-to-most-of-the-obitools-commands",
     "href": "commands.html#options-common-to-most-of-the-obitools-commands",
-    "title": "3  The OBITools V4 commands",
-    "section": "3.2 Options common to most of the OBITools commands",
-    "text": "3.2 Options common to most of the OBITools commands\n\n3.2.1 Specifying input format\nFive sequence formats are accepted for input files. Fasta and Fastq are the main ones, EMBL and Genbank allow the use of flat files produced by these two international databases. The last one, ecoPCR, is maintained for compatibility with previous OBITools and allows to read ecoPCR outputs as sequence files.\n\n--ecopcr : Read data following the ecoPCR output format.\n--embl Read data following the EMBL flatfile format.\n--genbank Read data following the Genbank flatfile format.\n\nSeveral encoding schemes have been proposed for quality scores in Fastq format. Currently, OBITools considers Sanger encoding as the standard. For reasons of compatibility with older datasets produced with Solexa sequencers, it is possible, by using the following option, to force the use of the corresponding quality encoding scheme when reading these older files.\n\n--solexa Decodes quality string according to the Solexa specification. (default: false)\n\n\n\n3.2.2 Specifying output format\nOnly two output sequence formats are supported by OBITools, Fasta and Fastq. Fastq is used when output sequences are associated with quality information. Otherwise, Fasta is the default format. However, it is possible to force the output format by using one of the following two options. Forcing the use of Fasta results in the loss of quality information. Conversely, when the Fastq format is forced with sequences that have no quality data, dummy qualities set to 40 for each nucleotide are added.\n\n--fasta-output Read data following the ecoPCR output format.\n--fastq-output Read data following the EMBL flatfile format.\n\nOBITools allows multiple input files to be specified for a single command.\n\n--no-order When several input files are provided, indicates that there is no order among them. (default: false)\n\n\n\n3.2.3 Format of the annotations in Fasta and Fastq files\nOBITools extend the Fasta and Fastq formats by introducing a format for the title lines of these formats allowing to annotate every sequence. While the previous version of OBITools used an ad-hoc format for these annotation, this new version introduce the usage of the standard JSON format to store them.\nOn input, OBITools automatically recognize the format of the annotations, but two options allows to force the parsing following one of them. You should normally not need to use these options.\n\n--input-OBI-header FASTA/FASTQ title line annotations follow OBI format. (default: false)\n--input-json-header FASTA/FASTQ title line annotations follow json format. (default: false)\n\nOn output, by default annotation are formatted using the new JSON format. For compatibility with previous version of OBITools and with external scripts and software, it is possible to force the usage of the previous OBITools format.\n\n--output-OBI-header|-O output FASTA/FASTQ title line annotations follow OBI format. (default: false)\n--output-json-header output FASTA/FASTQ title line annotations follow json format. (default: false)\n\n\n3.2.3.1 System related options\n\n--debug (default: false)\n--help\\|-h\\|-? (default: false)\n--max-cpu <int> Number of parallele threads computing the result (default: 10)\n--workers\\|-w <int> Number of parallele threads computing the result (default: 9)"
+    "title": "4  The OBITools V4 commands",
+    "section": "4.2 Options common to most of the OBITools commands",
+    "text": "4.2 Options common to most of the OBITools commands\n\n4.2.1 Specifying input format\nFive sequence formats are accepted for input files. Fasta (Section 2.1.2) and Fastq (Section 2.1.3) are the main ones, EMBL and Genbank allow the use of flat files produced by these two international databases. The last one, ecoPCR, is maintained for compatibility with previous OBITools and allows to read ecoPCR outputs as sequence files.\n\n--ecopcr : Read data following the ecoPCR output format.\n--embl Read data following the EMBL flatfile format.\n--genbank Read data following the Genbank flatfile format.\n\nSeveral encoding schemes have been proposed for quality scores in Fastq format. Currently, OBITools considers Sanger encoding as the standard. For reasons of compatibility with older datasets produced with Solexa sequencers, it is possible, by using the following option, to force the use of the corresponding quality encoding scheme when reading these older files.\n\n--solexa Decodes quality string according to the Solexa specification. (default: false)\n\n\n\n4.2.2 Specifying output format\nOnly two output sequence formats are supported by OBITools, Fasta and Fastq. Fastq is used when output sequences are associated with quality information. Otherwise, Fasta is the default format. However, it is possible to force the output format by using one of the following two options. Forcing the use of Fasta results in the loss of quality information. Conversely, when the Fastq format is forced with sequences that have no quality data, dummy qualities set to 40 for each nucleotide are added.\n\n--fasta-output Read data following the ecoPCR output format.\n--fastq-output Read data following the EMBL flatfile format.\n\nOBITools allows multiple input files to be specified for a single command.\n\n--no-order When several input files are provided, indicates that there is no order among them. (default: false). Using such option can increase a lot the processing of the data.\n\n\n\n4.2.3 The Fasta and Fastq annotations format\nOBITools extend the Fasta and Fastq formats by introducing a format for the title lines of these formats allowing to annotate every sequence. While the previous version of OBITools used an ad-hoc format for these annotation, this new version introduce the usage of the standard JSON format to store them.\nOn input, OBITools automatically recognize the format of the annotations, but two options allows to force the parsing following one of them. You should normally not need to use these options.\n\n--input-OBI-header FASTA/FASTQ title line annotations follow OBI format. (default: false)\n--input-json-header FASTA/FASTQ title line annotations follow json format. (default: false)\n\nOn output, by default annotation are formatted using the new JSON format. For compatibility with previous version of OBITools and with external scripts and software, it is possible to force the usage of the previous OBITools format.\n\n--output-OBI-header|-O output FASTA/FASTQ title line annotations follow OBI format. (default: false)\n--output-json-header output FASTA/FASTQ title line annotations follow json format. (default: false)\n\n\n4.2.3.1 System related options\n\n--debug (default: false)\n--help\\|-h\\|-? (default: false)\n--max-cpu <int> Number of parallele threads computing the result (default: 10)\n--workers\\|-w <int> Number of parallele threads computing the result (default: 9)"
   },
   {
     "objectID": "commands.html#obitools-expression-language",
     "href": "commands.html#obitools-expression-language",
-    "title": "3  The OBITools V4 commands",
-    "section": "3.3 OBITools expression language",
-    "text": "3.3 OBITools expression language\nSeveral OBITools (e.g. obigrep, obiannotate) allow the user to specify some simple expressions to compute values or define predicates. This expressions are parsed and evaluated using the gval go package, which allows for evaluating go-Like expression.\n\n3.3.1 Variables usable in the expression\n\n3.3.1.1 sequence\nsequence is the sequence object on which the expression is evaluated\n\n\n3.3.1.2 annotation\n\n\n\n3.3.2 Function defined in the language\n\n3.3.2.1 len\n\n\n3.3.2.2 ismap\n\n\n3.3.2.3 hasattribute\n\n\n3.3.2.4 min\n\n\n3.3.2.5 max\n\n\n\n3.3.3 Accessing to the sequence annotations"
+    "title": "4  The OBITools V4 commands",
+    "section": "4.3 OBITools expression language",
+    "text": "4.3 OBITools expression language\nSeveral OBITools (e.g. obigrep, obiannotate) allow the user to specify some simple expressions to compute values or define predicates. This expressions are parsed and evaluated using the gval go package, which allows for evaluating go-Like expression.\n\n4.3.1 Variables usable in the expression\n\n4.3.1.1 sequence\nsequence is the sequence object on which the expression is evaluated\n\n\n4.3.1.2 annotation\n\n\n\n4.3.2 Function defined in the language\n\n4.3.2.1 len\n\n\n4.3.2.2 ismap\n\n\n4.3.2.3 hasattribute\n\n\n4.3.2.4 min\n\n\n4.3.2.5 max\n\n\n\n4.3.3 Accessing to the sequence annotations"
   },
   {
     "objectID": "commands.html#metabarcode-design-and-quality-assessment",
     "href": "commands.html#metabarcode-design-and-quality-assessment",
-    "title": "3  The OBITools V4 commands",
-    "section": "3.4 Metabarcode design and quality assessment",
-    "text": "3.4 Metabarcode design and quality assessment\n\n3.4.0.1 obipcr\n\nReplace the ecoPCR original OBITools"
+    "title": "4  The OBITools V4 commands",
+    "section": "4.4 Metabarcode design and quality assessment",
+    "text": "4.4 Metabarcode design and quality assessment\n\n4.4.0.1 obipcr\n\nReplace the ecoPCR original OBITools"
   },
   {
     "objectID": "commands.html#file-format-conversions",
     "href": "commands.html#file-format-conversions",
-    "title": "3  The OBITools V4 commands",
-    "section": "3.5 File format conversions",
-    "text": "3.5 File format conversions\n\n3.5.0.1 obiconvert"
+    "title": "4  The OBITools V4 commands",
+    "section": "4.5 File format conversions",
+    "text": "4.5 File format conversions\n\n4.5.0.1 obiconvert"
   },
   {
     "objectID": "commands.html#sequence-annotations",
     "href": "commands.html#sequence-annotations",
-    "title": "3  The OBITools V4 commands",
-    "section": "3.6 Sequence annotations",
-    "text": "3.6 Sequence annotations\n\n3.6.0.1 obitag"
+    "title": "4  The OBITools V4 commands",
+    "section": "4.6 Sequence annotations",
+    "text": "4.6 Sequence annotations\n\n4.6.0.1 obitag"
   },
   {
     "objectID": "commands.html#computations-on-sequences",
     "href": "commands.html#computations-on-sequences",
-    "title": "3  The OBITools V4 commands",
-    "section": "3.7 Computations on sequences",
-    "text": "3.7 Computations on sequences\n\n3.7.1 obipairing\n\nReplace the illuminapairedends original OBITools\n\n\n3.7.1.1 Alignment procedure\nobipairing is introducing a new alignment algorithm compared to the illuminapairedend command of the OBITools V2. Nethertheless this new algorithm has been design to produce the same results than the previous, except in very few cases.\nThe new algorithm is a two-step procedure. First, a FASTN-type algorithm (Lipman and Pearson 1985) identifies the best offset between the two matched readings. This identifies the region of overlap.\nIn the second step, the matching regions of the two reads are extracted along with a flanking sequence of \\(\\Delta\\) base pairs. The two subsequences are then aligned using a “one side free end-gap” dynamic programming algorithm. This latter step is only called if at least one mismatch is detected by the FASTP step.\nUnless the similarity between the two reads at their overlap region is very low, the addition of the flanking regions in the second step of the alignment ensures the same alignment as if the dynamic programming alignment was performed on the full reads.\n\n\n3.7.1.2 The scoring system\nIn the dynamic programming step, the match and mismatch scores take into account the quality scores of the two aligned nucleotides. By taking these into account, the probability of a true match can be calculated for each aligned base pair.\nIf we consider a nucleotide read with a quality score \\(Q\\), the probability of misreading this base (\\(P_E\\)) is : \\[\nP_E = 10^{-\\frac{Q}{10}}\n\\]\nThus, when a given nucleotide \\(X\\) is observed with the quality score \\(Q\\). The probability that \\(X\\) is really an \\(X\\) is :\n\\[\nP(X=X) = 1 - P_E\n\\]\nOtherwise, \\(X\\) is actually one of the three other possible nucleotides (\\(X_{E1}\\), \\(X_{E2}\\) or \\(X_{E3}\\)). If we suppose that the three reading error have the same probability :\n\\[\nP(X=X_{E1}) = P(X=X_{E3}) = P(X=X_{E3}) = \\frac{P_E}{3}\n\\]\nAt each position in an alignment where the two nucleotides \\(X_1\\) and \\(X_2\\) face each other (not a gapped position), the probability of a true match varies depending on whether \\(X_1=X_2\\), an observed match, or \\(X_1 \\neq X_2\\), an observed mismatch.\nProbability of a true match when \\(X_1=X_2\\)\nThat probability can be divided in two parts. First \\(X_1\\) and \\(X_2\\) have been correctly read. The corresponding probability is :\n\\[\n\\begin{aligned}\nP_{TM} &= (1- PE_1)(1-PE_2)\\\\\n       &=(1 - 10^{-\\frac{Q_1}{10} } )(1 - 10^{-\\frac{Q_2}{10}} )\n\\end{aligned}\n\\]\nSecondly, a match can occure if the true nucleotides read as \\(X_1\\) and \\(X_2\\) are not \\(X_1\\) and \\(X_2\\) but identical.\n\\[\n\\begin{aligned}\nP(X_1==X_{E1}) \\cap P(X_2==X_{E1}) &= \\frac{P_{E1} P_{E2}}{9} \\\\\nP(X_1==X_{Ex}) \\cap P(X_2==X_{Ex}) & = \\frac{P_{E1} P_{E2}}{3}\n\\end{aligned}\n\\]\nThe probability of a true match between \\(X_1\\) and \\(X_2\\) when \\(X_1 = X_2\\) an observed match :\n\\[\n\\begin{aligned}\nP(MATCH | X_1 = X_2) = (1- PE_1)(1-PE_2) + \\frac{P_{E1} P_{E2}}{3}\n\\end{aligned}\n\\]\nProbability of a true match when \\(X_1 \\neq X_2\\)\nThat probability can be divided in three parts.\n\n\\(X_1\\) has been correctly read and \\(X_2\\) is a sequencing error and is actually equal to \\(X_1\\). \\[\nP_a =  (1-P_{E1})\\frac{P_{E2}}{3}\n\\]\n\\(X_2\\) has been correctly read and \\(X_1\\) is a sequencing error and is actually equal to \\(X_2\\). \\[\nP_b =  (1-P_{E2})\\frac{P_{E1}}{3}\n\\]\n\\(X_1\\) and \\(X_2\\) corresponds to sequencing error but are actually the same base \\(X_{Ex}\\) \\[\nP_c = 2\\frac{P_{E1} P_{E2}}{9}\n\\]\n\nConsequently : \\[\n\\begin{aligned}\nP(MATCH | X_1 \\neq X_2) =  (1-P_{E1})\\frac{P_{E2}}{3} +  (1-P_{E2})\\frac{P_{E1}}{3} + 2\\frac{P_{E1} P_{E2}}{9}\n\\end{aligned}\n\\]\nProbability of a match under the random model\n\n\n\n\n\nEvolution of the match and mismatch scores when the quality of base is 20 while the second range from 10 to 40.\n\n\n\n\n\n\n3.7.1.3 obimultiplex\n\nReplace the ngsfilter original OBITools\n\n\n\n3.7.1.4 obicomplement\n\n\n3.7.1.5 obiclean\n\n\n3.7.1.6 obiuniq"
+    "title": "4  The OBITools V4 commands",
+    "section": "4.7 Computations on sequences",
+    "text": "4.7 Computations on sequences\n\n4.7.1 obipairing\n\nReplace the illuminapairedends original OBITools\n\n\n4.7.1.1 Alignment procedure\nobipairing is introducing a new alignment algorithm compared to the illuminapairedend command of the OBITools V2. Nethertheless this new algorithm has been design to produce the same results than the previous, except in very few cases.\nThe new algorithm is a two-step procedure. First, a FASTN-type algorithm (Lipman and Pearson 1985) identifies the best offset between the two matched readings. This identifies the region of overlap.\nIn the second step, the matching regions of the two reads are extracted along with a flanking sequence of \\(\\Delta\\) base pairs. The two subsequences are then aligned using a “one side free end-gap” dynamic programming algorithm. This latter step is only called if at least one mismatch is detected by the FASTP step.\nUnless the similarity between the two reads at their overlap region is very low, the addition of the flanking regions in the second step of the alignment ensures the same alignment as if the dynamic programming alignment was performed on the full reads.\n\n\n4.7.1.2 The scoring system\nIn the dynamic programming step, the match and mismatch scores take into account the quality scores of the two aligned nucleotides. By taking these into account, the probability of a true match can be calculated for each aligned base pair.\nIf we consider a nucleotide read with a quality score \\(Q\\), the probability of misreading this base (\\(P_E\\)) is : \\[\nP_E = 10^{-\\frac{Q}{10}}\n\\]\nThus, when a given nucleotide \\(X\\) is observed with the quality score \\(Q\\). The probability that \\(X\\) is really an \\(X\\) is :\n\\[\nP(X=X) = 1 - P_E\n\\]\nOtherwise, \\(X\\) is actually one of the three other possible nucleotides (\\(X_{E1}\\), \\(X_{E2}\\) or \\(X_{E3}\\)). If we suppose that the three reading error have the same probability :\n\\[\nP(X=X_{E1}) = P(X=X_{E3}) = P(X=X_{E3}) = \\frac{P_E}{3}\n\\]\nAt each position in an alignment where the two nucleotides \\(X_1\\) and \\(X_2\\) face each other (not a gapped position), the probability of a true match varies depending on whether \\(X_1=X_2\\), an observed match, or \\(X_1 \\neq X_2\\), an observed mismatch.\nProbability of a true match when \\(X_1=X_2\\)\nThat probability can be divided in two parts. First \\(X_1\\) and \\(X_2\\) have been correctly read. The corresponding probability is :\n\\[\n\\begin{aligned}\nP_{TM} &= (1- PE_1)(1-PE_2)\\\\\n       &=(1 - 10^{-\\frac{Q_1}{10} } )(1 - 10^{-\\frac{Q_2}{10}} )\n\\end{aligned}\n\\]\nSecondly, a match can occure if the true nucleotides read as \\(X_1\\) and \\(X_2\\) are not \\(X_1\\) and \\(X_2\\) but identical.\n\\[\n\\begin{aligned}\nP(X_1==X_{E1}) \\cap P(X_2==X_{E1}) &= \\frac{P_{E1} P_{E2}}{9} \\\\\nP(X_1==X_{Ex}) \\cap P(X_2==X_{Ex}) & = \\frac{P_{E1} P_{E2}}{3}\n\\end{aligned}\n\\]\nThe probability of a true match between \\(X_1\\) and \\(X_2\\) when \\(X_1 = X_2\\) an observed match :\n\\[\n\\begin{aligned}\nP(MATCH | X_1 = X_2) = (1- PE_1)(1-PE_2) + \\frac{P_{E1} P_{E2}}{3}\n\\end{aligned}\n\\]\nProbability of a true match when \\(X_1 \\neq X_2\\)\nThat probability can be divided in three parts.\n\n\\(X_1\\) has been correctly read and \\(X_2\\) is a sequencing error and is actually equal to \\(X_1\\). \\[\nP_a =  (1-P_{E1})\\frac{P_{E2}}{3}\n\\]\n\\(X_2\\) has been correctly read and \\(X_1\\) is a sequencing error and is actually equal to \\(X_2\\). \\[\nP_b =  (1-P_{E2})\\frac{P_{E1}}{3}\n\\]\n\\(X_1\\) and \\(X_2\\) corresponds to sequencing error but are actually the same base \\(X_{Ex}\\) \\[\nP_c = 2\\frac{P_{E1} P_{E2}}{9}\n\\]\n\nConsequently : \\[\n\\begin{aligned}\nP(MATCH | X_1 \\neq X_2) =  (1-P_{E1})\\frac{P_{E2}}{3} +  (1-P_{E2})\\frac{P_{E1}}{3} + 2\\frac{P_{E1} P_{E2}}{9}\n\\end{aligned}\n\\]\nProbability of a match under the random model\n\n\n\n\n\nEvolution of the match and mismatch scores when the quality of base is 20 while the second range from 10 to 40.\n\n\n\n\n\n\n4.7.1.3 obimultiplex\n\nReplace the ngsfilter original OBITools\n\n\n\n4.7.1.4 obicomplement\n\n\n4.7.1.5 obiclean\n\n\n4.7.1.6 obiuniq"
   },
   {
     "objectID": "commands.html#sequence-sampling-and-filtering",
     "href": "commands.html#sequence-sampling-and-filtering",
-    "title": "3  The OBITools V4 commands",
-    "section": "3.8 Sequence sampling and filtering",
-    "text": "3.8 Sequence sampling and filtering\n\n3.8.0.1 obigrep\n\n\n3.8.1 Utilities\n\n3.8.1.1 obicount\n\n\n3.8.1.2 obidistribute\n\n\n3.8.1.3 obifind\n\nReplace the ecofind original OBITools.\n\n\n\n\n\nLipman, D J, and W R Pearson. 1985. “Rapid and sensitive protein similarity searches.” Science 227 (4693): 1435–41. http://www.ncbi.nlm.nih.gov/pubmed/2983426."
+    "title": "4  The OBITools V4 commands",
+    "section": "4.8 Sequence sampling and filtering",
+    "text": "4.8 Sequence sampling and filtering\n\n4.8.0.1 obigrep\n\n\n4.8.1 Utilities\n\n4.8.1.1 obicount\n\n\n4.8.1.2 obidistribute\n\n\n4.8.1.3 obifind\n\nReplace the ecofind original OBITools.\n\n\n\n\n\nLipman, D J, and W R Pearson. 1985. “Rapid and sensitive protein similarity searches.” Science 227 (4693): 1435–41. http://www.ncbi.nlm.nih.gov/pubmed/2983426."
   },
   {
     "objectID": "library.html#biosequence",
     "href": "library.html#biosequence",
-    "title": "4  The GO OBITools library",
-    "section": "4.1 BioSequence",
-    "text": "4.1 BioSequence\nThe BioSequence class is used to represent biological sequences. It allows for storing : - the sequence itself as a []byte - the sequencing quality score as a []byte if needed - an identifier as a string - a definition as a string - a set of (key, value) pairs in a map[sting]interface{}\nBioSequence is defined in the obiseq module and is included using the code\nimport (\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiseq\"\n)\n\n4.1.1 Creating new instances\nTo create new instance, use\n\nMakeBioSequence(id string, sequence []byte, definition string) obiseq.BioSequence\nNewBioSequence(id string, sequence []byte, definition string) *obiseq.BioSequence\n\nBoth create a BioSequence instance, but when the first one returns the instance, the second returns a pointer on the new instance. Two other functions MakeEmptyBioSequence, and NewEmptyBioSequence do the same job but provide an uninitialized objects.\n\nid parameters corresponds to the unique identifier of the sequence. It mist be a string constituted of a single word (not containing any space).\nsequence is the DNA sequence itself, provided as a byte array ([]byte).\ndefinition is a string, potentially empty, but usualy containing a sentence explaining what is that sequence.\n\nimport (\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiseq\"\n)\n\nfunc main() {\n    myseq := obiseq.NewBiosequence(\n        \"seq_GH0001\",\n        bytes.FromString(\"ACGTGTCAGTCG\"),\n        \"A short test sequence\",\n        )\n}\nWhen formated as fasta the parameters correspond to the following schema\n>id definition containing potentially several words\nsequence\n\n\n4.1.2 End of life of a BioSequence instance\nWhen an instance of BioSequence is no longer in use, it is normally taken over by the GO garbage collector. If you know that an instance will never be used again, you can, if you wish, call the Recycle method on it to store the allocated memory elements in a pool to limit the allocation effort when many sequences are being handled. Once the recycle method has been called on an instance, you must ensure that no other method is called on it.\n\n\n4.1.3 Accessing to the elements of a sequence\nThe different elements of an obiseq.BioSequence must be accessed using a set of methods. For the three main elements provided during the creation of a new instance methodes are :\n\nId() string\nSequence() []byte\nDefinition() string\n\nIt exists pending method to change the value of these elements\n\nSetId(id string)\nSetSequence(sequence []byte)\nSetDefinition(definition string)\n\nimport (\n    \"fmt\"\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiseq\"\n)\n\nfunc main() {\n    myseq := obiseq.NewBiosequence(\n        \"seq_GH0001\",\n        bytes.FromString(\"ACGTGTCAGTCG\"),\n        \"A short test sequence\",\n        )\n\n    fmt.Println(myseq.Id())\n    myseq.SetId(\"SPE01_0001\")\n    fmt.Println(myseq.Id())\n}\n\n4.1.3.1 Different ways for accessing an editing the sequence\nIf Sequence()and SetSequence(sequence []byte) methods are the basic ones, several other methods exist.\n\nString() string return the sequence directly converted to a string instance.\nThe Write method family allows for extending an existing sequence following the buffer protocol.\n\nWrite(data []byte) (int, error) allows for appending a byte array on 3’ end of the sequence.\nWriteString(data string) (int, error) allows for appending a string.\nWriteByte(data byte) error allows for appending a single byte.\n\n\nThe Clear method empties the sequence buffer.\nimport (\n    \"fmt\"\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiseq\"\n)\n\nfunc main() {\n    myseq := obiseq.NewEmptyBiosequence()\n\n    myseq.WriteString(\"accc\")\n    myseq.WriteByte(byte('c'))\n    fmt.Println(myseq.String())\n}\n\n\n4.1.3.2 Sequence quality scores\nSequence quality scores cannot be initialized at the time of instance creation. You must use dedicated methods to add quality scores to a sequence.\nTo be coherent the length of both the DNA sequence and que quality score sequence must be equal. But assessment of this constraint is realized. It is of the programmer responsability to check that invariant.\nWhile accessing to the quality scores relies on the method Quality() []byte, setting the quality need to call one of the following method. They run similarly to their sequence dedicated conterpart.\n\nSetQualities(qualities Quality)\nWriteQualities(data []byte) (int, error)\nWriteByteQualities(data byte) error\n\nIn a way analogous to the Clear method, ClearQualities() empties the sequence of quality scores.\n\n\n\n4.1.4 The annotations of a sequence\nA sequence can be annotated with attributes. Each attribute is associated with a value. An attribute is identified by its name. The name of an attribute consists of a character string containing no spaces or blank characters. Values can be of several types.\n\nScalar types:\n\ninteger\nnumeric\ncharacter\nboolean\n\nContainer types:\n\nvector\nmap\n\n\nVectors can contain any type of scalar. Maps are compulsorily indexed by strings and can contain any scalar type. It is not possible to have nested container type.\nAnnotations are stored in an object of type bioseq.Annotation which is an alias of map[string]interface{}. This map can be retrieved using the Annotations() Annotation method. If no annotation has been defined for this sequence, the method returns an empty map. It is possible to test an instance of BioSequence using its HasAnnotation() bool method to see if it has any annotations associated with it.\n\nGetAttribute(key string) (interface{}, bool)"
+    "title": "5  The GO OBITools library",
+    "section": "5.1 BioSequence",
+    "text": "5.1 BioSequence\nThe BioSequence class is used to represent biological sequences. It allows for storing : - the sequence itself as a []byte - the sequencing quality score as a []byte if needed - an identifier as a string - a definition as a string - a set of (key, value) pairs in a map[sting]interface{}\nBioSequence is defined in the obiseq module and is included using the code\nimport (\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiseq\"\n)\n\n5.1.1 Creating new instances\nTo create new instance, use\n\nMakeBioSequence(id string, sequence []byte, definition string) obiseq.BioSequence\nNewBioSequence(id string, sequence []byte, definition string) *obiseq.BioSequence\n\nBoth create a BioSequence instance, but when the first one returns the instance, the second returns a pointer on the new instance. Two other functions MakeEmptyBioSequence, and NewEmptyBioSequence do the same job but provide an uninitialized objects.\n\nid parameters corresponds to the unique identifier of the sequence. It mist be a string constituted of a single word (not containing any space).\nsequence is the DNA sequence itself, provided as a byte array ([]byte).\ndefinition is a string, potentially empty, but usualy containing a sentence explaining what is that sequence.\n\nimport (\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiseq\"\n)\n\nfunc main() {\n    myseq := obiseq.NewBiosequence(\n        \"seq_GH0001\",\n        bytes.FromString(\"ACGTGTCAGTCG\"),\n        \"A short test sequence\",\n        )\n}\nWhen formated as fasta the parameters correspond to the following schema\n>id definition containing potentially several words\nsequence\n\n\n5.1.2 End of life of a BioSequence instance\nWhen an instance of BioSequence is no longer in use, it is normally taken over by the GO garbage collector. If you know that an instance will never be used again, you can, if you wish, call the Recycle method on it to store the allocated memory elements in a pool to limit the allocation effort when many sequences are being handled. Once the recycle method has been called on an instance, you must ensure that no other method is called on it.\n\n\n5.1.3 Accessing to the elements of a sequence\nThe different elements of an obiseq.BioSequence must be accessed using a set of methods. For the three main elements provided during the creation of a new instance methodes are :\n\nId() string\nSequence() []byte\nDefinition() string\n\nIt exists pending method to change the value of these elements\n\nSetId(id string)\nSetSequence(sequence []byte)\nSetDefinition(definition string)\n\nimport (\n    \"fmt\"\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiseq\"\n)\n\nfunc main() {\n    myseq := obiseq.NewBiosequence(\n        \"seq_GH0001\",\n        bytes.FromString(\"ACGTGTCAGTCG\"),\n        \"A short test sequence\",\n        )\n\n    fmt.Println(myseq.Id())\n    myseq.SetId(\"SPE01_0001\")\n    fmt.Println(myseq.Id())\n}\n\n5.1.3.1 Different ways for accessing an editing the sequence\nIf Sequence()and SetSequence(sequence []byte) methods are the basic ones, several other methods exist.\n\nString() string return the sequence directly converted to a string instance.\nThe Write method family allows for extending an existing sequence following the buffer protocol.\n\nWrite(data []byte) (int, error) allows for appending a byte array on 3’ end of the sequence.\nWriteString(data string) (int, error) allows for appending a string.\nWriteByte(data byte) error allows for appending a single byte.\n\n\nThe Clear method empties the sequence buffer.\nimport (\n    \"fmt\"\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiseq\"\n)\n\nfunc main() {\n    myseq := obiseq.NewEmptyBiosequence()\n\n    myseq.WriteString(\"accc\")\n    myseq.WriteByte(byte('c'))\n    fmt.Println(myseq.String())\n}\n\n\n5.1.3.2 Sequence quality scores\nSequence quality scores cannot be initialized at the time of instance creation. You must use dedicated methods to add quality scores to a sequence.\nTo be coherent the length of both the DNA sequence and que quality score sequence must be equal. But assessment of this constraint is realized. It is of the programmer responsability to check that invariant.\nWhile accessing to the quality scores relies on the method Quality() []byte, setting the quality need to call one of the following method. They run similarly to their sequence dedicated conterpart.\n\nSetQualities(qualities Quality)\nWriteQualities(data []byte) (int, error)\nWriteByteQualities(data byte) error\n\nIn a way analogous to the Clear method, ClearQualities() empties the sequence of quality scores.\n\n\n\n5.1.4 The annotations of a sequence\nA sequence can be annotated with attributes. Each attribute is associated with a value. An attribute is identified by its name. The name of an attribute consists of a character string containing no spaces or blank characters. Values can be of several types.\n\nScalar types:\n\ninteger\nnumeric\ncharacter\nboolean\n\nContainer types:\n\nvector\nmap\n\n\nVectors can contain any type of scalar. Maps are compulsorily indexed by strings and can contain any scalar type. It is not possible to have nested container type.\nAnnotations are stored in an object of type bioseq.Annotation which is an alias of map[string]interface{}. This map can be retrieved using the Annotations() Annotation method. If no annotation has been defined for this sequence, the method returns an empty map. It is possible to test an instance of BioSequence using its HasAnnotation() bool method to see if it has any annotations associated with it.\n\nGetAttribute(key string) (interface{}, bool)"
   },
   {
     "objectID": "library.html#the-sequence-iterator",
     "href": "library.html#the-sequence-iterator",
-    "title": "4  The GO OBITools library",
-    "section": "4.2 The sequence iterator",
-    "text": "4.2 The sequence iterator\nThe pakage obiter provides an iterator mecanism for manipulating sequences. The main class provided by this package is obiiter.IBioSequence. An IBioSequence iterator provides batch of sequences.\n\n4.2.1 Basic usage of a sequence iterator\nMany functions, among them functions reading sequences from a text file, return a IBioSequence iterator. The iterator class provides two main methods:\n\nNext() bool\nGet() obiiter.BioSequenceBatch\n\nThe Next method moves the iterator to the next value, while the Get method returns the currently pointed value. Using them, it is possible to loop over the data as in the following code chunk.\nimport (\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiformats\"\n)\n\nfunc main() {\n    mydata := obiformats.ReadFastSeqFromFile(\"myfile.fasta\")\n       \n    for mydata.Next() {\n        data := mydata.Get()\n        //\n        // Whatever you want to do with the data chunk\n        //\n    }\n}\nAn obiseq.BioSequenceBatch instance is a set of sequences stored in an obiseq.BioSequenceSlice and a sequence number. The number of sequences in a batch is not defined. A batch can even contain zero sequences, if for example all sequences initially included in the batch have been filtered out at some stage of their processing.\n\n\n4.2.2 The Pipable functions\nA function consuming a obiiter.IBioSequence and returning a obiiter.IBioSequence is of class obiiter.Pipable.\n\n\n4.2.3 The Teeable functions\nA function consuming a obiiter.IBioSequence and returning two obiiter.IBioSequence instance is of class obiiter.Teeable."
+    "title": "5  The GO OBITools library",
+    "section": "5.2 The sequence iterator",
+    "text": "5.2 The sequence iterator\nThe pakage obiter provides an iterator mecanism for manipulating sequences. The main class provided by this package is obiiter.IBioSequence. An IBioSequence iterator provides batch of sequences.\n\n5.2.1 Basic usage of a sequence iterator\nMany functions, among them functions reading sequences from a text file, return a IBioSequence iterator. The iterator class provides two main methods:\n\nNext() bool\nGet() obiiter.BioSequenceBatch\n\nThe Next method moves the iterator to the next value, while the Get method returns the currently pointed value. Using them, it is possible to loop over the data as in the following code chunk.\nimport (\n    \"git.metabarcoding.org/lecasofts/go/obitools/pkg/obiformats\"\n)\n\nfunc main() {\n    mydata := obiformats.ReadFastSeqFromFile(\"myfile.fasta\")\n       \n    for mydata.Next() {\n        data := mydata.Get()\n        //\n        // Whatever you want to do with the data chunk\n        //\n    }\n}\nAn obiseq.BioSequenceBatch instance is a set of sequences stored in an obiseq.BioSequenceSlice and a sequence number. The number of sequences in a batch is not defined. A batch can even contain zero sequences, if for example all sequences initially included in the batch have been filtered out at some stage of their processing.\n\n\n5.2.2 The Pipable functions\nA function consuming a obiiter.IBioSequence and returning a obiiter.IBioSequence is of class obiiter.Pipable.\n\n\n5.2.3 The Teeable functions\nA function consuming a obiiter.IBioSequence and returning two obiiter.IBioSequence instance is of class obiiter.Teeable."
   },
   {
     "objectID": "annexes.html",
     "href": "annexes.html",
-    "title": "5  Annexes",
+    "title": "6  Annexes",
     "section": "",
-    "text": "5.0.1 Sequence attributes\n\n5.0.1.1 Reserved sequence attributes\n\n5.0.1.1.1 ali_dir\n\n5.0.1.1.1.1 Type : string\nThe attribute can contain 2 string values \"left\" or \"right\".\n\n\n5.0.1.1.1.2 Set by the obipairing tool\nThe alignment generated by obipairing is a 3’-end gap free algorithm. Two cases can occur when aligning the forward and reverse reads. If the barcode is long enough, both the reads overlap only on their 3’ ends. In such case, the alignment direction ali_dir is set to left. If the barcode is shorter than the read length, the paired reads overlap by their 5’ ends, and the complete barcode is sequenced by both the reads. In that later case, ali_dir is set to right.\n\n\n\n5.0.1.1.2 ali_length\n\n5.0.1.1.2.1 Set by the obipairing tool\nLength of the aligned parts when merging forward and reverse reads\n\n\n\n5.0.1.1.3 count : the number of sequence occurrences\n\n5.0.1.1.3.1 Set by the obiuniq tool\nThe count attribute indicates how-many strictly identical sequences have been merged in a single record. It contains an integer value. If it is absent this means that the sequence record represents a single occurrence of the sequence.\n\n\n5.0.1.1.3.2 Getter : method Count()\nThe Count() method allows to access to the count attribute as an integer value. If the count attribute is not defined for the given sequence, the value 1 is returned\n\n\n\n5.0.1.1.4 merged_*\n\n5.0.1.1.4.1 Type : map[string]int\n\n\n5.0.1.1.4.2 Set by the obiuniq tool\nThe -m option of the obiuniq tools allows for keeping track of the distribution of the values stored in given attribute of interest. Often this option is used to summarise distribution of a sequence variant accross samples when obiuniq is run after running obimultiplex. The actual name of the attribute depends on the name of the monitored attribute. If -m option is used with the attribute sample, then this attribute names merged_sample.\n\n\n\n5.0.1.1.5 mode\n\n5.0.1.1.5.1 Set by the obipairing tool\nobitag_ref_index\n\n\n5.0.1.1.5.2 Set by the obirefidx tool.\nIt resumes to which taxonomic annotation a match to that sequence must lead according to the number of differences existing between the query sequence and the reference sequence having that tag.\n\n\n5.0.1.1.5.3 Getter : method Count()\n\n\n\n5.0.1.1.6 pairing_mismatches\n\n5.0.1.1.6.1 Set by the obipairing tool\n\n\n\n5.0.1.1.7 score\n\n5.0.1.1.7.1 Set by the obipairing tool\n\n\n\n5.0.1.1.8 score_norm\n\n5.0.1.1.8.1 Set by the obipairing tool"
+    "text": "6.0.1 Sequence attributes\n\n6.0.1.1 Reserved sequence attributes\n\n6.0.1.1.1 ali_dir\n\n6.0.1.1.1.1 Type : string\nThe attribute can contain 2 string values \"left\" or \"right\".\n\n\n6.0.1.1.1.2 Set by the obipairing tool\nThe alignment generated by obipairing is a 3’-end gap free algorithm. Two cases can occur when aligning the forward and reverse reads. If the barcode is long enough, both the reads overlap only on their 3’ ends. In such case, the alignment direction ali_dir is set to left. If the barcode is shorter than the read length, the paired reads overlap by their 5’ ends, and the complete barcode is sequenced by both the reads. In that later case, ali_dir is set to right.\n\n\n\n6.0.1.1.2 ali_length\n\n6.0.1.1.2.1 Set by the obipairing tool\nLength of the aligned parts when merging forward and reverse reads\n\n\n\n6.0.1.1.3 count : the number of sequence occurrences\n\n6.0.1.1.3.1 Set by the obiuniq tool\nThe count attribute indicates how-many strictly identical sequences have been merged in a single record. It contains an integer value. If it is absent this means that the sequence record represents a single occurrence of the sequence.\n\n\n6.0.1.1.3.2 Getter : method Count()\nThe Count() method allows to access to the count attribute as an integer value. If the count attribute is not defined for the given sequence, the value 1 is returned\n\n\n\n6.0.1.1.4 merged_*\n\n6.0.1.1.4.1 Type : map[string]int\n\n\n6.0.1.1.4.2 Set by the obiuniq tool\nThe -m option of the obiuniq tools allows for keeping track of the distribution of the values stored in given attribute of interest. Often this option is used to summarise distribution of a sequence variant accross samples when obiuniq is run after running obimultiplex. The actual name of the attribute depends on the name of the monitored attribute. If -m option is used with the attribute sample, then this attribute names merged_sample.\n\n\n\n6.0.1.1.5 mode\n\n6.0.1.1.5.1 Set by the obipairing tool\nobitag_ref_index\n\n\n6.0.1.1.5.2 Set by the obirefidx tool.\nIt resumes to which taxonomic annotation a match to that sequence must lead according to the number of differences existing between the query sequence and the reference sequence having that tag.\n\n\n6.0.1.1.5.3 Getter : method Count()\n\n\n\n6.0.1.1.6 pairing_mismatches\n\n6.0.1.1.6.1 Set by the obipairing tool\n\n\n\n6.0.1.1.7 score\n\n6.0.1.1.7.1 Set by the obipairing tool\n\n\n\n6.0.1.1.8 score_norm\n\n6.0.1.1.8.1 Set by the obipairing tool"
   },
   {
     "objectID": "references.html",
     "href": "references.html",
     "title": "References",
     "section": "",
-    "text": "Boyer, Frédéric, Céline Mercier, Aurélie Bonin, Yvan Le Bras, Pierre\nTaberlet, and Eric Coissac. 2016. “obitools:\na unix-inspired software package for DNA metabarcoding.”\nMolecular Ecology Resources 16 (1): 176–82. https://doi.org/10.1111/1755-0998.12428.\n\n\nCock, Peter JA, Christopher J Fields, Naohisa Goto, Michael L Heuer, and\nPeter M Rice. 2010. “The Sanger FASTQ File Format for Sequences\nwith Quality Scores, and the Solexa/Illumina FASTQ Variants.”\nNucleic Acids Research 38 (6): 1767–71.\n\n\nLipman, D J, and W R Pearson. 1985. “Rapid\nand sensitive protein similarity searches.”\nScience 227 (4693): 1435–41. http://www.ncbi.nlm.nih.gov/pubmed/2983426.\n\n\nRiaz, Tiayyba, Wasim Shehzad, Alain Viari, François Pompanon, Pierre\nTaberlet, and Eric Coissac. 2011. “ecoPrimers: inference of new DNA barcode markers from\nwhole genome sequence analysis.” Nucleic Acids\nResearch 39 (21): e145. https://doi.org/10.1093/nar/gkr732.\n\n\nSeguritan, V, and F Rohwer. 2001. “FastGroup:\na program to dereplicate libraries of 16S rDNA sequences.”\nBMC Bioinformatics 2 (October): 9. https://doi.org/10.1186/1471-2105-2-9.\n\n\nShehzad, Wasim, Tiayyba Riaz, Muhammad A Nawaz, Christian Miquel, Carole\nPoillot, Safdar A Shah, Francois Pompanon, Eric Coissac, and Pierre\nTaberlet. 2012. “Carnivore diet analysis\nbased on next-generation sequencing: Application to the leopard cat\n(Prionailurus bengalensis) in Pakistan.” Molecular\nEcology 21 (8): 1951–65. https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x."
+    "text": "Andersen, Kenneth, Karen Lise Bird, Morten Rasmussen, James Haile,\nHenrik Breuning-Madsen, Kurt H Kjaer, Ludovic Orlando, M Thomas P\nGilbert, and Eske Willerslev. 2012. “Meta-barcoding of ëdirtı́DNA from soil reflects vertebrate\nbiodiversity.” Molecular Ecology 21 (8): 1966–79.\n\n\nBaldwin, Darren S, Matthew J Colloff, Gavin N Rees, Anthony A Chariton,\nGarth O Watson, Leon N Court, Diana M Hartley, et al. 2013. “Impacts of inundation and drought on eukaryote\nbiodiversity in semi-arid floodplain soils.” Molecular\nEcology 22 (6): 1746–58. https://doi.org/10.1111/mec.12190.\n\n\nBoyer, Frédéric, Céline Mercier, Aurélie Bonin, Yvan Le Bras, Pierre\nTaberlet, and Eric Coissac. 2016. “obitools:\na unix-inspired software package for DNA metabarcoding.”\nMolecular Ecology Resources 16 (1): 176–82. https://doi.org/10.1111/1755-0998.12428.\n\n\nCaporaso, J Gregory, Justin Kuczynski, Jesse Stombaugh, Kyle Bittinger,\nFrederic D Bushman, Elizabeth K Costello, Noah Fierer, et al. 2010.\n“QIIME allows analysis of high-throughput\ncommunity sequencing data.” Nature Methods 7 (5):\n335–36. https://doi.org/10.1038/nmeth.f.303.\n\n\nChariton, Anthony A, Anthony C Roach, Stuart L Simpson, and Graeme E\nBatley. 2010. “Influence of the choice of\nphysical and chemistry variables on interpreting patterns of sediment\ncontaminants and their relationships with estuarine macrobenthic\ncommunities.” Marine and Freshwater Research. https://doi.org/10.1071/mf09263.\n\n\nCock, Peter JA, Christopher J Fields, Naohisa Goto, Michael L Heuer, and\nPeter M Rice. 2010. “The Sanger FASTQ File Format for Sequences\nwith Quality Scores, and the Solexa/Illumina FASTQ Variants.”\nNucleic Acids Research 38 (6): 1767–71.\n\n\nDeagle, Bruce E, Roger Kirkwood, and Simon N Jarman. 2009. “Analysis of Australian fur seal diet by pyrosequencing\nprey DNA in faeces.” Molecular Ecology 18 (9):\n2022–38. https://doi.org/10.1111/j.1365-294X.2009.04158.x.\n\n\nKowalczyk, Rafał, Pierre Taberlet, Eric Coissac, Alice Valentini,\nChristian Miquel, Tomasz Kamiński, and Jan M Wójcik. 2011. “Influence of management practices on large herbivore\ndiet—Case of European bison in Białowieża Primeval Forest (Poland).”\nForest Ecology and Management 261 (4): 821–28. https://doi.org/10.1016/j.foreco.2010.11.026.\n\n\nLipman, D J, and W R Pearson. 1985. “Rapid\nand sensitive protein similarity searches.”\nScience 227 (4693): 1435–41. http://www.ncbi.nlm.nih.gov/pubmed/2983426.\n\n\nParducci, Laura, Tina Jørgensen, Mari Mette Tollefsrud, Ellen Elverland,\nTorbjørn Alm, Sonia L Fontana, K D Bennett, et al. 2012. “Glacial survival of boreal trees in northern\nScandinavia.” Science 335 (6072): 1083–86. https://doi.org/10.1126/science.1216043.\n\n\nRiaz, Tiayyba, Wasim Shehzad, Alain Viari, François Pompanon, Pierre\nTaberlet, and Eric Coissac. 2011. “ecoPrimers: inference of new DNA barcode markers from\nwhole genome sequence analysis.” Nucleic Acids\nResearch 39 (21): e145. https://doi.org/10.1093/nar/gkr732.\n\n\nSchloss, Patrick D, Sarah L Westcott, Thomas Ryabin, Justine R Hall,\nMartin Hartmann, Emily B Hollister, Ryan A Lesniewski, et al. 2009.\n“Introducing mothur: open-source,\nplatform-independent, community-supported software for describing and\ncomparing microbial communities.” Applied and\nEnvironmental Microbiology 75 (23): 7537–41. https://doi.org/10.1128/AEM.01541-09.\n\n\nSeguritan, V, and F Rohwer. 2001. “FastGroup:\na program to dereplicate libraries of 16S rDNA sequences.”\nBMC Bioinformatics 2 (October): 9. https://doi.org/10.1186/1471-2105-2-9.\n\n\nShehzad, Wasim, Tiayyba Riaz, Muhammad A Nawaz, Christian Miquel, Carole\nPoillot, Safdar A Shah, Francois Pompanon, Eric Coissac, and Pierre\nTaberlet. 2012. “Carnivore diet analysis\nbased on next-generation sequencing: Application to the leopard cat\n(Prionailurus bengalensis) in Pakistan.” Molecular\nEcology 21 (8): 1951–65. https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x.\n\n\nSogin, Mitchell L, Hilary G Morrison, Julie A Huber, David Mark Welch,\nSusan M Huse, Phillip R Neal, Jesus M Arrieta, and Gerhard J Herndl.\n2006. “Microbial diversity in the deep sea\nand the underexplored \"rare biosphere\".” Proceedings\nof the National Academy of Sciences of the United States of America\n103 (32): 12115–20. https://doi.org/10.1073/pnas.0605127103.\n\n\nSønstebø, J H, L Gielly, A K Brysting, R Elven, M Edwards, J Haile, E\nWillerslev, et al. 2010. “Using\nnext-generation sequencing for molecular reconstruction of past Arctic\nvegetation and climate.” Molecular Ecology\nResources 10 (6): 1009–18. https://doi.org/10.1111/j.1755-0998.2010.02855.x.\n\n\nTaberlet, Pierre, Eric Coissac, Mehrdad Hajibabaei, and Loren H\nRieseberg. 2012. “Environmental DNA.”\nMolecular Ecology 21 (8): 1789–93. https://doi.org/10.1111/j.1365-294X.2012.05542.x.\n\n\nThomsen, Philip Francis, Jos Kielgast, Lars L Iversen, Carsten Wiuf,\nMorten Rasmussen, M Thomas P Gilbert, Ludovic Orlando, and Eske\nWillerslev. 2012. “Monitoring endangered\nfreshwater biodiversity using environmental DNA.”\nMolecular Ecology 21 (11): 2565–73. https://doi.org/10.1111/j.1365-294X.2011.05418.x.\n\n\nValentini, Alice, Christian Miquel, Muhammad Ali Nawaz, Eva Bellemain,\nEric Coissac, François Pompanon, Ludovic Gielly, et al. 2009.\n“New perspectives in diet analysis based on\nDNA barcoding and parallel pyrosequencing: the trnL\napproach.” Molecular Ecology Resources 9 (1):\n51–60. https://doi.org/10.1111/j.1755-0998.2008.02352.x.\n\n\nYoccoz, N G, K A Bråthen, L Gielly, J Haile, M E Edwards, T Goslar, H\nVon Stedingk, et al. 2012. “DNA from soil\nmirrors plant taxonomic and growth form diversity.”\nMolecular Ecology 21 (15): 3647–55. https://doi.org/10.1111/j.1365-294X.2012.05545.x."
   }
 ]
\ No newline at end of file
diff --git a/doc/_book/tutorial.html b/doc/_book/tutorial.html
index 758392f..34538b8 100644
--- a/doc/_book/tutorial.html
+++ b/doc/_book/tutorial.html
@@ -7,7 +7,7 @@
 <meta name="viewport" content="width=device-width, initial-scale=1.0, user-scalable=yes">
 
 
-<title>OBITools V4 - 2&nbsp; OBITools V4 Tutorial</title>
+<title>OBITools V4 - 3&nbsp; OBITools V4 Tutorial</title>
 <style>
 code{white-space: pre-wrap;}
 span.smallcaps{font-variant: small-caps;}
@@ -113,7 +113,7 @@ div.csl-indent {
 <script src="site_libs/quarto-search/quarto-search.js"></script>
 <meta name="quarto:offset" content="./">
 <link href="./commands.html" rel="next">
-<link href="./intro.html" rel="prev">
+<link href="./formats.html" rel="prev">
 <script src="site_libs/quarto-html/quarto.js"></script>
 <script src="site_libs/quarto-html/popper.min.js"></script>
 <script src="site_libs/quarto-html/tippy.umd.min.js"></script>
@@ -153,7 +153,7 @@ div.csl-indent {
   <header id="quarto-header" class="headroom fixed-top">
   <nav class="quarto-secondary-nav" data-bs-toggle="collapse" data-bs-target="#quarto-sidebar" aria-controls="quarto-sidebar" aria-expanded="false" aria-label="Toggle sidebar navigation" onclick="if (window.quartoToggleHeadroom) { window.quartoToggleHeadroom(); }">
     <div class="container-fluid d-flex justify-content-between">
-      <h1 class="quarto-secondary-nav-title"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></h1>
+      <h1 class="quarto-secondary-nav-title"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></h1>
       <button type="button" class="quarto-btn-toggle btn" aria-label="Show secondary navigation">
         <i class="bi bi-chevron-right"></i>
       </button>
@@ -188,22 +188,27 @@ div.csl-indent {
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./tutorial.html" class="sidebar-item-text sidebar-link active"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
+  <a href="./formats.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
+  <a href="./tutorial.html" class="sidebar-item-text sidebar-link active"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  <a href="./commands.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></a>
   </div>
 </li>
         <li class="sidebar-item">
   <div class="sidebar-item-container"> 
-  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">Annexes</span></a>
+  <a href="./library.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">5</span>&nbsp; <span class="chapter-title">The GO <em>OBITools</em> library</span></a>
+  </div>
+</li>
+        <li class="sidebar-item">
+  <div class="sidebar-item-container"> 
+  <a href="./annexes.html" class="sidebar-item-text sidebar-link"><span class="chapter-number">6</span>&nbsp; <span class="chapter-title">Annexes</span></a>
   </div>
 </li>
         <li class="sidebar-item">
@@ -220,18 +225,18 @@ div.csl-indent {
     <h2 id="toc-title">Table of contents</h2>
    
   <ul>
-  <li><a href="#wolves-diet-based-on-dna-metabarcoding" id="toc-wolves-diet-based-on-dna-metabarcoding" class="nav-link active" data-scroll-target="#wolves-diet-based-on-dna-metabarcoding"><span class="toc-section-number">2.1</span>  Wolves’ diet based on DNA metabarcoding</a></li>
-  <li><a href="#step-by-step-analysis" id="toc-step-by-step-analysis" class="nav-link" data-scroll-target="#step-by-step-analysis"><span class="toc-section-number">2.2</span>  Step by step analysis</a>
+  <li><a href="#wolves-diet-based-on-dna-metabarcoding" id="toc-wolves-diet-based-on-dna-metabarcoding" class="nav-link active" data-scroll-target="#wolves-diet-based-on-dna-metabarcoding"><span class="toc-section-number">3.1</span>  Wolves’ diet based on DNA metabarcoding</a></li>
+  <li><a href="#step-by-step-analysis" id="toc-step-by-step-analysis" class="nav-link" data-scroll-target="#step-by-step-analysis"><span class="toc-section-number">3.2</span>  Step by step analysis</a>
   <ul class="collapse">
-  <li><a href="#recover-full-sequence-reads-from-forward-and-reverse-partial-reads" id="toc-recover-full-sequence-reads-from-forward-and-reverse-partial-reads" class="nav-link" data-scroll-target="#recover-full-sequence-reads-from-forward-and-reverse-partial-reads"><span class="toc-section-number">2.2.1</span>  Recover full sequence reads from forward and reverse partial reads</a></li>
-  <li><a href="#remove-unaligned-sequence-records" id="toc-remove-unaligned-sequence-records" class="nav-link" data-scroll-target="#remove-unaligned-sequence-records"><span class="toc-section-number">2.2.2</span>  Remove unaligned sequence records</a></li>
-  <li><a href="#assign-each-sequence-record-to-the-corresponding-samplemarker-combination" id="toc-assign-each-sequence-record-to-the-corresponding-samplemarker-combination" class="nav-link" data-scroll-target="#assign-each-sequence-record-to-the-corresponding-samplemarker-combination"><span class="toc-section-number">2.2.3</span>  Assign each sequence record to the corresponding sample/marker combination</a></li>
-  <li><a href="#dereplicate-reads-into-uniq-sequences" id="toc-dereplicate-reads-into-uniq-sequences" class="nav-link" data-scroll-target="#dereplicate-reads-into-uniq-sequences"><span class="toc-section-number">2.2.4</span>  Dereplicate reads into uniq sequences</a></li>
-  <li><a href="#denoise-the-sequence-dataset" id="toc-denoise-the-sequence-dataset" class="nav-link" data-scroll-target="#denoise-the-sequence-dataset"><span class="toc-section-number">2.2.5</span>  Denoise the sequence dataset</a></li>
-  <li><a href="#taxonomic-assignment-of-sequences" id="toc-taxonomic-assignment-of-sequences" class="nav-link" data-scroll-target="#taxonomic-assignment-of-sequences"><span class="toc-section-number">2.2.6</span>  Taxonomic assignment of sequences</a></li>
-  <li><a href="#assign-each-sequence-to-a-taxon" id="toc-assign-each-sequence-to-a-taxon" class="nav-link" data-scroll-target="#assign-each-sequence-to-a-taxon"><span class="toc-section-number">2.2.7</span>  Assign each sequence to a taxon</a></li>
-  <li><a href="#generate-the-final-result-table" id="toc-generate-the-final-result-table" class="nav-link" data-scroll-target="#generate-the-final-result-table"><span class="toc-section-number">2.2.8</span>  Generate the final result table</a></li>
-  <li><a href="#looking-at-the-data-in-r" id="toc-looking-at-the-data-in-r" class="nav-link" data-scroll-target="#looking-at-the-data-in-r"><span class="toc-section-number">2.2.9</span>  Looking at the data in R</a></li>
+  <li><a href="#recover-full-sequence-reads-from-forward-and-reverse-partial-reads" id="toc-recover-full-sequence-reads-from-forward-and-reverse-partial-reads" class="nav-link" data-scroll-target="#recover-full-sequence-reads-from-forward-and-reverse-partial-reads"><span class="toc-section-number">3.2.1</span>  Recover full sequence reads from forward and reverse partial reads</a></li>
+  <li><a href="#remove-unaligned-sequence-records" id="toc-remove-unaligned-sequence-records" class="nav-link" data-scroll-target="#remove-unaligned-sequence-records"><span class="toc-section-number">3.2.2</span>  Remove unaligned sequence records</a></li>
+  <li><a href="#assign-each-sequence-record-to-the-corresponding-samplemarker-combination" id="toc-assign-each-sequence-record-to-the-corresponding-samplemarker-combination" class="nav-link" data-scroll-target="#assign-each-sequence-record-to-the-corresponding-samplemarker-combination"><span class="toc-section-number">3.2.3</span>  Assign each sequence record to the corresponding sample/marker combination</a></li>
+  <li><a href="#dereplicate-reads-into-uniq-sequences" id="toc-dereplicate-reads-into-uniq-sequences" class="nav-link" data-scroll-target="#dereplicate-reads-into-uniq-sequences"><span class="toc-section-number">3.2.4</span>  Dereplicate reads into uniq sequences</a></li>
+  <li><a href="#denoise-the-sequence-dataset" id="toc-denoise-the-sequence-dataset" class="nav-link" data-scroll-target="#denoise-the-sequence-dataset"><span class="toc-section-number">3.2.5</span>  Denoise the sequence dataset</a></li>
+  <li><a href="#taxonomic-assignment-of-sequences" id="toc-taxonomic-assignment-of-sequences" class="nav-link" data-scroll-target="#taxonomic-assignment-of-sequences"><span class="toc-section-number">3.2.6</span>  Taxonomic assignment of sequences</a></li>
+  <li><a href="#assign-each-sequence-to-a-taxon" id="toc-assign-each-sequence-to-a-taxon" class="nav-link" data-scroll-target="#assign-each-sequence-to-a-taxon"><span class="toc-section-number">3.2.7</span>  Assign each sequence to a taxon</a></li>
+  <li><a href="#generate-the-final-result-table" id="toc-generate-the-final-result-table" class="nav-link" data-scroll-target="#generate-the-final-result-table"><span class="toc-section-number">3.2.8</span>  Generate the final result table</a></li>
+  <li><a href="#looking-at-the-data-in-r" id="toc-looking-at-the-data-in-r" class="nav-link" data-scroll-target="#looking-at-the-data-in-r"><span class="toc-section-number">3.2.9</span>  Looking at the data in R</a></li>
   </ul></li>
   </ul>
 </nav>
@@ -241,7 +246,7 @@ div.csl-indent {
 
 <header id="title-block-header" class="quarto-title-block default">
 <div class="quarto-title">
-<h1 class="title d-none d-lg-block"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></h1>
+<h1 class="title d-none d-lg-block"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">OBITools V4 Tutorial</span></h1>
 </div>
 
 
@@ -261,8 +266,8 @@ div.csl-indent {
 <li>the OBITools</li>
 <li>some basic Unix commands</li>
 </ul>
-<section id="wolves-diet-based-on-dna-metabarcoding" class="level2" data-number="2.1">
-<h2 data-number="2.1" class="anchored" data-anchor-id="wolves-diet-based-on-dna-metabarcoding"><span class="header-section-number">2.1</span> Wolves’ diet based on DNA metabarcoding</h2>
+<section id="wolves-diet-based-on-dna-metabarcoding" class="level2" data-number="3.1">
+<h2 data-number="3.1" class="anchored" data-anchor-id="wolves-diet-based-on-dna-metabarcoding"><span class="header-section-number">3.1</span> Wolves’ diet based on DNA metabarcoding</h2>
 <p>The data used in this tutorial correspond to the analysis of four wolf scats, using the protocol published in <span class="citation" data-cites="Shehzad2012-pn">Shehzad et al. (<a href="references.html#ref-Shehzad2012-pn" role="doc-biblioref">2012</a>)</span> for assessing carnivore diet. After extracting DNA from the faeces, the DNA amplifications were carried out using the primers <code>TTAGATACCCCACTATGC</code> and <code>TAGAACAGGCTCCTCTAG</code> amplifiying the <em>12S-V5</em> region <span class="citation" data-cites="Riaz2011-gn">(<a href="references.html#ref-Riaz2011-gn" role="doc-biblioref">Riaz et al. 2011</a>)</span>, together with a wolf blocking oligonucleotide.</p>
 <p>The complete data set can be downloaded here: <a href="wolf_diet.tgz">the tutorial dataset</a></p>
 <p>Once the data file is downloaded, using a UNIX terminal unarchive the data from the <code>tgz</code> file.</p>
@@ -293,10 +298,10 @@ div.csl-indent {
 <div class="sourceCode cell-code" id="cb2"><pre class="sourceCode bash code-with-copy"><code class="sourceCode bash"><span id="cb2-1"><a href="#cb2-1" aria-hidden="true" tabindex="-1"></a><span class="fu">mkdir</span> results</span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
 </div>
 </section>
-<section id="step-by-step-analysis" class="level2" data-number="2.2">
-<h2 data-number="2.2" class="anchored" data-anchor-id="step-by-step-analysis"><span class="header-section-number">2.2</span> Step by step analysis</h2>
-<section id="recover-full-sequence-reads-from-forward-and-reverse-partial-reads" class="level3" data-number="2.2.1">
-<h3 data-number="2.2.1" class="anchored" data-anchor-id="recover-full-sequence-reads-from-forward-and-reverse-partial-reads"><span class="header-section-number">2.2.1</span> Recover full sequence reads from forward and reverse partial reads</h3>
+<section id="step-by-step-analysis" class="level2" data-number="3.2">
+<h2 data-number="3.2" class="anchored" data-anchor-id="step-by-step-analysis"><span class="header-section-number">3.2</span> Step by step analysis</h2>
+<section id="recover-full-sequence-reads-from-forward-and-reverse-partial-reads" class="level3" data-number="3.2.1">
+<h3 data-number="3.2.1" class="anchored" data-anchor-id="recover-full-sequence-reads-from-forward-and-reverse-partial-reads"><span class="header-section-number">3.2.1</span> Recover full sequence reads from forward and reverse partial reads</h3>
 <p>When using the result of a paired-end sequencing assay with supposedly overlapping forward and reverse reads, the first step is to recover the assembled sequence.</p>
 <p>The forward and reverse reads of the same fragment are <em>at the same line position</em> in the two fastq files obtained after sequencing. Based on these two files, the assembly of the forward and reverse reads is done with the <code>obipairing</code> utility that aligns the two reads and returns the reconstructed sequence.</p>
 <p>In our case, the command is:</p>
@@ -309,8 +314,8 @@ div.csl-indent {
 </div>
 <p>The <code>--min-identity</code> and <code>--min-overlap</code> options allow discarding sequences with low alignment quality. If after the aligment, the overlaping parts of the reads is shorter than 10 base pairs or the similarity over this aligned region is below 80% of identity, in the output file, the forward and reverse reads are not aligned but concatenated, and the value of the <code>mode</code> attribute in the sequence header is set to <code>joined</code> instead of <code>alignment</code>.</p>
 </section>
-<section id="remove-unaligned-sequence-records" class="level3" data-number="2.2.2">
-<h3 data-number="2.2.2" class="anchored" data-anchor-id="remove-unaligned-sequence-records"><span class="header-section-number">2.2.2</span> Remove unaligned sequence records</h3>
+<section id="remove-unaligned-sequence-records" class="level3" data-number="3.2.2">
+<h3 data-number="3.2.2" class="anchored" data-anchor-id="remove-unaligned-sequence-records"><span class="header-section-number">3.2.2</span> Remove unaligned sequence records</h3>
 <p>Unaligned sequences (:py<code class="interpreted-text" role="mod">mode=joined</code>) cannot be used. The following command allows removing them from the dataset:</p>
 <div class="cell">
 <div class="sourceCode cell-code" id="cb4"><pre class="sourceCode bash code-with-copy"><code class="sourceCode bash"><span id="cb4-1"><a href="#cb4-1" aria-hidden="true" tabindex="-1"></a><span class="ex">obigrep</span> <span class="at">-p</span> <span class="st">'annotations.mode != "join"'</span> <span class="dt">\</span></span>
@@ -327,8 +332,8 @@ ccgcctcctttagataccccactatgcttagccctaaacacaagtaattaatataacaaaattgttcgccagagtactac
 +
 CCCCCCCBCCCCCCCCCCCCCCCCCCCCCCBCCCCCBCCCCCCC&lt;CcCccbe[`F`accXV&lt;TA\RYU\\ee_e[XZ[XEEEEEEEEEE?EEEEEEEEEEDEEEEEEECCCCCCCCCCCCCCCCCCCCCCCACCCCCACCCCCCCCCCCCCCCC</code></pre>
 </section>
-<section id="assign-each-sequence-record-to-the-corresponding-samplemarker-combination" class="level3" data-number="2.2.3">
-<h3 data-number="2.2.3" class="anchored" data-anchor-id="assign-each-sequence-record-to-the-corresponding-samplemarker-combination"><span class="header-section-number">2.2.3</span> Assign each sequence record to the corresponding sample/marker combination</h3>
+<section id="assign-each-sequence-record-to-the-corresponding-samplemarker-combination" class="level3" data-number="3.2.3">
+<h3 data-number="3.2.3" class="anchored" data-anchor-id="assign-each-sequence-record-to-the-corresponding-samplemarker-combination"><span class="header-section-number">3.2.3</span> Assign each sequence record to the corresponding sample/marker combination</h3>
 <p>Each sequence record is assigned to its corresponding sample and marker using the data provided in a text file (here <code>wolf_diet_ngsfilter.txt</code>). This text file contains one line per sample, with the name of the experiment (several experiments can be included in the same file), the name of the tags (for example: <code>aattaac</code> if the same tag has been used on each extremity of the PCR products, or <code>aattaac:gaagtag</code> if the tags were different), the sequence of the forward primer, the sequence of the reverse primer, the letter <code>T</code> or <code>F</code> for sample identification using the forward primer and tag only or using both primers and both tags, respectively (see <code>obimultiplex</code> for details).</p>
 <div class="cell">
 <div class="sourceCode cell-code" id="cb7"><pre class="sourceCode bash code-with-copy"><code class="sourceCode bash"><span id="cb7-1"><a href="#cb7-1" aria-hidden="true" tabindex="-1"></a><span class="ex">obimultiplex</span> <span class="at">-t</span> wolf_data/wolf_diet_ngsfilter.txt <span class="dt">\</span></span>
@@ -348,8 +353,8 @@ ttagccctaaacacaagtaattaatataacaaaattgttcgccagagtactaccggcaatagcttaaaactcaaaggact
 +
 CCCBCCCCCBCCCCCCC&lt;CcCccbe[`F`accXV&lt;TA\RYU\\ee_e[XZ[XEEEEEEEEEE?EEEEEEEEEEDEEEEEEECCCCCCCCCCCCCCCCCCC</code></pre>
 </section>
-<section id="dereplicate-reads-into-uniq-sequences" class="level3" data-number="2.2.4">
-<h3 data-number="2.2.4" class="anchored" data-anchor-id="dereplicate-reads-into-uniq-sequences"><span class="header-section-number">2.2.4</span> Dereplicate reads into uniq sequences</h3>
+<section id="dereplicate-reads-into-uniq-sequences" class="level3" data-number="3.2.4">
+<h3 data-number="3.2.4" class="anchored" data-anchor-id="dereplicate-reads-into-uniq-sequences"><span class="header-section-number">3.2.4</span> Dereplicate reads into uniq sequences</h3>
 <p>The same DNA molecule can be sequenced several times. In order to reduce both file size and computations time, and to get easier interpretable results, it is convenient to work with unique <em>sequences</em> instead of <em>reads</em>. To <em>dereplicate</em> such <em>reads</em> into unique <em>sequences</em>, we use the <code>obiuniq</code> command.</p>
 <table class="table">
 <colgroup>
@@ -408,8 +413,8 @@ gcctgaaactcaaaggacttggcggtgctttacatccct
 ttagccctaaacacaagtaattaatataacaaaattattcgccagagtactaccggcaat
 agcttaaaactcaaaggactcggcggtgctttataccctt</code></pre>
 </section>
-<section id="denoise-the-sequence-dataset" class="level3" data-number="2.2.5">
-<h3 data-number="2.2.5" class="anchored" data-anchor-id="denoise-the-sequence-dataset"><span class="header-section-number">2.2.5</span> Denoise the sequence dataset</h3>
+<section id="denoise-the-sequence-dataset" class="level3" data-number="3.2.5">
+<h3 data-number="3.2.5" class="anchored" data-anchor-id="denoise-the-sequence-dataset"><span class="header-section-number">3.2.5</span> Denoise the sequence dataset</h3>
 <p>To have a set of sequences assigned to their corresponding samples does not mean that all sequences are <em>biologically</em> meaningful i.e.&nbsp;some of these sequences can contains PCR and/or sequencing errors, or chimeras.</p>
 <section id="tag-the-sequences-for-pcr-errors-sequence-variants" class="level4 unnumbered">
 <h4 class="unnumbered anchored" data-anchor-id="tag-the-sequences-for-pcr-errors-sequence-variants">Tag the sequences for PCR errors (sequence variants)</h4>
@@ -431,7 +436,7 @@ agcccaaaactcaaaggacttggcggtgcttcacaccctt</code></pre>
 <div class="cell">
 <div class="sourceCode cell-code" id="cb15"><pre class="sourceCode bash code-with-copy"><code class="sourceCode bash"><span id="cb15-1"><a href="#cb15-1" aria-hidden="true" tabindex="-1"></a><span class="ex">obicount</span> results/wolf.ali.assigned.simple.clean.fasta</span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
 <div class="cell-output cell-output-stdout">
-<pre><code>time="2023-02-02T23:07:30+01:00" level=info msg="Appending results/wolf.ali.assigned.simple.clean.fasta file\n"
+<pre><code>time="2023-02-03T11:50:36+01:00" level=info msg="Appending results/wolf.ali.assigned.simple.clean.fasta file\n"
  2749 36409 273387</code></pre>
 </div>
 </div>
@@ -440,10 +445,10 @@ agcccaaaactcaaaggacttggcggtgcttcacaccctt</code></pre>
 <div class="sourceCode cell-code" id="cb17"><pre class="sourceCode bash code-with-copy"><code class="sourceCode bash"><span id="cb17-1"><a href="#cb17-1" aria-hidden="true" tabindex="-1"></a><span class="ex">obigrep</span> <span class="at">-p</span> <span class="st">'sequence.Count() == 1'</span> results/wolf.ali.assigned.simple.clean.fasta <span class="dt">\</span></span>
 <span id="cb17-2"><a href="#cb17-2" aria-hidden="true" tabindex="-1"></a>    <span class="kw">|</span> <span class="ex">obicount</span></span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
 <div class="cell-output cell-output-stdout">
-<pre><code>time="2023-02-02T23:07:30+01:00" level=info msg="Reading sequences from stdin in guessed\n"
-time="2023-02-02T23:07:30+01:00" level=info msg="Appending results/wolf.ali.assigned.simple.clean.fasta file\n"
-time="2023-02-02T23:07:30+01:00" level=info msg="On output use JSON headers"
- 2309 2309 229912</code></pre>
+<pre><code>time="2023-02-03T11:50:36+01:00" level=info msg="Reading sequences from stdin in guessed\n"
+time="2023-02-03T11:50:36+01:00" level=info msg="Appending results/wolf.ali.assigned.simple.clean.fasta file\n"
+time="2023-02-03T11:50:36+01:00" level=info msg="On output use JSON headers"
+ 2229 2229 221982</code></pre>
 </div>
 </div>
 <p>In that dataset sigletons corresponds to <span class="math inline">\(3511\)</span> variants.</p>
@@ -491,14 +496,14 @@ agcttaaaactcaaaggacttggcggtgctttataccctt</code></pre>
 <div class="cell">
 <div class="sourceCode cell-code" id="cb23"><pre class="sourceCode bash code-with-copy"><code class="sourceCode bash"><span id="cb23-1"><a href="#cb23-1" aria-hidden="true" tabindex="-1"></a><span class="ex">obicount</span> results/wolf.ali.assigned.simple.clean.c10.l80.fasta</span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
 <div class="cell-output cell-output-stdout">
-<pre><code>time="2023-02-02T23:07:31+01:00" level=info msg="Appending results/wolf.ali.assigned.simple.clean.c10.l80.fasta file\n"
+<pre><code>time="2023-02-03T11:50:38+01:00" level=info msg="Appending results/wolf.ali.assigned.simple.clean.c10.l80.fasta file\n"
  26 31337 2585</code></pre>
 </div>
 </div>
 </section>
 </section>
-<section id="taxonomic-assignment-of-sequences" class="level3" data-number="2.2.6">
-<h3 data-number="2.2.6" class="anchored" data-anchor-id="taxonomic-assignment-of-sequences"><span class="header-section-number">2.2.6</span> Taxonomic assignment of sequences</h3>
+<section id="taxonomic-assignment-of-sequences" class="level3" data-number="3.2.6">
+<h3 data-number="3.2.6" class="anchored" data-anchor-id="taxonomic-assignment-of-sequences"><span class="header-section-number">3.2.6</span> Taxonomic assignment of sequences</h3>
 <p>Once denoising has been done, the next step in diet analysis is to assign the barcodes to the corresponding species in order to get the complete list of species associated to each sample.</p>
 <p>Taxonomic assignment of sequences requires a reference database compiling all possible species to be identified in the sample. Assignment is then done based on sequence comparison between sample sequences and reference sequences.</p>
 <section id="download-the-taxonomy" class="level4 unnumbered">
@@ -582,8 +587,8 @@ agcttaaaactcaaaggacttggcggtgctttataccctt</code></pre>
 </section>
 </section>
 </section>
-<section id="assign-each-sequence-to-a-taxon" class="level3" data-number="2.2.7">
-<h3 data-number="2.2.7" class="anchored" data-anchor-id="assign-each-sequence-to-a-taxon"><span class="header-section-number">2.2.7</span> Assign each sequence to a taxon</h3>
+<section id="assign-each-sequence-to-a-taxon" class="level3" data-number="3.2.7">
+<h3 data-number="3.2.7" class="anchored" data-anchor-id="assign-each-sequence-to-a-taxon"><span class="header-section-number">3.2.7</span> Assign each sequence to a taxon</h3>
 <p>Once the reference database is built, taxonomic assignment can be carried out using the <code>obitag</code> command.</p>
 <div class="cell">
 <div class="sourceCode cell-code" id="cb30"><pre class="sourceCode bash code-with-copy"><code class="sourceCode bash"><span id="cb30-1"><a href="#cb30-1" aria-hidden="true" tabindex="-1"></a><span class="ex">obitag</span> <span class="at">-t</span> TAXO <span class="at">-R</span> wolf_data/db_v05_r117.indexed.fasta <span class="dt">\</span></span>
@@ -602,8 +607,8 @@ agcttaaaactcaaaggacttggcggtgctttataccctt</code></pre>
 <span id="cb31-2"><a href="#cb31-2" aria-hidden="true" tabindex="-1"></a><span class="ex">ttagccctaaacataaacattcaataaacaagaatgttcgccagaggactactagcaata</span></span>
 <span id="cb31-3"><a href="#cb31-3" aria-hidden="true" tabindex="-1"></a><span class="ex">gcttaaaactcaaaggacttggcggtgctttatatccct</span></span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
 </section>
-<section id="generate-the-final-result-table" class="level3" data-number="2.2.8">
-<h3 data-number="2.2.8" class="anchored" data-anchor-id="generate-the-final-result-table"><span class="header-section-number">2.2.8</span> Generate the final result table</h3>
+<section id="generate-the-final-result-table" class="level3" data-number="3.2.8">
+<h3 data-number="3.2.8" class="anchored" data-anchor-id="generate-the-final-result-table"><span class="header-section-number">3.2.8</span> Generate the final result table</h3>
 <p>Some unuseful attributes can be removed at this stage.</p>
 <ul>
 <li>obiclean_head</li>
@@ -626,8 +631,8 @@ agcttaaaactcaaaggacttggcggtgctttataccctt</code></pre>
 ttagccctaaacataagctattccataacaaaataattcgccagagaactactagcaaca
 gattaaacctcaaaggacttggcagtgctttatacccct</code></pre>
 </section>
-<section id="looking-at-the-data-in-r" class="level3" data-number="2.2.9">
-<h3 data-number="2.2.9" class="anchored" data-anchor-id="looking-at-the-data-in-r"><span class="header-section-number">2.2.9</span> Looking at the data in R</h3>
+<section id="looking-at-the-data-in-r" class="level3" data-number="3.2.9">
+<h3 data-number="3.2.9" class="anchored" data-anchor-id="looking-at-the-data-in-r"><span class="header-section-number">3.2.9</span> Looking at the data in R</h3>
 <div class="cell">
 <div class="sourceCode cell-code" id="cb34"><pre class="sourceCode r code-with-copy"><code class="sourceCode r"><span id="cb34-1"><a href="#cb34-1" aria-hidden="true" tabindex="-1"></a><span class="fu">library</span>(ROBIFastread)</span>
 <span id="cb34-2"><a href="#cb34-2" aria-hidden="true" tabindex="-1"></a><span class="fu">library</span>(vegan)</span></code><button title="Copy to Clipboard" class="code-copy-button"><i class="bi"></i></button></pre></div>
@@ -652,19 +657,19 @@ gattaaacctcaaaggacttggcagtgctttatacccct</code></pre>
 <pre><code>4 x 26 sparse Matrix of class "dgCMatrix"</code></pre>
 </div>
 <div class="cell-output cell-output-stderr">
-<pre><code>  [[ suppressing 26 column names 'HELIUM_000100422_612GNAAXX:7:30:17945:19531#0/1_sub[28..126]', 'HELIUM_000100422_612GNAAXX:7:94:16908:11285#0/1_sub[28..127]', 'HELIUM_000100422_612GNAAXX:7:100:4828:3492#0/1_sub[28..127]' ... ]]</code></pre>
+<pre><code>  [[ suppressing 26 column names 'HELIUM_000100422_612GNAAXX:7:5:15939:5437#0/1_sub[28..126]', 'HELIUM_000100422_612GNAAXX:7:100:4828:3492#0/1_sub[28..127]', 'HELIUM_000100422_612GNAAXX:7:113:17236:15166#0/1_sub[28..126]' ... ]]</code></pre>
 </div>
 <div class="cell-output cell-output-stdout">
-<pre><code>                                                                            
-26a_F040644 43    .  .  .   . 88   . 52 208 15 31  .    . 14 481 72 17  .  .
-13a_F730603  . 8409 22  1   .  .   .  .   .  .  . 20    .  .  19  .  . 15  .
-29a_F260619  .    .  . 13 353  . 391  .   .  .  .  . 6275  .   1  .  .  . 44
-15a_F730814  .    .  .  .   .  .   .  .   .  .  .  . 9165  .   5  .  .  .  .
-                                   
-26a_F040644 12830  14  . . 18  .  .
-13a_F730603     .   .  . 9  .  . 25
-29a_F260619     . 110 16 .  . 25  .
-15a_F730814     .   .  . 4  .  .  .</code></pre>
+<pre><code>                                                                             
+26a_F040644 12787  . 72   .  .  . 31 . 208 18 14 88 15  14   . 52 481    .  .
+13a_F730603     . 22  .   .  .  .  . 9   .  .  .  .  .   .   .  .  19    .  .
+29a_F260619     .  .  . 353 25 16  . .   .  .  .  .  . 110 391  .   1 6246 44
+15a_F730814     .  .  .   .  .  .  . 4   .  .  .  .  .   .   .  .   5 9165  .
+                                  
+26a_F040644  .  . 17  .  .    . 43
+13a_F730603  1 25  . 15 20 8409  .
+29a_F260619 13  .  .  .  .    .  .
+15a_F730814  .  .  .  .  .    .  .</code></pre>
 </div>
 </div>
 <dl>
@@ -831,13 +836,13 @@ window.document.addEventListener("DOMContentLoaded", function (event) {
 </script>
 <nav class="page-navigation">
   <div class="nav-page nav-page-previous">
-      <a href="./intro.html" class="pagination-link">
-        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">1</span>&nbsp; <span class="chapter-title">The OBITools</span></span>
+      <a href="./formats.html" class="pagination-link">
+        <i class="bi bi-arrow-left-short"></i> <span class="nav-page-text"><span class="chapter-number">2</span>&nbsp; <span class="chapter-title">File formats usable with <em>OBITools</em></span></span>
       </a>          
   </div>
   <div class="nav-page nav-page-next">
       <a href="./commands.html" class="pagination-link">
-        <span class="nav-page-text"><span class="chapter-number">3</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></span> <i class="bi bi-arrow-right-short"></i>
+        <span class="nav-page-text"><span class="chapter-number">4</span>&nbsp; <span class="chapter-title">The <em>OBITools V4</em> commands</span></span> <i class="bi bi-arrow-right-short"></i>
       </a>
   </div>
 </nav>
diff --git a/doc/_quarto.yml b/doc/_quarto.yml
index 6fe21ea..a81ddc9 100644
--- a/doc/_quarto.yml
+++ b/doc/_quarto.yml
@@ -10,6 +10,7 @@ book:
   chapters:
     - index.qmd
     - intro.qmd
+    - formats.qmd
     - tutorial.qmd
     - commands.qmd
     - library.qmd
@@ -26,6 +27,8 @@ format:
     theme: zephyr
   pdf:
     documentclass: scrreprt
+  epub: default
+
 
 
 
diff --git a/doc/book.bib b/doc/book.bib
index 1258c96..534bddc 100644
--- a/doc/book.bib
+++ b/doc/book.bib
@@ -178,3 +178,532 @@
   doi      = "10.1186/1471-2105-2-9",
   pmc      = "PMC59723"
 }
+
+
+@ARTICLE{Taberlet2012-pf,
+  title    = "{Environmental DNA}",
+  author   = "Taberlet, Pierre and Coissac, Eric and Hajibabaei, Mehrdad and
+              Rieseberg, Loren H",
+  journal  = "Molecular ecology",
+  volume   =  21,
+  number   =  8,
+  pages    = "1789--1793",
+  month    =  apr,
+  year     =  2012,
+  url      = "http://dx.doi.org/10.1111/j.1365-294X.2012.05542.x",
+  language = "en",
+  issn     = "0962-1083, 1365-294X",
+  pmid     = "22486819",
+  doi      = "10.1111/j.1365-294X.2012.05542.x"
+}
+
+
+@ARTICLE{Sogin2006-ab,
+  title    = "{Microbial diversity in the deep sea and the underexplored "rare
+              biosphere"}",
+  author   = "Sogin, Mitchell L and Morrison, Hilary G and Huber, Julie A and
+              Mark Welch, David and Huse, Susan M and Neal, Phillip R and
+              Arrieta, Jesus M and Herndl, Gerhard J",
+  abstract = "The evolution of marine microbes over billions of years predicts
+              that the composition of microbial communities should be much
+              greater than the published estimates of a few thousand distinct
+              kinds of microbes per liter of seawater. By adopting a massively
+              parallel tag sequencing strategy, we show that bacterial
+              communities of deep water masses of the North Atlantic and
+              diffuse flow hydrothermal vents are one to two orders of
+              magnitude more complex than previously reported for any microbial
+              environment. A relatively small number of different populations
+              dominate all samples, but thousands of low-abundance populations
+              account for most of the observed phylogenetic diversity. This
+              ``rare biosphere'' is very ancient and may represent a nearly
+              inexhaustible source of genomic innovation. Members of the rare
+              biosphere are highly divergent from each other and, at different
+              times in earth's history, may have had a profound impact on
+              shaping planetary processes.",
+  journal  = "Proceedings of the National Academy of Sciences of the United
+              States of America",
+  volume   =  103,
+  number   =  32,
+  pages    = "12115--12120",
+  month    =  aug,
+  year     =  2006,
+  url      = "http://dx.doi.org/10.1073/pnas.0605127103",
+  issn     = "0027-8424",
+  pmid     = "16880384",
+  doi      = "10.1073/pnas.0605127103",
+  pmc      = "PMC1524930"
+}
+
+
+@ARTICLE{Sonstebo2010-vv,
+  title    = "{Using next-generation sequencing for molecular reconstruction of
+              past Arctic vegetation and climate}",
+  author   = "S{\o}nsteb{\o}, J H and Gielly, L and Brysting, A K and Elven, R
+              and Edwards, M and Haile, J and Willerslev, E and Coissac, E and
+              Rioux, D and Sannier, J and Taberlet, P and Brochmann, C",
+  abstract = "Palaeoenvironments and former climates are typically inferred
+              from pollen and macrofossil records. This approach is
+              time-consuming and suffers from low taxonomic resolution and
+              biased taxon sampling. Here, we test an alternative DNA-based
+              approach utilizing the P6 loop in the chloroplast trnL (UAA)
+              intron; a short (13-158 bp) and variable region with highly
+              conserved flanking sequences. For taxonomic reference, a whole
+              trnL intron sequence database was constructed from recently
+              collected material of 842 species, representing all widespread
+              and/or ecologically important taxa of the species-poor arctic
+              flora. The P6 loop alone allowed identification of all families,
+              most genera (>75\%) and one-third of the species, thus providing
+              much higher taxonomic resolution than pollen records. The
+              suitability of the P6 loop for analysis of samples containing
+              degraded ancient DNA from a mixture of species is demonstrated by
+              high-throughput parallel pyrosequencing of permafrost-preserved
+              DNA and reconstruction of two plant communities from the last
+              glacial period. Our approach opens new possibilities for
+              DNA-based assessment of ancient as well as modern biodiversity of
+              many groups of organisms using environmental samples.",
+  journal  = "Molecular ecology resources",
+  volume   =  10,
+  number   =  6,
+  pages    = "1009--1018",
+  month    =  nov,
+  year     =  2010,
+  url      = "http://dx.doi.org/10.1111/j.1755-0998.2010.02855.x",
+  language = "en",
+  issn     = "1755-098X, 1755-0998",
+  pmid     = "21565110",
+  doi      = "10.1111/j.1755-0998.2010.02855.x"
+}
+
+@ARTICLE{Yoccoz2012-ix,
+  title    = "{DNA from soil mirrors plant taxonomic and growth form diversity}",
+  author   = "Yoccoz, N G and Br{\aa}then, K A and Gielly, L and Haile, J and
+              Edwards, M E and Goslar, T and Von Stedingk, H and Brysting, A K
+              and Coissac, E and Pompanon, F and S{\o}nsteb{\o}, J H and
+              Miquel, C and Valentini, A and De Bello, F and Chave, J and
+              Thuiller, W and Wincker, P and Cruaud, C and Gavory, F and
+              Rasmussen, M and Gilbert, M T P and Orlando, L and Brochmann, C
+              and Willerslev, E and Taberlet, P",
+  abstract = "Ecosystems across the globe are threatened by climate change and
+              human activities. New rapid survey approaches for monitoring
+              biodiversity would greatly advance assessment and understanding
+              of these threats. Taking advantage of next-generation DNA
+              sequencing, we tested an approach we call metabarcoding:
+              high-throughput and simultaneous taxa identification based on a
+              very short (usually <100 base pairs) but informative DNA
+              fragment. Short DNA fragments allow the use of degraded DNA from
+              environmental samples. All analyses included amplification using
+              plant-specific versatile primers, sequencing and estimation of
+              taxonomic diversity. We tested in three steps whether degraded
+              DNA from dead material in soil has the potential of efficiently
+              assessing biodiversity in different biomes. First, soil DNA from
+              eight boreal plant communities located in two different
+              vegetation types (meadow and heath) was amplified. Plant
+              diversity detected from boreal soil was highly consistent with
+              plant taxonomic and growth form diversity estimated from
+              conventional above-ground surveys. Second, we assessed DNA
+              persistence using samples from formerly cultivated soils in
+              temperate environments. We found that the number of crop DNA
+              sequences retrieved strongly varied with years since last
+              cultivation, and crop sequences were absent from nearby,
+              uncultivated plots. Third, we assessed the universal
+              applicability of DNA metabarcoding using soil samples from
+              tropical environments: a large proportion of species and families
+              from the study site were efficiently recovered. The results open
+              unprecedented opportunities for large-scale DNA-based
+              biodiversity studies across a range of taxonomic groups using
+              standardized metabarcoding approaches.",
+  journal  = "Molecular ecology",
+  volume   =  21,
+  number   =  15,
+  pages    = "3647--3655",
+  month    =  aug,
+  year     =  2012,
+  url      = "http://dx.doi.org/10.1111/j.1365-294X.2012.05545.x",
+  language = "en",
+  issn     = "0962-1083, 1365-294X",
+  pmid     = "22507540",
+  doi      = "10.1111/j.1365-294X.2012.05545.x"
+}
+
+
+@ARTICLE{Parducci2012-rn,
+  title    = "{Glacial survival of boreal trees in northern Scandinavia}",
+  author   = "Parducci, Laura and J{\o}rgensen, Tina and Tollefsrud, Mari Mette
+              and Elverland, Ellen and Alm, Torbj{\o}rn and Fontana, Sonia L
+              and Bennett, K D and Haile, James and Matetovici, Irina and
+              Suyama, Yoshihisa and Edwards, Mary E and Andersen, Kenneth and
+              Rasmussen, Morten and Boessenkool, Sanne and Coissac, Eric and
+              Brochmann, Christian and Taberlet, Pierre and Houmark-Nielsen,
+              Michael and Larsen, Nicolaj Krog and Orlando, Ludovic and
+              Gilbert, M Thomas P and Kj{\ae}r, Kurt H and Alsos, Inger Greve
+              and Willerslev, Eske",
+  abstract = "It is commonly believed that trees were absent in Scandinavia
+              during the last glaciation and first recolonized the Scandinavian
+              Peninsula with the retreat of its ice sheet some 9000 years ago.
+              Here, we show the presence of a rare mitochondrial DNA haplotype
+              of spruce that appears unique to Scandinavia and with its highest
+              frequency to the west-an area believed to sustain ice-free
+              refugia during most of the last ice age. We further show the
+              survival of DNA from this haplotype in lake sediments and pollen
+              of Tr{\o}ndelag in central Norway dating back ~10,300 years and
+              chloroplast DNA of pine and spruce in lake sediments adjacent to
+              the ice-free And{\o}ya refugium in northwestern Norway as early
+              as ~22,000 and 17,700 years ago, respectively. Our findings imply
+              that conifer trees survived in ice-free refugia of Scandinavia
+              during the last glaciation, challenging current views on survival
+              and spread of trees as a response to climate changes.",
+  journal  = "Science",
+  volume   =  335,
+  number   =  6072,
+  pages    = "1083--1086",
+  month    =  mar,
+  year     =  2012,
+  url      = "http://dx.doi.org/10.1126/science.1216043",
+  language = "en",
+  issn     = "0036-8075, 1095-9203",
+  pmid     = "22383845",
+  doi      = "10.1126/science.1216043"
+}
+
+
+@MISC{Chariton2010-cz,
+  title   = "{Influence of the choice of physical and chemistry variables on
+             interpreting patterns of sediment contaminants and their
+             relationships with estuarine macrobenthic communities}",
+  author  = "Chariton, Anthony A and Roach, Anthony C and Simpson, Stuart L and
+             Batley, Graeme E",
+  journal = "Marine and Freshwater Research",
+  volume  =  61,
+  number  =  10,
+  pages   = "1109",
+  year    =  2010,
+  url     = "http://dx.doi.org/10.1071/mf09263",
+  doi     = "10.1071/mf09263"
+}
+
+
+@ARTICLE{Baldwin2013-yc,
+  title     = "{Impacts of inundation and drought on eukaryote biodiversity in
+               semi-arid floodplain soils}",
+  author    = "Baldwin, Darren S and Colloff, Matthew J and Rees, Gavin N and
+               Chariton, Anthony A and Watson, Garth O and Court, Leon N and
+               Hartley, Diana M and Morgan, Matthew J and King, Andrew J and
+               Wilson, Jessica S and Hodda, Michael and Hardy, Christopher M",
+  abstract  = "Floodplain ecosystems are characterized by alternating wet and
+               dry phases and periodic inundation defines their ecological
+               character. Climate change, river regulation and the construction
+               of levees have substantially altered natural flooding and drying
+               regimes worldwide with uncertain effects on key biotic groups.
+               In southern Australia, we hypothesized that soil eukaryotic
+               communities in climate change affected areas of a semi-arid
+               floodplain would transition towards comprising mainly dry-soil
+               specialist species with increasing drought severity. Here, we
+               used 18S rRNA amplicon pyrosequencing to measure the eukaryote
+               community composition in soils that had been depleted of water
+               to varying degrees to confirm that reproducible transitional
+               changes occur in eukaryotic biodiversity on this floodplain.
+               Interflood community structures (3 years post-flood) were
+               dominated by persistent rather than either aquatic or
+               dry-specialist organisms. Only 2\% of taxa were unique to dry
+               locations by 8 years post-flood, and 10\% were restricted to wet
+               locations (inundated a year to 2 weeks post-flood). Almost half
+               (48\%) of the total soil biota were detected in both these
+               environments. The discovery of a large suite of organisms able
+               to survive nearly a decade of drought, and up to a year
+               submerged supports the concept of inherent resilience of
+               Australian semi-arid floodplain soil communities under
+               increasing pressure from climatic induced changes in water
+               availability.",
+  journal   = "Molecular ecology",
+  publisher = "Wiley Online Library",
+  volume    =  22,
+  number    =  6,
+  pages     = "1746--1758",
+  month     =  mar,
+  year      =  2013,
+  url       = "http://dx.doi.org/10.1111/mec.12190",
+  issn      = "0962-1083, 1365-294X",
+  pmid      = "23379967",
+  doi       = "10.1111/mec.12190"
+}
+
+
+@ARTICLE{Andersen2012-gj,
+  title     = "{Meta-barcoding of {\"e}dirt{\'\i}DNA from soil reflects
+               vertebrate biodiversity}",
+  author    = "Andersen, Kenneth and Bird, Karen Lise and Rasmussen, Morten and
+               Haile, James and Breuning-Madsen, Henrik and Kjaer, Kurt H and
+               Orlando, Ludovic and Gilbert, M Thomas P and Willerslev, Eske",
+  journal   = "Molecular ecology",
+  publisher = "Wiley Online Library",
+  volume    =  21,
+  number    =  8,
+  pages     = "1966--1979",
+  year      =  2012,
+  issn      = "0962-1083"
+}
+
+
+@ARTICLE{Thomsen2012-au,
+  title    = "{Monitoring endangered freshwater biodiversity using environmental
+              DNA}",
+  author   = "Thomsen, Philip Francis and Kielgast, Jos and Iversen, Lars L and
+              Wiuf, Carsten and Rasmussen, Morten and Gilbert, M Thomas P and
+              Orlando, Ludovic and Willerslev, Eske",
+  abstract = "Freshwater ecosystems are among the most endangered habitats on
+              Earth, with thousands of animal species known to be threatened or
+              already extinct. Reliable monitoring of threatened organisms is
+              crucial for data-driven conservation actions but remains a
+              challenge owing to nonstandardized methods that depend on
+              practical and taxonomic expertise, which is rapidly declining.
+              Here, we show that a diversity of rare and threatened freshwater
+              animals--representing amphibians, fish, mammals, insects and
+              crustaceans--can be detected and quantified based on DNA obtained
+              directly from small water samples of lakes, ponds and streams. We
+              successfully validate our findings in a controlled mesocosm
+              experiment and show that DNA becomes undetectable within 2 weeks
+              after removal of animals, indicating that DNA traces are near
+              contemporary with presence of the species. We further demonstrate
+              that entire faunas of amphibians and fish can be detected by
+              high-throughput sequencing of DNA extracted from pond water. Our
+              findings underpin the ubiquitous nature of DNA traces in the
+              environment and establish environmental DNA as a tool for
+              monitoring rare and threatened species across a wide range of
+              taxonomic groups.",
+  journal  = "Molecular ecology",
+  volume   =  21,
+  number   =  11,
+  pages    = "2565--2573",
+  month    =  jun,
+  year     =  2012,
+  url      = "http://dx.doi.org/10.1111/j.1365-294X.2011.05418.x",
+  issn     = "0962-1083, 1365-294X",
+  pmid     = "22151771",
+  doi      = "10.1111/j.1365-294X.2011.05418.x"
+}
+
+
+@ARTICLE{Valentini2009-ay,
+  title     = "{New perspectives in diet analysis based on DNA barcoding and
+               parallel pyrosequencing: the trnL approach}",
+  author    = "Valentini, Alice and Miquel, Christian and Nawaz, Muhammad Ali
+               and Bellemain, Eva and Coissac, Eric and Pompanon, Fran{\c c}ois
+               and Gielly, Ludovic and Cruaud, Corinne and Nascetti, Giuseppe
+               and Wincker, Patrick and Swenson, Jon E and Taberlet, Pierre",
+  abstract  = "The development of DNA barcoding (species identification using a
+               standardized DNA sequence), and the availability of recent DNA
+               sequencing techniques offer new possibilities in diet analysis.
+               DNA fragments shorter than 100-150 bp remain in a much higher
+               proportion in degraded DNA samples and can be recovered from
+               faeces. As a consequence, by using universal primers that
+               amplify a very short but informative DNA fragment, it is
+               possible to reliably identify the plant taxon that has been
+               eaten. According to our experience and using this identification
+               system, about 50\% of the taxa can be identified to species
+               using the trnL approach, that is, using the P6 loop of the
+               chloroplast trnL (UAA) intron. We demonstrated that this new
+               method is fast, simple to implement, and very robust. It can be
+               applied for diet analyses of a wide range of phytophagous
+               species at large scales. We also demonstrated that our approach
+               is efficient for mammals, birds, insects and molluscs. This
+               method opens new perspectives in ecology, not only by allowing
+               large-scale studies on diet, but also by enhancing studies on
+               resource partitioning among competing species, and describing
+               food webs in ecosystems.",
+  journal   = "Molecular ecology resources",
+  publisher = "WILEY-BLACKWELL PUBLISHING, INC",
+  volume    =  9,
+  number    =  1,
+  pages     = "51--60",
+  month     =  jan,
+  year      =  2009,
+  url       = "http://dx.doi.org/10.1111/j.1755-0998.2008.02352.x",
+  address   = "COMMERCE PLACE, 350 MAIN ST, MALDEN 02148, MA USA",
+  keywords  = "chloroplast DNA; diet analysis; DNA barcoding; faeces;
+               pyrosequencing; trnL (UAA) intron; universal primers",
+  language  = "en",
+  issn      = "1755-098X",
+  pmid      = "21564566",
+  doi       = "10.1111/j.1755-0998.2008.02352.x"
+}
+
+
+@ARTICLE{Kowalczyk2011-kg,
+  title     = "{Influence of management practices on large herbivore diet---Case
+               of European bison in Bia{\l}owie{\.z}a Primeval Forest (Poland)}",
+  author    = "Kowalczyk, Rafa{\l} and Taberlet, Pierre and Coissac, Eric and
+               Valentini, Alice and Miquel, Christian and Kami{\'n}ski, Tomasz
+               and W{\'o}jcik, Jan M",
+  abstract  = "Large herbivores are keystone species in many forest areas, as
+               they shape the structure, species diversity and functioning of
+               those ecosystems. The European bison Bison bonasus has been
+               successfully restored after extinction in the wild at the
+               beginning of 20th century. As free-ranging populations of the
+               species were re-established mainly in forest habitats, knowledge
+               of the impact by the largest European terrestrial mammal on tree
+               stands is essential. This helps to make management and
+               conservation decisions for viable population maintenance of the
+               species in the wild. Using a novel DNA-based method of herbivore
+               diet analysis, the trnL approach (DNA-barcoding), we
+               investigated the influence of different foraging conditions
+               (access to supplementary fodder) on bison diet in winter and its
+               potential impact on woody species. Faecal samples were collected
+               from different bison treatment groups: (1) intensively fed; (2)
+               less intensively fed; (3) non-fed utilising forest habitats; and
+               (4) non-fed utilising agricultural areas surrounding the Forest.
+               These were analysed to estimate the proportion of different
+               plant groups consumed by bison. Bison groups differed
+               significantly in their diet. The amount of woody materials
+               (trees and shrubs) consumed by bison increased with decreasing
+               access to supplementary fodder, ranging from 16\% in intensively
+               fed bison to 65\% in non-fed bison utilising forest habitats.
+               Inversely, the amount of herbs, grasses and sedges decreased
+               from 82\% in intensively fed bison to 32\% in non fed bison
+               utilising forest habitats. The species of trees mainly browsed
+               by bison, Carpinus/Corylus, Betula sp. and Salix sp., were of
+               lower economic importance for forest management. The impact of
+               bison on tree species needs further investigation, however, we
+               can predict that browsing by bison, mainly on Carpinus/Corylus,
+               makes an insignificant impact on forestry due to the high and
+               increasing representation of this species in the forest
+               understory. Supplementary feeding has several negative effects
+               on bison ecology and health, therefore reduced and distributed
+               supplementary feeding should be applied as the management
+               practice in the Bia{\l}owie{\.z}a Forest.",
+  journal   = "Forest ecology and management",
+  publisher = "ELSEVIER SCIENCE BV",
+  volume    =  261,
+  number    =  4,
+  pages     = "821--828",
+  month     =  feb,
+  year      =  2011,
+  url       = "http://www.sciencedirect.com/science/article/pii/S0378112710006961",
+  annote    = "di",
+  address   = "PO BOX 211, 1000 AE AMSTERDAM, NETHERLANDS",
+  keywords  = "Bison bonasus; trnL approach; DNA barcoding; Diet analysis;
+               Large ungulates; Wildlife management; L approach",
+  language  = "English",
+  issn      = "0378-1127",
+  doi       = "10.1016/j.foreco.2010.11.026"
+}
+
+
+@ARTICLE{Deagle2009-yh,
+  title    = "{Analysis of Australian fur seal diet by pyrosequencing prey DNA
+              in faeces}",
+  author   = "Deagle, Bruce E and Kirkwood, Roger and Jarman, Simon N",
+  abstract = "DNA-based techniques have proven useful for defining trophic
+              links in a variety of ecosystems and recently developed
+              sequencing technologies provide new opportunities for dietary
+              studies. We investigated the diet of Australian fur seals
+              (Arctocephalus pusillus doriferus) by pyrosequencing prey DNA
+              from faeces collected at three breeding colonies across the
+              seals' range. DNA from 270 faecal samples was amplified with four
+              polymerase chain reaction primer sets and a blocking primer was
+              used to limit amplification of fur seal DNA. Pooled amplicons
+              from each colony were sequenced using the Roche GS-FLX platform,
+              generating > 20,000 sequences. Software was developed to sort and
+              group similar sequences. A total of 54 bony fish, 4 cartilaginous
+              fish and 4 cephalopods were identified based on the most
+              taxonomically informative amplicons sequenced (mitochondrial
+              16S). The prevalence of sequences from redbait (Emmelichthys
+              nitidus) and jack mackerel (Trachurus declivis) confirm the
+              importance of these species in the seals' diet. A third fish
+              species, blue mackerel (Scomber australasicus), may be a more
+              important prey species than previously recognised. There were
+              major differences in the proportions of prey DNA recovered in
+              faeces from different colonies, probably reflecting differences
+              in prey availability. Parallel hard-part analysis identified
+              largely the same main prey species as did the DNA-based
+              technique, but with lower species diversity and no remains from
+              cartilaginous prey. The pyrosequencing approach presented
+              significantly expands the capabilities of DNA-based methods of
+              dietary analysis and is suitable for large-scale diet
+              investigations on a broad range of animals.",
+  journal  = "Molecular ecology",
+  volume   =  18,
+  number   =  9,
+  pages    = "2022--2038",
+  month    =  may,
+  year     =  2009,
+  url      = "http://dx.doi.org/10.1111/j.1365-294X.2009.04158.x",
+  issn     = "0962-1083, 1365-294X",
+  pmid     = "19317847",
+  doi      = "10.1111/j.1365-294X.2009.04158.x"
+}
+
+
+@ARTICLE{Shehzad2012-pn,
+  title     = "{Carnivore diet analysis based on next-generation sequencing:
+               Application to the leopard cat (Prionailurus bengalensis) in
+               Pakistan}",
+  author    = "Shehzad, Wasim and Riaz, Tiayyba and Nawaz, Muhammad A and
+               Miquel, Christian and Poillot, Carole and Shah, Safdar A and
+               Pompanon, Francois and Coissac, Eric and Taberlet, Pierre",
+  journal   = "Molecular ecology",
+  publisher = "Wiley Online Library",
+  volume    =  21,
+  number    =  8,
+  pages     = "1951--1965",
+  year      =  2012,
+  url       = "https://onlinelibrary.wiley.com/doi/abs/10.1111/j.1365-294X.2011.05424.x",
+  issn      = "0962-1083"
+}
+
+
+@ARTICLE{Schloss2009-qy,
+  title    = "{Introducing mothur: open-source, platform-independent,
+              community-supported software for describing and comparing
+              microbial communities}",
+  author   = "Schloss, Patrick D and Westcott, Sarah L and Ryabin, Thomas and
+              Hall, Justine R and Hartmann, Martin and Hollister, Emily B and
+              Lesniewski, Ryan A and Oakley, Brian B and Parks, Donovan H and
+              Robinson, Courtney J and Sahl, Jason W and Stres, Blaz and
+              Thallinger, Gerhard G and Van Horn, David J and Weber, Carolyn F",
+  abstract = "mothur aims to be a comprehensive software package that allows
+              users to use a single piece of software to analyze community
+              sequence data. It builds upon previous tools to provide a
+              flexible and powerful software package for analyzing sequencing
+              data. As a case study, we used mothur to trim, screen, and align
+              sequences; calculate distances; assign sequences to operational
+              taxonomic units; and describe the alpha and beta diversity of
+              eight marine samples previously characterized by pyrosequencing
+              of 16S rRNA gene fragments. This analysis of more than 222,000
+              sequences was completed in less than 2 h with a laptop computer.",
+  journal  = "Applied and environmental microbiology",
+  volume   =  75,
+  number   =  23,
+  pages    = "7537--7541",
+  month    =  dec,
+  year     =  2009,
+  url      = "http://dx.doi.org/10.1128/AEM.01541-09",
+  issn     = "0099-2240, 1098-5336",
+  pmid     = "19801464",
+  doi      = "10.1128/AEM.01541-09",
+  pmc      = "PMC2786419"
+}
+
+
+@ARTICLE{Caporaso2010-ii,
+  title   = "{QIIME allows analysis of high-throughput community sequencing data}",
+  author  = "Caporaso, J Gregory and Kuczynski, Justin and Stombaugh, Jesse and
+             Bittinger, Kyle and Bushman, Frederic D and Costello, Elizabeth K
+             and Fierer, Noah and Pe{\~n}a, Antonio Gonzalez and Goodrich,
+             Julia K and Gordon, Jeffrey I and Huttley, Gavin A and Kelley,
+             Scott T and Knights, Dan and Koenig, Jeremy E and Ley, Ruth E and
+             Lozupone, Catherine A and McDonald, Daniel and Muegge, Brian D and
+             Pirrung, Meg and Reeder, Jens and Sevinsky, Joel R and Turnbaugh,
+             Peter J and Walters, William A and Widmann, Jeremy and Yatsunenko,
+             Tanya and Zaneveld, Jesse and Knight, Rob",
+  journal = "Nature methods",
+  volume  =  7,
+  number  =  5,
+  pages   = "335--336",
+  month   =  may,
+  year    =  2010,
+  url     = "http://dx.doi.org/10.1038/nmeth.f.303",
+  issn    = "1548-7091, 1548-7105",
+  pmid    = "20383131",
+  doi     = "10.1038/nmeth.f.303",
+  pmc     = "PMC3156573"
+}
diff --git a/doc/commands.qmd b/doc/commands.qmd
index d83d322..08f95bd 100644
--- a/doc/commands.qmd
+++ b/doc/commands.qmd
@@ -6,13 +6,13 @@
 
 ### Specifying input format
 
-Five sequence formats are accepted for input files. [Fasta](#fasta-classical "Fasta format description") and [Fastq](#fastq-classical "Fastq format description") are the main ones, EMBL and Genbank allow the use of flat files produced by these two international databases. The last one, ecoPCR, is maintained for compatibility with previous *OBITools* and allows to read *ecoPCR* outputs as sequence files.
+Five sequence formats are accepted for input files. *Fasta* (@sec-fasta) and *Fastq* (@sec-fastq) are the main ones, EMBL and Genbank allow the use of flat files produced by these two international databases. The last one, ecoPCR, is maintained for compatibility with previous *OBITools* and allows to read *ecoPCR* outputs as sequence files.
 
 -   `--ecopcr` : Read data following the *ecoPCR* output format.
 -   `--embl` Read data following the *EMBL* flatfile format.
 -   `--genbank` Read data following the *Genbank* flatfile format.
 
-Several encoding schemes have been proposed for quality scores in [Fastq](#fastq-classical "Fastq format description") format. Currently, *OBITools* considers Sanger encoding as the standard. For reasons of compatibility with older datasets produced with *Solexa* sequencers, it is possible, by using the following option, to force the use of the corresponding quality encoding scheme when reading these older files.
+Several encoding schemes have been proposed for quality scores in *Fastq* format. Currently, *OBITools* considers Sanger encoding as the standard. For reasons of compatibility with older datasets produced with *Solexa* sequencers, it is possible, by using the following option, to force the use of the corresponding quality encoding scheme when reading these older files.
 
 -   `--solexa` Decodes quality string according to the Solexa specification. (default: false)
 
@@ -25,11 +25,12 @@ Only two output sequence formats are supported by OBITools, Fasta and Fastq. Fas
 
 OBITools allows multiple input files to be specified for a single command.
 
--   `--no-order` When several input files are provided, indicates that there is no order among them. (default: false)
+-   `--no-order` When several input files are provided, indicates that there is no order among them. (default: false). 
+                 Using such option can increase a lot the processing of the data.
 
-### Format of the annotations in Fasta and Fastq files
+### The Fasta and Fastq annotations format
 
-OBITools extend the [Fasta](#fasta-classical "Fasta format description") and [Fastq](#fastq-classical "Fastq format description") formats by introducing a format for the title lines of these formats allowing to annotate every sequence. While the previous version of OBITools used an *ad-hoc* format for these annotation, this new version introduce the usage of the standard JSON format to store them.
+OBITools extend the [Fasta](#the-fasta-sequence-format) and [Fastq](#the-fastq-sequence-format) formats by introducing a format for the title lines of these formats allowing to annotate every sequence. While the previous version of OBITools used an *ad-hoc* format for these annotation, this new version introduce the usage of the standard JSON format to store them.
 
 On input, OBITools automatically recognize the format of the annotations, but two options allows to force the parsing following one of them. You should normally not need to use these options.
 
diff --git a/doc/formats.qmd b/doc/formats.qmd
new file mode 100644
index 0000000..7c7b7c2
--- /dev/null
+++ b/doc/formats.qmd
@@ -0,0 +1,145 @@
+## File formats usable with *OBITools*
+
+OBITools manipulate have to manipulate DNA sequence data and taxonomical data. 
+They can use some supplentary metadata describing the experiment and produce
+some stats about the processed DNA data. All the manipulated data are stored in
+text files, following standard data format.
+
+## The DNA sequence data
+
+Sequences can be stored following various format. OBITools knows some of them. The central formats for sequence files manipulated by OBITools scripts are the [`fasta`](#the-fasta-sequence-format) and [`fastq`](#the-fastq-sequence-format) format. OBITools extends the both these formats by specifying a syntax to include in the definition line data qualifying the sequence. All file formats use the `IUPAC` code for encoding nucleotides.
+
+Moreover these two formats that can be used as input and output formats, **OBITools4** can read the following format : 
+
+- [EBML flat file](https://ena-docs.readthedocs.io/en/latest/submit/fileprep/flat-file-example.html) format (use by ENA)
+- [Genbank flat file format](https://www.ncbi.nlm.nih.gov/Sitemap/samplerecord.html)
+- [ecoPCR output files](https://pythonhosted.org/OBITools/scripts/ecoPCR.html)
+
+### The IUPAC Code
+
+The International Union of Pure and Applied Chemistry (IUPAC\_) defined the standard code for representing protein or DNA sequences.
+
+| **Code** | **Nucleotide**              |
+|----------|-----------------------------|
+| A        | Adenine                     |
+| C        | Cytosine                    |
+| G        | Guanine                     |
+| T        | Thymine                     |
+| U        | Uracil                      |
+| R        | Purine (A or G)             |
+| Y        | Pyrimidine (C, T, or U)     |
+| M        | C or A                      |
+| K        | T, U, or G                  |
+| W        | T, U, or A                  |
+| S        | C or G                      |
+| B        | C, T, U, or G (not A)       |
+| D        | A, T, U, or G (not C)       |
+| H        | A, T, U, or C (not G)       |
+| V        | A, C, or G (not T, not U)   |
+| N        | Any base (A, C, G, T, or U) |
+
+### The *fasta* sequence format {#sec-fasta}
+
+The **fasta format** is certainly the most widely used sequence file format. This is certainly due to its great simplicity. It was originally created for the Lipman and Pearson [FASTA program](http://www.ncbi.nlm.nih.gov/pubmed/3162770?dopt=Citation). OBITools use in more of the classical `fasta` format an `extended version` of this format where structured data are included in the title line.
+
+In *fasta* format a sequence is represented by a title line beginning with a **`>`** character and the sequences by itself following the :doc:`iupac` code. The sequence is usually split other severals lines of the same length (expect for the last one)
+
+    >my_sequence this is my pretty sequence
+    ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
+    GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
+    AACGACGTTGCAGTACGTTGCAGT
+
+This is no special format for the title line excepting that this line should be unique. Usually the first word following the **\>** character is considered as the sequence identifier. The end of the title line corresponding to a description of the sequence. Several sequences can be concatenated in a same file. The description of the next sequence is just pasted at the end of the record of the previous one
+
+    >sequence_A this is my first pretty sequence
+    ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
+    GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
+    AACGACGTTGCAGTACGTTGCAGT
+    >sequence_B this is my second pretty sequence
+    ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
+    GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
+    AACGACGTTGCAGTACGTTGCAGT
+    >sequence_C this is my third pretty sequence
+    ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
+    GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
+    AACGACGTTGCAGTACGTTGCAGT
+
+### The *fastq* sequence format[^01_obitools_doc-1]{#sec-fastq}
+
+The **FASTQ** format is a text file format for storing both biological sequences (only nucleic acid sequences) and the associated quality scores. The sequence and score are each encoded by a single ASCII character. This format was originally developed by the Wellcome Trust Sanger Institute to link a [FASTA](#the-fasta-sequence-format) sequence file to the corresponding quality data, but has recently become the de facto standard for storing results from high-throughput sequencers [@cock2010sanger].
+
+[^01_obitools_doc-1]: This article uses material from the Wikipedia article [`FASTQ format`](http://en.wikipedia.org/wiki/FASTQ_format) which is released under the `Creative Commons Attribution-Share-Alike License 3.0`
+
+A fastq file normally uses four lines per sequence.
+
+-   Line 1 begins with a '\@' character and is followed by a sequence identifier and an *optional* description (like a :ref:`fasta` title line).
+-   Line 2 is the raw sequence letters.
+-   Line 3 begins with a '+' character and is *optionally* followed by the same sequence identifier (and any description) again.
+-   Line 4 encodes the quality values for the sequence in Line 2, and must contain the same number of symbols as letters in the sequence.
+
+A fastq file containing a single sequence might look like this:
+
+```
+@SEQ_ID
+GATTTGGGGTTCAAAGCAGTATCGATCAAATAGTAAATCCATTTGTTCAACTCACAGTTT
++
+!''*((((***+))%%%++)(%%%%).1***-+*''))**55CCF>>>>>>CCCCCCC65
+```
+
+The character '!' represents the lowest quality while '\~' is the highest. Here are the quality value characters in left-to-right increasing order of quality (`ASCII`):
+
+    !"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\]^_`abcdefghijklmnopqrstuvwxyz{|}~
+
+The original Sanger FASTQ files also allowed the sequence and quality strings to be wrapped (split over multiple lines), but this is generally discouraged as it can make parsing complicated due to the unfortunate choice of "\@" and "+" as markers (these characters can also occur in the quality string).
+
+#### Sequence quality scores{.unnumbered}
+
+The Phred quality value *Q* is an integer mapping of *p* (i.e., the probability that the corresponding base call is incorrect). Two different equations have been in use. The first is the standard Sanger variant to assess reliability of a base call, otherwise known as Phred quality score:
+
+$$
+Q_\text{sanger} = -10 \, \log_{10} p
+$$
+
+The Solexa pipeline (i.e., the software delivered with the Illumina Genome Analyzer) earlier used a different mapping, encoding the odds $\mathbf{p}/(1-\mathbf{p})$ instead of the probability $\mathbf{p}$:
+
+$$
+Q_\text{solexa-prior to v.1.3} = -10 \; \log_{10} \frac{p}{1-p}
+$$
+
+Although both mappings are asymptotically identical at higher quality values, they differ at lower quality levels (i.e., approximately $\mathbf{p} > 0.05$, or equivalently, $\mathbf{Q} < 13$).
+
+![Relationship between *Q* and *p* using the Sanger (red) and Solexa (black) equations (described above). The vertical dotted line indicates $\mathbf{p}= 0.05$, or equivalently, $Q = 13$.](Probabilitymetrics.png){#fig-Probabilitymetrics}
+
+
+##### Encoding{.unnumbered}
+
+The *fastq* format had differente way of encoding the Phred quality score along the time. Here a breif history of these changes is presented. 
+
+-   Sanger format can encode a Phred quality score from 0 to 93 using ASCII 33 to 126 (although in raw read data the Phred quality score rarely exceeds 60, higher scores are possible in assemblies or read maps).
+-   Solexa/Illumina 1.0 format can encode a Solexa/Illumina quality score from -5 to 62 using ASCII 59 to 126 (although in raw read data Solexa scores from -5 to 40 only are expected)
+-   Starting with Illumina 1.3 and before Illumina 1.8, the format encoded a Phred quality score from 0 to 62 using ASCII 64 to 126 (although in raw read data Phred scores from 0 to 40 only are expected).
+-   Starting in Illumina 1.5 and before Illumina 1.8, the Phred scores 0 to 2 have a slightly different meaning. The values 0 and 1 are no longer used and the value 2, encoded by ASCII 66 "B".
+
+> Sequencing Control Software, Version 2.6, (Catalog \# SY-960-2601, Part \# 15009921 Rev. A, November 2009, page 30) 
+> states the following: *If a read ends with a segment of mostly low quality (Q15 or below), then all of the quality 
+> values in the segment are replaced with a value of 2 (encoded as the letter B in Illumina's text-based encoding of 
+> quality scores)... This Q2 indicator does not predict a specific error rate, but rather indicates that a specific 
+> final portion of the read should not be used in further analyses.* Also, the quality score encoded as "B" letter 
+> may occur internally within reads at least as late as pipeline version 1.6, as shown in the following example:
+
+```
+@HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1
+TTAATTGGTAAATAAATCTCCTAATAGCTTAGATNTTACCTTNNNNNNNNNNTAGTTTCTTGAGATTTGTTGGGGGAGACATTTTTGTGATTGCCTTGAT
++HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1
+efcfffffcfeefffcffffffddf`feed]`]_Ba_^__[YBBBBBBBBBBRTT\]][]dddd`ddd^dddadd^BBBBBBBBBBBBBBBBBBBBBBBB
+```
+
+An alternative interpretation of this ASCII encoding has been proposed. Also, in Illumina runs using PhiX controls, the character 'B' was observed to represent an "unknown quality score". The error rate of 'B' reads was roughly 3 phred scores lower the mean observed score of a given run.
+
+-   Starting in Illumina 1.8, the quality scores have basically returned to the use of the Sanger format (Phred+33).
+
+OBItools support the Sanger format. It is nevertheless to read files encoded following the Solexa/Illumina format, that are still possible to find in old files, by applying a shift of 62.
+
+### File extension
+
+There is no standard file extension for a FASTQ file, but .fq and .fastq, are commonly used.
diff --git a/doc/intro.qmd b/doc/intro.qmd
index 4b5845d..32d0b4f 100644
--- a/doc/intro.qmd
+++ b/doc/intro.qmd
@@ -1,142 +1,57 @@
 # The OBITools
 
+The *OBITools4* are programs specifically designed for analyzing NGS data in a DNA metabarcoding context, taking into account taxonomic information. It is distributed as an open source software available on the following website: http://metabarcoding.org/obitools4.
+
+
 ## Aims of *OBITools*
 
-## File formats usable with *OBITools*
+DNA metabarcoding is an efficient approach for biodiversity studies [@Taberlet2012-pf]. Originally mainly developed by microbiologists [*e.g.* @Sogin2006-ab], it is now widely used for plants [*e.g.* @Sonstebo2010-vv;@Yoccoz2012-ix;@Parducci2012-rn] and animals from meiofauna [*e.g.* @Chariton2010-cz;@Baldwin2013-yc] to larger organisms [*e.g.* @Andersen2012-gj;@Thomsen2012-au]. Interestingly, this method is not limited to *sensu
+stricto* biodiversity surveys, but it can also be implemented in other
+ecological contexts such as for herbivore [e.g. @Valentini2009-ay;@Kowalczyk2011-kg] or carnivore [e.g. @Deagle2009-yh;@Shehzad2012-pn] diet
+analyses.
 
-### The sequence files
+Whatever the biological question under consideration, the DNA metabarcoding
+methodology relies heavily on next-generation sequencing (NGS), and generates
+considerable numbers of DNA sequence reads (typically million of reads).
+Manipulation of such large datasets requires dedicated programs usually running
+on a Unix system. Unix is an operating system, whose first version was created 
+during the sixties. Since its early stages, it is dedicated to scientific
+computing and includes a large set of simple tools to efficiently process text
+files. Most of those programs can be viewed as filters extracting information
+from a text file to create a new text file. These programs process text files as
+streams, line per line, therefore allowing computation on a huge dataset without
+requiring a large memory. Unix programs usually print their results to their
+standard output (*stdout*), which by default is the terminal, so the results can
+be examined on screen. The main philosophy of the Unix environment is to allow
+easy redirection of the *stdout* either to a file, for saving the results, or to
+the standard input (*stdin*) of a second program thus allowing to easily create
+complex processing from simple base commands. Access to Unix computers is
+increasingly easier for scientists nowadays. Indeed, the Linux operating system,
+an open source version of Unix, can be freely installed on every PC machine and
+the MacOS operating system, running on Apple computers, is also a Unix system. 
+The *OBITools* programs imitate Unix standard programs because they usually act as
+filters, reading their data from text files or the *stdin* and writing their
+results to the *stdout*. The main difference with classical Unix programs is that
+text files are not analyzed line per line but sequence record per sequence
+record (see below for a detailed description of a sequence record).
+Compared to packages for similar purposes like mothur [@Schloss2009-qy] or
+QIIME [@Caporaso2010-ii], the *OBITools* mainly rely on filtering and sorting
+algorithms. This allows users to set up versatile data analysis pipelines
+(Figure 1), adjustable to the broad range of DNA metabarcoding applications. 
+The innovation of the *OBITools* is their ability to take into account the
+taxonomic annotations, ultimately allowing sorting and filtering of sequence
+records based on the taxonomy.
 
-Sequences can be stored following various format. OBITools knows some of them. The central formats for sequence files manipulated by OBITools scripts are the `fasta` and fastq format. OBITools extends the both these formats by specifying a syntax to include in the definition line data qualifying the sequence. All file formats use the `IUPAC` code for encoding nucleotides.
+## Installation of the obitools
 
-### The IUPAC Code
+### Availability of the OBITools
 
-The International Union of Pure and Applied Chemistry (IUPAC\_) defined the standard code for representing protein or DNA sequences.
+The *OBITools* are open source and protected by the [CeCILL 2.1 license](http://www.cecill.info/licences/Licence_CeCILL_V2.1-en.html).
 
-#### Nucleic IUPAC Code {#DNA-IUPAC}
+All the sources of the [*OBITools4*](http://metabarcoding.org/obitools4) can be downloaded from the metabarcoding git server (https://git.metabarcoding.org).
 
-| **Code** | **Nucleotide**              |
-|----------|-----------------------------|
-| A        | Adenine                     |
-| C        | Cytosine                    |
-| G        | Guanine                     |
-| T        | Thymine                     |
-| U        | Uracil                      |
-| R        | Purine (A or G)             |
-| Y        | Pyrimidine (C, T, or U)     |
-| M        | C or A                      |
-| K        | T, U, or G                  |
-| W        | T, U, or A                  |
-| S        | C or G                      |
-| B        | C, T, U, or G (not A)       |
-| D        | A, T, U, or G (not C)       |
-| H        | A, T, U, or C (not G)       |
-| V        | A, C, or G (not T, not U)   |
-| N        | Any base (A, C, G, T, or U) |
+### Prerequisites
 
-### The *fasta* format {#classical-fasta}
+The *OBITools4* are developped using the [GO programming language](https://go.dev/), we stick to the latest version of the language, today the $1.19.5$. If you want to download and compile the sources yourself, you first need to install the corresponding compiler on your system. Some parts of the soft are also written in C, therefore a recent C compiler is also requested, GCC on Linux or Windows, the Developer Tools on Mac.
 
-The **fasta format** is certainly the most widely used sequence file format. This is certainly due to its great simplicity. It was originally created for the Lipman and Pearson [FASTA program](http://www.ncbi.nlm.nih.gov/pubmed/3162770?dopt=Citation). OBITools use in more of the classical :ref:`fasta` format an :ref:`extended version` of this format where structured data are included in the title line.
-
-In *fasta* format a sequence is represented by a title line beginning with a **`>`** character and the sequences by itself following the :doc:`iupac` code. The sequence is usually split other severals lines of the same length (expect for the last one)
-
-    >my_sequence this is my pretty sequence
-    ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
-    GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
-    AACGACGTTGCAGTACGTTGCAGT
-
-This is no special format for the title line excepting that this line should be unique. Usually the first word following the **\>** character is considered as the sequence identifier. The end of the title line corresponding to a description of the sequence. Several sequences can be concatenated in a same file. The description of the next sequence is just pasted at the end of the record of the previous one
-
-    >sequence_A this is my first pretty sequence
-    ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
-    GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
-    AACGACGTTGCAGTACGTTGCAGT
-    >sequence_B this is my second pretty sequence
-    ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
-    GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
-    AACGACGTTGCAGTACGTTGCAGT
-    >sequence_C this is my third pretty sequence
-    ACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGT
-    GTGCTGACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTACGTTGCAGTGTTT
-    AACGACGTTGCAGTACGTTGCAGT
-
-### The *fastq* sequence format[^01_obitools_doc-1]  {#classical-fastq}
-
-**fastq format** is a text-based format for storing both a biological sequence (usually nucleotide sequence) and its corresponding quality scores. Both the sequence letter and quality score are encoded with a single ASCII character for brevity. It was originally developed at the `Wellcome Trust Sanger Institute` to bundle a [fasta](#classical-fasta) sequence and its quality data, but has recently become the *de facto* standard for storing the output of high throughput sequencing instruments such as the Illumina Genome Analyzer Illumina [@cock2010sanger] . 
-
-[^01_obitools_doc-1]: This article uses material from the Wikipedia article [`FASTQ format`](http://en.wikipedia.org/wiki/FASTQ_format) which is released under the `Creative Commons Attribution-Share-Alike License 3.0`
-
-A fastq file normally uses four lines per sequence.
-
--   Line 1 begins with a '\@' character and is followed by a sequence identifier and an *optional* description (like a :ref:`fasta` title line).
--   Line 2 is the raw sequence letters.
--   Line 3 begins with a '+' character and is *optionally* followed by the same sequence identifier (and any description) again.
--   Line 4 encodes the quality values for the sequence in Line 2, and must contain the same number of symbols as letters in the sequence.
-
-A fastq file containing a single sequence might look like this:
-
-    @SEQ_ID
-    GATTTGGGGTTCAAAGCAGTATCGATCAAATAGTAAATCCATTTGTTCAACTCACAGTTT
-    +
-    !''*((((***+))%%%++)(%%%%).1***-+*''))**55CCF>>>>>>CCCCCCC65
-
-The character '!' represents the lowest quality while '\~' is the highest. Here are the quality value characters in left-to-right increasing order of quality (`ASCII`):
-
-    !"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\]^_`abcdefghijklmnopqrstuvwxyz{|}~
-
-The original Sanger FASTQ files also allowed the sequence and quality strings to be wrapped (split over multiple lines), but this is generally discouraged as it can make parsing complicated due to the unfortunate choice of "\@" and "+" as markers (these characters can also occur in the quality string).
-
-#### Variations
-
-##### Quality
-
-A quality value *Q* is an integer mapping of *p* (i.e., the probability that the corresponding base call is incorrect). Two different equations have been in use. The first is the standard Sanger variant to assess reliability of a base call, otherwise known as Phred quality score:
-
-$$
-Q_\text{sanger} = -10 \, \log_{10} p
-$$
-
-The Solexa pipeline (i.e., the software delivered with the Illumina Genome Analyzer) earlier used a different mapping, encoding the odds $\mathbf{p}/(1-\mathbf{p})$ instead of the probability $\mathbf{p}$:
-
-$$
-Q_\text{solexa-prior to v.1.3} = -10 \, \log_{10} \frac{p}{1-p}
-$$
-
-Although both mappings are asymptotically identical at higher quality values, they differ at lower quality levels (i.e., approximately $\mathbf{p} > 0.05$, or equivalently, $\mathbf{Q} < 13$).
-
-\|Relationship between *Q* and *p* using the Sanger (red) and Solexa (black) equations (described above). The vertical dotted line indicates $\mathbf{p}= 0.05$, or equivalently, $Q = 13$.\|
-
-#### Encoding
-
--   Sanger format can encode a Phred quality score from 0 to 93 using ASCII 33 to 126 (although in raw read data the Phred quality score rarely exceeds 60, higher scores are possible in assemblies or read maps).
--   Solexa/Illumina 1.0 format can encode a Solexa/Illumina quality score from -5 to 62 using ASCII 59 to 126 (although in raw read data Solexa scores from -5 to 40 only are expected)
--   Starting with Illumina 1.3 and before Illumina 1.8, the format encoded a Phred quality score from 0 to 62 using ASCII 64 to 126 (although in raw read data Phred scores from 0 to 40 only are expected).
--   Starting in Illumina 1.5 and before Illumina 1.8, the Phred scores 0 to 2 have a slightly different meaning. The values 0 and 1 are no longer used and the value 2, encoded by ASCII 66 "B".
-
-Sequencing Control Software, Version 2.6, Catalog \# SY-960-2601, Part \# 15009921 Rev. A, November 2009] [[http://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf\\\\](http://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf){.uri}](%5Bhttp://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf\%5D(http://watson.nci.nih.gov/solexa/Using_SCSv2.6_15009921_A.pdf)%7B.uri%7D){.uri} (page 30) states the following: *If a read ends with a segment of mostly low quality (Q15 or below), then all of the quality values in the segment are replaced with a value of 2 (encoded as the letter B in Illumina's text-based encoding of quality scores)... This Q2 indicator does not predict a specific error rate, but rather indicates that a specific final portion of the read should not be used in further analyses.* Also, the quality score encoded as "B" letter may occur internally within reads at least as late as pipeline version 1.6, as shown in the following example:
-
-    @HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1
-    TTAATTGGTAAATAAATCTCCTAATAGCTTAGATNTTACCTTNNNNNNNNNNTAGTTTCTTGAGATTTGTTGGGGGAGACATTTTTGTGATTGCCTTGAT
-    +HWI-EAS209_0006_FC706VJ:5:58:5894:21141#ATCACG/1
-    efcfffffcfeefffcffffffddf`feed]`]_Ba_^__[YBBBBBBBBBBRTT\]][]dddd`ddd^dddadd^BBBBBBBBBBBBBBBBBBBBBBBB
-
-An alternative interpretation of this ASCII encoding has been proposed. Also, in Illumina runs using PhiX controls, the character 'B' was observed to represent an "unknown quality score". The error rate of 'B' reads was roughly 3 phred scores lower the mean observed score of a given run.
-
--   Starting in Illumina 1.8, the quality scores have basically returned to the use of the Sanger format (Phred+33).
-
-## File extension
-
-There is no standard file extension for a FASTQ file, but .fq and .fastq, are commonly used.
-
-## See also
-
--   :ref:`fasta`
-
-## References
-
-.. [1] Cock et al (2009) The Sanger FASTQ file format for sequences with quality scores, and the Solexa/Illumina FASTQ variants. Nucleic Acids Research,
-
-.. [2] Illumina Quality Scores, Tobias Mann, Bioinformatics, San Diego, Illumina `1`\_\_
-
-.. \|Relationship between *Q* and *p* using the Sanger (red) and Solexa (black) equations (described above). The vertical dotted line indicates *p* = 0.05, or equivalently, *Q* Å 13.\| image:: Probability metrics.png
-
-See <http://en.wikipedia.org/wiki/FASTQ_format>
+Whatever the installation you decide for, you will have to ensure that a C compiler is available on your system.
\ No newline at end of file