2022-01-13 23:27:39 +01:00
|
|
|
package ncbitaxdump
|
|
|
|
|
|
|
|
import (
|
|
|
|
"bufio"
|
|
|
|
"encoding/csv"
|
|
|
|
"fmt"
|
|
|
|
"io"
|
2022-02-24 12:14:52 +01:00
|
|
|
log "github.com/sirupsen/logrus"
|
2022-01-13 23:27:39 +01:00
|
|
|
"os"
|
|
|
|
"path"
|
|
|
|
"strconv"
|
|
|
|
"strings"
|
|
|
|
|
2022-01-13 23:43:01 +01:00
|
|
|
"git.metabarcoding.org/lecasofts/go/obitools/pkg/obitax"
|
2022-01-13 23:27:39 +01:00
|
|
|
)
|
|
|
|
|
|
|
|
func loadNodeTable(reader io.Reader, taxonomy *obitax.Taxonomy) {
|
|
|
|
file := csv.NewReader(reader)
|
|
|
|
file.Comma = '|'
|
|
|
|
file.Comment = '#'
|
|
|
|
file.TrimLeadingSpace = true
|
|
|
|
file.ReuseRecord = true
|
|
|
|
|
|
|
|
for record, err := file.Read(); err == nil; record, err = file.Read() {
|
|
|
|
taxid, _ := strconv.Atoi(strings.TrimSpace(record[0]))
|
|
|
|
parent, _ := strconv.Atoi(strings.TrimSpace(record[1]))
|
|
|
|
rank := strings.TrimSpace(record[2])
|
|
|
|
|
|
|
|
taxonomy.AddNewTaxa(taxid, parent, rank, true, true)
|
|
|
|
}
|
|
|
|
|
|
|
|
taxonomy.ReindexParent()
|
|
|
|
}
|
|
|
|
|
|
|
|
func loadNameTable(reader io.Reader, taxonomy *obitax.Taxonomy, onlysn bool) int {
|
|
|
|
// file := csv.NewReader(reader)
|
|
|
|
// file.Comma = '|'
|
|
|
|
// file.Comment = '#'
|
|
|
|
// file.TrimLeadingSpace = true
|
|
|
|
// file.ReuseRecord = true
|
|
|
|
// file.LazyQuotes = true
|
|
|
|
file := bufio.NewReader(reader)
|
|
|
|
|
|
|
|
n := 0
|
|
|
|
|
|
|
|
for line, prefix, err := file.ReadLine(); err == nil; line, prefix, err = file.ReadLine() {
|
|
|
|
|
|
|
|
if prefix {
|
|
|
|
return -1
|
|
|
|
}
|
|
|
|
|
|
|
|
record := strings.Split(string(line), "|")
|
|
|
|
taxid, _ := strconv.Atoi(strings.TrimSpace(record[0]))
|
|
|
|
name := strings.TrimSpace(record[1])
|
|
|
|
classname := strings.TrimSpace(record[3])
|
|
|
|
|
|
|
|
if !onlysn || classname == "scientific name" {
|
|
|
|
n++
|
|
|
|
taxonomy.AddNewName(taxid, &name, &classname)
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
return n
|
|
|
|
}
|
|
|
|
|
|
|
|
func loadMergedTable(reader io.Reader, taxonomy *obitax.Taxonomy) int {
|
|
|
|
file := csv.NewReader(reader)
|
|
|
|
file.Comma = '|'
|
|
|
|
file.Comment = '#'
|
|
|
|
file.TrimLeadingSpace = true
|
|
|
|
file.ReuseRecord = true
|
|
|
|
|
|
|
|
n := 0
|
|
|
|
|
|
|
|
for record, err := file.Read(); err == nil; record, err = file.Read() {
|
|
|
|
oldtaxid, _ := strconv.Atoi(strings.TrimSpace(record[0]))
|
|
|
|
newtaxid, _ := strconv.Atoi(strings.TrimSpace(record[1]))
|
|
|
|
n++
|
|
|
|
taxonomy.AddNewAlias(newtaxid, oldtaxid)
|
|
|
|
}
|
|
|
|
|
|
|
|
return n
|
|
|
|
}
|
|
|
|
|
|
|
|
func LoadNCBITaxDump(directory string, onlysn bool) (*obitax.Taxonomy, error) {
|
|
|
|
|
|
|
|
taxonomy := obitax.NewTaxonomy()
|
|
|
|
|
|
|
|
//
|
|
|
|
// Load the Taxonomy nodes
|
|
|
|
//
|
|
|
|
|
|
|
|
log.Printf("Loading Taxonomy nodes\n")
|
|
|
|
|
|
|
|
nodefile, err := os.Open(path.Join(directory, "nodes.dmp"))
|
|
|
|
if err != nil {
|
2022-01-14 16:10:19 +01:00
|
|
|
return nil, fmt.Errorf("cannot open nodes file from '%s'",
|
|
|
|
directory)
|
2022-01-13 23:27:39 +01:00
|
|
|
}
|
|
|
|
defer nodefile.Close()
|
|
|
|
|
|
|
|
buffered := bufio.NewReader(nodefile)
|
|
|
|
loadNodeTable(buffered, taxonomy)
|
|
|
|
log.Printf("%d Taxonomy nodes read\n", taxonomy.Length())
|
|
|
|
|
|
|
|
//
|
|
|
|
// Load the Taxonomy nodes
|
|
|
|
//
|
|
|
|
|
|
|
|
log.Printf("Loading Taxon names\n")
|
|
|
|
|
|
|
|
namefile, nerr := os.Open(path.Join(directory, "names.dmp"))
|
|
|
|
if nerr != nil {
|
2022-01-14 16:10:19 +01:00
|
|
|
return nil, fmt.Errorf("cannot open names file from '%s'",
|
|
|
|
directory)
|
2022-01-13 23:27:39 +01:00
|
|
|
}
|
|
|
|
defer namefile.Close()
|
|
|
|
|
|
|
|
n := loadNameTable(namefile, taxonomy, onlysn)
|
|
|
|
log.Printf("%d taxon names read\n", n)
|
|
|
|
|
|
|
|
//
|
|
|
|
// Load the merged taxa
|
|
|
|
//
|
|
|
|
|
|
|
|
log.Printf("Loading Merged taxa\n")
|
|
|
|
|
|
|
|
aliasfile, aerr := os.Open(path.Join(directory, "merged.dmp"))
|
|
|
|
if aerr != nil {
|
2022-01-14 16:10:19 +01:00
|
|
|
return nil, fmt.Errorf("cannot open merged file from '%s'",
|
|
|
|
directory)
|
2022-01-13 23:27:39 +01:00
|
|
|
}
|
|
|
|
defer aliasfile.Close()
|
|
|
|
|
|
|
|
buffered = bufio.NewReader(aliasfile)
|
|
|
|
n = loadMergedTable(buffered, taxonomy)
|
|
|
|
log.Printf("%d merged taxa read\n", n)
|
|
|
|
|
|
|
|
return taxonomy, nil
|
|
|
|
}
|