Taxonomy: new functions to find taxa by name

2020-10-30 10:45:20 +01:00
parent b9b4cec5b5
commit 112e12cab0
7 changed files with 174 additions and 58 deletions
--- a/src/obidms_taxonomy.c
+++ b/src/obidms_taxonomy.c
@ -3649,6 +3649,18 @@ ecotx_t* obi_taxo_get_taxon_with_taxid(OBIDMS_taxonomy_p taxonomy, int32_t taxid
 }


+char* obi_taxo_get_name_from_name_idx(OBIDMS_taxonomy_p taxonomy, int32_t idx)
+{
+	return (((taxonomy->names)->names)[idx]).name;
+}
+
+
+ecotx_t* obi_taxo_get_taxon_from_name_idx(OBIDMS_taxonomy_p taxonomy, int32_t idx)
+{
+	return (((taxonomy->names)->names)[idx]).taxon;
+}
+
+
 int obi_taxo_is_taxon_under_taxid(ecotx_t* taxon, int32_t other_taxid)		// TODO discuss that this doesn't work with deprecated taxids
 {
 	ecotx_t* next_parent;
--- a/src/obidms_taxonomy.h
+++ b/src/obidms_taxonomy.h
@ -447,8 +447,51 @@ ecotx_t* obi_taxo_get_superkingdom(ecotx_t* taxon, OBIDMS_taxonomy_p taxonomy);
 const char* obi_taxo_rank_index_to_label(int32_t rank_idx, ecorankidx_t* ranks);


-// TODO
+/**
+ * @brief Function checking whether a taxid is included in a subset of the taxonomy.
+ *
+ * @param taxonomy A pointer on the taxonomy structure.
+ * @param restrict_to_taxids An array of taxids. The researched taxid must be under at least one of those array taxids.
+ * @param count Number of taxids in restrict_to_taxids.
+ * @param taxid The taxid to check.
+ *
+ * @returns A value indicating whether the taxid is included in the chosen subset of the taxonomy.
+ * @retval 0 if the taxid is not included in the subset of the taxonomy.
+ * @retval 1 if the taxid is included in the subset of the taxonomy.
+ *
+ * @since October 2020
+ * @author Celine Mercier (celine.mercier@metabarcoding.org)
+ */
 int obi_taxo_is_taxid_included(OBIDMS_taxonomy_p taxonomy,
 			     			   int32_t* restrict_to_taxids,
 				    		   int32_t count,
 					    	   int32_t taxid);
+
+
+/**
+ * @brief Function returning the name of a taxon from its index in the taxonomy name index (econameidx_t).
+ *
+ * @param taxonomy A pointer on the taxonomy structure.
+ * @param idx The index at which the name is in the taxonomy name index (econameidx_t).
+ *
+ * @returns The taxon name.
+ *
+ * @since October 2020
+ * @author Celine Mercier (celine.mercier@metabarcoding.org)
+ */
+char* obi_taxo_get_name_from_name_idx(OBIDMS_taxonomy_p taxonomy, int32_t idx);
+
+
+/**
+ * @brief Function returning a taxon structure from its index in the taxonomy name index (econameidx_t).
+ *
+ * @param taxonomy A pointer on the taxonomy structure.
+ * @param idx The index at which the taxon is in the taxonomy name index (econameidx_t).
+ *
+ * @returns The taxon structure.
+ *
+ * @since October 2020
+ * @author Celine Mercier (celine.mercier@metabarcoding.org)
+ */
+ecotx_t* obi_taxo_get_taxon_from_name_idx(OBIDMS_taxonomy_p taxonomy, int32_t idx);
+
--- a/src/obiview.h
+++ b/src/obiview.h
@ -30,54 +30,56 @@
 #include "obiblob.h"


-#define OBIVIEW_NAME_MAX_LENGTH (249)   		/**< The maximum length of an OBIDMS view name, without the extension.
-                                	 	  	  	 */
-#define VIEW_TYPE_MAX_LENGTH (1024)   			/**< The maximum length of the type name of a view.
-                                	 	  	  	 */
-#define LINES_COLUMN_NAME "LINES"				/**< The name of the column containing the line selections
- 	 	 	 	 	 	 	 	 	 	 	 	 *   in all views.
-                                	 	  	  	 */
-#define VIEW_TYPE_NUC_SEQS "NUC_SEQS_VIEW"   	/**< The type name of views based on nucleotide sequences
-												 *   and their metadata.
-                                	 	  	  	 */
-#define NUC_SEQUENCE_COLUMN "NUC_SEQ"			/**< The name of the column containing the nucleotide sequences
- 	 	 	 	 	 	 	 	 	 	 	 	 *   in NUC_SEQS_VIEW views.
-                                	 	  	  	 */
-#define ID_COLUMN "ID"							/**< The name of the column containing the sequence identifiers
- 	 	 	 	 	 	 	 	 	 	 	 	 *   in NUC_SEQS_VIEW views.
-                                	 	  	  	 */
-#define DEFINITION_COLUMN "DEFINITION"			/**< The name of the column containing the sequence definitions
- 	 	 	 	 	 	 	 	 	 	 	 	 *   in NUC_SEQS_VIEW views.
-                                	 	  	  	 */
-#define QUALITY_COLUMN "QUALITY"				/**< The name of the column containing the sequence qualities
- 	 	 	 	 	 	 	 	 	 	 	 	 *   in NUC_SEQS_VIEW views.
-                                	 	  	   	 */
-#define REVERSE_QUALITY_COLUMN "REVERSE_QUALITY" /**< The name of the column containing the sequence qualities
- 	 	 	 	 	 	 	 	 	 	 	 	 *    of the reverse read (generated by ngsfilter, used by alignpairedend).
-                                	 	  	   	 */
+#define OBIVIEW_NAME_MAX_LENGTH (249)   		   /**< The maximum length of an OBIDMS view name, without the extension.
+                                	 	  	  	    */
+#define VIEW_TYPE_MAX_LENGTH (1024)   			   /**< The maximum length of the type name of a view.
+                                	 	  	  	    */
+#define LINES_COLUMN_NAME "LINES"				   /**< The name of the column containing the line selections
+ 	 	 	 	 	 	 	 	 	 	 	 	    *   in all views.
+                                	 	  	  	    */
+#define VIEW_TYPE_NUC_SEQS "NUC_SEQS_VIEW"   	   /**< The type name of views based on nucleotide sequences
+												    *   and their metadata.
+                                	 	  	  	    */
+#define NUC_SEQUENCE_COLUMN "NUC_SEQ"			   /**< The name of the column containing the nucleotide sequences
+ 	 	 	 	 	 	 	 	 	 	 	 	    *   in NUC_SEQS_VIEW views.
+                                	 	  	  	    */
+#define ID_COLUMN "ID"							   /**< The name of the column containing the sequence identifiers
+ 	 	 	 	 	 	 	 	 	 	 	 	    *   in NUC_SEQS_VIEW views.
+                                	 	  	  	    */
+#define DEFINITION_COLUMN "DEFINITION"			   /**< The name of the column containing the sequence definitions
+ 	 	 	 	 	 	 	 	 	 	 	 	    *   in NUC_SEQS_VIEW views.
+                                	 	  	  	    */
+#define QUALITY_COLUMN "QUALITY"				   /**< The name of the column containing the sequence qualities
+ 	 	 	 	 	 	 	 	 	 	 	 	    *   in NUC_SEQS_VIEW views.
+                                	 	  	   	    */
+#define REVERSE_QUALITY_COLUMN "REVERSE_QUALITY"   /**< The name of the column containing the sequence qualities
+ 	 	 	 	 	 	 	 	 	 	 	 	    *    of the reverse read (generated by ngsfilter, used by alignpairedend).
+                                	 	  	   	    */
 #define REVERSE_SEQUENCE_COLUMN "REVERSE_SEQUENCE" /**< The name of the column containing the sequence
- 	 	 	 	 	 	 	 	 	 	 	 	 *    of the reverse read (generated by ngsfilter, used by alignpairedend).
-                                	 	  	   	 */
-#define QUALITY_COLUMN "QUALITY"				/**< The name of the column containing the sequence qualities
- 	 	 	 	 	 	 	 	 	 	 	 	 *   in NUC_SEQS_VIEW views.
-                                	 	  	   	 */
-#define COUNT_COLUMN "COUNT"				    /**< The name of the column containing the sequence counts
- 	 	 	 	 	 	 	 	 	 	 	 	 *   in NUC_SEQS_VIEW views.
-                                	 	  	  	 */
-#define TAXID_COLUMN "TAXID"				    /**< The name of the column containing the taxids.       TODO subtype of INT column?
-                                	             */
-#define MERGED_TAXID_COLUMN "MERGED_TAXID"		/**< The name of the column containing the merged taxids information.
-                                	             */
-#define MERGED_PREFIX "MERGED_"		            /**< The prefix to prepend to column names when merging informations during obi uniq.
-                                	             */
-#define TAXID_DIST_COLUMN "TAXID_DIST"			/**< The name of the column containing a dictionary of taxid:[list of ids] when merging informations during obi uniq.
-                                	             */
-#define MERGED_COLUMN "MERGED"					/**< The name of the column containing a list of ids when merging informations during obi uniq.
-                                	             */
-#define ID_PREFIX "seq"						    /**< The default prefix of sequence identifiers in automatic ID columns.
-                                	 	  	  	 */
-#define PREDICATE_KEY "predicates"		        /**< The key used in the json-formatted view comments to store predicates.
-                                	 	  	  	 */
+ 	 	 	 	 	 	 	 	 	 	 	 	    *    of the reverse read (generated by ngsfilter, used by alignpairedend).
+                                	 	  	   	    */
+#define QUALITY_COLUMN "QUALITY"				   /**< The name of the column containing the sequence qualities
+ 	 	 	 	 	 	 	 	 	 	 	 	    *   in NUC_SEQS_VIEW views.
+                                	 	  	   	    */
+#define COUNT_COLUMN "COUNT"				       /**< The name of the column containing the sequence counts
+ 	 	 	 	 	 	 	 	 	 	 	 	    *   in NUC_SEQS_VIEW views.
+                                	 	  	  	    */
+#define SCIENTIFIC_NAME_COLUMN "SCIENTIFIC_NAME"   /**< The name of the column containing the taxon scientific name.
+                                	                */
+#define TAXID_COLUMN "TAXID"				       /**< The name of the column containing the taxids.       TODO subtype of INT column?
+                                	                */
+#define MERGED_TAXID_COLUMN "MERGED_TAXID"		   /**< The name of the column containing the merged taxids information.
+                                	                */
+#define MERGED_PREFIX "MERGED_"		               /**< The prefix to prepend to column names when merging informations during obi uniq.
+                                	                */
+#define TAXID_DIST_COLUMN "TAXID_DIST"			   /**< The name of the column containing a dictionary of taxid:[list of ids] when merging informations during obi uniq.
+                                	                */
+#define MERGED_COLUMN "MERGED"					   /**< The name of the column containing a list of ids when merging informations during obi uniq.
+                                	                */
+#define ID_PREFIX "seq"						       /**< The default prefix of sequence identifiers in automatic ID columns.
+                                	 	  	  	    */
+#define PREDICATE_KEY "predicates"		           /**< The key used in the json-formatted view comments to store predicates.
+                                	 	  	  	    */


 /**