Created
May 4, 2026 12:32
-
-
Save samuell/8c6a3eff3f633b515e08a2588171326d to your computer and use it in GitHub Desktop.
Prepare the UNITE database for Emu
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| from Bio import SeqIO | |
| import pandas as pd | |
| ## define variables | |
| input_fa_path = "./sh_general_release_dynamic_19.02.2025_dev.fasta" | |
| output_dir_path = "./" | |
| database_name = "unite-eukaryotes" | |
| tax_headers = ['superkingdom', 'phylum', 'class', 'order', 'family', 'genus', 'species'] | |
| output_records = [] | |
| file_type = "fasta" | |
| seq_counter, tax_counter = 1, 1 | |
| seq_id_list, tax_id_list = [], [] | |
| tax_dict = {} | |
| # gather dict of unique tax lineages and assign incremented tax_id | |
| for record in SeqIO.parse(input_fa_path, file_type): | |
| rid = record.id | |
| tax_lineage = rid.split("|")[4] | |
| # if new tax_lineage, add new entry | |
| if tax_lineage not in tax_dict: | |
| tax_dict[tax_lineage] = tax_counter | |
| tax_counter = tax_counter + 1 | |
| # update seq_id info for new fasta file | |
| seq_id = f"{database_name}_{seq_counter}" | |
| record.id = seq_id | |
| seq_counter = seq_counter + 1 | |
| seq_id_list.append(seq_id) | |
| tax_id_list.append(tax_dict[tax_lineage]) | |
| output_records.append(record) | |
| # write seq2tax map | |
| SeqIO.write(output_records,f"{output_dir_path}/emu-unite-fungi-input.fa", "fasta") | |
| seq_tax_df = pd.DataFrame({'seq_id':seq_id_list, | |
| 'tax_id':tax_id_list}) | |
| # write taxononmy lineage .tsv | |
| seq_tax_df.to_csv(f"{output_dir_path}/seq2tax.map", sep='\t', index=False) | |
| taxonomy_df = pd.DataFrame({'tax_id':tax_dict.values(), | |
| 'lineage':tax_dict.keys()}) | |
| tax_temp_df = taxonomy_df['lineage'].str.split(';', expand=True) | |
| tax_temp_df = tax_temp_df.rename(columns={0:'kingdom', 1:'phylum', | |
| 2:'class', 3:'order', | |
| 4:'family', 5:'genus', | |
| 6:'species'}) | |
| taxonomy_df[tax_headers] = tax_temp_df | |
| taxonomy_df = taxonomy_df.drop(columns='lineage') | |
| # remove rank labeling in tax rank names | |
| for col in taxonomy_df.columns: | |
| if col != 'tax_id': | |
| taxonomy_df[col] = taxonomy_df[col].apply(lambda x:x.split("__",1)[1]) | |
| taxonomy_df = taxonomy_df.iloc[:, [0,7,6,5,4,3,2,1]] | |
| taxonomy_df.to_csv(f"{output_dir_path}/tax.tsv", sep='\t', index=False) |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment