-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathgbf_to_prot.py
More file actions
31 lines (24 loc) · 1.31 KB
/
Copy pathgbf_to_prot.py
File metadata and controls
31 lines (24 loc) · 1.31 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
# This code take a .gbf GeneBank file and extract protein sequences from it.
from Bio import SeqIO
# Specifica il percorso del tuo file GenBank
file_path = ""
# Crea un dizionario per memorizzare le sequenze proteiche
protein_sequences = {}
# Apre il file GenBank e lo legge
with open(file_path, "r") as gb_file:
for record in SeqIO.parse(gb_file, "genbank"):
# Estrai il nome dell'organismo
organism = record.annotations["organism"]
# Estrai le sequenze proteiche contrassegnate come "translation"
for feature in record.features:
if "translation" in feature.qualifiers:
protein_sequence = feature.qualifiers["translation"][0]
locus_tag = feature.qualifiers.get("locus_tag", ["N/A"])[0]
product = feature.qualifiers.get("product", ["N/A"])[0]
protein_id = feature.qualifiers.get("protein_id", ["N/A"])[0]
# Aggiungi la sequenza proteica al dizionario
protein_sequences[f">{organism}|{locus_tag}|{product}|{protein_id}"] = protein_sequence
# Scrivi le sequenze proteiche nel file multifasta
with open("protein_output.fasta", "w") as protein_output_file:
for header, sequence in protein_sequences.items():
protein_output_file.write(f"{header}\n{sequence}\n")