Skip to content

Commit 0d429aa

Browse files
authored
Merge pull request #217 from BioGeek/fix/proforma-spec-conformance
Enforce terminal grammar boundaries
2 parents a955db2 + 4c1b19b commit 0d429aa

2 files changed

Lines changed: 87 additions & 4 deletions

File tree

‎pyteomics/proforma.py‎

Lines changed: 45 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -2947,7 +2947,8 @@ def handle_tag(self, c: str):
29472947
elif self.state == TAG_BEFORE:
29482948
self.state = POST_TAG_BEFORE
29492949
elif self.state == TAG_AFTER:
2950-
self.c_term = self.current_tag()
2950+
tags = self.current_tag()
2951+
self.c_term.extend(tags)
29512952
self.state = POST_TAG_AFTER
29522953
elif self.state == GLOBAL:
29532954
self.state = POST_GLOBAL
@@ -3035,6 +3036,17 @@ def handle_post_tag_before(self, c: str):
30353036
self.unlocalized_modifications.extend(self.current_tag())
30363037
self.state = BEFORE
30373038
elif c == "-":
3039+
if self.n_term:
3040+
raise ProFormaError(
3041+
(
3042+
f"Error In State {self.state}, found a second N-terminal modification group "
3043+
f"at index {self.index}; ProForma allows only one N-terminal group. "
3044+
"Put multiple N-terminal modifications in that group, e.g. "
3045+
"'[UNIMOD:5][UNIMOD:385]-PEPTIDE'"
3046+
),
3047+
self.index,
3048+
self.state,
3049+
)
30383050
self.n_term = self.current_tag()
30393051
self.state = BEFORE
30403052
elif c == "^":
@@ -3125,6 +3137,15 @@ def handle_post_tag_after(self, c: str):
31253137
self.charge_buffer = NumberParser()
31263138
elif c == "+":
31273139
self._handle_chimeric_separator()
3140+
elif c == "[":
3141+
self.state = TAG_AFTER
3142+
self.depth = 1
3143+
else:
3144+
raise ProFormaError(
3145+
f"Error In State {self.state}, unexpected {c} found at index {self.index}",
3146+
self.index,
3147+
self.state,
3148+
)
31283149

31293150
def handle_charge_start(self, c: str):
31303151
if c in "+-":
@@ -3184,6 +3205,12 @@ def handle_adduct_start(self, c: str):
31843205
def handle_adduct_end(self, c: str):
31853206
if c == "+":
31863207
self._handle_chimeric_separator()
3208+
else:
3209+
raise ProFormaError(
3210+
f"Error In State {self.state}, unexpected {c} found at index {self.index}",
3211+
self.index,
3212+
self.state,
3213+
)
31873214

31883215
def handle_name_level(self, c: str):
31893216
if c == '>' and self.name_level < 3:
@@ -3283,6 +3310,23 @@ def step(self) -> bool:
32833310
return self.index < self.length
32843311

32853312
def _finish_component(self) -> ProFormaParseResult:
3313+
if not self.positions and self.current_aa is None:
3314+
if self.chimeric:
3315+
raise ProFormaError("Empty peptidoform in chimeric ProForma string", self.index, self.state)
3316+
raise ProFormaError(
3317+
"A ProForma peptidoform must contain at least one amino acid",
3318+
self.index,
3319+
self.state,
3320+
)
3321+
3322+
complete_states = (SEQ, POST_INTERVAL_TAG, POST_TAG_AFTER, CHARGE_NUMBER, ADDUCT_END)
3323+
if self.state not in complete_states:
3324+
raise ProFormaError(
3325+
f"Error In State {self.state}, incomplete ProForma string reached end of input",
3326+
self.index,
3327+
self.state,
3328+
)
3329+
32863330
if self.charge_buffer:
32873331
charge_number = self.charge_buffer()
32883332
if self.adduct_buffer:
@@ -3298,9 +3342,6 @@ def _finish_component(self) -> ProFormaParseResult:
32983342
if self.current_aa:
32993343
self.pack_sequence_position()
33003344

3301-
if not self.positions and self.chimeric:
3302-
raise ProFormaError("Empty peptidoform in chimeric ProForma string", self.index, self.state)
3303-
33043345
z, k = self._local_charges()
33053346
if k:
33063347
if charge_state is None:

‎tests/test_proforma.py‎

Lines changed: 42 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -4,6 +4,7 @@
44
import unittest
55
import pickle
66
import math
7+
from itertools import product
78
import pyteomics
89
pyteomics.__path__ = [path.abspath(
910
path.join(path.dirname(__file__), path.pardir, 'pyteomics'))]
@@ -26,6 +27,47 @@
2627
class ProFormaTest(unittest.TestCase):
2728
maxDiff = None
2829

30+
def test_spec_shaped_peptidoform_conformance_matrix(self):
31+
"""Exercise combinations assembled from the ProForma grammar.
32+
33+
This deterministic matrix covers optional prefix and suffix productions.
34+
Each input has a ``sequence`` production, so the parser must accept it.
35+
"""
36+
prefixes = (
37+
"", "{+1}", "[+1]?", "[+1]^2?", "[+1]-", "[+1][+2]-",
38+
"[+1][+2][+3]-", "<13C>", "<[+1]@C>",
39+
)
40+
sequences = ("PEPTIDE", "PEP[+1]TIDE", "(PEP)[+1]TIDE", "(?PEP)TIDE")
41+
suffixes = ("", "-[+1]", "-[+1][+2]", "-[+1][+2][+3]", "/2", "/2[+H+]")
42+
for prefix, sequence, suffix in product(prefixes, sequences, suffixes):
43+
with self.subTest(peptidoform=prefix + sequence + suffix):
44+
ProForma.parse(prefix + sequence + suffix)
45+
46+
# Positive examples added with HUPO-PSI/ProForma issue #33.
47+
for peptidoform in (
48+
"[Acetyl][Acetyl][Carbamyl]-QPEPTIDE",
49+
"PEPTIDEG-[Methyl][Amidated][INFO:A lot of C terminal mods]",
50+
):
51+
with self.subTest(peptidoform=peptidoform):
52+
ProForma.parse(peptidoform)
53+
54+
def test_spec_shaped_invalid_peptidoform_conformance_matrix(self):
55+
"""Reject grammar-breaking mutations of otherwise valid productions."""
56+
invalid_peptidoforms = (
57+
"", # no required sequence
58+
"[+1]", # a pre-sequence tag needs ?/^n? or -
59+
"[+1]-", # N-terminus needs a sequence afterwards
60+
"{+1}", # labile tag without a sequence
61+
"<13C>", # isotope rule without a sequence
62+
"[+1]-[+2]-PEPTIDE", # repeated modNTerm group
63+
"PEPTIDE-[+1]X", # no content may follow modCTerm
64+
"PEPTIDE/2[+H+]X", # no content may follow an adduct list
65+
)
66+
for peptidoform in invalid_peptidoforms:
67+
with self.subTest(peptidoform=peptidoform):
68+
with self.assertRaises(ProFormaError):
69+
ProForma.parse(peptidoform)
70+
2971
def test_complicated_short(self):
3072
complicated_short = r"<[Carbamidomethyl]@C><13C>[Hydroxylation]?{HexNAc}[Hex]-ST[UNIMOD:Oxidation](EPP)[+18.15]ING"
3173
tokens, properties = parse(complicated_short)

0 commit comments

Comments
 (0)