Skip to content

Commit 886ea0d

Browse files
authored
Merge pull request #195 from frescobaldi/fix-ascii-digit-regexes
Match only ASCII digits in the lexer number patterns
2 parents ac3db74 + e05d8dd commit 886ea0d

4 files changed

Lines changed: 79 additions & 13 deletions

File tree

‎ly/lex/html.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -95,8 +95,9 @@ class StringSQEnd(String, _token.StringEnd, _token.Leaver):
9595
rx = r"'"
9696

9797

98+
# [0-9], not \d: a numeric character reference is ASCII digits only
9899
class EntityRef(_token.Character):
99-
rx = r"\&(#\d+|#[xX][0-9A-Fa-f]+|[A-Za-z_:][\w.:_-]*);"
100+
rx = r"\&(#[0-9]+|#[xX][0-9A-Fa-f]+|[A-Za-z_:][\w.:_-]*);"
100101

101102

102103
class LilyPondTag(Tag):

‎ly/lex/lilypond.py‎

Lines changed: 12 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -28,6 +28,9 @@
2828
from . import _token
2929
from . import Parser, FallthroughParser
3030

31+
# digit patterns use [0-9]: \d also matches Unicode digits that are not legal
32+
# LilyPond. \d in a negated class excludes those too, so it stays
33+
3134
# an identifier allowing letters and single hyphens in between
3235
re_identifier = r"[^\W\d_]+([_-][^\W\d_]+)*"
3336

@@ -45,7 +48,7 @@
4548
re_duration = rf"(\\(maxima|longa|breve){re_identifier_end}|(1|2|4|8|16|32|64|128|256|512|1024|2048)(?!\d))"
4649

4750
re_dot = r"\."
48-
re_scaling = r"\*[\t ]*\d+(/\d+)?"
51+
re_scaling = r"\*[\t ]*[0-9]+(/[0-9]+)?"
4952

5053

5154

@@ -72,15 +75,15 @@ class Value(_token.Item, _token.Numeric):
7275

7376

7477
class DecimalValue(Value):
75-
rx = r"-?\d+(\.\d+)?"
78+
rx = r"-?[0-9]+(\.[0-9]+)?"
7679

7780

7881
class IntegerValue(DecimalValue):
79-
rx = r"\d+"
82+
rx = r"[0-9]+"
8083

8184

8285
class Fraction(Value):
83-
rx = r"\d+/\d+"
86+
rx = r"[0-9]+/[0-9]+"
8487

8588

8689
class Delimiter(_token.Token):
@@ -289,11 +292,11 @@ class ScriptAbbreviation(Articulation, _token.Leaver):
289292

290293

291294
class Fingering(Articulation, _token.Leaver):
292-
rx = r"\d+"
295+
rx = r"[0-9]+"
293296

294297

295298
class StringNumber(Articulation):
296-
rx = r"\\\d+"
299+
rx = r"\\[0-9]+"
297300

298301

299302
class Slur(_token.Token):
@@ -379,7 +382,7 @@ class ChordSeparator(ChordItem):
379382

380383

381384
class ChordStepNumber(ChordItem):
382-
rx = r"\d+[-+]?"
385+
rx = r"[0-9]+[-+]?"
383386

384387

385388
class DotChord(ChordItem):
@@ -571,7 +574,7 @@ def update_state(self, state):
571574

572575

573576
class TempoSeparator(Delimiter):
574-
rx = r"[-~](?=\s*\d)"
577+
rx = r"[-~](?=\s*[0-9])"
575578

576579

577580
class Partial(Command):
@@ -740,7 +743,7 @@ class FigureBracket(Figure):
740743

741744
class FigureStep(Figure):
742745
"""A step figure number or the underscore."""
743-
rx = r"_|\d+"
746+
rx = r"_|[0-9]+"
744747

745748

746749
class FigureAccidental(Figure):

‎ly/lex/scheme.py‎

Lines changed: 4 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -138,20 +138,21 @@ def test_match(cls, match):
138138
return match.group() in data.scheme_constants()
139139

140140

141+
# [0-9], not \d, which also matches Unicode digits that are not legal LilyPond
141142
class Number(_token.Item, _token.Numeric):
142143
rx = (r"("
143-
r"-?\d+|"
144+
r"-?[0-9]+|"
144145
r"#(b[0-1]+|o[0-7]+|x[0-9a-fA-F]+)|"
145146
r"[-+]inf.0|[-+]?nan.0"
146147
r")(?=$|[)\s])")
147148

148149

149150
class Fraction(Number):
150-
rx = r"-?\d+/\d+(?=$|[)\s])"
151+
rx = r"-?[0-9]+/[0-9]+(?=$|[)\s])"
151152

152153

153154
class Float(Number):
154-
rx = r"-?((\d+(\.\d*)|\.\d+)(E\d+)?)(?=$|[)\s])"
155+
rx = r"-?(([0-9]+(\.[0-9]*)|\.[0-9]+)(E[0-9]+)?)(?=$|[)\s])"
155156

156157

157158
class VectorStart(OpenParen):

‎tests/test_lex.py‎

Lines changed: 61 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,61 @@
1+
"""Tests for the lexers."""
2+
import ly.document
3+
import ly.lex._token
4+
import ly.lex.html
5+
import ly.lex.lilypond
6+
7+
8+
DIGIT_TOKENS = (ly.lex._token.Numeric,
9+
ly.lex.lilypond.Fingering,
10+
ly.lex.lilypond.StringNumber,
11+
ly.lex.lilypond.ChordStepNumber,
12+
ly.lex.lilypond.FigureStep,
13+
ly.lex.lilypond.TempoSeparator,
14+
ly.lex.lilypond.Scaling)
15+
16+
TO_FULLWIDTH = str.maketrans('0123456789', '0123456789')
17+
18+
SOURCES = [
19+
'#(display 42)',
20+
'#(x 1/2)',
21+
'#(x 1.5)',
22+
'\\tempo 4 = 60 - 80',
23+
'\\time 3/4',
24+
'c4',
25+
'c-4',
26+
'\\override NoteHead.font-size = 2',
27+
'\\chords { c1:5 }',
28+
'\\figures { <1> }',
29+
'\\score { { c1*2 } }',
30+
'\\score { \\new TabStaff { c4\\4 } }',
31+
]
32+
33+
34+
def digit_tokens(text, mode=None):
35+
"""Return the text of every token the lexer only reads as such because of
36+
the digits in it."""
37+
doc = ly.document.Document(text, mode)
38+
return [str(t) for block in doc
39+
for t in doc.tokens(block) if isinstance(t, DIGIT_TOKENS)]
40+
41+
42+
def test_ascii_digits_are_read():
43+
for source in SOURCES:
44+
assert digit_tokens(source), source
45+
46+
47+
def test_fullwidth_digits_are_not_read():
48+
for source in SOURCES:
49+
fullwidth = source.translate(TO_FULLWIDTH)
50+
assert digit_tokens(fullwidth) == [], fullwidth
51+
52+
53+
def entity_refs(text):
54+
doc = ly.document.Document(text, 'html')
55+
return [str(t) for block in doc
56+
for t in doc.tokens(block) if isinstance(t, ly.lex.html.EntityRef)]
57+
58+
59+
def test_only_ascii_digits_form_a_numeric_entity():
60+
assert entity_refs('&#65;') == ['&#65;']
61+
assert entity_refs('&#65;'.translate(TO_FULLWIDTH)) == []

0 commit comments

Comments
 (0)