From b186bc13e02b8aab02c78297de05127a7dfd5116 Mon Sep 17 00:00:00 2001 From: lindsay stevens Date: Tue, 15 Sep 2026 21:58:41 +1000 Subject: [PATCH] fix: missing opening square bracket on unicode character set - test case includes characters in the broken range as an example. The intent of the regex is to match XML so it should have allowed them. - not quite a regression since although typo has been present since 2024, the previous regex validation only allowed ascii characters. Also the error message just says "letters" are allowed rather than specifying that a particular type of letter is supported. --- pyxform/parsing/expression.py | 2 +- tests/test_survey_element.py | 4 ++++ 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/pyxform/parsing/expression.py b/pyxform/parsing/expression.py index c43a5dac..9ce27f87 100644 --- a/pyxform/parsing/expression.py +++ b/pyxform/parsing/expression.py @@ -9,7 +9,7 @@ # They in turn adapted it from https://www.w3.org/TR/REC-xml/#NT-NameStartChar # and https://www.w3.org/TR/REC-xml-names/#NT-NCName namestartchar = ( - r"(?:[A-Z]|_|[a-z]|\xc0-\xd6]|[\xd8-\xf6]|[\xf8-\u02ff]|" + r"(?:[A-Z]|_|[a-z]|[\xc0-\xd6]|[\xd8-\xf6]|[\xf8-\u02ff]|" + r"[\u0370-\u037d]|[\u037f-\u1fff]|[\u200c-\u200d]|[\u2070-\u218f]|" + r"[\u2c00-\u2fef]|[\u3001-\uD7FF]|[\uF900-\uFDCF]|[\uFDF0-\uFFFD]" + r"|[\U00010000-\U000EFFFF])" diff --git a/tests/test_survey_element.py b/tests/test_survey_element.py index 41754411..5c2d6880 100644 --- a/tests/test_survey_element.py +++ b/tests/test_survey_element.py @@ -75,3 +75,7 @@ def test_validate__invalid_name__error(self): self.assertEqual( ErrorCode.NAMES_009.value.format(name=co.NAME), err.exception.args[0] ) + + def test_validate__name_in_valid_unicode_range__ok(self): + """Should not raise an error if the 'name' includes valid unicode characters.""" + SurveyElement(name="SäöüÄÖÜß", label="Q1").validate()