From 01d8b5f999ec6eed569389a94a7f86a1ec8383aa Mon Sep 17 00:00:00 2001 From: Sylvester Kaczmarek <16242628+sylvesterkaczmarek@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:35:08 +0100 Subject: [PATCH 1/4] FIX: validate Unicode substitution code points --- pyrit/converter/unicode_sub_converter.py | 19 ++++++++++++++++--- 1 file changed, 16 insertions(+), 3 deletions(-) diff --git a/pyrit/converter/unicode_sub_converter.py b/pyrit/converter/unicode_sub_converter.py index 79ce3eef22..6e77d98991 100644 --- a/pyrit/converter/unicode_sub_converter.py +++ b/pyrit/converter/unicode_sub_converter.py @@ -20,7 +20,12 @@ def __init__(self, *, start_value: int = 0xE0000) -> None: Args: start_value (int): The unicode starting point to use for encoding. + + Raises: + ValueError: If ``start_value`` is outside the Unicode code point range. """ + if not 0 <= start_value <= 0x10FFFF: + raise ValueError("start_value must be a valid Unicode code point between 0 and 0x10FFFF") self.startValue = start_value def _build_identifier(self) -> ComponentIdentifier: @@ -49,10 +54,18 @@ async def convert_async(self, *, prompt: str, input_type: PromptDataType = "text ConverterResult: The result containing the converted output and its type. Raises: - ValueError: If the input type is not supported. + ValueError: If the input type is not supported or a substitution falls outside the Unicode range. """ if not self.input_supported(input_type): raise ValueError("Input type not supported") - ret_text = "".join(chr(self.startValue + ord(ch)) for ch in prompt) - return ConverterResult(output_text=ret_text, output_type="text") + output_chars: list[str] = [] + for char in prompt: + code_point = self.startValue + ord(char) + if code_point > 0x10FFFF: + raise ValueError( + f"Unicode substitution for character {char!r} exceeds the maximum code point U+10FFFF" + ) + output_chars.append(chr(code_point)) + + return ConverterResult(output_text="".join(output_chars), output_type="text") From 9e2d571c0078036176178f53335a467c12d091d9 Mon Sep 17 00:00:00 2001 From: Sylvester Kaczmarek <16242628+sylvesterkaczmarek@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:35:30 +0100 Subject: [PATCH 2/4] TST: cover invalid Unicode substitution ranges --- tests/unit/converter/test_unicode_sub_converter.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/tests/unit/converter/test_unicode_sub_converter.py b/tests/unit/converter/test_unicode_sub_converter.py index f76a2df660..1d2f1860ef 100644 --- a/tests/unit/converter/test_unicode_sub_converter.py +++ b/tests/unit/converter/test_unicode_sub_converter.py @@ -22,6 +22,19 @@ async def test_unicode_sub_custom_start(): assert result.output_type == "text" +@pytest.mark.parametrize("start_value", [-1, 0x110000]) +def test_unicode_sub_rejects_invalid_start_value(start_value): + with pytest.raises(ValueError, match="valid Unicode code point"): + UnicodeSubstitutionConverter(start_value=start_value) + + +async def test_unicode_sub_rejects_derived_code_point_overflow(): + converter = UnicodeSubstitutionConverter(start_value=0x10FFFF) + + with pytest.raises(ValueError, match="exceeds the maximum code point"): + await converter.convert_async(prompt="a", input_type="text") + + async def test_unicode_sub_empty(): converter = UnicodeSubstitutionConverter() result = await converter.convert_async(prompt="", input_type="text") From 1391c74b807222d099258356e49f3b6b242215b9 Mon Sep 17 00:00:00 2001 From: Sylvester Kaczmarek <16242628+sylvesterkaczmarek@users.noreply.github.com> Date: Sat, 12 Sep 2026 07:55:32 -0400 Subject: [PATCH 3/4] Address review: keep Unicode validation in constructor --- pyrit/converter/unicode_sub_converter.py | 14 +++----------- 1 file changed, 3 insertions(+), 11 deletions(-) diff --git a/pyrit/converter/unicode_sub_converter.py b/pyrit/converter/unicode_sub_converter.py index 6e77d98991..f16a1ed8c0 100644 --- a/pyrit/converter/unicode_sub_converter.py +++ b/pyrit/converter/unicode_sub_converter.py @@ -54,18 +54,10 @@ async def convert_async(self, *, prompt: str, input_type: PromptDataType = "text ConverterResult: The result containing the converted output and its type. Raises: - ValueError: If the input type is not supported or a substitution falls outside the Unicode range. + ValueError: If the input type is not supported. """ if not self.input_supported(input_type): raise ValueError("Input type not supported") - output_chars: list[str] = [] - for char in prompt: - code_point = self.startValue + ord(char) - if code_point > 0x10FFFF: - raise ValueError( - f"Unicode substitution for character {char!r} exceeds the maximum code point U+10FFFF" - ) - output_chars.append(chr(code_point)) - - return ConverterResult(output_text="".join(output_chars), output_type="text") + ret_text = "".join(chr(self.startValue + ord(ch)) for ch in prompt) + return ConverterResult(output_text=ret_text, output_type="text") From ed510044f2329e6c834eb6495b2e4579ca0f1bb8 Mon Sep 17 00:00:00 2001 From: Sylvester Kaczmarek <16242628+sylvesterkaczmarek@users.noreply.github.com> Date: Sat, 12 Sep 2026 07:55:40 -0400 Subject: [PATCH 4/4] Address review: remove runtime overflow regression test --- tests/unit/converter/test_unicode_sub_converter.py | 7 ------- 1 file changed, 7 deletions(-) diff --git a/tests/unit/converter/test_unicode_sub_converter.py b/tests/unit/converter/test_unicode_sub_converter.py index 1d2f1860ef..3084550fc6 100644 --- a/tests/unit/converter/test_unicode_sub_converter.py +++ b/tests/unit/converter/test_unicode_sub_converter.py @@ -28,13 +28,6 @@ def test_unicode_sub_rejects_invalid_start_value(start_value): UnicodeSubstitutionConverter(start_value=start_value) -async def test_unicode_sub_rejects_derived_code_point_overflow(): - converter = UnicodeSubstitutionConverter(start_value=0x10FFFF) - - with pytest.raises(ValueError, match="exceeds the maximum code point"): - await converter.convert_async(prompt="a", input_type="text") - - async def test_unicode_sub_empty(): converter = UnicodeSubstitutionConverter() result = await converter.convert_async(prompt="", input_type="text")