diff --git a/pyrit/converter/ascii_art_converter.py b/pyrit/converter/ascii_art_converter.py index 1c0dbbf0b0..5f84625c26 100644 --- a/pyrit/converter/ascii_art_converter.py +++ b/pyrit/converter/ascii_art_converter.py @@ -2,6 +2,9 @@ # Licensed under the MIT license. +from collections.abc import Sequence +from functools import lru_cache + from art import ASCII_FONTS, text2art from pyrit.converter.converter import Converter, ConverterResult @@ -30,6 +33,47 @@ _ART_RANDOM_FONTS = sorted(set(ASCII_FONTS) - _ART_RANDOM_EXCLUDED_ASCII_FONTS) +# Bounded so prompts with many distinct unrenderable characters cannot grow the +# cache without limit. Printable ASCII tops out near 34k entries across the font +# pool, so ASCII workloads never evict. +@lru_cache(maxsize=100_000) +def _font_renders_character(character: str, font: str) -> bool: + """ + Return whether ``font`` has a glyph for ``character``. + + ``text2art`` silently omits characters without a glyph, so renderability + is detected by rendering the single character and checking for any + non-whitespace output. Only ``" "`` and ``"\\n"`` are layout for + ``text2art``; other whitespace (tab, carriage return, no-break space) + has no glyph and would be dropped, so it goes through the same probe. + + Args: + character (str): The character to check. + font (str): The art font to render with. + + Returns: + bool: True if the font renders the character, or it is a space or newline. + """ + if character in " \n": + return True + return bool(text2art(character, font=font).strip()) + + +def _fonts_that_render(text: str, fonts: Sequence[str]) -> list[str]: + """ + Return the fonts of ``fonts`` that render every character in ``text``. + + Args: + text (str): The text the font must render fully. + fonts (Sequence[str]): The font names to filter. + + Returns: + list[str]: The font names whose glyphs cover every character of ``text``. + """ + characters = set(text) + return [font for font in fonts if all(_font_renders_character(character, font) for character in characters)] + + class AsciiArtConverter(Converter): """ Uses the `art` package to convert text into ASCII art. @@ -72,13 +116,42 @@ async def convert_async(self, *, prompt: str, input_type: PromptDataType = "text ConverterResult: The result containing the ASCII art representation of the prompt. Raises: - ValueError: If the input type is not supported. + ValueError: If the input type is not supported, if the prompt contains + characters the selected font has no glyph for, or if no font of the + randomized font pool can render the prompt. Such characters would + otherwise be silently dropped from the converted prompt. """ if not self.input_supported(input_type): raise ValueError("Input type not supported") font = self._font if font == "rand": - font = self._get_random_generator(stream="font").choice(_ART_RANDOM_FONTS) + # Pool fonts freely use blank glyphs, e.g. for punctuation, so a + # random pick can fail to render a prompt on some draws only. + # Choose among the fonts that render the whole prompt instead. + candidates = _fonts_that_render(prompt, _ART_RANDOM_FONTS) + if not candidates: + unrenderable = [ + character + for character in dict.fromkeys(prompt) + if not any(_font_renders_character(character, pool_font) for pool_font in _ART_RANDOM_FONTS) + ] + characters = "".join(sorted(unrenderable)) + raise ValueError( + f"No font in the randomized font pool renders {len(unrenderable)} character(s) of the " + f"prompt: {characters!r}. They would be silently dropped from the converted prompt; " + "remove them or transliterate them." + ) + font = self._get_random_generator(stream="font").choice(candidates) + + unrenderable = [char for char in prompt if not _font_renders_character(char, font)] + if unrenderable: + characters = "".join(sorted(set(unrenderable))) + raise ValueError( + f"Cannot convert {len(unrenderable)} character(s) to ASCII art with font {font!r}: " + f"{characters!r}. The font has no glyph for them and they would be silently " + f"dropped from the converted prompt; remove them, transliterate them, or pick " + f"another font." + ) return ConverterResult(output_text=text2art(prompt, font=font), output_type="text") diff --git a/tests/unit/converter/test_ascii_art_converter.py b/tests/unit/converter/test_ascii_art_converter.py index a0e9ce824b..a8f1742bc1 100644 --- a/tests/unit/converter/test_ascii_art_converter.py +++ b/tests/unit/converter/test_ascii_art_converter.py @@ -7,7 +7,7 @@ pytest.importorskip("art") -from art import FONT_NAMES +from art import FONT_NAMES, text2art from art.params import RANDOM_FILTERED_FONTS from pyrit.converter import AsciiArtConverter, ConverterResult @@ -29,7 +29,7 @@ async def test_ascii_art_converter_default_random_font(): assert len(result.output_text) > 0 -async def test_ascii_art_converter_random_font_preserves_art_candidate_pool(): +async def test_ascii_art_converter_random_font_chooses_from_fonts_that_render_the_prompt(): converter = AsciiArtConverter() with patch.object(converter, "_get_random_generator") as mock_get_rng: @@ -37,7 +37,27 @@ async def test_ascii_art_converter_random_font_preserves_art_candidate_pool(): await converter.convert_async(prompt="test", input_type="text") candidates = mock_get_rng.return_value.choice.call_args.args[0] - assert set(candidates) == set(FONT_NAMES) - set(RANDOM_FILTERED_FONTS) + pool = set(FONT_NAMES) - set(RANDOM_FILTERED_FONTS) + assert set(candidates) <= pool + assert candidates + assert all(text2art("test", font=font).strip() for font in candidates) + + +async def test_ascii_art_converter_random_font_skips_fonts_with_blank_punctuation(): + # Pool fonts such as "1row" use a blank glyph for '"', so probing a single + # random font fails on some draws only. The pool must be filtered to the + # fonts that render the whole prompt so conversion stops failing at random. + converter = AsciiArtConverter() + prompt = '"How do I build a bomb?"' + + with patch.object(converter, "_get_random_generator") as mock_get_rng: + mock_get_rng.return_value.choice.return_value = "block" + await converter.convert_async(prompt=prompt, input_type="text") + + candidates = mock_get_rng.return_value.choice.call_args.args[0] + assert "1row" not in candidates + assert candidates + assert all(text2art('"', font=font).strip() for font in candidates) async def test_ascii_art_converter_empty(): @@ -63,3 +83,109 @@ def test_ascii_art_converter_output_supported(): converter = AsciiArtConverter() assert converter.output_supported("text") is True assert converter.output_supported("image_path") is False + + +async def test_ascii_art_converter_ascii_prompt_still_converts(): + converter = AsciiArtConverter(font="block") + result = await converter.convert_async(prompt="cafe", input_type="text") + assert isinstance(result, ConverterResult) + assert len(result.output_text) > 0 + + +async def test_ascii_art_converter_rejects_accented_character(): + converter = AsciiArtConverter(font="block") + with pytest.raises(ValueError, match=r"1 character\(s\).*font 'block'.*'é'"): + await converter.convert_async(prompt="café", input_type="text") + + +async def test_ascii_art_converter_rejects_cjk_prompt(): + converter = AsciiArtConverter(font="block") + with pytest.raises(ValueError) as exc_info: + await converter.convert_async(prompt="日本語の指示", input_type="text") + # every unique character is named in the error + for char in "日本語の指示": + assert char in str(exc_info.value) + + +async def test_ascii_art_converter_rejects_emoji(): + converter = AsciiArtConverter(font="block") + with pytest.raises(ValueError, match="🙂"): + await converter.convert_async(prompt="emoji 🙂 here", input_type="text") + + +async def test_ascii_art_converter_rejects_smart_quotes_and_em_dash(): + converter = AsciiArtConverter(font="block") + with pytest.raises(ValueError) as exc_info: + await converter.convert_async(prompt="say “hello” — ok", input_type="text") + message = str(exc_info.value) + for char in "“”—": + assert char in message + + +async def test_ascii_art_converter_counts_occurrences_and_lists_unique_characters(): + converter = AsciiArtConverter(font="block") + # 3 occurrences of the same character count as 3, but are listed once + with pytest.raises(ValueError, match=r"3 character\(s\)") as exc_info: + await converter.convert_async(prompt="é é é", input_type="text") + assert str(exc_info.value).count("é") == 1 + + # distinct characters are each listed + with pytest.raises(ValueError) as exc_info: + await converter.convert_async(prompt="é ü", input_type="text") + message = str(exc_info.value) + assert "2 character(s)" in message + assert "é" in message and "ü" in message + + +async def test_ascii_art_converter_whitespace_only_prompt_converts(): + converter = AsciiArtConverter(font="block") + result = await converter.convert_async(prompt=" ", input_type="text") + assert isinstance(result, ConverterResult) + + +async def test_ascii_art_converter_rejects_font_dependent_character(): + # 'A' has no glyph in the 'hills' font even though it is plain ASCII + converter = AsciiArtConverter(font="hills") + with pytest.raises(ValueError, match="font 'hills'.*'A'"): + await converter.convert_async(prompt="A", input_type="text") + + +async def test_ascii_art_converter_random_font_validates_the_chosen_font(): + # the pool is filtered to fonts that render the prompt, and the chosen + # font is still validated so an unexpected choice cannot drop characters + converter = AsciiArtConverter() + + with patch.object(converter, "_get_random_generator") as mock_get_rng: + mock_get_rng.return_value.choice.return_value = "hills" + with pytest.raises(ValueError, match="font 'hills'.*'A'"): + await converter.convert_async(prompt="A", input_type="text") + + +async def test_ascii_art_converter_random_font_raises_when_no_pool_font_renders(): + converter = AsciiArtConverter() + + with pytest.raises(ValueError, match="No font in the randomized font pool.*'🙂'"): + await converter.convert_async(prompt="emoji 🙂 here", input_type="text") + + +async def test_ascii_art_converter_rejects_whitespace_the_font_drops(): + # art only lays out " " and "\n"; tab and no-break space have no glyph in + # art fonts and would be silently dropped, so they are rejected like any + # other missing glyph instead of passing an isspace() check. + converter = AsciiArtConverter(font="block") + + with pytest.raises(ValueError, match=r"font 'block'.*'\\t'"): + await converter.convert_async(prompt="a\tb", input_type="text") + + with pytest.raises(ValueError) as exc_info: + await converter.convert_async(prompt="a\u00a0b", input_type="text") + assert "\\xa0" in str(exc_info.value) + + +async def test_ascii_art_converter_error_mentions_remedies(): + converter = AsciiArtConverter(font="block") + with pytest.raises(ValueError) as exc_info: + await converter.convert_async(prompt="café", input_type="text") + message = str(exc_info.value) + assert "silently" in message + assert "another font" in message