From aab21be979c6dd3a459247025345dd8766719e65 Mon Sep 17 00:00:00 2001 From: Serhiy Storchaka Date: Fri, 24 Jul 2026 10:24:41 +0300 Subject: [PATCH] gh-154582: Fix test_invalid_utf8_arg in non-UTF-8 multibyte locales Arbitrary bytes round-trip through surrogateescape only in UTF-8 and single-byte encodings, not in a stateful multibyte encoding such as EUC-JP. Skip the run_default and run_no_utf8_mode variants when their encoding is not lossless. Co-Authored-By: Claude Opus 4.8 --- Lib/test/test_cmd_line.py | 25 ++++++++++++++++++++++--- 1 file changed, 22 insertions(+), 3 deletions(-) diff --git a/Lib/test/test_cmd_line.py b/Lib/test/test_cmd_line.py index 7640e50f19b7839..25d6d1a248b4577 100644 --- a/Lib/test/test_cmd_line.py +++ b/Lib/test/test_cmd_line.py @@ -2,6 +2,7 @@ # Most tests are executed with environment variables ignored # See test_cmd_line_script.py for testing of script execution +import locale import os import re import subprocess @@ -369,9 +370,27 @@ def run_no_utf8_mode(arg): ) test_args = [valid_utf8, invalid_utf8] - for run_cmd in (run_default, run_c_locale, run_utf8_mode, - run_no_utf8_mode): - with self.subTest(run_cmd=run_cmd): + for run_cmd, encoding in ( + (run_default, sys.getfilesystemencoding()), + (run_c_locale, None), + (run_utf8_mode, None), + (run_no_utf8_mode, locale.getencoding()) + ): + with self.subTest(run_cmd=run_cmd.__name__): + # Arbitrary bytes round-trip through surrogateescape only in + # UTF-8 and single-byte encodings, not in a multibyte encoding + # such as EUC-JP. + if encoding is not None: + try: + lossless = len(bytes(range(256)).decode( + encoding, 'surrogateescape')) == 256 + except UnicodeError: + lossless = False + else: + lossless = True + if not lossless: + self.skipTest(f'{encoding} cannot losslessly ' + f'round-trip arbitrary bytes') for arg in test_args: proc = run_cmd(arg) self.assertEqual(proc.stdout.rstrip(), ascii(arg))