From 61dffa79eba0bdd713291f00aeca8ad7a5cc365f Mon Sep 17 00:00:00 2001 From: Vyron Vasileiadis Date: Sat, 22 Aug 2026 01:41:04 +0300 Subject: [PATCH] gh-156207: Fix curses Textbox.gather() for double-width characters gather() read the window one cell at a time, so the second cell of a double-width character was reported as another copy of it. Read each line with in_wchstr() instead: its count is a cell count, and it skips the continuation cell. --- Lib/curses/textpad.py | 6 ++---- Lib/test/test_curses.py | 16 ++++++++++++++++ 2 files changed, 18 insertions(+), 4 deletions(-) diff --git a/Lib/curses/textpad.py b/Lib/curses/textpad.py index f6dfc990901d995..54d13cf72fbcbcd 100644 --- a/Lib/curses/textpad.py +++ b/Lib/curses/textpad.py @@ -204,10 +204,8 @@ def gather(self): stop = self._end_of_line(y) if stop == 0 and self.stripspaces: continue - for x in range(self.maxx+1): - if self.stripspaces and x > stop: - break - result = result + str(self.win.in_wch(y, x)) + count = stop+1 if self.stripspaces else self.maxx+1 + result = result + str(self.win.in_wchstr(y, 0, count)) if self.maxy > 0: result = result + "\n" return result diff --git a/Lib/test/test_curses.py b/Lib/test/test_curses.py index ea2dcd76b585a99..32a39e60aaedfae 100644 --- a/Lib/test/test_curses.py +++ b/Lib/test/test_curses.py @@ -2660,6 +2660,22 @@ def test_textbox_combining(self): box.do_command(ch) self.assertEqual(box.gather(), text + ' ') + @requires_wide_build + def test_textbox_double_width(self): + # A double-width (East Asian) character occupies two cells. gather() + # reads a whole line at a time so that the second cell, which holds + # the same character, is not reported as another one. + text = '你好' + if self._encodable(text): + box, win = self._make_textbox(1, 12) + for ch in text: + box.do_command(ch) + self.assertEqual(box.gather(), text + ' ') + box, win = self._make_textbox(1, 12, stripspaces=0) + for ch in text: + box.do_command(ch) + self.assertEqual(box.gather(), text + ' ' * 8) + def test_textbox_edit_wide(self): # edit() reads characters through get_wch(). Each character is pushed # with unget_wch(), which on a narrow build requires it to encode to a