From 7a465cb5ee7e298cae626ace1fc3e7d97df79f2e Mon Sep 17 00:00:00 2001 From: Serhiy Storchaka Date: Sat, 30 Mar 2019 08:23:38 +0200 Subject: bpo-24214: Fixed the UTF-8 incremental decoder. (GH-12603) The bug occurred when the encoded surrogate character is passed to the incremental decoder in two chunks. --- Objects/unicodeobject.c | 3 +++ 1 file changed, 3 insertions(+) (limited to 'Objects') diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index 8ab3943e61..c0b345be7e 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -4883,6 +4883,9 @@ PyUnicode_DecodeUTF8Stateful(const char *s, case 2: case 3: case 4: + if (s == end || consumed) { + goto End; + } errmsg = "invalid continuation byte"; startinpos = s - starts; endinpos = startinpos + ch - 1; -- cgit v1.2.1