summaryrefslogtreecommitdiff
path: root/Lib/test
diff options
context:
space:
mode:
authorMichael W. Hudson <mwh@python.net>2002-10-07 12:32:57 +0000
committerMichael W. Hudson <mwh@python.net>2002-10-07 12:32:57 +0000
commitd1bb75505fbb5b47f4f3d420e00964b38c24b203 (patch)
tree301416738736d193214c3d6639c94682faf225af /Lib/test
parentb45532abd26a9752e42ca589b5b0e49552761811 (diff)
downloadcpython-git-d1bb75505fbb5b47f4f3d420e00964b38c24b203.tar.gz
Backport:
2002/08/11 12:23:04 lemburg Python/bltinmodule.c 2.262 2002/08/11 12:23:04 lemburg Objects/unicodeobject.c 2.162 2002/08/11 12:23:03 lemburg Misc/NEWS 1.461 2002/08/11 12:23:03 lemburg Lib/test/test_unicode.py 1.65 2002/08/11 12:23:03 lemburg Include/unicodeobject.h 2.39 Add C API PyUnicode_FromOrdinal() which exposes unichr() at C level. u'%c' will now raise a ValueError in case the argument is an integer outside the valid range of Unicode code point ordinals. Closes SF bug #593581.
Diffstat (limited to 'Lib/test')
-rw-r--r--Lib/test/test_unicode.py16
1 files changed, 16 insertions, 0 deletions
diff --git a/Lib/test/test_unicode.py b/Lib/test/test_unicode.py
index 6125c92efc..d4209249b6 100644
--- a/Lib/test/test_unicode.py
+++ b/Lib/test/test_unicode.py
@@ -442,6 +442,14 @@ except KeyError:
else:
verify(value == u'abc, def')
+for ordinal in (-100, 0x200000):
+ try:
+ u"%c" % ordinal
+ except ValueError:
+ pass
+ else:
+ print '*** formatting u"%%c" %% %i should give a ValueError' % ordinal
+
# formatting jobs delegated from the string implementation:
verify('...%(foo)s...' % {'foo':u"abc"} == u'...abc...')
verify('...%(foo)s...' % {'foo':"abc"} == '...abc...')
@@ -737,6 +745,14 @@ for encoding in (
except ValueError,why:
print '*** codec for "%s" failed: %s' % (encoding, why)
+# UTF-8 must be roundtrip safe for all UCS-2 code points
+# This excludes surrogates: in the full range, there would be
+# a surrogate pair (\udbff\udc00), which gets converted back
+# to a non-BMP character (\U0010fc00)
+u = u''.join(map(unichr, range(0,0xd800)+range(0xe000,0x10000)))
+for encoding in ('utf-8',):
+ verify(unicode(u.encode(encoding),encoding) == u)
+
print 'done.'
print 'Testing Unicode string concatenation...',