X-Git-Url: http://git.shadowcat.co.uk/gitweb/gitweb.cgi?a=blobdiff_plain;f=utf8.c;h=21d0f08a19d31c996d59515d7c149402c591765b;hb=5d25492c931e03949e966bf309e602c3fc6aad65;hp=2afcbeb7879a89f583faa9db92529ae4e8e55880;hpb=2c4547fe91306627634f7dd6526eed3a43460162;p=p5sagit%2Fp5-mst-13.2.git

diff --git a/utf8.c b/utf8.c
index 2afcbeb..21d0f08 100644
--- a/utf8.c
+++ b/utf8.c
@@ -1,6 +1,6 @@
 /*    utf8.c
  *
- *    Copyright (c) 1998-2002, Larry Wall
+ *    Copyright (C) 2000, 2001, 2002, 2003, by Larry Wall and others
  *
  *    You may distribute under the terms of either the GNU General Public
  *    License or the Artistic License, as specified in the README file.
@@ -24,6 +24,8 @@
 #define PERL_IN_UTF8_C
 #include "perl.h"
 
+static char unees[] = "Malformed UTF-8 character (unexpected end of string)";
+
 /* 
 =head1 Unicode Support
 
@@ -57,26 +59,23 @@ Perl_uvuni_to_utf8_flags(pTHX_ U8 *d, UV uv, UV flags)
     if (ckWARN(WARN_UTF8)) {
 	 if (UNICODE_IS_SURROGATE(uv) &&
 	     !(flags & UNICODE_ALLOW_SURROGATE))
-	      Perl_warner(aTHX_ WARN_UTF8, "UTF-16 surrogate 0x%04"UVxf, uv);
+	      Perl_warner(aTHX_ packWARN(WARN_UTF8), "UTF-16 surrogate 0x%04"UVxf, uv);
 	 else if (
 		  ((uv >= 0xFDD0 && uv <= 0xFDEF &&
 		    !(flags & UNICODE_ALLOW_FDD0))
 		   ||
-		   ((uv & 0xFFFF) == 0xFFFE &&
-		    !(flags & UNICODE_ALLOW_FFFE))
-		   ||
-		   ((uv & 0xFFFF) == 0xFFFF &&
+		   ((uv & 0xFFFE) == 0xFFFE && /* Either FFFE or FFFF. */
 		    !(flags & UNICODE_ALLOW_FFFF))) &&
 		  /* UNICODE_ALLOW_SUPER includes
 		   * FFFEs and FFFFs beyond 0x10FFFF. */
 		  ((uv <= PERL_UNICODE_MAX) ||
 		   !(flags & UNICODE_ALLOW_SUPER))
 		  )
-	      Perl_warner(aTHX_ WARN_UTF8,
+	      Perl_warner(aTHX_ packWARN(WARN_UTF8),
 			 "Unicode character 0x%04"UVxf" is illegal", uv);
     }
     if (UNI_IS_INVARIANT(uv)) {
-	*d++ = UTF_TO_NATIVE(uv);
+	*d++ = (U8)UTF_TO_NATIVE(uv);
 	return d;
     }
 #if defined(EBCDIC)
@@ -84,76 +83,76 @@ Perl_uvuni_to_utf8_flags(pTHX_ U8 *d, UV uv, UV flags)
 	STRLEN len  = UNISKIP(uv);
 	U8 *p = d+len-1;
 	while (p > d) {
-	    *p-- = UTF_TO_NATIVE((uv & UTF_CONTINUATION_MASK) | UTF_CONTINUATION_MARK);
+	    *p-- = (U8)UTF_TO_NATIVE((uv & UTF_CONTINUATION_MASK) | UTF_CONTINUATION_MARK);
 	    uv >>= UTF_ACCUMULATION_SHIFT;
 	}
-	*p = UTF_TO_NATIVE((uv & UTF_START_MASK(len)) | UTF_START_MARK(len));
+	*p = (U8)UTF_TO_NATIVE((uv & UTF_START_MASK(len)) | UTF_START_MARK(len));
 	return d+len;
     }
 #else /* Non loop style */
     if (uv < 0x800) {
-	*d++ = (( uv >>  6)         | 0xc0);
-	*d++ = (( uv        & 0x3f) | 0x80);
+	*d++ = (U8)(( uv >>  6)         | 0xc0);
+	*d++ = (U8)(( uv        & 0x3f) | 0x80);
 	return d;
     }
     if (uv < 0x10000) {
-	*d++ = (( uv >> 12)         | 0xe0);
-	*d++ = (((uv >>  6) & 0x3f) | 0x80);
-	*d++ = (( uv        & 0x3f) | 0x80);
+	*d++ = (U8)(( uv >> 12)         | 0xe0);
+	*d++ = (U8)(((uv >>  6) & 0x3f) | 0x80);
+	*d++ = (U8)(( uv        & 0x3f) | 0x80);
 	return d;
     }
     if (uv < 0x200000) {
-	*d++ = (( uv >> 18)         | 0xf0);
-	*d++ = (((uv >> 12) & 0x3f) | 0x80);
-	*d++ = (((uv >>  6) & 0x3f) | 0x80);
-	*d++ = (( uv        & 0x3f) | 0x80);
+	*d++ = (U8)(( uv >> 18)         | 0xf0);
+	*d++ = (U8)(((uv >> 12) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >>  6) & 0x3f) | 0x80);
+	*d++ = (U8)(( uv        & 0x3f) | 0x80);
 	return d;
     }
     if (uv < 0x4000000) {
-	*d++ = (( uv >> 24)         | 0xf8);
-	*d++ = (((uv >> 18) & 0x3f) | 0x80);
-	*d++ = (((uv >> 12) & 0x3f) | 0x80);
-	*d++ = (((uv >>  6) & 0x3f) | 0x80);
-	*d++ = (( uv        & 0x3f) | 0x80);
+	*d++ = (U8)(( uv >> 24)         | 0xf8);
+	*d++ = (U8)(((uv >> 18) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 12) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >>  6) & 0x3f) | 0x80);
+	*d++ = (U8)(( uv        & 0x3f) | 0x80);
 	return d;
     }
     if (uv < 0x80000000) {
-	*d++ = (( uv >> 30)         | 0xfc);
-	*d++ = (((uv >> 24) & 0x3f) | 0x80);
-	*d++ = (((uv >> 18) & 0x3f) | 0x80);
-	*d++ = (((uv >> 12) & 0x3f) | 0x80);
-	*d++ = (((uv >>  6) & 0x3f) | 0x80);
-	*d++ = (( uv        & 0x3f) | 0x80);
+	*d++ = (U8)(( uv >> 30)         | 0xfc);
+	*d++ = (U8)(((uv >> 24) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 18) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 12) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >>  6) & 0x3f) | 0x80);
+	*d++ = (U8)(( uv        & 0x3f) | 0x80);
 	return d;
     }
 #ifdef HAS_QUAD
     if (uv < UTF8_QUAD_MAX)
 #endif
     {
-	*d++ =                        0xfe;	/* Can't match U+FEFF! */
-	*d++ = (((uv >> 30) & 0x3f) | 0x80);
-	*d++ = (((uv >> 24) & 0x3f) | 0x80);
-	*d++ = (((uv >> 18) & 0x3f) | 0x80);
-	*d++ = (((uv >> 12) & 0x3f) | 0x80);
-	*d++ = (((uv >>  6) & 0x3f) | 0x80);
-	*d++ = (( uv        & 0x3f) | 0x80);
+	*d++ =                            0xfe;	/* Can't match U+FEFF! */
+	*d++ = (U8)(((uv >> 30) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 24) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 18) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 12) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >>  6) & 0x3f) | 0x80);
+	*d++ = (U8)(( uv        & 0x3f) | 0x80);
 	return d;
     }
 #ifdef HAS_QUAD
     {
-	*d++ =                        0xff;	/* Can't match U+FFFE! */
-	*d++ =                        0x80;	/* 6 Reserved bits */
-	*d++ = (((uv >> 60) & 0x0f) | 0x80);	/* 2 Reserved bits */
-	*d++ = (((uv >> 54) & 0x3f) | 0x80);
-	*d++ = (((uv >> 48) & 0x3f) | 0x80);
-	*d++ = (((uv >> 42) & 0x3f) | 0x80);
-	*d++ = (((uv >> 36) & 0x3f) | 0x80);
-	*d++ = (((uv >> 30) & 0x3f) | 0x80);
-	*d++ = (((uv >> 24) & 0x3f) | 0x80);
-	*d++ = (((uv >> 18) & 0x3f) | 0x80);
-	*d++ = (((uv >> 12) & 0x3f) | 0x80);
-	*d++ = (((uv >>  6) & 0x3f) | 0x80);
-	*d++ = (( uv        & 0x3f) | 0x80);
+	*d++ =                            0xff;		/* Can't match U+FFFE! */
+	*d++ =                            0x80;		/* 6 Reserved bits */
+	*d++ = (U8)(((uv >> 60) & 0x0f) | 0x80);	/* 2 Reserved bits */
+	*d++ = (U8)(((uv >> 54) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 48) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 42) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 36) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 30) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 24) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 18) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >> 12) & 0x3f) | 0x80);
+	*d++ = (U8)(((uv >>  6) & 0x3f) | 0x80);
+	*d++ = (U8)(( uv        & 0x3f) | 0x80);
 	return d;
     }
 #endif
@@ -171,12 +170,11 @@ Perl_uvuni_to_utf8(pTHX_ U8 *d, UV uv)
 =for apidoc A|STRLEN|is_utf8_char|U8 *s
 
 Tests if some arbitrary number of bytes begins in a valid UTF-8
-character.  Note that an INVARIANT (i.e. ASCII) character is a valid UTF-8 character.
-The actual number of bytes in the UTF-8 character will be returned if
-it is valid, otherwise 0.
+character.  Note that an INVARIANT (i.e. ASCII) character is a valid
+UTF-8 character.  The actual number of bytes in the UTF-8 character
+will be returned if it is valid, otherwise 0.
 
-=cut
-*/
+=cut */
 STRLEN
 Perl_is_utf8_char(pTHX_ U8 *s)
 {
@@ -210,7 +208,7 @@ Perl_is_utf8_char(pTHX_ U8 *s)
 	s++;
     }
 
-    if (UNISKIP(uv) < len)
+    if ((STRLEN)UNISKIP(uv) < len)
 	return 0;
 
     return len;
@@ -219,10 +217,10 @@ Perl_is_utf8_char(pTHX_ U8 *s)
 /*
 =for apidoc A|bool|is_utf8_string|U8 *s|STRLEN len
 
-Returns true if first C<len> bytes of the given string form a valid UTF8
-string, false otherwise.  Note that 'a valid UTF8 string' does not mean
-'a string that contains UTF8' because a valid ASCII string is a valid
-UTF8 string.
+Returns true if first C<len> bytes of the given string form a valid
+UTF8 string, false otherwise.  Note that 'a valid UTF8 string' does
+not mean 'a string that contains code points above 0x7F encoded in
+UTF8' because a valid ASCII string is a valid UTF8 string.
 
 =cut
 */
@@ -239,9 +237,17 @@ Perl_is_utf8_string(pTHX_ U8 *s, STRLEN len)
     send = s + len;
 
     while (x < send) {
-        c = is_utf8_char(x);
-	if (!c)
-	    return FALSE;
+	 /* Inline the easy bits of is_utf8_char() here for speed... */
+	 if (UTF8_IS_INVARIANT(*x))
+	      c = 1;
+	 else if (!UTF8_IS_START(*x))
+	      return FALSE;
+	 else {
+	      /* ... and call is_utf8_char() only if really needed. */
+	      c = is_utf8_char(x);
+	      if (!c)
+		   return FALSE;
+	 }
         x += c;
     }
     if (x != send)
@@ -294,9 +300,8 @@ Perl_utf8n_to_uvuni(pTHX_ U8 *s, STRLEN curlen, STRLEN *retlen, U32 flags)
 #define UTF8_WARN_SHORT				 5
 #define UTF8_WARN_OVERFLOW			 6
 #define UTF8_WARN_SURROGATE			 7
-#define UTF8_WARN_BOM				 8
-#define UTF8_WARN_LONG				 9
-#define UTF8_WARN_FFFF				10
+#define UTF8_WARN_LONG				 8
+#define UTF8_WARN_FFFF				 9 /* Also FFFE. */
 
     if (curlen == 0 &&
 	!(flags & UTF8_ALLOW_EMPTY)) {
@@ -391,11 +396,7 @@ Perl_utf8n_to_uvuni(pTHX_ U8 *s, STRLEN curlen, STRLEN *retlen, U32 flags)
 	!(flags & UTF8_ALLOW_SURROGATE)) {
 	warning = UTF8_WARN_SURROGATE;
 	goto malformed;
-    } else if (UNICODE_IS_BYTE_ORDER_MARK(uv) &&
-	       !(flags & UTF8_ALLOW_BOM)) {
-	warning = UTF8_WARN_BOM;
-	goto malformed;
-    } else if ((expectlen > UNISKIP(uv)) &&
+    } else if ((expectlen > (STRLEN)UNISKIP(uv)) &&
 	       !(flags & UTF8_ALLOW_LONG)) {
 	warning = UTF8_WARN_LONG;
 	goto malformed;
@@ -450,9 +451,6 @@ malformed:
 	case UTF8_WARN_SURROGATE:
 	    Perl_sv_catpvf(aTHX_ sv, "(UTF-16 surrogate 0x%04"UVxf")", uv);
 	    break;
-	case UTF8_WARN_BOM:
-	    Perl_sv_catpvf(aTHX_ sv, "(byte order mark 0x%04"UVxf")", uv);
-	    break;
 	case UTF8_WARN_LONG:
 	    Perl_sv_catpvf(aTHX_ sv, "(%d byte%s, need %d, after start byte 0x%02"UVxf")",
 			   expectlen, expectlen == 1 ? "": "s", UNISKIP(uv), startbyte);
@@ -469,10 +467,10 @@ malformed:
 	    char *s = SvPVX(sv);
 
 	    if (PL_op)
-		Perl_warner(aTHX_ WARN_UTF8,
+		Perl_warner(aTHX_ packWARN(WARN_UTF8),
 			    "%s in %s", s,  OP_DESC(PL_op));
 	    else
-		Perl_warner(aTHX_ WARN_UTF8, "%s", s);
+		Perl_warner(aTHX_ packWARN(WARN_UTF8), "%s", s);
 	}
     }
 
@@ -498,7 +496,8 @@ returned and retlen is set, if possible, to -1.
 UV
 Perl_utf8_to_uvchr(pTHX_ U8 *s, STRLEN *retlen)
 {
-    return Perl_utf8n_to_uvchr(aTHX_ s, UTF8_MAXLEN, retlen, 0);
+    return Perl_utf8n_to_uvchr(aTHX_ s, UTF8_MAXLEN, retlen,
+			       ckWARN(WARN_UTF8) ? 0 : UTF8_ALLOW_ANY);
 }
 
 /*
@@ -521,7 +520,8 @@ UV
 Perl_utf8_to_uvuni(pTHX_ U8 *s, STRLEN *retlen)
 {
     /* Call the low level routine asking for checks */
-    return Perl_utf8n_to_uvuni(aTHX_ s, UTF8_MAXLEN, retlen, 0);
+    return Perl_utf8n_to_uvuni(aTHX_ s, UTF8_MAXLEN, retlen,
+			       ckWARN(WARN_UTF8) ? 0 : UTF8_ALLOW_ANY);
 }
 
 /*
@@ -543,13 +543,29 @@ Perl_utf8_length(pTHX_ U8 *s, U8 *e)
      * the bitops (especially ~) can create illegal UTF-8.
      * In other words: in Perl UTF-8 is not just for Unicode. */
 
-    if (e < s)
-	Perl_croak(aTHX_ "panic: utf8_length: unexpected end");
+    if (e < s) {
+        if (ckWARN_d(WARN_UTF8)) {
+	    if (PL_op)
+	        Perl_warner(aTHX_ packWARN(WARN_UTF8),
+			    "%s in %s", unees, OP_DESC(PL_op));
+	    else
+	        Perl_warner(aTHX_ packWARN(WARN_UTF8), unees);
+	}
+	return 0;
+    }
     while (s < e) {
 	U8 t = UTF8SKIP(s);
 
-	if (e - s < t)
-	    Perl_croak(aTHX_ "panic: utf8_length: unaligned end");
+	if (e - s < t) {
+	    if (ckWARN_d(WARN_UTF8)) {
+	        if (PL_op)
+		    Perl_warner(aTHX_ packWARN(WARN_UTF8),
+				unees, OP_DESC(PL_op));
+		else
+		    Perl_warner(aTHX_ packWARN(WARN_UTF8), unees);
+	    }
+	    return len;
+	}
 	s += t;
 	len++;
     }
@@ -582,8 +598,16 @@ Perl_utf8_distance(pTHX_ U8 *a, U8 *b)
 	while (a < b) {
 	    U8 c = UTF8SKIP(a);
 
-	    if (b - a < c)
-		Perl_croak(aTHX_ "panic: utf8_distance: unaligned end");
+	    if (b - a < c) {
+	        if (ckWARN_d(WARN_UTF8)) {
+		    if (PL_op)
+		        Perl_warner(aTHX_ packWARN(WARN_UTF8),
+				    "%s in %s", unees, OP_DESC(PL_op));
+		    else
+		        Perl_warner(aTHX_ packWARN(WARN_UTF8), unees);
+		}
+		return off;
+	    }
 	    a += c;
 	    off--;
 	}
@@ -592,8 +616,16 @@ Perl_utf8_distance(pTHX_ U8 *a, U8 *b)
 	while (b < a) {
 	    U8 c = UTF8SKIP(b);
 
-	    if (a - b < c)
-		Perl_croak(aTHX_ "panic: utf8_distance: unaligned end");
+	    if (a - b < c) {
+	        if (ckWARN_d(WARN_UTF8)) {
+		    if (PL_op)
+		        Perl_warner(aTHX_ packWARN(WARN_UTF8),
+				    "%s in %s", unees, OP_DESC(PL_op));
+		    else
+		        Perl_warner(aTHX_ packWARN(WARN_UTF8), unees);
+		}
+		return off;
+	    }
 	    b += c;
 	    off++;
 	}
@@ -738,6 +770,9 @@ Converts a string C<s> of length C<len> from ASCII into UTF8 encoding.
 Returns a pointer to the newly-created string, and sets C<len> to
 reflect the new length.
 
+If you want to convert to UTF8 from other encodings than ASCII,
+see sv_recode_to_utf8().
+
 =cut
 */
 
@@ -755,10 +790,10 @@ Perl_bytes_to_utf8(pTHX_ U8 *s, STRLEN *len)
     while (s < send) {
         UV uv = NATIVE_TO_ASCII(*s++);
         if (UNI_IS_INVARIANT(uv))
-            *d++ = UTF_TO_NATIVE(uv);
+            *d++ = (U8)UTF_TO_NATIVE(uv);
         else {
-            *d++ = UTF8_EIGHT_BIT_HI(uv);
-            *d++ = UTF8_EIGHT_BIT_LO(uv);
+            *d++ = (U8)UTF8_EIGHT_BIT_HI(uv);
+            *d++ = (U8)UTF8_EIGHT_BIT_LO(uv);
         }
     }
     *d = '\0';
@@ -787,31 +822,32 @@ Perl_utf16_to_utf8(pTHX_ U8* p, U8* d, I32 bytelen, I32 *newlen)
 	UV uv = (p[0] << 8) + p[1]; /* UTF-16BE */
 	p += 2;
 	if (uv < 0x80) {
-	    *d++ = uv;
+	    *d++ = (U8)uv;
 	    continue;
 	}
 	if (uv < 0x800) {
-	    *d++ = (( uv >>  6)         | 0xc0);
-	    *d++ = (( uv        & 0x3f) | 0x80);
+	    *d++ = (U8)(( uv >>  6)         | 0xc0);
+	    *d++ = (U8)(( uv        & 0x3f) | 0x80);
 	    continue;
 	}
 	if (uv >= 0xd800 && uv < 0xdbff) {	/* surrogates */
-	    UV low = *p++;
+	    UV low = (p[0] << 8) + p[1];
+	    p += 2;
 	    if (low < 0xdc00 || low >= 0xdfff)
 		Perl_croak(aTHX_ "Malformed UTF-16 surrogate");
 	    uv = ((uv - 0xd800) << 10) + (low - 0xdc00) + 0x10000;
 	}
 	if (uv < 0x10000) {
-	    *d++ = (( uv >> 12)         | 0xe0);
-	    *d++ = (((uv >>  6) & 0x3f) | 0x80);
-	    *d++ = (( uv        & 0x3f) | 0x80);
+	    *d++ = (U8)(( uv >> 12)         | 0xe0);
+	    *d++ = (U8)(((uv >>  6) & 0x3f) | 0x80);
+	    *d++ = (U8)(( uv        & 0x3f) | 0x80);
 	    continue;
 	}
 	else {
-	    *d++ = (( uv >> 18)         | 0xf0);
-	    *d++ = (((uv >> 12) & 0x3f) | 0x80);
-	    *d++ = (((uv >>  6) & 0x3f) | 0x80);
-	    *d++ = (( uv        & 0x3f) | 0x80);
+	    *d++ = (U8)(( uv >> 18)         | 0xf0);
+	    *d++ = (U8)(((uv >> 12) & 0x3f) | 0x80);
+	    *d++ = (U8)(((uv >>  6) & 0x3f) | 0x80);
+	    *d++ = (U8)(( uv        & 0x3f) | 0x80);
 	    continue;
 	}
     }
@@ -841,7 +877,7 @@ bool
 Perl_is_uni_alnum(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_alnum(tmpbuf);
 }
 
@@ -849,7 +885,7 @@ bool
 Perl_is_uni_alnumc(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_alnumc(tmpbuf);
 }
 
@@ -857,7 +893,7 @@ bool
 Perl_is_uni_idfirst(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_idfirst(tmpbuf);
 }
 
@@ -865,7 +901,7 @@ bool
 Perl_is_uni_alpha(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_alpha(tmpbuf);
 }
 
@@ -873,7 +909,7 @@ bool
 Perl_is_uni_ascii(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_ascii(tmpbuf);
 }
 
@@ -881,7 +917,7 @@ bool
 Perl_is_uni_space(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_space(tmpbuf);
 }
 
@@ -889,7 +925,7 @@ bool
 Perl_is_uni_digit(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_digit(tmpbuf);
 }
 
@@ -897,7 +933,7 @@ bool
 Perl_is_uni_upper(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_upper(tmpbuf);
 }
 
@@ -905,7 +941,7 @@ bool
 Perl_is_uni_lower(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_lower(tmpbuf);
 }
 
@@ -913,7 +949,7 @@ bool
 Perl_is_uni_cntrl(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_cntrl(tmpbuf);
 }
 
@@ -921,7 +957,7 @@ bool
 Perl_is_uni_graph(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_graph(tmpbuf);
 }
 
@@ -929,7 +965,7 @@ bool
 Perl_is_uni_print(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_print(tmpbuf);
 }
 
@@ -937,7 +973,7 @@ bool
 Perl_is_uni_punct(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_punct(tmpbuf);
 }
 
@@ -945,40 +981,36 @@ bool
 Perl_is_uni_xdigit(pTHX_ UV c)
 {
     U8 tmpbuf[UTF8_MAXLEN_UCLC+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
+    uvchr_to_utf8(tmpbuf, c);
     return is_utf8_xdigit(tmpbuf);
 }
 
 UV
 Perl_to_uni_upper(pTHX_ UV c, U8* p, STRLEN *lenp)
 {
-    U8 tmpbuf[UTF8_MAXLEN_UCLC+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
-    return to_utf8_upper(tmpbuf, p, lenp);
+    uvchr_to_utf8(p, c);
+    return to_utf8_upper(p, p, lenp);
 }
 
 UV
 Perl_to_uni_title(pTHX_ UV c, U8* p, STRLEN *lenp)
 {
-    U8 tmpbuf[UTF8_MAXLEN_UCLC+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
-    return to_utf8_title(tmpbuf, p, lenp);
+    uvchr_to_utf8(p, c);
+    return to_utf8_title(p, p, lenp);
 }
 
 UV
 Perl_to_uni_lower(pTHX_ UV c, U8* p, STRLEN *lenp)
 {
-    U8 tmpbuf[UTF8_MAXLEN_UCLC+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
-    return to_utf8_lower(tmpbuf, p, lenp);
+    uvchr_to_utf8(p, c);
+    return to_utf8_lower(p, p, lenp);
 }
 
 UV
 Perl_to_uni_fold(pTHX_ UV c, U8* p, STRLEN *lenp)
 {
-    U8 tmpbuf[UTF8_MAXLEN_FOLD+1];
-    uvchr_to_utf8(tmpbuf, (UV)c);
-    return to_utf8_fold(tmpbuf, p, lenp);
+    uvchr_to_utf8(p, c);
+    return to_utf8_fold(p, p, lenp);
 }
 
 /* for now these all assume no locale info available for Unicode > 255 */
@@ -1107,13 +1139,13 @@ Perl_is_utf8_alnum(pTHX_ U8 *p)
 	 * descendant of isalnum(3), in other words, it doesn't
 	 * contain the '_'. --jhi */
 	PL_utf8_alnum = swash_init("utf8", "IsWord", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_alnum, p, TRUE);
+    return swash_fetch(PL_utf8_alnum, p, TRUE) != 0;
 /*    return *p == '_' || is_utf8_alpha(p) || is_utf8_digit(p); */
 #ifdef SURPRISINGLY_SLOWER  /* probably because alpha is usually true */
     if (!PL_utf8_alnum)
 	PL_utf8_alnum = swash_init("utf8", "",
 	    sv_2mortal(newSVpv("+utf8::IsAlpha\n+utf8::IsDigit\n005F\n",0)), 0, 0);
-    return swash_fetch(PL_utf8_alnum, p, TRUE);
+    return swash_fetch(PL_utf8_alnum, p, TRUE) != 0;
 #endif
 }
 
@@ -1124,20 +1156,38 @@ Perl_is_utf8_alnumc(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_alnum)
 	PL_utf8_alnum = swash_init("utf8", "IsAlnumC", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_alnum, p, TRUE);
+    return swash_fetch(PL_utf8_alnum, p, TRUE) != 0;
 /*    return is_utf8_alpha(p) || is_utf8_digit(p); */
 #ifdef SURPRISINGLY_SLOWER  /* probably because alpha is usually true */
     if (!PL_utf8_alnum)
 	PL_utf8_alnum = swash_init("utf8", "",
 	    sv_2mortal(newSVpv("+utf8::IsAlpha\n+utf8::IsDigit\n005F\n",0)), 0, 0);
-    return swash_fetch(PL_utf8_alnum, p, TRUE);
+    return swash_fetch(PL_utf8_alnum, p, TRUE) != 0;
 #endif
 }
 
 bool
-Perl_is_utf8_idfirst(pTHX_ U8 *p)
+Perl_is_utf8_idfirst(pTHX_ U8 *p) /* The naming is historical. */
 {
-    return *p == '_' || is_utf8_alpha(p);
+    if (*p == '_')
+	return TRUE;
+    if (!is_utf8_char(p))
+	return FALSE;
+    if (!PL_utf8_idstart) /* is_utf8_idstart would be more logical. */
+	PL_utf8_idstart = swash_init("utf8", "IdStart", &PL_sv_undef, 0, 0);
+    return swash_fetch(PL_utf8_idstart, p, TRUE) != 0;
+}
+
+bool
+Perl_is_utf8_idcont(pTHX_ U8 *p)
+{
+    if (*p == '_')
+	return TRUE;
+    if (!is_utf8_char(p))
+	return FALSE;
+    if (!PL_utf8_idcont)
+	PL_utf8_idcont = swash_init("utf8", "IdContinue", &PL_sv_undef, 0, 0);
+    return swash_fetch(PL_utf8_idcont, p, TRUE) != 0;
 }
 
 bool
@@ -1147,7 +1197,7 @@ Perl_is_utf8_alpha(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_alpha)
 	PL_utf8_alpha = swash_init("utf8", "IsAlpha", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_alpha, p, TRUE);
+    return swash_fetch(PL_utf8_alpha, p, TRUE) != 0;
 }
 
 bool
@@ -1157,7 +1207,7 @@ Perl_is_utf8_ascii(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_ascii)
 	PL_utf8_ascii = swash_init("utf8", "IsAscii", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_ascii, p, TRUE);
+    return swash_fetch(PL_utf8_ascii, p, TRUE) != 0;
 }
 
 bool
@@ -1167,7 +1217,7 @@ Perl_is_utf8_space(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_space)
 	PL_utf8_space = swash_init("utf8", "IsSpacePerl", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_space, p, TRUE);
+    return swash_fetch(PL_utf8_space, p, TRUE) != 0;
 }
 
 bool
@@ -1177,7 +1227,7 @@ Perl_is_utf8_digit(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_digit)
 	PL_utf8_digit = swash_init("utf8", "IsDigit", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_digit, p, TRUE);
+    return swash_fetch(PL_utf8_digit, p, TRUE) != 0;
 }
 
 bool
@@ -1186,8 +1236,8 @@ Perl_is_utf8_upper(pTHX_ U8 *p)
     if (!is_utf8_char(p))
 	return FALSE;
     if (!PL_utf8_upper)
-	PL_utf8_upper = swash_init("utf8", "IsUpper", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_upper, p, TRUE);
+	PL_utf8_upper = swash_init("utf8", "IsUppercase", &PL_sv_undef, 0, 0);
+    return swash_fetch(PL_utf8_upper, p, TRUE) != 0;
 }
 
 bool
@@ -1196,8 +1246,8 @@ Perl_is_utf8_lower(pTHX_ U8 *p)
     if (!is_utf8_char(p))
 	return FALSE;
     if (!PL_utf8_lower)
-	PL_utf8_lower = swash_init("utf8", "IsLower", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_lower, p, TRUE);
+	PL_utf8_lower = swash_init("utf8", "IsLowercase", &PL_sv_undef, 0, 0);
+    return swash_fetch(PL_utf8_lower, p, TRUE) != 0;
 }
 
 bool
@@ -1207,7 +1257,7 @@ Perl_is_utf8_cntrl(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_cntrl)
 	PL_utf8_cntrl = swash_init("utf8", "IsCntrl", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_cntrl, p, TRUE);
+    return swash_fetch(PL_utf8_cntrl, p, TRUE) != 0;
 }
 
 bool
@@ -1217,7 +1267,7 @@ Perl_is_utf8_graph(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_graph)
 	PL_utf8_graph = swash_init("utf8", "IsGraph", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_graph, p, TRUE);
+    return swash_fetch(PL_utf8_graph, p, TRUE) != 0;
 }
 
 bool
@@ -1227,7 +1277,7 @@ Perl_is_utf8_print(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_print)
 	PL_utf8_print = swash_init("utf8", "IsPrint", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_print, p, TRUE);
+    return swash_fetch(PL_utf8_print, p, TRUE) != 0;
 }
 
 bool
@@ -1237,7 +1287,7 @@ Perl_is_utf8_punct(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_punct)
 	PL_utf8_punct = swash_init("utf8", "IsPunct", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_punct, p, TRUE);
+    return swash_fetch(PL_utf8_punct, p, TRUE) != 0;
 }
 
 bool
@@ -1247,7 +1297,7 @@ Perl_is_utf8_xdigit(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_xdigit)
 	PL_utf8_xdigit = swash_init("utf8", "IsXDigit", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_xdigit, p, TRUE);
+    return swash_fetch(PL_utf8_xdigit, p, TRUE) != 0;
 }
 
 bool
@@ -1257,7 +1307,7 @@ Perl_is_utf8_mark(pTHX_ U8 *p)
 	return FALSE;
     if (!PL_utf8_mark)
 	PL_utf8_mark = swash_init("utf8", "IsM", &PL_sv_undef, 0, 0);
-    return swash_fetch(PL_utf8_mark, p, TRUE);
+    return swash_fetch(PL_utf8_mark, p, TRUE) != 0;
 }
 
 /*
@@ -1270,73 +1320,109 @@ The "ustrp" is a pointer to the character buffer to put the
 conversion result to.  The "lenp" is a pointer to the length
 of the result.
 
-The "swash" is a pointer to the swash to use.
+The "swashp" is a pointer to the swash to use.
 
-The "normal" is a string like "ToLower" which means the swash
-$utf8::ToLower, which is stored in lib/unicore/To/Lower.pl,
-and loaded by SWASHGET, using lib/utf8_heavy.pl.
+Both the special and normal mappings are stored lib/unicore/To/Foo.pl,
+and loaded by SWASHGET, using lib/utf8_heavy.pl.  The special (usually,
+but not always, a multicharacter mapping), is tried first.
 
-The "special" is a string like "utf8::ToSpecLower", which means
-the hash %utf8::ToSpecLower, which is stored in the same file,
-lib/unicore/To/Lower.pl, and also loaded by SWASHGET.  The access
-to the hash is by Perl_to_utf8_case().
+The "special" is a string like "utf8::ToSpecLower", which means the
+hash %utf8::ToSpecLower.  The access to the hash is through
+Perl_to_utf8_case().
 
-=cut
- */
+The "normal" is a string like "ToLower" which means the swash
+%utf8::ToLower.
+
+=cut */
 
 UV
 Perl_to_utf8_case(pTHX_ U8 *p, U8* ustrp, STRLEN *lenp, SV **swashp, char *normal, char *special)
 {
-    UV uv;
+    UV uv0, uv1;
+    U8 tmpbuf[UTF8_MAXLEN_FOLD+1];
+    STRLEN len = 0;
+
+    uv0 = utf8_to_uvchr(p, 0);
+    /* The NATIVE_TO_UNI() and UNI_TO_NATIVE() mappings
+     * are necessary in EBCDIC, they are redundant no-ops
+     * in ASCII-ish platforms, and hopefully optimized away. */
+    uv1 = NATIVE_TO_UNI(uv0);
+    uvuni_to_utf8(tmpbuf, uv1);
 
-    if (!*swashp)
-        *swashp = swash_init("utf8", normal, &PL_sv_undef, 4, 0);
-    uv = swash_fetch(*swashp, p, TRUE);
-    if (!uv) {
+    if (!*swashp) /* load on-demand */
+         *swashp = swash_init("utf8", normal, &PL_sv_undef, 4, 0);
+
+    if (special) {
+         /* It might be "special" (sometimes, but not always,
+	  * a multicharacter mapping) */
 	 HV *hv;
 	 SV *keysv;
 	 HE *he;
-
-	 uv = utf8_to_uvchr(p, 0);
-
+	 SV *val;
+	
 	 if ((hv    = get_hv(special, FALSE)) &&
-	     (keysv = sv_2mortal(Perl_newSVpvf(aTHX_ "%04"UVXf, uv))) &&
-	     (he    = hv_fetch_ent(hv, keysv, FALSE, 0))) {
-	      SV *val = HeVAL(he);
-	      char *s = SvPV(val, *lenp);
-	      U8 c = *(U8*)s;
-
-	      if (*lenp > 1 || UNI_IS_INVARIANT(c))
-		   Copy(s, ustrp, *lenp, U8);
+	     (keysv = sv_2mortal(Perl_newSVpvf(aTHX_ "%04"UVXf, uv1))) &&
+	     (he    = hv_fetch_ent(hv, keysv, FALSE, 0)) &&
+	     (val   = HeVAL(he))) {
+	     char *s;
+
+	      s = SvPV(val, len);
+	      if (len == 1)
+		   len = uvuni_to_utf8(ustrp, NATIVE_TO_UNI(*(U8*)s)) - ustrp;
 	      else {
-		   /* something in the 0x80..0xFF range */
-		   ustrp[0] = UTF8_EIGHT_BIT_HI(c);
-		   ustrp[1] = UTF8_EIGHT_BIT_LO(c);
-		   *lenp = 2;
-	      }
 #ifdef EBCDIC
-	      {
-		   U8 tmpbuf[UTF8_MAXLEN_FOLD+1];
-		   U8 *d = tmpbuf;
-		   U8 *t, *tend;
-		   STRLEN tlen;
-
-		   for (t = ustrp, tend = t + *lenp; t < tend; t += tlen) {
-			UV c = utf8_to_uvchr(t, &tlen);
-			d = uvchr_to_utf8(d, UNI_TO_NATIVE(c));
+		   /* If we have EBCDIC we need to remap the characters
+		    * since any characters in the low 256 are Unicode
+		    * code points, not EBCDIC. */
+		   U8 *t = (U8*)s, *tend = t + len, *d;
+		
+		   d = tmpbuf;
+		   if (SvUTF8(val)) {
+			STRLEN tlen = 0;
+			
+			while (t < tend) {
+			     UV c = utf8_to_uvchr(t, &tlen);
+			     if (tlen > 0) {
+				  d = uvchr_to_utf8(d, UNI_TO_NATIVE(c));
+				  t += tlen;
+			     }
+			     else
+				  break;
+			}
 		   }
-		   *lenp = d - tmpbuf; 
-		   Copy(tmpbuf, ustrp, *lenp, U8);
-	      }
+		   else {
+			while (t < tend) {
+			     d = uvchr_to_utf8(d, UNI_TO_NATIVE(*t));
+			     t++;
+			}
+		   }
+		   len = d - tmpbuf;
+		   Copy(tmpbuf, ustrp, len, U8);
+#else
+		   Copy(s, ustrp, len, U8);
 #endif
-	      return utf8_to_uvchr(ustrp, 0);
+	      }
+	 }
+    }
+
+    if (!len && *swashp) {
+	 UV uv2 = swash_fetch(*swashp, tmpbuf, TRUE);
+	 
+	 if (uv2) {
+	      /* It was "normal" (a single character mapping). */
+	      UV uv3 = UNI_TO_NATIVE(uv2);
+	      
+	      len = uvchr_to_utf8(ustrp, uv3) - ustrp;
 	 }
-	 uv  = NATIVE_TO_UNI(uv);
     }
+
+    if (!len) /* Neither: just copy. */
+	 len = uvchr_to_utf8(ustrp, uv0) - ustrp;
+
     if (lenp)
-       *lenp = UNISKIP(uv);
-    uvuni_to_utf8(ustrp, uv);
-    return uv;
+	 *lenp = len;
+
+    return len ? utf8_to_uvchr(ustrp, 0) : 0;
 }
 
 /*
@@ -1457,9 +1543,12 @@ Perl_swash_init(pTHX_ char* pkg, char* name, SV *listsv, I32 minbits, I32 none)
     SAVEI32(PL_hints);
     PL_hints = 0;
     save_re_context();
-    if (PL_curcop == &PL_compiling)
+    if (PL_curcop == &PL_compiling) {
 	/* XXX ought to be handled by lex_start */
+	SAVEI32(PL_in_my);
+	PL_in_my = 0;
 	sv_setpv(tokenbufsv, PL_tokenbuf);
+    }
     errsv_save = newSVsv(ERRSV);
     if (call_method("SWASHNEW", G_SCALAR))
 	retval = newSVsv(*PL_stack_sp--);
@@ -1475,10 +1564,14 @@ Perl_swash_init(pTHX_ char* pkg, char* name, SV *listsv, I32 minbits, I32 none)
 	char* pv = SvPV(tokenbufsv, len);
 
 	Copy(pv, PL_tokenbuf, len+1, char);
-	PL_curcop->op_private = PL_hints;
+	PL_curcop->op_private = (U8)(PL_hints & HINT_PRIVATE_MASK);
     }
-    if (!SvROK(retval) || SvTYPE(SvRV(retval)) != SVt_PVHV)
+    if (!SvROK(retval) || SvTYPE(SvRV(retval)) != SVt_PVHV) {
+        if (SvPOK(retval))
+	    Perl_croak(aTHX_ "Can't find Unicode property definition \"%"SVf"\"",
+		       retval);
 	Perl_croak(aTHX_ "SWASHNEW didn't return an HV ref");
+    }
     return retval;
 }
 
@@ -1503,8 +1596,8 @@ Perl_swash_fetch(pTHX_ SV *sv, U8 *ptr, bool do_utf8)
     UV c = NATIVE_TO_ASCII(*ptr);
 
     if (!do_utf8 && !UNI_IS_INVARIANT(c)) {
-        tmputf8[0] = UTF8_EIGHT_BIT_HI(c);
-        tmputf8[1] = UTF8_EIGHT_BIT_LO(c);
+        tmputf8[0] = (U8)UTF8_EIGHT_BIT_HI(c);
+        tmputf8[1] = (U8)UTF8_EIGHT_BIT_LO(c);
         ptr = tmputf8;
     }
     /* Given a UTF-X encoded char 0xAA..0xYY,0xZZ
@@ -1556,7 +1649,9 @@ Perl_swash_fetch(pTHX_ SV *sv, U8 *ptr, bool do_utf8)
 	    /* We use utf8n_to_uvuni() as we want an index into
 	       Unicode tables, not a native character number.
 	     */
-	    UV code_point = utf8n_to_uvuni(ptr, UTF8_MAXLEN, NULL, 0);
+	    UV code_point = utf8n_to_uvuni(ptr, UTF8_MAXLEN, 0,
+					   ckWARN(WARN_UTF8) ?
+					   0 : UTF8_ALLOW_ANY);
 	    SV *errsv_save;
 	    ENTER;
 	    SAVETMPS;
@@ -1582,7 +1677,7 @@ Perl_swash_fetch(pTHX_ SV *sv, U8 *ptr, bool do_utf8)
 	    FREETMPS;
 	    LEAVE;
 	    if (PL_curcop == &PL_compiling)
-		PL_curcop->op_private = PL_hints;
+		PL_curcop->op_private = (U8)(PL_hints & HINT_PRIVATE_MASK);
 
 	    svp = hv_store(hv, (char*)ptr, klen, retval, 0);
 
@@ -1719,7 +1814,7 @@ Perl_pv_uni_display(pTHX_ SV *dsv, U8 *spv, STRLEN len, STRLEN pvlim, UV flags)
 		 case '\a':
 		     Perl_sv_catpvf(aTHX_ dsv, "\\a"); ok = TRUE; break;
 		 case '\\':
-		     Perl_sv_catpvf(aTHX_ dsv, "\\" ); ok = TRUE; break;
+		     Perl_sv_catpvf(aTHX_ dsv, "\\\\" ); ok = TRUE; break;
 		 default: break;
 		 }
 	     }
@@ -1798,11 +1893,11 @@ Perl_ibcmp_utf8(pTHX_ const char *s1, char **pe1, register UV l1, bool u1, const
      
      if (pe1)
 	  e1 = *(U8**)pe1;
-     if (e1 == 0 || (l1 && l1 < e1 - (U8*)s1))
+     if (e1 == 0 || (l1 && l1 < (UV)(e1 - (U8*)s1)))
 	  f1 = (U8*)s1 + l1;
      if (pe2)
 	  e2 = *(U8**)pe2;
-     if (e2 == 0 || (l2 && l2 < e2 - (U8*)s2))
+     if (e2 == 0 || (l2 && l2 < (UV)(e2 - (U8*)s2)))
 	  f2 = (U8*)s2 + l2;
 
      if ((e1 == 0 && f1 == 0) || (e2 == 0 && f2 == 0) || (f1 == 0 && f2 == 0))
@@ -1819,7 +1914,7 @@ Perl_ibcmp_utf8(pTHX_ const char *s1, char **pe1, register UV l1, bool u1, const
 	       if (u1)
 		    to_utf8_fold(p1, foldbuf1, &foldlen1);
 	       else {
-		    natbuf[0] = NATIVE_TO_UNI(*p1);
+		    natbuf[0] = *p1;
 		    to_utf8_fold(natbuf, foldbuf1, &foldlen1);
 	       }
 	       q1 = foldbuf1;
@@ -1829,7 +1924,7 @@ Perl_ibcmp_utf8(pTHX_ const char *s1, char **pe1, register UV l1, bool u1, const
 	       if (u2)
 		    to_utf8_fold(p2, foldbuf2, &foldlen2);
 	       else {
-		    natbuf[0] = NATIVE_TO_UNI(*p2);
+		    natbuf[0] = *p2;
 		    to_utf8_fold(natbuf, foldbuf2, &foldlen2);
 	       }
 	       q2 = foldbuf2;