threads::shared disabling

[p5sagit/p5-mst-13.2.git] / regexec.c
diff --git a/regexec.c b/regexec.c

index 4db4729..51b55f6 100644 (file)
--- a/regexec.c
+++ b/regexec.c
@@ -980,37 +980,28 @@ S_find_byclass(pTHX_ regexp * prog, regnode *c, char *s, char *strend, char *sta
                U8 tmpbuf [UTF8_MAXLEN+1];
                U8 foldbuf[UTF8_MAXLEN_FOLD+1];
                STRLEN len, foldlen;
-
-               /* The ibcmp_utf8() uses to_uni_fold() which is more
-                * correct folding for Unicode than using lowercase.
-                * However, it doesn't work quite fully since the folding
-                * is a one-to-many mapping and the regex optimizer is
-                * unaware of this, so it may throw out good matches.
-                * Fortunately, not getting this right is allowed
-                * for Unicode Regular Expression Support level 1,
-                * only one-to-one matching is required. --jhi */
-
+               char* se;
+               
                if (c1 == c2) {
                    while (s <= e) {
                        c = utf8_to_uvchr((U8*)s, &len);
                        if ( c == c1
                             && (ln == len ||
-                                !ibcmp_utf8(s, do_utf8,
-                                            strend - s > ln ? ln : strend - s,
-                                            m, UTF, ln))
+                                ((se = e + 1) &&
+                                 !ibcmp_utf8(s, &se, 0,  do_utf8,
+                                             m, 0  , ln, UTF)))
                             && (norun || regtry(prog, s)) )
                            goto got_it;
                        else {
                             uvchr_to_utf8(tmpbuf, c);
-                            to_utf8_fold(tmpbuf, foldbuf, &foldlen);
-                            f = utf8_to_uvchr(foldbuf, 0);
+                            f = to_utf8_fold(tmpbuf, foldbuf, &foldlen);
                             if ( f != c
                                  && (f == c1 || f == c2)
                                  && (ln == foldlen ||
                                      !ibcmp_utf8((char *)foldbuf,
-                                                 do_utf8,
-                                                 foldlen > ln ? ln : foldlen,
-                                                 m, UTF, ln))
+                                                 0, foldlen, do_utf8,
+                                                 m,
+                                                 0, ln,      UTF))
                                  && (norun || regtry(prog, s)) )
                                  goto got_it;
                        }
@@ -1034,22 +1025,21 @@ S_find_byclass(pTHX_ regexp * prog, regnode *c, char *s, char *strend, char *sta
 
                        if ( (c == c1 || c == c2)
                             && (ln == len ||
-                                !ibcmp_utf8(s, do_utf8,
-                                            strend - s > ln ? ln : strend - s,
-                                            m, UTF, ln))
+                                ((se = e + 1) &&
+                                 !ibcmp_utf8(s, &se, 0,  do_utf8,
+                                             m, 0,   ln, UTF)))
                             && (norun || regtry(prog, s)) )
                            goto got_it;
                        else {
                             uvchr_to_utf8(tmpbuf, c);
-                            to_utf8_fold(tmpbuf, foldbuf, &foldlen);
-                            f = utf8_to_uvchr(foldbuf, 0);
+                            f = to_utf8_fold(tmpbuf, foldbuf, &foldlen);
                             if ( f != c
                                  && (f == c1 || f == c2)
                                  && (ln == foldlen ||
                                      !ibcmp_utf8((char *)foldbuf,
-                                                 do_utf8,
-                                                 foldlen > ln ? ln : foldlen,
-                                                 m, UTF, ln))
+                                                 0, foldlen, do_utf8,
+                                                 m,
+                                                 0, ln,      UTF))
                                  && (norun || regtry(prog, s)) )
                                  goto got_it;
                        }
@@ -2351,99 +2341,17 @@ S_regmatch(pTHX_ regnode *prog)
            s = STRING(scan);
            ln = STR_LEN(scan);
 
-           {
+           if (do_utf8 || UTF) {
+             /* Either target or the pattern are utf8. */
                char *l = locinput;
-               char *e = s + ln;
+               char *e = PL_regeol;
 
-               if (do_utf8 != (UTF!=0)) {
-                    /* The target and the pattern have differing utf8ness. */
-                    STRLEN ulen1, ulen2;
-                    UV cs, cl;
-
-                    if (do_utf8) {
-                         /* The target is utf8, the pattern is not utf8. */
-                         while (s < e) {
-                              if (l >= PL_regeol)
-                                   sayNO;
-
-                              cs = to_uni_fold(NATIVE_TO_UNI(*(U8*)s),
-                                               (U8*)s, &ulen1);
-                              cl = utf8_to_uvchr((U8*)l, &ulen2);
-
-                              if (cs != cl) {
-                                   cl = to_uni_fold(cl, (U8*)l, &ulen2);
-                                   if (ulen1 != ulen2 || cs != cl)
-                                        sayNO;
-                              }
-                              l += ulen1;
-                              s ++;
-                         }
-                    }
-                    else {
-                         /* The target is not utf8, the pattern is utf8. */
-                         while (s < e) {
-                              if (l >= PL_regeol)
-                                   sayNO;
-
-                              cs = utf8_to_uvchr((U8*)s, &ulen1);
-
-                              cl = to_uni_fold(NATIVE_TO_UNI(*(U8*)l),
-                                               (U8*)l, &ulen2);
-
-                              if (cs != cl) {
-                                   cs = to_uni_fold(cs, (U8*)s, &ulen1);
-                                   if (ulen1 != ulen2 || cs != cl)
-                                        sayNO;
-                              }
-                              l ++;
-                              s += ulen1;
-                         }
-                    }
-                    locinput = l;
-                    nextchr = UCHARAT(locinput);
-                    break;
-               }
-
-               if (do_utf8 && UTF) {
-                    /* Both the target and the pattern are utf8. */
-                    U8 lfoldbuf[UTF8_MAXLEN_FOLD+1], *lf;
-                    U8 sfoldbuf[UTF8_MAXLEN_FOLD+1], *sf;
-                    STRLEN lfoldlen, sfoldlen;
-                    STRLEN llen = 0;
-                    STRLEN slen = 0;
-
-                    while (s < e) {
-                         /* Fold them and walk them characterwise.  */
-
-                         if (llen == 0) {
-                              to_utf8_fold((U8*)l, lfoldbuf, &lfoldlen);
-                              lf   = lfoldbuf;
-                              llen = lfoldlen;
-                         }
-
-                         if (slen == 0) {
-                              to_utf8_fold((U8*)s, sfoldbuf, &sfoldlen);
-                              sf   = sfoldbuf;
-                              slen = sfoldlen;
-                         }
-
-                         while (llen && slen) {
-                              if (UTF8SKIP(lf) != UTF8SKIP(sf) ||
-                                  memNE((char*)lf, (char*)sf, UTF8SKIP(lf)))
-                                   sayNO;
-                              llen -= UTF8SKIP(lf);
-                              lf   += UTF8SKIP(lf);
-                              slen -= UTF8SKIP(sf);
-                              sf   += UTF8SKIP(sf);
-                         }
-                         
-                         l += UTF8SKIP(l);
-                         s += UTF8SKIP(s);
-                    }
-                    locinput = l;
-                    nextchr = UCHARAT(locinput);
-                    break;
-               }
+               if (ibcmp_utf8(s, 0,  ln, do_utf8,
+                              l, &e, 0,  UTF))
+                    sayNO;
+               locinput = e;
+               nextchr = UCHARAT(locinput);
+               break;
            }
 
            /* Neither the target and the pattern are utf8. */
@@ -4261,14 +4169,14 @@ S_reginclass(pTHX_ register regnode *n, register U8* p, register bool do_utf8)
                if (swash_fetch(sw, p, do_utf8))
                    match = TRUE;
                else if (flags & ANYOF_FOLD) {
-                   STRLEN ulen;
-                   U8 tmpbuf[UTF8_MAXLEN_FOLD+1];
+                   U8 foldbuf[UTF8_MAXLEN_FOLD+1];
+                   STRLEN foldlen;
 
-                   to_utf8_fold(p, tmpbuf, &ulen);
-                   if (swash_fetch(sw, tmpbuf, do_utf8))
+                   to_utf8_fold(p, foldbuf, &foldlen);
+                   if (swash_fetch(sw, foldbuf, do_utf8))
                        match = TRUE;
-                   to_utf8_upper(p, tmpbuf, &ulen);
-                   if (swash_fetch(sw, tmpbuf, do_utf8))
+                   to_utf8_upper(p, foldbuf, &foldlen);
+                   if (swash_fetch(sw, foldbuf, do_utf8))
                        match = TRUE;
                }
            }