#define HOP3c(pos,off,lim) ((char*)HOP3(pos,off,lim))
#define HOPMAYBE3c(pos,off,lim) ((char*)HOPMAYBE3(pos,off,lim))
-#define LOAD_UTF8_CHARCLASS(a,b) STMT_START { if (!CAT2(PL_utf8_,a)) { ENTER; save_re_context(); (void)CAT2(is_utf8_, a)((const U8*)b); LEAVE; } } STMT_END
+#define LOAD_UTF8_CHARCLASS(class,str) STMT_START { \
+ if (!CAT2(PL_utf8_,class)) { bool ok; ENTER; save_re_context(); ok=CAT2(is_utf8_,class)((const U8*)str); assert(ok); LEAVE; } } STMT_END
+#define LOAD_UTF8_CHARCLASS_ALNUM() LOAD_UTF8_CHARCLASS(alnum,"a")
+#define LOAD_UTF8_CHARCLASS_DIGIT() LOAD_UTF8_CHARCLASS(digit,"0")
+#define LOAD_UTF8_CHARCLASS_SPACE() LOAD_UTF8_CHARCLASS(space," ")
+#define LOAD_UTF8_CHARCLASS_MARK() LOAD_UTF8_CHARCLASS(mark, "\xcd\x86")
/* for use after a quantifier and before an EXACT-like node -- japhy */
#define JUMPABLE(rn) ( \
}
tmp = ((OP(c) == BOUND ?
isALNUM_uni(tmp) : isALNUM_LC_uvchr(UNI_TO_NATIVE(tmp))) != 0);
- LOAD_UTF8_CHARCLASS(alnum,"a");
+ LOAD_UTF8_CHARCLASS_ALNUM();
while (s + (uskip = UTF8SKIP(s)) <= strend) {
if (tmp == !(OP(c) == BOUND ?
swash_fetch(PL_utf8_alnum, (U8*)s, do_utf8) :
}
tmp = ((OP(c) == NBOUND ?
isALNUM_uni(tmp) : isALNUM_LC_uvchr(UNI_TO_NATIVE(tmp))) != 0);
- LOAD_UTF8_CHARCLASS(alnum,"a");
+ LOAD_UTF8_CHARCLASS_ALNUM();
while (s + (uskip = UTF8SKIP(s)) <= strend) {
if (tmp == !(OP(c) == NBOUND ?
swash_fetch(PL_utf8_alnum, (U8*)s, do_utf8) :
break;
case ALNUM:
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(alnum,"a");
+ LOAD_UTF8_CHARCLASS_ALNUM();
while (s + (uskip = UTF8SKIP(s)) <= strend) {
if (swash_fetch(PL_utf8_alnum, (U8*)s, do_utf8)) {
if (tmp && (norun || regtry(prog, s)))
break;
case NALNUM:
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(alnum,"a");
+ LOAD_UTF8_CHARCLASS_ALNUM();
while (s + (uskip = UTF8SKIP(s)) <= strend) {
if (!swash_fetch(PL_utf8_alnum, (U8*)s, do_utf8)) {
if (tmp && (norun || regtry(prog, s)))
break;
case SPACE:
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(space," ");
+ LOAD_UTF8_CHARCLASS_SPACE();
while (s + (uskip = UTF8SKIP(s)) <= strend) {
if (*s == ' ' || swash_fetch(PL_utf8_space,(U8*)s, do_utf8)) {
if (tmp && (norun || regtry(prog, s)))
break;
case NSPACE:
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(space," ");
+ LOAD_UTF8_CHARCLASS_SPACE();
while (s + (uskip = UTF8SKIP(s)) <= strend) {
if (!(*s == ' ' || swash_fetch(PL_utf8_space,(U8*)s, do_utf8))) {
if (tmp && (norun || regtry(prog, s)))
break;
case DIGIT:
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(digit,"0");
+ LOAD_UTF8_CHARCLASS_DIGIT();
while (s + (uskip = UTF8SKIP(s)) <= strend) {
if (swash_fetch(PL_utf8_digit,(U8*)s, do_utf8)) {
if (tmp && (norun || regtry(prog, s)))
break;
case NDIGIT:
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(digit,"0");
+ LOAD_UTF8_CHARCLASS_DIGIT();
while (s + (uskip = UTF8SKIP(s)) <= strend) {
if (!swash_fetch(PL_utf8_digit,(U8*)s, do_utf8)) {
if (tmp && (norun || regtry(prog, s)))
if (!nextchr)
sayNO;
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(alnum,"a");
+ LOAD_UTF8_CHARCLASS_ALNUM();
if (!(OP(scan) == ALNUM
? swash_fetch(PL_utf8_alnum, (U8*)locinput, do_utf8)
: isALNUM_LC_utf8((U8*)locinput)))
if (!nextchr && locinput >= PL_regeol)
sayNO;
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(alnum,"a");
+ LOAD_UTF8_CHARCLASS_ALNUM();
if (OP(scan) == NALNUM
? swash_fetch(PL_utf8_alnum, (U8*)locinput, do_utf8)
: isALNUM_LC_utf8((U8*)locinput))
}
if (OP(scan) == BOUND || OP(scan) == NBOUND) {
ln = isALNUM_uni(ln);
- LOAD_UTF8_CHARCLASS(alnum,"a");
+ LOAD_UTF8_CHARCLASS_ALNUM();
n = swash_fetch(PL_utf8_alnum, (U8*)locinput, do_utf8);
}
else {
sayNO;
if (do_utf8) {
if (UTF8_IS_CONTINUED(nextchr)) {
- LOAD_UTF8_CHARCLASS(space," ");
+ LOAD_UTF8_CHARCLASS_SPACE();
if (!(OP(scan) == SPACE
? swash_fetch(PL_utf8_space, (U8*)locinput, do_utf8)
: isSPACE_LC_utf8((U8*)locinput)))
if (!nextchr && locinput >= PL_regeol)
sayNO;
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(space," ");
+ LOAD_UTF8_CHARCLASS_SPACE();
if (OP(scan) == NSPACE
? swash_fetch(PL_utf8_space, (U8*)locinput, do_utf8)
: isSPACE_LC_utf8((U8*)locinput))
if (!nextchr)
sayNO;
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(digit,"0");
+ LOAD_UTF8_CHARCLASS_DIGIT();
if (!(OP(scan) == DIGIT
? swash_fetch(PL_utf8_digit, (U8*)locinput, do_utf8)
: isDIGIT_LC_utf8((U8*)locinput)))
if (!nextchr && locinput >= PL_regeol)
sayNO;
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(digit,"0");
+ LOAD_UTF8_CHARCLASS_DIGIT();
if (OP(scan) == NDIGIT
? swash_fetch(PL_utf8_digit, (U8*)locinput, do_utf8)
: isDIGIT_LC_utf8((U8*)locinput))
if (locinput >= PL_regeol)
sayNO;
if (do_utf8) {
- LOAD_UTF8_CHARCLASS(mark,"~");
+ LOAD_UTF8_CHARCLASS_MARK();
if (swash_fetch(PL_utf8_mark,(U8*)locinput, do_utf8))
sayNO;
locinput += PL_utf8skip[nextchr];
case ALNUM:
if (do_utf8) {
loceol = PL_regeol;
- LOAD_UTF8_CHARCLASS(alnum,"a");
+ LOAD_UTF8_CHARCLASS_ALNUM();
while (hardcount < max && scan < loceol &&
swash_fetch(PL_utf8_alnum, (U8*)scan, do_utf8)) {
scan += UTF8SKIP(scan);
case NALNUM:
if (do_utf8) {
loceol = PL_regeol;
- LOAD_UTF8_CHARCLASS(alnum,"a");
+ LOAD_UTF8_CHARCLASS_ALNUM();
while (hardcount < max && scan < loceol &&
!swash_fetch(PL_utf8_alnum, (U8*)scan, do_utf8)) {
scan += UTF8SKIP(scan);
case SPACE:
if (do_utf8) {
loceol = PL_regeol;
- LOAD_UTF8_CHARCLASS(space," ");
+ LOAD_UTF8_CHARCLASS_SPACE();
while (hardcount < max && scan < loceol &&
(*scan == ' ' ||
swash_fetch(PL_utf8_space,(U8*)scan, do_utf8))) {
case NSPACE:
if (do_utf8) {
loceol = PL_regeol;
- LOAD_UTF8_CHARCLASS(space," ");
+ LOAD_UTF8_CHARCLASS_SPACE();
while (hardcount < max && scan < loceol &&
!(*scan == ' ' ||
swash_fetch(PL_utf8_space,(U8*)scan, do_utf8))) {
case DIGIT:
if (do_utf8) {
loceol = PL_regeol;
- LOAD_UTF8_CHARCLASS(digit,"0");
+ LOAD_UTF8_CHARCLASS_DIGIT();
while (hardcount < max && scan < loceol &&
swash_fetch(PL_utf8_digit, (U8*)scan, do_utf8)) {
scan += UTF8SKIP(scan);
case NDIGIT:
if (do_utf8) {
loceol = PL_regeol;
- LOAD_UTF8_CHARCLASS(digit,"0");
+ LOAD_UTF8_CHARCLASS_DIGIT();
while (hardcount < max && scan < loceol &&
!swash_fetch(PL_utf8_digit, (U8*)scan, do_utf8)) {
scan += UTF8SKIP(scan);