mirror of
https://github.com/cloudwu/skynet.git
synced 2026-07-23 19:43:09 +00:00
upgrade lpeg to 1.1.0 (#1766)
* upgrade lpeg to 1.1.0 * fix lpeg compiler error
This commit is contained in:
@@ -1,7 +1,3 @@
|
||||
/*
|
||||
** $Id: lpcode.c $
|
||||
** Copyright 2007, Lua.org & PUC-Rio (see 'lpeg.html' for license)
|
||||
*/
|
||||
|
||||
#include <limits.h>
|
||||
|
||||
@@ -11,6 +7,7 @@
|
||||
|
||||
#include "lptypes.h"
|
||||
#include "lpcode.h"
|
||||
#include "lpcset.h"
|
||||
|
||||
|
||||
/* signals a "no-instruction */
|
||||
@@ -26,61 +23,13 @@ static const Charset fullset_ =
|
||||
|
||||
static const Charset *fullset = &fullset_;
|
||||
|
||||
|
||||
/*
|
||||
** {======================================================
|
||||
** Analysis and some optimizations
|
||||
** =======================================================
|
||||
*/
|
||||
|
||||
/*
|
||||
** Check whether a charset is empty (returns IFail), singleton (IChar),
|
||||
** full (IAny), or none of those (ISet). When singleton, '*c' returns
|
||||
** which character it is. (When generic set, the set was the input,
|
||||
** so there is no need to return it.)
|
||||
*/
|
||||
static Opcode charsettype (const byte *cs, int *c) {
|
||||
int count = 0; /* number of characters in the set */
|
||||
int i;
|
||||
int candidate = -1; /* candidate position for the singleton char */
|
||||
for (i = 0; i < CHARSETSIZE; i++) { /* for each byte */
|
||||
int b = cs[i];
|
||||
if (b == 0) { /* is byte empty? */
|
||||
if (count > 1) /* was set neither empty nor singleton? */
|
||||
return ISet; /* neither full nor empty nor singleton */
|
||||
/* else set is still empty or singleton */
|
||||
}
|
||||
else if (b == 0xFF) { /* is byte full? */
|
||||
if (count < (i * BITSPERCHAR)) /* was set not full? */
|
||||
return ISet; /* neither full nor empty nor singleton */
|
||||
else count += BITSPERCHAR; /* set is still full */
|
||||
}
|
||||
else if ((b & (b - 1)) == 0) { /* has byte only one bit? */
|
||||
if (count > 0) /* was set not empty? */
|
||||
return ISet; /* neither full nor empty nor singleton */
|
||||
else { /* set has only one char till now; track it */
|
||||
count++;
|
||||
candidate = i;
|
||||
}
|
||||
}
|
||||
else return ISet; /* byte is neither empty, full, nor singleton */
|
||||
}
|
||||
switch (count) {
|
||||
case 0: return IFail; /* empty set */
|
||||
case 1: { /* singleton; find character bit inside byte */
|
||||
int b = cs[candidate];
|
||||
*c = candidate * BITSPERCHAR;
|
||||
if ((b & 0xF0) != 0) { *c += 4; b >>= 4; }
|
||||
if ((b & 0x0C) != 0) { *c += 2; b >>= 2; }
|
||||
if ((b & 0x02) != 0) { *c += 1; }
|
||||
return IChar;
|
||||
}
|
||||
default: {
|
||||
assert(count == CHARSETSIZE * BITSPERCHAR); /* full set */
|
||||
return IAny;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** A few basic operations on Charsets
|
||||
@@ -89,42 +38,12 @@ static void cs_complement (Charset *cs) {
|
||||
loopset(i, cs->cs[i] = ~cs->cs[i]);
|
||||
}
|
||||
|
||||
static int cs_equal (const byte *cs1, const byte *cs2) {
|
||||
loopset(i, if (cs1[i] != cs2[i]) return 0);
|
||||
return 1;
|
||||
}
|
||||
|
||||
static int cs_disjoint (const Charset *cs1, const Charset *cs2) {
|
||||
loopset(i, if ((cs1->cs[i] & cs2->cs[i]) != 0) return 0;)
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** If 'tree' is a 'char' pattern (TSet, TChar, TAny), convert it into a
|
||||
** charset and return 1; else return 0.
|
||||
*/
|
||||
int tocharset (TTree *tree, Charset *cs) {
|
||||
switch (tree->tag) {
|
||||
case TSet: { /* copy set */
|
||||
loopset(i, cs->cs[i] = treebuffer(tree)[i]);
|
||||
return 1;
|
||||
}
|
||||
case TChar: { /* only one char */
|
||||
assert(0 <= tree->u.n && tree->u.n <= UCHAR_MAX);
|
||||
loopset(i, cs->cs[i] = 0); /* erase all chars */
|
||||
setchar(cs->cs, tree->u.n); /* add that one */
|
||||
return 1;
|
||||
}
|
||||
case TAny: {
|
||||
loopset(i, cs->cs[i] = 0xFF); /* add all characters to the set */
|
||||
return 1;
|
||||
}
|
||||
default: return 0;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** Visit a TCall node taking care to stop recursion. If node not yet
|
||||
** visited, return 'f(sib2(tree))', otherwise return 'def' (default
|
||||
@@ -196,7 +115,7 @@ int hascaptures (TTree *tree) {
|
||||
int checkaux (TTree *tree, int pred) {
|
||||
tailcall:
|
||||
switch (tree->tag) {
|
||||
case TChar: case TSet: case TAny:
|
||||
case TChar: case TSet: case TAny: case TUTFR:
|
||||
case TFalse: case TOpenCall:
|
||||
return 0; /* not nullable */
|
||||
case TRep: case TTrue:
|
||||
@@ -220,7 +139,7 @@ int checkaux (TTree *tree, int pred) {
|
||||
if (checkaux(sib2(tree), pred)) return 1;
|
||||
/* else return checkaux(sib1(tree), pred); */
|
||||
tree = sib1(tree); goto tailcall;
|
||||
case TCapture: case TGrammar: case TRule:
|
||||
case TCapture: case TGrammar: case TRule: case TXInfo:
|
||||
/* return checkaux(sib1(tree), pred); */
|
||||
tree = sib1(tree); goto tailcall;
|
||||
case TCall: /* return checkaux(sib2(tree), pred); */
|
||||
@@ -239,11 +158,13 @@ int fixedlen (TTree *tree) {
|
||||
switch (tree->tag) {
|
||||
case TChar: case TSet: case TAny:
|
||||
return len + 1;
|
||||
case TUTFR:
|
||||
return (tree->cap == sib1(tree)->cap) ? len + tree->cap : -1;
|
||||
case TFalse: case TTrue: case TNot: case TAnd: case TBehind:
|
||||
return len;
|
||||
case TRep: case TRunTime: case TOpenCall:
|
||||
return -1;
|
||||
case TCapture: case TRule: case TGrammar:
|
||||
case TCapture: case TRule: case TGrammar: case TXInfo:
|
||||
/* return fixedlen(sib1(tree)); */
|
||||
tree = sib1(tree); goto tailcall;
|
||||
case TCall: {
|
||||
@@ -294,18 +215,21 @@ int fixedlen (TTree *tree) {
|
||||
static int getfirst (TTree *tree, const Charset *follow, Charset *firstset) {
|
||||
tailcall:
|
||||
switch (tree->tag) {
|
||||
case TChar: case TSet: case TAny: {
|
||||
case TChar: case TSet: case TAny: case TFalse: {
|
||||
tocharset(tree, firstset);
|
||||
return 0;
|
||||
}
|
||||
case TUTFR: {
|
||||
int c;
|
||||
clearset(firstset->cs); /* erase all chars */
|
||||
for (c = tree->key; c <= sib1(tree)->key; c++)
|
||||
setchar(firstset->cs, c);
|
||||
return 0;
|
||||
}
|
||||
case TTrue: {
|
||||
loopset(i, firstset->cs[i] = follow->cs[i]);
|
||||
return 1; /* accepts the empty string */
|
||||
}
|
||||
case TFalse: {
|
||||
loopset(i, firstset->cs[i] = 0);
|
||||
return 0;
|
||||
}
|
||||
case TChoice: {
|
||||
Charset csaux;
|
||||
int e1 = getfirst(sib1(tree), follow, firstset);
|
||||
@@ -334,7 +258,7 @@ static int getfirst (TTree *tree, const Charset *follow, Charset *firstset) {
|
||||
loopset(i, firstset->cs[i] |= follow->cs[i]);
|
||||
return 1; /* accept the empty string */
|
||||
}
|
||||
case TCapture: case TGrammar: case TRule: {
|
||||
case TCapture: case TGrammar: case TRule: case TXInfo: {
|
||||
/* return getfirst(sib1(tree), follow, firstset); */
|
||||
tree = sib1(tree); goto tailcall;
|
||||
}
|
||||
@@ -356,9 +280,8 @@ static int getfirst (TTree *tree, const Charset *follow, Charset *firstset) {
|
||||
if (tocharset(sib1(tree), firstset)) {
|
||||
cs_complement(firstset);
|
||||
return 1;
|
||||
}
|
||||
/* else go through */
|
||||
}
|
||||
} /* else */
|
||||
} /* FALLTHROUGH */
|
||||
case TBehind: { /* instruction gives no new information */
|
||||
/* call 'getfirst' only to check for math-time captures */
|
||||
int e = getfirst(sib1(tree), follow, firstset);
|
||||
@@ -380,9 +303,9 @@ static int headfail (TTree *tree) {
|
||||
case TChar: case TSet: case TAny: case TFalse:
|
||||
return 1;
|
||||
case TTrue: case TRep: case TRunTime: case TNot:
|
||||
case TBehind:
|
||||
case TBehind: case TUTFR:
|
||||
return 0;
|
||||
case TCapture: case TGrammar: case TRule: case TAnd:
|
||||
case TCapture: case TGrammar: case TRule: case TXInfo: case TAnd:
|
||||
tree = sib1(tree); goto tailcall; /* return headfail(sib1(tree)); */
|
||||
case TCall:
|
||||
tree = sib2(tree); goto tailcall; /* return headfail(sib2(tree)); */
|
||||
@@ -407,7 +330,7 @@ static int headfail (TTree *tree) {
|
||||
static int needfollow (TTree *tree) {
|
||||
tailcall:
|
||||
switch (tree->tag) {
|
||||
case TChar: case TSet: case TAny:
|
||||
case TChar: case TSet: case TAny: case TUTFR:
|
||||
case TFalse: case TTrue: case TAnd: case TNot:
|
||||
case TRunTime: case TGrammar: case TCall: case TBehind:
|
||||
return 0;
|
||||
@@ -418,7 +341,7 @@ static int needfollow (TTree *tree) {
|
||||
case TSeq:
|
||||
tree = sib2(tree); goto tailcall;
|
||||
default: assert(0); return 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* }====================================================== */
|
||||
@@ -437,10 +360,11 @@ static int needfollow (TTree *tree) {
|
||||
*/
|
||||
int sizei (const Instruction *i) {
|
||||
switch((Opcode)i->i.code) {
|
||||
case ISet: case ISpan: return CHARSETINSTSIZE;
|
||||
case ITestSet: return CHARSETINSTSIZE + 1;
|
||||
case ISet: case ISpan: return 1 + i->i.aux2.set.size;
|
||||
case ITestSet: return 2 + i->i.aux2.set.size;
|
||||
case ITestChar: case ITestAny: case IChoice: case IJmp: case ICall:
|
||||
case IOpenCall: case ICommit: case IPartialCommit: case IBackCommit:
|
||||
case IUTFR:
|
||||
return 2;
|
||||
default: return 1;
|
||||
}
|
||||
@@ -468,23 +392,67 @@ static void codegen (CompileState *compst, TTree *tree, int opt, int tt,
|
||||
const Charset *fl);
|
||||
|
||||
|
||||
void realloccode (lua_State *L, Pattern *p, int nsize) {
|
||||
void *ud;
|
||||
lua_Alloc f = lua_getallocf(L, &ud);
|
||||
void *newblock = f(ud, p->code, p->codesize * sizeof(Instruction),
|
||||
nsize * sizeof(Instruction));
|
||||
if (newblock == NULL && nsize > 0)
|
||||
static void finishrelcode (lua_State *L, Pattern *p, Instruction *block,
|
||||
int size) {
|
||||
if (block == NULL)
|
||||
luaL_error(L, "not enough memory");
|
||||
p->code = (Instruction *)newblock;
|
||||
p->codesize = nsize;
|
||||
block->codesize = size;
|
||||
p->code = (Instruction *)block + 1;
|
||||
}
|
||||
|
||||
|
||||
static int nextinstruction (CompileState *compst) {
|
||||
int size = compst->p->codesize;
|
||||
if (compst->ncode >= size)
|
||||
realloccode(compst->L, compst->p, size * 2);
|
||||
return compst->ncode++;
|
||||
/*
|
||||
** Initialize array 'p->code'
|
||||
*/
|
||||
static void newcode (lua_State *L, Pattern *p, int size) {
|
||||
void *ud;
|
||||
Instruction *block;
|
||||
lua_Alloc f = lua_getallocf(L, &ud);
|
||||
size++; /* slot for 'codesize' */
|
||||
block = (Instruction*) f(ud, NULL, 0, size * sizeof(Instruction));
|
||||
finishrelcode(L, p, block, size);
|
||||
}
|
||||
|
||||
|
||||
void freecode (lua_State *L, Pattern *p) {
|
||||
if (p->code != NULL) {
|
||||
void *ud;
|
||||
lua_Alloc f = lua_getallocf(L, &ud);
|
||||
uint osize = p->code[-1].codesize;
|
||||
f(ud, p->code - 1, osize * sizeof(Instruction), 0); /* free block */
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** Assume that 'nsize' is not zero and that 'p->code' already exists.
|
||||
*/
|
||||
static void realloccode (lua_State *L, Pattern *p, int nsize) {
|
||||
void *ud;
|
||||
lua_Alloc f = lua_getallocf(L, &ud);
|
||||
Instruction *block = p->code - 1;
|
||||
uint osize = block->codesize;
|
||||
nsize++; /* add the 'codesize' slot to size */
|
||||
block = (Instruction*) f(ud, block, osize * sizeof(Instruction),
|
||||
nsize * sizeof(Instruction));
|
||||
finishrelcode(L, p, block, nsize);
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** Add space for an instruction with 'n' slots and return its index.
|
||||
*/
|
||||
static int nextinstruction (CompileState *compst, int n) {
|
||||
int size = compst->p->code[-1].codesize - 1;
|
||||
int ncode = compst->ncode;
|
||||
if (ncode > size - n) {
|
||||
uint nsize = size + (size >> 1) + n;
|
||||
if (nsize >= INT_MAX)
|
||||
luaL_error(compst->L, "pattern code too large");
|
||||
realloccode(compst->L, compst->p, nsize);
|
||||
}
|
||||
compst->ncode = ncode + n;
|
||||
return ncode;
|
||||
}
|
||||
|
||||
|
||||
@@ -492,9 +460,9 @@ static int nextinstruction (CompileState *compst) {
|
||||
|
||||
|
||||
static int addinstruction (CompileState *compst, Opcode op, int aux) {
|
||||
int i = nextinstruction(compst);
|
||||
int i = nextinstruction(compst, 1);
|
||||
getinstr(compst, i).i.code = op;
|
||||
getinstr(compst, i).i.aux = aux;
|
||||
getinstr(compst, i).i.aux1 = aux;
|
||||
return i;
|
||||
}
|
||||
|
||||
@@ -518,6 +486,16 @@ static void setoffset (CompileState *compst, int instruction, int offset) {
|
||||
}
|
||||
|
||||
|
||||
static void codeutfr (CompileState *compst, TTree *tree) {
|
||||
int i = addoffsetinst(compst, IUTFR);
|
||||
int to = sib1(tree)->u.n;
|
||||
assert(sib1(tree)->tag == TXInfo);
|
||||
getinstr(compst, i + 1).offset = tree->u.n;
|
||||
getinstr(compst, i).i.aux1 = to & 0xff;
|
||||
getinstr(compst, i).i.aux2.key = to >> 8;
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** Add a capture instruction:
|
||||
** 'op' is the capture instruction; 'cap' the capture kind;
|
||||
@@ -527,7 +505,7 @@ static void setoffset (CompileState *compst, int instruction, int offset) {
|
||||
static int addinstcap (CompileState *compst, Opcode op, int cap, int key,
|
||||
int aux) {
|
||||
int i = addinstruction(compst, op, joinkindoff(cap, aux));
|
||||
getinstr(compst, i).i.key = key;
|
||||
getinstr(compst, i).i.aux2.key = key;
|
||||
return i;
|
||||
}
|
||||
|
||||
@@ -560,7 +538,7 @@ static void jumptohere (CompileState *compst, int instruction) {
|
||||
*/
|
||||
static void codechar (CompileState *compst, int c, int tt) {
|
||||
if (tt >= 0 && getinstr(compst, tt).i.code == ITestChar &&
|
||||
getinstr(compst, tt).i.aux == c)
|
||||
getinstr(compst, tt).i.aux1 == c)
|
||||
addinstruction(compst, IAny, 0);
|
||||
else
|
||||
addinstruction(compst, IChar, c);
|
||||
@@ -568,45 +546,63 @@ static void codechar (CompileState *compst, int c, int tt) {
|
||||
|
||||
|
||||
/*
|
||||
** Add a charset posfix to an instruction
|
||||
** Add a charset posfix to an instruction.
|
||||
*/
|
||||
static void addcharset (CompileState *compst, const byte *cs) {
|
||||
int p = gethere(compst);
|
||||
static void addcharset (CompileState *compst, int inst, charsetinfo *info) {
|
||||
int p;
|
||||
Instruction *I = &getinstr(compst, inst);
|
||||
byte *charset;
|
||||
int isize = instsize(info->size); /* size in instructions */
|
||||
int i;
|
||||
for (i = 0; i < (int)CHARSETINSTSIZE - 1; i++)
|
||||
nextinstruction(compst); /* space for buffer */
|
||||
/* fill buffer with charset */
|
||||
loopset(j, getinstr(compst, p).buff[j] = cs[j]);
|
||||
I->i.aux2.set.offset = info->offset * 8; /* offset in bits */
|
||||
I->i.aux2.set.size = isize;
|
||||
I->i.aux1 = info->deflt;
|
||||
p = nextinstruction(compst, isize); /* space for charset */
|
||||
charset = getinstr(compst, p).buff; /* charset buffer */
|
||||
for (i = 0; i < isize * (int)sizeof(Instruction); i++)
|
||||
charset[i] = getbytefromcharset(info, i); /* copy the buffer */
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** code a char set, optimizing unit sets for IChar, "complete"
|
||||
** sets for IAny, and empty sets for IFail; also use an IAny
|
||||
** when instruction is dominated by an equivalent test.
|
||||
** Check whether charset 'info' is dominated by instruction 'p'
|
||||
*/
|
||||
static void codecharset (CompileState *compst, const byte *cs, int tt) {
|
||||
int c = 0; /* (=) to avoid warnings */
|
||||
Opcode op = charsettype(cs, &c);
|
||||
switch (op) {
|
||||
case IChar: codechar(compst, c, tt); break;
|
||||
case ISet: { /* non-trivial set? */
|
||||
if (tt >= 0 && getinstr(compst, tt).i.code == ITestSet &&
|
||||
cs_equal(cs, getinstr(compst, tt + 2).buff))
|
||||
addinstruction(compst, IAny, 0);
|
||||
else {
|
||||
addinstruction(compst, ISet, 0);
|
||||
addcharset(compst, cs);
|
||||
}
|
||||
break;
|
||||
static int cs_equal (Instruction *p, charsetinfo *info) {
|
||||
if (p->i.code != ITestSet)
|
||||
return 0;
|
||||
else if (p->i.aux2.set.offset != info->offset * 8 ||
|
||||
p->i.aux2.set.size != instsize(info->size) ||
|
||||
p->i.aux1 != info->deflt)
|
||||
return 0;
|
||||
else {
|
||||
int i;
|
||||
for (i = 0; i < instsize(info->size) * (int)sizeof(Instruction); i++) {
|
||||
if ((p + 2)->buff[i] != getbytefromcharset(info, i))
|
||||
return 0;
|
||||
}
|
||||
default: addinstruction(compst, op, c); break;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** Code a char set, using IAny when instruction is dominated by an
|
||||
** equivalent test.
|
||||
*/
|
||||
static void codecharset (CompileState *compst, TTree *tree, int tt) {
|
||||
charsetinfo info;
|
||||
tree2cset(tree, &info);
|
||||
if (tt >= 0 && cs_equal(&getinstr(compst, tt), &info))
|
||||
addinstruction(compst, IAny, 0);
|
||||
else {
|
||||
int i = addinstruction(compst, ISet, 0);
|
||||
addcharset(compst, i, &info);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** code a test set, optimizing unit sets for ITestChar, "complete"
|
||||
** Code a test set, optimizing unit sets for ITestChar, "complete"
|
||||
** sets for ITestAny, and empty sets for IJmp (always fails).
|
||||
** 'e' is true iff test should accept the empty string. (Test
|
||||
** instructions in the current VM never accept the empty string.)
|
||||
@@ -614,22 +610,22 @@ static void codecharset (CompileState *compst, const byte *cs, int tt) {
|
||||
static int codetestset (CompileState *compst, Charset *cs, int e) {
|
||||
if (e) return NOINST; /* no test */
|
||||
else {
|
||||
int c = 0;
|
||||
Opcode op = charsettype(cs->cs, &c);
|
||||
charsetinfo info;
|
||||
Opcode op = charsettype(cs->cs, &info);
|
||||
switch (op) {
|
||||
case IFail: return addoffsetinst(compst, IJmp); /* always jump */
|
||||
case IAny: return addoffsetinst(compst, ITestAny);
|
||||
case IChar: {
|
||||
int i = addoffsetinst(compst, ITestChar);
|
||||
getinstr(compst, i).i.aux = c;
|
||||
getinstr(compst, i).i.aux1 = info.offset;
|
||||
return i;
|
||||
}
|
||||
case ISet: {
|
||||
default: { /* regular set */
|
||||
int i = addoffsetinst(compst, ITestSet);
|
||||
addcharset(compst, cs->cs);
|
||||
addcharset(compst, i, &info);
|
||||
assert(op == ISet);
|
||||
return i;
|
||||
}
|
||||
default: assert(0); return 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -665,11 +661,11 @@ static void codebehind (CompileState *compst, TTree *tree) {
|
||||
|
||||
/*
|
||||
** Choice; optimizations:
|
||||
** - when p1 is headfail or
|
||||
** when first(p1) and first(p2) are disjoint, than
|
||||
** a character not in first(p1) cannot go to p1, and a character
|
||||
** in first(p1) cannot go to p2 (at it is not in first(p2)).
|
||||
** (The optimization is not valid if p1 accepts the empty string,
|
||||
** - when p1 is headfail or when first(p1) and first(p2) are disjoint,
|
||||
** than a character not in first(p1) cannot go to p1 and a character
|
||||
** in first(p1) cannot go to p2, either because p1 will accept
|
||||
** (headfail) or because it is not in first(p2) (disjoint).
|
||||
** (The second case is not valid if p1 accepts the empty string,
|
||||
** as then there is no character at all...)
|
||||
** - when p2 is empty and opt is true; a IPartialCommit can reuse
|
||||
** the Choice already active in the stack.
|
||||
@@ -686,7 +682,7 @@ static void codechoice (CompileState *compst, TTree *p1, TTree *p2, int opt,
|
||||
int jmp = NOINST;
|
||||
codegen(compst, p1, 0, test, fl);
|
||||
if (!emptyp2)
|
||||
jmp = addoffsetinst(compst, IJmp);
|
||||
jmp = addoffsetinst(compst, IJmp);
|
||||
jumptohere(compst, test);
|
||||
codegen(compst, p2, opt, NOINST, fl);
|
||||
jumptohere(compst, jmp);
|
||||
@@ -697,7 +693,7 @@ static void codechoice (CompileState *compst, TTree *p1, TTree *p2, int opt,
|
||||
codegen(compst, p1, 1, NOINST, fullset);
|
||||
}
|
||||
else {
|
||||
/* <p1 / p2> ==
|
||||
/* <p1 / p2> ==
|
||||
test(first(p1)) -> L1; choice L1; <p1>; commit L2; L1: <p2>; L2: */
|
||||
int pcommit;
|
||||
int test = codetestset(compst, &cs1, e1);
|
||||
@@ -764,9 +760,53 @@ static void coderuntime (CompileState *compst, TTree *tree, int tt) {
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** Create a jump to 'test' and fix 'test' to jump to next instruction
|
||||
*/
|
||||
static void closeloop (CompileState *compst, int test) {
|
||||
int jmp = addoffsetinst(compst, IJmp);
|
||||
jumptohere(compst, test);
|
||||
jumptothere(compst, jmp, test);
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** Try repetition of charsets:
|
||||
** For an empty set, repetition of fail is a no-op;
|
||||
** For any or char, code a tight loop;
|
||||
** For generic charset, use a span instruction.
|
||||
*/
|
||||
static int coderepcharset (CompileState *compst, TTree *tree) {
|
||||
switch (tree->tag) {
|
||||
case TFalse: return 1; /* 'fail*' is a no-op */
|
||||
case TAny: { /* L1: testany -> L2; any; jmp L1; L2: */
|
||||
int test = addoffsetinst(compst, ITestAny);
|
||||
addinstruction(compst, IAny, 0);
|
||||
closeloop(compst, test);
|
||||
return 1;
|
||||
}
|
||||
case TChar: { /* L1: testchar c -> L2; any; jmp L1; L2: */
|
||||
int test = addoffsetinst(compst, ITestChar);
|
||||
getinstr(compst, test).i.aux1 = tree->u.n;
|
||||
addinstruction(compst, IAny, 0);
|
||||
closeloop(compst, test);
|
||||
return 1;
|
||||
}
|
||||
case TSet: { /* regular set */
|
||||
charsetinfo info;
|
||||
int i = addinstruction(compst, ISpan, 0);
|
||||
tree2cset(tree, &info);
|
||||
addcharset(compst, i, &info);
|
||||
return 1;
|
||||
}
|
||||
default: return 0; /* not a charset */
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
** Repetion; optimizations:
|
||||
** When pattern is a charset, can use special instruction ISpan.
|
||||
** When pattern is a charset, use special code.
|
||||
** When pattern is head fail, or if it starts with characters that
|
||||
** are disjoint from what follows the repetions, a simple test
|
||||
** is enough (a fail inside the repetition would backtrack to fail
|
||||
@@ -776,21 +816,14 @@ static void coderuntime (CompileState *compst, TTree *tree, int tt) {
|
||||
*/
|
||||
static void coderep (CompileState *compst, TTree *tree, int opt,
|
||||
const Charset *fl) {
|
||||
Charset st;
|
||||
if (tocharset(tree, &st)) {
|
||||
addinstruction(compst, ISpan, 0);
|
||||
addcharset(compst, st.cs);
|
||||
}
|
||||
else {
|
||||
if (!coderepcharset(compst, tree)) {
|
||||
Charset st;
|
||||
int e1 = getfirst(tree, fullset, &st);
|
||||
if (headfail(tree) || (!e1 && cs_disjoint(&st, fl))) {
|
||||
/* L1: test (fail(p1)) -> L2; <p>; jmp L1; L2: */
|
||||
int jmp;
|
||||
int test = codetestset(compst, &st, 0);
|
||||
codegen(compst, tree, 0, test, fullset);
|
||||
jmp = addoffsetinst(compst, IJmp);
|
||||
jumptohere(compst, test);
|
||||
jumptothere(compst, jmp, test);
|
||||
closeloop(compst, test);
|
||||
}
|
||||
else {
|
||||
/* test(fail(p1)) -> L2; choice L2; L1: <p>; partialcommit L1; L2: */
|
||||
@@ -847,7 +880,7 @@ static void correctcalls (CompileState *compst, int *positions,
|
||||
Instruction *code = compst->p->code;
|
||||
for (i = from; i < to; i += sizei(&code[i])) {
|
||||
if (code[i].i.code == IOpenCall) {
|
||||
int n = code[i].i.key; /* rule number */
|
||||
int n = code[i].i.aux2.key; /* rule number */
|
||||
int rule = positions[n]; /* rule position */
|
||||
assert(rule == from || code[rule - 1].i.code == IRet);
|
||||
if (code[finaltarget(code, i + 2)].i.code == IRet) /* call; ret ? */
|
||||
@@ -874,8 +907,10 @@ static void codegrammar (CompileState *compst, TTree *grammar) {
|
||||
int start = gethere(compst); /* here starts the initial rule */
|
||||
jumptohere(compst, firstcall);
|
||||
for (rule = sib1(grammar); rule->tag == TRule; rule = sib2(rule)) {
|
||||
TTree *r = sib1(rule);
|
||||
assert(r->tag == TXInfo);
|
||||
positions[rulenumber++] = gethere(compst); /* save rule position */
|
||||
codegen(compst, sib1(rule), 0, NOINST, fullset); /* code rule */
|
||||
codegen(compst, sib1(r), 0, NOINST, fullset); /* code rule */
|
||||
addinstruction(compst, IRet, 0);
|
||||
}
|
||||
assert(rule->tag == TTrue);
|
||||
@@ -886,8 +921,8 @@ static void codegrammar (CompileState *compst, TTree *grammar) {
|
||||
|
||||
static void codecall (CompileState *compst, TTree *call) {
|
||||
int c = addoffsetinst(compst, IOpenCall); /* to be corrected later */
|
||||
getinstr(compst, c).i.key = sib2(call)->cap; /* rule number */
|
||||
assert(sib2(call)->tag == TRule);
|
||||
assert(sib1(sib2(call))->tag == TXInfo);
|
||||
getinstr(compst, c).i.aux2.key = sib1(sib2(call))->u.n; /* rule number */
|
||||
}
|
||||
|
||||
|
||||
@@ -922,9 +957,10 @@ static void codegen (CompileState *compst, TTree *tree, int opt, int tt,
|
||||
switch (tree->tag) {
|
||||
case TChar: codechar(compst, tree->u.n, tt); break;
|
||||
case TAny: addinstruction(compst, IAny, 0); break;
|
||||
case TSet: codecharset(compst, treebuffer(tree), tt); break;
|
||||
case TSet: codecharset(compst, tree, tt); break;
|
||||
case TTrue: break;
|
||||
case TFalse: addinstruction(compst, IFail, 0); break;
|
||||
case TUTFR: codeutfr(compst, tree); break;
|
||||
case TChoice: codechoice(compst, sib1(tree), sib2(tree), opt, fl); break;
|
||||
case TRep: coderep(compst, sib1(tree), opt, fl); break;
|
||||
case TBehind: codebehind(compst, tree); break;
|
||||
@@ -971,7 +1007,7 @@ static void peephole (CompileState *compst) {
|
||||
case IRet: case IFail: case IFailTwice:
|
||||
case IEnd: { /* instructions with unconditional implicit jumps */
|
||||
code[i] = code[ft]; /* jump becomes that instruction */
|
||||
code[i + 1].i.code = IAny; /* 'no-op' for target position */
|
||||
code[i + 1].i.code = IEmpty; /* 'no-op' for target position */
|
||||
break;
|
||||
}
|
||||
case ICommit: case IPartialCommit:
|
||||
@@ -996,12 +1032,13 @@ static void peephole (CompileState *compst) {
|
||||
|
||||
|
||||
/*
|
||||
** Compile a pattern
|
||||
** Compile a pattern. 'size' is the size of the pattern's tree,
|
||||
** which gives a hint for the size of the final code.
|
||||
*/
|
||||
Instruction *compile (lua_State *L, Pattern *p) {
|
||||
Instruction *compile (lua_State *L, Pattern *p, uint size) {
|
||||
CompileState compst;
|
||||
compst.p = p; compst.ncode = 0; compst.L = L;
|
||||
realloccode(L, p, 2); /* minimum initial size */
|
||||
newcode(L, p, size/2u + 2); /* set initial size */
|
||||
codegen(&compst, p->tree, 0, NOINST, fullset);
|
||||
addinstruction(&compst, IEnd, 0);
|
||||
realloccode(L, p, compst.ncode); /* set final size */
|
||||
|
||||
Reference in New Issue
Block a user