minimize number of capturing groups
This commit is contained in:
parent
cfa3171ac4
commit
903b7a1951
1 changed files with 73 additions and 43 deletions
|
|
@ -62,20 +62,41 @@ var Tokenizer = function(rules, flag) {
|
||||||
var mapping = this.matchMappings[key] = {};
|
var mapping = this.matchMappings[key] = {};
|
||||||
|
|
||||||
for (var i = 0; i < state.length; i++) {
|
for (var i = 0; i < state.length; i++) {
|
||||||
if (state[i].regex instanceof RegExp)
|
var rule = state[i];
|
||||||
state[i].regex = state[i].regex.toString().slice(1, -1);
|
if (rule.regex instanceof RegExp)
|
||||||
|
rule.regex = rule.regex.toString().slice(1, -1);
|
||||||
|
|
||||||
// Count number of matching groups. 2 extra groups from the full match
|
// Count number of matching groups. 2 extra groups from the full match
|
||||||
// And the catch-all on the end (used to force a match);
|
// And the catch-all on the end (used to force a match);
|
||||||
var matchcount = new RegExp("(?:(" + state[i].regex + ")|(.))").exec("a").length - 2;
|
var adjustedregex = rule.regex
|
||||||
|
var matchcount = new RegExp("(?:(" + adjustedregex + ")|(.))").exec("a").length - 2;
|
||||||
|
if (Array.isArray(rule.token)) {
|
||||||
|
if (rule.token.length != matchcount - 1) {
|
||||||
|
throw new Error(
|
||||||
|
"For " + rule.regex +
|
||||||
|
" the matching groups (" +(matchcount-1) +
|
||||||
|
") and length of the token array (" + rule.token.length +
|
||||||
|
") don't match (rule #" + i + " of state " + key + ")"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
rule.tokenArray = rule.token;
|
||||||
|
rule.token = this.$splitHelper;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (matchcount > 1) {
|
||||||
|
if (/\\\d/.test(rule.regex)) {
|
||||||
// Replace any backreferences and offset appropriately.
|
// Replace any backreferences and offset appropriately.
|
||||||
var adjustedregex = state[i].regex.replace(/\\([0-9]+)/g, function (match, digit) {
|
adjustedregex = rule.regex.replace(/\\([0-9]+)/g, function (match, digit) {
|
||||||
return "\\" + (parseInt(digit, 10) + matchTotal + 1);
|
return "\\" + (parseInt(digit, 10) + matchTotal + 1);
|
||||||
});
|
});
|
||||||
|
} else {
|
||||||
if (matchcount > 1 && typeof state[i].token == "string" && state[i].token.length !== matchcount-1)
|
matchcount = 1;
|
||||||
throw new Error("For " + state[i].regex + " the matching groups (" +(matchcount-1) + ") and length of the token array (" + state[i].token.length + ") don't match (rule #" + i + " of state " + key + ")");
|
adjustedregex = this.nonCapturingRegexp(rule.regex);
|
||||||
|
}
|
||||||
|
if (!rule.splitRegex)
|
||||||
|
rule.splitRegex = new RegExp(rule.regex)
|
||||||
|
}
|
||||||
|
|
||||||
mapping[matchTotal] = {
|
mapping[matchTotal] = {
|
||||||
rule: i,
|
rule: i,
|
||||||
|
|
@ -85,12 +106,33 @@ var Tokenizer = function(rules, flag) {
|
||||||
|
|
||||||
ruleRegExps.push(adjustedregex);
|
ruleRegExps.push(adjustedregex);
|
||||||
}
|
}
|
||||||
|
console.log(key, ruleRegExps)
|
||||||
|
|
||||||
this.regExps[key] = new RegExp("(?:(" + ruleRegExps.join(")|(") + ")|(.))", flag);
|
this.regExps[key] = new RegExp("(?:(" + ruleRegExps.join(")|(") + ")|(.))", flag);
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
(function() {
|
(function() {
|
||||||
|
this.$splitHelper = function(str) {
|
||||||
|
var values = str.split(this.splitRegex)
|
||||||
|
var tokens = [];
|
||||||
|
var types = this.tokenArray;
|
||||||
|
for (var i = 0; i < types.length; i++) {
|
||||||
|
if (values[i + 1]) {
|
||||||
|
tokens[tokens.length] = {
|
||||||
|
type: types[i],
|
||||||
|
value: values[i + 1]
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return tokens;
|
||||||
|
}
|
||||||
|
this.nonCapturingRegexp = function(src) {
|
||||||
|
return src.replace(
|
||||||
|
/\[(?:\\.|[^\]])*?\]|\\.|\((?:\?:|\?=|!=)|(\()/g,
|
||||||
|
function(x, y) {return y ? "(?:" : x;}
|
||||||
|
);
|
||||||
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Returns an object containing two properties: `tokens`, which contains all the tokens; and `state`, the current state.
|
* Returns an object containing two properties: `tokens`, which contains all the tokens; and `state`, the current state.
|
||||||
|
|
@ -100,9 +142,9 @@ var Tokenizer = function(rules, flag) {
|
||||||
if (startState && typeof startState != "string") {
|
if (startState && typeof startState != "string") {
|
||||||
var stack = startState.slice(0);
|
var stack = startState.slice(0);
|
||||||
startState = stack[0];
|
startState = stack[0];
|
||||||
} else {
|
} else
|
||||||
var stack = []
|
var stack = []
|
||||||
}
|
|
||||||
var currentState = startState || "start";
|
var currentState = startState || "start";
|
||||||
var state = this.rules[currentState];
|
var state = this.rules[currentState];
|
||||||
var mapping = this.matchMappings[currentState];
|
var mapping = this.matchMappings[currentState];
|
||||||
|
|
@ -110,18 +152,14 @@ var Tokenizer = function(rules, flag) {
|
||||||
re.lastIndex = 0;
|
re.lastIndex = 0;
|
||||||
|
|
||||||
var match, tokens = [];
|
var match, tokens = [];
|
||||||
|
|
||||||
var lastIndex = 0;
|
var lastIndex = 0;
|
||||||
|
|
||||||
var token = {
|
var token = {type: null, value: ""};
|
||||||
type: null,
|
|
||||||
value: ""
|
|
||||||
};
|
|
||||||
|
|
||||||
while (match = re.exec(line)) {
|
while (match = re.exec(line)) {
|
||||||
var type = "text";
|
var type = "text";
|
||||||
var rule = null;
|
var rule = null;
|
||||||
var value = [match[0]];
|
var value = match[0];
|
||||||
|
|
||||||
for (var i = 0; i < match.length-2; i++) {
|
for (var i = 0; i < match.length-2; i++) {
|
||||||
if (match[i + 1] === undefined)
|
if (match[i + 1] === undefined)
|
||||||
|
|
@ -129,12 +167,9 @@ var Tokenizer = function(rules, flag) {
|
||||||
|
|
||||||
rule = state[mapping[i].rule];
|
rule = state[mapping[i].rule];
|
||||||
|
|
||||||
if (mapping[i].len > 1)
|
|
||||||
value = match.slice(i+2, i+1+mapping[i].len);
|
|
||||||
|
|
||||||
// compute token type
|
// compute token type
|
||||||
if (typeof rule.token == "function")
|
if (typeof rule.token == "function")
|
||||||
type = rule.token(value.length == 1 ? value[0] : value, currentState, stack);
|
type = rule.token(value, currentState, stack);
|
||||||
else
|
else
|
||||||
type = rule.token;
|
type = rule.token;
|
||||||
|
|
||||||
|
|
@ -154,26 +189,21 @@ var Tokenizer = function(rules, flag) {
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (value[0]) {
|
if (value) {
|
||||||
if (typeof type == "string") {
|
if (typeof type == "string") {
|
||||||
value = [value.join("")];
|
if ((!rule || rule.merge !== false) && token.type === type) {
|
||||||
type = [type];
|
token.value += value;
|
||||||
}
|
|
||||||
for (var i = 0; i < value.length; i++) {
|
|
||||||
if (!value[i])
|
|
||||||
continue;
|
|
||||||
|
|
||||||
if ((!rule || rule.merge !== false) && token.type === type[i]) {
|
|
||||||
token.value += value[i];
|
|
||||||
} else {
|
} else {
|
||||||
if (token.type)
|
if (token.type)
|
||||||
tokens.push(token);
|
tokens.push(token);
|
||||||
|
token = {type: type, value: value};
|
||||||
token = {
|
|
||||||
type: type[i],
|
|
||||||
value: value[i]
|
|
||||||
};
|
|
||||||
}
|
}
|
||||||
|
} else {
|
||||||
|
if (token.type)
|
||||||
|
tokens.push(token);
|
||||||
|
token = {type: null, value: ""};
|
||||||
|
for (var i = 0; i < type.length; i++)
|
||||||
|
tokens.push(type[i]);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue