add {defaultToken: "..."} rules for skipped text
This commit is contained in:
parent
0e7a7b9ebe
commit
ea857f23c2
1 changed files with 52 additions and 44 deletions
|
|
@ -50,19 +50,22 @@ define(function(require, exports, module) {
|
|||
**/
|
||||
var Tokenizer = function(rules, flag) {
|
||||
flag = flag ? "g" + flag : "g";
|
||||
this.rules = rules;
|
||||
this.states = rules;
|
||||
|
||||
this.regExps = {};
|
||||
this.matchMappings = {};
|
||||
for (var key in this.rules) {
|
||||
var rule = this.rules[key];
|
||||
var state = rule;
|
||||
for (var key in this.states) {
|
||||
var state = this.states[key];
|
||||
var ruleRegExps = [];
|
||||
var matchTotal = 0;
|
||||
var mapping = this.matchMappings[key] = {};
|
||||
var mapping = this.matchMappings[key] = {defaultToken: "text"};
|
||||
|
||||
for (var i = 0; i < state.length; i++) {
|
||||
var rule = state[i];
|
||||
if (rule.defaultToken) {
|
||||
mapping.defaultToken = rule.defaultToken;
|
||||
continue;
|
||||
}
|
||||
if (rule.regex instanceof RegExp)
|
||||
rule.regex = rule.regex.toString().slice(1, -1);
|
||||
|
||||
|
|
@ -71,17 +74,12 @@ var Tokenizer = function(rules, flag) {
|
|||
var adjustedregex = rule.regex
|
||||
var matchcount = new RegExp("(?:(" + adjustedregex + ")|(.))").exec("a").length - 2;
|
||||
if (Array.isArray(rule.token)) {
|
||||
if (rule.token.length != matchcount - 1) {
|
||||
throw new Error(
|
||||
"For " + rule.regex +
|
||||
" the matching groups (" +(matchcount-1) +
|
||||
") and length of the token array (" + rule.token.length +
|
||||
") don't match (rule #" + i + " of state " + key + ")"
|
||||
);
|
||||
if (rule.token.length == 1) {
|
||||
rule.token = rule.token[0];
|
||||
} else {
|
||||
rule.tokenArray = rule.token;
|
||||
rule.token = this.$arrayTokens;
|
||||
}
|
||||
|
||||
rule.tokenArray = rule.token;
|
||||
rule.token = this.$splitHelper;
|
||||
}
|
||||
|
||||
if (matchcount > 1) {
|
||||
|
|
@ -89,19 +87,16 @@ var Tokenizer = function(rules, flag) {
|
|||
// Replace any backreferences and offset appropriately.
|
||||
adjustedregex = rule.regex.replace(/\\([0-9]+)/g, function (match, digit) {
|
||||
return "\\" + (parseInt(digit, 10) + matchTotal + 1);
|
||||
});
|
||||
});
|
||||
} else {
|
||||
matchcount = 1;
|
||||
adjustedregex = this.nonCapturingRegexp(rule.regex);
|
||||
adjustedregex = this.removeCapturingGroups(rule.regex);
|
||||
}
|
||||
if (!rule.splitRegex)
|
||||
rule.splitRegex = new RegExp(rule.regex)
|
||||
rule.splitRegex = this.createSplitterRegexp(rule.regex, flag);
|
||||
}
|
||||
|
||||
mapping[matchTotal] = {
|
||||
rule: i,
|
||||
len: matchcount
|
||||
};
|
||||
|
||||
mapping[matchTotal] = i;
|
||||
matchTotal += matchcount;
|
||||
|
||||
ruleRegExps.push(adjustedregex);
|
||||
|
|
@ -112,10 +107,14 @@ var Tokenizer = function(rules, flag) {
|
|||
};
|
||||
|
||||
(function() {
|
||||
this.$splitHelper = function(str) {
|
||||
this.$arrayTokens = function(str) {
|
||||
var values = str.split(this.splitRegex)
|
||||
var tokens = [];
|
||||
var types = this.tokenArray;
|
||||
if (types.length != values.length - 2) {
|
||||
console.log(types.length , values.length - 2, str, this.splitRegex)
|
||||
return [{type: "error.invalid", value: str}];
|
||||
}
|
||||
for (var i = 0; i < types.length; i++) {
|
||||
if (values[i + 1]) {
|
||||
tokens[tokens.length] = {
|
||||
|
|
@ -125,12 +124,19 @@ var Tokenizer = function(rules, flag) {
|
|||
}
|
||||
}
|
||||
return tokens;
|
||||
}
|
||||
this.nonCapturingRegexp = function(src) {
|
||||
return src.replace(
|
||||
/\[(?:\\.|[^\]])*?\]|\\.|\((?:\?:|\?=|!=)|(\()/g,
|
||||
};
|
||||
|
||||
this.removeCapturingGroups = function(src) {
|
||||
var r = src.replace(
|
||||
/\[(?:\\.|[^\]])*?\]|\\.|\(\?[:=!]|(\()/g,
|
||||
function(x, y) {return y ? "(?:" : x;}
|
||||
);
|
||||
return r;
|
||||
};
|
||||
|
||||
this.createSplitterRegexp = function(src, flag) {
|
||||
src = src.replace(/\(\?=([^()]|\\.)*?\)$/, "");
|
||||
return new RegExp(src, flag);
|
||||
};
|
||||
|
||||
/**
|
||||
|
|
@ -145,7 +151,7 @@ var Tokenizer = function(rules, flag) {
|
|||
var stack = [];
|
||||
|
||||
var currentState = startState || "start";
|
||||
var state = this.rules[currentState];
|
||||
var state = this.states[currentState];
|
||||
var mapping = this.matchMappings[currentState];
|
||||
var re = this.regExps[currentState];
|
||||
re.lastIndex = 0;
|
||||
|
|
@ -156,38 +162,40 @@ var Tokenizer = function(rules, flag) {
|
|||
var token = {type: null, value: ""};
|
||||
|
||||
while (match = re.exec(line)) {
|
||||
var type = "text";
|
||||
var type = mapping.defaultToken;
|
||||
var rule = null;
|
||||
var value = match[0];
|
||||
var index = re.lastIndex;
|
||||
|
||||
|
||||
if (index - value.length > lastIndex) {
|
||||
if (token.type)
|
||||
tokens.push(token);
|
||||
token = {
|
||||
type: "regex",
|
||||
value: line.substring(lastIndex, index - value.length)
|
||||
};
|
||||
var skipped = line.substring(lastIndex, index - value.length);
|
||||
if (token.type == type) {
|
||||
token.value += skipped;
|
||||
} else {
|
||||
if (token.type)
|
||||
tokens.push(token);
|
||||
token = {type: type, value: skipped};
|
||||
}
|
||||
}
|
||||
|
||||
for (var i = 0; i < match.length-2; i++) {
|
||||
if (match[i + 1] === undefined)
|
||||
continue;
|
||||
|
||||
rule = state[mapping[i].rule];
|
||||
rule = state[mapping[i]];
|
||||
|
||||
// compute token type
|
||||
if (typeof rule.token == "function")
|
||||
type = rule.token(value, currentState, stack);
|
||||
else
|
||||
type = rule.token;
|
||||
type = typeof rule.token == "function"
|
||||
? rule.token(value, currentState, stack)
|
||||
: rule.token;
|
||||
|
||||
if (rule.next) {
|
||||
currentState = rule.next;
|
||||
state = this.rules[currentState];
|
||||
state = this.states[currentState];
|
||||
if (!state) {
|
||||
window.console && console.error && console.error(currentState, "doesn't exist");
|
||||
currentState = "start";
|
||||
state = this.rules[currentState];
|
||||
state = this.states[currentState];
|
||||
}
|
||||
mapping = this.matchMappings[currentState];
|
||||
lastIndex = index;
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue