properly tonekize regexp literals in JavaScript.

fix #140
This commit is contained in:
Fabian Jakobs 2011-07-26 16:00:40 +02:00
commit 4012c32a55
4 changed files with 76 additions and 27 deletions

View file

@ -115,7 +115,8 @@ var Editor =function(renderer, session) {
};
this.setSession = function(session) {
if (this.session == session) return;
if (this.session == session)
return;
if (this.session) {
var oldSession = this.session;

View file

@ -82,9 +82,6 @@ var JavaScriptHighlightRules = function() {
token : "comment", // multi line comment
regex : "\\/\\*",
next : "comment"
}, {
token : "string.regexp",
regex : "[/](?:(?:\\[(?:\\\\]|[^\\]])+\\])|(?:\\\\/|[^\\]/]))*[/]\\w*\\s*(?=[).,;]|$)"
}, {
token : "string", // single line
regex : '["](?:(?:\\\\.)|(?:[^"\\\\]))*?["]'
@ -126,13 +123,19 @@ var JavaScriptHighlightRules = function() {
regex : identifierRe
}, {
token : "keyword.operator",
regex : "!|\\$|%|&|\\*|\\-\\-|\\-|\\+\\+|\\+|~|===|==|=|!=|!==|<=|>=|<<=|>>=|>>>=|<>|<|>|!|&&|\\|\\||\\?\\:|\\*=|%=|\\+=|\\-=|&=|\\^=|\\b(?:in|instanceof|new|delete|typeof|void)"
regex : "!|\\$|%|&|\\*|\\-\\-|\\-|\\+\\+|\\+|~|===|==|=|!=|!==|<=|>=|<<=|>>=|>>>=|<>|<|>|!|&&|\\|\\||\\?\\:|\\*=|%=|\\+=|\\-=|&=|\\^=|\\b(?:in|instanceof|new|delete|typeof|void)",
next : "regex_allowed"
}, {
token : "lparen",
regex : "[[({]"
regex : "[[({]",
next : "regex_allowed"
}, {
token : "rparen",
regex : "[\\])}]"
}, {
token : "keyword.operator",
regex : "\\/=?",
next : "regex_allowed"
}, {
token: "comment",
regex: "^#!.*$"
@ -141,6 +144,26 @@ var JavaScriptHighlightRules = function() {
regex : "\\s+"
}
],
// regular expressions are only allowed after certain tokens. This
// makes sure we don't mix up regexps with the divison operator
"regex_allowed": [
{
token: "string.regexp",
regex: "\\/(?:(?:\\[(?:\\\\]|[^\\]])+\\])"
+ "|(?:\\\\/|[^\\]/]))*"
+ "[/]\\w*",
next: "start"
}, {
token : "text",
regex : "\\s+"
}, {
// immediately return to the start mode without mathcing
// anything
token: "empty",
regex: "",
next: "start"
}
],
"comment" : [
{
token : "comment", // closing comment

View file

@ -96,19 +96,42 @@ module.exports = {
assert.equal("text", tokens[1].type);
assert.equal("rparen", tokens[2].type);
},
"test for last rule in ruleset to catch capturing group bugs" : function() {
var tokens = this.tokenizer.getLineTokens("}", "start").tokens;
assert.equal(1, tokens.length);
assert.equal("rparen", tokens[0].type);
},
"test tokenize regular expressions": function() {
"test tokenize arithmetic expression which looks like a regexp": function() {
var tokens = this.tokenizer.getLineTokens("a/b/c", "start").tokens;
assert.equal(5, tokens.length);
var tokens = this.tokenizer.getLineTokens("a/=b/c", "start").tokens;
assert.equal(5, tokens.length);
},
"test tokenize reg exps" : function() {
var tokens = this.tokenizer.getLineTokens("a=/b/g", "start").tokens;
assert.equal(3, tokens.length);
assert.equal("string.regexp", tokens[2].type);
var tokens = this.tokenizer.getLineTokens("a+/b/g", "start").tokens;
assert.equal(3, tokens.length);
assert.equal("string.regexp", tokens[2].type);
var tokens = this.tokenizer.getLineTokens("a = 1 + /2 + 1/b", "start").tokens;
assert.equal(9, tokens.length);
assert.equal("string.regexp", tokens[8].type);
var tokens = this.tokenizer.getLineTokens("a=/a/ / /a/", "start").tokens;
assert.equal(7, tokens.length);
assert.equal("string.regexp", tokens[2].type);
assert.equal("string.regexp", tokens[6].type);
},
"test tokenize identifier with umlauts": function() {
var tokens = this.tokenizer.getLineTokens("füße", "start").tokens;
assert.equal(1, tokens.length);

View file

@ -103,13 +103,15 @@ var Tokenizer = function(rules) {
value = match.slice(i+2, i+1+mapping[i].len);
}
// compute token type
if (typeof rule.token == "function")
type = rule.token.apply(this, value);
else
type = rule.token;
if (rule.next && rule.next !== currentState) {
currentState = rule.next;
var next = rule.next;
if (next && next !== currentState) {
currentState = next;
state = this.rules[currentState];
mapping = this.matchMappings[currentState];
lastIndex = re.lastIndex;
@ -120,26 +122,26 @@ var Tokenizer = function(rules) {
break;
}
};
if (typeof type == "string") {
if (typeof value != "string") {
if (value[0]) {
if (typeof type == "string") {
value = [value.join("")];
type = [type];
}
type = [type];
}
for ( var i = 0; i < value.length; i++) {
if (token.type !== type[i]) {
if (token.type) {
tokens.push(token);
}
for (var i = 0; i < value.length; i++) {
if (token.type !== type[i]) {
if (token.type) {
tokens.push(token);
}
token = {
type: type[i],
value: value[i]
token = {
type: type[i],
value: value[i]
}
} else {
token.value += value[i];
}
} else {
token.value += value[i];
}
}