properly tonekize regexp literals in JavaScript.

fix #140
This commit is contained in:
Fabian Jakobs 2011-07-26 16:00:40 +02:00
commit 4012c32a55
4 changed files with 76 additions and 27 deletions

View file

@ -115,7 +115,8 @@ var Editor =function(renderer, session) {
}; };
this.setSession = function(session) { this.setSession = function(session) {
if (this.session == session) return; if (this.session == session)
return;
if (this.session) { if (this.session) {
var oldSession = this.session; var oldSession = this.session;

View file

@ -82,9 +82,6 @@ var JavaScriptHighlightRules = function() {
token : "comment", // multi line comment token : "comment", // multi line comment
regex : "\\/\\*", regex : "\\/\\*",
next : "comment" next : "comment"
}, {
token : "string.regexp",
regex : "[/](?:(?:\\[(?:\\\\]|[^\\]])+\\])|(?:\\\\/|[^\\]/]))*[/]\\w*\\s*(?=[).,;]|$)"
}, { }, {
token : "string", // single line token : "string", // single line
regex : '["](?:(?:\\\\.)|(?:[^"\\\\]))*?["]' regex : '["](?:(?:\\\\.)|(?:[^"\\\\]))*?["]'
@ -126,13 +123,19 @@ var JavaScriptHighlightRules = function() {
regex : identifierRe regex : identifierRe
}, { }, {
token : "keyword.operator", token : "keyword.operator",
regex : "!|\\$|%|&|\\*|\\-\\-|\\-|\\+\\+|\\+|~|===|==|=|!=|!==|<=|>=|<<=|>>=|>>>=|<>|<|>|!|&&|\\|\\||\\?\\:|\\*=|%=|\\+=|\\-=|&=|\\^=|\\b(?:in|instanceof|new|delete|typeof|void)" regex : "!|\\$|%|&|\\*|\\-\\-|\\-|\\+\\+|\\+|~|===|==|=|!=|!==|<=|>=|<<=|>>=|>>>=|<>|<|>|!|&&|\\|\\||\\?\\:|\\*=|%=|\\+=|\\-=|&=|\\^=|\\b(?:in|instanceof|new|delete|typeof|void)",
next : "regex_allowed"
}, { }, {
token : "lparen", token : "lparen",
regex : "[[({]" regex : "[[({]",
next : "regex_allowed"
}, { }, {
token : "rparen", token : "rparen",
regex : "[\\])}]" regex : "[\\])}]"
}, {
token : "keyword.operator",
regex : "\\/=?",
next : "regex_allowed"
}, { }, {
token: "comment", token: "comment",
regex: "^#!.*$" regex: "^#!.*$"
@ -141,6 +144,26 @@ var JavaScriptHighlightRules = function() {
regex : "\\s+" regex : "\\s+"
} }
], ],
// regular expressions are only allowed after certain tokens. This
// makes sure we don't mix up regexps with the divison operator
"regex_allowed": [
{
token: "string.regexp",
regex: "\\/(?:(?:\\[(?:\\\\]|[^\\]])+\\])"
+ "|(?:\\\\/|[^\\]/]))*"
+ "[/]\\w*",
next: "start"
}, {
token : "text",
regex : "\\s+"
}, {
// immediately return to the start mode without mathcing
// anything
token: "empty",
regex: "",
next: "start"
}
],
"comment" : [ "comment" : [
{ {
token : "comment", // closing comment token : "comment", // closing comment

View file

@ -104,11 +104,34 @@ module.exports = {
assert.equal("rparen", tokens[0].type); assert.equal("rparen", tokens[0].type);
}, },
"test tokenize regular expressions": function() { "test tokenize arithmetic expression which looks like a regexp": function() {
var tokens = this.tokenizer.getLineTokens("a/b/c", "start").tokens; var tokens = this.tokenizer.getLineTokens("a/b/c", "start").tokens;
assert.equal(5, tokens.length); assert.equal(5, tokens.length);
var tokens = this.tokenizer.getLineTokens("a/=b/c", "start").tokens;
assert.equal(5, tokens.length);
}, },
"test tokenize reg exps" : function() {
var tokens = this.tokenizer.getLineTokens("a=/b/g", "start").tokens;
assert.equal(3, tokens.length);
assert.equal("string.regexp", tokens[2].type);
var tokens = this.tokenizer.getLineTokens("a+/b/g", "start").tokens;
assert.equal(3, tokens.length);
assert.equal("string.regexp", tokens[2].type);
var tokens = this.tokenizer.getLineTokens("a = 1 + /2 + 1/b", "start").tokens;
assert.equal(9, tokens.length);
assert.equal("string.regexp", tokens[8].type);
var tokens = this.tokenizer.getLineTokens("a=/a/ / /a/", "start").tokens;
assert.equal(7, tokens.length);
assert.equal("string.regexp", tokens[2].type);
assert.equal("string.regexp", tokens[6].type);
},
"test tokenize identifier with umlauts": function() { "test tokenize identifier with umlauts": function() {
var tokens = this.tokenizer.getLineTokens("füße", "start").tokens; var tokens = this.tokenizer.getLineTokens("füße", "start").tokens;
assert.equal(1, tokens.length); assert.equal(1, tokens.length);

View file

@ -103,13 +103,15 @@ var Tokenizer = function(rules) {
value = match.slice(i+2, i+1+mapping[i].len); value = match.slice(i+2, i+1+mapping[i].len);
} }
// compute token type
if (typeof rule.token == "function") if (typeof rule.token == "function")
type = rule.token.apply(this, value); type = rule.token.apply(this, value);
else else
type = rule.token; type = rule.token;
if (rule.next && rule.next !== currentState) { var next = rule.next;
currentState = rule.next; if (next && next !== currentState) {
currentState = next;
state = this.rules[currentState]; state = this.rules[currentState];
mapping = this.matchMappings[currentState]; mapping = this.matchMappings[currentState];
lastIndex = re.lastIndex; lastIndex = re.lastIndex;
@ -121,25 +123,25 @@ var Tokenizer = function(rules) {
} }
}; };
if (typeof type == "string") { if (value[0]) {
if (typeof value != "string") { if (typeof type == "string") {
value = [value.join("")]; value = [value.join("")];
type = [type];
} }
type = [type];
}
for ( var i = 0; i < value.length; i++) { for (var i = 0; i < value.length; i++) {
if (token.type !== type[i]) { if (token.type !== type[i]) {
if (token.type) { if (token.type) {
tokens.push(token); tokens.push(token);
} }
token = { token = {
type: type[i], type: type[i],
value: value[i] value: value[i]
}
} else {
token.value += value[i];
} }
} else {
token.value += value[i];
} }
} }