diff --git a/hscript/Error.hx b/hscript/Error.hx index 5fafa8a..a022650 100644 --- a/hscript/Error.hx +++ b/hscript/Error.hx @@ -48,6 +48,7 @@ enum Error EUnexpected( s : String ); EUnterminatedString; EUnterminatedComment; + EUnterminatedRegex; EInvalidPreprocessor( msg : String ); EUnknownVariable( v : String ); EInvalidIterator( v : String ); @@ -79,16 +80,4 @@ enum abstract ErrorMessage(Int) from Int to Int { case EXPECT_KEY_VALUE_SYNTAX: "Expected a => b"; } } - - public static function fromString(s:String):ErrorMessage { - return switch(s) { - case "INVALID_CHAR_CODE_MULTI": INVALID_CHAR_CODE_MULTI; - case "FROM_CHAR_CODE_NON_INT": FROM_CHAR_CODE_NON_INT; - case "EMPTY_INTERPOLATION": EMPTY_INTERPOLATION; - case "UNKNOWN_MAP_TYPE": UNKNOWN_MAP_TYPE; - case "UNKNOWN_MAP_TYPE_RUNTIME": UNKNOWN_MAP_TYPE_RUNTIME; - case "EXPECT_KEY_VALUE_SYNTAX": EXPECT_KEY_VALUE_SYNTAX; - default: throw "Unknown ErrorMessage"; - } - } } \ No newline at end of file diff --git a/hscript/Parser.hx b/hscript/Parser.hx index 17b57c6..1e9a10f 100644 --- a/hscript/Parser.hx +++ b/hscript/Parser.hx @@ -46,6 +46,7 @@ enum Token { TEof; TConst( c : Const ); TInterpString( tkl: Array ); // Stores the tokens that make up the interpolation string + TRegex( r: String, opt: String ); TId( s : String ); TOp( s : String ); TPOpen; @@ -103,7 +104,7 @@ class Parser { /** resume from parsing errors (when parsing incomplete code, during completion for example) **/ - public var resumeErrors : Bool; + public var resumeErrors : Bool = false; // implementation var input : String; @@ -479,6 +480,9 @@ class Parser { } } + return parseExprNext(e); + case TRegex(r, opt): + var e = mk(EParent(mk(ENew("EReg", [mk(EConst(CString(r))), mk(EConst(CString(opt)))]), p1)), p1); return parseExprNext(e); case TPOpen: tk = token(); @@ -2315,6 +2319,27 @@ class Parser { } invalidChar(char); default: + if(char == '~'.code) { + var char = readChar(); + if( char == "/".code ) { + var regex = getRegexBody(); + var opt = ""; + while( true ) { + char = readChar(); + if( char != 'g'.code && char != 'i'.code && char != 'm'.code && char != 's'.code && char != 'u'.code ) { + if( char >= 'a'.code && char <= 'z'.code ) { + error(ECustom('Invalid regex expression option "' + String.fromCharCode(char) + '"'), readPos, readPos); + } + this.char = char; + return TRegex(regex, opt); + } + opt += String.fromCharCode(char); + } + } + readPos--; + //this.char = peekChar(); + } + if( ops[char] ) { var op = String.fromCharCode(char); while( true ) { @@ -2354,6 +2379,57 @@ class Parser { return null; } + function getRegexBody() { + var regex = new StringBuf(); + var esc = false; + var start = readPos-2; + while( true ) { + var char = readChar(); + if(StringTools.isEof(char)) + error(EUnterminatedRegex, start, start); + if( char == "\n".code || char == "\r".code ) + error(EUnterminatedRegex, start, readPos-1); + + if( esc ) { + esc = false; + switch( char ) { + case '/'.code: regex.addChar("/".code); + case 'n'.code: regex.addChar("\n".code); + case 'r'.code: regex.addChar("\r".code); + case 't'.code: regex.addChar("\t".code); + case '\\'.code, '$'.code, '.'.code, '*'.code, '+'.code, '^'.code, '|'.code, '{'.code, '}'.code, '['.code, ']'.code, '('.code, ')'.code, '?'.code, '-'.code: + regex.addChar(char); + case '0'.code | '1'.code | '2'.code | '3'.code | '4'.code | '5'.code | '6'.code | '7'.code | '8'.code | '9'.code: + regex.addChar("\\".code); + regex.addChar(char); + case 'w'.code, 'W'.code, 'b'.code, 'B'.code, 's'.code, 'S'.code, 'd'.code, 'D'.code, 'x'.code: + regex.addChar("\\".code); + regex.addChar(char); + case 'u'.code, 'U'.code: // UNICODE + regex.addChar("\\".code); + for( i in 0...4 ) { + var c = readChar(); + if( StringTools.isEof(c) ) + error(EUnterminatedRegex, start, readPos-1); + var h = convertHex(c); + if( h == -1 ) + invalidChar(c); + regex.addChar(c); + } + + default: invalidChar(char); + } + } else if( char == "\\".code ) { + esc = true; + continue; + } else if( char == "/".code ) + break; + else + regex.addChar(char); + } + return regex.toString(); + } + function preprocValue( id : String ) : Dynamic { return preprocesorValues.get(id); } @@ -2531,6 +2607,7 @@ class Parser { case TEof: ""; case TConst(c): constString(c); case TInterpString(tg): constInterpString(tg); + case TRegex(r, opt): '~/' + r + '/' + opt; case TId(s): s; case TOp(s): s; case TPOpen: "("; diff --git a/hscript/Preprocessor.hx b/hscript/Preprocessor.hx index 7cb10e8..62f25e7 100644 --- a/hscript/Preprocessor.hx +++ b/hscript/Preprocessor.hx @@ -9,19 +9,54 @@ import hscript.Parser; class Preprocessor { static inline function expr(e:Expr) return Tools.expr(e); + private static var importStack:Array> = []; + private static function addImport(e:String, ?as:String) { + for(i in importStack) + if(i[0] == e) + return; + importStack.push([e, as]); + } + /** * Preprocesses the expression, like 'a'.code => 97 * Also for transforming any abstracts into their implementations (TODO) * Also for transforming any static extensions into their real form (TODO) + * Also to automatically add imports for stuff that is not imported **/ - public static function process(e:Expr, doBlock:Bool = true):Expr { + public static function process(e:Expr, top:Bool = true):Expr { + importStack = []; + var e = _process(e, top); + + // Automatically add imports for stuff + switch(expr(e)) { + case EBlock(exprs): + while(importStack.length > 0) { + var im = importStack.pop(); + exprs.unshift(mk(EImport(im[0], im[1]), e)); + } + return mk(EBlock(exprs), e); + default: + if(importStack.length > 0) { + var exprs = []; + while(importStack.length > 0) { + var im = importStack.pop(); + exprs.unshift(mk(EImport(im[0], im[1]), e)); + } + exprs.push(e); + return mk(EBlock(exprs), e); + } + } + return e; + } + + private static function _process(e:Expr, top:Bool = true):Expr { if(e == null) return null; //trace(expr(e)); e = Tools.map(e, function(e) { - return process(e); + return _process(e, false); }); switch(expr(e)) { @@ -42,6 +77,12 @@ class Preprocessor { #else throw Parser.getBaseError(EPreset(FROM_CHAR_CODE_NON_INT)); #end + case ENew("String", _): + addImport("String"); + case ENew("EReg", _): + addImport("EReg"); + case EField(expr(_) => EIdent("EReg"), "escape", _): + addImport("EReg"); default: } diff --git a/hscript/Printer.hx b/hscript/Printer.hx index 011f8ab..3b34613 100644 --- a/hscript/Printer.hx +++ b/hscript/Printer.hx @@ -534,6 +534,7 @@ class Printer { case EUnexpected(s): "Unexpected token: \""+s+"\""; case EUnterminatedString: "Unterminated string"; case EUnterminatedComment: "Unterminated comment"; + case EUnterminatedRegex: "Unterminated regular expression"; case EInvalidPreprocessor(str): "Invalid preprocessor (" + str + ")"; case EUnknownVariable(v): "Unknown variable: "+v; case EInvalidIterator(v): "Invalid iterator: "+v; diff --git a/tests/src/Main.hx b/tests/src/Main.hx index 3a7acd4..c665d31 100644 --- a/tests/src/Main.hx +++ b/tests/src/Main.hx @@ -22,6 +22,7 @@ class Main { runTest("BinOp", new BinOpCase()); runTest("Enum", new EnumCase()); runTest("EvalOrder", new EvalOrderCase()); + runTest("Error", new ErrorCase()); runTest("Float", new FloatCase()); runTest("IntIterator", new IntIteratorCase()); runTest("Lambda", new LambdaCase()); diff --git a/tests/src/tests/ErrorCase.hx b/tests/src/tests/ErrorCase.hx new file mode 100644 index 0000000..ba8f092 --- /dev/null +++ b/tests/src/tests/ErrorCase.hx @@ -0,0 +1,20 @@ +package tests; + +class ErrorCase extends TestCase { + override function setup() { + super.setup(); + } + + override function getNewInterp() { + var interp = super.getNewInterp(); + return interp; + } + + override function run() { + Sys.println("TODO: write tests for errors"); + } + + override function teardown() { + super.teardown(); + } +} \ No newline at end of file diff --git a/tests/src/tests/RegexCase.hx b/tests/src/tests/RegexCase.hx index 70ea851..4a690b0 100644 --- a/tests/src/tests/RegexCase.hx +++ b/tests/src/tests/RegexCase.hx @@ -19,24 +19,18 @@ class RegexCase extends TestCase { } override function run() { - var r = ~/a/; - var rg = ~/a/g; - var rg2 = ~/aa/g; - - Util.runKnownBug("Regex Syntax doesn't work", () -> { - assertEq("~/a/;", r); - assertEq("~/a/g;", rg); - assertEq("~/aa/g;", rg2); - }); + var r = new EReg("a", ""); + var rg = new EReg("a", "g"); + var rg2 = new EReg("aa", "g"); // TODO: implement better checking for regexes //assertEq("new EReg('a', '');", r); //assertEq("new EReg('a', 'g');", rg); //assertEq("new EReg('aa', 'g');", rg2); - @:privateAccess assertEq("new EReg('a', '').toString();", r.toString()); - @:privateAccess assertEq("new EReg('a', 'g').toString();", rg.toString()); - @:privateAccess assertEq("new EReg('aa', 'g').toString();", rg2.toString()); + //@:privateAccess assertEq("new EReg('a', '').toString();", r.toString()); + //@:privateAccess assertEq("new EReg('a', 'g').toString();", rg.toString()); + //@:privateAccess assertEq("new EReg('aa', 'g').toString();", rg2.toString()); //assertEq(r.match("") == false); //assertEq(r.match("b") == false); @@ -48,92 +42,76 @@ class RegexCase extends TestCase { //assertEq(pos.pos == 0); //assertEq(pos.len == 1); - /*execute(' - assertEq(r.match("") == false; - r.match("b") == false; - r.match("a") == true; - r.matched(0) == "a"; - r.matchedLeft() == ""; - r.matchedRight() == ""; - var pos = r.matchedPos(); - pos.pos == 0; - pos.len == 1; + headerCode = "var r = ~/a/;"; - r.match("aa") == true; - r.matched(0) == "a"; - r.matchedLeft() == ""; - r.matchedRight() == "a"; - var pos = r.matchedPos(); - pos.pos == 0; - pos.len == 1; + assertEq('[r.match(""), r.match("b"), r.match("a"), r.matched(0), r.matchedLeft(), r.matchedRight(), {var pos = r.matchedPos(); [pos.pos, pos.len];}]', [r.match(""), r.match("b"), r.match("a"), r.matched(0), r.matchedLeft(), r.matchedRight(), {var pos = r.matchedPos(); [pos.pos, pos.len];}]); + assertEq('[r.match("aa"), r.matched(0), r.match("a"), r.matchedLeft(), r.matchedRight(), {var pos = r.matchedPos(); [pos.pos, pos.len];}]', [r.match("aa"), r.matched(0), r.match("a"), r.matchedLeft(), r.matchedRight(), {var pos = r.matchedPos(); [pos.pos, pos.len];}]); - rg.match("aa") == true; - rg.matched(0) == "a"; - rg.matchedLeft() == ""; - rg.matchedRight() == "a"; - var pos = rg.matchedPos(); - pos.pos == 0; - pos.len == 1; - rg2.match("aa") == true; - rg2.matched(0) == "aa"; - rg2.matchedLeft() == ""; - rg2.matchedRight() == ""; - var pos = rg2.matchedPos(); - pos.pos == 0; - pos.len == 2; + headerCode = "var rg = ~/a/g;"; - rg2.match("AaaBaaC") == true; - rg2.matched(0) == "aa"; - rg2.matchedLeft() == "A"; - rg2.matchedRight() == "BaaC"; - var pos = rg2.matchedPos(); - pos.pos == 1; - pos.len == 2; + assertEq('[rg.match("aa"), rg.matched(0), rg.matchedLeft(), rg.matchedRight(), {var pos = rg.matchedPos(); [pos.pos, pos.len];}]', [rg.match("aa"), rg.matched(0), rg.matchedLeft(), rg.matchedRight(), {var pos = rg.matchedPos(); [pos.pos, pos.len];}]); + + headerCode = "var rg2 = ~/aa/g;"; + + assertEq('[rg2.match("aa"), rg2.matched(0), rg2.matchedLeft(), rg2.matchedRight(), {var pos = rg2.matchedPos(); [pos.pos, pos.len];}]', [rg2.match("aa"), rg2.matched(0), rg2.matchedLeft(), rg2.matchedRight(), {var pos = rg2.matchedPos(); [pos.pos, pos.len];}]); + assertEq('[rg2.match("AaaBaaC"), rg2.matched(0), rg2.matchedLeft(), rg2.matchedRight(), {var pos = rg2.matchedPos(); [pos.pos, pos.len];}]', [rg2.match("AaaBaaC"), rg2.matched(0), rg2.matchedLeft(), rg2.matchedRight(), {var pos = rg2.matchedPos(); [pos.pos, pos.len];}]); + + headerCode = ""; // split - ~/a/.split("") == [""]; - ~/a/.split("a") == ["",""]; - ~/a/.split("aa") == ["","a"]; - ~/a/.split("b") == ["b"]; - ~/a/.split("ab") == ["","b"]; - ~/a/.split("ba") == ["b",""]; - ~/a/.split("aba") == ["","ba"]; - ~/a/.split("bab") == ["b","b"]; - ~/a/.split("baba") == ["b","ba"]; + assertEq('~/a/.split("")', ~/a/.split("")); + assertEq('~/a/.split("a")', ~/a/.split("a")); + assertEq('~/a/.split("aa")', ~/a/.split("aa")); + assertEq('~/a/.split("b")', ~/a/.split("b")); + assertEq('~/a/.split("ab")', ~/a/.split("ab")); + assertEq('~/a/.split("ba")', ~/a/.split("ba")); + assertEq('~/a/.split("aba")', ~/a/.split("aba")); + assertEq('~/a/.split("bab")', ~/a/.split("bab")); + assertEq('~/a/.split("baba")', ~/a/.split("baba")); // split + g - ~/a/g.split("") == [""]; - ~/a/g.split("a") == ["",""]; - ~/a/g.split("aa") == ["","",""]; - ~/a/g.split("b") == ["b"]; - ~/a/g.split("ab") == ["","b"]; - ~/a/g.split("ba") == ["b",""]; - ~/a/g.split("aba") == ["","b",""]; - ~/a/g.split("bab") == ["b","b"]; - ~/a/g.split("baba") == ["b","b",""]; + assertEq('~/a/g.split("")', ~/a/g.split("")); + assertEq('~/a/g.split("a")', ~/a/g.split("a")); + assertEq('~/a/g.split("aa")', ~/a/g.split("aa")); + assertEq('~/a/g.split("b")', ~/a/g.split("b")); + assertEq('~/a/g.split("ab")', ~/a/g.split("ab")); + assertEq('~/a/g.split("ba")', ~/a/g.split("ba")); + assertEq('~/a/g.split("aba")', ~/a/g.split("aba")); + assertEq('~/a/g.split("bab")', ~/a/g.split("bab")); + assertEq('~/a/g.split("baba")', ~/a/g.split("baba")); // replace - ~/a/.replace("", "z") == ""; - ~/a/.replace("a", "z") == "z"; - ~/a/.replace("aa", "z") == "za"; - ~/a/.replace("b", "z") == "b"; - ~/a/.replace("ab", "z") == "zb"; - ~/a/.replace("ba", "z") == "bz"; - ~/a/.replace("aba", "z") == "zba"; - ~/a/.replace("bab", "z") == "bzb"; - ~/a/.replace("baba", "z") == "bzba"; + assertEq('~/a/.replace("", "z")', ~/a/.replace("", "z")); + assertEq('~/a/.replace("a", "z")', ~/a/.replace("a", "z")); + assertEq('~/a/.replace("aa", "z")', ~/a/.replace("aa", "z")); + assertEq('~/a/.replace("b", "z")', ~/a/.replace("b", "z")); + assertEq('~/a/.replace("ab", "z")', ~/a/.replace("ab", "z")); + assertEq('~/a/.replace("ba", "z")', ~/a/.replace("ba", "z")); + assertEq('~/a/.replace("aba", "z")', ~/a/.replace("aba", "z")); + assertEq('~/a/.replace("bab", "z")', ~/a/.replace("bab", "z")); + assertEq('~/a/.replace("baba", "z")', ~/a/.replace("baba", "z")); + + //Util.runKnownBug("Regex Syntax doesn't work", () -> { + assertCompiles("~/a/;"); + assertCompiles("~/(a)a\\0\\//g;"); + assertCompiles("~/(a)a\\0\\//g+5;"); + assertError("~/(a)a\\0\\//ga;", Parser.getBaseError(ECustom("Invalid regex expression option \"a\""))); + //assertEq("~/a/+5", null);//~/a/+5); + //}); // replace + g - ~/a/g.replace("", "z") == ""; - ~/a/g.replace("a", "z") == "z"; - ~/a/g.replace("aa", "z") == "zz"; - ~/a/g.replace("b", "z") == "b"; - ~/a/g.replace("ab", "z") == "zb"; - ~/a/g.replace("ba", "z") == "bz"; - ~/a/g.replace("aba", "z") == "zbz"; - ~/a/g.replace("bab", "z") == "bzb"; - ~/a/g.replace("baba", "z") == "bzbz";*/ + assertEq('~/a/g.replace("", "z")', ~/a/g.replace("", "z")); + assertEq('~/a/g.replace("a", "z")', ~/a/g.replace("a", "z")); + assertEq('~/a/g.replace("aa", "z")', ~/a/g.replace("aa", "z")); + assertEq('~/a/g.replace("b", "z")', ~/a/g.replace("b", "z")); + assertEq('~/a/g.replace("ab", "z")', ~/a/g.replace("ab", "z")); + assertEq('~/a/g.replace("ba", "z")', ~/a/g.replace("ba", "z")); + assertEq('~/a/g.replace("aba", "z")', ~/a/g.replace("aba", "z")); + assertEq('~/a/g.replace("bab", "z")', ~/a/g.replace("bab", "z")); + assertEq('~/a/g.replace("baba", "z")', ~/a/g.replace("baba", "z")); + + // var 0 = 5; // Missing variable identifier } override function teardown() { diff --git a/tests/src/tests/TestCase.hx b/tests/src/tests/TestCase.hx index 5b8dff8..f1177f3 100644 --- a/tests/src/tests/TestCase.hx +++ b/tests/src/tests/TestCase.hx @@ -20,6 +20,20 @@ class TestCase extends HScriptRunner { return true; } + public function assertCompiles(script:String, ?message:String, ?pos:haxe.PosInfos) { + if(message == null) + message = script; + try { + var result = Util.parseUnsafe(script); + } catch(e:hscript.Error) { + var e = Printer.getPrintableError(e); + Sys.println("# Error trying to compile: " + script); + Sys.println("## Error: " + e); + return Util.failed(); + } + return Util.passed(); + } + public function assertError(script:String, expectedError:hscript.Error, ?message:String, ?vars:Dynamic, ?pos:haxe.PosInfos) { var expectedError = Printer.getPrintableError(expectedError); if(message == null)