From 2a644d161599703277a2058b304b9fd0071df74a Mon Sep 17 00:00:00 2001 From: Ruben Bridgewater Date: Wed, 1 Jul 2026 19:17:43 +0200 Subject: [PATCH] chore: remove the unmaintained pure-JS lexer port lexer.js reimplemented the WebAssembly lexer in hand-written JS, kept in sync by hand, but was unreachable through the package `exports` (only `.` and `./js` are exposed - the wasm and asm.js builds) and `chomp test` never ran it, so it could drift silently. The asm.js build already covers the no-WebAssembly case, and the removed `if (!js)` test guards now run unconditionally, matching their existing behavior under the wasm and asm suites. --- chompfile.toml | 4 - lexer.js | 1009 ------------------------------------------------ package.json | 3 +- test/_unit.cjs | 48 +-- 4 files changed, 5 insertions(+), 1059 deletions(-) delete mode 100644 lexer.js diff --git a/chompfile.toml b/chompfile.toml index 5330958..8b3d74d 100644 --- a/chompfile.toml +++ b/chompfile.toml @@ -250,10 +250,6 @@ run = ''' name = 'test' deps = ['test:wasm', 'test:asm'] -[[task]] -name = 'test:js' -run = 'mocha -b -u tdd test/*.cjs' - [[task]] name = 'test:asm' deps = ['dist/lexer.asm.js'] diff --git a/lexer.js b/lexer.js deleted file mode 100644 index fae6cab..0000000 --- a/lexer.js +++ /dev/null @@ -1,1009 +0,0 @@ -let source, pos, end, - openTokenDepth, - lastTokenPos, - openTokenPosStack, - openClassPosStack, - curDynamicImport, - templateStackDepth, - facade, - lastSlashWasDivision, - nextBraceIsClass, - templateDepth, - templateStack, - imports, - exports, - exportStatementStart, - name; - -function addImport (ss, s, e, d) { - const impt = { ss, se: d === -2 ? e : d === -1 ? e + 1 : 0, s, e, d, a: -1, n: undefined, at: null }; - imports.push(impt); - return impt; -} - -function addExport (s, e, ls, le) { - exports.push({ - s, - e, - ls, - le, - ss: exportStatementStart, - n: s[0] === '"' ? readString(s, '"') : s[0] === "'" ? readString(s, "'") : source.slice(s, e), - ln: ls[0] === '"' ? readString(ls, '"') : ls[0] === "'" ? readString(ls, "'") : source.slice(ls, le) - }); -} - -function readName (impt) { - let { d, s } = impt; - if (d !== -1) - s++; - impt.n = readString(s, source.charCodeAt(s - 1)); -} - -// Note: parsing is based on the _assumption_ that the source is already valid -export function parse (_source, _name) { - openTokenDepth = 0; - curDynamicImport = null; - templateDepth = -1; - lastTokenPos = -1; - lastSlashWasDivision = false; - templateStack = Array(1024); - templateStackDepth = 0; - openTokenPosStack = Array(1024); - openClassPosStack = Array(1024); - nextBraceIsClass = false; - facade = true; - name = _name || '@'; - - imports = []; - exports = []; - - source = _source; - pos = -1; - end = source.length - 1; - let ch = 0; - - // start with a pure "module-only" parser - m: while (pos++ < end) { - ch = source.charCodeAt(pos); - - if (ch === 32 || ch < 14 && ch > 8) - continue; - - switch (ch) { - case 101/*e*/: - if (openTokenDepth === 0 && keywordStart(pos) && source.startsWith('xport', pos + 1)) { - tryParseExportStatement(); - // export might have been a non-pure declaration - if (!facade) { - lastTokenPos = pos; - break m; - } - } - break; - case 105/*i*/: - if (keywordStart(pos) && source.startsWith('mport', pos + 1)) - tryParseImportStatement(); - break; - case 59/*;*/: - break; - case 47/*/*/: { - const next_ch = source.charCodeAt(pos + 1); - if (next_ch === 47/*/*/) { - lineComment(); - // dont update lastToken - continue; - } - else if (next_ch === 42/***/) { - blockComment(true); - // dont update lastToken - continue; - } - // fallthrough - } - default: - // as soon as we hit a non-module token, we go to main parser - facade = false; - pos--; - break m; - } - lastTokenPos = pos; - } - - while (pos++ < end) { - ch = source.charCodeAt(pos); - - if (ch === 32 || ch < 14 && ch > 8) - continue; - - switch (ch) { - case 101/*e*/: - if (openTokenDepth === 0 && keywordStart(pos) && source.startsWith('xport', pos + 1)) - tryParseExportStatement(); - break; - case 105/*i*/: - if (keywordStart(pos) && source.startsWith('mport', pos + 1)) - tryParseImportStatement(); - break; - case 99/*c*/: - if (keywordStart(pos) && source.startsWith('lass', pos + 1) && isBrOrWs(source.charCodeAt(pos + 5))) - nextBraceIsClass = true; - break; - case 40/*(*/: - openTokenPosStack[openTokenDepth++] = lastTokenPos; - break; - case 41/*)*/: - if (openTokenDepth === 0) - syntaxError(); - openTokenDepth--; - if (curDynamicImport && curDynamicImport.d === openTokenPosStack[openTokenDepth]) { - if (curDynamicImport.e === 0) - curDynamicImport.e = pos; - curDynamicImport.se = pos; - curDynamicImport = null; - } - break; - case 91/*[*/: - openTokenPosStack[openTokenDepth++] = lastTokenPos; - break; - case 93/*]*/: - if (openTokenDepth === 0) - syntaxError(); - openTokenDepth--; - break; - case 44/*,*/: - if (curDynamicImport && curDynamicImport.e === 0 && curDynamicImport.d === openTokenPosStack[openTokenDepth - 1]) { - curDynamicImport.e = lastTokenPos + 1; - pos++; - commentWhitespace(true); - curDynamicImport.a = pos; - pos--; - } - break; - case 123/*{*/: - // dynamic import followed by { is not a dynamic import (so remove) - // this is a sneaky way to get around { import () {} } v { import () } - // block / object ambiguity without a parser (assuming source is valid) - // se marks the closing paren; e is moved before the first comma for import(a, b) - if (source.charCodeAt(lastTokenPos) === 41/*)*/ && imports.length && imports[imports.length - 1].se === lastTokenPos) { - imports.pop(); - } - openClassPosStack[openTokenDepth] = nextBraceIsClass; - nextBraceIsClass = false; - openTokenPosStack[openTokenDepth++] = lastTokenPos; - break; - case 125/*}*/: - if (openTokenDepth === 0) - syntaxError(); - if (openTokenDepth-- === templateDepth) { - templateDepth = templateStack[--templateStackDepth]; - templateString(); - } - else { - if (templateDepth !== -1 && openTokenDepth < templateDepth) - syntaxError(); - } - break; - case 39/*'*/: - case 34/*"*/: - stringLiteral(ch); - break; - case 47/*/*/: { - const next_ch = source.charCodeAt(pos + 1); - if (next_ch === 47/*/*/) { - lineComment(); - // dont update lastToken - continue; - } - else if (next_ch === 42/***/) { - blockComment(true); - // dont update lastToken - continue; - } - else { - // Division / regex ambiguity handling based on checking backtrack analysis of: - // - what token came previously (lastToken) - // - if a closing brace or paren, what token came before the corresponding - // opening brace or paren (lastOpenTokenIndex) - const lastToken = source.charCodeAt(lastTokenPos); - const lastExport = exports[exports.length - 1]; - if (isExpressionPunctuator(lastToken) && - !(lastToken === 46/*.*/ && (source.charCodeAt(lastTokenPos - 1) >= 48/*0*/ && source.charCodeAt(lastTokenPos - 1) <= 57/*9*/)) && - !(lastToken === 43/*+*/ && source.charCodeAt(lastTokenPos - 1) === 43/*+*/) && !(lastToken === 45/*-*/ && source.charCodeAt(lastTokenPos - 1) === 45/*-*/) || - lastToken === 41/*)*/ && isParenKeyword(openTokenPosStack[openTokenDepth]) || - openTokenDepth > 0 && lastToken === 102/*f*/ && source.charCodeAt(lastTokenPos - 1) === 111/*o*/ && isForOfBinding(lastTokenPos - 2) && isForParen(openTokenPosStack[openTokenDepth - 1]) || - lastToken === 125/*}*/ && (isExpressionTerminator(openTokenPosStack[openTokenDepth]) || openClassPosStack[openTokenDepth]) || - lastToken === 47/*/*/ && lastSlashWasDivision || - isExpressionKeyword(lastTokenPos) || - !lastToken) { - regularExpression(); - lastSlashWasDivision = false; - } - else if (lastExport && lastTokenPos >= lastExport.s && lastTokenPos <= lastExport.e) { - // export default /some-regexp/ - regularExpression(); - lastSlashWasDivision = false; - } - else { - lastSlashWasDivision = true; - } - } - break; - } - case 96/*`*/: - templateString(); - break; - } - lastTokenPos = pos; - } - - if (templateDepth !== -1 || openTokenDepth) - syntaxError(); - - return [imports, exports, facade]; -} - -function tryParseImportStatement () { - const startPos = pos; - - pos += 6; - - let ch = commentWhitespace(true); - - switch (ch) { - // dynamic import - case 40/*(*/: - openTokenPosStack[openTokenDepth++] = startPos; - if (source.charCodeAt(lastTokenPos) === 46/*.*/) - return; - // dynamic import indicated by positive d - // try parse a string, to record a safe dynamic import string - pos++; - ch = commentWhitespace(true); - // The specifier start is recorded after leading whitespace/comments so it - // points at the literal, matching the C lexer (src/lexer.c). - const impt = addImport(startPos, pos, 0, startPos); - curDynamicImport = impt; - if (ch === 39/*'*/ || ch === 34/*"*/) { - stringLiteral(ch); - } - else if (ch === 96/*`*/ && noSubstitutionTemplate()) { - // A no-substitution template literal is a constant string, so it is a - // safe specifier exactly like a quoted one. An interpolated template - // leaves noSubstitutionTemplate() false and falls through to the open- - // token machinery, which records the import as unsafe (n stays unset). - } - else { - pos--; - return; - } - pos++; - ch = commentWhitespace(true); - if (ch === 44/*,*/) { - impt.e = pos; - pos++; - ch = commentWhitespace(true); - impt.a = pos; - readName(impt); - pos--; - } - else if (ch === 41/*)*/) { - openTokenDepth--; - impt.e = pos; - impt.se = pos; - readName(impt); - } - else { - pos--; - } - return; - // import.meta - case 46/*.*/: - pos++; - ch = commentWhitespace(true); - // import.meta indicated by d === -2 - if (ch === 109/*m*/ && source.startsWith('eta', pos + 1) && source.charCodeAt(lastTokenPos) !== 46/*.*/) - addImport(startPos, startPos, pos + 4, -2); - return; - - default: - // no space after "import" -> not an import keyword - if (pos === startPos + 6) - break; - case 34/*"*/: - case 39/*'*/: - case 123/*{*/: - case 42/***/: - // import statement only permitted at base-level - if (openTokenDepth !== 0) { - pos--; - return; - } - while (pos < end) { - ch = source.charCodeAt(pos); - if (ch === 39/*'*/ || ch === 34/*"*/) { - readImportString(startPos, ch); - return; - } - pos++; - } - syntaxError(); - } -} - -function tryParseExportStatement () { - const sStartPos = pos; - const prevExport = exports.length; - - pos += 6; - - const curPos = pos; - - let ch = commentWhitespace(true); - - // Only commit the statement start once this is a real export: skipExpression - // re-enters here for an `export`-prefixed identifier (e.g. `exports`) in an - // initializer, which would otherwise clobber the start for later bindings. - if (pos === curPos && !isPunctuator(ch)) - return; - - exportStatementStart = sStartPos; - - switch (ch) { - // export default ... - case 100/*d*/: - addExport(pos, pos + 7, -1, -1); - return; - - // export async? function*? name () { - case 97/*a*/: - pos += 5; - commentWhitespace(true); - // fallthrough - case 102/*f*/: - pos += 8; - ch = commentWhitespace(true); - if (ch === 42/***/) { - pos++; - ch = commentWhitespace(true); - } - const startPos = pos; - ch = readToWsOrPunctuator(ch); - addExport(startPos, pos, startPos, pos); - pos--; - return; - - // export class name ... - case 99/*c*/: - if (source.startsWith('lass', pos + 1) && isBrOrWsOrPunctuatorNotDot(source.charCodeAt(pos + 5))) { - pos += 5; - ch = commentWhitespace(true); - const startPos = pos; - ch = readToWsOrPunctuator(ch); - addExport(startPos, pos, startPos, pos); - pos--; - return; - } - pos += 2; - // fallthrough - - // export var/let/const name = ...(, name = ...)+ - case 118/*v*/: - case 109/*l*/: - // destructured initializations not currently supported (skipped for { or [) - // also, lexing names after variable equals is skipped (export var p = function () { ... }, q = 5 skips "q") - pos += 2; - facade = false; - do { - pos++; - ch = commentWhitespace(true); - const startPos = pos; - ch = readToWsOrPunctuator(ch); - // dont yet handle [ { destructurings - if (ch === 123/*{*/ || ch === 91/*[*/) { - pos--; - return; - } - if (pos === startPos) - return; - addExport(startPos, pos, startPos, pos); - ch = commentWhitespace(true); - if (ch === 61/*=*/) { - pos--; - return; - } - } while (ch === 44/*,*/); - pos--; - return; - - - // export {...} - case 123/*{*/: - pos++; - ch = commentWhitespace(true); - while (true) { - const startPos = pos; - readToWsOrPunctuator(ch); - const endPos = pos; - commentWhitespace(true); - ch = readExportAs(startPos, endPos); - // , - if (ch === 44/*,*/) { - pos++; - ch = commentWhitespace(true); - } - if (ch === 125/*}*/) - break; - if (pos === startPos) - return syntaxError(); - if (pos > end) - return syntaxError(); - } - pos++; - ch = commentWhitespace(true); - break; - - // export * - // export * as X - case 42/***/: - pos++; - commentWhitespace(true); - ch = readExportAs(pos, pos); - ch = commentWhitespace(true); - break; - } - - // from ... - if (ch === 102/*f*/ && source.startsWith('rom', pos + 1)) { - pos += 4; - readImportString(sStartPos, commentWhitespace(true)); - - // There were no local names. - for (let i = prevExport; i < exports.length; ++i) { - exports[i].ls = exports[i].le = -1; - exports[i].ln = undefined; - } - } - else { - pos--; - } -} - -/* - * Ported from Acorn - * - * MIT License - - * Copyright (C) 2012-2020 by various contributors (see AUTHORS) - - * Permission is hereby granted, free of charge, to any person obtaining a copy - * of this software and associated documentation files (the "Software"), to deal - * in the Software without restriction, including without limitation the rights - * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell - * copies of the Software, and to permit persons to whom the Software is - * furnished to do so, subject to the following conditions: - - * The above copyright notice and this permission notice shall be included in - * all copies or substantial portions of the Software. - - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN - * THE SOFTWARE. - */ -let acornPos; -function readString (start, quote) { - acornPos = start; - let out = '', chunkStart = acornPos; - for (;;) { - if (acornPos >= source.length) syntaxError(); - const ch = source.charCodeAt(acornPos); - if (ch === quote) break; - if (ch === 92) { // '\' - out += source.slice(chunkStart, acornPos); - out += readEscapedChar(); - chunkStart = acornPos; - } - else if (ch === 0x2028 || ch === 0x2029) { - ++acornPos; - } - else { - // Template literals (backtick quote) permit raw line breaks; string - // literals do not. - if (isBr(ch) && quote !== 96/*`*/) syntaxError(); - ++acornPos; - } - } - out += source.slice(chunkStart, acornPos++); - return out; -} - -// Used to read escaped characters - -function readEscapedChar () { - let ch = source.charCodeAt(++acornPos); - ++acornPos; - switch (ch) { - case 110: return '\n'; // 'n' -> '\n' - case 114: return '\r'; // 'r' -> '\r' - case 120: return String.fromCharCode(readHexChar(2)); // 'x' - case 117: return readCodePointToString(); // 'u' - case 116: return '\t'; // 't' -> '\t' - case 98: return '\b'; // 'b' -> '\b' - case 118: return '\u000b'; // 'v' -> '\u000b' - case 102: return '\f'; // 'f' -> '\f' - case 13: if (source.charCodeAt(acornPos) === 10) ++acornPos; // '\r\n' - case 10: // ' \n' - return ''; - case 56: - case 57: - syntaxError(); - default: - if (ch >= 48 && ch <= 55) { - let octalStr = source.substr(acornPos - 1, 3).match(/^[0-7]+/)[0]; - let octal = parseInt(octalStr, 8); - if (octal > 255) { - octalStr = octalStr.slice(0, -1); - octal = parseInt(octalStr, 8); - } - acornPos += octalStr.length - 1; - ch = source.charCodeAt(acornPos); - if (octalStr !== '0' || ch === 56 || ch === 57) - syntaxError(); - return String.fromCharCode(octal); - } - if (isBr(ch)) { - // Unicode new line characters after \ get removed from output in both - // template literals and strings - return ''; - } - return String.fromCharCode(ch); - } -} - -// Used to read character escape sequences ('\x', '\u', '\U'). - -function readHexChar (len) { - const start = acornPos; - let total = 0, lastCode = 0; - for (let i = 0; i < len; ++i, ++acornPos) { - let code = source.charCodeAt(acornPos), val; - - if (code === 95) { - if (lastCode === 95 || i === 0) syntaxError(); - lastCode = code; - continue; - } - - if (code >= 97) val = code - 97 + 10; // a - else if (code >= 65) val = code - 65 + 10; // A - else if (code >= 48 && code <= 57) val = code - 48; // 0-9 - else break; - if (val >= 16) break; - lastCode = code; - total = total * 16 + val; - } - - if (lastCode === 95 || acornPos - start !== len) syntaxError(); - - return total; -} - -// Read a string value, interpreting backslash-escapes. - -function readCodePointToString () { - const ch = source.charCodeAt(acornPos); - let code; - if (ch === 123) { // '{' - ++acornPos; - code = readHexChar(source.indexOf('}', acornPos) - acornPos); - ++acornPos; - if (code > 0x10FFFF) syntaxError(); - } else { - code = readHexChar(4); - } - // UTF-16 Decoding - if (code <= 0xFFFF) return String.fromCharCode(code); - code -= 0x10000; - return String.fromCharCode((code >> 10) + 0xD800, (code & 1023) + 0xDC00); -} - -/* - * - */ - -function readExportAs (startPos, endPos) { - let ch = source.charCodeAt(pos); - let ls = startPos, le = endPos; - if (ch === 97 /*a*/) { - pos += 2; - ch = commentWhitespace(true); - startPos = pos; - readToWsOrPunctuator(ch); - endPos = pos; - ch = commentWhitespace(true); - } - if (pos !== startPos) - addExport(startPos, endPos, ls, le); - return ch; -} - -function readImportString (ss, ch) { - const startPos = pos + 1; - if (ch === 39/*'*/ || ch === 34/*"*/) { - stringLiteral(ch); - } - else { - syntaxError(); - return; - } - const impt = addImport(ss, startPos, pos, -1); - readName(impt); - pos++; - ch = commentWhitespace(false); - if (ch !== 119/*w*/ || !source.startsWith('ith', pos + 1)) { - pos--; - return; - } - const attrIndex = pos; - - pos += 4; - ch = commentWhitespace(true); - if (ch !== 123/*{*/) { - pos = attrIndex; - return; - } - const attrStart = pos; - const attrs = []; - do { - pos++; - ch = commentWhitespace(true); - let key, keyStart, keyEnd; - if (ch === 39/*'*/ || ch === 34/*"*/) { - keyStart = pos; - stringLiteral(ch); - keyEnd = pos + 1; - key = readString(keyStart, ch); - pos++; - ch = commentWhitespace(true); - } - else { - keyStart = pos; - ch = readToWsOrPunctuator(ch); - keyEnd = pos; - key = source.slice(keyStart, keyEnd); - } - if (ch !== 58/*:*/) { - pos = attrIndex; - return; - } - pos++; - ch = commentWhitespace(true); - let value, valueStart; - if (ch === 39/*'*/ || ch === 34/*"*/) { - valueStart = pos; - stringLiteral(ch); - value = readString(valueStart, ch); - } - else { - pos = attrIndex; - return; - } - attrs.push([key, value]); - pos++; - ch = commentWhitespace(true); - if (ch === 44/*,*/) { - pos++; - continue; - } - if (ch === 125/*}*/) - break; - pos = attrIndex; - return; - } while (true); - impt.a = attrStart; - impt.at = attrs; - impt.se = pos + 1; -} - -function commentWhitespace (br) { - let ch; - do { - ch = source.charCodeAt(pos); - if (ch === 47/*/*/) { - const next_ch = source.charCodeAt(pos + 1); - if (next_ch === 47/*/*/) - lineComment(); - else if (next_ch === 42/***/) - blockComment(br); - else - return ch; - } - else if (br ? !isBrOrWs(ch): !isWsNotBr(ch)) { - return ch; - } - } while (pos++ < end); - return ch; -} - -function templateString () { - while (pos++ < end) { - const ch = source.charCodeAt(pos); - if (ch === 36/*$*/ && source.charCodeAt(pos + 1) === 123/*{*/) { - pos++; - templateStack[templateStackDepth++] = templateDepth; - templateDepth = ++openTokenDepth; - return; - } - if (ch === 96/*`*/) - return; - if (ch === 92/*\*/) - pos++; - } - syntaxError(); -} - -// pos AT the opening backtick. A no-substitution template literal (no ${...}) -// is a constant string, so a dynamic import can record it as a safe specifier. -// On success consumes it, leaves pos AT the closing backtick and returns true. -// On a substitution or EOF restores pos and returns false, leaving the literal -// to the main loop's template handling. -function noSubstitutionTemplate () { - const startPos = pos; - while (pos++ < end) { - const ch = source.charCodeAt(pos); - if (ch === 96/*`*/) - return true; - if (ch === 92/*\*/) { - pos++; - continue; - } - if (ch === 36/*$*/ && source.charCodeAt(pos + 1) === 123/*{*/) - break; - } - pos = startPos; - return false; -} - -function blockComment (br) { - pos++; - while (pos++ < end) { - const ch = source.charCodeAt(pos); - if (!br && isBr(ch)) - return; - if (ch === 42/***/ && source.charCodeAt(pos + 1) === 47/*/*/) { - pos++; - return; - } - } -} - -function lineComment () { - while (pos++ < end) { - const ch = source.charCodeAt(pos); - if (ch === 10/*\n*/ || ch === 13/*\r*/) - return; - } -} - -function stringLiteral (quote) { - while (pos++ < end) { - let ch = source.charCodeAt(pos); - if (ch === quote) - return; - if (ch === 92/*\*/) { - ch = source.charCodeAt(++pos); - if (ch === 13/*\r*/ && source.charCodeAt(pos + 1) === 10/*\n*/) - pos++; - } - else if (isBr(ch)) - break; - } - syntaxError(); -} - -function regexCharacterClass () { - while (pos++ < end) { - let ch = source.charCodeAt(pos); - if (ch === 93/*]*/) - return ch; - if (ch === 92/*\*/) - pos++; - else if (ch === 10/*\n*/ || ch === 13/*\r*/) - break; - } - syntaxError(); -} - -function regularExpression () { - while (pos++ < end) { - let ch = source.charCodeAt(pos); - if (ch === 47/*/*/) - return; - if (ch === 91/*[*/) - ch = regexCharacterClass(); - else if (ch === 92/*\*/) - pos++; - else if (ch === 10/*\n*/ || ch === 13/*\r*/) - break; - } - syntaxError(); -} - -function readToWsOrPunctuator (ch) { - do { - if (isBrOrWs(ch) || isPunctuator(ch)) - return ch; - } while (ch = source.charCodeAt(++pos)); - return ch; -} - -// Note: non-asii BR and whitespace checks omitted for perf / footprint -// if there is a significant user need this can be reconsidered -function isBr (c) { - return c === 13/*\r*/ || c === 10/*\n*/; -} - -function isWsNotBr (c) { - return c === 9 || c === 11 || c === 12 || c === 32 || c === 160; -} - -function isBrOrWs (c) { - return c > 8 && c < 14 || c === 32 || c === 160; -} - -function isBrOrWsOrPunctuatorNotDot (c) { - return c > 8 && c < 14 || c === 32 || c === 160 || isPunctuator(c) && c !== 46/*.*/; -} - -function keywordStart (pos) { - return pos === 0 || isBrOrWsOrPunctuatorNotDot(source.charCodeAt(pos - 1)); -} - -function readPrecedingKeyword (pos, match) { - if (pos < match.length - 1) - return false; - return source.startsWith(match, pos - match.length + 1) && (pos === 0 || isBrOrWsOrPunctuatorNotDot(source.charCodeAt(pos - match.length))); -} - -function readPrecedingKeyword1 (pos, ch) { - return source.charCodeAt(pos) === ch && (pos === 0 || isBrOrWsOrPunctuatorNotDot(source.charCodeAt(pos - 1))); -} - -// Detects one of case, debugger, delete, do, else, in, instanceof, new, -// return, throw, typeof, void, yield, await -function isExpressionKeyword (pos) { - switch (source.charCodeAt(pos)) { - case 100/*d*/: - switch (source.charCodeAt(pos - 1)) { - case 105/*i*/: - // void - return readPrecedingKeyword(pos - 2, 'vo'); - case 108/*l*/: - // yield - return readPrecedingKeyword(pos - 2, 'yie'); - default: - return false; - } - case 101/*e*/: - switch (source.charCodeAt(pos - 1)) { - case 115/*s*/: - switch (source.charCodeAt(pos - 2)) { - case 108/*l*/: - // else - return readPrecedingKeyword1(pos - 3, 101/*e*/); - case 97/*a*/: - // case - return readPrecedingKeyword1(pos - 3, 99/*c*/); - default: - return false; - } - case 116/*t*/: - // delete - return readPrecedingKeyword(pos - 2, 'dele'); - default: - return false; - } - case 102/*f*/: - if (source.charCodeAt(pos - 1) !== 111/*o*/ || source.charCodeAt(pos - 2) !== 101/*e*/) - return false; - switch (source.charCodeAt(pos - 3)) { - case 99/*c*/: - // instanceof - return readPrecedingKeyword(pos - 4, 'instan'); - case 112/*p*/: - // typeof - return readPrecedingKeyword(pos - 4, 'ty'); - default: - return false; - } - case 110/*n*/: - // in, return - return readPrecedingKeyword1(pos - 1, 105/*i*/) || readPrecedingKeyword(pos - 1, 'retur'); - case 111/*o*/: - // do - return readPrecedingKeyword1(pos - 1, 100/*d*/); - case 114/*r*/: - // debugger - return readPrecedingKeyword(pos - 1, 'debugge'); - case 116/*t*/: - // await - return readPrecedingKeyword(pos - 1, 'awai'); - case 119/*w*/: - switch (source.charCodeAt(pos - 1)) { - case 101/*e*/: - // new - return readPrecedingKeyword1(pos - 2, 110/*n*/); - case 111/*o*/: - // throw - return readPrecedingKeyword(pos - 2, 'thr'); - default: - return false; - } - } - return false; -} - -function isParenKeyword (curPos) { - return source.charCodeAt(curPos) === 101/*e*/ && source.startsWith('whil', curPos - 4) || - source.charCodeAt(curPos) === 114/*r*/ && source.startsWith('fo', curPos - 2) || - source.charCodeAt(curPos - 1) === 105/*i*/ && source.charCodeAt(curPos) === 102/*f*/; -} - -function isForParen (curPos) { - return source.charCodeAt(curPos) === 114/*r*/ && source.startsWith('fo', curPos - 2); -} - -// In valid JS, the for-of `of` keyword always follows a binding, -// which ends with an identifier-tail char, ']', '}', or ')'. -function isForOfBinding (pos) { - const ch = source.charCodeAt(pos); - if (!isBrOrWs(ch) && ch !== 93/*]*/ && ch !== 125/*}*/ && ch !== 41/*)*/) - return false; - while (pos > 0 && isBrOrWs(source.charCodeAt(pos))) - pos--; - const c = source.charCodeAt(pos); - return c === 93/*]*/ || c === 125/*}*/ || c === 41/*)*/ || !isPunctuator(c); -} - -function isPunctuator (ch) { - // 23 possible punctuator endings: !%&()*+,-./:;<=>?[]^{}|~ - return ch === 33/*!*/ || ch === 37/*%*/ || ch === 38/*&*/ || - ch > 39 && ch < 48 || ch > 57 && ch < 64 || - ch === 91/*[*/ || ch === 93/*]*/ || ch === 94/*^*/ || - ch > 122 && ch < 127; -} - -function isExpressionPunctuator (ch) { - // 20 possible expression endings: !%&(*+,-.:;<=>?[^{|~ - return ch === 33/*!*/ || ch === 37/*%*/ || ch === 38/*&*/ || - ch > 39 && ch < 47 && ch !== 41 || ch > 57 && ch < 64 || - ch === 91/*[*/ || ch === 94/*^*/ || ch > 122 && ch < 127 && ch !== 125/*}*/; -} - -function isExpressionTerminator (curPos) { - // detects: - // => ; ) finally catch else - // as all of these followed by a { will indicate a statement brace - switch (source.charCodeAt(curPos)) { - case 62/*>*/: - return source.charCodeAt(curPos - 1) === 61/*=*/; - case 59/*;*/: - case 41/*)*/: - return true; - case 104/*h*/: - return source.startsWith('catc', curPos - 4); - case 121/*y*/: - return source.startsWith('finall', curPos - 6); - case 101/*e*/: - return source.startsWith('els', curPos - 3); - } - return false; -} - -function syntaxError () { - throw Object.assign(new Error(`Parse error ${name}:${source.slice(0, pos).split('\n').length}:${pos - source.lastIndexOf('\n', pos - 1)}`), { idx: pos }); -} \ No newline at end of file diff --git a/package.json b/package.json index 61e14d3..4fe3665 100755 --- a/package.json +++ b/package.json @@ -37,8 +37,7 @@ }, "files": [ "dist", - "types", - "lexer.js" + "types" ], "type": "module", "repository": { diff --git a/test/_unit.cjs b/test/_unit.cjs index 39110f4..973565c 100755 --- a/test/_unit.cjs +++ b/test/_unit.cjs @@ -1,20 +1,15 @@ const assert = require('assert'); -let js = false; let parse; const init = (async () => { if (parse) return; - if (process.env.WASM) { - const m = await import('../dist/lexer.js'); - await m.init; - parse = m.parse; - } - else if (process.env.ASM) { + if (process.env.ASM) { ({ parse } = await import('../dist/lexer.asm.js')); } else { - js = true; - ({ parse } = await import('../lexer.js')); + const m = await import('../dist/lexer.js'); + await m.init; + parse = m.parse; } })(); @@ -673,7 +668,6 @@ suite('Lexer', () => { parse(source); }); - if (!js) test('Multiline dynamic import on windows', () => { const source = `import(\n"./statehash\\u1011.js"\r)`; const [imports] = parse(source); @@ -682,7 +676,6 @@ suite('Lexer', () => { assert.strictEqual(source.slice(imports[0].s, imports[0].e), '"./statehash\\u1011.js"'); }); - if (!js) test('Basic nested dynamic import support', () => { const source = `await import (await import ('foo'))`; const [imports] = parse(source); @@ -695,7 +688,6 @@ suite('Lexer', () => { assert.strictEqual(source.slice(imports[1].s, imports[1].e), '\'foo\''); }); - if (!js) test('Import attributes', () => { const source = ` import json from "./foo.json" with { type: "json" }; @@ -719,7 +711,6 @@ suite('Lexer', () => { assertExportIs(source, exports[0], {n: 'p', ln: 'p', a: false}); }); - if (!js) test('Import attributes', () => { const source = ` import json from "./foo.json" with { type: "json" }; @@ -804,7 +795,6 @@ suite('Lexer', () => { assert.deepStrictEqual(exports.map(expt => expt.n), ['default', 'default', 'default', 'default', 'default', 'default']); }); - if (!js) test('Regexp keyword prefixes', () => { const [imports] = parse(` x: while (true) { @@ -1051,7 +1041,6 @@ suite('Lexer', () => { assert.strictEqual(imports.length, 0); }); - if (!js) test('dynamic import edge cases', () => { const source = ` ({ @@ -1140,7 +1129,6 @@ function x() { assertExportIs(source, exports[0], { n: 'a', ln: 'a' }); }); - if (!js) test('Strings', () => { const source = ` ""; @@ -1290,7 +1278,6 @@ function x() { }); suite('Import From', () => { - if (!js) test('non-identifier-string as (doubleQuote)', () => { const source = ` import { "~123" as foo0 } from './mod0.js'; @@ -1319,7 +1306,6 @@ function x() { assert.strictEqual(imports[8].n, './mod8.js'); }); - if (!js) test('non-identifier-string as (singleQuote)', () => { const source = ` import { '~123' as foo0 } from './mod0.js'; @@ -1346,7 +1332,6 @@ function x() { assert.strictEqual(imports[8].n, './mod8.js'); }); - if (!js) test('with-backslash-keywords as (doubleQuote)', () => { const source = String.raw` import { " slash\\ " as foo0 } from './mod0.js'; @@ -1363,7 +1348,6 @@ function x() { assert.strictEqual(imports[3].n, './mod3.js'); }); - if (!js) test('with-backslash-keywords as (singleQuote)', () => { const source = String.raw` import { ' slash\\ ' as foo0 } from './mod0.js'; @@ -1380,7 +1364,6 @@ function x() { assert.strictEqual(imports[3].n, './mod3.js'); }); - if (!js) test('with-emoji as', () => { const source = ` import { "hm🤔" as foo0 } from './mod0.js'; @@ -1393,7 +1376,6 @@ function x() { assert.strictEqual(imports[1].n, './mod1.js'); }); - if (!js) test('double-quotes-and-curly-bracket', () => { const source = ` import { asdf as "b} from 'wrong'" } from 'mod0';`; @@ -1404,7 +1386,6 @@ function x() { assert.strictEqual(imports[0].n, 'mod0'); }); - if (!js) test('single-quotes-and-curly-bracket', () => { const source = ` import { asdf as 'b} from "wrong"' } from 'mod0';`; @@ -1436,7 +1417,6 @@ function x() { assertExportIs(source, exports[6], { n: 'LionCombobox', ln: undefined }); }); - if (!js) test('non-identifier-string as variable (doubleQuote)', () => { const source = ` export { "~123" as foo0 } from './mod0.js'; @@ -1463,7 +1443,6 @@ function x() { assertExportIs(source, exports[8], { n: 'foo8', ln: undefined }); }); - if (!js) test('non-identifier-string as variable (singleQuote)', () => { const source = ` export { '~123' as foo0 } from './mod0.js'; @@ -1490,7 +1469,6 @@ function x() { assertExportIs(source, exports[8], { n: 'foo8', ln: undefined }); }); - if (!js) test('with-backslash-keywords as variable (doubleQuote)', () => { const source = String.raw` export { " slash\\ " as foo0 } from './mod0.js'; @@ -1507,7 +1485,6 @@ function x() { assertExportIs(source, exports[3], { n: 'foo3', ln: undefined }); }); - if (!js) test('with-backslash-keywords as variable (singleQuote)', () => { const source = String.raw` export { ' slash\\ ' as foo0 } from './mod0.js'; @@ -1524,7 +1501,6 @@ function x() { assertExportIs(source, exports[3], { n: 'foo3', ln: undefined }); }); - if (!js) test('with-emoji as', () => { const source = ` export { "hm🤔" as foo0 } from './mod0.js'; @@ -1537,7 +1513,6 @@ function x() { assertExportIs(source, exports[1], { n: 'foo1', ln: undefined }); }); - if (!js) test('non-identifier-string (doubleQuote)', () => { const source = ` export { "~123" } from './mod0.js'; @@ -1564,7 +1539,6 @@ function x() { assertExportIs(source, exports[8], { n: ' notidentifier ', ln: undefined }); }); - if (!js) test('non-identifier-string (singleQuote)', () => { const source = ` export { '~123' } from './mod0.js'; @@ -1591,7 +1565,6 @@ function x() { assertExportIs(source, exports[8], { n: ' notidentifier ', ln: undefined }); }); - if (!js) test('with-backslash-keywords (doubleQuote)', () => { const source = String.raw` export { " slash\\ " } from './mod0.js'; @@ -1608,7 +1581,6 @@ function x() { assertExportIs(source, exports[3], { n: String.raw` quote' `, ln: undefined }); }); - if (!js) test('with-backslash-keywords (singleQuote)', () => { const source = String.raw` export { ' slash\\ ' } from './mod0.js'; @@ -1625,7 +1597,6 @@ function x() { assertExportIs(source, exports[3], { n: String.raw` quote' `, ln: undefined }); }); - if (!js) test('variable as non-identifier-string (doubleQuote)', () => { const source = ` export { foo0 as "~123" } from './mod0.js'; @@ -1652,7 +1623,6 @@ function x() { assertExportIs(source, exports[8], { n: ' notidentifier ', ln: undefined }); }); - if (!js) test('variable as non-identifier-string (singleQuote)', () => { const source = ` export { foo0 as '~123' } from './mod0.js'; @@ -1679,7 +1649,6 @@ function x() { assertExportIs(source, exports[8], { n: ' notidentifier ', ln: undefined }); }); - if (!js) test('variable as with-backslash-keywords (doubleQuote)', () => { const source = String.raw` export { foo0 as " slash\\ " } from './mod0.js'; @@ -1696,7 +1665,6 @@ function x() { assertExportIs(source, exports[3], { n: String.raw` quote' `, ln: undefined }); }); - if (!js) test('variable as with-backslash-keywords (singleQuote)', () => { const source = String.raw` export { foo0 as ' slash\\ ' } from './mod0.js'; @@ -1713,7 +1681,6 @@ function x() { assertExportIs(source, exports[3], { n: String.raw` quote' `, ln: undefined }); }); - if (!js) test('non-identifier-string as non-identifier-string (doubleQuote)', () => { const source = ` export { "~123" as "~123" } from './mod0.js'; @@ -1740,7 +1707,6 @@ function x() { assertExportIs(source, exports[8], { n: ' notidentifier ', ln: undefined }); }); - if (!js) test('non-identifier-string as non-identifier-string (singleQuote)', () => { const source = ` export { '~123' as '~123' } from './mod0.js'; @@ -1767,7 +1733,6 @@ function x() { assertExportIs(source, exports[8], { n: ' notidentifier ', ln: undefined }); }); - if (!js) test('with-backslash-keywords as with-backslash-keywords (doubleQuote)', () => { const source = String.raw` export { " slash\\ " as " slash\\ " } from './mod0.js'; @@ -1784,7 +1749,6 @@ function x() { assertExportIs(source, exports[3], { n: String.raw` quote' `, ln: undefined, a: true}); }); - if (!js) test('with-backslash-keywords as with-backslash-keywords (singleQuote)', () => { const source = String.raw` export { ' slash\\ ' as ' slash\\ ' } from './mod0.js'; @@ -1801,7 +1765,6 @@ function x() { assertExportIs(source, exports[3], { n: String.raw` quote' `, ln: undefined }); }); - if (!js) test('curly-brace (doubleQuote)', () => { const source = ` export { " right-curlybrace} " } from './mod0.js'; @@ -1822,7 +1785,6 @@ function x() { assertExportIs(source, exports[5], { n: ' {curlybrackets} ', ln: undefined }); }); - if (!js) test('* as curly-brace (doubleQuote)', () => { const source = ` export { foo as " right-curlybrace} " } from './mod0.js'; @@ -1843,7 +1805,6 @@ function x() { assertExportIs(source, exports[5], { n: ' {curlybrackets} ', ln: undefined }); }); - if (!js) test('curly-brace as curly-brace (doubleQuote)', () => { const source = ` export { " right-curlybrace} " as " right-curlybrace} " } from './mod0.js'; @@ -1864,7 +1825,6 @@ function x() { assertExportIs(source, exports[5], { n: ' {curlybrackets} ', ln: undefined }); }); - if (!js) test('complex & edge cases', () => { const source = ` export {