diff --git a/src/lexer.ts b/src/lexer.ts index 4418ddf..cfcfef5 100644 --- a/src/lexer.ts +++ b/src/lexer.ts @@ -122,6 +122,9 @@ export class TokenValue { // True when `value` is exactly the raw source span [pos, end) — lets consumers // reuse the string instead of slicing the source again. raw = false; + // True when a word contains no quoting, escaped characters, or expansions. + // Backslash-newline continuations preserve keyword eligibility. + keywordEligible = false; constructor(owner: Lexer | null = null) { this._owner = owner; @@ -148,6 +151,7 @@ export class TokenValue { this.targetEnd = 0; this.assignmentOperatorPos = -1; this.raw = false; + this.keywordEligible = false; } copyFrom(other: TokenValue): void { @@ -162,6 +166,7 @@ export class TokenValue { this.targetEnd = other.targetEnd; this.assignmentOperatorPos = other.assignmentOperatorPos; this.raw = other.raw; + this.keywordEligible = other.keywordEligible; } } @@ -391,6 +396,7 @@ function setToken(out: TokenValue, token: Token, value: string, pos: number = 0, out.content = undefined; out.assignmentOperatorPos = -1; out.raw = false; + out.keywordEligible = false; } // Word token over [pos, end) whose value materializes on first access. @@ -404,6 +410,7 @@ function setSpanToken(out: TokenValue, token: Token, pos: number, end: number, r out.content = undefined; out.assignmentOperatorPos = -1; out.raw = raw; + out.keywordEligible = false; } // Positional reserved-word match over src[start, start+len) — no slice. Mirrors @@ -783,7 +790,8 @@ export class Lexer { commandStart = true; } else if (frame.phase === "coproc-body") { if (token === Token.Newline) continue; - frame.phase = token === Token.Word && value.value === "time" ? "time-command" : "commands"; + frame.phase = + token === Token.Word && value.keywordEligible && value.value === "time" ? "time-command" : "commands"; commandStart = true; if (frame.phase === "time-command") continue; } else if (frame.phase === "time-command") { @@ -903,7 +911,7 @@ export class Lexer { frame.phase = "coproc-command"; break; default: - if (token === Token.Word && value.value === "time") { + if (token === Token.Word && value.keywordEligible && value.value === "time") { frame.phase = "time-command"; commandStart = true; } else { @@ -1683,6 +1691,7 @@ export class Lexer { private _wordRaw = false; private _wordQuoted = false; private _wordHasExpansions = false; + private _wordKeywordEligible = false; private _wordIsAssignment: boolean | undefined; private _wordAssignmentOperatorPos: number | undefined; _wordParts: WordPart[] | null = null; @@ -1708,6 +1717,7 @@ export class Lexer { const raw = this._wordRaw; const hasExpansions = this._wordHasExpansions; const quoted = this._wordQuoted; + const keywordEligible = this._wordKeywordEligible; const isAssignment = this._wordIsAssignment; let assignmentOpPos = this._wordAssignmentOperatorPos; const wordEnd = this.pos; @@ -1726,7 +1736,7 @@ export class Lexer { } if (ctx === LexContext.CommandStart) { - if (!hasExpansions && !quoted) { + if (keywordEligible) { if (raw) { if (wordLen <= 8) { const reserved = matchReservedWord(src, tokenStart, wordLen); @@ -1784,7 +1794,7 @@ export class Lexer { return; } } - if (!hasExpansions && !quoted) { + if (keywordEligible) { if (raw) { if ( wordLen === 2 && @@ -1845,6 +1855,7 @@ export class Lexer { setSpanToken(out, Token.Word, tokenStart, wordEnd, raw); if (value !== null) out._value = value; + out.keywordEligible = keywordEligible; } private readWordText(): void { @@ -1872,6 +1883,7 @@ export class Lexer { this._wordRaw = true; this._wordQuoted = false; this._wordHasExpansions = false; + this._wordKeywordEligible = true; this._wordIsAssignment = undefined; this._wordAssignmentOperatorPos = undefined; if (this._buildParts) this._wordParts = null; @@ -1887,6 +1899,7 @@ export class Lexer { let text = bt && pos > fastStart ? src.slice(fastStart, pos) : ""; let quoted = false; let hasExpansions = false; + let keywordEligible = true; let valueIsRaw = true; let lastValueChar = pos > fastStart ? src.charCodeAt(pos - 1) : 0; let assignmentState = scanAssignmentPrefix(src, fastStart, pos, ASSIGNMENT_NAME_START); @@ -1922,6 +1935,7 @@ export class Lexer { if (charType[ch] & 1) { if (ch === CH_LPAREN && lastValueChar < 128 && extglobPrefix[lastValueChar]) { + keywordEligible = false; const prefixChar = lastValueChar; pos++; const innerStart = pos; @@ -1971,6 +1985,7 @@ export class Lexer { } else { if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID; quoted = true; + keywordEligible = false; valueIsRaw = false; lastValueChar = src.charCodeAt(pos); if (bt) { @@ -1987,6 +2002,7 @@ export class Lexer { const sqStart = pos; if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID; quoted = true; + keywordEligible = false; valueIsRaw = false; pos++; const start = pos; @@ -2011,6 +2027,7 @@ export class Lexer { const dqStart = pos; if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID; quoted = true; + keywordEligible = false; valueIsRaw = false; pos++; this.pos = pos; @@ -2038,6 +2055,7 @@ export class Lexer { } if (ch === CH_DOLLAR) { + keywordEligible = false; const dollarStart = pos; if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID; this.pos = pos; @@ -2063,6 +2081,7 @@ export class Lexer { } if (ch === CH_BACKTICK) { + keywordEligible = false; const btStart = pos; if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID; this.pos = pos; @@ -2087,6 +2106,7 @@ export class Lexer { if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID; const braceEnd = scanBraceExpansion(src, pos, len); if (braceEnd > 0) { + keywordEligible = false; lastValueChar = src.charCodeAt(braceEnd - 1); if (bt) { const braceText = src.slice(pos, braceEnd); @@ -2128,6 +2148,7 @@ export class Lexer { this._wordRaw = valueIsRaw; this._wordQuoted = quoted; this._wordHasExpansions = hasExpansions; + this._wordKeywordEligible = keywordEligible; this._wordIsAssignment = isMatchedAssignment(assignmentState); this._wordAssignmentOperatorPos = this._wordIsAssignment ? assignmentOperatorPos(assignmentState) : undefined; if (bp) { @@ -2288,6 +2309,7 @@ export class Lexer { this._wordRaw = false; this._wordQuoted = false; this._wordHasExpansions = false; + this._wordKeywordEligible = false; if (bp) { this._wordParts = parts!.length > 1 || (parts!.length === 1 && parts![0].type !== "Literal") ? parts! : null; } @@ -2309,6 +2331,7 @@ export class Lexer { const savedText = this._wordText; const savedParts = this._wordParts; const savedQuoted = this._wordQuoted; + const savedKeywordEligible = this._wordKeywordEligible; this.srcEnd = end; this.pos = start; @@ -2326,6 +2349,7 @@ export class Lexer { this._wordText = savedText; this._wordParts = savedParts; this._wordQuoted = savedQuoted; + this._wordKeywordEligible = savedKeywordEligible; this._nestingDepth--; return word; } diff --git a/src/parser.ts b/src/parser.ts index 9ab3ba8..4b6db9f 100644 --- a/src/parser.ts +++ b/src/parser.ts @@ -343,6 +343,24 @@ class Parser { shebang = nl === -1 ? this.source : this.source.slice(0, nl); } const commands = this.list(); + for (;;) { + const unexpected = this.tok.peek(LexContext.CommandStart); + if (unexpected.token === Token.EOF) break; + + this.error(`unexpected token '${unexpected.value}'`, unexpected.pos); + if (!listTerminators[unexpected.token] && unexpected.token !== Token.In) break; + + this.tok.next(LexContext.CommandStart); + let separator = this.tok.peek(LexContext.CommandStart).token; + if (separator !== Token.Semi && separator !== Token.Newline && separator !== Token.Amp) break; + + while (separator === Token.Semi || separator === Token.Newline || separator === Token.Amp) { + this.tok.next(LexContext.CommandStart); + separator = this.tok.peek(LexContext.CommandStart).token; + } + const recovered = this.list(); + for (let i = 0; i < recovered.length; i++) commands.push(recovered[i]); + } const lexerErrors = this.tok._errors; if (lexerErrors !== null && lexerErrors.length > 0) { const errors = (this.errors ??= []); @@ -483,10 +501,8 @@ class Parser { private pipeline(): Node | null { let time = false; let pipelinePos = 0; - if ( - this.tok.peek(LexContext.CommandStart).token === Token.Word && - this.tok.peek(LexContext.CommandStart).value === "time" - ) { + const firstToken = this.tok.peek(LexContext.CommandStart); + if (firstToken.token === Token.Word && firstToken.keywordEligible && firstToken.value === "time") { time = true; pipelinePos = this.tok.next(LexContext.CommandStart).pos; if ( diff --git a/test/deep-nesting.test.ts b/test/deep-nesting.test.ts index cb787e6..1784c73 100644 --- a/test/deep-nesting.test.ts +++ b/test/deep-nesting.test.ts @@ -306,6 +306,8 @@ test("compound recovery tracks command prefixes before compound bodies", () => { "time -p ( echo timed )", "time -p -- { echo timed; }", "time coproc worker { echo coproc; }", + "ti\\" + "\nme -p { echo timed; }", + "coproc worker ti\\" + "\nme -p { echo timed; }", ]) { const ast = parse(`${nestedBraceGroups(2_000, body)}; echo after`); diff --git a/test/error-recovery.test.ts b/test/error-recovery.test.ts index 60965d8..4c7094a 100644 --- a/test/error-recovery.test.ts +++ b/test/error-recovery.test.ts @@ -148,6 +148,42 @@ test("valid compound commands have no errors", () => { assert.equal(ast.errors, undefined); }); +test("unexpected root list terminators collect errors and recover", () => { + for (const terminator of ["then", "else", "elif", "fi", "do", "done", "in", "esac", ")", "}", ";;", ";&", ";;&"]) { + const source = `safe\n${terminator}; recovered`; + const ast = parse(source); + + assert.deepEqual( + ast.errors, + [{ message: `unexpected token '${terminator}'`, pos: source.indexOf(terminator) }], + source, + ); + assert.equal(ast.commands.length, 2, source); + assert.equal(ast.commands[0].command.type === "Command" && ast.commands[0].command.name?.value, "safe", source); + assert.equal( + ast.commands[1].command.type === "Command" && ast.commands[1].command.name?.value, + "recovered", + source, + ); + } +}); + +test("valid trailing separators, comments, and whitespace reach EOF", () => { + for (const source of ["safe;", "safe &", "safe\n", "safe; # comment\n\n\t# final comment"]) { + const ast = parse(source); + assert.equal(ast.errors, undefined, source); + assert.equal(ast.commands.length, 1, source); + } +}); + +test("root recovery appends large statement lists without exceeding the argument limit", () => { + const statementCount = 130_000; + const ast = parse("fi;\n" + "x\n".repeat(statementCount)); + + assert.deepEqual(ast.errors, [{ message: "unexpected token 'fi'", pos: 0 }]); + assert.equal(ast.commands.length, statementCount); +}); + test("trailing command operators collect errors", () => { for (const operator of ["&&", "||", "|", "|&"]) { const source = `echo ${operator}`; diff --git a/test/mvdan-sh-compat.test.ts b/test/mvdan-sh-compat.test.ts index c48691a..37c21e5 100644 --- a/test/mvdan-sh-compat.test.ts +++ b/test/mvdan-sh-compat.test.ts @@ -12,6 +12,14 @@ const knownInvalidInputs = new Set([ "foo |&", "foo \\" + "\n\t|&", "foo >!a >>|b >>!c &>|d &>!e &>>|f &>>!g", + "function f1 f2 f3() {\n\ta\n}", + "function {\n\ta\n}", + "() {\n\ta\n}", + "$foo[(r)pattern]", + "echo *.txt(@)", + "echo /bin/sh(:t)", + '@test "desc" { body; }', + "@test 'desc' {\n\tmultiple\n\tstatements\n}", "echo <->", "echo <5-10>", ]); diff --git a/test/pipelines.test.ts b/test/pipelines.test.ts index 31c7e03..9378af4 100644 --- a/test/pipelines.test.ts +++ b/test/pipelines.test.ts @@ -123,6 +123,47 @@ test("time with negation", () => { assert.equal(p.negated, true); }); +test("time allows unquoted backslash-newline continuations", () => { + const cases: [string, boolean | undefined, number][] = [ + ["ti\\" + "\nme echo", undefined, 1], + ["time\\" + "\n -p echo", undefined, 1], + ["ti\\" + "\nme ! echo", true, 1], + ["ti\\" + "\nme echo | cat", undefined, 2], + ]; + for (const [source, negated, commandCount] of cases) { + const ast = parse(source); + const pipeline = ast.commands[0].command; + + assert.equal(pipeline.type, "Pipeline", source); + assert.equal(pipeline.type === "Pipeline" && pipeline.time, true, source); + assert.equal(pipeline.type === "Pipeline" && pipeline.negated, negated, source); + assert.equal(pipeline.type === "Pipeline" && pipeline.commands.length, commandCount, source); + assert.equal(ast.errors, undefined, source); + } +}); + +test("quoted and escaped time spellings remain command names", () => { + for (const source of [ + "'time' echo", + '"time" echo', + "$'time' echo", + '$"time" echo', + "ti$''me echo", + "\\time echo", + "t\\ime echo", + "ti'me' echo", + 'ti"me" echo', + ]) { + const ast = parse(source); + const command = ast.commands[0].command; + + assert.equal(command.type, "Command", source); + assert.equal(command.type === "Command" && command.name?.value, "time", source); + assert.deepEqual(command.type === "Command" && command.suffix.map((word) => word.value), ["echo"], source); + assert.equal(ast.errors, undefined, source); + } +}); + test("time alone produces a node", () => { const ast = parse("time"); assert.equal(ast.commands.length, 1); diff --git a/test/quoting.test.ts b/test/quoting.test.ts index 0a05a7d..93d4113 100644 --- a/test/quoting.test.ts +++ b/test/quoting.test.ts @@ -58,6 +58,17 @@ test("backslash-escaped reserved word is not a keyword", () => { assert.equal(c.name?.text, "\\if"); }); +test("dollar-quoted reserved words are not keywords", () => { + for (const source of ["$'if' true", '$"if" true', "i$''f true"]) { + const ast = parse(source); + const c = getCmd(ast); + + assert.equal(c.name?.value, "if", source); + assert.equal(c.suffix[0].value, "true", source); + assert.equal(ast.errors, undefined, source); + } +}); + test("single quote inside double quotes is literal", () => { const c = getCmd(parse(`echo "TEST1 'TEST2"`)); assert.equal(c.suffix[0].text, '"TEST1 \'TEST2"'); diff --git a/test/test-expressions.test.ts b/test/test-expressions.test.ts index d99bd79..a626771 100644 --- a/test/test-expressions.test.ts +++ b/test/test-expressions.test.ts @@ -98,6 +98,15 @@ test("binary =", () => { assert.equal(binary(t.expression).operator, "="); }); +test("ANSI-C quoted closing brackets remain an operand", () => { + const ast = parse("[[ value == $']]' ]]"); + const t = ast.commands[0].command; + + assert.equal(t.type, "TestCommand"); + assert.equal(t.type === "TestCommand" && binary(t.expression).right.value, "]]"); + assert.equal(ast.errors, undefined); +}); + test("binary -eq -ne -lt -le -gt -ge", () => { for (const op of ["-eq", "-ne", "-lt", "-le", "-gt", "-ge"]) { const t = getTest(`[[ $num ${op} 42 ]]`);