Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
32 changes: 28 additions & 4 deletions src/lexer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -122,6 +122,9 @@ export class TokenValue {
// True when `value` is exactly the raw source span [pos, end) — lets consumers
// reuse the string instead of slicing the source again.
raw = false;
// True when a word contains no quoting, escaped characters, or expansions.
// Backslash-newline continuations preserve keyword eligibility.
keywordEligible = false;

constructor(owner: Lexer | null = null) {
this._owner = owner;
Expand All @@ -148,6 +151,7 @@ export class TokenValue {
this.targetEnd = 0;
this.assignmentOperatorPos = -1;
this.raw = false;
this.keywordEligible = false;
}

copyFrom(other: TokenValue): void {
Expand All @@ -162,6 +166,7 @@ export class TokenValue {
this.targetEnd = other.targetEnd;
this.assignmentOperatorPos = other.assignmentOperatorPos;
this.raw = other.raw;
this.keywordEligible = other.keywordEligible;
}
}

Expand Down Expand Up @@ -391,6 +396,7 @@ function setToken(out: TokenValue, token: Token, value: string, pos: number = 0,
out.content = undefined;
out.assignmentOperatorPos = -1;
out.raw = false;
out.keywordEligible = false;
}

// Word token over [pos, end) whose value materializes on first access.
Expand All @@ -404,6 +410,7 @@ function setSpanToken(out: TokenValue, token: Token, pos: number, end: number, r
out.content = undefined;
out.assignmentOperatorPos = -1;
out.raw = raw;
out.keywordEligible = false;
}

// Positional reserved-word match over src[start, start+len) — no slice. Mirrors
Expand Down Expand Up @@ -783,7 +790,8 @@ export class Lexer {
commandStart = true;
} else if (frame.phase === "coproc-body") {
if (token === Token.Newline) continue;
frame.phase = token === Token.Word && value.value === "time" ? "time-command" : "commands";
frame.phase =
token === Token.Word && value.keywordEligible && value.value === "time" ? "time-command" : "commands";
commandStart = true;
if (frame.phase === "time-command") continue;
} else if (frame.phase === "time-command") {
Expand Down Expand Up @@ -903,7 +911,7 @@ export class Lexer {
frame.phase = "coproc-command";
break;
default:
if (token === Token.Word && value.value === "time") {
if (token === Token.Word && value.keywordEligible && value.value === "time") {
frame.phase = "time-command";
commandStart = true;
} else {
Expand Down Expand Up @@ -1683,6 +1691,7 @@ export class Lexer {
private _wordRaw = false;
private _wordQuoted = false;
private _wordHasExpansions = false;
private _wordKeywordEligible = false;
private _wordIsAssignment: boolean | undefined;
private _wordAssignmentOperatorPos: number | undefined;
_wordParts: WordPart[] | null = null;
Expand All @@ -1708,6 +1717,7 @@ export class Lexer {
const raw = this._wordRaw;
const hasExpansions = this._wordHasExpansions;
const quoted = this._wordQuoted;
const keywordEligible = this._wordKeywordEligible;
const isAssignment = this._wordIsAssignment;
let assignmentOpPos = this._wordAssignmentOperatorPos;
const wordEnd = this.pos;
Expand All @@ -1726,7 +1736,7 @@ export class Lexer {
}

if (ctx === LexContext.CommandStart) {
if (!hasExpansions && !quoted) {
if (keywordEligible) {
if (raw) {
if (wordLen <= 8) {
const reserved = matchReservedWord(src, tokenStart, wordLen);
Expand Down Expand Up @@ -1784,7 +1794,7 @@ export class Lexer {
return;
}
}
if (!hasExpansions && !quoted) {
if (keywordEligible) {
if (raw) {
if (
wordLen === 2 &&
Expand Down Expand Up @@ -1845,6 +1855,7 @@ export class Lexer {

setSpanToken(out, Token.Word, tokenStart, wordEnd, raw);
if (value !== null) out._value = value;
out.keywordEligible = keywordEligible;
}

private readWordText(): void {
Expand Down Expand Up @@ -1872,6 +1883,7 @@ export class Lexer {
this._wordRaw = true;
this._wordQuoted = false;
this._wordHasExpansions = false;
this._wordKeywordEligible = true;
this._wordIsAssignment = undefined;
this._wordAssignmentOperatorPos = undefined;
if (this._buildParts) this._wordParts = null;
Expand All @@ -1887,6 +1899,7 @@ export class Lexer {
let text = bt && pos > fastStart ? src.slice(fastStart, pos) : "";
let quoted = false;
let hasExpansions = false;
let keywordEligible = true;
let valueIsRaw = true;
let lastValueChar = pos > fastStart ? src.charCodeAt(pos - 1) : 0;
let assignmentState = scanAssignmentPrefix(src, fastStart, pos, ASSIGNMENT_NAME_START);
Expand Down Expand Up @@ -1922,6 +1935,7 @@ export class Lexer {

if (charType[ch] & 1) {
if (ch === CH_LPAREN && lastValueChar < 128 && extglobPrefix[lastValueChar]) {
keywordEligible = false;
const prefixChar = lastValueChar;
pos++;
const innerStart = pos;
Expand Down Expand Up @@ -1971,6 +1985,7 @@ export class Lexer {
} else {
if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID;
quoted = true;
keywordEligible = false;
valueIsRaw = false;
lastValueChar = src.charCodeAt(pos);
if (bt) {
Expand All @@ -1987,6 +2002,7 @@ export class Lexer {
const sqStart = pos;
if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID;
quoted = true;
keywordEligible = false;
valueIsRaw = false;
pos++;
const start = pos;
Expand All @@ -2011,6 +2027,7 @@ export class Lexer {
const dqStart = pos;
if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID;
quoted = true;
keywordEligible = false;
valueIsRaw = false;
pos++;
this.pos = pos;
Expand Down Expand Up @@ -2038,6 +2055,7 @@ export class Lexer {
}

if (ch === CH_DOLLAR) {
keywordEligible = false;
const dollarStart = pos;
if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID;
this.pos = pos;
Expand All @@ -2063,6 +2081,7 @@ export class Lexer {
}

if (ch === CH_BACKTICK) {
keywordEligible = false;
const btStart = pos;
if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID;
this.pos = pos;
Expand All @@ -2087,6 +2106,7 @@ export class Lexer {
if (assignmentState >= 0 && assignmentState < ASSIGNMENT_INDEX_BASE) assignmentState = ASSIGNMENT_INVALID;
const braceEnd = scanBraceExpansion(src, pos, len);
if (braceEnd > 0) {
keywordEligible = false;
lastValueChar = src.charCodeAt(braceEnd - 1);
if (bt) {
const braceText = src.slice(pos, braceEnd);
Expand Down Expand Up @@ -2128,6 +2148,7 @@ export class Lexer {
this._wordRaw = valueIsRaw;
this._wordQuoted = quoted;
this._wordHasExpansions = hasExpansions;
this._wordKeywordEligible = keywordEligible;
this._wordIsAssignment = isMatchedAssignment(assignmentState);
this._wordAssignmentOperatorPos = this._wordIsAssignment ? assignmentOperatorPos(assignmentState) : undefined;
if (bp) {
Expand Down Expand Up @@ -2288,6 +2309,7 @@ export class Lexer {
this._wordRaw = false;
this._wordQuoted = false;
this._wordHasExpansions = false;
this._wordKeywordEligible = false;
if (bp) {
this._wordParts = parts!.length > 1 || (parts!.length === 1 && parts![0].type !== "Literal") ? parts! : null;
}
Expand All @@ -2309,6 +2331,7 @@ export class Lexer {
const savedText = this._wordText;
const savedParts = this._wordParts;
const savedQuoted = this._wordQuoted;
const savedKeywordEligible = this._wordKeywordEligible;

this.srcEnd = end;
this.pos = start;
Expand All @@ -2326,6 +2349,7 @@ export class Lexer {
this._wordText = savedText;
this._wordParts = savedParts;
this._wordQuoted = savedQuoted;
this._wordKeywordEligible = savedKeywordEligible;
this._nestingDepth--;
return word;
}
Expand Down
24 changes: 20 additions & 4 deletions src/parser.ts
Original file line number Diff line number Diff line change
Expand Up @@ -343,6 +343,24 @@ class Parser {
shebang = nl === -1 ? this.source : this.source.slice(0, nl);
}
const commands = this.list();
for (;;) {
const unexpected = this.tok.peek(LexContext.CommandStart);
if (unexpected.token === Token.EOF) break;

this.error(`unexpected token '${unexpected.value}'`, unexpected.pos);
if (!listTerminators[unexpected.token] && unexpected.token !== Token.In) break;

this.tok.next(LexContext.CommandStart);
let separator = this.tok.peek(LexContext.CommandStart).token;
if (separator !== Token.Semi && separator !== Token.Newline && separator !== Token.Amp) break;

while (separator === Token.Semi || separator === Token.Newline || separator === Token.Amp) {
this.tok.next(LexContext.CommandStart);
separator = this.tok.peek(LexContext.CommandStart).token;
}
const recovered = this.list();
for (let i = 0; i < recovered.length; i++) commands.push(recovered[i]);
}
const lexerErrors = this.tok._errors;
if (lexerErrors !== null && lexerErrors.length > 0) {
const errors = (this.errors ??= []);
Expand Down Expand Up @@ -483,10 +501,8 @@ class Parser {
private pipeline(): Node | null {
let time = false;
let pipelinePos = 0;
if (
this.tok.peek(LexContext.CommandStart).token === Token.Word &&
this.tok.peek(LexContext.CommandStart).value === "time"
) {
const firstToken = this.tok.peek(LexContext.CommandStart);
if (firstToken.token === Token.Word && firstToken.keywordEligible && firstToken.value === "time") {
time = true;
pipelinePos = this.tok.next(LexContext.CommandStart).pos;
if (
Expand Down
2 changes: 2 additions & 0 deletions test/deep-nesting.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -306,6 +306,8 @@ test("compound recovery tracks command prefixes before compound bodies", () => {
"time -p ( echo timed )",
"time -p -- { echo timed; }",
"time coproc worker { echo coproc; }",
"ti\\" + "\nme -p { echo timed; }",
"coproc worker ti\\" + "\nme -p { echo timed; }",
]) {
const ast = parse(`${nestedBraceGroups(2_000, body)}; echo after`);

Expand Down
36 changes: 36 additions & 0 deletions test/error-recovery.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -148,6 +148,42 @@ test("valid compound commands have no errors", () => {
assert.equal(ast.errors, undefined);
});

test("unexpected root list terminators collect errors and recover", () => {
for (const terminator of ["then", "else", "elif", "fi", "do", "done", "in", "esac", ")", "}", ";;", ";&", ";;&"]) {
const source = `safe\n${terminator}; recovered`;
const ast = parse(source);

assert.deepEqual(
ast.errors,
[{ message: `unexpected token '${terminator}'`, pos: source.indexOf(terminator) }],
source,
);
assert.equal(ast.commands.length, 2, source);
assert.equal(ast.commands[0].command.type === "Command" && ast.commands[0].command.name?.value, "safe", source);
assert.equal(
ast.commands[1].command.type === "Command" && ast.commands[1].command.name?.value,
"recovered",
source,
);
}
});

test("valid trailing separators, comments, and whitespace reach EOF", () => {
for (const source of ["safe;", "safe &", "safe\n", "safe; # comment\n\n\t# final comment"]) {
const ast = parse(source);
assert.equal(ast.errors, undefined, source);
assert.equal(ast.commands.length, 1, source);
}
});

test("root recovery appends large statement lists without exceeding the argument limit", () => {
const statementCount = 130_000;
const ast = parse("fi;\n" + "x\n".repeat(statementCount));

assert.deepEqual(ast.errors, [{ message: "unexpected token 'fi'", pos: 0 }]);
assert.equal(ast.commands.length, statementCount);
});

test("trailing command operators collect errors", () => {
for (const operator of ["&&", "||", "|", "|&"]) {
const source = `echo ${operator}`;
Expand Down
8 changes: 8 additions & 0 deletions test/mvdan-sh-compat.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,14 @@ const knownInvalidInputs = new Set([
"foo |&",
"foo \\" + "\n\t|&",
"foo >!a >>|b >>!c &>|d &>!e &>>|f &>>!g",
"function f1 f2 f3() {\n\ta\n}",
"function {\n\ta\n}",
"() {\n\ta\n}",
"$foo[(r)pattern]",
"echo *.txt(@)",
"echo /bin/sh(:t)",
'@test "desc" { body; }',
"@test 'desc' {\n\tmultiple\n\tstatements\n}",
"echo <->",
"echo <5-10>",
]);
Expand Down
41 changes: 41 additions & 0 deletions test/pipelines.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -123,6 +123,47 @@ test("time with negation", () => {
assert.equal(p.negated, true);
});

test("time allows unquoted backslash-newline continuations", () => {
const cases: [string, boolean | undefined, number][] = [
["ti\\" + "\nme echo", undefined, 1],
["time\\" + "\n -p echo", undefined, 1],
["ti\\" + "\nme ! echo", true, 1],
["ti\\" + "\nme echo | cat", undefined, 2],
];
for (const [source, negated, commandCount] of cases) {
const ast = parse(source);
const pipeline = ast.commands[0].command;

assert.equal(pipeline.type, "Pipeline", source);
assert.equal(pipeline.type === "Pipeline" && pipeline.time, true, source);
assert.equal(pipeline.type === "Pipeline" && pipeline.negated, negated, source);
assert.equal(pipeline.type === "Pipeline" && pipeline.commands.length, commandCount, source);
assert.equal(ast.errors, undefined, source);
}
});

test("quoted and escaped time spellings remain command names", () => {
for (const source of [
"'time' echo",
'"time" echo',
"$'time' echo",
'$"time" echo',
"ti$''me echo",
"\\time echo",
"t\\ime echo",
"ti'me' echo",
'ti"me" echo',
]) {
const ast = parse(source);
const command = ast.commands[0].command;

assert.equal(command.type, "Command", source);
assert.equal(command.type === "Command" && command.name?.value, "time", source);
assert.deepEqual(command.type === "Command" && command.suffix.map((word) => word.value), ["echo"], source);
assert.equal(ast.errors, undefined, source);
}
});

test("time alone produces a node", () => {
const ast = parse("time");
assert.equal(ast.commands.length, 1);
Expand Down
11 changes: 11 additions & 0 deletions test/quoting.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -58,6 +58,17 @@ test("backslash-escaped reserved word is not a keyword", () => {
assert.equal(c.name?.text, "\\if");
});

test("dollar-quoted reserved words are not keywords", () => {
for (const source of ["$'if' true", '$"if" true', "i$''f true"]) {
const ast = parse(source);
const c = getCmd(ast);

assert.equal(c.name?.value, "if", source);
assert.equal(c.suffix[0].value, "true", source);
assert.equal(ast.errors, undefined, source);
}
});

test("single quote inside double quotes is literal", () => {
const c = getCmd(parse(`echo "TEST1 'TEST2"`));
assert.equal(c.suffix[0].text, '"TEST1 \'TEST2"');
Expand Down
9 changes: 9 additions & 0 deletions test/test-expressions.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -98,6 +98,15 @@ test("binary =", () => {
assert.equal(binary(t.expression).operator, "=");
});

test("ANSI-C quoted closing brackets remain an operand", () => {
const ast = parse("[[ value == $']]' ]]");
const t = ast.commands[0].command;

assert.equal(t.type, "TestCommand");
assert.equal(t.type === "TestCommand" && binary(t.expression).right.value, "]]");
assert.equal(ast.errors, undefined);
});

test("binary -eq -ne -lt -le -gt -ge", () => {
for (const op of ["-eq", "-ne", "-lt", "-le", "-gt", "-ge"]) {
const t = getTest(`[[ $num ${op} 42 ]]`);
Expand Down
Loading