diff --git a/README.md b/README.md index afe8185..d8b9fb0 100644 --- a/README.md +++ b/README.md @@ -153,6 +153,19 @@ root: check `errors` on every nested `script` while traversing. A consumer that only reads the root `errors` array cannot tell that a substitution body failed to parse. +Comments follow the same rule. A script's `comments` lists every comment the +parser skipped in it, in source order, as `[pos, end)` spans from the `#` up to +the newline; it is absent when there are none, and a shebang line is `shebang` +rather than a comment: + +```js +const src = "echo one # trailing\n# alone\n"; +const script = parse(src); + +script.comments.map(({ pos, end }) => src.slice(pos, end)); +// ["# trailing", "# alone"] +``` + ### Print Basic opinionated printer, does not preserve whitespace or comments (except diff --git a/src/lexer.ts b/src/lexer.ts index 6f8d404..6dce67d 100644 --- a/src/lexer.ts +++ b/src/lexer.ts @@ -5,6 +5,7 @@ import type { ExtGlobOperator, ParameterExpansionPart, ParseError, + Comment, Word, WordPart, } from "./types.ts"; @@ -545,6 +546,7 @@ export class Lexer { private pendingHereDocs: PendingHereDoc[] | null; private collectedExpansions: [DeferredCommandExpansion, number][] | null; _errors: ParseError[] | null = null; + _comments: Comment[] | null = null; _buildParts = false; // Build processed text while scanning. Off on the normal token path (values // materialize lazily); on for redirect targets, heredoc delimiters, arithmetic @@ -1375,8 +1377,9 @@ export class Lexer { const ch = src.charCodeAt(pos); if (ch === CH_HASH) { - // Skip comment + // Skip comment, recording its span for consumers that want it while (this.pos < len && src.charCodeAt(this.pos) !== CH_NL) this.pos++; + (this._comments ??= []).push({ pos: tokenStart, end: this.pos }); this.readNext(out, ctx); return; } diff --git a/src/parser.ts b/src/parser.ts index 1a4d76b..4d5a2b0 100644 --- a/src/parser.ts +++ b/src/parser.ts @@ -20,6 +20,7 @@ import type { LogicalOperator, Node, ParseError, + Comment, ParsedScript, PipeOperator, Pipeline, @@ -366,6 +367,8 @@ class Parser { private end: number; private depth: number; private errors: ParseError[] | null = null; + // Comments from sub-lexers (array bodies); the main lexer's own are merged in run(). + private comments: Comment[] | null = null; private _redirects: Redirect[] = EMPTY_REDIRECTS; private syntaxDepth = 0; @@ -417,6 +420,14 @@ class Parser { for (let i = 0; i < lexerErrors.length; i++) errors.push(lexerErrors[i]); } if (this.errors !== null && this.errors.length > 1) this.errors.sort((a, b) => a.pos - b.pos); + const lexerComments = this.tok._comments; + if (lexerComments !== null && lexerComments.length > 0) { + const comments = (this.comments ??= []); + for (let i = 0; i < lexerComments.length; i++) comments.push(lexerComments[i]); + } + // An array body's comments are lexed when its assignment is parsed, after the main + // lexer may already have read past it, so source order is restored here. + if (this.comments !== null && this.comments.length > 1) this.comments.sort((a, b) => a.pos - b.pos); const result = { type: "Script", pos: start, @@ -424,6 +435,7 @@ class Parser { shebang, commands, errors: this.errors ?? undefined, + comments: this.comments ?? undefined, } as ParsedScript; return result; } @@ -1571,6 +1583,11 @@ class Parser { elements.push(new WordImpl(text, t.pos, t.end, this.source, undefined, this.depth)); } } + const subComments = subTok._comments; + if (subComments !== null) { + const comments = (this.comments ??= []); + for (let i = 0; i < subComments.length; i++) comments.push(subComments[i]); + } return elements; } diff --git a/src/types.ts b/src/types.ts index 4ad1be1..0dafb54 100644 --- a/src/types.ts +++ b/src/types.ts @@ -449,6 +449,12 @@ export interface ParsedScript extends Script { */ readonly source?: string; errors?: ParseError[]; + /** + * Every comment the parser skipped in this script, in source order, as a `[pos, end)` span + * from the `#` up to (not including) the newline. Absent when there are none. A shebang is + * `shebang`, not a comment. Comments inside a lazily parsed script are on that script. + */ + comments?: Comment[]; } export interface ParseError { @@ -456,4 +462,9 @@ export interface ParseError { pos: number; } +export interface Comment { + pos: number; + end: number; +} + export type DeferredCommandExpansion = CommandExpansionPart | ProcessSubstitutionPart | ArithmeticCommandExpansion; diff --git a/test/comments.test.ts b/test/comments.test.ts new file mode 100644 index 0000000..753ee9c --- /dev/null +++ b/test/comments.test.ts @@ -0,0 +1,59 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { parse } from "../src/parser.ts"; +import type { Command, ParsedScript } from "../src/types.ts"; + +const spans = (src: string, script: ParsedScript = parse(src)) => + (script.comments ?? []).map(({ pos, end }) => src.slice(pos, end)); + +test("comments are reported as source spans in order", () => { + const src = "# first\necho one # trailing\n\n\t# indented\necho two"; + assert.deepEqual(spans(src), ["# first", "# trailing", "# indented"]); +}); + +test("a script without comments has no comments field", () => { + assert.equal(parse("echo one\necho two\n").comments, undefined); +}); + +test("a comment span ends before the newline, or at end of input", () => { + const src = "echo a # no newline"; + const [comment] = parse(src).comments!; + assert.equal(comment.pos, 7); + assert.equal(comment.end, src.length); + assert.deepEqual(spans("echo a # crlf\r\necho b"), ["# crlf\r"]); +}); + +test("a shebang is not a comment", () => { + const src = "#!/bin/bash\n# real\necho x"; + const script = parse(src); + assert.equal(script.shebang, "#!/bin/bash"); + assert.deepEqual(spans(src, script), ["# real"]); +}); + +test("hash inside a word, quotes, or a heredoc body is not a comment", () => { + const src = "echo a#b \"# not\" '# not' $'# not'\ncat < { + const src = "if true; then # then\n\t:\nfi # fi\nx=(\n a # a\n b\n)\n{ # brace\n:; }\n"; + assert.deepEqual(spans(src), ["# then", "# fi", "# a", "# brace"]); + assert.equal(parse(src).errors, undefined); +}); + +test("a comment inside a substitution is on the nested script", () => { + const src = "echo $(\n# inner\nid\n) # outer\n"; + const script = parse(src); + assert.deepEqual(spans(src, script), ["# outer"]); + const cmd = script.commands[0].command as Command; + const part = cmd.suffix[0].parts!.find((p) => p.type === "CommandExpansion")!; + assert.equal(part.type, "CommandExpansion"); + assert.deepEqual(spans(src, part.script!), ["# inner"]); +}); + +test("comments survive error recovery", () => { + const src = "# before\nfi\n# after\necho x\n"; + const script = parse(src); + assert.ok(script.errors && script.errors.length > 0); + assert.deepEqual(spans(src, script), ["# before", "# after"]); +});