v0.2.0 · 1e731ad

Parsing @prose comments

Turns one source file’s text into its file prose, preamble and chunks (spec §3.2): find the @prose comments (each language its own way, §3.1), then slice the code between consecutive comments into chunks. Each chunk keeps its comment’s byte span, so a renderer can lay the file out in source order with the prose where the comment was.

import { parseSync } from "oxc-parser";
import { declaredAfter, declaredIdentifiers, type DeclaredName } from "./names.js";

export type CodeLang = "js" | "css" | "html" | "yaml" | "toml";

export interface ProseChunk {
  /** Content-derived address within the file (spec §3.2): `file`, the heading slug, the
   *  first declared name, or `chunk-N`; unique per file. */
  anchor: string;
  heading: string | null;
  prose: string;
  code: string;
  pending: boolean;
  /** 1-based line the `@prose` comment starts on, for a link into the editor. */
  startLine: number;
  /** Byte span `[startIndex, endIndex)` of the `@prose` comment itself. */
  startIndex: number;
  endIndex: number;
  /** The language of the chunk's code. Not always the comment's: a `.svelte` file's `<style>`
   *  uses `/** *\/` comments, but its code is CSS. */
  codeLang: CodeLang;
}

export interface ProseSection {
  heading: string | null;
  slug: string;
  chunks: ProseChunk[];
}

export interface FileParse {
  fileProse: string | null;
  /** The file prose as a block in its own right (anchor `file`), so it can be laid out in place
   *  like any chunk. Its `code` is the preamble. Null for a file with no block. */
  fileBlock: ProseChunk | null;
  preamble: string;
  sections: ProseSection[];
}

interface RawBlock {
  codeLang: CodeLang;
  body: string;
  startIndex: number;
  endIndex: number;
  startLine: number;
  /** A `@prose` comment below the top level that doesn't start its own line: not a block, left
   *  as an ordinary comment (§3.1). */
  nested?: boolean;
  /** For a block inside a class or function: the first name declared after it in the whole file,
   *  since its own code isn't a program `declaredIdentifiers` can read. */
  inner?: { declared: DeclaredName | null };
  /** For `.svelte` files: the end of this block's own part (script, style or markup), so its
   *  trailing code never bleeds across a part boundary into the next `<script>` or `<style>`. */
  partEnd?: number;
}

const MARKER = "@prose";
const HEADING_RE = /^#{1,6}\s+(.*)$/;

function lineAt(source: string, index: number): number {
  let line = 1;
  for (let i = 0; i < index; i++) {
    if (source[i] === "\n") line++;
  }
  return line;
}

function slugify(text: string): string {
  return (
    text
      .toLowerCase()
      .replaceAll("`", "")
      .replace(/[^a-z0-9]+/g, "-")
      .replace(/^-+|-+$/g, "") || "section"
  );
}

#Recognizing a prose block

A comment counts only if its first line, trimmed, starts with @prose; any other comment (unmarked JSDoc, //, a tool pragma) returns null and stays an ordinary code comment (spec §3.1). Each style strips its own gutter before this runs: JS’s *, YAML’s # , and none for HTML. Blank lines at either edge of the body are dropped, and a CRLF file’s \r stays out of it.

function extractMarkedFromLines(rawLines: string[]): string | null {
  const lines = rawLines.map((line) => line.replace(/\r$/, ""));
  const trimmedFirst = lines[0]!.trim();
  if (!trimmedFirst.startsWith(MARKER)) return null;
  let rest = trimmedFirst.slice(MARKER.length);
  if (rest.startsWith(" ")) rest = rest.slice(1);
  const out: string[] = [];
  if (rest.length > 0) out.push(rest);
  for (let i = 1; i < lines.length; i++) out.push(lines[i]!);
  while (out.length && out[0]!.trim() === "") out.shift();
  while (out.length && out.at(-1)!.trim() === "") out.pop();
  return out.join("\n");
}

/** Whether only indentation precedes `index` on its line. */
function startsLine(source: string, index: number): boolean {
  const lineStart = source.lastIndexOf("\n", index - 1) + 1;
  return /^[ \t]*$/.test(source.slice(lineStart, index));
}

function extractMarkedBlock(inner: string, stripContinuation: boolean): string | null {
  const lines = inner.split("\n");
  const stripped = lines.map((line, i) =>
    i === 0 || !stripContinuation ? line : line.replace(/^[ \t]*\*[ \t]?/, ""),
  );
  return extractMarkedFromLines(stripped);
}

#Scanning JS and TS

The comments come from oxc-parser, so a regex literal, a template string or a comment-shaped string can’t be misread as a comment. A /** comment that starts with the marker counts at the top level, and at any depth when it starts its own line: markz’s class Parser reads method by method. One that shares its line with code, inside a call or an object literal, is nested, and stays part of the code around it (spec §3.1). A file oxc can’t parse at all yields no blocks.

function scanJs(source: string, lang: "js" | "ts"): RawBlock[] {
  let parsed: ReturnType<typeof parseSync>;
  try {
    parsed = parseSync(`file.${lang}`, source);
  } catch {
    return [];
  }
  const statements = parsed.program.body.map((node): [number, number] => [node.start, node.end]);
  const blocks: RawBlock[] = [];
  let after: ((index: number) => DeclaredName | null) | undefined;
  for (const comment of parsed.comments) {
    if (comment.type !== "Block" || !comment.value.startsWith("*")) continue;
    const text = source.slice(comment.start, comment.end);
    const body = extractMarkedBlock(text.slice(3, -2), true);
    if (body === null) continue;
    const inside = statements.some(([start, end]) => comment.start >= start && comment.end <= end);
    const ownLine = startsLine(source, comment.start);
    if (inside && ownLine) after ??= declaredAfter(parsed.program);
    blocks.push({
      body,
      codeLang: "js",
      startIndex: comment.start,
      endIndex: comment.end,
      startLine: lineAt(source, comment.start),
      nested: inside && !ownLine,
      inner: inside && ownLine ? { declared: after!(comment.end) } : undefined,
    });
  }
  return blocks;
}

#Scanning CSS

CSS has no parser here, only a pass that skips strings and comments and counts {}, so a /** @prose *\/ counts at depth 0, and inside a rule when it starts its own line; otherwise it’s nested (spec §3.1).

function scanCss(source: string): RawBlock[] {
  const blocks: RawBlock[] = [];
  let i = 0;
  let depth = 0;
  const n = source.length;
  while (i < n) {
    const c = source[i];
    if (c === '"' || c === "'") {
      i++;
      while (i < n && source[i] !== c && source[i] !== "\n") i += source[i] === "\\" ? 2 : 1;
      i++;
    } else if (c === "/" && source[i + 1] === "*") {
      const close = source.indexOf("*/", i + 2);
      const end = close === -1 ? n : close + 2;
      const text = source.slice(i, end);
      const body = text.startsWith("/**")
        ? extractMarkedBlock(text.slice(3, text.endsWith("*/") ? -2 : undefined), true)
        : null;
      if (body !== null) {
        blocks.push({
          body,
          codeLang: "css",
          startIndex: i,
          endIndex: end,
          startLine: lineAt(source, i),
          nested: depth > 0 && !startsLine(source, i),
        });
      }
      i = end;
    } else {
      if (c === "{") depth++;
      else if (c === "}") depth = Math.max(0, depth - 1);
      i++;
    }
  }
  return blocks;
}

#Scanning YAML and TOML

Both use # line comments with no closing delimiter, so the marker defines the boundary: a block starts at a column-0 # @prose line and runs through the following # lines, up to the next marker line or the first line that isn’t a comment. Two blocks can then sit back to back without the second’s lines joining the first’s body. An indented # comment, inside a nested mapping or table, is an ordinary comment, as braces make one in CSS (spec §3.1).

function scanHashComments(source: string, codeLang: "yaml" | "toml"): RawBlock[] {
  const blocks: RawBlock[] = [];
  const lines = source.split("\n");
  const lineOffsets: number[] = [];
  let offset = 0;
  for (const line of lines) {
    lineOffsets.push(offset);
    offset += line.length + 1;
  }
  const stripHash = (line: string): string => line.replace(/^#[ \t]?/, "");
  const isMarkerLine = (stripped: string): boolean => stripped.trim().startsWith(MARKER);

  let i = 0;
  while (i < lines.length) {
    if (!lines[i]!.startsWith("#") || !isMarkerLine(stripHash(lines[i]!))) {
      i++;
      continue;
    }
    const startLine = i;
    const commentLines = [stripHash(lines[i]!)];
    i++;
    while (i < lines.length && lines[i]!.startsWith("#")) {
      const next = stripHash(lines[i]!);
      if (isMarkerLine(next)) break;
      commentLines.push(next);
      i++;
    }
    const body = extractMarkedFromLines(commentLines);
    if (body !== null) {
      // The block ends right after its last line's content, before the line ending, as a
      // `*/` or `-->` does in the other styles.
      const lastLine = i - 1;
      blocks.push({
        body,
        codeLang,
        startIndex: lineOffsets[startLine]!,
        endIndex: lineOffsets[lastLine]! + lines[lastLine]!.replace(/\r$/, "").length,
        startLine: startLine + 1,
      });
    }
  }
  return blocks;
}

/** Scans HTML source for `<!-- @prose … -->` comments. */
function scanHtml(source: string): RawBlock[] {
  const blocks: RawBlock[] = [];
  const re = /<!--([\s\S]*?)-->/g;
  let match: RegExpExecArray | null;
  while ((match = re.exec(source))) {
    const body = extractMarkedBlock(match[1]!, false);
    if (body !== null) {
      blocks.push({
        body,
        codeLang: "html",
        startIndex: match.index,
        endIndex: match.index + match[0].length,
        startLine: lineAt(source, match.index),
      });
    }
  }
  return blocks;
}

/** Shifts a block found in an extracted sub-range back into the full source's coordinates. */
function shiftBlock(
  block: RawBlock,
  offset: number,
  fullSource: string,
  partEnd: number,
): RawBlock {
  const startIndex = block.startIndex + offset;
  return {
    ...block,
    startIndex,
    endIndex: block.endIndex + offset,
    startLine: lineAt(fullSource, startIndex),
    partEnd,
    inner: block.inner && {
      declared: block.inner.declared && {
        ...block.inner.declared,
        start: block.inner.declared.start + offset,
      },
    },
  };
}

#Scanning .svelte files

Each part follows its own language’s rule (spec §3.1): <script> and <style> bodies as TS and CSS, everything else as HTML, merged back in source order. Each block carries its part’s end, so a chunk’s code is clipped there instead of running past </script> into the next part.

function scanSvelte(source: string): RawBlock[] {
  const blocks: RawBlock[] = [];
  for (const part of svelteParts(source)) {
    const text = source.slice(part.start, part.end);
    const found =
      part.kind === "markup"
        ? scanHtml(text)
        : part.kind === "style"
          ? scanCss(text)
          : scanJs(text, "ts");
    for (const block of found) blocks.push(shiftBlock(block, part.start, source, part.end));
  }
  blocks.sort((a, b) => a.startIndex - b.startIndex);
  return blocks;
}

/** A `.svelte` file's parts in order: each `<script>` and `<style>` body, and the markup between
 *  them (tags included in the markup, which is scanned as HTML). */
function svelteParts(
  source: string,
): { kind: "script" | "style" | "markup"; start: number; end: number }[] {
  const parts: { kind: "script" | "style" | "markup"; start: number; end: number }[] = [];
  const tagRe = /<(script|style)\b[^>]*>([\s\S]*?)<\/\1>/g;
  let last = 0;
  let match: RegExpExecArray | null;
  while ((match = tagRe.exec(source))) {
    const [full, tagName, inner] = match as unknown as [string, string, string];
    const inside = match.index + full.indexOf(inner);
    parts.push({ kind: "markup", start: last, end: match.index });
    parts.push({ kind: tagName as "script" | "style", start: inside, end: inside + inner.length });
    last = match.index + full.length;
  }
  parts.push({ kind: "markup", start: last, end: source.length });
  return parts;
}

#Building chunks from blocks

The first block is the file prose; the code before the next block is the preamble. Every later block starts a chunk that runs to the next block, or to the end of the file. A block whose first line is a Markdown heading also opens a new section, and stays a chunk itself, so a heading with no other prose isn’t lost. A chunk with no code is pending: a plan item (spec §3.2).

export function parseFile(source: string, extension: string): FileParse {
  const found =
    extension === "html"
      ? scanHtml(source)
      : extension === "svelte"
        ? scanSvelte(source)
        : extension === "yaml" || extension === "yml"
          ? scanHashComments(source, "yaml")
          : extension === "toml"
            ? scanHashComments(source, "toml")
            : extension === "css"
              ? scanCss(source)
              : scanJs(source, extension === "ts" ? "ts" : "js");
  const blocks = found.filter((b) => !b.nested);

  if (blocks.length === 0) {
    return { fileProse: null, fileBlock: null, preamble: source.trim(), sections: [] };
  }

  const fileBlock = blocks[0]!;
  const rest = blocks.slice(1);
  const nextStart = rest[0]?.startIndex ?? source.length;
  const preambleEnd =
    fileBlock.partEnd !== undefined ? Math.min(nextStart, fileBlock.partEnd) : nextStart;
  const preamble = source.slice(fileBlock.endIndex, preambleEnd).trim();

  const sections: ProseSection[] = [];
  /** A block inside a class or function names its chunk from the file's AST (`scanJs`), when the
   *  name it found is in the chunk's own code; `null` means it has no name, not "look it up". */
  const innerNames = new Map<ProseChunk, string | null>();
  let currentSection: ProseSection = { heading: null, slug: "top", chunks: [] };
  let hasOpenedSection = false;

  for (let i = 0; i < rest.length; i++) {
    const block = rest[i]!;
    const nextBlockStart = i + 1 < rest.length ? rest[i + 1]!.startIndex : source.length;
    const codeEnd =
      block.partEnd !== undefined ? Math.min(nextBlockStart, block.partEnd) : nextBlockStart;
    const code = source.slice(block.endIndex, codeEnd).trim();

    const headingMatch = block.body.split("\n")[0]?.match(HEADING_RE);
    if (headingMatch) {
      if (hasOpenedSection || currentSection.chunks.length > 0) sections.push(currentSection);
      const heading = headingMatch[1]!.trim();
      currentSection = { heading, slug: slugify(heading), chunks: [] };
      hasOpenedSection = true;
    }

    const chunk: ProseChunk = {
      anchor: "",
      heading: headingMatch ? currentSection.heading : null,
      prose: block.body,
      code,
      pending: code.length === 0,
      startLine: block.startLine,
      startIndex: block.startIndex,
      endIndex: block.endIndex,
      codeLang: block.codeLang,
    };
    currentSection.chunks.push(chunk);
    if (block.inner) {
      const declared = block.inner.declared;
      innerNames.set(chunk, declared && declared.start < codeEnd ? declared.name : null);
    }
  }
  sections.push(currentSection);

  assignAnchors(sections, innerNames);

  return {
    fileProse: fileBlock.body,
    fileBlock: {
      anchor: FILE_ANCHOR,
      heading: null,
      prose: fileBlock.body,
      code: preamble,
      pending: false,
      startLine: fileBlock.startLine,
      startIndex: fileBlock.startIndex,
      endIndex: fileBlock.endIndex,
      codeLang: fileBlock.codeLang,
    },
    preamble,
    sections,
  };
}

/** The anchor of a file's own prose block: `src/store.ts#file`. Reserved: a chunk whose name
 *  would be `file` becomes `file-2`. */
export const FILE_ANCHOR = "file";

#Content-derived anchors (spec §3.2)

In order: the slug of the chunk’s heading, then the first name its code declares (JS and TS only; for a block inside a class, the method or field below it), then chunk-N by position, the only kind that moves when a block is inserted above. The heading comes first because it’s what the reader sees: a ## Table rows block is #table-rows, not whichever helper its code happens to declare first. A repeat within the file takes -2, -3 in source order, and file is taken from the start. Names keep their case, so #addTodo reads as the code does.

function assignAnchors(sections: ProseSection[], innerNames: Map<ProseChunk, string | null>): void {
  const used = new Set([FILE_ANCHOR]);
  const seen = new Map<string, number>();
  let position = 0;
  for (const section of sections) {
    for (const chunk of section.chunks) {
      position++;
      const base = chunk.heading
        ? slugify(chunk.heading)
        : ((innerNames.has(chunk)
            ? innerNames.get(chunk)
            : [...declaredIdentifiers(chunk.code, chunk.codeLang)][0]) ?? `chunk-${position}`);
      let n = seen.get(base) ?? 1;
      let anchor = n === 1 && !used.has(base) ? base : `${base}-${++n}`;
      while (used.has(anchor)) anchor = `${base}-${++n}`;
      seen.set(base, n);
      used.add(anchor);
      chunk.anchor = anchor;
    }
  }
}

/** Returns the first Markdown paragraph of `text`, skipping a leading heading line, for use as a summary. */
export function firstParagraph(text: string): string {
  const withoutTitle = text.replace(/^#{1,6}\s+.*(\n|$)/, "").trim();
  const paragraph = withoutTitle.split(/\n\s*\n/)[0] ?? "";
  return paragraph.trim();
}