markz

Walking

The one traversal consumers need, since every transformation of a markz document is a fold (design: Read-only, and transformations are folds). walk visits a subtree depth-first, calling enter before a node's children and exit after them. It follows the parent and sibling columns rather than recursing, so a deeply nested document can't overflow the stack, and it allocates nothing. enter returning false skips that node's children; its exit still runs.

import { NONE, type Document, type NodeId } from "./ast";

/** What `walk` calls for each node. `enter` returning `false` skips the node's children. */
export interface Visitor {
  enter?(node: NodeId): boolean | void;
  exit?(node: NodeId): void;
}

/** Visits a subtree depth-first, calling `enter` before a node's children and `exit` after. */
export function walk(doc: Document, visitor: Visitor, from: NodeId = doc.root): void {
  let node = from;
  for (;;) {
    const child = visitor.enter?.(node) === false ? NONE : doc.firstChild(node);
    if (child !== NONE) {
      node = child;
      continue;
    }
    for (;;) {
      visitor.exit?.(node);
      if (node === from) return;
      const next = doc.nextSibling(node);
      if (next !== NONE) {
        node = next;
        break;
      }
      node = doc.parent(node);
    }
  }
}

#Text content

The text html() writes for a node, as a browser's textContent would read it back: escapes decoded, code, math and expressions as written, a hard break as a line ending. Images, comments, metadata and raw blocks add nothing. Blocks are joined with nothing between them, as in the DOM.

export function textContent(doc: Document, node: NodeId = doc.root): string {
  let out = "";
  walk(
    doc,
    {
      enter(n) {
        switch (doc.type(n)) {
          case "text":
            out += doc.data(n, "text").value;
            break;
          case "inlineCode":
            out += doc.data(n, "inlineCode").value;
            break;
          case "code":
            out += doc.data(n, "code").value;
            break;
          case "math":
            out += doc.data(n, "math").value;
            break;
          case "expression":
            out += doc.source.slice(doc.start(n), doc.end(n));
            break;
          case "break":
            out += "\n";
            break;
          case "image":
            return false;
        }
        return true;
      },
    },
    node,
  );
  return out;
}

/** One heading in a document's outline, with the id `html()` gives it. */
export interface Heading {
  node: NodeId;
  depth: 1 | 2 | 3 | 4 | 5 | 6;
  id: string;
  text: string;
}

#Headings

The outline of a document, for a table of contents or a "jump to" list: every heading in source order, wherever it sits (a heading in a blockquote, list item or element counts), with the id html() writes for it, so a link to #id lands on it. It is a flat list, not a tree: depth is data, and how to nest, number or filter by it is the consumer's choice, which is why it isn't part of html(). node gives the source range through doc.start and doc.end.

export function headings(doc: Document): Heading[] {
  const out: Heading[] = [];
  walk(doc, {
    enter(node) {
      const type = doc.type(node);
      if (type === "heading") {
        const { depth, id } = doc.data(node, "heading");
        out.push({ node, depth, id, text: textContent(doc, node) });
        return false;
      }
      // Headings are blocks, so nothing inside a paragraph, a table or code holds one.
      return type !== "paragraph" && type !== "table" && type !== "code";
    },
  });
  return out;
}