parseInline function

List<MdNode> parseInline(
  1. String text,
  2. bool useDollar, [
  3. MarkdownDefinitions defs = MarkdownDefinitions.empty
])

Implementation

List<MdNode> parseInline(
  String text,
  bool useDollar, [
  MarkdownDefinitions defs = MarkdownDefinitions.empty,
]) {
  final n = text.length;

  // Most runs of an assistant's prose contain no markup at all. Finding that
  // out costs one scan, and skips the buffer, the delimiter tables and the
  // dispatch loop entirely.
  var plainUntil = 0;
  while (plainUntil < n) {
    final unit = text.codeUnitAt(plainUntil);
    if (_canStartConstruct(unit) &&
        (unit != _underscore || !_intrawordUnderscore(text, plainUntil))) {
      break;
    }
    plainUntil += 1;
  }
  if (plainUntil == n) {
    return n == 0 ? <MdNode>[] : <MdNode>[MdText(text: text)];
  }

  final delims = _Delims(text);
  final closers = _Closers(text);
  final dollars = useDollar ? DollarMathCloser(text) : null;
  final urls = _BareUrls(text);
  final nodes = <MdNode>[];
  final buf = StringBuffer();
  var i = 0;

  void flush() {
    if (buf.isNotEmpty) {
      nodes.add(MdText(text: buf.toString()));
      buf.clear();
    }
  }

  while (i < n) {
    final c = text.codeUnitAt(i);

    // Fast path: a run of characters that cannot begin a construct is copied
    // through in one piece. This is the bulk of ordinary prose, and skipping
    // the dispatch chain for it is what keeps the parser's cost close to a
    // scan.
    if (!_canStartConstruct(c) ||
        (c == _underscore && _intrawordUnderscore(text, i))) {
      var j = i + 1;
      while (j < n) {
        final unit = text.codeUnitAt(j);
        if (_canStartConstruct(unit) &&
            (unit != _underscore || !_intrawordUnderscore(text, j))) {
          break;
        }
        j += 1;
      }
      buf.write(text.substring(i, j));
      i = j;
      continue;
    }

    var matched = false;

    // ![alt](url)  or  ![alt][label]
    if (c == _bang &&
        i + 1 < n &&
        text.codeUnitAt(i + 1) == _openBracket &&
        delims.bracket.containsKey(i + 1)) {
      final r =
          _tryImage(text, i, delims) ??
          _tryReferenceImage(text, i, delims, defs);
      if (r != null) {
        flush();
        nodes.add(r.node);
        i = r.next;
        matched = true;
      }
    }

    // [text](url), [text][label], [label], [^note]  or  [123] source tag
    if (!matched && c == _openBracket) {
      final link =
          _tryLink(text, i, useDollar, delims, defs) ??
          _tryReferenceLink(text, i, useDollar, delims, defs) ??
          _tryFootnoteReference(text, i, delims, defs) ??
          _trySourceTag(text, i, defs);
      if (link != null) {
        flush();
        nodes.add(link.node);
        i = link.next;
        matched = true;
      }
    }

    // **bold**, *italic*, __bold__, _italic_
    if (!matched && (c == _star || c == _underscore)) {
      // Emphasis is decided by the length of the *run* of delimiters, not by
      // the first one. Reading `***both***` as `**` starting at the second
      // asterisk left a stray `*` inside the bold, and closing a single `*`
      // with `indexOf('*')` landed on the opening half of a nested `**`, which
      // dropped the bold and cut the italic into pieces.
      var run = 1;
      while (i + run < n && text.codeUnitAt(i + run) == c) {
        run += 1;
      }

      if (_canOpen(text, i, run, c, urls)) {
        // `***x***` is both.
        if (run >= 3) {
          final close = closers.find(i + run, c, 3);
          if (close != -1) {
            final inner = text.substring(i + 3, close);
            if (inner.trim().isNotEmpty) {
              flush();
              nodes.add(
                MdBold(
                  children: [
                    MdItalic(children: parseInline(inner, useDollar, defs)),
                  ],
                ),
              );
              i = close + 3;
              matched = true;
            }
          }
        }
        if (!matched && run == 2) {
          final close = closers.find(i + 2, c, 2);
          if (close != -1) {
            final inner = text.substring(i + 2, close);
            if (inner.trim().isNotEmpty) {
              flush();
              nodes.add(MdBold(children: parseInline(inner, useDollar, defs)));
              i = close + 2;
              matched = true;
            }
          }
        }
        if (!matched && run == 1) {
          // Only a lone delimiter closes an italic; a `**` inside it opens a
          // bold, which the recursive parse below then claims.
          final close = closers.find(i + 1, c, 1, exact: true);
          if (close != -1) {
            final inner = text.substring(i + 1, close);
            if (inner.trim().isNotEmpty) {
              flush();
              nodes.add(
                MdItalic(children: parseInline(inner, useDollar, defs)),
              );
              i = close + 1;
              matched = true;
            }
          }
        }
      }
    }

    // ~~strike~~
    if (!matched &&
        c == _tilde &&
        i + 1 < n &&
        text.codeUnitAt(i + 1) == _tilde) {
      final end = text.indexOf('~~', i + 2);
      if (end != -1) {
        flush();
        nodes.add(
          MdStrike(
            children: parseInline(text.substring(i + 2, end), useDollar, defs),
          ),
        );
        i = end + 2;
        matched = true;
      }
    }

    // `code`, ``co`de`` — a run of N backticks closes at the next run of
    // exactly N. An opening run with no closer is literal, all of it.
    if (!matched && c == _backtick) {
      final span = codeSpanAt(text, i);
      if (span != null) {
        flush();
        nodes.add(
          MdInlineCode(
            text: codeSpanContent(text.substring(i + span.run, span.close)),
          ),
        );
        i = span.close + span.run;
      } else {
        final run = backtickRunAt(text, i);
        buf.write(text.substring(i, i + run));
        i += run;
      }
      matched = true;
    }

    // <u>underline</u>
    if (!matched && c == _lt && text.startsWith('<u>', i)) {
      final end = text.indexOf('</u>', i + 3);
      if (end != -1) {
        flush();
        nodes.add(
          MdUnderline(
            children: parseInline(text.substring(i + 3, end), useDollar, defs),
          ),
        );
        i = end + 4;
        matched = true;
      }
    }

    // <!-- comment --> — dropped, as an HTML renderer would hide it.
    if (!matched && c == _lt && text.startsWith('<!--', i)) {
      final end = text.indexOf('-->', i + 4);
      if (end != -1) {
        i = end + 3;
        matched = true;
      }
    }

    // \[ block latex \] in an inline position.
    //
    // The block parser claims `\[` only when it opens a line, so block maths
    // written mid-sentence — or after a list marker, `1. Result: \[ x^2 \]` —
    // used to survive as literal text. The syntax is recognised wherever it
    // appears; it still renders as a block, because that is what it is.
    if (!matched &&
        c == _backslash &&
        i + 1 < n &&
        text.codeUnitAt(i + 1) == _openBracket) {
      final end = text.indexOf('\\]', i + 2);
      if (end != -1) {
        flush();
        nodes.add(MdBlockLatex(tex: text.substring(i + 2, end).trim()));
        i = end + 2;
        matched = true;
      }
    }

    // \( inline latex \)
    if (!matched &&
        c == _backslash &&
        i + 1 < n &&
        text.codeUnitAt(i + 1) == _openParen) {
      final end = text.indexOf('\\)', i + 2);
      if (end != -1) {
        flush();
        nodes.add(MdInlineLatex(tex: text.substring(i + 2, end).trim()));
        i = end + 2;
        matched = true;
      }
    }

    // Backslash escapes. Any ASCII punctuation after a backslash is literal —
    // `\*`, `\_`, `\#`, `\$`, and `\|`, the GFM escape a table cell uses for a
    // pipe (cells are split before this runs; see _splitPipes in
    // block_parser.dart). A backslash before a line break is a hard break.
    //
    // Not `\(`, `\)`, `\[` or `\]`: in this dialect those are maths
    // delimiters, matched above when they pair up. An unpaired one stays
    // exactly as written, which is also what the streaming reveal expects of
    // a formula still arriving.
    if (!matched && c == _backslash && i + 1 < n) {
      final next = text.codeUnitAt(i + 1);
      if (next == _newline) {
        buf.writeCharCode(_newline);
        i += 2;
        matched = true;
      } else if (isAsciiPunctuation(next) &&
          next != _openParen &&
          next != _closeParen &&
          next != _openBracket &&
          next != _closeBracket) {
        buf.writeCharCode(next);
        i += 2;
        matched = true;
      }
    }

    // $$ … $$  /  $ … $  (only when enabled)
    if (!matched && useDollar && c == _dollar) {
      if (i + 1 < n && text.codeUnitAt(i + 1) == _dollar) {
        final end = text.indexOf(r'$$', i + 2);
        if (end != -1) {
          flush();
          nodes.add(MdInlineLatex(tex: text.substring(i + 2, end).trim()));
          i = end + 2;
          matched = true;
        }
      }
      if (!matched) {
        final end = dollars!.find(i);
        if (end != -1) {
          flush();
          nodes.add(MdInlineLatex(tex: text.substring(i + 1, end).trim()));
          i = end + 1;
          matched = true;
        }
      }
    }

    // &amp;  &#169;  &#x1F600; — but not inside a bare URL. The autolinker
    // reads the raw text there, and GFM has it leave a trailing `&amp;` out
    // of the link, which it can only do if the reference is still written.
    if (!matched && c == _amp && !urls.contains(i)) {
      final entity = entityAt(text, i);
      if (entity != null) {
        buf.write(entity.value);
        i = entity.next;
        matched = true;
      }
    }

    if (!matched) {
      buf.writeCharCode(c);
      i += 1;
    }
  }

  flush();
  return nodes;
}