Files
multica/packages/views/editor/utils/highlight-match.ts
Jiayuan Zhang 0d51614c9c feat(editor): text highlight (==text==) in description & comments [MUL-2934] (#3661)
* feat(editor): support text highlight (==text==) in description & comments

Adds a single-color (yellow) text highlight mark to the shared rich-text
editor, round-tripped through stored Markdown as ==text==.

- HighlightExtension: @tiptap/extension-highlight + @tiptap/markdown hooks
  (markdownTokenizer/parseMarkdown/renderMarkdown) so ==text== <-> <mark>
  round-trips; inner inline formatting preserved via inlineTokens.
- Bubble menu: highlight toggle button (Mod-Shift-H), i18n in 4 locales.
- Read-only renderer: highlightToHtml lowers ==text== -> <mark> (skips code
  and math); rehype-sanitize schema whitelists <mark>. Nested Markdown inside
  a highlight still parses via the existing rehype-raw step.
- prose.css: single yellow <mark> style, legible in light/dark.

Pinned @tiptap/extension-highlight to exact 3.22.1 to match @tiptap/core
(>=3.23 expects a getStyleProperty export core 3.22.1 doesn't have).

Web/desktop only. Mobile (native md4c, no == syntax, no custom renderers)
is tracked as a follow-up. MUL-2934.

Tests: editor round-trip (cross-process serialization protocol), readonly
<mark> rendering + sanitize, and the ==->mark transform incl. code-skip.

Co-authored-by: multica-agent <github@multica.ai>

* fix(editor): align highlight boundary rules across editor & readonly

Addresses two boundary bugs from review (PR #3661):

1. A == inside inline code/math could close a highlight when the opening
   == was outside the literal span (e.g. ==a `b==c` d== wrongly became
   <mark>a `b</mark>c` d==). Both the editor tokenizer's lazy regex and the
   readonly transform only guarded the opening fence, not the closing one.
2. The readonly transform matched across blank lines (==a\n\nb==) while the
   editor lexes those as two literal paragraphs — a storage↔editor↔readonly
   mismatch.

Fix: extract one shared matcher (utils/highlight-match.ts) used by BOTH the
editor tokenizer and the readonly lowering, so the rules can't drift. It skips
fences that fall inside code/math literal ranges (open or close) and caps the
inner span at the first blank line.

Tests: shared-matcher unit tests + both repros covered on the editor
(round-trip/HTML) and readonly (transform + rendered DOM) sides.

Co-authored-by: multica-agent <github@multica.ai>

* fix(editor): handle CRLF in highlight blank-line boundary

BLANK_LINE_RE only matched LF, so a CRLF blank line (==a\r\n\r\nb==) was not
recognized as a block boundary and got highlighted. Widen to \r?\n[ \t]*\r?\n.

Tests: CRLF blank-line (no highlight) + CRLF soft-break (still highlights) on
the matcher, readonly transform, and editor sides.

Co-authored-by: multica-agent <github@multica.ai>

---------

Co-authored-by: Lambda <lambda@multica.ai>
Co-authored-by: multica-agent <github@multica.ai>
2026-06-02 17:24:55 +02:00

93 lines
3.7 KiB
TypeScript

/**
* highlight-match — the single source of truth for what `==text==` highlight
* spans look like. BOTH the editor tokenizer (extensions/highlight.ts) and the
* read-only lowering (utils/highlight-markdown.ts) use this so the storage ↔
* editor ↔ read-only boundary rules can never drift.
*
* Rules for a highlight `==INNER==`:
* - no whitespace directly inside the fences (`==x==` ok, `== x ==` not)
* - INNER is non-empty (`====` stays literal)
* - neither fence may sit inside a code/math literal span — a `==` inside
* inline code/math or a fenced/`$$` block is literal and must not open or
* close a highlight (so ``==a `b==c` d=="`` highlights the whole thing, the
* inner `==` stays code)
* - INNER may not cross a blank line / block boundary (`==a\n\nb==` is two
* literal paragraphs, matching how the editor lexes it)
*/
export interface Range {
start: number;
end: number;
}
/**
* A blank line: a line break, optional spaces/tabs, then another line break.
* Handles both LF and CRLF (Windows / pasted) line endings so the boundary
* rule holds regardless of how the content was stored.
*/
const BLANK_LINE_RE = /\r?\n[ \t]*\r?\n/;
/**
* Spans where `==` must NOT be interpreted as highlight syntax: fenced code,
* inline code, and inline/display math. Mirrors the code-range scan in
* @multica/ui/markdown's linkify so highlight and autolink skip the same
* literal contexts. Earlier (block-level) ranges win so a `==`/backtick inside
* a fenced block is not also counted as inline.
*/
export function findLiteralRanges(text: string): Range[] {
const ranges: Range[] = [];
const add = (re: RegExp) => {
let m: RegExpExecArray | null;
re.lastIndex = 0;
while ((m = re.exec(text)) !== null) {
const start = m.index;
if (ranges.some((r) => start >= r.start && start < r.end)) continue;
ranges.push({ start, end: start + m[0].length });
}
};
add(/```[\s\S]*?```/g); // fenced code
add(/\$\$[\s\S]*?\$\$/g); // display math
add(/(?<!\$)\$(?!\$)[^$\n]+\$(?!\$)/g); // inline math
add(/(?<!`)`(?!`)[^`\n]+`(?!`)/g); // inline code
return ranges;
}
function isInside(pos: number, ranges: Range[]): boolean {
return ranges.some((r) => pos >= r.start && pos < r.end);
}
/**
* Try to match a highlight whose opening `==` begins at `i`. Returns the
* exclusive end offset (just past the closing `==`) and the inner text, or
* null if no valid highlight starts here.
*
* `ranges` may be passed in when the caller already computed literal ranges for
* the whole text (read-only path); otherwise they are computed from `text`.
*/
export function matchHighlightAt(
text: string,
i: number,
ranges?: Range[],
): { end: number; inner: string } | null {
if (text[i] !== "=" || text[i + 1] !== "=") return null;
const innerStart = i + 2;
// Non-empty and no whitespace directly after the opening fence.
if (innerStart >= text.length) return null;
if (/\s/.test(text[innerStart]!)) return null;
const r = ranges ?? findLiteralRanges(text);
if (isInside(i, r)) return null; // opening fence is literal (inside code/math)
// INNER may not cross a blank line: cap the scan at the first one.
const blankRel = text.slice(innerStart).search(BLANK_LINE_RE);
const scanLimit = blankRel === -1 ? text.length : innerStart + blankRel;
for (let j = innerStart + 1; j <= scanLimit && j + 1 < text.length; j++) {
if (text[j] !== "=" || text[j + 1] !== "=") continue;
if (isInside(j, r)) continue; // closing fence is literal — keep scanning
if (/\s/.test(text[j - 1]!)) continue; // no whitespace directly before fence
return { end: j + 2, inner: text.slice(innerStart, j) };
}
return null;
}