@tanstack/highlight
Version:
Tiny class-based syntax highlighting for documentation.
103 lines (102 loc) • 4.96 kB
JavaScript
import { defineLanguage } from '../core.js';
import { collectPatternRanges } from '../internal/patterns.js';
const patterns = [
{ className: 'attr', regex: /(?<![\w@])@(?:[a-z]+:)?[A-Za-z_][\w.]*/g },
{
className: 'keyword',
regex: /(?<!\.)(?:\bas\?|!(?:in|is)\b|\b(?:abstract|as|break|by|catch|class|companion|const|constructor|continue|crossinline|do|else|external|final|finally|for|fun|if|import|in|infix|init|inline|interface|internal|is|lateinit|noinline|object|operator|override|package|private|protected|public|reified|return|sealed|super|suspend|tailrec|this|throw|try|typealias|val|var|vararg|when|where|while)\b)/g,
},
{ className: 'keyword', regex: /(?<!\.)\b(?:data|value|enum|annotation|inner|open|expect|actual)(?=[ \t]+(?:[a-z]+[ \t]+){0,3}(?:class|fun|val|var|interface|object|constructor)\b)|(?<![\w.])(?:get(?=\(\)\s*[:={])|set(?=\(\w+\)\s*[:={]))|(?<=\b(?:private|protected|internal)[ \t]+)set\b|\bout(?=\s+[A-Z])/g },
{ className: 'literal', regex: /\b(?:true|false|null)\b/g },
{ className: 'type', regex: /\b(?:class|interface|object|typealias)\s+(\w+)/g, group: 1 },
{ className: 'function', regex: /\b[A-Za-z_]\w*(?=\s*\()|\b[a-z_]\w*(?=(?:<[\w.?*, ]*(?:<[\w.?*, ]*>[\w.?*, ]*)?>)?[ \t]*[({])|(?<=^[ \t]*|[{=][ \t]*)[A-Z][A-Z\d_]*[a-z]\w*(?=[ \t]*\{)/gm },
{ className: 'type', regex: /\b[A-Z][A-Z\d_]*[a-z]\w*/g },
{ className: 'number', regex: /(?<!\w|[^.]\.)(?:0[xX][\da-fA-F_]+|0[bB][01_]+|\d[\d_]*(?:\.\d[\d_]*)?(?:[eE][+-]?\d[\d_]*)?)(?:[fF]|L|[uU]L?)?(?!\w|\.\d)/g },
{ className: 'property', regex: /\.\s*([A-Za-z_]\w*)/g, group: 1 },
{ className: 'operator', regex: /===|!==|\?[.:]|!!|\.\.<?|->|::|&&|\|\||\+\+|--|[+\-*/%!<>=]=?/g },
];
export const kotlin = defineLanguage({
name: 'kotlin',
aliases: ['kt', 'kts'],
tokenize(code) {
const ranges = [];
const occupied = new Uint8Array(code.length);
const lexical = /\/\/.*|\/\*|"""|"|'(?:\\(?:u[\da-fA-F]{4}|.)|[^'\\\n])'|`[^`\n]+`(\s*\()?/g;
let match;
while ((match = lexical.exec(code))) {
const token = match[0];
let from = match.index;
let end = lexical.lastIndex;
let className = token[0] === '/' ? 'comment' : 'string';
if (token === '/*') {
const nested = /\/\*|\*\//g;
nested.lastIndex = end;
for (let depth = 1; depth && (match = nested.exec(code));)
depth += match[0] === '/*' ? 1 : -1;
end = match ? nested.lastIndex : code.length;
}
else if (token[0] === '`') {
if (!match[1]) {
occupied.fill(1, from, end);
continue;
}
end = from + token.lastIndexOf('`') + 1;
className = 'function';
}
else if (token[0] === '"') {
while (code[from - 1] === '$')
from--;
end = stringEnd(code, end, token.length > 1);
}
ranges.push({ start: from, end, className });
lexical.lastIndex = end;
}
for (const path of code.matchAll(/^[ \t]*(?:package|import)[ \t]+([\w.*`]+)/gm)) {
occupied.fill(1, path.index + path[0].length - path[1].length, path.index + path[0].length);
}
return collectPatternRanges(code, patterns, ranges, occupied);
},
});
// Stack of open strings (raw flag) and `${` templates (brace depth), so nested quotes stay inside.
function stringEnd(code, index, raw) {
const stack = [raw];
while (index < code.length) {
const top = stack[stack.length - 1];
const char = code[index++];
if (typeof top === 'number') {
if (char === '"') {
const nested = code.startsWith('""', index);
stack.push(nested);
if (nested)
index += 2;
}
else if (char === "'")
index = code.indexOf("'", index + (code[index] === '\\' ? 2 : 1)) + 1 || code.length;
else if (char === '{')
stack[stack.length - 1] = top + 1;
else if (char === '}') {
if (top)
stack[stack.length - 1] = top - 1;
else
stack.pop();
}
}
else if (char === '"' && (!top || code.startsWith('""', index)) || char === '\n' && !top) {
if (char === '\n')
index--;
else if (top)
while (code[index] === '"')
index++;
stack.pop();
if (!stack.length)
return index;
}
else if (char === '\\' && !top)
index++;
else if (char === '$' && code[index] === '{') {
stack.push(0);
index++;
}
}
return code.length;
}