theme_guess/src/tokenizer.js

125 lines
4.7 KiB
JavaScript

// Minimal hand-rolled JS-ish lexer. It doesn't need to be a real parser —
// just consistent enough to bucket every word in the sample into the
// same "kind" categories a syntax theme would color.
const KEYWORDS = new Set([
'const', 'let', 'var', 'function', 'return', 'if', 'else', 'for', 'while',
'of', 'in', 'new', 'class', 'extends', 'constructor', 'import', 'from',
'export', 'default', 'static', 'async', 'await', 'try', 'catch', 'finally',
'throw', 'typeof', 'instanceof', 'break', 'continue', 'switch', 'case',
'yield', 'super', 'do',
]);
const CONSTANTS = new Set(['true', 'false', 'null', 'undefined', 'this']);
const OPS3 = ['===', '!==', '**='];
const OPS2 = ['=>', '==', '!=', '<=', '>=', '&&', '||', '+=', '-=', '*=', '/=', '++', '--'];
const PUNCT = '{}()[];,.';
const OP_CHARS = '+-*/%=<>!?:&|^~';
function scan(src) {
const raw = [];
let i = 0;
const n = src.length;
while (i < n) {
const ch = src[i];
if (ch === '\n') { raw.push({ type: 'newline', text: '\n' }); i++; continue; }
if (ch === ' ' || ch === '\t') {
let j = i;
while (j < n && (src[j] === ' ' || src[j] === '\t')) j++;
raw.push({ type: 'whitespace', text: src.slice(i, j) }); i = j; continue;
}
if (ch === '/' && src[i + 1] === '/') {
let j = i;
while (j < n && src[j] !== '\n') j++;
raw.push({ type: 'comment', text: src.slice(i, j) }); i = j; continue;
}
if (ch === '/' && src[i + 1] === '*') {
let j = i + 2;
while (j < n && !(src[j] === '*' && src[j + 1] === '/')) j++;
j = Math.min(j + 2, n);
raw.push({ type: 'comment', text: src.slice(i, j) }); i = j; continue;
}
if (ch === '"' || ch === "'" || ch === '`') {
const quote = ch;
let j = i + 1;
while (j < n && src[j] !== quote) { if (src[j] === '\\') j++; j++; }
j = Math.min(j + 1, n);
raw.push({ type: 'string', text: src.slice(i, j) }); i = j; continue;
}
if (/[0-9]/.test(ch)) {
let j = i;
while (j < n && /[0-9.]/.test(src[j])) j++;
raw.push({ type: 'number', text: src.slice(i, j) }); i = j; continue;
}
if (/[A-Za-z_$]/.test(ch)) {
let j = i;
while (j < n && /[A-Za-z0-9_$]/.test(src[j])) j++;
raw.push({ type: 'identifier', text: src.slice(i, j) }); i = j; continue;
}
const three = src.slice(i, i + 3);
const two = src.slice(i, i + 2);
if (OPS3.includes(three)) { raw.push({ type: 'operator', text: three }); i += 3; continue; }
if (OPS2.includes(two)) { raw.push({ type: 'operator', text: two }); i += 2; continue; }
if (PUNCT.includes(ch)) { raw.push({ type: 'punctuation', text: ch }); i++; continue; }
if (OP_CHARS.includes(ch)) { raw.push({ type: 'operator', text: ch }); i++; continue; }
raw.push({ type: 'punctuation', text: ch }); i++;
}
return raw;
}
// Defensive: split any token that smuggled in a newline (block comments,
// multi-line strings) so layout math stays a simple line/col grid.
function expandNewlines(raw) {
const tokens = [];
for (const t of raw) {
if (t.type !== 'newline' && t.text.includes('\n')) {
const parts = t.text.split('\n');
parts.forEach((p, idx) => {
if (p.length) tokens.push({ type: t.type, text: p });
if (idx < parts.length - 1) tokens.push({ type: 'newline', text: '\n' });
});
} else {
tokens.push(t);
}
}
return tokens;
}
function classify(tokens) {
let line = 0;
let col = 0;
for (let k = 0; k < tokens.length; k++) {
const t = tokens[k];
if (t.type === 'newline') { line++; col = 0; continue; }
if (t.type === 'identifier') {
const word = t.text;
let p = k - 1;
while (p >= 0 && (tokens[p].type === 'whitespace' || tokens[p].type === 'newline')) p--;
let nx = k + 1;
while (nx < tokens.length && (tokens[nx].type === 'whitespace' || tokens[nx].type === 'newline')) nx++;
const prev = p >= 0 ? tokens[p] : null;
const next = nx < tokens.length ? tokens[nx] : null;
const isCall = !!next && next.type === 'punctuation' && next.text === '(';
const isProp = !!prev && prev.type === 'punctuation' && prev.text === '.';
const isClassCtx = !!prev && prev.type === 'keyword' && ['class', 'new', 'extends'].includes(prev.text);
if (KEYWORDS.has(word)) t.type = 'keyword';
else if (CONSTANTS.has(word)) t.type = 'constant';
else if (/^[A-Z]/.test(word) && (isClassCtx || !isCall)) t.type = 'type';
else if (isCall) t.type = 'function';
else if (isProp) t.type = 'property';
else t.type = 'variable';
}
t.line = line;
t.col = col;
col += t.text.length;
}
return tokens;
}
export function tokenize(source) {
return classify(expandNewlines(scan(source)));
}