core/normalizers/word.normalizer.ts
core/normalizers/word.normalizer.ts is a file in Bane's Lab Site. 89 lines of code and 27 definitions.
import { ANGLE_CLOSE, ANGLE_OPEN, CLOSING_MARK } from "#configuration/constants/syntax.constants";
import { PROTECTED_TAGS, WORD_CHARACTERS } from "#configuration/constants/vocabulary.constants";
import { SPACE } from "#configuration/constants/document.constants";
import type { WordToken } from "#types/vocabulary.types";
import { normalizeWord } from "@govlab/constants";
interface Span {
readonly end: number;
readonly start: number;
}
interface Guard {
readonly name: string;
readonly start: number;
}
interface Step {
readonly guard: Guard | null;
readonly span: Span | null;
}
export const wordsOf = function wordsOf(phrase: string): readonly string[] {
const words: string[] = [];
let current = "";
for (const char of phrase.toLowerCase() + SPACE) {
if (WORD_CHARACTERS.includes(char)) {
current += char;
continue;
}
if (current.length > 0) {
words.push(normalizeWord(current));
current = "";
}
}
return words;
};
const tagOf = function tagOf(body: string): { readonly closing: boolean; readonly name: string } {
const closing = body.startsWith(CLOSING_MARK);
const trimmed = closing ? body.slice(1) : body;
const space = trimmed.indexOf(SPACE);
return { closing, name: (space === -1 ? trimmed : trimmed.slice(0, space)).toLowerCase() };
};
const stepOf = function stepOf(guard: Guard | null, body: string, open: number, close: number): Step {
const tag = tagOf(body);
if (guard !== null) {
const closes = tag.closing && tag.name === guard.name;
return closes ? { guard: null, span: { end: close + 1, start: guard.start } } : { guard, span: null };
}
if (!tag.closing && PROTECTED_TAGS.has(tag.name)) {
return { guard: { name: tag.name, start: open }, span: null };
}
return { guard: null, span: { end: close + 1, start: open } };
};
const protectedSpans = function protectedSpans(text: string): readonly Span[] {
const spans: Span[] = [];
let cursor = 0;
let guard: Guard | null = null;
while (cursor < text.length) {
const open = text.indexOf(ANGLE_OPEN, cursor);
const close = open === -1 ? -1 : text.indexOf(ANGLE_CLOSE, open);
if (open === -1 || close === -1) {
break;
}
const step = stepOf(guard, text.slice(open + 1, close), open, close);
if (step.span !== null) {
spans.push(step.span);
}
({ guard } = step);
cursor = close + 1;
}
return guard === null ? spans : [...spans, { end: text.length, start: guard.start }];
};
const isProtected = function isProtected(spans: readonly Span[], at: number): boolean {
return spans.some((span) => at >= span.start && at < span.end);
};
export const tokensOf = function tokensOf(text: string): readonly WordToken[] {
const spans = protectedSpans(text);
const tokens: WordToken[] = [];
let start = -1;
for (let at = 0; at <= text.length; at += 1) {
const char = at < text.length ? text.charAt(at).toLowerCase() : SPACE;
const inWord = WORD_CHARACTERS.includes(char) && !isProtected(spans, at);
if (inWord && start === -1) {
start = at;
continue;
}
if (!inWord && start !== -1) {
tokens.push({ end: at, start, word: normalizeWord(text.slice(start, at)) });
start = -1;
}
}
return tokens;
};