core/normalizers/word.normalizer.ts

core/normalizers/word.normalizer.ts is a file in Bane's Lab Site. 89 lines of code and 27 definitions.

import { ANGLE_CLOSE, ANGLE_OPEN, CLOSING_MARK } from "#configuration/constants/syntax.constants";
import { PROTECTED_TAGS, WORD_CHARACTERS } from "#configuration/constants/vocabulary.constants";
import { SPACE } from "#configuration/constants/document.constants";
import type { WordToken } from "#types/vocabulary.types";
import { normalizeWord } from "@govlab/constants";

interface Span {
    readonly end: number;
    readonly start: number;
}

interface Guard {
    readonly name: string;
    readonly start: number;
}

interface Step {
    readonly guard: Guard | null;
    readonly span: Span | null;
}

export const wordsOf = function wordsOf(phrase: string): readonly string[] {
    const words: string[] = [];
    let current = "";
    for (const char of phrase.toLowerCase() + SPACE) {
        if (WORD_CHARACTERS.includes(char)) {
            current += char;
            continue;
        }
        if (current.length > 0) {
            words.push(normalizeWord(current));
            current = "";
        }
    }
    return words;
};

const tagOf = function tagOf(body: string): { readonly closing: boolean; readonly name: string } {
    const closing = body.startsWith(CLOSING_MARK);
    const trimmed = closing ? body.slice(1) : body;
    const space = trimmed.indexOf(SPACE);
    return { closing, name: (space === -1 ? trimmed : trimmed.slice(0, space)).toLowerCase() };
};

const stepOf = function stepOf(guard: Guard | null, body: string, open: number, close: number): Step {
    const tag = tagOf(body);
    if (guard !== null) {
        const closes = tag.closing && tag.name === guard.name;
        return closes ? { guard: null, span: { end: close + 1, start: guard.start } } : { guard, span: null };
    }
    if (!tag.closing && PROTECTED_TAGS.has(tag.name)) {
        return { guard: { name: tag.name, start: open }, span: null };
    }
    return { guard: null, span: { end: close + 1, start: open } };
};

const protectedSpans = function protectedSpans(text: string): readonly Span[] {
    const spans: Span[] = [];
    let cursor = 0;
    let guard: Guard | null = null;
    while (cursor < text.length) {
        const open = text.indexOf(ANGLE_OPEN, cursor);
        const close = open === -1 ? -1 : text.indexOf(ANGLE_CLOSE, open);
        if (open === -1 || close === -1) {
            break;
        }
        const step = stepOf(guard, text.slice(open + 1, close), open, close);
        if (step.span !== null) {
            spans.push(step.span);
        }
        ({ guard } = step);
        cursor = close + 1;
    }
    return guard === null ? spans : [...spans, { end: text.length, start: guard.start }];
};

const isProtected = function isProtected(spans: readonly Span[], at: number): boolean {
    return spans.some((span) => at >= span.start && at < span.end);
};

export const tokensOf = function tokensOf(text: string): readonly WordToken[] {
    const spans = protectedSpans(text);
    const tokens: WordToken[] = [];
    let start = -1;
    for (let at = 0; at <= text.length; at += 1) {
        const char = at < text.length ? text.charAt(at).toLowerCase() : SPACE;
        const inWord = WORD_CHARACTERS.includes(char) && !isProtected(spans, at);
        if (inWord && start === -1) {
            start = at;
            continue;
        }
        if (!inWord && start !== -1) {
            tokens.push({ end: at, start, word: normalizeWord(text.slice(start, at)) });
            start = -1;
        }
    }
    return tokens;
};