core/matchers/search.matcher.ts

core/matchers/search.matcher.ts is a file in Bane's Lab Site. 93 lines of code and 20 definitions.

import {
    FUZZY_MIN_TERM_LENGTH,
    HEX_CHARACTERS,
    HEX_RUN_MIN_LENGTH,
    INDEXED_WORD_MAX_LENGTH,
} from "#configuration/constants/search.constants";
import { identifierParts, normalizeWord } from "@govlab/constants";
import { DIGITS } from "#configuration/constants/syntax.constants";
import { SPACE } from "#configuration/constants/document.constants";
import { WORD_CHARACTERS } from "#configuration/constants/vocabulary.constants";
import { convertMarkup } from "#core/converters/markup.converter";
import { wordsOf } from "#core/normalizers/word.normalizer";

export const queryTerms = function queryTerms(query: string): readonly string[] {
    return [...new Set(wordsOf(query))];
};

const everyIn = function everyIn(word: string, characters: string): boolean {
    for (const character of word) {
        if (!characters.includes(character)) {
            return false;
        }
    }
    return true;
};

const isNoise = function isNoise(word: string): boolean {
    return (
        word.length > INDEXED_WORD_MAX_LENGTH ||
        everyIn(word, DIGITS) ||
        (word.length >= HEX_RUN_MIN_LENGTH && everyIn(word, HEX_CHARACTERS))
    );
};

const rawTokens = function rawTokens(text: string): readonly string[] {
    const tokens: string[] = [];
    let current = "";
    for (const character of text + SPACE) {
        const inWord = WORD_CHARACTERS.includes(character.toLowerCase());
        if (!inWord && current.length > 0) {
            tokens.push(current);
            current = "";
        }
        if (inWord) {
            current += character;
        }
    }
    return tokens;
};

export const markupWords = function markupWords(markup: string): readonly string[] {
    const text = convertMarkup(markup)
        .map((run) => run.text)
        .join(SPACE);
    return rawTokens(text).flatMap((token) => {
        const joined = normalizeWord(token);
        if (isNoise(joined)) {
            return [];
        }
        const parts = identifierParts(token);
        return parts.length > 1 ? [joined, ...parts.map(normalizeWord).filter((part) => !isNoise(part))] : [joined];
    });
};

const skipsOne = function skipsOne(longer: string, shorter: string): boolean {
    let at = 0;
    while (at < shorter.length && longer.charAt(at) === shorter.charAt(at)) {
        at += 1;
    }
    return longer.slice(at + 1) === shorter.slice(at);
};

const swapsOne = function swapsOne(left: string, right: string): boolean {
    let differences = 0;
    for (let at = 0; at < left.length; at += 1) {
        differences += left.charAt(at) === right.charAt(at) ? 0 : 1;
    }
    return differences <= 1;
};

export const withinOneEdit = function withinOneEdit(left: string, right: string): boolean {
    if (left.length === right.length) {
        return swapsOne(left, right);
    }
    if (Math.abs(left.length - right.length) > 1) {
        return false;
    }
    return left.length > right.length ? skipsOne(left, right) : skipsOne(right, left);
};

const PREFIX_SLACK = [-1, 0, 1];

const fuzzyPrefix = function fuzzyPrefix(term: string, word: string): boolean {
    return PREFIX_SLACK.some((slack) => withinOneEdit(term, word.slice(0, term.length + slack)));
};

export const termMatches = function termMatches(term: string, words: readonly string[]): boolean {
    return words.some(
        (word) => word.startsWith(term) || (term.length >= FUZZY_MIN_TERM_LENGTH && fuzzyPrefix(term, word)),
    );
};

export const matchesAll = function matchesAll(terms: readonly string[], words: readonly string[]): boolean {
    return terms.length > 0 && terms.every((term) => termMatches(term, words));
};