core/matchers/leak.matcher.ts

core/matchers/leak.matcher.ts is a file in Bane's Lab Content. 153 lines of code and 42 definitions.

import type { LeakMatch, PrivateHit, PrivateTerm } from "#types/leak.types";
import { MIN_TOKEN_LENGTH, PATH_SEPARATOR, TOKEN_STOPS, WORD_SEPARATOR } from "#configuration/constants/leak.constants";
import { BACKTICK } from "#configuration/constants/inventory.constants";
import { fingerprintOf } from "@govlab/content-fingerprint";

interface Word {
    readonly line: number;
    readonly text: string;
}

const NEWLINE_CODE = 10;
const DIGIT_CODES: readonly [number, number] = [48, 57];
const UPPER_CODES: readonly [number, number] = [65, 90];
const LOWER_CODES: readonly [number, number] = [97, 122];

const IDENTIFIER_MARKS: ReadonlySet<string> = new Set(["_", "-", ".", "@", PATH_SEPARATOR]);

const inlineCodeSpans = function inlineCodeSpans(text: string): string[] {
    const spans: string[] = [];
    let open = text.indexOf(BACKTICK);
    while (open !== -1) {
        const close = text.indexOf(BACKTICK, open + 1);
        if (close === -1) {
            break;
        }
        const span = text.slice(open + 1, close).trim();
        if (span.length > 0) {
            spans.push(span);
        }
        open = text.indexOf(BACKTICK, close + 1);
    }
    return spans;
};

const plainTokens = function plainTokens(text: string): string[] {
    const tokens: string[] = [];
    let current = "";
    for (const char of text) {
        if (TOKEN_STOPS.has(char)) {
            if (current.length >= MIN_TOKEN_LENGTH) {
                tokens.push(current);
            }
            current = "";
        } else {
            current += char;
        }
    }
    if (current.length >= MIN_TOKEN_LENGTH) {
        tokens.push(current);
    }
    return tokens;
};

const hasIdentifierShape = function hasIdentifierShape(token: string): boolean {
    for (const char of token) {
        if (IDENTIFIER_MARKS.has(char)) {
            return true;
        }
    }
    return false;
};

export const identifierTokens = function identifierTokens(text: string): string[] {
    return plainTokens(text).filter(hasIdentifierShape);
};

const isWorkspacePath = function isWorkspacePath(token: string, roots: ReadonlySet<string>): boolean {
    const cut = token.indexOf(PATH_SEPARATOR);
    return cut > 0 && roots.has(token.slice(0, cut));
};

export const leaksIn = function leaksIn(
    text: string,
    tokens: ReadonlySet<string>,
    roots: ReadonlySet<string>,
): LeakMatch[] {
    const spanReason = function spanReason(span: string): LeakMatch["reason"] | null {
        if (tokens.has(span)) {
            return "inline-code";
        }
        return isWorkspacePath(span, roots) ? "path" : null;
    };
    const tokenReason = function tokenReason(token: string): LeakMatch["reason"] | null {
        if (isWorkspacePath(token, roots)) {
            return "path";
        }
        return hasIdentifierShape(token) && tokens.has(token) ? "identifier" : null;
    };
    const spans = inlineCodeSpans(text).flatMap((span) => {
        const reason = spanReason(span);
        return reason === null ? [] : [{ reason, token: span }];
    });
    const words = plainTokens(text).flatMap((token) => {
        const reason = tokenReason(token);
        return reason === null ? [] : [{ reason, token }];
    });
    const seen = new Set<string>();
    return [...spans, ...words].filter((match) => {
        const fresh = !seen.has(match.token);
        seen.add(match.token);
        return fresh;
    });
};

const within = function within(code: number, [low, high]: readonly [number, number]): boolean {
    return code >= low && code <= high;
};

const isWordCode = function isWordCode(code: number): boolean {
    return within(code, DIGIT_CODES) || within(code, UPPER_CODES) || within(code, LOWER_CODES);
};

const wordsOf = function wordsOf(text: string): Word[] {
    const words: Word[] = [];
    let line = 1;
    let start = -1;
    for (let at = 0; at <= text.length; at += 1) {
        const code = at < text.length ? (text.codePointAt(at) ?? 0) : NEWLINE_CODE;
        if (isWordCode(code)) {
            start = start === -1 ? at : start;
            continue;
        }
        if (start !== -1) {
            words.push({ line, text: text.slice(start, at).toLowerCase() });
            start = -1;
        }
        line += code === NEWLINE_CODE ? 1 : 0;
    }
    return words;
};

const blanked = function blanked(text: string, allowances: readonly string[]): string {
    return allowances.reduce((held, allowed) => held.split(allowed).join(WORD_SEPARATOR.repeat(allowed.length)), text);
};

export const privateTermScanner = function privateTermScanner(
    terms: readonly PrivateTerm[],
    allowances: readonly string[],
): (text: string) => PrivateHit[] {
    const cache = new Map<string, string>();
    const digestOf = function digestOf(text: string): string {
        const known = cache.get(text) ?? fingerprintOf([text]);
        cache.set(text, known);
        return known;
    };
    const firsts = new Set(terms.map((term) => term.first));
    const lengths = new Set(terms.map((term) => term.length));
    const matchAt = function matchAt(words: readonly Word[], index: number, term: PrivateTerm): boolean {
        const run = words.slice(index, index + term.words);
        return run.length === term.words && digestOf(run.map((word) => word.text).join(WORD_SEPARATOR)) === term.digest;
    };
    return (text: string): PrivateHit[] => {
        const words = wordsOf(blanked(text, allowances));
        return words.flatMap((word, index) => {
            if (!lengths.has(word.text.length)) {
                return [];
            }
            const first = digestOf(word.text);
            if (!firsts.has(first)) {
                return [];
            }
            return terms
                .filter((term) => term.first === first && matchAt(words, index, term))
                .map((term) => ({ digest: term.digest, line: word.line }));
        });
    };
};