core/matchers/leak.matcher.ts
core/matchers/leak.matcher.ts is a file in Bane's Lab Content. 153 lines of code and 42 definitions.
import type { LeakMatch, PrivateHit, PrivateTerm } from "#types/leak.types";
import { MIN_TOKEN_LENGTH, PATH_SEPARATOR, TOKEN_STOPS, WORD_SEPARATOR } from "#configuration/constants/leak.constants";
import { BACKTICK } from "#configuration/constants/inventory.constants";
import { fingerprintOf } from "@govlab/content-fingerprint";
interface Word {
readonly line: number;
readonly text: string;
}
const NEWLINE_CODE = 10;
const DIGIT_CODES: readonly [number, number] = [48, 57];
const UPPER_CODES: readonly [number, number] = [65, 90];
const LOWER_CODES: readonly [number, number] = [97, 122];
const IDENTIFIER_MARKS: ReadonlySet<string> = new Set(["_", "-", ".", "@", PATH_SEPARATOR]);
const inlineCodeSpans = function inlineCodeSpans(text: string): string[] {
const spans: string[] = [];
let open = text.indexOf(BACKTICK);
while (open !== -1) {
const close = text.indexOf(BACKTICK, open + 1);
if (close === -1) {
break;
}
const span = text.slice(open + 1, close).trim();
if (span.length > 0) {
spans.push(span);
}
open = text.indexOf(BACKTICK, close + 1);
}
return spans;
};
const plainTokens = function plainTokens(text: string): string[] {
const tokens: string[] = [];
let current = "";
for (const char of text) {
if (TOKEN_STOPS.has(char)) {
if (current.length >= MIN_TOKEN_LENGTH) {
tokens.push(current);
}
current = "";
} else {
current += char;
}
}
if (current.length >= MIN_TOKEN_LENGTH) {
tokens.push(current);
}
return tokens;
};
const hasIdentifierShape = function hasIdentifierShape(token: string): boolean {
for (const char of token) {
if (IDENTIFIER_MARKS.has(char)) {
return true;
}
}
return false;
};
export const identifierTokens = function identifierTokens(text: string): string[] {
return plainTokens(text).filter(hasIdentifierShape);
};
const isWorkspacePath = function isWorkspacePath(token: string, roots: ReadonlySet<string>): boolean {
const cut = token.indexOf(PATH_SEPARATOR);
return cut > 0 && roots.has(token.slice(0, cut));
};
export const leaksIn = function leaksIn(
text: string,
tokens: ReadonlySet<string>,
roots: ReadonlySet<string>,
): LeakMatch[] {
const spanReason = function spanReason(span: string): LeakMatch["reason"] | null {
if (tokens.has(span)) {
return "inline-code";
}
return isWorkspacePath(span, roots) ? "path" : null;
};
const tokenReason = function tokenReason(token: string): LeakMatch["reason"] | null {
if (isWorkspacePath(token, roots)) {
return "path";
}
return hasIdentifierShape(token) && tokens.has(token) ? "identifier" : null;
};
const spans = inlineCodeSpans(text).flatMap((span) => {
const reason = spanReason(span);
return reason === null ? [] : [{ reason, token: span }];
});
const words = plainTokens(text).flatMap((token) => {
const reason = tokenReason(token);
return reason === null ? [] : [{ reason, token }];
});
const seen = new Set<string>();
return [...spans, ...words].filter((match) => {
const fresh = !seen.has(match.token);
seen.add(match.token);
return fresh;
});
};
const within = function within(code: number, [low, high]: readonly [number, number]): boolean {
return code >= low && code <= high;
};
const isWordCode = function isWordCode(code: number): boolean {
return within(code, DIGIT_CODES) || within(code, UPPER_CODES) || within(code, LOWER_CODES);
};
const wordsOf = function wordsOf(text: string): Word[] {
const words: Word[] = [];
let line = 1;
let start = -1;
for (let at = 0; at <= text.length; at += 1) {
const code = at < text.length ? (text.codePointAt(at) ?? 0) : NEWLINE_CODE;
if (isWordCode(code)) {
start = start === -1 ? at : start;
continue;
}
if (start !== -1) {
words.push({ line, text: text.slice(start, at).toLowerCase() });
start = -1;
}
line += code === NEWLINE_CODE ? 1 : 0;
}
return words;
};
const blanked = function blanked(text: string, allowances: readonly string[]): string {
return allowances.reduce((held, allowed) => held.split(allowed).join(WORD_SEPARATOR.repeat(allowed.length)), text);
};
export const privateTermScanner = function privateTermScanner(
terms: readonly PrivateTerm[],
allowances: readonly string[],
): (text: string) => PrivateHit[] {
const cache = new Map<string, string>();
const digestOf = function digestOf(text: string): string {
const known = cache.get(text) ?? fingerprintOf([text]);
cache.set(text, known);
return known;
};
const firsts = new Set(terms.map((term) => term.first));
const lengths = new Set(terms.map((term) => term.length));
const matchAt = function matchAt(words: readonly Word[], index: number, term: PrivateTerm): boolean {
const run = words.slice(index, index + term.words);
return run.length === term.words && digestOf(run.map((word) => word.text).join(WORD_SEPARATOR)) === term.digest;
};
return (text: string): PrivateHit[] => {
const words = wordsOf(blanked(text, allowances));
return words.flatMap((word, index) => {
if (!lengths.has(word.text.length)) {
return [];
}
const first = digestOf(word.text);
if (!firsts.has(first)) {
return [];
}
return terms
.filter((term) => term.first === first && matchAt(words, index, term))
.map((term) => ({ digest: term.digest, line: word.line }));
});
};
};