shared/matchers/word.matcher.ts
shared/matchers/word.matcher.ts is a file in GovLab Extension Host. 153 lines of code and 47 definitions.
import type { RenameRules, RenamedSpan } from "../../types/writing.types.ts";
interface IdSite {
readonly after: number;
readonly at: number;
readonly before: string;
readonly next: string;
readonly rules: RenameRules;
readonly text: string;
}
const LOWER = "abcdefghijklmnopqrstuvwxyz";
const UPPER = "ABCDEFGHIJKLMNOPQRSTUVWXYZ";
const DIGITS = "0123456789";
const JOINERS = "-_";
const REF_MARK = ":";
const ANCHOR_MARK = "#";
const ANCHOR_JOINER = "-";
const SLASH = "/";
const SEGMENT_DOT = ".";
const SEGMENT_ENDS: readonly string[] = [SLASH, ".md", ".json"];
const PLACEHOLDER_OPENERS = "<{";
const COLLECTION_NOUN_TAILS: readonly string[] = [" collection", "` collection"];
const isLetter = function isLetter(char: string): boolean {
return char.length === 1 && (LOWER.includes(char) || UPPER.includes(char));
};
const isWordPart = function isWordPart(char: string): boolean {
return isLetter(char) || (char.length === 1 && (DIGITS.includes(char) || JOINERS.includes(char)));
};
const isIdStart = function isIdStart(char: string): boolean {
return char.length === 1 && (LOWER.includes(char) || DIGITS.includes(char) || PLACEHOLDER_OPENERS.includes(char));
};
const closesText = function closesText(site: IdSite): boolean {
return site.at === 0 && site.after + 1 === site.text.length;
};
const precedingSegment = function precedingSegment(text: string, slashAt: number): string {
let open = slashAt;
while (open > 0 && isWordPart(text.charAt(open - 1))) {
open -= 1;
}
return text.slice(open, slashAt);
};
const endsSegment = function endsSegment(text: string, at: number): boolean {
const next = text.charAt(at);
return SEGMENT_ENDS.some((end) => text.startsWith(end, at)) || (!isWordPart(next) && next !== SEGMENT_DOT);
};
const anchorShape = function anchorShape(site: IdSite): boolean | null {
return site.before === ANCHOR_MARK
? site.next === ANCHOR_JOINER && isIdStart(site.text.charAt(site.after + 1))
: null;
};
const pathShape = function pathShape(site: IdSite): boolean | null {
if (site.before !== SLASH) {
return null;
}
return (
site.rules.pathSegments.includes(precedingSegment(site.text, site.at - 1)) && endsSegment(site.text, site.after)
);
};
const closingPrefix = function closingPrefix(site: IdSite): boolean {
return site.rules.wholeId && closesText(site) && site.next === ANCHOR_JOINER;
};
const namesCollection = function namesCollection(site: IdSite): boolean {
return COLLECTION_NOUN_TAILS.some(
(tail) => site.text.startsWith(tail, site.after) && !isLetter(site.text.charAt(site.after + tail.length)),
);
};
const reference = function reference(site: IdSite): boolean {
return site.next === REF_MARK && (isIdStart(site.text.charAt(site.after + 1)) || closesText(site));
};
const openShape = function openShape(site: IdSite): boolean {
return closingPrefix(site) || (!isWordPart(site.before) && (namesCollection(site) || reference(site)));
};
const idSpanAt = function idSpanAt(text: string, at: number, id: string, rules: RenameRules): boolean {
const after = at + id.length;
const site: IdSite = {
after,
at,
before: at === 0 ? "" : text.charAt(at - 1),
next: text.charAt(after),
rules,
text,
};
return anchorShape(site) ?? pathShape(site) ?? openShape(site);
};
const casedLike = function casedLike(original: string, replacement: string): string {
if (original.length > 1 && original === original.toUpperCase()) {
return replacement.toUpperCase();
}
const first = original.charAt(0);
return first === first.toUpperCase() ? replacement.charAt(0).toUpperCase() + replacement.slice(1) : replacement;
};
const wordEnd = function wordEnd(text: string, at: number): number {
let end = at;
while (end < text.length && isLetter(text.charAt(end))) {
end += 1;
}
return end;
};
const wordSpanAt = function wordSpanAt(
text: string,
at: number,
words: ReadonlyMap<string, string>,
): RenamedSpan | null {
if (!isLetter(text.charAt(at)) || (at > 0 && isWordPart(text.charAt(at - 1)))) {
return null;
}
const end = wordEnd(text, at);
if (end < text.length && isWordPart(text.charAt(end))) {
return null;
}
const from = text.slice(at, end);
const to = words.get(from.toLowerCase());
return to === undefined ? null : { end, from, start: at, to: casedLike(from, to) };
};
const spanAt = function spanAt(
text: string,
at: number,
ids: readonly string[],
rules: RenameRules,
): RenamedSpan | null {
const id = ids.find((candidate) => text.startsWith(candidate, at) && idSpanAt(text, at, candidate, rules));
if (id !== undefined) {
return { end: at + id.length, from: id, start: at, to: rules.ids.get(id) ?? id };
}
return rules.words === null ? null : wordSpanAt(text, at, rules.words);
};
export const renamedSpans = function renamedSpans(text: string, rules: RenameRules): readonly RenamedSpan[] {
const whole = rules.ids.get(text);
if (rules.wholeId && whole !== undefined) {
return [{ end: text.length, from: text, start: 0, to: whole }];
}
const ids = [...rules.ids.keys()].toSorted((left, right) => right.length - left.length);
const spans: RenamedSpan[] = [];
let at = 0;
while (at < text.length) {
const span = spanAt(text, at, ids, rules);
if (span === null) {
at += 1;
} else {
spans.push(span);
at = span.end;
}
}
return spans;
};
export const renamedText = function renamedText(text: string, rules: RenameRules): string {
let renamed = "";
let at = 0;
for (const span of renamedSpans(text, rules)) {
renamed += text.slice(at, span.start) + span.to;
at = span.end;
}
return renamed + text.slice(at);
};