# shared/matchers/word.matcher.ts

> 153 lines of code and 47 definitions.

Tree: Governance tree
Language: typescript
Layer: processing
Canonical: https://banes-lab.com/anatomy/governance#file-governance-shared-matchers-word-matcher-ts
Source text: https://banes-lab.com/source/governance/shared/matchers/word.matcher.ts.txt

Listed in [shared/matchers](https://banes-lab.com/api/source/governance/shared/matchers.md), after [shared/matchers/vocabulary.matcher.ts](https://banes-lab.com/source/governance/shared/matchers/vocabulary.matcher.ts.md).

## Definitions

- `isWordPart` (lexical_declaration, line 29)
- `isLetter` (lexical_declaration, line 25)
- `isIdStart` (lexical_declaration, line 33)
- `closesText` (lexical_declaration, line 37)
- `renamedSpans` (lexical_declaration, line 146, exported)
- `openShape` (lexical_declaration, line 83)
- `wordSpanAt` (lexical_declaration, line 116)
- `pathShape` (lexical_declaration, line 60)
- `reference` (lexical_declaration, line 79)
- `precedingSegment` (lexical_declaration, line 41)
- `endsSegment` (lexical_declaration, line 49)
- `anchorShape` (lexical_declaration, line 54)
- `closingPrefix` (lexical_declaration, line 69)
- `namesCollection` (lexical_declaration, line 73)
- `casedLike` (lexical_declaration, line 100)
- `idSpanAt` (lexical_declaration, line 87)
- `wordEnd` (lexical_declaration, line 108)
- `spanAt` (lexical_declaration, line 133)
- `renamedText` (lexical_declaration, line 166, exported)
- `IdSite` (interface_declaration, line 3)
- `LOWER` (lexical_declaration, line 12)
- `UPPER` (lexical_declaration, line 13)
- `DIGITS` (lexical_declaration, line 14)
- `JOINERS` (lexical_declaration, line 15)
- `REF_MARK` (lexical_declaration, line 16)
- `ANCHOR_MARK` (lexical_declaration, line 17)
- `ANCHOR_JOINER` (lexical_declaration, line 18)
- `SLASH` (lexical_declaration, line 19)
- `SEGMENT_DOT` (lexical_declaration, line 20)
- `SEGMENT_ENDS` (lexical_declaration, line 21)
- `PLACEHOLDER_OPENERS` (lexical_declaration, line 22)
- `COLLECTION_NOUN_TAILS` (lexical_declaration, line 23)
- `open` (lexical_declaration, line 42)
- `next` (lexical_declaration, line 50)
- `after` (lexical_declaration, line 88)
- `site` (lexical_declaration, line 89)
- `first` (lexical_declaration, line 104)
- `end` (lexical_declaration, line 124)
- `from` (lexical_declaration, line 128)
- `to` (lexical_declaration, line 129)
- `id` (lexical_declaration, line 139)
- `whole` (lexical_declaration, line 147, exported)
- `ids` (lexical_declaration, line 151, exported)
- `spans` (lexical_declaration, line 152, exported)
- `span` (lexical_declaration, line 155, exported)
- `renamed` (lexical_declaration, line 167, exported)
- `at` (lexical_declaration, line 168, exported)

## Used by

- [codemods/analyzers/vocabulary.analyzer.ts](https://banes-lab.com/source/governance/codemods/analyzers/vocabulary.analyzer.ts.md)

## Source

```typescript
import type { RenameRules, RenamedSpan } from "../../types/writing.types.ts";

interface IdSite {
    readonly after: number;
    readonly at: number;
    readonly before: string;
    readonly next: string;
    readonly rules: RenameRules;
    readonly text: string;
}

const LOWER = "abcdefghijklmnopqrstuvwxyz";
const UPPER = "ABCDEFGHIJKLMNOPQRSTUVWXYZ";
const DIGITS = "0123456789";
const JOINERS = "-_";
const REF_MARK = ":";
const ANCHOR_MARK = "#";
const ANCHOR_JOINER = "-";
const SLASH = "/";
const SEGMENT_DOT = ".";
const SEGMENT_ENDS: readonly string[] = [SLASH, ".md", ".json"];
const PLACEHOLDER_OPENERS = "<{";
const COLLECTION_NOUN_TAILS: readonly string[] = [" collection", "` collection"];

const isLetter = function isLetter(char: string): boolean {
    return char.length === 1 && (LOWER.includes(char) || UPPER.includes(char));
};

const isWordPart = function isWordPart(char: string): boolean {
    return isLetter(char) || (char.length === 1 && (DIGITS.includes(char) || JOINERS.includes(char)));
};

const isIdStart = function isIdStart(char: string): boolean {
    return char.length === 1 && (LOWER.includes(char) || DIGITS.includes(char) || PLACEHOLDER_OPENERS.includes(char));
};

const closesText = function closesText(site: IdSite): boolean {
    return site.at === 0 && site.after + 1 === site.text.length;
};

const precedingSegment = function precedingSegment(text: string, slashAt: number): string {
    let open = slashAt;
    while (open > 0 && isWordPart(text.charAt(open - 1))) {
        open -= 1;
    }
    return text.slice(open, slashAt);
};

const endsSegment = function endsSegment(text: string, at: number): boolean {
    const next = text.charAt(at);
    return SEGMENT_ENDS.some((end) => text.startsWith(end, at)) || (!isWordPart(next) && next !== SEGMENT_DOT);
};

const anchorShape = function anchorShape(site: IdSite): boolean | null {
    return site.before === ANCHOR_MARK
        ? site.next === ANCHOR_JOINER && isIdStart(site.text.charAt(site.after + 1))
        : null;
};

const pathShape = function pathShape(site: IdSite): boolean | null {
    if (site.before !== SLASH) {
        return null;
    }
    return (
        site.rules.pathSegments.includes(precedingSegment(site.text, site.at - 1)) && endsSegment(site.text, site.after)
    );
};

const closingPrefix = function closingPrefix(site: IdSite): boolean {
    return site.rules.wholeId && closesText(site) && site.next === ANCHOR_JOINER;
};

const namesCollection = function namesCollection(site: IdSite): boolean {
    return COLLECTION_NOUN_TAILS.some(
        (tail) => site.text.startsWith(tail, site.after) && !isLetter(site.text.charAt(site.after + tail.length)),
    );
};

const reference = function reference(site: IdSite): boolean {
    return site.next === REF_MARK && (isIdStart(site.text.charAt(site.after + 1)) || closesText(site));
};

const openShape = function openShape(site: IdSite): boolean {
    return closingPrefix(site) || (!isWordPart(site.before) && (namesCollection(site) || reference(site)));
};

const idSpanAt = function idSpanAt(text: string, at: number, id: string, rules: RenameRules): boolean {
    const after = at + id.length;
    const site: IdSite = {
        after,
        at,
        before: at === 0 ? "" : text.charAt(at - 1),
        next: text.charAt(after),
        rules,
        text,
    };
    return anchorShape(site) ?? pathShape(site) ?? openShape(site);
};

const casedLike = function casedLike(original: string, replacement: string): string {
    if (original.length > 1 && original === original.toUpperCase()) {
        return replacement.toUpperCase();
    }
    const first = original.charAt(0);
    return first === first.toUpperCase() ? replacement.charAt(0).toUpperCase() + replacement.slice(1) : replacement;
};

const wordEnd = function wordEnd(text: string, at: number): number {
    let end = at;
    while (end < text.length && isLetter(text.charAt(end))) {
        end += 1;
    }
    return end;
};

const wordSpanAt = function wordSpanAt(
    text: string,
    at: number,
    words: ReadonlyMap<string, string>,
): RenamedSpan | null {
    if (!isLetter(text.charAt(at)) || (at > 0 && isWordPart(text.charAt(at - 1)))) {
        return null;
    }
    const end = wordEnd(text, at);
    if (end < text.length && isWordPart(text.charAt(end))) {
        return null;
    }
    const from = text.slice(at, end);
    const to = words.get(from.toLowerCase());
    return to === undefined ? null : { end, from, start: at, to: casedLike(from, to) };
};

const spanAt = function spanAt(
    text: string,
    at: number,
    ids: readonly string[],
    rules: RenameRules,
): RenamedSpan | null {
    const id = ids.find((candidate) => text.startsWith(candidate, at) && idSpanAt(text, at, candidate, rules));
    if (id !== undefined) {
        return { end: at + id.length, from: id, start: at, to: rules.ids.get(id) ?? id };
    }
    return rules.words === null ? null : wordSpanAt(text, at, rules.words);
};

export const renamedSpans = function renamedSpans(text: string, rules: RenameRules): readonly RenamedSpan[] {
    const whole = rules.ids.get(text);
    if (rules.wholeId && whole !== undefined) {
        return [{ end: text.length, from: text, start: 0, to: whole }];
    }
    const ids = [...rules.ids.keys()].toSorted((left, right) => right.length - left.length);
    const spans: RenamedSpan[] = [];
    let at = 0;
    while (at < text.length) {
        const span = spanAt(text, at, ids, rules);
        if (span === null) {
            at += 1;
        } else {
            spans.push(span);
            at = span.end;
        }
    }
    return spans;
};

export const renamedText = function renamedText(text: string, rules: RenameRules): string {
    let renamed = "";
    let at = 0;
    for (const span of renamedSpans(text, rules)) {
        renamed += text.slice(at, span.start) + span.to;
        at = span.end;
    }
    return renamed + text.slice(at);
};
```
