# codemods/analyzers/word.analyzer.ts

> 135 lines of code and 35 definitions.

Tree: Governance tree
Language: typescript
Layer: processing
Canonical: https://banes-lab.com/anatomy/governance#file-governance-codemods-analyzers-word-analyzer-ts
Source text: https://banes-lab.com/source/governance/codemods/analyzers/word.analyzer.ts.txt

Listed in [codemods/analyzers](https://banes-lab.com/api/source/governance/codemods/analyzers.md), after [codemods/analyzers/vocabulary.analyzer.ts](https://banes-lab.com/source/governance/codemods/analyzers/vocabulary.analyzer.ts.md).

## Definitions

- `isLetter` (lexical_declaration, line 34)
- `isWordPart` (lexical_declaration, line 40)
- `wordsOf` (lexical_declaration, line 67)
- `inCase` (lexical_declaration, line 26)
- `insideSpan` (lexical_declaration, line 58)
- `spellingFindingsIn` (lexical_declaration, line 116, exported)
- `filesUnder` (lexical_declaration, line 131, exported)
- `lineFindings` (lexical_declaration, line 94)
- `collectSpellingFindings` (lexical_declaration, line 144, exported)
- `MARKDOWN_EXTENSION` (lexical_declaration, line 8)
- `FENCE` (lexical_declaration, line 9)
- `TICK` (lexical_declaration, line 10)
- `NEWLINE` (lexical_declaration, line 11)
- `americanOf` (lexical_declaration, line 13)
- `tail` (lexical_declaration, line 18)
- `first` (lexical_declaration, line 30)
- `WORD_JOINERS` (lexical_declaration, line 38)
- `codeSpans` (lexical_declaration, line 44)
- `open` (lexical_declaration, line 46)
- `close` (lexical_declaration, line 48)
- `WordSite` (interface_declaration, line 62)
- `sites` (lexical_declaration, line 68)
- `index` (lexical_declaration, line 69)
- `end` (lexical_declaration, line 75)
- `LineSite` (interface_declaration, line 87)
- `spans` (lexical_declaration, line 95)
- `american` (lexical_declaration, line 97)
- `markdown` (lexical_declaration, line 117, exported)
- `findings` (lexical_declaration, line 118, exported)
- `offset` (lexical_declaration, line 119, exported)
- `fenced` (lexical_declaration, line 120, exported)
- `fence` (lexical_declaration, line 122, exported)
- `scanned` (lexical_declaration, line 124, exported)
- `full` (lexical_declaration, line 136, exported)
- `wanted` (lexical_declaration, line 145, exported)

## Uses

- [codemods/selectors/program.selector.ts](https://banes-lab.com/source/governance/codemods/selectors/program.selector.ts.md)

## Source

````typescript
import { AMERICAN_WORDS, IZE_STEMS, IZE_SUFFIXES } from "@govlab/constants";
import type { LiteralFinding, WordScope } from "../../types/analyzer.types.ts";
import { readFileSync, readdirSync, statSync } from "node:fs";
import { ROOT } from "@ssot/paths";
import path from "node:path";
import { relPath } from "../selectors/program.selector.ts";

const MARKDOWN_EXTENSION = ".md";
const FENCE = "```";
const TICK = "`";
const NEWLINE = "\n";

const americanOf = function americanOf(lower: string): string | null {
    if (Object.hasOwn(AMERICAN_WORDS, lower)) {
        return AMERICAN_WORDS[lower] ?? null;
    }
    for (const stem of IZE_STEMS) {
        const tail = lower.slice(stem.length);
        if (lower.startsWith(stem) && Object.hasOwn(IZE_SUFFIXES, tail)) {
            return stem + (IZE_SUFFIXES[tail] ?? "");
        }
    }
    return null;
};

const inCase = function inCase(original: string, american: string): string {
    if (original === original.toUpperCase()) {
        return american.toUpperCase();
    }
    const first = original.charAt(0);
    return first === first.toUpperCase() ? american.charAt(0).toUpperCase() + american.slice(1) : american;
};

const isLetter = function isLetter(char: string): boolean {
    return (char >= "a" && char <= "z") || (char >= "A" && char <= "Z");
};

const WORD_JOINERS: ReadonlySet<string> = new Set(["_", "-", "0", "1", "2", "3", "4", "5", "6", "7", "8", "9"]);

const isWordPart = function isWordPart(char: string): boolean {
    return isLetter(char) || WORD_JOINERS.has(char);
};

const codeSpans = function codeSpans(line: string): [number, number][] {
    const spans: [number, number][] = [];
    let open = line.indexOf(TICK);
    while (open !== -1) {
        const close = line.indexOf(TICK, open + 1);
        if (close === -1) {
            break;
        }
        spans.push([open, close]);
        open = line.indexOf(TICK, close + 1);
    }
    return spans;
};

const insideSpan = function insideSpan(spans: readonly [number, number][], index: number): boolean {
    return spans.some(([open, close]) => index > open && index < close);
};

interface WordSite {
    readonly start: number;
    readonly word: string;
}

const wordsOf = function wordsOf(line: string): WordSite[] {
    const sites: WordSite[] = [];
    let index = 0;
    while (index < line.length) {
        if (!isLetter(line.charAt(index)) || (index > 0 && isWordPart(line.charAt(index - 1)))) {
            index += 1;
            continue;
        }
        let end = index;
        while (end < line.length && isLetter(line.charAt(end))) {
            end += 1;
        }
        if (end >= line.length || !isWordPart(line.charAt(end))) {
            sites.push({ start: index, word: line.slice(index, end) });
        }
        index = end;
    }
    return sites;
};

interface LineSite {
    readonly fileName: string;
    readonly line: string;
    readonly index: number;
    readonly offset: number;
}

const lineFindings = function lineFindings(site: LineSite): LiteralFinding[] {
    const spans = path.extname(site.fileName) === MARKDOWN_EXTENSION ? codeSpans(site.line) : [];
    return wordsOf(site.line).flatMap((word) => {
        const american = americanOf(word.word.toLowerCase());
        if (american === null || insideSpan(spans, word.start)) {
            return [];
        }
        return [
            {
                end: site.offset + word.start + word.word.length,
                file: relPath(site.fileName),
                fileName: site.fileName,
                from: word.word,
                line: site.index + 1,
                reason: null,
                start: site.offset + word.start,
                to: inCase(word.word, american),
            },
        ];
    });
};

export const spellingFindingsIn = function spellingFindingsIn(fileName: string, content: string): LiteralFinding[] {
    const markdown = path.extname(fileName) === MARKDOWN_EXTENSION;
    const findings: LiteralFinding[] = [];
    let offset = 0;
    let fenced = false;
    for (const [index, line] of content.split(NEWLINE).entries()) {
        const fence = markdown && line.trimStart().startsWith(FENCE);
        fenced = fence ? !fenced : fenced;
        const scanned = fence || fenced ? [] : lineFindings({ fileName, index, line, offset });
        findings.push(...scanned);
        offset += line.length + NEWLINE.length;
    }
    return findings;
};

export const filesUnder = function filesUnder(target: string, skipped: (relPath: string) => boolean): string[] {
    if (!statSync(target).isDirectory()) {
        return [target];
    }
    return readdirSync(target, { withFileTypes: true }).flatMap((entry) => {
        const full = path.join(target, entry.name);
        if (skipped(full)) {
            return [];
        }
        return entry.isDirectory() ? filesUnder(full, skipped) : [full];
    });
};

export const collectSpellingFindings = function collectSpellingFindings(scope: WordScope): LiteralFinding[] {
    const wanted = (fileName: string): boolean => scope.extensions.includes(path.extname(fileName));
    return scope.roots
        .flatMap((root) => filesUnder(path.resolve(ROOT, root), scope.skipped))
        .filter(wanted)
        .flatMap((fileName) => spellingFindingsIn(fileName, readFileSync(fileName, "utf8")));
};
````
