# core/validators/algorithm.validator.ts

> 126 lines of code and 31 definitions.

Tree: GovLab Context
Language: typescript
Layer: processing
Canonical: https://banes-lab.com/anatomy/context#file-context-core-validators-algorithm-validator-ts
Source text: https://banes-lab.com/source/context/core/validators/algorithm.validator.ts.txt

Listed in [core/validators](https://banes-lab.com/api/source/context/core/validators.md), after [core/validators/algorithm.reference.validator.ts](https://banes-lab.com/source/context/core/validators/algorithm.reference.validator.ts.md) and before [core/validators/alias.validator.ts](https://banes-lab.com/source/context/core/validators/alias.validator.ts.md).

## Definitions

- `groupSymbolsByGrammar` (lexical_declaration, line 39)
- `titleTokens` (lexical_declaration, line 68)
- `tokenOverlap` (lexical_declaration, line 78)
- `crossCatalogRedundancyOf` (lexical_declaration, line 82)
- `intraGrammarDivergentOf` (lexical_declaration, line 56)
- `IntegrityFaces` (interface_declaration, line 7)
- `GrammarSymbolSite` (interface_declaration, line 11)
- `CROSS_CATALOG_A` (lexical_declaration, line 16)
- `CROSS_CATALOG_B` (lexical_declaration, line 17)
- `TITLE_STOPWORDS` (lexical_declaration, line 18)
- `TITLE_OVERLAP_MIN` (lexical_declaration, line 31)
- `MIN_TOKEN_LENGTH` (lexical_declaration, line 32)
- `WORD_SEPARATOR` (lexical_declaration, line 33)
- `byName` (lexical_declaration, line 35)
- `byGrammar` (lexical_declaration, line 42)
- `grammar` (lexical_declaration, line 44)
- `site` (lexical_declaration, line 46)
- `catalogA` (lexical_declaration, line 83)
- `catalogB` (lexical_declaration, line 84)
- `tokensA` (lexical_declaration, line 87)
- `declaresDistinct` (lexical_declaration, line 95)
- `invalidDistinctDeclarationsOf` (lexical_declaration, line 99)
- `ids` (lexical_declaration, line 102)
- `integrityIssuesOf` (lexical_declaration, line 115, exported)
- `intraGrammarDivergentSymbols` (lexical_declaration, line 116, exported)
- `byId` (lexical_declaration, line 117, exported)
- `crossCatalogRedundancy` (lexical_declaration, line 118, exported)
- `invalidDistinctDeclarations` (lexical_declaration, line 121, exported)
- `generatedIndex` (lexical_declaration, line 122, exported)
- `committedIndex` (lexical_declaration, line 123, exported)
- `symbolIndexStale` (lexical_declaration, line 124, exported)

## Contained in

- [core/validators](https://banes-lab.com/anatomy/context/folder-context-core-validators.md)

## Linked from

- [core/validators](https://banes-lab.com/anatomy/context/folder-context-core-validators.md)

## Source

```typescript
import type { Contract, SymbolEntry } from "#types/algorithm.types";
import { DISTINCT_UNKNOWN_CONTRACT, DISTINCT_UNREASONED } from "#configuration/strings/validation.strings";
import type { IntegrityIssues } from "#types/validation.types";
import { buildSymbolIndex } from "#core/converters/algorithm.index.converter";
import { symbolIndexesMatch } from "#core/predicates/algorithm.predicate";

interface IntegrityFaces {
    algo: { all: () => readonly Contract[]; symbols: () => readonly SymbolEntry[] };
}

interface GrammarSymbolSite {
    rhsSet: Set<string>;
    records: Set<string>;
}

const CROSS_CATALOG_A = "architecture";
const CROSS_CATALOG_B = "arch-relationships";
const TITLE_STOPWORDS: ReadonlySet<string> = new Set([
    "and",
    "or",
    "the",
    "of",
    "a",
    "an",
    "to",
    "for",
    "with",
    "in",
    "on",
]);
const TITLE_OVERLAP_MIN = 2;
const MIN_TOKEN_LENGTH = 2;
const WORD_SEPARATOR = " ";

const byName = function byName(a: string, b: string): number {
    return a.localeCompare(b);
};

const groupSymbolsByGrammar = function groupSymbolsByGrammar(
    faces: IntegrityFaces,
): Map<string, Map<string, GrammarSymbolSite>> {
    const byGrammar = new Map<string, Map<string, GrammarSymbolSite>>();
    for (const contract of faces.algo.all()) {
        const grammar = byGrammar.get(contract.domain) ?? new Map<string, GrammarSymbolSite>();
        for (const production of contract.productions) {
            const site = grammar.get(production.lhs) ?? { records: new Set<string>(), rhsSet: new Set<string>() };
            site.rhsSet.add(production.rhs);
            site.records.add(contract.id);
            grammar.set(production.lhs, site);
        }
        byGrammar.set(contract.domain, grammar);
    }
    return byGrammar;
};

const intraGrammarDivergentOf = function intraGrammarDivergentOf(
    faces: IntegrityFaces,
): IntegrityIssues["intraGrammarDivergentSymbols"] {
    return [...groupSymbolsByGrammar(faces)]
        .flatMap(([grammar, symbols]) =>
            [...symbols]
                .filter(([, site]) => site.rhsSet.size > 1)
                .map(([name, site]) => ({ definedIn: [...site.records].toSorted(byName), grammar, name })),
        )
        .toSorted((a, b) => a.grammar.localeCompare(b.grammar) || a.name.localeCompare(b.name));
};

const titleTokens = function titleTokens(title: string): Set<string> {
    return new Set(
        title
            .toLowerCase()
            .split(WORD_SEPARATOR)
            .map((raw) => raw.trim())
            .filter((token) => token.length > MIN_TOKEN_LENGTH && !TITLE_STOPWORDS.has(token)),
    );
};

const tokenOverlap = function tokenOverlap(tokensA: Set<string>, tokensB: Set<string>): number {
    return [...tokensA].filter((token) => tokensB.has(token)).length;
};

const crossCatalogRedundancyOf = function crossCatalogRedundancyOf(faces: IntegrityFaces): { a: string; b: string }[] {
    const catalogA = faces.algo.all().filter((contract) => contract.domain === CROSS_CATALOG_A);
    const catalogB = faces.algo.all().filter((contract) => contract.domain === CROSS_CATALOG_B);
    return catalogA
        .flatMap((a) => {
            const tokensA = titleTokens(a.title);
            return catalogB
                .filter((b) => tokenOverlap(tokensA, titleTokens(b.title)) >= TITLE_OVERLAP_MIN)
                .map((b) => ({ a: a.id, b: b.id }));
        })
        .toSorted((x, y) => x.a.localeCompare(y.a) || x.b.localeCompare(y.b));
};

const declaresDistinct = function declaresDistinct(contract: Contract | undefined, other: string): boolean {
    return (contract?.distinctFrom ?? []).some((entry) => entry.id === other && entry.reason.trim().length > 0);
};

const invalidDistinctDeclarationsOf = function invalidDistinctDeclarationsOf(
    faces: IntegrityFaces,
): IntegrityIssues["invalidDistinctDeclarations"] {
    const ids = new Set(faces.algo.all().map((contract) => contract.id));
    return faces.algo.all().flatMap((contract) =>
        (contract.distinctFrom ?? []).flatMap((entry) => {
            if (!ids.has(entry.id)) {
                return [{ from: contract.id, reason: DISTINCT_UNKNOWN_CONTRACT, target: entry.id }];
            }
            return entry.reason.trim().length === 0
                ? [{ from: contract.id, reason: DISTINCT_UNREASONED, target: entry.id }]
                : [];
        }),
    );
};

export const integrityIssuesOf = function integrityIssuesOf(faces: IntegrityFaces): IntegrityIssues {
    const intraGrammarDivergentSymbols = intraGrammarDivergentOf(faces);
    const byId = new Map(faces.algo.all().map((contract) => [contract.id, contract]));
    const crossCatalogRedundancy = crossCatalogRedundancyOf(faces).filter(
        (pair) => !declaresDistinct(byId.get(pair.a), pair.b) && !declaresDistinct(byId.get(pair.b), pair.a),
    );
    const invalidDistinctDeclarations = invalidDistinctDeclarationsOf(faces);
    const generatedIndex = buildSymbolIndex(faces.algo.all());
    const committedIndex = faces.algo.symbols();
    const symbolIndexStale = !symbolIndexesMatch(committedIndex, generatedIndex);
    return {
        crossCatalogRedundancy,
        indexedSymbolCount: committedIndex.length,
        intraGrammarDivergentSymbols,
        invalidDistinctDeclarations,
        liveSymbolCount: generatedIndex.length,
        subtotal:
            intraGrammarDivergentSymbols.length +
            crossCatalogRedundancy.length +
            invalidDistinctDeclarations.length +
            (symbolIndexStale ? 1 : 0),
        symbolIndexStale,
    };
};
```
