core/coordinators/comment.coordinator.ts

core/coordinators/comment.coordinator.ts is a file in GovLab Quality. 163 lines of code and 33 definitions.

import type { CommentGrammar, ExtractResult, FileComment, StripResult } from "#types/comment.types";
import {
    EXTRACT_OUT,
    MAX_PARSE_BYTES,
    STRIP_CONCURRENCY,
    STRIP_INDEX,
} from "#configuration/constants/comment.constants";
import { type FingerprintIndex, cacheFile, createFingerprintIndex, fingerprintOf } from "@govlab/content-fingerprint";
import {
    cleanedLine,
    extractSummary,
    extractedLine,
    skippedLine,
    stripSummary,
} from "#configuration/strings/comment.strings";
import { collectFiles, commentGrammars, langOf, preload } from "#core/loaders/comment.loader";
import { extractComments, stripComments } from "#core/converters/comment.converter";
import type { PathExclusion } from "#types/exclusions.types";
import { appendExtracted } from "#core/persistence/comment.persistence";
import { promises as fs } from "node:fs";
import path from "node:path";
import { relativePath } from "@ssot/paths";
import { writeVerbatim } from "@govlab/canonical-write";

type Grammars = Map<string, CommentGrammar>;

interface StripCtx {
    grammars: Grammars;
    index: FingerprintIndex;
    root: string;
}

interface StripJob {
    filePath: string;
    hash: string;
    key: string;
    result: StripResult | null;
}

const errorDetail = function errorDetail(error: unknown): string {
    return error instanceof Error ? `${error.name}: ${error.message}` : String(error);
};

const pool = async function pool<T>(items: readonly T[], worker: (item: T) => Promise<void>): Promise<void> {
    const queue = [...items];
    const drain = async function drain(): Promise<void> {
        const next = queue.shift();
        if (next === undefined) {
            return;
        }
        await worker(next);
        await drain();
    };
    await Promise.all(Array.from({ length: Math.min(STRIP_CONCURRENCY, queue.length) }, drain));
};

const oversized = async function oversized(filePath: string, key: string): Promise<boolean> {
    const { size } = await fs.stat(filePath);
    if (size <= MAX_PARSE_BYTES) {
        return false;
    }
    process.stderr.write(
        skippedLine(key, `${String(size)} bytes exceeds the ${String(MAX_PARSE_BYTES)}-byte parse limit`),
    );
    return true;
};

const guarded = function guarded<T>(key: string, run: () => T): T | null {
    try {
        return run();
    } catch (error) {
        process.stderr.write(skippedLine(key, errorDetail(error)));
        return null;
    }
};

const stripGrammar = function stripGrammar(filePath: string, grammars: Grammars): CommentGrammar | null {
    const lang = langOf(filePath);
    return lang === null ? null : (grammars.get(lang) ?? { directivePrefixes: [], lang });
};

const extractGrammar = function extractGrammar(filePath: string, grammars: Grammars): CommentGrammar | null {
    const lang = langOf(filePath);
    return lang === null ? null : (grammars.get(lang) ?? null);
};

const prepareStrip = async function prepareStrip(
    filePath: string,
    key: string,
    index: FingerprintIndex,
): Promise<{ content: string; hash: string } | null> {
    if (await oversized(filePath, key)) {
        return null;
    }
    const content = await fs.readFile(filePath, "utf8");
    const hash = fingerprintOf([content]);
    return index.unchanged(key, hash) ? null : { content, hash };
};

const commitStrip = function commitStrip(job: StripJob, ctx: StripCtx): number {
    if (job.result?.changed !== true) {
        ctx.index.update(job.key, job.hash);
        return 0;
    }
    writeVerbatim(job.filePath, job.result.content);
    process.stdout.write(cleanedLine(path.relative(ctx.root, job.filePath)));
    ctx.index.update(job.key, fingerprintOf([job.result.content]));
    return 1;
};

const stripFile = async function stripFile(filePath: string, ctx: StripCtx): Promise<number> {
    const grammar = stripGrammar(filePath, ctx.grammars);
    const key = path.relative(ctx.root, filePath);
    const prepared = grammar === null ? null : await prepareStrip(filePath, key, ctx.index);
    if (grammar === null || prepared === null) {
        return 0;
    }
    const result = guarded(key, () => stripComments(prepared.content, grammar));
    return commitStrip({ filePath, hash: prepared.hash, key, result }, ctx);
};

const extractFile = async function extractFile(
    filePath: string,
    root: string,
    grammars: Grammars,
): Promise<FileComment[]> {
    const grammar = extractGrammar(filePath, grammars);
    const key = path.relative(root, filePath);
    if (grammar === null || (await oversized(filePath, key))) {
        return [];
    }
    const content = await fs.readFile(filePath, "utf8");
    const result: ExtractResult | null = guarded(key, () => extractComments(content, grammar));
    if (result?.changed !== true) {
        return [];
    }
    writeVerbatim(filePath, result.content);
    process.stdout.write(extractedLine(key));
    return result.comments.map((comment) => ({ file: key, line: comment.line, text: comment.text }));
};

export const runStrip = async function runStrip(root: string, excluded: PathExclusion): Promise<void> {
    const files = collectFiles(root, excluded);
    await preload(files);
    const ctx: StripCtx = {
        grammars: commentGrammars(),
        index: createFingerprintIndex({ file: cacheFile(STRIP_INDEX) }),
        root,
    };
    const counts: number[] = [];
    await pool(files, async (filePath) => {
        counts.push(await stripFile(filePath, ctx));
    });
    ctx.index.flush();
    process.stdout.write(
        stripSummary(
            counts.reduce((sum, count) => sum + count, 0),
            files.length,
        ),
    );
};

export const runExtract = async function runExtract(
    root: string,
    out: string | null,
    excluded: PathExclusion,
): Promise<void> {
    const files = collectFiles(root, excluded);
    await preload(files);
    const grammars = commentGrammars();
    const collected: FileComment[] = [];
    await pool(files, async (filePath) => {
        collected.push(...(await extractFile(filePath, root, grammars)));
    });
    const outPath = out === null ? path.join(root, relativePath("govlabHost.root"), EXTRACT_OUT) : path.resolve(out);
    const added = await appendExtracted(outPath, collected);
    process.stdout.write(extractSummary(added, path.relative(root, outPath)));
};