configuration/lexicon/data/model.data.json
configuration/lexicon/data/model.data.json is a file in GovLab Context. 375 lines of code and 0 definitions.
{
"category": "model-architecture",
"records": [
{
"name": "Ad-Hoc Notebook-to-Production",
"kind": "anti-pattern",
"definition": "Promoting exploratory notebook code straight to production without engineering it into a reliable pipeline."
},
{
"name": "Deploy-and-Forget Models",
"distinctFrom": [
{
"id": "lexicon:model-drift",
"reason": "Deploy-and-forget is never watching a model after release, while model drift is the loss of fidelity that goes unseen as a result."
}
],
"kind": "anti-pattern",
"definition": "Deploying a model and never monitoring it, so degradation as the data drifts goes unnoticed."
},
{
"name": "Exact Keyword Search Only",
"kind": "anti-pattern",
"definition": "Relying solely on exact keyword matching for retrieval, missing semantically related results."
},
{
"name": "Flat Document-Only Knowledge",
"kind": "anti-pattern",
"definition": "Representing knowledge as unlinked flat documents, losing the relationships a graph would capture."
},
{
"name": "Opaque Black-Box Decisions",
"kind": "anti-pattern",
"definition": "Producing model decisions with no explanation, so their reasoning cannot be inspected or trusted."
},
{
"name": "Opaque Ungoverned Model Use",
"kind": "anti-pattern",
"definition": "Using models with no governance or oversight, leaving their behavior and risks unmanaged."
},
{
"name": "Training-Time-Only Model Logic",
"kind": "anti-pattern",
"definition": "Building logic that exists only during training, with no counterpart to serve predictions at inference."
},
{
"name": "Unapproved Model Deployment",
"distinctFrom": [
{
"id": "architecture:model-version-ambiguity",
"reason": "Unapproved deployment skips the approval gate, while version ambiguity loses track of which version and configuration is running."
}
],
"kind": "anti-pattern",
"definition": "Deploying a model to production without passing the required review and approval gates."
},
{
"name": "Ungrounded Generation",
"kind": "anti-pattern",
"definition": "Generating output from a model alone without grounding it in retrieved facts, inviting hallucination."
},
{
"name": "Unguarded Model Autonomy",
"kind": "anti-pattern",
"definition": "Letting a model act autonomously with no safety guardrails on what it can do."
},
{
"name": "Untested Model Deployment",
"kind": "anti-pattern",
"definition": "Deploying a model without evaluating it, so its real-world quality is unknown until it fails."
},
{
"name": "Approval Policy",
"kind": "constraint",
"definition": "The declared rules and gates a model must pass before it may be deployed."
},
{
"name": "Document Store",
"kind": "artifact",
"definition": "The body of documents a retrieval system searches to ground a model's generation."
},
{
"name": "Data Pipeline",
"kind": "mechanism",
"definition": "The stages that ingest, clean, and transform data into a form suitable for training or inference."
},
{
"name": "Data/Model Boundaries",
"kind": "constraint",
"definition": "The lines separating data preparation, model training, and serving so each concern stays isolated."
},
{
"name": "Dataset",
"kind": "artifact",
"definition": "A curated collection of examples used to train or evaluate a model."
},
{
"name": "Embeddings",
"distinctFrom": [
{
"id": "lexicon:vector-index",
"reason": "Embeddings are the vectors themselves, while a vector index organizes them for nearest-neighbor lookup."
}
],
"kind": "artifact",
"definition": "Numeric vector representations of data that place semantically similar items near each other."
},
{
"name": "Entities",
"distinctFrom": [
{
"id": "lexicon:relations",
"reason": "Entities are a knowledge graph's nodes, while relations are its edges."
},
{
"id": "lexicon:schema-ontology",
"reason": "Entities are the instances in the graph, while the schema defines the entity types they belong to."
}
],
"kind": "artifact",
"definition": "The distinct things, such as people, places or concepts, that a knowledge graph represents as nodes."
},
{
"name": "Grounding Strategy",
"kind": "approach",
"definition": "A scheme for anchoring a model's output in retrieved, authoritative sources rather than its parameters alone."
},
{
"name": "Guardrails",
"kind": "mechanism",
"definition": "Constraints and filters that bound what a model is permitted to output or do at runtime."
},
{
"name": "Input/Output Contract",
"kind": "constraint",
"definition": "The agreed schema of the inputs a model accepts and the outputs it returns."
},
{
"name": "Model Artifact",
"kind": "artifact",
"definition": "The trained model file, with its learned weights, that is loaded to serve predictions."
},
{
"name": "Model Registry",
"kind": "artifact",
"definition": "A catalog that tracks model versions, their metadata, and their deployment status."
},
{
"name": "Rationale/Evidence",
"kind": "artifact",
"definition": "The reasons and supporting evidence recorded for a model's decision."
},
{
"name": "Relations",
"distinctFrom": [
{
"id": "lexicon:schema-ontology",
"reason": "Relations are the edges present in the graph, while the schema defines the relationship types allowed."
}
],
"kind": "artifact",
"definition": "The typed connections between entities that a knowledge graph represents as edges."
},
{
"name": "Retriever",
"kind": "mechanism",
"definition": "A component that finds and returns the most relevant documents for a query."
},
{
"name": "Schema/Ontology",
"kind": "artifact",
"definition": "A formal definition of the entity types and relationship types a knowledge graph may contain."
},
{
"name": "Tool Interface",
"kind": "constraint",
"definition": "The defined contract through which an agent invokes external tools and receives their results."
},
{
"name": "Training/Inference Separation",
"kind": "constraint",
"definition": "The requirement that model training and prediction serving be distinct, separately-managed phases."
},
{
"name": "Vector Index",
"kind": "artifact",
"definition": "A data structure that organizes embedding vectors for fast nearest-neighbor lookup."
},
{
"name": "Retraining Triggers",
"kind": "mechanism",
"definition": "Signals that automatically initiate model retraining when measured drift crosses a threshold."
},
{
"name": "Model-Integrated Systems",
"kind": "capability",
"definition": "The ability of a software system to incorporate models as integral parts of its behavior."
},
{
"name": "Audit and Debugging",
"kind": "capability",
"definition": "The ability to inspect and trace a model's decisions for auditing and debugging."
},
{
"name": "Bounded Tool-Using Agents",
"distinctFrom": [
{
"id": "lexicon:governed-autonomy",
"reason": "Bounded tool use limits which tools an agent may call, while governed autonomy limits which decisions it may take without approval."
}
],
"kind": "capability",
"definition": "The ability to run agents that use external tools within defined, safe limits."
},
{
"name": "Contextual Generation",
"kind": "capability",
"definition": "The ability to generate output informed by retrieved, task-specific context."
},
{
"name": "Controlled Model Deployment",
"kind": "capability",
"definition": "The ability to release models through a governed, approved process."
},
{
"name": "Degradation Detection",
"kind": "capability",
"definition": "The ability to detect when a model's accuracy declines as data shifts."
},
{
"name": "Governed Autonomy",
"kind": "capability",
"definition": "The ability to let an agent act autonomously within enforced governance limits."
},
{
"name": "Model Selection/Regression Detection",
"kind": "capability",
"definition": "The ability to compare models and catch quality regressions before deployment."
},
{
"name": "Relationship-Aware Retrieval/Reasoning",
"kind": "capability",
"definition": "The ability to retrieve and reason over the relationships between entities, not just isolated facts."
},
{
"name": "Reliable Model Lifecycle",
"kind": "capability",
"definition": "The ability to manage a model's data, training, deployment and monitoring reliably and repeatably."
},
{
"name": "Safe Model Deployment",
"kind": "capability",
"definition": "The ability to deploy models with safeguards that bound their behavior."
},
{
"name": "Semantic Search",
"kind": "capability",
"definition": "The ability to find results by meaning and similarity rather than exact keyword match."
},
{
"name": "Similarity Retrieval",
"kind": "capability",
"definition": "The ability to retrieve items nearest to a query in an embedding space."
},
{
"name": "Structured, Versioned Prompts",
"kind": "artifact",
"definition": "Prompts authored as structured, version-controlled artifacts rather than ad-hoc strings."
},
{
"name": "Knowledge Freshness",
"kind": "quality-attribute",
"definition": "The degree to which a system's knowledge reflects current rather than stale information."
},
{
"name": "Trust",
"kind": "quality-attribute",
"definition": "The degree to which users are willing to rely on a system's outputs."
},
{
"name": "Capability/Utility",
"kind": "quality-attribute",
"definition": "The degree of usefulness a model offers, which strict safety limits can constrain."
},
{
"name": "Curation Cost",
"kind": "quality-attribute",
"definition": "The degree of ongoing effort required to build and maintain a curated knowledge graph."
},
{
"name": "Experiment Velocity",
"kind": "metric",
"definition": "The rate at which model experiments can be run and iterated, which governance can slow."
},
{
"name": "Experimentation Speed",
"kind": "metric",
"definition": "The rate at which new modeling ideas can be tried and evaluated."
},
{
"name": "Explainability/Recall",
"kind": "quality-attribute",
"definition": "The degree to which retrieval stays explainable and complete, traded against pure similarity ranking."
},
{
"name": "Latency/Cost",
"kind": "quality-attribute",
"definition": "The degree of latency and expense incurred to serve model predictions."
},
{
"name": "Metric Completeness",
"kind": "quality-attribute",
"definition": "The degree to which evaluation metrics capture every dimension of a model's quality."
},
{
"name": "Model Complexity",
"kind": "quality-attribute",
"definition": "The degree of intricacy in a model, which raises accuracy but lowers explainability."
},
{
"name": "Monitoring Cost",
"kind": "quality-attribute",
"definition": "The degree of ongoing expense of continuously monitoring a deployed model."
},
{
"name": "Retrieval Quality/Latency",
"kind": "quality-attribute",
"definition": "The degree to which retrieval must trade result quality against speed."
},
{
"name": "Prompt Registry",
"kind": "technique",
"definition": "A technique for keeping prompts in one versioned registry instead of as literals across the code."
},
{
"name": "Prompt Versioning",
"kind": "technique",
"definition": "A technique for giving each prompt, dataset and index a version that inference logs record."
},
{
"name": "Evaluation Suite",
"kind": "technique",
"definition": "A technique for scoring model output against a fixed set of cases on every change to the model, prompt or data."
},
{
"name": "Centralized Model Configuration",
"kind": "technique",
"definition": "A technique for holding model names, parameters and limits in one configuration read by every call site."
},
{
"name": "Inference Context Record",
"kind": "technique",
"definition": "A technique for logging with each model output the model version, prompt version and inputs that produced it."
},
{
"name": "Evidence Citation",
"kind": "technique",
"definition": "A technique for attaching to each generated claim the retrieved source that supports it."
},
{
"name": "Claim Validation",
"kind": "technique",
"definition": "A technique for checking generated claims against their cited sources before they are returned."
},
{
"name": "Abstain or Disclose Uncertainty",
"kind": "technique",
"definition": "A technique for having a model decline or state its uncertainty when the evidence does not support an answer."
},
{
"name": "Governance Log",
"kind": "technique",
"definition": "A technique for recording each approval, deployment and retirement of a model with who decided it."
}
]
}