configuration/lexicon/data/observability.data.json

configuration/lexicon/data/observability.data.json is a file in GovLab Context. 342 lines of code and 0 definitions.

{
    "category": "observability-auditability-traceability",
    "records": [
        {
            "name": "Cost/Noise",
            "kind": "quality-attribute",
            "definition": "The degree to which collecting more telemetry adds cost and noise that can obscure the signals that matter."
        },
        {
            "name": "Incident Diagnosis",
            "kind": "capability",
            "definition": "The ability to diagnose the cause of an incident from a system's observable signals."
        },
        {
            "name": "Logs",
            "distinctFrom": [
                {
                    "id": "lexicon:traces",
                    "reason": "Logs record discrete events, while traces record the path and timing of one request across components."
                }
            ],
            "kind": "artifact",
            "definition": "A record of discrete, timestamped events a system emits about what it did."
        },
        {
            "name": "Traces",
            "kind": "artifact",
            "definition": "A record of the path and timing of a request as it moves through a system's components."
        },
        {
            "name": "Incident Analysis",
            "kind": "capability",
            "definition": "The ability to reconstruct and analyze an incident from the events a system logged."
        },
        {
            "name": "Noise/Personal Data Leakage",
            "kind": "quality-attribute",
            "definition": "The degree to which verbose logging adds noise and risks leaking personal or sensitive data."
        },
        {
            "name": "Structured Events",
            "kind": "artifact",
            "definition": "Log entries emitted as structured, machine-parsable records rather than free-form text."
        },
        {
            "name": "Alert Noise",
            "kind": "quality-attribute",
            "definition": "The degree to which monitoring many signals generates alerts that drown out the meaningful ones."
        },
        {
            "name": "Blind Operation",
            "kind": "anti-pattern",
            "definition": "Running a system in production with no monitoring, so problems are noticed only when users report them."
        },
        {
            "name": "Failure Detection",
            "kind": "capability",
            "definition": "The ability to detect that a system has failed or degraded by watching its monitored signals."
        },
        {
            "name": "Alert Fatigue",
            "kind": "quality-attribute",
            "definition": "The degree to which too many alerts desensitize responders, so the ones that matter are ignored."
        },
        {
            "name": "Incident Response",
            "kind": "activity",
            "definition": "Detecting, triaging, and resolving an operational incident once an alert fires."
        },
        {
            "name": "Timely Intervention",
            "kind": "capability",
            "definition": "The ability to intervene on a problem quickly by being alerted the moment it arises."
        },
        {
            "name": "Accountability",
            "kind": "capability",
            "definition": "The ability to attribute every consequential action to the actor responsible for it."
        },
        {
            "name": "Opaque Mutation",
            "kind": "anti-pattern",
            "definition": "Changing state with no record of who changed what or when, so the change cannot be audited."
        },
        {
            "name": "Storage/Privacy",
            "kind": "quality-attribute",
            "definition": "The degree to which retaining a full audit history grows storage and raises privacy concerns."
        },
        {
            "name": "Action",
            "distinctFrom": [
                {
                    "id": "lexicon:target",
                    "reason": "The action field names the operation, while the target field names the resource it was performed on."
                },
                {
                    "id": "lexicon:timestamp",
                    "reason": "The action field names the operation, while the timestamp field marks when it happened."
                }
            ],
            "kind": "artifact",
            "definition": "A field of an audit record identifying the operation that was performed."
        },
        {
            "name": "Actor",
            "distinctFrom": [
                {
                    "id": "lexicon:action",
                    "reason": "The actor field names who acted, while the action field names what was done."
                },
                {
                    "id": "lexicon:target",
                    "reason": "The actor field names who acted, while the target field names what was acted on."
                },
                {
                    "id": "lexicon:timestamp",
                    "reason": "The actor field names who acted, while the timestamp field marks when."
                }
            ],
            "kind": "artifact",
            "definition": "A field of an audit record identifying who or what performed an action."
        },
        {
            "name": "Forensics",
            "kind": "capability",
            "definition": "The ability to reconstruct after the fact what happened from an immutable audit record."
        },
        {
            "name": "Privacy",
            "kind": "quality-attribute",
            "definition": "The degree to which a system limits the collection and exposure of personal or sensitive information."
        },
        {
            "name": "Target",
            "kind": "artifact",
            "definition": "A field of an audit record identifying the resource an action was performed on."
        },
        {
            "name": "Timestamp",
            "distinctFrom": [
                {
                    "id": "lexicon:target",
                    "reason": "The timestamp field marks when an action happened, while the target field names what it was performed on."
                }
            ],
            "kind": "artifact",
            "definition": "A field of an audit record marking when an action occurred."
        },
        {
            "name": "Untracked Mutation",
            "kind": "anti-pattern",
            "definition": "Mutating state without writing an audit entry, so the change leaves no trace."
        },
        {
            "name": "Anonymous Flow",
            "kind": "anti-pattern",
            "definition": "Data or requests flowing through a system with no identifiers, so their path cannot be reconstructed."
        },
        {
            "name": "End-to-End Causality",
            "kind": "capability",
            "definition": "The ability to follow a request's cause-and-effect chain across every component it touches."
        },
        {
            "name": "Logs/Traces",
            "kind": "artifact",
            "definition": "The combined log and trace records that let a request be followed from end to end."
        },
        {
            "name": "Metadata Propagation Overhead",
            "kind": "quality-attribute",
            "definition": "The degree to which carrying trace metadata through every call adds size and processing cost."
        },
        {
            "name": "Context Propagation",
            "kind": "constraint",
            "definition": "The requirement that request context be carried across service boundaries so related calls can be correlated."
        },
        {
            "name": "Header/Metadata Management",
            "kind": "quality-attribute",
            "definition": "The degree to which threading correlation identifiers through headers adds handling to every call."
        },
        {
            "name": "Request-Level Traceability",
            "kind": "capability",
            "definition": "The ability to trace all work belonging to one request by a shared correlation identifier."
        },
        {
            "name": "Uncorrelated Events",
            "kind": "anti-pattern",
            "definition": "Emitting events with no shared identifier, so those belonging to one request cannot be tied together."
        },
        {
            "name": "Audit Trail",
            "kind": "artifact",
            "definition": "A chronological record linking each event to the one that caused it."
        },
        {
            "name": "Cause-Effect Reconstruction",
            "kind": "capability",
            "definition": "The ability to reconstruct which event triggered which by following causation identifiers."
        },
        {
            "name": "Event Metadata",
            "kind": "artifact",
            "definition": "Descriptive fields attached to an event, such as its identifiers, timestamps, and causation links."
        },
        {
            "name": "Metadata Verbosity",
            "kind": "quality-attribute",
            "definition": "The degree to which stamping every event with causation metadata makes the event payload verbose."
        },
        {
            "name": "Unlinked Events",
            "kind": "anti-pattern",
            "definition": "Recording events with no link to their cause, so cause-and-effect chains cannot be rebuilt."
        },
        {
            "name": "Latency/Failure Root Cause Analysis",
            "kind": "capability",
            "definition": "The ability to pinpoint which service caused a request's latency or failure by tracing it across hops."
        },
        {
            "name": "Opaque Distributed Calls",
            "kind": "anti-pattern",
            "definition": "Calls crossing service boundaries with no tracing, so a request's path and bottlenecks are invisible."
        },
        {
            "name": "Overhead/Sampling",
            "kind": "quality-attribute",
            "definition": "The degree to which tracing every request adds overhead, forcing sampling that can miss rare cases."
        },
        {
            "name": "Trace Context Propagation",
            "kind": "constraint",
            "definition": "The requirement that trace identifiers be propagated across every hop of a distributed request."
        },
        {
            "name": "Error-Budget Decisions",
            "kind": "capability",
            "definition": "The ability to decide how much risk to take by spending against a defined reliability error budget."
        },
        {
            "name": "Feature Velocity",
            "kind": "quality-attribute",
            "definition": "The degree to which holding to strict reliability targets limits how fast new features can ship."
        },
        {
            "name": "Vague Reliability Goals",
            "kind": "anti-pattern",
            "definition": "Stating reliability aims in vague, unmeasurable terms, so no check can tell whether they are met."
        },
        {
            "name": "At-a-Glance System Health",
            "distinctFrom": [
                {
                    "id": "lexicon:trend-visibility",
                    "reason": "At-a-glance health shows the state now, while trend visibility shows how a metric has moved over time."
                }
            ],
            "kind": "capability",
            "definition": "The ability to see a system's overall health at a glance from a consolidated visual display."
        },
        {
            "name": "Dashboard Sprawl",
            "kind": "quality-attribute",
            "definition": "The degree to which unchecked creation of dashboards scatters attention across too many redundant views."
        },
        {
            "name": "Log-Grep-Only Diagnosis",
            "kind": "anti-pattern",
            "definition": "Diagnosing problems solely by grepping raw logs, with no aggregated view of system health."
        },
        {
            "name": "Trend Visibility",
            "kind": "capability",
            "definition": "The ability to see how a metric is trending over time from a visualized history."
        },
        {
            "name": "Structured Log",
            "kind": "technique",
            "definition": "A technique for writing log entries as typed fields, so they are queried by field instead of by text."
        },
        {
            "name": "Trace Span",
            "kind": "technique",
            "definition": "A technique for recording the start, end and context of an operation as a span within a request's trace."
        },
        {
            "name": "Alert on Critical Failure",
            "kind": "technique",
            "definition": "A technique for raising an alert with an owner and a severity when a critical path fails."
        },
        {
            "name": "Cardinality Reduction",
            "kind": "technique",
            "definition": "A technique for limiting the distinct label values a metric carries, so its storage and queries stay bounded."
        },
        {
            "name": "Sampling and Aggregation",
            "kind": "technique",
            "definition": "A technique for keeping a representative share of telemetry or a summary of it instead of every event."
        },
        {
            "name": "Runbook Owner",
            "kind": "technique",
            "definition": "A technique for naming the person or team responsible for each alert and its response steps."
        },
        {
            "name": "Causation Tracing",
            "kind": "technique",
            "definition": "A technique for carrying the id of the event or request that caused each effect, so a chain can be followed back."
        },
        {
            "name": "Usage Instrumentation",
            "kind": "technique",
            "definition": "A technique for counting calls to a code path in production, so its use is measured before it is removed."
        },
        {
            "name": "Data Change Audit",
            "kind": "technique",
            "definition": "A technique for recording who changed which data, when, and from what value to what value."
        },
        {
            "name": "Execution Log",
            "kind": "technique",
            "definition": "A technique for recording each step an automated or manual procedure performed and its result."
        },
        {
            "name": "Define Signal Quality",
            "kind": "technique",
            "definition": "A technique for stating what makes a log, metric or alert useful before it is emitted."
        },
        {
            "name": "Service Level Objectives",
            "kind": "technique",
            "definition": "A technique for choosing measured indicators of reliability and setting the targets a service must meet on them."
        }
    ]
}