configuration/principle/data/observability.data.json
configuration/principle/data/observability.data.json is a file in GovLab Context. 516 lines of code and 0 definitions.
{
"category": "Observability / Auditability / Traceability",
"check": {
"population": "every critical path, sensitive operation, request and message the telemetry is meant to cover",
"freshness": "a verdict stands until a code path, a telemetry field or an alert threshold changes",
"refusal": "the logging, tracing or audit policy check fails a change that leaves a covered path without its signal",
"observation": "emitted logs, metrics, spans and audit records, compared with the paths that should emit them",
"evidence": "none: the catalog states this check as a class, so a watched run belongs to each system that adopts it",
"authority": "the declared telemetry coverage, which the emitted signals are compared against"
},
"records": [
{
"id": "observability",
"distinctFrom": [
{
"id": "architecture:fault-tolerance",
"reason": "Observability is seeing internal behavior, while fault tolerance is operating correctly through component failure."
},
{
"id": "architecture:auditability",
"reason": "Observability infers internal behavior from signals, while auditability ties sensitive actions to their actors afterwards."
},
{
"id": "architecture:traceability",
"reason": "Observability is the whole view from logs, metrics and traces, while traceability is following one request or change end to end."
},
{
"id": "lexicon:cost-noise",
"reason": "Observability is the insight telemetry gives, while cost and noise are what collecting more of it adds."
},
{
"id": "lexicon:debuggability",
"reason": "Observability is inferring behavior in production, while debuggability is how easily the developer locates a fault."
},
{
"id": "lexicon:fault-isolation",
"reason": "Observability is seeing a failure, while fault isolation is keeping it from spreading."
}
],
"name": "Observability",
"definition": "The degree to which a system's internal behavior in production can be inferred from the logs, metrics and traces it emits.",
"type": "quality-attribute",
"scope": [
"service",
"system",
"runtime"
],
"requires": [
"Logs",
"Metrics",
"Traces"
],
"reinforces": [
"Resilience",
"Debuggability"
],
"enables": ["Incident Diagnosis"],
"conflicts_with": [
"Opaque System",
"Unobservable Failure"
],
"tensions_with": ["Cost/Noise"],
"violated_by": ["lexicon:blind-operation"],
"detected_by": ["missing telemetry around critical paths"],
"measured_by": [
"telemetry coverage",
"MTTR"
],
"refactored_by": [
"lexicon:structured-log",
"lexicon:metrics",
"architecture:distributed-tracing"
],
"enforced_by": ["observability standards"],
"severity": "contextual",
"mandatoryFor": "production systems",
"exemplar": {
"before": "async function processFoo(foo: Foo) {\n await fooStore.save(foo);\n}",
"after": "async function processFoo(foo: Foo, telemetry: Telemetry) {\n return telemetry.trace(\"foo.process\", { fooId: foo.id }, async span => {\n await fooStore.save(foo);\n span.event(\"foo.saved\");\n telemetry.count(\"foo.processed\", 1);\n });\n}",
"lang": "ts"
}
},
{
"id": "logging",
"name": "Logging",
"definition": "A mechanism that records structured events with their context as the code runs.",
"type": "mechanism",
"scope": [
"application",
"service"
],
"requires": [
"Structured Events",
"Context"
],
"reinforces": [
"Traceability",
"Debugging"
],
"enables": ["Incident Analysis"],
"conflicts_with": [
"Silent Failure",
"Log-as-Control-Flow"
],
"tensions_with": ["Noise/Personal Data Leakage"],
"violated_by": ["architecture:unobservable-failure"],
"detected_by": ["absence of logs on error/business events"],
"measured_by": [
"log coverage",
"signal/noise ratio"
],
"refactored_by": [
"lexicon:structured-log",
"lexicon:pass-context-explicitly"
],
"enforced_by": [
"logging policy",
"linting"
],
"severity": "mandatory",
"exemplar": {
"before": "console.log(\"saved\", foo);",
"after": "logger.info(\"foo.saved\", { fooId: foo.id, version: foo.version });",
"lang": "ts"
}
},
{
"id": "monitoring",
"distinctFrom": [
{
"id": "architecture:alerting",
"reason": "Monitoring collects and compares metrics over time, while alerting notifies a responder when a condition holds past its threshold."
},
{
"id": "lexicon:metrics",
"reason": "Monitoring is the mechanism that collects and compares, while metrics are the measurements it collects."
},
{
"id": "lexicon:guardrails",
"reason": "Monitoring observes behavior after the fact, while guardrails bound what a model may output or do as it runs."
}
],
"name": "Monitoring",
"definition": "A mechanism that collects metrics about resources and service levels over time and compares them with thresholds.",
"type": "mechanism",
"scope": [
"service",
"infrastructure"
],
"requires": [
"Metrics",
"Thresholds"
],
"reinforces": ["Reliability"],
"enables": ["Failure Detection"],
"conflicts_with": ["Blind Operation"],
"tensions_with": ["Alert Noise"],
"violated_by": ["lexicon:blind-operation"],
"detected_by": ["missing dashboards/SLI metrics"],
"measured_by": [
"metric coverage",
"detection latency"
],
"refactored_by": [
"lexicon:metrics",
"lexicon:service-level-objectives"
],
"enforced_by": ["production readiness checklist"],
"severity": "mandatory",
"exemplar": {
"before": "setInterval(() => report(fooQueue.length), 60_000);",
"after": "metrics.gauge(\"foo.queue.depth\", () => fooQueue.length);\nmetrics.histogram(\"foo.processing.duration_ms\", fooProcessingDuration);",
"lang": "ts"
}
},
{
"id": "alerting",
"name": "Alerting",
"definition": "A mechanism that notifies an on-call responder when a monitored condition holds for longer than its threshold.",
"type": "mechanism",
"scope": [
"operations",
"service"
],
"requires": [
"Monitoring",
"Thresholds"
],
"reinforces": ["Incident Response"],
"enables": ["Timely Intervention"],
"conflicts_with": [
"Silent Failure",
"Observability Noise"
],
"tensions_with": ["Alert Fatigue"],
"violated_by": ["architecture:unobservable-failure"],
"detected_by": ["incident reported by customers before monitoring caught it"],
"measured_by": [
"MTTD",
"alert precision"
],
"refactored_by": [
"lexicon:alert-on-critical-failure",
"lexicon:track-rule-metrics"
],
"enforced_by": ["on-call policy"],
"severity": "mandatory",
"exemplar": {
"before": "if (errorRate > 0.05) notify(\"foo errors\");",
"after": "alerts.define(\"FooErrorBudgetBurn\", {\n condition: rate(\"foo.errors\", \"5m\") > 0.05,\n for: \"10m\",\n severity: \"critical\",\n});",
"lang": "ts"
}
},
{
"id": "auditability",
"distinctFrom": [
{
"id": "architecture:traceability",
"reason": "Auditability ties each sensitive action to its actor, target, time and reason, while traceability follows a request or change end to end."
},
{
"id": "lexicon:storage-privacy",
"reason": "Auditability is keeping the history, while storage and privacy are the costs of keeping it."
},
{
"id": "architecture:reproducibility",
"reason": "Auditability records what happened, while reproducibility reruns it and gets the same output."
}
],
"name": "Auditability",
"definition": "The degree to which each sensitive action can be traced afterwards to its actor, target, time and reason.",
"type": "quality-attribute",
"scope": [
"system",
"data",
"security"
],
"requires": [
"Audit Logging",
"Traceability"
],
"reinforces": ["Compliance"],
"enables": ["Accountability"],
"conflicts_with": ["Opaque Mutation"],
"tensions_with": ["Storage/Privacy"],
"violated_by": ["lexicon:untracked-mutation"],
"detected_by": ["missing audit event for sensitive operation"],
"measured_by": ["audit event coverage"],
"refactored_by": [
"architecture:audit-logging",
"lexicon:data-change-audit"
],
"enforced_by": ["compliance gates"],
"severity": "contextual",
"mandatoryFor": "regulated or sensitive systems",
"exemplar": {
"before": "function renameFoo(foo: Foo, name: string) { foo.name = name; }",
"after": "function renameFoo(foo: Foo, name: string, actor: Actor) {\n const previous = foo.name;\n foo.rename(name);\n audit.append({ action: \"FooRenamed\", fooId: foo.id, actorId: actor.id, previous, next: name });\n}",
"lang": "ts"
}
},
{
"id": "audit-logging",
"name": "Audit Logging",
"definition": "A mechanism that appends an immutable record of each sensitive operation, naming the actor, the action, the target and the time.",
"type": "mechanism",
"scope": [
"security",
"compliance",
"data mutation"
],
"requires": [
"Actor",
"Action",
"Timestamp",
"Target"
],
"reinforces": [
"Auditability",
"Traceability"
],
"enables": ["Forensics"],
"conflicts_with": ["Untracked Mutation"],
"tensions_with": ["Privacy"],
"violated_by": ["lexicon:opaque-mutation"],
"detected_by": ["missing audit instrumentation"],
"measured_by": ["audit coverage"],
"refactored_by": ["architecture:append-only-log"],
"enforced_by": [
"policy-as-code",
"tests"
],
"severity": "contextual",
"mandatoryFor": "sensitive systems",
"exemplar": {
"before": "logger.info(`user ${user.id} changed foo ${foo.id}`);",
"after": "auditLog.append({\n eventId: eventId(),\n action: \"FOO_UPDATE\",\n actorId: user.id,\n resourceId: foo.id,\n occurredAt: clock.now().toISOString(),\n});",
"lang": "ts"
}
},
{
"id": "traceability",
"distinctFrom": [
{
"id": "lexicon:metadata-propagation-overhead",
"reason": "Traceability is following a request end to end, while propagation overhead is the size and processing that carrying its metadata costs."
}
],
"name": "Traceability",
"definition": "The degree to which a request, workflow or change can be followed end to end through correlated records.",
"type": "quality-attribute",
"scope": [
"request",
"workflow",
"change"
],
"requires": [
"Correlation ID",
"Logs/Traces"
],
"reinforces": [
"Debuggability",
"Auditability"
],
"enables": ["End-to-End Causality"],
"conflicts_with": ["Anonymous Flow"],
"tensions_with": ["Metadata Propagation Overhead"],
"violated_by": ["lexicon:uncorrelated-events"],
"detected_by": ["missing correlation propagation"],
"measured_by": ["trace completeness"],
"refactored_by": [
"architecture:correlation-id",
"architecture:distributed-tracing"
],
"enforced_by": [
"middleware",
"tracing policy"
],
"severity": "mandatory",
"exemplar": {
"before": "await processFoo(foo);",
"after": "const trace = traceContext.start({ operation: \"processFoo\", fooId: foo.id });\nawait processFoo(foo, trace);\ntrace.finish();",
"lang": "ts"
}
},
{
"id": "correlation-id",
"distinctFrom": [
{
"id": "architecture:distributed-tracing",
"reason": "A correlation id is one identifier carried through a request, while distributed tracing records each call as a timed span of one trace."
}
],
"name": "Correlation ID",
"definition": "A mechanism that attaches one identifier to a request and carries it through every call, message and log line the request causes.",
"type": "mechanism",
"scope": [
"request",
"message",
"workflow"
],
"requires": ["Context Propagation"],
"reinforces": ["Distributed Tracing"],
"enables": ["Request-Level Traceability"],
"conflicts_with": ["Uncorrelated Events"],
"tensions_with": ["Header/Metadata Management"],
"violated_by": ["lexicon:uncorrelated-events"],
"detected_by": ["missing correlation field"],
"measured_by": ["correlation coverage"],
"refactored_by": ["architecture:chain-of-responsibility-pattern"],
"enforced_by": ["logging/tracing standards"],
"severity": "contextual",
"mandatoryFor": "distributed systems",
"exemplar": {
"before": "await http.post(\"/bar\", { fooId: foo.id });",
"after": "const correlationId = request.headers.get(\"X-Correlation-ID\") ?? idSource.next();\nawait http.post(\"/bar\", { fooId: foo.id }, { headers: { \"X-Correlation-ID\": correlationId } });",
"lang": "ts"
}
},
{
"id": "causation-id",
"name": "Causation ID",
"definition": "A mechanism that stamps each event with the identifier of the event that caused it.",
"type": "mechanism",
"scope": [
"event",
"message",
"workflow"
],
"requires": ["Event Metadata"],
"reinforces": [
"Causality",
"Audit Trail"
],
"enables": ["Cause-Effect Reconstruction"],
"conflicts_with": ["Unlinked Events"],
"tensions_with": ["Metadata Verbosity"],
"violated_by": ["lexicon:unlinked-events"],
"detected_by": ["missing causation field in event metadata"],
"measured_by": ["causation coverage"],
"refactored_by": [],
"enforced_by": ["event schema rules"],
"severity": "recommended",
"exemplar": {
"before": "events.publish({ id: eventId(), type: \"BarCreated\", fooId: event.fooId });",
"after": "events.publish({\n id: eventId(),\n type: \"BarCreated\",\n fooId: event.fooId,\n correlationId: event.correlationId,\n causationId: event.id,\n});",
"lang": "ts"
}
},
{
"id": "distributed-tracing",
"name": "Distributed Tracing",
"definition": "A mechanism that propagates a trace context across service calls and records each call as a timed span of one trace.",
"type": "mechanism",
"scope": [
"distributed system",
"service mesh"
],
"requires": ["Trace Context Propagation"],
"reinforces": [
"Observability",
"Causality"
],
"enables": ["Latency/Failure Root Cause Analysis"],
"conflicts_with": ["Opaque Distributed Calls"],
"tensions_with": ["Overhead/Sampling"],
"violated_by": ["lexicon:opaque-distributed-calls"],
"detected_by": [
"broken traces",
"missing spans"
],
"measured_by": [
"trace completeness",
"span coverage"
],
"refactored_by": [],
"enforced_by": ["observability policy"],
"severity": "contextual",
"mandatoryFor": "distributed systems",
"exemplar": {
"before": "await fooService.call();\nawait barService.call();",
"after": "await tracer.span(\"foo.request\", async span => {\n await fooService.call({ traceparent: span.traceparent });\n await barService.call({ traceparent: span.traceparent });\n});",
"lang": "ts"
}
},
{
"id": "slo-sli",
"name": "SLO/SLI",
"aliases": [
"Service Level Objective",
"Service Level Indicator",
"Objective Reliability Targets"
],
"definition": "A rule or precondition that a service's reliability is stated as a measured indicator with an objective and an error budget over a window.",
"type": "constraint",
"scope": [
"service",
"reliability",
"operations"
],
"requires": ["Monitoring"],
"reinforces": [
"Observability",
"Alerting"
],
"enables": ["Error-Budget Decisions"],
"conflicts_with": ["Vague Reliability Goals"],
"tensions_with": ["Feature Velocity"],
"violated_by": ["lexicon:vague-reliability-goals"],
"detected_by": ["no measured indicator behind reliability claims"],
"measured_by": ["SLO attainment vs error budget"],
"refactored_by": ["lexicon:service-level-objectives"],
"enforced_by": ["reliability review"],
"severity": "contextual",
"mandatoryFor": "production systems",
"exemplar": {
"before": "alert.when(latency > 1000);",
"after": "const fooLatencySli = ratio(\"foo.requests.fast\", \"foo.requests.total\");\ndefineSLO(\"foo-latency\", { sli: fooLatencySli, objective: 0.99, window: \"30d\", errorBudget: 0.01 });",
"lang": "ts"
}
},
{
"id": "dashboards",
"name": "Dashboards",
"definition": "Descriptive data about a system's key signals, arranged as panels that show health and trends at a glance.",
"type": "artifact",
"scope": [
"service",
"operations",
"visibility"
],
"requires": ["Monitoring"],
"reinforces": [
"Observability",
"Traceability"
],
"enables": [
"At-a-Glance System Health",
"Trend Visibility"
],
"conflicts_with": ["Log-Grep-Only Diagnosis"],
"tensions_with": ["Dashboard Sprawl"],
"violated_by": ["lexicon:log-grep-only-diagnosis"],
"detected_by": ["no curated view of key signals"],
"measured_by": ["time-to-diagnose during incidents"],
"refactored_by": ["architecture:monitoring"],
"enforced_by": ["operations review"],
"severity": "recommended",
"exemplar": {
"before": "grepLogsForFooErrors();",
"after": "const fooDashboard = dashboard(\"foo-health\", {\n panels: [rate(\"foo.errors\"), histogram(\"foo.latency\"), gauge(\"foo.queue.depth\")],\n});",
"lang": "ts"
}
}
]
}