configuration/lexicon/data/recovery.data.json

configuration/lexicon/data/recovery.data.json is a file in GovLab Context. 305 lines of code and 0 definitions.

{
    "category": "self-healing-recovery-deployment-safety",
    "records": [
        {
            "name": "Health Signal",
            "kind": "artifact",
            "definition": "A machine-readable signal that reports whether a component is currently healthy, or that a fault or anomaly in it has been detected.",
            "aliases": ["Detection Signal"]
        },
        {
            "name": "Reduced Mean Time to Recovery",
            "kind": "capability",
            "definition": "The ability to shorten the mean time to recover from a failure by acting automatically."
        },
        {
            "name": "Blind Routing",
            "kind": "anti-pattern",
            "definition": "Routing traffic to instances without checking their health, so requests hit dead or degraded nodes."
        },
        {
            "name": "Observable Health Criteria",
            "kind": "constraint",
            "definition": "The requirement that explicit, observable criteria define when a component counts as healthy."
        },
        {
            "name": "Readiness/Liveness Routing",
            "kind": "capability",
            "definition": "The ability to route traffic only to instances that report themselves ready and alive."
        },
        {
            "name": "Continuity During Failure",
            "kind": "capability",
            "definition": "The ability to keep serving requests by switching to a standby when the primary fails."
        },
        {
            "name": "Health Detection",
            "kind": "capability",
            "definition": "The ability to detect that a component has failed so a switchover can be triggered."
        },
        {
            "name": "Single Instance Dependency",
            "kind": "anti-pattern",
            "definition": "Depending on a single instance with no standby, so its failure takes down the whole service."
        },
        {
            "name": "Replication or Alternate Capacity",
            "kind": "constraint",
            "definition": "The requirement that duplicate copies or spare capacity exist to take over when a component fails."
        },
        {
            "name": "Consistency Lag",
            "kind": "quality-attribute",
            "definition": "The degree to which replicas trail the primary, so reads served from them may return stale data."
        },
        {
            "name": "Consistency Policy",
            "kind": "constraint",
            "definition": "The requirement that a defined policy specify how and when replicas converge to a consistent state."
        },
        {
            "name": "Read Scaling",
            "kind": "capability",
            "definition": "The ability to serve more read traffic by distributing it across replicas."
        },
        {
            "name": "Single Copy State",
            "kind": "anti-pattern",
            "definition": "Keeping only one copy of state, so its loss or unavailability takes down the whole system."
        },
        {
            "name": "Cost/Cold Start",
            "kind": "quality-attribute",
            "definition": "The degree to which scaling capacity up and down incurs added cost and cold-start latency."
        },
        {
            "name": "Fixed Capacity",
            "kind": "approach",
            "definition": "Provisioning a fixed, preset amount of capacity, rather than adapting it to demand."
        },
        {
            "name": "Horizontal Scalability",
            "kind": "quality-attribute",
            "definition": "The degree to which a system can grow by adding more interchangeable instances rather than enlarging one."
        },
        {
            "name": "Incident Reduction",
            "distinctFrom": [
                {
                    "id": "lexicon:reduced-mean-time-to-recovery",
                    "reason": "Incident reduction lowers how many incidents reach a responder, while reduced mean time to recovery shortens each one that happens."
                }
            ],
            "kind": "capability",
            "definition": "The ability to reduce the number of incidents that reach human responders by fixing them automatically."
        },
        {
            "name": "Manual Remediation",
            "kind": "approach",
            "definition": "Recovering from incidents through human-operated fixes, rather than automated remediation."
        },
        {
            "name": "Remediation Workflow",
            "kind": "activity",
            "definition": "A defined corrective action, or sequence of steps, carried out to return a system to a healthy state after a fault is detected.",
            "aliases": ["Remediation Action"]
        },
        {
            "name": "Unsafe Automation",
            "kind": "quality-attribute",
            "definition": "The degree to which automated remediation may take wrong or harmful corrective actions, for example on a faulty signal, faster than the developer can intervene.",
            "aliases": [
                "Automation Risk",
                "False Recovery Actions"
            ]
        },
        {
            "name": "Data Migration Compatibility",
            "kind": "quality-attribute",
            "definition": "The degree to which reverting code is constrained by forward data migrations that cannot easily be undone."
        },
        {
            "name": "Fast Failure Recovery",
            "kind": "capability",
            "definition": "The ability to recover quickly from a bad release by reverting to the last good version."
        },
        {
            "name": "Irreversible Deployment",
            "distinctFrom": [
                {
                    "id": "architecture:irreversible-migration",
                    "reason": "An irreversible deployment cannot roll back a release, while an irreversible migration cannot roll back a schema or data change."
                }
            ],
            "kind": "anti-pattern",
            "definition": "Deploying in a way that cannot be undone, so a bad release cannot be rolled back."
        },
        {
            "name": "Reversible Deployment",
            "kind": "constraint",
            "definition": "The requirement that a deployment be structured so it can be safely reverted to a prior version."
        },
        {
            "name": "Versioned Artifact",
            "kind": "artifact",
            "definition": "A build artifact tagged with a distinct version so a prior one can be redeployed."
        },
        {
            "name": "In-Place Mutation Only",
            "kind": "anti-pattern",
            "definition": "Upgrading by mutating the running environment in place, with no parallel target to cut over to or fall back from."
        },
        {
            "name": "Infrastructure Cost",
            "kind": "quality-attribute",
            "definition": "The degree to which running two full parallel environments doubles infrastructure cost during a cutover."
        },
        {
            "name": "Low-Risk Cutover",
            "kind": "capability",
            "definition": "The ability to switch traffic to a new version with low risk by keeping the old one ready to fall back to."
        },
        {
            "name": "Parallel Environments",
            "kind": "constraint",
            "definition": "The requirement that two full production-equivalent environments run side by side for cutover."
        },
        {
            "name": "Controlled Exposure",
            "kind": "capability",
            "definition": "The ability to expose a new version to a small, controlled fraction of traffic first."
        },
        {
            "name": "Progressive Delivery",
            "kind": "approach",
            "definition": "A release strategy that rolls out changes gradually to widening audiences while monitoring for regressions."
        },
        {
            "name": "Rollout Complexity",
            "kind": "quality-attribute",
            "definition": "The degree to which staging a release in gradual increments adds orchestration complexity."
        },
        {
            "name": "Traffic Splitting",
            "kind": "mechanism",
            "definition": "A facility that routes a configurable proportion of traffic to different versions of a service."
        },
        {
            "name": "Empirical Resilience Verification",
            "kind": "capability",
            "definition": "The ability to verify a system's resilience empirically by injecting faults and observing recovery."
        },
        {
            "name": "Production Risk",
            "kind": "quality-attribute",
            "definition": "The degree to which deliberately injecting faults in production risks causing user-facing incidents."
        },
        {
            "name": "Untested Failure Assumptions",
            "kind": "anti-pattern",
            "definition": "Assuming a system will survive failures without ever testing those assumptions against injected faults."
        },
        {
            "name": "Connection Cleanup",
            "kind": "capability",
            "definition": "The ability to close open connections cleanly when a process shuts down."
        },
        {
            "name": "Hard Process Kill",
            "kind": "anti-pattern",
            "definition": "Terminating a process abruptly without draining work, dropping in-flight requests and risking corrupt state."
        },
        {
            "name": "In-Flight Work Drain",
            "distinctFrom": [
                {
                    "id": "lexicon:connection-cleanup",
                    "reason": "Draining finishes or hands off work in progress, while connection cleanup closes the open connections."
                }
            ],
            "kind": "capability",
            "definition": "The ability to finish or safely hand off in-progress work before a process exits."
        },
        {
            "name": "Lifecycle Signals",
            "kind": "constraint",
            "definition": "The requirement that a process receive lifecycle signals telling it when to start draining and stop."
        },
        {
            "name": "Shutdown Latency",
            "kind": "quality-attribute",
            "definition": "The degree to which draining in-flight work before exit lengthens the time a shutdown takes."
        },
        {
            "name": "Disk-Failure Survival",
            "distinctFrom": [
                {
                    "id": "lexicon:parity-based-recovery",
                    "reason": "Disk-failure survival is continuing to serve data after a disk is lost, while parity-based recovery is one way the lost data is rebuilt."
                }
            ],
            "kind": "capability",
            "definition": "The ability to keep serving data after one or more disks fail, by reconstructing from redundancy."
        },
        {
            "name": "Multiple Physical Disks",
            "kind": "constraint",
            "definition": "The requirement that data span several physical disks so redundancy can survive a single-disk loss."
        },
        {
            "name": "Parity-Based Recovery",
            "kind": "capability",
            "definition": "The ability to reconstruct lost data from parity information stored across the disk array."
        },
        {
            "name": "Single-Disk Point of Failure",
            "kind": "anti-pattern",
            "definition": "Storing data on a single disk with no redundancy, so that one disk's failure loses everything."
        },
        {
            "name": "Write Amplification",
            "kind": "quality-attribute",
            "definition": "The degree to which maintaining parity on writes multiplies the underlying disk writes for each logical write."
        },
        {
            "name": "Recovery Policy",
            "kind": "technique",
            "definition": "A technique for declaring, per failure class, the action the system takes to restore service and who is told."
        },
        {
            "name": "Expand-Contract Migration",
            "kind": "technique",
            "definition": "A technique for changing a schema in steps, adding the new shape, moving every reader and writer, then removing the old shape."
        },
        {
            "name": "Dual Read and Write",
            "kind": "technique",
            "definition": "A technique for writing to both the old and the new store and comparing reads during a migration, until the new store is trusted."
        },
        {
            "name": "Rollback Plan",
            "kind": "technique",
            "definition": "A technique for stating before a release how it is undone and testing that the undo works."
        },
        {
            "name": "Small-Batch Release",
            "kind": "technique",
            "definition": "A technique for shipping changes in small increments, so each release has a small blast radius."
        },
        {
            "name": "Backup",
            "kind": "technique",
            "definition": "A technique for copying data to separate storage on a schedule, so it can be restored after loss or corruption."
        },
        {
            "name": "Automate the Runbook",
            "kind": "technique",
            "definition": "A technique for turning repeated manual operational steps into a script the system runs and logs."
        },
        {
            "name": "Autoscaling Policy",
            "kind": "technique",
            "definition": "A technique for declaring the load signals and limits by which the number of instances grows and shrinks."
        }
    ]
}