configuration/principle/data/performance.data.json

configuration/principle/data/performance.data.json is a file in GovLab Context. 1067 lines of code and 0 definitions.

{
    "category": "Scalability / Performance / Optimization",
    "check": {
        "population": "every performance-sensitive path, resource and scaling policy under a declared workload",
        "freshness": "a verdict stands for the measured workload and build, and goes stale when either changes",
        "refusal": "the load test, benchmark or SLO gate fails the change that breaks its budget",
        "observation": "profiles, benchmark results, load-test metrics and production telemetry measured against their budgets",
        "evidence": "none: the catalog states this check as a class, so a watched run belongs to each system that adopts it",
        "authority": "the declared budget, which every measurement is compared against"
    },
    "records": [
        {
            "id": "scalability",
            "distinctFrom": [
                {
                    "id": "architecture:elasticity",
                    "reason": "Scalability is keeping throughput as load grows by adding resources, while elasticity is those resources following demand automatically in both directions."
                },
                {
                    "id": "architecture:low-coupling",
                    "reason": "Scalability is growing with load, while low coupling is changing without ripple."
                },
                {
                    "id": "architecture:memory-efficiency",
                    "reason": "Scalability is growth by adding resources, while memory efficiency is bounded use of the resources one process has."
                },
                {
                    "id": "architecture:resilience",
                    "reason": "Scalability is holding performance under load, while resilience is containing and recovering from failure."
                },
                {
                    "id": "architecture:testability",
                    "reason": "Scalability is growing with load, while testability is running code in isolation."
                },
                {
                    "id": "lexicon:cost-efficiency",
                    "reason": "Scalability is keeping throughput as load grows, while cost efficiency is the resource cost of the work."
                },
                {
                    "id": "lexicon:resource-efficiency",
                    "reason": "Scalability adds resources to meet load, while resource efficiency uses as few as possible for the same work."
                },
                {
                    "id": "lexicon:simplicity",
                    "reason": "Scalability is growth with load, while simplicity is the absence of the structure that growth tends to add."
                },
                {
                    "id": "lexicon:availability",
                    "reason": "Scalability is performance under growing load, while availability is the share of time the system serves at all."
                }
            ],
            "name": "Scalability",
            "definition": "The degree to which a system keeps its throughput and latency as load grows, by adding resources.",
            "type": "quality-attribute",
            "scope": [
                "service",
                "system",
                "infrastructure"
            ],
            "requires": [
                "Load Model",
                "Bottleneck Awareness"
            ],
            "reinforces": ["Performance Engineering"],
            "enables": ["Growth Handling"],
            "conflicts_with": ["Fixed-Capacity Design"],
            "tensions_with": [
                "Simplicity",
                "Consistency"
            ],
            "violated_by": ["lexicon:bottlenecks"],
            "detected_by": ["saturation under load test"],
            "measured_by": ["throughput under increasing load"],
            "refactored_by": [
                "architecture:caching",
                "architecture:partitioning",
                "lexicon:introduce-async-event",
                "architecture:horizontal-scaling"
            ],
            "enforced_by": [
                "load tests",
                "SLO gates"
            ],
            "severity": "contextual",
            "exemplar": {
                "before": "class FooServer {\n  private readonly foos = new Map<FooId, Foo>();\n  handle(request: FooRequest) { return processFoo(request, this.foos); }\n}",
                "after": "class FooServer {\n  constructor(private readonly store: DistributedFooStore) {}\n  handle(request: FooRequest) { return processFoo(request, this.store); }\n}",
                "lang": "ts"
            }
        },
        {
            "id": "horizontal-scaling",
            "name": "Horizontal Scaling",
            "definition": "A technique for adding capacity by running more stateless instances of a service behind a load balancer.",
            "type": "technique",
            "scope": [
                "service",
                "infrastructure"
            ],
            "requires": ["Statelessness or Shared State Strategy"],
            "reinforces": [
                "Elasticity",
                "Availability"
            ],
            "enables": ["Scale-Out"],
            "conflicts_with": ["Instance-Local State"],
            "tensions_with": ["Distributed Coordination"],
            "violated_by": ["lexicon:instance-affinity"],
            "detected_by": ["local session/state coupling"],
            "measured_by": ["scale-out efficiency"],
            "refactored_by": [
                "lexicon:externalize-session-state",
                "architecture:load-balancing"
            ],
            "enforced_by": ["deployment tests"],
            "severity": "contextual",
            "exemplar": {
                "before": "deployFoo({ replicas: 1, cpu: 32, memoryGb: 128 });",
                "after": "deployFoo({ replicas: 12, cpu: 2, memoryGb: 4, stateless: true });",
                "lang": "ts"
            }
        },
        {
            "id": "vertical-scaling",
            "name": "Vertical Scaling",
            "definition": "A technique for adding capacity by giving one instance more CPU, memory or I/O.",
            "type": "technique",
            "scope": [
                "infrastructure",
                "process"
            ],
            "requires": ["Resource Headroom"],
            "reinforces": ["Simplicity"],
            "enables": ["Capacity Increase without Distribution"],
            "conflicts_with": ["Hard Resource Ceiling"],
            "tensions_with": ["Cost/Limit"],
            "violated_by": ["lexicon:hard-resource-ceiling"],
            "detected_by": ["resource saturation trends"],
            "measured_by": ["utilization/headroom"],
            "refactored_by": [
                "architecture:profiling",
                "architecture:horizontal-scaling"
            ],
            "enforced_by": ["capacity planning"],
            "severity": "contextual",
            "exemplar": {
                "before": "deployFoo({ cpu: 1, memoryGb: 1 });\nqueueFooWhenSaturated();",
                "after": "deployFoo({ cpu: 8, memoryGb: 32 });\nverifyFooCapacity({ targetConcurrency: 200 });",
                "lang": "ts"
            }
        },
        {
            "id": "elasticity",
            "distinctFrom": [
                {
                    "id": "lexicon:availability",
                    "reason": "Elasticity is capacity following demand, while availability is the share of time the system can serve."
                },
                {
                    "id": "lexicon:cost-efficiency",
                    "reason": "Elasticity is capacity following demand, while cost efficiency is the work done per unit of resource cost."
                },
                {
                    "id": "lexicon:warm-up-latency",
                    "reason": "Elasticity adds capacity when demand rises, while warm-up latency is the delay before that capacity can serve."
                }
            ],
            "name": "Elasticity",
            "definition": "The degree to which a system's provisioned capacity follows demand up and down automatically.",
            "type": "quality-attribute",
            "scope": [
                "deployment",
                "infrastructure"
            ],
            "requires": [
                "Auto-Scaling",
                "Metrics"
            ],
            "reinforces": [
                "Scalability",
                "Cost Efficiency"
            ],
            "enables": ["Dynamic Capacity"],
            "conflicts_with": ["Fixed Provisioning"],
            "tensions_with": ["Warm-Up Latency"],
            "violated_by": ["lexicon:fixed-provisioning"],
            "detected_by": ["under/over-provisioning patterns"],
            "measured_by": [
                "scale response time",
                "utilization"
            ],
            "refactored_by": [
                "lexicon:autoscaling-policy",
                "lexicon:externalize-session-state"
            ],
            "enforced_by": ["infrastructure policy"],
            "severity": "contextual",
            "exemplar": {
                "before": "deployFooWorkers({ replicas: 10 });",
                "after": "deployFooWorkers({\n  minReplicas: 2,\n  maxReplicas: 50,\n  target: { queueDepthPerReplica: 100 },\n});",
                "lang": "ts"
            }
        },
        {
            "id": "load-balancing",
            "name": "Load Balancing",
            "definition": "A mechanism that spreads incoming requests across healthy instances of a service.",
            "type": "mechanism",
            "scope": [
                "traffic",
                "service"
            ],
            "requires": [
                "Multiple Targets",
                "Health Checks"
            ],
            "reinforces": [
                "Availability",
                "Scalability"
            ],
            "enables": ["Traffic Distribution"],
            "conflicts_with": ["Single Target Routing"],
            "tensions_with": ["Session Affinity"],
            "violated_by": ["lexicon:single-target-routing"],
            "detected_by": ["skewed instance utilization"],
            "measured_by": [
                "request distribution",
                "latency"
            ],
            "refactored_by": ["lexicon:externalize-session-state"],
            "enforced_by": ["infrastructure config checks"],
            "severity": "contextual",
            "exemplar": {
                "before": "const endpoint = fooServers[0];\nendpoint.handle(request);",
                "after": "const endpoint = fooLoadBalancer.next({ key: request.fooId });\nendpoint.handle(request);",
                "lang": "ts"
            }
        },
        {
            "id": "sharding",
            "name": "Sharding",
            "definition": "A design pattern that splits a dataset across separate stores by a shard key, and routes each request to the shard that holds its key.",
            "type": "pattern",
            "scope": [
                "database",
                "storage",
                "messaging"
            ],
            "requires": ["Partition Key"],
            "reinforces": ["Horizontal Scaling"],
            "enables": ["Large Dataset Scaling"],
            "conflicts_with": ["Single Monolithic Store"],
            "tensions_with": ["Cross-Shard Queries"],
            "violated_by": ["lexicon:single-monolithic-store"],
            "detected_by": [
                "hotspot partitions",
                "storage bottleneck"
            ],
            "measured_by": [
                "shard balance",
                "query fan-out"
            ],
            "refactored_by": ["architecture:partitioning"],
            "enforced_by": ["data architecture review"],
            "severity": "contextual",
            "exemplar": {
                "before": "const foo = await singleFooDatabase.find(id);",
                "after": "const shard = fooShardMap.resolve(id);\nconst foo = await shard.find(id);",
                "lang": "ts"
            }
        },
        {
            "id": "partitioning",
            "distinctFrom": [
                {
                    "id": "architecture:parallelism",
                    "reason": "Partitioning divides data or work by a key, while parallelism runs the independent parts at the same time."
                }
            ],
            "name": "Partitioning",
            "definition": "A technique for dividing data or work into independent partitions by a key, so each can be processed in parallel.",
            "type": "technique",
            "scope": [
                "data",
                "workload",
                "service"
            ],
            "requires": ["Partition Strategy"],
            "reinforces": [
                "Scalability",
                "Isolation"
            ],
            "enables": ["Parallelism"],
            "conflicts_with": ["Global Shared State"],
            "tensions_with": ["Rebalancing Complexity"],
            "violated_by": ["lexicon:single-monolithic-store"],
            "detected_by": ["hotspot resource usage"],
            "measured_by": ["partition balance"],
            "refactored_by": [],
            "enforced_by": ["architecture review"],
            "severity": "contextual",
            "exemplar": {
                "before": "const events = await fooLog.readAll();",
                "after": "const partition = hash(fooId) % partitionCount;\nconst events = await fooLog.readPartition(partition);",
                "lang": "ts"
            }
        },
        {
            "id": "caching",
            "name": "Caching",
            "definition": "A design pattern that stores the result of an expensive read or computation under a key derived from every input it depends on, and serves it again while that key matches.",
            "type": "pattern",
            "scope": [
                "data access",
                "computation",
                "API"
            ],
            "requires": ["Invalidation Policy"],
            "reinforces": [
                "Latency Reduction",
                "Scalability"
            ],
            "enables": ["Reduced Load"],
            "conflicts_with": ["Cache Poisoning by Design"],
            "tensions_with": [
                "Consistency",
                "Always-Fresh Reads"
            ],
            "violated_by": ["lexicon:repeated-stable-computation"],
            "detected_by": [
                "hot repeated reads",
                "high latency calls"
            ],
            "measured_by": [
                "hit ratio",
                "stale read rate"
            ],
            "refactored_by": [
                "lexicon:context-keyed-cache",
                "lexicon:cache-invalidation-on-write"
            ],
            "enforced_by": ["performance tests"],
            "severity": "contextual",
            "exemplar": {
                "before": "async function loadFoo(id: FooId) { return fooStore.find(id); }",
                "after": "async function loadFoo(id: FooId) {\n  const key = fingerprint(id, await fooStore.versionOf(id));\n  const cached = await fooCache.get(key);\n  if (cached) return cached;\n  const foo = await fooStore.find(id);\n  if (foo) await fooCache.set(key, foo);\n  return foo;\n}",
                "lang": "ts"
            }
        },
        {
            "id": "statelessness",
            "name": "Statelessness",
            "definition": "A design rule that a handler keeps no request state between calls, so any instance can serve any request.",
            "type": "principle",
            "scope": [
                "service",
                "process",
                "handler"
            ],
            "requires": ["Externalized State"],
            "reinforces": [
                "Horizontal Scaling",
                "Resilience"
            ],
            "enables": [
                "Load Balancing",
                "Auto-Scaling"
            ],
            "conflicts_with": [
                "Instance Affinity",
                "Temporal Coupling"
            ],
            "tensions_with": ["State Access Latency"],
            "violated_by": ["lexicon:instance-local-state"],
            "detected_by": ["mutable static/session-local state"],
            "measured_by": ["state externalization coverage"],
            "refactored_by": [
                "lexicon:externalize-session-state",
                "architecture:session-management"
            ],
            "enforced_by": ["architecture tests"],
            "severity": "recommended",
            "exemplar": {
                "before": "class FooHandler {\n  private currentUser?: User;\n  handle(request: Request) {\n    this.currentUser = request.user;\n    return processFoo(request, this.currentUser);\n  }\n}",
                "after": "class FooHandler {\n  handle(request: Request) {\n    return processFoo(request, request.user);\n  }\n}",
                "lang": "ts"
            }
        },
        {
            "id": "concurrency",
            "name": "Concurrency",
            "definition": "A conceptual representation of several tasks in progress over overlapping time, with their access to shared state coordinated.",
            "canon": ["concurrency"],
            "type": "model",
            "scope": [
                "runtime",
                "service",
                "algorithm"
            ],
            "requires": ["Concurrency Control"],
            "reinforces": ["Throughput"],
            "enables": ["Overlapping Work"],
            "conflicts_with": ["Race Conditions"],
            "tensions_with": ["Complexity"],
            "violated_by": ["lexicon:race-conditions"],
            "detected_by": [
                "data races",
                "flaky concurrent tests"
            ],
            "measured_by": [
                "throughput",
                "race count"
            ],
            "refactored_by": [
                "lexicon:apply-concurrency-control",
                "lexicon:make-immutable"
            ],
            "enforced_by": [
                "race detectors",
                "tests"
            ],
            "severity": "contextual",
            "exemplar": {
                "before": "for (const foo of foos) await processFoo(foo);",
                "after": "await Promise.all(foos.map(foo => processFoo(foo)));",
                "lang": "ts"
            }
        },
        {
            "id": "parallelism",
            "name": "Parallelism",
            "definition": "A technique for running independent units of work at the same time on several cores or workers.",
            "type": "technique",
            "scope": [
                "algorithm",
                "processing",
                "runtime"
            ],
            "requires": ["Independent Work Units"],
            "reinforces": [
                "Throughput",
                "Performance"
            ],
            "enables": ["Multi-Core Utilization"],
            "conflicts_with": ["Sequential Bottleneck"],
            "tensions_with": ["Coordination Overhead"],
            "violated_by": ["lexicon:sequential-bottleneck"],
            "detected_by": ["CPU bottlenecks with independent work"],
            "measured_by": [
                "speedup",
                "utilization"
            ],
            "refactored_by": [],
            "enforced_by": ["performance benchmarks"],
            "severity": "contextual",
            "exemplar": {
                "before": "const results = foos.map(foo => cpuHeavyFoo(foo));",
                "after": "const results = await workerPool.map(foos, foo => cpuHeavyFoo(foo));",
                "lang": "ts"
            }
        },
        {
            "id": "throughput",
            "distinctFrom": [
                {
                    "id": "architecture:latency",
                    "reason": "Throughput counts completions per unit of time, while latency measures the time of one request."
                }
            ],
            "name": "Throughput",
            "definition": "The rate at which a system completes requests, messages or items, counted per unit of time.",
            "type": "metric",
            "scope": [
                "service",
                "pipeline",
                "system"
            ],
            "requires": ["Capacity Model"],
            "reinforces": ["Scalability"],
            "enables": ["Load Handling"],
            "conflicts_with": ["Bottlenecks"],
            "tensions_with": ["Latency"],
            "violated_by": ["lexicon:bottlenecks"],
            "detected_by": ["load test failures"],
            "measured_by": ["requests/messages/items per second"],
            "refactored_by": [
                "architecture:bottleneck-analysis",
                "architecture:parallelism",
                "architecture:horizontal-scaling"
            ],
            "enforced_by": ["performance gates"],
            "severity": "contextual",
            "exemplar": {
                "before": "for (const foo of foos) await fooStore.save(foo);",
                "after": "for (const batch of chunk(foos, 500)) await fooStore.saveBatch(batch);",
                "lang": "ts"
            }
        },
        {
            "id": "latency",
            "name": "Latency",
            "definition": "A measure of the time between a request and its response, usually reported at percentiles.",
            "type": "metric",
            "scope": [
                "API",
                "service",
                "user flow"
            ],
            "requires": ["Time Budget"],
            "reinforces": [
                "User Experience",
                "Performance Engineering"
            ],
            "enables": ["Responsiveness"],
            "conflicts_with": ["Long Blocking Work"],
            "tensions_with": ["Throughput/Batching"],
            "violated_by": ["lexicon:long-blocking-work"],
            "detected_by": ["trace span delays"],
            "measured_by": ["p50/p95/p99 latency"],
            "refactored_by": [
                "architecture:caching",
                "lexicon:introduce-async-event",
                "lexicon:add-index"
            ],
            "enforced_by": ["SLO gates"],
            "severity": "contextual",
            "exemplar": {
                "before": "async function renderFoo(id: FooId) {\n  const foo = await fooStore.find(id);\n  const bar = await barStore.find(foo.barId);\n  const baz = await bazStore.find(foo.bazId);\n  return render(foo, bar, baz);\n}",
                "after": "async function renderFoo(id: FooId) {\n  const foo = await fooStore.find(id);\n  const [bar, baz] = await Promise.all([\n    barStore.find(foo.barId),\n    bazStore.find(foo.bazId),\n  ]);\n  return render(foo, bar, baz);\n}",
                "lang": "ts"
            }
        },
        {
            "id": "performance-engineering",
            "distinctFrom": [
                {
                    "id": "architecture:benchmarking",
                    "reason": "Performance engineering is the practice that sets budgets and acts on measurements, while benchmarking is the controlled timing it measures with."
                }
            ],
            "name": "Performance Engineering",
            "definition": "The practice of setting performance budgets, measuring against a representative workload and changing code only where a measurement points.",
            "aliases": ["Evidence-Based Optimization"],
            "type": "activity",
            "scope": [
                "codebase",
                "service",
                "system"
            ],
            "requires": [
                "Profiling",
                "Benchmarking"
            ],
            "reinforces": [
                "Scalability",
                "Resource Efficiency"
            ],
            "enables": [],
            "conflicts_with": ["Guess-Based Optimization"],
            "tensions_with": ["Maintainability"],
            "violated_by": ["lexicon:guess-based-optimization"],
            "detected_by": ["performance changes lacking benchmark"],
            "measured_by": [
                "benchmark trend",
                "SLO compliance"
            ],
            "refactored_by": [
                "architecture:profiling",
                "architecture:bottleneck-analysis",
                "architecture:benchmarking"
            ],
            "enforced_by": ["performance CI"],
            "severity": "recommended",
            "exemplar": {
                "before": "optimizeFooCode();",
                "after": "const budget = { p95LatencyMs: 150, throughputPerSecond: 1000 } as const;\nconst profile = await measureFooWorkload(representativeLoad);\nconst change = optimize(profile.hotspot);\nassertPerformance(change, budget);",
                "lang": "ts"
            }
        },
        {
            "id": "algorithmic-efficiency",
            "name": "Algorithmic Efficiency",
            "definition": "A design rule that an algorithm and its data structures are chosen for how their cost grows with input size.",
            "type": "principle",
            "scope": [
                "algorithm",
                "data structure"
            ],
            "requires": ["Complexity Awareness"],
            "reinforces": ["Scalability"],
            "enables": ["Efficient Processing"],
            "conflicts_with": [
                "Inefficient Algorithm Choice",
                "N Plus One Query"
            ],
            "tensions_with": ["Implementation Simplicity"],
            "violated_by": ["lexicon:inefficient-algorithm-choice"],
            "detected_by": [
                "complexity analysis",
                "benchmark slope"
            ],
            "measured_by": ["time/space complexity"],
            "refactored_by": [
                "lexicon:replace-algorithm",
                "lexicon:add-index"
            ],
            "enforced_by": [
                "review",
                "benchmarks"
            ],
            "severity": "contextual",
            "exemplar": {
                "before": "function hasFoo(foos: Foo[], id: FooId) {\n  return foos.some(foo => foo.id === id);\n}",
                "after": "function indexFoos(foos: readonly Foo[]) {\n  return new Map(foos.map(foo => [foo.id, foo]));\n}\nconst hasFoo = (index: ReadonlyMap<FooId, Foo>, id: FooId) => index.has(id);",
                "lang": "ts"
            }
        },
        {
            "id": "time-complexity",
            "name": "Time Complexity",
            "definition": "A measure of how an algorithm's running time grows as its input grows.",
            "type": "metric",
            "scope": [
                "algorithm",
                "function"
            ],
            "requires": ["Input Size Model"],
            "reinforces": ["Algorithmic Efficiency"],
            "enables": ["Scalability Analysis"],
            "conflicts_with": ["Unbounded Runtime Growth"],
            "tensions_with": ["Space Complexity"],
            "violated_by": ["lexicon:inefficient-algorithm-choice"],
            "detected_by": [
                "nested loops over large inputs",
                "benchmark slope"
            ],
            "measured_by": [
                "Big O",
                "runtime scaling"
            ],
            "refactored_by": [
                "lexicon:replace-algorithm",
                "lexicon:add-index",
                "architecture:caching"
            ],
            "enforced_by": ["benchmark thresholds"],
            "severity": "contextual",
            "exemplar": {
                "before": "function duplicateFooIds(foos: Foo[]) {\n  return foos.filter((foo, index) => foos.findIndex(x => x.id === foo.id) !== index);\n}",
                "after": "function duplicateFooIds(foos: readonly Foo[]) {\n  const seen = new Set<FooId>();\n  return foos.filter(foo => seen.has(foo.id) || !seen.add(foo.id));\n}",
                "lang": "ts"
            }
        },
        {
            "id": "space-complexity",
            "distinctFrom": [
                {
                    "id": "architecture:resource-utilization",
                    "reason": "Space complexity is how an algorithm's memory grows with input, while resource utilization is how much provisioned capacity is in use now."
                },
                {
                    "id": "architecture:time-complexity",
                    "reason": "Space complexity measures memory growth, while time complexity measures running-time growth."
                }
            ],
            "name": "Space Complexity",
            "definition": "A measure of how an algorithm's memory use grows as its input grows.",
            "type": "metric",
            "scope": [
                "algorithm",
                "process"
            ],
            "requires": ["Memory Model"],
            "reinforces": ["Resource Utilization"],
            "enables": ["Memory Scalability"],
            "conflicts_with": ["Unbounded Memory Growth"],
            "tensions_with": ["Time Complexity"],
            "violated_by": ["lexicon:unbounded-memory-growth"],
            "detected_by": [
                "memory profiling",
                "full materialization"
            ],
            "measured_by": [
                "Big O space",
                "peak memory"
            ],
            "refactored_by": [
                "architecture:sequential-access",
                "architecture:iterator-pattern",
                "lexicon:chunked-processing"
            ],
            "enforced_by": ["memory benchmarks"],
            "severity": "contextual",
            "exemplar": {
                "before": "function processFoos(stream: AsyncIterable<Foo>) {\n  return collectAll(stream).then(foos => foos.map(transformFoo));\n}",
                "after": "async function* processFoos(stream: AsyncIterable<Foo>) {\n  for await (const foo of stream) yield transformFoo(foo);\n}",
                "lang": "ts"
            }
        },
        {
            "id": "big-o-notation",
            "name": "Big O Notation",
            "definition": "A method for classifying an algorithm by the upper bound on how its cost grows with input size, ignoring constant factors.",
            "type": "technique",
            "scope": ["algorithm"],
            "requires": ["Complexity Model"],
            "reinforces": ["Algorithmic Efficiency"],
            "enables": ["Comparative Analysis"],
            "conflicts_with": ["Anecdotal Performance Claims"],
            "tensions_with": ["Constant-Factor Practicality"],
            "violated_by": ["lexicon:inefficient-algorithm-choice"],
            "detected_by": ["missing complexity note for critical algorithm"],
            "measured_by": ["asymptotic classification"],
            "refactored_by": ["lexicon:replace-algorithm"],
            "enforced_by": ["review checklist"],
            "severity": "contextual",
            "exemplar": {
                "before": "function pairFoosWithBars(foos: Foo[], bars: Bar[]) {\n  return foos.flatMap(foo => bars.filter(bar => bar.fooId === foo.id).map(bar => [foo, bar]));\n}",
                "after": "function pairFoosWithBars(foos: readonly Foo[], bars: readonly Bar[]) {\n  const barsByFoo = groupBy(bars, bar => bar.fooId);\n  return foos.flatMap(foo => (barsByFoo.get(foo.id) ?? []).map(bar => [foo, bar]));\n}",
                "lang": "ts"
            }
        },
        {
            "id": "optimization",
            "distinctFrom": [
                {
                    "id": "architecture:bottleneck-analysis",
                    "reason": "Optimization changes the code at a bottleneck, while bottleneck analysis finds where it is."
                },
                {
                    "id": "architecture:performance-engineering",
                    "reason": "Optimization is one change at a located bottleneck, while performance engineering is the whole practice of budgets, measurement and change."
                }
            ],
            "name": "Optimization",
            "definition": "The activity of changing code or configuration to reduce the cost of a bottleneck that a measurement has located.",
            "type": "activity",
            "scope": [
                "code",
                "database",
                "system"
            ],
            "requires": [
                "Profiling",
                "Bottleneck Evidence"
            ],
            "reinforces": ["Performance Engineering"],
            "enables": ["Resource Efficiency"],
            "conflicts_with": ["Premature Optimization"],
            "tensions_with": ["Readability/Maintainability"],
            "violated_by": ["lexicon:premature-optimization"],
            "detected_by": ["complex code without performance evidence"],
            "measured_by": [
                "benchmark delta",
                "SLO improvement"
            ],
            "refactored_by": [
                "architecture:bottleneck-analysis",
                "lexicon:cleanup-simplification"
            ],
            "enforced_by": ["benchmark review"],
            "severity": "contextual",
            "exemplar": {
                "before": "const fooCache = new Map<FooId, Foo>();\nfunction loadFoo(id: FooId) { return fooCache.get(id) ?? expensiveLoad(id); }",
                "after": "const profile = profiler.measure(\"foo.load\", representativeFooIds);\nif (profile.hotspot === \"foo-store-read\") {\n  enableBoundedFooCache({ maxEntries: 10_000, keyBy: fooVersionFingerprint });\n}",
                "lang": "ts"
            }
        },
        {
            "id": "profiling",
            "name": "Profiling",
            "definition": "A technique for measuring where a running program spends its time and memory under a representative workload.",
            "type": "technique",
            "scope": [
                "runtime",
                "code path"
            ],
            "requires": ["Representative Workload"],
            "reinforces": ["Performance Engineering"],
            "enables": ["Bottleneck Detection"],
            "conflicts_with": ["Guesswork"],
            "tensions_with": ["Measurement Overhead"],
            "violated_by": ["lexicon:guesswork"],
            "detected_by": ["missing profile evidence"],
            "measured_by": ["hotspot attribution"],
            "refactored_by": [],
            "enforced_by": ["performance review"],
            "severity": "recommended",
            "exemplar": {
                "before": "rewriteFooParserForSpeed();",
                "after": "const profile = await profiler.capture(() => parseFooBatch(batch));\nconst hotspot = profile.topFrame();\noptimizeFooFrame(hotspot);",
                "lang": "ts"
            }
        },
        {
            "id": "benchmarking",
            "name": "Benchmarking",
            "definition": "The activity of timing a fixed workload repeatedly in a controlled environment, so results can be compared across changes.",
            "type": "activity",
            "scope": [
                "function",
                "service",
                "system"
            ],
            "requires": ["Repeatable Test Environment"],
            "reinforces": [
                "Reproducibility",
                "Performance Engineering"
            ],
            "enables": ["Regression Detection"],
            "conflicts_with": ["Anecdotal Timing"],
            "tensions_with": ["Environment Drift"],
            "violated_by": ["lexicon:anecdotal-performance-claims"],
            "detected_by": ["missing benchmark for perf-sensitive changes"],
            "measured_by": ["benchmark score/trend"],
            "refactored_by": ["architecture:immutable-infrastructure"],
            "enforced_by": ["benchmark CI"],
            "severity": "recommended",
            "exemplar": {
                "before": "const start = clock.now();\nrunFoo();\nreport(clock.now() - start);",
                "after": "benchmark(\"foo.parse\", {\n  warmup: 100,\n  iterations: 10_000,\n  run: () => parseFoo(fixture),\n});",
                "lang": "ts"
            }
        },
        {
            "id": "bottleneck-analysis",
            "name": "Bottleneck Analysis",
            "definition": "The activity of finding the stage that limits a system's overall throughput or latency, from traces and profiles.",
            "type": "activity",
            "scope": [
                "code path",
                "system"
            ],
            "requires": [
                "Profiling",
                "Metrics"
            ],
            "reinforces": ["Optimization"],
            "enables": ["Targeted Improvement"],
            "conflicts_with": ["Local Micro-Optimization"],
            "tensions_with": ["Distributed Complexity"],
            "violated_by": ["lexicon:local-micro-optimization"],
            "detected_by": ["performance work without hotspot evidence"],
            "measured_by": ["bottleneck contribution percentage"],
            "refactored_by": [
                "architecture:parallelism",
                "architecture:caching"
            ],
            "enforced_by": ["performance review"],
            "severity": "recommended",
            "exemplar": {
                "before": "addMoreFooWorkers();",
                "after": "const trace = await measureFooPipeline();\nconst bottleneck = trace.stages.sort((a, b) => b.waitMs - a.waitMs)[0];\nremoveBottleneck(bottleneck);",
                "lang": "ts"
            }
        },
        {
            "id": "resource-utilization",
            "name": "Resource Utilization",
            "definition": "A measure of how much of the provisioned CPU, memory, I/O and network capacity is in use.",
            "type": "metric",
            "scope": [
                "CPU",
                "memory",
                "IO",
                "network"
            ],
            "requires": ["Monitoring"],
            "reinforces": ["Performance Engineering"],
            "enables": ["Capacity Planning"],
            "conflicts_with": ["Resource Waste/Saturation"],
            "tensions_with": ["Over-Provisioning"],
            "violated_by": ["lexicon:resource-waste-saturation"],
            "detected_by": ["monitoring metrics"],
            "measured_by": ["CPU/memory/IO/network utilization"],
            "refactored_by": [
                "architecture:profiling",
                "architecture:horizontal-scaling",
                "lexicon:externalize-configuration"
            ],
            "enforced_by": ["SLO/capacity policy"],
            "severity": "contextual",
            "exemplar": {
                "before": "deployFoo({ cpu: 16, memoryGb: 64 });",
                "after": "const sizing = rightSizeFoo({\n  cpuP95: metrics.cpu(\"foo\", \"p95\"),\n  memoryP95: metrics.memory(\"foo\", \"p95\"),\n  headroom: 0.25,\n});\ndeployFoo(sizing);",
                "lang": "ts"
            }
        },
        {
            "id": "rate-limiting",
            "name": "Rate Limiting",
            "definition": "A mechanism that caps how many requests a caller can make in a time window and rejects the excess.",
            "type": "mechanism",
            "scope": [
                "API",
                "service",
                "queue"
            ],
            "requires": ["Quota Policy"],
            "reinforces": [
                "Backpressure",
                "Security"
            ],
            "enables": ["Abuse/Overload Protection"],
            "conflicts_with": ["Unbounded Access"],
            "tensions_with": ["User Experience"],
            "violated_by": ["lexicon:unbounded-access"],
            "detected_by": ["missing rate limiter on public/expensive endpoints"],
            "measured_by": [
                "limit hit rate",
                "overload incidents"
            ],
            "refactored_by": [],
            "enforced_by": ["API gateway/policy"],
            "severity": "contextual",
            "mandatoryFor": "public APIs",
            "exemplar": {
                "before": "app.post(\"/foo\", createFoo);",
                "after": "app.post(\"/foo\", rateLimit({\n  key: request => request.identity.id,\n  limit: 100,\n  windowMs: 60_000,\n}), createFoo);",
                "lang": "ts"
            }
        },
        {
            "id": "memory-efficiency",
            "distinctFrom": [
                {
                    "id": "lexicon:cpu-cost",
                    "reason": "Memory efficiency bounds the memory a process uses, while CPU cost is the processor time that streaming or chunking may add."
                }
            ],
            "name": "Memory Efficiency",
            "definition": "The degree to which a process handles its input with bounded memory, by streaming or chunking instead of loading it whole.",
            "type": "quality-attribute",
            "scope": [
                "algorithm",
                "process",
                "stream"
            ],
            "requires": ["Space Complexity Awareness"],
            "reinforces": ["Scalability"],
            "enables": ["Large Input Handling"],
            "conflicts_with": ["Full Materialization"],
            "tensions_with": ["CPU Cost"],
            "violated_by": ["lexicon:unbounded-memory-growth"],
            "detected_by": ["memory profile spikes"],
            "measured_by": [
                "peak memory",
                "allocation rate"
            ],
            "refactored_by": [
                "architecture:event-stream",
                "lexicon:chunked-processing",
                "architecture:iterator-pattern"
            ],
            "enforced_by": ["memory benchmarks"],
            "severity": "contextual",
            "exemplar": {
                "before": "const copies = foos.map(foo => structuredClone(foo));",
                "after": "function* fooViews(foos: readonly Foo[]) {\n  for (const foo of foos) yield { id: foo.id, name: foo.name };\n}",
                "lang": "ts"
            }
        },
        {
            "id": "cdn-edge-caching",
            "name": "CDN / Edge Caching",
            "aliases": ["Content Delivery Network"],
            "definition": "A mechanism that serves cacheable content from servers near the requester, keyed by a fingerprint of the content.",
            "type": "mechanism",
            "scope": [
                "service",
                "infrastructure",
                "latency"
            ],
            "requires": ["Cacheable Content"],
            "reinforces": [
                "Caching",
                "Latency"
            ],
            "enables": [
                "Origin Offload",
                "Geographically-Local Delivery"
            ],
            "conflicts_with": ["Origin-Only Serving"],
            "tensions_with": ["Cache Invalidation"],
            "violated_by": ["lexicon:origin-only-serving"],
            "detected_by": ["static assets served from origin per request"],
            "measured_by": ["origin request rate / cache hit ratio"],
            "refactored_by": [],
            "enforced_by": ["performance review"],
            "severity": "contextual",
            "exemplar": {
                "before": "app.get(\"/foo/:id/avatar\", serveFooAvatarFromOrigin);",
                "after": "app.get(\"/foo/:id/avatar\",\n  edgeCache({ immutable: true, key: request => avatarFingerprint(request.params.id) }),\n  serveFooAvatarFromOrigin,\n);",
                "lang": "ts"
            }
        },
        {
            "id": "read-replica",
            "distinctFrom": [
                {
                    "id": "architecture:horizontal-scaling",
                    "reason": "A read replica scales database reads by copying data, while horizontal scaling adds stateless service instances behind a load balancer."
                }
            ],
            "name": "Read Replica",
            "definition": "A technique for routing read queries to replicated copies of a database, so the primary handles only writes.",
            "type": "technique",
            "scope": [
                "service",
                "database",
                "scalability"
            ],
            "requires": ["Replication"],
            "reinforces": [
                "Horizontal Scaling",
                "Load Balancing"
            ],
            "enables": ["Read Traffic Offload"],
            "conflicts_with": ["Single-Primary Read Contention"],
            "tensions_with": ["Read-Your-Writes Consistency"],
            "violated_by": ["lexicon:single-primary-read-contention"],
            "detected_by": ["read load saturating the write primary"],
            "measured_by": ["primary read/write contention ratio"],
            "refactored_by": [],
            "enforced_by": ["database design review"],
            "severity": "contextual",
            "exemplar": {
                "before": "const foo = await primaryDb.query(fooQuery);\nawait primaryDb.write(fooCommand);",
                "after": "const foo = await replicaRouter.read(fooQuery);\nawait primaryDb.write(fooCommand);",
                "lang": "ts"
            }
        },
        {
            "id": "queuing-theory",
            "name": "Queuing Theory",
            "definition": "A conceptual representation of a system as queues with arrival and service rates, used to predict waiting time and size capacity.",
            "type": "model",
            "scope": [
                "performance",
                "capacity",
                "system"
            ],
            "requires": ["Arrival and Service Rates"],
            "reinforces": [
                "Capacity Planning",
                "Latency"
            ],
            "enables": [
                "Wait-Time Prediction",
                "Utilization-Based Sizing"
            ],
            "conflicts_with": ["Guess-Based Capacity"],
            "tensions_with": ["Model Assumptions"],
            "violated_by": ["lexicon:guess-based-capacity"],
            "detected_by": ["latency collapsing as utilization approaches saturation"],
            "measured_by": ["predicted vs actual queue depth and wait time"],
            "refactored_by": [],
            "enforced_by": ["capacity review"],
            "severity": "contextual",
            "exemplar": {
                "before": "const workers = 4;",
                "after": "const rho = arrivalRate / (workers * serviceRate);\nif (rho >= 1) throw new Error(\"unstable queue: utilization >= 1\");\nconst avgWaitMs = mm1WaitTime({ arrivalRate, serviceRate, servers: workers });",
                "lang": "ts"
            }
        }
    ]
}