[
  {
    "id": "01-system-context",
    "title": "SLO and Error Budget Service — System Context",
    "layout": "context",
    "canvas": {
      "width": 1700
    },
    "colWidth": 268,
    "system": {
      "label": "SLO & Error Budget Service",
      "sub": "computes the verdict; owns no signal"
    },
    "groups": [
      {
        "side": "left",
        "title": "People",
        "nodes": [
          {
            "id": "team",
            "label": "Service team lead",
            "kind": "actor",
            "rel": "declares",
            "dir": "in"
          },
          {
            "id": "rel",
            "label": "Release manager",
            "kind": "actor",
            "rel": "asks the gate",
            "dir": "in"
          },
          {
            "id": "sre",
            "label": "Platform SRE",
            "kind": "actor",
            "rel": "on call",
            "dir": "in"
          },
          {
            "id": "head",
            "label": "Reliability lead",
            "kind": "actor",
            "rel": "owns policy",
            "dir": "in"
          }
        ]
      },
      {
        "side": "right",
        "title": "Who acts on the verdict",
        "nodes": [
          {
            "id": "gate",
            "label": "CI/CD release gate",
            "kind": "external",
            "rel": "verdict",
            "dir": "out"
          },
          {
            "id": "page",
            "label": "On-call paging",
            "kind": "external",
            "rel": "pages",
            "dir": "out",
            "kind2": "async"
          },
          {
            "id": "inc",
            "label": "Incident platform",
            "kind": "external",
            "rel": "correlates",
            "dir": "out",
            "kind2": "async"
          },
          {
            "id": "exec",
            "label": "Executive reporting",
            "kind": "external",
            "rel": "monthly",
            "dir": "out",
            "kind2": "batch"
          }
        ]
      },
      {
        "side": "top",
        "title": "Measurement plane — not ours",
        "nodes": [
          {
            "id": "mon",
            "label": "Azure Monitor",
            "kind": "platform",
            "rel": "buckets",
            "dir": "in"
          },
          {
            "id": "prom",
            "label": "Managed Prometheus",
            "kind": "platform",
            "rel": "rules",
            "dir": "in"
          },
          {
            "id": "probe",
            "label": "Synthetic probes",
            "kind": "external",
            "rel": "probes",
            "dir": "in",
            "kind2": "async"
          }
        ]
      },
      {
        "side": "bottom",
        "title": "Platform dependencies",
        "nodes": [
          {
            "id": "git",
            "label": "Definitions repository",
            "kind": "external",
            "rel": "SLO as code",
            "dir": "in"
          },
          {
            "id": "entra",
            "label": "Microsoft Entra ID",
            "kind": "security",
            "rel": "identity",
            "dir": "in"
          },
          {
            "id": "kv",
            "label": "Key Vault Managed HSM",
            "kind": "security",
            "rel": "signing",
            "dir": "in",
            "kind2": "bidirectional"
          }
        ]
      }
    ],
    "note": "In scope: deriving an error budget from someone else's aggregates and publishing a signed verdict. Out of scope: collecting telemetry, storing traces, drawing dashboards, and blocking a deployment. The audit ledger and key custody are omitted here and appear in views 11, 19 and 20.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "02-high-level-architecture",
    "title": "SLO and Error Budget Service — High-Level Architecture",
    "layout": "flow",
    "canvas": {
      "width": 1780
    },
    "chain": true,
    "align": "middle",
    "nodeWidth": 186,
    "stages": [
      {
        "title": "Declare",
        "nodes": [
          {
            "id": "recon",
            "label": "Definition reconciler",
            "sub": "Container Apps job",
            "kind": "app"
          },
          {
            "id": "admit",
            "label": "Admission checks",
            "sub": "resolution, cardinality",
            "kind": "decision"
          }
        ]
      },
      {
        "title": "Remember",
        "nodes": [
          {
            "id": "reg",
            "label": "SLO registry",
            "sub": "Azure SQL",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Ingest",
        "nodes": [
          {
            "id": "eh",
            "label": "Indicator stream",
            "sub": "Event Hubs",
            "kind": "queue"
          },
          {
            "id": "bkt",
            "label": "Event-time bucketer",
            "sub": "10 min lateness",
            "kind": "app"
          }
        ]
      },
      {
        "title": "Store",
        "nodes": [
          {
            "id": "adx",
            "label": "SLI aggregates",
            "sub": "Data Explorer, 13 mo",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Compute",
        "nodes": [
          {
            "id": "calc",
            "label": "Budget calculator",
            "sub": "watermarked",
            "kind": "app"
          },
          {
            "id": "snap",
            "label": "Snapshot writer",
            "sub": "at window close",
            "kind": "app"
          }
        ]
      },
      {
        "title": "Publish",
        "nodes": [
          {
            "id": "vapi",
            "label": "Verdict API",
            "sub": "signed, p99 150 ms",
            "kind": "integration"
          },
          {
            "id": "alerts",
            "label": "Burn-rate evaluator",
            "sub": "multi-window",
            "kind": "app"
          }
        ]
      },
      {
        "title": "Act",
        "nodes": [
          {
            "id": "cd",
            "label": "Release gate",
            "sub": "outside the platform",
            "kind": "external"
          },
          {
            "id": "oncall",
            "label": "On-call",
            "sub": "page or ticket",
            "kind": "external"
          }
        ]
      }
    ],
    "note": "Seven stages on one spine. The platform begins at an aggregate it did not produce and ends at a document it does not enforce; everything in between is derivation.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "03-actors-and-journeys",
    "title": "Who the Budget Is For, and What They Get to Do",
    "layout": "actors",
    "canvas": {
      "width": 1740
    },
    "cardWidth": 268,
    "groups": [
      {
        "title": "The teams being measured",
        "kind": "boundary",
        "actors": [
          {
            "id": "team",
            "label": "Service team lead",
            "kind": "actor",
            "sub": "400 services",
            "goal": "I want to know whether my service is reliable enough to ship, in one number I can disagree with — and if it says no, I want to see the forty minutes that spent the budget.",
            "journeys": [
              {
                "id": "j-declare",
                "label": "Declare an SLO for a journey",
                "sub": "see view 06"
              },
              {
                "label": "Read the intervals that spent it"
              },
              {
                "label": "Correct a wrong definition"
              }
            ]
          },
          {
            "id": "relm",
            "label": "Release manager",
            "kind": "actor",
            "sub": "ships twice a day",
            "goal": "It is Monday and the train is loaded. Tell me yes or no before stand-up, and make it the same answer the SRE lead would give me.",
            "journeys": [
              {
                "id": "j-ship",
                "label": "Can we ship today?",
                "sub": "see view 04"
              },
              {
                "label": "Request an override, on the record"
              }
            ]
          }
        ]
      },
      {
        "title": "The people who carry it",
        "kind": "cloud",
        "actors": [
          {
            "id": "sre",
            "label": "Platform SRE",
            "kind": "actor",
            "sub": "12 on rotation",
            "goal": "Wake me when the budget is genuinely going, not when one minute looked bad. And when the metrics pipeline breaks, tell me that instead of telling me the service is fine.",
            "journeys": [
              {
                "id": "j-page",
                "label": "The 02:00 burn-rate page",
                "sub": "see view 05"
              },
              {
                "label": "Suppress a blind SLO"
              },
              {
                "label": "Replay a quarantined batch"
              }
            ]
          },
          {
            "id": "head",
            "label": "Reliability lead",
            "kind": "actor",
            "sub": "owns the policy",
            "goal": "I want the freeze to be arithmetic rather than seniority, and I want to see how often we grant an exception — because a policy with a routine exception has stopped existing.",
            "journeys": [
              {
                "label": "Set a budget policy"
              },
              {
                "label": "Approve an exclusion window"
              },
              {
                "label": "Read the override register"
              }
            ]
          }
        ]
      },
      {
        "title": "Assurance",
        "kind": "trust",
        "actors": [
          {
            "id": "aud",
            "label": "Internal auditor",
            "kind": "actor",
            "sub": "quarterly",
            "goal": "Show me that last quarter's published attainment cannot have been edited after the fact, and show me who removed each excluded minute and why.",
            "journeys": [
              {
                "label": "Verify a closed window"
              },
              {
                "label": "Trace an exclusion to its approver"
              }
            ]
          }
        ]
      },
      {
        "title": "Machines in the cast",
        "kind": "plain",
        "actors": [
          {
            "id": "gate",
            "label": "Release gate",
            "kind": "external",
            "sub": "CI/CD, 2,000 rps peak",
            "goal": "Give me a signed verdict in under 150 ms, or tell me nothing at all — but never give me a figure you cannot stand behind, because I will act on it without a human reading it.",
            "journeys": [
              {
                "label": "Fetch a verdict per deployment"
              },
              {
                "label": "Fail static on a stale cache"
              }
            ]
          },
          {
            "id": "src",
            "label": "Measurement plane",
            "kind": "platform",
            "sub": "Monitor, Prometheus",
            "goal": "I will give you buckets, sometimes late and sometimes not at all. Do not pretend my silence was a good minute.",
            "journeys": [
              {
                "label": "Emit the outcome cube"
              },
              {
                "label": "Go quiet during an incident"
              }
            ]
          }
        ]
      }
    ],
    "note": "Six parties, two of them machines. The release gate is in the cast because it is the only actor that acts on a verdict without a human reading it, which is what forces the verdict to be signed and typed.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "04-journey-can-we-ship-today",
    "title": "Journey — Monday Morning: Can We Ship Today?",
    "layout": "journey",
    "canvas": {
      "width": 1740
    },
    "cellWidth": 215,
    "actor": {
      "label": "Release manager",
      "sub": "checkout service, Monday 08:40",
      "goal": "A defensible yes or no before stand-up",
      "trigger": "Weekend tickets say checkout failed; the dashboards are green",
      "success": "The train ships, or it does not, and nobody argues about which"
    },
    "phases": [
      {
        "title": "Ask",
        "sub": "open the journey"
      },
      {
        "title": "See the number",
        "sub": "budget remaining"
      },
      {
        "title": "Find the damage",
        "moment": true
      },
      {
        "title": "Read the verdict",
        "moment": true
      },
      {
        "title": "Ship or hold",
        "sub": "gate decides"
      }
    ],
    "lanes": [
      {
        "title": "What they do",
        "kind": "step",
        "cells": [
          [
            {
              "label": "Opens checkout journey"
            }
          ],
          [
            {
              "label": "Reads 11% remaining"
            }
          ],
          [
            {
              "label": "Clicks the bad interval"
            },
            {
              "label": "Sees Sat 22:10–22:48"
            }
          ],
          [
            {
              "label": "Reads \"exhausted\""
            }
          ],
          [
            {
              "label": "Holds features, ships the fix"
            }
          ]
        ]
      },
      {
        "title": "What the platform does",
        "kind": "system",
        "cells": [
          [
            {
              "label": "Resolves journey to SLOs"
            }
          ],
          [
            {
              "label": "Serves rolling + calendar"
            },
            {
              "label": "Labels coverage 99.8%"
            }
          ],
          [
            {
              "label": "Attributes to intervals"
            },
            {
              "label": "Links the incident record"
            }
          ],
          [
            {
              "label": "Signs the verdict"
            }
          ],
          [
            {
              "label": "Exempts rollback class"
            }
          ]
        ]
      },
      {
        "title": "How it feels",
        "kind": "emotion",
        "levels": [
          "Confident",
          "Fine",
          "Uneasy"
        ],
        "points": [
          1,
          2,
          2,
          1,
          0
        ]
      },
      {
        "title": "Where it hurts",
        "kind": "pain",
        "cells": [
          [],
          [
            {
              "label": "11% of what, exactly?"
            }
          ],
          [
            {
              "label": "Was that minute really ours?"
            }
          ],
          [],
          []
        ]
      },
      {
        "title": "What answers it",
        "kind": "gain",
        "cells": [
          [
            {
              "label": "Journey, not endpoint"
            }
          ],
          [
            {
              "label": "Budget in failed requests"
            }
          ],
          [
            {
              "label": "Interval links to the incident"
            }
          ],
          [
            {
              "label": "Typed verdict, one source"
            }
          ],
          [
            {
              "label": "Fixes are never frozen"
            }
          ]
        ]
      }
    ],
    "chain": true,
    "note": "The trough is not the number — it is the forty minutes behind it. A budget figure nobody can trace to an interval gets argued with, which is the failure mode this view exists to prevent.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "05-journey-the-burn-rate-page",
    "title": "Journey — 02:00, the Burn-Rate Page",
    "layout": "journey",
    "canvas": {
      "width": 1740
    },
    "cellWidth": 215,
    "actor": {
      "label": "Platform SRE",
      "sub": "on call, asleep",
      "goal": "Know within a minute whether this is real",
      "trigger": "A fast-burn page on the sign-in SLO",
      "success": "Either a fixed outage or a page that should not have fired, named as such"
    },
    "phases": [
      {
        "title": "Page",
        "sub": "14.4× burn"
      },
      {
        "title": "Triage",
        "sub": "is it real?"
      },
      {
        "title": "The other answer",
        "moment": true
      },
      {
        "title": "Act",
        "sub": "fix or suppress"
      },
      {
        "title": "After",
        "sub": "alert quality"
      }
    ],
    "lanes": [
      {
        "title": "What they do",
        "kind": "step",
        "cells": [
          [
            {
              "label": "Reads the page"
            }
          ],
          [
            {
              "label": "Checks coverage first"
            }
          ],
          [
            {
              "label": "Sees coverage 41%"
            },
            {
              "label": "Not an outage — blind"
            }
          ],
          [
            {
              "label": "Fixes the scrape, not the service"
            }
          ],
          [
            {
              "label": "Logs it as measurement"
            }
          ]
        ]
      },
      {
        "title": "What the platform does",
        "kind": "system",
        "cells": [
          [
            {
              "label": "Fast + confirm window agree"
            }
          ],
          [
            {
              "label": "Publishes coverage with figure"
            }
          ],
          [
            {
              "label": "Raises measurement-failure"
            },
            {
              "label": "Suppresses reliability alert"
            }
          ],
          [
            {
              "label": "Keeps buckets as no-data"
            }
          ],
          [
            {
              "label": "Counts it outside precision"
            }
          ]
        ]
      },
      {
        "title": "How it feels",
        "kind": "emotion",
        "levels": [
          "Clear",
          "Workable",
          "Lost"
        ],
        "points": [
          1,
          1,
          2,
          0,
          0
        ]
      },
      {
        "title": "Where it hurts",
        "kind": "pain",
        "cells": [
          [],
          [
            {
              "label": "Three dashboards, three answers"
            }
          ],
          [
            {
              "label": "Did users notice or not?"
            }
          ],
          [],
          []
        ]
      },
      {
        "title": "What answers it",
        "kind": "gain",
        "cells": [
          [
            {
              "label": "Burn rate, not error %"
            }
          ],
          [
            {
              "label": "Coverage on every figure"
            }
          ],
          [
            {
              "label": "Two alert classes, not one"
            }
          ],
          [
            {
              "label": "No-data is never good"
            }
          ],
          [
            {
              "label": "Precision measured per SLO"
            }
          ]
        ]
      }
    ],
    "chain": true,
    "note": "The trough is the moment the SRE cannot tell an outage from a blind SLO. A platform with one alert class puts every reader in that trough; two classes is the whole fix.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "06-journey-declare-an-slo",
    "title": "Journey — Declaring an SLO for a Journey",
    "layout": "journey",
    "canvas": {
      "width": 1740
    },
    "cellWidth": 215,
    "actor": {
      "label": "Service team lead",
      "sub": "with the product owner",
      "goal": "An SLO that means what a user would mean",
      "trigger": "Checkout is now a critical journey; it needs a target",
      "success": "A merged definition that the team will not quietly widen later"
    },
    "phases": [
      {
        "title": "Agree the journey",
        "sub": "with product"
      },
      {
        "title": "Write the predicate",
        "moment": true
      },
      {
        "title": "Name the denominator",
        "moment": true
      },
      {
        "title": "Review",
        "sub": "pull request"
      },
      {
        "title": "Live",
        "sub": "reconciled"
      }
    ],
    "lanes": [
      {
        "title": "What they do",
        "kind": "step",
        "cells": [
          [
            {
              "label": "Picks \"complete a checkout\""
            }
          ],
          [
            {
              "label": "Drafts good = 2xx under 800 ms"
            }
          ],
          [
            {
              "label": "Argues about 499s"
            },
            {
              "label": "Excludes bot traffic"
            }
          ],
          [
            {
              "label": "Opens a PR"
            }
          ],
          [
            {
              "label": "Watches first buckets"
            }
          ]
        ]
      },
      {
        "title": "What the platform does",
        "kind": "system",
        "cells": [
          [
            {
              "label": "Offers the journey catalogue"
            }
          ],
          [
            {
              "label": "Checks the outcome cube covers it"
            }
          ],
          [
            {
              "label": "Rejects implicit denominator"
            },
            {
              "label": "Rejects 99.999% monthly"
            }
          ],
          [
            {
              "label": "Dry-runs over 28 days of history"
            }
          ],
          [
            {
              "label": "Versions it, effective-from"
            }
          ]
        ]
      },
      {
        "title": "How it feels",
        "kind": "emotion",
        "levels": [
          "Clear",
          "Workable",
          "Stuck"
        ],
        "points": [
          0,
          1,
          2,
          1,
          0
        ]
      },
      {
        "title": "Where it hurts",
        "kind": "pain",
        "cells": [
          [],
          [
            {
              "label": "Is 800 ms the user's number?"
            }
          ],
          [
            {
              "label": "Nobody owns the 499 answer"
            }
          ],
          [],
          []
        ]
      },
      {
        "title": "What answers it",
        "kind": "gain",
        "cells": [
          [
            {
              "label": "Journey catalogue, not endpoints"
            }
          ],
          [
            {
              "label": "Cube dimensions are declared"
            }
          ],
          [
            {
              "label": "Implicit denominators rejected"
            }
          ],
          [
            {
              "label": "Dry-run on real history"
            }
          ],
          [
            {
              "label": "Versioned, so it can be corrected"
            }
          ]
        ]
      }
    ],
    "chain": true,
    "note": "The trough is the valid-event denominator. It is where an SLO is really defined, where it is later gamed, and the one thing the platform refuses to let a team leave implicit.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "07-layered-architecture",
    "title": "SLO and Error Budget Service — Layered Architecture",
    "layout": "bands",
    "canvas": {
      "width": 1720
    },
    "layerHeaderWidth": 162,
    "bands": [
      {
        "name": "Reader surface",
        "nodes": [
          {
            "id": "ui",
            "label": "Reliability console",
            "kind": "app"
          },
          {
            "id": "vapi",
            "label": "Verdict API",
            "sub": "signed, typed",
            "kind": "integration"
          },
          {
            "id": "qapi",
            "label": "Budget query API",
            "kind": "integration"
          },
          {
            "id": "rep",
            "label": "Monthly reports",
            "kind": "app"
          }
        ]
      },
      {
        "name": "Control plane",
        "nodes": [
          {
            "id": "recon",
            "label": "Definition reconciler",
            "sub": "from Git",
            "kind": "app"
          },
          {
            "id": "admit",
            "label": "Admission checks",
            "kind": "decision"
          },
          {
            "id": "pol",
            "label": "Policy engine",
            "kind": "app"
          },
          {
            "id": "excl",
            "label": "Exclusion workflow",
            "sub": "two approvers",
            "kind": "security"
          }
        ]
      },
      {
        "name": "Computation plane",
        "nodes": [
          {
            "id": "calc",
            "label": "Budget calculator",
            "kind": "app"
          },
          {
            "id": "wm",
            "label": "Window manager",
            "kind": "app"
          },
          {
            "id": "burn",
            "label": "Burn-rate evaluator",
            "kind": "app"
          },
          {
            "id": "rc",
            "label": "Recompute engine",
            "sub": "preemptible",
            "kind": "app"
          }
        ]
      },
      {
        "name": "Derived state",
        "nodes": [
          {
            "id": "bs",
            "label": "Budget projection",
            "sub": "cache, rebuildable",
            "kind": "store"
          },
          {
            "id": "rules",
            "label": "Compiled alert rules",
            "kind": "store"
          }
        ]
      },
      {
        "name": "State of record",
        "nodes": [
          {
            "id": "reg",
            "label": "SLO registry",
            "sub": "Azure SQL",
            "kind": "store"
          },
          {
            "id": "adx",
            "label": "SLI aggregates",
            "sub": "Data Explorer",
            "kind": "store"
          },
          {
            "id": "snap",
            "label": "Window snapshots",
            "sub": "immutable blob",
            "kind": "store"
          },
          {
            "id": "aud",
            "label": "Audit ledger",
            "sub": "tamper-evident",
            "kind": "store"
          }
        ]
      },
      {
        "name": "Ingest",
        "nodes": [
          {
            "id": "eh",
            "label": "Indicator stream",
            "sub": "Event Hubs",
            "kind": "queue"
          },
          {
            "id": "bkt",
            "label": "Event-time bucketer",
            "kind": "app"
          },
          {
            "id": "qt",
            "label": "Quarantine",
            "sub": "typed reasons",
            "kind": "risk"
          }
        ]
      },
      {
        "name": "Measurement plane — outside",
        "nodes": [
          {
            "id": "mon",
            "label": "Azure Monitor",
            "kind": "platform"
          },
          {
            "id": "prom",
            "label": "Managed Prometheus",
            "kind": "platform"
          },
          {
            "id": "probe",
            "label": "Synthetic probes",
            "kind": "external"
          }
        ]
      }
    ],
    "edges": [
      {
        "from": "mon",
        "to": "eh",
        "label": "outcome cube",
        "kind": "async"
      },
      {
        "from": "bkt",
        "to": "adx",
        "label": "minute buckets"
      },
      {
        "from": "recon",
        "to": "reg",
        "label": "version write"
      },
      {
        "from": "calc",
        "to": "adx",
        "label": "range scan",
        "kind": "bidirectional"
      },
      {
        "from": "calc",
        "to": "bs"
      },
      {
        "from": "wm",
        "to": "snap",
        "label": "close window"
      }
    ],
    "note": "Seven layers, and one seam that matters: the computation plane reads the state of record and writes only derived state. Nothing above the state-of-record layer is backed up, because nothing above it is data. The verdict API's read of the projection is omitted here and drawn in views 08 and 13.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "08-platform-components",
    "title": "Platform Components — Container View",
    "layout": "nested",
    "canvas": {
      "width": 1780
    },
    "boxes": [
      {
        "title": "Microsoft Azure — primary region",
        "kind": "cloud",
        "dir": "col",
        "children": [
          {
            "title": "Reader surface (Container Apps behind Front Door and API Management)",
            "kind": "boundary",
            "nodes": [
              {
                "id": "ui",
                "label": "Reliability console",
                "kind": "app"
              },
              {
                "id": "vapi",
                "label": "Verdict API",
                "sub": "signed, typed",
                "kind": "integration"
              },
              {
                "id": "qapi",
                "label": "Budget query API",
                "kind": "integration"
              },
              {
                "id": "rep",
                "label": "Report renderer",
                "sub": "monthly",
                "kind": "app"
              }
            ]
          },
          {
            "title": "Control plane (Container Apps)",
            "kind": "boundary",
            "nodes": [
              {
                "id": "recon",
                "label": "Definition reconciler",
                "kind": "app"
              },
              {
                "id": "admit",
                "label": "Admission checks",
                "kind": "decision"
              },
              {
                "id": "pol",
                "label": "Policy engine",
                "kind": "app"
              },
              {
                "id": "excl",
                "label": "Exclusion workflow",
                "kind": "security"
              }
            ]
          },
          {
            "title": "Computation plane (Container Apps, KEDA-scaled)",
            "kind": "boundary",
            "nodes": [
              {
                "id": "bkt",
                "label": "Event-time bucketer",
                "kind": "app"
              },
              {
                "id": "calc",
                "label": "Budget calculator",
                "kind": "app"
              },
              {
                "id": "wm",
                "label": "Window manager",
                "kind": "app"
              },
              {
                "id": "burn",
                "label": "Burn-rate evaluator",
                "kind": "app"
              },
              {
                "id": "rc",
                "label": "Recompute runner",
                "sub": "spot pool",
                "kind": "app"
              }
            ]
          },
          {
            "title": "State",
            "kind": "boundary",
            "nodes": [
              {
                "id": "reg",
                "label": "SLO registry",
                "sub": "Azure SQL",
                "kind": "store"
              },
              {
                "id": "adx",
                "label": "SLI aggregates",
                "sub": "Data Explorer",
                "kind": "store"
              },
              {
                "id": "bs",
                "label": "Budget projection",
                "sub": "Cache for Redis",
                "kind": "store"
              },
              {
                "id": "snap",
                "label": "Window snapshots",
                "sub": "immutable Blob",
                "kind": "store"
              },
              {
                "id": "aud",
                "label": "Audit ledger",
                "sub": "SQL ledger tables",
                "kind": "store"
              }
            ]
          }
        ]
      }
    ],
    "outside": [
      {
        "id": "eh",
        "label": "Indicator stream",
        "sub": "Event Hubs",
        "kind": "queue"
      },
      {
        "id": "git",
        "label": "Definitions repository",
        "kind": "external"
      },
      {
        "id": "gate",
        "label": "Release gate",
        "kind": "external"
      }
    ],
    "edges": [
      {
        "from": "calc",
        "to": "bs",
        "label": "materialise"
      },
      {
        "from": "vapi",
        "to": "bs",
        "label": "read"
      },
      {
        "from": "gate",
        "to": "vapi",
        "label": "verdict?"
      }
    ],
    "note": "Four tiers in one region, carrying only the three edges this view alone can show: a gate asks, the verdict API reads the projection, the calculator writes it. Nothing else a reader depends on is written by the computation plane. The inbound stream and the definitions repository are drawn in views 09 and 10, the signing path in view 20, and the secondary region in view 16.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "09-integration-surface",
    "title": "Integration Surface — Who Calls, Who Is Called",
    "layout": "hub",
    "canvas": {
      "width": 1700
    },
    "left": {
      "title": "Inbound",
      "nodes": [
        {
          "id": "eh",
          "label": "Indicator stream",
          "kind": "queue",
          "rel": "buckets",
          "kind2": "async"
        },
        {
          "id": "git",
          "label": "Definitions repository",
          "kind": "external",
          "rel": "SLOs"
        },
        {
          "id": "ui",
          "label": "Reliability console",
          "kind": "app",
          "rel": "HTTPS"
        },
        {
          "id": "inc",
          "label": "Incident platform",
          "kind": "external",
          "rel": "incidents",
          "kind2": "async"
        }
      ]
    },
    "centre": {
      "title": "SLO & Error Budget Service",
      "nodes": [
        {
          "id": "vapi",
          "label": "Verdict API",
          "sub": "signed, typed, cacheable",
          "kind": "integration"
        },
        {
          "id": "qapi",
          "label": "Budget query API",
          "sub": "budget, intervals, coverage",
          "kind": "integration"
        },
        {
          "id": "padm",
          "label": "Privileged API",
          "sub": "exclusions, overrides, policy",
          "kind": "security"
        }
      ]
    },
    "right": {
      "title": "Outbound",
      "nodes": [
        {
          "id": "gate",
          "label": "Release gate",
          "kind": "external",
          "rel": "verdict",
          "dir": "out"
        },
        {
          "id": "page",
          "label": "Action groups",
          "kind": "platform",
          "dir": "out",
          "kind2": "async"
        },
        {
          "id": "bi",
          "label": "Reporting sink",
          "kind": "external",
          "dir": "out",
          "kind2": "batch"
        }
      ]
    },
    "note": "Three surfaces separated by their authorisation story: anyone may read a verdict, a team may read its own detail, and three named roles may change what the arithmetic is allowed to say. The ingest path is deliberately not an API — it is a stream the platform consumes. The audit export is omitted here and appears in views 11 and 20.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "10-indicator-to-verdict-data-flow",
    "title": "Data Flow — Outcome Cube to Published Verdict",
    "layout": "flow",
    "canvas": {
      "width": 1780
    },
    "chain": true,
    "align": "middle",
    "nodeWidth": 184,
    "stages": [
      {
        "title": "Emit",
        "nodes": [
          {
            "id": "rr",
            "label": "Recording rules",
            "sub": "status × latency bucket",
            "kind": "platform"
          },
          {
            "id": "pr",
            "label": "Probe results",
            "sub": "journey checks",
            "kind": "external"
          }
        ]
      },
      {
        "title": "Transport",
        "nodes": [
          {
            "id": "eh",
            "label": "Indicator stream",
            "sub": "Event Hubs, partitioned",
            "kind": "queue"
          }
        ]
      },
      {
        "title": "Normalise",
        "nodes": [
          {
            "id": "late",
            "label": "Lateness gate",
            "sub": "10 min horizon",
            "kind": "decision"
          },
          {
            "id": "bkt",
            "label": "Event-time bucketer",
            "sub": "idempotent per bucket",
            "kind": "app"
          },
          {
            "id": "qt",
            "label": "Quarantine",
            "sub": "typed reason",
            "kind": "risk"
          }
        ]
      },
      {
        "title": "Measured truth",
        "nodes": [
          {
            "id": "adx",
            "label": "Minute buckets",
            "sub": "good, valid, coverage",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Derive",
        "nodes": [
          {
            "id": "red",
            "label": "Window reduce",
            "sub": "counters summed",
            "kind": "app"
          },
          {
            "id": "ov",
            "label": "Exclusion overlay",
            "sub": "annotation, not deletion",
            "kind": "app"
          },
          {
            "id": "br",
            "label": "Burn rate",
            "sub": "spend per window",
            "kind": "app"
          }
        ]
      },
      {
        "title": "Derived state",
        "nodes": [
          {
            "id": "bs",
            "label": "Budget projection",
            "sub": "watermarked",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Publish",
        "nodes": [
          {
            "id": "vd",
            "label": "Signed verdict",
            "sub": "with coverage",
            "kind": "integration"
          },
          {
            "id": "sn",
            "label": "Window snapshot",
            "sub": "immutable at close",
            "kind": "store"
          }
        ]
      }
    ],
    "note": "Good and valid counts stay separate the whole way across; no ratio is stored anywhere. The only irreversible step is the window close, and the only lossy one is a bucket that arrives past the horizon — which becomes a coverage deficit rather than a dropped event.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "11-storage-zones",
    "title": "Storage Zones — Record, Input, Derivation, Evidence",
    "layout": "nested",
    "canvas": {
      "width": 1760
    },
    "boxes": [
      {
        "title": "System of record — backed up, point-in-time restore",
        "kind": "boundary",
        "nodes": [
          {
            "id": "reg",
            "label": "SLO registry",
            "sub": "Azure SQL, RPO 1 min",
            "kind": "store"
          },
          {
            "id": "pol",
            "label": "Budget policies",
            "sub": "versioned",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Measured input — retained and replayable",
        "kind": "boundary",
        "nodes": [
          {
            "id": "adx",
            "label": "Minute buckets",
            "sub": "13 months hot/warm",
            "kind": "store"
          },
          {
            "id": "qt",
            "label": "Quarantined batches",
            "sub": "replayable, 30 d",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Derived — rebuildable, deliberately not backed up",
        "kind": "boundary",
        "nodes": [
          {
            "id": "bs",
            "label": "Budget projection",
            "sub": "Redis",
            "kind": "store"
          },
          {
            "id": "rules",
            "label": "Compiled alert rules",
            "kind": "store"
          },
          {
            "id": "rx",
            "label": "Report extracts",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Evidence — immutable, write-once",
        "kind": "trust",
        "nodes": [
          {
            "id": "snap",
            "label": "Window snapshots",
            "sub": "Blob WORM, 5 yr",
            "kind": "store"
          },
          {
            "id": "aud",
            "label": "Audit ledger",
            "sub": "SQL ledger, 5 yr",
            "kind": "store"
          },
          {
            "id": "ad",
            "label": "Alert decisions",
            "sub": "13 months",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Cold",
        "kind": "plain",
        "nodes": [
          {
            "id": "arch",
            "label": "Compressed aggregates",
            "sub": "archive tier",
            "kind": "store"
          }
        ]
      }
    ],
    "edges": [
      {
        "from": "adx",
        "to": "bs",
        "label": "project",
        "kind": "async"
      },
      {
        "from": "adx",
        "to": "snap",
        "label": "close window",
        "kind": "batch"
      },
      {
        "from": "adx",
        "to": "arch",
        "label": "tier at 90 d",
        "kind": "batch"
      },
      {
        "from": "qt",
        "to": "adx",
        "label": "replay",
        "kind": "batch"
      }
    ],
    "note": "Four classes, and the classification is the decision: anything rebuildable from the two left-hand zones is not backed up at all, and anything in the evidence zone has no update path. A restore of the derived zone is a recompute, which is why its objective is stated in minutes of compute rather than as an RPO.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "12-data-model",
    "title": "Data Model — Definition, Bucket, Budget, Verdict",
    "layout": "er",
    "canvas": {
      "width": 1740,
      "cols": 4
    },
    "rowGap": 258,
    "entities": [
      {
        "id": "service",
        "name": "service",
        "kind": "store",
        "row": 0,
        "col": 0,
        "attrs": [
          "service_id  PK",
          "name",
          "owning_team",
          "tier"
        ]
      },
      {
        "id": "journey",
        "name": "critical_journey",
        "kind": "store",
        "row": 0,
        "col": 1,
        "attrs": [
          "journey_id  PK",
          "name",
          "product_owner",
          "composite_rule  nullable"
        ]
      },
      {
        "id": "slo",
        "name": "slo",
        "kind": "store",
        "row": 0,
        "col": 2,
        "attrs": [
          "slo_id  PK",
          "journey_id  FK",
          "service_id  FK",
          "owning_team",
          "current_version  FK"
        ]
      },
      {
        "id": "ver",
        "name": "slo_version",
        "kind": "store",
        "row": 0,
        "col": 3,
        "attrs": [
          "version_id  PK",
          "slo_id  FK",
          "indicator_type",
          "good_predicate",
          "valid_predicate",
          "objective_pct",
          "window_len, alignment",
          "effective_from",
          "label_dims  bounded"
        ]
      },
      {
        "id": "bucket",
        "name": "sli_bucket",
        "kind": "store",
        "row": 1,
        "col": 0,
        "attrs": [
          "slo_id  PK1",
          "minute_utc  PK2",
          "good_count",
          "valid_count",
          "coverage  ok|no-data",
          "source_class",
          "ingested_at"
        ]
      },
      {
        "id": "win",
        "name": "budget_window",
        "kind": "store",
        "row": 1,
        "col": 1,
        "attrs": [
          "window_id  PK",
          "slo_id  FK",
          "version_id  FK",
          "kind  rolling|calendar",
          "opened_at, closes_at",
          "watermark_minute"
        ]
      },
      {
        "id": "snap",
        "name": "budget_snapshot",
        "kind": "store",
        "row": 1,
        "col": 2,
        "attrs": [
          "snapshot_id  PK",
          "window_id  FK",
          "total, consumed",
          "remaining_pct",
          "coverage_pct",
          "sealed_at  immutable"
        ]
      },
      {
        "id": "excl",
        "name": "exclusion_window",
        "kind": "store",
        "row": 1,
        "col": 3,
        "attrs": [
          "exclusion_id  PK",
          "slo_id  FK",
          "from_minute, to_minute",
          "reason_code",
          "approver_1, approver_2",
          "expires_at"
        ]
      },
      {
        "id": "pol",
        "name": "budget_policy",
        "kind": "store",
        "row": 2,
        "col": 0,
        "attrs": [
          "policy_id  PK",
          "scope  slo|service",
          "thresholds[]",
          "gate_mode  auto|advisory",
          "unreachable  open|closed",
          "exempt_classes[]"
        ]
      },
      {
        "id": "verd",
        "name": "verdict",
        "kind": "store",
        "row": 2,
        "col": 1,
        "attrs": [
          "verdict_id  PK",
          "slo_id  FK",
          "state  healthy|warning|",
          "  exhausted|insufficient",
          "version_id  FK",
          "computed_at",
          "coverage_pct",
          "signature, valid_until"
        ]
      },
      {
        "id": "rule",
        "name": "alert_rule",
        "kind": "store",
        "row": 2,
        "col": 2,
        "attrs": [
          "rule_id  PK",
          "slo_id  FK",
          "burn_multiple",
          "long_window, short_window",
          "class  page|ticket"
        ]
      },
      {
        "id": "dec",
        "name": "alert_decision",
        "kind": "store",
        "row": 2,
        "col": 3,
        "attrs": [
          "decision_id  PK",
          "rule_id  FK",
          "evaluated_at",
          "inputs  snapshot",
          "fired  bool",
          "suppressed_reason"
        ]
      }
    ],
    "relations": [
      {
        "from": "service",
        "to": "journey",
        "label": "N : M",
        "from_side": "e",
        "to_side": "w"
      },
      {
        "from": "journey",
        "to": "slo",
        "label": "1 : N",
        "from_side": "e",
        "to_side": "w"
      },
      {
        "from": "slo",
        "to": "ver",
        "label": "1 : N",
        "from_side": "e",
        "to_side": "w"
      },
      {
        "from": "slo",
        "to": "bucket",
        "label": "1 : N",
        "from_side": "s",
        "to_side": "n"
      },
      {
        "from": "win",
        "to": "snap",
        "label": "1 : 1",
        "from_side": "e",
        "to_side": "w"
      },
      {
        "from": "snap",
        "to": "excl",
        "label": "N : M",
        "from_side": "e",
        "to_side": "w"
      },
      {
        "from": "win",
        "to": "verd",
        "label": "1 : N",
        "from_side": "s",
        "to_side": "n"
      },
      {
        "from": "pol",
        "to": "verd",
        "label": "1 : N",
        "from_side": "e",
        "to_side": "w"
      },
      {
        "from": "verd",
        "to": "rule",
        "label": "N : M",
        "from_side": "e",
        "to_side": "w"
      },
      {
        "from": "rule",
        "to": "dec",
        "label": "1 : N",
        "from_side": "e",
        "to_side": "w"
      }
    ],
    "note": "Two keys carry the design. (slo_id, minute_utc) makes bucket writes idempotent, so a replayed batch cannot double-count. A sealed budget_snapshot has no update path, so a later exclusion produces a new annotation rather than a new history. budget_window carries version_id as an FK without a drawn relation, to keep the vertical runs legible; the override register and audit ledger appear in views 19 and 20.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "13-verdict-request-sequence",
    "title": "Critical Flow — A Release Asks Whether It Can Ship",
    "layout": "sequence",
    "canvas": {
      "width": 1740
    },
    "lifelines": [
      {
        "id": "gate",
        "label": "Release gate",
        "kind": "external"
      },
      {
        "id": "fd",
        "label": "Front Door / APIM",
        "kind": "integration"
      },
      {
        "id": "vapi",
        "label": "Verdict API",
        "kind": "app"
      },
      {
        "id": "bs",
        "label": "Budget projection",
        "kind": "store"
      },
      {
        "id": "reg",
        "label": "SLO registry",
        "kind": "store"
      },
      {
        "id": "hsm",
        "label": "Key Vault HSM",
        "kind": "security"
      }
    ],
    "messages": [
      {
        "from": "gate",
        "to": "fd",
        "label": "verdict? + workload token",
        "kind": "call"
      },
      {
        "from": "fd",
        "to": "fd",
        "label": "validate OIDC, rate limit",
        "kind": "self"
      },
      {
        "from": "fd",
        "to": "vapi",
        "label": "GET /verdict/checkout",
        "kind": "call"
      },
      {
        "from": "vapi",
        "to": "reg",
        "label": "policy + version",
        "kind": "call"
      },
      {
        "from": "reg",
        "to": "vapi",
        "label": "thresholds, gate mode",
        "kind": "return"
      },
      {
        "from": "vapi",
        "to": "bs",
        "label": "budget state",
        "kind": "call"
      },
      {
        "from": "bs",
        "to": "vapi",
        "label": "remaining, watermark",
        "kind": "return"
      },
      {
        "from": "vapi",
        "to": "vapi",
        "label": "staleness > ceiling?",
        "kind": "self"
      },
      {
        "from": "vapi",
        "to": "vapi",
        "label": "coverage < 95%?",
        "kind": "self"
      },
      {
        "from": "vapi",
        "to": "hsm",
        "label": "sign document",
        "kind": "call"
      },
      {
        "from": "hsm",
        "to": "vapi",
        "label": "signature",
        "kind": "return"
      },
      {
        "from": "vapi",
        "to": "gate",
        "label": "exhausted + coverage 99.8%",
        "kind": "return"
      },
      {
        "from": "gate",
        "to": "gate",
        "label": "verify signature, apply policy",
        "kind": "self"
      },
      {
        "from": "gate",
        "to": "gate",
        "label": "rollback class? exempt",
        "kind": "self"
      },
      {
        "from": "vapi",
        "to": "gate",
        "label": "unreachable: gate fails static",
        "kind": "error"
      }
    ],
    "note": "Fourteen messages, and the last one is the design. The platform never tells the gate to stop; it hands over a signed claim and the gate applies its own declared policy — including the one for the case where this conversation does not happen at all.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "14-burn-rate-evaluation",
    "title": "Burn-Rate Evaluation — From Buckets to a Page",
    "layout": "flow",
    "canvas": {
      "width": 1780
    },
    "chain": true,
    "align": "middle",
    "nodeWidth": 188,
    "stages": [
      {
        "title": "Read",
        "nodes": [
          {
            "id": "red",
            "label": "Window reduce",
            "sub": "sum counters",
            "kind": "app"
          },
          {
            "id": "cov",
            "label": "Coverage check",
            "sub": "floor 95%",
            "kind": "decision"
          }
        ]
      },
      {
        "title": "Normalise",
        "nodes": [
          {
            "id": "br",
            "label": "Burn rate",
            "sub": "1x exhausts at close",
            "kind": "app"
          }
        ]
      },
      {
        "title": "Evaluate",
        "nodes": [
          {
            "id": "fast",
            "label": "Fast pair",
            "sub": "14.4x over 1 h",
            "kind": "app"
          },
          {
            "id": "slow",
            "label": "Slow pair",
            "sub": "3x over 6 h",
            "kind": "app"
          }
        ]
      },
      {
        "title": "Confirm",
        "nodes": [
          {
            "id": "conf",
            "label": "Confirmation window",
            "sub": "5 min must agree",
            "kind": "decision"
          }
        ]
      },
      {
        "title": "Classify",
        "nodes": [
          {
            "id": "cls",
            "label": "Alert class",
            "sub": "reliability or measurement",
            "kind": "decision"
          },
          {
            "id": "grp",
            "label": "Grouping",
            "sub": "by dependency",
            "kind": "app"
          }
        ]
      },
      {
        "title": "Route",
        "nodes": [
          {
            "id": "pg",
            "label": "Page",
            "sub": "capped per incident",
            "kind": "external"
          },
          {
            "id": "tk",
            "label": "Ticket",
            "sub": "never pages",
            "kind": "external"
          },
          {
            "id": "mf",
            "label": "Measurement failure",
            "sub": "separate rotation",
            "kind": "risk"
          }
        ]
      },
      {
        "title": "Record",
        "nodes": [
          {
            "id": "dec",
            "label": "Alert decision",
            "sub": "inputs retained",
            "kind": "store"
          }
        ]
      }
    ],
    "note": "Two rules per SLO, generated from a template rather than authored, and one decision ahead of both: a blind SLO produces a measurement-failure alert and no reliability alert at all. Every evaluation is recorded with its inputs, which is what makes a missed page reconstructible instead of arguable.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "15-recompute-lanes",
    "title": "Recompute Lanes — Five Reasons History Changes",
    "layout": "swimlane",
    "canvas": {
      "width": 1780
    },
    "laneHeaderWidth": 170,
    "stages": [
      "Trigger",
      "Scope",
      "Schedule",
      "Compute",
      "Publish"
    ],
    "lanes": [
      {
        "title": "Late data",
        "cells": [
          [
            {
              "id": "t1",
              "label": "Bucket within horizon",
              "kind": "app"
            }
          ],
          [
            {
              "id": "s1",
              "label": "One minute",
              "kind": "app"
            }
          ],
          [
            {
              "id": "c1",
              "label": "Inline, continuous",
              "kind": "app"
            }
          ],
          [
            {
              "id": "p1",
              "label": "Re-sum the window",
              "sub": "seconds",
              "kind": "app"
            }
          ],
          [
            {
              "id": "u1",
              "label": "Projection advances",
              "kind": "store"
            }
          ]
        ]
      },
      {
        "title": "Definition change",
        "cells": [
          [
            {
              "id": "t2",
              "label": "New version merged",
              "kind": "app"
            }
          ],
          [
            {
              "id": "s2",
              "label": "One SLO, 28 days",
              "kind": "app"
            }
          ],
          [
            {
              "id": "c2",
              "label": "On demand",
              "kind": "decision"
            }
          ],
          [
            {
              "id": "c2b",
              "label": "Re-project the cube",
              "sub": "p95 90 s",
              "kind": "app"
            }
          ],
          [
            {
              "id": "u2",
              "label": "Corrected history",
              "kind": "store"
            }
          ]
        ]
      },
      {
        "title": "Exclusion approved",
        "cells": [
          [
            {
              "id": "t3",
              "label": "Two approvers",
              "kind": "security"
            }
          ],
          [
            {
              "id": "s3",
              "label": "An interval",
              "kind": "app"
            }
          ],
          [
            {
              "id": "c3",
              "label": "Immediate",
              "kind": "app"
            }
          ],
          [
            {
              "id": "c3b",
              "label": "Overlay, no rewrite",
              "sub": "buckets untouched",
              "kind": "app"
            }
          ],
          [
            {
              "id": "u3",
              "label": "Both figures kept",
              "kind": "store"
            }
          ]
        ]
      },
      {
        "title": "Full-estate backfill",
        "cells": [
          [
            {
              "id": "t4",
              "label": "Cube schema change",
              "kind": "app"
            }
          ],
          [
            {
              "id": "s4",
              "label": "5,000 SLOs, 13 mo",
              "kind": "app"
            }
          ],
          [
            {
              "id": "c4",
              "label": "Scheduled window",
              "sub": "spot capacity",
              "kind": "decision"
            }
          ],
          [
            {
              "id": "c4b",
              "label": "Partitioned by SLO",
              "sub": "≤ 6 h",
              "kind": "app"
            }
          ],
          [
            {
              "id": "u4",
              "label": "Read path unaffected",
              "kind": "store"
            }
          ]
        ]
      },
      {
        "title": "Shadow verification",
        "cells": [
          [
            {
              "id": "t5",
              "label": "Continuous",
              "kind": "platform"
            }
          ],
          [
            {
              "id": "s5",
              "label": "Sampled SLOs",
              "kind": "app"
            }
          ],
          [
            {
              "id": "c5",
              "label": "Always on",
              "kind": "app"
            }
          ],
          [
            {
              "id": "c5b",
              "label": "Rebuild and compare",
              "kind": "app"
            }
          ],
          [
            {
              "id": "u5",
              "label": "Divergence alert",
              "kind": "risk"
            }
          ]
        ]
      }
    ],
    "note": "Five reasons a published figure changes, through one engine with five schedules. Only the second and fourth cost real compute, and neither is allowed to touch the read path — which is why the recompute runner is a separate deployment on preemptible capacity rather than a mode of the calculator.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "16-deployment-architecture",
    "title": "Deployment — One Write Region, Recompute-Based Recovery",
    "layout": "nested",
    "canvas": {
      "width": 1780
    },
    "boxes": [
      {
        "title": "Microsoft Azure — primary write region",
        "kind": "cloud",
        "dir": "col",
        "children": [
          {
            "title": "Zone-redundant across three availability zones",
            "kind": "boundary",
            "nodes": [
              {
                "id": "ca",
                "label": "Container Apps env",
                "sub": "all tiers, KEDA",
                "kind": "app"
              },
              {
                "id": "sql",
                "label": "Azure SQL",
                "sub": "zone-redundant, PITR",
                "kind": "store"
              },
              {
                "id": "adx",
                "label": "Data Explorer",
                "sub": "3-node cluster",
                "kind": "store"
              },
              {
                "id": "redis",
                "label": "Cache for Redis",
                "sub": "zone-redundant",
                "kind": "store"
              }
            ]
          },
          {
            "title": "Regional shared services",
            "kind": "boundary",
            "nodes": [
              {
                "id": "fd",
                "label": "Front Door",
                "sub": "global anycast",
                "kind": "integration"
              },
              {
                "id": "apim",
                "label": "API Management",
                "kind": "integration"
              },
              {
                "id": "eh",
                "label": "Event Hubs",
                "sub": "zone-redundant",
                "kind": "queue"
              },
              {
                "id": "hsm",
                "label": "Key Vault Managed HSM",
                "kind": "security"
              }
            ]
          }
        ]
      },
      {
        "title": "Microsoft Azure — secondary region (warm, no writes)",
        "kind": "cloud",
        "nodes": [
          {
            "id": "sqlr",
            "label": "Registry geo-replica",
            "sub": "read-only",
            "kind": "store"
          },
          {
            "id": "adxf",
            "label": "Data Explorer follower",
            "sub": "aggregates",
            "kind": "store"
          },
          {
            "id": "car",
            "label": "Container Apps",
            "sub": "scaled to zero",
            "kind": "app"
          },
          {
            "id": "blob",
            "label": "Snapshots + audit",
            "sub": "RA-GRS, immutable",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Third region — independent watchdog",
        "kind": "trust",
        "nodes": [
          {
            "id": "wd",
            "label": "Platform watchdog",
            "sub": "Functions, own SLOs",
            "kind": "platform"
          }
        ]
      }
    ],
    "edges": [
      {
        "from": "sql",
        "to": "sqlr",
        "label": "geo-replicate",
        "kind": "async"
      },
      {
        "from": "wd",
        "to": "fd",
        "label": "probes the verdict API",
        "kind": "bidirectional"
      }
    ],
    "note": "One write region, because two regions computing the same budget from partially overlapping buckets produce two figures that no key reconciles. The aggregate cluster is followed in the secondary alongside the registry replica. Failover is a promotion plus a recompute, targeting 90 minutes to a trustworthy projection rather than seconds to a stale one. The watchdog is deliberately elsewhere: a platform cannot be the sole authority on its own availability.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "17-observability",
    "title": "Observability — Signal Type by Pipeline Stage",
    "layout": "grid",
    "canvas": {
      "width": 1780
    },
    "laneHeaderWidth": 164,
    "columns": [
      "Ingest",
      "Compute",
      "Publish",
      "Alert",
      "Policy"
    ],
    "rows": [
      {
        "title": "The alert",
        "cells": [
          [
            {
              "label": "Oldest unbucketed minute",
              "sub": "> 10 min pages",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Projection staleness",
              "sub": "> 120 s pages",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Verdict error rate",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Evaluation lag",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Override rate",
              "sub": "weekly review",
              "kind": "risk"
            }
          ]
        ]
      },
      {
        "title": "Metrics",
        "cells": [
          [
            {
              "label": "Buckets/s by source",
              "kind": "platform"
            },
            {
              "label": "Late fraction",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Watermark lag",
              "kind": "platform"
            },
            {
              "label": "Recompute minutes",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Verdict p99",
              "kind": "platform"
            },
            {
              "label": "Cache hit rate",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Rules evaluated/min",
              "kind": "platform"
            },
            {
              "label": "Pages per rotation",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Exclusion minutes",
              "kind": "platform"
            }
          ]
        ]
      },
      {
        "title": "Logs and traces",
        "cells": [
          [
            {
              "label": "Quarantine reasons",
              "kind": "store"
            }
          ],
          [
            {
              "label": "Window close record",
              "kind": "store"
            }
          ],
          [
            {
              "label": "Signed verdict log",
              "kind": "store"
            }
          ],
          [
            {
              "label": "Alert decision inputs",
              "kind": "store"
            }
          ],
          [
            {
              "label": "Audit ledger",
              "kind": "store"
            }
          ]
        ]
      },
      {
        "title": "The platform's own SLOs",
        "cells": [
          [
            {
              "label": "Bucket freshness",
              "sub": "p95 60 s",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Recompute within 90 s",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Verdict availability",
              "sub": "99.95%",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Detection within 5 min",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Two-approver compliance",
              "kind": "app"
            }
          ]
        ]
      },
      {
        "title": "What nothing else sees",
        "cells": [
          [
            {
              "label": "Coverage per SLO",
              "sub": "the blind-SLO signal",
              "kind": "decision"
            }
          ],
          [
            {
              "label": "Shadow divergence",
              "kind": "decision"
            }
          ],
          [
            {
              "label": "Insufficient-data rate",
              "kind": "decision"
            }
          ],
          [
            {
              "label": "Page precision",
              "sub": "vs incident records",
              "kind": "decision"
            }
          ],
          [
            {
              "label": "Exception frequency",
              "kind": "decision"
            }
          ]
        ]
      }
    ],
    "note": "The bottom row is the reason this view exists. A platform that is up, fast and wrong passes every row above it: only coverage, shadow divergence and page precision can tell you that the arithmetic has quietly stopped describing anything.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "18-slo-lifecycle",
    "title": "SLO Lifecycle — The Loop That Has to Close",
    "layout": "cycle",
    "canvas": {
      "width": 1620
    },
    "centre": {
      "label": "One SLO",
      "sub": "one version at a time"
    },
    "nodes": [
      {
        "id": "decl",
        "label": "Declared",
        "sub": "in the repository",
        "kind": "app"
      },
      {
        "id": "adm",
        "label": "Admitted",
        "sub": "resolution checked",
        "kind": "decision"
      },
      {
        "id": "meas",
        "label": "Measured",
        "sub": "buckets + coverage",
        "kind": "store"
      },
      {
        "id": "der",
        "label": "Derived",
        "sub": "budget + burn rate",
        "kind": "app"
      },
      {
        "id": "pub",
        "label": "Published",
        "sub": "signed verdict",
        "kind": "integration"
      },
      {
        "id": "act",
        "label": "Acted on",
        "sub": "gate, page, report",
        "kind": "external"
      },
      {
        "id": "rev",
        "label": "Reviewed",
        "sub": "attainment + precision",
        "kind": "platform"
      },
      {
        "id": "am",
        "label": "Amended",
        "sub": "new version",
        "kind": "app"
      }
    ],
    "ringLabels": [
      "reject the unmeasurable",
      "event-time, 10 min",
      "counters summed",
      "sign and type",
      "someone decides",
      "was the page real?",
      "effective-from",
      "pull request"
    ],
    "rx": 452,
    "ry": 218,
    "note": "The loop only closes at \"reviewed\": without measuring whether the pages were real and whether attainment is drifting, an SLO becomes a number that is maintained rather than used. The amendment step is a pull request, which is why a correction produces a corrected history rather than a silent rewrite.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "19-security-trust-zones",
    "title": "Security — Trust Zones and What Crosses Them",
    "layout": "zones",
    "canvas": {
      "width": 1740
    },
    "zones": [
      {
        "title": "Internet — untrusted",
        "kind": "trust",
        "nodes": [
          {
            "id": "eng",
            "label": "Engineer",
            "kind": "actor"
          },
          {
            "id": "gate",
            "label": "Release gate",
            "kind": "external"
          },
          {
            "id": "forge",
            "label": "Replayed verdict",
            "kind": "risk"
          }
        ]
      },
      {
        "title": "Perimeter",
        "kind": "trust",
        "nodes": [
          {
            "id": "fd",
            "label": "Front Door",
            "sub": "TLS, WAF",
            "kind": "integration"
          },
          {
            "id": "apim",
            "label": "API Management",
            "sub": "quota, throttle",
            "kind": "integration"
          },
          {
            "id": "oidc",
            "label": "Entra ID verification",
            "sub": "workload federation",
            "kind": "security"
          }
        ]
      },
      {
        "title": "Read zone — broad access",
        "kind": "trust",
        "nodes": [
          {
            "id": "vapi",
            "label": "Verdict API",
            "kind": "integration"
          },
          {
            "id": "qapi",
            "label": "Budget query API",
            "sub": "tenant-scoped",
            "kind": "integration"
          },
          {
            "id": "bs",
            "label": "Budget projection",
            "kind": "store"
          }
        ]
      },
      {
        "title": "Privilege zone — four named grants",
        "kind": "trust",
        "nodes": [
          {
            "id": "rbac",
            "label": "Role check",
            "sub": "author ǀ approve ǀ grant ǀ read",
            "kind": "security"
          },
          {
            "id": "padm",
            "label": "Privileged API",
            "kind": "security"
          },
          {
            "id": "sod",
            "label": "Separation of duties",
            "sub": "author ≠ approver",
            "kind": "decision"
          },
          {
            "id": "hsm",
            "label": "Key Vault HSM",
            "sub": "signing key, non-exportable",
            "kind": "security"
          }
        ]
      },
      {
        "title": "Evidence zone — no update path",
        "kind": "trust",
        "nodes": [
          {
            "id": "snap",
            "label": "Window snapshots",
            "sub": "WORM",
            "kind": "store"
          },
          {
            "id": "aud",
            "label": "Audit ledger",
            "sub": "tamper-evident",
            "kind": "store"
          }
        ]
      }
    ],
    "edges": [
      {
        "from": "eng",
        "to": "fd",
        "label": "HTTPS"
      },
      {
        "from": "gate",
        "to": "apim",
        "label": "workload token"
      },
      {
        "from": "apim",
        "to": "oidc",
        "label": "verify"
      },
      {
        "from": "oidc",
        "to": "vapi",
        "label": "authenticated"
      },
      {
        "from": "oidc",
        "to": "rbac",
        "label": "claims"
      },
      {
        "from": "rbac",
        "to": "padm",
        "label": "authorise"
      },
      {
        "from": "padm",
        "to": "sod",
        "label": "second approver"
      },
      {
        "from": "sod",
        "to": "aud",
        "label": "prior value"
      },
      {
        "from": "forge",
        "to": "apim",
        "label": "valid_until rejects",
        "kind": "error"
      }
    ],
    "note": "Five zones, and the asymmetry is the point: reading a verdict is deliberately easy, and the two privileges that can make a breached objective appear met — approving an exclusion and granting an override — sit behind separation of duties and an append-only record. A verdict carries a validity period precisely because a signed document is otherwise replayable forever. The verdict API's read of the projection is drawn in view 08.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "20-privilege-and-signing-flow",
    "title": "Identity and Privilege — Excluding Forty Minutes",
    "layout": "sequence",
    "canvas": {
      "width": 1740
    },
    "lifelines": [
      {
        "id": "a1",
        "label": "SLO author",
        "kind": "actor"
      },
      {
        "id": "a2",
        "label": "Reliability lead",
        "kind": "actor"
      },
      {
        "id": "entra",
        "label": "Entra ID",
        "kind": "security"
      },
      {
        "id": "padm",
        "label": "Privileged API",
        "kind": "security"
      },
      {
        "id": "reg",
        "label": "SLO registry",
        "kind": "store"
      },
      {
        "id": "aud",
        "label": "Audit ledger",
        "kind": "store"
      },
      {
        "id": "hsm",
        "label": "Key Vault HSM",
        "kind": "security"
      }
    ],
    "messages": [
      {
        "from": "a1",
        "to": "entra",
        "label": "sign in + MFA",
        "kind": "call"
      },
      {
        "from": "entra",
        "to": "a1",
        "label": "claims: author",
        "kind": "return"
      },
      {
        "from": "a1",
        "to": "padm",
        "label": "exclude 22:10–22:48",
        "kind": "call"
      },
      {
        "from": "padm",
        "to": "padm",
        "label": "reason code present?",
        "kind": "self"
      },
      {
        "from": "padm",
        "to": "a1",
        "label": "author cannot approve",
        "kind": "error"
      },
      {
        "from": "a2",
        "to": "entra",
        "label": "sign in + MFA",
        "kind": "call"
      },
      {
        "from": "a2",
        "to": "padm",
        "label": "approve, expiry 30 d",
        "kind": "call"
      },
      {
        "from": "padm",
        "to": "padm",
        "label": "> 60 min? second approver",
        "kind": "self"
      },
      {
        "from": "padm",
        "to": "aud",
        "label": "actor, reason, prior value",
        "kind": "call"
      },
      {
        "from": "padm",
        "to": "reg",
        "label": "annotate, never delete",
        "kind": "call"
      },
      {
        "from": "reg",
        "to": "padm",
        "label": "both figures retrievable",
        "kind": "return"
      },
      {
        "from": "padm",
        "to": "hsm",
        "label": "re-sign current verdict",
        "kind": "call"
      },
      {
        "from": "hsm",
        "to": "padm",
        "label": "signature, valid 5 min",
        "kind": "return"
      },
      {
        "from": "padm",
        "to": "a2",
        "label": "applied, counted in register",
        "kind": "return"
      }
    ],
    "note": "The refused message is the design. The person who wrote the SLO cannot approve the removal of a minute from it, because an exclusion privilege in the hands of the measured party is not a control. Everything afterwards is recorded before it is applied.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  },
  {
    "id": "21-failure-classes",
    "title": "Failure Classes and Their Structural Answers",
    "layout": "grid",
    "canvas": {
      "width": 1780
    },
    "laneHeaderWidth": 176,
    "columns": [
      "How it shows",
      "Detection",
      "Structural answer",
      "Accepted residual"
    ],
    "rows": [
      {
        "title": "Telemetry gap",
        "cells": [
          [
            {
              "label": "Buckets stop arriving",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Coverage per SLO",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "no-data + measurement alert",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Blind minutes unrecoverable",
              "kind": "risk"
            }
          ]
        ]
      },
      {
        "title": "Late data",
        "cells": [
          [
            {
              "label": "Bucket arrives at +12 min",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Lateness histogram",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Horizon, then coverage deficit",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Figure slightly pessimistic",
              "kind": "risk"
            }
          ]
        ]
      },
      {
        "title": "Poison input",
        "cells": [
          [
            {
              "label": "Implausible counter jump",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Tolerance on closed windows",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Quarantine, typed, replayable",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Coverage drops meanwhile",
              "kind": "risk"
            }
          ]
        ]
      },
      {
        "title": "Definition error",
        "cells": [
          [
            {
              "label": "Budget wrong, not missing",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Dry-run, shadow divergence",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Versioned rollback + recompute",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Verdicts served meanwhile",
              "kind": "risk"
            }
          ]
        ]
      },
      {
        "title": "Projection staleness",
        "cells": [
          [
            {
              "label": "Budget lags the buckets",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Watermark vs now",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Ceiling → insufficient-data",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Gate loses its input",
              "kind": "risk"
            }
          ]
        ]
      },
      {
        "title": "Verdict API down",
        "cells": [
          [
            {
              "label": "Gate cannot ask",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Independent watchdog",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Gate fails static on cache",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Policy weakens with age",
              "kind": "risk"
            }
          ]
        ]
      },
      {
        "title": "Alert storm",
        "cells": [
          [
            {
              "label": "200 SLOs breach at once",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Pages per rotation",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Group by dependency, cap pages",
              "kind": "app"
            }
          ],
          [
            {
              "label": "One real page may be grouped",
              "kind": "risk"
            }
          ]
        ]
      },
      {
        "title": "Regional loss",
        "cells": [
          [
            {
              "label": "Write region gone",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Health probes, watchdog",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Promote, then recompute",
              "kind": "app"
            }
          ],
          [
            {
              "label": "90 min of insufficient-data",
              "kind": "risk"
            }
          ]
        ]
      },
      {
        "title": "Privilege misuse",
        "cells": [
          [
            {
              "label": "Breach made to look met",
              "kind": "risk"
            }
          ],
          [
            {
              "label": "Exclusion minutes, override rate",
              "kind": "platform"
            }
          ],
          [
            {
              "label": "Two approvers, append-only",
              "kind": "app"
            }
          ],
          [
            {
              "label": "Collusion defeats it",
              "kind": "risk"
            }
          ]
        ]
      }
    ],
    "note": "Nine classes, each with a residual nobody designs away. The two that would change the architecture are in the right-hand column: if blind minutes became common the platform would have to own collection after all, and if collusion on exclusions were plausible the figure would need an external witness.",
    "meta": {
      "v": "1.0",
      "owner": "Reliability Architecture",
      "date": "2026-10"
    }
  }
]
