{
  "format": "bayeto-decision-record",
  "formatVersion": 3,
  "formatNote": "v3 adds decisionHash. Additive: v2 records remain valid and re-derivable — no v2 field changed meaning, moved, or was removed.",
  "record": {
    "action": "Mark the shared prefix with cache_control on the 60.2% of Research Copilot calls not yet caching it",
    "applicationId": "research-copilot",
    "beforeAfter": {
      "addedEdges": [],
      "addedNodes": [],
      "after": "Research Copilot: Mark the shared prefix with cache_control on the 60.2% of Research Copilot calls not yet caching it",
      "before": "Research Copilot: static prompt prefix re-billed at the full input rate on every request",
      "changedEdges": [
        "app:research-copilot"
      ]
    },
    "benchmark": {
      "context": "Share of requests reading a warm prompt-cache prefix on cache-priced models. (illustrative)",
      "median": 0.55,
      "metric": "prompt_cache_coverage",
      "topQuartile": 0.8,
      "unit": "ratio",
      "yourValue": 0.3984
    },
    "businessValueEur": 4964.24,
    "complexity": "Low",
    "confidence": 0.6712,
    "contextCompleteness": {
      "absent": [
        "Codebase",
        "Quality evaluation"
      ],
      "present": [
        "Telemetry",
        "Provider bill",
        "Per-app criticality",
        "Prior experiments"
      ],
      "presentCount": 4,
      "totalDims": 6
    },
    "costModelBasis": {
      "agentRunAllowanceEur": 20,
      "byHand": {
        "effortCostEur": 1600,
        "effortDays": 2,
        "engineeringDayRateEur": 800
      },
      "effortCostEur": 220,
      "rateBasis": "prior",
      "reviewHourRateEur": 100,
      "reviewHours": 2,
      "reviewHoursBasis": "prior",
      "version": "agent-assisted-v1-2026-08"
    },
    "criticalReview": {
      "alternativeExplanations": [
        "The prompt prefix may rotate per request (dynamic context first) — then there is nothing stable to cache."
      ],
      "assumptions": [
        "Assumes the uncovered requests share the SAME prefix the covered ones already cache."
      ],
      "failureConditions": [
        "Wrong if request cadence exceeds the cache TTL (the prefix expires before reuse), or if the prefix genuinely varies per call."
      ],
      "missingEvidence": [
        "No proof the uncovered cohort's prefix is byte-identical to the covered cohort's — same template is inferred, not measured.",
        "No independent quality evaluation — the quality impact is unmeasured."
      ]
    },
    "demoted": null,
    "effort": "Low",
    "evidenceIds": [
      "ev:research-copilot:prompt_cache_adoption"
    ],
    "expectedImpact": {
      "annualSavingsEur": 2606.46,
      "costDeltaPct": -0.1031,
      "latencyDeltaPct": -0.05,
      "qualityDeltaPct": 0,
      "qualityRisk": "none"
    },
    "id": "rec:research-copilot:prompt_cache_adoption",
    "inferenceId": "inf:research-copilot:prompt_cache_adoption",
    "latencyValueEur": 2357.78,
    "memoryNote": "No prior implementations yet — confidence derived from evidence only.",
    "objectiveKey": "highest_quality",
    "observationIds": [
      "obs:research-copilot:cache_hit_rate",
      "obs:research-copilot:avg_input_tokens"
    ],
    "pareto": true,
    "paybackDays": 31,
    "priceBasis": "scenario",
    "priorityScore": 1749.46,
    "qualityOfEvidence": "Medium",
    "roi": {
      "affectedRequestsMonthly": 31200,
      "annualSavingsEur": 2606.46,
      "effortDays": 2,
      "monthlySavingsEur": 217.21,
      "paybackDays": 31,
      "verified": false
    },
    "ruleId": "prompt_cache_adoption",
    "sampleSize": 24960,
    "targetNodeId": "app:research-copilot",
    "title": "Extend prompt caching across Research Copilot",
    "whyNot": [
      {
        "costDeltaPct": -0.15,
        "latencyDeltaPct": -0.05,
        "option": "Compress the prompt instead",
        "paretoOptimal": false,
        "qualityDeltaPct": null,
        "qualityRisk": "low",
        "reason": "Cuts the same tokens but risks quality and has a learned under-delivery penalty. Caching KEEPS the tokens and just stops paying full rate for them."
      },
      {
        "costDeltaPct": -0.3,
        "latencyDeltaPct": 0,
        "option": "Right-size the model instead",
        "paretoOptimal": false,
        "qualityDeltaPct": null,
        "qualityRisk": "medium",
        "reason": "A bigger cut, but it carries an UNMEASURED quality risk — do the model move first if warranted, then cache on the new model (caches are per-model). Not Pareto here because caching is quality-neutral and this isn't."
      },
      {
        "costDeltaPct": -0.2,
        "latencyDeltaPct": -0.15,
        "option": "Semantic response cache",
        "paretoOptimal": false,
        "qualityDeltaPct": 0,
        "qualityRisk": "none",
        "reason": "Serves repeated ANSWERS, not shared prefixes — a different mechanism needing a similarity threshold and staleness policy. Orthogonal; both can be right."
      }
    ]
  },
  "decisionHash": "sha256:e4c52b2cdeda12dc2165511750c76818bfcae98d744f23e383acc613c5822897",
  "narrative": {
    "note": "Phrased by the AI Fabric from deterministic drafts. Not part of the decision; changing it cannot change any number above. `provenance` records which provider source control approved, whether it served, and why not — source:\"draft\" means you are reading the engine's own deterministic wording.",
    "business": "Research Copilot already caches on some traffic but pays full input rate on the rest of the same prefix; closing that gap is quality-neutral savings — provided the shared prefix is verified byte-identical across the uncovered cohort. Estimated €2,606/yr at the current run-rate, payback 31 days, confidence 67%.",
    "technical": "Place the stable prefix first and mark it with the provider's cache_control; verify cache-read tokens climb in the next export. Caches are per-model, so re-do this after any model change. Affects ~31,200 req/mo; expected cost -10.3%, latency -5%, quality unchanged by construction.",
    "phrasedBy": "mock",
    "provenance": {
      "narrativeProvenanceVersion": 1,
      "attempted": "mock",
      "policyVersion": "fabric-v1-2026-08",
      "source": "provider",
      "degraded": false
    }
  },
  "ledgerState": "lead",
  "ledgerVersion": "22c35e346df765c3",
  "overlapGroup": {
    "applicationId": "research-copilot",
    "memberIds": [
      "rec:research-copilot:prompt_cache_adoption",
      "rec:research-copilot:overpowered_model",
      "rec:research-copilot:prompt_compression"
    ],
    "leadId": "rec:research-copilot:prompt_cache_adoption"
  },
  "simulation": {
    "formulaVersion": "sim-v1",
    "formula": "realization = 0.7 + 0.3 × confidence (scenario assumption)",
    "inputs": {
      "baseAnnualSavingsEur": 2606.46,
      "confidence": 0.6712
    },
    "intermediates": {
      "realization": 0.901
    },
    "outputs": {
      "annualSavingsEur": 2349.36,
      "downsideAnnualSavingsEur": 1824.52,
      "upsideAnnualSavingsEur": 2606.46
    }
  },
  "measured": null,
  "run": {
    "engineVersion": "0.1.0",
    "workspaceSourceKind": "demo",
    "stateVersion": "none|none|0|0:none",
    "datasetHash": "c8030438f36955eebab44a8b2fe9a468df7d9592992ba71e8145de0a9965ce23",
    "pricing": {
      "catalogVersion": "thornbury-scenario-v0",
      "unresolved": null,
      "rows": [
        {
          "family": "claude-haiku",
          "rates": {
            "inputPer1M": 1,
            "outputPer1M": 5,
            "cacheReadPer1M": 0.1,
            "cacheWritePer1M": 1.25
          },
          "provenance": {
            "source": "bayeto-scenario",
            "capturedAt": "2026-07-21",
            "quality": "scenario"
          }
        },
        {
          "family": "claude-sonnet",
          "rates": {
            "inputPer1M": 2,
            "outputPer1M": 10,
            "cacheReadPer1M": 0.2,
            "cacheWritePer1M": 2.5
          },
          "provenance": {
            "source": "bayeto-scenario",
            "capturedAt": "2026-07-21",
            "quality": "scenario"
          }
        }
      ]
    },
    "windowDays": 24,
    "windowStart": "2026-06-06T00:00:00.000Z",
    "windowEnd": "2026-06-29T23:58:51.723Z",
    "coverage": {
      "rows": 124800,
      "activeDays": 24,
      "unattributedRows": 0
    },
    "fxRates": {
      "USD_EUR": 0.92,
      "GBP_EUR": 1.17,
      "effectiveDate": "2026-07-06",
      "basis": "fixed reference rates (determinism; never live FX)"
    },
    "implementationCostModel": {
      "version": "agent-assisted-v1-2026-08",
      "reviewHourRateEur": 100,
      "agentRunAllowanceEur": 20,
      "basis": "agent-assisted priors (review hours × rate + agent-run allowance); byHand comparison at the blended day rate"
    },
    "engineeringDayRateEur": 800,
    "evidenceTaxonomyVersion": "evtax-v1",
    "formulaVersions": {
      "accuracy": "pvm-v1",
      "simulation": "sim-v1"
    },
    "generatedAt": "2026-08-28T18:33:21.126Z"
  }
}
