{
  "module": "tokensaver",
  "version": "2.2.0",
  "what_it_is": "A deterministic gate in front of a model. It decides whether a request is answered from store, sent to the model, sent to a cheaper model, held for a person, or refused. It also names the waste inside every request it sees.",
  "model_calls_made_to_reach_a_decision": 0,
  "layers": {
    "1_hard_rules": {
      "why": "A cost gate that depends only on tuned weights is a cost gate nobody can defend. These are absolute.",
      "rules": {
        "budget_exhausted": "spend has reached the key's ceiling",
        "exceeds_remaining_budget": "this one call could cost more than the budget left. Output uses the exact ceiling you set; input is estimated from characters, so this rule is deliberately cautious. Held for a person when a human is declared, refused when one is not.",
        "runaway_loop": "the same request 8 times in 120 seconds",
        "runaway_loop_unattended": "the same request 4 times in 120 seconds with no human declared",
        "runaway_burst": "120 requests from one key in 60 seconds"
      }
    },
    "2_the_score": {
      "spend_signals": {
        "exposure": {
          "weight": 0.18,
          "measures": "worst case cost of this call against the budget left"
        },
        "size": {
          "weight": 0.14,
          "saturates_at": 100000,
          "measures": "prompt characters, log scaled"
        },
        "ask": {
          "weight": 0.14,
          "saturates_at": 8000,
          "measures": "the max_tokens ceiling the caller set"
        },
        "depth": {
          "weight": 0.1,
          "saturates_at": 40,
          "measures": "turns re-sent on every call"
        },
        "tools": {
          "weight": 0.06,
          "saturates_at": 24,
          "measures": "tool definitions re-sent on every call"
        }
      },
      "waste_signals": {
        "loop": {
          "weight": 0.16,
          "saturates_at": 5
        },
        "burst": {
          "weight": 0.09,
          "saturates_at": 20
        },
        "grind": {
          "weight": 0.07,
          "saturates_at": 200
        },
        "novelty": {
          "weight": 0.06
        }
      },
      "spend_weight_total": 0.62,
      "waste_weight_total": 0.38,
      "base_weights_sum_to": 1.0,
      "outside_the_base_sum": {
        "unattended": 0.1
      },
      "bands": {
        "CHALLENGE": ">= 0.55",
        "BLOCK": ">= 0.80"
      },
      "downgrade_is_earned_not_suspected": {
        "max_score": 0.3,
        "max_prompt_characters": 4000,
        "max_turns": 6,
        "max_output_tokens": 1000,
        "tools_allowed": 0,
        "why": "a suspicious request is never sent to a weaker model. Only a genuinely small one is."
      }
    },
    "3_findings": {
      "why": "The verdict saves money on this call. The findings change what the caller sends next time, which saves far more.",
      "codes": [
        "repeating_without_recording",
        "varied_output_blocks_reuse",
        "carrying_old_turns",
        "unused_tool_definitions",
        "large_system_prompt_resent",
        "output_ceiling_authorised",
        "duplicate_turns_in_context",
        "budget_nearly_gone"
      ]
    }
  },
  "verdict_vocabulary": {
    "SERVE": "answered from an identical earlier request; nothing was bought",
    "ALLOW": "send it to the model as asked",
    "DOWNGRADE": "small and simple enough for the cheap model",
    "CHALLENGE": "hold it for a person before spending",
    "BLOCK": "refused; it never reaches the model, so no completion is paid for"
  },
  "certainty_tiers": {
    "tokens_not_bought": "exact, provider reported, the only figure that enters a savings total",
    "worst_case_tokens_avoided": "a ceiling on what a refused request could have cost, reported separately",
    "findings_tokens": "estimated from characters where marked, never entering any total"
  },
  "two_ways_to_call_it": {
    "request": "send the provider request body. This platform sees your prompt.",
    "digest": "send only a fingerprint and counts. Your prompts and answers never leave your building, the decision is identical, and the receipt is the same. The downloaded client uses this path by default."
  },
  "honest_limits": [
    "Matching is exact. A reworded prompt is a different request and goes to the model.",
    "Savings totals use only token counts the provider itself reported. Nothing in a total is estimated.",
    "A refused request has a worst case cost, not a known cost. It is reported separately and never added to the savings total.",
    "Token figures inside findings are estimated from character counts and are marked as estimates. They never enter a total.",
    "A stored answer is returned unchanged. This module does not judge whether it is still correct.",
    "No model is called to reach any decision here."
  ],
  "routes": {
    "public": [
      "spec",
      "stats",
      "verify"
    ],
    "keyed": [
      "estimate",
      "gate",
      "record",
      "ledger",
      "budget",
      "prices",
      "forget"
    ]
  }
}