# Opt-in local quota example. Nothing here calls a real supplier.
# See docs/developer/backend-extensions.md#quota-reporting for startup commands.
# The example extension queries the fixture's /usage endpoint independently
# of inference; matching `provider`, `usage_label`, and `unit` joins the two.
models:
  - id: quota-demo
    name: Local Quota Demo
    provider: openai_compat
    context_length: 4096
    max_output_length: 1024
    supports_tools: false
    supported_params: [max_tokens, stream]
    input_modalities: [text]
    output_modalities: [text]
    pricing:
      prompt: "1.00"
      completion: "2.00"
    router: routewise
    router_params:
      budget_alpha: 1.0
      quota_snapshot_refresh_interval_sec: 5.0
      routewise_probe_enabled: true
      routewise_probe_interval_sec: 5.0
      routewise_probe_idle_only: false
    route:
      # A priced fallback supplies cost-envelope samples on cold start and
      # continues serving if the quota is exhausted or never becomes ready.
      - kind: openai_compat
        provider: example_fallback
        weight: 1.0
        base_url: http://127.0.0.1:18352/v1
        api_key: example-local-key
        provider_model_id: example-upstream
        provider_type: on_demand
        pricing:
          prompt: "1.00"
          completion: "2.00"
      - kind: openai_compat
        provider: example_quota
        weight: 1.0
        base_url: http://127.0.0.1:18353/v1
        api_key: example-local-key
        provider_model_id: example-upstream
        provider_type: quota
        quota_pool: example-daily
        quota:
          limit: 100
        quota_source:
          provider: example_quota
          usage_label: Daily requests
          unit: requests
