# yaml-language-server: $schema=https://promptfoo.dev/config-schema.json
#
# Controlled pure-text experiment: the prompt and six cases stay fixed;
# only the model changes between the three runs.

description: "The Switch Test - controlled text-model comparison"

prompts:
  - file://tests/task_prompt.txt

providers:
  # Open-weights model
  - id: openrouter:deepseek/deepseek-v4-pro
    config:
      temperature: 0
      max_tokens: 3000
      maxRetries: 0
      showThinking: false
      passthrough:
        reasoning:
          effort: high
          exclude: true

  # Open-weights model; formally replaces the API-incomplete Kimi K3 row
  - id: openrouter:z-ai/glm-5.2
    config:
      temperature: 0
      max_tokens: 3000
      maxRetries: 0
      showThinking: false
      passthrough:
        reasoning:
          effort: high
          exclude: true

  # Closed model; Anthropic is unavailable for the account billing region
  - id: openrouter:x-ai/grok-4.5
    config:
      temperature: 0
      max_tokens: 3000
      maxRetries: 0
      showThinking: false
      passthrough:
        reasoning:
          effort: high
          exclude: true

tests: file://tests/cases.yaml

commandLineOptions:
  maxConcurrency: 1
  repeat: 1
  share: false
  cache: false
