{
  "schemaVersion": 3,
  "id": "tus-ttbt",
  "kind": "exam",
  "title": {
    "tr": "TUS · TTBT",
    "en": "TUS · TTBT"
  },
  "subtitle": {
    "tr": "2023-I–2026-I · 72 soruluk yayımlanmış soru derlemesi",
    "en": "2023-I–2026-I · 72-question published-item collection"
  },
  "category": {
    "tr": "Mesleki sınav",
    "en": "Professional exam"
  },
  "description": {
    "tr": "30 dil modelinin 2023-I–2026-I dönemlerinde yayımlanan 72 TTBT sorusundaki net, maliyet ve token kullanımını karşılaştırır.",
    "en": "Compares 30 language models on net score, cost, and token use across 72 TTBT items published from 2023-I through 2026-I."
  },
  "status": "published",
  "examDate": "2026-03-15",
  "runId": "2026-08-15T15-44-17Z",
  "runCompletedAt": "2026-08-15T15:56:53Z",
  "generatedAt": "2026-08-15T15:56:53Z",
  "itemCount": 72,
  "responseCount": 2160,
  "totalQuestions": 72,
  "score": {
    "label": {
      "tr": "TTBT Neti",
      "en": "TTBT Net"
    },
    "metric": "net",
    "min": -18.0,
    "max": 72.0,
    "formula": {
      "tr": "doğru − (yanlış + geçersiz)/4",
      "en": "correct − (wrong + invalid)/4"
    }
  },
  "scopeNote": {
    "tr": "72 soruluk bu derleme, 2023-I–2026-I dönemlerinde yayımlanan soruları kapsar; tek bir resmî 120 soruluk TUS oturumu veya resmî TUS puanı değildir.",
    "en": "This 72-question collection covers published items from 2023-I–2026-I; it is not one official 120-question TUS sitting or an official TUS score."
  },
  "netRule": {
    "tr": "net = doğru − (yanlış + geçersiz)/4",
    "en": "net = correct − (wrong + invalid)/4"
  },
  "provenance": {
    "settingsSource": "run",
    "scoring": {
      "penaltyDivisor": 4.0,
      "base": 0.0,
      "coefficients": {
        "TTBT": 1.0
      }
    },
    "invalidPolicy": "wrong"
  },
  "sections": [
    {
      "code": "TTBT",
      "name": {
        "tr": "Temel Tıp Bilimleri Testi",
        "en": "Basic Medical Sciences Test"
      },
      "questionCount": 72
    }
  ],
  "models": [
    {
      "modelId": "google/gemini-3.7-flash",
      "baseModelId": "google/gemini-3.7-flash",
      "modelSlug": "gemini-3-7-flash",
      "developerId": "google",
      "name": "Gemini 3.7 Flash",
      "score": 70.75,
      "net": 70.75,
      "sections": {
        "TTBT": {
          "correct": 71,
          "wrong": 1,
          "invalid": 0,
          "net": 70.75
        }
      },
      "cost": {
        "usd": 0.0586,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 41180,
        "known": true,
        "approximate": false,
        "prompt": 12380,
        "completion": 28800,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 1207.3379,
        "netPer1kTokens": 1.7181
      },
      "rank": 1
    },
    {
      "modelId": "minimax/minimax-m3",
      "baseModelId": "minimax/minimax-m3",
      "modelSlug": "minimax-m3",
      "developerId": "minimax",
      "name": "MiniMax M3",
      "score": 70.75,
      "net": 70.75,
      "sections": {
        "TTBT": {
          "correct": 71,
          "wrong": 1,
          "invalid": 0,
          "net": 70.75
        }
      },
      "cost": {
        "usd": 0.0654,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 77139,
        "known": true,
        "approximate": false,
        "prompt": 27674,
        "completion": 49465,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": null,
        "requestedRoutingPolicy": {
          "quantizations": [
            "fp8"
          ],
          "allow_fallbacks": true
        },
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 1081.8043,
        "netPer1kTokens": 0.9172
      },
      "rank": 2
    },
    {
      "modelId": "google/gemini-3.6-flash",
      "baseModelId": "google/gemini-3.6-flash",
      "modelSlug": "gemini-3-6-flash",
      "developerId": "google",
      "name": "Gemini 3.6 Flash",
      "score": 70.75,
      "net": 70.75,
      "sections": {
        "TTBT": {
          "correct": 71,
          "wrong": 1,
          "invalid": 0,
          "net": 70.75
        }
      },
      "cost": {
        "usd": 0.1728,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 55990,
        "known": true,
        "approximate": false,
        "prompt": 12380,
        "completion": 43610,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 409.4329,
        "netPer1kTokens": 1.2636
      },
      "rank": 3
    },
    {
      "modelId": "z-ai/glm-5.2",
      "baseModelId": "z-ai/glm-5.2",
      "modelSlug": "glm-5-2",
      "developerId": "zai",
      "name": "GLM 5.2",
      "score": 70.75,
      "net": 70.75,
      "sections": {
        "TTBT": {
          "correct": 71,
          "wrong": 1,
          "invalid": 0,
          "net": 70.75
        }
      },
      "cost": {
        "usd": 0.1947,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 55218,
        "known": true,
        "approximate": false,
        "prompt": 15322,
        "completion": 39896,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "high",
        "requestedRoutingPolicy": {
          "quantizations": [
            "fp8"
          ],
          "ignore": [
            "akashml",
            "ambient"
          ],
          "allow_fallbacks": true
        },
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 363.3796,
        "netPer1kTokens": 1.2813
      },
      "rank": 4
    },
    {
      "modelId": "anthropic/claude-opus-5",
      "baseModelId": "anthropic/claude-opus-5",
      "modelSlug": "claude-opus-5",
      "developerId": "anthropic",
      "name": "Claude Opus 5",
      "score": 70.75,
      "net": 70.75,
      "sections": {
        "TTBT": {
          "correct": 71,
          "wrong": 1,
          "invalid": 0,
          "net": 70.75
        }
      },
      "cost": {
        "usd": 0.2614,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 28167,
        "known": true,
        "approximate": false,
        "prompt": 22140,
        "completion": 6027,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 270.658,
        "netPer1kTokens": 2.5118
      },
      "rank": 5
    },
    {
      "modelId": "meta/muse-spark-1.2",
      "baseModelId": "meta/muse-spark-1.2",
      "modelSlug": "muse-spark-1-2",
      "developerId": "meta",
      "name": "Muse Spark 1.2",
      "score": 70.75,
      "net": 70.75,
      "sections": {
        "TTBT": {
          "correct": 71,
          "wrong": 1,
          "invalid": 0,
          "net": 70.75
        }
      },
      "cost": {
        "usd": 0.3094,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 82936,
        "known": true,
        "approximate": false,
        "prompt": 13131,
        "completion": 69805,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 228.6684,
        "netPer1kTokens": 0.8531
      },
      "rank": 6
    },
    {
      "modelId": "openai/gpt-5.6-sol",
      "baseModelId": "openai/gpt-5.6-sol",
      "modelSlug": "gpt-5-6-sol",
      "developerId": "openai",
      "name": "GPT-5.6 Sol",
      "score": 70.75,
      "net": 70.75,
      "sections": {
        "TTBT": {
          "correct": 71,
          "wrong": 1,
          "invalid": 0,
          "net": 70.75
        }
      },
      "cost": {
        "usd": 0.3179,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 22022,
        "known": true,
        "approximate": false,
        "prompt": 13712,
        "completion": 8310,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 222.5543,
        "netPer1kTokens": 3.2127
      },
      "rank": 7
    },
    {
      "modelId": "anthropic/claude-opus-4.8",
      "baseModelId": "anthropic/claude-opus-4.8",
      "modelSlug": "claude-opus-4-8",
      "developerId": "anthropic",
      "name": "Claude Opus 4.8",
      "score": 70.75,
      "net": 70.75,
      "sections": {
        "TTBT": {
          "correct": 71,
          "wrong": 1,
          "invalid": 0,
          "net": 70.75
        }
      },
      "cost": {
        "usd": 0.4004,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 33730,
        "known": true,
        "approximate": false,
        "prompt": 22140,
        "completion": 11590,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 176.6983,
        "netPer1kTokens": 2.0975
      },
      "rank": 8
    },
    {
      "modelId": "google/gemini-3.1-pro-preview",
      "baseModelId": "google/gemini-3.1-pro-preview",
      "modelSlug": "gemini-3-1-pro",
      "developerId": "google",
      "name": "Gemini 3.1 Pro Preview",
      "score": 70.75,
      "net": 70.75,
      "sections": {
        "TTBT": {
          "correct": 71,
          "wrong": 1,
          "invalid": 0,
          "net": 70.75
        }
      },
      "cost": {
        "usd": 0.4699,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 49474,
        "known": true,
        "approximate": false,
        "prompt": 12380,
        "completion": 37094,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 150.5639,
        "netPer1kTokens": 1.43
      },
      "rank": 9
    },
    {
      "modelId": "deepseek/deepseek-v4-flash-0731",
      "baseModelId": "deepseek/deepseek-v4-flash-0731",
      "modelSlug": "deepseek-v4-flash-0731",
      "developerId": "deepseek",
      "name": "DeepSeek V4 Flash 0731",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.007,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 49965,
        "known": true,
        "approximate": false,
        "prompt": 16811,
        "completion": 33154,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "high",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 9928.5714,
        "netPer1kTokens": 1.391
      },
      "rank": 10
    },
    {
      "modelId": "openai/gpt-5.6-luna",
      "baseModelId": "openai/gpt-5.6-luna",
      "modelSlug": "gpt-5-6-luna",
      "developerId": "openai",
      "name": "GPT-5.6 Luna",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.007,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 23046,
        "known": true,
        "approximate": false,
        "prompt": 13712,
        "completion": 9334,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 9928.5714,
        "netPer1kTokens": 3.0157
      },
      "rank": 11
    },
    {
      "modelId": "google/gemma-4-31b-it",
      "baseModelId": "google/gemma-4-31b-it",
      "modelSlug": "gemma-4-31b",
      "developerId": "google",
      "name": "Gemma 4 31B",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.0331,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 92667,
        "known": true,
        "approximate": false,
        "prompt": 13532,
        "completion": 79135,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": null,
        "requestedRoutingPolicy": {
          "quantizations": [
            "fp8"
          ],
          "allow_fallbacks": true
        },
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 2099.6979,
        "netPer1kTokens": 0.75
      },
      "rank": 12
    },
    {
      "modelId": "thinkingmachines/inkling-small",
      "baseModelId": "thinkingmachines/inkling-small",
      "modelSlug": "inkling-small",
      "developerId": "thinking-machines",
      "name": "Inkling Small",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.0332,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 36552,
        "known": true,
        "approximate": false,
        "prompt": 14216,
        "completion": 22336,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 2093.3735,
        "netPer1kTokens": 1.9014
      },
      "rank": 13
    },
    {
      "modelId": "openai/gpt-5.6-terra",
      "baseModelId": "openai/gpt-5.6-terra",
      "modelSlug": "gpt-5-6-terra",
      "developerId": "openai",
      "name": "GPT-5.6 Terra",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.0546,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 20529,
        "known": true,
        "approximate": false,
        "prompt": 13712,
        "completion": 6817,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 1272.8938,
        "netPer1kTokens": 3.3855
      },
      "rank": 14
    },
    {
      "modelId": "anthropic/claude-sonnet-5",
      "baseModelId": "anthropic/claude-sonnet-5",
      "modelSlug": "claude-sonnet-5",
      "developerId": "anthropic",
      "name": "Claude Sonnet 5",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.0834,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 26051,
        "known": true,
        "approximate": false,
        "prompt": 22140,
        "completion": 3911,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 833.3333,
        "netPer1kTokens": 2.6678
      },
      "rank": 15
    },
    {
      "modelId": "thinkingmachines/inkling",
      "baseModelId": "thinkingmachines/inkling",
      "modelSlug": "thinky-inkling",
      "developerId": "thinking-machines",
      "name": "Inkling",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.0848,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 31687,
        "known": true,
        "approximate": false,
        "prompt": 14288,
        "completion": 17399,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 819.5755,
        "netPer1kTokens": 2.1933
      },
      "rank": 16
    },
    {
      "modelId": "deepseek/deepseek-v4-pro-0813",
      "baseModelId": "deepseek/deepseek-v4-pro-0813",
      "modelSlug": "deepseek-v4-pro-0813",
      "developerId": "deepseek",
      "name": "DeepSeek V4 Pro 0813",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.1775,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 59682,
        "known": true,
        "approximate": false,
        "prompt": 22020,
        "completion": 37662,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "high",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 391.5493,
        "netPer1kTokens": 1.1645
      },
      "rank": 17
    },
    {
      "modelId": "x-ai/grok-4.5",
      "baseModelId": "x-ai/grok-4.5",
      "modelSlug": "grok-4-5",
      "developerId": "xai",
      "name": "Grok 4.5",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.2726,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 67237,
        "known": true,
        "approximate": false,
        "prompt": 27594,
        "completion": 39643,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 254.9523,
        "netPer1kTokens": 1.0337
      },
      "rank": 18
    },
    {
      "modelId": "x-ai/grok-4.6",
      "baseModelId": "x-ai/grok-4.6",
      "modelSlug": "grok-4-6",
      "developerId": "xai",
      "name": "Grok 4.6",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.3014,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 71473,
        "known": true,
        "approximate": false,
        "prompt": 27594,
        "completion": 43879,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 230.5906,
        "netPer1kTokens": 0.9724
      },
      "rank": 19
    },
    {
      "modelId": "qwen/qwen3.8-max",
      "baseModelId": "qwen/qwen3.8-max",
      "modelSlug": "qwen3-8-max",
      "developerId": "alibaba-qwen",
      "name": "Qwen3.8 Max",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.3773,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 72374,
        "known": true,
        "approximate": false,
        "prompt": 14235,
        "completion": 58139,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 184.2036,
        "netPer1kTokens": 0.9603
      },
      "rank": 20
    },
    {
      "modelId": "moonshotai/kimi-k3",
      "baseModelId": "moonshotai/kimi-k3",
      "modelSlug": "kimi-k3",
      "developerId": "moonshot",
      "name": "Kimi K3",
      "score": 69.5,
      "net": 69.5,
      "sections": {
        "TTBT": {
          "correct": 70,
          "wrong": 2,
          "invalid": 0,
          "net": 69.5
        }
      },
      "cost": {
        "usd": 0.4677,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 51524,
        "known": true,
        "approximate": false,
        "prompt": 23518,
        "completion": 28006,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "high",
        "requestedRoutingPolicy": {
          "ignore": [
            "nebius",
            "moonshotai"
          ],
          "allow_fallbacks": true
        },
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 148.5995,
        "netPer1kTokens": 1.3489
      },
      "rank": 21
    },
    {
      "modelId": "meta/muse-glimmer-30b",
      "baseModelId": "meta/muse-glimmer-30b",
      "modelSlug": "muse-glimmer-30b",
      "developerId": "meta",
      "name": "Muse Glimmer 30B",
      "score": 68.25,
      "net": 68.25,
      "sections": {
        "TTBT": {
          "correct": 69,
          "wrong": 3,
          "invalid": 0,
          "net": 68.25
        }
      },
      "cost": {
        "usd": 0.0502,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 45990,
        "known": true,
        "approximate": false,
        "prompt": 14283,
        "completion": 31707,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": {
          "ignore": [
            "deepinfra"
          ],
          "allow_fallbacks": true
        },
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 1359.5618,
        "netPer1kTokens": 1.484
      },
      "rank": 22
    },
    {
      "modelId": "tencent/hy3",
      "baseModelId": "tencent/hy3",
      "modelSlug": "tencent-hy3",
      "developerId": "tencent",
      "name": "Hy3",
      "score": 68.25,
      "net": 68.25,
      "sections": {
        "TTBT": {
          "correct": 69,
          "wrong": 2,
          "invalid": 1,
          "net": 68.25
        }
      },
      "cost": {
        "usd": 0.0567,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 114775,
        "known": true,
        "approximate": false,
        "prompt": 16017,
        "completion": 98758,
        "orchestration": 0
      },
      "errors": {
        "invalid": 1,
        "apiFailures": 0
      },
      "note": {
        "tr": "1 yanıt 16.384 token tavanında kesildi; zorunlu CEVAP satırı üretilemedi. Yanıt geçersiz sayıldı ve net hesabında yanlış olarak cezalandırıldı.",
        "en": "1 response was cut off at the 16,384-token ceiling before producing the required CEVAP line. It was counted as invalid and penalized as wrong in the net calculation."
      },
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "high",
        "requestedRoutingPolicy": {
          "quantizations": [
            "fp8"
          ],
          "allow_fallbacks": true
        },
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 1203.7037,
        "netPer1kTokens": 0.5946
      },
      "rank": 23
    },
    {
      "modelId": "nvidia/nemotron-3-ultra-550b-a55b",
      "baseModelId": "nvidia/nemotron-3-ultra-550b-a55b",
      "modelSlug": "nemotron-3-ultra",
      "developerId": "nvidia",
      "name": "Nemotron 3 Ultra",
      "score": 68.25,
      "net": 68.25,
      "sections": {
        "TTBT": {
          "correct": 69,
          "wrong": 2,
          "invalid": 1,
          "net": 68.25
        }
      },
      "cost": {
        "usd": 0.2397,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 78978,
        "known": true,
        "approximate": false,
        "prompt": 14819,
        "completion": 64159,
        "orchestration": 0
      },
      "errors": {
        "invalid": 1,
        "apiFailures": 0
      },
      "note": {
        "tr": "1 yanıt 16.384 token tavanında kesildi; zorunlu CEVAP satırı üretilemedi. Yanıt geçersiz sayıldı ve net hesabında yanlış olarak cezalandırıldı.",
        "en": "1 response was cut off at the 16,384-token ceiling before producing the required CEVAP line. It was counted as invalid and penalized as wrong in the net calculation."
      },
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": {
          "ignore": [
            "baseten"
          ],
          "allow_fallbacks": true
        },
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 284.7309,
        "netPer1kTokens": 0.8642
      },
      "rank": 24
    },
    {
      "modelId": "xiaomi/mimo-v2.5-pro",
      "baseModelId": "xiaomi/mimo-v2.5-pro",
      "modelSlug": "mimo-v2-5-pro",
      "developerId": "xiaomi",
      "name": "MiMo-V2.5-Pro",
      "score": 67.0,
      "net": 67.0,
      "sections": {
        "TTBT": {
          "correct": 68,
          "wrong": 3,
          "invalid": 1,
          "net": 67.0
        }
      },
      "cost": {
        "usd": 0.1762,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 70466,
        "known": true,
        "approximate": false,
        "prompt": 15004,
        "completion": 55462,
        "orchestration": 0
      },
      "errors": {
        "invalid": 1,
        "apiFailures": 0
      },
      "note": {
        "tr": "Yanıt tamamlandı ancak zorunlu CEVAP: X satırı yoktu. Yanıt geçersiz sayıldı ve net hesabında yanlış olarak cezalandırıldı.",
        "en": "The response completed without the required CEVAP: X line. It was counted as invalid and penalized as wrong in the net calculation."
      },
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": null,
        "requestedRoutingPolicy": {
          "quantizations": [
            "fp8"
          ],
          "allow_fallbacks": true
        },
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 380.2497,
        "netPer1kTokens": 0.9508
      },
      "rank": 25
    },
    {
      "modelId": "openai/gpt-oss-120b",
      "baseModelId": "openai/gpt-oss-120b",
      "modelSlug": "gpt-oss-120b",
      "developerId": "openai",
      "name": "gpt-oss-120b",
      "score": 65.75,
      "net": 65.75,
      "sections": {
        "TTBT": {
          "correct": 67,
          "wrong": 5,
          "invalid": 0,
          "net": 65.75
        }
      },
      "cost": {
        "usd": 0.0215,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 38443,
        "known": true,
        "approximate": false,
        "prompt": 18320,
        "completion": 20123,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 3058.1395,
        "netPer1kTokens": 1.7103
      },
      "rank": 26
    },
    {
      "modelId": "qwen/qwen3.8-27b",
      "baseModelId": "qwen/qwen3.8-27b",
      "modelSlug": "qwen3-8-27b",
      "developerId": "alibaba-qwen",
      "name": "Qwen3.8 27B",
      "score": 65.75,
      "net": 65.75,
      "sections": {
        "TTBT": {
          "correct": 67,
          "wrong": 5,
          "invalid": 0,
          "net": 65.75
        }
      },
      "cost": {
        "usd": 0.208,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 77238,
        "known": true,
        "approximate": false,
        "prompt": 14235,
        "completion": 63003,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": "medium",
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 316.1058,
        "netPer1kTokens": 0.8513
      },
      "rank": 27
    },
    {
      "modelId": "anthropic/claude-haiku-4.5",
      "baseModelId": "anthropic/claude-haiku-4.5",
      "modelSlug": "claude-haiku-4-5",
      "developerId": "anthropic",
      "name": "Claude Haiku 4.5",
      "score": 65.75,
      "net": 65.75,
      "sections": {
        "TTBT": {
          "correct": 67,
          "wrong": 5,
          "invalid": 0,
          "net": 65.75
        }
      },
      "cost": {
        "usd": 0.5003,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 116334,
        "known": true,
        "approximate": false,
        "prompt": 20342,
        "completion": 95992,
        "orchestration": 0
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 131.4211,
        "netPer1kTokens": 0.5652
      },
      "rank": 28
    },
    {
      "modelId": "google/gemma-4-26b-a4b-it",
      "baseModelId": "google/gemma-4-26b-a4b-it",
      "modelSlug": "gemma-4-26b-a4b",
      "developerId": "google",
      "name": "Gemma 4 26B A4B",
      "score": 62.0,
      "net": 62.0,
      "sections": {
        "TTBT": {
          "correct": 64,
          "wrong": 1,
          "invalid": 7,
          "net": 62.0
        }
      },
      "cost": {
        "usd": 0.1208,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 211431,
        "known": true,
        "approximate": false,
        "prompt": 13460,
        "completion": 197971,
        "orchestration": 0
      },
      "errors": {
        "invalid": 7,
        "apiFailures": 0
      },
      "note": {
        "tr": "7 yanıt 16.384 token tavanında kesildi; zorunlu CEVAP satırı üretilemedi. Bu yanıtlar geçersiz sayıldı ve net hesabında yanlış olarak cezalandırıldı.",
        "en": "7 responses were cut off at the 16,384-token ceiling before producing the required CEVAP line. They were counted as invalid and penalized as wrong in the net calculation."
      },
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 513.245,
        "netPer1kTokens": 0.2932
      },
      "rank": 29
    },
    {
      "modelId": "nvidia/nemotron-3.5-lightning",
      "baseModelId": "nvidia/nemotron-3.5-lightning",
      "modelSlug": "nemotron-3-5-lightning",
      "developerId": "nvidia",
      "name": "Nemotron 3.5 Lightning",
      "score": 60.75,
      "net": 60.75,
      "sections": {
        "TTBT": {
          "correct": 63,
          "wrong": 8,
          "invalid": 1,
          "net": 60.75
        }
      },
      "cost": {
        "usd": 0.0295,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 153652,
        "known": true,
        "approximate": false,
        "prompt": 14811,
        "completion": 138841,
        "orchestration": 0
      },
      "errors": {
        "invalid": 1,
        "apiFailures": 0
      },
      "note": {
        "tr": "1 yanıt 16.384 token tavanında kesildi; zorunlu CEVAP satırı üretilemedi. Yanıt geçersiz sayıldı ve net hesabında yanlış olarak cezalandırıldı.",
        "en": "1 response was cut off at the 16,384-token ceiling before producing the required CEVAP line. It was counted as invalid and penalized as wrong in the net calculation."
      },
      "settings": {
        "temperature": 0.0,
        "maxTokens": 16384,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "efficiency": {
        "netPerUsd": 2059.322,
        "netPer1kTokens": 0.3954
      },
      "rank": 30
    }
  ],
  "validationStudies": [
    {
      "kind": "knowledge_cutoff_control",
      "id": "tus-ttbt-cutoff-control",
      "title": {
        "tr": "Bilgi kesimi analizi",
        "en": "Knowledge cutoff analysis"
      },
      "method": {
        "tr": "Kontaminasyon açıklamasını sınayan kontrol. Skorlar 70 soruluk analiz kümesine; maliyet ve çıktı tokenı 72 soruluk tam kontrol koşusuna aittir.",
        "en": "Control testing the contamination explanation. Scores apply to the 70-question analysis set; cost and completion tokens apply to the full 72-question control run."
      },
      "sourceRuns": [
        {
          "runId": "2026-08-15T15-44-17Z",
          "role": "primary_reference"
        },
        {
          "runId": "2026-08-16T11-54-45Z",
          "role": "legacy_control"
        }
      ],
      "itemCount": 72,
      "scoredItemCount": 70,
      "responseCount": 504,
      "scoredResponseCount": 490,
      "excludedItems": [
        {
          "questionId": "TTBT-15",
          "reason": {
            "tr": "Hatalı cevap anahtarı",
            "en": "Incorrect answer key"
          }
        },
        {
          "questionId": "TTBT-45",
          "reason": {
            "tr": "Tek doğru cevabı tartışmalı",
            "en": "Ambiguous single-best answer"
          }
        }
      ],
      "slices": {
        "earlier": {
          "label": {
            "tr": "İlk 60",
            "en": "First 60"
          },
          "itemCount": 60
        },
        "latest": {
          "label": {
            "tr": "2026-I",
            "en": "2026-I"
          },
          "itemCount": 10,
          "examDate": "2026-03-15"
        }
      },
      "controlRunCostUsd": 4.823276,
      "controlRows": [
        {
          "modelId": "openai/o1",
          "name": "o1",
          "cutoff": {
            "value": "2023-10",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "overall": {
            "correct": 70,
            "wrong": 0,
            "invalid": 0,
            "net": 70.0,
            "netRatio": 1.0,
            "total": 70
          },
          "earlierSlice": {
            "correct": 60,
            "total": 60
          },
          "latestSlice": {
            "correct": 10,
            "total": 10
          },
          "completionTokens": 29878,
          "costUsd": 1.99836
        },
        {
          "modelId": "openai/gpt-4o",
          "name": "GPT-4o",
          "cutoff": {
            "value": "2023-10",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "overall": {
            "correct": 69,
            "wrong": 1,
            "invalid": 0,
            "net": 68.75,
            "netRatio": 0.9821428571428571,
            "total": 70
          },
          "earlierSlice": {
            "correct": 60,
            "total": 60
          },
          "latestSlice": {
            "correct": 9,
            "total": 10
          },
          "completionTokens": 17549,
          "costUsd": 0.20995
        },
        {
          "modelId": "anthropic/claude-sonnet-4",
          "name": "Claude Sonnet 4",
          "cutoff": {
            "value": "2025-01",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "overall": {
            "correct": 69,
            "wrong": 1,
            "invalid": 0,
            "net": 68.75,
            "netRatio": 0.9821428571428571,
            "total": 70
          },
          "earlierSlice": {
            "correct": 60,
            "total": 60
          },
          "latestSlice": {
            "correct": 9,
            "total": 10
          },
          "completionTokens": 78362,
          "costUsd": 1.236456
        },
        {
          "modelId": "deepseek/deepseek-r1",
          "name": "DeepSeek R1",
          "cutoff": {
            "value": "2024-07",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "overall": {
            "correct": 68,
            "wrong": 2,
            "invalid": 0,
            "net": 67.5,
            "netRatio": 0.9642857142857143,
            "total": 70
          },
          "earlierSlice": {
            "correct": 59,
            "total": 60
          },
          "latestSlice": {
            "correct": 9,
            "total": 10
          },
          "completionTokens": 102050,
          "costUsd": 0.26654
        },
        {
          "modelId": "openai/gpt-4",
          "name": "GPT-4",
          "cutoff": {
            "value": "2021-09",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "overall": {
            "correct": 61,
            "wrong": 9,
            "invalid": 0,
            "net": 58.75,
            "netRatio": 0.8392857142857143,
            "total": 70
          },
          "earlierSlice": {
            "correct": 53,
            "total": 60
          },
          "latestSlice": {
            "correct": 8,
            "total": 10
          },
          "completionTokens": 9611,
          "costUsd": 1.08114
        },
        {
          "modelId": "anthropic/claude-3-haiku",
          "name": "Claude 3 Haiku",
          "cutoff": {
            "value": "2023-08",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "overall": {
            "correct": 54,
            "wrong": 16,
            "invalid": 0,
            "net": 50.0,
            "netRatio": 0.7142857142857143,
            "total": 70
          },
          "earlierSlice": {
            "correct": 47,
            "total": 60
          },
          "latestSlice": {
            "correct": 7,
            "total": 10
          },
          "completionTokens": 3878,
          "costUsd": 0.009393
        },
        {
          "modelId": "openai/gpt-3.5-turbo",
          "name": "GPT-3.5 Turbo",
          "cutoff": {
            "value": "2021-09",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "overall": {
            "correct": 47,
            "wrong": 22,
            "invalid": 1,
            "net": 41.25,
            "netRatio": 0.5892857142857143,
            "total": 70
          },
          "earlierSlice": {
            "correct": 40,
            "total": 60
          },
          "latestSlice": {
            "correct": 7,
            "total": 10
          },
          "completionTokens": 8686,
          "costUsd": 0.021437
        }
      ],
      "comparisonRows": [
        {
          "modelId": "openai/gpt-3.5-turbo",
          "name": "GPT-3.5 Turbo",
          "cutoff": {
            "value": "2021-09",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 40,
            "total": 60
          },
          "latestSlice": {
            "correct": 7,
            "total": 10
          },
          "percentagePointDifference": 3.3333333333333326
        },
        {
          "modelId": "openai/gpt-4",
          "name": "GPT-4",
          "cutoff": {
            "value": "2021-09",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 53,
            "total": 60
          },
          "latestSlice": {
            "correct": 8,
            "total": 10
          },
          "percentagePointDifference": -8.333333333333325
        },
        {
          "modelId": "anthropic/claude-3-haiku",
          "name": "Claude 3 Haiku",
          "cutoff": {
            "value": "2023-08",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 47,
            "total": 60
          },
          "latestSlice": {
            "correct": 7,
            "total": 10
          },
          "percentagePointDifference": -8.333333333333337
        },
        {
          "modelId": "openai/o1",
          "name": "o1",
          "cutoff": {
            "value": "2023-10",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 60,
            "total": 60
          },
          "latestSlice": {
            "correct": 10,
            "total": 10
          },
          "percentagePointDifference": 0.0
        },
        {
          "modelId": "openai/gpt-4o",
          "name": "GPT-4o",
          "cutoff": {
            "value": "2023-10",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 60,
            "total": 60
          },
          "latestSlice": {
            "correct": 9,
            "total": 10
          },
          "percentagePointDifference": -9.999999999999998
        },
        {
          "modelId": "openai/gpt-oss-120b",
          "name": "gpt-oss-120b",
          "cutoff": {
            "value": "2024-06",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 57,
            "total": 60
          },
          "latestSlice": {
            "correct": 10,
            "total": 10
          },
          "percentagePointDifference": 5.000000000000004
        },
        {
          "modelId": "deepseek/deepseek-r1",
          "name": "DeepSeek R1",
          "cutoff": {
            "value": "2024-07",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 59,
            "total": 60
          },
          "latestSlice": {
            "correct": 9,
            "total": 10
          },
          "percentagePointDifference": -8.333333333333325
        },
        {
          "modelId": "anthropic/claude-sonnet-4",
          "name": "Claude Sonnet 4",
          "cutoff": {
            "value": "2025-01",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 60,
            "total": 60
          },
          "latestSlice": {
            "correct": 9,
            "total": 10
          },
          "percentagePointDifference": -9.999999999999998
        },
        {
          "modelId": "openai/gpt-5.6-luna",
          "name": "GPT-5.6 Luna",
          "cutoff": {
            "value": "2026-02",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 60,
            "total": 60
          },
          "latestSlice": {
            "correct": 10,
            "total": 10
          },
          "percentagePointDifference": 0.0
        },
        {
          "modelId": "openai/gpt-5.6-terra",
          "name": "GPT-5.6 Terra",
          "cutoff": {
            "value": "2026-02",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 60,
            "total": 60
          },
          "latestSlice": {
            "correct": 10,
            "total": 10
          },
          "percentagePointDifference": 0.0
        },
        {
          "modelId": "openai/gpt-5.6-sol",
          "name": "GPT-5.6 Sol",
          "cutoff": {
            "value": "2026-02",
            "precision": "month",
            "status": "study_assumption",
            "sourceUrl": null
          },
          "earlierSlice": {
            "correct": 60,
            "total": 60
          },
          "latestSlice": {
            "correct": 10,
            "total": 10
          },
          "percentagePointDifference": 0.0
        }
      ],
      "caveats": [
        {
          "tr": "Bilgi kesimi, nihai eğitim verisi bileşimini tek başına kanıtlamaz.",
          "en": "A knowledge cutoff alone does not establish final training-data composition."
        },
        {
          "tr": "2026-I dilimi yalnız 10 sorudur; tek hata 10 yüzde puan değiştirir.",
          "en": "The 2026-I slice has only 10 questions; one error changes the result by 10 percentage points."
        },
        {
          "tr": "İlk 60 geçerli sorunun soru-bazlı dönem etiketleri henüz ayrıntılı değildir.",
          "en": "Question-level period labels are not yet detailed for the first 60 valid questions."
        },
        {
          "tr": "Ana ve kontrol koşuları farklı tarihlerde çalıştırılmıştır.",
          "en": "The primary and control runs were executed on different dates."
        },
        {
          "tr": "İki soru yalnız bu analizde, açık gerekçelerle kapsam dışıdır.",
          "en": "Two questions are out of scope only for this analysis, with explicit reasons."
        },
        {
          "tr": "Sağlayıcı çözüm alanları unknown/null olduğundan pinned provider veya checkpoint iddiası yapılmaz.",
          "en": "Provider resolution is unknown/null, so no pinned-provider or checkpoint claim is made."
        },
        {
          "tr": "Skorlar 70 soruluk analiz kümesine; maliyet ve çıktı tokenı 72 soruluk tam kontrol koşusuna aittir.",
          "en": "Scores apply to the 70-question analysis set; cost and completion tokens apply to the full 72-question control run."
        }
      ]
    }
  ],
  "questionStats": [
    {
      "id": "TTBT-1",
      "number": 1,
      "section": "TTBT",
      "cancelled": false,
      "correct": 29,
      "responses": 30
    },
    {
      "id": "TTBT-2",
      "number": 2,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-3",
      "number": 3,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-4",
      "number": 4,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-5",
      "number": 5,
      "section": "TTBT",
      "cancelled": false,
      "correct": 28,
      "responses": 30
    },
    {
      "id": "TTBT-6",
      "number": 6,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-7",
      "number": 7,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-8",
      "number": 8,
      "section": "TTBT",
      "cancelled": false,
      "correct": 29,
      "responses": 30
    },
    {
      "id": "TTBT-9",
      "number": 9,
      "section": "TTBT",
      "cancelled": false,
      "correct": 29,
      "responses": 30
    },
    {
      "id": "TTBT-10",
      "number": 10,
      "section": "TTBT",
      "cancelled": false,
      "correct": 28,
      "responses": 30
    },
    {
      "id": "TTBT-11",
      "number": 11,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-12",
      "number": 12,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-13",
      "number": 13,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-14",
      "number": 14,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-15",
      "number": 15,
      "section": "TTBT",
      "cancelled": false,
      "correct": 0,
      "responses": 30
    },
    {
      "id": "TTBT-16",
      "number": 16,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-17",
      "number": 17,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-18",
      "number": 18,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-19",
      "number": 19,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-20",
      "number": 20,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-21",
      "number": 21,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-22",
      "number": 22,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-23",
      "number": 23,
      "section": "TTBT",
      "cancelled": false,
      "correct": 28,
      "responses": 30
    },
    {
      "id": "TTBT-24",
      "number": 24,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-25",
      "number": 25,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-26",
      "number": 26,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-27",
      "number": 27,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-28",
      "number": 28,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-29",
      "number": 29,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-30",
      "number": 30,
      "section": "TTBT",
      "cancelled": false,
      "correct": 29,
      "responses": 30
    },
    {
      "id": "TTBT-31",
      "number": 31,
      "section": "TTBT",
      "cancelled": false,
      "correct": 28,
      "responses": 30
    },
    {
      "id": "TTBT-32",
      "number": 32,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-33",
      "number": 33,
      "section": "TTBT",
      "cancelled": false,
      "correct": 26,
      "responses": 30
    },
    {
      "id": "TTBT-34",
      "number": 34,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-35",
      "number": 35,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-36",
      "number": 36,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-37",
      "number": 37,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-38",
      "number": 38,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-39",
      "number": 39,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-40",
      "number": 40,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-41",
      "number": 41,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-42",
      "number": 42,
      "section": "TTBT",
      "cancelled": false,
      "correct": 28,
      "responses": 30
    },
    {
      "id": "TTBT-43",
      "number": 43,
      "section": "TTBT",
      "cancelled": false,
      "correct": 27,
      "responses": 30
    },
    {
      "id": "TTBT-44",
      "number": 44,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-45",
      "number": 45,
      "section": "TTBT",
      "cancelled": false,
      "correct": 10,
      "responses": 30
    },
    {
      "id": "TTBT-46",
      "number": 46,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-47",
      "number": 47,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-48",
      "number": 48,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-49",
      "number": 49,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-50",
      "number": 50,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-51",
      "number": 51,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-52",
      "number": 52,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-53",
      "number": 53,
      "section": "TTBT",
      "cancelled": false,
      "correct": 27,
      "responses": 30
    },
    {
      "id": "TTBT-54",
      "number": 54,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-55",
      "number": 55,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-56",
      "number": 56,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-57",
      "number": 57,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-58",
      "number": 58,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-59",
      "number": 59,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-60",
      "number": 60,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-61",
      "number": 61,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-62",
      "number": 62,
      "section": "TTBT",
      "cancelled": false,
      "correct": 28,
      "responses": 30
    },
    {
      "id": "TTBT-63",
      "number": 63,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-64",
      "number": 64,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-65",
      "number": 65,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-66",
      "number": 66,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-67",
      "number": 67,
      "section": "TTBT",
      "cancelled": false,
      "correct": 28,
      "responses": 30
    },
    {
      "id": "TTBT-68",
      "number": 68,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-69",
      "number": 69,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-70",
      "number": 70,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-71",
      "number": 71,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    },
    {
      "id": "TTBT-72",
      "number": 72,
      "section": "TTBT",
      "cancelled": false,
      "correct": 30,
      "responses": 30
    }
  ]
}
