{
  "schemaVersion": 3,
  "kind": "exam",
  "id": "tus-2026",
  "title": {
    "tr": "2026 TUS · Açık sorular",
    "en": "2026 TUS · open questions"
  },
  "subtitle": {
    "tr": "1. ve 2. dönem derlemesi",
    "en": "first and second session compilation"
  },
  "category": {
    "tr": "Ulusal sınav",
    "en": "National exam"
  },
  "description": {
    "tr": "40 soruluk model karşılaştırması.",
    "en": "A 40-question model comparison."
  },
  "status": "published",
  "itemCount": 40,
  "responseCount": 1560,
  "examDate": "2026-08-29",
  "runId": "2026-08-29T02-19-52Z",
  "runCompletedAt": "2026-08-29T10:39:53Z",
  "generatedAt": "2026-09-07T00:00:00Z",
  "totalQuestions": 40,
  "score": {
    "label": {
      "tr": "Net",
      "en": "Net"
    },
    "metric": "net",
    "min": -10,
    "max": 40,
    "formula": {
      "tr": "doğru − (yanlış + geçersiz)/4",
      "en": "correct − (wrong + invalid)/4"
    }
  },
  "netRule": {
    "tr": "Geçersiz yanıtlar yanlış sayılır.",
    "en": "Invalid answers count as wrong."
  },
  "sections": [
    {
      "code": "TBTT",
      "name": {
        "tr": "Temel Bilimler",
        "en": "Basic sciences"
      },
      "questionCount": 20
    },
    {
      "code": "KTBT",
      "name": {
        "tr": "Klinik Bilimler",
        "en": "Clinical sciences"
      },
      "questionCount": 20
    }
  ],
  "models": [
    {
      "modelId": "deepseek/deepseek-v4-flash-0731",
      "baseModelId": "deepseek/deepseek-v4-flash-0731",
      "modelSlug": "deepseek-v4-flash-0731",
      "developerId": "deepseek",
      "rank": 1,
      "name": "DeepSeek V4 Flash",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.008769,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 22864,
        "known": true,
        "approximate": false,
        "prompt": 12918,
        "completion": 9946,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 4561.5235,
        "netPer1kTokens": 1.7495
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 9.57,
        "meanQuestionMs": 1808.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "openai/gpt-5.6-luna",
      "baseModelId": "openai/gpt-5.6-luna",
      "modelSlug": "gpt-5-6-luna",
      "developerId": "openai",
      "rank": 2,
      "name": "GPT-5.6 Luna",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.010074,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 15175,
        "known": true,
        "approximate": false,
        "prompt": 8136,
        "completion": 7039,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 3970.6174,
        "netPer1kTokens": 2.6359
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 13.04,
        "meanQuestionMs": 2534.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "qwen/qwen3.8-flash",
      "baseModelId": "qwen/qwen3.8-flash",
      "modelSlug": "qwen-qwen3-8-flash",
      "developerId": "alibaba-qwen",
      "rank": 3,
      "name": "Qwen3.8 Flash",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.012392,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 33211,
        "known": true,
        "approximate": false,
        "prompt": 9948,
        "completion": 23263,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 3227.889,
        "netPer1kTokens": 1.2044
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 57.13,
        "meanQuestionMs": 9497.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "minimax/minimax-m3",
      "baseModelId": "minimax/minimax-m3",
      "modelSlug": "minimax-m3",
      "developerId": "minimax",
      "rank": 4,
      "name": "MiniMax M3",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.034191,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 48636,
        "known": true,
        "approximate": false,
        "prompt": 15924,
        "completion": 32712,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 1169.8985,
        "netPer1kTokens": 0.8224
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 25.47,
        "meanQuestionMs": 5341.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "anthropic/claude-sonnet-5",
      "baseModelId": "anthropic/claude-sonnet-5",
      "modelSlug": "claude-sonnet-5",
      "developerId": "anthropic",
      "rank": 5,
      "name": "Claude Sonnet 5",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.043454,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 14803,
        "known": true,
        "approximate": false,
        "prompt": 13072,
        "completion": 1731,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 920.5136,
        "netPer1kTokens": 2.7022
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 19.43,
        "meanQuestionMs": 2906.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "thinkingmachines/inkling",
      "baseModelId": "thinkingmachines/inkling",
      "modelSlug": "thinky-inkling",
      "developerId": "thinking-machines",
      "rank": 6,
      "name": "Inkling",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.051248,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 19022,
        "known": true,
        "approximate": false,
        "prompt": 8456,
        "completion": 10566,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 780.5183,
        "netPer1kTokens": 2.1028
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 16.58,
        "meanQuestionMs": 3115.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "openai/gpt-5.6-sol",
      "baseModelId": "openai/gpt-5.6-sol",
      "modelSlug": "gpt-5-6-sol",
      "developerId": "openai",
      "rank": 7,
      "name": "GPT-5.6 Sol",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.065302,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 13039,
        "known": true,
        "approximate": false,
        "prompt": 8136,
        "completion": 4903,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 612.5387,
        "netPer1kTokens": 3.0677
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 16.23,
        "meanQuestionMs": 2747.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "nvidia/nemotron-3-ultra-550b-a55b",
      "baseModelId": "nvidia/nemotron-3-ultra-550b-a55b",
      "modelSlug": "nemotron-3-ultra",
      "developerId": "nvidia",
      "rank": 8,
      "name": "Nemotron 3 Ultra 550B-A55B",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.067212,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 34647,
        "known": true,
        "approximate": false,
        "prompt": 8856,
        "completion": 25791,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 595.1318,
        "netPer1kTokens": 1.1545
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 33.9,
        "meanQuestionMs": 7300.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "openai/gpt-5.6-terra",
      "baseModelId": "openai/gpt-5.6-terra",
      "modelSlug": "gpt-5-6-terra",
      "developerId": "openai",
      "rank": 9,
      "name": "GPT-5.6 Terra",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.07032,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 12640,
        "known": true,
        "approximate": false,
        "prompt": 8136,
        "completion": 4504,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 568.8282,
        "netPer1kTokens": 3.1646
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 11.33,
        "meanQuestionMs": 2294.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "google/gemini-3.7-flash",
      "baseModelId": "google/gemini-3.7-flash",
      "modelSlug": "gemini-3-7-flash",
      "developerId": "google",
      "rank": 10,
      "name": "Gemini 3.7 Flash",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.071858,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 25047,
        "known": true,
        "approximate": false,
        "prompt": 7356,
        "completion": 17691,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 556.6534,
        "netPer1kTokens": 1.597
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 17.81,
        "meanQuestionMs": 4047.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "anthropic/claude-opus-5",
      "baseModelId": "anthropic/claude-opus-5",
      "modelSlug": "claude-opus-5",
      "developerId": "anthropic",
      "rank": 11,
      "name": "Claude Opus 5",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.11266,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 14964,
        "known": true,
        "approximate": false,
        "prompt": 13072,
        "completion": 1892,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 355.0506,
        "netPer1kTokens": 2.6731
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 16.55,
        "meanQuestionMs": 3673.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "x-ai/grok-4.6",
      "baseModelId": "x-ai/grok-4.6",
      "modelSlug": "grok-4-6",
      "developerId": "xai",
      "rank": 12,
      "name": "Grok 4.6",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.21296,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 47164,
        "known": true,
        "approximate": false,
        "prompt": 15778,
        "completion": 31386,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 187.8287,
        "netPer1kTokens": 0.8481
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 134.84,
        "meanQuestionMs": 17018.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "anthropic/claude-opus-4.8",
      "baseModelId": "anthropic/claude-opus-4.8",
      "modelSlug": "claude-opus-4-8",
      "developerId": "anthropic",
      "rank": 13,
      "name": "Claude Opus 4.8",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.21941,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 19234,
        "known": true,
        "approximate": false,
        "prompt": 13072,
        "completion": 6162,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 182.3071,
        "netPer1kTokens": 2.0797
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 19.37,
        "meanQuestionMs": 4356.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "qwen/qwen3.8-max",
      "baseModelId": "qwen/qwen3.8-max",
      "modelSlug": "qwen3-8-max",
      "developerId": "alibaba-qwen",
      "rank": 14,
      "name": "Qwen3.8 Max",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.22433,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 43007,
        "known": true,
        "approximate": false,
        "prompt": 8428,
        "completion": 34579,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 178.3087,
        "netPer1kTokens": 0.9301
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 73.58,
        "meanQuestionMs": 16640.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "google/gemini-3.1-pro-preview",
      "baseModelId": "google/gemini-3.1-pro-preview",
      "modelSlug": "gemini-3-1-pro",
      "developerId": "google",
      "rank": 15,
      "name": "Gemini 3.1 Pro (preview)",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.30696,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 31710,
        "known": true,
        "approximate": false,
        "prompt": 7356,
        "completion": 24354,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 130.3101,
        "netPer1kTokens": 1.2614
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 26.7,
        "meanQuestionMs": 6201.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "moonshotai/kimi-k3",
      "baseModelId": "moonshotai/kimi-k3",
      "modelSlug": "kimi-k3",
      "developerId": "moonshot",
      "rank": 16,
      "name": "Kimi K3",
      "score": 40.0,
      "net": 40.0,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        }
      },
      "cost": {
        "usd": 0.321771,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 33077,
        "known": true,
        "approximate": false,
        "prompt": 13560,
        "completion": 19517,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 124.312,
        "netPer1kTokens": 1.2093
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 17.94,
        "meanQuestionMs": 3651.0
      },
      "cohortScores": {
        "t": 55.1717,
        "k": 55.4335
      }
    },
    {
      "modelId": "xiaomi/mimo-v2.5-pro",
      "baseModelId": "xiaomi/mimo-v2.5-pro",
      "modelSlug": "mimo-v2-5-pro",
      "developerId": "xiaomi",
      "rank": 17,
      "name": "MiMo v2.5 Pro",
      "score": 38.75,
      "net": 38.75,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 19,
          "wrong": 1,
          "invalid": 0,
          "net": 18.75
        }
      },
      "cost": {
        "usd": 0.061555,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 48375,
        "known": true,
        "approximate": false,
        "prompt": 8880,
        "completion": 39495,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 629.5183,
        "netPer1kTokens": 0.801
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 110.48,
        "meanQuestionMs": 22081.0
      },
      "cohortScores": {
        "t": 54.1036,
        "k": 53.8312
      }
    },
    {
      "modelId": "deepseek/deepseek-v4-pro-0813",
      "baseModelId": "deepseek/deepseek-v4-pro-0813",
      "modelSlug": "deepseek-v4-pro-0813",
      "developerId": "deepseek",
      "rank": 18,
      "name": "DeepSeek V4 Pro",
      "score": 38.75,
      "net": 38.75,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 19,
          "wrong": 0,
          "invalid": 1,
          "net": 18.75
        }
      },
      "cost": {
        "usd": 0.154566,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 47673,
        "known": true,
        "approximate": false,
        "prompt": 12913,
        "completion": 34760,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 250.702,
        "netPer1kTokens": 0.8128
      },
      "errors": {
        "invalid": 1,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 83.67,
        "meanQuestionMs": 6113.0
      },
      "cohortScores": {
        "t": 54.1036,
        "k": 53.8312
      }
    },
    {
      "modelId": "meta/muse-spark-1.2",
      "baseModelId": "meta/muse-spark-1.2",
      "modelSlug": "muse-spark-1-2",
      "developerId": "meta",
      "rank": 19,
      "name": "Muse Spark 1.2",
      "score": 38.75,
      "net": 38.75,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 19,
          "wrong": 0,
          "invalid": 1,
          "net": 18.75
        }
      },
      "cost": {
        "usd": 0.188078,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 49640,
        "known": true,
        "approximate": false,
        "prompt": 7541,
        "completion": 42099,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 206.0315,
        "netPer1kTokens": 0.7806
      },
      "errors": {
        "invalid": 1,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 67.66,
        "meanQuestionMs": 12357.0
      },
      "cohortScores": {
        "t": 54.1036,
        "k": 53.8312
      }
    },
    {
      "modelId": "qwen/qwen3.8-2.4t-a95b",
      "baseModelId": "qwen/qwen3.8-2.4t-a95b",
      "modelSlug": "qwen-qwen3-8-2-4t-a95b",
      "developerId": "alibaba-qwen",
      "rank": 20,
      "name": "Qwen3.8 2.4T-A95B",
      "score": 38.75,
      "net": 38.75,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 19,
          "wrong": 1,
          "invalid": 0,
          "net": 18.75
        }
      },
      "cost": {
        "usd": 0.226712,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 43908,
        "known": true,
        "approximate": false,
        "prompt": 8428,
        "completion": 35480,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 170.9217,
        "netPer1kTokens": 0.8825
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 37.56,
        "meanQuestionMs": 8449.0
      },
      "cohortScores": {
        "t": 54.1036,
        "k": 53.8312
      }
    },
    {
      "modelId": "anthropic/claude-haiku-4.5",
      "baseModelId": "anthropic/claude-haiku-4.5",
      "modelSlug": "claude-haiku-4-5",
      "developerId": "anthropic",
      "rank": 21,
      "name": "Claude Haiku 4.5",
      "score": 38.75,
      "net": 38.75,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 19,
          "wrong": 1,
          "invalid": 0,
          "net": 18.75
        }
      },
      "cost": {
        "usd": 0.302896,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 70048,
        "known": true,
        "approximate": false,
        "prompt": 11836,
        "completion": 58212,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 127.9317,
        "netPer1kTokens": 0.5532
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 56.15,
        "meanQuestionMs": 12039.0
      },
      "cohortScores": {
        "t": 54.1036,
        "k": 53.8312
      }
    },
    {
      "modelId": "z-ai/glm-5.3-flash",
      "baseModelId": "z-ai/glm-5.3-flash",
      "modelSlug": "z-ai-glm-5-3-flash",
      "developerId": "zai",
      "rank": 22,
      "name": "GLM-5.3 Flash",
      "score": 37.5,
      "net": 37.5,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 18,
          "wrong": 2,
          "invalid": 0,
          "net": 17.5
        }
      },
      "cost": {
        "usd": 0.007582,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 23816,
        "known": true,
        "approximate": false,
        "prompt": 9042,
        "completion": 14774,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 4945.9246,
        "netPer1kTokens": 1.5746
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 55.77,
        "meanQuestionMs": 10449.0
      },
      "cohortScores": {
        "t": 53.0354,
        "k": 52.229
      }
    },
    {
      "modelId": "xiaomi/mimo-v2.5",
      "baseModelId": "xiaomi/mimo-v2.5",
      "modelSlug": "xiaomi-mimo-v2-5",
      "developerId": "xiaomi",
      "rank": 23,
      "name": "MiMo v2.5",
      "score": 37.5,
      "net": 37.5,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 18,
          "wrong": 2,
          "invalid": 0,
          "net": 17.5
        }
      },
      "cost": {
        "usd": 0.013462,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 41886,
        "known": true,
        "approximate": false,
        "prompt": 8880,
        "completion": 33006,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 2785.6188,
        "netPer1kTokens": 0.8953
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 171.11,
        "meanQuestionMs": 23797.0
      },
      "cohortScores": {
        "t": 53.0354,
        "k": 52.229
      }
    },
    {
      "modelId": "thinkingmachines/inkling-small",
      "baseModelId": "thinkingmachines/inkling-small",
      "modelSlug": "inkling-small",
      "developerId": "thinking-machines",
      "rank": 24,
      "name": "Inkling Small",
      "score": 37.5,
      "net": 37.5,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 18,
          "wrong": 2,
          "invalid": 0,
          "net": 17.5
        }
      },
      "cost": {
        "usd": 0.018485,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 20642,
        "known": true,
        "approximate": false,
        "prompt": 8450,
        "completion": 12192,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 2028.6719,
        "netPer1kTokens": 1.8167
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 216.41,
        "meanQuestionMs": 20601.0
      },
      "cohortScores": {
        "t": 53.0354,
        "k": 52.229
      }
    },
    {
      "modelId": "google/gemma-4-31b-it",
      "baseModelId": "google/gemma-4-31b-it",
      "modelSlug": "gemma-4-31b",
      "developerId": "google",
      "rank": 25,
      "name": "Gemma 4 31B",
      "score": 37.5,
      "net": 37.5,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 18,
          "wrong": 2,
          "invalid": 0,
          "net": 17.5
        }
      },
      "cost": {
        "usd": 0.032745,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 53582,
        "known": true,
        "approximate": false,
        "prompt": 8068,
        "completion": 45514,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 1145.213,
        "netPer1kTokens": 0.6999
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 138.1,
        "meanQuestionMs": 21097.0
      },
      "cohortScores": {
        "t": 53.0354,
        "k": 52.229
      }
    },
    {
      "modelId": "z-ai/glm-5.3",
      "baseModelId": "z-ai/glm-5.3",
      "modelSlug": "z-ai-glm-5-3",
      "developerId": "zai",
      "rank": 26,
      "name": "GLM-5.3",
      "score": 37.5,
      "net": 37.5,
      "sections": {
        "TBTT": {
          "correct": 20,
          "wrong": 0,
          "invalid": 0,
          "net": 20.0
        },
        "KTBT": {
          "correct": 18,
          "wrong": 2,
          "invalid": 0,
          "net": 17.5
        }
      },
      "cost": {
        "usd": 0.099415,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 29489,
        "known": true,
        "approximate": false,
        "prompt": 9042,
        "completion": 20447,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 377.2067,
        "netPer1kTokens": 1.2717
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 24.59,
        "meanQuestionMs": 4075.0
      },
      "cohortScores": {
        "t": 53.0354,
        "k": 52.229
      }
    },
    {
      "modelId": "qwen/qwen3.8-27b",
      "baseModelId": "qwen/qwen3.8-27b",
      "modelSlug": "qwen3-8-27b",
      "developerId": "alibaba-qwen",
      "rank": 27,
      "name": "Qwen3.8 27B",
      "score": 36.25,
      "net": 36.25,
      "sections": {
        "TBTT": {
          "correct": 19,
          "wrong": 1,
          "invalid": 0,
          "net": 18.75
        },
        "KTBT": {
          "correct": 18,
          "wrong": 2,
          "invalid": 0,
          "net": 17.5
        }
      },
      "cost": {
        "usd": 0.129226,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 47626,
        "known": true,
        "approximate": false,
        "prompt": 8428,
        "completion": 39198,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 280.5163,
        "netPer1kTokens": 0.7611
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 66.06,
        "meanQuestionMs": 13803.0
      },
      "cohortScores": {
        "t": 51.2523,
        "k": 51.0403
      }
    },
    {
      "modelId": "mistralai/mistral-medium-3-5",
      "baseModelId": "mistralai/mistral-medium-3-5",
      "modelSlug": "mistralai-mistral-medium-3-5",
      "developerId": "mistral",
      "rank": 28,
      "name": "Mistral Medium 3.5",
      "score": 36.25,
      "net": 36.25,
      "sections": {
        "TBTT": {
          "correct": 19,
          "wrong": 1,
          "invalid": 0,
          "net": 18.75
        },
        "KTBT": {
          "correct": 18,
          "wrong": 2,
          "invalid": 0,
          "net": 17.5
        }
      },
      "cost": {
        "usd": 0.321931,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 50009,
        "known": true,
        "approximate": false,
        "prompt": 8856,
        "completion": 41153,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 112.6018,
        "netPer1kTokens": 0.7249
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 66.02,
        "meanQuestionMs": 9825.0
      },
      "cohortScores": {
        "t": 51.2523,
        "k": 51.0403
      }
    },
    {
      "modelId": "inception/mercury-2",
      "baseModelId": "inception/mercury-2",
      "modelSlug": "inception-mercury-2",
      "developerId": "inception",
      "rank": 29,
      "name": "Mercury 2",
      "score": 35.0,
      "net": 35.0,
      "sections": {
        "TBTT": {
          "correct": 19,
          "wrong": 1,
          "invalid": 0,
          "net": 18.75
        },
        "KTBT": {
          "correct": 17,
          "wrong": 3,
          "invalid": 0,
          "net": 16.25
        }
      },
      "cost": {
        "usd": 0.011883,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 21737,
        "known": true,
        "approximate": false,
        "prompt": 7836,
        "completion": 13901,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 2945.3842,
        "netPer1kTokens": 1.6102
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 6.0,
        "meanQuestionMs": 1239.0
      },
      "cohortScores": {
        "t": 50.1842,
        "k": 49.4381
      }
    },
    {
      "modelId": "meta/muse-glimmer-30b",
      "baseModelId": "meta/muse-glimmer-30b",
      "modelSlug": "muse-glimmer-30b",
      "developerId": "meta",
      "rank": 30,
      "name": "Muse Glimmer 30B",
      "score": 35.0,
      "net": 35.0,
      "sections": {
        "TBTT": {
          "correct": 19,
          "wrong": 1,
          "invalid": 0,
          "net": 18.75
        },
        "KTBT": {
          "correct": 17,
          "wrong": 3,
          "invalid": 0,
          "net": 16.25
        }
      },
      "cost": {
        "usd": 0.031774,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 28512,
        "known": true,
        "approximate": false,
        "prompt": 8413,
        "completion": 20099,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 1101.5296,
        "netPer1kTokens": 1.2276
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 23.95,
        "meanQuestionMs": 5365.0
      },
      "cohortScores": {
        "t": 50.1842,
        "k": 49.4381
      }
    },
    {
      "modelId": "meta-llama/llama-3.3-70b-instruct",
      "baseModelId": "meta-llama/llama-3.3-70b-instruct",
      "modelSlug": "meta-llama-llama-3-3-70b-instruct",
      "developerId": "meta",
      "rank": 31,
      "name": "Llama 3.3 70B",
      "score": 32.5,
      "net": 32.5,
      "sections": {
        "TBTT": {
          "correct": 17,
          "wrong": 3,
          "invalid": 0,
          "net": 16.25
        },
        "KTBT": {
          "correct": 17,
          "wrong": 3,
          "invalid": 0,
          "net": 16.25
        }
      },
      "cost": {
        "usd": 0.006369,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 8970,
        "known": true,
        "approximate": false,
        "prompt": 8279,
        "completion": 691,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 5102.8419,
        "netPer1kTokens": 3.6232
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 6.68,
        "meanQuestionMs": 579.0
      },
      "cohortScores": {
        "t": 46.618,
        "k": 47.0606
      }
    },
    {
      "modelId": "nvidia/nemotron-3-nano-30b-a3b",
      "baseModelId": "nvidia/nemotron-3-nano-30b-a3b",
      "modelSlug": "nvidia-nemotron-3-nano-30b-a3b",
      "developerId": "nvidia",
      "rank": 32,
      "name": "Nemotron 3 Nano 30B-A3B",
      "score": 32.5,
      "net": 32.5,
      "sections": {
        "TBTT": {
          "correct": 18,
          "wrong": 2,
          "invalid": 0,
          "net": 17.5
        },
        "KTBT": {
          "correct": 16,
          "wrong": 4,
          "invalid": 0,
          "net": 15.0
        }
      },
      "cost": {
        "usd": 0.008653,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 44916,
        "known": true,
        "approximate": false,
        "prompt": 8816,
        "completion": 36100,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 3755.9228,
        "netPer1kTokens": 0.7236
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 29.69,
        "meanQuestionMs": 4091.0
      },
      "cohortScores": {
        "t": 47.333,
        "k": 46.6472
      }
    },
    {
      "modelId": "google/gemma-4-26b-a4b-it",
      "baseModelId": "google/gemma-4-26b-a4b-it",
      "modelSlug": "gemma-4-26b-a4b",
      "developerId": "google",
      "rank": 33,
      "name": "Gemma 4 26B-A4B",
      "score": 31.25,
      "net": 31.25,
      "sections": {
        "TBTT": {
          "correct": 16,
          "wrong": 0,
          "invalid": 4,
          "net": 15.0
        },
        "KTBT": {
          "correct": 17,
          "wrong": 0,
          "invalid": 3,
          "net": 16.25
        }
      },
      "cost": {
        "usd": 0.054713,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 167301,
        "known": true,
        "approximate": false,
        "prompt": 8036,
        "completion": 159265,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 571.1622,
        "netPer1kTokens": 0.1868
      },
      "errors": {
        "invalid": 7,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 793.39,
        "meanQuestionMs": 156812.0
      },
      "cohortScores": {
        "t": 44.8349,
        "k": 45.8719
      }
    },
    {
      "modelId": "nvidia/nemotron-3.5-lightning",
      "baseModelId": "nvidia/nemotron-3.5-lightning",
      "modelSlug": "nemotron-3-5-lightning",
      "developerId": "nvidia",
      "rank": 34,
      "name": "Nemotron 3.5 Lightning",
      "score": 30.0,
      "net": 30.0,
      "sections": {
        "TBTT": {
          "correct": 15,
          "wrong": 5,
          "invalid": 0,
          "net": 13.75
        },
        "KTBT": {
          "correct": 17,
          "wrong": 3,
          "invalid": 0,
          "net": 16.25
        }
      },
      "cost": {
        "usd": 0.018929,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 81006,
        "known": true,
        "approximate": false,
        "prompt": 8816,
        "completion": 72190,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 1584.8698,
        "netPer1kTokens": 0.3703
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 49.43,
        "meanQuestionMs": 7061.0
      },
      "cohortScores": {
        "t": 43.0518,
        "k": 44.6832
      }
    },
    {
      "modelId": "tencent/hy4-preview",
      "baseModelId": "tencent/hy4-preview",
      "modelSlug": "tencent-hy4-preview",
      "developerId": "tencent",
      "rank": 35,
      "name": "Hunyuan 4 (preview)",
      "score": 27.5,
      "net": 27.5,
      "sections": {
        "TBTT": {
          "correct": 13,
          "wrong": 0,
          "invalid": 7,
          "net": 11.25
        },
        "KTBT": {
          "correct": 17,
          "wrong": 0,
          "invalid": 3,
          "net": 16.25
        }
      },
      "cost": {
        "usd": 0.237176,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 100082,
        "known": true,
        "approximate": false,
        "prompt": 7815,
        "completion": 92267,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 115.9477,
        "netPer1kTokens": 0.2748
      },
      "errors": {
        "invalid": 10,
        "apiFailures": 10
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 1010.47,
        "meanQuestionMs": 165417.0
      },
      "cohortScores": {
        "t": 39.4856,
        "k": 42.3057
      }
    },
    {
      "modelId": "tencent/hy-mt2-30b-a3b",
      "baseModelId": "tencent/hy-mt2-30b-a3b",
      "modelSlug": "tencent-hy-mt2-30b-a3b",
      "developerId": "tencent",
      "rank": 36,
      "name": "Hunyuan MT2 30B-A3B",
      "score": 26.25,
      "net": 26.25,
      "sections": {
        "TBTT": {
          "correct": 15,
          "wrong": 5,
          "invalid": 0,
          "net": 13.75
        },
        "KTBT": {
          "correct": 14,
          "wrong": 6,
          "invalid": 0,
          "net": 12.5
        }
      },
      "cost": {
        "usd": 0.003352,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 18448,
        "known": true,
        "approximate": false,
        "prompt": 9458,
        "completion": 8990,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 7831.1456,
        "netPer1kTokens": 1.4229
      },
      "errors": {
        "invalid": 0,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 10.72,
        "meanQuestionMs": 2477.0
      },
      "cohortScores": {
        "t": 39.8474,
        "k": 39.8765
      }
    },
    {
      "modelId": "mistralai/ministral-3b-2512",
      "baseModelId": "mistralai/ministral-3b-2512",
      "modelSlug": "mistralai-ministral-3b-2512",
      "developerId": "mistral",
      "rank": 37,
      "name": "Ministral 3B",
      "score": 23.75,
      "net": 23.75,
      "sections": {
        "TBTT": {
          "correct": 16,
          "wrong": 4,
          "invalid": 0,
          "net": 15.0
        },
        "KTBT": {
          "correct": 11,
          "wrong": 8,
          "invalid": 1,
          "net": 8.75
        }
      },
      "cost": {
        "usd": 0.005106,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 55839,
        "known": true,
        "approximate": false,
        "prompt": 8376,
        "completion": 47463,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 4651.3905,
        "netPer1kTokens": 0.4253
      },
      "errors": {
        "invalid": 1,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 163.15,
        "meanQuestionMs": 10422.0
      },
      "cohortScores": {
        "t": 38.4261,
        "k": 36.2586
      }
    },
    {
      "modelId": "meta-llama/llama-3.1-8b-instruct",
      "baseModelId": "meta-llama/llama-3.1-8b-instruct",
      "modelSlug": "meta-llama-llama-3-1-8b-instruct",
      "developerId": "meta",
      "rank": 38,
      "name": "Llama 3.1 8B",
      "score": 13.75,
      "net": 13.75,
      "sections": {
        "TBTT": {
          "correct": 10,
          "wrong": 9,
          "invalid": 1,
          "net": 7.5
        },
        "KTBT": {
          "correct": 9,
          "wrong": 9,
          "invalid": 2,
          "net": 6.25
        }
      },
      "cost": {
        "usd": 0.001873,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 8514,
        "known": true,
        "approximate": false,
        "prompt": 8279,
        "completion": 235,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": 7341.1639,
        "netPer1kTokens": 1.615
      },
      "errors": {
        "invalid": 3,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 1.56,
        "meanQuestionMs": 333.0
      },
      "cohortScores": {
        "t": 25.5912,
        "k": 25.9218
      }
    },
    {
      "modelId": "meta-llama/llama-3.2-1b-instruct",
      "baseModelId": "meta-llama/llama-3.2-1b-instruct",
      "modelSlug": "meta-llama-llama-3-2-1b-instruct",
      "developerId": "meta",
      "rank": 39,
      "name": "Llama 3.2 1B",
      "score": -6.25,
      "net": -6.25,
      "sections": {
        "TBTT": {
          "correct": 3,
          "wrong": 10,
          "invalid": 7,
          "net": -1.25
        },
        "KTBT": {
          "correct": 0,
          "wrong": 8,
          "invalid": 12,
          "net": -5.0
        }
      },
      "cost": {
        "usd": 0.000279,
        "known": true,
        "approximate": false
      },
      "tokens": {
        "billed": 8556,
        "known": true,
        "approximate": false,
        "prompt": 8279,
        "completion": 277,
        "orchestration": 0
      },
      "efficiency": {
        "netPerUsd": -22401.4337,
        "netPer1kTokens": -0.7305
      },
      "errors": {
        "invalid": 19,
        "apiFailures": 0
      },
      "note": null,
      "settings": {
        "temperature": null,
        "maxTokens": null,
        "reasoningEffort": null,
        "requestedRoutingPolicy": null,
        "endpointBaseUrl": null,
        "resolvedInferenceProviderId": null,
        "resolutionSource": "unknown",
        "settingsSource": "run"
      },
      "timing": {
        "wallClockSec": 2.03,
        "meanQuestionMs": 409.0
      },
      "cohortScores": {
        "t": 3.4963,
        "k": 3.1808
      }
    }
  ],
  "provenance": {
    "settingsSource": "run",
    "scoring": {
      "cohortSize": 39,
      "standardDeviation": "population",
      "sectionMeans": {
        "TBTT": 18.044871794871796,
        "KTBT": 17.21153846153846
      },
      "sectionPopulationStandardDeviations": {
        "TBTT": 4.2061736505326595,
        "KTBT": 4.681023632553109
      },
      "weights": {
        "t": {
          "TBTT": 0.6,
          "KTBT": 0.4
        },
        "k": {
          "TBTT": 0.4,
          "KTBT": 0.6
        }
      },
      "sourceSummarySha256": "2e90fd0b79d12cb4b1716485d001e7ed17d45dca4323d504a61bec375b78cce3"
    },
    "invalidPolicy": "invalid answers are penalized as wrong"
  }
}
