{
  "schemaVersion": 3,
  "kind": "inference",
  "id": "glm53-flash-nvfp4-nextn-20260905",
  "title": {
    "tr": "GLM-5.3 Flash çıkarım throughput",
    "en": "GLM-5.3 Flash inference throughput"
  },
  "subtitle": {
    "tr": "NVFP4 · NextN/MTP",
    "en": "NVFP4 · NextN/MTP"
  },
  "category": {
    "tr": "Sistem ölçümü",
    "en": "Systems measurement"
  },
  "description": {
    "tr": "Üç giriş bağlamında eşzamanlı çıktı üretim ölçümü.",
    "en": "Concurrent output-generation measurement across three input contexts."
  },
  "status": "published",
  "itemCount": 15,
  "responseCount": 186,
  "runId": "20260905T170046Z-nextn-bf16-restart-context-sweep",
  "runCompletedAt": "2026-09-05T17:00:46Z",
  "generatedAt": "2026-09-07T00:00:00Z",
  "sources": [
    {
      "name": "throughput.csv",
      "sha256": "fe679f3f1e7618904c8f42e731a7afc9d4da28785b127f5fa5bea9e11a70317a"
    },
    {
      "name": "verified-summary.json",
      "sha256": "dc66cd7be5b49e9653840e67102233078c68b74f876962869327fd014e637658"
    }
  ],
  "method": {
    "tr": "Toplu çıktı throughput'u, tamamlanan çıktı tokenlarının batch duvar saatine oranıdır ve giriş işlemeyi içerir. Her hücre iki batch'in aritmetik ortalamasıdır; her ölçüm isteği akıl yürütme dahil 1.024 tamamlanma tokenı üretir ve üç 64-token warmup isteği hariçtir.",
    "en": "Aggregate output throughput is completed output tokens divided by batch wall time and includes input processing. Each cell is the arithmetic mean of two batches; every measured request produces 1,024 completion tokens including reasoning, and the three 64-token warmup requests are excluded."
  },
  "caveats": [
    {
      "tr": "128K servis sınırıdır; ölçülen girişler yalnız 1K, 4K ve 16K'dır.",
      "en": "128K is the serving limit; measured inputs are only 1K, 4K, and 16K."
    },
    {
      "tr": "Bu, 5 Eylül 2026 tarihli tarihsel bir ölçümdür; servisin bugün canlı olduğunu göstermez.",
      "en": "This is a historical measurement from September 5, 2026; it does not establish that the service is live today."
    },
    {
      "tr": "Throughput sonucu kaliteyi, doğruluğu veya bağlam kapasitesini ölçmez.",
      "en": "The throughput result does not measure quality, accuracy, or context capacity."
    },
    {
      "tr": "Spekülatif olmayan bir kontrol koşusu bulunmadığından NextN/MTP için nedensel hızlanma iddiası desteklenmez.",
      "en": "Without a non-speculative control run, no causal speedup claim for NextN/MTP is supported."
    }
  ],
  "profile": {
    "modelId": "glm-5.3-flash",
    "modelName": "GLM-5.3 Flash",
    "modelRevision": "NVFP4",
    "tensorParallel": 4,
    "maxConcurrency": 16,
    "contextLimitTokens": 131072,
    "weightDtype": "NVFP4",
    "kvDtype": "BF16",
    "speculativeMethod": "native NextN/MTP through SGLang EAGLE",
    "reasoningEffort": "high",
    "temperature": 0,
    "outputTokens": 1024,
    "ignoreEos": true
  },
  "cells": [
    {
      "contextTokens": 1024,
      "concurrency": 1,
      "repetitions": 2,
      "aggregateTokensPerSecond": 46.92754002611305,
      "meanRequestTokensPerSecond": 46.92792736203499,
      "meanTtftSeconds": 0.5718681104990537,
      "runTokensPerSecond": [
        46.48148674453251,
        47.373593307693596
      ]
    },
    {
      "contextTokens": 1024,
      "concurrency": 2,
      "repetitions": 2,
      "aggregateTokensPerSecond": 66.01695830674277,
      "meanRequestTokensPerSecond": 33.4214131256884,
      "meanTtftSeconds": 3.7495509640011733,
      "runTokensPerSecond": [
        61.130180890766,
        70.90373572271952
      ]
    },
    {
      "contextTokens": 1024,
      "concurrency": 4,
      "repetitions": 2,
      "aggregateTokensPerSecond": 124.10339723994677,
      "meanRequestTokensPerSecond": 31.661094906543205,
      "meanTtftSeconds": 1.9721906392496749,
      "runTokensPerSecond": [
        125.4370136938364,
        122.76978078605714
      ]
    },
    {
      "contextTokens": 1024,
      "concurrency": 8,
      "repetitions": 2,
      "aggregateTokensPerSecond": 149.23716609031905,
      "meanRequestTokensPerSecond": 19.18623498035486,
      "meanTtftSeconds": 3.2919611612501285,
      "runTokensPerSecond": [
        147.9930201601125,
        150.4813120205256
      ]
    },
    {
      "contextTokens": 1024,
      "concurrency": 16,
      "repetitions": 2,
      "aggregateTokensPerSecond": 197.07587252205633,
      "meanRequestTokensPerSecond": 12.81377834602376,
      "meanTtftSeconds": 6.0177714189376275,
      "runTokensPerSecond": [
        196.47380828050186,
        197.67793676361083
      ]
    },
    {
      "contextTokens": 4096,
      "concurrency": 1,
      "repetitions": 2,
      "aggregateTokensPerSecond": 44.907612692526925,
      "meanRequestTokensPerSecond": 44.90795674334873,
      "meanTtftSeconds": 1.6833485040006053,
      "runTokensPerSecond": [
        46.040053209711374,
        43.775172175342476
      ]
    },
    {
      "contextTokens": 4096,
      "concurrency": 2,
      "repetitions": 2,
      "aggregateTokensPerSecond": 67.69536766437191,
      "meanRequestTokensPerSecond": 33.96403917980952,
      "meanTtftSeconds": 3.223521661999257,
      "runTokensPerSecond": [
        69.52826386535463,
        65.8624714633892
      ]
    },
    {
      "contextTokens": 4096,
      "concurrency": 4,
      "repetitions": 2,
      "aggregateTokensPerSecond": 93.24848172296007,
      "meanRequestTokensPerSecond": 23.65841965943336,
      "meanTtftSeconds": 6.002756428750217,
      "runTokensPerSecond": [
        92.21379023549932,
        94.28317321042083
      ]
    },
    {
      "contextTokens": 4096,
      "concurrency": 8,
      "repetitions": 2,
      "aggregateTokensPerSecond": 121.98383830985904,
      "meanRequestTokensPerSecond": 15.63257358657528,
      "meanTtftSeconds": 10.068398736123527,
      "runTokensPerSecond": [
        125.1685681506482,
        118.79910846906986
      ]
    },
    {
      "contextTokens": 4096,
      "concurrency": 16,
      "repetitions": 2,
      "aggregateTokensPerSecond": 158.72274476717496,
      "meanRequestTokensPerSecond": 10.177706854605848,
      "meanTtftSeconds": 16.825040088344622,
      "runTokensPerSecond": [
        158.1732215799646,
        159.2722679543853
      ]
    },
    {
      "contextTokens": 16384,
      "concurrency": 1,
      "repetitions": 2,
      "aggregateTokensPerSecond": 37.54048928274629,
      "meanRequestTokensPerSecond": 37.54080441972414,
      "meanTtftSeconds": 6.3824825284973485,
      "runTokensPerSecond": [
        37.67872361697234,
        37.40225494852024
      ]
    },
    {
      "contextTokens": 16384,
      "concurrency": 2,
      "repetitions": 2,
      "aggregateTokensPerSecond": 49.77536028877646,
      "meanRequestTokensPerSecond": 25.463886318362,
      "meanTtftSeconds": 11.057915566252632,
      "runTokensPerSecond": [
        48.09628682057236,
        51.45443375698056
      ]
    },
    {
      "contextTokens": 16384,
      "concurrency": 4,
      "repetitions": 2,
      "aggregateTokensPerSecond": 66.4177990236277,
      "meanRequestTokensPerSecond": 16.854475114643446,
      "meanTtftSeconds": 18.14204138187415,
      "runTokensPerSecond": [
        66.72734968055028,
        66.10824836670511
      ]
    },
    {
      "contextTokens": 16384,
      "concurrency": 8,
      "repetitions": 2,
      "aggregateTokensPerSecond": 80.47181332538699,
      "meanRequestTokensPerSecond": 10.220245269244746,
      "meanTtftSeconds": 31.082480875000783,
      "runTokensPerSecond": [
        80.98960311888379,
        79.95402353189019
      ]
    },
    {
      "contextTokens": 16384,
      "concurrency": 16,
      "repetitions": 2,
      "aggregateTokensPerSecond": 93.1577273671081,
      "meanRequestTokensPerSecond": 5.909783415116328,
      "meanTtftSeconds": 56.44599598812465,
      "runTokensPerSecond": [
        92.3743290750869,
        93.94112565912928
      ]
    }
  ],
  "totals": {
    "batches": 30,
    "requests": 186,
    "completionTokens": 190464,
    "excludedWarmupRequests": 3
  }
}
