{
  "campaign": {
    "schemaVersion": "horizon.site-campaign.v1",
    "runId": "residency-executors-39-20260911-01",
    "sources": [
      {
        "path": "artifacts/residency-executors-39-20260911-01/contract.json",
        "sha256": "bc0ea65fc12b1612fc41c71bc679f19679da08014933d4ab1b9bfadbd38d2455"
      },
      {
        "path": "artifacts/residency-executors-39-20260911-01/consolidation-01/report.md",
        "sha256": "3be6d95d89fb312b4e6af0dbe1e64990212ea4369178f5acf2a56197102694f0"
      },
      {
        "path": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
        "sha256": "e87b76aa9b84eb936ea6bbd0e971506becc86caa165f4e4b3edbb18ddd8c9ba7"
      },
      {
        "path": "artifacts/residency-executors-39-20260911-01/consolidation-01/per-prompt.json",
        "sha256": "5053725f887d822527d7ef782064a9b8ce1f5584a330cc757636f1078f23999e"
      },
      {
        "path": "artifacts/residency-executors-39-20260911-01/consolidation-01/per-worker.json",
        "sha256": "f6a41673245761d4bd0ad85880f7fa33f536ee4f151cd93af7c5aa7e2ff5a600"
      },
      {
        "path": "artifacts/residency-executors-39-20260911-01/consolidation-01/source-hashes.json",
        "sha256": "7873fe23a5e1440b33f55131327cf3a2149d12c77af549565e3a477d630b7a3d"
      },
      {
        "path": "docs/evidence/grouped-moe-real-model-task0-4-20260904-01/preflight.json",
        "sha256": "79d83ccc3721c80f624c074aa507a4f3c65b1b1a52974b8869f05af973798639"
      },
      {
        "path": "artifacts/residency-executors-39-20260911-01/source/packages/backends/offload-qpacked/src/horizon_offload_qpacked/catalog.py",
        "sha256": "5a1b8e5011b9e6a8a4a2f83dabdfb1107b1cd214d592a5bb42ac8ef4c143781d"
      },
      {
        "path": "artifacts/residency-executors-39-20260911-01/01-olmoe_0924_instruct-c64-olmoe-t3/telemetry.jsonl",
        "sha256": "79235e2502fe852cbb6be677083a503dd3beb9f0bda6edbde5e9e6763da08ff9"
      }
    ],
    "workers": 39,
    "measuredResponses": 468,
    "warmups": 39,
    "configurations": 13,
    "responsesPerConfiguration": 36,
    "tokensPerResponse": 128,
    "promptsPerBlock": 12,
    "blocks": 3,
    "controllerSeconds": 6621.088021400001,
    "decodeSeconds": 2226.1549032,
    "ttftSeconds": 2109.6774162,
    "otherSeconds": 2285.2557020000004,
    "peakTemperatureC": 58,
    "minBlockSpreadPercent": 0.2746388397249761,
    "maxBlockSpreadPercent": 4.7839350075771,
    "driverVramTotalBytes": 12820938752,
    "systemRamBytes": 68472115200
  },
  "results": [
    {
      "resultId": "olmoe-0924-instruct-c30-residency39",
      "model": "allenai/OLMoE-1B-7B-0924-Instruct",
      "revision": "7f1c97f440f06ce36705e4f2b843edb5925f4498",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 30,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 23.001858869666933,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "olmoe0924-scalar-partial",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0924_instruct",
      "displayName": "OLMoE 0924",
      "arm": "olmoe-core",
      "executor": "PRODUCT-core scalar",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 6.170728905555555,
      "responseSeconds": 11.692021802777779,
      "peakVramGiB": 4.113231658935547,
      "peakRssGiB": 4.946601867675781,
      "overallTokensPerSecond": 10.94763610255926,
      "h2dGiB": 1108.095474243164,
      "blockRates": [
        23.124099285796564,
        23.143095333208343,
        22.742840088711702
      ],
      "blockSpreadPercent": 1.7400995578860226,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "olmoe-0924-instruct-c40-residency39",
      "model": "allenai/OLMoE-1B-7B-0924-Instruct",
      "revision": "7f1c97f440f06ce36705e4f2b843edb5925f4498",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 40,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 27.691961601492615,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "olmoe0924-scalar-partial",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0924_instruct",
      "displayName": "OLMoE 0924",
      "arm": "olmoe-core",
      "executor": "PRODUCT-core scalar",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 4.486795727777778,
      "responseSeconds": 9.072964155555557,
      "peakVramGiB": 4.618438720703125,
      "peakRssGiB": 4.950050354003906,
      "overallTokensPerSecond": 14.107848086407689,
      "h2dGiB": 576.1717987060547,
      "blockRates": [
        27.755722642658483,
        27.631184933885375,
        27.68925764762307
      ],
      "blockSpreadPercent": 0.44972512444331575,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "olmoe-0924-instruct-c50-residency39",
      "model": "allenai/OLMoE-1B-7B-0924-Instruct",
      "revision": "7f1c97f440f06ce36705e4f2b843edb5925f4498",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 50,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 31.501747677273897,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "olmoe0924-scalar-partial",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0924_instruct",
      "displayName": "OLMoE 0924",
      "arm": "olmoe-core",
      "executor": "PRODUCT-core scalar",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 3.6184148333333335,
      "responseSeconds": 7.649937188888888,
      "peakVramGiB": 5.147003173828125,
      "peakRssGiB": 4.950908660888672,
      "overallTokensPerSecond": 16.732163524938866,
      "h2dGiB": 230.88111877441406,
      "blockRates": [
        31.745368813643385,
        31.57048724976466,
        31.19443376348943
      ],
      "blockSpreadPercent": 1.7489031268935293,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "olmoe-0924-instruct-c60-residency39",
      "model": "allenai/OLMoE-1B-7B-0924-Instruct",
      "revision": "7f1c97f440f06ce36705e4f2b843edb5925f4498",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 60,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 35.01159000794033,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "olmoe0924-scalar-partial",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0924_instruct",
      "displayName": "OLMoE 0924",
      "arm": "olmoe-core",
      "executor": "PRODUCT-core scalar",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 3.0013486361111115,
      "responseSeconds": 6.628718886111112,
      "peakVramGiB": 5.670684814453125,
      "peakRssGiB": 4.956043243408203,
      "overallTokensPerSecond": 19.30991526404797,
      "h2dGiB": 37.615814208984375,
      "blockRates": [
        35.47005949236675,
        34.72178886437509,
        34.85199722045717
      ],
      "blockSpreadPercent": 2.137208358209259,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "olmoe-0924-instruct-c64-residency39",
      "model": "allenai/OLMoE-1B-7B-0924-Instruct",
      "revision": "7f1c97f440f06ce36705e4f2b843edb5925f4498",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 64,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 61.25592960111646,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": null,
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0924_instruct",
      "displayName": "OLMoE 0924",
      "arm": "olmoe-t3",
      "executor": "Grouped T3",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 0.7455810444444444,
      "responseSeconds": 2.818849719444444,
      "peakVramGiB": 5.990997314453125,
      "peakRssGiB": 4.983863830566406,
      "overallTokensPerSecond": 45.40859312827326,
      "h2dGiB": 0,
      "blockRates": [
        59.72004372071484,
        62.650487581119435,
        61.46853484386853
      ],
      "blockSpreadPercent": 4.7839350075771,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "olmoe-0125-instruct-c30-residency39",
      "model": "allenai/OLMoE-1B-7B-0125-Instruct",
      "revision": "b89a7c4bc24fb9e55ce2543c9458ce0ca5c4650e",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 30,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 23.04059913975746,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "olmoe0125-scalar-partial",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0125_instruct",
      "displayName": "OLMoE 0125",
      "arm": "olmoe-core",
      "executor": "PRODUCT-core scalar",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 6.070801397222223,
      "responseSeconds": 11.58281083888889,
      "peakVramGiB": 4.148338317871094,
      "peakRssGiB": 4.968273162841797,
      "overallTokensPerSecond": 11.050858188087162,
      "h2dGiB": 1115.870361328125,
      "blockRates": [
        22.649647054072243,
        22.978154001056666,
        23.510298077489338
      ],
      "blockSpreadPercent": 3.7353673756339414,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "olmoe-0125-instruct-c40-residency39",
      "model": "allenai/OLMoE-1B-7B-0125-Instruct",
      "revision": "b89a7c4bc24fb9e55ce2543c9458ce0ca5c4650e",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 40,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 27.12311586468119,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "olmoe0125-scalar-partial",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0125_instruct",
      "displayName": "OLMoE 0125",
      "arm": "olmoe-core",
      "executor": "PRODUCT-core scalar",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 4.526311008333334,
      "responseSeconds": 9.2086639,
      "peakVramGiB": 4.615997314453125,
      "peakRssGiB": 4.951240539550781,
      "overallTokensPerSecond": 13.899953499225875,
      "h2dGiB": 580.71533203125,
      "blockRates": [
        27.206097164862253,
        26.750809453778892,
        27.4211141686653
      ],
      "blockSpreadPercent": 2.471341118146598,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "olmoe-0125-instruct-c50-residency39",
      "model": "allenai/OLMoE-1B-7B-0125-Instruct",
      "revision": "b89a7c4bc24fb9e55ce2543c9458ce0ca5c4650e",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 50,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 31.22055391154392,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "olmoe0125-scalar-partial",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0125_instruct",
      "displayName": "OLMoE 0125",
      "arm": "olmoe-core",
      "executor": "PRODUCT-core scalar",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 3.5939765083333333,
      "responseSeconds": 7.661809525,
      "peakVramGiB": 5.150886535644531,
      "peakRssGiB": 4.954097747802734,
      "overallTokensPerSecond": 16.706236246456413,
      "h2dGiB": 235.35324096679688,
      "blockRates": [
        31.102916395160722,
        31.309963832451416,
        31.2495085866114
      ],
      "blockSpreadPercent": 0.6631766940372479,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "olmoe-0125-instruct-c60-residency39",
      "model": "allenai/OLMoE-1B-7B-0125-Instruct",
      "revision": "b89a7c4bc24fb9e55ce2543c9458ce0ca5c4650e",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 60,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 35.442592698091474,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "olmoe0125-scalar-partial",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0125_instruct",
      "displayName": "OLMoE 0125",
      "arm": "olmoe-core",
      "executor": "PRODUCT-core scalar",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 2.9682318777777774,
      "responseSeconds": 6.551491180555555,
      "peakVramGiB": 5.670722961425781,
      "peakRssGiB": 4.959484100341797,
      "overallTokensPerSecond": 19.537536794660816,
      "h2dGiB": 40.81146240234375,
      "blockRates": [
        35.452487389435085,
        35.48638382795515,
        35.389044702600664
      ],
      "blockSpreadPercent": 0.2746388397249761,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "olmoe-0125-instruct-c64-residency39",
      "model": "allenai/OLMoE-1B-7B-0125-Instruct",
      "revision": "b89a7c4bc24fb9e55ce2543c9458ce0ca5c4650e",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 64,
      "totalExperts": 64,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 62.07316072245486,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": null,
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "olmoe_0125_instruct",
      "displayName": "OLMoE 0125",
      "arm": "olmoe-t3",
      "executor": "Grouped T3",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 0.6966347027777778,
      "responseSeconds": 2.7426075277777775,
      "peakVramGiB": 6.010833740234375,
      "peakRssGiB": 4.962856292724609,
      "overallTokensPerSecond": 46.670913976420515,
      "h2dGiB": 0,
      "blockRates": [
        62.69202801631041,
        61.39069798315202,
        62.15054844800415
      ],
      "blockSpreadPercent": 2.096445578108982,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "qwen15-moe-chat-c20-residency39",
      "model": "Qwen/Qwen1.5-MoE-A2.7B-Chat",
      "revision": "ec052fda178e241c7c443468d2fa1db6618996be",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 20,
      "totalExperts": 60,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 15.22236659827508,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "qwen-top4",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "qwen15_moe_chat",
      "displayName": "Qwen",
      "arm": "qwen-top4",
      "executor": "top4-fused-v1",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 9.41710652777778,
      "responseSeconds": 17.760093091666665,
      "peakVramGiB": 7.110137939453125,
      "peakRssGiB": 7.967433929443359,
      "overallTokensPerSecond": 7.2071694297627165,
      "h2dGiB": 2702.7495861053467,
      "blockRates": [
        15.08248905711235,
        15.032487454343652,
        15.563286946268972
      ],
      "blockSpreadPercent": 3.486970889174798,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "qwen15-moe-chat-c30-residency39",
      "model": "Qwen/Qwen1.5-MoE-A2.7B-Chat",
      "revision": "ec052fda178e241c7c443468d2fa1db6618996be",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 30,
      "totalExperts": 60,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 17.170567206372947,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "qwen-top4",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "qwen15_moe_chat",
      "displayName": "Qwen",
      "arm": "qwen-top4",
      "executor": "top4-fused-v1",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 7.506468494444444,
      "responseSeconds": 14.902846172222224,
      "peakVramGiB": 8.077144622802734,
      "peakRssGiB": 7.992816925048828,
      "overallTokensPerSecond": 8.58896337792054,
      "h2dGiB": 1852.0655822753906,
      "blockRates": [
        17.45256915909483,
        17.022197127115316,
        17.04372912713543
      ],
      "blockSpreadPercent": 2.5064520397426224,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    },
    {
      "resultId": "qwen15-moe-chat-c39-residency39",
      "model": "Qwen/Qwen1.5-MoE-A2.7B-Chat",
      "revision": "ec052fda178e241c7c443468d2fa1db6618996be",
      "evidenceLevel": "E3",
      "status": "exploratory",
      "residentExperts": 39,
      "totalExperts": 60,
      "generatedTokens": 128,
      "metric": "post_first_token_decode_tokens_per_second",
      "value": 19.945904586527387,
      "unit": "token/s",
      "hardwareSummary": "NVIDIA GeForce RTX 5070 (frozen campaign declaration); shared Windows desktop.",
      "runDate": "2026-09-11",
      "sourceArtifact": "artifacts/residency-executors-39-20260911-01/consolidation-01/summary.json",
      "runContractSummary": "PERF-REAL-v1; 12 prompts repeated in 3 blocks; 36 measured responses of 128 tokens per configuration; greedy selection, EOS suppressed; 64-token warmup excluded. Canonical initial placement, evolving LRU through fixed prompts; concurrency 1.",
      "comparisonGroup": "qwen-top4",
      "limitations": [
        "Exploratory E3 on a shared desktop; no quality evaluation or E4 promotion.",
        "Driver VRAM and process RSS are whole-worker peaks including preparation and warmup; VRAM includes desktop use.",
        "OLMoE c60 to c64 changes executor as well as residency; the transition is not a residency-only effect.",
        "Block spread is an observed range, not a confidence interval. Token parity concerns repetitions within each configuration only.",
        "TTFT and response time use internal request boundaries; isolated prefill is UNKNOWN and OLMoE HTTP observation is UNSUPPORTED."
      ],
      "modelKey": "qwen15_moe_chat",
      "displayName": "Qwen",
      "arm": "qwen-top4",
      "executor": "top4-fused-v1",
      "aggregation": "pooled-post-first-token",
      "measuredResponses": 36,
      "blocks": 3,
      "ttftSeconds": 5.799750786111111,
      "responseSeconds": 12.16697266111111,
      "peakVramGiB": 8.989349365234375,
      "peakRssGiB": 7.973819732666016,
      "overallTokensPerSecond": 10.520283357677142,
      "h2dGiB": 1172.502737045288,
      "blockRates": [
        20.308577805967435,
        19.775902868488615,
        19.762866403445862
      ],
      "blockSpreadPercent": 2.735957149269516,
      "storagePrecision": "INT4 / block 256 / FP16 scales",
      "executionPrecision": "FP16 execution values from low-bit source; original checkpoint dtype is a separate plane."
    }
  ]
}
