[
  {
    "name": "mainline64k-27b-q3xl-mtp-np3-q8q8",
    "model": "C:\\Users\\sjake\\LLM\\models\\unsloth-qwen3.6-27b-mtp-gguf\\Qwen3.6-27B-UD-Q3_K_XL.gguf",
    "mtp": true,
    "n_ok": 10,
    "n_total": 10,
    "avg_wall_s": 12.799441590000423,
    "avg_prompt_tps": 571.9423534012651,
    "avg_predicted_tps": 49.14135616945792,
    "avg_prompt_ms": 2233.2393,
    "avg_predicted_ms": 10461.8102,
    "total_prompt_tokens": 13132,
    "total_completion_tokens": 5120,
    "draft_tokens": 4062,
    "accepted_draft_tokens": 3075,
    "draft_accept_rate": 0.7570162481536189,
    "vram_before": {
      "used_mib": 1151,
      "free_mib": 23163,
      "power_w": 29.6,
      "temp_c": 45
    },
    "vram_ready": {
      "used_mib": 18762,
      "free_mib": 5552,
      "power_w": 81.06,
      "temp_c": 47
    },
    "vram_after": {
      "used_mib": 19108,
      "free_mib": 5206,
      "power_w": 468.12,
      "temp_c": 68
    },
    "net_ready_mib": 17611,
    "props_n_ctx": 21504,
    "total_slots": 3,
    "log": "C:\\Users\\sjake\\LLM\\logs\\mainline64k-27b-q3xl-mtp-np3-q8q8-20260526-194600.server.log",
    "interesting_lines": [
      "0.00.577.252 I common_memory_breakdown_print: | memory breakdown [MiB]  | total    free     self   model   context   compute    unaccounted |",
      "0.00.577.255 I common_memory_breakdown_print: |   - CUDA0 (RTX 3090 Ti) | 24563 = 23285 + (13684 = 13410 +     133 +     140) +      -12406 |",
      "0.00.577.256 I common_memory_breakdown_print: |   - Host                |                    697 =   682 +       0 +      15                |",
      "0.00.992.077 I common_memory_breakdown_print: | memory breakdown [MiB]  | total    free     self   model   context   compute    unaccounted |",
      "0.00.992.080 I common_memory_breakdown_print: |   - CUDA0 (RTX 3090 Ti) | 24563 = 21937 + (17026 = 13410 +    3488 +     127) +      -14399 |",
      "0.00.992.081 I common_memory_breakdown_print: |   - Host                |                    697 =   682 +       0 +      15                |",
      "0.02.454.777 I load_tensors:   CPU_Mapped model buffer size =   682.03 MiB",
      "0.02.454.781 I load_tensors:        CUDA0 model buffer size = 13410.41 MiB",
      "0.07.100.164 I llama_kv_cache:      CUDA0 KV buffer size =  2142.00 MiB",
      "0.07.299.042 I llama_memory_recurrent:      CUDA0 RS buffer size =  1346.62 MiB",
      "0.07.310.725 I sched_reserve:      CUDA0 compute buffer size =   127.24 MiB",
      "0.07.310.729 I sched_reserve:  CUDA_Host compute buffer size =    15.89 MiB",
      "0.07.705.568 I llama_kv_cache:      CUDA0 KV buffer size =   133.88 MiB",
      "0.07.720.020 I sched_reserve:      CUDA0 compute buffer size =   140.61 MiB",
      "0.07.720.024 I sched_reserve:  CUDA_Host compute buffer size =    15.89 MiB",
      "0.09.340.728 I sched_reserve:      CUDA0 compute buffer size =   263.75 MiB",
      "0.09.340.732 I sched_reserve:  CUDA_Host compute buffer size =    15.89 MiB",
      "0.21.241.648 I slot print_timing: id  2 | task 0 | draft acceptance = 0.77114 (  310 accepted /   402 generated)",
      "0.31.992.050 I slot print_timing: id  1 | task 205 | draft acceptance = 0.74634 (  306 accepted /   410 generated)",
      "0.42.797.355 I slot print_timing: id  0 | task 414 | draft acceptance = 0.75862 (  308 accepted /   406 generated)",
      "0.42.928.830 W slot update_slots: id  2 | task 622 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "0.56.321.669 I slot print_timing: id  2 | task 622 | draft acceptance = 0.78392 (  312 accepted /   398 generated)",
      "0.56.458.507 W slot update_slots: id  0 | task 829 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.08.817.866 I slot print_timing: id  0 | task 829 | draft acceptance = 0.76049 (  308 accepted /   405 generated)",
      "1.08.952.604 W slot update_slots: id  1 | task 1036 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.21.338.314 I slot print_timing: id  1 | task 1036 | draft acceptance = 0.75430 (  307 accepted /   407 generated)",
      "1.21.503.907 W slot update_slots: id  2 | task 1245 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.34.524.225 I slot print_timing: id  2 | task 1245 | draft acceptance = 0.77945 (  311 accepted /   399 generated)",
      "1.34.665.119 W slot update_slots: id  0 | task 1451 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.47.410.557 I slot print_timing: id  0 | task 1451 | draft acceptance = 0.80102 (  314 accepted /   392 generated)",
      "1.47.545.891 W slot update_slots: id  1 | task 1654 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "2.00.468.533 I slot print_timing: id  1 | task 1654 | draft acceptance = 0.72010 (  301 accepted /   418 generated)",
      "2.00.610.882 W slot update_slots: id  2 | task 1869 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "2.16.586.145 I slot print_timing: id  2 | task 1869 | draft acceptance = 0.70118 (  298 accepted /   425 generated)"
    ],
    "errors": [],
    "jsonl": "C:\\Users\\sjake\\LLM\\logs\\swebench-lite-first10-mainline64k-27b-q3xl-mtp-np3-q8q8-20260526-194600.jsonl"
  },
  {
    "name": "mainline64k-35a3b-q3xl-mtp-np3-q8q8",
    "model": "C:\\Users\\sjake\\LLM\\models\\unsloth-qwen3.6-35b-a3b-mtp-gguf\\Qwen3.6-35B-A3B-UD-Q3_K_XL.gguf",
    "mtp": true,
    "n_ok": 10,
    "n_total": 10,
    "avg_wall_s": 8.052431630000138,
    "avg_prompt_tps": 703.6676584213677,
    "avg_predicted_tps": 81.36078570012702,
    "avg_prompt_ms": 1701.4,
    "avg_predicted_ms": 6293.9525,
    "total_prompt_tokens": 13132,
    "total_completion_tokens": 5120,
    "draft_tokens": 4159,
    "accepted_draft_tokens": 3022,
    "draft_accept_rate": 0.7266169752344314,
    "vram_before": {
      "used_mib": 1252,
      "free_mib": 23062,
      "power_w": 29.11,
      "temp_c": 48
    },
    "vram_ready": {
      "used_mib": 19037,
      "free_mib": 5277,
      "power_w": 83.08,
      "temp_c": 48
    },
    "vram_after": {
      "used_mib": 19198,
      "free_mib": 5116,
      "power_w": 131.42,
      "temp_c": 47
    },
    "net_ready_mib": 17785,
    "props_n_ctx": 21504,
    "total_slots": 3,
    "log": "C:\\Users\\sjake\\LLM\\logs\\mainline64k-35a3b-q3xl-mtp-np3-q8q8-20260526-194826.server.log",
    "interesting_lines": [
      "0.00.541.862 I common_memory_breakdown_print: | memory breakdown [MiB]  | total    free     self   model   context   compute    unaccounted |",
      "0.00.541.867 I common_memory_breakdown_print: |   - CUDA0 (RTX 3090 Ti) | 24563 = 23285 + (16095 = 15903 +      66 +     125) +      -14817 |",
      "0.00.541.867 I common_memory_breakdown_print: |   - Host                |                    528 =   515 +       0 +      12                |",
      "0.00.921.992 I common_memory_breakdown_print: | memory breakdown [MiB]  | total    free     self   model   context   compute    unaccounted |",
      "0.00.921.996 I common_memory_breakdown_print: |   - CUDA0 (RTX 3090 Ti) | 24563 = 22719 + (17263 = 15903 +    1234 +     125) +      -15419 |",
      "0.00.921.997 I common_memory_breakdown_print: |   - Host                |                    528 =   515 +       0 +      12                |",
      "0.04.447.608 I load_tensors:   CPU_Mapped model buffer size =   515.31 MiB",
      "0.04.447.609 I load_tensors:        CUDA0 model buffer size = 15903.70 MiB",
      "0.10.129.782 I llama_kv_cache:      CUDA0 KV buffer size =   669.38 MiB",
      "0.10.249.805 I llama_memory_recurrent:      CUDA0 RS buffer size =   565.31 MiB",
      "0.10.259.724 I sched_reserve:      CUDA0 compute buffer size =   125.22 MiB",
      "0.10.259.729 I sched_reserve:  CUDA_Host compute buffer size =    12.87 MiB",
      "0.10.536.756 I llama_kv_cache:      CUDA0 KV buffer size =    66.94 MiB",
      "0.10.548.193 I sched_reserve:      CUDA0 compute buffer size =   125.22 MiB",
      "0.10.548.197 I sched_reserve:  CUDA_Host compute buffer size =    12.87 MiB",
      "0.12.055.874 I sched_reserve:      CUDA0 compute buffer size =   248.37 MiB",
      "0.12.055.879 I sched_reserve:  CUDA_Host compute buffer size =    12.87 MiB",
      "0.18.599.541 I slot print_timing: id  2 | task 0 | draft acceptance = 0.77500 (  310 accepted /   400 generated)",
      "0.25.788.914 I slot print_timing: id  1 | task 205 | draft acceptance = 0.75616 (  307 accepted /   406 generated)",
      "0.33.372.644 I slot print_timing: id  0 | task 413 | draft acceptance = 0.75000 (  306 accepted /   408 generated)",
      "0.33.447.518 W slot update_slots: id  2 | task 623 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "0.42.571.207 I slot print_timing: id  2 | task 623 | draft acceptance = 0.71599 (  300 accepted /   419 generated)",
      "0.42.638.348 W slot update_slots: id  0 | task 842 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "0.49.584.079 I slot print_timing: id  0 | task 842 | draft acceptance = 0.70853 (  299 accepted /   422 generated)",
      "0.49.658.547 W slot update_slots: id  1 | task 1058 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "0.57.028.700 I slot print_timing: id  1 | task 1058 | draft acceptance = 0.74209 (  305 accepted /   411 generated)",
      "0.57.102.471 W slot update_slots: id  2 | task 1269 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.05.302.438 I slot print_timing: id  2 | task 1269 | draft acceptance = 0.74209 (  305 accepted /   411 generated)",
      "1.05.376.919 W slot update_slots: id  0 | task 1481 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.13.836.915 I slot print_timing: id  0 | task 1481 | draft acceptance = 0.70283 (  298 accepted /   424 generated)",
      "1.13.908.320 W slot update_slots: id  1 | task 1700 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.21.733.672 I slot print_timing: id  1 | task 1700 | draft acceptance = 0.72249 (  302 accepted /   418 generated)",
      "1.21.812.837 W slot update_slots: id  2 | task 1914 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.32.068.707 I slot print_timing: id  2 | task 1914 | draft acceptance = 0.65909 (  290 accepted /   440 generated)"
    ],
    "errors": [],
    "jsonl": "C:\\Users\\sjake\\LLM\\logs\\swebench-lite-first10-mainline64k-35a3b-q3xl-mtp-np3-q8q8-20260526-194826.jsonl"
  },
  {
    "name": "mainline64k-35a3b-q3xl-nonmtp-np3-q8q8",
    "model": "C:\\Users\\sjake\\LLM\\models\\unsloth-qwen3.6-35b-a3b-gguf\\Qwen3.6-35B-A3B-UD-Q3_K_XL.gguf",
    "mtp": false,
    "n_ok": 10,
    "n_total": 10,
    "avg_wall_s": 7.277327670000159,
    "avg_prompt_tps": 877.7555848446085,
    "avg_predicted_tps": 89.36095540313923,
    "avg_prompt_ms": 1397.7045,
    "avg_predicted_ms": 5826.3127,
    "total_prompt_tokens": 13132,
    "total_completion_tokens": 5120,
    "draft_tokens": 0,
    "accepted_draft_tokens": 0,
    "draft_accept_rate": null,
    "vram_before": {
      "used_mib": 1248,
      "free_mib": 23066,
      "power_w": 29.18,
      "temp_c": 41
    },
    "vram_ready": {
      "used_mib": 18110,
      "free_mib": 6204,
      "power_w": 81.94,
      "temp_c": 43
    },
    "vram_after": {
      "used_mib": 18107,
      "free_mib": 6207,
      "power_w": 215.13,
      "temp_c": 56
    },
    "net_ready_mib": 16862,
    "props_n_ctx": 21504,
    "total_slots": 3,
    "log": "C:\\Users\\sjake\\LLM\\logs\\mainline64k-35a3b-q3xl-nonmtp-np3-q8q8-20260526-195008.server.log",
    "interesting_lines": [
      "0.00.539.592 I common_memory_breakdown_print: | memory breakdown [MiB]  | total    free     self   model   context   compute    unaccounted |",
      "0.00.539.596 I common_memory_breakdown_print: |   - CUDA0 (RTX 3090 Ti) | 24563 = 23095 + (16522 = 15539 +     857 +     125) +      -15053 |",
      "0.00.539.597 I common_memory_breakdown_print: |   - Host                |                    528 =   515 +       0 +      12                |",
      "0.03.963.331 I load_tensors:   CPU_Mapped model buffer size =   515.31 MiB",
      "0.03.963.333 I load_tensors:        CUDA0 model buffer size = 15539.34 MiB",
      "0.09.725.542 I llama_kv_cache:      CUDA0 KV buffer size =   669.38 MiB",
      "0.09.771.316 I llama_memory_recurrent:      CUDA0 RS buffer size =   188.44 MiB",
      "0.09.780.799 I sched_reserve:      CUDA0 compute buffer size =   125.22 MiB",
      "0.09.780.804 I sched_reserve:  CUDA_Host compute buffer size =    12.87 MiB",
      "0.31.176.106 W slot update_slots: id  2 | task 1546 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "0.38.949.396 W slot update_slots: id  0 | task 2065 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "0.46.144.051 W slot update_slots: id  1 | task 2580 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "0.53.013.538 W slot update_slots: id  2 | task 3096 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.00.779.925 W slot update_slots: id  0 | task 3613 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.07.700.976 W slot update_slots: id  1 | task 4130 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)",
      "1.15.252.573 W slot update_slots: id  2 | task 4646 | forcing full prompt re-processing due to lack of cache data (likely due to SWA or hybrid/recurrent memory, see https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)"
    ],
    "errors": [],
    "jsonl": "C:\\Users\\sjake\\LLM\\logs\\swebench-lite-first10-mainline64k-35a3b-q3xl-nonmtp-np3-q8q8-20260526-195008.jsonl"
  }
]