[System] Starting Qwen3.8-Couture-Engine-27B with Advanced Optimizations...
0.00.080.805 W DEPRECATED: --mmap and --no-mmap are deprecated. use --load-mode mmap instead
0.00.087.073 I cmn common_param: common_params_print_info: verbosity = 3 (adjust with the `-lv N` CLI arg)
0.00.141.618 W srv llama_server: -----------------
0.00.141.625 W srv llama_server: CORS is set to allow all origins ('*') and no API key is set
0.00.141.625 W srv llama_server: this can be a security risk (cross-origin attacks)
0.00.141.625 W srv llama_server: more info: https://github.com/ggml-org/llama.cpp/pull/25655
0.00.141.626 W srv llama_server: -----------------
0.00.145.260 I srv load_model: loading model 'Qwen3.8-Couture-Engine-27B-Q4_K.gguf'
0.11.569.589 I cmn init: llama threadpool init, n_threads = 8
0.11.648.942 I common_speculative_init_result: creating MTP draft context against the target model 'Qwen3.8-Couture-Engine-27B-Q4_K.gguf'
0.11.717.997 W load_hparams: Qwen-VL models require at minimum 1024 image tokens to function correctly on grounding tasks
0.11.718.002 W load_hparams: if you encounter problems with accuracy, try adding --image-min-tokens 1024
0.11.718.003 W load_hparams: more info: https://github.com/ggml-org/llama.cpp/issues/16842
0.15.654.843 I srv load_model: loaded multimodal model, 'mmproj-Qwen3.8-Couture-Engine-27B-BF16.gguf'
0.15.787.036 I srv load_model: initializing, n_slots = 1, n_ctx_slot = 131072, kv_unified = 'true'
0.15.931.596 I srv init: chat template supports preserving reasoning, consider enabling it via --reasoning-preserve
0.15.931.639 I srv llama_server: model loaded
0.15.931.643 I srv llama_server: listening on http://127.0.0.1:8080
0.15.931.644 W srv llama_server: NOTICE: server default port will be changed to :9931 in a future release
0.15.931.644 W srv llama_server: ref: https://github.com/ggml-org/llama.cpp/pull/26508
0.59.702.850 I slot get_availabl: id 0 | task -1 | selected slot by LRU, t_last = -1
0.59.798.051 I slot launch_slot_: id 0 | task 0 | processing task, is_child = 0
1.01.065.764 I slot print_timing: id 0 | task 0 | prompt eval time = 589.09 ms / 798 tokens ( 0.74 ms per token, 1354.64 tokens per second)
1.01.065.770 I slot print_timing: id 0 | task 0 | eval time = 678.57 ms / 64 tokens ( 10.77 ms per token, 92.84 tokens per second)
1.01.065.771 I slot print_timing: id 0 | task 0 | total time = 1267.65 ms / 862 tokens
1.01.065.772 I slot print_timing: id 0 | task 0 | graphs reused = 21
1.01.065.778 I slot print_timing: id 0 | task 0 | draft acceptance = 0.62500 ( 40 accepted / 64 generated), mean len = 2.82
1.01.065.876 I slot release: id 0 | task 0 | stop processing: n_tokens = 861, truncated = 0
1.01.065.886 I slot get_availabl: id 0 | task -1 | selected slot by LRU, t_last = 60988403
1.01.295.729 I slot launch_slot_: id 0 | task 1 | processing task, is_child = 0
1.07.389.341 I slot print_timing: id 0 | task 1 | n_gen = 345, tg = 114.09 t/s, tg_3s = 114.41 t/s
1.10.405.097 I slot print_timing: id 0 | task 1 | n_gen = 667, tg = 110.43 t/s, tg_3s = 106.77 t/s
1.13.413.678 I slot print_timing: id 0 | task 1 | n_gen = 965, tg = 106.64 t/s, tg_3s = 99.05 t/s
1.16.415.350 I slot print_timing: id 0 | task 1 | n_gen = 1302, tg = 108.05 t/s, tg_3s = 112.27 t/s
1.19.419.948 I slot print_timing: id 0 | task 1 | n_gen = 1621, tg = 107.67 t/s, tg_3s = 106.17 t/s
1.22.452.598 I slot print_timing: id 0 | task 1 | n_gen = 1928, tg = 106.59 t/s, tg_3s = 101.23 t/s
1.25.477.881 I slot print_timing: id 0 | task 1 | n_gen = 2232, tg = 105.72 t/s, tg_3s = 100.49 t/s
1.28.488.909 I slot print_timing: id 0 | task 1 | n_gen = 2559, tg = 106.08 t/s, tg_3s = 108.60 t/s
1.31.506.796 I slot print_timing: id 0 | task 1 | n_gen = 2888, tg = 106.40 t/s, tg_3s = 109.02 t/s
1.34.515.451 I slot print_timing: id 0 | task 1 | n_gen = 3213, tg = 106.57 t/s, tg_3s = 108.02 t/s
1.37.530.266 I slot print_timing: id 0 | task 1 | n_gen = 3523, tg = 106.22 t/s, tg_3s = 102.83 t/s
1.40.552.797 I slot print_timing: id 0 | task 1 | n_gen = 3815, tg = 105.42 t/s, tg_3s = 96.61 t/s
1.43.554.898 I slot print_timing: id 0 | task 1 | n_gen = 4105, tg = 104.75 t/s, tg_3s = 96.60 t/s
1.46.559.961 I slot print_timing: id 0 | task 1 | n_gen = 4372, tg = 103.61 t/s, tg_3s = 88.85 t/s
1.49.571.488 I slot print_timing: id 0 | task 1 | n_gen = 4674, tg = 103.39 t/s, tg_3s = 100.28 t/s
1.52.603.529 I slot print_timing: id 0 | task 1 | n_gen = 4942, tg = 102.45 t/s, tg_3s = 88.39 t/s
1.55.627.142 I slot print_timing: id 0 | task 1 | n_gen = 5237, tg = 102.16 t/s, tg_3s = 97.57 t/s
1.58.650.338 I slot print_timing: id 0 | task 1 | n_gen = 5566, tg = 102.53 t/s, tg_3s = 108.82 t/s
2.01.659.790 I slot print_timing: id 0 | task 1 | n_gen = 5932, tg = 103.53 t/s, tg_3s = 121.62 t/s
2.04.685.254 I slot print_timing: id 0 | task 1 | n_gen = 6269, tg = 103.93 t/s, tg_3s = 111.39 t/s
2.07.699.298 I slot print_timing: id 0 | task 1 | n_gen = 6544, tg = 103.32 t/s, tg_3s = 91.24 t/s
2.10.735.241 I slot print_timing: id 0 | task 1 | n_gen = 6834, tg = 102.97 t/s, tg_3s = 95.52 t/s
2.13.761.201 I slot print_timing: id 0 | task 1 | n_gen = 7119, tg = 102.58 t/s, tg_3s = 94.18 t/s
2.16.769.773 I slot print_timing: id 0 | task 1 | n_gen = 7433, tg = 102.66 t/s, tg_3s = 104.37 t/s
2.19.798.442 I slot print_timing: id 0 | task 1 | n_gen = 7773, tg = 103.04 t/s, tg_3s = 112.26 t/s
2.22.816.186 I slot print_timing: id 0 | task 1 | n_gen = 8129, tg = 103.62 t/s, tg_3s = 117.97 t/s
2.25.838.711 I slot print_timing: id 0 | task 1 | n_gen = 8423, tg = 103.38 t/s, tg_3s = 97.27 t/s
2.28.839.157 I slot print_timing: id 0 | task 1 | n_gen = 8720, tg = 103.23 t/s, tg_3s = 98.99 t/s
2.31.840.950 I slot print_timing: id 0 | task 1 | n_gen = 9029, tg = 103.22 t/s, tg_3s = 102.94 t/s
2.34.846.240 I slot print_timing: id 0 | task 1 | n_gen = 9361, tg = 103.46 t/s, tg_3s = 110.47 t/s
2.37.878.572 I slot print_timing: id 0 | task 1 | n_gen = 9749, tg = 104.25 t/s, tg_3s = 127.95 t/s
2.39.030.755 I slot print_timing: id 0 | task 1 | prompt eval time = 3078.20 ms / 8937 tokens ( 0.34 ms per token, 2903.32 tokens per second)
2.39.030.771 I slot print_timing: id 0 | task 1 | eval time = 94656.20 ms / 9882 tokens ( 9.58 ms per token, 104.39 tokens per second)
2.39.030.775 I slot print_timing: id 0 | task 1 | total time = 97734.40 ms / 18819 tokens
2.39.030.778 I slot print_timing: id 0 | task 1 | graphs reused = 3164
2.39.030.794 I slot print_timing: id 0 | task 1 | draft acceptance = 0.70164 ( 6700 accepted / 9549 generated), mean len = 3.10
2.39.032.126 I slot release: id 0 | task 1 | stop processing: n_tokens = 18820, truncated = 0
2.39.190.578 I slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.880 (> 0.100 thold), f_keep = 0.475
2.39.623.937 I slot launch_slot_: id 0 | task 3219 | processing task, is_child = 0
2.40.887.873 I slot print_timing: id 0 | task 3219 | prompt eval time = 656.38 ms / 1227 tokens ( 0.53 ms per token, 1869.36 tokens per second)
2.40.887.877 I slot print_timing: id 0 | task 3219 | eval time = 607.30 ms / 68 tokens ( 9.06 ms per token, 110.32 tokens per second)
2.40.887.878 I slot print_timing: id 0 | task 3219 | total time = 1263.68 ms / 1295 tokens
2.40.887.879 I slot print_timing: id 0 | task 3219 | graphs reused = 3183
2.40.887.884 I slot print_timing: id 0 | task 3219 | draft acceptance = 0.81667 ( 49 accepted / 60 generated), mean len = 3.45
2.40.888.537 I slot release: id 0 | task 3219 | stop processing: n_tokens = 10229, truncated = 0
2.42.890.504 I slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.987 (> 0.100 thold), f_keep = 0.993
2.42.977.112 I slot launch_slot_: id 0 | task 3243 | processing task, is_child = 0
2.43.977.832 I slot print_timing: id 0 | task 3243 | prompt eval time = 293.66 ms / 141 tokens ( 2.08 ms per token, 480.14 tokens per second)
2.43.977.840 I slot print_timing: id 0 | task 3243 | eval time = 706.59 ms / 78 tokens ( 9.18 ms per token, 108.97 tokens per second)
2.43.977.843 I slot print_timing: id 0 | task 3243 | total time = 1000.26 ms / 219 tokens
2.43.977.844 I slot print_timing: id 0 | task 3243 | graphs reused = 3204
2.43.977.852 I slot print_timing: id 0 | task 3243 | draft acceptance = 0.86364 ( 57 accepted / 66 generated), mean len = 3.59
2.43.978.559 I slot release: id 0 | task 3243 | stop processing: n_tokens = 10376, truncated = 0
3.00.000.935 I slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.960 (> 0.100 thold), f_keep = 0.992
3.00.083.611 I slot launch_slot_: id 0 | task 3268 | processing task, is_child = 0
3.01.814.248 I slot print_timing: id 0 | task 3268 | prompt eval time = 519.69 ms / 436 tokens ( 1.19 ms per token, 838.96 tokens per second)
3.01.814.255 I slot print_timing: id 0 | task 3268 | eval time = 1210.68 ms / 130 tokens ( 9.39 ms per token, 106.55 tokens per second)
3.01.814.257 I slot print_timing: id 0 | task 3268 | total time = 1730.37 ms / 566 tokens
3.01.814.258 I slot print_timing: id 0 | task 3268 | graphs reused = 3239
3.01.814.264 I slot print_timing: id 0 | task 3268 | draft acceptance = 0.84685 ( 94 accepted / 111 generated), mean len = 3.54
3.01.814.967 I slot release: id 0 | task 3268 | stop processing: n_tokens = 10860, truncated = 0
3.01.900.833 I slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.986 (> 0.100 thold), f_keep = 0.988
3.02.006.600 I slot launch_slot_: id 0 | task 3308 | processing task, is_child = 0
3.03.748.409 I slot print_timing: id 0 | task 3308 | prompt eval time = 277.26 ms / 154 tokens ( 1.80 ms per token, 555.43 tokens per second)
3.03.748.415 I slot print_timing: id 0 | task 3308 | eval time = 1464.31 ms / 167 tokens ( 8.82 ms per token, 113.36 tokens per second)
3.03.748.416 I slot print_timing: id 0 | task 3308 | total time = 1741.58 ms / 321 tokens
3.03.748.417 I slot print_timing: id 0 | task 3308 | graphs reused = 3283
3.03.748.422 I slot print_timing: id 0 | task 3308 | draft acceptance = 0.88406 ( 122 accepted / 138 generated), mean len = 3.65
3.03.749.113 I slot release: id 0 | task 3308 | stop processing: n_tokens = 11047, truncated = 0
3.24.061.514 I slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.987 (> 0.100 thold), f_keep = 0.985
3.24.160.843 I slot launch_slot_: id 0 | task 3357 | processing task, is_child = 0
3.25.260.230 I slot print_timing: id 0 | task 3357 | prompt eval time = 576.40 ms / 143 tokens ( 4.03 ms per token, 248.09 tokens per second)
3.25.260.235 I slot print_timing: id 0 | task 3357 | eval time = 522.68 ms / 69 tokens ( 7.69 ms per token, 130.10 tokens per second)
3.25.260.236 I slot print_timing: id 0 | task 3357 | total time = 1099.09 ms / 212 tokens
3.25.260.237 I slot print_timing: id 0 | task 3357 | graphs reused = 3299
3.25.260.242 I slot print_timing: id 0 | task 3357 | draft acceptance = 1.00000 ( 51 accepted / 51 generated), mean len = 4.00
3.25.260.977 I slot release: id 0 | task 3357 | stop processing: n_tokens = 11086, truncated = 0
3.30.709.825 I slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.949 (> 0.100 thold), f_keep = 0.994
3.30.783.674 I slot launch_slot_: id 0 | task 3377 | processing task, is_child = 0
3.31.792.151 I slot print_timing: id 0 | task 3377 | prompt eval time = 385.90 ms / 598 tokens ( 0.65 ms per token, 1549.64 tokens per second)
3.31.792.157 I slot print_timing: id 0 | task 3377 | eval time = 622.36 ms / 65 tokens ( 9.72 ms per token, 102.83 tokens per second)
3.31.792.158 I slot print_timing: id 0 | task 3377 | total time = 1008.26 ms / 663 tokens
3.31.792.159 I slot print_timing: id 0 | task 3377 | graphs reused = 3318
3.31.792.164 I slot print_timing: id 0 | task 3377 | draft acceptance = 0.73333 ( 44 accepted / 60 generated), mean len = 3.20
3.31.792.784 I slot release: id 0 | task 3377 | stop processing: n_tokens = 11676, truncated = 0
3.31.889.111 I slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.990 (> 0.100 thold), f_keep = 0.994
3.31.999.290 I slot launch_slot_: id 0 | task 3400 | processing task, is_child = 0
3.35.289.917 I slot print_timing: id 0 | task 3400 | n_gen = 303, tg = 100.05 t/s, tg_3s = 100.37 t/s
3.38.318.301 I slot print_timing: id 0 | task 3400 | n_gen = 548, tg = 90.46 t/s, tg_3s = 80.90 t/s
3.41.337.932 I slot print_timing: id 0 | task 3400 | n_gen = 860, tg = 94.74 t/s, tg_3s = 103.32 t/s
3.44.363.880 I slot print_timing: id 0 | task 3400 | n_gen = 1168, tg = 96.51 t/s, tg_3s = 101.79 t/s
3.47.364.422 I slot print_timing: id 0 | task 3400 | n_gen = 1468, tg = 97.20 t/s, tg_3s = 99.98 t/s
3.50.390.516 I slot print_timing: id 0 | task 3400 | n_gen = 1822, tg = 100.50 t/s, tg_3s = 116.98 t/s
3.53.408.340 I slot print_timing: id 0 | task 3400 | n_gen = 2137, tg = 101.06 t/s, tg_3s = 104.38 t/s
3.56.431.692 I slot print_timing: id 0 | task 3400 | n_gen = 2416, tg = 99.96 t/s, tg_3s = 92.28 t/s
3.59.444.218 I slot print_timing: id 0 | task 3400 | n_gen = 2667, tg = 98.11 t/s, tg_3s = 83.32 t/s
4.02.450.740 I slot print_timing: id 0 | task 3400 | n_gen = 2933, tg = 97.15 t/s, tg_3s = 88.47 t/s
4.05.458.471 I slot print_timing: id 0 | task 3400 | n_gen = 3214, tg = 96.81 t/s, tg_3s = 93.43 t/s
4.08.466.165 I slot print_timing: id 0 | task 3400 | n_gen = 3525, tg = 97.36 t/s, tg_3s = 103.40 t/s
4.11.499.516 I slot print_timing: id 0 | task 3400 | n_gen = 3814, tg = 97.20 t/s, tg_3s = 95.27 t/s
4.14.526.050 I slot print_timing: id 0 | task 3400 | n_gen = 4116, tg = 97.39 t/s, tg_3s = 99.78 t/s
4.17.541.465 I slot print_timing: id 0 | task 3400 | n_gen = 4436, tg = 97.97 t/s, tg_3s = 106.12 t/s
4.20.560.899 I slot print_timing: id 0 | task 3400 | n_gen = 4815, tg = 99.69 t/s, tg_3s = 125.52 t/s
4.23.588.078 I slot print_timing: id 0 | task 3400 | n_gen = 5199, tg = 101.29 t/s, tg_3s = 126.85 t/s
4.26.613.916 I slot print_timing: id 0 | task 3400 | n_gen = 5555, tg = 102.20 t/s, tg_3s = 117.65 t/s
4.27.644.311 I slot print_timing: id 0 | task 3400 | prompt eval time = 271.83 ms / 124 tokens ( 2.19 ms per token, 456.16 tokens per second)
4.27.644.316 I slot print_timing: id 0 | task 3400 | eval time = 55372.84 ms / 5653 tokens ( 9.80 ms per token, 102.07 tokens per second)
4.27.644.317 I slot print_timing: id 0 | task 3400 | total time = 55644.67 ms / 5777 tokens
4.27.644.318 I slot print_timing: id 0 | task 3400 | graphs reused = 5171
4.27.644.322 I slot print_timing: id 0 | task 3400 | draft acceptance = 0.67111 ( 3777 accepted / 5628 generated), mean len = 3.01
4.27.645.058 I slot release: id 0 | task 3400 | stop processing: n_tokens = 17385, truncated = 0
5.42.302.525 I slot get_availabl: id 0 | task -1 | selected slot by LRU, t_last = 267567585
5.42.828.527 I slot launch_slot_: id 0 | task 5279 | processing task, is_child = 0
5.43.536.817 W find_slot: non-consecutive token position 46 after 45 for sequence 0 with 1024 new tokens
5.43.536.825 W find_slot: non-consecutive token position 46 after 46 for sequence 0 with 992 new tokens
5.43.537.832 W find_slot: non-consecutive token position 46 after 45 for sequence 0 with 1024 new tokens
5.43.711.882 W find_slot: non-consecutive token position 46 after 46 for sequence 0 with 992 new tokens
5.43.992.335 W find_slot: non-consecutive token position 114 after 46 for sequence 0 with 6 new tokens
5.43.992.391 W find_slot: non-consecutive token position 114 after 46 for sequence 0 with 6 new tokens
5.47.200.004 I slot print_timing: id 0 | task 5279 | n_gen = 293, tg = 97.33 t/s, tg_3s = 97.66 t/s
5.50.206.077 I slot print_timing: id 0 | task 5279 | n_gen = 546, tg = 90.75 t/s, tg_3s = 84.16 t/s
5.53.217.348 I slot print_timing: id 0 | task 5279 | n_gen = 812, tg = 89.94 t/s, tg_3s = 88.33 t/s
5.56.224.654 I slot print_timing: id 0 | task 5279 | n_gen = 1099, tg = 91.31 t/s, tg_3s = 95.43 t/s
5.59.233.818 I slot print_timing: id 0 | task 5279 | n_gen = 1387, tg = 92.19 t/s, tg_3s = 95.71 t/s
6.02.235.630 I slot print_timing: id 0 | task 5279 | n_gen = 1618, tg = 89.66 t/s, tg_3s = 76.95 t/s
6.02.643.556 I slot print_timing: id 0 | task 5279 | prompt eval time = 1371.26 ms / 2072 tokens ( 0.66 ms per token, 1511.02 tokens per second)
6.02.643.560 I slot print_timing: id 0 | task 5279 | eval time = 18443.55 ms / 1643 tokens ( 11.23 ms per token, 89.03 tokens per second)
6.02.643.561 I slot print_timing: id 0 | task 5279 | total time = 19814.81 ms / 3715 tokens
6.02.643.561 I slot print_timing: id 0 | task 5279 | graphs reused = 5840
6.02.643.565 I slot print_timing: id 0 | task 5279 | draft acceptance = 0.47633 ( 966 accepted / 2028 generated), mean len = 2.43
6.02.643.600 I slot release: id 0 | task 5279 | stop processing: n_tokens = 3714, truncated = 0