time=2026-10-05T09:30:03.736+02:00 level=INFO source=routes.go:2117 msg="server config" env="map[CUDA_VISIBLE_DEVICES: GGML_VK_VISIBLE_DEVICES: GPU_DEVICE_ORDINAL: HIP_VISIBLE_DEVICES: HSA_OVERRIDE_GFX_VERSION: HTTPS_PROXY: HTTP_PROXY: LLAMA_ARG_FIT: LLAMA_ARG_FIT_TARGET: NO_PROXY: OLLAMA_CONTEXT_LENGTH:0 OLLAMA_CREATE_REMOTE:false OLLAMA_DEBUG:INFO OLLAMA_DEBUG_LOG_REQUESTS:false OLLAMA_EDITOR: OLLAMA_FLASH_ATTENTION:false OLLAMA_GO_TEMPLATE:true OLLAMA_GPU_OVERHEAD:0 OLLAMA_HOST:http://127.0.0.1:11434 OLLAMA_IGPU_ENABLE: OLLAMA_KEEP_ALIVE:5m0s OLLAMA_KV_CACHE_TYPE: OLLAMA_LLM_LIBRARY: OLLAMA_LOAD_TIMEOUT:5m0s OLLAMA_MAX_LOADED_MODELS:0 OLLAMA_MAX_QUEUE:512 OLLAMA_MAX_TRANSFER_STREAMS:4 OLLAMA_MODELS:C:\\Users\\jerom\\.ollama\\models OLLAMA_NOHISTORY:false OLLAMA_NOPRUNE:false OLLAMA_NO_CLOUD:false OLLAMA_NUM_PARALLEL:1 OLLAMA_ORIGINS:[http://localhost https://localhost http://localhost:* https://localhost:* http://127.0.0.1 https://127.0.0.1 http://127.0.0.1:* https://127.0.0.1:* http://0.0.0.0 https://0.0.0.0 http://0.0.0.0:* https://0.0.0.0:* app://* file://* tauri://* vscode-webview://* vscode-file://*] OLLAMA_REMOTES:[ollama.com] OLLAMA_SCHED_SPREAD:false OLLAMA_VULKAN:true ROCR_VISIBLE_DEVICES:]"
time=2026-10-05T09:30:03.781+02:00 level=INFO source=routes.go:2119 msg="Ollama cloud disabled: false"
time=2026-10-05T09:30:03.785+02:00 level=INFO source=images.go:947 msg="total blobs: 8"
time=2026-10-05T09:30:03.786+02:00 level=INFO source=images.go:955 msg="total unused blobs removed: 0"
time=2026-10-05T09:30:03.790+02:00 level=INFO source=routes.go:2177 msg="Listening on 127.0.0.1:11434 (version 0.35.1)"
time=2026-10-05T09:30:03.796+02:00 level=INFO source=runner.go:60 msg="discovering available GPUs..."
time=2026-10-05T09:30:03.994+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h21m45.61569341s consecutive_failures=0
time=2026-10-05T09:30:12.974+02:00 level=INFO source=runner.go:405 msg="dropping integrated GPU; to enable, set OLLAMA_IGPU_ENABLE=1" id=0 library=Vulkan compute=0.0 name=Vulkan0 description="Intel(R) UHD Graphics 770" pci_id=""
time=2026-10-05T09:30:14.050+02:00 level=INFO source=types.go:32 msg="inference compute" id=0 filter_id=0 library=CUDA compute=8.6 name=CUDA0 description="NVIDIA GeForce RTX 3080" libdirs=ollama,cuda_v13 driver=13.4 pci_id=0000:01:00.0 type=discrete total="10.0 GiB" available="8.9 GiB"
time=2026-10-05T09:30:14.050+02:00 level=INFO source=routes.go:2227 msg="vram-based default context" total_vram="10.0 GiB" default_num_ctx=4096
[GIN] 2026/10/05 - 09:30:14 | 500 |      2.5009ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:30:14 | 500 |      3.0004ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:30:14 | 500 |      3.0004ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:30:14 | 500 |      5.3519ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:30:14 | 500 |      6.5092ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:30:14 | 200 |      9.0071ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:30:14 | 200 |      3.4993ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:30:14 | 200 |     57.7955ms |       127.0.0.1 | POST     "/api/show"
[GIN] 2026/10/05 - 09:30:17 | 200 |      6.0158ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:30:47 | 200 |      5.5113ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:31:17 | 200 |      5.4419ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:31:47 | 200 |      3.9959ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:32:17 | 200 |      10.536ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:32:47 | 200 |      4.0916ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:32:48 | 200 |      3.6425ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:32:48 | 200 |      5.0084ms |       127.0.0.1 | POST     "/api/show"
[GIN] 2026/10/05 - 09:32:48 | 200 |     52.5524ms |       127.0.0.1 | POST     "/api/show"
[GIN] 2026/10/05 - 09:32:48 | 200 |      4.5145ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:32:48 | 200 |      2.5924ms |       127.0.0.1 | POST     "/api/show"
[GIN] 2026/10/05 - 09:32:48 | 200 |      3.5007ms |       127.0.0.1 | POST     "/api/show"
time=2026-10-05T09:32:49.671+02:00 level=INFO source=sched.go:1152 msg="disabling mmap for llama-server load by default" model=C:\Users\jerom\.ollama\models\blobs\sha256-1194192cf2a187eb02722edcc3f77b11d21f537048ce04b67ccf8ba78863006a reason=windows_cuda
time=2026-10-05T09:32:49.671+02:00 level=INFO source=server.go:100 msg="using llama-server for model" model=C:\Users\jerom\.ollama\models\blobs\sha256-1194192cf2a187eb02722edcc3f77b11d21f537048ce04b67ccf8ba78863006a
time=2026-10-05T09:32:49.691+02:00 level=INFO source=llama_server.go:436 msg="starting llama-server" cmd="C:\\Users\\jerom\\AppData\\Local\\Programs\\Ollama\\lib\\ollama\\llama-server.exe --model C:\\Users\\jerom\\.ollama\\models\\blobs\\sha256-1194192cf2a187eb02722edcc3f77b11d21f537048ce04b67ccf8ba78863006a --port 53541 --host 127.0.0.1 --no-webui --offline -c 16384 -np 1 --log-verbosity 4 --no-log-prefix --no-log-timestamps --no-jinja --chat-template chatml --load-mode none --flash-attn auto -b 512 -ub 512 --context-shift --keep 4"
time=2026-10-05T09:32:49.699+02:00 level=INFO source=sched.go:618 msg="system memory" total="31.7 GiB" free="4.1 GiB" free_swap="2.9 GiB"
time=2026-10-05T09:32:49.699+02:00 level=INFO source=sched.go:625 msg="gpu memory" id=0 library=CUDA available="8.4 GiB" free="8.9 GiB" minimum="457.0 MiB" overhead="0 B"
time=2026-10-05T09:32:49.699+02:00 level=INFO source=llama_server.go:1054 msg="loading model via llama-server" model=C:\Users\jerom\.ollama\models\blobs\sha256-1194192cf2a187eb02722edcc3f77b11d21f537048ce04b67ccf8ba78863006a
time=2026-10-05T09:32:49.699+02:00 level=INFO source=llama_server.go:1307 msg="waiting for llama-server to start responding"
time=2026-10-05T09:32:49.700+02:00 level=INFO source=llama_server.go:1362 msg="waiting for llama-server to become available" status="llm server error"
srv  llama_server: initializing ...
load_backend: loaded CPU backend from C:\Users\jerom\AppData\Local\Programs\Ollama\lib\ollama\ggml-cpu-alderlake.dll
ggml_cuda_init: found 1 CUDA devices (Total VRAM: 10239 MiB):
  Device 0: NVIDIA GeForce RTX 3080, compute capability 8.6, VMM: yes, VRAM: 10239 MiB
load_backend: loaded CUDA backend from C:\Users\jerom\AppData\Local\Programs\Ollama\lib\ollama\cuda_v13\ggml-cuda.dll
cmn  common_param: common_params_print_info: build 1 (6f767fe96) with Clang 18.1.8 for Windows AMD64
cmn  common_param: common_params_print_info: verbosity = 4 (adjust with the `-lv N` CLI arg)
cmn  common_param: device_info:
cmn  common_param:   - CPU     : 12th Gen Intel(R) Core(TM) i7-12700K (32489 MiB, 3994 MiB free)
cmn  common_param:   - CUDA0   : NVIDIA GeForce RTX 3080 (10239 MiB, 9095 MiB free)
cmn  common_param: system_info: n_threads = 10 (n_threads_batch = 10) / 20 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX_VNNI = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | BMI2 = 1 | LLAMAFILE = 1 | OPENMP = 1 | REPACK = 1 | CUDA : ARCHS = 750,800,860,890,1000,1200 | USE_GRAPHS = 1 | FA_QUANTS = q4_0-q4_0,q8_0-q8_0,f16-f16,bf16-bf16 | 
srv  init_listene: The UI is disabled
srv  init_listene: Use --ui/--no-ui (or deprecated --webui/--no-webui) to enable/disable
srv          init: using 19 threads for HTTP server
srv  llama_server: security: no API key is set and CORS allows all origins (see https://github.com/ggml-org/llama.cpp/pull/25655)
srv    load_model: loading model 'C:\Users\jerom\.ollama\models\blobs\sha256-1194192cf2a187eb02722edcc3f77b11d21f537048ce04b67ccf8ba78863006a'
srv    load_model: local path 'C:\Users\jerom\.ollama\models\blobs\sha256-1194192cf2a187eb02722edcc3f77b11d21f537048ce04b67ccf8ba78863006a'
cmn  common_init_: fitting params to device memory ...
cmn  common_init_: (for bugs during this step try to reproduce them with -fit off, or provide --verbose logs if the bug only occurs with -fit on)
common_params_fit_impl: getting device memory data for initial parameters:
time=2026-10-05T09:32:50.208+02:00 level=INFO source=llama_server.go:1362 msg="waiting for llama-server to become available" status="llm server loading model"
common_memory_breakdown_print: | memory breakdown [MiB] | total   free     self   model   context   compute    unaccounted |
common_memory_breakdown_print: |   - CUDA0 (RTX 3080)   | 10239 = 9095 + (19160 = 17524 +    1536 +     100) +      -18015 |
common_memory_breakdown_print: |   - Host               |                   190 =   166 +       0 +      24                |
common_params_fit_impl: projected to use 19160 MiB of device memory vs. 9095 MiB of free device memory
common_params_fit_impl: cannot meet free memory target of 1024 MiB, need to reduce device memory by 11089 MiB
common_params_fit_impl: context size set by user to 16384 -> no change
common_params_fit_impl: getting device memory data with all MoE tensors moved to system memory:
common_memory_breakdown_print: | memory breakdown [MiB] | total   free     self   model   context   compute    unaccounted |
common_memory_breakdown_print: |   - CUDA0 (RTX 3080)   | 10239 = 9095 + ( 2553 =   784 +    1536 +     233) +       -1409 |
common_memory_breakdown_print: |   - Host               |                 16930 = 16906 +       0 +      24                |
common_params_fit_impl: with only dense weights in device memory there is a total surplus of 5517 MiB
common_params_fit_impl: id=0, target=8071 MiB
common_memory_breakdown_print: | memory breakdown [MiB] | total   free     self   model   context   compute    unaccounted |
common_memory_breakdown_print: |   - CUDA0 (RTX 3080)   | 10239 = 9095 + (  266 =     0 +       0 +     266) +         878 |
common_memory_breakdown_print: |   - Host               |                 19251 = 17691 +    1536 +      24                |
common_params_fit_impl: memory for test allocation by device:
common_params_fit_impl: id=0, n_layer= 0, n_part= 0, overflow_type=4, mem=   266 MiB
common_params_fit_impl: filling dense-only layers back-to-front:
common_memory_breakdown_print: | memory breakdown [MiB] | total   free     self   model   context   compute    unaccounted |
common_memory_breakdown_print: |   - CUDA0 (RTX 3080)   | 10239 = 9095 + ( 2920 =  1157 +    1536 +     226) +       -1775 |
common_memory_breakdown_print: |   - Host               |                 16557 = 16533 +       0 +      24                |
common_params_fit_impl: memory for test allocation by device:
common_params_fit_impl: id=0, n_layer=49, n_part=48, overflow_type=4, mem=  2920 MiB
common_params_fit_impl: set ngl_per_device[0].n_layer=49
common_params_fit_impl:   - CUDA0 (NVIDIA GeForce RTX 3080): 49 layers,   2920 MiB used,   6174 MiB free
common_params_fit_impl: converting dense-only layers to full layers and filling them front-to-back with overflow to next device/system memory:
common_memory_breakdown_print: | memory breakdown [MiB] | total   free     self   model   context   compute    unaccounted |
common_memory_breakdown_print: |   - CUDA0 (RTX 3080)   | 10239 = 9095 + (19160 = 17524 +    1536 +     100) +      -18015 |
common_memory_breakdown_print: |   - Host               |                   190 =   166 +       0 +      24                |
common_params_fit_impl: memory for test allocation by device:
common_params_fit_impl: id=0, n_layer=49, n_part= 0, overflow_type=4, mem= 19160 MiB
common_memory_breakdown_print: | memory breakdown [MiB] | total   free     self   model   context   compute    unaccounted |
common_memory_breakdown_print: |   - CUDA0 (RTX 3080)   | 10239 = 9095 + ( 8176 =  6413 +    1536 +     226) +       -7031 |
common_memory_breakdown_print: |   - Host               |                 11301 = 11277 +       0 +      24                |
common_params_fit_impl: memory for test allocation by device:
common_params_fit_impl: id=0, n_layer=49, n_part=33, overflow_type=4, mem=  8176 MiB
common_params_fit_impl: set ngl_per_device_high[0].(n_layer, n_part)=(49, 33), id_dense_start_high=0
common_memory_breakdown_print: | memory breakdown [MiB] | total   free     self   model   context   compute    unaccounted |
common_memory_breakdown_print: |   - CUDA0 (RTX 3080)   | 10239 = 9095 + ( 7852 =  6089 +    1536 +     226) +       -6707 |
common_memory_breakdown_print: |   - Host               |                 11625 = 11601 +       0 +      24                |
common_params_fit_impl: memory for test allocation by device:
common_params_fit_impl: id=0, n_layer=49, n_part=34, overflow_type=4, mem=  7852 MiB
common_params_fit_impl: set ngl_per_device[0].(n_layer, n_part)=(49, 34), id_dense_start=0
common_params_fit_impl: trying to fit one extra layer with overflow_type=LAYER_FRACTION_UP
common_memory_breakdown_print: | memory breakdown [MiB] | total   free     self   model   context   compute    unaccounted |
common_memory_breakdown_print: |   - CUDA0 (RTX 3080)   | 10239 = 9095 + ( 7959 =  6196 +    1536 +     226) +       -6814 |
common_memory_breakdown_print: |   - Host               |                 11518 = 11494 +       0 +      24                |
common_params_fit_impl: memory for test allocation by device:
common_params_fit_impl: id=0, n_layer=49, n_part=34, overflow_type=2, mem=  7959 MiB
common_params_fit_impl: set ngl_per_device[0].(n_layer, n_part, overflow_type)=(49, 34, UP), id_dense_start=0
common_params_fit_impl: trying to fit one extra layer with overflow_type=LAYER_FRACTION_GATE
common_memory_breakdown_print: | memory breakdown [MiB] | total   free     self   model   context   compute    unaccounted |
common_memory_breakdown_print: |   - CUDA0 (RTX 3080)   | 10239 = 9095 + ( 8068 =  6305 +    1536 +     226) +       -6923 |
common_memory_breakdown_print: |   - Host               |                 11409 = 11385 +       0 +      24                |
common_params_fit_impl: memory for test allocation by device:
common_params_fit_impl: id=0, n_layer=49, n_part=34, overflow_type=3, mem=  8068 MiB
common_params_fit_impl: set ngl_per_device[0].(n_layer, n_part, overflow_type)=(49, 34, GATE), id_dense_start=0
common_params_fit_impl:   - CUDA0 (NVIDIA GeForce RTX 3080): 49 layers (34 overflowing),   8068 MiB used,   1026 MiB free
common_fit_params: successfully fit params to free device memory
common_fit_params: fitting params to free memory took 4.55 seconds
llama_model_loader: loaded meta data with 35 key-value pairs and 579 tensors from C:\Users\jerom\.ollama\models\blobs\sha256-1194192cf2a187eb02722edcc3f77b11d21f537048ce04b67ccf8ba78863006a (version GGUF V3 (latest))
llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output.
llama_model_loader: - kv   0:                       general.architecture str              = qwen3moe
llama_model_loader: - kv   1:                           general.basename str              = Qwen3-Coder
llama_model_loader: - kv   2:                          general.file_type u32              = 15
llama_model_loader: - kv   3:                           general.finetune str              = Instruct
llama_model_loader: - kv   4:                            general.license str              = apache-2.0
llama_model_loader: - kv   5:                       general.license.link str              = https://huggingface.co/Qwen/Qwen3-Cod...
llama_model_loader: - kv   6:                               general.name str              = Qwen3 Coder 30B A3B Instruct
llama_model_loader: - kv   7:                    general.parameter_count u64              = 30532122624
llama_model_loader: - kv   8:               general.quantization_version u32              = 2
llama_model_loader: - kv   9:                         general.size_label str              = 30B-A3B
llama_model_loader: - kv  10:                               general.tags arr[str,1]       = ["text-generation"]
llama_model_loader: - kv  11:                               general.type str              = model
llama_model_loader: - kv  12:              qwen3moe.attention.head_count u32              = 32
llama_model_loader: - kv  13:           qwen3moe.attention.head_count_kv u32              = 4
llama_model_loader: - kv  14:              qwen3moe.attention.key_length u32              = 128
llama_model_loader: - kv  15:  qwen3moe.attention.layer_norm_rms_epsilon f32              = 0.000001
llama_model_loader: - kv  16:            qwen3moe.attention.value_length u32              = 128
llama_model_loader: - kv  17:                       qwen3moe.block_count u32              = 48
llama_model_loader: - kv  18:                    qwen3moe.context_length u32              = 262144
llama_model_loader: - kv  19:                  qwen3moe.embedding_length u32              = 2048
llama_model_loader: - kv  20:                      qwen3moe.expert_count u32              = 128
llama_model_loader: - kv  21:        qwen3moe.expert_feed_forward_length u32              = 768
llama_model_loader: - kv  22: qwen3moe.expert_shared_feed_forward_length u32              = 0
llama_model_loader: - kv  23:                 qwen3moe.expert_used_count u32              = 8
llama_model_loader: - kv  24:               qwen3moe.feed_forward_length u32              = 5472
llama_model_loader: - kv  25:                    qwen3moe.rope.freq_base f32              = 10000000.000000
llama_model_loader: - kv  26:                    tokenizer.chat_template str              = {% macro render_item_list(item_list, ...
llama_model_loader: - kv  27:               tokenizer.ggml.add_bos_token bool             = false
llama_model_loader: - kv  28:                tokenizer.ggml.eos_token_id u32              = 151645
llama_model_loader: - kv  29:                      tokenizer.ggml.merges arr[str,151387]  = ["Ġ Ġ", "ĠĠ ĠĠ", "i n", "Ġ t",...
llama_model_loader: - kv  30:                       tokenizer.ggml.model str              = gpt2
llama_model_loader: - kv  31:            tokenizer.ggml.padding_token_id u32              = 151643
llama_model_loader: - kv  32:                         tokenizer.ggml.pre str              = qwen2
llama_model_loader: - kv  33:                  tokenizer.ggml.token_type arr[i32,151936]  = [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, ...
llama_model_loader: - kv  34:                      tokenizer.ggml.tokens arr[str,151936]  = ["!", "\"", "#", "$", "%", "&", "'", ...
llama_model_loader: - type  f32:  241 tensors
llama_model_loader: - type q4_K:  289 tensors
llama_model_loader: - type q6_K:   49 tensors
print_info: file format = GGUF V3 (latest)
print_info: file type   = Q4_K - Medium
print_info: file size   = 17.28 GiB (4.86 BPW) 
llama_prepare_model_devices: using device CUDA0 (NVIDIA GeForce RTX 3080) (0000:01:00.0) - 9095 MiB free
load: 0 unused tokens
load: control-looking token: 128247 '</s>' was not control-type; this is probably a bug in the model. its type will be overridden
load: printing all EOG tokens:
load:   - 128247 ('</s>')
load:   - 151643 ('<|endoftext|>')
load:   - 151645 ('<|im_end|>')
load:   - 151662 ('<|fim_pad|>')
load:   - 151663 ('<|repo_name|>')
load:   - 151664 ('<|file_sep|>')
load: special tokens cache size = 27
load: token to piece cache size = 0.9311 MB
print_info: arch                  = qwen3moe
print_info: vocab_only            = 0
print_info: no_alloc              = 0
print_info: n_ctx_train           = 262144
print_info: n_embd_inp            = 2048
print_info: n_embd                = 2048
print_info: n_embd_out            = 2048
print_info: n_layer               = 48
print_info: n_layer_all           = 48
print_info: n_head                = 32
print_info: n_head_kv             = 4
print_info: n_rot                 = 128
print_info: n_swa                 = 0
print_info: is_swa_any            = 0
print_info: non_causal_type       = 0
print_info: n_embd_head_k         = 128
print_info: n_embd_head_v         = 128
print_info: n_gqa                 = 8
print_info: n_embd_k_gqa          = 512
print_info: n_embd_v_gqa          = 512
print_info: f_norm_eps            = 0.0e+00
print_info: f_norm_rms_eps        = 1.0e-06
print_info: f_clamp_kqv           = 0.0e+00
print_info: f_max_alibi_bias      = 0.0e+00
print_info: f_logit_scale         = 0.0e+00
print_info: f_attn_scale          = 0.0e+00
print_info: f_attn_value_scale    = 0.0000
print_info: n_ff                  = 5472
print_info: n_expert              = 128
print_info: n_expert_used         = 8
print_info: n_expert_groups       = 0
print_info: n_group_used          = 0
print_info: causal attn           = 1
print_info: pooling type          = -1
print_info: rope type             = 2
print_info: rope scaling          = linear
print_info: freq_base_train       = 10000000.0
print_info: freq_scale_train      = 1
print_info: n_ctx_orig_yarn       = 262144
print_info: rope_yarn_log_mul     = 0.0000
print_info: rope_finetuned        = unknown
print_info: model type            = 30B.A3B
print_info: model params          = 30.53 B
print_info: general.name          = Qwen3 Coder 30B A3B Instruct
print_info: n_ff_exp              = 768
print_info: vocab type            = BPE
print_info: n_vocab               = 151936
print_info: n_merges              = 151387
print_info: BOS token             = 11 ','
print_info: EOS token             = 151645 '<|im_end|>'
print_info: EOT token             = 151645 '<|im_end|>'
print_info: PAD token             = 151643 '<|endoftext|>'
print_info: LF token              = 198 'Ċ'
print_info: FIM PRE token         = 151659 '<|fim_prefix|>'
print_info: FIM SUF token         = 151661 '<|fim_suffix|>'
print_info: FIM MID token         = 151660 '<|fim_middle|>'
print_info: FIM PAD token         = 151662 '<|fim_pad|>'
print_info: FIM REP token         = 151663 '<|repo_name|>'
print_info: FIM SEP token         = 151664 '<|file_sep|>'
print_info: EOG token             = 128247 '</s>'
print_info: EOG token             = 151643 '<|endoftext|>'
print_info: EOG token             = 151645 '<|im_end|>'
print_info: EOG token             = 151662 '<|fim_pad|>'
print_info: EOG token             = 151663 '<|repo_name|>'
print_info: EOG token             = 151664 '<|file_sep|>'
print_info: max token length      = 256
load_tensors: loading model tensors, this can take a while... (load_mode = none)
load_tensors: offloading output layer to GPU
load_tensors: offloading 47 repeating layers to GPU
load_tensors: offloaded 49/49 layers to GPU
load_tensors:        CUDA0 model buffer size =  6305.92 MiB
load_tensors:    CUDA_Host model buffer size = 11385.42 MiB
[GIN] 2026/10/05 - 09:33:17 | 200 |     20.2058ms |       127.0.0.1 | GET      "/api/tags"
cmn  common_init_: added </s> logit bias = -inf
cmn  common_init_: added <|endoftext|> logit bias = -inf
cmn  common_init_: added <|im_end|> logit bias = -inf
cmn  common_init_: added <|fim_pad|> logit bias = -inf
cmn  common_init_: added <|repo_name|> logit bias = -inf
cmn  common_init_: added <|file_sep|> logit bias = -inf
llama_context: constructing llama_context
llama_context: n_seq_max             = 1
llama_context: n_ctx                 = 16384
llama_context: n_ctx_seq             = 16384
llama_context: n_batch               = 512
llama_context: n_ubatch              = 512
llama_context: causal_attn           = 1
llama_context: flash_attn            = auto
llama_context: kv_unified            = false
llama_context: freq_base             = 10000000.0
llama_context: freq_scale            = 1
llama_context: n_rs_seq              = 0
llama_context: n_outputs_max         = 1
llama_context: n_outputs_max_per_seq = 1
llama_context: n_ctx_seq (16384) < n_ctx_train (262144) -- the full capacity of the model will not be utilized
llama_context:  CUDA_Host  output buffer size =     0.58 MiB
llama_kv_cache:      CUDA0 KV buffer size =  1536.00 MiB
llama_kv_cache: size = 1536.00 MiB ( 16384 cells,  48 layers,  1/1 seqs), K (f16):  768.00 MiB, V (f16):  768.00 MiB
llama_kv_cache: attn_rot_k = 0, n_embd_head_k_all = 128
llama_kv_cache: attn_rot_v = 0, n_embd_head_k_all = 128
sched_reserve: reserving ...
resolve_fused_ops: Flash Attention enabled
resolve_fused_ops: resolving fused DeepSeek V4 HC support:
resolve_fused_ops: fused DeepSeek V4 HC pre enabled
resolve_fused_ops: fused DeepSeek V4 HC comb enabled
resolve_fused_ops: fused DeepSeek V4 HC post enabled
sched_reserve:      CUDA0 compute buffer size =   226.30 MiB
sched_reserve:  CUDA_Host compute buffer size =    24.01 MiB
sched_reserve: graph (pp bs=512, tg bs=1): nodes = 3030 / 3030, splits = 98 / 68, input objects = 4 / 4, input tensors = 6 / 6
sched_reserve: reserve took 34.92 ms, sched copies = 1
cmn          init: llama threadpool init, n_threads = 10
cmn  common_init_: warming up the model with an empty run - please wait ... (--no-warmup to disable)
srv    load_model: initializing, n_slots = 1, n_ctx_slot = 16384, kv_unified = 'false'
spec common_specu: no implementations specified for speculative decoding
slot   load_model: id  0 | task -1 | new slot, n_ctx = 16384
srv    load_model: prompt cache is enabled, size limit: 8192 MiB
srv    load_model: use `--cache-ram 0` to disable the prompt cache
srv    load_model: for more info see https://github.com/ggml-org/llama.cpp/pull/16391
srv    load_model: context checkpoints enabled, max = 32, min spacing = 8192
srv          init: idle slots will be saved to prompt cache upon starting a new task
srv          init: init: chat template, example_format: '<|im_start|>system
You are a helpful assistant<|im_end|>
<|im_start|>user
Hello<|im_end|>
<|im_start|>assistant
Hi there<|im_end|>
<|im_start|>user
How are you?<|im_end|>
<|im_start|>assistant
'
srv          init: init: chat template, thinking = 0
srv          init: preserve_reasoning kwarg: not supported by template
srv  llama_server: model loaded
srv  llama_server: listening on http://127.0.0.1:53541
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:33:48 | 200 |     69.9818ms |       127.0.0.1 | GET      "/api/tags"
time=2026-10-05T09:33:48.197+02:00 level=INFO source=llama_server.go:1374 msg="llama-server started in 58.45 seconds"
time=2026-10-05T09:33:48.197+02:00 level=INFO source=images.go:395 msg="template selection" model=registry.ollama.ai/library/qwen3-coder:30b selected=renderer_parser renderer=qwen3-coder parser=qwen3-coder go_template=null chat_template="[tools completion]" harmony=null renderer_parser="[completion tools]"
time=2026-10-05T09:33:48.210+02:00 level=INFO source=sched.go:733 msg="loaded runners" count=1
time=2026-10-05T09:33:48.212+02:00 level=INFO source=llama_server.go:1307 msg="waiting for llama-server to start responding"
time=2026-10-05T09:33:48.215+02:00 level=INFO source=llama_server.go:1374 msg="llama-server started in 58.52 seconds"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - skipping, slot is empty
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = -1
srv  get_availabl: updating prompt cache
srv          load:  - looking for better prompt, base f_keep = -1.000, f_sim = 0.000
srv        update:  - cache state: 0 prompts, 0.000 MiB (limits: 8192.000 MiB, 16384 tokens, 8589934592 est)
srv  get_availabl: prompt cache update took 0.02 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 0 | processing task, is_child = 0
slot   operator(): id  0 | task 0 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1517
slot   operator(): id  0 | task 0 | cached n_tokens = 0, memory_seq_rm [0, end)
slot print_timing: id  0 | task 0 | prompt processing, n_tokens =    512, progress = 0.34, t =  13.30 s / 38.50 tokens per second
slot   operator(): id  0 | task 0 | cached n_tokens = 512, memory_seq_rm [512, end)
slot print_timing: id  0 | task 0 | prompt processing, n_tokens =   1024, progress = 0.68, t =  24.57 s / 41.67 tokens per second
slot   operator(): id  0 | task 0 | cached n_tokens = 1024, memory_seq_rm [1024, end)
slot init_sampler: id  0 | task 0 | init sampler, took 0.75 ms, tokens: text = 1517, total = 1517
[GIN] 2026/10/05 - 09:34:18 | 200 |    222.6001ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:34:49 | 200 |      39.562ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 0 | n_gen =    100, tg =   4.09 t/s, tg_3s =   4.13 t/s
slot print_timing: id  0 | task 0 | n_gen =    129, tg =   4.70 t/s, tg_3s =   9.51 t/s
slot print_timing: id  0 | task 0 | n_gen =    151, tg =   4.82 t/s, tg_3s =   5.66 t/s
slot print_timing: id  0 | task 0 | n_gen =    169, tg =   4.63 t/s, tg_3s =   3.49 t/s
slot print_timing: id  0 | task 0 | n_gen =    184, tg =   4.48 t/s, tg_3s =   3.31 t/s
[GIN] 2026/10/05 - 09:35:19 | 200 |     31.5214ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 0 | n_gen =    187, tg =   4.15 t/s, tg_3s =   0.75 t/s
slot print_timing: id  0 | task 0 | n_gen =    208, tg =   4.32 t/s, tg_3s =   6.90 t/s
slot print_timing: id  0 | task 0 | n_gen =    259, tg =   5.06 t/s, tg_3s =  16.82 t/s
slot print_timing: id  0 | task 0 | prompt eval time =   50170.70 ms /  1517 tokens (   33.07 ms per token,    30.24 tokens per second)
slot print_timing: id  0 | task 0 |        eval time =   51182.87 ms /   263 tokens (  195.35 ms per token,     5.12 tokens per second)
slot print_timing: id  0 | task 0 |       total time =  101353.57 ms /  1780 tokens
slot print_timing: id  0 | task 0 |    graphs reused =        261
slot      release: id  0 | task 0 | stop processing: n_tokens = 1779, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:35:29 | 200 |         2m41s |       127.0.0.1 | POST     "/api/chat"
[GIN] 2026/10/05 - 09:35:49 | 200 |     19.9915ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:36:19 | 200 |     33.0771ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:36:49 | 200 |     28.0722ms |       127.0.0.1 | GET      "/api/tags"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.003 (4/1437) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 160002594
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1779, total state size = 166.803 MiB (draft: 0.000 MiB)
[GIN] 2026/10/05 - 09:37:21 | 200 |     50.1016ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:37:21 | 200 |     11.0155ms |       127.0.0.1 | POST     "/api/show"
[GIN] 2026/10/05 - 09:37:21 | 200 |     16.5395ms |       127.0.0.1 | POST     "/api/show"
[GIN] 2026/10/05 - 09:37:25 | 200 |     56.6571ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:37:25 | 200 |     27.5415ms |       127.0.0.1 | POST     "/api/show"
[GIN] 2026/10/05 - 09:37:25 | 200 |    121.6917ms |       127.0.0.1 | POST     "/api/show"
[GIN] 2026/10/05 - 09:37:55 | 200 |     28.5504ms |       127.0.0.1 | GET      "/api/tags"
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1779, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv        update:  - cache state: 1 prompts, 166.803 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53A28070:    1779 tokens, checkpoints:  0,   166.803 MiB
srv  get_availabl: prompt cache update took 53860.66 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 266 | processing task, is_child = 0
slot   operator(): id  0 | task 266 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1437
slot   operator(): id  0 | task 266 | cached n_tokens = 4, memory_seq_rm [4, end)
slot print_timing: id  0 | task 266 | prompt processing, n_tokens =    512, progress = 0.36, t =   6.03 s / 84.89 tokens per second
slot   operator(): id  0 | task 266 | cached n_tokens = 516, memory_seq_rm [516, end)
slot print_timing: id  0 | task 266 | prompt processing, n_tokens =   1024, progress = 0.72, t =  12.35 s / 82.91 tokens per second
slot   operator(): id  0 | task 266 | cached n_tokens = 1028, memory_seq_rm [1028, end)
slot init_sampler: id  0 | task 266 | init sampler, took 1.17 ms, tokens: text = 1437, total = 1437
[GIN] 2026/10/05 - 09:38:25 | 200 |     11.0999ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 266 | prompt eval time =   18132.71 ms /  1433 tokens (   12.65 ms per token,    79.03 tokens per second)
slot print_timing: id  0 | task 266 |        eval time =    2314.78 ms /    44 tokens (   53.83 ms per token,    18.58 tokens per second)
slot print_timing: id  0 | task 266 |       total time =   20447.49 ms /  1477 tokens
slot print_timing: id  0 | task 266 |    graphs reused =        303
slot      release: id  0 | task 266 | stop processing: n_tokens = 1480, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:38:26 | 200 |         1m14s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.794 (1436/1808) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.794 (> 0.100 thold), f_keep = 0.970
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 313 | processing task, is_child = 0
slot   operator(): id  0 | task 313 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1808
slot   operator(): id  0 | task 313 | cached n_tokens = 1436, memory_seq_rm [1436, end)
slot init_sampler: id  0 | task 313 | init sampler, took 0.47 ms, tokens: text = 1808, total = 1808
slot print_timing: id  0 | task 313 | n_gen =    100, tg =  17.30 t/s, tg_3s =  17.47 t/s
slot print_timing: id  0 | task 313 | n_gen =    157, tg =  17.80 t/s, tg_3s =  18.74 t/s
slot print_timing: id  0 | task 313 | n_gen =    208, tg =  17.57 t/s, tg_3s =  16.91 t/s
slot print_timing: id  0 | task 313 | n_gen =    261, tg =  17.55 t/s, tg_3s =  17.45 t/s
slot print_timing: id  0 | task 313 | n_gen =    313, tg =  17.50 t/s, tg_3s =  17.26 t/s
slot print_timing: id  0 | task 313 | n_gen =    364, tg =  17.40 t/s, tg_3s =  16.83 t/s
[GIN] 2026/10/05 - 09:38:55 | 200 |      9.1963ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 313 | n_gen =    412, tg =  17.22 t/s, tg_3s =  15.99 t/s
slot print_timing: id  0 | task 313 | n_gen =    460, tg =  17.09 t/s, tg_3s =  16.00 t/s
slot print_timing: id  0 | task 313 | n_gen =    479, tg =  15.98 t/s, tg_3s =   6.22 t/s
slot print_timing: id  0 | task 313 | n_gen =    495, tg =  14.55 t/s, tg_3s =   3.96 t/s
slot print_timing: id  0 | task 313 | n_gen =    507, tg =  13.49 t/s, tg_3s =   3.38 t/s
slot print_timing: id  0 | task 313 | n_gen =    523, tg =  12.88 t/s, tg_3s =   5.32 t/s
slot print_timing: id  0 | task 313 | prompt eval time =    5905.45 ms /   372 tokens (   15.87 ms per token,    62.99 tokens per second)
slot print_timing: id  0 | task 313 |        eval time =   40885.55 ms /   528 tokens (   77.58 ms per token,    12.89 tokens per second)
slot print_timing: id  0 | task 313 |       total time =   46791.00 ms /   900 tokens
slot print_timing: id  0 | task 313 |    graphs reused =        827
slot      release: id  0 | task 313 | stop processing: n_tokens = 2335, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:39:13 | 200 |   46.8643037s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.004 (5/1269) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 383840099
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 2335, total state size = 218.934 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.004
srv          load:    - prompt with length    1779, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    2335, lcp =       5, f_keep = 0.002, f_sim = 0.004
srv        update:  - cache state: 2 prompts, 385.737 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53A28070:    1779 tokens, checkpoints:  0,   166.803 MiB
srv        update:    - prompt 000001EC4EA3E9D0:    2335 tokens, checkpoints:  0,   218.934 MiB
srv  get_availabl: prompt cache update took 130.51 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 842 | processing task, is_child = 0
slot   operator(): id  0 | task 842 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1269
slot   operator(): id  0 | task 842 | cached n_tokens = 5, memory_seq_rm [5, end)
slot print_timing: id  0 | task 842 | prompt processing, n_tokens =    512, progress = 0.41, t =   5.49 s / 93.19 tokens per second
slot   operator(): id  0 | task 842 | cached n_tokens = 517, memory_seq_rm [517, end)
slot print_timing: id  0 | task 842 | prompt processing, n_tokens =   1024, progress = 0.81, t =  10.99 s / 93.14 tokens per second
slot   operator(): id  0 | task 842 | cached n_tokens = 1029, memory_seq_rm [1029, end)
slot init_sampler: id  0 | task 842 | init sampler, took 0.36 ms, tokens: text = 1269, total = 1269
[GIN] 2026/10/05 - 09:39:25 | 200 |     29.5958ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 842 | n_gen =    100, tg =  16.33 t/s, tg_3s =  16.50 t/s
slot print_timing: id  0 | task 842 | prompt eval time =   14760.58 ms /  1264 tokens (   11.68 ms per token,    85.63 tokens per second)
slot print_timing: id  0 | task 842 |        eval time =    8619.16 ms /   151 tokens (   57.46 ms per token,    17.40 tokens per second)
slot print_timing: id  0 | task 842 |       total time =   23379.75 ms /  1415 tokens
slot print_timing: id  0 | task 842 |    graphs reused =        975
slot      release: id  0 | task 842 | stop processing: n_tokens = 1419, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:39:37 | 200 |   23.5736524s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.003 (4/1517) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 407545944
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1419, total state size = 133.049 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    1779, lcp =    1517, f_keep = 0.853, f_sim = 1.000
srv          load:    - prompt with length    2335, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1419, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:  - found better prompt with f_keep = 0.853, f_sim = 1.000
srv        update:  - cache state: 2 prompts, 351.983 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC4EA3E9D0:    2335 tokens, checkpoints:  0,   218.934 MiB
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv  get_availabl: prompt cache update took 405.40 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 996 | processing task, is_child = 0
slot   operator(): id  0 | task 996 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1517
slot   operator(): id  0 | task 996 | need to evaluate at least 1 token for each active slot (n_past = 1517, task.n_tokens() = 1517)
slot   operator(): id  0 | task 996 | n_past was set to 1516
slot   operator(): id  0 | task 996 | cached n_tokens = 1516, memory_seq_rm [1516, end)
slot init_sampler: id  0 | task 996 | init sampler, took 0.37 ms, tokens: text = 1517, total = 1517
slot print_timing: id  0 | task 996 | n_gen =    100, tg =  14.03 t/s, tg_3s =  14.17 t/s
slot print_timing: id  0 | task 996 | n_gen =    143, tg =  14.06 t/s, tg_3s =  14.12 t/s
slot print_timing: id  0 | task 996 | n_gen =    185, tg =  14.02 t/s, tg_3s =  13.90 t/s
slot print_timing: id  0 | task 996 | n_gen =    233, tg =  14.38 t/s, tg_3s =  15.96 t/s
[GIN] 2026/10/05 - 09:39:55 | 200 |      6.5147ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 996 | prompt eval time =      61.01 ms /     1 tokens (   61.01 ms per token,    16.39 tokens per second)
slot print_timing: id  0 | task 996 |        eval time =   18035.34 ms /   268 tokens (   67.55 ms per token,    14.80 tokens per second)
slot print_timing: id  0 | task 996 |       total time =   18096.35 ms /   269 tokens
slot print_timing: id  0 | task 996 |    graphs reused =       1242
slot      release: id  0 | task 996 | stop processing: n_tokens = 1784, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:39:56 | 200 |   18.5590018s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 1.000 (1517/1517) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LCP similarity, f_sim_best = 1.000 (> 0.100 thold), f_keep = 0.850
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 1265 | processing task, is_child = 0
slot   operator(): id  0 | task 1265 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1517
slot   operator(): id  0 | task 1265 | need to evaluate at least 1 token for each active slot (n_past = 1517, task.n_tokens() = 1517)
slot   operator(): id  0 | task 1265 | n_past was set to 1516
slot   operator(): id  0 | task 1265 | cached n_tokens = 1516, memory_seq_rm [1516, end)
slot init_sampler: id  0 | task 1265 | init sampler, took 0.34 ms, tokens: text = 1517, total = 1517
slot print_timing: id  0 | task 1265 | n_gen =    100, tg =   7.46 t/s, tg_3s =   7.54 t/s
slot print_timing: id  0 | task 1265 | n_gen =    123, tg =   6.88 t/s, tg_3s =   5.15 t/s
slot print_timing: id  0 | task 1265 | n_gen =    142, tg =   6.78 t/s, tg_3s =   6.17 t/s
slot print_timing: id  0 | task 1265 | n_gen =    163, tg =   6.80 t/s, tg_3s =   6.97 t/s
slot print_timing: id  0 | task 1265 | n_gen =    187, tg =   6.63 t/s, tg_3s =   5.67 t/s
[GIN] 2026/10/05 - 09:40:25 | 200 |     12.0154ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 1265 | n_gen =    210, tg =   6.72 t/s, tg_3s =   7.62 t/s
slot print_timing: id  0 | task 1265 | n_gen =    236, tg =   6.63 t/s, tg_3s =   5.99 t/s
slot print_timing: id  0 | task 1265 | n_gen =    254, tg =   6.58 t/s, tg_3s =   5.94 t/s
slot print_timing: id  0 | task 1265 | prompt eval time =      57.91 ms /     1 tokens (   57.91 ms per token,    17.27 tokens per second)
slot print_timing: id  0 | task 1265 |        eval time =   40852.28 ms /   290 tokens (  141.36 ms per token,     7.07 tokens per second)
slot print_timing: id  0 | task 1265 |       total time =   40910.18 ms /   291 tokens
slot print_timing: id  0 | task 1265 |    graphs reused =       1529
slot      release: id  0 | task 1265 | stop processing: n_tokens = 1806, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:40:37 | 200 |   40.9457201s |       127.0.0.1 | POST     "/api/chat"
[GIN] 2026/10/05 - 09:40:55 | 200 |      6.0012ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:41:25 | 200 |      8.0111ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:41:56 | 200 |    267.7074ms |       127.0.0.1 | GET      "/api/tags"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.003 (4/1547) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 468178308
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1806, total state size = 169.334 MiB (draft: 0.000 MiB)
[GIN] 2026/10/05 - 09:42:26 | 200 |    140.0457ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:42:56 | 200 |     52.2605ms |       127.0.0.1 | GET      "/api/tags"
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    2335, lcp =     628, f_keep = 0.269, f_sim = 0.406
srv          load:    - prompt with length    1419, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    1806, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:  - found better prompt with f_keep = 0.269, f_sim = 0.406
srv        update:  - cache state: 2 prompts, 302.383 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EC53D699B0:    1806 tokens, checkpoints:  0,   169.334 MiB
srv  get_availabl: prompt cache update took 59636.18 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 1556 | processing task, is_child = 0
slot   operator(): id  0 | task 1556 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1547
slot   operator(): id  0 | task 1556 | cached n_tokens = 628, memory_seq_rm [628, end)
[GIN] 2026/10/05 - 09:43:26 | 200 |      39.687ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 1556 | prompt processing, n_tokens =    512, progress = 0.74, t =   8.49 s / 60.28 tokens per second
slot   operator(): id  0 | task 1556 | cached n_tokens = 1140, memory_seq_rm [1140, end)
slot init_sampler: id  0 | task 1556 | init sampler, took 1.18 ms, tokens: text = 1547, total = 1547
[GIN] 2026/10/05 - 09:43:56 | 200 |     37.1222ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 1556 | prompt eval time =   34002.53 ms /   919 tokens (   37.00 ms per token,    27.03 tokens per second)
slot print_timing: id  0 | task 1556 |        eval time =    2809.15 ms /    44 tokens (   65.33 ms per token,    15.31 tokens per second)
slot print_timing: id  0 | task 1556 |       total time =   36811.68 ms /   963 tokens
slot print_timing: id  0 | task 1556 |    graphs reused =       1571
slot      release: id  0 | task 1556 | stop processing: n_tokens = 1590, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:43:58 | 200 |         1m37s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.762 (1546/2028) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.762 (> 0.100 thold), f_keep = 0.972
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 1602 | processing task, is_child = 0
slot   operator(): id  0 | task 1602 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 2028
slot   operator(): id  0 | task 1602 | cached n_tokens = 1546, memory_seq_rm [1546, end)
slot init_sampler: id  0 | task 1602 | init sampler, took 0.62 ms, tokens: text = 2028, total = 2028
slot print_timing: id  0 | task 1602 | n_gen =    100, tg =  16.36 t/s, tg_3s =  16.52 t/s
slot print_timing: id  0 | task 1602 | n_gen =    149, tg =  16.32 t/s, tg_3s =  16.26 t/s
slot print_timing: id  0 | task 1602 | n_gen =    196, tg =  16.10 t/s, tg_3s =  15.43 t/s
slot print_timing: id  0 | task 1602 | n_gen =    244, tg =  16.06 t/s, tg_3s =  15.91 t/s
slot print_timing: id  0 | task 1602 | n_gen =    296, tg =  16.27 t/s, tg_3s =  17.33 t/s
slot print_timing: id  0 | task 1602 | n_gen =    344, tg =  16.21 t/s, tg_3s =  15.83 t/s
slot print_timing: id  0 | task 1602 | n_gen =    381, tg =  15.69 t/s, tg_3s =  12.13 t/s
[GIN] 2026/10/05 - 09:44:27 | 200 |     299.011ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 1602 | n_gen =    417, tg =  15.25 t/s, tg_3s =  11.76 t/s
slot print_timing: id  0 | task 1602 | n_gen =    454, tg =  14.96 t/s, tg_3s =  12.30 t/s
slot print_timing: id  0 | task 1602 | n_gen =    503, tg =  15.08 t/s, tg_3s =  16.28 t/s
slot print_timing: id  0 | task 1602 | n_gen =    542, tg =  14.90 t/s, tg_3s =  12.98 t/s
slot print_timing: id  0 | task 1602 | prompt eval time =    4274.11 ms /   482 tokens (    8.87 ms per token,   112.77 tokens per second)
slot print_timing: id  0 | task 1602 |        eval time =   37341.68 ms /   554 tokens (   67.53 ms per token,    14.81 tokens per second)
slot print_timing: id  0 | task 1602 |       total time =   41615.79 ms /  1036 tokens
slot print_timing: id  0 | task 1602 |    graphs reused =       2120
slot      release: id  0 | task 1602 | stop processing: n_tokens = 2581, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:44:39 | 200 |   41.7163771s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.004 (5/1278) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 710158692
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 2581, total state size = 241.999 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.004
srv          load:    - prompt with length    1419, lcp =     101, f_keep = 0.071, f_sim = 0.079
srv          load:    - prompt with length    1806, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    2581, lcp =       5, f_keep = 0.002, f_sim = 0.004
srv        update:  - cache state: 3 prompts, 544.382 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EC53D699B0:    1806 tokens, checkpoints:  0,   169.334 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv  get_availabl: prompt cache update took 204.24 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 2157 | processing task, is_child = 0
slot   operator(): id  0 | task 2157 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1278
slot   operator(): id  0 | task 2157 | cached n_tokens = 5, memory_seq_rm [5, end)
slot print_timing: id  0 | task 2157 | prompt processing, n_tokens =    512, progress = 0.40, t =   3.72 s / 137.54 tokens per second
slot   operator(): id  0 | task 2157 | cached n_tokens = 517, memory_seq_rm [517, end)
slot print_timing: id  0 | task 2157 | prompt processing, n_tokens =   1024, progress = 0.81, t =   7.46 s / 137.26 tokens per second
slot   operator(): id  0 | task 2157 | cached n_tokens = 1029, memory_seq_rm [1029, end)
slot init_sampler: id  0 | task 2157 | init sampler, took 0.50 ms, tokens: text = 1278, total = 1278
slot print_timing: id  0 | task 2157 | n_gen =    100, tg =  15.13 t/s, tg_3s =  15.28 t/s
[GIN] 2026/10/05 - 09:44:57 | 200 |     13.5336ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 2157 | prompt eval time =   10461.61 ms /  1273 tokens (    8.22 ms per token,   121.68 tokens per second)
slot print_timing: id  0 | task 2157 |        eval time =    8532.70 ms /   136 tokens (   63.21 ms per token,    15.82 tokens per second)
slot print_timing: id  0 | task 2157 |       total time =   18994.31 ms /  1409 tokens
slot print_timing: id  0 | task 2157 |    graphs reused =       2253
slot      release: id  0 | task 2157 | stop processing: n_tokens = 1413, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:44:59 | 200 |   19.2595037s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.003 (4/1517) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 729529888
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1413, total state size = 132.486 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    1419, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    1806, lcp =    1517, f_keep = 0.840, f_sim = 1.000
srv          load:    - prompt with length    2581, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1413, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:  - found better prompt with f_keep = 0.840, f_sim = 1.000
srv        update:  - cache state: 3 prompts, 507.534 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv  get_availabl: prompt cache update took 853.42 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 2296 | processing task, is_child = 0
slot   operator(): id  0 | task 2296 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1517
slot   operator(): id  0 | task 2296 | need to evaluate at least 1 token for each active slot (n_past = 1517, task.n_tokens() = 1517)
slot   operator(): id  0 | task 2296 | n_past was set to 1516
slot   operator(): id  0 | task 2296 | cached n_tokens = 1516, memory_seq_rm [1516, end)
slot init_sampler: id  0 | task 2296 | init sampler, took 0.37 ms, tokens: text = 1517, total = 1517
slot print_timing: id  0 | task 2296 | n_gen =    100, tg =  17.38 t/s, tg_3s =  17.55 t/s
slot print_timing: id  0 | task 2296 | n_gen =    142, tg =  14.11 t/s, tg_3s =   9.78 t/s
slot print_timing: id  0 | task 2296 | n_gen =    178, tg =  13.63 t/s, tg_3s =  12.00 t/s
slot print_timing: id  0 | task 2296 | n_gen =    224, tg =  13.89 t/s, tg_3s =  14.99 t/s
slot print_timing: id  0 | task 2296 | n_gen =    268, tg =  13.99 t/s, tg_3s =  14.56 t/s
slot print_timing: id  0 | task 2296 | prompt eval time =     102.09 ms /     1 tokens (  102.09 ms per token,     9.80 tokens per second)
slot print_timing: id  0 | task 2296 |        eval time =   19714.97 ms /   277 tokens (   71.43 ms per token,    14.00 tokens per second)
slot print_timing: id  0 | task 2296 |       total time =   19817.06 ms /   278 tokens
slot print_timing: id  0 | task 2296 |    graphs reused =       2528
slot      release: id  0 | task 2296 | stop processing: n_tokens = 1793, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:45:21 | 200 |    20.699706s |       127.0.0.1 | POST     "/api/chat"
[GIN] 2026/10/05 - 09:45:27 | 200 |      8.0084ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:45:57 | 200 |      8.6081ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:46:27 | 200 |     18.8066ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:46:57 | 200 |      11.063ms |       127.0.0.1 | GET      "/api/tags"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.003 (4/1537) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 751312767
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1793, total state size = 168.115 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1419, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    2581, lcp =     630, f_keep = 0.244, f_sim = 0.410
srv          load:    - prompt with length    1413, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    1793, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv        update:  - cache state: 4 prompts, 675.649 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EE3017E160:    1793 tokens, checkpoints:  0,   168.115 MiB
srv  get_availabl: prompt cache update took 20145.57 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 2574 | processing task, is_child = 0
slot   operator(): id  0 | task 2574 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1537
slot   operator(): id  0 | task 2574 | cached n_tokens = 4, memory_seq_rm [4, end)
[GIN] 2026/10/05 - 09:47:27 | 200 |      9.0083ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 2574 | prompt processing, n_tokens =    512, progress = 0.34, t =  14.74 s / 34.74 tokens per second
slot   operator(): id  0 | task 2574 | cached n_tokens = 516, memory_seq_rm [516, end)
[GIN] 2026/10/05 - 09:47:57 | 200 |      9.0135ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:48:28 | 200 |    131.3441ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 2574 | prompt processing, n_tokens =   1024, progress = 0.67, t =  72.94 s / 14.04 tokens per second
slot   operator(): id  0 | task 2574 | cached n_tokens = 1028, memory_seq_rm [1028, end)
slot init_sampler: id  0 | task 2574 | init sampler, took 1.67 ms, tokens: text = 1537, total = 1537
[GIN] 2026/10/05 - 09:48:58 | 200 |     47.4019ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 2574 | prompt eval time =   85030.20 ms /  1533 tokens (   55.47 ms per token,    18.03 tokens per second)
slot print_timing: id  0 | task 2574 |        eval time =   35854.27 ms /    44 tokens (  833.82 ms per token,     1.20 tokens per second)
slot print_timing: id  0 | task 2574 |       total time =  120884.48 ms /  1577 tokens
slot print_timing: id  0 | task 2574 |    graphs reused =       2570
slot      release: id  0 | task 2574 | stop processing: n_tokens = 1580, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:49:24 | 200 |         2m21s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.761 (1536/2018) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.761 (> 0.100 thold), f_keep = 0.972
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 2621 | processing task, is_child = 0
slot   operator(): id  0 | task 2621 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 2018
slot   operator(): id  0 | task 2621 | cached n_tokens = 1536, memory_seq_rm [1536, end)
slot init_sampler: id  0 | task 2621 | init sampler, took 0.56 ms, tokens: text = 2018, total = 2018
[GIN] 2026/10/05 - 09:49:28 | 200 |      8.0354ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:49:58 | 200 |      10.501ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 2621 | n_gen =    100, tg =  17.24 t/s, tg_3s =  17.42 t/s
slot print_timing: id  0 | task 2621 | n_gen =    151, tg =  17.11 t/s, tg_3s =  16.85 t/s
slot print_timing: id  0 | task 2621 | n_gen =    196, tg =  16.53 t/s, tg_3s =  14.87 t/s
slot print_timing: id  0 | task 2621 | n_gen =    210, tg =  14.04 t/s, tg_3s =   4.52 t/s
slot print_timing: id  0 | task 2621 | n_gen =    226, tg =  12.54 t/s, tg_3s =   5.23 t/s
slot print_timing: id  0 | task 2621 | n_gen =    233, tg =   8.24 t/s, tg_3s =   0.68 t/s
slot print_timing: id  0 | task 2621 | n_gen =    244, tg =   7.64 t/s, tg_3s =   3.04 t/s
[GIN] 2026/10/05 - 09:50:28 | 200 |          14ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 2621 | n_gen =    259, tg =   7.41 t/s, tg_3s =   4.97 t/s
slot print_timing: id  0 | task 2621 | n_gen =    294, tg =   7.73 t/s, tg_3s =  11.37 t/s
slot print_timing: id  0 | task 2621 | n_gen =    331, tg =   8.06 t/s, tg_3s =  12.17 t/s
slot print_timing: id  0 | task 2621 | n_gen =    360, tg =   8.16 t/s, tg_3s =   9.50 t/s
slot print_timing: id  0 | task 2621 | n_gen =    399, tg =   8.47 t/s, tg_3s =  12.91 t/s
slot print_timing: id  0 | task 2621 | n_gen =    444, tg =   8.85 t/s, tg_3s =  14.76 t/s
slot print_timing: id  0 | task 2621 | n_gen =    484, tg =   9.10 t/s, tg_3s =  13.13 t/s
slot print_timing: id  0 | task 2621 | prompt eval time =   31685.88 ms /   482 tokens (   65.74 ms per token,    15.21 tokens per second)
slot print_timing: id  0 | task 2621 |        eval time =   53892.46 ms /   494 tokens (  109.32 ms per token,     9.15 tokens per second)
slot print_timing: id  0 | task 2621 |       total time =   85578.34 ms /   976 tokens
slot print_timing: id  0 | task 2621 |    graphs reused =       3060
slot      release: id  0 | task 2621 | stop processing: n_tokens = 2511, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:50:50 | 200 |         1m25s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.004 (5/1158) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 1080600907
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 2511, total state size = 235.436 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.004
srv          load:    - prompt with length    1419, lcp =     101, f_keep = 0.071, f_sim = 0.087
srv          load:    - prompt with length    2581, lcp =       5, f_keep = 0.002, f_sim = 0.004
srv          load:    - prompt with length    1413, lcp =     102, f_keep = 0.072, f_sim = 0.088
srv          load:    - prompt with length    1793, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    2511, lcp =       5, f_keep = 0.002, f_sim = 0.004
srv        update:  - cache state: 5 prompts, 911.086 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EE3017E160:    1793 tokens, checkpoints:  0,   168.115 MiB
srv        update:    - prompt 000001EC53D69B90:    2511 tokens, checkpoints:  0,   235.436 MiB
srv  get_availabl: prompt cache update took 773.64 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 3116 | processing task, is_child = 0
slot   operator(): id  0 | task 3116 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1158
slot   operator(): id  0 | task 3116 | cached n_tokens = 5, memory_seq_rm [5, end)
slot print_timing: id  0 | task 3116 | prompt processing, n_tokens =    512, progress = 0.45, t =   3.90 s / 131.13 tokens per second
slot   operator(): id  0 | task 3116 | cached n_tokens = 517, memory_seq_rm [517, end)
[GIN] 2026/10/05 - 09:50:58 | 200 |      9.0734ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 3116 | prompt processing, n_tokens =   1024, progress = 0.89, t =   8.57 s / 119.47 tokens per second
slot   operator(): id  0 | task 3116 | cached n_tokens = 1029, memory_seq_rm [1029, end)
slot init_sampler: id  0 | task 3116 | init sampler, took 0.76 ms, tokens: text = 1158, total = 1158
slot print_timing: id  0 | task 3116 | n_gen =    100, tg =  16.25 t/s, tg_3s =  16.42 t/s
slot print_timing: id  0 | task 3116 | n_gen =    137, tg =  14.88 t/s, tg_3s =  12.14 t/s
slot print_timing: id  0 | task 3116 | prompt eval time =   11264.72 ms /  1153 tokens (    9.77 ms per token,   102.35 tokens per second)
slot print_timing: id  0 | task 3116 |        eval time =    9972.38 ms /   147 tokens (   68.30 ms per token,    14.64 tokens per second)
slot print_timing: id  0 | task 3116 |       total time =   21237.10 ms /  1300 tokens
slot print_timing: id  0 | task 3116 |    graphs reused =       3204
slot      release: id  0 | task 3116 | stop processing: n_tokens = 1304, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:51:14 | 200 |    23.514128s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.003 (4/1517) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 1104371779
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1304, total state size = 122.266 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    1419, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    2581, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1413, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    1793, lcp =    1517, f_keep = 0.846, f_sim = 1.000
srv          load:    - prompt with length    2511, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1304, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:  - found better prompt with f_keep = 0.846, f_sim = 1.000
srv        update:  - cache state: 5 prompts, 865.236 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EC53D69B90:    2511 tokens, checkpoints:  0,   235.436 MiB
srv        update:    - prompt 000001EE40A30850:    1304 tokens, checkpoints:  0,   122.266 MiB
srv  get_availabl: prompt cache update took 1069.14 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 3266 | processing task, is_child = 0
slot   operator(): id  0 | task 3266 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1517
slot   operator(): id  0 | task 3266 | need to evaluate at least 1 token for each active slot (n_past = 1517, task.n_tokens() = 1517)
slot   operator(): id  0 | task 3266 | n_past was set to 1516
slot   operator(): id  0 | task 3266 | cached n_tokens = 1516, memory_seq_rm [1516, end)
slot init_sampler: id  0 | task 3266 | init sampler, took 0.34 ms, tokens: text = 1517, total = 1517
slot print_timing: id  0 | task 3266 | n_gen =    100, tg =   9.52 t/s, tg_3s =   9.61 t/s
[GIN] 2026/10/05 - 09:51:28 | 200 |     15.4995ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 3266 | n_gen =    142, tg =  10.50 t/s, tg_3s =  13.86 t/s
slot print_timing: id  0 | task 3266 | n_gen =    186, tg =  11.20 t/s, tg_3s =  14.28 t/s
slot print_timing: id  0 | task 3266 | n_gen =    194, tg =   7.39 t/s, tg_3s =   0.83 t/s
slot print_timing: id  0 | task 3266 | n_gen =    222, tg =   7.58 t/s, tg_3s =   9.28 t/s
slot print_timing: id  0 | task 3266 | n_gen =    243, tg =   7.12 t/s, tg_3s =   4.31 t/s
slot print_timing: id  0 | task 3266 | n_gen =    251, tg =   6.62 t/s, tg_3s =   2.14 t/s
slot print_timing: id  0 | task 3266 | n_gen =    259, tg =   6.30 t/s, tg_3s =   2.49 t/s
[GIN] 2026/10/05 - 09:52:00 | 200 |   45.5228062s |       127.0.0.1 | POST     "/api/chat"
slot print_timing: id  0 | task 3266 | prompt eval time =     210.85 ms /     1 tokens (  210.85 ms per token,     4.74 tokens per second)
slot print_timing: id  0 | task 3266 |        eval time =   44033.25 ms /   262 tokens (  168.71 ms per token,     5.93 tokens per second)
slot print_timing: id  0 | task 3266 |       total time =   44244.10 ms /   263 tokens
slot print_timing: id  0 | task 3266 |    graphs reused =       3465
slot      release: id  0 | task 3266 | stop processing: n_tokens = 1778, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:52:01 | 200 |    2.3444969s |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:52:31 | 200 |      7.5011ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:53:01 | 200 |     17.5714ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:53:31 | 200 |    234.5008ms |       127.0.0.1 | GET      "/api/tags"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.003 (4/1525) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 1150904984
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1778, total state size = 166.709 MiB (draft: 0.000 MiB)
[GIN] 2026/10/05 - 09:54:01 | 200 |     23.0299ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:54:31 | 200 |      7.2632ms |       127.0.0.1 | GET      "/api/tags"
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1419, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    2581, lcp =     628, f_keep = 0.243, f_sim = 0.412
srv          load:    - prompt with length    1413, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    2511, lcp =     628, f_keep = 0.250, f_sim = 0.412
srv          load:    - prompt with length    1304, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    1778, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:  - found better prompt with f_keep = 0.250, f_sim = 0.412
srv        update:  - cache state: 5 prompts, 796.509 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EE40A30850:    1304 tokens, checkpoints:  0,   122.266 MiB
srv        update:    - prompt 000001EE40A30CB0:    1778 tokens, checkpoints:  0,   166.709 MiB
srv  get_availabl: prompt cache update took 51743.00 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 3529 | processing task, is_child = 0
slot   operator(): id  0 | task 3529 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1525
slot   operator(): id  0 | task 3529 | cached n_tokens = 628, memory_seq_rm [628, end)
slot print_timing: id  0 | task 3529 | prompt processing, n_tokens =    512, progress = 0.75, t =  13.58 s / 37.71 tokens per second
slot   operator(): id  0 | task 3529 | cached n_tokens = 1140, memory_seq_rm [1140, end)
slot init_sampler: id  0 | task 3529 | init sampler, took 0.99 ms, tokens: text = 1525, total = 1525
slot print_timing: id  0 | task 3529 | prompt eval time =   17054.73 ms /   897 tokens (   19.01 ms per token,    52.60 tokens per second)
slot print_timing: id  0 | task 3529 |        eval time =    2051.74 ms /    44 tokens (   47.71 ms per token,    20.96 tokens per second)
slot print_timing: id  0 | task 3529 |       total time =   19106.47 ms /   941 tokens
slot print_timing: id  0 | task 3529 |    graphs reused =       3506
slot      release: id  0 | task 3529 | stop processing: n_tokens = 1568, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:54:54 | 200 |         1m11s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.760 (1524/2006) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.760 (> 0.100 thold), f_keep = 0.972
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 3575 | processing task, is_child = 0
slot   operator(): id  0 | task 3575 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 2006
slot   operator(): id  0 | task 3575 | cached n_tokens = 1524, memory_seq_rm [1524, end)
slot init_sampler: id  0 | task 3575 | init sampler, took 0.63 ms, tokens: text = 2006, total = 2006
[GIN] 2026/10/05 - 09:55:01 | 200 |     42.7036ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 3575 | n_gen =    100, tg =  22.11 t/s, tg_3s =  22.33 t/s
slot print_timing: id  0 | task 3575 | n_gen =    161, tg =  21.35 t/s, tg_3s =  20.22 t/s
slot print_timing: id  0 | task 3575 | n_gen =    217, tg =  20.50 t/s, tg_3s =  18.41 t/s
slot print_timing: id  0 | task 3575 | n_gen =    271, tg =  19.90 t/s, tg_3s =  17.84 t/s
slot print_timing: id  0 | task 3575 | n_gen =    312, tg =  18.18 t/s, tg_3s =  11.58 t/s
slot print_timing: id  0 | task 3575 | n_gen =    323, tg =  15.99 t/s, tg_3s =   3.63 t/s
slot print_timing: id  0 | task 3575 | n_gen =    338, tg =  14.52 t/s, tg_3s =   4.87 t/s
slot print_timing: id  0 | task 3575 | n_gen =    352, tg =  13.03 t/s, tg_3s =   3.76 t/s
slot print_timing: id  0 | task 3575 | n_gen =    368, tg =  11.59 t/s, tg_3s =   3.39 t/s
[GIN] 2026/10/05 - 09:55:31 | 200 |      6.0077ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 3575 | n_gen =    386, tg =  11.09 t/s, tg_3s =   5.92 t/s
slot print_timing: id  0 | task 3575 | n_gen =    402, tg =  10.32 t/s, tg_3s =   3.86 t/s
slot print_timing: id  0 | task 3575 | n_gen =    418, tg =   9.76 t/s, tg_3s =   4.14 t/s
slot print_timing: id  0 | task 3575 | n_gen =    433, tg =   9.45 t/s, tg_3s =   4.96 t/s
slot print_timing: id  0 | task 3575 | n_gen =    456, tg =   9.33 t/s, tg_3s =   7.61 t/s
slot print_timing: id  0 | task 3575 | prompt eval time =    3821.68 ms /   482 tokens (    7.93 ms per token,   126.12 tokens per second)
slot print_timing: id  0 | task 3575 |        eval time =   50357.05 ms /   482 tokens (  104.69 ms per token,     9.55 tokens per second)
slot print_timing: id  0 | task 3575 |       total time =   54178.73 ms /   964 tokens
slot print_timing: id  0 | task 3575 |    graphs reused =       3984
slot      release: id  0 | task 3575 | stop processing: n_tokens = 2487, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:55:48 | 200 |   54.2285418s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.004 (5/1191) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 1379033858
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 2487, total state size = 233.186 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.004
srv          load:    - prompt with length    1419, lcp =     101, f_keep = 0.071, f_sim = 0.085
srv          load:    - prompt with length    2581, lcp =       5, f_keep = 0.002, f_sim = 0.004
srv          load:    - prompt with length    1413, lcp =     102, f_keep = 0.072, f_sim = 0.086
srv          load:    - prompt with length    1304, lcp =     102, f_keep = 0.078, f_sim = 0.086
srv          load:    - prompt with length    1778, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    2487, lcp =       5, f_keep = 0.002, f_sim = 0.004
srv        update:  - cache state: 6 prompts, 1029.695 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EE40A30850:    1304 tokens, checkpoints:  0,   122.266 MiB
srv        update:    - prompt 000001EE40A30CB0:    1778 tokens, checkpoints:  0,   166.709 MiB
srv        update:    - prompt 000001EE34181330:    2487 tokens, checkpoints:  0,   233.186 MiB
srv  get_availabl: prompt cache update took 181.40 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 4058 | processing task, is_child = 0
slot   operator(): id  0 | task 4058 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1191
slot   operator(): id  0 | task 4058 | cached n_tokens = 5, memory_seq_rm [5, end)
[GIN] 2026/10/05 - 09:56:01 | 200 |     65.7136ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:56:31 | 200 |     64.9076ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 4058 | prompt processing, n_tokens =    512, progress = 0.43, t =  49.79 s / 10.28 tokens per second
slot   operator(): id  0 | task 4058 | cached n_tokens = 517, memory_seq_rm [517, end)
slot print_timing: id  0 | task 4058 | prompt processing, n_tokens =   1024, progress = 0.86, t =  57.72 s / 17.74 tokens per second
slot   operator(): id  0 | task 4058 | cached n_tokens = 1029, memory_seq_rm [1029, end)
slot init_sampler: id  0 | task 4058 | init sampler, took 0.92 ms, tokens: text = 1191, total = 1191
slot print_timing: id  0 | task 4058 | n_gen =    100, tg =  21.33 t/s, tg_3s =  21.55 t/s
slot print_timing: id  0 | task 4058 | prompt eval time =   60668.86 ms /  1186 tokens (   51.15 ms per token,    19.55 tokens per second)
slot print_timing: id  0 | task 4058 |        eval time =    5124.01 ms /   110 tokens (   47.01 ms per token,    21.27 tokens per second)
slot print_timing: id  0 | task 4058 |       total time =   65792.87 ms /  1296 tokens
slot print_timing: id  0 | task 4058 |    graphs reused =       4091
slot      release: id  0 | task 4058 | stop processing: n_tokens = 1300, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:56:54 | 200 |          1m6s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.003 (4/1517) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 1445164918
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1300, total state size = 121.891 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    1419, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    2581, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1413, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    1304, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:    - prompt with length    1778, lcp =    1517, f_keep = 0.853, f_sim = 1.000
srv          load:    - prompt with length    2487, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1300, lcp =       4, f_keep = 0.003, f_sim = 0.003
srv          load:  - found better prompt with f_keep = 0.853, f_sim = 1.000
srv        update:  - cache state: 6 prompts, 984.877 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EE40A30850:    1304 tokens, checkpoints:  0,   122.266 MiB
srv        update:    - prompt 000001EE34181330:    2487 tokens, checkpoints:  0,   233.186 MiB
srv        update:    - prompt 000001EE34180250:    1300 tokens, checkpoints:  0,   121.891 MiB
srv  get_availabl: prompt cache update took 1020.63 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 4171 | processing task, is_child = 0
slot   operator(): id  0 | task 4171 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1517
slot   operator(): id  0 | task 4171 | need to evaluate at least 1 token for each active slot (n_past = 1517, task.n_tokens() = 1517)
slot   operator(): id  0 | task 4171 | n_past was set to 1516
slot   operator(): id  0 | task 4171 | cached n_tokens = 1516, memory_seq_rm [1516, end)
slot init_sampler: id  0 | task 4171 | init sampler, took 0.36 ms, tokens: text = 1517, total = 1517
[GIN] 2026/10/05 - 09:57:01 | 200 |     20.5224ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 4171 | n_gen =    100, tg =  17.11 t/s, tg_3s =  17.28 t/s
slot print_timing: id  0 | task 4171 | n_gen =    143, tg =  16.11 t/s, tg_3s =  14.21 t/s
slot print_timing: id  0 | task 4171 | n_gen =    187, tg =  15.67 t/s, tg_3s =  14.41 t/s
slot print_timing: id  0 | task 4171 | n_gen =    232, tg =  15.49 t/s, tg_3s =  14.78 t/s
slot print_timing: id  0 | task 4171 | prompt eval time =      60.35 ms /     1 tokens (   60.35 ms per token,    16.57 tokens per second)
slot print_timing: id  0 | task 4171 |        eval time =   17847.85 ms /   285 tokens (   62.84 ms per token,    15.91 tokens per second)
slot print_timing: id  0 | task 4171 |       total time =   17908.20 ms /   286 tokens
slot print_timing: id  0 | task 4171 |    graphs reused =       4374
slot      release: id  0 | task 4171 | stop processing: n_tokens = 1801, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:57:15 | 200 |   19.2206668s |       127.0.0.1 | POST     "/api/chat"
[GIN] 2026/10/05 - 09:57:31 | 200 |      5.4997ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:58:01 | 200 |     15.5028ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 09:58:31 | 200 |      13.936ms |       127.0.0.1 | GET      "/api/tags"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.003 (4/1554) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 1465488063
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1801, total state size = 168.865 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    1419, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    2581, lcp =     630, f_keep = 0.244, f_sim = 0.405
srv          load:    - prompt with length    1413, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    1304, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    2487, lcp =     628, f_keep = 0.253, f_sim = 0.404
srv          load:    - prompt with length    1300, lcp =       5, f_keep = 0.004, f_sim = 0.003
srv          load:    - prompt with length    1801, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:  - found better prompt with f_keep = 0.253, f_sim = 0.404
srv        update:  - cache state: 6 prompts, 920.557 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EE40A30850:    1304 tokens, checkpoints:  0,   122.266 MiB
srv        update:    - prompt 000001EE34180250:    1300 tokens, checkpoints:  0,   121.891 MiB
srv        update:    - prompt 000001EC4E9D5DA0:    1801 tokens, checkpoints:  0,   168.865 MiB
srv  get_availabl: prompt cache update took 1105.38 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 4457 | processing task, is_child = 0
slot   operator(): id  0 | task 4457 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1554
slot   operator(): id  0 | task 4457 | cached n_tokens = 628, memory_seq_rm [628, end)
[GIN] 2026/10/05 - 09:59:01 | 200 |       7.979ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 4457 | prompt processing, n_tokens =    512, progress = 0.73, t =   6.22 s / 82.33 tokens per second
slot   operator(): id  0 | task 4457 | cached n_tokens = 1140, memory_seq_rm [1140, end)
slot init_sampler: id  0 | task 4457 | init sampler, took 0.45 ms, tokens: text = 1554, total = 1554
slot print_timing: id  0 | task 4457 | prompt eval time =   11898.03 ms /   926 tokens (   12.85 ms per token,    77.83 tokens per second)
slot print_timing: id  0 | task 4457 |        eval time =    2474.32 ms /    44 tokens (   57.54 ms per token,    17.38 tokens per second)
slot print_timing: id  0 | task 4457 |       total time =   14372.35 ms /   970 tokens
slot print_timing: id  0 | task 4457 |    graphs reused =       4416
slot      release: id  0 | task 4457 | stop processing: n_tokens = 1597, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:59:13 | 200 |   15.5538588s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.763 (1553/2035) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.763 (> 0.100 thold), f_keep = 0.972
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 4503 | processing task, is_child = 0
slot   operator(): id  0 | task 4503 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 2035
slot   operator(): id  0 | task 4503 | cached n_tokens = 1553, memory_seq_rm [1553, end)
slot init_sampler: id  0 | task 4503 | init sampler, took 0.64 ms, tokens: text = 2035, total = 2035
slot print_timing: id  0 | task 4503 | n_gen =    100, tg =  15.94 t/s, tg_3s =  16.10 t/s
slot print_timing: id  0 | task 4503 | n_gen =    148, tg =  15.88 t/s, tg_3s =  15.75 t/s
[GIN] 2026/10/05 - 09:59:31 | 200 |     30.0614ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 4503 | n_gen =    194, tg =  15.68 t/s, tg_3s =  15.06 t/s
slot print_timing: id  0 | task 4503 | n_gen =    240, tg =  15.56 t/s, tg_3s =  15.10 t/s
slot print_timing: id  0 | task 4503 | n_gen =    274, tg =  14.85 t/s, tg_3s =  11.22 t/s
slot print_timing: id  0 | task 4503 | n_gen =    307, tg =  14.23 t/s, tg_3s =  10.58 t/s
slot print_timing: id  0 | task 4503 | n_gen =    339, tg =  13.78 t/s, tg_3s =  10.59 t/s
slot print_timing: id  0 | task 4503 | n_gen =    383, tg =  13.39 t/s, tg_3s =  11.01 t/s
slot print_timing: id  0 | task 4503 | n_gen =    423, tg =  13.36 t/s, tg_3s =  13.08 t/s
slot print_timing: id  0 | task 4503 | n_gen =    464, tg =  13.38 t/s, tg_3s =  13.58 t/s
slot print_timing: id  0 | task 4503 | n_gen =    506, tg =  13.42 t/s, tg_3s =  13.91 t/s
slot print_timing: id  0 | task 4503 | prompt eval time =    7315.34 ms /   482 tokens (   15.18 ms per token,    65.89 tokens per second)
slot print_timing: id  0 | task 4503 |        eval time =   38984.93 ms /   525 tokens (   74.40 ms per token,    13.44 tokens per second)
slot print_timing: id  0 | task 4503 |       total time =   46300.27 ms /  1007 tokens
slot print_timing: id  0 | task 4503 |    graphs reused =       4937
slot      release: id  0 | task 4503 | stop processing: n_tokens = 2559, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 09:59:59 | 200 |   46.3280122s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.004 (5/1177) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 1629665517
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 2559, total state size = 239.937 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.002, f_sim = 0.004
srv          load:    - prompt with length    1419, lcp =     101, f_keep = 0.071, f_sim = 0.086
srv          load:    - prompt with length    2581, lcp =       5, f_keep = 0.002, f_sim = 0.004
srv          load:    - prompt with length    1413, lcp =     102, f_keep = 0.072, f_sim = 0.087
srv          load:    - prompt with length    1304, lcp =     102, f_keep = 0.078, f_sim = 0.087
srv          load:    - prompt with length    1300, lcp =     102, f_keep = 0.078, f_sim = 0.087
srv          load:    - prompt with length    1801, lcp =       4, f_keep = 0.002, f_sim = 0.003
srv          load:    - prompt with length    2559, lcp =       5, f_keep = 0.002, f_sim = 0.004
srv        update:  - cache state: 7 prompts, 1160.493 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EE40A30850:    1304 tokens, checkpoints:  0,   122.266 MiB
srv        update:    - prompt 000001EE34180250:    1300 tokens, checkpoints:  0,   121.891 MiB
srv        update:    - prompt 000001EC4E9D5DA0:    1801 tokens, checkpoints:  0,   168.865 MiB
srv        update:    - prompt 000001EE34180C50:    2559 tokens, checkpoints:  0,   239.937 MiB
srv  get_availabl: prompt cache update took 289.37 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 5029 | processing task, is_child = 0
slot   operator(): id  0 | task 5029 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 1177
slot   operator(): id  0 | task 5029 | cached n_tokens = 5, memory_seq_rm [5, end)
[GIN] 2026/10/05 - 10:00:01 | 200 |      6.1657ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 5029 | prompt processing, n_tokens =    512, progress = 0.44, t =   8.95 s / 57.22 tokens per second
slot   operator(): id  0 | task 5029 | cached n_tokens = 517, memory_seq_rm [517, end)
slot print_timing: id  0 | task 5029 | prompt processing, n_tokens =   1024, progress = 0.87, t =  28.44 s / 36.00 tokens per second
slot   operator(): id  0 | task 5029 | cached n_tokens = 1029, memory_seq_rm [1029, end)
slot init_sampler: id  0 | task 5029 | init sampler, took 0.35 ms, tokens: text = 1177, total = 1177
[GIN] 2026/10/05 - 10:00:31 | 200 |      6.5075ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 5029 | n_gen =    100, tg =   8.25 t/s, tg_3s =   8.33 t/s
slot print_timing: id  0 | task 5029 | n_gen =    129, tg =   8.51 t/s, tg_3s =   9.51 t/s
slot print_timing: id  0 | task 5029 | n_gen =    140, tg =   7.52 t/s, tg_3s =   3.20 t/s
slot print_timing: id  0 | task 5029 | prompt eval time =   34607.16 ms /  1172 tokens (   29.53 ms per token,    33.87 tokens per second)
slot print_timing: id  0 | task 5029 |        eval time =   19215.31 ms /   152 tokens (  127.25 ms per token,     7.86 tokens per second)
slot print_timing: id  0 | task 5029 |       total time =   53822.47 ms /  1324 tokens
slot print_timing: id  0 | task 5029 |    graphs reused =       5086
slot      release: id  0 | task 5029 | stop processing: n_tokens = 1328, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 10:00:53 | 200 |   54.1695953s |       127.0.0.1 | POST     "/api/chat"
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.002 (5/2147) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 1683931349
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 1328, total state size = 124.516 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    1419, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    2581, lcp =      43, f_keep = 0.017, f_sim = 0.020
srv          load:    - prompt with length    1413, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    1304, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    1300, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    1801, lcp =       4, f_keep = 0.002, f_sim = 0.002
srv          load:    - prompt with length    2559, lcp =      43, f_keep = 0.017, f_sim = 0.020
srv          load:    - prompt with length    1328, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv        update:  - cache state: 8 prompts, 1285.010 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EE40A30850:    1304 tokens, checkpoints:  0,   122.266 MiB
srv        update:    - prompt 000001EE34180250:    1300 tokens, checkpoints:  0,   121.891 MiB
srv        update:    - prompt 000001EC4E9D5DA0:    1801 tokens, checkpoints:  0,   168.865 MiB
srv        update:    - prompt 000001EE34180C50:    2559 tokens, checkpoints:  0,   239.937 MiB
srv        update:    - prompt 000001EE34181330:    1328 tokens, checkpoints:  0,   124.516 MiB
srv  get_availabl: prompt cache update took 1461.98 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 5184 | processing task, is_child = 0
slot   operator(): id  0 | task 5184 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 2147
slot   operator(): id  0 | task 5184 | cached n_tokens = 5, memory_seq_rm [5, end)
[GIN] 2026/10/05 - 10:01:01 | 200 |      5.0172ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 5184 | prompt processing, n_tokens =    512, progress = 0.24, t =  13.37 s / 38.29 tokens per second
slot   operator(): id  0 | task 5184 | cached n_tokens = 517, memory_seq_rm [517, end)
slot print_timing: id  0 | task 5184 | prompt processing, n_tokens =   1024, progress = 0.48, t =  18.50 s / 55.36 tokens per second
slot   operator(): id  0 | task 5184 | cached n_tokens = 1029, memory_seq_rm [1029, end)
slot print_timing: id  0 | task 5184 | prompt processing, n_tokens =   1536, progress = 0.72, t =  30.53 s / 50.30 tokens per second
slot   operator(): id  0 | task 5184 | cached n_tokens = 1541, memory_seq_rm [1541, end)
[GIN] 2026/10/05 - 10:01:31 | 200 |      8.5734ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 5184 | prompt processing, n_tokens =   2048, progress = 0.96, t =  51.55 s / 39.73 tokens per second
slot   operator(): id  0 | task 5184 | cached n_tokens = 2053, memory_seq_rm [2053, end)
slot init_sampler: id  0 | task 5184 | init sampler, took 0.57 ms, tokens: text = 2147, total = 2147
slot print_timing: id  0 | task 5184 | prompt eval time =   59349.18 ms /  2142 tokens (   27.71 ms per token,    36.09 tokens per second)
slot print_timing: id  0 | task 5184 |        eval time =    3330.24 ms /    64 tokens (   52.86 ms per token,    18.92 tokens per second)
slot print_timing: id  0 | task 5184 |       total time =   62679.42 ms /  2206 tokens
slot print_timing: id  0 | task 5184 |    graphs reused =       5148
[GIN] 2026/10/05 - 10:01:58 | 200 |          1m4s |       127.0.0.1 | POST     "/api/chat"
slot      release: id  0 | task 5184 | stop processing: n_tokens = 2210, truncated = 0
srv  update_slots: all slots are idle
srv  server_strea: conv_id= (empty=1)
slot get_availabl: id  0 | task -1 |  - checking sim = 0.025 (61/2422) > 0.100
slot get_availabl: id  0 | task -1 | selected slot by LRU, t_last = 1748275744
srv  get_availabl: updating prompt cache
srv   prompt_save:  - saving prompt with length 2210, total state size = 207.214 MiB (draft: 0.000 MiB)
srv          load:  - looking for better prompt, base f_keep = 0.028, f_sim = 0.025
srv          load:    - prompt with length    1419, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    2581, lcp =      43, f_keep = 0.017, f_sim = 0.018
srv          load:    - prompt with length    1413, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    1304, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    1300, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    1801, lcp =       4, f_keep = 0.002, f_sim = 0.002
srv          load:    - prompt with length    2559, lcp =      43, f_keep = 0.017, f_sim = 0.018
srv          load:    - prompt with length    1328, lcp =       5, f_keep = 0.004, f_sim = 0.002
srv          load:    - prompt with length    2210, lcp =      61, f_keep = 0.028, f_sim = 0.025
srv        update:  - cache state: 9 prompts, 1492.223 MiB (limits: 8192.000 MiB, 16384 tokens, 87370 est)
srv        update:    - prompt 000001EC53D688D0:    1419 tokens, checkpoints:  0,   133.049 MiB
srv        update:    - prompt 000001EE34252960:    2581 tokens, checkpoints:  0,   241.999 MiB
srv        update:    - prompt 000001EE342530E0:    1413 tokens, checkpoints:  0,   132.486 MiB
srv        update:    - prompt 000001EE40A30850:    1304 tokens, checkpoints:  0,   122.266 MiB
srv        update:    - prompt 000001EE34180250:    1300 tokens, checkpoints:  0,   121.891 MiB
srv        update:    - prompt 000001EC4E9D5DA0:    1801 tokens, checkpoints:  0,   168.865 MiB
srv        update:    - prompt 000001EE34180C50:    2559 tokens, checkpoints:  0,   239.937 MiB
srv        update:    - prompt 000001EE34181330:    1328 tokens, checkpoints:  0,   124.516 MiB
srv        update:    - prompt 000001EC4EA3EBB0:    2210 tokens, checkpoints:  0,   207.214 MiB
srv  get_availabl: prompt cache update took 225.18 ms
slot launch_slot_: id  0 | task -1 | sampler chain: logits -> penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist 
slot launch_slot_: id  0 | task -1 | sampler params: 
	repeat_last_n = 64, repeat_penalty = 1.050, frequency_penalty = 0.000, presence_penalty = 0.000
	dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64
	top_k = 20, top_p = 0.800, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.700
	mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900
slot launch_slot_: id  0 | task 5253 | processing task, is_child = 0
slot   operator(): id  0 | task 5253 | new prompt, n_ctx_slot = 16384, n_keep = 4, task.n_tokens = 2422
slot   operator(): id  0 | task 5253 | cached n_tokens = 61, memory_seq_rm [61, end)
[GIN] 2026/10/05 - 10:02:01 | 200 |      8.4997ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 5253 | prompt processing, n_tokens =    512, progress = 0.24, t =   5.58 s / 91.79 tokens per second
slot   operator(): id  0 | task 5253 | cached n_tokens = 573, memory_seq_rm [573, end)
slot print_timing: id  0 | task 5253 | prompt processing, n_tokens =   1024, progress = 0.45, t =  14.42 s / 71.00 tokens per second
slot   operator(): id  0 | task 5253 | cached n_tokens = 1085, memory_seq_rm [1085, end)
slot print_timing: id  0 | task 5253 | prompt processing, n_tokens =   1536, progress = 0.66, t =  26.12 s / 58.81 tokens per second
slot   operator(): id  0 | task 5253 | cached n_tokens = 1597, memory_seq_rm [1597, end)
[GIN] 2026/10/05 - 10:02:32 | 200 |     83.6372ms |       127.0.0.1 | GET      "/api/tags"
slot print_timing: id  0 | task 5253 | prompt processing, n_tokens =   2048, progress = 0.87, t =  34.86 s / 58.75 tokens per second
slot   operator(): id  0 | task 5253 | cached n_tokens = 2109, memory_seq_rm [2109, end)
slot init_sampler: id  0 | task 5253 | init sampler, took 0.64 ms, tokens: text = 2422, total = 2422
slot print_timing: id  0 | task 5253 | n_gen =    100, tg =  10.40 t/s, tg_3s =  10.50 t/s
[GIN] 2026/10/05 - 10:02:51 | 200 |   53.6479506s |       127.0.0.1 | POST     "/api/chat"
srv          stop: cancel task, id_task = 5253
slot      release: id  0 | task 5253 | stop processing: n_tokens = 2546, truncated = 0
srv  update_slots: all slots are idle
[GIN] 2026/10/05 - 10:03:03 | 200 |    889.3398ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:03:33 | 200 |     11.0292ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:04:03 | 200 |      5.4677ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:04:33 | 200 |      7.5866ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:05:03 | 200 |      4.9995ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:05:34 | 200 |    713.5056ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:06:04 | 200 |    114.6115ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:06:34 | 200 |      3.5356ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:07:04 | 200 |     12.0147ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:07:34 | 200 |    148.4639ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:08:04 | 200 |      7.0025ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:08:34 | 200 |      8.5083ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:09:04 | 200 |      3.0002ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:09:34 | 200 |      5.9939ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:10:04 | 200 |      3.4991ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:10:34 | 200 |      5.0073ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:11:04 | 200 |      3.5005ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:11:34 | 200 |     12.5355ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:12:04 | 200 |      3.5026ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:12:34 | 200 |      3.0064ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:13:04 | 200 |      4.0115ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:13:34 | 200 |       4.415ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:14:04 | 200 |      2.9989ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:14:34 | 200 |      6.0119ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:15:04 | 200 |      4.9941ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:15:35 | 200 |    178.0968ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:16:05 | 200 |     10.4766ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:16:35 | 200 |      3.5313ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:17:05 | 200 |      3.0129ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:17:35 | 200 |      4.8663ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:18:05 | 200 |      7.5074ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:18:35 | 200 |      4.0181ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:19:05 | 200 |         3.5ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:19:35 | 200 |      4.5196ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:20:05 | 200 |     45.5421ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:20:35 | 200 |      42.045ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:21:05 | 200 |      3.5934ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:21:35 | 200 |     10.0207ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:22:05 | 200 |      9.9995ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:22:35 | 200 |    210.6627ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:23:05 | 200 |      17.508ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:23:35 | 200 |      3.9999ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:24:05 | 200 |     12.0216ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:24:36 | 200 |     38.0291ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:25:06 | 200 |      4.0077ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:25:36 | 200 |       3.612ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:26:06 | 200 |      2.9999ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:26:36 | 200 |      4.5031ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:27:06 | 200 |      5.5005ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:27:36 | 200 |      4.0083ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:28:06 | 200 |      3.9985ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:28:36 | 200 |      4.4996ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:29:06 | 200 |      3.5083ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:29:36 | 200 |       6.041ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:30:06 | 200 |      4.0112ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:30:36 | 200 |       3.393ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:31:06 | 200 |      5.0937ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:31:36 | 200 |     31.5873ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:32:06 | 200 |      4.5045ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:32:36 | 200 |      3.4998ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:33:06 | 200 |      3.8543ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:33:36 | 200 |      4.0361ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:34:06 | 200 |      3.3663ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:34:36 | 200 |      3.0398ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:35:06 | 200 |      3.8636ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:35:36 | 200 |      3.0583ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:36:06 | 200 |      6.6848ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:36:36 | 200 |      4.0084ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:37:06 | 200 |      4.7544ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:37:36 | 200 |      4.9303ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:38:06 | 200 |      3.4996ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:38:37 | 200 |      4.4515ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:39:07 | 200 |      5.0044ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:39:37 | 200 |     12.3629ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:40:07 | 200 |       4.999ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:40:37 | 200 |      3.5077ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:41:07 | 200 |      4.5037ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:41:37 | 200 |      3.5844ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:42:07 | 200 |      3.0001ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:42:37 | 200 |      5.2904ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:43:07 | 200 |      3.9992ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:43:37 | 200 |      9.0152ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:44:07 | 200 |      3.4039ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:44:37 | 200 |      2.9976ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:45:08 | 200 |      4.4997ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:45:38 | 200 |      5.0055ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:46:08 | 200 |      5.5032ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:46:38 | 200 |         3.5ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:47:08 | 200 |      3.5082ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:47:38 | 200 |      3.5625ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:48:08 | 200 |      3.5046ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:48:38 | 200 |      3.4996ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:49:08 | 200 |      4.0032ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:49:38 | 200 |       3.999ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:50:08 | 200 |      5.5005ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:50:38 | 200 |      5.3171ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:51:08 | 200 |      4.5039ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:51:38 | 200 |      4.0038ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:52:08 | 200 |           3ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:52:38 | 200 |      2.9999ms |       127.0.0.1 | GET      "/api/tags"
[GIN] 2026/10/05 - 10:53:08 | 200 |    331.1531ms |       127.0.0.1 | GET      "/api/tags"
