time=2026-09-17T02:13:13.052+02:00 level=INFO source=routes.go:1941 msg="server config" env="map[CUDA_VISIBLE_DEVICES: GGML_VK_VISIBLE_DEVICES: GPU_DEVICE_ORDINAL: HIP_VISIBLE_DEVICES: HSA_OVERRIDE_GFX_VERSION: HTTPS_PROXY: HTTP_PROXY: LLAMA_ARG_FIT: LLAMA_ARG_FIT_TARGET: NO_PROXY: OLLAMA_CONTEXT_LENGTH:8192 OLLAMA_CREATE_REMOTE:false OLLAMA_DEBUG:INFO OLLAMA_DEBUG_LOG_REQUESTS:false OLLAMA_EDITOR: OLLAMA_FLASH_ATTENTION:false OLLAMA_GO_TEMPLATE:true OLLAMA_GPU_OVERHEAD:0 OLLAMA_HOST:http://127.0.0.1:11550 OLLAMA_IGPU_ENABLE: OLLAMA_KEEP_ALIVE:30m0s OLLAMA_KV_CACHE_TYPE: OLLAMA_LLM_LIBRARY: OLLAMA_LOAD_TIMEOUT:5m0s OLLAMA_MAX_LOADED_MODELS:1 OLLAMA_MAX_QUEUE:512 OLLAMA_MAX_TRANSFER_STREAMS:4 OLLAMA_MODELS:/work/acb/my-deepseek/models OLLAMA_NOHISTORY:false OLLAMA_NOPRUNE:false OLLAMA_NO_CLOUD:true OLLAMA_NUM_PARALLEL:1 OLLAMA_ORIGINS:[http://localhost https://localhost http://localhost:* https://localhost:* http://127.0.0.1 https://127.0.0.1 http://127.0.0.1:* https://127.0.0.1:* http://0.0.0.0 https://0.0.0.0 http://0.0.0.0:* https://0.0.0.0:* app://* file://* tauri://* vscode-webview://* vscode-file://*] OLLAMA_REMOTES:[ollama.com] OLLAMA_SCHED_SPREAD:false OLLAMA_VULKAN:true ROCR_VISIBLE_DEVICES: http_proxy: https_proxy: no_proxy:]" time=2026-09-17T02:13:13.052+02:00 level=INFO source=routes.go:1943 msg="Ollama cloud disabled: true" time=2026-09-17T02:13:13.052+02:00 level=INFO source=images.go:929 msg="total blobs: 0" time=2026-09-17T02:13:13.052+02:00 level=INFO source=images.go:937 msg="total unused blobs removed: 0" time=2026-09-17T02:13:13.052+02:00 level=INFO source=routes.go:2001 msg="Listening on 127.0.0.1:11550 (version 0.34.1)" time=2026-09-17T02:13:13.053+02:00 level=INFO source=runner.go:60 msg="discovering available GPUs..." time=2026-09-17T02:13:13.053+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h54m0.074241062s consecutive_failures=0 time=2026-09-17T02:13:13.263+02:00 level=INFO source=types.go:50 msg="inference compute" id=cpu library=cpu compute="" name=cpu description=cpu libdirs=ollama driver="" pci_id="" type="" total="125.2 GiB" available="40.9 GiB" time=2026-09-17T02:13:13.263+02:00 level=INFO source=routes.go:2051 msg="vram-based default context" total_vram="0 B" default_num_ctx=4096 [GIN] 2026/09/17 - 02:13:14 | 200 | 64.239µs | 127.0.0.1 | GET "/api/version" [GIN] 2026/09/17 - 02:13:14 | 200 | 30.316µs | 127.0.0.1 | HEAD "/" time=2026-09-17T02:13:14.930+02:00 level=INFO source=download.go:181 msg="downloading 6e9f90f02bb3 in 16 561 MB part(s)" time=2026-09-17T02:14:33.746+02:00 level=INFO source=download.go:181 msg="downloading c5ad996bda6e in 1 556 B part(s)" time=2026-09-17T02:14:34.310+02:00 level=INFO source=download.go:181 msg="downloading 6e4c38e1172f in 1 1.1 KB part(s)" time=2026-09-17T02:14:34.837+02:00 level=INFO source=download.go:181 msg="downloading f4d24e9138dd in 1 148 B part(s)" time=2026-09-17T02:14:35.382+02:00 level=INFO source=download.go:181 msg="downloading 3c24b0c80794 in 1 488 B part(s)" [GIN] 2026/09/17 - 02:14:40 | 200 | 1m26s | 127.0.0.1 | POST "/api/pull" [GIN] 2026/09/17 - 02:14:40 | 200 | 33.923µs | 127.0.0.1 | HEAD "/" [GIN] 2026/09/17 - 02:14:40 | 200 | 56.766176ms | 127.0.0.1 | POST "/api/show" [GIN] 2026/09/17 - 02:14:40 | 200 | 22.913µs | 127.0.0.1 | HEAD "/" [GIN] 2026/09/17 - 02:14:40 | 200 | 409.208µs | 127.0.0.1 | POST "/api/show" time=2026-09-17T02:14:57.954+02:00 level=INFO source=routes.go:1941 msg="server config" env="map[CUDA_VISIBLE_DEVICES: GGML_VK_VISIBLE_DEVICES: GPU_DEVICE_ORDINAL: HIP_VISIBLE_DEVICES: HSA_OVERRIDE_GFX_VERSION: HTTPS_PROXY: HTTP_PROXY: LLAMA_ARG_FIT: LLAMA_ARG_FIT_TARGET: NO_PROXY: OLLAMA_CONTEXT_LENGTH:8192 OLLAMA_CREATE_REMOTE:false OLLAMA_DEBUG:INFO OLLAMA_DEBUG_LOG_REQUESTS:false OLLAMA_EDITOR: OLLAMA_FLASH_ATTENTION:false OLLAMA_GO_TEMPLATE:true OLLAMA_GPU_OVERHEAD:0 OLLAMA_HOST:http://127.0.0.1:11550 OLLAMA_IGPU_ENABLE: OLLAMA_KEEP_ALIVE:30m0s OLLAMA_KV_CACHE_TYPE: OLLAMA_LLM_LIBRARY: OLLAMA_LOAD_TIMEOUT:5m0s OLLAMA_MAX_LOADED_MODELS:1 OLLAMA_MAX_QUEUE:512 OLLAMA_MAX_TRANSFER_STREAMS:4 OLLAMA_MODELS:/work/acb/my-deepseek/models OLLAMA_NOHISTORY:false OLLAMA_NOPRUNE:false OLLAMA_NO_CLOUD:true OLLAMA_NUM_PARALLEL:1 OLLAMA_ORIGINS:[http://localhost https://localhost http://localhost:* https://localhost:* http://127.0.0.1 https://127.0.0.1 http://127.0.0.1:* https://127.0.0.1:* http://0.0.0.0 https://0.0.0.0 http://0.0.0.0:* https://0.0.0.0:* app://* file://* tauri://* vscode-webview://* vscode-file://*] OLLAMA_REMOTES:[ollama.com] OLLAMA_SCHED_SPREAD:false OLLAMA_VULKAN:true ROCR_VISIBLE_DEVICES: http_proxy: https_proxy: no_proxy:]" time=2026-09-17T02:14:57.954+02:00 level=INFO source=routes.go:1943 msg="Ollama cloud disabled: true" time=2026-09-17T02:14:57.954+02:00 level=INFO source=images.go:929 msg="total blobs: 0" time=2026-09-17T02:14:57.955+02:00 level=INFO source=images.go:937 msg="total unused blobs removed: 0" time=2026-09-17T02:14:57.955+02:00 level=INFO source=routes.go:2001 msg="Listening on 127.0.0.1:11550 (version 0.34.1)" time=2026-09-17T02:14:57.955+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h13m49.717714036s consecutive_failures=0 time=2026-09-17T02:14:57.955+02:00 level=INFO source=runner.go:60 msg="discovering available GPUs..." time=2026-09-17T02:14:58.130+02:00 level=INFO source=types.go:50 msg="inference compute" id=cpu library=cpu compute="" name=cpu description=cpu libdirs=ollama driver="" pci_id="" type="" total="125.2 GiB" available="42.6 GiB" time=2026-09-17T02:14:58.130+02:00 level=INFO source=routes.go:2051 msg="vram-based default context" total_vram="0 B" default_num_ctx=4096 [GIN] 2026/09/17 - 02:14:58 | 200 | 67.295µs | 127.0.0.1 | GET "/api/version" time=2026-09-17T02:14:59.008+02:00 level=INFO source=sched.go:1152 msg="disabling mmap for llama-server load by default" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e reason=cpu time=2026-09-17T02:14:59.008+02:00 level=INFO source=server.go:100 msg="using llama-server for model" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e time=2026-09-17T02:14:59.009+02:00 level=INFO source=llama_server.go:434 msg="starting llama-server" cmd="/work/acb/my-deepseek/runtime/lib/ollama/llama-server --model /work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e --port 40049 --host 127.0.0.1 --no-webui --offline -c 8192 -np 1 --log-verbosity 4 --no-log-prefix --no-log-timestamps --load-mode none --flash-attn auto -b 1024 -ub 1024 --context-shift --keep 4" time=2026-09-17T02:14:59.009+02:00 level=INFO source=sched.go:618 msg="system memory" total="125.2 GiB" free="42.6 GiB" free_swap="252.0 KiB" time=2026-09-17T02:14:59.009+02:00 level=INFO source=llama_server.go:1052 msg="loading model via llama-server" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e time=2026-09-17T02:14:59.009+02:00 level=INFO source=llama_server.go:1305 msg="waiting for llama-server to start responding" time=2026-09-17T02:14:59.009+02:00 level=INFO source=llama_server.go:1360 msg="waiting for llama-server to become available" status="llm server not responding" cmn common_param: common_params_print_info: build 1 (5d806aa25) with GNU 13.3.1 for Linux x86_64 cmn common_param: common_params_print_info: verbosity = 4 (adjust with the `-lv N` CLI arg) cmn common_param: device_info: cmn common_param: - CPU : AMD Ryzen 9 5950X 16-Core Processor (128213 MiB, 128213 MiB free) cmn common_param: system_info: n_threads = 16 (n_threads_batch = 16) / 32 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | BMI2 = 1 | LLAMAFILE = 1 | REPACK = 1 | srv init: using 31 threads for HTTP server srv init: The UI is disabled srv init: Use --ui/--no-ui (or deprecated --webui/--no-webui) to enable/disable srv llama_server: ----------------- srv llama_server: CORS is set to allow all origins ('*') and no API key is set srv llama_server: this can be a security risk (cross-origin attacks) srv llama_server: more info: https://github.com/ggml-org/llama.cpp/pull/25655 srv llama_server: ----------------- srv start: binding port with default address family srv load_model: loading model '/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e' srv load_model: local path '/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e' cmn common_init_: fitting params to device memory ... cmn common_init_: (for bugs during this step try to reproduce them with -fit off, or provide --verbose logs if the bug only occurs with -fit on) common_params_fit_impl: getting device memory data for initial parameters: common_memory_breakdown_print: | memory breakdown [MiB] | total free self model context compute unaccounted | common_memory_breakdown_print: | - Host | 4251 = 2457 + 1536 + 258 | common_memory_breakdown_print: | - CPU_REPACK | 6108 = 6108 + 0 + 0 | common_params_fit_impl: projected to use 4251 MiB of host memory vs. 128213 MiB of total host memory common_params_fit_impl: will leave 123962 >= 1024 MiB of system memory, no changes needed common_fit_params: successfully fit params to free device memory common_fit_params: fitting params to free memory took 0.19 seconds llama_model_loader: loaded meta data with 26 key-value pairs and 579 tensors from /work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e (version GGUF V3 (latest)) llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output. llama_model_loader: - kv 0: general.architecture str = qwen2 llama_model_loader: - kv 1: general.type str = model llama_model_loader: - kv 2: general.name str = DeepSeek R1 Distill Qwen 14B llama_model_loader: - kv 3: general.basename str = DeepSeek-R1-Distill-Qwen llama_model_loader: - kv 4: general.size_label str = 14B llama_model_loader: - kv 5: qwen2.block_count u32 = 48 llama_model_loader: - kv 6: qwen2.context_length u32 = 131072 llama_model_loader: - kv 7: qwen2.embedding_length u32 = 5120 llama_model_loader: - kv 8: qwen2.feed_forward_length u32 = 13824 llama_model_loader: - kv 9: qwen2.attention.head_count u32 = 40 llama_model_loader: - kv 10: qwen2.attention.head_count_kv u32 = 8 llama_model_loader: - kv 11: qwen2.rope.freq_base f32 = 1000000.000000 llama_model_loader: - kv 12: qwen2.attention.layer_norm_rms_epsilon f32 = 0.000010 llama_model_loader: - kv 13: general.file_type u32 = 15 llama_model_loader: - kv 14: tokenizer.ggml.model str = gpt2 llama_model_loader: - kv 15: tokenizer.ggml.pre str = qwen2 llama_model_loader: - kv 16: tokenizer.ggml.tokens arr[str,152064] = ["!", "\"", "#", "$", "%", "&", "'", ... llama_model_loader: - kv 17: tokenizer.ggml.token_type arr[i32,152064] = [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, ... llama_model_loader: - kv 18: tokenizer.ggml.merges arr[str,151387] = ["Ġ Ġ", "ĠĠ ĠĠ", "i n", "Ġ t",... llama_model_loader: - kv 19: tokenizer.ggml.bos_token_id u32 = 151646 llama_model_loader: - kv 20: tokenizer.ggml.eos_token_id u32 = 151643 llama_model_loader: - kv 21: tokenizer.ggml.padding_token_id u32 = 151643 llama_model_loader: - kv 22: tokenizer.ggml.add_bos_token bool = true llama_model_loader: - kv 23: tokenizer.ggml.add_eos_token bool = false llama_model_loader: - kv 24: tokenizer.chat_template str = {% if not add_generation_prompt is de... llama_model_loader: - kv 25: general.quantization_version u32 = 2 llama_model_loader: - type f32: 241 tensors llama_model_loader: - type q4_K: 289 tensors llama_model_loader: - type q6_K: 49 tensors print_info: file format = GGUF V3 (latest) print_info: file type = Q4_K - Medium print_info: file size = 8.37 GiB (4.87 BPW) time=2026-09-17T02:14:59.260+02:00 level=INFO source=llama_server.go:1360 msg="waiting for llama-server to become available" status="llm server loading model" load: 0 unused tokens load: control-looking token: 128247 '' was not control-type; this is probably a bug in the model. its type will be overridden load: printing all EOG tokens: load: - 128247 ('') load: - 151643 ('<|end▁of▁sentence|>') load: - 151662 ('<|fim_pad|>') load: - 151663 ('<|repo_name|>') load: - 151664 ('<|file_sep|>') load: special tokens cache size = 23 load: token to piece cache size = 0.9310 MB print_info: arch = qwen2 print_info: vocab_only = 0 print_info: no_alloc = 0 print_info: n_ctx_train = 131072 print_info: n_embd_inp = 5120 print_info: n_embd = 5120 print_info: n_embd_out = 5120 print_info: n_layer = 48 print_info: n_layer_all = 48 print_info: n_head = 40 print_info: n_head_kv = 8 print_info: n_rot = 128 print_info: n_swa = 0 print_info: is_swa_any = 0 print_info: non_causal_type = 0 print_info: n_embd_head_k = 128 print_info: n_embd_head_v = 128 print_info: n_gqa = 5 print_info: n_embd_k_gqa = 1024 print_info: n_embd_v_gqa = 1024 print_info: f_norm_eps = 0.0e+00 print_info: f_norm_rms_eps = 1.0e-05 print_info: f_clamp_kqv = 0.0e+00 print_info: f_max_alibi_bias = 0.0e+00 print_info: f_logit_scale = 0.0e+00 print_info: f_attn_scale = 0.0e+00 print_info: f_attn_value_scale = 0.0000 print_info: n_ff = 13824 print_info: n_expert = 0 print_info: n_expert_used = 0 print_info: n_expert_groups = 0 print_info: n_group_used = 0 print_info: causal attn = 1 print_info: pooling type = -1 print_info: rope type = 2 print_info: rope scaling = linear print_info: freq_base_train = 1000000.0 print_info: freq_scale_train = 1 print_info: n_ctx_orig_yarn = 131072 print_info: rope_yarn_log_mul = 0.0000 print_info: rope_finetuned = unknown print_info: model type = 14B print_info: model params = 14.77 B print_info: general.name = DeepSeek R1 Distill Qwen 14B print_info: vocab type = BPE print_info: n_vocab = 152064 print_info: n_merges = 151387 print_info: BOS token = 151646 '<|begin▁of▁sentence|>' print_info: EOS token = 151643 '<|end▁of▁sentence|>' print_info: EOT token = 151643 '<|end▁of▁sentence|>' print_info: PAD token = 151643 '<|end▁of▁sentence|>' print_info: LF token = 198 'Ċ' print_info: FIM PRE token = 151659 '<|fim_prefix|>' print_info: FIM SUF token = 151661 '<|fim_suffix|>' print_info: FIM MID token = 151660 '<|fim_middle|>' print_info: FIM PAD token = 151662 '<|fim_pad|>' print_info: FIM REP token = 151663 '<|repo_name|>' print_info: FIM SEP token = 151664 '<|file_sep|>' print_info: EOG token = 128247 '' print_info: EOG token = 151643 '<|end▁of▁sentence|>' print_info: EOG token = 151662 '<|fim_pad|>' print_info: EOG token = 151663 '<|repo_name|>' print_info: EOG token = 151664 '<|file_sep|>' print_info: max token length = 256 load_tensors: loading model tensors, this can take a while... (load_mode = none) load_tensors: CPU model buffer size = 2457.29 MiB load_tensors: CPU_REPACK model buffer size = 6108.75 MiB cmn common_init_: added logit bias = -inf cmn common_init_: added <|end▁of▁sentence|> logit bias = -inf cmn common_init_: added <|fim_pad|> logit bias = -inf cmn common_init_: added <|repo_name|> logit bias = -inf cmn common_init_: added <|file_sep|> logit bias = -inf llama_context: constructing llama_context llama_context: n_seq_max = 1 llama_context: n_ctx = 8192 llama_context: n_ctx_seq = 8192 llama_context: n_batch = 1024 llama_context: n_ubatch = 1024 llama_context: causal_attn = 1 llama_context: flash_attn = auto llama_context: kv_unified = false llama_context: freq_base = 1000000.0 llama_context: freq_scale = 1 llama_context: n_rs_seq = 0 llama_context: n_outputs_max = 1 llama_context: n_outputs_max_per_seq = 1 llama_context: n_ctx_seq (8192) < n_ctx_train (131072) -- the full capacity of the model will not be utilized llama_context: CPU output buffer size = 0.58 MiB llama_kv_cache: CPU KV buffer size = 1536.00 MiB llama_kv_cache: size = 1536.00 MiB ( 8192 cells, 48 layers, 1/1 seqs), K (f16): 768.00 MiB, V (f16): 768.00 MiB llama_kv_cache: attn_rot_k = 0, n_embd_head_k_all = 128 llama_kv_cache: attn_rot_v = 0, n_embd_head_k_all = 128 sched_reserve: reserving ... resolve_fused_ops: Flash Attention enabled resolve_fused_ops: resolving fused DeepSeek V4 HC support: resolve_fused_ops: fused DeepSeek V4 HC pre enabled resolve_fused_ops: fused DeepSeek V4 HC comb enabled resolve_fused_ops: fused DeepSeek V4 HC post enabled sched_reserve: CPU compute buffer size = 258.02 MiB sched_reserve: graph nodes = 1638 sched_reserve: graph splits = 1 sched_reserve: reserve took 3.83 ms, sched copies = 1 cmn init: llama threadpool init, n_threads = 16 cmn common_init_: warming up the model with an empty run - please wait ... (--no-warmup to disable) srv load_model: initializing, n_slots = 1, n_ctx_slot = 8192, kv_unified = 'false' spec common_specu: no implementations specified for speculative decoding slot load_model: id 0 | task -1 | new slot, n_ctx = 8192 srv load_model: prompt cache is enabled, size limit: 8192 MiB srv load_model: use `--cache-ram 0` to disable the prompt cache srv load_model: for more info see https://github.com/ggml-org/llama.cpp/pull/16391 srv load_model: context checkpoints enabled, max = 32, min spacing = 8192 srv init: idle slots will be saved to prompt cache upon starting a new task srv init: init: chat template, example_format: 'You are a helpful assistant<|User|>Hello<|Assistant|>Hi there<|end▁of▁sentence|><|User|>How are you?<|Assistant|>' srv init: init: chat template, thinking = 1 srv init: preserve_reasoning kwarg: not supported by template srv llama_server: model loaded srv llama_server: listening on http://127.0.0.1:40049 srv update_slots: all slots are idle time=2026-09-17T02:15:11.551+02:00 level=INFO source=llama_server.go:1372 msg="llama-server started in 12.54 seconds" time=2026-09-17T02:15:11.551+02:00 level=INFO source=images.go:382 msg="template selection" model=registry.ollama.ai/library/deepseek-r1:14b selected=gguf_chat_template renderer="" parser="" go_template="[completion thinking]" chat_template="[tools thinking completion]" harmony=null renderer_parser=null time=2026-09-17T02:15:11.551+02:00 level=INFO source=sched.go:733 msg="loaded runners" count=1 time=2026-09-17T02:15:11.551+02:00 level=INFO source=llama_server.go:1305 msg="waiting for llama-server to start responding" time=2026-09-17T02:15:11.551+02:00 level=INFO source=llama_server.go:1372 msg="llama-server started in 12.54 seconds" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - skipping, slot is empty slot get_availabl: id 0 | task -1 | selected slot by LRU, t_last = -1 srv get_availabl: updating prompt cache srv load: - looking for better prompt, base f_keep = -1.000, f_sim = 0.000 srv update: - cache state: 0 prompts, 0.000 MiB (limits: 8192.000 MiB, 8192 tokens, 8589934592 est) srv get_availabl: prompt cache update took 0.01 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 0 | processing task, is_child = 0 slot operator(): id 0 | task 0 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 19 slot operator(): id 0 | task 0 | cached n_tokens = 0, memory_seq_rm [0, end) slot init_sampler: id 0 | task 0 | init sampler, took 0.01 ms, tokens: text = 19, total = 19 slot print_timing: id 0 | task 0 | n_gen = 100, tg = 4.06 t/s, tg_3s = 4.11 t/s slot print_timing: id 0 | task 0 | n_gen = 113, tg = 4.06 t/s, tg_3s = 4.06 t/s slot print_timing: id 0 | task 0 | n_gen = 126, tg = 4.06 t/s, tg_3s = 4.05 t/s slot print_timing: id 0 | task 0 | n_gen = 139, tg = 4.06 t/s, tg_3s = 4.05 t/s slot print_timing: id 0 | task 0 | n_gen = 152, tg = 4.06 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 0 | n_gen = 165, tg = 4.06 t/s, tg_3s = 4.04 t/s slot print_timing: id 0 | task 0 | n_gen = 178, tg = 4.06 t/s, tg_3s = 4.05 t/s slot print_timing: id 0 | task 0 | n_gen = 190, tg = 4.05 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 0 | n_gen = 203, tg = 4.05 t/s, tg_3s = 4.04 t/s slot print_timing: id 0 | task 0 | n_gen = 215, tg = 4.04 t/s, tg_3s = 3.98 t/s slot print_timing: id 0 | task 0 | n_gen = 228, tg = 4.04 t/s, tg_3s = 4.00 t/s slot print_timing: id 0 | task 0 | n_gen = 240, tg = 4.04 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 0 | prompt eval time = 542.87 ms / 19 tokens ( 28.57 ms per token, 35.00 tokens per second) slot print_timing: id 0 | task 0 | eval time = 60423.81 ms / 245 tokens ( 247.64 ms per token, 4.04 tokens per second) slot print_timing: id 0 | task 0 | total time = 60966.68 ms / 264 tokens slot print_timing: id 0 | task 0 | graphs reused = 243 slot release: id 0 | task 0 | stop processing: n_tokens = 263, truncated = 0 srv update_slots: all slots are idle [GIN] 2026/09/17 - 02:16:12 | 200 | 1m13s | 127.0.0.1 | POST "/v1/chat/completions" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - checking sim = 0.133 (2/15) > 0.100 slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.133 (> 0.100 thold), f_keep = 0.008 srv get_availabl: updating prompt cache srv prompt_save: - saving prompt with length 263, total state size = 49.317 MiB (draft: 0.000 MiB) srv load: - looking for better prompt, base f_keep = 0.008, f_sim = 0.133 srv load: - prompt with length 263, lcp = 2, f_keep = 0.008, f_sim = 0.133 srv update: - cache state: 1 prompts, 49.317 MiB (limits: 8192.000 MiB, 8192 tokens, 43687 est) srv update: - prompt 0x3d72aa0: 263 tokens, checkpoints: 0, 49.317 MiB srv get_availabl: prompt cache update took 24.61 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 246 | processing task, is_child = 0 slot operator(): id 0 | task 246 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 15 slot operator(): id 0 | task 246 | cached n_tokens = 2, memory_seq_rm [2, end) slot init_sampler: id 0 | task 246 | init sampler, took 0.00 ms, tokens: text = 15, total = 15 slot print_timing: id 0 | task 246 | n_gen = 100, tg = 4.08 t/s, tg_3s = 4.12 t/s slot print_timing: id 0 | task 246 | n_gen = 113, tg = 4.08 t/s, tg_3s = 4.07 t/s slot print_timing: id 0 | task 246 | n_gen = 126, tg = 4.07 t/s, tg_3s = 4.05 t/s slot print_timing: id 0 | task 246 | n_gen = 139, tg = 4.07 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 246 | n_gen = 152, tg = 4.07 t/s, tg_3s = 4.08 t/s slot print_timing: id 0 | task 246 | n_gen = 164, tg = 4.06 t/s, tg_3s = 3.96 t/s slot print_timing: id 0 | task 246 | n_gen = 177, tg = 4.06 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 246 | n_gen = 190, tg = 4.06 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 246 | n_gen = 203, tg = 4.06 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 246 | n_gen = 216, tg = 4.05 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 246 | n_gen = 229, tg = 4.05 t/s, tg_3s = 4.02 t/s slot print_timing: id 0 | task 246 | n_gen = 242, tg = 4.05 t/s, tg_3s = 4.04 t/s slot print_timing: id 0 | task 246 | n_gen = 254, tg = 4.05 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 246 | n_gen = 267, tg = 4.05 t/s, tg_3s = 4.02 t/s slot print_timing: id 0 | task 246 | n_gen = 279, tg = 4.04 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 246 | n_gen = 291, tg = 4.04 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 246 | n_gen = 303, tg = 4.03 t/s, tg_3s = 3.89 t/s slot print_timing: id 0 | task 246 | n_gen = 315, tg = 4.03 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 328, tg = 4.03 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 246 | n_gen = 340, tg = 4.03 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 353, tg = 4.03 t/s, tg_3s = 4.04 t/s slot print_timing: id 0 | task 246 | n_gen = 365, tg = 4.02 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 246 | n_gen = 378, tg = 4.02 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 246 | n_gen = 390, tg = 4.02 t/s, tg_3s = 3.94 t/s slot print_timing: id 0 | task 246 | n_gen = 402, tg = 4.02 t/s, tg_3s = 3.96 t/s slot print_timing: id 0 | task 246 | n_gen = 415, tg = 4.02 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 246 | n_gen = 427, tg = 4.02 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 246 | n_gen = 439, tg = 4.01 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 246 | n_gen = 451, tg = 4.01 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 464, tg = 4.01 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 246 | n_gen = 476, tg = 4.01 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 246 | n_gen = 488, tg = 4.01 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 246 | n_gen = 500, tg = 4.01 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 512, tg = 4.01 t/s, tg_3s = 3.87 t/s slot print_timing: id 0 | task 246 | n_gen = 524, tg = 4.00 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 536, tg = 4.00 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 548, tg = 4.00 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 246 | n_gen = 560, tg = 4.00 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 246 | n_gen = 572, tg = 4.00 t/s, tg_3s = 3.96 t/s slot print_timing: id 0 | task 246 | n_gen = 584, tg = 4.00 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 596, tg = 4.00 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 608, tg = 3.99 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 246 | n_gen = 620, tg = 3.99 t/s, tg_3s = 3.96 t/s slot print_timing: id 0 | task 246 | n_gen = 632, tg = 3.99 t/s, tg_3s = 3.94 t/s slot print_timing: id 0 | task 246 | n_gen = 644, tg = 3.99 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 656, tg = 3.99 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 246 | n_gen = 668, tg = 3.99 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 680, tg = 3.99 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 246 | n_gen = 692, tg = 3.99 t/s, tg_3s = 3.91 t/s slot print_timing: id 0 | task 246 | n_gen = 704, tg = 3.98 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 246 | n_gen = 716, tg = 3.98 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 728, tg = 3.98 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 246 | n_gen = 740, tg = 3.98 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 246 | n_gen = 752, tg = 3.98 t/s, tg_3s = 3.89 t/s slot print_timing: id 0 | task 246 | n_gen = 764, tg = 3.98 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 246 | n_gen = 776, tg = 3.98 t/s, tg_3s = 3.82 t/s slot print_timing: id 0 | task 246 | n_gen = 788, tg = 3.97 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 246 | n_gen = 800, tg = 3.97 t/s, tg_3s = 3.89 t/s slot print_timing: id 0 | task 246 | n_gen = 812, tg = 3.97 t/s, tg_3s = 3.83 t/s slot print_timing: id 0 | task 246 | n_gen = 824, tg = 3.97 t/s, tg_3s = 3.91 t/s slot print_timing: id 0 | task 246 | n_gen = 836, tg = 3.97 t/s, tg_3s = 3.87 t/s slot print_timing: id 0 | task 246 | n_gen = 848, tg = 3.97 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 246 | n_gen = 860, tg = 3.97 t/s, tg_3s = 3.84 t/s slot print_timing: id 0 | task 246 | n_gen = 872, tg = 3.96 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 246 | n_gen = 884, tg = 3.96 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 246 | n_gen = 896, tg = 3.96 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 246 | n_gen = 908, tg = 3.96 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 246 | n_gen = 920, tg = 3.96 t/s, tg_3s = 3.80 t/s slot print_timing: id 0 | task 246 | n_gen = 932, tg = 3.95 t/s, tg_3s = 3.80 t/s slot print_timing: id 0 | task 246 | n_gen = 944, tg = 3.95 t/s, tg_3s = 3.76 t/s slot print_timing: id 0 | task 246 | n_gen = 956, tg = 3.95 t/s, tg_3s = 3.74 t/s slot print_timing: id 0 | task 246 | n_gen = 968, tg = 3.95 t/s, tg_3s = 3.68 t/s slot print_timing: id 0 | task 246 | n_gen = 979, tg = 3.94 t/s, tg_3s = 3.66 t/s slot print_timing: id 0 | task 246 | n_gen = 991, tg = 3.94 t/s, tg_3s = 3.80 t/s slot print_timing: id 0 | task 246 | n_gen = 1003, tg = 3.94 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 246 | n_gen = 1015, tg = 3.94 t/s, tg_3s = 3.84 t/s slot print_timing: id 0 | task 246 | n_gen = 1027, tg = 3.94 t/s, tg_3s = 3.75 t/s slot print_timing: id 0 | task 246 | n_gen = 1039, tg = 3.93 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 246 | n_gen = 1051, tg = 3.93 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 246 | n_gen = 1063, tg = 3.93 t/s, tg_3s = 3.82 t/s slot print_timing: id 0 | task 246 | prompt eval time = 446.20 ms / 13 tokens ( 34.32 ms per token, 29.13 tokens per second) slot print_timing: id 0 | task 246 | eval time = 272979.32 ms / 1074 tokens ( 254.41 ms per token, 3.93 tokens per second) slot print_timing: id 0 | task 246 | total time = 273425.52 ms / 1087 tokens slot print_timing: id 0 | task 246 | graphs reused = 1311 slot release: id 0 | task 246 | stop processing: n_tokens = 1088, truncated = 0 srv update_slots: all slots are idle [GIN] 2026/09/17 - 02:20:46 | 200 | 4m33s | 127.0.0.1 | POST "/v1/chat/completions" [GIN] 2026/09/17 - 02:20:46 | 200 | 26.95µs | 127.0.0.1 | HEAD "/" [GIN] 2026/09/17 - 02:20:46 | 200 | 535.702µs | 127.0.0.1 | GET "/api/tags" [GIN] 2026/09/17 - 02:20:46 | 200 | 13.966µs | 127.0.0.1 | HEAD "/" [GIN] 2026/09/17 - 02:20:46 | 200 | 55.903µs | 127.0.0.1 | GET "/api/ps" time=2026-09-17T02:20:58.672+02:00 level=INFO source=routes.go:1941 msg="server config" env="map[CUDA_VISIBLE_DEVICES: GGML_VK_VISIBLE_DEVICES: GPU_DEVICE_ORDINAL: HIP_VISIBLE_DEVICES: HSA_OVERRIDE_GFX_VERSION: HTTPS_PROXY: HTTP_PROXY: LLAMA_ARG_FIT: LLAMA_ARG_FIT_TARGET: NO_PROXY: OLLAMA_CONTEXT_LENGTH:8192 OLLAMA_CREATE_REMOTE:false OLLAMA_DEBUG:INFO OLLAMA_DEBUG_LOG_REQUESTS:false OLLAMA_EDITOR: OLLAMA_FLASH_ATTENTION:false OLLAMA_GO_TEMPLATE:true OLLAMA_GPU_OVERHEAD:0 OLLAMA_HOST:http://127.0.0.1:11550 OLLAMA_IGPU_ENABLE: OLLAMA_KEEP_ALIVE:30m0s OLLAMA_KV_CACHE_TYPE: OLLAMA_LLM_LIBRARY: OLLAMA_LOAD_TIMEOUT:5m0s OLLAMA_MAX_LOADED_MODELS:1 OLLAMA_MAX_QUEUE:512 OLLAMA_MAX_TRANSFER_STREAMS:4 OLLAMA_MODELS:/work/acb/my-deepseek/models OLLAMA_NOHISTORY:false OLLAMA_NOPRUNE:false OLLAMA_NO_CLOUD:true OLLAMA_NUM_PARALLEL:1 OLLAMA_ORIGINS:[http://localhost https://localhost http://localhost:* https://localhost:* http://127.0.0.1 https://127.0.0.1 http://127.0.0.1:* https://127.0.0.1:* http://0.0.0.0 https://0.0.0.0 http://0.0.0.0:* https://0.0.0.0:* app://* file://* tauri://* vscode-webview://* vscode-file://*] OLLAMA_REMOTES:[ollama.com] OLLAMA_SCHED_SPREAD:false OLLAMA_VULKAN:true ROCR_VISIBLE_DEVICES: http_proxy: https_proxy: no_proxy:]" time=2026-09-17T02:20:58.672+02:00 level=INFO source=routes.go:1943 msg="Ollama cloud disabled: true" time=2026-09-17T02:20:58.672+02:00 level=INFO source=images.go:929 msg="total blobs: 0" time=2026-09-17T02:20:58.673+02:00 level=INFO source=images.go:937 msg="total unused blobs removed: 0" time=2026-09-17T02:20:58.673+02:00 level=INFO source=routes.go:2001 msg="Listening on 127.0.0.1:11550 (version 0.34.1)" time=2026-09-17T02:20:58.673+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h52m33.344258389s consecutive_failures=0 time=2026-09-17T02:20:58.674+02:00 level=INFO source=runner.go:60 msg="discovering available GPUs..." time=2026-09-17T02:20:58.838+02:00 level=INFO source=types.go:50 msg="inference compute" id=cpu library=cpu compute="" name=cpu description=cpu libdirs=ollama driver="" pci_id="" type="" total="125.2 GiB" available="42.0 GiB" time=2026-09-17T02:20:58.838+02:00 level=INFO source=routes.go:2051 msg="vram-based default context" total_vram="0 B" default_num_ctx=4096 [GIN] 2026/09/17 - 02:20:59 | 200 | 62.987µs | 127.0.0.1 | GET "/api/version" time=2026-09-17T02:20:59.704+02:00 level=INFO source=sched.go:1152 msg="disabling mmap for llama-server load by default" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e reason=cpu time=2026-09-17T02:20:59.704+02:00 level=INFO source=server.go:100 msg="using llama-server for model" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e time=2026-09-17T02:20:59.705+02:00 level=INFO source=llama_server.go:434 msg="starting llama-server" cmd="/work/acb/my-deepseek/runtime/lib/ollama/llama-server --model /work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e --port 40003 --host 127.0.0.1 --no-webui --offline -c 8192 -np 1 --log-verbosity 4 --no-log-prefix --no-log-timestamps --load-mode none --flash-attn auto -b 1024 -ub 1024 --context-shift --keep 4" time=2026-09-17T02:20:59.705+02:00 level=INFO source=sched.go:618 msg="system memory" total="125.2 GiB" free="42.0 GiB" free_swap="252.0 KiB" time=2026-09-17T02:20:59.705+02:00 level=INFO source=llama_server.go:1052 msg="loading model via llama-server" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e time=2026-09-17T02:20:59.705+02:00 level=INFO source=llama_server.go:1305 msg="waiting for llama-server to start responding" time=2026-09-17T02:20:59.706+02:00 level=INFO source=llama_server.go:1360 msg="waiting for llama-server to become available" status="llm server not responding" cmn common_param: common_params_print_info: build 1 (5d806aa25) with GNU 13.3.1 for Linux x86_64 cmn common_param: common_params_print_info: verbosity = 4 (adjust with the `-lv N` CLI arg) cmn common_param: device_info: cmn common_param: - CPU : AMD Ryzen 9 5950X 16-Core Processor (128213 MiB, 128213 MiB free) cmn common_param: system_info: n_threads = 16 (n_threads_batch = 16) / 32 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | BMI2 = 1 | LLAMAFILE = 1 | REPACK = 1 | srv init: using 31 threads for HTTP server srv init: The UI is disabled srv init: Use --ui/--no-ui (or deprecated --webui/--no-webui) to enable/disable srv llama_server: ----------------- srv llama_server: CORS is set to allow all origins ('*') and no API key is set srv llama_server: this can be a security risk (cross-origin attacks) srv llama_server: more info: https://github.com/ggml-org/llama.cpp/pull/25655 srv llama_server: ----------------- srv start: binding port with default address family srv load_model: loading model '/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e' srv load_model: local path '/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e' cmn common_init_: fitting params to device memory ... cmn common_init_: (for bugs during this step try to reproduce them with -fit off, or provide --verbose logs if the bug only occurs with -fit on) common_params_fit_impl: getting device memory data for initial parameters: common_memory_breakdown_print: | memory breakdown [MiB] | total free self model context compute unaccounted | common_memory_breakdown_print: | - Host | 4251 = 2457 + 1536 + 258 | common_memory_breakdown_print: | - CPU_REPACK | 6108 = 6108 + 0 + 0 | common_params_fit_impl: projected to use 4251 MiB of host memory vs. 128213 MiB of total host memory common_params_fit_impl: will leave 123962 >= 1024 MiB of system memory, no changes needed common_fit_params: successfully fit params to free device memory common_fit_params: fitting params to free memory took 0.19 seconds llama_model_loader: loaded meta data with 26 key-value pairs and 579 tensors from /work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e (version GGUF V3 (latest)) llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output. llama_model_loader: - kv 0: general.architecture str = qwen2 llama_model_loader: - kv 1: general.type str = model llama_model_loader: - kv 2: general.name str = DeepSeek R1 Distill Qwen 14B llama_model_loader: - kv 3: general.basename str = DeepSeek-R1-Distill-Qwen llama_model_loader: - kv 4: general.size_label str = 14B llama_model_loader: - kv 5: qwen2.block_count u32 = 48 llama_model_loader: - kv 6: qwen2.context_length u32 = 131072 llama_model_loader: - kv 7: qwen2.embedding_length u32 = 5120 llama_model_loader: - kv 8: qwen2.feed_forward_length u32 = 13824 llama_model_loader: - kv 9: qwen2.attention.head_count u32 = 40 llama_model_loader: - kv 10: qwen2.attention.head_count_kv u32 = 8 llama_model_loader: - kv 11: qwen2.rope.freq_base f32 = 1000000.000000 llama_model_loader: - kv 12: qwen2.attention.layer_norm_rms_epsilon f32 = 0.000010 llama_model_loader: - kv 13: general.file_type u32 = 15 llama_model_loader: - kv 14: tokenizer.ggml.model str = gpt2 llama_model_loader: - kv 15: tokenizer.ggml.pre str = qwen2 llama_model_loader: - kv 16: tokenizer.ggml.tokens arr[str,152064] = ["!", "\"", "#", "$", "%", "&", "'", ... llama_model_loader: - kv 17: tokenizer.ggml.token_type arr[i32,152064] = [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, ... llama_model_loader: - kv 18: tokenizer.ggml.merges arr[str,151387] = ["Ġ Ġ", "ĠĠ ĠĠ", "i n", "Ġ t",... llama_model_loader: - kv 19: tokenizer.ggml.bos_token_id u32 = 151646 llama_model_loader: - kv 20: tokenizer.ggml.eos_token_id u32 = 151643 llama_model_loader: - kv 21: tokenizer.ggml.padding_token_id u32 = 151643 llama_model_loader: - kv 22: tokenizer.ggml.add_bos_token bool = true llama_model_loader: - kv 23: tokenizer.ggml.add_eos_token bool = false llama_model_loader: - kv 24: tokenizer.chat_template str = {% if not add_generation_prompt is de... llama_model_loader: - kv 25: general.quantization_version u32 = 2 llama_model_loader: - type f32: 241 tensors llama_model_loader: - type q4_K: 289 tensors llama_model_loader: - type q6_K: 49 tensors print_info: file format = GGUF V3 (latest) print_info: file type = Q4_K - Medium print_info: file size = 8.37 GiB (4.87 BPW) time=2026-09-17T02:20:59.957+02:00 level=INFO source=llama_server.go:1360 msg="waiting for llama-server to become available" status="llm server loading model" load: 0 unused tokens load: control-looking token: 128247 '' was not control-type; this is probably a bug in the model. its type will be overridden load: printing all EOG tokens: load: - 128247 ('') load: - 151643 ('<|end▁of▁sentence|>') load: - 151662 ('<|fim_pad|>') load: - 151663 ('<|repo_name|>') load: - 151664 ('<|file_sep|>') load: special tokens cache size = 23 load: token to piece cache size = 0.9310 MB print_info: arch = qwen2 print_info: vocab_only = 0 print_info: no_alloc = 0 print_info: n_ctx_train = 131072 print_info: n_embd_inp = 5120 print_info: n_embd = 5120 print_info: n_embd_out = 5120 print_info: n_layer = 48 print_info: n_layer_all = 48 print_info: n_head = 40 print_info: n_head_kv = 8 print_info: n_rot = 128 print_info: n_swa = 0 print_info: is_swa_any = 0 print_info: non_causal_type = 0 print_info: n_embd_head_k = 128 print_info: n_embd_head_v = 128 print_info: n_gqa = 5 print_info: n_embd_k_gqa = 1024 print_info: n_embd_v_gqa = 1024 print_info: f_norm_eps = 0.0e+00 print_info: f_norm_rms_eps = 1.0e-05 print_info: f_clamp_kqv = 0.0e+00 print_info: f_max_alibi_bias = 0.0e+00 print_info: f_logit_scale = 0.0e+00 print_info: f_attn_scale = 0.0e+00 print_info: f_attn_value_scale = 0.0000 print_info: n_ff = 13824 print_info: n_expert = 0 print_info: n_expert_used = 0 print_info: n_expert_groups = 0 print_info: n_group_used = 0 print_info: causal attn = 1 print_info: pooling type = -1 print_info: rope type = 2 print_info: rope scaling = linear print_info: freq_base_train = 1000000.0 print_info: freq_scale_train = 1 print_info: n_ctx_orig_yarn = 131072 print_info: rope_yarn_log_mul = 0.0000 print_info: rope_finetuned = unknown print_info: model type = 14B print_info: model params = 14.77 B print_info: general.name = DeepSeek R1 Distill Qwen 14B print_info: vocab type = BPE print_info: n_vocab = 152064 print_info: n_merges = 151387 print_info: BOS token = 151646 '<|begin▁of▁sentence|>' print_info: EOS token = 151643 '<|end▁of▁sentence|>' print_info: EOT token = 151643 '<|end▁of▁sentence|>' print_info: PAD token = 151643 '<|end▁of▁sentence|>' print_info: LF token = 198 'Ċ' print_info: FIM PRE token = 151659 '<|fim_prefix|>' print_info: FIM SUF token = 151661 '<|fim_suffix|>' print_info: FIM MID token = 151660 '<|fim_middle|>' print_info: FIM PAD token = 151662 '<|fim_pad|>' print_info: FIM REP token = 151663 '<|repo_name|>' print_info: FIM SEP token = 151664 '<|file_sep|>' print_info: EOG token = 128247 '' print_info: EOG token = 151643 '<|end▁of▁sentence|>' print_info: EOG token = 151662 '<|fim_pad|>' print_info: EOG token = 151663 '<|repo_name|>' print_info: EOG token = 151664 '<|file_sep|>' print_info: max token length = 256 load_tensors: loading model tensors, this can take a while... (load_mode = none) load_tensors: CPU model buffer size = 2457.29 MiB load_tensors: CPU_REPACK model buffer size = 6108.75 MiB cmn common_init_: added logit bias = -inf cmn common_init_: added <|end▁of▁sentence|> logit bias = -inf cmn common_init_: added <|fim_pad|> logit bias = -inf cmn common_init_: added <|repo_name|> logit bias = -inf cmn common_init_: added <|file_sep|> logit bias = -inf llama_context: constructing llama_context llama_context: n_seq_max = 1 llama_context: n_ctx = 8192 llama_context: n_ctx_seq = 8192 llama_context: n_batch = 1024 llama_context: n_ubatch = 1024 llama_context: causal_attn = 1 llama_context: flash_attn = auto llama_context: kv_unified = false llama_context: freq_base = 1000000.0 llama_context: freq_scale = 1 llama_context: n_rs_seq = 0 llama_context: n_outputs_max = 1 llama_context: n_outputs_max_per_seq = 1 llama_context: n_ctx_seq (8192) < n_ctx_train (131072) -- the full capacity of the model will not be utilized llama_context: CPU output buffer size = 0.58 MiB llama_kv_cache: CPU KV buffer size = 1536.00 MiB llama_kv_cache: size = 1536.00 MiB ( 8192 cells, 48 layers, 1/1 seqs), K (f16): 768.00 MiB, V (f16): 768.00 MiB llama_kv_cache: attn_rot_k = 0, n_embd_head_k_all = 128 llama_kv_cache: attn_rot_v = 0, n_embd_head_k_all = 128 sched_reserve: reserving ... resolve_fused_ops: Flash Attention enabled resolve_fused_ops: resolving fused DeepSeek V4 HC support: resolve_fused_ops: fused DeepSeek V4 HC pre enabled resolve_fused_ops: fused DeepSeek V4 HC comb enabled resolve_fused_ops: fused DeepSeek V4 HC post enabled sched_reserve: CPU compute buffer size = 258.02 MiB sched_reserve: graph nodes = 1638 sched_reserve: graph splits = 1 sched_reserve: reserve took 4.21 ms, sched copies = 1 cmn init: llama threadpool init, n_threads = 16 cmn common_init_: warming up the model with an empty run - please wait ... (--no-warmup to disable) srv load_model: initializing, n_slots = 1, n_ctx_slot = 8192, kv_unified = 'false' spec common_specu: no implementations specified for speculative decoding slot load_model: id 0 | task -1 | new slot, n_ctx = 8192 srv load_model: prompt cache is enabled, size limit: 8192 MiB srv load_model: use `--cache-ram 0` to disable the prompt cache srv load_model: for more info see https://github.com/ggml-org/llama.cpp/pull/16391 srv load_model: context checkpoints enabled, max = 32, min spacing = 8192 srv init: idle slots will be saved to prompt cache upon starting a new task srv init: init: chat template, example_format: 'You are a helpful assistant<|User|>Hello<|Assistant|>Hi there<|end▁of▁sentence|><|User|>How are you?<|Assistant|>' srv init: init: chat template, thinking = 1 srv init: preserve_reasoning kwarg: not supported by template srv llama_server: model loaded srv llama_server: listening on http://127.0.0.1:40003 srv update_slots: all slots are idle time=2026-09-17T02:21:13.753+02:00 level=INFO source=llama_server.go:1372 msg="llama-server started in 14.05 seconds" time=2026-09-17T02:21:13.753+02:00 level=INFO source=images.go:382 msg="template selection" model=registry.ollama.ai/library/deepseek-r1:14b selected=gguf_chat_template renderer="" parser="" go_template="[completion thinking]" chat_template="[tools thinking completion]" harmony=null renderer_parser=null time=2026-09-17T02:21:13.753+02:00 level=INFO source=sched.go:733 msg="loaded runners" count=1 time=2026-09-17T02:21:13.753+02:00 level=INFO source=llama_server.go:1305 msg="waiting for llama-server to start responding" time=2026-09-17T02:21:13.754+02:00 level=INFO source=llama_server.go:1372 msg="llama-server started in 14.05 seconds" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - skipping, slot is empty slot get_availabl: id 0 | task -1 | selected slot by LRU, t_last = -1 srv get_availabl: updating prompt cache srv load: - looking for better prompt, base f_keep = -1.000, f_sim = 0.000 srv update: - cache state: 0 prompts, 0.000 MiB (limits: 8192.000 MiB, 8192 tokens, 8589934592 est) srv get_availabl: prompt cache update took 0.01 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> ?min-p -> ?xtc -> temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 0.950, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.600 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 0 | processing task, is_child = 0 slot operator(): id 0 | task 0 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 11 slot operator(): id 0 | task 0 | cached n_tokens = 0, memory_seq_rm [0, end) slot init_sampler: id 0 | task 0 | init sampler, took 0.01 ms, tokens: text = 11, total = 11 slot print_timing: id 0 | task 0 | n_gen = 100, tg = 4.07 t/s, tg_3s = 4.11 t/s slot print_timing: id 0 | task 0 | n_gen = 113, tg = 4.06 t/s, tg_3s = 4.04 t/s slot print_timing: id 0 | task 0 | n_gen = 126, tg = 4.07 t/s, tg_3s = 4.09 t/s slot print_timing: id 0 | task 0 | n_gen = 138, tg = 4.06 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 0 | n_gen = 151, tg = 4.06 t/s, tg_3s = 4.07 t/s slot print_timing: id 0 | task 0 | n_gen = 164, tg = 4.05 t/s, tg_3s = 4.02 t/s slot print_timing: id 0 | task 0 | prompt eval time = 411.74 ms / 11 tokens ( 37.43 ms per token, 26.72 tokens per second) slot print_timing: id 0 | task 0 | eval time = 40706.36 ms / 166 tokens ( 246.71 ms per token, 4.05 tokens per second) slot print_timing: id 0 | task 0 | total time = 41118.10 ms / 177 tokens slot print_timing: id 0 | task 0 | graphs reused = 165 slot release: id 0 | task 0 | stop processing: n_tokens = 176, truncated = 0 srv update_slots: all slots are idle [GIN] 2026/09/17 - 02:21:54 | 200 | 55.199018736s | 127.0.0.1 | POST "/api/chat" time=2026-09-17T02:24:19.757+02:00 level=INFO source=routes.go:1941 msg="server config" env="map[CUDA_VISIBLE_DEVICES: GGML_VK_VISIBLE_DEVICES: GPU_DEVICE_ORDINAL: HIP_VISIBLE_DEVICES: HSA_OVERRIDE_GFX_VERSION: HTTPS_PROXY: HTTP_PROXY: LLAMA_ARG_FIT: LLAMA_ARG_FIT_TARGET: NO_PROXY: OLLAMA_CONTEXT_LENGTH:8192 OLLAMA_CREATE_REMOTE:false OLLAMA_DEBUG:INFO OLLAMA_DEBUG_LOG_REQUESTS:false OLLAMA_EDITOR: OLLAMA_FLASH_ATTENTION:false OLLAMA_GO_TEMPLATE:true OLLAMA_GPU_OVERHEAD:0 OLLAMA_HOST:http://127.0.0.1:11550 OLLAMA_IGPU_ENABLE: OLLAMA_KEEP_ALIVE:30m0s OLLAMA_KV_CACHE_TYPE: OLLAMA_LLM_LIBRARY: OLLAMA_LOAD_TIMEOUT:5m0s OLLAMA_MAX_LOADED_MODELS:1 OLLAMA_MAX_QUEUE:512 OLLAMA_MAX_TRANSFER_STREAMS:4 OLLAMA_MODELS:/work/acb/my-deepseek/models OLLAMA_NOHISTORY:false OLLAMA_NOPRUNE:false OLLAMA_NO_CLOUD:true OLLAMA_NUM_PARALLEL:1 OLLAMA_ORIGINS:[http://localhost https://localhost http://localhost:* https://localhost:* http://127.0.0.1 https://127.0.0.1 http://127.0.0.1:* https://127.0.0.1:* http://0.0.0.0 https://0.0.0.0 http://0.0.0.0:* https://0.0.0.0:* app://* file://* tauri://* vscode-webview://* vscode-file://*] OLLAMA_REMOTES:[ollama.com] OLLAMA_SCHED_SPREAD:false OLLAMA_VULKAN:true ROCR_VISIBLE_DEVICES: http_proxy: https_proxy: no_proxy:]" time=2026-09-17T02:24:19.757+02:00 level=INFO source=routes.go:1943 msg="Ollama cloud disabled: true" time=2026-09-17T02:24:19.757+02:00 level=INFO source=images.go:929 msg="total blobs: 0" time=2026-09-17T02:24:19.757+02:00 level=INFO source=images.go:937 msg="total unused blobs removed: 0" time=2026-09-17T02:24:19.757+02:00 level=INFO source=routes.go:2001 msg="Listening on 127.0.0.1:11550 (version 0.34.1)" time=2026-09-17T02:24:19.758+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h20m53.873696419s consecutive_failures=0 time=2026-09-17T02:24:19.758+02:00 level=INFO source=runner.go:60 msg="discovering available GPUs..." time=2026-09-17T02:24:19.951+02:00 level=INFO source=types.go:50 msg="inference compute" id=cpu library=cpu compute="" name=cpu description=cpu libdirs=ollama driver="" pci_id="" type="" total="125.2 GiB" available="42.1 GiB" time=2026-09-17T02:24:19.951+02:00 level=INFO source=routes.go:2051 msg="vram-based default context" total_vram="0 B" default_num_ctx=4096 [GIN] 2026/09/17 - 02:24:20 | 200 | 61.144µs | 127.0.0.1 | GET "/api/version" time=2026-09-17T02:24:20.811+02:00 level=INFO source=sched.go:1152 msg="disabling mmap for llama-server load by default" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e reason=cpu time=2026-09-17T02:24:20.811+02:00 level=INFO source=server.go:100 msg="using llama-server for model" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e time=2026-09-17T02:24:20.812+02:00 level=INFO source=llama_server.go:434 msg="starting llama-server" cmd="/work/acb/my-deepseek/runtime/lib/ollama/llama-server --model /work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e --port 41669 --host 127.0.0.1 --no-webui --offline -c 8192 -np 1 --log-verbosity 4 --no-log-prefix --no-log-timestamps --load-mode none --flash-attn auto -b 1024 -ub 1024 --context-shift --keep 4" time=2026-09-17T02:24:20.812+02:00 level=INFO source=sched.go:618 msg="system memory" total="125.2 GiB" free="42.1 GiB" free_swap="264.0 KiB" time=2026-09-17T02:24:20.812+02:00 level=INFO source=llama_server.go:1052 msg="loading model via llama-server" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e time=2026-09-17T02:24:20.812+02:00 level=INFO source=llama_server.go:1305 msg="waiting for llama-server to start responding" time=2026-09-17T02:24:20.812+02:00 level=INFO source=llama_server.go:1360 msg="waiting for llama-server to become available" status="llm server not responding" cmn common_param: common_params_print_info: build 1 (5d806aa25) with GNU 13.3.1 for Linux x86_64 cmn common_param: common_params_print_info: verbosity = 4 (adjust with the `-lv N` CLI arg) cmn common_param: device_info: cmn common_param: - CPU : AMD Ryzen 9 5950X 16-Core Processor (128213 MiB, 128213 MiB free) cmn common_param: system_info: n_threads = 16 (n_threads_batch = 16) / 32 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | BMI2 = 1 | LLAMAFILE = 1 | REPACK = 1 | srv init: using 31 threads for HTTP server srv init: The UI is disabled srv init: Use --ui/--no-ui (or deprecated --webui/--no-webui) to enable/disable srv llama_server: ----------------- srv llama_server: CORS is set to allow all origins ('*') and no API key is set srv llama_server: this can be a security risk (cross-origin attacks) srv llama_server: more info: https://github.com/ggml-org/llama.cpp/pull/25655 srv llama_server: ----------------- srv start: binding port with default address family srv load_model: loading model '/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e' srv load_model: local path '/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e' cmn common_init_: fitting params to device memory ... cmn common_init_: (for bugs during this step try to reproduce them with -fit off, or provide --verbose logs if the bug only occurs with -fit on) common_params_fit_impl: getting device memory data for initial parameters: common_memory_breakdown_print: | memory breakdown [MiB] | total free self model context compute unaccounted | common_memory_breakdown_print: | - Host | 4251 = 2457 + 1536 + 258 | common_memory_breakdown_print: | - CPU_REPACK | 6108 = 6108 + 0 + 0 | common_params_fit_impl: projected to use 4251 MiB of host memory vs. 128213 MiB of total host memory common_params_fit_impl: will leave 123962 >= 1024 MiB of system memory, no changes needed common_fit_params: successfully fit params to free device memory common_fit_params: fitting params to free memory took 0.18 seconds llama_model_loader: loaded meta data with 26 key-value pairs and 579 tensors from /work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e (version GGUF V3 (latest)) llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output. llama_model_loader: - kv 0: general.architecture str = qwen2 llama_model_loader: - kv 1: general.type str = model llama_model_loader: - kv 2: general.name str = DeepSeek R1 Distill Qwen 14B llama_model_loader: - kv 3: general.basename str = DeepSeek-R1-Distill-Qwen llama_model_loader: - kv 4: general.size_label str = 14B llama_model_loader: - kv 5: qwen2.block_count u32 = 48 llama_model_loader: - kv 6: qwen2.context_length u32 = 131072 llama_model_loader: - kv 7: qwen2.embedding_length u32 = 5120 llama_model_loader: - kv 8: qwen2.feed_forward_length u32 = 13824 llama_model_loader: - kv 9: qwen2.attention.head_count u32 = 40 llama_model_loader: - kv 10: qwen2.attention.head_count_kv u32 = 8 llama_model_loader: - kv 11: qwen2.rope.freq_base f32 = 1000000.000000 llama_model_loader: - kv 12: qwen2.attention.layer_norm_rms_epsilon f32 = 0.000010 llama_model_loader: - kv 13: general.file_type u32 = 15 llama_model_loader: - kv 14: tokenizer.ggml.model str = gpt2 llama_model_loader: - kv 15: tokenizer.ggml.pre str = qwen2 llama_model_loader: - kv 16: tokenizer.ggml.tokens arr[str,152064] = ["!", "\"", "#", "$", "%", "&", "'", ... llama_model_loader: - kv 17: tokenizer.ggml.token_type arr[i32,152064] = [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, ... llama_model_loader: - kv 18: tokenizer.ggml.merges arr[str,151387] = ["Ġ Ġ", "ĠĠ ĠĠ", "i n", "Ġ t",... llama_model_loader: - kv 19: tokenizer.ggml.bos_token_id u32 = 151646 llama_model_loader: - kv 20: tokenizer.ggml.eos_token_id u32 = 151643 llama_model_loader: - kv 21: tokenizer.ggml.padding_token_id u32 = 151643 llama_model_loader: - kv 22: tokenizer.ggml.add_bos_token bool = true llama_model_loader: - kv 23: tokenizer.ggml.add_eos_token bool = false llama_model_loader: - kv 24: tokenizer.chat_template str = {% if not add_generation_prompt is de... llama_model_loader: - kv 25: general.quantization_version u32 = 2 llama_model_loader: - type f32: 241 tensors llama_model_loader: - type q4_K: 289 tensors llama_model_loader: - type q6_K: 49 tensors print_info: file format = GGUF V3 (latest) print_info: file type = Q4_K - Medium print_info: file size = 8.37 GiB (4.87 BPW) time=2026-09-17T02:24:21.064+02:00 level=INFO source=llama_server.go:1360 msg="waiting for llama-server to become available" status="llm server loading model" load: 0 unused tokens load: control-looking token: 128247 '' was not control-type; this is probably a bug in the model. its type will be overridden load: printing all EOG tokens: load: - 128247 ('') load: - 151643 ('<|end▁of▁sentence|>') load: - 151662 ('<|fim_pad|>') load: - 151663 ('<|repo_name|>') load: - 151664 ('<|file_sep|>') load: special tokens cache size = 23 load: token to piece cache size = 0.9310 MB print_info: arch = qwen2 print_info: vocab_only = 0 print_info: no_alloc = 0 print_info: n_ctx_train = 131072 print_info: n_embd_inp = 5120 print_info: n_embd = 5120 print_info: n_embd_out = 5120 print_info: n_layer = 48 print_info: n_layer_all = 48 print_info: n_head = 40 print_info: n_head_kv = 8 print_info: n_rot = 128 print_info: n_swa = 0 print_info: is_swa_any = 0 print_info: non_causal_type = 0 print_info: n_embd_head_k = 128 print_info: n_embd_head_v = 128 print_info: n_gqa = 5 print_info: n_embd_k_gqa = 1024 print_info: n_embd_v_gqa = 1024 print_info: f_norm_eps = 0.0e+00 print_info: f_norm_rms_eps = 1.0e-05 print_info: f_clamp_kqv = 0.0e+00 print_info: f_max_alibi_bias = 0.0e+00 print_info: f_logit_scale = 0.0e+00 print_info: f_attn_scale = 0.0e+00 print_info: f_attn_value_scale = 0.0000 print_info: n_ff = 13824 print_info: n_expert = 0 print_info: n_expert_used = 0 print_info: n_expert_groups = 0 print_info: n_group_used = 0 print_info: causal attn = 1 print_info: pooling type = -1 print_info: rope type = 2 print_info: rope scaling = linear print_info: freq_base_train = 1000000.0 print_info: freq_scale_train = 1 print_info: n_ctx_orig_yarn = 131072 print_info: rope_yarn_log_mul = 0.0000 print_info: rope_finetuned = unknown print_info: model type = 14B print_info: model params = 14.77 B print_info: general.name = DeepSeek R1 Distill Qwen 14B print_info: vocab type = BPE print_info: n_vocab = 152064 print_info: n_merges = 151387 print_info: BOS token = 151646 '<|begin▁of▁sentence|>' print_info: EOS token = 151643 '<|end▁of▁sentence|>' print_info: EOT token = 151643 '<|end▁of▁sentence|>' print_info: PAD token = 151643 '<|end▁of▁sentence|>' print_info: LF token = 198 'Ċ' print_info: FIM PRE token = 151659 '<|fim_prefix|>' print_info: FIM SUF token = 151661 '<|fim_suffix|>' print_info: FIM MID token = 151660 '<|fim_middle|>' print_info: FIM PAD token = 151662 '<|fim_pad|>' print_info: FIM REP token = 151663 '<|repo_name|>' print_info: FIM SEP token = 151664 '<|file_sep|>' print_info: EOG token = 128247 '' print_info: EOG token = 151643 '<|end▁of▁sentence|>' print_info: EOG token = 151662 '<|fim_pad|>' print_info: EOG token = 151663 '<|repo_name|>' print_info: EOG token = 151664 '<|file_sep|>' print_info: max token length = 256 load_tensors: loading model tensors, this can take a while... (load_mode = none) load_tensors: CPU model buffer size = 2457.29 MiB load_tensors: CPU_REPACK model buffer size = 6108.75 MiB cmn common_init_: added logit bias = -inf cmn common_init_: added <|end▁of▁sentence|> logit bias = -inf cmn common_init_: added <|fim_pad|> logit bias = -inf cmn common_init_: added <|repo_name|> logit bias = -inf cmn common_init_: added <|file_sep|> logit bias = -inf llama_context: constructing llama_context llama_context: n_seq_max = 1 llama_context: n_ctx = 8192 llama_context: n_ctx_seq = 8192 llama_context: n_batch = 1024 llama_context: n_ubatch = 1024 llama_context: causal_attn = 1 llama_context: flash_attn = auto llama_context: kv_unified = false llama_context: freq_base = 1000000.0 llama_context: freq_scale = 1 llama_context: n_rs_seq = 0 llama_context: n_outputs_max = 1 llama_context: n_outputs_max_per_seq = 1 llama_context: n_ctx_seq (8192) < n_ctx_train (131072) -- the full capacity of the model will not be utilized llama_context: CPU output buffer size = 0.58 MiB llama_kv_cache: CPU KV buffer size = 1536.00 MiB llama_kv_cache: size = 1536.00 MiB ( 8192 cells, 48 layers, 1/1 seqs), K (f16): 768.00 MiB, V (f16): 768.00 MiB llama_kv_cache: attn_rot_k = 0, n_embd_head_k_all = 128 llama_kv_cache: attn_rot_v = 0, n_embd_head_k_all = 128 sched_reserve: reserving ... resolve_fused_ops: Flash Attention enabled resolve_fused_ops: resolving fused DeepSeek V4 HC support: resolve_fused_ops: fused DeepSeek V4 HC pre enabled resolve_fused_ops: fused DeepSeek V4 HC comb enabled resolve_fused_ops: fused DeepSeek V4 HC post enabled sched_reserve: CPU compute buffer size = 258.02 MiB sched_reserve: graph nodes = 1638 sched_reserve: graph splits = 1 sched_reserve: reserve took 3.92 ms, sched copies = 1 cmn init: llama threadpool init, n_threads = 16 cmn common_init_: warming up the model with an empty run - please wait ... (--no-warmup to disable) srv load_model: initializing, n_slots = 1, n_ctx_slot = 8192, kv_unified = 'false' spec common_specu: no implementations specified for speculative decoding slot load_model: id 0 | task -1 | new slot, n_ctx = 8192 srv load_model: prompt cache is enabled, size limit: 8192 MiB srv load_model: use `--cache-ram 0` to disable the prompt cache srv load_model: for more info see https://github.com/ggml-org/llama.cpp/pull/16391 srv load_model: context checkpoints enabled, max = 32, min spacing = 8192 srv init: idle slots will be saved to prompt cache upon starting a new task srv init: init: chat template, example_format: 'You are a helpful assistant<|User|>Hello<|Assistant|>Hi there<|end▁of▁sentence|><|User|>How are you?<|Assistant|>' srv init: init: chat template, thinking = 1 srv init: preserve_reasoning kwarg: not supported by template srv llama_server: model loaded srv llama_server: listening on http://127.0.0.1:41669 srv update_slots: all slots are idle time=2026-09-17T02:24:32.351+02:00 level=INFO source=llama_server.go:1372 msg="llama-server started in 11.54 seconds" time=2026-09-17T02:24:32.351+02:00 level=INFO source=images.go:382 msg="template selection" model=registry.ollama.ai/library/deepseek-r1:14b selected=gguf_chat_template renderer="" parser="" go_template="[completion thinking]" chat_template="[tools thinking completion]" harmony=null renderer_parser=null time=2026-09-17T02:24:32.351+02:00 level=INFO source=sched.go:733 msg="loaded runners" count=1 time=2026-09-17T02:24:32.351+02:00 level=INFO source=llama_server.go:1305 msg="waiting for llama-server to start responding" time=2026-09-17T02:24:32.352+02:00 level=INFO source=llama_server.go:1372 msg="llama-server started in 11.54 seconds" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - skipping, slot is empty slot get_availabl: id 0 | task -1 | selected slot by LRU, t_last = -1 srv get_availabl: updating prompt cache srv load: - looking for better prompt, base f_keep = -1.000, f_sim = 0.000 srv update: - cache state: 0 prompts, 0.000 MiB (limits: 8192.000 MiB, 8192 tokens, 8589934592 est) srv get_availabl: prompt cache update took 0.01 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 0 | processing task, is_child = 0 slot operator(): id 0 | task 0 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 19 slot operator(): id 0 | task 0 | cached n_tokens = 0, memory_seq_rm [0, end) slot init_sampler: id 0 | task 0 | init sampler, took 0.01 ms, tokens: text = 19, total = 19 slot print_timing: id 0 | task 0 | n_gen = 100, tg = 3.97 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 0 | n_gen = 112, tg = 3.98 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 0 | n_gen = 124, tg = 3.94 t/s, tg_3s = 3.67 t/s slot print_timing: id 0 | task 0 | n_gen = 136, tg = 3.94 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 0 | n_gen = 148, tg = 3.95 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 0 | n_gen = 161, tg = 3.95 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 0 | n_gen = 173, tg = 3.95 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 0 | n_gen = 186, tg = 3.96 t/s, tg_3s = 4.06 t/s slot print_timing: id 0 | task 0 | n_gen = 199, tg = 3.96 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 0 | n_gen = 212, tg = 3.97 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 0 | n_gen = 224, tg = 3.97 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 0 | prompt eval time = 527.54 ms / 19 tokens ( 27.77 ms per token, 36.02 tokens per second) slot print_timing: id 0 | task 0 | eval time = 58688.92 ms / 234 tokens ( 251.88 ms per token, 3.97 tokens per second) slot print_timing: id 0 | task 0 | total time = 59216.45 ms / 253 tokens slot print_timing: id 0 | task 0 | graphs reused = 233 slot release: id 0 | task 0 | stop processing: n_tokens = 252, truncated = 0 srv update_slots: all slots are idle [GIN] 2026/09/17 - 02:25:31 | 200 | 1m10s | 127.0.0.1 | POST "/v1/chat/completions" time=2026-09-17T02:30:07.498+02:00 level=INFO source=routes.go:1941 msg="server config" env="map[CUDA_VISIBLE_DEVICES: GGML_VK_VISIBLE_DEVICES: GPU_DEVICE_ORDINAL: HIP_VISIBLE_DEVICES: HSA_OVERRIDE_GFX_VERSION: HTTPS_PROXY: HTTP_PROXY: LLAMA_ARG_FIT: LLAMA_ARG_FIT_TARGET: NO_PROXY: OLLAMA_CONTEXT_LENGTH:8192 OLLAMA_CREATE_REMOTE:false OLLAMA_DEBUG:INFO OLLAMA_DEBUG_LOG_REQUESTS:false OLLAMA_EDITOR: OLLAMA_FLASH_ATTENTION:false OLLAMA_GO_TEMPLATE:true OLLAMA_GPU_OVERHEAD:0 OLLAMA_HOST:http://127.0.0.1:11550 OLLAMA_IGPU_ENABLE: OLLAMA_KEEP_ALIVE:30m0s OLLAMA_KV_CACHE_TYPE: OLLAMA_LLM_LIBRARY: OLLAMA_LOAD_TIMEOUT:5m0s OLLAMA_MAX_LOADED_MODELS:1 OLLAMA_MAX_QUEUE:512 OLLAMA_MAX_TRANSFER_STREAMS:4 OLLAMA_MODELS:/work/acb/my-deepseek/models OLLAMA_NOHISTORY:false OLLAMA_NOPRUNE:false OLLAMA_NO_CLOUD:true OLLAMA_NUM_PARALLEL:1 OLLAMA_ORIGINS:[http://localhost https://localhost http://localhost:* https://localhost:* http://127.0.0.1 https://127.0.0.1 http://127.0.0.1:* https://127.0.0.1:* http://0.0.0.0 https://0.0.0.0 http://0.0.0.0:* https://0.0.0.0:* app://* file://* tauri://* vscode-webview://* vscode-file://*] OLLAMA_REMOTES:[ollama.com] OLLAMA_SCHED_SPREAD:false OLLAMA_VULKAN:true ROCR_VISIBLE_DEVICES: http_proxy: https_proxy: no_proxy:]" time=2026-09-17T02:30:07.498+02:00 level=INFO source=routes.go:1943 msg="Ollama cloud disabled: true" time=2026-09-17T02:30:07.499+02:00 level=INFO source=images.go:929 msg="total blobs: 0" time=2026-09-17T02:30:07.499+02:00 level=INFO source=images.go:937 msg="total unused blobs removed: 0" time=2026-09-17T02:30:07.499+02:00 level=INFO source=routes.go:2001 msg="Listening on 127.0.0.1:11550 (version 0.34.1)" time=2026-09-17T02:30:07.499+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h26m27.021718695s consecutive_failures=0 time=2026-09-17T02:30:07.500+02:00 level=INFO source=runner.go:60 msg="discovering available GPUs..." time=2026-09-17T02:30:07.664+02:00 level=INFO source=types.go:50 msg="inference compute" id=cpu library=cpu compute="" name=cpu description=cpu libdirs=ollama driver="" pci_id="" type="" total="125.2 GiB" available="41.1 GiB" time=2026-09-17T02:30:07.664+02:00 level=INFO source=routes.go:2051 msg="vram-based default context" total_vram="0 B" default_num_ctx=4096 [GIN] 2026/09/17 - 02:30:08 | 200 | 54.341µs | 127.0.0.1 | GET "/api/version" [GIN] 2026/09/17 - 02:30:20 | 200 | 31.518µs | 127.0.0.1 | HEAD "/" [GIN] 2026/09/17 - 02:30:20 | 200 | 678.656µs | 127.0.0.1 | GET "/api/tags" [GIN] 2026/09/17 - 02:30:21 | 200 | 22.652µs | 127.0.0.1 | HEAD "/" [GIN] 2026/09/17 - 02:30:21 | 200 | 67.556µs | 127.0.0.1 | GET "/api/ps" time=2026-09-17T02:30:51.519+02:00 level=INFO source=sched.go:1152 msg="disabling mmap for llama-server load by default" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e reason=cpu time=2026-09-17T02:30:51.519+02:00 level=INFO source=server.go:100 msg="using llama-server for model" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e time=2026-09-17T02:30:51.520+02:00 level=INFO source=llama_server.go:434 msg="starting llama-server" cmd="/work/acb/my-deepseek/runtime/lib/ollama/llama-server --model /work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e --port 1175 --host 127.0.0.1 --no-webui --offline -c 8192 -np 1 --log-verbosity 4 --no-log-prefix --no-log-timestamps --load-mode none --flash-attn auto -b 1024 -ub 1024 --context-shift --keep 4" time=2026-09-17T02:30:51.520+02:00 level=INFO source=sched.go:618 msg="system memory" total="125.2 GiB" free="41.1 GiB" free_swap="4.0 KiB" time=2026-09-17T02:30:51.520+02:00 level=INFO source=llama_server.go:1052 msg="loading model via llama-server" model=/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e time=2026-09-17T02:30:51.520+02:00 level=INFO source=llama_server.go:1305 msg="waiting for llama-server to start responding" time=2026-09-17T02:30:51.520+02:00 level=INFO source=llama_server.go:1360 msg="waiting for llama-server to become available" status="llm server not responding" cmn common_param: common_params_print_info: build 1 (5d806aa25) with GNU 13.3.1 for Linux x86_64 cmn common_param: common_params_print_info: verbosity = 4 (adjust with the `-lv N` CLI arg) cmn common_param: device_info: cmn common_param: - CPU : AMD Ryzen 9 5950X 16-Core Processor (128213 MiB, 128213 MiB free) cmn common_param: system_info: n_threads = 16 (n_threads_batch = 16) / 32 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | BMI2 = 1 | LLAMAFILE = 1 | REPACK = 1 | srv init: using 31 threads for HTTP server srv init: The UI is disabled srv init: Use --ui/--no-ui (or deprecated --webui/--no-webui) to enable/disable srv llama_server: ----------------- srv llama_server: CORS is set to allow all origins ('*') and no API key is set srv llama_server: this can be a security risk (cross-origin attacks) srv llama_server: more info: https://github.com/ggml-org/llama.cpp/pull/25655 srv llama_server: ----------------- srv start: binding port with default address family srv load_model: loading model '/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e' srv load_model: local path '/work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e' cmn common_init_: fitting params to device memory ... cmn common_init_: (for bugs during this step try to reproduce them with -fit off, or provide --verbose logs if the bug only occurs with -fit on) common_params_fit_impl: getting device memory data for initial parameters: common_memory_breakdown_print: | memory breakdown [MiB] | total free self model context compute unaccounted | common_memory_breakdown_print: | - Host | 4251 = 2457 + 1536 + 258 | common_memory_breakdown_print: | - CPU_REPACK | 6108 = 6108 + 0 + 0 | common_params_fit_impl: projected to use 4251 MiB of host memory vs. 128213 MiB of total host memory common_params_fit_impl: will leave 123962 >= 1024 MiB of system memory, no changes needed common_fit_params: successfully fit params to free device memory common_fit_params: fitting params to free memory took 0.19 seconds llama_model_loader: loaded meta data with 26 key-value pairs and 579 tensors from /work/acb/my-deepseek/models/blobs/sha256-6e9f90f02bb3b39b59e81916e8cfce9deb45aeaeb9a54a5be4414486b907dc1e (version GGUF V3 (latest)) llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output. llama_model_loader: - kv 0: general.architecture str = qwen2 llama_model_loader: - kv 1: general.type str = model llama_model_loader: - kv 2: general.name str = DeepSeek R1 Distill Qwen 14B llama_model_loader: - kv 3: general.basename str = DeepSeek-R1-Distill-Qwen llama_model_loader: - kv 4: general.size_label str = 14B llama_model_loader: - kv 5: qwen2.block_count u32 = 48 llama_model_loader: - kv 6: qwen2.context_length u32 = 131072 llama_model_loader: - kv 7: qwen2.embedding_length u32 = 5120 llama_model_loader: - kv 8: qwen2.feed_forward_length u32 = 13824 llama_model_loader: - kv 9: qwen2.attention.head_count u32 = 40 llama_model_loader: - kv 10: qwen2.attention.head_count_kv u32 = 8 llama_model_loader: - kv 11: qwen2.rope.freq_base f32 = 1000000.000000 llama_model_loader: - kv 12: qwen2.attention.layer_norm_rms_epsilon f32 = 0.000010 llama_model_loader: - kv 13: general.file_type u32 = 15 llama_model_loader: - kv 14: tokenizer.ggml.model str = gpt2 llama_model_loader: - kv 15: tokenizer.ggml.pre str = qwen2 llama_model_loader: - kv 16: tokenizer.ggml.tokens arr[str,152064] = ["!", "\"", "#", "$", "%", "&", "'", ... llama_model_loader: - kv 17: tokenizer.ggml.token_type arr[i32,152064] = [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, ... llama_model_loader: - kv 18: tokenizer.ggml.merges arr[str,151387] = ["Ġ Ġ", "ĠĠ ĠĠ", "i n", "Ġ t",... llama_model_loader: - kv 19: tokenizer.ggml.bos_token_id u32 = 151646 llama_model_loader: - kv 20: tokenizer.ggml.eos_token_id u32 = 151643 llama_model_loader: - kv 21: tokenizer.ggml.padding_token_id u32 = 151643 llama_model_loader: - kv 22: tokenizer.ggml.add_bos_token bool = true llama_model_loader: - kv 23: tokenizer.ggml.add_eos_token bool = false llama_model_loader: - kv 24: tokenizer.chat_template str = {% if not add_generation_prompt is de... llama_model_loader: - kv 25: general.quantization_version u32 = 2 llama_model_loader: - type f32: 241 tensors llama_model_loader: - type q4_K: 289 tensors llama_model_loader: - type q6_K: 49 tensors print_info: file format = GGUF V3 (latest) print_info: file type = Q4_K - Medium print_info: file size = 8.37 GiB (4.87 BPW) time=2026-09-17T02:30:51.772+02:00 level=INFO source=llama_server.go:1360 msg="waiting for llama-server to become available" status="llm server loading model" load: 0 unused tokens load: control-looking token: 128247 '' was not control-type; this is probably a bug in the model. its type will be overridden load: printing all EOG tokens: load: - 128247 ('') load: - 151643 ('<|end▁of▁sentence|>') load: - 151662 ('<|fim_pad|>') load: - 151663 ('<|repo_name|>') load: - 151664 ('<|file_sep|>') load: special tokens cache size = 23 load: token to piece cache size = 0.9310 MB print_info: arch = qwen2 print_info: vocab_only = 0 print_info: no_alloc = 0 print_info: n_ctx_train = 131072 print_info: n_embd_inp = 5120 print_info: n_embd = 5120 print_info: n_embd_out = 5120 print_info: n_layer = 48 print_info: n_layer_all = 48 print_info: n_head = 40 print_info: n_head_kv = 8 print_info: n_rot = 128 print_info: n_swa = 0 print_info: is_swa_any = 0 print_info: non_causal_type = 0 print_info: n_embd_head_k = 128 print_info: n_embd_head_v = 128 print_info: n_gqa = 5 print_info: n_embd_k_gqa = 1024 print_info: n_embd_v_gqa = 1024 print_info: f_norm_eps = 0.0e+00 print_info: f_norm_rms_eps = 1.0e-05 print_info: f_clamp_kqv = 0.0e+00 print_info: f_max_alibi_bias = 0.0e+00 print_info: f_logit_scale = 0.0e+00 print_info: f_attn_scale = 0.0e+00 print_info: f_attn_value_scale = 0.0000 print_info: n_ff = 13824 print_info: n_expert = 0 print_info: n_expert_used = 0 print_info: n_expert_groups = 0 print_info: n_group_used = 0 print_info: causal attn = 1 print_info: pooling type = -1 print_info: rope type = 2 print_info: rope scaling = linear print_info: freq_base_train = 1000000.0 print_info: freq_scale_train = 1 print_info: n_ctx_orig_yarn = 131072 print_info: rope_yarn_log_mul = 0.0000 print_info: rope_finetuned = unknown print_info: model type = 14B print_info: model params = 14.77 B print_info: general.name = DeepSeek R1 Distill Qwen 14B print_info: vocab type = BPE print_info: n_vocab = 152064 print_info: n_merges = 151387 print_info: BOS token = 151646 '<|begin▁of▁sentence|>' print_info: EOS token = 151643 '<|end▁of▁sentence|>' print_info: EOT token = 151643 '<|end▁of▁sentence|>' print_info: PAD token = 151643 '<|end▁of▁sentence|>' print_info: LF token = 198 'Ċ' print_info: FIM PRE token = 151659 '<|fim_prefix|>' print_info: FIM SUF token = 151661 '<|fim_suffix|>' print_info: FIM MID token = 151660 '<|fim_middle|>' print_info: FIM PAD token = 151662 '<|fim_pad|>' print_info: FIM REP token = 151663 '<|repo_name|>' print_info: FIM SEP token = 151664 '<|file_sep|>' print_info: EOG token = 128247 '' print_info: EOG token = 151643 '<|end▁of▁sentence|>' print_info: EOG token = 151662 '<|fim_pad|>' print_info: EOG token = 151663 '<|repo_name|>' print_info: EOG token = 151664 '<|file_sep|>' print_info: max token length = 256 load_tensors: loading model tensors, this can take a while... (load_mode = none) load_tensors: CPU model buffer size = 2457.29 MiB load_tensors: CPU_REPACK model buffer size = 6108.75 MiB cmn common_init_: added logit bias = -inf cmn common_init_: added <|end▁of▁sentence|> logit bias = -inf cmn common_init_: added <|fim_pad|> logit bias = -inf cmn common_init_: added <|repo_name|> logit bias = -inf cmn common_init_: added <|file_sep|> logit bias = -inf llama_context: constructing llama_context llama_context: n_seq_max = 1 llama_context: n_ctx = 8192 llama_context: n_ctx_seq = 8192 llama_context: n_batch = 1024 llama_context: n_ubatch = 1024 llama_context: causal_attn = 1 llama_context: flash_attn = auto llama_context: kv_unified = false llama_context: freq_base = 1000000.0 llama_context: freq_scale = 1 llama_context: n_rs_seq = 0 llama_context: n_outputs_max = 1 llama_context: n_outputs_max_per_seq = 1 llama_context: n_ctx_seq (8192) < n_ctx_train (131072) -- the full capacity of the model will not be utilized llama_context: CPU output buffer size = 0.58 MiB llama_kv_cache: CPU KV buffer size = 1536.00 MiB llama_kv_cache: size = 1536.00 MiB ( 8192 cells, 48 layers, 1/1 seqs), K (f16): 768.00 MiB, V (f16): 768.00 MiB llama_kv_cache: attn_rot_k = 0, n_embd_head_k_all = 128 llama_kv_cache: attn_rot_v = 0, n_embd_head_k_all = 128 sched_reserve: reserving ... resolve_fused_ops: Flash Attention enabled resolve_fused_ops: resolving fused DeepSeek V4 HC support: resolve_fused_ops: fused DeepSeek V4 HC pre enabled resolve_fused_ops: fused DeepSeek V4 HC comb enabled resolve_fused_ops: fused DeepSeek V4 HC post enabled sched_reserve: CPU compute buffer size = 258.02 MiB sched_reserve: graph nodes = 1638 sched_reserve: graph splits = 1 sched_reserve: reserve took 3.98 ms, sched copies = 1 cmn init: llama threadpool init, n_threads = 16 cmn common_init_: warming up the model with an empty run - please wait ... (--no-warmup to disable) srv load_model: initializing, n_slots = 1, n_ctx_slot = 8192, kv_unified = 'false' spec common_specu: no implementations specified for speculative decoding slot load_model: id 0 | task -1 | new slot, n_ctx = 8192 srv load_model: prompt cache is enabled, size limit: 8192 MiB srv load_model: use `--cache-ram 0` to disable the prompt cache srv load_model: for more info see https://github.com/ggml-org/llama.cpp/pull/16391 srv load_model: context checkpoints enabled, max = 32, min spacing = 8192 srv init: idle slots will be saved to prompt cache upon starting a new task srv init: init: chat template, example_format: 'You are a helpful assistant<|User|>Hello<|Assistant|>Hi there<|end▁of▁sentence|><|User|>How are you?<|Assistant|>' srv init: init: chat template, thinking = 1 srv init: preserve_reasoning kwarg: not supported by template srv llama_server: model loaded srv llama_server: listening on http://127.0.0.1:1175 srv update_slots: all slots are idle time=2026-09-17T02:31:03.566+02:00 level=INFO source=llama_server.go:1372 msg="llama-server started in 12.05 seconds" time=2026-09-17T02:31:03.566+02:00 level=INFO source=images.go:382 msg="template selection" model=registry.ollama.ai/library/deepseek-r1:14b selected=gguf_chat_template renderer="" parser="" go_template="[completion thinking]" chat_template="[tools thinking completion]" harmony=null renderer_parser=null time=2026-09-17T02:31:03.566+02:00 level=INFO source=sched.go:733 msg="loaded runners" count=1 time=2026-09-17T02:31:03.566+02:00 level=INFO source=llama_server.go:1305 msg="waiting for llama-server to start responding" time=2026-09-17T02:31:03.567+02:00 level=INFO source=llama_server.go:1372 msg="llama-server started in 12.05 seconds" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - skipping, slot is empty slot get_availabl: id 0 | task -1 | selected slot by LRU, t_last = -1 srv get_availabl: updating prompt cache srv load: - looking for better prompt, base f_keep = -1.000, f_sim = 0.000 srv update: - cache state: 0 prompts, 0.000 MiB (limits: 8192.000 MiB, 8192 tokens, 8589934592 est) srv get_availabl: prompt cache update took 0.01 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 0 | processing task, is_child = 0 slot operator(): id 0 | task 0 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 4 slot operator(): id 0 | task 0 | cached n_tokens = 0, memory_seq_rm [0, end) slot init_sampler: id 0 | task 0 | init sampler, took 0.01 ms, tokens: text = 4, total = 4 slot print_timing: id 0 | task 0 | prompt eval time = 275.35 ms / 4 tokens ( 68.84 ms per token, 14.53 tokens per second) slot print_timing: id 0 | task 0 | eval time = 3682.69 ms / 16 tokens ( 245.51 ms per token, 4.07 tokens per second) slot print_timing: id 0 | task 0 | total time = 3958.04 ms / 20 tokens slot print_timing: id 0 | task 0 | graphs reused = 15 slot release: id 0 | task 0 | stop processing: n_tokens = 19, truncated = 0 srv update_slots: all slots are idle [GIN] 2026/09/17 - 02:31:07 | 200 | 16.037905079s | 127.0.0.1 | POST "/v1/chat/completions" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - checking sim = 0.286 (2/7) > 0.100 slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.286 (> 0.100 thold), f_keep = 0.105 srv get_availabl: updating prompt cache srv prompt_save: - saving prompt with length 19, total state size = 3.564 MiB (draft: 0.000 MiB) srv load: - looking for better prompt, base f_keep = 0.105, f_sim = 0.286 srv load: - prompt with length 19, lcp = 2, f_keep = 0.105, f_sim = 0.286 srv update: - cache state: 1 prompts, 3.564 MiB (limits: 8192.000 MiB, 8192 tokens, 43674 est) srv update: - prompt 0x2fec280: 19 tokens, checkpoints: 0, 3.564 MiB srv get_availabl: prompt cache update took 0.49 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 17 | processing task, is_child = 0 slot operator(): id 0 | task 17 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 7 slot operator(): id 0 | task 17 | cached n_tokens = 2, memory_seq_rm [2, end) slot init_sampler: id 0 | task 17 | init sampler, took 0.00 ms, tokens: text = 7, total = 7 slot print_timing: id 0 | task 17 | prompt eval time = 277.33 ms / 5 tokens ( 55.47 ms per token, 18.03 tokens per second) slot print_timing: id 0 | task 17 | eval time = 10691.08 ms / 44 tokens ( 248.63 ms per token, 4.02 tokens per second) slot print_timing: id 0 | task 17 | total time = 10968.41 ms / 49 tokens slot print_timing: id 0 | task 17 | graphs reused = 57 slot release: id 0 | task 17 | stop processing: n_tokens = 50, truncated = 0 srv update_slots: all slots are idle [GIN] 2026/09/17 - 02:32:48 | 200 | 10.972985869s | 127.0.0.1 | POST "/v1/chat/completions" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - checking sim = 0.222 (2/9) > 0.100 slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.222 (> 0.100 thold), f_keep = 0.040 srv get_availabl: updating prompt cache srv prompt_save: - saving prompt with length 50, total state size = 9.377 MiB (draft: 0.000 MiB) srv load: - looking for better prompt, base f_keep = 0.040, f_sim = 0.222 srv load: - prompt with length 19, lcp = 2, f_keep = 0.105, f_sim = 0.222 srv load: - prompt with length 50, lcp = 2, f_keep = 0.040, f_sim = 0.222 srv update: - cache state: 2 prompts, 12.941 MiB (limits: 8192.000 MiB, 8192 tokens, 43680 est) srv update: - prompt 0x2fec280: 19 tokens, checkpoints: 0, 3.564 MiB srv update: - prompt 0x2ff1330: 50 tokens, checkpoints: 0, 9.377 MiB srv get_availabl: prompt cache update took 2.54 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 62 | processing task, is_child = 0 slot operator(): id 0 | task 62 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 9 slot operator(): id 0 | task 62 | cached n_tokens = 2, memory_seq_rm [2, end) slot init_sampler: id 0 | task 62 | init sampler, took 0.00 ms, tokens: text = 9, total = 9 slot print_timing: id 0 | task 62 | prompt eval time = 310.08 ms / 7 tokens ( 44.30 ms per token, 22.57 tokens per second) slot print_timing: id 0 | task 62 | eval time = 10104.74 ms / 42 tokens ( 246.46 ms per token, 4.06 tokens per second) slot print_timing: id 0 | task 62 | total time = 10414.82 ms / 49 tokens slot print_timing: id 0 | task 62 | graphs reused = 97 slot release: id 0 | task 62 | stop processing: n_tokens = 50, truncated = 0 srv update_slots: all slots are idle [GIN] 2026/09/17 - 02:33:55 | 200 | 10.421812795s | 127.0.0.1 | POST "/v1/chat/completions" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - checking sim = 0.143 (2/14) > 0.100 slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.143 (> 0.100 thold), f_keep = 0.040 srv get_availabl: updating prompt cache srv prompt_save: - saving prompt with length 50, total state size = 9.377 MiB (draft: 0.000 MiB) srv load: - looking for better prompt, base f_keep = 0.040, f_sim = 0.143 srv load: - prompt with length 19, lcp = 2, f_keep = 0.105, f_sim = 0.143 srv load: - prompt with length 50, lcp = 2, f_keep = 0.040, f_sim = 0.143 srv load: - prompt with length 50, lcp = 2, f_keep = 0.040, f_sim = 0.143 srv update: - cache state: 3 prompts, 22.317 MiB (limits: 8192.000 MiB, 8192 tokens, 43681 est) srv update: - prompt 0x2fec280: 19 tokens, checkpoints: 0, 3.564 MiB srv update: - prompt 0x2ff1330: 50 tokens, checkpoints: 0, 9.377 MiB srv update: - prompt 0x2fe3fd0: 50 tokens, checkpoints: 0, 9.377 MiB srv get_availabl: prompt cache update took 4.44 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 105 | processing task, is_child = 0 slot operator(): id 0 | task 105 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 14 slot operator(): id 0 | task 105 | cached n_tokens = 2, memory_seq_rm [2, end) slot init_sampler: id 0 | task 105 | init sampler, took 0.00 ms, tokens: text = 14, total = 14 slot print_timing: id 0 | task 105 | n_gen = 100, tg = 4.06 t/s, tg_3s = 4.10 t/s slot print_timing: id 0 | task 105 | n_gen = 113, tg = 4.07 t/s, tg_3s = 4.11 t/s slot print_timing: id 0 | task 105 | n_gen = 126, tg = 4.06 t/s, tg_3s = 4.02 t/s slot print_timing: id 0 | task 105 | n_gen = 139, tg = 4.06 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 105 | n_gen = 152, tg = 4.05 t/s, tg_3s = 4.02 t/s slot print_timing: id 0 | task 105 | n_gen = 165, tg = 4.05 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 105 | n_gen = 178, tg = 4.05 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 105 | n_gen = 191, tg = 4.05 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 105 | n_gen = 204, tg = 4.05 t/s, tg_3s = 4.02 t/s slot print_timing: id 0 | task 105 | n_gen = 216, tg = 4.04 t/s, tg_3s = 3.94 t/s slot print_timing: id 0 | task 105 | n_gen = 229, tg = 4.04 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 105 | n_gen = 242, tg = 4.04 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 105 | n_gen = 254, tg = 4.04 t/s, tg_3s = 3.98 t/s slot print_timing: id 0 | task 105 | n_gen = 266, tg = 4.03 t/s, tg_3s = 3.98 t/s slot print_timing: id 0 | task 105 | n_gen = 278, tg = 4.03 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 105 | n_gen = 290, tg = 4.03 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 105 | n_gen = 302, tg = 4.02 t/s, tg_3s = 3.89 t/s slot print_timing: id 0 | task 105 | n_gen = 315, tg = 4.02 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 105 | n_gen = 327, tg = 4.02 t/s, tg_3s = 3.96 t/s slot print_timing: id 0 | task 105 | n_gen = 339, tg = 4.02 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 105 | n_gen = 351, tg = 4.01 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 105 | n_gen = 363, tg = 4.01 t/s, tg_3s = 3.98 t/s slot print_timing: id 0 | task 105 | n_gen = 375, tg = 4.01 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 105 | n_gen = 387, tg = 4.01 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 105 | n_gen = 399, tg = 4.00 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 105 | n_gen = 411, tg = 4.00 t/s, tg_3s = 3.96 t/s slot print_timing: id 0 | task 105 | n_gen = 423, tg = 4.00 t/s, tg_3s = 3.98 t/s slot print_timing: id 0 | task 105 | n_gen = 435, tg = 4.00 t/s, tg_3s = 3.91 t/s slot print_timing: id 0 | task 105 | n_gen = 447, tg = 4.00 t/s, tg_3s = 3.98 t/s slot print_timing: id 0 | task 105 | n_gen = 459, tg = 4.00 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 105 | n_gen = 472, tg = 4.00 t/s, tg_3s = 4.00 t/s slot print_timing: id 0 | task 105 | n_gen = 484, tg = 3.99 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 105 | n_gen = 496, tg = 3.99 t/s, tg_3s = 3.94 t/s slot print_timing: id 0 | task 105 | n_gen = 508, tg = 3.99 t/s, tg_3s = 3.94 t/s slot print_timing: id 0 | task 105 | n_gen = 520, tg = 3.99 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 105 | n_gen = 532, tg = 3.99 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 105 | n_gen = 544, tg = 3.99 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 105 | n_gen = 556, tg = 3.99 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 105 | n_gen = 568, tg = 3.98 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 105 | n_gen = 580, tg = 3.98 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 105 | n_gen = 592, tg = 3.98 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 105 | n_gen = 604, tg = 3.98 t/s, tg_3s = 3.91 t/s slot print_timing: id 0 | task 105 | n_gen = 616, tg = 3.98 t/s, tg_3s = 3.87 t/s slot print_timing: id 0 | task 105 | n_gen = 628, tg = 3.98 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 105 | n_gen = 640, tg = 3.98 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 105 | n_gen = 652, tg = 3.98 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 105 | n_gen = 664, tg = 3.98 t/s, tg_3s = 3.96 t/s slot print_timing: id 0 | task 105 | n_gen = 676, tg = 3.97 t/s, tg_3s = 3.91 t/s slot print_timing: id 0 | task 105 | n_gen = 688, tg = 3.97 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 105 | n_gen = 700, tg = 3.97 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 105 | n_gen = 712, tg = 3.97 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 105 | n_gen = 724, tg = 3.97 t/s, tg_3s = 3.89 t/s slot print_timing: id 0 | task 105 | n_gen = 736, tg = 3.97 t/s, tg_3s = 3.83 t/s slot print_timing: id 0 | task 105 | n_gen = 748, tg = 3.97 t/s, tg_3s = 3.91 t/s slot print_timing: id 0 | task 105 | n_gen = 760, tg = 3.96 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 105 | n_gen = 772, tg = 3.96 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 105 | n_gen = 784, tg = 3.96 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 105 | n_gen = 796, tg = 3.96 t/s, tg_3s = 3.95 t/s slot print_timing: id 0 | task 105 | n_gen = 808, tg = 3.96 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 105 | n_gen = 820, tg = 3.96 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 105 | n_gen = 832, tg = 3.96 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 105 | n_gen = 844, tg = 3.96 t/s, tg_3s = 3.85 t/s [GIN] 2026/09/17 - 02:38:22 | 500 | 3m33s | 127.0.0.1 | POST "/v1/chat/completions" srv stop: cancel task, id_task = 105 slot release: id 0 | task 105 | stop processing: n_tokens = 860, truncated = 0 srv update_slots: all slots are idle srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - checking sim = 1.000 (14/14) > 0.100 slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 1.000 (> 0.100 thold), f_keep = 0.016 srv get_availabl: updating prompt cache srv prompt_save: - saving prompt with length 860, total state size = 161.261 MiB (draft: 0.000 MiB) srv load: - looking for better prompt, base f_keep = 0.016, f_sim = 1.000 srv load: - prompt with length 19, lcp = 2, f_keep = 0.105, f_sim = 0.143 srv load: - prompt with length 50, lcp = 2, f_keep = 0.040, f_sim = 0.143 srv load: - prompt with length 50, lcp = 2, f_keep = 0.040, f_sim = 0.143 srv load: - prompt with length 860, lcp = 14, f_keep = 0.016, f_sim = 1.000 srv update: - cache state: 4 prompts, 183.578 MiB (limits: 8192.000 MiB, 8192 tokens, 43686 est) srv update: - prompt 0x2fec280: 19 tokens, checkpoints: 0, 3.564 MiB srv update: - prompt 0x2ff1330: 50 tokens, checkpoints: 0, 9.377 MiB srv update: - prompt 0x2fe3fd0: 50 tokens, checkpoints: 0, 9.377 MiB srv update: - prompt 0x2ff4740: 860 tokens, checkpoints: 0, 161.261 MiB srv get_availabl: prompt cache update took 77.32 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 954 | processing task, is_child = 0 slot operator(): id 0 | task 954 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 14 slot operator(): id 0 | task 954 | need to evaluate at least 1 token for each active slot (n_past = 14, task.n_tokens() = 14) slot operator(): id 0 | task 954 | n_past was set to 13 slot operator(): id 0 | task 954 | cached n_tokens = 13, memory_seq_rm [13, end) slot init_sampler: id 0 | task 954 | init sampler, took 0.00 ms, tokens: text = 14, total = 14 slot print_timing: id 0 | task 954 | prompt eval time = 242.45 ms / 1 tokens ( 242.45 ms per token, 4.12 tokens per second) slot print_timing: id 0 | task 954 | eval time = 10109.54 ms / 42 tokens ( 246.57 ms per token, 4.06 tokens per second) slot print_timing: id 0 | task 954 | total time = 10351.99 ms / 43 tokens slot print_timing: id 0 | task 954 | graphs reused = 980 slot release: id 0 | task 954 | stop processing: n_tokens = 55, truncated = 0 srv update_slots: all slots are idle [GIN] 2026/09/17 - 02:38:36 | 200 | 10.433511165s | 127.0.0.1 | POST "/v1/chat/completions" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - checking sim = 1.000 (14/14) > 0.100 slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 1.000 (> 0.100 thold), f_keep = 0.255 srv get_availabl: updating prompt cache srv prompt_save: - saving prompt with length 55, total state size = 10.314 MiB (draft: 0.000 MiB) srv load: - looking for better prompt, base f_keep = 0.255, f_sim = 1.000 srv load: - prompt with length 19, lcp = 2, f_keep = 0.105, f_sim = 0.143 srv load: - prompt with length 50, lcp = 2, f_keep = 0.040, f_sim = 0.143 srv load: - prompt with length 50, lcp = 2, f_keep = 0.040, f_sim = 0.143 srv load: - prompt with length 860, lcp = 14, f_keep = 0.016, f_sim = 1.000 srv load: - prompt with length 55, lcp = 14, f_keep = 0.255, f_sim = 1.000 srv update: - cache state: 5 prompts, 193.892 MiB (limits: 8192.000 MiB, 8192 tokens, 43686 est) srv update: - prompt 0x2fec280: 19 tokens, checkpoints: 0, 3.564 MiB srv update: - prompt 0x2ff1330: 50 tokens, checkpoints: 0, 9.377 MiB srv update: - prompt 0x2fe3fd0: 50 tokens, checkpoints: 0, 9.377 MiB srv update: - prompt 0x2ff4740: 860 tokens, checkpoints: 0, 161.261 MiB srv update: - prompt 0x2feaba0: 55 tokens, checkpoints: 0, 10.314 MiB srv get_availabl: prompt cache update took 5.33 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 997 | processing task, is_child = 0 slot operator(): id 0 | task 997 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 14 slot operator(): id 0 | task 997 | need to evaluate at least 1 token for each active slot (n_past = 14, task.n_tokens() = 14) slot operator(): id 0 | task 997 | n_past was set to 13 slot operator(): id 0 | task 997 | cached n_tokens = 13, memory_seq_rm [13, end) slot init_sampler: id 0 | task 997 | init sampler, took 0.00 ms, tokens: text = 14, total = 14 slot print_timing: id 0 | task 997 | prompt eval time = 246.03 ms / 1 tokens ( 246.03 ms per token, 4.06 tokens per second) slot print_timing: id 0 | task 997 | eval time = 23655.65 ms / 97 tokens ( 246.41 ms per token, 4.06 tokens per second) slot print_timing: id 0 | task 997 | total time = 23901.67 ms / 98 tokens slot print_timing: id 0 | task 997 | graphs reused = 1077 slot release: id 0 | task 997 | stop processing: n_tokens = 110, truncated = 0 srv update_slots: all slots are idle [GIN] 2026/09/17 - 02:39:16 | 200 | 23.911188178s | 127.0.0.1 | POST "/v1/chat/completions" srv server_strea: conv_id= (empty=1) srv operator(): chat format: peg-native slot get_availabl: id 0 | task -1 | - checking sim = 0.143 (2/14) > 0.100 slot get_availabl: id 0 | task -1 | selected slot by LCP similarity, f_sim_best = 0.143 (> 0.100 thold), f_keep = 0.018 srv get_availabl: updating prompt cache srv prompt_save: - saving prompt with length 110, total state size = 20.627 MiB (draft: 0.000 MiB) srv load: - looking for better prompt, base f_keep = 0.018, f_sim = 0.143 srv load: - prompt with length 19, lcp = 2, f_keep = 0.105, f_sim = 0.143 srv load: - prompt with length 50, lcp = 2, f_keep = 0.040, f_sim = 0.143 srv load: - prompt with length 50, lcp = 2, f_keep = 0.040, f_sim = 0.143 srv load: - prompt with length 860, lcp = 2, f_keep = 0.002, f_sim = 0.143 srv load: - prompt with length 55, lcp = 2, f_keep = 0.036, f_sim = 0.143 srv load: - prompt with length 110, lcp = 2, f_keep = 0.018, f_sim = 0.143 srv update: - cache state: 6 prompts, 214.520 MiB (limits: 8192.000 MiB, 8192 tokens, 43686 est) srv update: - prompt 0x2fec280: 19 tokens, checkpoints: 0, 3.564 MiB srv update: - prompt 0x2ff1330: 50 tokens, checkpoints: 0, 9.377 MiB srv update: - prompt 0x2fe3fd0: 50 tokens, checkpoints: 0, 9.377 MiB srv update: - prompt 0x2ff4740: 860 tokens, checkpoints: 0, 161.261 MiB srv update: - prompt 0x2feaba0: 55 tokens, checkpoints: 0, 10.314 MiB srv update: - prompt 0x2fe8dc0: 110 tokens, checkpoints: 0, 20.627 MiB srv get_availabl: prompt cache update took 10.45 ms slot launch_slot_: id 0 | task -1 | sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> ?top-p -> ?min-p -> ?xtc -> ?temp-ext -> dist slot launch_slot_: id 0 | task -1 | sampler params: repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000 dry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = 64 top_k = 40, top_p = 1.000, min_p = 0.000, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 1.000 mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900 slot launch_slot_: id 0 | task 1095 | processing task, is_child = 0 slot operator(): id 0 | task 1095 | new prompt, n_ctx_slot = 8192, n_keep = 4, task.n_tokens = 14 slot operator(): id 0 | task 1095 | cached n_tokens = 2, memory_seq_rm [2, end) slot init_sampler: id 0 | task 1095 | init sampler, took 0.00 ms, tokens: text = 14, total = 14 slot print_timing: id 0 | task 1095 | n_gen = 100, tg = 4.06 t/s, tg_3s = 4.10 t/s slot print_timing: id 0 | task 1095 | n_gen = 113, tg = 4.05 t/s, tg_3s = 4.04 t/s slot print_timing: id 0 | task 1095 | n_gen = 126, tg = 4.06 t/s, tg_3s = 4.08 t/s slot print_timing: id 0 | task 1095 | n_gen = 139, tg = 4.05 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 1095 | n_gen = 152, tg = 4.05 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 1095 | n_gen = 165, tg = 4.05 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 1095 | n_gen = 178, tg = 4.05 t/s, tg_3s = 4.01 t/s slot print_timing: id 0 | task 1095 | n_gen = 191, tg = 4.05 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 1095 | n_gen = 203, tg = 4.04 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 1095 | n_gen = 216, tg = 4.04 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 1095 | n_gen = 228, tg = 4.04 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 1095 | n_gen = 241, tg = 4.04 t/s, tg_3s = 4.00 t/s slot print_timing: id 0 | task 1095 | n_gen = 253, tg = 4.03 t/s, tg_3s = 3.98 t/s slot print_timing: id 0 | task 1095 | n_gen = 266, tg = 4.03 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 1095 | n_gen = 278, tg = 4.03 t/s, tg_3s = 3.97 t/s slot print_timing: id 0 | task 1095 | n_gen = 291, tg = 4.03 t/s, tg_3s = 4.03 t/s slot print_timing: id 0 | task 1095 | n_gen = 303, tg = 4.03 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 1095 | n_gen = 315, tg = 4.02 t/s, tg_3s = 3.94 t/s slot print_timing: id 0 | task 1095 | n_gen = 327, tg = 4.02 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 1095 | n_gen = 339, tg = 4.02 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 1095 | n_gen = 351, tg = 4.02 t/s, tg_3s = 3.96 t/s slot print_timing: id 0 | task 1095 | n_gen = 363, tg = 4.01 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 1095 | n_gen = 375, tg = 4.01 t/s, tg_3s = 3.99 t/s slot print_timing: id 0 | task 1095 | n_gen = 387, tg = 4.01 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 1095 | n_gen = 399, tg = 4.00 t/s, tg_3s = 3.79 t/s slot print_timing: id 0 | task 1095 | n_gen = 410, tg = 3.99 t/s, tg_3s = 3.66 t/s slot print_timing: id 0 | task 1095 | n_gen = 421, tg = 3.98 t/s, tg_3s = 3.66 t/s slot print_timing: id 0 | task 1095 | n_gen = 432, tg = 3.97 t/s, tg_3s = 3.63 t/s slot print_timing: id 0 | task 1095 | n_gen = 444, tg = 3.97 t/s, tg_3s = 3.71 t/s slot print_timing: id 0 | task 1095 | n_gen = 456, tg = 3.96 t/s, tg_3s = 3.68 t/s slot print_timing: id 0 | task 1095 | n_gen = 467, tg = 3.95 t/s, tg_3s = 3.67 t/s slot print_timing: id 0 | task 1095 | n_gen = 478, tg = 3.94 t/s, tg_3s = 3.61 t/s slot print_timing: id 0 | task 1095 | n_gen = 489, tg = 3.93 t/s, tg_3s = 3.65 t/s slot print_timing: id 0 | task 1095 | n_gen = 501, tg = 3.93 t/s, tg_3s = 3.75 t/s slot print_timing: id 0 | task 1095 | n_gen = 513, tg = 3.93 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 1095 | n_gen = 525, tg = 3.93 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 1095 | n_gen = 537, tg = 3.93 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 1095 | n_gen = 549, tg = 3.93 t/s, tg_3s = 3.94 t/s slot print_timing: id 0 | task 1095 | n_gen = 561, tg = 3.93 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 1095 | n_gen = 573, tg = 3.93 t/s, tg_3s = 3.89 t/s slot print_timing: id 0 | task 1095 | n_gen = 585, tg = 3.93 t/s, tg_3s = 3.91 t/s slot print_timing: id 0 | task 1095 | n_gen = 597, tg = 3.93 t/s, tg_3s = 3.94 t/s slot print_timing: id 0 | task 1095 | n_gen = 609, tg = 3.93 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 621, tg = 3.93 t/s, tg_3s = 3.94 t/s slot print_timing: id 0 | task 1095 | n_gen = 632, tg = 3.92 t/s, tg_3s = 3.59 t/s slot print_timing: id 0 | task 1095 | n_gen = 643, tg = 3.91 t/s, tg_3s = 3.53 t/s slot print_timing: id 0 | task 1095 | n_gen = 654, tg = 3.91 t/s, tg_3s = 3.66 t/s slot print_timing: id 0 | task 1095 | n_gen = 666, tg = 3.91 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 1095 | n_gen = 678, tg = 3.91 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 1095 | n_gen = 690, tg = 3.91 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 1095 | n_gen = 702, tg = 3.91 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 1095 | n_gen = 714, tg = 3.91 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 1095 | n_gen = 726, tg = 3.91 t/s, tg_3s = 3.91 t/s slot print_timing: id 0 | task 1095 | n_gen = 738, tg = 3.91 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 750, tg = 3.91 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 1095 | n_gen = 762, tg = 3.91 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 1095 | n_gen = 774, tg = 3.91 t/s, tg_3s = 3.88 t/s slot print_timing: id 0 | task 1095 | n_gen = 786, tg = 3.90 t/s, tg_3s = 3.84 t/s slot print_timing: id 0 | task 1095 | n_gen = 798, tg = 3.90 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 1095 | n_gen = 810, tg = 3.90 t/s, tg_3s = 3.93 t/s slot print_timing: id 0 | task 1095 | n_gen = 822, tg = 3.90 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 834, tg = 3.90 t/s, tg_3s = 3.92 t/s slot print_timing: id 0 | task 1095 | n_gen = 846, tg = 3.90 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 1095 | n_gen = 858, tg = 3.90 t/s, tg_3s = 3.84 t/s slot print_timing: id 0 | task 1095 | n_gen = 870, tg = 3.90 t/s, tg_3s = 3.90 t/s slot print_timing: id 0 | task 1095 | n_gen = 882, tg = 3.90 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 1095 | n_gen = 894, tg = 3.90 t/s, tg_3s = 3.79 t/s slot print_timing: id 0 | task 1095 | n_gen = 906, tg = 3.90 t/s, tg_3s = 3.78 t/s slot print_timing: id 0 | task 1095 | n_gen = 918, tg = 3.90 t/s, tg_3s = 3.89 t/s slot print_timing: id 0 | task 1095 | n_gen = 930, tg = 3.90 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 1095 | n_gen = 942, tg = 3.90 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 954, tg = 3.90 t/s, tg_3s = 3.83 t/s slot print_timing: id 0 | task 1095 | n_gen = 966, tg = 3.90 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 1095 | n_gen = 978, tg = 3.90 t/s, tg_3s = 3.87 t/s slot print_timing: id 0 | task 1095 | n_gen = 990, tg = 3.89 t/s, tg_3s = 3.82 t/s slot print_timing: id 0 | task 1095 | n_gen = 1002, tg = 3.89 t/s, tg_3s = 3.87 t/s slot print_timing: id 0 | task 1095 | n_gen = 1014, tg = 3.89 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 1026, tg = 3.89 t/s, tg_3s = 3.82 t/s slot print_timing: id 0 | task 1095 | n_gen = 1038, tg = 3.89 t/s, tg_3s = 3.78 t/s slot print_timing: id 0 | task 1095 | n_gen = 1050, tg = 3.89 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 1062, tg = 3.89 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 1095 | n_gen = 1074, tg = 3.89 t/s, tg_3s = 3.86 t/s slot print_timing: id 0 | task 1095 | n_gen = 1086, tg = 3.89 t/s, tg_3s = 3.83 t/s slot print_timing: id 0 | task 1095 | n_gen = 1098, tg = 3.89 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 1110, tg = 3.89 t/s, tg_3s = 3.84 t/s slot print_timing: id 0 | task 1095 | n_gen = 1122, tg = 3.89 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 1095 | n_gen = 1134, tg = 3.89 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 1146, tg = 3.89 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 1095 | n_gen = 1158, tg = 3.88 t/s, tg_3s = 3.79 t/s slot print_timing: id 0 | task 1095 | n_gen = 1170, tg = 3.88 t/s, tg_3s = 3.79 t/s slot print_timing: id 0 | task 1095 | n_gen = 1182, tg = 3.88 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 1095 | n_gen = 1194, tg = 3.88 t/s, tg_3s = 3.83 t/s slot print_timing: id 0 | task 1095 | n_gen = 1206, tg = 3.88 t/s, tg_3s = 3.75 t/s slot print_timing: id 0 | task 1095 | n_gen = 1218, tg = 3.88 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 1230, tg = 3.88 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 1095 | n_gen = 1242, tg = 3.88 t/s, tg_3s = 3.80 t/s slot print_timing: id 0 | task 1095 | n_gen = 1254, tg = 3.88 t/s, tg_3s = 3.79 t/s slot print_timing: id 0 | task 1095 | n_gen = 1266, tg = 3.88 t/s, tg_3s = 3.80 t/s slot print_timing: id 0 | task 1095 | n_gen = 1278, tg = 3.88 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 1290, tg = 3.88 t/s, tg_3s = 3.79 t/s slot print_timing: id 0 | task 1095 | n_gen = 1302, tg = 3.88 t/s, tg_3s = 3.85 t/s slot print_timing: id 0 | task 1095 | n_gen = 1314, tg = 3.88 t/s, tg_3s = 3.77 t/s slot print_timing: id 0 | task 1095 | n_gen = 1326, tg = 3.87 t/s, tg_3s = 3.75 t/s slot print_timing: id 0 | task 1095 | n_gen = 1338, tg = 3.87 t/s, tg_3s = 3.78 t/s slot print_timing: id 0 | task 1095 | n_gen = 1350, tg = 3.87 t/s, tg_3s = 3.76 t/s slot print_timing: id 0 | task 1095 | n_gen = 1362, tg = 3.87 t/s, tg_3s = 3.78 t/s slot print_timing: id 0 | task 1095 | n_gen = 1374, tg = 3.87 t/s, tg_3s = 3.73 t/s slot print_timing: id 0 | task 1095 | n_gen = 1386, tg = 3.87 t/s, tg_3s = 3.82 t/s slot print_timing: id 0 | task 1095 | n_gen = 1398, tg = 3.87 t/s, tg_3s = 3.74 t/s slot print_timing: id 0 | task 1095 | n_gen = 1410, tg = 3.87 t/s, tg_3s = 3.74 t/s slot print_timing: id 0 | task 1095 | n_gen = 1422, tg = 3.87 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 1095 | n_gen = 1434, tg = 3.87 t/s, tg_3s = 3.76 t/s slot print_timing: id 0 | task 1095 | n_gen = 1446, tg = 3.87 t/s, tg_3s = 3.78 t/s slot print_timing: id 0 | task 1095 | n_gen = 1458, tg = 3.86 t/s, tg_3s = 3.75 t/s slot print_timing: id 0 | task 1095 | n_gen = 1470, tg = 3.86 t/s, tg_3s = 3.82 t/s slot print_timing: id 0 | task 1095 | n_gen = 1482, tg = 3.86 t/s, tg_3s = 3.75 t/s slot print_timing: id 0 | task 1095 | n_gen = 1494, tg = 3.86 t/s, tg_3s = 3.69 t/s slot print_timing: id 0 | task 1095 | n_gen = 1506, tg = 3.86 t/s, tg_3s = 3.81 t/s slot print_timing: id 0 | task 1095 | n_gen = 1518, tg = 3.86 t/s, tg_3s = 3.78 t/s slot print_timing: id 0 | task 1095 | n_gen = 1530, tg = 3.86 t/s, tg_3s = 3.71 t/s slot print_timing: id 0 | task 1095 | n_gen = 1542, tg = 3.86 t/s, tg_3s = 3.67 t/s slot print_timing: id 0 | task 1095 | n_gen = 1554, tg = 3.86 t/s, tg_3s = 3.68 t/s slot print_timing: id 0 | task 1095 | n_gen = 1566, tg = 3.86 t/s, tg_3s = 3.73 t/s slot print_timing: id 0 | task 1095 | n_gen = 1578, tg = 3.85 t/s, tg_3s = 3.72 t/s slot print_timing: id 0 | task 1095 | n_gen = 1590, tg = 3.85 t/s, tg_3s = 3.76 t/s slot print_timing: id 0 | task 1095 | n_gen = 1602, tg = 3.85 t/s, tg_3s = 3.70 t/s slot print_timing: id 0 | task 1095 | n_gen = 1614, tg = 3.85 t/s, tg_3s = 3.71 t/s slot print_timing: id 0 | task 1095 | n_gen = 1626, tg = 3.85 t/s, tg_3s = 3.69 t/s slot print_timing: id 0 | task 1095 | n_gen = 1638, tg = 3.85 t/s, tg_3s = 3.74 t/s slot print_timing: id 0 | task 1095 | n_gen = 1650, tg = 3.85 t/s, tg_3s = 3.72 t/s slot print_timing: id 0 | task 1095 | n_gen = 1662, tg = 3.85 t/s, tg_3s = 3.70 t/s slot print_timing: id 0 | task 1095 | n_gen = 1674, tg = 3.85 t/s, tg_3s = 3.75 t/s slot print_timing: id 0 | task 1095 | n_gen = 1686, tg = 3.85 t/s, tg_3s = 3.72 t/s slot print_timing: id 0 | task 1095 | n_gen = 1698, tg = 3.84 t/s, tg_3s = 3.69 t/s slot print_timing: id 0 | task 1095 | n_gen = 1710, tg = 3.84 t/s, tg_3s = 3.76 t/s slot print_timing: id 0 | task 1095 | n_gen = 1722, tg = 3.84 t/s, tg_3s = 3.71 t/s slot print_timing: id 0 | task 1095 | n_gen = 1734, tg = 3.84 t/s, tg_3s = 3.76 t/s slot print_timing: id 0 | task 1095 | n_gen = 1746, tg = 3.84 t/s, tg_3s = 3.70 t/s slot print_timing: id 0 | task 1095 | n_gen = 1758, tg = 3.84 t/s, tg_3s = 3.73 t/s slot print_timing: id 0 | task 1095 | n_gen = 1770, tg = 3.84 t/s, tg_3s = 3.72 t/s slot print_timing: id 0 | task 1095 | n_gen = 1782, tg = 3.84 t/s, tg_3s = 3.70 t/s slot print_timing: id 0 | task 1095 | n_gen = 1794, tg = 3.84 t/s, tg_3s = 3.74 t/s slot print_timing: id 0 | task 1095 | n_gen = 1806, tg = 3.84 t/s, tg_3s = 3.70 t/s slot print_timing: id 0 | task 1095 | n_gen = 1818, tg = 3.84 t/s, tg_3s = 3.72 t/s slot print_timing: id 0 | task 1095 | n_gen = 1829, tg = 3.83 t/s, tg_3s = 3.66 t/s slot print_timing: id 0 | task 1095 | n_gen = 1841, tg = 3.83 t/s, tg_3s = 3.68 t/s slot print_timing: id 0 | task 1095 | n_gen = 1853, tg = 3.83 t/s, tg_3s = 3.73 t/s slot print_timing: id 0 | task 1095 | n_gen = 1865, tg = 3.83 t/s, tg_3s = 3.69 t/s slot print_timing: id 0 | task 1095 | n_gen = 1877, tg = 3.83 t/s, tg_3s = 3.74 t/s slot print_timing: id 0 | task 1095 | n_gen = 1889, tg = 3.83 t/s, tg_3s = 3.67 t/s slot print_timing: id 0 | task 1095 | n_gen = 1901, tg = 3.83 t/s, tg_3s = 3.67 t/s slot print_timing: id 0 | task 1095 | n_gen = 1913, tg = 3.83 t/s, tg_3s = 3.73 t/s slot print_timing: id 0 | task 1095 | n_gen = 1925, tg = 3.83 t/s, tg_3s = 3.69 t/s slot print_timing: id 0 | task 1095 | n_gen = 1937, tg = 3.83 t/s, tg_3s = 3.72 t/s slot print_timing: id 0 | task 1095 | n_gen = 1949, tg = 3.83 t/s, tg_3s = 3.67 t/s slot print_timing: id 0 | task 1095 | n_gen = 1961, tg = 3.83 t/s, tg_3s = 3.71 t/s slot print_timing: id 0 | task 1095 | n_gen = 1973, tg = 3.82 t/s, tg_3s = 3.67 t/s slot print_timing: id 0 | task 1095 | n_gen = 1985, tg = 3.82 t/s, tg_3s = 3.69 t/s slot print_timing: id 0 | task 1095 | n_gen = 1997, tg = 3.82 t/s, tg_3s = 3.71 t/s slot print_timing: id 0 | task 1095 | n_gen = 2009, tg = 3.82 t/s, tg_3s = 3.68 t/s slot print_timing: id 0 | task 1095 | n_gen = 2021, tg = 3.82 t/s, tg_3s = 3.70 t/s [GIN] 2026/09/17 - 02:49:36 | 500 | 8m50s | 127.0.0.1 | POST "/v1/chat/completions" srv stop: cancel task, id_task = 1095 slot release: id 0 | task 1095 | stop processing: n_tokens = 2040, truncated = 0 srv update_slots: all slots are idle time=2026-09-17T05:56:34.624+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h8m6.016160826s consecutive_failures=0 time=2026-09-17T10:04:40.732+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h58m51.028585992s consecutive_failures=0 time=2026-09-17T14:03:31.860+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h51m31.224546427s consecutive_failures=0 time=2026-09-17T17:55:03.084+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h34m26.054971116s consecutive_failures=0 time=2026-09-17T22:29:29.140+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h28m16.287574833s consecutive_failures=0 time=2026-09-18T02:57:45.428+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h39m34.159230128s consecutive_failures=0 time=2026-09-18T07:37:19.686+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h17m40.747838868s consecutive_failures=0 time=2026-09-18T10:55:00.440+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h51m32.057477887s consecutive_failures=0 time=2026-09-18T14:46:32.589+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h19m0.404623608s consecutive_failures=0 time=2026-09-18T18:05:33.076+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h27m1.349681019s consecutive_failures=0 time=2026-09-18T22:32:34.526+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h14m47.580448034s consecutive_failures=0 time=2026-09-19T02:47:22.205+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h12m45.1289559s consecutive_failures=0 time=2026-09-19T07:00:07.434+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h57m5.813730285s consecutive_failures=0 time=2026-09-19T10:57:13.251+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h14m5.260215603s consecutive_failures=0 time=2026-09-19T15:11:18.514+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h22m13.494793154s consecutive_failures=0 time=2026-09-19T19:33:32.011+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h53m19.957568623s consecutive_failures=0 time=2026-09-19T23:26:52.069+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h39m15.135216728s consecutive_failures=0 time=2026-09-20T04:06:07.204+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h28m5.711177009s consecutive_failures=0 time=2026-09-20T08:34:12.916+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h31m21.882189541s consecutive_failures=0 time=2026-09-20T13:05:34.898+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h31m3.280986973s consecutive_failures=0 time=2026-09-20T17:36:38.179+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h48m16.966215284s consecutive_failures=0 time=2026-09-20T21:24:55.146+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h26m43.513272951s consecutive_failures=0 time=2026-09-21T01:51:38.659+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h12m1.219538471s consecutive_failures=0 time=2026-09-21T05:03:39.979+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h2m46.54880107s consecutive_failures=0 time=2026-09-21T09:06:26.627+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h17m3.780301676s consecutive_failures=0 time=2026-09-21T12:23:29.407+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h9m31.238869238s consecutive_failures=0 time=2026-09-21T16:33:00.740+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h50m44.171124562s consecutive_failures=0 time=2026-09-21T20:23:45.006+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h29m48.35776765s consecutive_failures=0 time=2026-09-21T23:53:33.364+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h21m1.352672491s consecutive_failures=0 time=2026-09-22T04:14:34.717+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h24m23.594707362s consecutive_failures=0 time=2026-09-22T07:38:58.314+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h5m28.554222226s consecutive_failures=0 time=2026-09-22T11:44:26.963+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h36m48.124800241s consecutive_failures=0 time=2026-09-22T15:21:15.181+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h36m49.337049213s consecutive_failures=0 time=2026-09-22T18:58:04.614+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h14m17.072403179s consecutive_failures=0 time=2026-09-22T23:12:21.692+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h29m7.879349843s consecutive_failures=0 time=2026-09-23T02:41:29.571+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h37m33.099932175s consecutive_failures=0 time=2026-09-23T07:19:02.770+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h17m43.889919722s consecutive_failures=0 time=2026-09-23T10:36:46.660+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h23m41.169267757s consecutive_failures=0 time=2026-09-23T15:00:27.929+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h28m57.608925555s consecutive_failures=0 time=2026-09-23T19:29:25.637+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h29m5.235704159s consecutive_failures=0 time=2026-09-23T22:58:30.873+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h57m29.990875425s consecutive_failures=0 time=2026-09-24T02:56:00.961+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h36m4.087726082s consecutive_failures=0 time=2026-09-24T06:32:05.145+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h35m5.887434895s consecutive_failures=0 time=2026-09-24T11:07:11.033+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h8m45.469588445s consecutive_failures=0 time=2026-09-24T15:15:56.503+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h28m57.578459041s consecutive_failures=0 time=2026-09-24T19:44:54.082+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h13m15.032036878s consecutive_failures=0 time=2026-09-24T22:58:09.208+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h5m24.355394111s consecutive_failures=0 time=2026-09-25T03:03:33.564+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h45m55.235029415s consecutive_failures=0 time=2026-09-25T07:49:28.897+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h42m20.816064404s consecutive_failures=0 time=2026-09-25T12:31:49.813+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h34m10.332413113s consecutive_failures=0 time=2026-09-25T17:06:00.245+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h15m6.690957946s consecutive_failures=0 time=2026-09-25T21:21:06.936+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h40m18.390255782s consecutive_failures=0 time=2026-09-26T01:01:25.327+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h35m23.717508146s consecutive_failures=0 time=2026-09-26T04:36:49.142+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h13m20.517793874s consecutive_failures=0 time=2026-09-26T07:50:09.660+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h30m53.845312231s consecutive_failures=0 time=2026-09-26T12:21:03.506+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h6m7.854898605s consecutive_failures=0 time=2026-09-26T16:27:11.362+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h28m20.347611991s consecutive_failures=0 time=2026-09-26T20:55:31.709+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h47m50.497562384s consecutive_failures=0 time=2026-09-27T01:43:22.307+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h35m21.762511541s consecutive_failures=0 time=2026-09-27T06:18:44.070+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h17m41.981786584s consecutive_failures=0 time=2026-09-27T10:36:26.150+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h1m23.462871808s consecutive_failures=0 time=2026-09-27T14:37:49.614+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h40m6.939737904s consecutive_failures=0 time=2026-09-27T18:17:56.554+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h41m37.1122095s consecutive_failures=0 time=2026-09-27T22:59:33.766+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h9m25.901274006s consecutive_failures=0 time=2026-09-28T03:08:59.667+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h14m8.130904592s consecutive_failures=0 time=2026-09-28T07:23:07.799+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h34m50.34812322s consecutive_failures=0 time=2026-09-28T10:57:58.147+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h10m59.651852283s consecutive_failures=0 time=2026-09-28T15:08:57.799+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h46m50.801952983s consecutive_failures=0 time=2026-09-28T18:55:48.602+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h43m13.059157851s consecutive_failures=0 time=2026-09-28T23:39:01.761+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h21m54.284201358s consecutive_failures=0 time=2026-09-29T04:00:56.046+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h22m0.816573297s consecutive_failures=0 time=2026-09-29T08:22:56.961+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h40m53.053380524s consecutive_failures=0 time=2026-09-29T12:03:50.113+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h31m14.346113806s consecutive_failures=0 time=2026-09-29T15:35:04.459+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h16m18.790744584s consecutive_failures=0 time=2026-09-29T19:51:23.250+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h23m18.399535206s consecutive_failures=0 time=2026-09-29T23:14:41.749+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h11m42.692588467s consecutive_failures=0 time=2026-09-30T03:26:24.442+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h13m10.450201514s consecutive_failures=0 time=2026-09-30T07:39:34.992+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h43m6.476719867s consecutive_failures=0 time=2026-09-30T12:22:41.469+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h13m39.489588453s consecutive_failures=0 time=2026-09-30T15:36:21.057+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h43m46.345655468s consecutive_failures=0 time=2026-09-30T20:20:07.403+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h13m38.763333957s consecutive_failures=0 time=2026-09-30T23:33:46.266+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h11m58.338966719s consecutive_failures=0 time=2026-10-01T03:45:44.605+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h48m44.594842448s consecutive_failures=0 time=2026-10-01T07:34:29.201+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h12m23.133291644s consecutive_failures=0 time=2026-10-01T10:46:52.334+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h38m32.373520504s consecutive_failures=0 time=2026-10-01T15:25:24.708+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h11m17.111435684s consecutive_failures=0 time=2026-10-01T19:36:41.919+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h40m6.147193232s consecutive_failures=0 time=2026-10-01T23:16:48.166+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h26m11.955331708s consecutive_failures=0 time=2026-10-02T02:43:00.221+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h59m43.475596803s consecutive_failures=0 time=2026-10-02T06:42:43.697+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h19m22.682285735s consecutive_failures=0 time=2026-10-02T11:02:06.470+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h15m4.412236479s consecutive_failures=0 time=2026-10-02T15:17:10.883+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h11m25.835221794s consecutive_failures=0 time=2026-10-02T19:28:36.818+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h8m33.644499543s consecutive_failures=0 time=2026-10-02T23:37:10.561+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h35m40.7232441s consecutive_failures=0 time=2026-10-03T04:12:51.285+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h20m34.664393205s consecutive_failures=0 time=2026-10-03T07:33:25.949+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h16m41.503078052s consecutive_failures=0 time=2026-10-03T10:50:07.453+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h47m58.564176642s consecutive_failures=0 time=2026-10-03T15:38:06.117+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h47m16.22545723s consecutive_failures=0 time=2026-10-03T20:25:22.343+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h29m27.497032694s consecutive_failures=0 time=2026-10-04T00:54:49.939+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h0m58.527640074s consecutive_failures=0 time=2026-10-04T04:55:48.548+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h16m28.528417964s consecutive_failures=0 time=2026-10-04T08:12:17.168+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h15m19.674131223s consecutive_failures=0 time=2026-10-04T12:27:36.843+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h34m3.785552234s consecutive_failures=0 time=2026-10-04T17:01:40.629+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h30m38.484163387s consecutive_failures=0 time=2026-10-04T21:32:19.114+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h41m23.915688496s consecutive_failures=0 time=2026-10-05T02:13:43.124+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h37m50.502971576s consecutive_failures=0 time=2026-10-05T06:51:33.627+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h26m33.991754628s consecutive_failures=0 time=2026-10-05T11:18:07.619+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h16m2.11529186s consecutive_failures=0 time=2026-10-05T14:34:09.735+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h43m11.398237731s consecutive_failures=0 time=2026-10-05T18:17:21.233+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h0m17.400572143s consecutive_failures=0 time=2026-10-05T22:17:38.732+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h13m45.102947867s consecutive_failures=0 time=2026-10-06T02:31:23.835+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h33m46.646316044s consecutive_failures=0 time=2026-10-06T07:05:10.582+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h29m28.101976775s consecutive_failures=0 time=2026-10-06T10:34:38.685+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h17m17.737710655s consecutive_failures=0 time=2026-10-06T13:51:56.518+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h25m4.366629461s consecutive_failures=0 time=2026-10-06T17:17:00.885+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h27m53.663583298s consecutive_failures=0 time=2026-10-06T20:44:54.639+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h24m27.687611736s consecutive_failures=0 time=2026-10-07T00:09:22.426+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=3h25m43.50088258s consecutive_failures=0 time=2026-10-07T03:35:05.927+02:00 level=INFO source=model_recommendations.go:177 msg="model recommendations cache sleep scheduled" wait=4h3m17.042882982s consecutive_failures=0