流程节点完善
This commit is contained in:
@@ -120,7 +120,6 @@ print_info: EOG token = 151663 '<|repo_name|>'
|
||||
print_info: EOG token = 151664 '<|file_sep|>'
|
||||
print_info: max token length = 256
|
||||
load_tensors: loading model tensors, this can take a while... (mmap = true)
|
||||
srv log_server_r: request: GET /health 127.0.0.1 503
|
||||
load_tensors: offloading 36 repeating layers to GPU
|
||||
load_tensors: offloaded 36/37 layers to GPU
|
||||
load_tensors: CUDA0 model buffer size = 2445.68 MiB
|
||||
@@ -221,227 +220,37 @@ How are you?<|im_end|>
|
||||
'
|
||||
main: server is listening on http://0.0.0.0:8090 - starting the main loop
|
||||
srv update_slots: all slots are idle
|
||||
slot launch_slot_: id 0 | task 0 | processing task
|
||||
slot update_slots: id 0 | task 0 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 38
|
||||
slot update_slots: id 0 | task 0 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 0 | prompt processing progress, n_past = 38, n_tokens = 38, progress = 1.000000
|
||||
slot update_slots: id 0 | task 0 | prompt done, n_past = 38, n_tokens = 38
|
||||
slot release: id 0 | task 0 | stop processing: n_past = 38, truncated = 0
|
||||
srv log_server_r: request: GET /health 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 1 | processing task
|
||||
slot update_slots: id 0 | task 1 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 50
|
||||
slot update_slots: id 0 | task 1 | kv cache rm [4, end)
|
||||
slot update_slots: id 0 | task 1 | prompt processing progress, n_past = 50, n_tokens = 46, progress = 0.920000
|
||||
slot update_slots: id 0 | task 1 | prompt done, n_past = 50, n_tokens = 46
|
||||
slot release: id 0 | task 1 | stop processing: n_past = 50, truncated = 0
|
||||
slot launch_slot_: id 0 | task 2 | processing task
|
||||
slot update_slots: id 0 | task 2 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 142
|
||||
slot update_slots: id 0 | task 2 | kv cache rm [4, end)
|
||||
slot update_slots: id 0 | task 2 | prompt processing progress, n_past = 142, n_tokens = 138, progress = 0.971831
|
||||
slot update_slots: id 0 | task 2 | prompt done, n_past = 142, n_tokens = 138
|
||||
slot release: id 0 | task 2 | stop processing: n_past = 142, truncated = 0
|
||||
slot launch_slot_: id 0 | task 3 | processing task
|
||||
slot update_slots: id 0 | task 3 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 27
|
||||
slot update_slots: id 0 | task 3 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 3 | prompt processing progress, n_past = 27, n_tokens = 27, progress = 1.000000
|
||||
slot update_slots: id 0 | task 3 | prompt done, n_past = 27, n_tokens = 27
|
||||
slot release: id 0 | task 3 | stop processing: n_past = 27, truncated = 0
|
||||
slot update_slots: id 0 | task 1 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 9
|
||||
slot update_slots: id 0 | task 1 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 1 | prompt processing progress, n_past = 9, n_tokens = 9, progress = 1.000000
|
||||
slot update_slots: id 0 | task 1 | prompt done, n_past = 9, n_tokens = 9
|
||||
slot release: id 0 | task 1 | stop processing: n_past = 9, truncated = 0
|
||||
slot launch_slot_: id 0 | task 0 | processing task
|
||||
slot update_slots: id 0 | task 0 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 9
|
||||
slot update_slots: id 0 | task 0 | need to evaluate at least 1 token for each active slot, n_past = 9, n_prompt_tokens = 9
|
||||
slot update_slots: id 0 | task 0 | kv cache rm [8, end)
|
||||
slot update_slots: id 0 | task 0 | prompt processing progress, n_past = 9, n_tokens = 1, progress = 0.111111
|
||||
slot update_slots: id 0 | task 0 | prompt done, n_past = 9, n_tokens = 1
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot release: id 0 | task 0 | stop processing: n_past = 9, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 4 | processing task
|
||||
slot update_slots: id 0 | task 4 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 20
|
||||
slot update_slots: id 0 | task 4 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 4 | prompt processing progress, n_past = 20, n_tokens = 20, progress = 1.000000
|
||||
slot update_slots: id 0 | task 4 | prompt done, n_past = 20, n_tokens = 20
|
||||
slot release: id 0 | task 4 | stop processing: n_past = 20, truncated = 0
|
||||
slot launch_slot_: id 0 | task 5 | processing task
|
||||
slot update_slots: id 0 | task 5 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 17
|
||||
slot update_slots: id 0 | task 5 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 5 | prompt processing progress, n_past = 17, n_tokens = 17, progress = 1.000000
|
||||
slot update_slots: id 0 | task 5 | prompt done, n_past = 17, n_tokens = 17
|
||||
slot release: id 0 | task 5 | stop processing: n_past = 17, truncated = 0
|
||||
slot update_slots: id 0 | task 4 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 9
|
||||
slot update_slots: id 0 | task 4 | need to evaluate at least 1 token for each active slot, n_past = 9, n_prompt_tokens = 9
|
||||
slot update_slots: id 0 | task 4 | kv cache rm [8, end)
|
||||
slot update_slots: id 0 | task 4 | prompt processing progress, n_past = 9, n_tokens = 1, progress = 0.111111
|
||||
slot update_slots: id 0 | task 4 | prompt done, n_past = 9, n_tokens = 1
|
||||
slot release: id 0 | task 4 | stop processing: n_past = 9, truncated = 0
|
||||
slot launch_slot_: id 0 | task 6 | processing task
|
||||
slot update_slots: id 0 | task 6 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 10
|
||||
slot update_slots: id 0 | task 6 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 6 | prompt processing progress, n_past = 10, n_tokens = 10, progress = 1.000000
|
||||
slot update_slots: id 0 | task 6 | prompt done, n_past = 10, n_tokens = 10
|
||||
slot release: id 0 | task 6 | stop processing: n_past = 10, truncated = 0
|
||||
slot launch_slot_: id 0 | task 7 | processing task
|
||||
slot update_slots: id 0 | task 7 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 17
|
||||
slot update_slots: id 0 | task 7 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 7 | prompt processing progress, n_past = 17, n_tokens = 17, progress = 1.000000
|
||||
slot update_slots: id 0 | task 7 | prompt done, n_past = 17, n_tokens = 17
|
||||
slot release: id 0 | task 7 | stop processing: n_past = 17, truncated = 0
|
||||
slot launch_slot_: id 0 | task 8 | processing task
|
||||
slot update_slots: id 0 | task 8 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 18
|
||||
slot update_slots: id 0 | task 8 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 8 | prompt processing progress, n_past = 18, n_tokens = 18, progress = 1.000000
|
||||
slot update_slots: id 0 | task 8 | prompt done, n_past = 18, n_tokens = 18
|
||||
slot release: id 0 | task 8 | stop processing: n_past = 18, truncated = 0
|
||||
slot launch_slot_: id 0 | task 9 | processing task
|
||||
slot update_slots: id 0 | task 9 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 23
|
||||
slot update_slots: id 0 | task 9 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 9 | prompt processing progress, n_past = 23, n_tokens = 23, progress = 1.000000
|
||||
slot update_slots: id 0 | task 9 | prompt done, n_past = 23, n_tokens = 23
|
||||
slot release: id 0 | task 9 | stop processing: n_past = 23, truncated = 0
|
||||
slot launch_slot_: id 0 | task 10 | processing task
|
||||
slot update_slots: id 0 | task 10 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 25
|
||||
slot update_slots: id 0 | task 10 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 10 | prompt processing progress, n_past = 25, n_tokens = 25, progress = 1.000000
|
||||
slot update_slots: id 0 | task 10 | prompt done, n_past = 25, n_tokens = 25
|
||||
slot release: id 0 | task 10 | stop processing: n_past = 25, truncated = 0
|
||||
slot launch_slot_: id 0 | task 11 | processing task
|
||||
slot update_slots: id 0 | task 11 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 24
|
||||
slot update_slots: id 0 | task 11 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 11 | prompt processing progress, n_past = 24, n_tokens = 24, progress = 1.000000
|
||||
slot update_slots: id 0 | task 11 | prompt done, n_past = 24, n_tokens = 24
|
||||
slot release: id 0 | task 11 | stop processing: n_past = 24, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 24 | processing task
|
||||
slot update_slots: id 0 | task 24 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 38
|
||||
slot update_slots: id 0 | task 24 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 24 | prompt processing progress, n_past = 38, n_tokens = 38, progress = 1.000000
|
||||
slot update_slots: id 0 | task 24 | prompt done, n_past = 38, n_tokens = 38
|
||||
slot release: id 0 | task 24 | stop processing: n_past = 38, truncated = 0
|
||||
slot launch_slot_: id 0 | task 25 | processing task
|
||||
slot update_slots: id 0 | task 25 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 50
|
||||
slot update_slots: id 0 | task 25 | kv cache rm [4, end)
|
||||
slot update_slots: id 0 | task 25 | prompt processing progress, n_past = 50, n_tokens = 46, progress = 0.920000
|
||||
slot update_slots: id 0 | task 25 | prompt done, n_past = 50, n_tokens = 46
|
||||
slot release: id 0 | task 25 | stop processing: n_past = 50, truncated = 0
|
||||
slot launch_slot_: id 0 | task 26 | processing task
|
||||
slot update_slots: id 0 | task 26 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 142
|
||||
slot update_slots: id 0 | task 26 | kv cache rm [4, end)
|
||||
slot update_slots: id 0 | task 26 | prompt processing progress, n_past = 142, n_tokens = 138, progress = 0.971831
|
||||
slot update_slots: id 0 | task 26 | prompt done, n_past = 142, n_tokens = 138
|
||||
slot release: id 0 | task 26 | stop processing: n_past = 142, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 30 | processing task
|
||||
slot update_slots: id 0 | task 30 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 27
|
||||
slot update_slots: id 0 | task 30 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 30 | prompt processing progress, n_past = 27, n_tokens = 27, progress = 1.000000
|
||||
slot update_slots: id 0 | task 30 | prompt done, n_past = 27, n_tokens = 27
|
||||
slot release: id 0 | task 30 | stop processing: n_past = 27, truncated = 0
|
||||
slot launch_slot_: id 0 | task 31 | processing task
|
||||
slot update_slots: id 0 | task 31 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 20
|
||||
slot update_slots: id 0 | task 31 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 31 | prompt processing progress, n_past = 20, n_tokens = 20, progress = 1.000000
|
||||
slot update_slots: id 0 | task 31 | prompt done, n_past = 20, n_tokens = 20
|
||||
slot release: id 0 | task 31 | stop processing: n_past = 20, truncated = 0
|
||||
slot launch_slot_: id 0 | task 32 | processing task
|
||||
slot update_slots: id 0 | task 32 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 17
|
||||
slot update_slots: id 0 | task 32 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 32 | prompt processing progress, n_past = 17, n_tokens = 17, progress = 1.000000
|
||||
slot update_slots: id 0 | task 32 | prompt done, n_past = 17, n_tokens = 17
|
||||
slot release: id 0 | task 32 | stop processing: n_past = 17, truncated = 0
|
||||
slot launch_slot_: id 0 | task 33 | processing task
|
||||
slot update_slots: id 0 | task 33 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 10
|
||||
slot update_slots: id 0 | task 33 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 33 | prompt processing progress, n_past = 10, n_tokens = 10, progress = 1.000000
|
||||
slot update_slots: id 0 | task 33 | prompt done, n_past = 10, n_tokens = 10
|
||||
slot release: id 0 | task 33 | stop processing: n_past = 10, truncated = 0
|
||||
slot launch_slot_: id 0 | task 34 | processing task
|
||||
slot update_slots: id 0 | task 34 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 17
|
||||
slot update_slots: id 0 | task 34 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 34 | prompt processing progress, n_past = 17, n_tokens = 17, progress = 1.000000
|
||||
slot update_slots: id 0 | task 34 | prompt done, n_past = 17, n_tokens = 17
|
||||
slot release: id 0 | task 34 | stop processing: n_past = 17, truncated = 0
|
||||
slot launch_slot_: id 0 | task 35 | processing task
|
||||
slot update_slots: id 0 | task 35 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 18
|
||||
slot update_slots: id 0 | task 35 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 35 | prompt processing progress, n_past = 18, n_tokens = 18, progress = 1.000000
|
||||
slot update_slots: id 0 | task 35 | prompt done, n_past = 18, n_tokens = 18
|
||||
slot release: id 0 | task 35 | stop processing: n_past = 18, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 42 | processing task
|
||||
slot update_slots: id 0 | task 42 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 23
|
||||
slot update_slots: id 0 | task 42 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 42 | prompt processing progress, n_past = 23, n_tokens = 23, progress = 1.000000
|
||||
slot update_slots: id 0 | task 42 | prompt done, n_past = 23, n_tokens = 23
|
||||
slot release: id 0 | task 42 | stop processing: n_past = 23, truncated = 0
|
||||
slot launch_slot_: id 0 | task 43 | processing task
|
||||
slot update_slots: id 0 | task 43 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 25
|
||||
slot update_slots: id 0 | task 43 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 43 | prompt processing progress, n_past = 25, n_tokens = 25, progress = 1.000000
|
||||
slot update_slots: id 0 | task 43 | prompt done, n_past = 25, n_tokens = 25
|
||||
slot release: id 0 | task 43 | stop processing: n_past = 25, truncated = 0
|
||||
slot launch_slot_: id 0 | task 44 | processing task
|
||||
slot update_slots: id 0 | task 44 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 24
|
||||
slot update_slots: id 0 | task 44 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 44 | prompt processing progress, n_past = 24, n_tokens = 24, progress = 1.000000
|
||||
slot update_slots: id 0 | task 44 | prompt done, n_past = 24, n_tokens = 24
|
||||
slot release: id 0 | task 44 | stop processing: n_past = 24, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 48 | processing task
|
||||
slot update_slots: id 0 | task 48 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 2
|
||||
slot update_slots: id 0 | task 48 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 48 | prompt processing progress, n_past = 2, n_tokens = 2, progress = 1.000000
|
||||
slot update_slots: id 0 | task 48 | prompt done, n_past = 2, n_tokens = 2
|
||||
slot release: id 0 | task 48 | stop processing: n_past = 2, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 50 | processing task
|
||||
slot update_slots: id 0 | task 50 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 12
|
||||
slot update_slots: id 0 | task 50 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 50 | prompt processing progress, n_past = 12, n_tokens = 12, progress = 1.000000
|
||||
slot update_slots: id 0 | task 50 | prompt done, n_past = 12, n_tokens = 12
|
||||
slot release: id 0 | task 50 | stop processing: n_past = 12, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 52 | processing task
|
||||
slot update_slots: id 0 | task 52 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 17
|
||||
slot update_slots: id 0 | task 52 | kv cache rm [3, end)
|
||||
slot update_slots: id 0 | task 52 | prompt processing progress, n_past = 17, n_tokens = 14, progress = 0.823529
|
||||
slot update_slots: id 0 | task 52 | prompt done, n_past = 17, n_tokens = 14
|
||||
slot release: id 0 | task 52 | stop processing: n_past = 17, truncated = 0
|
||||
slot launch_slot_: id 0 | task 53 | processing task
|
||||
slot update_slots: id 0 | task 53 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 17
|
||||
slot update_slots: id 0 | task 53 | need to evaluate at least 1 token for each active slot, n_past = 17, n_prompt_tokens = 17
|
||||
slot update_slots: id 0 | task 53 | kv cache rm [16, end)
|
||||
slot update_slots: id 0 | task 53 | prompt processing progress, n_past = 17, n_tokens = 1, progress = 0.058824
|
||||
slot update_slots: id 0 | task 53 | prompt done, n_past = 17, n_tokens = 1
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot release: id 0 | task 53 | stop processing: n_past = 17, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 56 | processing task
|
||||
slot update_slots: id 0 | task 56 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 16
|
||||
slot update_slots: id 0 | task 56 | kv cache rm [5, end)
|
||||
slot update_slots: id 0 | task 56 | prompt processing progress, n_past = 16, n_tokens = 11, progress = 0.687500
|
||||
slot update_slots: id 0 | task 56 | prompt done, n_past = 16, n_tokens = 11
|
||||
slot release: id 0 | task 56 | stop processing: n_past = 16, truncated = 0
|
||||
slot launch_slot_: id 0 | task 59 | processing task
|
||||
slot update_slots: id 0 | task 59 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 16
|
||||
slot update_slots: id 0 | task 59 | need to evaluate at least 1 token for each active slot, n_past = 16, n_prompt_tokens = 16
|
||||
slot update_slots: id 0 | task 59 | kv cache rm [15, end)
|
||||
slot update_slots: id 0 | task 59 | prompt processing progress, n_past = 16, n_tokens = 1, progress = 0.062500
|
||||
slot update_slots: id 0 | task 59 | prompt done, n_past = 16, n_tokens = 1
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot release: id 0 | task 59 | stop processing: n_past = 16, truncated = 0
|
||||
slot launch_slot_: id 0 | task 57 | processing task
|
||||
slot update_slots: id 0 | task 57 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 16
|
||||
slot update_slots: id 0 | task 57 | need to evaluate at least 1 token for each active slot, n_past = 16, n_prompt_tokens = 16
|
||||
slot update_slots: id 0 | task 57 | kv cache rm [15, end)
|
||||
slot update_slots: id 0 | task 57 | prompt processing progress, n_past = 16, n_tokens = 1, progress = 0.062500
|
||||
slot update_slots: id 0 | task 57 | prompt done, n_past = 16, n_tokens = 1
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot release: id 0 | task 57 | stop processing: n_past = 16, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot launch_slot_: id 0 | task 62 | processing task
|
||||
slot update_slots: id 0 | task 62 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 15
|
||||
slot update_slots: id 0 | task 62 | kv cache rm [5, end)
|
||||
slot update_slots: id 0 | task 62 | prompt processing progress, n_past = 15, n_tokens = 10, progress = 0.666667
|
||||
slot update_slots: id 0 | task 62 | prompt done, n_past = 15, n_tokens = 10
|
||||
slot release: id 0 | task 62 | stop processing: n_past = 15, truncated = 0
|
||||
slot launch_slot_: id 0 | task 64 | processing task
|
||||
slot update_slots: id 0 | task 64 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 15
|
||||
slot update_slots: id 0 | task 64 | need to evaluate at least 1 token for each active slot, n_past = 15, n_prompt_tokens = 15
|
||||
slot update_slots: id 0 | task 64 | kv cache rm [14, end)
|
||||
slot update_slots: id 0 | task 64 | prompt processing progress, n_past = 15, n_tokens = 1, progress = 0.066667
|
||||
slot update_slots: id 0 | task 64 | prompt done, n_past = 15, n_tokens = 1
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot release: id 0 | task 64 | stop processing: n_past = 15, truncated = 0
|
||||
slot update_slots: id 0 | task 6 | new prompt, n_ctx_slot = 4096, n_keep = 0, n_prompt_tokens = 9
|
||||
slot update_slots: id 0 | task 6 | need to evaluate at least 1 token for each active slot, n_past = 9, n_prompt_tokens = 9
|
||||
slot update_slots: id 0 | task 6 | kv cache rm [8, end)
|
||||
slot update_slots: id 0 | task 6 | prompt processing progress, n_past = 9, n_tokens = 1, progress = 0.111111
|
||||
slot update_slots: id 0 | task 6 | prompt done, n_past = 9, n_tokens = 1
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
slot release: id 0 | task 6 | stop processing: n_past = 9, truncated = 0
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/embeddings 127.0.0.1 200
|
||||
|
||||
148
logs/fastapi.log
148
logs/fastapi.log
@@ -1,47 +1,107 @@
|
||||
2026-02-20 22:25:24,889 - ERROR - 提示词文件未找到 -> classifier_prompt.txt
|
||||
2026-02-20 22:25:24,979 - INFO - Anonymized telemetry enabled. See https://docs.trychroma.com/telemetry for more information.
|
||||
2026-02-20 22:25:25,151 - INFO - 成功找到节点定义JSON代码块
|
||||
2026-02-20 22:25:25,152 - INFO - 成功解析出动作节点: ['approach_target', 'deliver_payload', 'fly_sequence', 'fly_to_waypoint', 'land', 'loiter', 'manual_confirmation', 'move_direction', 'object_detect', 'return_emergency', 'rotate', 'rotate_search', 'search_pattern', 'system_checks', 'take_photos', 'takeoff', 'track_object']
|
||||
2026-02-20 22:25:25,152 - INFO - 成功解析出条件节点: ['at_waypoint', 'object_detected']
|
||||
2026-02-20 22:25:25,152 - WARNING - 集合 location_kb 不可用: Collection [location_kb] does not exist
|
||||
2026-02-20 22:25:25,152 - WARNING - 集合 pattern_kb 不可用: Collection [pattern_kb] does not exist
|
||||
2026-02-20 22:25:25,153 - WARNING - 集合 rules_kb 不可用: Collection [rules_kb] does not exist
|
||||
INFO: Started server process [10575]
|
||||
2026-02-26 19:00:57,073 - INFO - Anonymized telemetry enabled. See https://docs.trychroma.com/telemetry for more information.
|
||||
2026-02-26 19:00:57,211 - INFO - 成功找到节点定义JSON代码块
|
||||
2026-02-26 19:00:57,211 - INFO - 成功解析出动作节点: ['approach_target', 'deliver_payload', 'fly_sequence', 'fly_to_waypoint', 'land', 'loiter', 'manual_confirmation', 'move_direction', 'object_detect', 'return_emergency', 'rotate', 'rotate_search', 'search_pattern', 'system_checks', 'take_photos', 'takeoff', 'track_object']
|
||||
2026-02-26 19:00:57,211 - INFO - 成功解析出条件节点: ['at_waypoint', 'object_detected']
|
||||
INFO: Started server process [26111]
|
||||
INFO: Waiting for application startup.
|
||||
2026-02-20 22:25:25,155 - INFO - WebSocket event loop configured.
|
||||
2026-02-26 19:00:57,217 - INFO - WebSocket event loop configured.
|
||||
INFO: Application startup complete.
|
||||
INFO: Uvicorn running on http://0.0.0.0:8000 (Press CTRL+C to quit)
|
||||
INFO: 127.0.0.1:45093 - "GET /docs HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:46985 - "GET /docs HTTP/1.1" 200 OK
|
||||
2026-02-20 22:29:37,482 - INFO - 接收到用户请求: 无人机当前在空中,往北飞50米
|
||||
2026-02-20 22:29:38,939 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-20 22:29:40,302 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-20 22:29:42,389 - INFO - ✅ 任务树可视化成功
|
||||
2026-02-20 22:29:42,389 - INFO - 图形已保存到: /home/huangfukk/DronePlanning/backend_service/generated_visualizations/py_tree.png
|
||||
2026-02-20 22:29:42,389 - INFO - 💾 历史记录已保存: /home/huangfukk/DronePlanning/backend_service/src/../history/20260220_222942_plan.json
|
||||
2026-02-20 22:29:42,389 - INFO - ✅ 成功生成并验证了Pytree(Pipeline)
|
||||
INFO: 127.0.0.1:44861 - "POST /generate_plan HTTP/1.1" 200 OK
|
||||
2026-02-20 22:30:15,774 - INFO - 接收到用户请求: 无人机当前在地面,到广场查找穿红色衣服的人,找到后拍照
|
||||
2026-02-20 22:30:16,805 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-20 22:30:29,109 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-20 22:30:29,161 - INFO - ✅ 任务树可视化成功
|
||||
2026-02-20 22:30:29,161 - INFO - 图形已保存到: /home/huangfukk/DronePlanning/backend_service/generated_visualizations/py_tree.png
|
||||
2026-02-20 22:30:29,161 - INFO - 💾 历史记录已保存: /home/huangfukk/DronePlanning/backend_service/src/../history/20260220_223029_plan.json
|
||||
2026-02-20 22:30:29,161 - INFO - ✅ 成功生成并验证了Pytree(Pipeline)
|
||||
INFO: 127.0.0.1:43927 - "POST /generate_plan HTTP/1.1" 200 OK
|
||||
2026-02-20 22:32:13,176 - INFO - 接收到用户请求: 无人机当前在地面,去面前大楼左边20米巡查并拍照
|
||||
2026-02-20 22:32:14,331 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-20 22:32:30,714 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-20 22:32:30,764 - INFO - ✅ 任务树可视化成功
|
||||
2026-02-20 22:32:30,764 - INFO - 图形已保存到: /home/huangfukk/DronePlanning/backend_service/generated_visualizations/py_tree.png
|
||||
2026-02-20 22:32:30,765 - INFO - 💾 历史记录已保存: /home/huangfukk/DronePlanning/backend_service/src/../history/20260220_223230_plan.json
|
||||
2026-02-20 22:32:30,765 - INFO - ✅ 成功生成并验证了Pytree(Pipeline)
|
||||
INFO: 127.0.0.1:45399 - "POST /generate_plan HTTP/1.1" 200 OK
|
||||
2026-02-20 22:36:36,485 - INFO - 接收到用户请求: 无人机当前在地面,到广场查找绿色公交车,找到就拍照
|
||||
2026-02-20 22:36:37,548 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-20 22:36:48,200 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-20 22:36:48,248 - INFO - ✅ 任务树可视化成功
|
||||
2026-02-20 22:36:48,248 - INFO - 图形已保存到: /home/huangfukk/DronePlanning/backend_service/generated_visualizations/py_tree.png
|
||||
2026-02-20 22:36:48,248 - INFO - 💾 历史记录已保存: /home/huangfukk/DronePlanning/backend_service/src/../history/20260220_223648_plan.json
|
||||
2026-02-20 22:36:48,248 - INFO - ✅ 成功生成并验证了Pytree(Pipeline)
|
||||
INFO: 127.0.0.1:47209 - "POST /generate_plan HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:46065 - "GET /docs HTTP/1.1" 200 OK
|
||||
2026-02-26 19:01:05,074 - INFO - ========== [Stage 1] Task Understanding ==========
|
||||
2026-02-26 19:01:05,881 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-26 19:01:05,886 - INFO - Task Understanding Results: mode=scene1, state=in_air, intent=generic_mission, risks=[]
|
||||
INFO: 127.0.0.1:46071 - "POST /debug_stage HTTP/1.1" 200 OK
|
||||
2026-02-26 19:01:23,412 - INFO - ========== [Stage 1] Task Understanding ==========
|
||||
2026-02-26 19:01:23,780 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-26 19:01:23,780 - INFO - Task Understanding Results: mode=scene1, state=in_air, intent=generic_mission, risks=[]
|
||||
INFO: 127.0.0.1:44329 - "POST /debug_stage HTTP/1.1" 200 OK
|
||||
2026-02-26 19:01:36,748 - INFO - ========== [Stage 1] Task Understanding ==========
|
||||
2026-02-26 19:01:37,168 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-26 19:01:37,169 - INFO - Task Understanding Results: mode=scene1, state=in_air, intent=generic_mission, risks=[]
|
||||
INFO: 127.0.0.1:44483 - "POST /debug_stage HTTP/1.1" 200 OK
|
||||
2026-02-26 19:15:34,050 - INFO - ========== [Stage 1] Task Understanding ==========
|
||||
2026-02-26 19:15:34,569 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-26 19:15:34,570 - INFO - Task Understanding Results: mode=scene1, state=in_air, intent=generic_mission, risks=[]
|
||||
2026-02-26 19:15:34,570 - INFO - ========== [Stage 2] Context Binding ==========
|
||||
2026-02-26 19:15:34,799 - INFO - Context Binding Results: Required Actions=['Selector', 'Sequence', 'fly_to_waypoint', 'land', 'move_direction', 'object_detect', 'object_detected', 'rotate_search', 'take_photos', 'takeoff'], RAG Scopes=['location', 'pattern']
|
||||
INFO: 127.0.0.1:47143 - "POST /debug_stage HTTP/1.1" 200 OK
|
||||
2026-02-26 19:18:26,812 - INFO - ========== [Stage 1] Task Understanding ==========
|
||||
2026-02-26 19:18:27,220 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-26 19:18:27,220 - INFO - Task Understanding Results: mode=scene1, state=in_air, intent=generic_mission, risks=[]
|
||||
2026-02-26 19:18:27,220 - INFO - ========== [Stage 2] Context Binding ==========
|
||||
2026-02-26 19:18:27,276 - INFO - Context Binding Results: Required Actions=['Selector', 'Sequence', 'fly_to_waypoint', 'land', 'move_direction', 'object_detect', 'object_detected', 'rotate_search', 'take_photos', 'takeoff'], RAG Scopes=['location', 'pattern']
|
||||
2026-02-26 19:18:27,276 - INFO - ========== [Stage 3] Macro Planning (Round 1) ==========
|
||||
2026-02-26 19:18:27,278 - INFO - [Round 1] System Prompt Preview (first 500 chars):
|
||||
任务:根据用户的自然语言指令,规划无人机的宏观执行流程结构,并提取执行该流程所需的外部参数。
|
||||
|
||||
你现在是第一阶段“宏观规划与意图提取”AI。你只需要做两件事:
|
||||
1. 分析意图并排出正确的骨架树(不需要填充任何 parameters/params)。
|
||||
2. 从用户指令中提取出需要查询确切位置或目标属性的实体清单(如地标、方向、距离、识别目标)。
|
||||
|
||||
**严格约束**:仅输出符合以下 JSON 格式的数据,**禁止**包含任何外部分析、Markdown 标记外的纯文本,或者多余的字段。
|
||||
|
||||
输出格式约定:
|
||||
```json
|
||||
{
|
||||
"macro_tree": { ... 纯结构树 ... },
|
||||
"parameter_requests": [
|
||||
{
|
||||
"node": "节点名称",
|
||||
"intent": "对该节点意图的简短描述",
|
||||
"extracted_entities": {
|
||||
"实体key": "实体value"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## 一、核心节点定义(裁剪后)
|
||||
#### 1. 可用节点定义 (必须遵守...
|
||||
2026-02-26 19:18:27,278 - INFO - [Round 1] User Prompt:
|
||||
飞到广场东边50米
|
||||
|
||||
---
|
||||
参考知识:
|
||||
【地点知识】
|
||||
{"property": "location", "information": {"name": "广场", "coordinates": {"x": 100, "y": 260, "z": 0}}}
|
||||
|
||||
{"property": "location", "information": {"name": "飞行场地", "coordinates": {"x": 0, "y": 0, "z": 0}}}
|
||||
|
||||
{"property": "location", "information": {"name": "大楼外围四个点东南天坐标系坐标", "coordinates": {"A": {"x": -24.0, "y": 241.8, "z": 0}, "B": {"x": -108.5, "y": 241.8, "z": 0}, "C": {"x": -108.5, "y": 289.8, "z": 0}, "D": {"x": -24.0, "y": 292.8, "z": 0}}}}
|
||||
|
||||
【任务模式】
|
||||
无人机当前在空中,往广场西边飞200米,持续监控5分钟,发现人就拍照告诉我,到时间可以返航。
|
||||
|
||||
无人机当前在地面,去研究所正大门,搜索扎辫子女子,找到后拍照。
|
||||
|
||||
地面起飞后到面前大楼约12米高度,沿外围巡查打开窗户;如发现窗户则拍照回传。
|
||||
---
|
||||
2026-02-26 19:18:36,039 - INFO - HTTP Request: POST http://localhost:8081/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
2026-02-26 19:18:36,039 - INFO - [Round 1] Output Macro Tree:
|
||||
{
|
||||
"root": {
|
||||
"type": "Sequence",
|
||||
"name": "FlyToSquareEast50",
|
||||
"children": [
|
||||
{
|
||||
"type": "action",
|
||||
"name": "fly_to_waypoint"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
2026-02-26 19:18:36,039 - INFO - [Round 1] Output Parameter Requests:
|
||||
[
|
||||
{
|
||||
"node": "fly_to_waypoint",
|
||||
"intent": "前往广场东边50米",
|
||||
"extracted_entities": {
|
||||
"landmark": "广场",
|
||||
"direction": "东",
|
||||
"distance": 50.0
|
||||
}
|
||||
}
|
||||
]
|
||||
INFO: 127.0.0.1:46185 - "POST /debug_stage HTTP/1.1" 200 OK
|
||||
|
||||
@@ -112,7 +112,6 @@ print_info: EOG token = 151663 '<|repo_name|>'
|
||||
print_info: EOG token = 151664 '<|file_sep|>'
|
||||
print_info: max token length = 256
|
||||
load_tensors: loading model tensors, this can take a while... (mmap = true)
|
||||
srv log_server_r: request: GET /health 127.0.0.1 503
|
||||
load_tensors: offloading 36 repeating layers to GPU
|
||||
load_tensors: offloaded 36/37 layers to GPU
|
||||
load_tensors: CUDA0 model buffer size = 2445.68 MiB
|
||||
@@ -243,122 +242,239 @@ How are you?<|im_end|>
|
||||
'
|
||||
main: server is listening on http://0.0.0.0:8081 - starting the main loop
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: GET /v1/models 127.0.0.1 200
|
||||
srv log_server_r: request: GET /health 127.0.0.1 200
|
||||
srv params_from_: Chat format: Content-only
|
||||
slot launch_slot_: id 0 | task 0 | processing task
|
||||
slot update_slots: id 0 | task 0 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 1569
|
||||
slot update_slots: id 0 | task 0 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 423
|
||||
slot update_slots: id 0 | task 0 | kv cache rm [0, end)
|
||||
slot update_slots: id 0 | task 0 | prompt processing progress, n_past = 1569, n_tokens = 1569, progress = 1.000000
|
||||
slot update_slots: id 0 | task 0 | prompt done, n_past = 1569, n_tokens = 1569
|
||||
slot release: id 0 | task 0 | stop processing: n_past = 1578, truncated = 0
|
||||
slot update_slots: id 0 | task 0 | prompt processing progress, n_past = 423, n_tokens = 423, progress = 1.000000
|
||||
slot update_slots: id 0 | task 0 | prompt done, n_past = 423, n_tokens = 423
|
||||
slot release: id 0 | task 0 | stop processing: n_past = 429, truncated = 0
|
||||
slot print_timing: id 0 | task 0 |
|
||||
prompt eval time = 1012.37 ms / 1569 tokens ( 0.65 ms per token, 1549.82 tokens per second)
|
||||
eval time = 258.11 ms / 10 tokens ( 25.81 ms per token, 38.74 tokens per second)
|
||||
total time = 1270.48 ms / 1579 tokens
|
||||
prompt eval time = 534.22 ms / 423 tokens ( 1.26 ms per token, 791.81 tokens per second)
|
||||
eval time = 155.33 ms / 7 tokens ( 22.19 ms per token, 45.07 tokens per second)
|
||||
total time = 689.55 ms / 430 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Content-only
|
||||
slot launch_slot_: id 0 | task 11 | processing task
|
||||
slot update_slots: id 0 | task 11 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 1611
|
||||
slot update_slots: id 0 | task 11 | kv cache rm [5, end)
|
||||
slot update_slots: id 0 | task 11 | prompt processing progress, n_past = 1611, n_tokens = 1606, progress = 0.996896
|
||||
slot update_slots: id 0 | task 11 | prompt done, n_past = 1611, n_tokens = 1606
|
||||
slot release: id 0 | task 11 | stop processing: n_past = 1636, truncated = 0
|
||||
slot print_timing: id 0 | task 11 |
|
||||
prompt eval time = 613.51 ms / 1606 tokens ( 0.38 ms per token, 2617.71 tokens per second)
|
||||
eval time = 653.35 ms / 26 tokens ( 25.13 ms per token, 39.80 tokens per second)
|
||||
total time = 1266.86 ms / 1632 tokens
|
||||
slot launch_slot_: id 0 | task 8 | processing task
|
||||
slot update_slots: id 0 | task 8 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 423
|
||||
slot update_slots: id 0 | task 8 | kv cache rm [409, end)
|
||||
slot update_slots: id 0 | task 8 | prompt processing progress, n_past = 423, n_tokens = 14, progress = 0.033097
|
||||
slot update_slots: id 0 | task 8 | prompt done, n_past = 423, n_tokens = 14
|
||||
slot release: id 0 | task 8 | stop processing: n_past = 429, truncated = 0
|
||||
slot print_timing: id 0 | task 8 |
|
||||
prompt eval time = 214.36 ms / 14 tokens ( 15.31 ms per token, 65.31 tokens per second)
|
||||
eval time = 146.82 ms / 7 tokens ( 20.97 ms per token, 47.68 tokens per second)
|
||||
total time = 361.19 ms / 21 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Content-only
|
||||
slot launch_slot_: id 0 | task 38 | processing task
|
||||
slot update_slots: id 0 | task 38 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 1574
|
||||
slot update_slots: id 0 | task 38 | kv cache rm [5, end)
|
||||
slot update_slots: id 0 | task 38 | prompt processing progress, n_past = 1574, n_tokens = 1569, progress = 0.996823
|
||||
slot update_slots: id 0 | task 38 | prompt done, n_past = 1574, n_tokens = 1569
|
||||
slot release: id 0 | task 38 | stop processing: n_past = 1584, truncated = 0
|
||||
slot print_timing: id 0 | task 38 |
|
||||
prompt eval time = 768.98 ms / 1569 tokens ( 0.49 ms per token, 2040.36 tokens per second)
|
||||
eval time = 254.75 ms / 11 tokens ( 23.16 ms per token, 43.18 tokens per second)
|
||||
total time = 1023.74 ms / 1580 tokens
|
||||
slot launch_slot_: id 0 | task 16 | processing task
|
||||
slot update_slots: id 0 | task 16 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 423
|
||||
slot update_slots: id 0 | task 16 | kv cache rm [408, end)
|
||||
slot update_slots: id 0 | task 16 | prompt processing progress, n_past = 423, n_tokens = 15, progress = 0.035461
|
||||
slot update_slots: id 0 | task 16 | prompt done, n_past = 423, n_tokens = 15
|
||||
slot release: id 0 | task 16 | stop processing: n_past = 429, truncated = 0
|
||||
slot print_timing: id 0 | task 16 |
|
||||
prompt eval time = 270.02 ms / 15 tokens ( 18.00 ms per token, 55.55 tokens per second)
|
||||
eval time = 144.12 ms / 7 tokens ( 20.59 ms per token, 48.57 tokens per second)
|
||||
total time = 414.14 ms / 22 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 24 | processing task
|
||||
slot update_slots: id 0 | task 24 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 422
|
||||
slot update_slots: id 0 | task 24 | kv cache rm [400, end)
|
||||
slot update_slots: id 0 | task 24 | prompt processing progress, n_past = 422, n_tokens = 22, progress = 0.052133
|
||||
slot update_slots: id 0 | task 24 | prompt done, n_past = 422, n_tokens = 22
|
||||
slot release: id 0 | task 24 | stop processing: n_past = 596, truncated = 0
|
||||
slot print_timing: id 0 | task 24 |
|
||||
prompt eval time = 399.30 ms / 22 tokens ( 18.15 ms per token, 55.10 tokens per second)
|
||||
eval time = 3861.49 ms / 175 tokens ( 22.07 ms per token, 45.32 tokens per second)
|
||||
total time = 4260.79 ms / 197 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 200 | processing task
|
||||
slot update_slots: id 0 | task 200 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 200 | kv cache rm [417, end)
|
||||
slot update_slots: id 0 | task 200 | prompt processing progress, n_past = 426, n_tokens = 9, progress = 0.021127
|
||||
slot update_slots: id 0 | task 200 | prompt done, n_past = 426, n_tokens = 9
|
||||
slot release: id 0 | task 200 | stop processing: n_past = 436, truncated = 0
|
||||
slot print_timing: id 0 | task 200 |
|
||||
prompt eval time = 253.59 ms / 9 tokens ( 28.18 ms per token, 35.49 tokens per second)
|
||||
eval time = 234.29 ms / 11 tokens ( 21.30 ms per token, 46.95 tokens per second)
|
||||
total time = 487.87 ms / 20 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 212 | processing task
|
||||
slot update_slots: id 0 | task 212 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 212 | need to evaluate at least 1 token for each active slot, n_past = 426, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 212 | kv cache rm [425, end)
|
||||
slot update_slots: id 0 | task 212 | prompt processing progress, n_past = 426, n_tokens = 1, progress = 0.002347
|
||||
slot update_slots: id 0 | task 212 | prompt done, n_past = 426, n_tokens = 1
|
||||
slot release: id 0 | task 212 | stop processing: n_past = 436, truncated = 0
|
||||
slot print_timing: id 0 | task 212 |
|
||||
prompt eval time = 24.00 ms / 1 tokens ( 24.00 ms per token, 41.67 tokens per second)
|
||||
eval time = 234.10 ms / 11 tokens ( 21.28 ms per token, 46.99 tokens per second)
|
||||
total time = 258.10 ms / 12 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 224 | processing task
|
||||
slot update_slots: id 0 | task 224 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 224 | need to evaluate at least 1 token for each active slot, n_past = 426, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 224 | kv cache rm [425, end)
|
||||
slot update_slots: id 0 | task 224 | prompt processing progress, n_past = 426, n_tokens = 1, progress = 0.002347
|
||||
slot update_slots: id 0 | task 224 | prompt done, n_past = 426, n_tokens = 1
|
||||
slot release: id 0 | task 224 | stop processing: n_past = 436, truncated = 0
|
||||
slot print_timing: id 0 | task 224 |
|
||||
prompt eval time = 42.68 ms / 1 tokens ( 42.68 ms per token, 23.43 tokens per second)
|
||||
eval time = 369.06 ms / 11 tokens ( 33.55 ms per token, 29.81 tokens per second)
|
||||
total time = 411.74 ms / 12 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 236 | processing task
|
||||
slot update_slots: id 0 | task 236 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 236 | need to evaluate at least 1 token for each active slot, n_past = 426, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 236 | kv cache rm [425, end)
|
||||
slot update_slots: id 0 | task 236 | prompt processing progress, n_past = 426, n_tokens = 1, progress = 0.002347
|
||||
slot update_slots: id 0 | task 236 | prompt done, n_past = 426, n_tokens = 1
|
||||
slot release: id 0 | task 236 | stop processing: n_past = 436, truncated = 0
|
||||
slot print_timing: id 0 | task 236 |
|
||||
prompt eval time = 35.69 ms / 1 tokens ( 35.69 ms per token, 28.02 tokens per second)
|
||||
eval time = 272.33 ms / 11 tokens ( 24.76 ms per token, 40.39 tokens per second)
|
||||
total time = 308.02 ms / 12 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 248 | processing task
|
||||
slot update_slots: id 0 | task 248 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 248 | need to evaluate at least 1 token for each active slot, n_past = 426, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 248 | kv cache rm [425, end)
|
||||
slot update_slots: id 0 | task 248 | prompt processing progress, n_past = 426, n_tokens = 1, progress = 0.002347
|
||||
slot update_slots: id 0 | task 248 | prompt done, n_past = 426, n_tokens = 1
|
||||
slot release: id 0 | task 248 | stop processing: n_past = 436, truncated = 0
|
||||
slot print_timing: id 0 | task 248 |
|
||||
prompt eval time = 333.94 ms / 1 tokens ( 333.94 ms per token, 2.99 tokens per second)
|
||||
eval time = 222.91 ms / 11 tokens ( 20.26 ms per token, 49.35 tokens per second)
|
||||
total time = 556.85 ms / 12 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 260 | processing task
|
||||
slot update_slots: id 0 | task 260 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 260 | need to evaluate at least 1 token for each active slot, n_past = 426, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 260 | kv cache rm [425, end)
|
||||
slot update_slots: id 0 | task 260 | prompt processing progress, n_past = 426, n_tokens = 1, progress = 0.002347
|
||||
slot update_slots: id 0 | task 260 | prompt done, n_past = 426, n_tokens = 1
|
||||
slot release: id 0 | task 260 | stop processing: n_past = 436, truncated = 0
|
||||
slot print_timing: id 0 | task 260 |
|
||||
prompt eval time = 24.23 ms / 1 tokens ( 24.23 ms per token, 41.27 tokens per second)
|
||||
eval time = 232.33 ms / 11 tokens ( 21.12 ms per token, 47.35 tokens per second)
|
||||
total time = 256.56 ms / 12 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 272 | processing task
|
||||
slot update_slots: id 0 | task 272 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 272 | need to evaluate at least 1 token for each active slot, n_past = 426, n_prompt_tokens = 426
|
||||
slot update_slots: id 0 | task 272 | kv cache rm [425, end)
|
||||
slot update_slots: id 0 | task 272 | prompt processing progress, n_past = 426, n_tokens = 1, progress = 0.002347
|
||||
slot update_slots: id 0 | task 272 | prompt done, n_past = 426, n_tokens = 1
|
||||
slot release: id 0 | task 272 | stop processing: n_past = 436, truncated = 0
|
||||
slot print_timing: id 0 | task 272 |
|
||||
prompt eval time = 32.14 ms / 1 tokens ( 32.14 ms per token, 31.11 tokens per second)
|
||||
eval time = 225.34 ms / 11 tokens ( 20.49 ms per token, 48.82 tokens per second)
|
||||
total time = 257.48 ms / 12 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 284 | processing task
|
||||
slot update_slots: id 0 | task 284 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 180
|
||||
slot update_slots: id 0 | task 284 | kv cache rm [4, end)
|
||||
slot update_slots: id 0 | task 284 | prompt processing progress, n_past = 180, n_tokens = 176, progress = 0.977778
|
||||
slot update_slots: id 0 | task 284 | prompt done, n_past = 180, n_tokens = 176
|
||||
slot release: id 0 | task 284 | stop processing: n_past = 189, truncated = 0
|
||||
slot print_timing: id 0 | task 284 |
|
||||
prompt eval time = 62.74 ms / 176 tokens ( 0.36 ms per token, 2805.14 tokens per second)
|
||||
eval time = 217.64 ms / 10 tokens ( 21.76 ms per token, 45.95 tokens per second)
|
||||
total time = 280.39 ms / 186 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 295 | processing task
|
||||
slot update_slots: id 0 | task 295 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 180
|
||||
slot update_slots: id 0 | task 295 | need to evaluate at least 1 token for each active slot, n_past = 180, n_prompt_tokens = 180
|
||||
slot update_slots: id 0 | task 295 | kv cache rm [179, end)
|
||||
slot update_slots: id 0 | task 295 | prompt processing progress, n_past = 180, n_tokens = 1, progress = 0.005556
|
||||
slot update_slots: id 0 | task 295 | prompt done, n_past = 180, n_tokens = 1
|
||||
slot release: id 0 | task 295 | stop processing: n_past = 189, truncated = 0
|
||||
slot print_timing: id 0 | task 295 |
|
||||
prompt eval time = 24.62 ms / 1 tokens ( 24.62 ms per token, 40.62 tokens per second)
|
||||
eval time = 215.17 ms / 10 tokens ( 21.52 ms per token, 46.47 tokens per second)
|
||||
total time = 239.79 ms / 11 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Hermes 2 Pro
|
||||
slot launch_slot_: id 0 | task 306 | processing task
|
||||
slot update_slots: id 0 | task 306 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 180
|
||||
slot update_slots: id 0 | task 306 | kv cache rm [164, end)
|
||||
slot update_slots: id 0 | task 306 | prompt processing progress, n_past = 180, n_tokens = 16, progress = 0.088889
|
||||
slot update_slots: id 0 | task 306 | prompt done, n_past = 180, n_tokens = 16
|
||||
slot release: id 0 | task 306 | stop processing: n_past = 189, truncated = 0
|
||||
slot print_timing: id 0 | task 306 |
|
||||
prompt eval time = 44.44 ms / 16 tokens ( 2.78 ms per token, 360.08 tokens per second)
|
||||
eval time = 210.16 ms / 10 tokens ( 21.02 ms per token, 47.58 tokens per second)
|
||||
total time = 254.60 ms / 26 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Content-only
|
||||
slot launch_slot_: id 0 | task 50 | processing task
|
||||
slot update_slots: id 0 | task 50 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 4929
|
||||
slot update_slots: id 0 | task 50 | kv cache rm [3, end)
|
||||
slot update_slots: id 0 | task 50 | prompt processing progress, n_past = 2051, n_tokens = 2048, progress = 0.415500
|
||||
slot update_slots: id 0 | task 50 | kv cache rm [2051, end)
|
||||
slot update_slots: id 0 | task 50 | prompt processing progress, n_past = 4099, n_tokens = 2048, progress = 0.831000
|
||||
slot update_slots: id 0 | task 50 | kv cache rm [4099, end)
|
||||
slot update_slots: id 0 | task 50 | prompt processing progress, n_past = 4929, n_tokens = 830, progress = 0.999391
|
||||
slot update_slots: id 0 | task 50 | prompt done, n_past = 4929, n_tokens = 830
|
||||
slot release: id 0 | task 50 | stop processing: n_past = 5254, truncated = 0
|
||||
slot print_timing: id 0 | task 50 |
|
||||
prompt eval time = 2682.34 ms / 4926 tokens ( 0.54 ms per token, 1836.46 tokens per second)
|
||||
eval time = 9502.10 ms / 326 tokens ( 29.15 ms per token, 34.31 tokens per second)
|
||||
total time = 12184.44 ms / 5252 tokens
|
||||
slot launch_slot_: id 0 | task 317 | processing task
|
||||
slot update_slots: id 0 | task 317 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 423
|
||||
slot update_slots: id 0 | task 317 | kv cache rm [4, end)
|
||||
slot update_slots: id 0 | task 317 | prompt processing progress, n_past = 423, n_tokens = 419, progress = 0.990544
|
||||
slot update_slots: id 0 | task 317 | prompt done, n_past = 423, n_tokens = 419
|
||||
slot release: id 0 | task 317 | stop processing: n_past = 429, truncated = 0
|
||||
slot print_timing: id 0 | task 317 |
|
||||
prompt eval time = 316.01 ms / 419 tokens ( 0.75 ms per token, 1325.92 tokens per second)
|
||||
eval time = 141.87 ms / 7 tokens ( 20.27 ms per token, 49.34 tokens per second)
|
||||
total time = 457.88 ms / 426 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Content-only
|
||||
slot launch_slot_: id 0 | task 379 | processing task
|
||||
slot update_slots: id 0 | task 379 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 1573
|
||||
slot update_slots: id 0 | task 379 | kv cache rm [3, end)
|
||||
slot update_slots: id 0 | task 379 | prompt processing progress, n_past = 1573, n_tokens = 1570, progress = 0.998093
|
||||
slot update_slots: id 0 | task 379 | prompt done, n_past = 1573, n_tokens = 1570
|
||||
slot release: id 0 | task 379 | stop processing: n_past = 1583, truncated = 0
|
||||
slot print_timing: id 0 | task 379 |
|
||||
prompt eval time = 862.21 ms / 1570 tokens ( 0.55 ms per token, 1820.91 tokens per second)
|
||||
eval time = 284.82 ms / 11 tokens ( 25.89 ms per token, 38.62 tokens per second)
|
||||
total time = 1147.03 ms / 1581 tokens
|
||||
slot launch_slot_: id 0 | task 325 | processing task
|
||||
slot update_slots: id 0 | task 325 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 423
|
||||
slot update_slots: id 0 | task 325 | need to evaluate at least 1 token for each active slot, n_past = 423, n_prompt_tokens = 423
|
||||
slot update_slots: id 0 | task 325 | kv cache rm [422, end)
|
||||
slot update_slots: id 0 | task 325 | prompt processing progress, n_past = 423, n_tokens = 1, progress = 0.002364
|
||||
slot update_slots: id 0 | task 325 | prompt done, n_past = 423, n_tokens = 1
|
||||
slot release: id 0 | task 325 | stop processing: n_past = 429, truncated = 0
|
||||
slot print_timing: id 0 | task 325 |
|
||||
prompt eval time = 256.66 ms / 1 tokens ( 256.66 ms per token, 3.90 tokens per second)
|
||||
eval time = 143.27 ms / 7 tokens ( 20.47 ms per token, 48.86 tokens per second)
|
||||
total time = 399.92 ms / 8 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Content-only
|
||||
slot launch_slot_: id 0 | task 391 | processing task
|
||||
slot update_slots: id 0 | task 391 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 7208
|
||||
slot update_slots: id 0 | task 391 | kv cache rm [3, end)
|
||||
slot update_slots: id 0 | task 391 | prompt processing progress, n_past = 2051, n_tokens = 2048, progress = 0.284129
|
||||
slot update_slots: id 0 | task 391 | kv cache rm [2051, end)
|
||||
slot update_slots: id 0 | task 391 | prompt processing progress, n_past = 4099, n_tokens = 2048, progress = 0.568258
|
||||
slot update_slots: id 0 | task 391 | kv cache rm [4099, end)
|
||||
slot update_slots: id 0 | task 391 | prompt processing progress, n_past = 6147, n_tokens = 2048, progress = 0.852386
|
||||
slot update_slots: id 0 | task 391 | kv cache rm [6147, end)
|
||||
slot update_slots: id 0 | task 391 | prompt processing progress, n_past = 7208, n_tokens = 1061, progress = 0.999584
|
||||
slot update_slots: id 0 | task 391 | prompt done, n_past = 7208, n_tokens = 1061
|
||||
slot release: id 0 | task 391 | stop processing: n_past = 7577, truncated = 0
|
||||
slot print_timing: id 0 | task 391 |
|
||||
prompt eval time = 4810.49 ms / 7205 tokens ( 0.67 ms per token, 1497.77 tokens per second)
|
||||
eval time = 11443.72 ms / 370 tokens ( 30.93 ms per token, 32.33 tokens per second)
|
||||
total time = 16254.21 ms / 7575 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Content-only
|
||||
slot launch_slot_: id 0 | task 765 | processing task
|
||||
slot update_slots: id 0 | task 765 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 1572
|
||||
slot update_slots: id 0 | task 765 | kv cache rm [3, end)
|
||||
slot update_slots: id 0 | task 765 | prompt processing progress, n_past = 1572, n_tokens = 1569, progress = 0.998092
|
||||
slot update_slots: id 0 | task 765 | prompt done, n_past = 1572, n_tokens = 1569
|
||||
slot release: id 0 | task 765 | stop processing: n_past = 1582, truncated = 0
|
||||
slot print_timing: id 0 | task 765 |
|
||||
prompt eval time = 826.51 ms / 1569 tokens ( 0.53 ms per token, 1898.34 tokens per second)
|
||||
eval time = 227.62 ms / 11 tokens ( 20.69 ms per token, 48.33 tokens per second)
|
||||
total time = 1054.13 ms / 1580 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
srv params_from_: Chat format: Content-only
|
||||
slot launch_slot_: id 0 | task 777 | processing task
|
||||
slot update_slots: id 0 | task 777 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 4927
|
||||
slot update_slots: id 0 | task 777 | kv cache rm [3, end)
|
||||
slot update_slots: id 0 | task 777 | prompt processing progress, n_past = 2051, n_tokens = 2048, progress = 0.415669
|
||||
slot update_slots: id 0 | task 777 | kv cache rm [2051, end)
|
||||
slot update_slots: id 0 | task 777 | prompt processing progress, n_past = 4099, n_tokens = 2048, progress = 0.831338
|
||||
slot update_slots: id 0 | task 777 | kv cache rm [4099, end)
|
||||
slot update_slots: id 0 | task 777 | prompt processing progress, n_past = 4927, n_tokens = 828, progress = 0.999391
|
||||
slot update_slots: id 0 | task 777 | prompt done, n_past = 4927, n_tokens = 828
|
||||
slot release: id 0 | task 777 | stop processing: n_past = 5246, truncated = 0
|
||||
slot print_timing: id 0 | task 777 |
|
||||
prompt eval time = 2382.24 ms / 4924 tokens ( 0.48 ms per token, 2066.96 tokens per second)
|
||||
eval time = 8172.57 ms / 320 tokens ( 25.54 ms per token, 39.16 tokens per second)
|
||||
total time = 10554.82 ms / 5244 tokens
|
||||
slot launch_slot_: id 0 | task 333 | processing task
|
||||
slot update_slots: id 0 | task 333 | new prompt, n_ctx_slot = 16384, n_keep = 0, n_prompt_tokens = 7547
|
||||
slot update_slots: id 0 | task 333 | kv cache rm [3, end)
|
||||
slot update_slots: id 0 | task 333 | prompt processing progress, n_past = 2051, n_tokens = 2048, progress = 0.271366
|
||||
slot update_slots: id 0 | task 333 | kv cache rm [2051, end)
|
||||
slot update_slots: id 0 | task 333 | prompt processing progress, n_past = 4099, n_tokens = 2048, progress = 0.542732
|
||||
slot update_slots: id 0 | task 333 | kv cache rm [4099, end)
|
||||
slot update_slots: id 0 | task 333 | prompt processing progress, n_past = 6147, n_tokens = 2048, progress = 0.814098
|
||||
slot update_slots: id 0 | task 333 | kv cache rm [6147, end)
|
||||
slot update_slots: id 0 | task 333 | prompt processing progress, n_past = 7547, n_tokens = 1400, progress = 0.999602
|
||||
slot update_slots: id 0 | task 333 | prompt done, n_past = 7547, n_tokens = 1400
|
||||
slot release: id 0 | task 333 | stop processing: n_past = 7680, truncated = 0
|
||||
slot print_timing: id 0 | task 333 |
|
||||
prompt eval time = 4829.54 ms / 7544 tokens ( 0.64 ms per token, 1562.05 tokens per second)
|
||||
eval time = 3913.28 ms / 134 tokens ( 29.20 ms per token, 34.24 tokens per second)
|
||||
total time = 8742.83 ms / 7678 tokens
|
||||
srv update_slots: all slots are idle
|
||||
srv log_server_r: request: POST /v1/chat/completions 127.0.0.1 200
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
10395
|
||||
10396
|
||||
10573
|
||||
26023
|
||||
26024
|
||||
26109
|
||||
|
||||
Reference in New Issue
Block a user