iop/agent-test/inventory-dev.yaml
toki 32c0754f91 refactor(inventory): 에이전트 host 프로필을 분리한다
환경별 노드와 provider runtime의 원본을 유지하면서 에이전트별 host 설정과 Pi 호환성 관측을 별도 인벤토리로 관리한다.
2026-08-02 11:06:54 +09:00

1335 lines
53 KiB
YAML

inventory_id: inventory-dev
common_inventory: agent-test/inventory.yaml
test_env: dev
profile: dev-runtime-provider-pool
last_updated_at: "2026-08-02"
source:
remote_runner:
ssh: toki@toki-labs.com
repo_root: /Users/toki/agent-work/iop-dev
clean_sync:
- git fetch origin dev main --tags
- git clean -fd
- git switch --force-create dev origin/dev
- git reset --hard origin/dev
dirty_policy: discard
compose:
project_name: iop-dev-agent
network: iop-dev-agent-net
subnet: 10.89.1.0/24
env_file: .env.dev.example
runtime_root: /Users/toki/agent-work/iop-dev
services:
- postgres
- redis
- control-plane
- web
- prometheus
- grafana
host_ports:
web: 13001
control_plane_http: 18001
control_plane_client_ws: 19001
control_plane_edge_wire: 19002
edge_node_tcp_compose: 19003
postgres: 15401
redis: 16301
edge_metrics_native: 19101
control_plane_metrics: 19103
prometheus: 19111
grafana: 19121
observability:
prometheus_config: configs/prometheus/prometheus.dev.yml
control_plane_metrics_target: control-plane:9093
edge_metrics_target: host.docker.internal:19101
grafana_bind: 127.0.0.1
prometheus_bind: 127.0.0.1
edge:
id: edge-toki-labs-dev
config_path: build/dev-runtime/edge.yaml
control_plane_http: http://127.0.0.1:18001
control_plane_status_url: http://127.0.0.1:18001/edges/edge-toki-labs-dev/status
bootstrap_http_public: http://toki-labs.com:18082
openai_base_url_public: http://toki-labs.com:18083/v1
openai_base_url_runner: http://127.0.0.1:18083/v1
edge_node_tcp_public: toki-labs.com:18084
admin_addr_runner: 127.0.0.1:19093
build:
binaries:
edge: build/dev-runtime/bin/edge
node_macos: build/dev-runtime/bin/iop-node
node_linux_arm64: build/dev-runtime/bin/iop-node-linux-arm64
node_windows_amd64: build/dev-runtime/bin/iop-node-windows-amd64.exe
model:
alias: laguna-s:2.1
aliases:
"qwen3.6:35b":
status: active_edge_model_group
display_name: Qwen 3.6 35B
capacity_total: 2
providers:
- id: mac-mlx-vllm
served_model: mlx-community/Qwen3.6-35B-A3B-4bit
"ornith:35b":
status: active_edge_model_group
display_name: Ornith 1.0 35B
capacity_total: 4
providers:
- id: onexplayer-lemonade
served_model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
- id: rtx5090-lemonade
served_model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
"laguna-s:2.1":
observed_at: "2026-07-24"
status: active_edge_model_group
display_name: Poolside Laguna S 2.1
context_window: 262144
default_max_tokens: 65536
capacity_total: 4
providers:
- id: gx10-vllm
served_model: laguna-s:2.1
ornith-fast:
observed_at: "2026-07-18"
status: active_edge_model_group
display_name: Ornith 1.0 35B Fast
context_window: 262144
default_max_tokens: 32768
min_max_tokens: 16384
capacity_total: 1
capacity_accounting: independent_model_group_not_shared_with_ornith_35b
providers:
- id: rtx5090-lemonade
served_model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
capacity: 1
priority: 0
rollout:
candidate_validation_source_commit: cedb7fd41b94515765e20c3d9eaedf24fad7d2e0
config_check: passed
refresh_dry_run: applied_without_restart
refresh_apply: passed
edge_process_restart: not_required_for_live_model_catalog_refresh
node_process_restart: not_required_for_model_catalog_only_change
changed_models:
- ornith-fast
models_endpoint: passed
chat_completions_single: passed
responses_single: passed
connected_nodes:
- mac-codex-node
- gx10-vllm-node
- onexplayer-lemonade-node
- rtx5090-lemonade-node
known_risk: ornith-fast and ornith:35b each admit against an independent model-group capacity counter, so one simultaneous request through each alias can target the same RTX5090 provider until provider-owned shared capacity is implemented.
provider_capacity_total: 4
provider_capacity_status: verified_with_edge_routing_and_provider_metrics_2026_07_24
context_window: 262144
default_max_tokens: 65536
min_max_tokens: 0
default_thinking_token_budget: 0
qwen3_6_reference:
alias: qwen3.6:35b
status: registered_mac_mlx_only_after_ornith_replacement
provider_capacity_total: 2
provider: mac-mlx-vllm
served_model: mlx-community/Qwen3.6-35B-A3B-4bit
context_window: 262144
default_max_tokens: 32768
min_max_tokens: 32768
active_edge_model_group:
observed_at: "2026-07-24"
id: laguna-s:2.1
display_name: Poolside Laguna S 2.1
status: active_iop_edge_group
openai_base_url_public: http://toki-labs.com:18083/v1
context_window: 262144
default_max_tokens: 65536
min_max_tokens: 0
thinking_control: chat_template_kwargs.enable_thinking
reasoning_history_policy: preserve_thinking_and_replay_reasoning
provider_capacity_total: 4
providers:
gx10-vllm:
runtime_type: vllm
endpoint: http://192.168.0.91:8001/v1
served_model: laguna-s:2.1
upstream_model: poolside/Laguna-S-2.1-NVFP4
upstream_url: https://huggingface.co/poolside/Laguna-S-2.1-NVFP4
upstream_revision: 07614121b31898586430f189d27a25a0be310843
draft_model: poolside/Laguna-S-2.1-DFlash-NVFP4
draft_revision: 723794750422b3efbf3a7b3af76dffb4ba035943
quantization: nvfp4
dtype: bfloat16
capacity: 4
priority: 1
long_context_capacity: 1
runtime_args:
max_model_len: 262144
max_num_seqs: 4
gpu_memory_utilization: 0.70
enable_prefix_caching: true
enable_auto_tool_choice: true
tool_call_parser: poolside_v1
reasoning_parser: poolside_v1
chat_template: /run/iop/laguna-s-2.1-thinking.jinja
thinking_generation_prefix: "<think>\n"
default_chat_template_kwargs:
enable_thinking: true
override_generation_config:
temperature: 0.7
top_p: 0.95
speculative_config:
model: poolside/Laguna-S-2.1-DFlash-NVFP4
num_speculative_tokens: 7
method: dflash
observed_kv:
available_kv_cache_memory_gib: 13.08
gpu_kv_cache_tokens: 373711
max_concurrency_for_262144_token_requests: 1.43
thinking_observation:
parser_output_field: reasoning
pi_accepted_fields:
- reasoning_content
- reasoning
- reasoning_text
stock_generation_prefix_closed_reasoning_immediately: true
active_generation_prefix: "<think>\n"
simple_pi_high_reasoning_delta: passed
simple_pi_off_reasoning_delta_absent: passed
agentic_tool_request_reasoning_delta: passed
pi_compat:
thinking_format: chat-template
enable_thinking_source: thinking.enabled
preserve_thinking: true
smoke:
models_endpoint: passed
direct_chat_completions: passed
edge_chat_completions: passed
pi_high_off_thinking_control: passed
pi_agentic_tool_reasoning_stream: passed
previous_ornith_group_observation:
observed_at: "2026-07-18"
id: ornith:35b
display_name: Ornith 1.0 35B
status: historical_replaced_on_gx10_by_laguna_s_2_1
openai_base_url_public: http://toki-labs.com:18083/v1
context_window: 262144
default_max_tokens: 32768
min_max_tokens: 16384
provider_capacity_total: 8
qwen_group_update: qwen3.6:35b remains registered with mac-mlx-vllm only; gx10-vllm, onexplayer-lemonade, and rtx5090-lemonade serve this Ornith group.
providers:
gx10-vllm:
runtime_type: vllm
endpoint: http://192.168.0.91:8001/v1
served_model: ornith:35b
upstream_model: deepreinforce-ai/Ornith-1.0-35B-FP8
upstream_url: https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B-FP8
upstream_revision: 1ab57ce0b44950e498a88756f40ad1ed4d0f30ca
quantization: fp8
dtype: bfloat16
capacity: 4
priority: 1
total_context_tokens: 1048576
long_context_capacity: 4
runtime_args:
max_model_len: 262144
max_num_seqs: 4
gpu_memory_utilization: 0.50
enable_prefix_caching: true
enable_auto_tool_choice: true
tool_call_parser: qwen3_xml
reasoning_parser: qwen3
trust_remote_code: true
language_model_only: true
override_generation_config:
temperature: 0.6
top_p: 0.95
top_k: 20
observed_kv:
available_kv_cache_memory_gib: 20.79
gpu_kv_cache_tokens: 1060912
max_concurrency_for_262144_token_requests: 4.05
onexplayer-lemonade:
runtime_type: lemonade
endpoint: http://192.168.0.59:13305/v1
served_model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
response_model: ornith-1.0-35b-Q5_K_M.gguf
upstream_model: LordNeel/Ornith-1.0-35B-GGUF-llamacpp-tp1
upstream_url: https://huggingface.co/LordNeel/Ornith-1.0-35B-GGUF-llamacpp-tp1
upstream_revision: c50d5d4407f70e43208dee836c66bb8a05c1be91
quantization: Q5_K_M
capacity: 3
priority: 2
total_context_tokens: 524288
long_context_capacity: 2
runtime_args:
ctx_size: 524288
llamacpp_backend: vulkan
llamacpp_args: "--spec-type none -cb -fa on -b 4096 -ub 1024 --kv-unified --temp 0.6 --top-p 0.95 --top-k 20"
fixed_parallel_slots_removed: true
observed_total_slots: 4
observed_slot_n_ctx: 262144
rtx5090-lemonade:
runtime_type: lemonade
endpoint: http://192.168.0.111:13305/v1
served_model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
response_model: ornith-1.0-35b-Q5_K_M.gguf
upstream_model: LordNeel/Ornith-1.0-35B-GGUF-llamacpp-tp1
upstream_url: https://huggingface.co/LordNeel/Ornith-1.0-35B-GGUF-llamacpp-tp1
upstream_revision: c50d5d4407f70e43208dee836c66bb8a05c1be91
quantization: Q5_K_M
capacity: 1
priority: 0
total_context_tokens: 262144
long_context_capacity: 1
runtime_args:
ctx_size: 262144
llamacpp_backend: cuda
llamacpp_args: "--spec-type none -np 1 -cb -fa on -b 512 -ub 256 --kv-unified -ctk q8_0 -ctv q8_0 --temp 0.6 --top-p 0.95 --top-k 20"
observed_total_slots: 1
observed_slot_n_ctx: 262144
smoke:
observed_at: "2026-07-18"
direct_tool_choice_auto: passed
edge_chat_completions_routed: passed
isolated_alias_http_status: 200
isolated_alias_elapsed_sec: 0.320
active_alias_estimated_input_tokens: 49306
active_alias_edge_elapsed_sec: 15.414
active_alias_lemonade_input_tokens: 60071
active_alias_lemonade_output_tokens: 89
active_alias_time_to_first_token_sec: 14.664143
active_alias_generation_tokens_per_sec: 174.713
concurrency_one_enforced: true
overlapping_requests_routed_to_other_providers: 2
priority_rollout:
observed_at: "2026-07-19"
config_check: passed
refresh_dry_run: applied_without_restart
refresh_apply: passed
restart_required: false
node_results:
mac-codex-node: applied
gx10-vllm-node: applied
onexplayer-lemonade-node: applied
rtx5090-lemonade-node: applied
providers:
rtx5090-lemonade: 0
gx10-vllm: 1
onexplayer-lemonade: 2
smoke:
models_endpoint: passed
chat_completions_single:
endpoint: /v1/chat/completions
request_model: ornith:35b
prompt: "2+2? Answer only the number."
result: passed
content: "4"
elapsed_sec: 3.32
completion_tokens: 127
responses_single:
endpoint: /v1/responses
request_model: ornith:35b
prompt: "2+2? Answer only the number."
result: passed
output_text: "4"
elapsed_sec: 4.58
completion_tokens: 1
chat_completions_capacity:
endpoint: /v1/chat/completions
concurrent_requests: 8
success_count: 8
failure_count: 0
wall_sec: 63.97
peak_total_in_flight: 7
peak_total_queued: 2
provider_peaks:
gx10-vllm:
capacity: 4
max_in_flight: 4
max_queued: 1
onexplayer-lemonade:
capacity: 3
max_in_flight: 3
max_queued: 1
mac-mlx-vllm:
capacity: 2
max_in_flight: 0
max_queued: 0
direct_throughput_benchmarks:
observed_at: "2026-07-12"
benchmark_shape: direct provider OpenAI-compatible /v1/chat/completions, max_tokens=1024, temperature=0, same long-list prompt, completion_tokens divided by wall time
benchmark_scope_note: throughput-only benchmark; all measured requests filled the 1024-token output budget and ended with finish_reason=length, so this is not a final-answer quality smoke.
latest_concurrency_1_comparison:
observed_at: "2026-07-18"
execution_mode: sequential direct-provider requests with one active request per provider
endpoint: /v1/chat/completions
prompt_tokens: 48
request_max_tokens: 1024
temperature: 0
thinking_enabled: false
metric: completion_tokens divided by total HTTP response time
metric_scope: end-to-end throughput including network and time-to-first-response; Lemonade server decode timing is recorded separately when present
target_resolution: current dev has no RTX4090 node; the comparison target is the newly added rtx5090-lemonade node
edge_action: no config refresh, process restart, or route mutation
common_result:
http_status: 200
completion_tokens: 1024
finish_reason: length
providers:
gx10-vllm:
engine: vllm
quantization: fp8
elapsed_sec: 26.241519
end_to_end_tok_s: 39.022
server_decode_tok_s: null
onexplayer-lemonade:
engine: lemonade_llamacpp_vulkan
quantization: Q5_K_M
elapsed_sec: 30.315311
end_to_end_tok_s: 33.778
server_decode_tok_s: 48.10214038553302
rtx5090-lemonade:
engine: lemonade_llamacpp_cuda
quantization: Q5_K_M
elapsed_sec: 5.092999
end_to_end_tok_s: 201.060
server_decode_tok_s: 245.83231403349612
relative_end_to_end:
rtx5090_vs_gx10: 5.15
rtx5090_vs_onexplayer: 5.95
gx10_vs_onexplayer: 1.16
post_condition:
edge_nodes_connected:
- mac-codex-node
- gx10-vllm-node
- onexplayer-lemonade-node
- rtx5090-lemonade-node
rtx5090_capacity: 1
rtx5090_priority: 3
gx10-vllm:
endpoint: http://192.168.0.91:8001/v1
model: ornith:35b
engine: vllm
quantization: fp8
results:
concurrency_1:
success_count: 1
failure_count: 0
avg_per_call_tok_s: 38.80
aggregate_tok_s: 38.80
wall_sec: 26.39
total_completion_tokens: 1024
concurrency_2:
success_count: 2
failure_count: 0
avg_per_call_tok_s: 36.95
aggregate_tok_s: 73.87
wall_sec: 27.72
total_completion_tokens: 2048
concurrency_3:
success_count: 3
failure_count: 0
avg_per_call_tok_s: 28.24
aggregate_tok_s: 84.68
wall_sec: 36.28
total_completion_tokens: 3072
concurrency_4:
success_count: 4
failure_count: 0
avg_per_call_tok_s: 31.46
aggregate_tok_s: 125.77
wall_sec: 32.57
total_completion_tokens: 4096
onexplayer-lemonade:
endpoint: http://192.168.0.59:13305/v1
model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
response_model: ornith-1.0-35b-Q5_K_M.gguf
engine: lemonade_llamacpp_vulkan
quantization: Q5_K_M
results:
concurrency_1:
success_count: 1
failure_count: 0
avg_per_call_tok_s: 45.07
aggregate_tok_s: 45.07
wall_sec: 22.72
total_completion_tokens: 1024
concurrency_2:
success_count: 2
failure_count: 0
avg_per_call_tok_s: 34.34
aggregate_tok_s: 68.67
wall_sec: 29.82
total_completion_tokens: 2048
concurrency_3:
success_count: 3
failure_count: 0
avg_per_call_tok_s: 23.27
aggregate_tok_s: 69.81
wall_sec: 44.01
total_completion_tokens: 3072
reasoning_policy: model_native_thinking
laguna_official_sampling:
source: https://huggingface.co/poolside/Laguna-S-2.1
source_section: vLLM and Controlling reasoning
temperature: 0.7
top_p: 0.95
top_k: 20
top_k_source: model_generation_config
precedence: caller-explicit sampling parameters override provider defaults
rollout_observed_at: "2026-07-24"
providers:
gx10-vllm:
status: applied
mechanism: vllm --override-generation-config plus model generation_config
thinking_control: --default-chat-template-kwargs enable_thinking=true
chat_template: /run/iop/laguna-s-2.1-thinking.jinja
thinking_generation_prefix: "<think>\n"
reasoning_parser: poolside_v1
tool_call_parser: poolside_v1
reasoning_output_field: reasoning
validation:
simple_no_tool_request:
result: passed_with_reasoning_delta
note: The stock generation prefix emitted the closing think tag immediately. The active local template adds a newline after the opening think tag, and Pi high now emits thinking events while Pi off emits none.
pi_agentic_tool_request:
result: passed
reasoning_delta: present
tool_call: passed
reasoning_after_tool_result: present
final_text: present
pi_thinking_control:
thinking_format: chat-template
enable_thinking_source: thinking.enabled
preserve_thinking: true
previous_ornith_official_sampling_observation:
status: historical_gx10_replaced_by_laguna_s_2_1
source: https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B
source_section: Quickstart and Chat Completions API examples
temperature: 0.6
top_p: 0.95
top_k: 20
repeat_penalty: not_set_by_official_example
precedence: caller-explicit sampling parameters override provider defaults
language_control_note: sampling does not force reasoning language; use the short prompt "Think in English. Final in Korean." and validate reasoning/final separately
rollout_observed_at: "2026-07-23"
providers:
gx10-vllm:
status: applied
mechanism: vllm --override-generation-config
onexplayer-lemonade:
status: applied
mechanism: saved Lemonade recipe_options llamacpp_args and run-onex-runtime.ps1 reload baseline
rtx5090-lemonade:
status: applied
mechanism: saved Lemonade recipe_options llamacpp_args
validation:
prompt: "Think in English. Final in Korean. What is 2+2?"
max_tokens: 1024
gx10_vllm_direct:
result: passed
finish_reason: stop
reasoning_language: english_with_korean_answer_candidates_quoted
final_content: "4입니다."
onexplayer_lemonade_direct:
result: passed
finish_reason: stop
reasoning_language: english_with_korean_answer_candidates_quoted
final_content: "4입니다."
rtx5090_lemonade_direct:
result: passed
finish_reason: stop
reasoning_language: english_with_korean_answer_candidates_quoted
final_content: "2 + 2 = 4입니다."
edge_ornith_group:
result: passed
finish_reason: stop
reasoning_language: english_with_korean_answer_candidates_quoted
final_content: "4"
strict_language_isolation: not_guaranteed_by_sampling_or_short_prompt
short_prompt_comparison:
recommended: "Think in English. Final in Korean."
result: GX10, OneXPlayer, and RTX5090 returned finish_reason=stop with Korean final content
avoid: literal think tags and longer no-quote prohibitions
avoid_reason: compared variants leaked reasoning into final content or ended with finish_reason=length
full_capacity_smoke:
scope: full Ornith capacity is GX10 4 plus OneXPlayer 3 plus RTX5090 1
concurrent_requests: 9
chat_completions:
http_200: 9
peak_in_flight: 8
peak_queued: 3
gx10_peak_in_flight: 4
onexplayer_peak_in_flight: 3
rtx5090_peak_in_flight: 1
recovered_to_zero: true
responses:
http_200: 9
peak_in_flight: 8
peak_queued: 3
gx10_peak_in_flight: 4
onexplayer_peak_in_flight: 3
rtx5090_peak_in_flight: 1
recovered_to_zero: true
agent_tooling_policy:
runtime_profile: provider_family_specific
mac_vllm_mlx:
tool_call_parser: qwen
reasoning_parser: qwen3
default_chat_template_kwargs:
enable_thinking: true
gx10_vllm:
model: laguna-s:2.1
tool_call_parser: poolside_v1
reasoning_parser: poolside_v1
chat_template_host_path: /home/toki/iop-gx10-vllm/laguna-s-2.1-thinking.jinja
chat_template_container_path: /run/iop/laguna-s-2.1-thinking.jinja
thinking_generation_prefix: "<think>\n"
default_chat_template_kwargs:
enable_thinking: true
pi_chat_template_kwargs:
enable_thinking:
source: thinking.enabled
preserve_thinking: true
tool_call_parser_status: Pi read-tool multi-turn smoke passed with reasoning before the tool call and after the tool result
separation_note: Keep Laguna poolside_v1, Qwen qwen/qwen3, and dev-corp Gemma parser/template profiles separate.
content_completion_policy:
enforce_min_max_tokens: true
reason: prevent reasoning-only completions from exhausting caller max_tokens before final content
dispatch_policy:
capacity_gate: providers at or above capacity are excluded from dispatch candidates
selection_order: lower in_flight level wins among providers with available capacity
same_in_flight_tiebreak: lower numeric priority first, then deterministic rotation
capacity_smoke:
model: laguna-s:2.1
endpoints:
- /v1/responses
- /v1/chat/completions
concurrent_requests: 4
expected_total_in_flight: 4
expected_min_queued: 0
capacity_plus_one_concurrent_requests: 5
capacity_plus_one_expected_total_in_flight: 4
capacity_plus_one_expected_min_queued: 1
extended_concurrent_requests: 8
extended_expected_min_queued: 4
prompt_policy: long_reasoning_allowed
exact_output_match: false
latest_provider_snapshot:
observed_at: "2026-07-24"
total_capacity: 10
all_nodes_connected: true
providers:
gx10-vllm:
capacity: 4
priority: 1
in_flight: 0
queued: 0
long_context_capacity: 1
health: healthy
served_model: laguna-s:2.1
mac-mlx-vllm:
capacity: 2
priority: 2
in_flight: 0
queued: 0
long_context_capacity: 1
health: healthy
served_model: mlx-community/Qwen3.6-35B-A3B-4bit
onexplayer-lemonade:
capacity: 3
priority: 2
in_flight: 0
queued: 0
long_context_capacity: 2
health: healthy
served_model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
rtx5090-lemonade:
capacity: 1
priority: 0
in_flight: 0
queued: 0
long_context_capacity: 1
health: healthy
served_model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
laguna_agent_tool_smoke:
observed_at: "2026-07-24"
model: laguna-s:2.1
request_shape: Pi openai-completions streaming with read tool
result: passed
reasoning_before_tool_call: present
tool_call: passed
reasoning_after_tool_result: present
final_content: present
provider_reasoning_field: reasoning
pi_event: thinking_delta
previous_openai_tool_call_smoke:
status: historical_qwen_provider_pool_observation
observed_at: "2026-07-08"
request_shape: tools_plus_tool_choice_auto
direct_provider_expected:
http_status: 200
minimum_tool_calls: 1
direct_provider_result:
gx10-vllm:
http_status: 200
tool_calls: 1
mac-mlx-vllm:
http_status: 200
tool_calls: 1
onexplayer-lemonade:
http_status: 200
tool_calls: 1
edge_concurrency_result:
endpoint: /v1/chat/completions
concurrent_requests: 10
success_count: 10
failure_count: 0
tool_call_response_count: 10
finish_reason_tool_calls: 10
peak_total_in_flight: 9
peak_total_queued: 3
provider_result_counts:
gx10-vllm: 5
onexplayer-lemonade: 3
mac-mlx-vllm: 2
provider_peaks:
gx10-vllm:
max_in_flight: 4
max_queued: 1
onexplayer-lemonade:
max_in_flight: 3
max_queued: 1
mac-mlx-vllm:
max_in_flight: 2
max_queued: 1
final_content_smoke:
model: laguna-s:2.1
endpoint: /v1/chat/completions
concurrent_requests: 5
current_capacity_total: 4
expected_peak_total_in_flight: 4
expected_min_queued: 1
request_max_tokens: 65536
effective_min_max_tokens: 65536
include_reasoning: true
expected_success_count: 5
expected_finish_reason: stop
expected_final_marker: true
expected_empty_content: 0
expected_iop_notice: 0
previous_observed_at: "2026-07-04"
previous_observed_capacity_total: 10
previous_observed_result:
http_ok: 10
success_count: 10
failure_count: 0
finish_reason_stop: 10
marker_missing_count: 0
empty_content_count: 0
iop_notice_count: 0
think_tag_leak_count: 0
reasoning_present_count: 0
aggregate_completion_tok_s: 62.5
total_completion_tokens: 17933
batch_wall_sec: 286.908
avg_latency_sec: 175.074
peak_total_in_flight: 10
peak_total_queued: 0
provider_peaks:
gx10-vllm:
capacity: 4
max_in_flight: 4
max_queued: 0
mac-mlx-vllm:
capacity: 3
max_in_flight: 3
max_queued: 0
onexplayer-lemonade:
capacity: 3
max_in_flight: 3
max_queued: 0
retained_test_runtimes:
- id: onexplayer-laguna-standalone
host: 192.168.0.59
purpose: standalone_laguna_s_2_1_llama_server_test_runtime
status: stopped_retained_for_possible_future_use
iop_registration: none
separation:
active_iop_provider: onexplayer-lemonade
relationship: separate_test_runtime_not_managed_by_iop_or_lemonade
lifecycle_api: none
note: Do not use an IOP /v1/load request to start or manage this runtime.
last_runtime:
endpoint: http://192.168.0.59:13305/v1
listen_host: 0.0.0.0
port: 13305
served_model: laguna-s:2.1
model_file: C:/Users/r0bin/.cache/huggingface/hub/models--poolside--Laguna-S-2.1-GGUF/snapshots/92b968eeba0fbb790ef4216e2a70ef079ed19b07/laguna-s-2.1-Q4_K_M.gguf
llama_server: C:/Users/r0bin/iop-field/runtime/llama.cpp-laguna-04b2b72c/build-vulkan-release/bin/llama-server.exe
llama_server_sha256: 9BE0DDB62ED5A88B83D64F106F052233875F994D828DFC09DFE685A2A51210A0
runtime_args: "--ctx-size 262144 --parallel 1 --kv-unified --cache-type-k q8_0 --cache-type-v q8_0 -ngl 0 -fa on -b 1024 -ub 512 --jinja --reasoning-format auto --temp 0.7 --top-p 0.95"
context_window: 262144
parallel_slots: 1
kv_cache_quantization: q8_0
gpu_layers: 0
retained_assets:
executable_present_observed_at: "2026-07-24"
model_present_observed_at: "2026-07-24"
launcher_backup: C:/Users/r0bin/iop-field/run-onex-runtime.ps1.pre-ornith-restore-20260724T053553Z
stdout_log: C:/Users/r0bin/iop-field/laguna-server.stdout.log
stderr_log: C:/Users/r0bin/iop-field/laguna-server.stderr.log
deactivation:
observed_at: "2026-07-24T05:34:24+09:00"
stopped_runtime_supervisor_pid: 40760
stopped_llama_server_pid: 27412
assets_deleted: false
reuse_requirements:
- assign_a_port_that_does_not_conflict_with_active_onexplayer_lemonade_13305
- use_a_dedicated_launcher_instead_of_overwriting_run-onex-runtime.ps1
- keep_it_out_of_the_iop_provider_pool_unless_explicitly_registered
- verify_model_endpoint_and_slot_profile_after_start
agent_inventory:
inventory_id: inventory-agent
path: agent-test/inventory-agent.yaml
environment: dev
nodes:
- id: mac-codex-node
alias: mac-codex
role: cli+mlx-provider
ssh: toki@toki-labs.com
workspace: /Users/toki/agent-work/iop-workspace/nomadcode
provider_pool_candidate: true
adapters:
- cli
- mac-mlx-vllm
providers:
- id: mac-mlx-vllm
type: vllm-mlx
endpoint: http://127.0.0.1:8002/v1
served_model: mlx-community/Qwen3.6-35B-A3B-4bit
capacity: 2
priority: 2
# Long-context admission policy (maps to edge.yaml nodes[].providers[]).
# Runtime KV budget is 262144 (max_kv_size / max_request_tokens), so only
# one full 262144-window long request fits. long_context_capacity is the
# long-slot count, distinct from the normal capacity (2).
total_context_tokens: 262144
long_context_capacity: 1
runtime:
api_key_policy: local_bearer_from_edge_yaml
workdir: /Users/toki/agent-work/iop-mlx-vllm
python: .venv/bin/python
package_baseline:
vllm_mlx: 0.3.0
mlx: 0.31.2
mlx_lm: 0.31.3
model_cache: /Users/toki/agent-work/iop-mlx-vllm/hf-cache
pid_file: /Users/toki/agent-work/iop-mlx-vllm/vllm-mlx.pid
stdout_log: /Users/toki/agent-work/iop-mlx-vllm/logs/vllm-mlx.stdout.log
stderr_log: /Users/toki/agent-work/iop-mlx-vllm/logs/vllm-mlx.stderr.log
max_num_seqs: 2
max_kv_size: 262144
max_request_tokens: 262144
default_max_tokens: 32768
paged_cache_block_size: 64
max_cache_blocks: 4096
total_kv_tokens: 262144
enable_auto_tool_choice: true
tool_call_parser: qwen
reasoning_parser: qwen3
default_chat_template_kwargs:
enable_thinking: true
- id: gx10-vllm-node
alias: gx10-vllm
role: vllm-provider
ssh: toki@192.168.0.91
workspace: /home/toki/iop-gx10-vllm
provider_pool_candidate: true
provider:
id: gx10-vllm
type: vllm
endpoint: http://192.168.0.91:8001/v1
served_model: laguna-s:2.1
capacity: 4
priority: 1
# Long-context admission policy (maps to edge.yaml nodes[].providers[]).
# Laguna NVFP4 runtime reports 373,570 GPU KV tokens and 1.43x maximum
# concurrency for full 262144-token requests, so only one long slot is allowed.
total_context_tokens: 373570
long_context_capacity: 1
runtime:
container_name: iop-vllm-laguna-s21
docker_image: iop-vllm-laguna-s21:v0.25.1-cu130
docker_image_id: sha256:4eb2bb71c0ffc6ceb54dd3381030169f5e24ac1477fb1faa1ba90e57e6996331
vllm_version: 0.25.1
max_model_len: 262144
max_num_seqs: 4
gpu_memory_utilization: 0.70
default_max_tokens: 65536
enable_auto_tool_choice: true
tool_call_parser: poolside_v1
reasoning_parser: poolside_v1
chat_template_host_path: /home/toki/iop-gx10-vllm/laguna-s-2.1-thinking.jinja
chat_template_container_path: /run/iop/laguna-s-2.1-thinking.jinja
thinking_generation_prefix: "<think>\n"
default_chat_template_kwargs:
enable_thinking: true
quantization: nvfp4
dtype: bfloat16
override_generation_config:
temperature: 0.7
top_p: 0.95
speculative_config:
model: poolside/Laguna-S-2.1-DFlash-NVFP4
num_speculative_tokens: 7
method: dflash
direct_providers:
- id: laguna-direct
family: poolside_laguna
type: vllm
endpoint: http://192.168.0.91:8001/v1
served_model: laguna-s:2.1
upstream_model: poolside/Laguna-S-2.1-NVFP4
upstream_url: https://huggingface.co/poolside/Laguna-S-2.1-NVFP4
upstream_revision: 07614121b31898586430f189d27a25a0be310843
status: direct_pi_provider_not_registered_runtime_shares_current_iop_laguna_gx10_provider
pi_model_parameters:
context_window: 262144
max_tokens: 65536
reasoning: true
thinking_format: chat-template
preserve_thinking: true
runtime:
container_name: iop-vllm-laguna-s21
docker_image: iop-vllm-laguna-s21:v0.25.1-cu130
docker_image_id: sha256:4eb2bb71c0ffc6ceb54dd3381030169f5e24ac1477fb1faa1ba90e57e6996331
vllm_version: 0.25.1
port_mapping: 0.0.0.0:8001->8000
hf_home: /models/.cache/huggingface
hf_cache_snapshot_path: /models/.cache/huggingface/hub/models--poolside--Laguna-S-2.1-NVFP4/snapshots/07614121b31898586430f189d27a25a0be310843
draft_cache_snapshot_path: /models/.cache/huggingface/hub/models--poolside--Laguna-S-2.1-DFlash-NVFP4/snapshots/723794750422b3efbf3a7b3af76dffb4ba035943
quantization: nvfp4
vllm_quantization_backend: compressed-tensors
dtype: bfloat16
max_model_len: 262144
max_num_seqs: 4
max_num_scheduled_tokens: 2024
gpu_memory_utilization: 0.70
enable_prefix_caching: true
enable_chunked_prefill: true
tensor_parallel_size: 1
pipeline_parallel_size: 1
data_parallel_size: 1
load_format: auto
kv_cache_dtype: float8_e4m3fn
enforce_eager: false
enable_auto_tool_choice: true
tool_call_parser: poolside_v1
reasoning_parser: poolside_v1
reasoning_parser_enable_in_reasoning: false
chat_template_host_path: /home/toki/iop-gx10-vllm/laguna-s-2.1-thinking.jinja
chat_template_container_path: /run/iop/laguna-s-2.1-thinking.jinja
thinking_generation_prefix: "<think>\n"
default_chat_template_kwargs:
enable_thinking: true
attention_backend: FLASHINFER
generation_config_override:
temperature: 0.7
top_p: 0.95
speculative_config:
model: poolside/Laguna-S-2.1-DFlash-NVFP4
num_speculative_tokens: 7
method: dflash
docker_runtime:
network_mode: bridge
ipc_mode: host
shm_size_bytes: 67108864
restart_policy: "no"
gpu_device_request: all
bind_mounts:
/home/toki/Data/models: /models
/home/toki/Data/vllm/cache/laguna-s21: /root/.cache
/home/toki/iop-gx10-vllm/laguna-s-2.1-thinking.jinja: /run/iop/laguna-s-2.1-thinking.jinja
available_kv_cache_memory_gib: 13.08
gpu_kv_cache_tokens: 373711
max_concurrency_for_262144_token_requests: 1.43
thinking_note: The poolside_v1 parser emits reasoning in the provider-native reasoning field. The local chat template uses a trailing newline after the generation-prefix think tag; Pi high emits thinking deltas and Pi off emits no thinking event.
- id: onexplayer-lemonade-node
alias: onexplayer-lemonade
role: lemonade-provider
ssh: r0bin@192.168.0.59
ssh_origin: current_host
workspace: C:/Users/r0bin/iop-field
provider_pool_candidate: true
provider:
id: onexplayer-lemonade
type: lemonade
endpoint: http://192.168.0.59:13305/v1
served_model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
served_model_response: ornith-1.0-35b-Q5_K_M.gguf
previous_served_model: Qwen3.6-35B-A3B-MTP-GGUF
status: active_ornith_q5_current_iop_ornith_provider_pool_member
qwen_reenable_profile:
status: not_loaded_ornith_q5_active
model_name: Qwen3.6-35B-A3B-MTP-GGUF
checkpoint: unsloth/Qwen3.6-35B-A3B-MTP-GGUF:Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf
prior_saved_recipe_options:
ctx_size: 524288
llamacpp_args: "--spec-type none -np 3 -cb -fa on -b 4096 -ub 1024"
partition_note: The prior -np 3 profile fixed slot windows at about 174848 tokens for ctx_size 524288.
recommended_recipe_options:
ctx_size: 524288
llamacpp_args: "--spec-type none -cb -fa on -b 4096 -ub 1024 --kv-unified"
removed_args: "-np 3"
expected_slot_n_ctx: 262144
expected_auto_slots: 4
iop_capacity_guard: 3
long_context_capacity: 2
validation_required_after_load: true
validation_basis: Same Lemonade llama.cpp backend and current Ornith Q5 runtime confirmed --kv-unified with omitted -np prevents fixed ctx_size/parallel partitioning.
capacity: 3
priority: 2
# Long-context admission policy (maps to edge.yaml nodes[].providers[]).
# Ornith Q5 direct runtime keeps ctx_size=524288 and removes fixed -np
# partitioning by using llama.cpp unified KV. Do NOT raise ctx_size above 524288.
total_context_tokens: 524288
long_context_capacity: 2
load:
endpoint: http://192.168.0.59:13305/v1/load
model_name: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
checkpoint: LordNeel/Ornith-1.0-35B-GGUF-llamacpp-tp1:Q5_K_M
hf_url: https://huggingface.co/LordNeel/Ornith-1.0-35B-GGUF-llamacpp-tp1
hf_revision: c50d5d4407f70e43208dee836c66bb8a05c1be91
hf_ref: main
gguf_file: ornith-1.0-35b-Q5_K_M.gguf
gguf_file_size_bytes: 24729130848
cache_snapshot_path: C:/Users/r0bin/.cache/huggingface/hub/models--LordNeel--Ornith-1.0-35B-GGUF-llamacpp-tp1/snapshots/c50d5d4407f70e43208dee836c66bb8a05c1be91
backend: vulkan
ctx_size: 524288
llamacpp_args: "--spec-type none -cb -fa on -b 4096 -ub 1024 --kv-unified --temp 0.6 --top-p 0.95 --top-k 20"
observed_process_args: "--ctx-size 524288 --port 8001 --jinja --context-shift --keep 16 --reasoning-format auto --no-webui --no-mmap -ngl 99 --kv-unified --spec-type none --temp 0.6 --top-k 20 --top-p 0.95 -b 4096 -cb -fa on -ub 1024"
observed_total_slots: 4
observed_total_ctx_size: 524288
observed_slot_n_ctx: 262144
context_per_slot_note: -np is intentionally omitted; llama.cpp auto slots with --kv-unified avoid the prior fixed 524288/3 partition and report 262144-token slot windows backed by shared KV.
residency:
observed_at: "2026-07-12"
process_resident_policy: keep llama-server loaded until explicit lemonade unload or service/process restart
observed_process_running: true
observed_is_sleeping: false
idle_sleep_seconds: not_set_llama_cpp_default_minus_1_disabled
mmap_policy: disabled_via_no_mmap
memory_lock_policy: not_enabled_no_mlock_argument
note: This keeps the model warm in the active llama-server process, but it is not an OS-level hard memory pin; Windows can still page memory under pressure because --mlock is not set.
save_options: true
mtp_runtime_policy: disabled_via_spec_type_none
default_max_tokens: 32768
tool_choice_auto_verified: true
smoke:
observed_at: "2026-07-12"
qwen_unloaded: true
load_status: passed
english_final_answer:
max_tokens: 512
result: passed
content: 2+2 equals 4.
completion_tokens: 414
generation_tok_s: 31.94
korean_final_answer:
max_tokens: 1024
result: passed
content: 2 더하기 2는 4입니다.
completion_tokens: 384
tool_choice_auto:
result: passed
finish_reason: tool_calls
tool_calls: 1
function_name: get_weather
arguments: "{\"city\":\"Seoul\"}"
unified_kv_concurrency:
observed_at: "2026-07-12"
result: passed
concurrent_requests: 3
finish_reason_stop: 3
final_content: "4"
total_elapsed_sec: 6.55
completion_tokens:
- 149
- 88
- 85
low_output_budget_note: max_tokens 128 and one Korean 512-token probe ended in reasoning_content with empty final content; keep Pi/Edge maxTokens at 32768 or use at least 1024 for short smoke prompts.
windows_process_start:
ownership: user_managed_manual_toggle
observed_at: "2026-07-26"
boot_autostart:
status: disabled
startup_shortcuts_matching_iop_or_lemonade: absent
registry_run_entries_matching_iop_or_lemonade: absent
scheduled_task: absent
windows_services_matching_iop_or_lemonade: absent
removed_task: IOP-OnexNode
recoverable_backup: C:/Users/r0bin/iop-field/IOP-OnexNode.pre-manual-lemonade-20260725T234232Z.xml
manual_remote_llm_toggle:
script: C:/Users/r0bin/iop-field/remote-llm-toggle.ps1
script_sha256: cd04150b460da9d3157e9431f0b25f2d80e63f957a7a14448feb6925b3ada11f
log: C:/Users/r0bin/iop-field/onex-remote-llm-toggle.log
default_action: toggle_by_complete_stack_readiness
process_start: Win32_Process.Create
readiness_requires:
- LemonadeServer.exe running and health status ok
- exact Ornith model with Vulkan ctx_size 524288 profile
- public listener 0.0.0.0:13305
- four llama slots with n_ctx 262144
- iop-node.exe running and established Edge TCP connection to port 18084
up_sequence:
- start LemonadeServer.exe
- load Ornith Q5 model with the saved Vulkan profile
- start iop-node.exe with node.yaml
down_sequence:
- stop iop-node.exe
- unload all Lemonade models
- stop residual llama-server and LemonadeServer processes
taskbar_pin:
status: pinned
visual_confirmation: deferred_no_active_explorer_session
shortcut: C:/Users/r0bin/AppData/Roaming/Microsoft/Internet Explorer/Quick Launch/User Pinned/TaskBar/OneX Remote LLM Toggle.lnk
shortcut_sha256: 6f148a9ef852091cefe86b227b6d81fc0c8b1a84090fb53d3b304453881505a8
start_menu_shortcut: C:/Users/r0bin/AppData/Roaming/Microsoft/Windows/Start Menu/Programs/Lemonade Server/OneX Remote LLM Toggle.lnk
target: C:/Windows/System32/WindowsPowerShell/v1.0/powershell.exe
launches_script: C:/Users/r0bin/iop-field/remote-llm-toggle.ps1
icon: C:/Users/r0bin/AppData/Roaming/Microsoft/Installer/{221D1879-DDCE-4E14-AC7C-ACA9F084FE21}/LemonadeIcon,0
boot_autostart: false
validation:
observed_at: "2026-07-26"
default_toggle_down: passed
default_toggle_up: passed
final_state: ready
listener_public: passed
model_profile_valid: passed
edge_connected: passed
slots: 4x262144
- id: rtx5090-lemonade-node
alias: rtx5090-lemonade
role: lemonade-provider
ssh: r0bin@192.168.0.111
ssh_origin: current_host
ssh_access:
alias: iop-dev-rtx5090
auth: public_key
identity_source: host_local_ssh_config
windows_authorized_keys: C:/ProgramData/ssh/administrators_authorized_keys
public_key_fingerprint: SHA256:70d8kwBt0azcGkHXIOZ6kLcbPaoVGWusEFDpl5UBH4k
batch_mode_verified_at: "2026-07-18"
workspace: C:/Users/r0bin/iop-field
provider_pool_candidate: true
provider:
id: rtx5090-lemonade
type: lemonade
endpoint: http://192.168.0.111:13305/v1
served_model: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
served_model_response: ornith-1.0-35b-Q5_K_M.gguf
status: active_ornith_q5_current_iop_ornith_provider_pool_member
capacity: 1
priority: 0
total_context_tokens: 262144
long_context_capacity: 1
hardware:
gpu: NVIDIA GeForce RTX 5090
vram_total_mib: 32607
memory_policy: Lemonade uses system RAM for host-side state and the model/KV allocation uses VRAM
load:
endpoint: http://192.168.0.111:13305/v1/load
model_name: Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
checkpoint: LordNeel/Ornith-1.0-35B-GGUF-llamacpp-tp1:Q5_K_M
hf_url: https://huggingface.co/LordNeel/Ornith-1.0-35B-GGUF-llamacpp-tp1
hf_revision: c50d5d4407f70e43208dee836c66bb8a05c1be91
hf_ref: main
gguf_file: ornith-1.0-35b-Q5_K_M.gguf
gguf_file_size_bytes: 24729130848
models_dir: D:/Models
cache_snapshot_path: D:/Models/models--LordNeel--Ornith-1.0-35B-GGUF-llamacpp-tp1/snapshots/c50d5d4407f70e43208dee836c66bb8a05c1be91
backend: cuda
listen_host: 0.0.0.0
lemonade_version: 11.0.0
llamacpp_backend_version: b9851
iop_node_source_commit: 0ffcb88
iop_node_binary_sha256: 8fffd06fa4c14e03859be3a0c04f004275cce44848edcfd6f4f71d3c514fa384
iop_node_binary_observed_at: "2026-07-23T10:11:52Z"
iop_node_version: "0.1.0"
ctx_size: 262144
llamacpp_args: "--chat-template-kwargs '{\"preserve_thinking\":true}' --kv-unified --min-p 0.00 --repeat-penalty 1.0 --spec-type none --temp 0.6 --top-k 20 --top-p 0.95 -b 512 -cb -ctk q8_0 -ctv q8_0 -fa on -np 1 -ub 256"
observed_process_args: "--ctx-size 262144 --port 8001 --jinja --metrics --reasoning-format auto --no-webui --chat-template-kwargs {\"preserve_thinking\":true} --kv-unified --min-p 0.00 --repeat-penalty 1.0 --spec-type none --temp 0.6 --top-k 20 --top-p 0.95 -b 512 -cb -ctk q8_0 -ctv q8_0 -fa on -np 1 -ub 256"
observed_total_slots: 1
observed_total_ctx_size: 262144
observed_slot_n_ctx: 262144
kv_cache_key_type: q8_0
kv_cache_value_type: q8_0
default_max_tokens: 32768
tool_choice_auto_verified: true
residency:
observed_at: "2026-07-18"
model_loaded: true
gpu_memory_used_mib: 28637
gpu_memory_free_mib: 3551
comparative_benchmark:
status: completed
observed_at: "2026-07-18"
requested_metric: output_tokens_per_second
comparison_scope: direct provider sequential concurrency-1 benchmark against GX10 and OneXPlayer; current dev has no RTX4090 node
request:
endpoint: /v1/chat/completions
prompt_tokens: 48
max_tokens: 1024
temperature: 0
thinking_enabled: false
result:
http_status: 200
finish_reason: length
completion_tokens: 1024
elapsed_sec: 5.092999
end_to_end_tok_s: 201.060
server_decode_tok_s: 245.83231403349612
relative_to_gx10: 5.15
relative_to_onexplayer: 5.95
edge_action: none
smoke:
observed_at: "2026-07-18"
direct_chat_completions: passed
direct_tool_choice_auto:
result: passed
finish_reason: tool_calls
tool_calls: 1
function_name: get_weather
edge_chat_completions:
isolated_alias_http_status: 200
isolated_alias_elapsed_sec: 0.320
active_alias_dispatch_closed: true
active_alias_estimated_input_tokens: 49306
active_alias_edge_elapsed_sec: 15.414
active_alias_lemonade_input_tokens: 60071
active_alias_lemonade_output_tokens: 89
active_alias_time_to_first_token_sec: 14.664143
active_alias_generation_tokens_per_sec: 174.713
concurrency_one_enforced: true
overlapping_requests_routed_to_other_providers: 2
windows_process_start:
ownership: user_managed_manual_toggle
observed_at: "2026-07-26"
boot_autostart:
status: disabled
iop_node_shortcut: absent
lemonade_server_shortcut: absent
automation_shortcut: absent
registry_run_entries_matching_iop_or_lemonade: absent
scheduled_tasks_matching_iop_or_lemonade: absent
windows_services_matching_iop_or_lemonade: absent
unrelated_startup_shortcuts_preserved: true
previous_automation_shortcut: C:/Users/r0bin/AppData/Roaming/Microsoft/Windows/Start Menu/Programs/Startup/startup.lnk
previous_automation_target: C:/Work/automation/iot/script/windows/startup.bat
recoverable_backup: C:/Users/r0bin/iop-field/startup.lnk.pre-manual-remote-llm-20260726.bak
task_scheduler:
status: absent
removed_at: "2026-07-20"
manual_remote_llm_toggle:
script: C:/Users/r0bin/iop-field/remote-llm-toggle.ps1
script_sha256: 2b3c7084b1105dc2c6809c2cb8b3440034422ef96e3b63bc8b8afa9a551b02e6
log: C:/Users/r0bin/iop-field/toggle-remote-llm.log
default_action: toggle_by_complete_stack_readiness
pre_review_backup: C:/Users/r0bin/iop-field/toggle-remote-llm.ps1.pre-integrated-review-20260726.bak
readiness_requires:
- LemonadeServer.exe running and health status ok
- exact Ornith model and full saved runtime profile
- public listener 0.0.0.0:13305
- iop-node.exe running
- established Edge TCP connection to port 18084
down_completion_requires:
- no iop-node.exe
- no LemonadeServer.exe
- no Lemonade-managed llama-server.exe
- no loaded expected model
- no listener on port 13305
up_sequence:
- LemonadeServer.exe
- load Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
- iop-node.exe serve --config C:/Users/r0bin/iop-field/node.yaml
down_sequence:
- stop iop-node.exe
- unload Ornith-1.0-35B-GGUF-llamacpp-tp1-Q5_K_M
- stop LemonadeServer.exe and residual llama-server.exe
process_start: Win32_Process.Create
invoker:
ahk: C:/Work/automation/iot/script/windows/ahk/RemoteLLM_mode.ahk
ahk_sha256: 2dac2dab58aac7f7c235ed2726b7e0867d7cfb23085ffbbd8625eed3d6ee87f7
run_count: 1
direct_lemonade_load: absent
removed_task_scheduler_reference: true
taskbar_pin:
status: pinned
shortcut: C:/Users/r0bin/AppData/Roaming/Microsoft/Internet Explorer/Quick Launch/User Pinned/TaskBar/Remote LLM Toggle.lnk
shortcut_sha256: 4a3a9670d2d626df41c04cf6e155ad83b2298872469d92cb6e8251112fd2b8e7
start_menu_shortcut: C:/Users/r0bin/AppData/Roaming/Microsoft/Windows/Start Menu/Programs/Lemonade Server/Remote LLM Toggle.lnk
start_menu_shortcut_sha256: 4a3a9670d2d626df41c04cf6e155ad83b2298872469d92cb6e8251112fd2b8e7
target: C:/Windows/System32/WindowsPowerShell/v1.0/powershell.exe
launches_script: C:/Users/r0bin/iop-field/remote-llm-toggle.ps1
icon: C:/Users/r0bin/AppData/Roaming/Microsoft/Installer/{FECA419B-2EB4-46C1-8FAE-D392B4059880}/LemonadeIcon,0
boot_autostart: false
validation:
observed_at: "2026-07-26"
profile_validation:
valid_saved_profile_accepted: passed
malformed_preserve_thinking_json_rejected: passed
wrong_temperature_rejected: passed
wrong_min_p_rejected: passed
default_toggle_down:
result: passed
related_process_count: 0
listener_13305: absent
health_endpoint: unreachable
edge_connected: false
provider_health: offline
provider_capacity: 0
explicit_down_already_down:
result: passed
default_toggle_up:
result: passed
stack_ready: true
listener: 0.0.0.0:13305
model_profile_valid: true
edge_connected: true
provider_health: healthy
provider_capacity: 1
direct_chat_completions:
result: passed
finish_reason: stop
final_content: READY.
final_state: up
legacy_node_launcher: C:/Users/r0bin/iop-field/run-iop-node.cmd
current_reconnect:
interval_sec: 10
max_attempts: 10
long_outage_autonomous_recovery: false
post_refresh_connection:
observed_at: "2026-07-23"
result: reconnected_to_edge_after_tcp_read_header_error
ornith_concurrency_observation:
observed_at: "2026-07-23"
model: ornith:35b
request_shape: concurrent_long_korean_streaming
result: provider_parser_error_before_downstream_release
provider_error:
http_status: 500
message_prefix: Failed to parse input at pos
repetition_loop: not_observed_or_confirmed
raw_evidence: ignored_agent_test_run_only
sanitized_evidence: agent-roadmap/sdd/knowledge-tool-optimization-extension/openai-compatible-output-validation-filters/evidence/2026-07-23-ornith-llama-server-parser-error.log