224 lines
7 KiB
YAML
224 lines
7 KiB
YAML
test_env: dev
|
|
profile: dev-runtime-provider-pool
|
|
last_updated_at: "2026-07-04"
|
|
|
|
source:
|
|
remote_runner:
|
|
ssh: toki@toki-labs.com
|
|
repo_root: /Users/toki/agent-work/iop-dev
|
|
clean_sync:
|
|
- git fetch origin main
|
|
- git reset --hard origin/main
|
|
- git clean -fd
|
|
dirty_policy: discard
|
|
|
|
edge:
|
|
id: edge-toki-labs-dev
|
|
config_path: build/dev-runtime/edge.yaml
|
|
control_plane_http: http://127.0.0.1:18001
|
|
control_plane_status_url: http://127.0.0.1:18001/edges/edge-toki-labs-dev/status
|
|
bootstrap_http_public: http://toki-labs.com:18082
|
|
openai_base_url_public: http://toki-labs.com:18083/v1
|
|
openai_base_url_runner: http://127.0.0.1:18083/v1
|
|
edge_node_tcp_public: toki-labs.com:18084
|
|
admin_addr_runner: 127.0.0.1:19093
|
|
|
|
build:
|
|
binaries:
|
|
edge: build/dev-runtime/bin/edge
|
|
node_macos: build/dev-runtime/bin/iop-node
|
|
node_linux_arm64: build/dev-runtime/bin/iop-node-linux-arm64
|
|
node_windows_amd64: build/dev-runtime/bin/iop-node-windows-amd64.exe
|
|
|
|
model:
|
|
alias: qwen3.6:35b
|
|
provider_capacity_total: 9
|
|
provider_capacity_status: verified_with_control_plane_provider_snapshots_2026_07_04
|
|
context_window: 262144
|
|
default_max_tokens: 32768
|
|
min_max_tokens: 32768
|
|
default_thinking_token_budget: 1024
|
|
reasoning_policy: bounded_thinking
|
|
content_completion_policy:
|
|
enforce_min_max_tokens: true
|
|
reason: prevent reasoning-only completions from exhausting caller max_tokens before final content
|
|
dispatch_policy:
|
|
capacity_gate: providers at or above capacity are excluded from dispatch candidates
|
|
selection_order: lower in_flight level wins among providers with available capacity
|
|
same_in_flight_tiebreak: lower numeric priority first, then deterministic rotation
|
|
capacity_smoke:
|
|
endpoints:
|
|
- /v1/responses
|
|
- /v1/chat/completions
|
|
concurrent_requests: 9
|
|
expected_total_in_flight: 9
|
|
expected_min_queued: 0
|
|
capacity_plus_one_concurrent_requests: 10
|
|
capacity_plus_one_expected_total_in_flight: 9
|
|
capacity_plus_one_expected_min_queued: 1
|
|
extended_concurrent_requests: 13
|
|
extended_expected_min_queued: 4
|
|
prompt_policy: long_reasoning_allowed
|
|
exact_output_match: false
|
|
latest_provider_snapshot:
|
|
observed_at: "2026-07-04"
|
|
total_capacity: 9
|
|
all_nodes_connected: true
|
|
providers:
|
|
gx10-vllm:
|
|
capacity: 4
|
|
priority: 0
|
|
in_flight: 0
|
|
queued: 0
|
|
health: healthy
|
|
served_model: nvidia/Qwen3.6-35B-A3B-NVFP4
|
|
mac-mlx-vllm:
|
|
capacity: 2
|
|
priority: 2
|
|
in_flight: 0
|
|
queued: 0
|
|
health: healthy
|
|
served_model: mlx-community/Qwen3.6-35B-A3B-4bit
|
|
onexplayer-lemonade:
|
|
capacity: 3
|
|
priority: 1
|
|
in_flight: 0
|
|
queued: 0
|
|
health: healthy
|
|
served_model: Qwen3.6-35B-A3B-MTP-GGUF
|
|
final_content_smoke:
|
|
endpoint: /v1/chat/completions
|
|
concurrent_requests: 10
|
|
current_capacity_total: 9
|
|
expected_peak_total_in_flight: 9
|
|
expected_min_queued: 1
|
|
request_max_tokens: 3072
|
|
effective_min_max_tokens: 32768
|
|
include_reasoning: false
|
|
expected_success_count: 10
|
|
expected_finish_reason: stop
|
|
expected_final_marker: true
|
|
expected_empty_content: 0
|
|
expected_iop_notice: 0
|
|
previous_observed_at: "2026-07-04"
|
|
previous_observed_capacity_total: 10
|
|
previous_observed_result:
|
|
http_ok: 10
|
|
success_count: 10
|
|
failure_count: 0
|
|
finish_reason_stop: 10
|
|
marker_missing_count: 0
|
|
empty_content_count: 0
|
|
iop_notice_count: 0
|
|
think_tag_leak_count: 0
|
|
reasoning_present_count: 0
|
|
aggregate_completion_tok_s: 62.5
|
|
total_completion_tokens: 17933
|
|
batch_wall_sec: 286.908
|
|
avg_latency_sec: 175.074
|
|
peak_total_in_flight: 10
|
|
peak_total_queued: 0
|
|
provider_peaks:
|
|
gx10-vllm:
|
|
capacity: 4
|
|
max_in_flight: 4
|
|
max_queued: 0
|
|
mac-mlx-vllm:
|
|
capacity: 3
|
|
max_in_flight: 3
|
|
max_queued: 0
|
|
onexplayer-lemonade:
|
|
capacity: 3
|
|
max_in_flight: 3
|
|
max_queued: 0
|
|
|
|
nodes:
|
|
- id: mac-codex-node
|
|
alias: mac-codex
|
|
role: cli+mlx-provider
|
|
ssh: toki@toki-labs.com
|
|
workspace: /Users/toki/agent-work/iop-workspace/nomadcode
|
|
provider_pool_candidate: true
|
|
adapters:
|
|
- cli
|
|
- mac-mlx-vllm
|
|
providers:
|
|
- id: mac-mlx-vllm
|
|
type: vllm-mlx
|
|
endpoint: http://127.0.0.1:8002/v1
|
|
served_model: mlx-community/Qwen3.6-35B-A3B-4bit
|
|
capacity: 2
|
|
priority: 2
|
|
runtime:
|
|
api_key_policy: local_bearer_from_edge_yaml
|
|
workdir: /Users/toki/agent-work/iop-mlx-vllm
|
|
python: .venv/bin/python
|
|
package_baseline:
|
|
vllm_mlx: 0.3.0
|
|
mlx: 0.31.2
|
|
mlx_lm: 0.31.3
|
|
model_cache: /Users/toki/agent-work/iop-mlx-vllm/hf-cache
|
|
pid_file: /Users/toki/agent-work/iop-mlx-vllm/vllm-mlx.pid
|
|
stdout_log: /Users/toki/agent-work/iop-mlx-vllm/logs/vllm-mlx.stdout.log
|
|
stderr_log: /Users/toki/agent-work/iop-mlx-vllm/logs/vllm-mlx.stderr.log
|
|
max_num_seqs: 2
|
|
max_kv_size: 262144
|
|
max_request_tokens: 262144
|
|
default_max_tokens: 32768
|
|
default_thinking_token_budget: 1024
|
|
paged_cache_block_size: 64
|
|
max_cache_blocks: 4096
|
|
total_kv_tokens: 262144
|
|
tool_call_parser: qwen
|
|
reasoning_parser: qwen3
|
|
default_chat_template_kwargs:
|
|
enable_thinking: true
|
|
- id: gx10-vllm-node
|
|
alias: gx10-vllm
|
|
role: vllm-provider
|
|
ssh: toki@192.168.0.91
|
|
workspace: /home/toki/iop-gx10-vllm
|
|
provider_pool_candidate: true
|
|
provider:
|
|
id: gx10-vllm
|
|
type: vllm
|
|
endpoint: http://192.168.0.91:8001/v1
|
|
served_model: nvidia/Qwen3.6-35B-A3B-NVFP4
|
|
capacity: 4
|
|
priority: 0
|
|
runtime:
|
|
container_name: iop-vllm-qwen36-prev-20260704063128
|
|
max_model_len: 262144
|
|
max_num_seqs: 4
|
|
gpu_memory_utilization: 0.30
|
|
default_max_tokens: 32768
|
|
default_thinking_token_budget: 1024
|
|
reasoning_parser: qwen3
|
|
default_chat_template_kwargs:
|
|
enable_thinking: true
|
|
- id: onexplayer-lemonade-node
|
|
alias: onexplayer-lemonade
|
|
role: lemonade-provider
|
|
ssh: r0bin@192.168.0.59
|
|
ssh_origin: current_host
|
|
workspace: C:/Users/r0bin/iop-field
|
|
provider_pool_candidate: true
|
|
provider:
|
|
id: onexplayer-lemonade
|
|
type: lemonade
|
|
endpoint: http://192.168.0.59:13305/v1
|
|
served_model: Qwen3.6-35B-A3B-MTP-GGUF
|
|
capacity: 3
|
|
priority: 1
|
|
load:
|
|
endpoint: http://192.168.0.59:13305/v1/load
|
|
model_name: Qwen3.6-35B-A3B-MTP-GGUF
|
|
backend: vulkan
|
|
ctx_size: 524288
|
|
llamacpp_args: "--spec-type none -np 3 -cb -fa on -b 4096 -ub 1024"
|
|
observed_slot_n_ctx: 174848
|
|
save_options: true
|
|
mtp_runtime_policy: disabled
|
|
default_max_tokens: 32768
|
|
default_thinking_token_budget: 1024
|
|
windows_process_start: Win32_Process.Create
|