iop/agent-test/dev-corp/inventory.yaml

729 lines
34 KiB
YAML

test_env: dev-corp
profile: dev-corp-provider-pool
last_updated_at: "2026-07-13"
source:
remote_runner:
ssh: fe@172.24.63.178
repo_root: /Users/fe/agent-work/iop-dev-corp
setup_required: true
setup_status: deployed
current_observation: /Users/fe/agent-work/iop-dev-corp checkout is the mac-mini source/build/provider SSH runner only. As of 2026-07-09 the active dev-corp Edge is the public host iop.ai.kr (115.21.224.82), and provider nodes must use iop.ai.kr:18087 directly.
clean_sync:
- git fetch origin main
- git reset --hard origin/main
- git clean -fd
dirty_policy: discard
compose:
project_name: iop-dev-corp-agent
network: iop-dev-corp-agent-net
subnet: 10.89.2.0/24
env_file: .env.dev-corp.example
runtime_root: /Users/fe/agent-work/iop-dev-corp
setup_status: deployed_control_plane_observability_web_blocked
deployed_at: "2026-07-13T15:32:00+09:00"
prometheus_config: ./configs/prometheus/prometheus.dev-corp.yml
deployment_note: Control Plane, PostgreSQL, Redis, Prometheus, and Grafana are running in the dev-corp compose project on the mac-mini runner. The compose web service is not started because the runner Flutter/Dart SDK is below the app SDK constraint and required sibling path dependencies are absent.
services:
- postgres
- redis
- control-plane
- web
- prometheus
- grafana
host_ports:
web: 13002
control_plane_http: 18002
control_plane_client_ws: 19004
control_plane_edge_wire: 19005
edge_node_tcp_compose: 19006
postgres: 15402
redis: 16302
edge_metrics: 19102
control_plane_metrics: 19104
prometheus: 19112
grafana: 19122
active_services:
- postgres
- redis
- control-plane
- prometheus
- grafana
blocked_services:
web:
status: not_started
reason: runner Flutter uses Dart 3.10.0 while apps/client requires ^3.11.3, and sibling path dependencies agent-shell/nexo are not present under /Users/fe/agent-work.
latest_verification:
observed_at: "2026-07-13T15:32:00+09:00"
control_plane_healthz: passed
control_plane_readyz: passed
client_ws_upgrade_runner: passed
edge_wire_hello: passed
edge_status_nodes: 0
prometheus_ready: passed
prometheus_targets:
iop-control-plane: up
iop-edge-native: up
prometheus: up
grafana_health: passed
public_control_plane_ports:
13002: closed
18002: closed
19004: closed
public_edge_openai_models:
- gemma4:26b
- ornith:35b
edge:
id: dev-corp-edge
host: iop.ai.kr
public_ip: 115.21.224.82
ssh: toki@iop.ai.kr
runtime_root: /Users/toki/agent-work/iop-dev-corp
config_path: /Users/toki/agent-work/iop-dev-corp/build/dev-corp-runtime/edge.yaml
config_note: public Edge config and bootstrap artifacts were normalized to iop.ai.kr on 2026-07-09.
control_plane_enabled_current_runtime: unknown
control_plane_http: http://iop.ai.kr:18002
control_plane_status_url: http://iop.ai.kr:18002/edges/dev-corp-edge/status
control_plane_client_ws_runner: ws://iop.ai.kr:19004/client
control_plane_edge_wire_addr_runner: iop.ai.kr:19005
public_route_policy: active_dev_corp_edge_uses_iop_ai_kr
runner_route_policy: mac-mini is source/build/provider management only, not an Edge route.
public_route_note: iop.ai.kr is the dev-corp Edge runtime and the only default route for bootstrap, OpenAI-compatible base URL, and Node edge_addr.
control_plane_http_public: http://iop.ai.kr:18002
control_plane_client_ws_public: ws://iop.ai.kr:19004/client
bootstrap_http_public: http://iop.ai.kr:18085
bootstrap_http_node_internal: http://iop.ai.kr:18085
openai_base_url_public: https://digitalplatform.iop.ai.kr/v1
openai_base_url_public_http: http://digitalplatform.iop.ai.kr/v1
openai_base_url_direct_edge_listener: http://digitalplatform.iop.ai.kr:18086/v1
openai_base_url_node_internal: http://digitalplatform.iop.ai.kr:18086/v1
openai_base_url_runner: https://digitalplatform.iop.ai.kr/v1
openai_api_key_required_current_runtime: true
openai_api_key_secret_path_remote: build/dev-corp-runtime/.secrets/openai_api_key
openai_api_key_value_tracked: false
edge_node_tcp_public: iop.ai.kr:18087
edge_node_tcp_node_internal: iop.ai.kr:18087
admin_addr_runner: 127.0.0.1:19094
latest_public_edge_reconnect:
observed_at: "2026-07-09"
node_edge_addr: iop.ai.kr:18087
moved_from:
- retired reverse tunnel route
- retired mac-mini local route
nodes_registered:
- corp-dgx-spark-01-vllm-node
- corp-dgx-spark-02-vllm-node
- corp-mac-studio-mlx-vllm-node
public_chat_completions_15_historical_routing_evidence:
ok: 15
concurrent_requests: 15
elapsed_sec: 5.678
passthrough_reasoning_stream_seen: 15
manual_iop_response_id_count: 0
remote_log: build/dev-corp-runtime/logs/public_latest_edge_node_chat_15_20260709_181222.json
latest_node_binary_deploy:
observed_at: "2026-07-09T18:12:22+09:00"
source_ref: e8c240ea23a29f1568326b6d162913863b3dca33
linux_arm64_sha256: 02441a57f97238bd650a568896c876fb32c94fcace0073f6068509ef90ac4327
darwin_arm64_sha256: 95c59ee0dacb134dc2c6d46fb34f2dac49c33d7da38abe96157583822364f57a
nodes:
corp-dgx-spark-01-vllm-node:
pid_after_restart: 545848
binary_sha256: 02441a57f97238bd650a568896c876fb32c94fcace0073f6068509ef90ac4327
corp-dgx-spark-02-vllm-node:
pid_after_restart: 701517
binary_sha256: 02441a57f97238bd650a568896c876fb32c94fcace0073f6068509ef90ac4327
corp-mac-studio-mlx-vllm-node:
pid_after_restart: 87325
binary_sha256: 95c59ee0dacb134dc2c6d46fb34f2dac49c33d7da38abe96157583822364f57a
note: provider-pool passthrough contract preserves selected provider OpenAI-compatible fields; use /v1/chat/completions for capacity smoke, and treat /v1/responses as provider-dependent passthrough/relay evidence rather than an IOP-level unsupported route.
latest_bootstrap_domain_fix:
observed_at: "2026-07-11"
public_host: iop.ai.kr
runtime_pid_after_restart: 19178
config:
advertise_host: iop.ai.kr
artifact_base_url: http://iop.ai.kr:18085
bootstrap_defaults:
artifact_base_url: http://iop.ai.kr:18085
edge_addr: iop.ai.kr:18087
verification:
public_bootstrap_defaults: passed
public_chat_completions_single: passed
public_chat_completions_new_domain_15: passed
public_chat_completions_default_passthrough_15: passed
build:
binaries:
control_plane: build/dev-corp-runtime/bin/control-plane
edge: build/dev-corp-runtime/bin/iop-edge
node_macos: build/dev-corp-runtime/bin/iop-node-darwin-arm64
node_linux_arm64: build/dev-corp-runtime/bin/iop-node-linux-arm64
node_windows_amd64: null
node_windows_amd64_note: not part of the default dev-corp provider pool
model:
alias: gemma4:26b
alias_status: active_multi_model_pool_primary_alias
primary_alias: gemma4:26b
additional_aliases:
- ornith:35b
alias_policy: expose IOP model aliases per served model and map each alias to its current device/provider pool.
provider_capacity_total: 13
provider_capacity_status: verified_with_control_plane_provider_snapshots_2026_07_13_device_sync
aliases:
gemma4:26b:
capacity_total: 5
providers:
- corp-mac-studio-mlx-vllm
served_model: mlx-community/gemma-4-26b-a4b-it-nvfp4
device_policy: served only from Mac Studio in current dev-corp provider pool.
default_thinking_token_budget: 1024
reasoning_policy: bounded_thinking_enabled
think_policy:
default: enabled
strict_output_behavior: provider_pool_catalog_default_overrides_strict_disable
adapter_mapping: vllm_mlx_uses_chat_template_kwargs_enable_thinking_true
ornith:35b:
capacity_total: 8
providers:
- corp-dgx-spark-01-ornith
- corp-dgx-spark-02-ornith
served_model: ornith:35b
device_policy: served from Spark01 and Spark02 in current dev-corp provider pool.
default_thinking_token_budget: null
reasoning_policy: reasoning_parser_enabled_without_catalog_thinking_budget
think_policy:
default: provider_runtime_default
adapter_mapping: qwen3_reasoning_parser_without_model_catalog_thinking_budget
context_window_max: 262144
thinking_policy_scope: per_alias
capacity_planning_policy:
observed_at: "2026-07-13"
applies_to:
- gemma4:26b
- ornith:35b
max_context_tokens: 262144
catalog_capacity_meaning: provider admission/concurrency target for typical requests, not a guarantee that every admitted request consumes 100% of max context.
typical_request_context_ratio_of_max: "0.50-0.70"
typical_request_context_scope: prompt tokens plus generated tokens that occupy KV cache during the request.
kv_lower_bound_ratio_per_capacity_slot: "0.50"
kv_operating_target_ratio_per_capacity_slot: "0.70-0.75"
full_context_concurrency_metric: "vLLM startup Maximum concurrency for 262,144 tokens per request is a diagnostic upper-bound metric, not the catalog capacity requirement."
capacity_4:
lower_bound_kv_tokens: 524288
lower_bound_full_context_concurrency: 2.00
operating_target_kv_tokens_min: 734004
operating_target_kv_tokens_max: 786432
operating_target_full_context_concurrency_min: 2.80
operating_target_full_context_concurrency_max: 3.00
interpretation: capacity 4 is operationally suitable when runtime KV covers roughly 2.8x-3.0x full 262144-token requests; 2.0x-2.8x is only a lower-bound/test-line range and should not be described as covering four 70% context requests.
operating_note: Do not mark a provider under-capacity solely because full-context concurrency is below catalog capacity; for capacity 4, prefer 2.8x-3.0x full-context concurrency and treat 2.0x as the minimum lower bound requiring workload-specific smoke evidence.
spark_ornith_profile:
observed_at: "2026-07-13"
status: active_iop_provider_pool_and_direct_test_lines
family: spark_ornith
model_id: ornith:35b
display_name: Ornith 1.0 35B FP8 vLLM
api: openai-completions
route_policy: Spark Ornith endpoints are connected to the dev-corp Edge/provider pool as IOP providers; device inventory uses provider-internal endpoints and node-local Edge adapter endpoints only.
upstream_model: deepreinforce-ai/Ornith-1.0-35B-FP8
upstream_url: https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B-FP8
upstream_revision: 1ab57ce0b44950e498a88756f40ad1ed4d0f30ca
docker_image: vllm/vllm-openai:nightly-aarch64
docker_image_id: sha256:cd3e070f40f57d23cb1aaad139aad87b52abccd5c5b9f4539186605b94e0fb9d
vllm_version: 0.23.1rc1.dev1042+g8e981630c
common_runtime_args:
max_model_len: 262144
max_num_seqs: 4
dtype: bfloat16
enable_prefix_caching: true
enable_auto_tool_choice: true
tool_call_parser: qwen3_xml
reasoning_parser: qwen3
trust_remote_code: true
runner: generate
language_model_only_equivalent: runner_generate
endpoints:
spark01:
provider: corp-dgx-spark-01-ornith
endpoint: http://192.168.2.2:8003/v1
edge_adapter_endpoint: http://127.0.0.1:8003/v1
health: http://192.168.2.2:8003/health
container_name: iop-vllm-ornith35b-fp8
start_script: /home/digitalcommerce_dgx_spark_01/start_vllm_ornith35b_docker_8003.sh
host_port: 8003
container_port: 8000
gpu_memory_utilization: 0.47
gpu_memory_utilization_note: 2026-07-13 single-node test selected 0.47; 0.50 failed KV init, 0.40 yielded 369057 KV tokens and 1.41x full-context concurrency, and 0.47 yielded 783347 KV tokens and 2.99x full-context concurrency. Under capacity_planning_policy this satisfies the capacity 4 operating target because it is about 74.7% KV per slot, near the 70-75% target, without requiring 4.0x full 262144-token requests.
hf_home: /models/.cache/huggingface
host_model_root: /home/digitalcommerce_dgx_spark_01/Data/models
model_load_memory_gib: 34.9
gpu_kv_cache_tokens: 783347
max_concurrency_for_262144_token_requests: 2.99
verification:
gemma4_8002_status: stopped_2026_07_13_for_spark_ornith_iop_provider_pool
ornith_8003_health: passed
models_endpoint: ornith:35b
direct_chat_completion: passed
spark02:
provider: corp-dgx-spark-02-ornith
endpoint: http://192.168.2.4:8005/v1
edge_adapter_endpoint: http://127.0.0.1:8005/v1
health: http://192.168.2.4:8005/health
container_name: iop-vllm-ornith35b-fp8
start_script: /home/dplab/start_vllm_ornith35b_docker_8005.sh
host_port: 8005
container_port: 8000
gpu_memory_utilization: 0.50
gpu_memory_utilization_note: 0.50 remains above the capacity 4 operating target because startup observed 1074276 KV tokens and 4.10x full-context concurrency.
hf_home: /models/.cache/huggingface
host_model_root: /home/dplab/Data/models
cache_source: copied_from_spark01_ornith_hf_cache_after_slow_hf_download
model_load_memory_gib: 34.9
gpu_kv_cache_tokens: 1074276
max_concurrency_for_262144_token_requests: 4.10
north_runtime_status: stopped_2026_07_12_for_ornith_direct_memory_headroom
verification:
gemma4_8004_status: stopped_2026_07_13_for_spark_ornith_iop_provider_pool
ornith_8005_health: passed
models_endpoint: ornith:35b
direct_chat_completion: passed
iop_connection_status: connected_as_dev_corp_edge_provider_pool
timeout_policy:
openai_timeout_sec: 1800
adapter_queue_timeout_ms: 1800000
provider_catalog_queue_timeout_ms: 0
provider_catalog_queue_timeout_policy: disabled_no_provider_admission_queue_timeout
backend_request_timeout_ms: 1800000
note: provider catalog queue timeout is disabled with queue_timeout_ms=0, while openai_compat adapter instances keep a 30 minute queue/request budget for long reasoning and long-context traffic.
runtime_capacity_targets:
dgx_spark:
capacity: 4
context_window_max: 262144
typical_request_context_ratio_of_max: "0.50-0.70"
typical_request_context_scope: prompt plus generated tokens
kv_lower_bound_ratio_per_capacity_slot: "0.50"
kv_operating_target_ratio_per_capacity_slot: "0.70-0.75"
kv_lower_bound_tokens: 524288
kv_operating_target_tokens_min: 734004
kv_operating_target_tokens_max: 786432
lower_bound_full_context_concurrency: 2.00
operating_target_full_context_concurrency_min: 2.80
operating_target_full_context_concurrency_max: 3.00
kv_size_basis: catalog capacity 4 assumes typical requests use 50-70% of max context including generated tokens; 50% KV per slot is a lower bound, while the operating target is 70-75% of four 262144-token slots. Verify actual vLLM KV cache from startup log because gpu_memory_utilization controls available KV cache.
full_context_concurrency_note: startup full-context concurrency below 4.0x is not under-capacity by itself, but capacity 4 should prefer 2.8x-3.0x; 2.0x-2.8x requires workload-specific smoke evidence before calling it operationally suitable.
legacy_fp4_moe_env_for_spark_gemma4_restore: VLLM_MAX_TOKENS_PER_EXPERT_FP4_MOE=4194304
mac_studio:
capacity: 5
context_window_max: 262144
kv_lower_bound_ratio_per_capacity_slot: "0.50"
kv_operating_target_ratio_per_capacity_slot: "0.70-0.75"
kv_size_tokens_effective: 786432
kv_size_planning_ratio_per_slot: 0.60
kv_size_basis: requested 262144x3 mapped to vLLM-MLX max_kv_size; for catalog capacity 5 this is 60% KV per slot, above the lower bound but below the 70-75% operating target, so capacity 5 should be read as admission/test-line capacity unless workload smoke confirms the expected context mix.
capacity_smoke:
endpoints:
- /v1/chat/completions
provider_dependent_endpoints:
/v1/responses: selected provider가 지원하면 raw passthrough로 검증하고, provider가 미지원하면 provider status/body relay 여부를 증거로 남긴다
aggregate_provider_capacity: 13
model_scenarios:
ornith_9:
model: ornith:35b
concurrent_requests: 9
expected_max_in_flight: 8
queue_observation: overflow request may queue briefly; do not fail the smoke solely because polling misses the queued moment.
ornith_6:
model: ornith:35b
concurrent_requests: 6
expected_max_in_flight: 6
queue_observation: not_expected
gemma4_9:
model: gemma4:26b
concurrent_requests: 9
expected_max_in_flight: 5
queue_observation: overflow requests should queue when polling catches the capacity window.
gemma4_6:
model: gemma4:26b
concurrent_requests: 6
expected_max_in_flight: 5
queue_observation: overflow request may queue briefly; do not fail the smoke solely because polling misses the queued moment.
prompt_policy: long_reasoning_allowed
think_policy: per_alias_provider_defaults
exact_output_match: false
capacity_interpretation: capacity smoke validates provider admission, queue behavior, health, and recovery for typical prompts; it does not require every catalog capacity slot to hold a full 262144-token request at the same time, but capacity 4 runtime suitability should prefer the 70-75% KV operating target.
latest_capacity_verification:
date: "2026-07-13"
route: public_digitalplatform_edge
host: iop.ai.kr
edge_addr: iop.ai.kr:18087
source_ref: live_dev_corp_runtime
thinking_policy:
gemma4:26b:
default_thinking_token_budget: 1024
ornith:35b:
default_thinking_token_budget: null
control_plane_enabled: true
control_plane_status_url_remote: http://127.0.0.1:18002/edges/dev-corp-edge/status
control_plane_public_status_url: http://iop.ai.kr:18002/edges/dev-corp-edge/status
capacity_snapshot_observed: true
capacity_snapshot_note: public 18002 was not reachable from the current runner, but Edge host local Control Plane status over SSH returned provider_snapshots during the 2026-07-13 model-specific concurrency smokes.
node_connection:
node_edge_addr: iop.ai.kr:18087
nodes_registered:
- corp-dgx-spark-01-vllm-node
- corp-dgx-spark-02-vllm-node
- corp-mac-studio-mlx-vllm-node
provider_snapshot_baseline:
corp-dgx-spark-01-ornith:
capacity: 4
health: healthy
served_models:
- ornith:35b
corp-dgx-spark-02-ornith:
capacity: 4
health: healthy
served_models:
- ornith:35b
corp-mac-studio-mlx-vllm:
capacity: 5
health: healthy
served_models:
- mlx-community/gemma-4-26b-a4b-it-nvfp4
chat_completions:
ornith_9:
model: ornith:35b
concurrent_requests: 9
ok: 9
http_200: 9
finish_reason: stop
provider_max:
corp-dgx-spark-01-ornith:
max_in_flight: 4
max_queued: 0
corp-dgx-spark-02-ornith:
max_in_flight: 4
max_queued: 0
final_recovery: in_flight_0_queued_0_healthy
ornith_6:
model: ornith:35b
concurrent_requests: 6
ok: 6
http_200: 6
finish_reason: stop
provider_max:
corp-dgx-spark-01-ornith:
max_in_flight: 3
max_queued: 0
corp-dgx-spark-02-ornith:
max_in_flight: 3
max_queued: 0
final_recovery: in_flight_0_queued_0_healthy
gemma4_9:
model: gemma4:26b
concurrent_requests: 9
ok: 9
http_200: 9
finish_reason: stop
provider_max:
corp-mac-studio-mlx-vllm:
max_in_flight: 5
max_queued: 4
final_recovery: in_flight_0_queued_0_healthy
gemma4_6:
model: gemma4:26b
concurrent_requests: 6
ok: 6
http_200: 6
finish_reason: stop
provider_max:
corp-mac-studio-mlx-vllm:
max_in_flight: 5
max_queued: 1
final_recovery: in_flight_0_queued_0_healthy
responses:
iop_passthrough_contract: true
provider_support: provider_dependent
reason: selected provider가 /v1/responses를 지원하면 raw passthrough로 검증하고, provider가 미지원하면 provider error relay를 확인한다
legacy_capacity_verification_2026_07_08:
date: "2026-07-08"
source_ref: c2437aaedefbac4312d69dfd10aa017c2739e187
route: mac_mini_provider_pool_runtime
edge_runtime_patch: docs_15_concurrency_standard_redeploy
control_plane_status_url: http://127.0.0.1:18002/edges/dev-corp-edge/status
remote_log: build/dev-corp-runtime/logs/capacity_15_20260708_155027.json
provider_snapshot_baseline:
total_capacity: 13
providers:
corp-dgx-spark-01-vllm:
capacity: 4
health: healthy
status: available
corp-dgx-spark-02-vllm:
capacity: 4
health: healthy
status: available
corp-mac-studio-mlx-vllm:
capacity: 5
health: healthy
status: available
chat_completions:
concurrent_requests: 15
ok: 15
timeout_errors: 0
http_errors: 0
max_total_in_flight: 13
max_total_queued: 6
elapsed_sec: 25.117
provider_max:
corp-dgx-spark-01-vllm:
max_in_flight: 4
max_queued: 2
corp-dgx-spark-02-vllm:
max_in_flight: 4
max_queued: 2
corp-mac-studio-mlx-vllm:
max_in_flight: 5
max_queued: 2
final_recovery: in_flight_0_queued_0
responses:
legacy_supported_in_mac_mini_runtime: true
current_public_provider_pool_supported: false
concurrent_requests: 15
ok: 15
max_total_in_flight: 13
max_total_queued: 6
legacy_gemma4_runtime_update_2026_06_25:
date: "2026-06-25"
status: historical_pre_2026_07_13_spark_ornith_swap
reason: user requested dev-corp node capacity/context/KV alignment
requested_targets:
dgx_spark:
capacity: 4
context_window_max: 262144
kv_size: 262144x2
mac_studio:
capacity: 5
context_window_max: 262144
kv_size: 262144x3
applied_mapping:
dgx_spark: vLLM has no vLLM-MLX style max_kv_size flag; requested KV 262144x2 is validated from startup log KV cache size and max_model_len 262144
mac_studio: vLLM-MLX requested KV 262144x3 is represented as max_kv_size 786432 with max_request_tokens 262144
remediation_applied:
- Historical Spark Gemma4 FP4 MoE first profile required VLLM_MAX_TOKENS_PER_EXPERT_FP4_MOE=4194304 for max_num_batched_tokens 524288
- DGX Spark 01/02 Gemma4 agent serving used vLLM 0.24.0 with text-only/eager profile to keep native tool-call streaming stable for agent/tool-call flows; this is legacy evidence after the 2026-07-13 Spark Ornith swap
- Historical Spark Gemma4 DGX Spark 01 start script prefixed /home/digitalcommerce_dgx_spark_01/vllm_env_0_24_0/bin
- Mac Studio vLLM-MLX pyexpat required DYLD_LIBRARY_PATH=/opt/homebrew/opt/expat/lib
verification:
last_rechecked_at: "2026-07-08"
mac_studio_health: passed
dgx01_health: passed
dgx02_health: passed
direct_health_observation:
dgx01: historical pre-2026-07-13 Spark Gemma4 evidence; mac-mini http://192.168.2.2:8002 returned /health 200 and /v1/models exposed gemma-4-26B-A4B-it-NVFP4 after vLLM 0.24.0 text-only/eager restart with default_chat_template_kwargs.enable_thinking=true; direct API forced/auto/streaming tool-call smoke passed; startup log reported GPU KV cache size 1,453,705 tokens and 5.55x concurrency for 262,144-token requests
dgx02: historical pre-2026-07-13 Spark Gemma4 evidence; node-local http://127.0.0.1:8004 and mac-mini http://192.168.2.4:8004 returned /health 200, /v1/models exposed gemma-4-26B-A4B-it-NVFP4 after Docker image vllm/vllm-openai:v0.24.0 text-only/eager restart with default_chat_template_kwargs.enable_thinking=true; direct API auto/streaming tool-call smoke passed; startup log reported GPU KV cache size 1,452,633 tokens and 5.54x concurrency for 262,144-token requests
mac_studio: node-local http://127.0.0.1:8004, mac-mini http://192.168.2.3:8004, and current-host http://172.24.63.178:8004 returned /health 200, /v1/models exposed mlx-community/gemma-4-26b-a4b-it-nvfp4 after vllm-mlx 0.4.0 restart with continuous batching and disable_prefix_cache; direct API non-stream/stream auto tool-call and streaming multi-turn tool-result final-answer passed
edge_openai_smoke: passed_with_control_plane_capacity_snapshot_2026_07_06
nodes:
- id: corp-dgx-spark-01-vllm-node
alias: corp-dgx-spark-01-vllm
role: vllm-provider
ssh: digitalcommerce_dgx_spark_01@192.168.2.2
ssh_origin: mac-mini
workspace: /home/digitalcommerce_dgx_spark_01
provider_pool_candidate: true
provider:
id: corp-dgx-spark-01-ornith
type: vllm
endpoint: http://192.168.2.2:8003/v1
edge_adapter_endpoint: http://127.0.0.1:8003/v1
health: http://192.168.2.2:8003/health
served_model: ornith:35b
capacity: 4
capacity_status: configured_health_passed_direct_and_iop_smoke_2026_07_13
last_runtime_observation: Spark01 Gemma4 8002 stopped on 2026-07-13 and Ornith 8003 Docker runtime is connected to IOP provider pool; startup with gpu_memory_utilization 0.47 reported GPU KV cache size 783,347 tokens and 2.99x concurrency for 262,144-token requests, then /health, /v1/models, direct chat, and IOP ornith:35b smoke passed.
capacity_basis: capacity 4 uses the dev-corp capacity_planning_policy; gpu_memory_utilization 0.47 provides roughly 75% KV per slot and satisfies the 734004-786432 token operating target without requiring 4.0x full-context concurrency.
queue_timeout_ms: 0
adapter_queue_timeout_ms: 1800000
request_timeout_ms: 1800000
edge_connectivity:
mode: direct_public_edge
edge_addr: iop.ai.kr:18087
reason: active dev-corp Edge is the public iop.ai.kr host
tunnel_pid_file: /Users/fe/agent-work/iop-dev-corp/build/dev-corp-runtime/node01-tunnel.pid
tunnel_status: stopped_obsolete_2026_07_09
retired_reverse_tunnels: true
runtime:
host: spark-0b30
os: Ubuntu 24.04.3 LTS / aarch64
hardware: NVIDIA DGX Spark / NVIDIA GB10 / 119GiB RAM
manager: docker
container_name: iop-vllm-ornith35b-fp8
start_script: /home/digitalcommerce_dgx_spark_01/start_vllm_ornith35b_docker_8003.sh
image: vllm/vllm-openai:nightly-aarch64
package_baseline:
vllm: 0.23.1rc1.dev1042+g8e981630c
args:
host: 0.0.0.0
port: 8003
dtype: bfloat16
gpu_memory_utilization: 0.47
max_model_len: 262144
max_num_seqs: 4
enable_prefix_caching: true
enable_auto_tool_choice: true
tool_call_parser: qwen3_xml
reasoning_parser: qwen3
trust_remote_code: true
runner: generate
context_window_max: 262144
kv_size_tokens_effective: 783347
full_context_concurrency_observed: 2.99
- id: corp-dgx-spark-02-vllm-node
alias: corp-dgx-spark-02-vllm
role: vllm-provider
ssh: dplab@192.168.2.4
ssh_origin: mac-mini
workspace: /home/dplab
provider_pool_candidate: true
provider:
id: corp-dgx-spark-02-ornith
type: vllm
endpoint: http://192.168.2.4:8005/v1
edge_adapter_endpoint: http://127.0.0.1:8005/v1
health: http://192.168.2.4:8005/health
served_model: ornith:35b
capacity: 4
capacity_status: configured_health_passed_direct_and_iop_smoke_2026_07_13
last_runtime_observation: Spark02 Gemma4 8004 stopped on 2026-07-13 and Ornith 8005 Docker runtime is connected to IOP provider pool; startup with gpu_memory_utilization 0.50 reported GPU KV cache size 1,074,276 tokens and 4.10x concurrency for 262,144-token requests, then /health, /v1/models, direct chat, and IOP ornith:35b smoke passed.
capacity_basis: capacity 4 uses the dev-corp capacity_planning_policy; gpu_memory_utilization 0.50 is above the 734004-786432 token operating target and provides 4.10x full-context concurrency.
queue_timeout_ms: 0
adapter_queue_timeout_ms: 1800000
request_timeout_ms: 1800000
edge_connectivity:
mode: direct_public_edge
edge_addr: iop.ai.kr:18087
reason: active dev-corp Edge is the public iop.ai.kr host
tunnel_pid_file: /Users/fe/agent-work/iop-dev-corp/build/dev-corp-runtime/node02-tunnel.pid
tunnel_status: stopped_obsolete_2026_07_09
retired_reverse_tunnels: true
runtime:
host: spark-fb94
os: Ubuntu 24.04.3 LTS / aarch64
hardware: NVIDIA DGX Spark / NVIDIA GB10 / 119GiB RAM
manager: docker
container_name: iop-vllm-ornith35b-fp8
start_script: /home/dplab/start_vllm_ornith35b_docker_8005.sh
image: vllm/vllm-openai:nightly-aarch64
container_port: 8000
host_port: 8005
package_baseline:
vllm: 0.23.1rc1.dev1042+g8e981630c
args:
gpu_memory_utilization: 0.50
max_model_len: 262144
max_num_seqs: 4
dtype: bfloat16
enable_prefix_caching: true
enable_auto_tool_choice: true
tool_call_parser: qwen3_xml
reasoning_parser: qwen3
trust_remote_code: true
runner: generate
context_window_max: 262144
kv_size_tokens_effective: 1074276
full_context_concurrency_observed: 4.10
- id: corp-mac-studio-mlx-vllm-node
alias: corp-mac-studio-mlx-vllm
role: vllm-mlx-provider
ssh: dc_dev@192.168.2.3
ssh_origin: mac-mini
workspace: /Users/dc_dev
provider_pool_candidate: true
provider:
id: corp-mac-studio-mlx-vllm
type: openai_compat
runtime_type: vllm-mlx
endpoint: http://192.168.2.3:8004/v1
edge_adapter_endpoint: http://127.0.0.1:8004/v1
edge_connectivity:
mode: direct_public_edge
edge_addr: iop.ai.kr:18087
reason: active dev-corp Edge is the public iop.ai.kr host
health: http://192.168.2.3:8004/health
served_model: mlx-community/gemma-4-26b-a4b-it-nvfp4
capacity: 5
capacity_status: configured_health_passed_direct_smoke_and_edge_capacity_smoke
last_runtime_observation: screen vllm_mlx_8004 running vllm-mlx 0.4.0 with provider catalog capacity 5, continuous batching enabled, runtime headroom max_num_seqs 6, max_request_tokens 262144, max_kv_size 786432, disable_prefix_cache, and default_chat_template_kwargs.enable_thinking=true; node-local, mac-mini, and current-host /health passed, /v1/models exposed mlx-community/gemma-4-26b-a4b-it-nvfp4, direct API non-stream/stream auto tool-call and streaming multi-turn tool-result final-answer smoke passed
capacity_basis: provider catalog capacity 5; vllm-mlx 0.4.0 runtime headroom uses --max-num-seqs 6 with requested KV 262144x3 mapped to --max-kv-size 786432; continuous batching is required for this provider, while prefix cache is disabled for Gemma4 stability
working_checkpoint:
status: final_working_baseline_as_of_2026_07_08
do_not_change_with_spark_vllm: true
micro_tuning_policy: only in a separate experiment with an explicit rollback point; do not change Spark Ornith vLLM settings as part of Mac Studio Gemma4 tuning
required_runtime_flags:
- --continuous-batching
- --max-num-seqs 6
- --prefill-batch-size 6
- --completion-batch-size 6
- --chunked-prefill-tokens 1024
- --disable-prefix-cache
- --max-kv-size 786432
- --max-request-tokens 262144
- --enable-auto-tool-choice
- --tool-call-parser gemma4
- --reasoning-parser gemma4
- --default-chat-template-kwargs '{"enable_thinking":true}'
residual_issue: Gemma4 thought channel delimiters can still leak into final assistant content after successful tool execution.
containment_preference: prefer a vllm-mlx Gemma-specific Edge/provider adapter sanitizer over further runtime option churn if the residual issue appears in assistant content
queue_timeout_ms: 0
adapter_queue_timeout_ms: 1800000
request_timeout_ms: 1800000
runtime:
host: dc-devui-MacStudio.local
os: macOS 26.2
hardware: Apple M3 Ultra / 512GB RAM / 80-core GPU
manager: screen
manager_status: detached screen session vllm_mlx_8004; DYLD_LIBRARY_PATH=/opt/homebrew/opt/expat/lib is required for pyexpat on this host
screen_session: vllm_mlx_8004
start_script: /Users/dc_dev/iop-dev-corp-field/start_vllm_mlx_8004.sh
python: /Users/dc_dev/vllm-env-0.4.0/bin/python
package_baseline:
vllm_mlx: 0.4.0
mlx: 0.32.0
mlx_lm: 0.31.3
mlx_vlm: 0.6.4
torch: 2.12.1
transformers: 5.12.1
huggingface_hub: 1.22.0
args:
host: 0.0.0.0
port: 8004
continuous_batching: true
max_num_seqs: 6
prefill_batch_size: 6
completion_batch_size: 6
chunked_prefill_tokens: 1024
disable_prefix_cache: true
max_kv_size: 786432
max_request_tokens: 262144
context_window_max: 262144
kv_size_tokens_effective: 786432
enable_auto_tool_choice: true
tool_call_parser: gemma4
reasoning_parser: gemma4
default_chat_template_kwargs:
enable_thinking: true
secondary_provider_candidate:
id: corp-mac-studio-mlx-vlm-diffusiongemma
type: mlx-vlm
endpoint: http://192.168.2.3:8005/v1
health: http://192.168.2.3:8005/health
served_model: mlx-community/diffusiongemma-26B-A4B-it-OptiQ-4bit
include_in_default_pool: false
note: exposes multiple models and should be routed only after alias/capacity policy is decided