inventory_id: inventory-dev-corp common_inventory: agent-test/inventory.yaml test_env: dev-corp profile: dev-corp-provider-pool last_updated_at: "2026-07-14" source: remote_runner: ssh: fe@172.24.63.178 repo_root: /Users/fe/agent-work/iop-dev-corp setup_required: true setup_status: deployed current_observation: /Users/fe/agent-work/iop-dev-corp checkout is the mac-mini source/build/provider SSH runner only. As of 2026-07-09 the active dev-corp Edge is the public host iop.ai.kr (115.21.224.82), and provider nodes must use iop.ai.kr:18087 directly. clean_sync: - git fetch origin main - git reset --hard origin/main - git clean -fd dirty_policy: discard compose: project_name: iop-dev-corp-agent network: iop-dev-corp-agent-net subnet: 10.89.2.0/24 env_file: .env.dev-corp.example runtime_root: /Users/fe/agent-work/iop-dev-corp setup_status: deployed_control_plane_observability_web_blocked deployed_at: "2026-07-13T15:32:00+09:00" prometheus_config: ./configs/prometheus/prometheus.dev-corp.yml deployment_note: Control Plane, PostgreSQL, Redis, Prometheus, and Grafana are running in the dev-corp compose project on the mac-mini runner. The compose web service is not started because the runner Flutter/Dart SDK is below the app SDK constraint and required sibling path dependencies are absent. services: - postgres - redis - control-plane - web - prometheus - grafana host_ports: web: 13002 control_plane_http: 18002 control_plane_client_ws: 19004 control_plane_edge_wire: 19005 edge_node_tcp_compose: 19006 postgres: 15402 redis: 16302 edge_metrics: 19105 control_plane_metrics: 19104 prometheus: 19112 grafana: 19122 active_services: - postgres - redis - control-plane - prometheus - grafana blocked_services: web: status: not_started reason: runner Flutter uses Dart 3.10.0 while apps/client requires ^3.11.3, and sibling path dependencies agent-shell/nexo are not present under /Users/fe/agent-work. latest_verification: observed_at: "2026-07-13T15:32:00+09:00" control_plane_healthz: passed control_plane_readyz: passed client_ws_upgrade_runner: passed edge_wire_hello: passed edge_status_nodes: 0 prometheus_ready: passed prometheus_targets: iop-control-plane: up iop-edge-native: up prometheus: up grafana_health: passed public_control_plane_ports: 13002: closed 18002: closed 19004: closed public_edge_openai_models: - gemma4:26b - ornith:35b edge: id: dev-corp-edge host: iop.ai.kr public_ip: 115.21.224.82 ssh: toki@iop.ai.kr runtime_root: /Users/toki/agent-work/iop-dev-corp config_path: /Users/toki/agent-work/iop-dev-corp/build/dev-corp-runtime/edge.yaml config_note: public Edge config and bootstrap artifacts were normalized to iop.ai.kr on 2026-07-09. control_plane_enabled_current_runtime: unknown control_plane_http: http://iop.ai.kr:18002 control_plane_status_url: http://iop.ai.kr:18002/edges/dev-corp-edge/status control_plane_client_ws_runner: ws://iop.ai.kr:19004/client control_plane_edge_wire_addr_runner: iop.ai.kr:19005 public_route_policy: active_dev_corp_edge_uses_iop_ai_kr runner_route_policy: mac-mini is source/build/provider management only, not an Edge route. public_route_note: iop.ai.kr is the dev-corp Edge runtime and the only default route for bootstrap, OpenAI-compatible base URL, and Node edge_addr. control_plane_http_public: http://iop.ai.kr:18002 control_plane_client_ws_public: ws://iop.ai.kr:19004/client bootstrap_http_public: http://iop.ai.kr:18085 bootstrap_http_node_internal: http://iop.ai.kr:18085 openai_base_url_public: https://digitalplatform.iop.ai.kr/v1 openai_base_url_public_http: http://digitalplatform.iop.ai.kr/v1 openai_base_url_direct_edge_listener: http://digitalplatform.iop.ai.kr:18086/v1 openai_base_url_node_internal: http://digitalplatform.iop.ai.kr:18086/v1 openai_base_url_runner: https://digitalplatform.iop.ai.kr/v1 openai_api_key_required_current_runtime: true openai_api_key_secret_path_remote: build/dev-corp-runtime/.secrets/openai_api_key openai_api_key_value_tracked: false edge_node_tcp_public: iop.ai.kr:18087 edge_node_tcp_node_internal: iop.ai.kr:18087 admin_addr_runner: 127.0.0.1:19094 latest_public_edge_reconnect: observed_at: "2026-07-09" node_edge_addr: iop.ai.kr:18087 moved_from: - retired reverse tunnel route - retired mac-mini local route nodes_registered: - corp-dgx-spark-01-vllm-node - corp-dgx-spark-02-vllm-node - corp-mac-studio-mlx-vllm-node public_chat_completions_15_historical_routing_evidence: ok: 15 concurrent_requests: 15 elapsed_sec: 5.678 passthrough_reasoning_stream_seen: 15 manual_iop_response_id_count: 0 remote_log: build/dev-corp-runtime/logs/public_latest_edge_node_chat_15_20260709_181222.json latest_node_binary_deploy: observed_at: "2026-07-09T18:12:22+09:00" source_ref: e8c240ea23a29f1568326b6d162913863b3dca33 linux_arm64_sha256: 02441a57f97238bd650a568896c876fb32c94fcace0073f6068509ef90ac4327 darwin_arm64_sha256: 95c59ee0dacb134dc2c6d46fb34f2dac49c33d7da38abe96157583822364f57a nodes: corp-dgx-spark-01-vllm-node: pid_after_restart: 545848 binary_sha256: 02441a57f97238bd650a568896c876fb32c94fcace0073f6068509ef90ac4327 corp-dgx-spark-02-vllm-node: pid_after_restart: 701517 binary_sha256: 02441a57f97238bd650a568896c876fb32c94fcace0073f6068509ef90ac4327 corp-mac-studio-mlx-vllm-node: pid_after_restart: 87325 binary_sha256: 95c59ee0dacb134dc2c6d46fb34f2dac49c33d7da38abe96157583822364f57a note: provider-pool passthrough contract preserves selected provider OpenAI-compatible fields; use /v1/chat/completions for capacity smoke, and treat /v1/responses as provider-dependent passthrough/relay evidence rather than an IOP-level unsupported route. latest_bootstrap_domain_fix: observed_at: "2026-07-11" public_host: iop.ai.kr runtime_pid_after_restart: 19178 config: advertise_host: iop.ai.kr artifact_base_url: http://iop.ai.kr:18085 bootstrap_defaults: artifact_base_url: http://iop.ai.kr:18085 edge_addr: iop.ai.kr:18087 verification: public_bootstrap_defaults: passed public_chat_completions_single: passed public_chat_completions_new_domain_15: passed public_chat_completions_default_passthrough_15: passed build: binaries: control_plane: build/dev-corp-runtime/bin/control-plane edge: build/dev-corp-runtime/bin/iop-edge node_macos: build/dev-corp-runtime/bin/iop-node-darwin-arm64 node_linux_arm64: build/dev-corp-runtime/bin/iop-node-linux-arm64 node_windows_amd64: null node_windows_amd64_note: not part of the default dev-corp provider pool model: alias: gemma4:26b alias_status: active_multi_model_pool_primary_alias primary_alias: gemma4:26b additional_aliases: - ornith:35b alias_policy: expose IOP model aliases per served model and map each alias to its current device/provider pool. provider_capacity_total: 13 provider_capacity_status: verified_with_control_plane_provider_snapshots_2026_07_13_device_sync aliases: gemma4:26b: capacity_total: 5 providers: - corp-mac-studio-mlx-vllm served_model: mlx-community/gemma-4-26b-a4b-it-nvfp4 device_policy: served only from Mac Studio in current dev-corp provider pool. default_thinking_token_budget: 1024 reasoning_policy: bounded_thinking_enabled think_policy: default: enabled strict_output_behavior: provider_pool_catalog_default_overrides_strict_disable adapter_mapping: vllm_mlx_uses_chat_template_kwargs_enable_thinking_true ornith:35b: capacity_total: 8 providers: - corp-dgx-spark-01-ornith - corp-dgx-spark-02-ornith served_model: ornith:35b device_policy: served from Spark01 and Spark02 in current dev-corp provider pool. default_thinking_token_budget: null reasoning_policy: reasoning_parser_enabled_without_catalog_thinking_budget think_policy: default: provider_runtime_default adapter_mapping: qwen3_reasoning_parser_without_model_catalog_thinking_budget context_window_max: 262144 thinking_policy_scope: per_alias capacity_planning_policy: observed_at: "2026-07-13" applies_to: - gemma4:26b - ornith:35b max_context_tokens: 262144 catalog_capacity_meaning: provider admission/concurrency target for typical requests, not a guarantee that every admitted request consumes 100% of max context. typical_request_context_ratio_of_max: "0.50-0.70" typical_request_context_scope: prompt tokens plus generated tokens that occupy KV cache during the request. kv_lower_bound_ratio_per_capacity_slot: "0.50" kv_operating_target_ratio_per_capacity_slot: "0.70-0.75" full_context_concurrency_metric: "vLLM startup Maximum concurrency for 262,144 tokens per request is a diagnostic upper-bound metric, not the catalog capacity requirement." capacity_4: lower_bound_kv_tokens: 524288 lower_bound_full_context_concurrency: 2.00 operating_target_kv_tokens_min: 734004 operating_target_kv_tokens_max: 786432 operating_target_full_context_concurrency_min: 2.80 operating_target_full_context_concurrency_max: 3.00 interpretation: capacity 4 is operationally suitable when runtime KV covers roughly 2.8x-3.0x full 262144-token requests; 2.0x-2.8x is only a lower-bound/test-line range and should not be described as covering four 70% context requests. operating_note: Do not mark a provider under-capacity solely because full-context concurrency is below catalog capacity; for capacity 4, prefer 2.8x-3.0x full-context concurrency and treat 2.0x as the minimum lower bound requiring workload-specific smoke evidence. spark_ornith_profile: observed_at: "2026-07-13" status: active_iop_provider_pool_and_direct_test_lines family: spark_ornith model_id: ornith:35b display_name: Ornith 1.0 35B FP8 vLLM api: openai-completions route_policy: Spark Ornith endpoints are connected to the dev-corp Edge/provider pool as IOP providers; device inventory uses provider-internal endpoints and node-local Edge adapter endpoints only. upstream_model: deepreinforce-ai/Ornith-1.0-35B-FP8 upstream_url: https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B-FP8 upstream_revision: 1ab57ce0b44950e498a88756f40ad1ed4d0f30ca docker_image: vllm/vllm-openai:nightly-aarch64 docker_image_id: sha256:cd3e070f40f57d23cb1aaad139aad87b52abccd5c5b9f4539186605b94e0fb9d vllm_version: 0.23.1rc1.dev1042+g8e981630c common_runtime_args: max_model_len: 262144 max_num_seqs: 4 dtype: bfloat16 enable_prefix_caching: true enable_auto_tool_choice: true tool_call_parser: qwen3_xml reasoning_parser: qwen3 trust_remote_code: true runner: generate language_model_only_equivalent: runner_generate ornith_official_sampling: source: https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B source_section: Quickstart and Chat Completions API examples recorded_at: "2026-07-23" status: documented_pending_dev_corp_rollout apply_mechanism: vllm --override-generation-config override_generation_config: temperature: 0.6 top_p: 0.95 top_k: 20 repeat_penalty: not_set_by_official_example precedence: caller-explicit sampling parameters override provider defaults rollout_gate: do not mark applied until both Spark runtimes are restarted with the override and direct plus Edge smoke passes language_control: prompt: "Think in English. Final in Korean." scope: short Pi/system instruction; sampling does not force reasoning language avoid: literal think tags, reasoning prefill, and longer no-quote prohibitions strict_isolation: not guaranteed by sampling, prompt, or chat-template adjustment; validate reasoning and final separately smoke_max_tokens_min: 1024 dev_validation_basis: source: agent-test/inventory-dev.yaml result: GX10 vLLM, OneXPlayer Lemonade, and RTX5090 Lemonade returned stop with Korean final content using the short prompt model_artifact_note: Spark uses the official FP8 model; the byte-identical GGUF comparison does not apply to this runtime endpoints: spark01: provider: corp-dgx-spark-01-ornith endpoint: http://192.168.2.2:8003/v1 edge_adapter_endpoint: http://127.0.0.1:8003/v1 health: http://192.168.2.2:8003/health container_name: iop-vllm-ornith35b-fp8 start_script: /home/digitalcommerce_dgx_spark_01/start_vllm_ornith35b_docker_8003.sh host_port: 8003 container_port: 8000 gpu_memory_utilization: 0.47 gpu_memory_utilization_note: 2026-07-13 single-node test selected 0.47; 0.50 failed KV init, 0.40 yielded 369057 KV tokens and 1.41x full-context concurrency, and 0.47 yielded 783347 KV tokens and 2.99x full-context concurrency. Under capacity_planning_policy this satisfies the capacity 4 operating target because it is about 74.7% KV per slot, near the 70-75% target, without requiring 4.0x full 262144-token requests. hf_home: /models/.cache/huggingface host_model_root: /home/digitalcommerce_dgx_spark_01/Data/models model_load_memory_gib: 34.9 gpu_kv_cache_tokens: 783347 max_concurrency_for_262144_token_requests: 2.99 verification: gemma4_8002_status: stopped_2026_07_13_for_spark_ornith_iop_provider_pool ornith_8003_health: passed models_endpoint: ornith:35b direct_chat_completion: passed spark02: provider: corp-dgx-spark-02-ornith endpoint: http://192.168.2.4:8005/v1 edge_adapter_endpoint: http://127.0.0.1:8005/v1 health: http://192.168.2.4:8005/health container_name: iop-vllm-ornith35b-fp8 start_script: /home/dplab/start_vllm_ornith35b_docker_8005.sh host_port: 8005 container_port: 8000 gpu_memory_utilization: 0.50 gpu_memory_utilization_note: 0.50 remains above the capacity 4 operating target because startup observed 1074276 KV tokens and 4.10x full-context concurrency. hf_home: /models/.cache/huggingface host_model_root: /home/dplab/Data/models cache_source: copied_from_spark01_ornith_hf_cache_after_slow_hf_download model_load_memory_gib: 34.9 gpu_kv_cache_tokens: 1074276 max_concurrency_for_262144_token_requests: 4.10 north_runtime_status: stopped_2026_07_12_for_ornith_direct_memory_headroom verification: gemma4_8004_status: stopped_2026_07_13_for_spark_ornith_iop_provider_pool ornith_8005_health: passed models_endpoint: ornith:35b direct_chat_completion: passed iop_connection_status: connected_as_dev_corp_edge_provider_pool timeout_policy: openai_timeout_sec: 1800 adapter_queue_timeout_ms: 1800000 provider_catalog_queue_timeout_ms: 0 provider_catalog_queue_timeout_policy: disabled_no_provider_admission_queue_timeout backend_request_timeout_ms: 1800000 note: provider catalog queue timeout is disabled with queue_timeout_ms=0, while openai_compat adapter instances keep a 30 minute queue/request budget for long reasoning and long-context traffic. runtime_capacity_targets: dgx_spark: capacity: 4 context_window_max: 262144 typical_request_context_ratio_of_max: "0.50-0.70" typical_request_context_scope: prompt plus generated tokens kv_lower_bound_ratio_per_capacity_slot: "0.50" kv_operating_target_ratio_per_capacity_slot: "0.70-0.75" kv_lower_bound_tokens: 524288 kv_operating_target_tokens_min: 734004 kv_operating_target_tokens_max: 786432 lower_bound_full_context_concurrency: 2.00 operating_target_full_context_concurrency_min: 2.80 operating_target_full_context_concurrency_max: 3.00 kv_size_basis: catalog capacity 4 assumes typical requests use 50-70% of max context including generated tokens; 50% KV per slot is a lower bound, while the operating target is 70-75% of four 262144-token slots. Verify actual vLLM KV cache from startup log because gpu_memory_utilization controls available KV cache. full_context_concurrency_note: startup full-context concurrency below 4.0x is not under-capacity by itself, but capacity 4 should prefer 2.8x-3.0x; 2.0x-2.8x requires workload-specific smoke evidence before calling it operationally suitable. legacy_fp4_moe_env_for_spark_gemma4_restore: VLLM_MAX_TOKENS_PER_EXPERT_FP4_MOE=4194304 mac_studio: capacity: 5 context_window_max: 262144 kv_lower_bound_ratio_per_capacity_slot: "0.50" kv_operating_target_ratio_per_capacity_slot: "0.70-0.75" kv_size_tokens_effective: 786432 kv_size_planning_ratio_per_slot: 0.60 kv_size_basis: requested 262144x3 mapped to vLLM-MLX max_kv_size; for catalog capacity 5 this is 60% KV per slot, above the lower bound but below the 70-75% operating target, so capacity 5 should be read as admission/test-line capacity unless workload smoke confirms the expected context mix. capacity_smoke: endpoints: - /v1/chat/completions provider_dependent_endpoints: /v1/responses: selected provider가 지원하면 raw passthrough로 검증하고, provider가 미지원하면 provider status/body relay 여부를 증거로 남긴다 aggregate_provider_capacity: 13 model_scenarios: ornith_9: model: ornith:35b concurrent_requests: 9 expected_max_in_flight: 8 queue_observation: overflow request may queue briefly; do not fail the smoke solely because polling misses the queued moment. ornith_6: model: ornith:35b concurrent_requests: 6 expected_max_in_flight: 6 queue_observation: not_expected gemma4_9: model: gemma4:26b concurrent_requests: 9 expected_max_in_flight: 5 queue_observation: overflow requests should queue when polling catches the capacity window. gemma4_6: model: gemma4:26b concurrent_requests: 6 expected_max_in_flight: 5 queue_observation: overflow request may queue briefly; do not fail the smoke solely because polling misses the queued moment. prompt_policy: long_reasoning_allowed think_policy: per_alias_provider_defaults exact_output_match: false capacity_interpretation: capacity smoke validates provider admission, queue behavior, health, and recovery for typical prompts; it does not require every catalog capacity slot to hold a full 262144-token request at the same time, but capacity 4 runtime suitability should prefer the 70-75% KV operating target. latest_capacity_verification: date: "2026-07-13" route: public_digitalplatform_edge host: iop.ai.kr edge_addr: iop.ai.kr:18087 source_ref: live_dev_corp_runtime thinking_policy: gemma4:26b: default_thinking_token_budget: 1024 ornith:35b: default_thinking_token_budget: null control_plane_enabled: true control_plane_status_url_remote: http://127.0.0.1:18002/edges/dev-corp-edge/status control_plane_public_status_url: http://iop.ai.kr:18002/edges/dev-corp-edge/status capacity_snapshot_observed: true capacity_snapshot_note: public 18002 was not reachable from the current runner, but Edge host local Control Plane status over SSH returned provider_snapshots during the 2026-07-13 model-specific concurrency smokes. node_connection: node_edge_addr: iop.ai.kr:18087 nodes_registered: - corp-dgx-spark-01-vllm-node - corp-dgx-spark-02-vllm-node - corp-mac-studio-mlx-vllm-node provider_snapshot_baseline: corp-dgx-spark-01-ornith: capacity: 4 health: healthy served_models: - ornith:35b corp-dgx-spark-02-ornith: capacity: 4 health: healthy served_models: - ornith:35b corp-mac-studio-mlx-vllm: capacity: 5 health: healthy served_models: - mlx-community/gemma-4-26b-a4b-it-nvfp4 chat_completions: ornith_9: model: ornith:35b concurrent_requests: 9 ok: 9 http_200: 9 finish_reason: stop provider_max: corp-dgx-spark-01-ornith: max_in_flight: 4 max_queued: 0 corp-dgx-spark-02-ornith: max_in_flight: 4 max_queued: 0 final_recovery: in_flight_0_queued_0_healthy ornith_6: model: ornith:35b concurrent_requests: 6 ok: 6 http_200: 6 finish_reason: stop provider_max: corp-dgx-spark-01-ornith: max_in_flight: 3 max_queued: 0 corp-dgx-spark-02-ornith: max_in_flight: 3 max_queued: 0 final_recovery: in_flight_0_queued_0_healthy gemma4_9: model: gemma4:26b concurrent_requests: 9 ok: 9 http_200: 9 finish_reason: stop provider_max: corp-mac-studio-mlx-vllm: max_in_flight: 5 max_queued: 4 final_recovery: in_flight_0_queued_0_healthy gemma4_6: model: gemma4:26b concurrent_requests: 6 ok: 6 http_200: 6 finish_reason: stop provider_max: corp-mac-studio-mlx-vllm: max_in_flight: 5 max_queued: 1 final_recovery: in_flight_0_queued_0_healthy responses: iop_passthrough_contract: true provider_support: provider_dependent reason: selected provider가 /v1/responses를 지원하면 raw passthrough로 검증하고, provider가 미지원하면 provider error relay를 확인한다 legacy_capacity_verification_2026_07_08: date: "2026-07-08" source_ref: c2437aaedefbac4312d69dfd10aa017c2739e187 route: mac_mini_provider_pool_runtime edge_runtime_patch: docs_15_concurrency_standard_redeploy control_plane_status_url: http://127.0.0.1:18002/edges/dev-corp-edge/status remote_log: build/dev-corp-runtime/logs/capacity_15_20260708_155027.json provider_snapshot_baseline: total_capacity: 13 providers: corp-dgx-spark-01-vllm: capacity: 4 health: healthy status: available corp-dgx-spark-02-vllm: capacity: 4 health: healthy status: available corp-mac-studio-mlx-vllm: capacity: 5 health: healthy status: available chat_completions: concurrent_requests: 15 ok: 15 timeout_errors: 0 http_errors: 0 max_total_in_flight: 13 max_total_queued: 6 elapsed_sec: 25.117 provider_max: corp-dgx-spark-01-vllm: max_in_flight: 4 max_queued: 2 corp-dgx-spark-02-vllm: max_in_flight: 4 max_queued: 2 corp-mac-studio-mlx-vllm: max_in_flight: 5 max_queued: 2 final_recovery: in_flight_0_queued_0 responses: legacy_supported_in_mac_mini_runtime: true current_public_provider_pool_supported: false concurrent_requests: 15 ok: 15 max_total_in_flight: 13 max_total_queued: 6 legacy_gemma4_runtime_update_2026_06_25: date: "2026-06-25" status: historical_pre_2026_07_13_spark_ornith_swap reason: user requested dev-corp node capacity/context/KV alignment requested_targets: dgx_spark: capacity: 4 context_window_max: 262144 kv_size: 262144x2 mac_studio: capacity: 5 context_window_max: 262144 kv_size: 262144x3 applied_mapping: dgx_spark: vLLM has no vLLM-MLX style max_kv_size flag; requested KV 262144x2 is validated from startup log KV cache size and max_model_len 262144 mac_studio: vLLM-MLX requested KV 262144x3 is represented as max_kv_size 786432 with max_request_tokens 262144 remediation_applied: - Historical Spark Gemma4 FP4 MoE first profile required VLLM_MAX_TOKENS_PER_EXPERT_FP4_MOE=4194304 for max_num_batched_tokens 524288 - DGX Spark 01/02 Gemma4 agent serving used vLLM 0.24.0 with text-only/eager profile to keep native tool-call streaming stable for agent/tool-call flows; this is legacy evidence after the 2026-07-13 Spark Ornith swap - Historical Spark Gemma4 DGX Spark 01 start script prefixed /home/digitalcommerce_dgx_spark_01/vllm_env_0_24_0/bin - Mac Studio vLLM-MLX pyexpat required DYLD_LIBRARY_PATH=/opt/homebrew/opt/expat/lib verification: last_rechecked_at: "2026-07-08" mac_studio_health: passed dgx01_health: passed dgx02_health: passed direct_health_observation: dgx01: historical pre-2026-07-13 Spark Gemma4 evidence; mac-mini http://192.168.2.2:8002 returned /health 200 and /v1/models exposed gemma-4-26B-A4B-it-NVFP4 after vLLM 0.24.0 text-only/eager restart with default_chat_template_kwargs.enable_thinking=true; direct API forced/auto/streaming tool-call smoke passed; startup log reported GPU KV cache size 1,453,705 tokens and 5.55x concurrency for 262,144-token requests dgx02: historical pre-2026-07-13 Spark Gemma4 evidence; node-local http://127.0.0.1:8004 and mac-mini http://192.168.2.4:8004 returned /health 200, /v1/models exposed gemma-4-26B-A4B-it-NVFP4 after Docker image vllm/vllm-openai:v0.24.0 text-only/eager restart with default_chat_template_kwargs.enable_thinking=true; direct API auto/streaming tool-call smoke passed; startup log reported GPU KV cache size 1,452,633 tokens and 5.54x concurrency for 262,144-token requests mac_studio: node-local http://127.0.0.1:8004, mac-mini http://192.168.2.3:8004, and current-host http://172.24.63.178:8004 returned /health 200, /v1/models exposed mlx-community/gemma-4-26b-a4b-it-nvfp4 after vllm-mlx 0.4.0 restart with continuous batching and disable_prefix_cache; direct API non-stream/stream auto tool-call and streaming multi-turn tool-result final-answer passed edge_openai_smoke: passed_with_control_plane_capacity_snapshot_2026_07_06 nodes: - id: corp-dgx-spark-01-vllm-node alias: corp-dgx-spark-01-vllm role: vllm-provider ssh: digitalcommerce_dgx_spark_01@192.168.2.2 ssh_origin: mac-mini workspace: /home/digitalcommerce_dgx_spark_01 provider_pool_candidate: true provider: id: corp-dgx-spark-01-ornith type: vllm endpoint: http://192.168.2.2:8003/v1 edge_adapter_endpoint: http://127.0.0.1:8003/v1 health: http://192.168.2.2:8003/health served_model: ornith:35b capacity: 4 capacity_status: configured_health_passed_direct_and_iop_smoke_2026_07_13 last_runtime_observation: Spark01 Gemma4 8002 stopped on 2026-07-13 and Ornith 8003 Docker runtime is connected to IOP provider pool; startup with gpu_memory_utilization 0.47 reported GPU KV cache size 783,347 tokens and 2.99x concurrency for 262,144-token requests, then /health, /v1/models, direct chat, and IOP ornith:35b smoke passed. capacity_basis: capacity 4 uses the dev-corp capacity_planning_policy; gpu_memory_utilization 0.47 provides roughly 75% KV per slot and satisfies the 734004-786432 token operating target without requiring 4.0x full-context concurrency. queue_timeout_ms: 0 adapter_queue_timeout_ms: 1800000 request_timeout_ms: 1800000 edge_connectivity: mode: direct_public_edge edge_addr: iop.ai.kr:18087 reason: active dev-corp Edge is the public iop.ai.kr host tunnel_pid_file: /Users/fe/agent-work/iop-dev-corp/build/dev-corp-runtime/node01-tunnel.pid tunnel_status: stopped_obsolete_2026_07_09 retired_reverse_tunnels: true runtime: host: spark-0b30 os: Ubuntu 24.04.3 LTS / aarch64 hardware: NVIDIA DGX Spark / NVIDIA GB10 / 119GiB RAM manager: docker container_name: iop-vllm-ornith35b-fp8 start_script: /home/digitalcommerce_dgx_spark_01/start_vllm_ornith35b_docker_8003.sh image: vllm/vllm-openai:nightly-aarch64 package_baseline: vllm: 0.23.1rc1.dev1042+g8e981630c args: host: 0.0.0.0 port: 8003 dtype: bfloat16 gpu_memory_utilization: 0.47 max_model_len: 262144 max_num_seqs: 4 enable_prefix_caching: true enable_auto_tool_choice: true tool_call_parser: qwen3_xml reasoning_parser: qwen3 trust_remote_code: true runner: generate context_window_max: 262144 kv_size_tokens_effective: 783347 full_context_concurrency_observed: 2.99 - id: corp-dgx-spark-02-vllm-node alias: corp-dgx-spark-02-vllm role: vllm-provider ssh: dplab@192.168.2.4 ssh_origin: mac-mini workspace: /home/dplab provider_pool_candidate: true provider: id: corp-dgx-spark-02-ornith type: vllm endpoint: http://192.168.2.4:8005/v1 edge_adapter_endpoint: http://127.0.0.1:8005/v1 health: http://192.168.2.4:8005/health served_model: ornith:35b capacity: 4 capacity_status: configured_health_passed_direct_and_iop_smoke_2026_07_13 last_runtime_observation: Spark02 Gemma4 8004 stopped on 2026-07-13 and Ornith 8005 Docker runtime is connected to IOP provider pool; startup with gpu_memory_utilization 0.50 reported GPU KV cache size 1,074,276 tokens and 4.10x concurrency for 262,144-token requests, then /health, /v1/models, direct chat, and IOP ornith:35b smoke passed. capacity_basis: capacity 4 uses the dev-corp capacity_planning_policy; gpu_memory_utilization 0.50 is above the 734004-786432 token operating target and provides 4.10x full-context concurrency. queue_timeout_ms: 0 adapter_queue_timeout_ms: 1800000 request_timeout_ms: 1800000 edge_connectivity: mode: direct_public_edge edge_addr: iop.ai.kr:18087 reason: active dev-corp Edge is the public iop.ai.kr host tunnel_pid_file: /Users/fe/agent-work/iop-dev-corp/build/dev-corp-runtime/node02-tunnel.pid tunnel_status: stopped_obsolete_2026_07_09 retired_reverse_tunnels: true runtime: host: spark-fb94 os: Ubuntu 24.04.3 LTS / aarch64 hardware: NVIDIA DGX Spark / NVIDIA GB10 / 119GiB RAM manager: docker container_name: iop-vllm-ornith35b-fp8 start_script: /home/dplab/start_vllm_ornith35b_docker_8005.sh image: vllm/vllm-openai:nightly-aarch64 container_port: 8000 host_port: 8005 package_baseline: vllm: 0.23.1rc1.dev1042+g8e981630c args: gpu_memory_utilization: 0.50 max_model_len: 262144 max_num_seqs: 4 dtype: bfloat16 enable_prefix_caching: true enable_auto_tool_choice: true tool_call_parser: qwen3_xml reasoning_parser: qwen3 trust_remote_code: true runner: generate context_window_max: 262144 kv_size_tokens_effective: 1074276 full_context_concurrency_observed: 4.10 - id: corp-mac-studio-mlx-vllm-node alias: corp-mac-studio-mlx-vllm role: vllm-mlx-provider ssh: dc_dev@192.168.2.3 ssh_origin: mac-mini workspace: /Users/dc_dev provider_pool_candidate: true provider: id: corp-mac-studio-mlx-vllm type: openai_compat runtime_type: vllm-mlx endpoint: http://192.168.2.3:8004/v1 edge_adapter_endpoint: http://127.0.0.1:8004/v1 edge_connectivity: mode: direct_public_edge edge_addr: iop.ai.kr:18087 reason: active dev-corp Edge is the public iop.ai.kr host health: http://192.168.2.3:8004/health served_model: mlx-community/gemma-4-26b-a4b-it-nvfp4 capacity: 5 capacity_status: configured_health_passed_direct_smoke_and_edge_capacity_smoke last_runtime_observation: screen vllm_mlx_8004 running vllm-mlx 0.4.0 with provider catalog capacity 5, continuous batching enabled, runtime headroom max_num_seqs 6, max_request_tokens 262144, max_kv_size 786432, disable_prefix_cache, and default_chat_template_kwargs.enable_thinking=true; node-local, mac-mini, and current-host /health passed, /v1/models exposed mlx-community/gemma-4-26b-a4b-it-nvfp4, direct API non-stream/stream auto tool-call and streaming multi-turn tool-result final-answer smoke passed capacity_basis: provider catalog capacity 5; vllm-mlx 0.4.0 runtime headroom uses --max-num-seqs 6 with requested KV 262144x3 mapped to --max-kv-size 786432; continuous batching is required for this provider, while prefix cache is disabled for Gemma4 stability working_checkpoint: status: final_working_baseline_as_of_2026_07_08 do_not_change_with_spark_vllm: true micro_tuning_policy: only in a separate experiment with an explicit rollback point; do not change Spark Ornith vLLM settings as part of Mac Studio Gemma4 tuning required_runtime_flags: - --continuous-batching - --max-num-seqs 6 - --prefill-batch-size 6 - --completion-batch-size 6 - --chunked-prefill-tokens 1024 - --disable-prefix-cache - --max-kv-size 786432 - --max-request-tokens 262144 - --enable-auto-tool-choice - --tool-call-parser gemma4 - --reasoning-parser gemma4 - --default-chat-template-kwargs '{"enable_thinking":true}' residual_issue: Gemma4 thought channel delimiters can still leak into final assistant content after successful tool execution. containment_preference: prefer a vllm-mlx Gemma-specific Edge/provider adapter sanitizer over further runtime option churn if the residual issue appears in assistant content queue_timeout_ms: 0 adapter_queue_timeout_ms: 1800000 request_timeout_ms: 1800000 runtime: host: dc-devui-MacStudio.local os: macOS 26.2 hardware: Apple M3 Ultra / 512GB RAM / 80-core GPU manager: screen manager_status: detached screen session vllm_mlx_8004; DYLD_LIBRARY_PATH=/opt/homebrew/opt/expat/lib is required for pyexpat on this host screen_session: vllm_mlx_8004 start_script: /Users/dc_dev/iop-dev-corp-field/start_vllm_mlx_8004.sh python: /Users/dc_dev/vllm-env-0.4.0/bin/python package_baseline: vllm_mlx: 0.4.0 mlx: 0.32.0 mlx_lm: 0.31.3 mlx_vlm: 0.6.4 torch: 2.12.1 transformers: 5.12.1 huggingface_hub: 1.22.0 args: host: 0.0.0.0 port: 8004 continuous_batching: true max_num_seqs: 6 prefill_batch_size: 6 completion_batch_size: 6 chunked_prefill_tokens: 1024 disable_prefix_cache: true max_kv_size: 786432 max_request_tokens: 262144 context_window_max: 262144 kv_size_tokens_effective: 786432 enable_auto_tool_choice: true tool_call_parser: gemma4 reasoning_parser: gemma4 default_chat_template_kwargs: enable_thinking: true secondary_provider_candidate: id: corp-mac-studio-mlx-vlm-diffusiongemma type: mlx-vlm endpoint: http://192.168.2.3:8005/v1 health: http://192.168.2.3:8005/health served_model: mlx-community/diffusiongemma-26B-A4B-it-OptiQ-4bit include_in_default_pool: false note: exposes multiple models and should be routed only after alias/capacity policy is decided