{
  "schema_version": 3,
  "recorded_at": "2026-08-03T05:30:52.713622+00:00",
  "mode": "invocation-average",
  "scope": "cuPHY LDPC invocation-average latency under a capped MPS-managed PyTorch matmul client; no MAC, scheduler, fronthaul, cell, or OTA",
  "timing_warning": "Each sample is cuphy_ex_ldpc's average across 100 inner decodes. This is not a per-slot latency distribution and cannot support p99.9.",
  "cap_semantics": "CUDA_MPS_ACTIVE_THREAD_PERCENTAGE PROVISIONS but does not RESERVE: the documentation is explicit that kernels from different clients may still execute on the same SM. cuPHY is 100%; the AI client is capped.",
  "priority_semantics": "CUDA_MPS_CLIENT_PRIORITY is a documented HINT, not a guarantee. A null result for this arm is expected, not anomalous.",
  "sm_partitioning": {
    "requested": null
  },
  "sm_partition_control_replies": [],
  "slot_budget_us": 500,
  "protocol": {
    "ai_caps_percent": [
      100,
      50,
      25
    ],
    "load_matrix_sizes": [
      8192
    ],
    "invocations_per_level": 12,
    "inner_decodes_per_invocation": 100,
    "ai_load_seconds": 240,
    "settle_seconds": 6,
    "ai_client_priority": 0,
    "sm_partition_chunks": null
  },
  "pins": {
    "hostname": "gb10-ref",
    "machine": "aarch64",
    "kernel": "6.17.0-1014-nvidia",
    "gpu_compute_driver": "NVIDIA GB10, 12.1, 590.48.01",
    "aerial_image": "nvcr.io/nvidia/aerial/aerial-cuda-accelerated-ran:26-1-cubb",
    "aerial_image_id": "sha256:a1021357228a8af8081f82f88bff1dbdc0cda9d2fee454b0d0f47eaf810ee3d1",
    "ai_image": "amini/wg3-sr-worker:latest",
    "ai_image_id": "sha256:1cb1336e163c120d816230d00df99c58f57e829d151c1adec0a55cf32f2ef1dc"
  },
  "unmanaged_background": [
    {
      "pid": 4355,
      "process_name": "~/engines/llama.cpp/build/bin/llama-server"
    },
    {
      "pid": 4349,
      "process_name": "~/.venvs/voxcpm-trial/bin/python"
    },
    {
      "pid": 24093,
      "process_name": "/app/.venv/bin/python3"
    },
    {
      "pid": 17650,
      "process_name": "python"
    },
    {
      "pid": 32440,
      "process_name": "~/Ulap/ULAP-ONE/radio-planner/.venv/bin/python"
    },
    {
      "pid": 18501,
      "process_name": "python"
    },
    {
      "pid": 2631448,
      "process_name": "/usr/local/lib/ollama/llama-server"
    }
  ],
  "unmanaged_processes_at_end": [],
  "end_of_run_gpu_check": null,
  "every_cell_mps_client_verified": true,
  "completed_cells": 2,
  "planned_cells": 4,
  "partial_run": true,
  "reportable_clean_mps_run": false,
  "deadline_guarantee_supported": false,
  "largest_observed_passing_ai_cap_percent": null,
  "results": [
    {
      "level": "mps-idle",
      "ai_cap_percent": null,
      "load_matrix_size": null,
      "n_invocation_averages": 12,
      "inner_decodes_per_invocation": 100,
      "failures": {},
      "block_errors": 0,
      "min_invocation_avg_us": 153.5,
      "p50_invocation_avg_us": 153.6,
      "p90_invocation_avg_us": 153.6,
      "max_invocation_avg_us": 154.1,
      "stdev_invocation_avg_us": 0.2,
      "mean_throughput_gbps": 4.4,
      "observed_p90_within_slot_budget": true,
      "mps_client_verified": true
    },
    {
      "level": "ai-8192-cap-100",
      "ai_cap_percent": 100,
      "load_matrix_size": 8192,
      "n_invocation_averages": 12,
      "inner_decodes_per_invocation": 100,
      "failures": {},
      "block_errors": 0,
      "min_invocation_avg_us": 14371.2,
      "p50_invocation_avg_us": 15264.2,
      "p90_invocation_avg_us": 15572.9,
      "max_invocation_avg_us": 15640.4,
      "stdev_invocation_avg_us": 388.3,
      "mean_throughput_gbps": 0.04,
      "observed_p90_within_slot_budget": false,
      "mps_client_verified": true,
      "ai_mps_client_pid": 2940476,
      "ai_client_priority": 0
    }
  ],
  "partial_run_note": "PARTIAL SWEEP: 2 of 4 planned cells completed. The caps that were never measured are absent, not passing, so no envelope may be read from this document and largest_observed_passing_ai_cap_percent is withheld.",
  "run_error": "AI load client exited before MPS join: container=ulap-cuphy-ai-2936710-50-8192; state=exited; exit_code=1; logs_tail='Traceback (most recent call last):\\n  File \"<string>\", line 4, in <module>\\ntorch.AcceleratorError: CUDA error: CUDA-capable device(s) is/are busy or unavailable\\nSearch for `cudaErrorDevicesUnavailable\\' in https://docs.nvidia.com/cuda/cuda-runtime-api/group__CUDART__TYPES.html for more information.\\nFor more detailed error information, run with CUDA_LOG_FILE=stderr'",
  "ai_client_failure": {
    "container": "ulap-cuphy-ai-2936710-50-8192",
    "container_present": true,
    "state": "exited",
    "exit_code": 1,
    "oom_killed": false,
    "docker_state_error": null,
    "logs_tail": "Traceback (most recent call last):\n  File \"<string>\", line 4, in <module>\ntorch.AcceleratorError: CUDA error: CUDA-capable device(s) is/are busy or unavailable\nSearch for `cudaErrorDevicesUnavailable' in https://docs.nvidia.com/cuda/cuda-runtime-api/group__CUDART__TYPES.html for more information.\nFor more detailed error information, run with CUDA_LOG_FILE=stderr",
    "note": null
  }
}
