{
  "scope": "Dated, measured JEPA execution. This is a systems result, not diagnostic validation.",
  "kernel_capture": {
    "captured_at_utc": "2026-09-26T13:10:43.531588Z",
    "checkpoint_update": 30000,
    "tool": "Nsight Compute 2026.3.1",
    "scope": "Eight filtered BF16 GEMM launches in an isolated checkpoint replay",
    "sm_tensor_throughput_percent_range": [78.66, 89.70],
    "memory_throughput_percent_range": [54.83, 61.31],
    "dram_throughput_percent_range": [14.58, 33.30],
    "monitor_verdict": "Selected kernels primarily limited by compute throughput",
    "monitor_criterion": "compute >= 70 percent and compute > 1.2 * memory throughput percent",
    "roofline_measurement": null
  },
  "optimized_dual_gpu_trace": {
    "date": "2026-09-30",
    "update": 47323,
    "gpus": 2,
    "gpu_model": "NVIDIA L40S",
    "aligned_profile_span_seconds": 8.89263,
    "non_nccl_kernel_seconds_per_gpu_approx": [7.914, 7.937],
    "rank0_kernel_union_percent_of_rank_step": 90.32,
    "both_gpus_idle_ms": 146.08,
    "both_gpus_idle_percent_of_aligned_span": 1.64,
    "rank0_all_gather_ms": 35.63,
    "interpretation": "GPU computation dominates the optimized measured step; no other large isolated avoidable delay was identified in this trace."
  },
  "matched_unprofiled_comparison": {
    "date": "2026-09-30",
    "design": "ABBA",
    "observations_per_variant": 10,
    "before_seconds_per_update": 10.24605,
    "after_seconds_per_update": 8.78729,
    "time_reduction_percent_rounded": 14.24,
    "intervention": "Overlap coordinator input preparation with GPU work",
    "boundary": "Input waiting and optimizer update included; periodic diagnostics, checkpoint work, startup and downtime excluded",
    "recipe_changed": false,
    "validation": "Input, gradient, optimizer/state and resume equivalence checks passed",
    "scope": "Matched workload and runtime variants; not a universal hardware speedup"
  }
}
