{
  "scope": "Dated, measured JEPA execution. This is a systems result, not diagnostic validation.",
  "kernel_capture": {
    "captured_at_utc": "2026-09-26T13:10:43.531588Z",
    "checkpoint_update": 30000,
    "tool": "Nsight Compute 2026.3.1",
    "scope": "Eight filtered BF16 GEMM launches in an isolated checkpoint replay",
    "sm_tensor_throughput_percent_range": [
      78.66,
      89.7
    ],
    "memory_throughput_percent_range": [
      54.83,
      61.31
    ],
    "dram_throughput_percent_range": [
      14.58,
      33.3
    ],
    "monitor_verdict": "Selected kernels primarily limited by compute throughput",
    "monitor_criterion": "compute >= 70 percent and compute > 1.2 * memory throughput percent",
    "roofline_measurement": null
  },
  "optimized_dual_gpu_trace": {
    "date": "2026-09-30",
    "update": 47323,
    "gpus": 2,
    "gpu_model": "NVIDIA L40S",
    "aligned_profile_span_seconds": 8.89263,
    "non_nccl_kernel_seconds_per_gpu_approx": [
      7.914,
      7.937
    ],
    "rank0_kernel_union_percent_of_rank_step": 90.32,
    "both_gpus_idle_ms": 146.08,
    "both_gpus_idle_percent_of_aligned_span": 1.64,
    "rank0_all_gather_ms": 35.63,
    "interpretation": "GPU computation dominates the optimized measured step; no other large isolated avoidable delay was identified in this trace.",
    "rank1_kernel_union_percent_of_rank_step": 91.33
  },
  "matched_unprofiled_comparison": {
    "date": "2026-09-30",
    "design": "ABBA",
    "observations_per_variant": 10,
    "before_seconds_per_update": 10.24605,
    "after_seconds_per_update": 8.78729,
    "time_reduction_percent_rounded": 14.24,
    "intervention": "Overlap coordinator input preparation with GPU work",
    "boundary": "Input waiting and optimizer update included; periodic diagnostics, checkpoint work, startup and downtime excluded",
    "recipe_changed": false,
    "validation": "Input, gradient, optimizer/state and resume equivalence checks passed",
    "scope": "Matched workload and runtime variants; not a universal hardware speedup",
    "before_update_seconds_range": [
      10.024178651161492,
      10.511369281448424
    ],
    "after_update_seconds_range": [
      8.618102742359042,
      8.935084652155638
    ],
    "speedup": 1.1660073736670344,
    "warmup_updates_excluded_per_trial": 2,
    "final_updates_excluded_per_trial": 1
  },
  "equivalence_checks": {
    "inputs_features_and_loss": {
      "status": "PASS",
      "loss_scalars_all_exact": true,
      "features_exact": true,
      "cpu_inputs_exact": true
    },
    "gradients": {
      "status": "PASS",
      "tensor_count": 303,
      "bit_identical": true,
      "max_absolute_difference": 0.0
    },
    "model_optimizer_state": {
      "status": "PASS",
      "tensor_count": 1221,
      "bit_identical": true,
      "max_absolute_difference": 0.0
    },
    "rank_agreement": {
      "status": "PASS",
      "tensor_count": 1221,
      "bit_identical": true,
      "max_absolute_difference": 0.0
    },
    "resume": {
      "status": "PASS",
      "tensor_count": 1221,
      "bit_identical": true,
      "max_absolute_difference": 0.0
    }
  },
  "native_after_resume": {
    "first_update": 47332,
    "last_update": 47351,
    "n": 20,
    "mean_seconds": 8.798639131058007,
    "median_seconds": 8.777870916761458,
    "p95_seconds": 8.946879019029438,
    "min_seconds": 8.663957857526839,
    "max_seconds": 8.961056376807392,
    "scope": "Twenty sampled optimizer updates after native resume; same update timing boundary as the matched comparison."
  },
  "initial_dual_gpu_trace": {
    "date": "2026-09-30",
    "rank0_all_gather_ms": 1482.844638671875,
    "interpretation": "Rank zero waited for the other rank to finish input and view preparation; the CPU prefetch intervention addressed that upstream wait.",
    "comparison_scope": "Diagnostic traces use different source batches; matched unprofiled ABBA supplies the speed comparison."
  }
}
