{
  "device": "cpu",
  "torch": "2.14.0+cu130",
  "threads": 2,
  "scope": "tiny training epoch including loading and transfer; not a hardware ranking",
  "warmup_epochs": 1,
  "repetitions": 5,
  "variants": [
    {
      "name": "physical FP32",
      "microbatch": 48,
      "accumulation": 1,
      "effective_batch": 48,
      "samples": 133,
      "updates": 3,
      "median_seconds": 0.001063500007148832,
      "seconds": [
        0.001063500007148832,
        0.0010576999629847705,
        0.0011971999774686992,
        0.0013412999687716365,
        0.000999599986243993
      ],
      "samples_per_second": 125058.76737750435,
      "peak_cuda_allocated_mib": null,
      "max_parameter_difference_from_physical_fp32": 0.0
    },
    {
      "name": "accumulated FP32",
      "microbatch": 16,
      "accumulation": 3,
      "effective_batch": 48,
      "samples": 133,
      "updates": 3,
      "median_seconds": 0.001901899988297373,
      "seconds": [
        0.0019200000097043812,
        0.0018913999665528536,
        0.001901899988297373,
        0.0017736000008881092,
        0.002365600026678294
      ],
      "samples_per_second": 69930.07036035834,
      "peak_cuda_allocated_mib": null,
      "max_parameter_difference_from_physical_fp32": 1.4901161193847656e-08
    }
  ],
  "cuda_memory": null,
  "optimizer": {
    "name": "SGD",
    "learning_rate": 0.1
  },
  "model": "Sequential(\n  (0): Linear(in_features=6, out_features=12, bias=True)\n  (1): ReLU()\n  (2): Linear(in_features=12, out_features=3, bias=True)\n)",
  "amp": "skipped: CUDA required"
}