{
  "schema_version": 1,
  "captured_at": "2026-09-18T13:30:21.434276+00:00",
  "model": {
    "id": "lerobot/diffusion_pusht",
    "revision": "84a7c23178445c6bbf7e1a884ff497017910f653",
    "sha256": "995d14d35db57d95c35ad9704c3d79c8612b7bc45f3877e5c46c2cdc516856a8",
    "converted_sha256": "3367c7236d7a0d6da74e28cde1514ecc7b31470ec028e244cf28c93f93e0b5d6"
  },
  "processor": {
    "id": "lerobot/diffusion_pusht",
    "revision": "84a7c23178445c6bbf7e1a884ff497017910f653",
    "stats": {
      "observation.image": {
        "mean": [
          [
            [
              0.48500001430511475
            ]
          ],
          [
            [
              0.4560000002384186
            ]
          ],
          [
            [
              0.4059999883174896
            ]
          ]
        ],
        "std": [
          [
            [
              0.2290000021457672
            ]
          ],
          [
            [
              0.2240000069141388
            ]
          ],
          [
            [
              0.22499999403953552
            ]
          ]
        ]
      },
      "observation.state": {
        "max": [
          496.14617919921875,
          510.9578857421875
        ],
        "min": [
          13.45642375946045,
          32.93829345703125
        ]
      }
    },
    "config_sha256": "d391a7bf488accd1c26b2043482f0060b0855b1ec236f3d7358486918472c0a5"
  },
  "contract": {
    "observation_schema_hash": "2dde368a238be30a27b8a7c243b724c51d7c744946f28e439c2afc625e2904fe",
    "observation_steps": 2,
    "action_steps": 8,
    "action_dim": 2,
    "batch_size": 1,
    "inference_steps": 10,
    "action_units": "dataset",
    "normalization": "minmax",
    "solver": "ddim"
  },
  "hardware": {
    "host": "Jankowski",
    "os": "Linux-5.15.153.1-microsoft-standard-WSL2-x86_64-with-glibc2.39",
    "cpu": "x86_64",
    "threads": 1,
    "accelerator": "cuda",
    "compiler": "GCC 13.3.0",
    "build_type": "Release",
    "cuda_device": "NVIDIA GeForce GTX 1650"
  },
  "commands": {
    "flowedge": "/home/igor/fe-cuda-venv/bin/python -m flowedge_lerobot.benchmark models/diffusion_pusht.flowedge.safetensors --source models/diffusion_pusht --revision 84a7c23178445c6bbf7e1a884ff497017910f653 --model-id lerobot/diffusion_pusht --steps 10 --iterations 10 --warmup 2 --threads 1 --device cuda --seed 7 --output bench/artifacts/policy/diffusion-pusht-cuda-replay.json --observations bench/artifacts/policy/diffusion-pusht-cpu-replay.observations.npz",
    "lerobot": "/home/igor/fe-cuda-venv/bin/python -m flowedge_lerobot.benchmark models/diffusion_pusht.flowedge.safetensors --source models/diffusion_pusht --revision 84a7c23178445c6bbf7e1a884ff497017910f653 --model-id lerobot/diffusion_pusht --steps 10 --iterations 10 --warmup 2 --threads 1 --device cuda --seed 7 --output bench/artifacts/policy/diffusion-pusht-cuda-replay.json --observations bench/artifacts/policy/diffusion-pusht-cpu-replay.observations.npz"
  },
  "comparison": {
    "candidate": "FlowEdge native runtime",
    "reference": "LeRobot policy executed with PyTorch",
    "reference_framework": "PyTorch",
    "scope": "matched observation encoder, history, noise, and DDIM schedule on CUDA; FlowEdge engine is freed before the PyTorch U-Net is loaded; not TensorRT/ONNX; not Jetson/ARM"
  },
  "measurements": {
    "flowedge": {
      "status": "measured",
      "startup_ms": 31560.926092,
      "encoder_ms": {
        "p50": 5.5849955,
        "p95": 12.07649145,
        "p99": 12.394522290000001
      },
      "policy_ms": {
        "p50": 130.82786099999998,
        "p95": 133.32144155,
        "p99": 133.60093871
      },
      "end_to_end_ms": {
        "p50": 136.3215035,
        "p95": 143.98982475,
        "p99": 144.38599935
      },
      "throughput_hz": 7.241434794717635,
      "rss_mb": 3142.43359375,
      "rss_reason": "",
      "allocations": {
        "setup": null,
        "hot_path": null,
        "reason": "Python/PyTorch/native process allocations are not instrumented by this runner"
      }
    },
    "lerobot": {
      "status": "measured",
      "startup_ms": 31001.944888,
      "encoder_ms": {
        "p50": 5.0590395,
        "p95": 5.3140995,
        "p99": 5.3285679
      },
      "policy_ms": {
        "p50": 345.365693,
        "p95": 348.80471025,
        "p99": 349.32449805000005
      },
      "end_to_end_ms": {
        "p50": 350.364927,
        "p95": 354.0728503,
        "p99": 354.61172206000003
      },
      "throughput_hz": 2.8493342452464225,
      "rss_mb": 3142.43359375,
      "rss_reason": "",
      "allocations": {
        "setup": null,
        "hot_path": null,
        "reason": "Python/PyTorch/native process allocations are not instrumented by this runner"
      }
    }
  },
  "raw_samples": {
    "flowedge": {
      "encoder_ms": [
        5.206602,
        5.871312,
        12.47403,
        7.255074,
        6.874895,
        11.590611,
        5.226484,
        5.298679,
        5.033643,
        4.96888
      ],
      "policy_ms": [
        130.171498,
        129.682737,
        130.910528,
        131.158094,
        133.670813,
        132.894432,
        131.37265,
        130.745194,
        130.347738,
        130.187906
      ],
      "end_to_end_ms": [
        135.3781,
        135.554049,
        143.384558,
        138.413168,
        140.545708,
        144.485043,
        136.599134,
        136.043873,
        135.381381,
        135.156786
      ]
    },
    "lerobot": {
      "encoder_ms": [
        5.211999,
        5.332185,
        5.238984,
        5.291995,
        4.87613,
        5.063489,
        4.830965,
        4.91431,
        5.05459,
        4.784364
      ],
      "policy_ms": [
        344.019586,
        344.847584,
        348.01059,
        349.454445,
        345.124987,
        345.486596,
        345.24479,
        346.686252,
        345.728403,
        344.38952
      ],
      "end_to_end_ms": [
        349.231585,
        350.179769,
        353.249574,
        354.74644,
        350.001117,
        350.550085,
        350.075755,
        351.600562,
        350.782993,
        349.173884
      ]
    }
  },
  "parity": {
    "max_abs_error": 7.62939453125e-05,
    "atol": 0.001,
    "rtol": 0.0001
  },
  "fixture": {
    "file": "diffusion-pusht-cuda-replay.observations.npz",
    "sha256": "0e2616b6312a6f0682a7307682e4701011499a473f80ee734f570d438cccd60f",
    "seed": 7
  },
  "versions": {
    "torch": "2.14.0+cu126",
    "lerobot": "0.4.4",
    "numpy": "2.5.3"
  },
  "iterations": 10,
  "warmup": 2,
  "limitations": [
    "CUDA full-policy replay of one fixed PushT observation history; not closed-loop task success.",
    "End-to-end includes preprocessing, encoder, history and sampling; excludes camera capture, IPC and robot delivery.",
    "RSS is peak process high-water memory; CUDA frees the FlowEdge engine before loading the PyTorch U-Net, so both U-Nets are not device-resident together.",
    "Allocations are unmeasured; no zero-allocation claim applies to this Python visual pipeline.",
    "Startup excludes interpreter/import time and uses the existing filesystem cache.",
    "Small iteration counts do not establish stable p99 estimates.",
    "CUDA device name is the headline; host threads record leftover CPU work, not a GPU fair-compare.",
    "Not Jetson/ARM evidence. Not TensorRT/ONNX.",
    "A 4GB GPU cannot hold the FlowEdge resident DDIM and the PyTorch U-Net at once."
  ]
}
