{
  "artifacts": [
    {
      "file": "ram_stage3_tail_fp16.tflite",
      "sha256": "c46e40cab69f070a4b7ba86b9de29f15148f9c67804c5e0888290d243f24ed4e",
      "size_mb": 117.326
    }
  ],
  "benchmarks": [],
  "conversion": {
    "command": "python build_hybrid.py",
    "quantization": "fp16",
    "tool": "litert-torch (ramplus-work build scripts, litert_torch.convert; fp16 via ai_edge_quantizer float_casting)",
    "tool_version": "0.10.0 (editable dev checkout 115a136 + local patches)"
  },
  "cross_runtime": [],
  "delegation": null,
  "device": {
    "records": [
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "gpu_mldrift",
          "context_length": null,
          "date": "2026-08-26",
          "decode_tokens_per_s": null,
          "delegated_ops": null,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-npubench",
            "os_build": "Android 16",
            "runtime": "litert",
            "runtime_version": "2.2.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "npubench sweep row: N=50 median, 1 backend = 1 process, accepted only with thermal NONE->NONE; NPU rows additionally required qnn_partition delegate evidence in logcat (a stock model asked for on the NPU can silently land on XNNPACK and report a CPU number)",
            "mode=jit thermal=NONE->NONE headroom=0.77666664->0.7733334 attempt=0"
          ],
          "failure_class": null,
          "full_delegation": null,
          "latency_p50_ms": 8.005,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "iterations": 50,
            "latency_max_ms": 9.075,
            "latency_min_ms": 7.592,
            "load_ms": 3334.8
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": null,
          "provenance": "measured",
          "runs": true,
          "total_ops": null,
          "ttft_ms": null
        },
        "source": "data/device_runs/2.2.0/2026-08-26/ram-plus__ram_stage3_tail_fp16__galaxy-s26.json"
      },
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "npu_qnn",
          "context_length": null,
          "date": "2026-08-26",
          "decode_tokens_per_s": null,
          "delegated_ops": null,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-npubench",
            "os_build": "Android 16",
            "runtime": "litert",
            "runtime_version": "2.2.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": "QAIRT (Hexagon, JIT on-device)"
          },
          "error": null,
          "evidence": [
            "npubench sweep row: N=50 median, 1 backend = 1 process, accepted only with thermal NONE->NONE; NPU rows additionally required qnn_partition delegate evidence in logcat (a stock model asked for on the NPU can silently land on XNNPACK and report a CPU number)",
            "mode=jit thermal=NONE->NONE headroom=0.78000003->0.77666664 attempt=0"
          ],
          "failure_class": null,
          "full_delegation": null,
          "latency_p50_ms": 14.568,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "iterations": 50,
            "jit_first_load_ms": 7336.6,
            "latency_max_ms": 14.683,
            "latency_min_ms": 14.403,
            "load_ms": 147.9,
            "qnn_partition_lines": 57
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": null,
          "provenance": "measured",
          "runs": true,
          "total_ops": null,
          "ttft_ms": null
        },
        "source": "data/device_runs/2.2.0/2026-08-26/ram-plus__ram_stage3_tail_fp16__galaxy-s26.json"
      },
      {
        "device": "raspberry-pi-5",
        "run": {
          "accelerator": "cpu_xnnpack",
          "context_length": null,
          "date": "2026-08-31",
          "decode_tokens_per_s": null,
          "delegated_ops": null,
          "env": {
            "device": "Raspberry Pi 5 Model B Rev 1.1",
            "machine_label": "raspberry-pi-5",
            "os_build": "Linux-6.18.34+rpt-rpi-2712-aarch64-with-glibc2.41",
            "runtime": "litert",
            "runtime_version": "2.2.0.dev20260804",
            "soc": null,
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "pi5 sweep row: LiteRT benchmark_model, CPU/XNNPACK at --num_threads=4, 3 invocations per file of 10 warm-up + 50 timed runs (the tool caps a phase at 150 s, so very slow graphs run fewer); latency = median of the three per-invocation medians over the timed phase; a row counts as measured only when every invocation exited 0 with XNNPACK engaged and vcgencmd get_throttled 0x0 before and after",
            "versions: ai-edge-litert-nightly=2.2.0.dev20260804, benchmark_model_sha256=babc9275addd8612caf0842294cb5b3f301c3cce46e6c5979cd2803f215f6bee, cpu=Raspberry Pi 5 Model B Rev 1.1, litert-cli-nightly=0.2.0.dev20260805, platform=Linux-6.18.34+rpt-rpi-2712-aarch64-with-glibc2.41, python=3.13.5",
            "invocation 0: exit=0 wall_s=14.3 temp 49.4->63.7C throttled=0x0 xnnpack=True median_us=235099.0 runs=50 footprint_peak_mb=390.23",
            "invocation 1: exit=0 wall_s=14.0 temp 51.0->63.7C throttled=0x0 xnnpack=True median_us=229520.0 runs=50 footprint_peak_mb=390.73",
            "invocation 2: exit=0 wall_s=14.6 temp 50.5->64.2C throttled=0x0 xnnpack=True median_us=240265.0 runs=50 footprint_peak_mb=390.53"
          ],
          "failure_class": null,
          "full_delegation": null,
          "latency_p50_ms": 235.099,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "init_ms": 152.874,
            "invocations": 3,
            "iterations": 150,
            "latency_max_ms": 246.768,
            "latency_min_ms": 224.436,
            "threads": 4,
            "warmup_runs_per_invocation": 10
          },
          "output_match": null,
          "peak_mem_mb": 390.73,
          "prefill_tokens_per_s": null,
          "provenance": "measured",
          "runs": true,
          "total_ops": null,
          "ttft_ms": null
        },
        "source": "data/device_runs/2.2.0.dev20260804/2026-08-31/ram-plus__ram_stage3_tail_fp16__raspberry-pi-5.json"
      }
    ]
  },
  "model": {
    "family": "ram-plus",
    "id": "ram-plus__ram_stage3_tail_fp16",
    "license": "apache-2.0",
    "source_url": "https://huggingface.co/litert-community/RAM-Plus-LiteRT",
    "task": "image-classification"
  },
  "pitfalls": [
    "Swin-L stage 3 must run on CPU: it fp16-miscomputes on the Mali GPU delegate (fp16 matmul accumulation in the deep, high-magnitude blocks — not overflow, not head_dim) (ram/README.md L33-41; HF card)."
  ],
  "schema_version": "1.2"
}
