{
  "artifacts": [
    {
      "file": "7eec7d677efb6166ff1ba99d6e92a08a2f0c41e2f06ea6c9bfa3f36ac29cd5fa.tflite",
      "sha256": "183c928cd1b109ad0b94d9540dbf4c2660393428d2104494b6047af4aa4eb1da",
      "size_mb": 284.145
    }
  ],
  "benchmarks": [],
  "browser": {
    "backends": [
      {
        "backend": "wasm_xnnpack",
        "date": "2026-08-20",
        "env": {
          "browser": "chromium",
          "browser_version": "151.0.7922.34",
          "headless": true,
          "jspi": true,
          "litertjs_core_version": "2.5.3",
          "machine_label": "mac-studio-m4-max",
          "os": "macOS",
          "os_version": "27.0.0",
          "webgpu_adapter": {
            "architecture": "metal-3",
            "description": "",
            "device": "",
            "vendor": "apple"
          }
        },
        "full_delegation": null,
        "latency_p50_ms": 3691.958,
        "loads": true,
        "max_rel_diff": null,
        "output_match": null,
        "provenance": "measured",
        "runs": true
      },
      {
        "backend": "webgpu_mldrift",
        "date": "2026-08-20",
        "env": {
          "browser": "chromium",
          "browser_version": "151.0.7922.34",
          "headless": true,
          "jspi": true,
          "litertjs_core_version": "2.5.3",
          "machine_label": "mac-studio-m4-max",
          "os": "macOS",
          "os_version": "27.0.0",
          "webgpu_adapter": {
            "architecture": "metal-3",
            "description": "",
            "device": "",
            "vendor": "apple"
          }
        },
        "full_delegation": true,
        "latency_p50_ms": 41.163,
        "loads": true,
        "max_rel_diff": 26909.085798816566,
        "output_match": false,
        "provenance": "measured",
        "runs": true
      }
    ],
    "demo_url": null,
    "sweep_source": "data/sweep/2.5.3/2026-08-20/zipformer-medium-cr-ctc__zipformer_ctc_large_fp16.json"
  },
  "conversion": {
    "command": "python build_zipformer_ctc.py all",
    "quantization": "fp16",
    "tool": "litert-torch",
    "tool_version": "0.10.0"
  },
  "cross_runtime": [],
  "delegation": null,
  "device": {
    "records": [
      {
        "device": "raspberry-pi-5",
        "run": {
          "accelerator": "cpu_xnnpack",
          "context_length": null,
          "date": "2026-08-31",
          "decode_tokens_per_s": null,
          "delegated_ops": null,
          "env": {
            "device": "Raspberry Pi 5 Model B Rev 1.1",
            "machine_label": "raspberry-pi-5",
            "os_build": "Linux-6.18.34+rpt-rpi-2712-aarch64-with-glibc2.41",
            "runtime": "litert",
            "runtime_version": "2.2.0.dev20260804",
            "soc": null,
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "pi5 sweep row: LiteRT benchmark_model, CPU/XNNPACK at --num_threads=4, 3 invocations per file of 10 warm-up + 50 timed runs (the tool caps a phase at 150 s, so very slow graphs run fewer); latency = median of the three per-invocation medians over the timed phase; a row counts as measured only when every invocation exited 0 with XNNPACK engaged and vcgencmd get_throttled 0x0 before and after",
            "versions: ai-edge-litert-nightly=2.2.0.dev20260804, benchmark_model_sha256=babc9275addd8612caf0842294cb5b3f301c3cce46e6c5979cd2803f215f6bee, cpu=Raspberry Pi 5 Model B Rev 1.1, litert-cli-nightly=0.2.0.dev20260805, platform=Linux-6.18.34+rpt-rpi-2712-aarch64-with-glibc2.41, python=3.13.5",
            "invocation 0: exit=0 wall_s=64.0 temp 49.4->64.8C throttled=0x0 xnnpack=True median_us=1056062.0 runs=50 footprint_peak_mb=1014.08",
            "invocation 1: exit=0 wall_s=63.8 temp 50.5->65.3C throttled=0x0 xnnpack=True median_us=1053382.0 runs=50 footprint_peak_mb=1014.08",
            "invocation 2: exit=0 wall_s=63.9 temp 51.0->65.3C throttled=0x0 xnnpack=True median_us=1055052.0 runs=50 footprint_peak_mb=1014.58"
          ],
          "failure_class": null,
          "full_delegation": null,
          "latency_p50_ms": 1055.052,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "init_ms": 485.088,
            "invocations": 3,
            "iterations": 150,
            "latency_max_ms": 1114.58,
            "latency_min_ms": 1051.279,
            "threads": 4,
            "warmup_runs_per_invocation": 10
          },
          "output_match": null,
          "peak_mem_mb": 1014.58,
          "prefill_tokens_per_s": null,
          "provenance": "measured",
          "runs": true,
          "total_ops": null,
          "ttft_ms": null
        },
        "source": "data/device_runs/2.2.0.dev20260804/2026-08-31/zipformer-medium-cr-ctc__zipformer_ctc_large_fp16__raspberry-pi-5.json"
      }
    ]
  },
  "model": {
    "family": "zipformer",
    "id": "zipformer-medium-cr-ctc__zipformer_ctc_large_fp16",
    "license": "apache-2.0",
    "source_url": "https://huggingface.co/litert-community/Zipformer-medium-CR-CTC-LiteRT",
    "task": "automatic-speech-recognition"
  },
  "pitfalls": [
    "Input is host-computed kaldi-fbank [1,1600,80] with the waveform kept in [-1,1] scale, plus 4 additive attention-bias inputs (0 = valid, -1000 = pad; one per downsampling rate) — HF card 'How it runs'.",
    "Output is raw CTC logits at 25 Hz (blank id 0): LogSoftmax is deliberately NOT in the graph (its REDUCE_MAX lowering fails the on-device GPU compile); greedy CTC is argmax-invariant — HF card.",
    "Fixed 16 s window; shorter audio is padded with log(1e-10) fbank frames — HF card."
  ],
  "schema_version": "1.2"
}
