{
  "artifacts": [
    {
      "file": "granite-4.1-3b_int8.litertlm",
      "sha256": "97bd368d56b29740d52532abcc8c8d1fb0bf8e0828c226c39ed52d1ef4551193",
      "size_mb": 3654.16
    }
  ],
  "benchmarks": [],
  "conversion": {
    "command": "hf-to-litertlm granite41_work/ reproduction script (pip-only stack, no PYTHONPATH patches)",
    "quantization": "int8 dynamic per-channel on linears + embedding",
    "tool": "litert-torch (pristine released stack; repro = hf-to-litertlm granite41_work/)",
    "tool_version": "0.9.3 (litert-converter 0.3.1, ai-edge-quantizer 0.8.0, litert-lm-builder 0.16.0 — FINDINGS.md .venv-vl093)"
  },
  "cross_runtime": [],
  "delegation": null,
  "device": {
    "records": [
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "cpu",
          "context_length": null,
          "date": "2026-09-06",
          "decode_tokens_per_s": 8.61,
          "delegated_ops": 2172,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-cold-cache-cooled",
            "os_build": "Android 16",
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "VERBOSE: Replacing 1619 out of 1784 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 238 partitions for subgraph 0 (prefill_1024).",
            "VERBOSE: Replacing 1619 out of 1784 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 238 partitions for subgraph 1 (prefill_512).",
            "VERBOSE: Replacing 1619 out of 1784 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 238 partitions for subgraph 2 (prefill_256).",
            "VERBOSE: Replacing 1619 out of 1784 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 238 partitions for subgraph 3 (prefill_128).",
            "results block: prefill=67.58 decode=8.61 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": false,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 306.0,
            "init_s": 6.44145,
            "prefill_tokens": 202.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 67.58,
          "provenance": "measured",
          "runs": true,
          "total_ops": 2258,
          "ttft_ms": 3110.0
        },
        "source": "data/device_runs/0.16.0/2026-09-06/granite-4.1-3b-int8__galaxy-s26.json"
      },
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "gpu",
          "context_length": null,
          "date": "2026-08-25",
          "decode_tokens_per_s": 9.52,
          "delegated_ops": 1784,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-cl-pinned",
            "os_build": "Android 16",
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "VERBOSE: Replacing 1784 out of 1784 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 0 (prefill_1024).",
            "VERBOSE: Replacing 1784 out of 1784 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 1 (prefill_512).",
            "VERBOSE: Replacing 1784 out of 1784 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 2 (prefill_256).",
            "VERBOSE: Replacing 1784 out of 1784 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 3 (prefill_128).",
            "results block: prefill=162.16 decode=9.52 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": true,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 275.0,
            "init_s": 23.93209,
            "prefill_tokens": 202.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 162.16,
          "provenance": "measured",
          "runs": true,
          "total_ops": 1784,
          "ttft_ms": 1350.0
        },
        "source": "data/device_runs/0.16.0/2026-08-25/granite-4.1-3b-int8__galaxy-s26.json"
      },
      {
        "device": "mac-m4-max",
        "run": {
          "accelerator": "gpu",
          "context_length": null,
          "date": "2026-08-17",
          "decode_tokens_per_s": 32.74,
          "delegated_ops": null,
          "env": {
            "device": "Mac Studio M4 Max",
            "machine_label": "mac-studio-m4-max",
            "os_build": null,
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": null,
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "results block: prefill=743.30 decode=32.74 tokens/s, init=5.9184 s"
          ],
          "failure_class": null,
          "full_delegation": null,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 256.0,
            "init_s": 5.9184,
            "max_num_tokens": 1024.0,
            "prefill_tokens": 256.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 743.3,
          "provenance": "measured",
          "runs": true,
          "total_ops": null,
          "ttft_ms": 398.9
        },
        "source": "data/device_runs/0.16.0/2026-08-17/granite-4.1-3b-int8__mac-m4-max.json"
      },
      {
        "device": "pixel-8a",
        "run": {
          "accelerator": "gpu",
          "context_length": null,
          "date": "2026-08-17",
          "decode_tokens_per_s": 3.86,
          "delegated_ops": 1784,
          "env": {
            "device": "Pixel 8a",
            "machine_label": "pixel-8a",
            "os_build": null,
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Tensor G3",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "VERBOSE: Replacing 1784 out of 1784 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 0 (prefill_1024).",
            "VERBOSE: Replacing 1784 out of 1784 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 1 (prefill_512).",
            "VERBOSE: Replacing 1784 out of 1784 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 2 (prefill_256).",
            "VERBOSE: Replacing 1784 out of 1784 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 3 (prefill_128).",
            "results block: prefill=15.48 decode=3.86 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": true,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 79.0,
            "init_s": 65.3055,
            "prefill_tokens": 17.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 15.48,
          "provenance": "measured",
          "runs": true,
          "total_ops": 1784,
          "ttft_ms": 1360.0
        },
        "source": "data/device_runs/0.16.0/2026-08-17/granite-4.1-3b-int8__pixel-8a.json"
      },
      {
        "device": "raspberry-pi-5",
        "run": {
          "accelerator": "cpu",
          "context_length": null,
          "date": "2026-09-01",
          "decode_tokens_per_s": 1.73,
          "delegated_ops": null,
          "env": {
            "device": "Raspberry Pi 5 Model B Rev 1.1",
            "machine_label": "raspberry-pi-5-cooled-52c",
            "os_build": "Linux-6.18.34+rpt-rpi-2712-aarch64-with-glibc2.41",
            "runtime": "litert-lm",
            "runtime_version": "0.16.1",
            "soc": "Broadcom BCM2712",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "pi5 LLM sweep row: `litert-lm benchmark --backend cpu --cpu-thread-count 4 -p 256 -d 256 --runs 1 --cache memory` (wave-2 driver pi5_llm_bench.py; --cache memory rather than the house --cache no, which OOM-kills every >=1.2B file on the 8 GB Pi — equivalence measured on granite-350m int8, +2-3%), 3 invocations per file with cool-down to <=52 C between them, vcgencmd measure_temp + get_throttled logged per invocation, peak RSS polled from /proc; throughput = median of the three invocations (spread in metrics); a row counts as measured only when the real-generation gate (`litert-lm run`, degenerate-output check) passed and every invocation exited 0 with get_throttled 0x0",
            "versions: cpu=Raspberry Pi 5 Model B Rev 1.1, litert-lm=0.16.1, litert-lm-api=0.16.1, platform=Linux-6.18.34+rpt-rpi-2712-aarch64-with-glibc2.41, python=3.13.5",
            "cache mode 'memory'; -p 256 -d 256 --runs 1 --cpu-thread-count 4",
            "gate ('What is 17 plus 26? Answer with the number only.'): status pass, exit 0, wall 140.0 s, output head '43'",
            "invocation 0: exit=0 wall_s=514.3 temp 42.8->53.2C throttled=0x0 prefill_tps=12.04 decode_tps=1.7 ttft_s=21.9243 init_s=90.569 peak_rss_mb=4983",
            "invocation 1: exit=0 wall_s=468.3 temp 51.6->52.1C throttled=0x0 prefill_tps=12.64 decode_tps=1.73 ttft_s=20.9089 init_s=89.9779 peak_rss_mb=4981",
            "invocation 2: exit=0 wall_s=466.3 temp 48.8->54.3C throttled=0x0 prefill_tps=12.83 decode_tps=1.74 ttft_s=20.6681 init_s=90.2437 peak_rss_mb=4984"
          ],
          "failure_class": null,
          "full_delegation": null,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 256.0,
            "decode_tps_max": 1.74,
            "decode_tps_min": 1.7,
            "init_s": 90.2437,
            "init_s_max": 90.569,
            "init_s_min": 89.9779,
            "invocations": 3.0,
            "prefill_tokens": 256.0,
            "prefill_tps_max": 12.83,
            "prefill_tps_min": 12.04,
            "runs_per_invocation": 1.0,
            "threads": 4.0,
            "ttft_s_max": 21.9243,
            "ttft_s_min": 20.6681
          },
          "output_match": null,
          "peak_mem_mb": 4984.0,
          "prefill_tokens_per_s": 12.64,
          "provenance": "measured",
          "runs": true,
          "total_ops": null,
          "ttft_ms": 20908.9
        },
        "source": "data/device_runs/0.16.1/2026-09-01/granite-4.1-3b-int8__raspberry-pi-5.json"
      }
    ]
  },
  "model": {
    "family": "granite",
    "id": "granite-4.1-3b-int8",
    "license": "apache-2.0",
    "source_url": "https://huggingface.co/litert-community/granite-4.1-3b",
    "task": "text-generation"
  },
  "pitfalls": [
    "Requires litert-lm >= 0.16 (HF card header).",
    "No start_token in the metadata, on purpose: Granite sets add_bos_token False and its BOS is its EOS (<|end_of_text|>); with the converter-written start_token the runtime prepends end-of-text and the model echoes the question back — 5/8 vs 8/8 on the sanity gate. Reproduces on bf16 PyTorch with the token prepended by hand (HF card conversion notes).",
    "Prefill ladder trimmed to six signatures (1024/256/64/16/4/1): every exported signature is charged engine memory whether or not called; the eleven-signature build was killed by iOS during Metal init unless context was capped at 1024 (HF card conversion notes).",
    "Embedder externalised so the tied 100352x2560 vocab table sits in its own section, clear of the ~2 GiB single-section mmap ceiling on iOS (HF card).",
    "GSM8K 0-shot CoT n=100: bf16 88.0 / int8 87.0 / int4 84.0 — int4 costs 4 points for 1.7x smaller bytes (HF card quality table)."
  ],
  "schema_version": "1.2"
}
