{
  "artifacts": [
    {
      "file": "LFM2.5-VL-1.6B_int8.litertlm",
      "sha256": "d6a254aa3edefab51684918e7c717d096c60e495c085fcc35fd845fb8c80c8bd",
      "size_mb": 1725.193
    }
  ],
  "benchmarks": [],
  "conversion": {
    "command": "python convert_lfm25_vl3b.py ../src_models/lfm25-vl-1.6b out_vl16b_int8 && python quantize_vl.py out_vl16b_int8/model.litertlm LFM2.5-VL-1.6B_int8.litertlm --recipe none  # export-time dynamic int8 is the convert default (no --fp); the quantize_vl pass adds ExecutorMetadata + the externalized-embedder int8 recipe (RESULTS.md)",
    "quantization": "int8 dynamic (text linears + convs + embedding, vision tower) — export-time dynamic_wi8_afp32; externalized embedder int8 via quantize_vl",
    "tool": "litert-torch export_hf --task image_text_to_text (via litertlm-convert lfm25vl_work/convert_lfm25_vl3b.py + quantize_vl.py; public mirror hf-to-litertlm lfm_work/convert_lfm25_vl.py + quantize_lfm25_vl.py)",
    "tool_version": "0.9.3 (.venv-vl093: transformers 5.14.1, torch 2.12.1, torchvision 0.27.1)"
  },
  "cross_runtime": [],
  "delegation": null,
  "device": {
    "records": [
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "cpu",
          "context_length": null,
          "date": "2026-09-05",
          "decode_tokens_per_s": 41.66,
          "delegated_ops": 760,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-cold-cache-cooled",
            "os_build": "Android 16",
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "VERBOSE: Replacing 459 out of 543 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 104 partitions for subgraph 0 (prefill_1024).",
            "VERBOSE: Replacing 459 out of 543 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 104 partitions for subgraph 1 (prefill_512).",
            "VERBOSE: Replacing 459 out of 543 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 104 partitions for subgraph 2 (prefill_256).",
            "VERBOSE: Replacing 459 out of 543 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 104 partitions for subgraph 3 (prefill_128).",
            "results block: prefill=525.76 decode=41.66 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": false,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 122.0,
            "init_s": 1.94013,
            "prefill_tokens": 205.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 525.76,
          "provenance": "measured",
          "runs": true,
          "total_ops": 801,
          "ttft_ms": 410.0
        },
        "source": "data/device_runs/0.16.0/2026-09-05/lfm25-vl-1.6b-int8__galaxy-s26.json"
      },
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "gpu",
          "context_length": null,
          "date": "2026-08-24",
          "decode_tokens_per_s": 42.27,
          "delegated_ops": 543,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-cl-pinned",
            "os_build": "Android 16",
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "VERBOSE: Replacing 543 out of 543 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 0 (prefill_1024).",
            "VERBOSE: Replacing 543 out of 543 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 1 (prefill_512).",
            "VERBOSE: Replacing 543 out of 543 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 2 (prefill_256).",
            "VERBOSE: Replacing 543 out of 543 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 3 (prefill_128).",
            "results block: prefill=1518.86 decode=42.27 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": true,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 113.0,
            "init_s": 2.15942,
            "prefill_tokens": 205.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 1518.86,
          "provenance": "measured",
          "runs": true,
          "total_ops": 543,
          "ttft_ms": 160.0
        },
        "source": "data/device_runs/0.16.0/2026-08-24/lfm25-vl-1.6b-int8__galaxy-s26.json"
      },
      {
        "device": "pixel-8a",
        "run": {
          "accelerator": "cpu",
          "context_length": null,
          "date": "2026-08-13",
          "decode_tokens_per_s": 8.38,
          "delegated_ops": null,
          "env": {
            "device": "pixel-8a",
            "machine_label": "pixel-8a-cl-pinned",
            "os_build": null,
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Google Tensor G3",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "results block: prefill=60.42 decode=8.38 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": null,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 1515.0,
            "init_s": 16.7728,
            "prefill_tokens": 296.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 60.42,
          "provenance": "measured",
          "runs": true,
          "total_ops": null,
          "ttft_ms": 5020.0
        },
        "source": "data/device_runs/0.16.0/2026-08-13/lfm25-vl-1.6b-int8__pixel-8a.json"
      },
      {
        "device": "pixel-8a",
        "run": {
          "accelerator": "gpu",
          "context_length": null,
          "date": "2026-08-13",
          "decode_tokens_per_s": 15.91,
          "delegated_ops": 543,
          "env": {
            "device": "pixel-8a",
            "machine_label": "pixel-8a-cl-pinned",
            "os_build": null,
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Google Tensor G3",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "VERBOSE: Replacing 543 out of 543 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 0 (prefill_1024).",
            "VERBOSE: Replacing 543 out of 543 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 1 (prefill_512).",
            "VERBOSE: Replacing 543 out of 543 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 2 (prefill_256).",
            "VERBOSE: Replacing 543 out of 543 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 3 (prefill_128).",
            "results block: prefill=403.31 decode=15.91 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": true,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 3758.0,
            "init_s": 80.45954,
            "prefill_tokens": 296.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 403.31,
          "provenance": "measured",
          "runs": true,
          "total_ops": 543,
          "ttft_ms": 800.0
        },
        "source": "data/device_runs/0.16.0/2026-08-13/lfm25-vl-1.6b-int8__pixel-8a.json"
      }
    ]
  },
  "model": {
    "family": "lfm2_vl",
    "id": "lfm25-vl-1.6b-int8",
    "license": "lfm-open-license-v1.0",
    "source_url": "https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B",
    "task": "image-text-to-text"
  },
  "pitfalls": [
    "int4 vs int8 trade on this model: int8 prefills faster on CPU, int4 decodes markedly faster (blockwise-int4 repacking cost sits in prefill) — pick by whether prompts or outputs dominate; text graph delegates fully on Android OpenCL (543/543, zero rejections) (HF card, measured).",
    "Same VLM-bundle mechanics as the siblings: transformers==5.14.1 pin; llm_model_type { lfm2 {} } with bare <image> template; externalized-embedder int8 pass; pack/unpack post-processing (single-tflite tools drop vision sections) (REPRODUCE.md).",
    "iPhone GPU blocked (LiteRT-LM#3129) — iOS runs CPU; use litert-lm >= 0.16.0 for Android OpenCL / macOS GPU (HF card).",
    "Android v0.16.0 litert_lm_main has no image flag (--max_num_images only) — image input is API-only on Android for now, so device rows are text-graph numbers (RESULTS.md)."
  ],
  "schema_version": "1.2"
}
