{
  "artifacts": [
    {
      "file": "model.litertlm",
      "sha256": "5ae4dbc96c8d4919e4776e9b8eee7f5ece8bdd2d1c057f73576c3d5f289d7ca4",
      "size_mb": 780.13
    }
  ],
  "benchmarks": [],
  "conversion": {
    "command": "bash scripts/ship_internvl3_5_1b.sh",
    "quantization": "vision tower int8; Qwen3-0.6B decoder int4 blockwise-32 symmetric + OCTAV; input embedding int8 (externalized section)",
    "tool": "litert-torch (litertlm-convert fast_vlm pipeline, scripts/ship_internvl3_5_1b.sh)",
    "tool_version": "0.10.0 (editable dev checkout 115a136 + local patches)"
  },
  "cross_runtime": [],
  "delegation": null,
  "device": {
    "records": [
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "cpu",
          "context_length": null,
          "date": "2026-09-05",
          "decode_tokens_per_s": 26.13,
          "delegated_ops": 1896,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-cold-cache-cooled",
            "os_build": "Android 16",
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "VERBOSE: Replacing 1126 out of 1298 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 221 partitions for subgraph 0 (prefill_128).",
            "VERBOSE: Replacing 1126 out of 1298 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 221 partitions for subgraph 1 (prefill_512).",
            "VERBOSE: Replacing 1070 out of 1244 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 226 partitions for subgraph 2 (decode).",
            "VERBOSE: Replacing 7 out of 7 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 1 partitions for subgraph 3 (odml.rms_norm.impl).",
            "results block: prefill=222.23 decode=26.13 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": false,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 179.0,
            "init_s": 1.12643,
            "prefill_tokens": 202.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 222.23,
          "provenance": "measured",
          "runs": true,
          "total_ops": 1958,
          "ttft_ms": 950.0
        },
        "source": "data/device_runs/0.16.0/2026-09-05/internvl3_5-1b__galaxy-s26.json"
      },
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "gpu",
          "context_length": null,
          "date": "2026-08-24",
          "decode_tokens_per_s": 43.45,
          "delegated_ops": 1298,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-cl-pinned",
            "os_build": "Android 16",
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "VERBOSE: Replacing 1298 out of 1298 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 0 (prefill_128).",
            "VERBOSE: Replacing 1298 out of 1298 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 1 (prefill_512).",
            "VERBOSE: Replacing 1244 out of 1244 node(s) with delegate (LITERT_CL) node, yielding 1 partitions for subgraph 2 (decode).",
            "results block: prefill=962.12 decode=43.45 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": true,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 114.0,
            "init_s": 1.90011,
            "prefill_tokens": 203.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 962.12,
          "provenance": "measured",
          "runs": true,
          "total_ops": 1298,
          "ttft_ms": 230.0
        },
        "source": "data/device_runs/0.16.0/2026-08-24/internvl3_5-1b__galaxy-s26.json"
      }
    ]
  },
  "model": {
    "family": "internvl",
    "id": "internvl3_5-1b",
    "license": "apache-2.0",
    "source_url": "https://huggingface.co/OpenGVLab/InternVL3_5-1B",
    "task": "image-text-to-text"
  },
  "pitfalls": [
    "GPU (Metal) fast_vlm path: a second image in the same conversation truncates the answer — ask about one image per chat on GPU; multi-image works on the CPU backend (also reproduces with litert-community/FastVLM-0.5B, so it is runtime-level, not model-level).",
    "Vision-only bundle (no audio tower): create the engine with the vision modality only — requesting the audio tower (.all) on a bundle with no audio section fails at session creation.",
    "Image input is resized to 448x448; ImageNet normalization and the NCHW transpose are baked into the vision encoder (the runtime feeds a [0,1] NHWC image).",
    "InternViT attention is rewritten 4D-clean (qkv split before the head reshape) to avoid a 5D intermediate for the GPU delegate.",
    "Decoder is extracted from the InternVLChat wrapper as a standalone Qwen3ForCausalLM with dynamic rope_scaling stripped; exported with cache <= base max so base RoPE is exact."
  ],
  "schema_version": "1.2"
}
