{
  "artifacts": [
    {
      "file": "Shieldstral-1.0-3B-vision_int4.litertlm",
      "sha256": "180b948a4d1d4eacb5219c1124429ee9be392367dcb558e255f7ed466f0f4092",
      "size_mb": 2654.392
    }
  ],
  "benchmarks": [],
  "conversion": {
    "command": "TODO (reproduction script in hf-to-litertlm; invocation not stated on the card)",
    "quantization": "int4 blockwise-32 (OCTAV) decoder + int8 embedding (externalised) + int8 pixtral vision tower, static 560x560 (HF card Variants)",
    "tool": "litert-torch via hf-to-litertlm (reproduction script there); the bundle embeds no Jinja — plain prefix/suffix turn markers only; minimum runtime litert-lm 0.15.0 (HF card Conversion)",
    "tool_version": "litert-torch 0.9.2 · litert-converter 0.3.0 · ai-edge-quantizer 0.8.0 · litert-lm-builder 0.15.0 · transformers 5.14.1 (HF card Conversion)"
  },
  "cross_runtime": [],
  "delegation": null,
  "device": {
    "records": [
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "cpu",
          "context_length": null,
          "date": "2026-09-05",
          "decode_tokens_per_s": 1.87,
          "delegated_ops": 1433,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-cold-cache-cooled",
            "os_build": "Android 16",
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "VERBOSE: Replacing 1051 out of 1187 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 203 partitions for subgraph 0 (prefill_2048).",
            "VERBOSE: Replacing 1051 out of 1187 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 203 partitions for subgraph 1 (prefill_1024).",
            "VERBOSE: Replacing 1051 out of 1187 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 203 partitions for subgraph 2 (prefill_512).",
            "VERBOSE: Replacing 1051 out of 1187 node(s) with delegate (TfLiteXNNPackDelegate) node, yielding 203 partitions for subgraph 3 (prefill_256).",
            "results block: prefill=33.24 decode=1.87 tokens/s"
          ],
          "failure_class": null,
          "full_delegation": false,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 2.0,
            "init_s": 10.05529,
            "prefill_tokens": 231.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": 33.24,
          "provenance": "measured",
          "runs": true,
          "total_ops": 1493,
          "ttft_ms": 7480.0
        },
        "source": "data/device_runs/0.16.0/2026-09-05/shieldstral-1.0-3b-vision-int4__galaxy-s26.json"
      },
      {
        "device": "galaxy-s26",
        "run": {
          "accelerator": "gpu",
          "context_length": null,
          "date": "2026-08-24",
          "decode_tokens_per_s": null,
          "delegated_ops": 52,
          "env": {
            "device": "Galaxy S26 (SM-S942Q)",
            "machine_label": "galaxy-s26-cl-pinned",
            "os_build": "Android 16",
            "runtime": "litert-lm",
            "runtime_version": "0.16.0",
            "soc": "Qualcomm SM8850",
            "vendor_sdk": null
          },
          "error": "runtime/core/engine_advanced_impl.cc:308",
          "evidence": [
            "VERBOSE: Replacing 52 out of 1187 node(s) with delegate (LITERT_CL) node, yielding 2 partitions for subgraph 0 (prefill_2048).",
            "STABLEHLO_COMPOSITE: odml.softmax",
            "52 operations will run on the GPU, and the remaining 1135 operations will run on the CPU.",
            "runtime/core/engine_advanced_impl.cc:308"
          ],
          "failure_class": "engine_create_failed",
          "full_delegation": false,
          "latency_p50_ms": null,
          "loads": false,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": null,
          "provenance": "measured",
          "runs": false,
          "total_ops": 1187,
          "ttft_ms": null
        },
        "source": "data/device_runs/0.16.0/2026-08-24/shieldstral-1.0-3b-vision-int4__galaxy-s26.json"
      }
    ]
  },
  "model": {
    "family": "shieldstral",
    "id": "shieldstral-1.0-3b-vision-int4",
    "license": "apache-2.0",
    "source_url": "https://huggingface.co/litert-community/Shieldstral-1.0-3B",
    "task": "image-text-to-text"
  },
  "pitfalls": [
    "Images are fixed at 560x560 and you must letterbox: the runtime stretches whatever you pass; measured on 100 labelled images stretching costs 8.5 F1 against the source model, padding 3.0, and agreement rises from 92% to 96% — this matters more than int8-vs-int4 (HF card Limitations + Usage).",
    "No continuous score for images (the scoring API is text-only on litert-lm 0.15.0): image documents give the binary verdict only; one image per call (HF card Limitations).",
    "Image quality gates (100 images from an ungated LlavaGuard-derived set): read agreement with the source, not small F1 differences — image margins are far flatter than text ones and a quarter of the items sit within ±1 of the boundary (HF card Image quality gates).",
    "Policy-adaptive safety classifier: one policy per call (ask a single yes/no question per call rather than combining policies); a continuous text score costs two prefills — if you only need a thresholded decision at 0.5, generate one token; the logit margin is faithful in ordering but not scale (engine ≈ 1.16 x reference − 0.43 on CPU int8, residual sd 1.33), so thresholds tuned on the source at values other than 0.5 need re-tuning (HF card Limitations).",
    "No safety guarantee: a moderation aid, not a moderation system — on the gate set precision ≈ 0.78 at recall ≈ 0.96 with a broad 'is this unsafe?' query; both the query wording and the threshold change that trade-off substantially (HF card Limitations).",
    "Text quality gates: every variant sits inside a 1.2-point F1 band with the unquantized reference and its bf16 control (label flips only near |margin| < 0.7); 8-item floor set 8/8 on int4 CPU, int4 GPU and int8 CPU; prefill-length sweep over 82 lengths (95-1067 tokens) with zero failures; upstream reports 81.4 F1 on the full 1,680-item benchmark (HF card Quality gates).",
    "Context exported at 4096 tokens (the source supports far more; longer documents need a re-export) (HF card Limitations).",
    "LiteRT-LM .litertlm bundle: LiteRT.js cannot run it, so delegation stays null and there is no browser block; device rows come from data/device_runs/ (Pi 5 / S26 / Pixel 8a / Mac / iPhone as measured)."
  ],
  "schema_version": "1.2"
}
