{
  "artifacts": [
    {
      "file": "DeepSeek-R1-Distill-Qwen-1.5B_multi-prefill-seq_q8_ekv4096.litertlm",
      "sha256": "69b35f01759eed765641ab4af589bbe98131fd2825662a086d9037409b8c1295",
      "size_mb": 1748.516
    }
  ],
  "benchmarks": [],
  "conversion": {
    "command": "TODO (owner) — not recorded in the source card",
    "quantization": "int4 weights — blockwise (block 32) + OCTAV optimal clipping, symmetric; embedding INT8; integer compute; KV cache 4096. File model.litertlm, ~1.0 GB",
    "tool": "official upstream litert-torch export_hf (clean worktree at upstream/main, no fork); Qwen2ForCausalLM rides the stock converter, no custom code",
    "tool_version": "TODO (owner) — the card pins the tree ('upstream/main'), not a version number"
  },
  "cross_runtime": [],
  "delegation": null,
  "device": {
    "records": [
      {
        "device": "mac-studio-m4-max",
        "run": {
          "accelerator": "gpu",
          "context_length": null,
          "date": "2026-07-22",
          "decode_tokens_per_s": null,
          "delegated_ops": null,
          "env": {
            "device": "Mac Studio (M4 Max)",
            "machine_label": "mac-studio-m4-max",
            "os_build": null,
            "runtime": "litert-lm",
            "runtime_version": "0.14.0",
            "soc": "Apple M4 Max",
            "vendor_sdk": null
          },
          "error": null,
          "evidence": [
            "speed figures withheld: this measurement batch was retracted as contaminated (parallel GPU load; owner correction in gpu_audit MATRIX.md, 2026-07-23) — the PASS/FAIL verdict is load-independent and stands"
          ],
          "failure_class": null,
          "full_delegation": null,
          "latency_p50_ms": null,
          "loads": true,
          "max_abs_diff": null,
          "max_rel_diff": null,
          "metrics": {
            "decode_tokens": 256.0,
            "max_num_tokens": 1024.0,
            "prefill_tokens": 256.0
          },
          "output_match": null,
          "peak_mem_mb": null,
          "prefill_tokens_per_s": null,
          "provenance": "measured",
          "runs": true,
          "total_ops": null,
          "ttft_ms": null
        },
        "source": "data/device_runs/0.14.0/2026-07-22/r1-distill-qwen-1.5b__mac-studio-m4-max.json"
      }
    ]
  },
  "model": {
    "family": "deepseek-r1",
    "id": "r1-distill-qwen-1.5b",
    "license": "mit",
    "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B",
    "task": "text-generation"
  },
  "pitfalls": [
    "Reasoning model: opens a <think> ... </think> chain before the answer. The card's GSM8K run uses max_new_tokens=2048 — budget for the thinking block.",
    "GSM8K (n=100, greedy, 0-shot, identical prompt + extraction): int4 73.0% vs bf16 81.0% — at 1.5B, int4 costs ~8 pt (small-model 4-bit sensitivity; the 7B sibling is at -1 pt parity). Shipped as int4 for the best on-device size/speed.",
    "Prompt template is DeepSeek's own, bundled with the tokenizer: <｜User｜> / <｜Assistant｜>, stop token <｜end▁of▁sentence｜>.",
    "Gallery import: Google AI Edge Gallery 1.0.15+ supports .litertlm; v1.0.16+ can import directly from Hugging Face inside the app — the adb sideload steps are only needed on older builds or for local files.",
    "Measured decode ~116 tok/s (Mac M-series, Metal GPU, greedy); runs on 8 GB phones (iPhone / Android) at ~1 GB file size."
  ],
  "schema_version": "1.2"
}
