Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
55 changes: 43 additions & 12 deletions .github/ci/benchmark_config.json
Original file line number Diff line number Diff line change
Expand Up @@ -107,12 +107,16 @@
},
{
"address": "0x02000000",
"paths": ["yolo/weights_region.bin"],
"paths": [
"yolo/weights_region.bin"
],
"required": true
},
{
"address": "0x04A00000",
"paths": ["yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -122,15 +126,21 @@
"image_count": 5,
"min_image_count": 5,
"reference_contract": ".github/ci/reference/yolo.json",
"source_shape": [480, 640, 3]
"source_shape": [
480,
640,
3
]
},
"benchmark_cases": [
{
"name": "coco_room_000139",
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -146,7 +156,9 @@
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_cat_524280_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_cat_524280_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -162,7 +174,9 @@
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_giraffes_296969_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_giraffes_296969_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -178,7 +192,9 @@
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_elephants_445248_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_elephants_445248_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -194,7 +210,9 @@
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_baseball_043816_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_baseball_043816_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand Down Expand Up @@ -234,17 +252,24 @@
"file_loads": [
{
"address": "0x0",
"paths": ["zero2m.bin", "common/zero2m.bin"],
"paths": [
"zero2m.bin",
"common/zero2m.bin"
],
"required": true
},
{
"address": "0x2000",
"paths": ["dncnn/dncnn20l64_input.bin"],
"paths": [
"dncnn/dncnn20l64_input.bin"
],
"required": true
},
{
"address": "0x14000",
"paths": ["dncnn/dncnn20l64_weights.bin"],
"paths": [
"dncnn/dncnn20l64_weights.bin"
],
"required": true
}
],
Expand All @@ -253,7 +278,10 @@
"accuracy": {
"kind": "uint8_npy",
"offset": "0x10000",
"shape": [64, 64],
"shape": [
64,
64
],
"max_abs": 2,
"reference_path": "ported_models/dncnn/refs/dncnn20l64_reference.npy",
"comment": "Gates the 64x64 denoised output @0x10000 against the PyTorch/deepinv oracle (refs/dncnn20l64_reference.npy, produced by scripts/gen_dncnn_oracle.py running deepinv.models.DnCNN on the pinned weights). Board-verified: the int8 kernel matches the FP32 oracle at max_abs=1 (3/3 board runs); gate max_abs<=2 is a 1-unit margin."
Expand Down Expand Up @@ -315,6 +343,9 @@
},
"smolvlm_500m": {
"config": "ported_models/llama_cpp_et/benchmarks/smolvlm_500m.json"
},
"granite4_h_micro": {
"config": "ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json"
}
}
}
60 changes: 60 additions & 0 deletions ported_models/granite4_h_micro/docs/RECIPE.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,60 @@
# Granite-4.0-H-Micro Porting Recipe

## Overview

Adds `ibm-granite/granite-4.0-h-micro` (IBM's fourth-generation Granite,
hybrid attention+Mamba architecture) to the `llama.cpp-et` framework.
Confirmed via local GGUF metadata inspection: `general.architecture =
granitehybrid` -- distinct from both `granite` (dense, already ported
elsewhere on this board) and `granitemoe` (MoE, already ported earlier in
this campaign as `granite_3_1b_a400m`). A third, genuinely different
Granite execution family.

## Model Reference

- **Source**: `ibm-granite/granite-4.0-h-micro` (Hugging Face), revision
`d5f01a3ea75f088947be3aae039f4ad52837dfde`
- **License**: Apache-2.0
- **GGUF source**: `ibm-granite/granite-4.0-h-micro-GGUF` (official), file
`granite-4.0-h-micro-Q8_0.gguf`
- **Quantization**: Q8_0,
`sha256=a009111abf2865b7aad1e66326a6c772cddc29bccd22898f470292068b27bb59`
(verified locally against the downloaded file)
- **Architecture**: `arch = granitehybrid` per GGUF metadata, 506 tensors.

## Verification performed this round

Host reference: built a plain CPU-only (`GGML_ET=OFF`) configuration of the
same vendored `llama.cpp-et` source and ran `llama-perplexity` against the
board-pinned WikiText-2 corpus (`wikitext2_raw_test`,
`sha256=173c87a53759e0201f33e0ccf978e510c2042d7f2cb78229d9a50d79b9e7dd08`),
context 128 / batch 128 / ubatch 128 / 4 chunks. The model loads and runs
cleanly, allocating **both** a small transformer KV cache (`llama_kv_cache`,
only 4 layers) **and** a much larger recurrent SSM state cache
(`llama_memory_recurrent`, 40 layers) -- a hybrid weighted heavily toward
recurrent layers (4 attention : 40 recurrent), a different ratio from
`falcon_h1_1_5b`'s hybrid mix earlier in this campaign:

```
Final estimate: PPL = 13.2058 +/- 2.78765
```

This is a third data point (after `mamba_1_4b`, `falcon_h1_1_5b`) on
`SSM_CONV`/`SSM_SCAN` working correctly on `ggml-cpu`.

## Why this port's ET-SoC1 kernel support is a real, open question

Same caveat as `mamba_1_4b`/`falcon_h1_1_5b`: CPU success doesn't prove
ET-SoC1 support for `SSM_CONV`/`SSM_SCAN`. Three independent hybrid/SSM
models now share this same open question in this campaign.

## Open items for maintainer review

- Registered in `artifacts.json`, `ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json`,
and `.github/ci/benchmark_config.json` (port 18142) -- board-testable
now, independent of the model-ports track claim below.
- `ported_models/submissions/model_ports/granite4_h_micro.json` is the
model-ports track claim, pending identity approval.
- No changes to any protected file or the vendored submodule.
- SSM ops confirmed on CPU, NOT live-verified against ET sysemu
specifically.
25 changes: 25 additions & 0 deletions ported_models/granite4_h_micro/oracle/perplexity_oracle.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
{
"oracle_type": "perplexity_threshold",
"model_artifact": {
"repo": "ibm-granite/granite-4.0-h-micro-GGUF",
"revision": "main",
"filename": "granite-4.0-h-micro-Q8_0.gguf",
"sha256": "a009111abf2865b7aad1e66326a6c772cddc29bccd22898f470292068b27bb59"
},
"corpus": {
"artifact": "wikitext2_raw_test",
"sha256": "173c87a53759e0201f33e0ccf978e510c2042d7f2cb78229d9a50d79b9e7dd08"
},
"command": "llama-perplexity --model granite-4.0-h-micro-Q8_0.gguf -f wiki.test.raw -c 128 -b 128 -ub 128 --chunks 4",
"reference_run": {
"final_ppl": 13.2058,
"final_ppl_stderr": 2.78765,
"measured_on": "CPU (ggml-cpu backend, GGML_ET=OFF build)",
"measured_date": "2026-07-25"
},
"comparison_threshold": {
"metric": "final_ppl",
"max_relative_deviation": 0.2,
"note": "Matches this repo's own leaderboard-gate policy (PPL must stay within 20% of best-seen value). A full-offload ET-SoC1 re-run against this exact command/corpus/artifact should land at final_ppl within [10.56, 15.85] to be considered consistent with this reference run. No ET-SoC1 hardware was available to this session to perform that re-run directly."
}
}
17 changes: 17 additions & 0 deletions ported_models/llama_cpp_et/artifacts.json
Original file line number Diff line number Diff line change
Expand Up @@ -485,6 +485,23 @@
"sha256": "d1eb8b6b23979205fdf63703ed10f788131a3f812c7b1f72e0119d5d81295150",
"size_bytes": 108783360,
"note": "SmolVLM 500M vision projector (SigLIP ~93M + MLP). Q8_0 quantized. Must be loaded alongside smolvlm_500m_q8_gguf."
},
"granite4_h_micro_q8_gguf": {
"kind": "model",
"framework": "llama.cpp-et",
"variant": "granite-4.0-h-micro-Q8_0",
"filename": "granite-4.0-h-micro-Q8_0.gguf",
"env": "GRANITE4_H_MICRO_MODEL_PATH",
"source": {
"type": "huggingface",
"repo": "ibm-granite/granite-4.0-h-micro-GGUF",
"revision": "main",
"filename": "granite-4.0-h-micro-Q8_0.gguf",
"url": "https://huggingface.co/ibm-granite/granite-4.0-h-micro-GGUF/resolve/main/granite-4.0-h-micro-Q8_0.gguf"
},
"sha256": "a009111abf2865b7aad1e66326a6c772cddc29bccd22898f470292068b27bb59",
"local_cache": "local-artifacts/models/granite-4.0-h-micro-Q8_0.gguf",
"board_path": "/data/models/granite-4.0-h-micro-Q8_0.gguf"
}
}
}
52 changes: 52 additions & 0 deletions ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
{
"runner": "llama_server",
"board": true,
"framework": {
"name": "llama.cpp-et",
"runner": "llama_server",
"source_artifact": "llama_cpp_source"
},
"artifacts_file": "../artifacts.json",
"canonical_variant": "granite-4.0-h-micro-Q8_0",
"score": {
"metric": "tokens_per_second",
"label": "Decode tokens/s",
"higher_is_better": true
},
"llama_server": {
"source_artifact": "llama_cpp_source",
"model_artifact": "granite4_h_micro_q8_gguf",
"server_artifact": "llama_server",
"workdir_artifact": "llama_cpp_build",
"host": "127.0.0.1",
"port": 18142,
"device": "ET",
"gpu_layers": 99,
"ctx_size": 2048,
"batch_size": 256,
"ubatch_size": 128,
"parallel": 1,
"cache_ram_mib": 0,
"ready_timeout_s": 300,
"request_timeout_s": 420,
"flash_attn": false,
"api": "completion",
"prompt": "Repeat this token sequence without commentary: OK OK OK OK OK OK OK OK OK OK",
"max_tokens": 96,
"temperature": 0,
"ignore_eos": true,
"min_completion_tokens": 32,
"perplexity": {
"enabled": true,
"perplexity_artifact": "llama_perplexity",
"corpus_artifact": "wikitext2_raw_test",
"ctx_size": 128,
"batch_size": 128,
"ubatch_size": 128,
"timeout_s": 420,
"min_ppl": 1.0,
"max_ppl": 1000.0,
"chunks": 4
}
}
}
16 changes: 16 additions & 0 deletions ported_models/submissions/model_ports/granite4_h_micro.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,16 @@
{
"schema_version": 1,
"track": "most_models_ported",
"benchmark_model": "granite4_h_micro",
"identity_id": "granitehybrid",
"source": {
"repo": "ibm-granite/granite-4.0-h-micro",
"revision": "d5f01a3ea75f088947be3aae039f4ad52837dfde",
"license": "apache-2.0"
},
"implementation_paths": [
"ported_models/granite4_h_micro"
],
"benchmark_config": "ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json",
"recipe": "ported_models/granite4_h_micro/docs/RECIPE.md"
}
Loading