Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
61 changes: 49 additions & 12 deletions .github/ci/benchmark_config.json
Original file line number Diff line number Diff line change
Expand Up @@ -107,12 +107,16 @@
},
{
"address": "0x02000000",
"paths": ["yolo/weights_region.bin"],
"paths": [
"yolo/weights_region.bin"
],
"required": true
},
{
"address": "0x04A00000",
"paths": ["yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -122,15 +126,21 @@
"image_count": 5,
"min_image_count": 5,
"reference_contract": ".github/ci/reference/yolo.json",
"source_shape": [480, 640, 3]
"source_shape": [
480,
640,
3
]
},
"benchmark_cases": [
{
"name": "coco_room_000139",
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -146,7 +156,9 @@
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_cat_524280_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_cat_524280_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -162,7 +174,9 @@
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_giraffes_296969_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_giraffes_296969_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -178,7 +192,9 @@
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_elephants_445248_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_elephants_445248_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand All @@ -194,7 +210,9 @@
"file_loads": [
{
"address": "0x04A00000",
"paths": ["yolo/coco_baseball_043816_raw_480x640x3_uint8_rgb.bin"],
"paths": [
"yolo/coco_baseball_043816_raw_480x640x3_uint8_rgb.bin"
],
"required": true
}
],
Expand Down Expand Up @@ -234,17 +252,24 @@
"file_loads": [
{
"address": "0x0",
"paths": ["zero2m.bin", "common/zero2m.bin"],
"paths": [
"zero2m.bin",
"common/zero2m.bin"
],
"required": true
},
{
"address": "0x2000",
"paths": ["dncnn/dncnn20l64_input.bin"],
"paths": [
"dncnn/dncnn20l64_input.bin"
],
"required": true
},
{
"address": "0x14000",
"paths": ["dncnn/dncnn20l64_weights.bin"],
"paths": [
"dncnn/dncnn20l64_weights.bin"
],
"required": true
}
],
Expand All @@ -253,7 +278,10 @@
"accuracy": {
"kind": "uint8_npy",
"offset": "0x10000",
"shape": [64, 64],
"shape": [
64,
64
],
"max_abs": 2,
"reference_path": "ported_models/dncnn/refs/dncnn20l64_reference.npy",
"comment": "Gates the 64x64 denoised output @0x10000 against the PyTorch/deepinv oracle (refs/dncnn20l64_reference.npy, produced by scripts/gen_dncnn_oracle.py running deepinv.models.DnCNN on the pinned weights). Board-verified: the int8 kernel matches the FP32 oracle at max_abs=1 (3/3 board runs); gate max_abs<=2 is a 1-unit margin."
Expand Down Expand Up @@ -315,6 +343,15 @@
},
"smolvlm_500m": {
"config": "ported_models/llama_cpp_et/benchmarks/smolvlm_500m.json"
},
"openelm_1_1b": {
"config": "ported_models/llama_cpp_et/benchmarks/openelm_1_1b.json"
},
"olmo_1b": {
"config": "ported_models/llama_cpp_et/benchmarks/olmo_1b.json"
},
"granite_3_1b_a400m": {
"config": "ported_models/llama_cpp_et/benchmarks/granite_3_1b_a400m.json"
}
}
}
60 changes: 60 additions & 0 deletions ported_models/granite_3_1b_a400m/docs/RECIPE.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,60 @@
# Granite-3.0-1B-A400M Porting Recipe

## Overview

Adds `ibm-granite/granite-3.0-1b-a400m-instruct` (1B-total-parameter /
400M-active-parameter sparse Mixture-of-Experts causal LM) to the
`llama.cpp-et` framework. Confirmed via local GGUF metadata inspection:
`general.architecture = granitemoe` -- distinct from IBM's dense `granite`
architecture (already claimed elsewhere on this board), and the first MoE
model attempted in this specific porting campaign.

## Model Reference

- **Source**: `ibm-granite/granite-3.0-1b-a400m-instruct` (Hugging Face),
revision `ffec3c35bdfd97a06f0b4cd5fcc92cd9b1584445`
- **License**: Apache-2.0
- **GGUF source**: `bartowski/granite-3.0-1b-a400m-instruct-GGUF`, file
`granite-3.0-1b-a400m-instruct-Q8_0.gguf`
- **Quantization**: Q8_0,
`sha256=8c37dd0c10b73e9304b98a242be4adcd3050c09e9042c4862d49e2cfccf35411`
(verified locally against the downloaded file)
- **Architecture**: `arch = granitemoe` per GGUF metadata, 242 tensors
(roughly double a similarly-sized dense model's tensor count, consistent
with per-expert FFN weight tensors).

## Verification performed this round

Host reference: built a plain CPU-only (`GGML_ET=OFF`) configuration of the
same vendored `llama.cpp-et` source and ran `llama-perplexity` against the
board-pinned WikiText-2 corpus (`wikitext2_raw_test`,
`sha256=173c87a53759e0201f33e0ccf978e510c2042d7f2cb78229d9a50d79b9e7dd08`),
context 128 / batch 128 / ubatch 128 / 4 chunks. The model loads and runs
cleanly on the CPU backend:

```
Final estimate: PPL = 5.7635 +/- 0.84793
```

This confirms the MoE routing path (`GGML_OP_MUL_MAT_ID`, indexed/batched
matmul against a per-token-selected expert weight subset) works correctly
on `ggml-cpu` -- the first MoE model actually exercised in this campaign.

## Open question: MoE routing on the ET backend specifically (not confirmed)

CPU-backend success does not prove the ET-SoC1 backend's own `MUL_MAT_ID`
implementation works -- `ggml-et.cpp` lists it as supported (per the
op-coverage audit in `falcon7b_recipe.md` from earlier in this campaign),
but that has never been exercised against a real MoE model on ET sysemu or
board. Flagging as a genuinely open verification item, not a formality.

## Open items for maintainer review

- Registered in `artifacts.json`, `ported_models/llama_cpp_et/benchmarks/granite_3_1b_a400m.json`,
and `.github/ci/benchmark_config.json` (port 18132) -- board-testable now,
independent of the model-ports track claim below.
- `ported_models/submissions/model_ports/granite_3_1b_a400m.json` is the
model-ports track claim, pending identity approval.
- No changes to any protected file or the vendored submodule.
- MoE routing (`MUL_MAT_ID`) confirmed on CPU, NOT live-verified against ET
sysemu specifically -- see note above. Genuinely unproven on ET, not a hedge.
25 changes: 25 additions & 0 deletions ported_models/granite_3_1b_a400m/oracle/perplexity_oracle.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
{
"oracle_type": "perplexity_threshold",
"model_artifact": {
"repo": "bartowski/granite-3.0-1b-a400m-instruct-GGUF",
"revision": "main",
"filename": "granite-3.0-1b-a400m-instruct-Q8_0.gguf",
"sha256": "8c37dd0c10b73e9304b98a242be4adcd3050c09e9042c4862d49e2cfccf35411"
},
"corpus": {
"artifact": "wikitext2_raw_test",
"sha256": "173c87a53759e0201f33e0ccf978e510c2042d7f2cb78229d9a50d79b9e7dd08"
},
"command": "llama-perplexity --model granite-3.0-1b-a400m-instruct-Q8_0.gguf -f wiki.test.raw -c 128 -b 128 -ub 128 --chunks 4",
"reference_run": {
"final_ppl": 5.7635,
"final_ppl_stderr": 0.84793,
"measured_on": "CPU (ggml-cpu backend, GGML_ET=OFF build)",
"measured_date": "2026-07-25"
},
"comparison_threshold": {
"metric": "final_ppl",
"max_relative_deviation": 0.2,
"note": "Matches this repo's own leaderboard-gate policy (PPL must stay within 20% of best-seen value). A full-offload ET-SoC1 re-run against this exact command/corpus/artifact should land at final_ppl within [4.61, 6.92] to be considered consistent with this reference run. No ET-SoC1 hardware was available to this session to perform that re-run directly."
}
}
54 changes: 54 additions & 0 deletions ported_models/llama_cpp_et/artifacts.json
Original file line number Diff line number Diff line change
Expand Up @@ -485,6 +485,60 @@
"sha256": "d1eb8b6b23979205fdf63703ed10f788131a3f812c7b1f72e0119d5d81295150",
"size_bytes": 108783360,
"note": "SmolVLM 500M vision projector (SigLIP ~93M + MLP). Q8_0 quantized. Must be loaded alongside smolvlm_500m_q8_gguf."
},
"openelm_1_1b_q8_gguf": {
"kind": "model",
"framework": "llama.cpp-et",
"variant": "OpenELM-1_1B-Instruct-Q8_0",
"filename": "OpenELM-1_1B-Instruct-Q8_0.gguf",
"env": "OPENELM_1_1B_MODEL_PATH",
"source": {
"type": "huggingface",
"repo": "LiteLLMs/OpenELM-1_1B-Instruct-GGUF",
"revision": "main",
"filename": "Q8_0/Q8_0-00001-of-00001.gguf",
"url": "https://huggingface.co/LiteLLMs/OpenELM-1_1B-Instruct-GGUF/resolve/main/Q8_0/Q8_0-00001-of-00001.gguf"
},
"sha256": "13dc6676b2355d0356d7892d8b34d8518233d79a28f9ea1d14f94e81d7decad5",
"size": 1148477504,
"local_cache": "local-artifacts/models/OpenELM-1_1B-Instruct-Q8_0.gguf",
"board_path": "/data/models/OpenELM-1_1B-Instruct-Q8_0.gguf"
},
"olmo_1b_q8_gguf": {
"kind": "model",
"framework": "llama.cpp-et",
"variant": "OLMo-1B-hf-Q8_0",
"filename": "OLMo-1B-hf-Q8_0.gguf",
"env": "OLMO_1B_MODEL_PATH",
"source": {
"type": "huggingface",
"repo": "RichardErkhov/allenai_-_OLMo-1B-hf-gguf",
"revision": "main",
"filename": "OLMo-1B-hf.Q8_0.gguf",
"url": "https://huggingface.co/RichardErkhov/allenai_-_OLMo-1B-hf-gguf/resolve/main/OLMo-1B-hf.Q8_0.gguf"
},
"sha256": "11ad66bb4b0c4b9d4b40ccef351a506e322fc8265b87e3638b18802768d8875e",
"size": 1252087840,
"local_cache": "local-artifacts/models/OLMo-1B-hf-Q8_0.gguf",
"board_path": "/data/models/OLMo-1B-hf-Q8_0.gguf"
},
"granite_1b_a400m_q8_gguf": {
"kind": "model",
"framework": "llama.cpp-et",
"variant": "granite-3.0-1b-a400m-instruct-Q8_0",
"filename": "granite-3.0-1b-a400m-instruct-Q8_0.gguf",
"env": "GRANITE_3_1B_A400M_MODEL_PATH",
"source": {
"type": "huggingface",
"repo": "bartowski/granite-3.0-1b-a400m-instruct-GGUF",
"revision": "main",
"filename": "granite-3.0-1b-a400m-instruct-Q8_0.gguf",
"url": "https://huggingface.co/bartowski/granite-3.0-1b-a400m-instruct-GGUF/resolve/main/granite-3.0-1b-a400m-instruct-Q8_0.gguf"
},
"sha256": "8c37dd0c10b73e9304b98a242be4adcd3050c09e9042c4862d49e2cfccf35411",
"size": 1422237440,
"local_cache": "local-artifacts/models/granite-3.0-1b-a400m-instruct-Q8_0.gguf",
"board_path": "/data/models/granite-3.0-1b-a400m-instruct-Q8_0.gguf"
}
}
}
52 changes: 52 additions & 0 deletions ported_models/llama_cpp_et/benchmarks/granite_3_1b_a400m.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
{
"runner": "llama_server",
"board": true,
"framework": {
"name": "llama.cpp-et",
"runner": "llama_server",
"source_artifact": "llama_cpp_source"
},
"artifacts_file": "../artifacts.json",
"canonical_variant": "granite-3.0-1b-a400m-instruct-Q8_0",
"score": {
"metric": "tokens_per_second",
"label": "Decode tokens/s",
"higher_is_better": true
},
"llama_server": {
"source_artifact": "llama_cpp_source",
"model_artifact": "granite_1b_a400m_q8_gguf",
"server_artifact": "llama_server",
"workdir_artifact": "llama_cpp_build",
"host": "127.0.0.1",
"port": 18132,
"device": "ET",
"gpu_layers": 99,
"ctx_size": 2048,
"batch_size": 256,
"ubatch_size": 128,
"parallel": 1,
"cache_ram_mib": 0,
"ready_timeout_s": 300,
"request_timeout_s": 420,
"flash_attn": false,
"api": "completion",
"prompt": "Repeat this token sequence without commentary: OK OK OK OK OK OK OK OK OK OK",
"max_tokens": 96,
"temperature": 0,
"ignore_eos": true,
"min_completion_tokens": 32,
"perplexity": {
"enabled": true,
"perplexity_artifact": "llama_perplexity",
"corpus_artifact": "wikitext2_raw_test",
"ctx_size": 128,
"batch_size": 128,
"ubatch_size": 128,
"timeout_s": 420,
"min_ppl": 1.0,
"max_ppl": 1000.0,
"chunks": 4
}
}
}
52 changes: 52 additions & 0 deletions ported_models/llama_cpp_et/benchmarks/olmo_1b.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
{
"runner": "llama_server",
"board": true,
"framework": {
"name": "llama.cpp-et",
"runner": "llama_server",
"source_artifact": "llama_cpp_source"
},
"artifacts_file": "../artifacts.json",
"canonical_variant": "OLMo-1B-hf-Q8_0",
"score": {
"metric": "tokens_per_second",
"label": "Decode tokens/s",
"higher_is_better": true
},
"llama_server": {
"source_artifact": "llama_cpp_source",
"model_artifact": "olmo_1b_q8_gguf",
"server_artifact": "llama_server",
"workdir_artifact": "llama_cpp_build",
"host": "127.0.0.1",
"port": 18131,
"device": "ET",
"gpu_layers": 99,
"ctx_size": 2048,
"batch_size": 256,
"ubatch_size": 128,
"parallel": 1,
"cache_ram_mib": 0,
"ready_timeout_s": 300,
"request_timeout_s": 420,
"flash_attn": false,
"api": "completion",
"prompt": "Repeat this token sequence without commentary: OK OK OK OK OK OK OK OK OK OK",
"max_tokens": 96,
"temperature": 0,
"ignore_eos": true,
"min_completion_tokens": 32,
"perplexity": {
"enabled": true,
"perplexity_artifact": "llama_perplexity",
"corpus_artifact": "wikitext2_raw_test",
"ctx_size": 128,
"batch_size": 128,
"ubatch_size": 128,
"timeout_s": 420,
"min_ppl": 1.0,
"max_ppl": 1000.0,
"chunks": 4
}
}
}
Loading
Loading