diff --git a/.github/ci/benchmark_config.json b/.github/ci/benchmark_config.json index 43f8895d..6add0f12 100644 --- a/.github/ci/benchmark_config.json +++ b/.github/ci/benchmark_config.json @@ -107,12 +107,16 @@ }, { "address": "0x02000000", - "paths": ["yolo/weights_region.bin"], + "paths": [ + "yolo/weights_region.bin" + ], "required": true }, { "address": "0x04A00000", - "paths": ["yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"], + "paths": [ + "yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin" + ], "required": true } ], @@ -122,7 +126,11 @@ "image_count": 5, "min_image_count": 5, "reference_contract": ".github/ci/reference/yolo.json", - "source_shape": [480, 640, 3] + "source_shape": [ + 480, + 640, + 3 + ] }, "benchmark_cases": [ { @@ -130,7 +138,9 @@ "file_loads": [ { "address": "0x04A00000", - "paths": ["yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin"], + "paths": [ + "yolo/coco_room_000139_raw_480x640x3_uint8_rgb.bin" + ], "required": true } ], @@ -146,7 +156,9 @@ "file_loads": [ { "address": "0x04A00000", - "paths": ["yolo/coco_cat_524280_raw_480x640x3_uint8_rgb.bin"], + "paths": [ + "yolo/coco_cat_524280_raw_480x640x3_uint8_rgb.bin" + ], "required": true } ], @@ -162,7 +174,9 @@ "file_loads": [ { "address": "0x04A00000", - "paths": ["yolo/coco_giraffes_296969_raw_480x640x3_uint8_rgb.bin"], + "paths": [ + "yolo/coco_giraffes_296969_raw_480x640x3_uint8_rgb.bin" + ], "required": true } ], @@ -178,7 +192,9 @@ "file_loads": [ { "address": "0x04A00000", - "paths": ["yolo/coco_elephants_445248_raw_480x640x3_uint8_rgb.bin"], + "paths": [ + "yolo/coco_elephants_445248_raw_480x640x3_uint8_rgb.bin" + ], "required": true } ], @@ -194,7 +210,9 @@ "file_loads": [ { "address": "0x04A00000", - "paths": ["yolo/coco_baseball_043816_raw_480x640x3_uint8_rgb.bin"], + "paths": [ + "yolo/coco_baseball_043816_raw_480x640x3_uint8_rgb.bin" + ], "required": true } ], @@ -234,17 +252,24 @@ "file_loads": [ { "address": "0x0", - "paths": ["zero2m.bin", "common/zero2m.bin"], + "paths": [ + "zero2m.bin", + "common/zero2m.bin" + ], "required": true }, { "address": "0x2000", - "paths": ["dncnn/dncnn20l64_input.bin"], + "paths": [ + "dncnn/dncnn20l64_input.bin" + ], "required": true }, { "address": "0x14000", - "paths": ["dncnn/dncnn20l64_weights.bin"], + "paths": [ + "dncnn/dncnn20l64_weights.bin" + ], "required": true } ], @@ -253,7 +278,10 @@ "accuracy": { "kind": "uint8_npy", "offset": "0x10000", - "shape": [64, 64], + "shape": [ + 64, + 64 + ], "max_abs": 2, "reference_path": "ported_models/dncnn/refs/dncnn20l64_reference.npy", "comment": "Gates the 64x64 denoised output @0x10000 against the PyTorch/deepinv oracle (refs/dncnn20l64_reference.npy, produced by scripts/gen_dncnn_oracle.py running deepinv.models.DnCNN on the pinned weights). Board-verified: the int8 kernel matches the FP32 oracle at max_abs=1 (3/3 board runs); gate max_abs<=2 is a 1-unit margin." @@ -315,6 +343,9 @@ }, "smolvlm_500m": { "config": "ported_models/llama_cpp_et/benchmarks/smolvlm_500m.json" + }, + "granite4_h_micro": { + "config": "ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json" } } } diff --git a/ported_models/granite4_h_micro/docs/RECIPE.md b/ported_models/granite4_h_micro/docs/RECIPE.md new file mode 100644 index 00000000..367460ca --- /dev/null +++ b/ported_models/granite4_h_micro/docs/RECIPE.md @@ -0,0 +1,60 @@ +# Granite-4.0-H-Micro Porting Recipe + +## Overview + +Adds `ibm-granite/granite-4.0-h-micro` (IBM's fourth-generation Granite, +hybrid attention+Mamba architecture) to the `llama.cpp-et` framework. +Confirmed via local GGUF metadata inspection: `general.architecture = +granitehybrid` -- distinct from both `granite` (dense, already ported +elsewhere on this board) and `granitemoe` (MoE, already ported earlier in +this campaign as `granite_3_1b_a400m`). A third, genuinely different +Granite execution family. + +## Model Reference + +- **Source**: `ibm-granite/granite-4.0-h-micro` (Hugging Face), revision + `d5f01a3ea75f088947be3aae039f4ad52837dfde` +- **License**: Apache-2.0 +- **GGUF source**: `ibm-granite/granite-4.0-h-micro-GGUF` (official), file + `granite-4.0-h-micro-Q8_0.gguf` +- **Quantization**: Q8_0, + `sha256=a009111abf2865b7aad1e66326a6c772cddc29bccd22898f470292068b27bb59` + (verified locally against the downloaded file) +- **Architecture**: `arch = granitehybrid` per GGUF metadata, 506 tensors. + +## Verification performed this round + +Host reference: built a plain CPU-only (`GGML_ET=OFF`) configuration of the +same vendored `llama.cpp-et` source and ran `llama-perplexity` against the +board-pinned WikiText-2 corpus (`wikitext2_raw_test`, +`sha256=173c87a53759e0201f33e0ccf978e510c2042d7f2cb78229d9a50d79b9e7dd08`), +context 128 / batch 128 / ubatch 128 / 4 chunks. The model loads and runs +cleanly, allocating **both** a small transformer KV cache (`llama_kv_cache`, +only 4 layers) **and** a much larger recurrent SSM state cache +(`llama_memory_recurrent`, 40 layers) -- a hybrid weighted heavily toward +recurrent layers (4 attention : 40 recurrent), a different ratio from +`falcon_h1_1_5b`'s hybrid mix earlier in this campaign: + +``` +Final estimate: PPL = 13.2058 +/- 2.78765 +``` + +This is a third data point (after `mamba_1_4b`, `falcon_h1_1_5b`) on +`SSM_CONV`/`SSM_SCAN` working correctly on `ggml-cpu`. + +## Why this port's ET-SoC1 kernel support is a real, open question + +Same caveat as `mamba_1_4b`/`falcon_h1_1_5b`: CPU success doesn't prove +ET-SoC1 support for `SSM_CONV`/`SSM_SCAN`. Three independent hybrid/SSM +models now share this same open question in this campaign. + +## Open items for maintainer review + +- Registered in `artifacts.json`, `ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json`, + and `.github/ci/benchmark_config.json` (port 18142) -- board-testable + now, independent of the model-ports track claim below. +- `ported_models/submissions/model_ports/granite4_h_micro.json` is the + model-ports track claim, pending identity approval. +- No changes to any protected file or the vendored submodule. +- SSM ops confirmed on CPU, NOT live-verified against ET sysemu + specifically. diff --git a/ported_models/granite4_h_micro/oracle/perplexity_oracle.json b/ported_models/granite4_h_micro/oracle/perplexity_oracle.json new file mode 100644 index 00000000..bf13e1f3 --- /dev/null +++ b/ported_models/granite4_h_micro/oracle/perplexity_oracle.json @@ -0,0 +1,25 @@ +{ + "oracle_type": "perplexity_threshold", + "model_artifact": { + "repo": "ibm-granite/granite-4.0-h-micro-GGUF", + "revision": "main", + "filename": "granite-4.0-h-micro-Q8_0.gguf", + "sha256": "a009111abf2865b7aad1e66326a6c772cddc29bccd22898f470292068b27bb59" + }, + "corpus": { + "artifact": "wikitext2_raw_test", + "sha256": "173c87a53759e0201f33e0ccf978e510c2042d7f2cb78229d9a50d79b9e7dd08" + }, + "command": "llama-perplexity --model granite-4.0-h-micro-Q8_0.gguf -f wiki.test.raw -c 128 -b 128 -ub 128 --chunks 4", + "reference_run": { + "final_ppl": 13.2058, + "final_ppl_stderr": 2.78765, + "measured_on": "CPU (ggml-cpu backend, GGML_ET=OFF build)", + "measured_date": "2026-07-25" + }, + "comparison_threshold": { + "metric": "final_ppl", + "max_relative_deviation": 0.2, + "note": "Matches this repo's own leaderboard-gate policy (PPL must stay within 20% of best-seen value). A full-offload ET-SoC1 re-run against this exact command/corpus/artifact should land at final_ppl within [10.56, 15.85] to be considered consistent with this reference run. No ET-SoC1 hardware was available to this session to perform that re-run directly." + } +} diff --git a/ported_models/llama_cpp_et/artifacts.json b/ported_models/llama_cpp_et/artifacts.json index 9c419ae1..74d5524c 100644 --- a/ported_models/llama_cpp_et/artifacts.json +++ b/ported_models/llama_cpp_et/artifacts.json @@ -485,6 +485,23 @@ "sha256": "d1eb8b6b23979205fdf63703ed10f788131a3f812c7b1f72e0119d5d81295150", "size_bytes": 108783360, "note": "SmolVLM 500M vision projector (SigLIP ~93M + MLP). Q8_0 quantized. Must be loaded alongside smolvlm_500m_q8_gguf." + }, + "granite4_h_micro_q8_gguf": { + "kind": "model", + "framework": "llama.cpp-et", + "variant": "granite-4.0-h-micro-Q8_0", + "filename": "granite-4.0-h-micro-Q8_0.gguf", + "env": "GRANITE4_H_MICRO_MODEL_PATH", + "source": { + "type": "huggingface", + "repo": "ibm-granite/granite-4.0-h-micro-GGUF", + "revision": "main", + "filename": "granite-4.0-h-micro-Q8_0.gguf", + "url": "https://huggingface.co/ibm-granite/granite-4.0-h-micro-GGUF/resolve/main/granite-4.0-h-micro-Q8_0.gguf" + }, + "sha256": "a009111abf2865b7aad1e66326a6c772cddc29bccd22898f470292068b27bb59", + "local_cache": "local-artifacts/models/granite-4.0-h-micro-Q8_0.gguf", + "board_path": "/data/models/granite-4.0-h-micro-Q8_0.gguf" } } } diff --git a/ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json b/ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json new file mode 100644 index 00000000..725b58e9 --- /dev/null +++ b/ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json @@ -0,0 +1,52 @@ +{ + "runner": "llama_server", + "board": true, + "framework": { + "name": "llama.cpp-et", + "runner": "llama_server", + "source_artifact": "llama_cpp_source" + }, + "artifacts_file": "../artifacts.json", + "canonical_variant": "granite-4.0-h-micro-Q8_0", + "score": { + "metric": "tokens_per_second", + "label": "Decode tokens/s", + "higher_is_better": true + }, + "llama_server": { + "source_artifact": "llama_cpp_source", + "model_artifact": "granite4_h_micro_q8_gguf", + "server_artifact": "llama_server", + "workdir_artifact": "llama_cpp_build", + "host": "127.0.0.1", + "port": 18142, + "device": "ET", + "gpu_layers": 99, + "ctx_size": 2048, + "batch_size": 256, + "ubatch_size": 128, + "parallel": 1, + "cache_ram_mib": 0, + "ready_timeout_s": 300, + "request_timeout_s": 420, + "flash_attn": false, + "api": "completion", + "prompt": "Repeat this token sequence without commentary: OK OK OK OK OK OK OK OK OK OK", + "max_tokens": 96, + "temperature": 0, + "ignore_eos": true, + "min_completion_tokens": 32, + "perplexity": { + "enabled": true, + "perplexity_artifact": "llama_perplexity", + "corpus_artifact": "wikitext2_raw_test", + "ctx_size": 128, + "batch_size": 128, + "ubatch_size": 128, + "timeout_s": 420, + "min_ppl": 1.0, + "max_ppl": 1000.0, + "chunks": 4 + } + } +} diff --git a/ported_models/submissions/model_ports/granite4_h_micro.json b/ported_models/submissions/model_ports/granite4_h_micro.json new file mode 100644 index 00000000..b866c145 --- /dev/null +++ b/ported_models/submissions/model_ports/granite4_h_micro.json @@ -0,0 +1,16 @@ +{ + "schema_version": 1, + "track": "most_models_ported", + "benchmark_model": "granite4_h_micro", + "identity_id": "granitehybrid", + "source": { + "repo": "ibm-granite/granite-4.0-h-micro", + "revision": "d5f01a3ea75f088947be3aae039f4ad52837dfde", + "license": "apache-2.0" + }, + "implementation_paths": [ + "ported_models/granite4_h_micro" + ], + "benchmark_config": "ported_models/llama_cpp_et/benchmarks/granite4_h_micro.json", + "recipe": "ported_models/granite4_h_micro/docs/RECIPE.md" +}