Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 17 additions & 0 deletions .github/workflows/smoke-reference.yml
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,10 @@ on:
description: "Direct URL to Apertus-8B-Instruct-2509-Q4_K_S.gguf (~4.6 GB). Enables the Apertus golden-token parity gate (QK-norm + xIELU + ungated FFN). Leave blank to skip."
required: false
default: ""
gemma3n_gguf_url:
description: "Direct URL to gemma-3n-E2B-it-Q4_K_M.gguf (~3.0 GB). Enables the Gemma 3n golden-token parity gate on the DSL lane (AltUp + Laurel + sparsity + PLE + shared KV; needs a large-memory runner: 20g test heap). Leave blank to skip."
required: false
default: ""
gemma4_safetensors_dir_url:
description: "Direct URL to a tar.gz containing the Gemma-4 E2B SafeTensors checkpoint directory. Leave blank to skip the kgemma test."
required: false
Expand Down Expand Up @@ -110,6 +114,18 @@ jobs:
echo "APERTUS_GGUF_PATH=$RUNNER_TEMP/models/apertus/Apertus-8B-Instruct-2509-Q4_K_S.gguf" >> "$GITHUB_ENV"
# The 8B parity gate needs more than the module's 6g default test heap.
echo "APERTUS_HEAP_ARG=-PapertusTestMaxHeap=12g" >> "$GITHUB_ENV"

- name: Stage Gemma 3n E2B GGUF
if: inputs.gemma3n_gguf_url != ''
env:
URL: ${{ inputs.gemma3n_gguf_url }}
run: |
set -euo pipefail
mkdir -p "$RUNNER_TEMP/models/gemma3n"
curl -fsSL "$URL" -o "$RUNNER_TEMP/models/gemma3n/gemma-3n-E2B-it-Q4_K_M.gguf"
echo "GEMMA3N_E2B_GGUF=$RUNNER_TEMP/models/gemma3n/gemma-3n-E2B-it-Q4_K_M.gguf" >> "$GITHUB_ENV"
# The E2B parity gate self-skips below 16 GB test heap.
echo "GEMMA3N_HEAP_ARG=-PgemmaTestMaxHeap=20g" >> "$GITHUB_ENV"
if: inputs.gemma4_safetensors_dir_url != ''
env:
URL: ${{ inputs.gemma4_safetensors_dir_url }}
Expand Down Expand Up @@ -149,6 +165,7 @@ jobs:
-Dorg.gradle.configuration-cache=true \
-PsmokeReference -PincludeIntegration \
${APERTUS_HEAP_ARG:-} \
${GEMMA3N_HEAP_ARG:-} \
test

- name: Disk space (after run)
Expand Down
27 changes: 27 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,33 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added — Gemma 3n runs on the DSL path, parity-gated (#377)

- **`gemma3nNetwork()` + `Gemma3nModel`** — the full Gemma 3n text architecture declared
in the DSL, faithful to HF `modeling_gemma3n.py`: **AltUp** (four parallel hidden
streams with the tanh modality router; `Gemma3nAltUpBlock` per layer,
`Gemma3nAltUpGlobals` for the magnitude-renormed stream init/merge), **Laurel**,
**Gaussian-top-k activation sparsity** on the first ten layers (driven by the GGUF's
precomputed per-layer std multipliers; `-inf` = off), **PLE feeding the non-active
streams** (reusing the gemma-4 lane's `PerLayerEmbedding` — the math is identical),
per-type **shared KV** for the last ten layers, hybrid sliding/global attention with
dual RoPE bases, q/k-norm + parameterless v-norm, attention scale 1.0. All math goes
through `ctx.ops`, so the model is traceable for the StableHLO → IREE mobile path.
- **The hand-rolled `Gemma3nRuntime` was never faithful to real checkpoints**: it loaded
the PLE tensors but never applied them, had no Laurel, ignored the AltUp router, and
its `E2B_DEFAULT` config claimed AltUp/sparsity were E4B-only — the real E2B GGUF has
`altup.num_inputs=4` and first-10-layer sparsity. The GGUF CLI paths (kgemma, unified
skainet-cli) now route gemma3n through the DSL lane; SafeTensors stays on the legacy
runtime until the DSL grows that leg.
- **`Gemma3nGoldenTokenParityTest`** (#346 gate, the last ungated generative family):
full 32-step greedy text equality vs mainline llama.cpp b10621 on
`gemma-3n-E2B-it-Q4_K_M.gguf`, on the exact CLI path — engine loading stays
packed/MAPPED (the PLE table row-dequants on demand). Wired into the smoke-reference
tier (`gemma3n_gguf_url` + 20g heap arg); `smoke-models.json` gains a Gemma3n-E2B row.
Metadata parsing now reads the real llama.cpp GGUF keys (`sliding_window_pattern`
booleans, per-layer `activation_sparsity_scale`, `rope.freq_base` fallback,
`rms_norm_eps`, per-layer `feed_forward_length`).

### Fixed — Qwen tool calling follows the official Qwen3 chat template

- **`QwenChatTemplate` rewritten against the official Qwen3 `chat_template`** (verified
Expand Down
1 change: 1 addition & 0 deletions llm-apps/skainet-cli/build.gradle.kts
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@ dependencies {
implementation(project(":llm-inference:qwen"))
implementation(project(":llm-inference:bitnet"))
implementation(project(":llm-inference:gemma"))
implementation(project(":llm-inference:gemma3n"))
implementation(project(":llm-inference:apertus"))

// SKaiNET core libraries
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -242,26 +242,34 @@ fun main(args: Array<String>) {

val runtime: InferenceRuntime<FP32> = if (modelInfo.family == ModelFamily.GEMMA) {
// ModelFamily.GEMMA claims every gemma* architecture, but this DSL lane serves
// gemma3/gemma4 only (#376): 3n needs the hand-rolled runtime (AltUp/PLE/activation
// sparsity — kgemma CLI, split tracked in #377), and gemma2 has no supported path.
when (modelInfo.architecture) {
"gemma3n" -> error(
"Gemma 3n is not supported by the unified CLI's DSL lane — use the kgemma " +
"CLI (:llm-runtime:kgemma), which carries its hand-rolled runtime (#377).",
)
"gemma2", "gemma" -> error(
"Architecture '${modelInfo.architecture}' has no supported path — the Gemma " +
"lane serves gemma3/gemma4 checkpoints (#376).",
)
}
println("Loading Gemma GGUF model from $modelPath via gemmaNetwork() + OptimizedLLMRuntime (engine loader, keep-packed, mapped)...")
if (cliArgs.contextLength != null) {
println(" --context flag currently ignored on the Gemma path; uses model default capped to 4096.")
// gemma3/gemma4/gemma3n (#376, #377): gemma3n runs its own DSL lane
// (gemma3nNetwork() — AltUp/Laurel/sparsity/PLE, parity-gated vs llama.cpp);
// gemma2 has no supported path.
if (modelInfo.architecture == "gemma3n") {
println("Loading Gemma 3n GGUF model from $modelPath via gemma3nNetwork() + OptimizedLLMRuntime (engine loader, keep-packed, mapped)...")
val model3n = kotlinx.coroutines.runBlocking {
sk.ainet.models.gemma3n.Gemma3nNetworkLoader.fromGguf<FP32, Float>(
ctx,
{ JvmRandomAccessSource.open(modelPath.toString()) },
)
}
OptimizedLLMRuntime(model3n, ctx, OptimizedLLMMode.DIRECT, FP32::class)
} else {
when (modelInfo.architecture) {
"gemma2", "gemma" -> error(
"Architecture '${modelInfo.architecture}' has no supported path — the Gemma " +
"lane serves gemma3/gemma4/gemma3n checkpoints (#376, #377).",
)
}
println("Loading Gemma GGUF model from $modelPath via gemmaNetwork() + OptimizedLLMRuntime (engine loader, keep-packed, mapped)...")
if (cliArgs.contextLength != null) {
println(" --context flag currently ignored on the Gemma path; uses model default capped to 4096.")
}
val model = GemmaNetworkLoader.fromGguf(
randomAccessProvider = { JvmRandomAccessSource.open(modelPath.toString()) }
).load<FP32, Float>(ctx)
OptimizedLLMRuntime(model, ctx, OptimizedLLMMode.DIRECT, FP32::class)
}
val model = GemmaNetworkLoader.fromGguf(
randomAccessProvider = { JvmRandomAccessSource.open(modelPath.toString()) }
).load<FP32, Float>(ctx)
OptimizedLLMRuntime(model, ctx, OptimizedLLMMode.DIRECT, FP32::class)
} else if (modelInfo.family == ModelFamily.APERTUS) {
println("Loading Apertus GGUF model from $modelPath via apertusNetwork() + OptimizedLLMRuntime (engine loader, keep-packed, mapped)...")
if (cliArgs.contextLength != null) {
Expand Down
Loading