-
Notifications
You must be signed in to change notification settings - Fork 1
191 lines (172 loc) · 8.03 KB
/
Copy pathsmoke-reference.yml
File metadata and controls
191 lines (172 loc) · 8.03 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
name: Smoke Reference
# Runs the three reference-model smoke tests tagged `@Tag("smoke-reference")`:
# - Qwen3-1.7B Q8_0 GGUF (kllama runner)
# - Gemma-4 E2B SafeTensors (kgemma runner)
# - Gemma-4 E2B GGUF golden-token + timed smoke (kgemma runner, optional input)
# - BERT + LEAF SafeTensors (llm-test-java)
#
# By default each test self-skips when its model artifact is not resolvable
# through the standard env-var / `~/.lmstudio/models/` / `~/.cache/huggingface/hub/`
# fallback chain, so this workflow is **green with empty inputs** — it merely
# proves the wiring compiles and the JUnit filter resolves the smoke tier.
#
# To make a run actually exercise the models, trigger it with the matching
# workflow inputs (URLs / paths). The job downloads each artifact, sets the
# corresponding env var the test reads (see Qwen3ReferenceSmokeTest.kt,
# Gemma4ReferenceSmokeTest.kt, BertLeafReferenceSmokeTest.java), and runs
# the same `./gradlew test -PsmokeReference -PincludeIntegration` invocation
# that's documented in the repo's CHANGELOG and reference-smoke tests.
#
# Trigger pattern is manual (`workflow_dispatch`) — wiring this onto every
# push would silently consume Actions minutes without doing meaningful work
# until artifacts are available. A self-hosted runner with the three
# checkpoints pre-cached on disk is the natural place to flip this to
# `push: branches: [develop]` later.
on:
workflow_dispatch:
inputs:
qwen3_gguf_url:
description: "Direct URL to Qwen3-1.7B-Q8_0.gguf (~1.9 GB). Leave blank to skip the kllama test."
required: false
default: ""
qwen25_gguf_url:
description: "Direct URL to qwen2.5-0.5b-instruct-q8_0.gguf (~0.5 GB). Enables the Qwen2.5 golden-token parity gate (the #352 bias-path regression gate). Leave blank to skip."
required: false
default: ""
apertus_gguf_url:
description: "Direct URL to Apertus-8B-Instruct-2509-Q4_K_S.gguf (~4.6 GB). Enables the Apertus golden-token parity gate (QK-norm + xIELU + ungated FFN). Leave blank to skip."
required: false
default: ""
gemma3n_gguf_url:
description: "Direct URL to gemma-3n-E2B-it-Q4_K_M.gguf (~3.0 GB). Enables the Gemma 3n golden-token parity gate on the DSL lane (AltUp + Laurel + sparsity + PLE + shared KV; needs a large-memory runner: 20g test heap). Leave blank to skip."
required: false
default: ""
gemma4_safetensors_dir_url:
description: "Direct URL to a tar.gz containing the Gemma-4 E2B SafeTensors checkpoint directory. Leave blank to skip the kgemma test."
required: false
default: ""
gemma4_gguf_url:
description: "Direct URL to gemma-4-E2B-it-Q4_K_M.gguf (~3.1 GB). Enables the GGUF golden-token + timed smoke tests (needs a large-memory runner: the probe wants ~24g test heap). Leave blank to skip."
required: false
default: ""
leaf_safetensors_dir_url:
description: "Direct URL to a tar.gz containing the MongoDB mdbr-leaf-ir SafeTensors checkpoint directory. Leave blank to skip the BERT+LEAF test."
required: false
default: ""
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
jobs:
smoke-reference:
runs-on: ubuntu-latest
timeout-minutes: 90
steps:
- name: Checkout
uses: actions/checkout@v7
- name: Copy CI gradle.properties
run: mkdir -p ~/.gradle ; cp .github/ci-gradle.properties ~/.gradle/gradle.properties
- name: Set up JDK 25
uses: actions/setup-java@v6.0.1
with:
distribution: 'zulu'
java-version: 25
- name: Disk space (before downloads)
run: df -h
- name: Stage Qwen3-1.7B Q8_0 GGUF
if: inputs.qwen3_gguf_url != ''
env:
URL: ${{ inputs.qwen3_gguf_url }}
run: |
set -euo pipefail
mkdir -p "$RUNNER_TEMP/models/qwen3"
curl -fsSL "$URL" -o "$RUNNER_TEMP/models/qwen3/Qwen3-1.7B-Q8_0.gguf"
echo "QWEN3_1B7_MODEL_PATH=$RUNNER_TEMP/models/qwen3/Qwen3-1.7B-Q8_0.gguf" >> "$GITHUB_ENV"
# Same file also feeds the Qwen3 golden-token parity gate.
echo "QWEN3_17B_GGUF=$RUNNER_TEMP/models/qwen3/Qwen3-1.7B-Q8_0.gguf" >> "$GITHUB_ENV"
- name: Stage Qwen2.5-0.5B-Instruct Q8_0 GGUF
if: inputs.qwen25_gguf_url != ''
env:
URL: ${{ inputs.qwen25_gguf_url }}
run: |
set -euo pipefail
mkdir -p "$RUNNER_TEMP/models/qwen25"
curl -fsSL "$URL" -o "$RUNNER_TEMP/models/qwen25/qwen2.5-0.5b-instruct-q8_0.gguf"
echo "QWEN25_05B_GGUF=$RUNNER_TEMP/models/qwen25/qwen2.5-0.5b-instruct-q8_0.gguf" >> "$GITHUB_ENV"
- name: Stage Apertus-8B-Instruct Q4_K_S GGUF
if: inputs.apertus_gguf_url != ''
env:
URL: ${{ inputs.apertus_gguf_url }}
run: |
set -euo pipefail
mkdir -p "$RUNNER_TEMP/models/apertus"
curl -fsSL "$URL" -o "$RUNNER_TEMP/models/apertus/Apertus-8B-Instruct-2509-Q4_K_S.gguf"
echo "APERTUS_GGUF_PATH=$RUNNER_TEMP/models/apertus/Apertus-8B-Instruct-2509-Q4_K_S.gguf" >> "$GITHUB_ENV"
# The 8B parity gate needs more than the module's 6g default test heap.
echo "APERTUS_HEAP_ARG=-PapertusTestMaxHeap=12g" >> "$GITHUB_ENV"
- name: Stage Gemma 3n E2B GGUF
if: inputs.gemma3n_gguf_url != ''
env:
URL: ${{ inputs.gemma3n_gguf_url }}
run: |
set -euo pipefail
mkdir -p "$RUNNER_TEMP/models/gemma3n"
curl -fsSL "$URL" -o "$RUNNER_TEMP/models/gemma3n/gemma-3n-E2B-it-Q4_K_M.gguf"
echo "GEMMA3N_E2B_GGUF=$RUNNER_TEMP/models/gemma3n/gemma-3n-E2B-it-Q4_K_M.gguf" >> "$GITHUB_ENV"
# The E2B parity gate self-skips below 16 GB test heap.
echo "GEMMA3N_HEAP_ARG=-PgemmaTestMaxHeap=20g" >> "$GITHUB_ENV"
- name: Stage Gemma-4 E2B SafeTensors
if: inputs.gemma4_safetensors_dir_url != ''
env:
URL: ${{ inputs.gemma4_safetensors_dir_url }}
run: |
set -euo pipefail
mkdir -p "$RUNNER_TEMP/models/gemma4"
curl -fsSL "$URL" | tar -xz -C "$RUNNER_TEMP/models/gemma4"
echo "GEMMA4_E2B_SAFETENSORS_PATH=$RUNNER_TEMP/models/gemma4" >> "$GITHUB_ENV"
- name: Stage Gemma-4 E2B GGUF
if: inputs.gemma4_gguf_url != ''
env:
URL: ${{ inputs.gemma4_gguf_url }}
run: |
set -euo pipefail
mkdir -p "$RUNNER_TEMP/models/gemma4-gguf"
curl -fsSL "$URL" -o "$RUNNER_TEMP/models/gemma4-gguf/gemma-4-E2B-it-Q4_K_M.gguf"
echo "GEMMA4_E2B_GGUF_PATH=$RUNNER_TEMP/models/gemma4-gguf/gemma-4-E2B-it-Q4_K_M.gguf" >> "$GITHUB_ENV"
echo "GEMMA4_PROBE=1" >> "$GITHUB_ENV"
- name: Stage MongoDB LEAF SafeTensors
if: inputs.leaf_safetensors_dir_url != ''
env:
URL: ${{ inputs.leaf_safetensors_dir_url }}
run: |
set -euo pipefail
mkdir -p "$RUNNER_TEMP/models/leaf"
curl -fsSL "$URL" | tar -xz -C "$RUNNER_TEMP/models/leaf"
echo "LEAF_MODEL_DIR=$RUNNER_TEMP/models/leaf" >> "$GITHUB_ENV"
- name: Run smoke-reference tier
env:
GRADLE_OPTS: -Dorg.gradle.jvmargs="-Xmx4g -Dfile.encoding=UTF-8"
run: |
./gradlew --no-daemon --stacktrace \
-Dorg.gradle.caching=true \
-Dorg.gradle.configuration-cache=true \
-PsmokeReference -PincludeIntegration \
${APERTUS_HEAP_ARG:-} \
${GEMMA3N_HEAP_ARG:-} \
test
- name: Disk space (after run)
if: always()
run: df -h || true
- name: Memory info (on failure)
if: failure()
run: |
free -h || true
cat /proc/meminfo | head -n 50 || true
- name: Upload smoke-reference test reports
if: always()
uses: actions/upload-artifact@v7
with:
name: smoke-reference-reports
path: |
**/build/reports/tests/**
**/build/test-results/**
retention-days: 14