-
Notifications
You must be signed in to change notification settings - Fork 0
388 lines (343 loc) · 17.1 KB
/
Copy pathperformance.yml
File metadata and controls
388 lines (343 loc) · 17.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
name: performance-diagnostic
on:
push:
branches: [main]
workflow_dispatch:
permissions:
contents: read
jobs:
four-subject-matrix:
name: ${{ matrix.name }}
strategy:
fail-fast: false
matrix:
include:
- name: macOS 15 ARM64 CPU
os: macos-15
runtime_identifier: osx-arm64
executable_suffix: ''
llama_binary_directory: build/bin
- name: Ubuntu 24.04 x64 CPU
os: ubuntu-24.04
runtime_identifier: linux-x64
executable_suffix: ''
llama_binary_directory: build/bin
- name: Windows Server 2025 x64 CPU
os: windows-2025
runtime_identifier: win-x64
executable_suffix: .exe
llama_binary_directory: build/bin/Release
runs-on: ${{ matrix.os }}
timeout-minutes: 45
env:
DOTNET_NOLOGO: true
DOTNET_SKIP_FIRST_TIME_EXPERIENCE: true
SYNAPSE_MODEL_ROOT: ${{ github.workspace }}/artifacts/models
SYNAPSE_BENCHMARK_RUNNER_LABEL: ${{ matrix.name }}
steps:
- name: Checkout Synapse
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- name: Checkout pinned dotLLM
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
repository: kkokosa/dotLLM
ref: d88040451d7db56e5dfef9d5754ad0955b0f7fe5
path: _external/dotLLM
persist-credentials: false
- name: Checkout pinned llama.cpp
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
repository: ggml-org/llama.cpp
ref: b29c606e28a01b1bc8c1351026a0fa6e616bf6c4
path: _external/llama.cpp
persist-credentials: false
- name: Install pinned .NET SDK
uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6.0.0
with:
global-json-file: global.json
- name: Install pinned Rust toolchain for the native kernel
run: rustup toolchain install 1.98.1 --profile minimal
- name: Build Synapse benchmark and CLI
run: |
dotnet restore Synapse.slnx --locked-mode
dotnet build Synapse.slnx --configuration Release --no-restore
- name: Restore verified model cache
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
path: ${{ env.SYNAPSE_MODEL_ROOT }}
key: synapse-models-${{ runner.os }}-${{ matrix.runtime_identifier }}-${{ hashFiles('models/catalog.json') }}
- name: Fetch pinned Qwen smoke model
shell: bash
run: dotnet run --project src/Synapse.Cli --configuration Release --no-build -- model fetch --id qwen2.5-0.5b-instruct-q8_0 --output "${SYNAPSE_MODEL_ROOT}"
- name: Build pinned dotLLM
run: dotnet build _external/dotLLM/src/DotLLM.Cli/DotLLM.Cli.csproj --configuration Release
- name: Build pinned CPU llama.cpp
shell: bash
run: |
cmake -S _external/llama.cpp -B _external/llama.cpp/build \
-DCMAKE_BUILD_TYPE=Release -DGGML_METAL=OFF \
-DLLAMA_BUILD_TESTS=OFF -DLLAMA_BUILD_SERVER=OFF
cmake --build _external/llama.cpp/build --config Release --parallel 4 \
--target llama-completion
- name: Run isolated four-subject performance matrix
shell: bash
run: |
dotnet experiments/Synapse.ReferenceBenchmarks/bin/Release/net10.0/Synapse.ReferenceBenchmarks.dll matrix \
--model "${SYNAPSE_MODEL_ROOT}/qwen2.5-0.5b-instruct-q8_0/qwen2.5-0.5b-instruct-q8_0.gguf" \
--prompt "The capital of France is" \
--prompt-token-ids "785,6722,315,9625,374" \
--expected-token-ids "12095,13,1084,374,279,7772,3283,304" \
--expected-text " Paris. It is the largest city in" \
--synapse-executable "${GITHUB_WORKSPACE}/src/Synapse.Cli/bin/Release/net10.0/synapse${{ matrix.executable_suffix }}" \
--synapse-backend native \
--dotllm-executable "${GITHUB_WORKSPACE}/_external/dotLLM/src/DotLLM.Cli/bin/Release/net10.0/DotLLM.Cli${{ matrix.executable_suffix }}" \
--dotllm-version d88040451d7db56e5dfef9d5754ad0955b0f7fe5 \
--llamacpp-executable "${GITHUB_WORKSPACE}/_external/llama.cpp/${{ matrix.llama_binary_directory }}/llama-completion${{ matrix.executable_suffix }}" \
--llamacpp-version b29c606e28a01b1bc8c1351026a0fa6e616bf6c4 \
--max-tokens 8 --threads 2 --warmups 3 --measurements 5 \
--output "${RUNNER_TEMP}/synapse-benchmark.json"
- name: Summarize measured performance and workload
shell: bash
run: |
dotnet experiments/Synapse.ReferenceBenchmarks/bin/Release/net10.0/Synapse.ReferenceBenchmarks.dll report \
--input "${RUNNER_TEMP}/synapse-benchmark.json" \
--summary "${GITHUB_STEP_SUMMARY}" --require-quality
- name: Run longer single-request and three-turn CPU diagnostics
shell: bash
run: |
common=(
--model "${SYNAPSE_MODEL_ROOT}/qwen2.5-0.5b-instruct-q8_0/qwen2.5-0.5b-instruct-q8_0.gguf"
--synapse-executable "${GITHUB_WORKSPACE}/src/Synapse.Cli/bin/Release/net10.0/synapse${{ matrix.executable_suffix }}"
--dotllm-executable "${GITHUB_WORKSPACE}/_external/dotLLM/src/DotLLM.Cli/bin/Release/net10.0/DotLLM.Cli${{ matrix.executable_suffix }}"
--dotllm-version d88040451d7db56e5dfef9d5754ad0955b0f7fe5
--llamacpp-executable "${GITHUB_WORKSPACE}/_external/llama.cpp/${{ matrix.llama_binary_directory }}/llama-completion${{ matrix.executable_suffix }}"
--llamacpp-version b29c606e28a01b1bc8c1351026a0fa6e616bf6c4
--threads 2 --warmups 1 --measurements 3
)
runner="experiments/Synapse.ReferenceBenchmarks/bin/Release/net10.0/Synapse.ReferenceBenchmarks.dll"
dotnet "$runner" dialogue "${common[@]}" \
--scenario benchmarks/scenarios/capitals-single-long.json \
--max-tokens 128 --output "${RUNNER_TEMP}/synapse-single.json"
dotnet "$runner" dialogue "${common[@]}" \
--scenario benchmarks/scenarios/capitals-france-us-uk-3-turns.json \
--max-tokens 64 --output "${RUNNER_TEMP}/synapse-dialogue.json"
- name: Summarize longer CPU diagnostics
shell: bash
run: |
runner="experiments/Synapse.ReferenceBenchmarks/bin/Release/net10.0/Synapse.ReferenceBenchmarks.dll"
dotnet "$runner" report-dialogue --input "${RUNNER_TEMP}/synapse-single.json" --summary "${GITHUB_STEP_SUMMARY}"
dotnet "$runner" report-dialogue --input "${RUNNER_TEMP}/synapse-dialogue.json" --summary "${GITHUB_STEP_SUMMARY}"
- name: Preserve raw per-round evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: performance-${{ matrix.runtime_identifier }}
path: ${{ runner.temp }}/synapse-benchmark.json
if-no-files-found: error
retention-days: 30
- name: Preserve longer CPU raw evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: performance-long-${{ matrix.runtime_identifier }}
path: |
${{ runner.temp }}/synapse-single.json
${{ runner.temp }}/synapse-dialogue.json
if-no-files-found: warn
retention-days: 30
mlx-metal:
name: macOS 15 ARM64 MLX Metal (separate cohort)
runs-on: macos-15
timeout-minutes: 45
env:
DOTNET_NOLOGO: true
DOTNET_SKIP_FIRST_TIME_EXPERIENCE: true
SYNAPSE_MODEL_ROOT: ${{ github.workspace }}/artifacts/models
steps:
- name: Checkout Synapse
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- name: Install pinned .NET SDK
uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6.0.0
with:
global-json-file: global.json
- name: Install pinned Rust toolchain for the native kernel
run: rustup toolchain install 1.98.1 --profile minimal
- name: Build C# benchmark runner and model fetcher
run: |
dotnet restore Synapse.slnx --locked-mode
dotnet build Synapse.slnx --configuration Release --no-restore
- name: Restore verified MLX model cache
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
path: ${{ env.SYNAPSE_MODEL_ROOT }}
key: synapse-mlx-model-macos-arm64-${{ hashFiles('models/catalog.json') }}
- name: Fetch pinned MLX 8-bit Qwen model
shell: bash
run: dotnet run --project src/Synapse.Cli --configuration Release --no-build -- model fetch --id qwen2.5-0.5b-instruct-mlx-8bit --output "${SYNAPSE_MODEL_ROOT}"
- name: Download verified prebuilt SwiftLM MLX binary
shell: bash
run: |
mkdir -p "${RUNNER_TEMP}/swiftlm"
curl --fail --location --silent --show-error \
--output "${RUNNER_TEMP}/SwiftLM-b795-macos-arm64.tar.gz" \
https://github.com/SharpAI/SwiftLM/releases/download/b795/SwiftLM-b795-macos-arm64.tar.gz
printf '%s %s\n' \
'2ed6b5539b24c5267931d46ea9973775b7d2a9b5ee2f82109afab60f9603675e' \
"${RUNNER_TEMP}/SwiftLM-b795-macos-arm64.tar.gz" | shasum -a 256 -c -
tar -xzf "${RUNNER_TEMP}/SwiftLM-b795-macos-arm64.tar.gz" -C "${RUNNER_TEMP}/swiftlm"
- name: Run isolated MLX Metal single-request and three-turn diagnostics
shell: bash
run: |
runner="experiments/Synapse.ReferenceBenchmarks/bin/Release/net10.0/Synapse.ReferenceBenchmarks.dll"
common=(
--binary "${RUNNER_TEMP}/swiftlm/SwiftLM" --binary-version b795
--model "${SYNAPSE_MODEL_ROOT}/qwen2.5-0.5b-instruct-mlx-8bit"
--warmups 1 --measurements 3
)
dotnet "$runner" mlx "${common[@]}" --port 15414 \
--scenario benchmarks/scenarios/capitals-single-long.json \
--max-tokens 128 --output "${RUNNER_TEMP}/mlx-single.json"
dotnet "$runner" mlx "${common[@]}" --port 15414 \
--scenario benchmarks/scenarios/capitals-france-us-uk-3-turns.json \
--max-tokens 64 --output "${RUNNER_TEMP}/mlx-dialogue.json"
- name: Summarize MLX Metal diagnostics
shell: bash
run: |
runner="experiments/Synapse.ReferenceBenchmarks/bin/Release/net10.0/Synapse.ReferenceBenchmarks.dll"
dotnet "$runner" report-dialogue --input "${RUNNER_TEMP}/mlx-single.json" --summary "${GITHUB_STEP_SUMMARY}"
dotnet "$runner" report-dialogue --input "${RUNNER_TEMP}/mlx-dialogue.json" --summary "${GITHUB_STEP_SUMMARY}"
- name: Preserve MLX Metal raw evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: performance-mlx-osx-arm64
path: |
${{ runner.temp }}/mlx-single.json
${{ runner.temp }}/mlx-dialogue.json
if-no-files-found: warn
retention-days: 30
foundry-local-plan:
name: Foundry Local plan (models that fit each runner)
runs-on: ubuntu-24.04
timeout-minutes: 15
outputs:
matrix: ${{ steps.plan.outputs.matrix }}
env:
DOTNET_NOLOGO: true
DOTNET_SKIP_FIRST_TIME_EXPERIENCE: true
steps:
- name: Checkout Synapse
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- name: Install pinned .NET SDK
uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6.0.0
with:
global-json-file: global.json
- name: Build the isolated Foundry Local runner
run: |
dotnet restore experiments/Synapse.FoundryLocalBenchmarks --locked-mode
dotnet build experiments/Synapse.FoundryLocalBenchmarks --configuration Release --no-restore
- name: Expand the model set into one job per runner and model
id: plan
shell: bash
run: |
matrix="$(dotnet experiments/Synapse.FoundryLocalBenchmarks/bin/Release/net10.0/Synapse.FoundryLocalBenchmarks.dll plan \
--set benchmarks/model-sets/foundry-local-families.json --summary "${GITHUB_STEP_SUMMARY}")"
echo "matrix=${matrix}" >> "${GITHUB_OUTPUT}"
foundry-local:
name: Foundry Local · ${{ matrix.runner_name }} · ${{ matrix.alias }}
needs: foundry-local-plan
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.foundry-local-plan.outputs.matrix) }}
runs-on: ${{ matrix.runner }}
timeout-minutes: 60
env:
DOTNET_NOLOGO: true
DOTNET_SKIP_FIRST_TIME_EXPERIENCE: true
FOUNDRY_CACHE: ${{ github.workspace }}/artifacts/foundry-local
FOUNDRY_RUNNER: experiments/Synapse.FoundryLocalBenchmarks/bin/Release/net10.0/Synapse.FoundryLocalBenchmarks.dll
steps:
- name: Checkout Synapse
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- name: Install pinned .NET SDK
uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6.0.0
with:
global-json-file: global.json
- name: Build the isolated Foundry Local runner
run: |
dotnet restore experiments/Synapse.FoundryLocalBenchmarks --locked-mode
dotnet build experiments/Synapse.FoundryLocalBenchmarks --configuration Release --no-restore
- name: Download only this job's model
shell: bash
run: dotnet "${FOUNDRY_RUNNER}" fetch --set benchmarks/model-sets/foundry-local-families.json --alias "${{ matrix.alias }}" --cache "${FOUNDRY_CACHE}"
- name: Measure single-request and three-turn scenarios in one fresh process each
shell: bash
run: |
common=(
--set benchmarks/model-sets/foundry-local-families.json
--alias "${{ matrix.alias }}" --cache "${FOUNDRY_CACHE}"
--runner-label "${{ matrix.runner_name }}" --warmups 1 --measurements 3
)
dotnet "${FOUNDRY_RUNNER}" run "${common[@]}" \
--scenario benchmarks/scenarios/capitals-single-long.json \
--max-tokens 128 --output "${RUNNER_TEMP}/foundry-single.json"
dotnet "${FOUNDRY_RUNNER}" run "${common[@]}" \
--scenario benchmarks/scenarios/capitals-france-us-uk-3-turns.json \
--max-tokens 64 --output "${RUNNER_TEMP}/foundry-dialogue.json"
- name: Summarize Foundry Local diagnostics
shell: bash
run: |
dotnet "${FOUNDRY_RUNNER}" report --input "${RUNNER_TEMP}/foundry-single.json" --summary "${GITHUB_STEP_SUMMARY}"
dotnet "${FOUNDRY_RUNNER}" report --input "${RUNNER_TEMP}/foundry-dialogue.json" --summary "${GITHUB_STEP_SUMMARY}"
- name: Preserve Foundry Local raw evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: foundry-local-${{ matrix.runner }}-${{ matrix.alias }}
path: |
${{ runner.temp }}/foundry-single.json
${{ runner.temp }}/foundry-dialogue.json
if-no-files-found: warn
retention-days: 30
combined-report:
name: Combined performance results
if: always()
needs: [four-subject-matrix, mlx-metal, foundry-local-plan, foundry-local]
runs-on: ubuntu-24.04
timeout-minutes: 15
env:
DOTNET_NOLOGO: true
DOTNET_SKIP_FIRST_TIME_EXPERIENCE: true
steps:
- name: Checkout Synapse
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- name: Install pinned .NET SDK
uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6.0.0
with:
global-json-file: global.json
- name: Download this run's raw artifacts
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
continue-on-error: true
with:
path: ${{ runner.temp }}/performance-artifacts
merge-multiple: false
- name: Build C# artifact reporter
run: |
dotnet restore experiments/Synapse.ReferenceBenchmarks/Synapse.ReferenceBenchmarks.csproj --locked-mode
dotnet build experiments/Synapse.ReferenceBenchmarks/Synapse.ReferenceBenchmarks.csproj --configuration Release --no-restore
- name: Assemble the combined results table
shell: bash
run: |
dotnet experiments/Synapse.ReferenceBenchmarks/bin/Release/net10.0/Synapse.ReferenceBenchmarks.dll aggregate \
--artifacts "${RUNNER_TEMP}/performance-artifacts" \
--model-set benchmarks/model-sets/foundry-local-families.json \
--output "${RUNNER_TEMP}/performance-summary.md" \
--summary "${GITHUB_STEP_SUMMARY}"
- name: Preserve combined results table
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: performance-summary
path: ${{ runner.temp }}/performance-summary.md
if-no-files-found: warn
retention-days: 30