diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml
index 3b063a6..af9cc0f 100644
--- a/.github/workflows/pages.yml
+++ b/.github/workflows/pages.yml
@@ -27,7 +27,7 @@ jobs:
- name: Check scoring, imports, and configuration
run: |
python3 -m unittest tests.test_data
- node --test tests/test_data.cjs tests/test_yaml.cjs
+ node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs
- name: Check the browser with public fixtures
run: |
google-chrome --headless --no-sandbox --disable-gpu \
@@ -40,9 +40,10 @@ jobs:
node tests/test_public_browser.mjs || { cat "$RUNNER_TEMP/quickdash-chrome.log"; exit 1; }
- name: Build shared results and the fictional demo
run: |
- python3 -m app.build --results-dir results --configs-dir configs \
+ python3 -m app.build --results-dir results \
--output output/shared > "$RUNNER_TEMP/quickdash-shared.json"
- python3 -m app.build examples/scores.csv --config configs/example.yaml \
+ python3 -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \
+ --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml \
--output output/demo > "$RUNNER_TEMP/quickdash-demo.json"
mkdir -p output/site
cp output/shared/index.html output/site/index.html
diff --git a/README.md b/README.md
index c3ae650..7310b2f 100644
--- a/README.md
+++ b/README.md
@@ -6,16 +6,19 @@ A standalone, offline dashboard for comparing model evaluation scores. Explore c
## Compare models
-1. Choose an **Eval configuration** at the top. The supplied OELLM config defines evals, selected metrics, normalization, category weights, and explicit language assignments.
-2. Select shared models as A and B, or use **Add model CSV** to open your own exports. With one real model, a clearly labelled synthetic comparison is supplied for exploring the interface.
-3. Review **Warnings**, then explore the score and breakdown tabs. To use a temporary YAML config, open **Eval configuration → Load config**.
+1. Select shared models as A and B, or use **Add model CSV** to open your exports. With one real model, a labelled synthetic comparison is supplied for exploring the interface.
+2. Choose a **Weighting profile**. Leave **Eval set** on **Any available** to compare the measurements both models have, or select **flagship-1** to check an expected set. Weights and eval sets are independent. The supplied sets exclude prompted Global PIQA pending scoring validation.
+3. Review **Warnings**, then explore the scores and breakdowns. The global catalogue determines how to interpret each eval: category, scoring field, normalization, and language assignments.
-Files opened in the dashboard stay in your browser; they are not uploaded. Changes last until reload, which restores the published models and settings. **Clear models** removes the loaded models from your session while keeping config choices. **Export config YAML** saves edited scoring settings, but does not save model data or modify the repository.
+**Original** is the startup weighting profile. **Code & math emphasis** gives Code and Math 20% each, with the other category weights adjusted as shown in the [configuration reference](docs/configuration.md#choose-weights-and-expected-coverage).
+
+Files opened here stay in your browser; they are not uploaded. Changes last until reload. **Clear models** removes the loaded models while keeping settings. Under **Eval configuration**, load or export the catalogue, weights, and eval set as separate YAML files. Weight exports include your edits and active score calculation.
## Share results and scoring configs
- Add public CSV exports to [results/](results/README.md) to offer their models in the shared dashboard.
-- Add YAML scoring configs to [configs/](configs/README.md) to offer them in the configuration selector. Each config needs a unique `name`. [configs/default.txt](configs/default.txt) names the one selected on startup, currently `oellm.yaml`.
+- Add interpretation rules to [configs/catalogue.yaml](configs/catalogue.yaml).
+- Add weighting profiles to [configs/weights/](configs/weights/), or optional named eval sets to [configs/sets/](configs/sets/). Each directory has a `default.txt` choosing its startup selection. See [contributing configs](configs/README.md).
Use a pull request or GitHub’s **Add file → Upload files**. Changes on `main` trigger tests and a GitHub Pages rebuild; pull requests are checked without publishing. Invalid inputs stop the update and leave the last successful site online. The repository and dashboard are public, so use browser imports for private comparisons.
@@ -28,7 +31,8 @@ Building requires Python 3.8+ and Node.js 18+. The YAML parser is [bundled with
```sh
git clone https://github.com/OpenEuroLLM/quickdash.git
cd quickdash
-python3 -m app.build examples/scores.csv --config configs/example.yaml --output output/example
+python3 -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \
+ --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml --output output/example
open output/example/index.html # macOS; elsewhere, open it in your browser
```
@@ -47,7 +51,7 @@ For private exports, put your CSV in the ignored `data/` directory:
```sh
mkdir -p data
# Copy your export to data/evals.csv, then:
-python3 -m app.build data/evals.csv --config configs/oellm.yaml --output output/private
+python3 -m app.build data/evals.csv --output output/private
```
The builder replaces the files it generates in the chosen output directory, including audits and score summaries. Use separate output directories to keep builds. `data/` and `output/` are ignored by Git. CSV columns and configuration rules are described in the [configuration reference](docs/configuration.md).
@@ -61,7 +65,7 @@ The builder replaces the files it generates in the chosen output directory, incl
- **Eval configuration:** inspect every task's category, languages, selected field, raw alternate fields, normalization, and source notes. Load or export YAML here.
- **Warnings:** review missing coverage, scoring inconsistencies, language fallbacks, sample-count issues, and caveats stored in the config.
-Filters affect inspection views, while composite scores and contribution weights use the models' full shared coverage. Unmatched measurements are excluded from both compared scores with warnings. Malformed input and invalid selected scores are rejected; failed imports preserve the active dashboard. See the [data-handling policy](docs/configuration.md#data-validation-and-failure-behavior).
+Filters affect inspection views, while composite scores and contribution weights use shared coverage within the selected eval set. A named set with missing requirements is labelled incomplete; it uses the shared subset with redistributed weights. Unmatched measurements are excluded from both compared scores with warnings. Malformed input and invalid selected scores are rejected; failed imports preserve the active dashboard. See the [data-handling policy](docs/configuration.md#data-validation-and-failure-behavior).
Language/category breakdowns show descriptive raw averages. Weighted scores use configured normalization, whose baselines and limitations are visible per eval. Shared numerical scales do not establish comparable difficulty across benchmarks. Unknown/mixed-language scores use the documented English fallback for balancing; this does not change their language labels.
@@ -71,7 +75,7 @@ Application code lives in `app/`, tests in `tests/`, and contributor documentati
```sh
python3 -m unittest tests.test_data
-node --test tests/test_data.cjs tests/test_yaml.cjs
+node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs
```
See [development and publishing](docs/development.md) for the source layout, browser tests, and GitHub Pages workflow.
diff --git a/app/app.js b/app/app.js
index 357b603..ba4ef4a 100644
--- a/app/app.js
+++ b/app/app.js
@@ -3,7 +3,8 @@ const key = r => JSON.stringify(['task','metric','filter','n_shot','harness','ba
const avg = xs => xs.length ? xs.reduce((a,b)=>a+b,0)/xs.length : null;
const fmt = (x,digits=2) => x === null || !Number.isFinite(x) ? '—' : x.toFixed(digits);
const esc = x => String(x??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c]));
-const {parseCSV,parseConfig,serializeConfig,validateConfig,normalizeScore,taskLanguage,auditRows,matchTask,demoModel}=typeof module!=='undefined'?require('./eval_config.js'):EvalConfig;
+const {parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel}=typeof module!=='undefined'?require('./eval_config.js'):EvalConfig;
+const {parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveConfig,inSuite,suiteCoverage}=typeof module!=='undefined'?require('./suite_config.js'):SuiteConfig;
function selectRows(rows,scheme){const selected=auditRows(rows,scheme).filter(r=>r.selected);if(!selected.length)throw Error('No selected measurements');return selected;}
function buildCatalogue(rows,scheme){
return scheme.evals.map(f=>{const tasks=new Map();for(const r of rows.filter(r=>r.eval===f.name)){if(!tasks.has(r.task))tasks.set(r.task,[]);tasks.get(r.task).push(r);}return {...f,tasks:[...tasks].map(([name,rows])=>({name,rows})).sort((a,b)=>a.name.localeCompare(b.name))};}).filter(f=>f.tasks.length);
@@ -142,14 +143,17 @@ function protocolWarning(rows,evalConfig,config,model){
});
return {type:'Inconsistent scoring settings',name:evalConfig.name,eval:evalConfig.name,model,detail:'Selected variants use different settings: '+variants.map(v=>v.settings+' ('+v.tasks.length+' tasks; '+v.tasks.slice(0,2).join(', ')+(v.tasks.length>2?', …':'')+')').join(' versus ')+'. These variants are still included in the aggregate. Review the protocol before comparing languages.',variants};
}
-function collectWarnings(audits,config,aggregate='standard',englishWeights=config.english_weights||{}){
- const warnings=config.evals.filter(e=>e.warning).map(e=>({type:'Config caveat',name:e.name,eval:e.name,model:'Config · all models',detail:e.warning}));
+function collectWarnings(audits,config,aggregate='standard',englishWeights=config.english_weights||{},suite={mode:'available'},displayedRows=null){
+ const displayed=new Set((displayedRows??[...audits.values()].flat().filter(r=>r.selected&&inSuite(r,suite))).map(r=>r.eval));
+ const warnings=config.evals.filter(e=>e.warning&&displayed.has(e.name)).map(e=>({type:'Config caveat',name:e.name,eval:e.name,model:'Selected comparison',detail:e.warning}));
for(const [model,rows] of audits){
if(aggregate!=='standard')for(const c of totals(rows.filter(r=>r.selected),config,config.weights,aggregate,englishWeights).categories)if(c.englishShare!==0&&c.issue)warnings.push({type:'English split unavailable',name:c.name,model,detail:c.issue});
for(const task of [...new Set(rows.filter(r=>!r.eval).map(r=>r.task))].sort())warnings.push({type:'No config',name:task,model,detail:'Excluded from scoring. Add an eval match to the YAML config.'});
for(const e of config.evals){
- const matching=rows.filter(r=>r.eval===e.name);
- if(!matching.length){warnings.push({type:'No eval data',name:e.name,model,detail:'This configured eval has no matching task in the loaded model. Exclude it from comparisons with this model and redistribute its weight among the available evals.'});continue;}
+ const all=rows.filter(r=>r.eval===e.name),outside=all.filter(r=>(!e.select||matchTask(e.select,r.task))&&!inSuite(r,suite));
+ if(outside.length)warnings.push({type:'Not used',name:e.name,eval:e.name,model,detail:'Eval data is present but not selected by '+suite.name+'. These results are excluded from the calculation; inspect them in Eval configuration.',variants:[{settings:'Not used by the selected eval set',tasks:[...new Set(outside.map(r=>r.task+' · '+r.n_shot+' shots'))]}]});
+ const matching=all.filter(r=>inSuite(r,suite));
+ if(!matching.length)continue;
const selected=matching.filter(r=>r.selected),unknown=[...new Set(selected.filter(r=>taskLanguage(r.task,config).status==='unknown').map(r=>r.task))];
if(unknown.length)warnings.push({type:'Unknown language',name:e.name,eval:e.name,model,detail:'No explicit language assignment for '+unknown.length+' selected task(s). Included in scoring; both English-balance modes use the English fallback. Language views retain Unknown.',variants:[{settings:'Add an explicit language assignment in YAML',tasks:unknown}]});
const badSamples=selected.filter(r=>r.n_samples!==undefined&&r.n_samples!==null&&String(r.n_samples)!==''&&sampleCount(r)===null);
@@ -171,12 +175,12 @@ if(typeof module!=='undefined')module.exports={parseCSV,auditRows,selectRows,bui
if(typeof document!=='undefined')start();
function start(){
- const $=id=>document.getElementById(id);let scheme=DATA.scheme,weights={...scheme.weights},englishWeights={...scheme.english_weights};
- const presets=DATA.configurations||[{file:DATA.config_file,config:DATA.scheme}];let activePreset='0';
+ const $=id=>document.getElementById(id);let catalogue=DATA.catalogue,suite=DATA.suite,profile=DATA.profile,scheme=DATA.scheme,weights={...scheme.weights},englishWeights={...scheme.english_weights};
+ const suites=DATA.suites,profiles=DATA.profiles;let activeSuite='0',activeProfile='0';
let catalogueGroups=new Map();
let models=new Map(),sourceAudits=new Map(),metadata=new Map(DATA.metadata.map(r=>[r.task,r]));
- for(const m of DATA.models){const audit=auditRows(DATA.rows.filter(r=>r.checkpoint===m.model),scheme);sourceAudits.set(m.model,audit);models.set(m.model,audit.filter(r=>r.selected));}
- if(models.size)models.set(demoModel,synthetic([...models.values()][0],scheme));
+ for(const m of DATA.models){const audit=auditRows(DATA.rows.filter(r=>r.checkpoint===m.model),catalogue);sourceAudits.set(m.model,audit);models.set(m.model,audit.filter(r=>r.selected));}
+ if(models.size)models.set(demoModel,synthetic([...models.values()][0],catalogue));
const state={aggregate:scheme.aggregate||'standard',view:'score',scoreCategory:Object.keys(weights)[0],group:'eval',expandedComparisons:new Set(),measure:'raw',sort:'descending',sortBy:'delta',languageSort:'label',languageOrder:'ascending'};
const languages=r=>languageRoles(r,metadata).map(m=>m.language);
const td=x=>'
'+esc(x)+'
';
@@ -187,12 +191,23 @@ function start(){
const options=(entries,value)=>entries.map(([k,v])=>'').join('');
const control=(id,label,entries,value)=>'';
function modelOptions(preferredB){const previousB=$('modelB').value;for(const id of ['modelA','modelB']){const previous=$(id).value;$(id).innerHTML=options([...models.keys()].map(k=>[k,k]),previous);if(models.has(previous))$(id).value=previous;$(id).disabled=!models.size;}$('modelB').value=preferredB??(models.has(previousB)?previousB:[...sourceAudits.keys()][1]??(models.has(demoModel)?demoModel:''));$('swap').disabled=!models.size;$('clearModels').disabled=!models.size;}
- function configOptions(){const entries=presets.map((p,i)=>[String(i),p.config.name]);if(activePreset==='custom')entries.push(['custom','Uploaded: '+scheme.name]);$('configPreset').innerHTML=options(entries,activePreset);}
+ function configOptions(){
+ for(const [id,presets,active,current] of [['suitePreset',suites,activeSuite,suite],['weightPreset',profiles,activeProfile,profile]]){
+ const entries=presets.map((p,i)=>[String(i),p.config.name]);if(active==='custom')entries.push(['custom','Uploaded: '+current.name]);$(id).innerHTML=options(entries,active);
+ }
+ }
function languageOptions(){const previous=$('language').value;const all=[...new Set([...sourceAudits.values()].flat().flatMap(languages))].sort();$('language').innerHTML=options([['','All languages'],...all.map(k=>[k,languageLabel(k)])],previous);}
function filters(r){const q=$('search').value.trim().toLowerCase();return (!$('category').value||r.category===$('category').value)&&(!$('eval').value||r.eval===$('eval').value)&&matchesLanguage(r,metadata,$('language').value,$('direction').value)&&(!q||(r.task+' '+r.eval+' '+r.metric).toLowerCase().includes(q));}
- function selected(){const coverage=comparisonCoverage(models.get($('modelA').value)||[],models.get($('modelB').value)||[],scheme);return {...coverage,shown:coverage.pairs.filter(filters)};}
+ function selected(){
+ const scope=suiteCoverage(models.get($('modelA').value)||[],models.get($('modelB').value)||[],suite);
+ const coverage=comparisonCoverage(scope.a,scope.b,scheme);
+ return {...coverage,scope,shown:coverage.pairs.filter(filters)};
+ }
function activeWarnings(){
- const coverage=selected(),warnings=collectWarnings(sourceAudits,scheme).concat(coverage.warnings.map(w=>({...w,model:'A: '+$('modelA').value+' · B: '+$('modelB').value})));
+ const coverage=selected(),chosen=new Map([...sourceAudits].filter(([name])=>[$('modelA').value,$('modelB').value].includes(name)));
+ const warnings=collectWarnings(chosen,catalogue,'standard',{},suite,coverage.a).concat(coverage.warnings.map(w=>({...w,model:'A: '+$('modelA').value+' · B: '+$('modelB').value})));
+ if(models.size)warnings.push(...coverage.scope.warnings.map(w=>({...w,model:w.model+': '+$(w.model==='A'?'modelA':'modelB').value})));
+ for(const category of new Set(coverage.a.map(r=>r.category)))if(!Object.hasOwn(profile.weights,category)&&weights[category]===0)warnings.push({type:'No category weight',name:category,model:'Selected comparison',detail:'This category is not in the weighting profile and contributes zero. Add it to the profile to include it in the weighted score.'});
if(state.aggregate!=='standard')for(const c of totals(coverage.a,scheme,weights,state.aggregate,englishWeights,metadata).categories)if(c.issue)warnings.push({type:'English split unavailable',name:c.name,model:'Selected comparison',detail:c.issue});
return warnings;
}
@@ -206,11 +221,11 @@ function start(){
return fmt(c.englishShare*100,0)+'% English · '+fmt((1-c.englishShare)*100,0)+'% other';
}
function renderScore(a,b,ta,tb,valid){
- if(!models.size)return '
Compare your evaluation results
Choose an eval configuration above, then use Add model CSV to open your results. Add a second model to compare training methods.
Active config: '+esc(scheme.name)+'. To use your own YAML, open and choose Load config.
CSV and YAML files opened here stay in your browser; they are not uploaded. Reloading restores the published models and settings.
';
+ if(!models.size)return '
Compare your evaluation results
Use Add model CSV to open your results. Add a second model to compare training methods.
Active config: '+esc(scheme.name)+'. To use your own YAML, open and choose Load catalogue.
CSV and YAML files opened here stay in your browser; they are not uploaded. Reloading restores the published models and settings.
';
let html='
Weighted score
Follow selected variants through eval means and category weights.
Normalize variant scores to 0–100→'+(state.aggregate==='english_eval'?'Balance languages within each eval':'Mean within each eval')+'→'+(state.aggregate==='english_category'?'Balance languages across each category':'Average evals equally within each category')+'→Apply category weights
'+esc(c.name)+(state.aggregate!=='standard'?''+esc(state.aggregate==='english_eval'?(c.englishShare?fmt(c.englishShare*100,0)+'% English within each eval':'Original · split off'):englishShareLabel(c))+'':'')+(c.excluded?'Excluded · no shared scores':c.excludedEvals.length?''+c.excludedEvals.length+' eval(s) excluded':'')+(c.issue?''+esc(c.issue)+'':'')+'
';}),[1,2,3,4,5,6]);
- html+='Adjust category weights and English shares
English shares apply only when either English-balance option is selected in Score calculation above. '+(state.aggregate!=='standard'?'They are active now: '+(state.aggregate==='english_eval'?'inside each eval':'across each category')+'.':'They are saved but inactive while Original weighted score is selected.')+'
Category weights sum to 1. English share gives English and other languages 0.5 each at a 50/50 setting. It is applied inside each eval or across the category, according to the selected calculation. A group with only one language side keeps its full weight automatically. Set 0 to disable the split.
Total category weight: '+fmt(Object.values(weights).reduce((s,w)=>s+w,0),6)+' · must equal 1.
';
+ html+='Adjust category weights and English shares
English shares apply only when either English-balance option is selected in Score calculation above. '+(state.aggregate!=='standard'?'They are active now: '+(state.aggregate==='english_eval'?'inside each eval':'across each category')+'.':'They are saved but inactive while Original weighted score is selected.')+'
Category weights sum to 1. English share gives English and other languages 0.5 each at a 50/50 setting. It is applied inside each eval or across the category, according to the selected calculation. A group with only one language side keeps its full weight automatically. Set 0 to disable the split.
';
@@ -274,7 +289,7 @@ function start(){
const explanation=n.min===0?(n.basis==='unresolved'?'Baseline unresolved; no chance correction is currently applied.':'No chance correction is applied. Zero is a configured floor, not a measured random-model score.'):'Chance baseline used for this eval.';
return '
No warnings. All loaded tasks have an eval config, and every configured eval has a selected score.
');}
+ function renderWarnings(){const warnings=activeWarnings();return '
Warnings
Coverage for the selected comparison, scoring consistency, and config caveats. A named eval set checks required measurements; unused global catalogue rules are allowed.
No warnings. The selected comparison has no detected data or configuration issues.
');}
function scoringOptions(e,rows){
const used=rows.filter(r=>r.selected),values=(rr,key)=>[...new Set(rr.map(r=>String(r[key])))].sort((a,b)=>key==='n_shot'?Number(a)-Number(b):a.localeCompare(b));
const shots=values(used,'n_shot'),otherMetrics=values(rows,'metric').filter(m=>m!==e.metric),otherShots=values(rows,'n_shot').filter(n=>!shots.includes(n)),otherFilters=values(rows,'filter').filter(f=>f!==e.filter);
@@ -287,7 +302,7 @@ function start(){
const m=metadata.get(t.name)||{};
const warning=missing.filter(w=>w.name===t.name).map(w=>'
'+esc(w.model+': '+w.detail)+'
').join('');
const info='
'+esc(m.provenance||'No language assignment available.')+(m.evidence?' Source':'')+'
'+esc(f.category)+' · language assignment: '+esc(m.status||'unknown')+'. Selected metrics use the eval’s score calculation above.
';
- return warning+info+'
English-balance group: '+esc(englishAssignment({task:t.name},metadata))+'. Used in both English-balance modes when this category’s English share is non-zero.
Raw source scores are shown for every metric. The 0–100 columns apply to selected scores only.
English-balance group: '+esc(englishAssignment({task:t.name},metadata))+'. Used in both English-balance modes when this category’s English share is non-zero.
Raw source scores are shown for every metric. The 0–100 columns apply to selected scores only.
Load a YAML config to change eval matching, categories, scoring fields, normalization, and explicit language assignments. Export preserves category weights, English shares, and the active aggregate.
Normalization: each eval lists its baseline, formula, and supporting sources below. Chance correction affects the weighted score; raw scores stay unchanged. acc_norm is length-normalized answer scoring, not chance correction.
English balancing: unknown or mixed-language scores (including Language ID) count as English for weighting. Known non-English pools, such as Croatian/Serbian, count as other languages. Language labels stay unchanged; each variant’s scoring details show its weighting group.
Active config: '+esc(scheme.name)+' · '+scheme.languages.reduce((n,g)=>n+g.tasks.length,0)+' explicit task language assignments. Changes last until reload; use --config when building for permanent defaults.
';
+ const configControls='Global eval catalogue
The catalogue defines matching, categories, scoring fields, normalization, and language assignments for every known eval. It does not require models to run all those evals.
Normalization: each eval lists its baseline, formula, and sources below. Chance correction affects weighted scores; raw scores stay unchanged. acc_norm is length-normalized answer scoring, not chance correction.
Active catalogue: '+esc(catalogue.name)+' · '+catalogue.languages.reduce((n,g)=>n+g.tasks.length,0)+' explicit task language assignments.
Weighting profile and optional eval set
Weights apply independently of the eval set. Any available compares shared recognized measurements and warns on differences. A named set also warns about missing requirements and excludes extra measurements. An incomplete named-set score uses the shared subset with redistributed weights.
Active eval set: '+esc(suite.name)+(suite.exclude?.length?' · Explicitly excluded: '+suite.exclude.map(esc).join(', '):'')+'.
All imports are temporary and stay in your browser. Exports save the three inputs separately. Weight exports include your edits and active score calculation.
';
html=html.replace('
',configControls+'
');
return html;
}
function render(){
const weightOpen=$('weightEditor')?.open,englishOpen=$('englishComponents')?.open;
- const {a,b,pairs,shown,onlyA,onlyB}=selected(),ta=totals(a,scheme,weights,state.aggregate,englishWeights,metadata),tb=totals(b,scheme,weights,state.aggregate,englishWeights,metadata),valid=sameCoverage(a,b)&&ta.score!==null&&tb.score!==null;
+ const {a,b,pairs,shown,onlyA,onlyB,scope}=selected(),ta=totals(a,scheme,weights,state.aggregate,englishWeights,metadata),tb=totals(b,scheme,weights,state.aggregate,englishWeights,metadata),valid=sameCoverage(a,b)&&ta.score!==null&&tb.score!==null;
const demo=[$('modelA').value,$('modelB').value].some(n=>n===demoModel);
configOptions();$('cards').hidden=!models.size;document.querySelector('.aggregate-controls').hidden=!models.size;
$('demo').textContent=!models.size?'Ready for your eval results. Load a CSV to begin.':demo?'Demo comparison: one model has synthetic scores. Replace it with a real eval CSV to compare training methods.':'Real-model comparison · Check evaluation settings and training budgets before drawing a conclusion.';
document.querySelectorAll('[data-aggregate]').forEach(button=>button.setAttribute('aria-pressed',String(button.dataset.aggregate===state.aggregate)));
$('aggregateNote').textContent=state.aggregate!=='standard'?(state.aggregate==='english_eval'?'Balance English and other languages inside each eval, then average evals equally within categories.':'Balance English and other languages across each category. Evals with non-English coverage can receive more weight.')+' Shares are set in the weight editor. Unknown or mixed-language scores count as English for weighting; known non-English pools count as other. Raw views stay unchanged.':'Original aggregate: average variants within each eval, then evals within each category.';
- $('cards').innerHTML=[[$('modelA').value,ta.score,'A · '+(state.aggregate==='english_eval'?'English balance per eval':state.aggregate==='english_category'?'English balance per category':'original weighted score')],[$('modelB').value,tb.score,'B · '+(state.aggregate==='english_eval'?'English balance per eval':state.aggregate==='english_category'?'English balance per category':'original weighted score')],['A − B',valid?ta.score-tb.score:null,'Weighted difference']].map(([name,value,label])=>'
'+esc(name)+''+fmt(value)+''+label+'
').join('');
- $('coverage').textContent=!models.size?'No models loaded · choose a config, then add your CSV.':pairs.length+' matched variants · excluded: '+onlyA.length+' from A, '+onlyB.length+' from B · weights use shared data only'+(valid?'':' · Score unavailable: check weights and language assignments.');
+ $('cards').innerHTML=[[$('modelA').value,ta.score,'A · '+(state.aggregate==='english_eval'?'English balance per eval':state.aggregate==='english_category'?'English balance per category':'original weighted score')],[$('modelB').value,tb.score,'B · '+(state.aggregate==='english_eval'?'English balance per eval':state.aggregate==='english_category'?'English balance per category':'original weighted score')],['A − B',valid?ta.score-tb.score:null,'Weighted difference'+(suite.mode==='fixed'&&!scope.complete?' · incomplete set':'')]].map(([name,value,label])=>'
'+esc(name)+''+fmt(value)+''+label+'
').join('');
+ $('coverage').classList.toggle('notice',models.size>0&&suite.mode==='fixed'&&!scope.complete);
+ $('coverage').textContent=!models.size?'No models loaded · add your CSV.':(suite.mode==='fixed'?suite.name+' · '+(scope.complete?'Complete':'INCOMPLETE')+' · '+scope.sharedRequired+'/'+scope.required+' requirements shared (A '+scope.presentA+', B '+scope.presentB+') · '+(scope.extrasA+scope.extrasB)+' extra measurements excluded · ':suite.name+' · ')+pairs.length+' matched variants · excluded: '+onlyA.length+' from A, '+onlyB.length+' from B · weights use shared data only'+(valid?'':' · Score unavailable: check weights and language assignments.');
$('filters').hidden=['score','warnings'].includes(state.view);
$('filterStatus').textContent=shown.length+' of '+pairs.length+' matched variants · filters do not change the full weighted score';
const warnings=activeWarnings();$('warningCount').textContent=warnings.length;$('warningCount').classList.toggle('has-warnings',warnings.length>0);
@@ -334,22 +350,36 @@ function start(){
markHorizontalScroll();
if(!shown.length&&['categories','languages','comparisons'].includes(state.view))$('view').insertAdjacentHTML('beforeend','
No matched variants pass these filters.
');
}
- function importConfig(config,preset='custom'){
- validateConfig(config);
+ function refreshConfig(){state.scoreCategory=Object.keys(weights)[0];filterOptions();clearFilters();languageOptions();}
+ function importCatalogue(config){
+ validateCatalogue(config);const nextScheme=resolveConfig(config,suite,profile);
const audits=new Map(),nextModels=new Map(),nextMetadata=new Map();
- for(const [name,rows] of sourceAudits){const audit=auditRows(rows,config),selected=audit.filter(r=>r.selected);audits.set(name,audit);nextModels.set(name,selected);for(const row of audit)if(!nextMetadata.has(row.task))nextMetadata.set(row.task,taskLanguage(row.task,config));}
+ for(const [name,rows] of sourceAudits){const audit=auditRows(rows,config);audits.set(name,audit);nextModels.set(name,audit.filter(r=>r.selected));for(const row of audit)if(!nextMetadata.has(row.task))nextMetadata.set(row.task,taskLanguage(row.task,config));}
if(nextModels.size)nextModels.set(demoModel,synthetic([...nextModels.values()][0],config));
- activePreset=preset;scheme=config;weights={...config.weights};englishWeights={...config.english_weights};state.aggregate=config.aggregate||'standard';sourceAudits=audits;models=nextModels;metadata=nextMetadata;
- state.scoreCategory=Object.keys(weights)[0];filterOptions();clearFilters();languageOptions();
+ catalogue=config;scheme=nextScheme;sourceAudits=audits;models=nextModels;metadata=nextMetadata;
+ weights={...nextScheme.weights,...weights};refreshConfig();
+ }
+ function importSuite(config,preset='custom'){
+ const next=resolveConfig(catalogue,config,profile);suite=config;scheme=next;activeSuite=preset;
+ // Changing membership preserves the user's weighting choices and calculation.
+ weights={...next.weights,...weights};refreshConfig();
+ }
+ function importWeights(config,preset='custom'){
+ const next=resolveConfig(catalogue,suite,config);profile=config;scheme=next;activeProfile=preset;
+ weights={...next.weights};englishWeights={...next.english_weights};state.aggregate=next.aggregate;refreshConfig();
}
- function exportConfig(){const config=validateConfig({...scheme,weights:{...weights},english_weights:{...englishWeights},aggregate:state.aggregate});const url=URL.createObjectURL(new Blob([serializeConfig(config)],{type:'application/yaml'})),a=document.createElement('a');a.href=url;a.download='eval-config.yaml';a.click();setTimeout(()=>URL.revokeObjectURL(url),1000);}
+ function exportFile(content,name){const url=URL.createObjectURL(new Blob([content],{type:'application/yaml'})),a=document.createElement('a');a.href=url;a.download=name;a.click();setTimeout(()=>URL.revokeObjectURL(url),1000);}
+ function exportConfig(){exportFile(serializeCatalogue(catalogue),'catalogue.yaml');}
+ function exportWeights(){exportFile(serializeWeightProfile({...profile,weights:{...weights},english_weights:{...englishWeights},aggregate:state.aggregate}),'weights.yaml');}
$('view').onchange=async e=>{try{
const map={compareGroup:'group',compareMeasure:'measure',compareSort:'sort'};
if(map[e.target.id]){state[map[e.target.id]]=e.target.value;if(e.target.id==='compareMeasure'||e.target.id==='compareSort'&&['absolute','name'].includes(e.target.value))state.sortBy=e.target.id==='compareSort'&&e.target.value==='name'?'label':'delta';}
else if(e.target.id==='scoreCategory')state.scoreCategory=e.target.value;
else if(e.target.dataset.englishWeight){englishWeights[e.target.dataset.englishWeight]=e.target.value===''?NaN:Number(e.target.value);}
else if(e.target.dataset.weight){weights[e.target.dataset.weight]=e.target.value===''?NaN:Number(e.target.value);}
- else if(e.target.id==='configFile'){const file=e.target.files[0];if(file)importConfig(parseConfig(await file.text()));}
+ else if(e.target.id==='configFile'){const file=e.target.files[0];if(file)importCatalogue(parseCatalogue(await file.text()));}
+ else if(e.target.id==='weightsFile'){const file=e.target.files[0];if(file)importWeights(parseWeightProfile(await file.text()));}
+ else if(e.target.id==='suiteFile'){const file=e.target.files[0];if(file)importSuite(parseSuite(await file.text()));}
$('error').textContent='';render();
}catch(err){$('error').textContent=err.message;}};
function markHorizontalScroll(){
@@ -394,28 +424,28 @@ function start(){
const replacement=$('view').querySelector('[data-sort-column="'+d.sortColumn+'"]');replacement.closest('.table-scroll').scrollLeft=position.left;window.scrollTo(position.x,position.y);replacement.focus({preventScroll:true});
}
else if(d.expandEval)toggleComparison(d.expandEval,e.target);
- else if(e.target.id==='resetWeights'){Object.assign(weights,scheme.weights);englishWeights={...scheme.english_weights};render();}
- else if(e.target.id==='exportConfig'){try{exportConfig();$('error').textContent='';}catch(err){$('error').textContent=err.message;}}
+ else if(e.target.id==='resetWeights'){weights={...scheme.weights};englishWeights={...scheme.english_weights};render();}
+ else if(['exportConfig','exportWeights','exportSuite'].includes(e.target.id)){try{if(e.target.id==='exportConfig')exportConfig();else if(e.target.id==='exportWeights')exportWeights();else exportFile(serializeSuite(suite),'eval-set.yaml');$('error').textContent='';}catch(err){$('error').textContent=err.message;}}
};
document.querySelectorAll('[data-view]').forEach(e=>e.onclick=()=>{state.view=e.dataset.view;render();});
for(const id of ['modelA','modelB','category','eval','language','direction'])$(id).onchange=render;
document.querySelectorAll('[data-aggregate]').forEach(button=>button.onclick=()=>{state.aggregate=button.dataset.aggregate;render();});
$('search').oninput=render;$('clear').onclick=()=>{clearFilters();render();};
$('swap').onclick=()=>{const value=$('modelA').value;$('modelA').value=$('modelB').value;$('modelB').value=value;render();};
- $('configPreset').onchange=()=>{const chosen=$('configPreset').value;if(chosen==='custom')return;try{importConfig(structuredClone(presets[Number(chosen)].config),chosen);$('error').textContent='';render();}catch(err){$('error').textContent=err.message;$('configPreset').value=activePreset;}};
+ for(const [id,presets,apply] of [['suitePreset',suites,importSuite],['weightPreset',profiles,importWeights]])$(id).onchange=()=>{const chosen=$(id).value;if(chosen==='custom')return;try{apply(structuredClone(presets[Number(chosen)].config),chosen);$('error').textContent='';render();}catch(err){$('error').textContent=err.message;configOptions();}};
$('clearModels').onclick=()=>{models=new Map();sourceAudits=new Map();metadata=new Map();modelOptions();clearFilters();languageOptions();state.view='score';$('error').textContent='';render();};
$('modelFile').onchange=async e=>{try{
const file=e.target.files[0];if(!file)return;
- const audit=auditRows(parseCSV(await file.text()),scheme),rows=audit.filter(r=>r.selected),names=[...new Set(audit.map(r=>r.checkpoint))];
+ const audit=auditRows(parseCSV(await file.text()),catalogue),rows=audit.filter(r=>r.selected),names=[...new Set(audit.map(r=>r.checkpoint))];
if(!audit.length)throw Error('No measurements in CSV');if(names.some(n=>models.has(n)))throw Error('Checkpoint name already loaded; use a distinct model label.');
const nextModels=new Map(models),nextAudits=new Map(sourceAudits),nextMetadata=new Map(metadata);
for(const name of names){nextModels.set(name,rows.filter(r=>r.checkpoint===name));nextAudits.set(name,audit.filter(r=>r.checkpoint===name));}
- for(const r of audit)nextMetadata.set(r.task,taskLanguage(r.task,scheme));
- if(!nextModels.has(demoModel))nextModels.set(demoModel,synthetic([...nextModels.values()][0],scheme));
+ for(const r of audit)nextMetadata.set(r.task,taskLanguage(r.task,catalogue));
+ if(!nextModels.has(demoModel))nextModels.set(demoModel,synthetic([...nextModels.values()][0],catalogue));
models=nextModels;sourceAudits=nextAudits;metadata=nextMetadata;
modelOptions(sourceAudits.size>1?names.at(-1):demoModel);languageOptions();$('error').textContent='';render();
}catch(err){$('error').textContent=err.message;}finally{e.target.value='';}};
- function filterOptions(){$('category').innerHTML=options([['','All categories'],...Object.keys(weights).map(k=>[k,k])],'');
- $('eval').innerHTML=options([['','All evals'],...scheme.evals.map(f=>[f.name,f.name])],'');}
+ function filterOptions(){$('category').innerHTML=options([['','All categories'],...[...new Set([...Object.keys(weights),...catalogue.evals.map(e=>e.category)])].map(k=>[k,k])],'');
+ $('eval').innerHTML=options([['','All evals'],...catalogue.evals.map(f=>[f.name,f.name])],'');}
filterOptions();modelOptions();languageOptions();render();
}
diff --git a/app/build.py b/app/build.py
index 4552e33..8618654 100644
--- a/app/build.py
+++ b/app/build.py
@@ -5,7 +5,7 @@
import json
from pathlib import Path
from statistics import mean
-from .config_engine import classify, task_language, load_config, load_csv
+from .config_engine import classify, task_language, load_catalogue, shared_config, load_csv
APP = Path(__file__).resolve().parent
ROOT = APP.parent
@@ -101,36 +101,52 @@ def default_config(directory):
return path
-def build(source, output, config_path=None, results_dir=None, configs_dir=None):
- if source is not None and results_dir is not None:raise ValueError('Choose a CSV or --results-dir, not both')
- if config_path is None:
- configs_dir = configs_dir if configs_dir is not None else ROOT/'configs'
- config_path = default_config(configs_dir)
- config = load_config(config_path)
- configurations=[dict(file=config_path.name,config=config)]
- if configs_dir is not None:
- for path in directory_files(configs_dir,{'.yaml','.yml'}):
- if path.resolve()==config_path.resolve():continue
- try:alternative=load_config(path)
- except ValueError as error:raise ValueError(f'{path.name}: {error}') from error
- if any(c['config']['name']==alternative['name'] for c in configurations):raise ValueError(f'{path.name}: duplicate config name {alternative["name"]!r}; use distinct names')
- configurations.append(dict(file=path.name,config=alternative))
+def config_choices(path, directory, kind, default_directory):
+ if path is None:
+ directory = directory or default_directory
+ path = default_config(directory)
+ choices = [dict(file=path.name, config=shared_config(kind, path))]
+ if directory is not None:
+ for candidate in directory_files(directory, {'.yaml', '.yml'}):
+ if candidate.resolve() == path.resolve(): continue
+ try: config = shared_config(kind, candidate)
+ except ValueError as error: raise ValueError(f'{candidate.name}: {error}') from error
+ if any(p['config']['name'] == config['name'] for p in choices):
+ raise ValueError(f'{candidate.name}: duplicate config name {config["name"]!r}; use distinct names')
+ choices.append(dict(file=candidate.name, config=config))
+ return path, choices
+
+
+def build(source, output, catalogue_path=None, results_dir=None, *, weights_path=None, weights_dir=None, suite_path=None, sets_dir=None):
+ if source is not None and results_dir is not None: raise ValueError('Choose a CSV or --results-dir, not both')
+ catalogue_path = catalogue_path or ROOT/'configs/catalogue.yaml'
+ catalogue = load_catalogue(catalogue_path)
+ weights_path, profiles = config_choices(weights_path, weights_dir, 'weights', ROOT/'configs/weights')
+ suite_path, suites = config_choices(suite_path, sets_dir, 'suite', ROOT/'configs/sets')
+ # Every offered combination must resolve before any output is replaced.
+ for profile in profiles:
+ for entry in suites:
+ try: shared_config('resolve', value=[catalogue, entry['config'], profile['config']])
+ except ValueError as error: raise ValueError(f'{entry["file"]} / {profile["file"]}: {error}') from error
+ suite, profile = suites[0]['config'], profiles[0]['config']
+ config = shared_config('resolve', value=[catalogue, suite, profile])
paths=[source] if source is not None else directory_files(results_dir,{'.csv'}) if results_dir is not None else []
rows=[];audit=[];sources=[];owners={}
for path in paths:
try:
rr=load_csv(path)
- classified=classify(rr,config)
+ classified=classify(rr,catalogue)
except ValueError as error:raise ValueError(f'{path.name}: {error}') from error
for model in {r['checkpoint'] for r in rr}:
if model in owners:raise ValueError(f'Duplicate model name {model!r} in {owners[model]} and {path.name}; combine its results in one file or rename the checkpoint')
owners[model]=path.name
rows.extend(rr);audit.extend(classified)
sources.append(dict(file=path.name,sha256=hashlib.sha256(path.read_bytes()).hexdigest()))
- for alternative in configurations[1:]:
- try:classify(rows,alternative['config'])
- except ValueError as error:raise ValueError(f'{alternative["file"]}: {error}') from error
- aggregates = {mode: summarize(audit, config, mode) for mode in ['standard', 'english_eval', 'english_category']}
+ scoped = shared_config('scope', value=[audit, suite])['rows']
+ identities = {tuple(r[k] for k in ['checkpoint','task','metric','filter','n_shot','harness','backend']) for r in scoped}
+ included = {id(r) for r in audit if tuple(r[k] for k in ['checkpoint','task','metric','filter','n_shot','harness','backend']) in identities}
+ scoped_audit = [dict(r, selected=r['selected'] and id(r) in included) for r in audit]
+ aggregates = {mode: summarize(scoped_audit, config, mode) for mode in ['standard', 'english_eval', 'english_category']}
summary = aggregates[config.get('aggregate', 'standard')]
output.mkdir(parents=True, exist_ok=True)
if audit:
@@ -140,23 +156,31 @@ def build(source, output, config_path=None, results_dir=None, configs_dir=None):
else:
for name in ['row-audit.csv', 'eval-scores.csv', 'category-scores.csv', 'language-metadata.csv']:
(output/name).unlink(missing_ok=True)
- metadata = [task_language(task, config) for task in sorted({r['task'] for r in rows})]
+ metadata = [task_language(task, catalogue) for task in sorted({r['task'] for r in rows})]
if metadata:write_csv(output/'language-metadata.csv', metadata)
- payload = dict(config_file=config_path.name, configurations=configurations, metadata=metadata, scheme=config, models=summary, aggregates=aggregates, rows=audit, sources=sources, source=source.name if source else results_dir.name if results_dir else '', sha256=sources[0]['sha256'] if len(sources)==1 else None)
- (output/'eval-config.yaml').write_text(config_path.read_text())
+ payload = dict(catalogue=catalogue, catalogue_file=catalogue_path.name, suite=suite, suite_file=suite_path.name,
+ profile=profile, profile_file=weights_path.name, suites=suites, profiles=profiles,
+ metadata=metadata, scheme=config, models=summary, aggregates=aggregates, rows=audit, sources=sources,
+ source=source.name if source else results_dir.name if results_dir else '', sha256=sources[0]['sha256'] if len(sources)==1 else None)
+ (output/'catalogue.yaml').write_text(catalogue_path.read_text())
+ (output/'weights.yaml').write_text(weights_path.read_text())
+ (output/'eval-set.yaml').write_text(suite_path.read_text())
(output/'analysis.json').write_text(json.dumps(payload, indent=2))
template = (APP/'template.html').read_text()
- (output/'index.html').write_text(template.replace('__APP__', (APP/'vendor/js-yaml.js').read_text()+'\n'+(APP/'eval_config.js').read_text()+'\n'+(APP/'app.js').read_text()).replace('__PAYLOAD__', json.dumps(payload).replace('<', '\\u003c')))
- print(json.dumps(dict(models=summary, rows=len(audit), selected=sum(r['selected'] for r in audit)), indent=2))
+ (output/'index.html').write_text(template.replace('__APP__', (APP/'vendor/js-yaml.js').read_text()+'\n'+(APP/'eval_config.js').read_text()+'\n'+(APP/'suite_config.js').read_text()+'\n'+(APP/'app.js').read_text()).replace('__PAYLOAD__', json.dumps(payload).replace('<', '\\u003c')))
+ print(json.dumps(dict(models=summary, rows=len(audit), selected=sum(r['selected'] for r in scoped_audit)), indent=2))
if __name__ == '__main__':
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('csv', type=Path, nargs='?', help='CSV to embed; omit to start without results')
parser.add_argument('--results-dir', type=Path, help='Embed all CSV files directly inside this directory')
- parser.add_argument('--configs-dir', type=Path, help='Offer YAML configs from this directory; default: repository configs/ when --config is omitted')
+ parser.add_argument('--catalogue', type=Path, help='Global eval interpretation YAML; default: configs/catalogue.yaml')
+ parser.add_argument('--weights', type=Path, help='Default weighting profile YAML; used alone, embed only this profile')
+ parser.add_argument('--weights-dir', type=Path, help='Offer weighting profiles from this directory (default: configs/weights)')
+ parser.add_argument('--eval-set', type=Path, help='Default named eval set YAML; used alone, embed only this set')
+ parser.add_argument('--sets-dir', type=Path, help='Offer eval sets from this directory (default: configs/sets)')
parser.add_argument('--output', type=Path, default=ROOT/'output')
- parser.add_argument('--config', type=Path, help='Use this config instead of the filename in configs/default.txt; used alone, embed only this config')
args = parser.parse_args()
- if args.csv is not None and args.results_dir is not None:parser.error('Choose a CSV or --results-dir, not both')
- build(args.csv, args.output, args.config, args.results_dir, args.configs_dir)
+ if args.csv is not None and args.results_dir is not None: parser.error('Choose a CSV or --results-dir, not both')
+ build(args.csv, args.output, args.catalogue, args.results_dir, weights_path=args.weights, weights_dir=args.weights_dir, suite_path=args.eval_set, sets_dir=args.sets_dir)
diff --git a/app/config_engine.py b/app/config_engine.py
index 177eb40..fa226a8 100644
--- a/app/config_engine.py
+++ b/app/config_engine.py
@@ -6,18 +6,21 @@
from pathlib import Path
-def load_config(path):
- """Use the same bundled YAML parser and validation as browser imports."""
- result=subprocess.run(["node",str(Path(__file__).with_name("config_io.cjs")),str(path)],capture_output=True,text=True)
- if result.returncode:raise ValueError(result.stderr.strip())
- return validate_config(json.loads(result.stdout))
+def shared_config(mode, path=None, value=None):
+ """Use bundled browser parsers and set validation during builds."""
+ result = subprocess.run(["node", str(Path(__file__).with_name("config_io.cjs")), str(path) if path else "-", mode],
+ input=json.dumps(value) if path is None else None, capture_output=True, text=True)
+ if result.returncode: raise ValueError(result.stderr.strip())
+ return json.loads(result.stdout)
+
+
+def load_catalogue(path):
+ return validate_catalogue(shared_config('catalogue', path))
def load_csv(path):
- """Use the browser's strict CSV parser so builds and imports accept the same files."""
- result=subprocess.run(["node",str(Path(__file__).with_name("config_io.cjs")),str(path),'csv'],capture_output=True,text=True)
- if result.returncode:raise ValueError(result.stderr.strip())
- return json.loads(result.stdout)
+ return shared_config('csv', path)
+
LANGUAGE_CODE = re.compile(r'(?:[a-z]{3}_[A-Z][a-z]{3}|mul)')
DECIMAL = re.compile(r'[+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+)(?:[eE][+-]?[0-9]+)?')
@@ -60,6 +63,19 @@ def validate_config(config):
if 'english_weights' in config:
object_keys(config['english_weights'],set(weights))
if any(not number(v) or v<0 or v>1 for v in config['english_weights'].values()):raise ValueError('English weights must be between 0 and 1')
+ validate_rules(config)
+ if any(e['category'] not in weights for e in config['evals']):raise ValueError('Eval category has no weight')
+ return config
+
+
+def validate_catalogue(config):
+ object_keys(config, {'version','name','evals','languages','notes'}, {'version','name','evals','languages'})
+ return validate_rules(config)
+
+
+def validate_rules(config):
+ if config['version'] != 1 or isinstance(config['version'], bool):raise ValueError('Unsupported config version')
+ if not isinstance(config['name'], str) or not config['name'].strip():raise ValueError('Config name is required')
if not isinstance(config.get('notes',[]),list) or any(not isinstance(n,str) for n in config.get('notes',[])):raise ValueError('Notes must be strings')
if not isinstance(config['evals'],list) or not config['evals']:raise ValueError('At least one eval is required')
names=set();categories=set()
@@ -67,7 +83,7 @@ def validate_config(config):
object_keys(e,{'name','category','match','metric','filter','shots','select','score','normalize','warning'}, {'name','category','match','metric','filter','score'})
if not isinstance(e['name'],str) or not e['name'] or e['name'] in names:raise ValueError('Eval names must be unique and nonempty')
names.add(e['name'])
- if not isinstance(e['category'],str) or e['category'] not in weights:raise ValueError('Eval category has no weight: '+str(e['category']))
+ if not isinstance(e['category'],str) or not e['category'].strip():raise ValueError('Eval category must be text')
categories.add(e['category'])
if not isinstance(e['metric'],str) or not e['metric'] or not isinstance(e['filter'],str):raise ValueError('Metric and filter must be strings')
validate_match(e['match'])
@@ -83,7 +99,6 @@ def validate_config(config):
if 'basis' in n and n['basis'] not in ['uniform_choice','uniform_integer','not_applicable','unresolved']:raise ValueError('Invalid normalization basis')
if 'note' in n and not isinstance(n['note'],str):raise ValueError('Normalization note must be text')
if 'sources' in n and (not isinstance(n['sources'],list) or any(not isinstance(u,str) or not re.match(r'^https?://',u) for u in n['sources'])):raise ValueError('Normalization sources must be HTTP(S) URLs')
- if categories!=set(weights):raise ValueError('Each weighted category needs at least one eval')
seen=set()
if not isinstance(config['languages'],list):raise ValueError('languages must be a list')
for group in config['languages']:
@@ -136,7 +151,7 @@ def task_language(task,config):
def classify(rows,config):
- validate_config(config);audit=[];seen=set()
+ (validate_config if 'weights' in config else validate_catalogue)(config);audit=[];seen=set()
for index,source in enumerate(rows):
r=dict(source)
for field in ['checkpoint','task','metric','filter','n_shot','harness','backend','value']:
diff --git a/app/config_io.cjs b/app/config_io.cjs
index 65795bb..b595c5c 100644
--- a/app/config_io.cjs
+++ b/app/config_io.cjs
@@ -1,5 +1,8 @@
-// Python builds share the browser's CSV/YAML parsers and schema validation.
-const fs=require('node:fs');
-const {parseConfig,parseCSV}=require('./eval_config.js');
-try{process.stdout.write(JSON.stringify((process.argv[3]==='csv'?parseCSV:parseConfig)(fs.readFileSync(process.argv[2],'utf8'))));}
-catch(error){process.stderr.write(error.message+'\n');process.exitCode=1;}
+// Python builds share the browser's parsers, validation and set membership.
+const fs=require('node:fs'),evals=require('./eval_config.js'),sets=require('./suite_config.js');
+try{
+ const source=fs.readFileSync(process.argv[2]==='-'?0:process.argv[2],'utf8'),mode=process.argv[3];
+ const parsers={csv:evals.parseCSV,catalogue:evals.parseCatalogue,weights:sets.parseWeightProfile,suite:sets.parseSuite};
+ const value=mode==='resolve'?sets.resolveConfig(...JSON.parse(source)):mode==='scope'?sets.scopeRows(...JSON.parse(source)):parsers[mode](source);
+ process.stdout.write(JSON.stringify(value));
+}catch(error){process.stderr.write(error.message+'\n');process.exitCode=1;}
diff --git a/app/eval_config.js b/app/eval_config.js
index 1b462b1..0772495 100644
--- a/app/eval_config.js
+++ b/app/eval_config.js
@@ -1,8 +1,6 @@
'use strict';
const EvalConfig=(()=>{
const yaml=typeof module!=='undefined'?require('./vendor/js-yaml.js'):jsyaml;
- function parseConfig(source){return validateConfig(yaml.load(source,{schema:yaml.CORE_SCHEMA}));}
- function serializeConfig(config){return yaml.dump(validateConfig(config),{schema:yaml.CORE_SCHEMA,lineWidth:110,noRefs:true,sortKeys:false});}
const canonical=/^(?:[a-z]{3}_[A-Z][a-z]{3}|mul)$/;
const objectKeys=(value,allowed,required=[])=>{if(!value||typeof value!=='object'||Array.isArray(value)||Object.keys(value).some(k=>!allowed.includes(k))||required.some(k=>!Object.hasOwn(value,k)))throw Error('Invalid config fields; allowed '+allowed.join(', ')+'; required '+required.join(', '));};
const number=v=>typeof v==='number'&&Number.isFinite(v);
@@ -38,21 +36,35 @@ const EvalConfig=(()=>{
}
function validateMatch(rule){objectKeys(rule,['name','regex']);const values=Object.values(rule);if(values.length!==1||typeof values[0]!=='string'||!values[0])throw Error('A match needs exactly one nonempty name or regex');if('regex'in rule){if(rule.regex.includes('(?P')||rule.regex.includes('(?<'))throw Error('Use portable regexes');new RegExp(rule.regex);}}
function matchTask(rule,task){return 'name'in rule?rule.name===task:new RegExp('^(?:'+rule.regex+')$(?![\\s\\S])').test(task);}
- function validateConfig(config){
- objectKeys(config,['version','name','weights','evals','languages','notes','aggregate','english_weights'],['version','name','weights','evals','languages']);
- if(config.version!==1)throw Error('Unsupported config version');
- if(typeof config.name!=='string'||!config.name)throw Error('Config name is required');
+ function validateWeights(config){
const w=config.weights;objectKeys(w,Object.keys(w||{}));
if(!Object.keys(w).length||Object.entries(w).some(([k,v])=>!k||!number(v)||v<0)||Math.abs(Object.values(w).reduce((a,b)=>a+b,0)-1)>1e-8)throw Error('Category weights must be nonnegative and sum to 1');
if('aggregate'in config&&!['standard','english_eval','english_category'].includes(config.aggregate))throw Error('Aggregate must be standard, english_eval, or english_category');
if('english_weights'in config){objectKeys(config.english_weights,Object.keys(w));if(Object.values(config.english_weights).some(v=>!number(v)||v<0||v>1))throw Error('English weights must be between 0 and 1');}
+ }
+ function validateCatalogue(config){
+ objectKeys(config,['version','name','evals','languages','notes'],['version','name','evals','languages']);
+ return validateRules(config);
+ }
+ // Resolved scoring configuration used by arithmetic helpers, never a YAML input format.
+ function validateConfig(config){
+ objectKeys(config,['version','name','weights','evals','languages','notes','aggregate','english_weights'],['version','name','weights','evals','languages']);
+ validateWeights(config);validateRules(config);
+ for(const e of config.evals)if(!Object.hasOwn(config.weights,e.category))throw Error('Eval category has no weight: '+e.category);
+ return config;
+ }
+ function parseCatalogue(source){return validateCatalogue(yaml.load(source,{schema:yaml.CORE_SCHEMA}));}
+ function serializeCatalogue(config){return yaml.dump(validateCatalogue(config),{schema:yaml.CORE_SCHEMA,lineWidth:110,noRefs:true});}
+ function validateRules(config){
+ if(config.version!==1)throw Error('Unsupported config version');
+ if(typeof config.name!=='string'||!config.name)throw Error('Config name is required');
if('notes'in config&&(!Array.isArray(config.notes)||config.notes.some(n=>typeof n!=='string')))throw Error('Notes must be strings');
if(!Array.isArray(config.evals)||!config.evals.length)throw Error('At least one eval is required');
- const names=new Set(),categories=new Set();
+ const names=new Set();
for(const e of config.evals){
objectKeys(e,['name','category','match','metric','filter','shots','select','score','normalize','warning'],['name','category','match','metric','filter','score']);
- if(typeof e.name!=='string'||!e.name||names.has(e.name))throw Error('Eval names must be unique and nonempty');names.add(e.name);categories.add(e.category);
- if(typeof e.category!=='string'||!Object.hasOwn(w,e.category))throw Error('Eval category has no weight: '+e.category);
+ if(typeof e.name!=='string'||!e.name||names.has(e.name))throw Error('Eval names must be unique and nonempty');names.add(e.name);
+ if(typeof e.category!=='string'||!e.category)throw Error('Eval category must be nonempty text');
if(typeof e.metric!=='string'||!e.metric||typeof e.filter!=='string')throw Error('Metric and filter must be strings');
validateMatch(e.match);if('select'in e)validateMatch(e.select);
if('shots'in e&&(!Number.isInteger(e.shots)||e.shots<0))throw Error('shots must be a nonnegative integer');
@@ -60,7 +72,6 @@ const EvalConfig=(()=>{
if('warning'in e&&(typeof e.warning!=='string'||!e.warning.trim()))throw Error('Eval warning must be nonempty text');
if('normalize'in e){const n=e.normalize;objectKeys(n,['min','max','clip','basis','note','sources'],['min','max']);if(!number(n.min)||!number(n.max)||!(0<=n.min&&n.mintypeof u!=='string'||!/^https?:\/\//.test(u))))throw Error('Normalization sources must be HTTP(S) URLs');}
}
- if(categories.size!==Object.keys(w).length)throw Error('Each weighted category needs at least one eval');
if(!Array.isArray(config.languages))throw Error('languages must be a list');const seen=new Set();
for(const g of config.languages){
objectKeys(g,['tasks','scope','language','source_language','target_language','evidence','note'],['tasks','scope']);
@@ -78,7 +89,7 @@ const EvalConfig=(()=>{
function normalizeScore(value,e){if(!number(value)&&(typeof value!=='string'||!decimal.test(value.trim())))throw Error('Invalid score: expected a finite decimal number');const raw=Number(value)/e.score.scale;if(!Number.isFinite(raw)||raw<0||raw>1)throw Error('Invalid score: outside the configured source scale');const n=e.normalize??{min:0,max:1};let adjusted=(raw-n.min)/(n.max-n.min);if(n.clip!==false)adjusted=Math.max(0,Math.min(1,adjusted));if(!Number.isFinite(adjusted*100))throw Error('Invalid score: normalization overflow');return {raw_score_100:raw*100,score_100:adjusted*100};}
function taskLanguage(task,config){const g=config.languages.find(g=>g.tasks.includes(task));return {task,language:g?.language??'',source_language:g?.source_language??'',target_language:g?.target_language??'',scope:g?.scope??'unknown',status:g?'resolved':'unknown',evidence:g?.evidence??'',provenance:g?(g.note??'Explicit language assignment in eval config.'):'No explicit language assignment for this task.'};}
function auditRows(rows,config){
- validateConfig(config);const seen=new Set();return rows.map((source,index)=>{
+ (Object.hasOwn(config,'weights')?validateConfig:validateCatalogue)(config);const seen=new Set();return rows.map((source,index)=>{
const r={...source};
for(const field of ['checkpoint','task','metric','filter','n_shot','harness','backend','value'])if(!Object.hasOwn(r,field))throw Error('Missing CSV column: '+field);
for(const field of ['checkpoint','task','metric','harness','backend'])if(typeof r[field]!=='string'||!r[field].trim())throw Error('CSV row '+(index+2)+': '+field+' must be nonempty text');
@@ -96,6 +107,6 @@ const EvalConfig=(()=>{
return {...r,eval:e.name,category:e.category,selected,decision,...scores};
});
}
- return {parseCSV,parseConfig,serializeConfig,validateConfig,matchTask,normalizeScore,taskLanguage,auditRows,demoModel};
+ return {parseCatalogue,serializeCatalogue,validateCatalogue,validateWeights,parseCSV,validateConfig,matchTask,normalizeScore,taskLanguage,auditRows,demoModel};
})();
if(typeof module!=='undefined')module.exports=EvalConfig;
diff --git a/app/suite_config.js b/app/suite_config.js
new file mode 100644
index 0000000..7979ed5
--- /dev/null
+++ b/app/suite_config.js
@@ -0,0 +1,78 @@
+'use strict';
+const SuiteConfig=(()=>{
+ const api=typeof module!=='undefined'?require('./eval_config.js'):EvalConfig;
+ const yaml=typeof module!=='undefined'?require('./vendor/js-yaml.js'):jsyaml;
+ const keys=(o,allowed,required=[])=>{if(!o||typeof o!=='object'||Array.isArray(o)||Object.keys(o).some(k=>!allowed.includes(k))||required.some(k=>!Object.hasOwn(o,k)))throw Error('Invalid config fields; allowed '+allowed.join(', '));};
+ const text=v=>typeof v==='string'&&v.trim().length>0;
+ function validateSuite(s){
+ keys(s,['version','name','mode','evals','exclude','notes'],['version','name','mode']);
+ if(s.version!==1||!text(s.name))throw Error('Suite requires version 1 and a name');
+ if(!['available','fixed'].includes(s.mode))throw Error('Suite mode must be available or fixed');
+ if('notes'in s&&(!Array.isArray(s.notes)||s.notes.some(n=>typeof n!=='string')))throw Error('Suite notes must be strings');
+ if('exclude'in s&&(s.mode!=='available'||!Array.isArray(s.exclude)||s.exclude.some(n=>!text(n))||new Set(s.exclude).size!==s.exclude.length))throw Error('exclude must be a unique list of eval names in available mode');
+ if(s.mode==='available'){if('evals'in s)throw Error('Available mode does not declare required evals');return s;}
+ if(!Array.isArray(s.evals)||!s.evals.length)throw Error('Fixed suite needs required evals');
+ const names=new Set();
+ for(const e of s.evals){
+ keys(e,['name','variants'],['name']);if(!text(e.name)||names.has(e.name))throw Error('Suite eval names must be unique');names.add(e.name);
+ if(!('variants'in e))continue;
+ if(!Array.isArray(e.variants)||!e.variants.length)throw Error('Required variants must be a nonempty list');
+ const seen=new Map();
+ for(const v of e.variants){
+ keys(v,['task','n_shot'],['task']);if(!text(v.task))throw Error('Required variant needs a task name');
+ if('n_shot'in v&&(!Number.isSafeInteger(v.n_shot)||v.n_shot<0))throw Error('Variant n_shot must be a nonnegative integer');
+ const shots=seen.get(v.task)||new Set(),shot=v.n_shot??'*';
+ if(shots.has(shot)||shots.has('*')||shot==='*'&&shots.size)throw Error('Duplicate or overlapping required variant: '+v.task);
+ shots.add(shot);seen.set(v.task,shots);
+ }
+ }
+ return s;
+ }
+ const parseSuite=source=>validateSuite(yaml.load(source,{schema:yaml.CORE_SCHEMA}));
+ const serializeSuite=s=>yaml.dump(validateSuite(s),{schema:yaml.CORE_SCHEMA,lineWidth:110,noRefs:true});
+ function validateWeightProfile(p){
+ keys(p,['version','name','weights','english_weights','aggregate','notes'],['version','name','weights']);
+ if(p.version!==1||!text(p.name))throw Error('Weight profile requires version 1 and a name');
+ if('notes'in p&&(!Array.isArray(p.notes)||p.notes.some(n=>typeof n!=='string')))throw Error('Weight profile notes must be strings');
+ api.validateWeights(p);return p;
+ }
+ const parseWeightProfile=source=>validateWeightProfile(yaml.load(source,{schema:yaml.CORE_SCHEMA}));
+ const serializeWeightProfile=p=>yaml.dump(validateWeightProfile(p),{schema:yaml.CORE_SCHEMA,lineWidth:110,noRefs:true});
+ function resolveConfig(catalogue,suite,profile){
+ api.validateCatalogue(catalogue);validateSuite(suite);validateWeightProfile(profile);
+ const evals=suite.mode==='available'?catalogue.evals:suite.evals.map(required=>{
+ const e=catalogue.evals.find(e=>e.name===required.name);if(!e)throw Error('Suite eval has no catalogue rule: '+required.name);
+ for(const v of required.variants||[]){
+ const matches=catalogue.evals.filter(rule=>api.matchTask(rule.match,v.task));
+ if(matches.length!==1||matches[0].name!==e.name||e.select&&!api.matchTask(e.select,v.task)||'shots'in e&&'n_shot'in v&&e.shots!==v.n_shot)throw Error('Required variant is not selected by its catalogue rule: '+v.task);
+ }
+ return e;
+ });
+ return api.validateConfig({version:1,name:catalogue.name,evals,languages:catalogue.languages,weights:{...profile.weights,...Object.fromEntries(evals.filter(e=>!Object.hasOwn(profile.weights,e.category)).map(e=>[e.category,0]))},english_weights:{...profile.english_weights},aggregate:profile.aggregate||'standard',notes:[...(catalogue.notes||[]),...(profile.notes||[])]});
+ }
+ function inSuite(row,suite){
+ if(suite.mode==='available')return !(suite.exclude||[]).includes(row.eval);
+ const e=suite.evals.find(e=>e.name===row.eval);
+ return !!e&&(!e.variants||e.variants.some(v=>v.task===row.task&&(!('n_shot'in v)||String(v.n_shot)===String(row.n_shot))));
+ }
+ function scopeRows(rows,suite){
+ const selected=rows.filter(r=>r.selected),included=selected.filter(r=>inSuite(r,suite)),extras=selected.filter(r=>!inSuite(r,suite)),missing=[];
+ if(suite.mode==='fixed')for(const e of suite.evals){
+ if(e.variants)for(const v of e.variants){if(!included.some(r=>r.eval===e.name&&r.task===v.task&&(!('n_shot'in v)||String(v.n_shot)===String(r.n_shot))))missing.push({eval:e.name,task:v.task,n_shot:v.n_shot});}
+ else if(!included.some(r=>r.eval===e.name))missing.push({eval:e.name});
+ }
+ return {rows:included,extras,missing};
+ }
+ function suiteCoverage(a,b,suite){
+ const left=scopeRows(a,suite),right=scopeRows(b,suite),warnings=[];
+ for(const [side,scope] of [['A',left],['B',right]]){
+ for(const name of new Set(scope.missing.map(r=>r.eval))){const rr=scope.missing.filter(r=>r.eval===name);warnings.push({type:'Missing suite data',name,eval:name,model:side,detail:rr.length+' requirement(s) missing from '+suite.name+'. Missing results are excluded from both calculations. Comparison is incomplete; shared scores are used with redistributed weights.',variants:[{settings:'Required in '+side,tasks:rr.map(r=>r.task?(r.task+('n_shot'in r&&r.n_shot!==undefined?' · '+r.n_shot+' shots':'')):'Any selected measurement for '+r.eval)}]});}
+ }
+ const identity=r=>JSON.stringify(['task','metric','filter','n_shot','harness','backend'].map(k=>String(r[k]??'')));
+ const rightKeys=new Set(right.rows.map(identity)),shared=scopeRows(left.rows.filter(r=>rightKeys.has(identity(r))),suite);
+ const required=suite.mode==='fixed'?suite.evals.reduce((n,e)=>n+(e.variants?.length||1),0):null;
+ return {a:left.rows,b:right.rows,warnings,complete:!shared.missing.length,required,sharedRequired:required===null?null:required-shared.missing.length,presentA:required===null?null:required-left.missing.length,presentB:required===null?null:required-right.missing.length,extrasA:left.extras.length,extrasB:right.extras.length};
+ }
+ return {validateWeightProfile,parseWeightProfile,serializeWeightProfile,validateSuite,parseSuite,serializeSuite,resolveConfig,inSuite,scopeRows,suiteCoverage};
+})();
+if(typeof module!=='undefined')module.exports=SuiteConfig;
diff --git a/app/template.html b/app/template.html
index 59d7604..8e2230e 100644
--- a/app/template.html
+++ b/app/template.html
@@ -13,7 +13,8 @@
Quickdash · OELLM evals
DRAFT SCORING SCHEME
-
+
Weights and eval sets are independent. “Any available” needs no list of evals.
+
Score calculation
diff --git a/configs/README.md b/configs/README.md
index 56f58b4..b4d0f60 100644
--- a/configs/README.md
+++ b/configs/README.md
@@ -1,27 +1,21 @@
-# Share evaluation configurations
+# Share eval rules, weights and named sets
-The dashboard’s **Eval configuration** selector offers the YAML files directly in this directory, including the [OELLM config](oellm.yaml). Each file’s `name` appears in the selector, so give it a distinct, descriptive name. [example.yaml](example.yaml) is for the fictional scores in [examples/scores.csv](../examples/scores.csv).
+Choose the file to edit based on what you want to change:
-[default.txt](default.txt) contains the filename selected when the page opens:
+- **Interpret a new eval:** add its task matching, category, scoring metric, normalization, and explicit language assignments to [catalogue.yaml](catalogue.yaml). Adding a rule does not require every model to run it.
+- **Try different weighting:** add a YAML profile to [weights/](weights/). It can be used with any eval set. Category weights, English shares, and the default calculation live here.
+- **Require a standard comparison set:** add a YAML file to [sets/](sets/). [flagship-1.yaml](sets/flagship-1.yaml) pins expected tasks and shot counts; [any-available.yaml](sets/any-available.yaml) needs no required-eval list and compares shared data, with an explicit exclusion for unvalidated prompted Global PIQA.
-```text
-oellm.yaml
-```
-
-To change the default, replace that line with another YAML filename from this directory and commit it. The build rejects missing files, paths outside this directory, and multiple filenames. The display label still comes from the selected YAML’s `name` field.
+Each profile or set needs a distinct `name` within its directory. To change a selector's startup choice, edit that directory's `default.txt` to name one YAML file. The catalogue is selected at build time with `--catalogue`, or temporarily loaded in the browser.
-To add a scoring scheme:
+The [configuration reference](../docs/configuration.md) describes all three formats with small examples. [examples/](examples/) contains the fictional catalogue and weights used by [examples/scores.csv](../examples/scores.csv).
-1. Copy an existing config and edit its eval matches, categories, metrics, normalization, and language assignments. The [config reference](../docs/configuration.md) explains each field and provides a small complete example.
-2. Save it here as a `.yaml` or `.yml` file, then submit a PR or use GitHub’s **Add file → Upload files**.
-3. Check the generated dashboard and its warnings before using the scores to choose a training method.
+Submit changes as a PR, then check the generated dashboard:
```sh
python3 -m app.build --results-dir results --output output/shared
```
-The build checks every config against the shared CSVs before publishing. Invalid schemas, duplicate config names, overlapping eval matches, or invalid selected score scales stop the update. A config can intentionally cover only part of the results: unmatched tasks and missing metrics are excluded with warnings when that config is selected.
-
-Switching configs recalculates all loaded models. If a temporary uploaded model is incompatible with a config, the browser reports the error and keeps the previous config and scores. **Clear models** starts a fresh browser comparison while retaining the config choices; reload restores the published results.
+The build validates every offered combination before replacing output. In the dashboard, inspect **Warnings** for the comparison you intend to use. Named-set scores with missing requirements are explicitly incomplete; extras are excluded but remain inspectable.
-For a temporary config, use **Eval configuration → Load config** in the dashboard instead of committing a file. It appears as an uploaded choice for that browser session. **Export config YAML** saves your edits, including adjusted weights and the chosen aggregation mode; it does not modify this repository.
+For temporary changes, use the separate load/export controls under **Eval configuration**. Files stay in the browser. Weight exports save edited weights and the active calculation; catalogue and eval-set exports are independent. None of these controls modify the repository.
diff --git a/configs/oellm.yaml b/configs/catalogue.yaml
similarity index 96%
rename from configs/oellm.yaml
rename to configs/catalogue.yaml
index 715e244..ea3f25b 100644
--- a/configs/oellm.yaml
+++ b/configs/catalogue.yaml
@@ -1,30 +1,5 @@
-# Initial chance baselines. Bounds are fractions after value / score.scale.
-# Set min: 0, max: 1 to disable a correction; clip: true clips below chance to zero.
-# Per-eval rationale and sources are also shown in the dashboard.
version: 1
-name: OELLM draft · initial chance normalization
-weights:
- Code: 0.15
- Math: 0.15
- Reasoning: 0.15
- Knowledge: 0.15
- Commonsense: 0.15
- Reading: 0.15
- Translation: 0.03333333333333333
- Language: 0.03333333333333333
- Instruction following: 0.03333333333333333
-# The selector starts on the original aggregate; these shares configure the alternative.
-aggregate: standard
-english_weights:
- Code: 0.5
- Math: 0.5
- Reasoning: 0.5
- Knowledge: 0.5
- Commonsense: 0.5
- Reading: 0.5
- Translation: 0.5
- Language: 0.5 # Unknown/mixed scores use the English side; the Croatian/Serbian pool assigned to srp_Latn uses the other side.
- Instruction following: 0.5
+name: OELLM eval catalogue
evals:
- category: Code
name: CS Algorithms
@@ -258,14 +233,18 @@ evals:
score:
scale: 1
normalize:
- min: 0
+ # Shared 10.55% baseline; Table 2 reports approximately 10.5%.
+ min: 0.1055
max: 1
clip: true
- basis: unresolved
note: >-
- The evaluator mixes single-choice, multiple-answer, integer, and numeric questions. A baseline
- requires per-type guessing rules and the evaluated mixture; one fixed 1/K would be incorrect.
+ Configured 10.55% overall random baseline, aligned with the shared scoring policy. Table 2 of the
+ JEEBench paper reports an approximate 10.5% baseline. It combines
+ uniform single-choice guessing and random option subsets with partial credit for multi-answer
+ questions, assigning zero expected score to integer and numeric answers. This assumes the full
+ 515-question benchmark and the paper's scoring rules.
sources:
+ - https://aclanthology.org/2023.emnlp-main.468.pdf#page=5
- >-
https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/JEEBench/eval_instruct.py
- category: Reasoning
@@ -336,15 +315,14 @@ evals:
score:
scale: 1
normalize:
- min: 0.25016133557800224
+ min: 0.25
max: 1
clip: true
basis: uniform_choice
note: >-
- The published ARC-V1-Feb2018 Easy test split has 2,376 questions: 2,365 with four choices, seven with
- three, and four with five. Mean uniform-guess accuracy is (2365/4 + 7/3 + 4/5)/2376. The CSV reports
- the same sample count; this initial baseline assumes the full published test split, not a modified
- subset.
+ Use the conventional four-choice approximation of 0.25. The published test split includes a few
+ three- and five-choice questions (mean guessing accuracy approximately 0.2501613); this small
+ difference is intentionally ignored for consistency.
sources:
- https://ai2-public-datasets.s3.amazonaws.com/arc/ARC-V1-Feb2018.zip
- https://huggingface.co/datasets/allenai/ai2_arc
@@ -538,24 +516,39 @@ evals:
metric: acc_norm
filter: none
match:
- regex: (?:piqa|global_piqa_.+)
+ regex: (?:piqa|global_piqa_completions_.+)
score:
scale: 1
- select:
- regex: (?:piqa|global_piqa_completions_.+)
normalize:
min: 0.5
max: 1
clip: true
basis: uniform_choice
note: >-
- The selected completion tasks compare two solutions. Uniform guessing gives 1/2; prompted protocols
- are excluded by the selection rule.
+ The completion tasks compare two solutions. Uniform guessing gives 1/2. Prompted Global PIQA is
+ configured separately and excluded from the supplied comparison sets.
sources:
- >-
https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/piqa/piqa.yaml
- >-
https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/global_piqa/completions/_template
+ - category: Commonsense
+ name: Global PIQA (prompted)
+ match:
+ regex: global_piqa_prompted_.+
+ metric: exact_match
+ filter: strict_match
+ score:
+ scale: 1
+ normalize:
+ min: 0
+ max: 1
+ clip: true
+ basis: unresolved
+ note: Provisional scale conversion only; no validated chance correction for the prompted protocol.
+ warning: >-
+ Normalization and metric selection have not been validated for prompted Global PIQA.
+ The supplied comparison sets exclude this eval. Review the protocol before including its scores.
- category: Commonsense
name: Social IQa
metric: acc
@@ -713,10 +706,8 @@ evals:
https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/lambada/lambada_openai.yaml
- category: Reading
name: SIB-200
- warning: >-
- Metric-selection exception: use acc pending investigation. In this export, acc_norm is exactly
- 0.25 for 34 of 36 languages. The cause is not established; review the evaluator and results
- before switching this eval to the usual acc_norm preference.
+ # Use acc: acc_norm is not a reliable/useful metric for this eval in our export
+ # (exactly 0.25 for 34 of 36 languages).
metric: acc
filter: none
match:
@@ -754,7 +745,7 @@ evals:
- https://rajpurkar.github.io/SQuAD-explorer/
- category: Translation
name: FLORES200
- metric: chrf
+ metric: chrf++
filter: rescored
match:
regex: flores200:.+
@@ -766,16 +757,12 @@ evals:
clip: true
basis: not_applicable
note: >-
- chrF has a defined 0–100 range. Retain its native scale with no chance correction;
- dividing by score.scale=100 yields the fraction used by the composite.
+ Select the explicit chrf++ field from the rescored export (preferred over chrf, then bleu).
+ Retain native 0–100 points with no chance correction. The CSV does not record the rescoring
+ implementation signature; the separate chrf++ label identifies the intended metric.
sources:
- https://github.com/mjpost/sacrebleu/blob/master/sacrebleu/metrics/chrf.py
- https://github.com/facebookresearch/flores
- warning: >-
- Translation calibration is unresolved. chrF is bounded by 0 and 100, but is not accuracy.
- Its chance floor and comparable performance levels across language pairs are not established here.
- The composite currently uses native chrF points with no chance correction. Review this treatment
- before using the composite to choose a training method.
- category: Translation
name: OpenSubtitles
metric: chrf
@@ -789,16 +776,15 @@ evals:
max: 1
clip: true
basis: not_applicable
- note: Translation chrF is bounded by 0 and 100; retain its native scale with no chance correction.
+ note: >-
+ Select chrf: the referenced harness uses SacreBLEU defaults (char_order=6, word_order=0, beta=2),
+ so this is plain chrF, not chrF++. Prefer an explicitly identified chrF++ result when available,
+ then chrF, then BLEU. Retain native 0–100 points with no chance correction.
sources:
- >-
https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/opensubtitles_multi40/utils.py
- https://github.com/mjpost/sacrebleu/blob/master/sacrebleu/metrics/chrf.py
- warning: >-
- Translation calibration is unresolved. chrF is bounded by 0 and 100, but is not accuracy.
- Its chance floor and comparable performance levels across language pairs are not established here.
- The composite currently uses native chrF points with no chance correction. Review this treatment
- before using the composite to choose a training method.
+ - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/api/metrics.py
- category: Language
name: Language ID
metric: acc
@@ -859,55 +845,6 @@ evals:
baseline.
sources:
- https://github.com/google-research/google-research/tree/master/instruction_following_eval
-notes:
- - >-
- This is a provisional composite index, not an overall accuracy or a training-method recommendation. Only
- one model has been supplied.
- - >-
- The category assignments follow the supplied screenshot, including its questionable placements of CS
- Algorithms, Dyck languages, Operators and Repeat Copy Logic. ARC Challenge and ARC Easy are separate
- equally weighted evals.
- - >-
- Average selected measurements equally within each eval, evals equally within each category, and categories
- with the specified weights. No sample-count weighting. Language variants and difficulty slices remain
- visible. Existing MMLU and INCLUDE top-level summaries replace their children in the composite; those
- supplied summaries may internally use different weighting.
- - >-
- Multiply fractional accuracy, exact match, pass@1 and CoQA F1 by 100. SQuAD v2 F1 and translation chrF
- retain their supplied 0–100 scale. Sharing a scale does not make these metrics equally difficult,
- calibrated or interchangeable.
- - >-
- Prefer acc_norm when both acc and acc_norm are exported for the selected protocol. SIB-200 retains acc
- as an explicit, warned exception pending investigation of its repeated acc_norm values. Other metrics: flexible-extract for GSM8K/MGSM; Python pass@1 for HumanEval; overall LiveCodeBench accuracy; F1
- for CoQA/SQuAD v2; chrF for translation; strict prompt-level accuracy for IFEval. Other metrics stay in
- the audit.
- - >-
- HellaSwag uses 0-shot results throughout; its English 10-shot result is excluded. PIQA uses completion
- scoring; prompted exact-match variants are excluded. Other evals retain the available protocols, including
- differing shot counts.
- - >-
- MGSM combines native-CoT and global variants. MMLU and Global MMLU are separate evals, each with its own
- equal share of the Knowledge category. MMLU uses its original English summary; Global MMLU averages
- its language summaries. They have related content and should not be treated as independent evidence.
- Agree language/protocol balance before using the composite for a decision.
- - >-
- SIB-200 acc_norm is 0.25 for 34 of 36 languages in this export; the draft selects raw acc. This deserves
- inspection, but the CSV alone cannot establish the cause.
- - >-
- No aggregate confidence interval is reported: the CSV cannot establish cross-task or cross-model
- covariance. A future comparison must align task, metric, filter, shot count, harness, backend, benchmark
- versions and coverage before reporting deltas.
- - >-
- Comparisons use shared measurements. Missing evals are excluded from both scores and remaining evals
- share their category weight. Empty categories are excluded and other category weights are rescaled.
- Unmatched measurements and unconfigured tasks are listed in Warnings. Duplicate selected measurements
- stop the build.
- - >-
- Initial normalization uses benchmark-defined uniform valid-choice baselines. AIME24 and AIME25 use
- a zero minimum with no random-integer correction. These are research-based starting assumptions, not empirical random-model scores. acc_norm
- length-normalizes option likelihoods; it is not chance correction. ARC Challenge uses an explicitly
- approximate 25% floor; JEEBench remains uncorrected pending per-type baseline details. Values below chance
- clip to zero at the selected-variant aggregate level, so correction does not reproduce per-item clipping.
languages:
- tasks:
- AIME24
diff --git a/configs/example.yaml b/configs/examples/catalogue.yaml
similarity index 85%
rename from configs/example.yaml
rename to configs/examples/catalogue.yaml
index d7316c4..fcce2f3 100644
--- a/configs/example.yaml
+++ b/configs/examples/catalogue.yaml
@@ -1,12 +1,5 @@
version: 1
-name: Fictional example — not benchmark results
-aggregate: standard
-weights:
- Reasoning: 0.5
- Math: 0.5
-english_weights:
- Reasoning: 0.5
- Math: 0.5
+name: Fictional example catalogue
evals:
- name: Example reasoning
category: Reasoning
diff --git a/configs/examples/weights.yaml b/configs/examples/weights.yaml
new file mode 100644
index 0000000..92b6914
--- /dev/null
+++ b/configs/examples/weights.yaml
@@ -0,0 +1,9 @@
+version: 1
+name: Fictional example weights
+aggregate: standard
+weights:
+ Reasoning: 0.5
+ Math: 0.5
+english_weights:
+ Reasoning: 0.5
+ Math: 0.5
diff --git a/configs/sets/any-available.yaml b/configs/sets/any-available.yaml
new file mode 100644
index 0000000..c5d83bd
--- /dev/null
+++ b/configs/sets/any-available.yaml
@@ -0,0 +1,7 @@
+version: 1
+name: Any available
+mode: available
+exclude:
+ - Global PIQA (prompted)
+notes:
+ - Compare recognized measurements shared by both models; warn about differences.
diff --git a/configs/sets/default.txt b/configs/sets/default.txt
new file mode 100644
index 0000000..3d9f4c6
--- /dev/null
+++ b/configs/sets/default.txt
@@ -0,0 +1 @@
+any-available.yaml
diff --git a/configs/sets/flagship-1.yaml b/configs/sets/flagship-1.yaml
new file mode 100644
index 0000000..280926f
--- /dev/null
+++ b/configs/sets/flagship-1.yaml
@@ -0,0 +1,904 @@
+version: 1
+name: flagship-1
+mode: fixed
+notes:
+ - >-
+ Expected evals, tasks and shot settings for the flagship-1 comparison. Missing requirements make the score
+ incomplete; extra measurements are excluded.
+evals:
+ - name: CS Algorithms
+ variants:
+ - task: bigbench_cs_algorithms_generate_until
+ n_shot: 10
+ - name: HumanEval
+ variants:
+ - task: HumanEval
+ n_shot: 0
+ - name: LiveCodeBench
+ variants:
+ - task: LiveCodeBench
+ n_shot: 0
+ - name: MBPP
+ variants:
+ - task: mbpp
+ n_shot: 3
+ - name: Dyck languages
+ variants:
+ - task: bigbench_dyck_languages_generate_until
+ n_shot: 10
+ - name: Operators
+ variants:
+ - task: bigbench_operators_generate_until
+ n_shot: 10
+ - name: Repeat Copy Logic
+ variants:
+ - task: bigbench_repeat_copy_logic_generate_until
+ n_shot: 10
+ - name: GSM8K
+ variants:
+ - task: gsm8k
+ n_shot: 4
+ - name: MGSM
+ variants:
+ - task: global_mgsm_ca
+ n_shot: 0
+ - task: global_mgsm_cs
+ n_shot: 0
+ - task: global_mgsm_de
+ n_shot: 0
+ - task: global_mgsm_el
+ n_shot: 0
+ - task: global_mgsm_en
+ n_shot: 0
+ - task: global_mgsm_es
+ n_shot: 0
+ - task: global_mgsm_eu
+ n_shot: 0
+ - task: global_mgsm_fr
+ n_shot: 0
+ - task: global_mgsm_gl
+ n_shot: 0
+ - task: global_mgsm_hu
+ n_shot: 0
+ - task: global_mgsm_sr
+ n_shot: 0
+ - task: mgsm_native_cot_de
+ n_shot: 5
+ - task: mgsm_native_cot_en
+ n_shot: 5
+ - task: mgsm_native_cot_es
+ n_shot: 5
+ - task: mgsm_native_cot_fr
+ n_shot: 5
+ - name: MATH-500
+ variants:
+ - task: MATH500
+ n_shot: 0
+ - name: AIME24
+ variants:
+ - task: AIME24
+ n_shot: 0
+ - name: AIME25
+ variants:
+ - task: AIME25
+ n_shot: 0
+ - name: AMC23
+ variants:
+ - task: AMC23
+ n_shot: 0
+ - name: JEEBench
+ variants:
+ - task: JEEBench
+ n_shot: 0
+ - name: LSAT AR
+ variants:
+ - task: agieval_lsat_ar
+ n_shot: 3
+ - name: PolyMath
+ variants:
+ - task: polymath_de_high
+ n_shot: 0
+ - task: polymath_de_low
+ n_shot: 0
+ - task: polymath_de_medium
+ n_shot: 0
+ - task: polymath_de_top
+ n_shot: 0
+ - task: polymath_en_high
+ n_shot: 0
+ - task: polymath_en_low
+ n_shot: 0
+ - task: polymath_en_medium
+ n_shot: 0
+ - task: polymath_en_top
+ n_shot: 0
+ - task: polymath_es_high
+ n_shot: 0
+ - task: polymath_es_low
+ n_shot: 0
+ - task: polymath_es_medium
+ n_shot: 0
+ - task: polymath_es_top
+ n_shot: 0
+ - task: polymath_fr_high
+ n_shot: 0
+ - task: polymath_fr_low
+ n_shot: 0
+ - task: polymath_fr_medium
+ n_shot: 0
+ - task: polymath_fr_top
+ n_shot: 0
+ - task: polymath_it_high
+ n_shot: 0
+ - task: polymath_it_low
+ n_shot: 0
+ - task: polymath_it_medium
+ n_shot: 0
+ - task: polymath_it_top
+ n_shot: 0
+ - task: polymath_pt_high
+ n_shot: 0
+ - task: polymath_pt_low
+ n_shot: 0
+ - task: polymath_pt_medium
+ n_shot: 0
+ - task: polymath_pt_top
+ n_shot: 0
+ - name: ARC Challenge
+ variants:
+ - task: arc_challenge
+ n_shot: 10
+ - task: arc_challenge_mt_bg
+ n_shot: 0
+ - task: arc_challenge_mt_cs
+ n_shot: 0
+ - task: arc_challenge_mt_da
+ n_shot: 0
+ - task: arc_challenge_mt_de
+ n_shot: 0
+ - task: arc_challenge_mt_el
+ n_shot: 0
+ - task: arc_challenge_mt_es
+ n_shot: 0
+ - task: arc_challenge_mt_et
+ n_shot: 0
+ - task: arc_challenge_mt_fi
+ n_shot: 0
+ - task: arc_challenge_mt_fr
+ n_shot: 0
+ - task: arc_challenge_mt_hu
+ n_shot: 0
+ - task: arc_challenge_mt_is
+ n_shot: 0
+ - task: arc_challenge_mt_it
+ n_shot: 0
+ - task: arc_challenge_mt_lt
+ n_shot: 0
+ - task: arc_challenge_mt_lv
+ n_shot: 0
+ - task: arc_challenge_mt_nb
+ n_shot: 0
+ - task: arc_challenge_mt_nl
+ n_shot: 0
+ - task: arc_challenge_mt_pl
+ n_shot: 0
+ - task: arc_challenge_mt_pt
+ n_shot: 0
+ - task: arc_challenge_mt_ro
+ n_shot: 0
+ - task: arc_challenge_mt_sk
+ n_shot: 0
+ - task: arc_challenge_mt_sl
+ n_shot: 0
+ - task: arc_challenge_mt_sv
+ n_shot: 0
+ - name: ARC Easy
+ variants:
+ - task: arc_easy
+ n_shot: 10
+ - name: GPQA Diamond
+ variants:
+ - task: GPQADiamond
+ n_shot: 0
+ - name: INCLUDE
+ variants:
+ - task: include_base_44_albanian
+ n_shot: 0
+ - task: include_base_44_basque
+ n_shot: 0
+ - task: include_base_44_bulgarian
+ n_shot: 0
+ - task: include_base_44_croatian
+ n_shot: 0
+ - task: include_base_44_dutch
+ n_shot: 0
+ - task: include_base_44_estonian
+ n_shot: 0
+ - task: include_base_44_finnish
+ n_shot: 0
+ - task: include_base_44_french
+ n_shot: 0
+ - task: include_base_44_georgian
+ n_shot: 0
+ - task: include_base_44_german
+ n_shot: 0
+ - task: include_base_44_greek
+ n_shot: 0
+ - task: include_base_44_hungarian
+ n_shot: 0
+ - task: include_base_44_italian
+ n_shot: 0
+ - task: include_base_44_lithuanian
+ n_shot: 0
+ - task: include_base_44_north macedonian
+ n_shot: 0
+ - task: include_base_44_polish
+ n_shot: 0
+ - task: include_base_44_portuguese
+ n_shot: 0
+ - task: include_base_44_serbian
+ n_shot: 0
+ - task: include_base_44_spanish
+ n_shot: 0
+ - task: include_base_44_turkish
+ n_shot: 0
+ - task: include_base_44_ukrainian
+ n_shot: 0
+ - name: Jeopardy
+ variants:
+ - task: jeopardy
+ n_shot: 10
+ - name: MMLU
+ variants:
+ - task: mmlu
+ n_shot: 5
+ - name: Global MMLU
+ variants:
+ - task: global_mmlu_full_cs
+ n_shot: 5
+ - task: global_mmlu_full_de
+ n_shot: 5
+ - task: global_mmlu_full_el
+ n_shot: 5
+ - task: global_mmlu_full_en
+ n_shot: 5
+ - task: global_mmlu_full_es
+ n_shot: 5
+ - task: global_mmlu_full_fr
+ n_shot: 5
+ - task: global_mmlu_full_it
+ n_shot: 5
+ - task: global_mmlu_full_lt
+ n_shot: 5
+ - task: global_mmlu_full_nl
+ n_shot: 5
+ - task: global_mmlu_full_pl
+ n_shot: 5
+ - task: global_mmlu_full_pt
+ n_shot: 5
+ - task: global_mmlu_full_ro
+ n_shot: 5
+ - task: global_mmlu_full_sr
+ n_shot: 5
+ - task: global_mmlu_full_sv
+ n_shot: 5
+ - task: global_mmlu_full_tr
+ n_shot: 5
+ - task: global_mmlu_full_uk
+ n_shot: 5
+ - name: OpenBookQA
+ variants:
+ - task: openbookqa
+ n_shot: 0
+ - name: QA Wikidata
+ variants:
+ - task: bigbench_qa_wikidata_generate_until
+ n_shot: 10
+ - name: CommonsenseQA
+ variants:
+ - task: commonsense_qa
+ n_shot: 10
+ - name: COPA
+ variants:
+ - task: copa
+ n_shot: 0
+ - name: HellaSwag
+ variants:
+ - task: hellaswag
+ n_shot: 0
+ - task: hellaswag_ca
+ n_shot: 0
+ - task: hellaswag_da
+ n_shot: 0
+ - task: hellaswag_de
+ n_shot: 0
+ - task: hellaswag_es
+ n_shot: 0
+ - task: hellaswag_eu
+ n_shot: 0
+ - task: hellaswag_fr
+ n_shot: 0
+ - task: hellaswag_hr
+ n_shot: 0
+ - task: hellaswag_hu
+ n_shot: 0
+ - task: hellaswag_it
+ n_shot: 0
+ - task: hellaswag_nl
+ n_shot: 0
+ - task: hellaswag_pt
+ n_shot: 0
+ - task: hellaswag_ro
+ n_shot: 0
+ - task: hellaswag_sk
+ n_shot: 0
+ - task: hellaswag_sr
+ n_shot: 0
+ - task: hellaswag_sv
+ n_shot: 0
+ - task: hellaswag_uk
+ n_shot: 0
+ - name: PIQA
+ variants:
+ - task: global_piqa_completions_als_latn
+ n_shot: 0
+ - task: global_piqa_completions_bos_latn
+ n_shot: 0
+ - task: global_piqa_completions_bul_cyrl
+ n_shot: 0
+ - task: global_piqa_completions_cat_latn
+ n_shot: 0
+ - task: global_piqa_completions_ces_latn
+ n_shot: 0
+ - task: global_piqa_completions_deu_latn
+ n_shot: 0
+ - task: global_piqa_completions_ekk_latn
+ n_shot: 0
+ - task: global_piqa_completions_ell_grek
+ n_shot: 0
+ - task: global_piqa_completions_eng_latn
+ n_shot: 0
+ - task: global_piqa_completions_fin_latn
+ n_shot: 0
+ - task: global_piqa_completions_fra_latn_fran
+ n_shot: 0
+ - task: global_piqa_completions_glg_latn
+ n_shot: 0
+ - task: global_piqa_completions_hrv_latn
+ n_shot: 0
+ - task: global_piqa_completions_hun_latn
+ n_shot: 0
+ - task: global_piqa_completions_isl_latn
+ n_shot: 0
+ - task: global_piqa_completions_ita_latn
+ n_shot: 0
+ - task: global_piqa_completions_kat_geor
+ n_shot: 0
+ - task: global_piqa_completions_lit_latn
+ n_shot: 0
+ - task: global_piqa_completions_mkd_cyrl
+ n_shot: 0
+ - task: global_piqa_completions_nld_latn
+ n_shot: 0
+ - task: global_piqa_completions_nno_latn
+ n_shot: 0
+ - task: global_piqa_completions_nob_latn
+ n_shot: 0
+ - task: global_piqa_completions_pol_latn
+ n_shot: 0
+ - task: global_piqa_completions_por_latn_port
+ n_shot: 0
+ - task: global_piqa_completions_ron_latn
+ n_shot: 0
+ - task: global_piqa_completions_slk_latn
+ n_shot: 0
+ - task: global_piqa_completions_slv_latn
+ n_shot: 0
+ - task: global_piqa_completions_spa_latn_spai
+ n_shot: 0
+ - task: global_piqa_completions_srp_cyrl
+ n_shot: 0
+ - task: global_piqa_completions_swe_latn
+ n_shot: 0
+ - task: global_piqa_completions_tur_latn
+ n_shot: 0
+ - task: global_piqa_completions_ukr_cyrl
+ n_shot: 0
+ - task: piqa
+ n_shot: 10
+ - name: Social IQa
+ variants:
+ - task: social_iqa
+ n_shot: 0
+ - name: WSC273
+ variants:
+ - task: wsc273
+ n_shot: 0
+ - name: WinoGrande
+ variants:
+ - task: winogrande
+ n_shot: 0
+ - name: X-CSQA
+ variants:
+ - task: xcsqa_deu_Latn
+ n_shot: 0
+ - task: xcsqa_eng_Latn
+ n_shot: 0
+ - task: xcsqa_fra_Latn
+ n_shot: 0
+ - task: xcsqa_ita_Latn
+ n_shot: 0
+ - task: xcsqa_nld_Latn
+ n_shot: 0
+ - task: xcsqa_pol_Latn
+ n_shot: 0
+ - task: xcsqa_por_Latn
+ n_shot: 0
+ - task: xcsqa_spa_Latn
+ n_shot: 0
+ - name: XCOPA
+ variants:
+ - task: xcopa:et
+ n_shot: 0
+ - task: xcopa:it
+ n_shot: 0
+ - task: xcopa:tr
+ n_shot: 0
+ - name: Belebele
+ variants:
+ - task: belebele_bul_Cyrl
+ n_shot: 5
+ - task: belebele_ces_Latn
+ n_shot: 5
+ - task: belebele_dan_Latn
+ n_shot: 5
+ - task: belebele_deu_Latn
+ n_shot: 5
+ - task: belebele_ell_Grek
+ n_shot: 5
+ - task: belebele_eng_Latn
+ n_shot: 5
+ - task: belebele_est_Latn
+ n_shot: 5
+ - task: belebele_fin_Latn
+ n_shot: 5
+ - task: belebele_fra_Latn
+ n_shot: 5
+ - task: belebele_hrv_Latn
+ n_shot: 5
+ - task: belebele_hun_Latn
+ n_shot: 5
+ - task: belebele_ita_Latn
+ n_shot: 5
+ - task: belebele_lit_Latn
+ n_shot: 5
+ - task: belebele_lvs_Latn
+ n_shot: 5
+ - task: belebele_mlt_Latn
+ n_shot: 5
+ - task: belebele_nld_Latn
+ n_shot: 5
+ - task: belebele_nob_Latn
+ n_shot: 5
+ - task: belebele_pol_Latn
+ n_shot: 5
+ - task: belebele_por_Latn
+ n_shot: 5
+ - task: belebele_ron_Latn
+ n_shot: 5
+ - task: belebele_slk_Latn
+ n_shot: 5
+ - task: belebele_slv_Latn
+ n_shot: 5
+ - task: belebele_spa_Latn
+ n_shot: 5
+ - task: belebele_swe_Latn
+ n_shot: 5
+ - name: BoolQ
+ variants:
+ - task: boolq
+ n_shot: 10
+ - name: CoQA
+ variants:
+ - task: coqa
+ n_shot: 0
+ - name: LAMBADA
+ variants:
+ - task: lambada_openai
+ n_shot: 0
+ - name: SIB-200
+ variants:
+ - task: sib200_als_Latn
+ n_shot: 0
+ - task: sib200_bos_Latn
+ n_shot: 0
+ - task: sib200_bul_Cyrl
+ n_shot: 0
+ - task: sib200_cat_Latn
+ n_shot: 0
+ - task: sib200_ces_Latn
+ n_shot: 0
+ - task: sib200_dan_Latn
+ n_shot: 0
+ - task: sib200_deu_Latn
+ n_shot: 0
+ - task: sib200_ell_Grek
+ n_shot: 0
+ - task: sib200_eng_Latn
+ n_shot: 0
+ - task: sib200_est_Latn
+ n_shot: 0
+ - task: sib200_eus_Latn
+ n_shot: 0
+ - task: sib200_fin_Latn
+ n_shot: 0
+ - task: sib200_fra_Latn
+ n_shot: 0
+ - task: sib200_gle_Latn
+ n_shot: 0
+ - task: sib200_glg_Latn
+ n_shot: 0
+ - task: sib200_hrv_Latn
+ n_shot: 0
+ - task: sib200_hun_Latn
+ n_shot: 0
+ - task: sib200_isl_Latn
+ n_shot: 0
+ - task: sib200_ita_Latn
+ n_shot: 0
+ - task: sib200_kat_Geor
+ n_shot: 0
+ - task: sib200_lit_Latn
+ n_shot: 0
+ - task: sib200_lvs_Latn
+ n_shot: 0
+ - task: sib200_mkd_Cyrl
+ n_shot: 0
+ - task: sib200_mlt_Latn
+ n_shot: 0
+ - task: sib200_nld_Latn
+ n_shot: 0
+ - task: sib200_nob_Latn
+ n_shot: 0
+ - task: sib200_pol_Latn
+ n_shot: 0
+ - task: sib200_por_Latn
+ n_shot: 0
+ - task: sib200_ron_Latn
+ n_shot: 0
+ - task: sib200_slk_Latn
+ n_shot: 0
+ - task: sib200_slv_Latn
+ n_shot: 0
+ - task: sib200_spa_Latn
+ n_shot: 0
+ - task: sib200_srp_Cyrl
+ n_shot: 0
+ - task: sib200_swe_Latn
+ n_shot: 0
+ - task: sib200_tur_Latn
+ n_shot: 0
+ - task: sib200_ukr_Cyrl
+ n_shot: 0
+ - name: SQuAD v2
+ variants:
+ - task: squadv2
+ n_shot: 10
+ - name: FLORES200
+ variants:
+ - task: flores200:als_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:bos_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:bul_Cyrl-eng_Latn
+ n_shot: 4
+ - task: flores200:cat_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:ces_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:dan_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:deu_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:ell_Grek-eng_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-als_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-bos_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-bul_Cyrl
+ n_shot: 4
+ - task: flores200:eng_Latn-cat_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-ces_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-dan_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-deu_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-ell_Grek
+ n_shot: 4
+ - task: flores200:eng_Latn-est_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-eus_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-fin_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-fra_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-gle_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-glg_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-hrv_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-hun_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-isl_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-ita_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-kat_Geor
+ n_shot: 4
+ - task: flores200:eng_Latn-lit_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-lvs_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-mkd_Cyrl
+ n_shot: 4
+ - task: flores200:eng_Latn-mlt_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-nld_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-nob_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-pol_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-por_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-ron_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-slk_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-slv_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-spa_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-srp_Cyrl
+ n_shot: 4
+ - task: flores200:eng_Latn-swe_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-tur_Latn
+ n_shot: 4
+ - task: flores200:eng_Latn-ukr_Cyrl
+ n_shot: 4
+ - task: flores200:est_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:eus_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:fin_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:fra_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:gle_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:glg_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:hrv_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:hun_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:isl_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:ita_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:kat_Geor-eng_Latn
+ n_shot: 4
+ - task: flores200:lit_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:lvs_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:mkd_Cyrl-eng_Latn
+ n_shot: 4
+ - task: flores200:mlt_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:nld_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:nob_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:pol_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:por_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:ron_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:slk_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:slv_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:spa_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:srp_Cyrl-eng_Latn
+ n_shot: 4
+ - task: flores200:swe_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:tur_Latn-eng_Latn
+ n_shot: 4
+ - task: flores200:ukr_Cyrl-eng_Latn
+ n_shot: 4
+ - name: OpenSubtitles
+ variants:
+ - task: opensubtitles_multi40_bg_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_cs_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_da_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_de_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_el_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_bg
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_cs
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_da
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_de
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_el
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_es
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_et
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_fi
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_fr
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_hr
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_hu
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_it
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_lt
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_lv
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_nl
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_no
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_pl
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_pt
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_ro
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_sk
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_sl
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_sr
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_sv
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_tr
+ n_shot: 0
+ - task: opensubtitles_multi40_en_to_uk
+ n_shot: 0
+ - task: opensubtitles_multi40_es_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_et_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_fi_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_fr_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_hr_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_hu_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_it_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_lt_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_lv_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_nl_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_no_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_pl_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_pt_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_ro_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_sk_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_sl_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_sr_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_sv_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_tr_to_en
+ n_shot: 0
+ - task: opensubtitles_multi40_uk_to_en
+ n_shot: 0
+ - name: Language ID
+ variants:
+ - task: bigbench_language_identification_multiple_choice
+ n_shot: 10
+ - name: MultiBlimp
+ variants:
+ - task: multiblimp_bul
+ n_shot: 0
+ - task: multiblimp_cat
+ n_shot: 0
+ - task: multiblimp_ces
+ n_shot: 0
+ - task: multiblimp_dan
+ n_shot: 0
+ - task: multiblimp_deu
+ n_shot: 0
+ - task: multiblimp_ell
+ n_shot: 0
+ - task: multiblimp_eng
+ n_shot: 0
+ - task: multiblimp_est
+ n_shot: 0
+ - task: multiblimp_eus
+ n_shot: 0
+ - task: multiblimp_fin
+ n_shot: 0
+ - task: multiblimp_fra
+ n_shot: 0
+ - task: multiblimp_gle
+ n_shot: 0
+ - task: multiblimp_glg
+ n_shot: 0
+ - task: multiblimp_hbs
+ n_shot: 0
+ - task: multiblimp_hun
+ n_shot: 0
+ - task: multiblimp_isl
+ n_shot: 0
+ - task: multiblimp_ita
+ n_shot: 0
+ - task: multiblimp_kat
+ n_shot: 0
+ - task: multiblimp_lav
+ n_shot: 0
+ - task: multiblimp_lit
+ n_shot: 0
+ - task: multiblimp_mkd
+ n_shot: 0
+ - task: multiblimp_nld
+ n_shot: 0
+ - task: multiblimp_pol
+ n_shot: 0
+ - task: multiblimp_por
+ n_shot: 0
+ - task: multiblimp_ron
+ n_shot: 0
+ - task: multiblimp_slk
+ n_shot: 0
+ - task: multiblimp_slv
+ n_shot: 0
+ - task: multiblimp_spa
+ n_shot: 0
+ - task: multiblimp_sqi
+ n_shot: 0
+ - task: multiblimp_swe
+ n_shot: 0
+ - task: multiblimp_tur
+ n_shot: 0
+ - task: multiblimp_ukr
+ n_shot: 0
+ - name: IFEval
+ variants:
+ - task: ifeval
+ n_shot: 0
diff --git a/configs/weights/code-math.yaml b/configs/weights/code-math.yaml
new file mode 100644
index 0000000..2bcda48
--- /dev/null
+++ b/configs/weights/code-math.yaml
@@ -0,0 +1,24 @@
+version: 1
+name: Code & math emphasis
+weights:
+ Code: 0.2
+ Math: 0.2
+ Reasoning: 0.1
+ Knowledge: 0.125
+ Commonsense: 0.125
+ Reading: 0.15
+ Translation: 0.03333333333333333
+ Language: 0.03333333333333333
+ Instruction following: 0.03333333333333333
+# The selector starts on the original aggregate; these shares configure the alternative.
+aggregate: standard
+english_weights:
+ Code: 0.5
+ Math: 0.5
+ Reasoning: 0.5
+ Knowledge: 0.5
+ Commonsense: 0.5
+ Reading: 0.5
+ Translation: 0.5
+ Language: 0.5 # Unknown/mixed scores use the English side; the Croatian/Serbian pool assigned to srp_Latn uses the other side.
+ Instruction following: 0.5
diff --git a/configs/default.txt b/configs/weights/default.txt
similarity index 100%
rename from configs/default.txt
rename to configs/weights/default.txt
diff --git a/configs/weights/oellm.yaml b/configs/weights/oellm.yaml
new file mode 100644
index 0000000..ce6814a
--- /dev/null
+++ b/configs/weights/oellm.yaml
@@ -0,0 +1,24 @@
+version: 1
+name: Original
+weights:
+ Code: 0.15
+ Math: 0.15
+ Reasoning: 0.15
+ Knowledge: 0.15
+ Commonsense: 0.15
+ Reading: 0.15
+ Translation: 0.03333333333333333
+ Language: 0.03333333333333333
+ Instruction following: 0.03333333333333333
+# The selector starts on the original aggregate; these shares configure the alternative.
+aggregate: standard
+english_weights:
+ Code: 0.5
+ Math: 0.5
+ Reasoning: 0.5
+ Knowledge: 0.5
+ Commonsense: 0.5
+ Reading: 0.5
+ Translation: 0.5
+ Language: 0.5 # Unknown/mixed scores use the English side; the Croatian/Serbian pool assigned to srp_Latn uses the other side.
+ Instruction following: 0.5
diff --git a/docs/configuration.md b/docs/configuration.md
index 8421dd4..a65acae 100644
--- a/docs/configuration.md
+++ b/docs/configuration.md
@@ -1,29 +1,99 @@
-# Eval config reference
+# Configuration reference
-Edit `configs/oellm.yaml`, then load it in **Eval configuration → Load config**, or rebuild with `python3 -m app.build DATA.csv --config CONFIG.yaml`. Browser imports apply to every loaded real model and regenerate the synthetic comparison. If validation fails, the active config and scores remain unchanged. **Export config YAML** includes current category weights, English shares, and the selected aggregate; use that exported file when building to persist the choices.
+Quickdash separates four inputs: model results, a weighting profile, an optional named eval set, and the global eval catalogue. You can compare new result files with existing interpretation rules without writing a new list of required evals.
-## Shared config choices and results
+| Input | Purpose | Default |
+| --- | --- | --- |
+| CSV results | Raw measurements for models A and B | Shared files in `results/`, or browser imports |
+| Weighting profile | Category weights, English shares, default calculation | `configs/weights/oellm.yaml` |
+| Eval set | Optional expected evals and variants | `configs/sets/any-available.yaml` |
+| Global catalogue | Match tasks to evals, categories, metrics, normalization and languages | `configs/catalogue.yaml` |
-By default, the builder embeds all `.yaml` and `.yml` files directly in `configs/`. The single filename in `configs/default.txt` chooses the startup config, currently `oellm.yaml`. To choose another default, edit that line; no application code changes are needed. A missing selection file, invalid filename, or missing selected config stops the build before replacing output.
+## Choose weights and expected coverage
-`--configs-dir PATH` uses another directory and its `default.txt`. `--config PATH` overrides the startup selection; used alone it embeds only that config, or combined with `--configs-dir` it also embeds the directory’s other choices. Every config’s `name` must be unique. See [adding shared configs](../configs/README.md).
+Two profiles are supplied. **Original** is selected on startup; **Code & math emphasis** shifts weight toward those two categories.
-The page’s **Eval configuration** selector applies one config to every loaded model. Switching resets scoring weights, English shares, and the aggregate to that config’s values. Export edits before switching if you want to keep them. A temporary YAML upload appears as an uploaded choice for the current session; switching to a shared config replaces it. Reload restores the published defaults.
+| Category | Original | Code & math emphasis |
+| --- | ---: | ---: |
+| Code | 0.15 | 0.20 |
+| Math | 0.15 | 0.20 |
+| Reasoning | 0.15 | 0.10 |
+| Knowledge | 0.15 | 0.125 |
+| Commonsense | 0.15 | 0.125 |
+| Reading | 0.15 | 0.15 |
+| Translation | 0.1/3 | 0.1/3 |
+| Language | 0.1/3 | 0.1/3 |
+| Instruction following | 0.1/3 | 0.1/3 |
-`--results-dir results` embeds all CSVs directly in that directory. Each checkpoint label must occur in only one file, although one file may contain multiple models. Every config is validated against all shared inputs before a build replaces output. Missing coverage is permitted with warnings; invalid selected scores, duplicate config names, or invalid config schemas stop the build. Browser config changes are also validated against all loaded models and roll back on failure.
+Both profiles default to the standard calculation and store an English share of 0.5 per category, which applies only when an English-balance calculation is selected.
-An empty results directory, or a build without a CSV or results directory, starts with no models. Users can select a config and import CSVs afterward. **Clear models** clears the current comparison without removing config choices. Build metadata in `analysis.json` records input filenames and SHA-256 hashes under `sources`, and available config choices under `configurations`.
+The **Weighting profile** and **Eval set** selectors operate independently. Switching a profile resets category weights, English shares, and the calculation to that profile's values, leaving the eval set unchanged. Switching eval sets preserves your current weights and calculation. Export edits before switching profiles if you want to keep them.
+
+**Any available** uses recognized selected measurements shared by A and B. Measurements present on only one side generate comparison warnings and are excluded from both scores. Catalogue entries absent from both models do not generate warnings. The supplied freeform set explicitly excludes prompted Global PIQA pending validation; present data for it generates a **Not used** warning.
+
+**flagship-1** names the expected 45 evals and 403 task/shot requirements for the flagship comparison. A required measurement missing from either or both models generates a warning. Extra selected measurements are excluded with warnings, but remain inspectable in **Eval configuration**. The dashboard labels the score **INCOMPLETE**, shows shared/required coverage, and redistributes weights across the shared subset. Do not interpret an incomplete score as covering the full named set. Exact comparison identity still includes metric, filter, shots, harness, and backend; incompatible protocols cannot satisfy shared coverage just by sharing a task name.
+
+A named set can require a whole eval, or exact tasks with optional shot counts:
+
+```yaml
+version: 1
+name: Example required set
+mode: fixed
+evals:
+ - name: Example eval
+ variants:
+ - {task: example_en, n_shot: 0}
+ - {task: example_fr, n_shot: 0}
+```
+
+Omitting `variants` requires at least one shared selected measurement for that eval and permits all its variants. Omitting `n_shot` accepts any shot count, but A and B must still match each other's protocol. Required tasks must match the named catalogue eval and its `select`/`shots` restrictions. Duplicate or overlapping requirements, unknown eval names, and unknown fields are errors. Metric and normalization choices belong only in the catalogue. Eval sets never contain weights.
+
+Freeform mode needs no required-eval list. It can optionally exclude evals by their exact catalogue names:
+
+```yaml
+version: 1
+name: Any available
+mode: available
+exclude:
+ - Global PIQA (prompted)
+```
+
+`exclude` is allowed only in available mode and must contain unique names. An excluded name absent from the active catalogue is harmless, so a set can be reused with another catalogue. Fixed sets already exclude everything outside their membership. Exclusions warn whenever matching eval data is present, including rows containing only an alternate metric. Alternate fields or summary children of a selected eval do not generate unused-eval warnings merely because a preferred field or summary is selected.
+
+A weighting profile works with either mode:
+
+```yaml
+version: 1
+name: Reasoning weights
+weights: {Reasoning: 1}
+aggregate: standard
+english_weights: {Reasoning: 0.5}
+```
+
+Weights must be nonnegative and sum to 1. Empty categories have their weight redistributed. A catalogue category omitted from the profile receives zero weight and produces a warning when shared measurements use it. The editor exposes that zero so it can be assigned weight. Weight profiles never list evals. All three YAML types accept optional `notes` (a list of strings), require `version: 1` and a nonempty `name`, and reject unknown fields.
+
+## Build defaults and browser imports
+
+The builder embeds YAML profiles directly in `configs/weights/` and sets directly in `configs/sets/`. Each directory's `default.txt` contains one YAML filename selected on startup. Missing or invalid defaults stop the build before replacing output. Names must be unique within each selector.
+
+- `--catalogue PATH` chooses the global interpretation file.
+- `--weights PATH` chooses a profile; used alone, it embeds only that profile. `--weights-dir DIR` offers the profiles in another directory and uses its `default.txt` unless an explicit profile is supplied.
+- `--eval-set PATH` and `--sets-dir DIR` work the same way for eval sets.
+- `--results-dir DIR` embeds CSVs directly in that directory. A checkpoint label may occur in only one file; one file can contain multiple models.
+
+Every offered set is validated against the catalogue and every profile before writing output. Raw results are classified once by the global catalogue; changing sets only changes comparison membership. An empty results directory, or no CSV input, starts without models.
+
+Under **Eval configuration**, load or export the catalogue, weights, and eval set separately. Uploaded choices are temporary; reload restores published defaults. Catalogue imports reinterpret all loaded real models and regenerate the synthetic comparison. Invalid imports preserve the previous models and settings. If a new catalogue does not contain the active named set's evals, switch to Any available before loading it. **Clear models** retains settings.
+
+`analysis.json` records the catalogue, selected profile and set, available `profiles` and `suites`, and source filenames/hashes. It also contains the resolved internal `scheme` used for arithmetic; that combined object is not a YAML input format. Generated `catalogue.yaml`, `weights.yaml`, and `eval-set.yaml` record the build inputs.
## Complete small example
-This config selects `acc_norm` for two explicit language variants of a four-choice eval. Additional categories and evals follow the same structure.
+This catalogue selects `acc_norm` for two explicit language variants of a four-choice eval. Additional categories and evals follow the same structure.
```yaml
version: 1
name: Example scoring
-weights:
- Reasoning: 1.0
-
evals:
- name: Example eval
category: Reasoning
@@ -56,19 +126,19 @@ For an input row with `value=0.625`, the raw score is 62.5 and the normalized sc
| Field | Meaning |
|---|---|
| `name` | Unique display name of the eval, grouping its variants. |
-| `category` | A key in `weights`. Every weighted category must have at least one eval. |
+| `category` | Category label used by weighting profiles and breakdowns. |
| `match` | Exactly one of `{name: exact task name}` or `{regex: 'full-match pattern'}`. Unmatched tasks are excluded and listed in Warnings; overlapping matches are an error. |
| `metric` | Exact CSV scoring field, such as `acc`, `acc_norm`, or `python_pass@1`. |
| `filter` | Exact CSV filter string, including `""` if empty. |
| `shots` | Optional nonnegative integer selecting the CSV `n_shot`. Omit to include all shot settings. |
| `select` | Optional name/regex rule restricting which matched tasks contribute. Useful for selecting summaries while retaining child-task audits. |
| `score.scale` | Raw metric's upper scale: 1 for fractional accuracy; 100 for percentage or chrF scores. Selected values must be finite and within 0..scale. |
-| `warning` | Optional nonempty text describing an unresolved scoring assumption. Appears once for all models in Warnings and in this eval’s configuration details; it does not change scores. |
+| `warning` | Optional nonempty text describing an unresolved scoring assumption. Appears once in Warnings when present in the selected comparison and in this eval’s configuration details; it does not change scores. |
| `normalize` | Optional object with `min`, `max`, and optional `clip`, `basis`, `note`, and `sources`. Thresholds are fractions after division by `score.scale`. |
Use the common Python/JavaScript regex subset: literal text, character classes, alternatives, groups, and ordinary quantifiers. Patterns match the entire task name. Named groups and lookbehind are rejected. Language extraction does not use these patterns.
-The supplied OELLM config gives Code, Math, Reasoning, Knowledge, Commonsense, and Reading a weight of 0.15 each; Translation, Language, and Instruction following each receive 0.1/3. Category weights must be nonnegative and sum to 1. Names, metrics, and task strings are case-sensitive. Unknown config fields are rejected to catch typos. `version` must be 1; `name` labels the active config. Optional top-level `notes` is a list of strings.
+The Original weighting profile gives Code, Math, Reasoning, Knowledge, Commonsense, and Reading a weight of 0.15 each; Translation, Language, and Instruction following each receive 0.1/3. Category weights must be nonnegative and sum to 1. Names, metrics, and task strings are case-sensitive. Unknown config fields are rejected to catch typos. `version` must be 1; `name` labels the active config. Optional top-level `notes` is a list of strings.
## Normalization and contributions
@@ -97,14 +167,17 @@ The supplied config applies the following baselines. Per-eval `normalize.sources
| 1/7 | SIB-200: seven topic labels |
| 1/11 | Language ID: 11 candidate names per question, despite 1,000 languages in the corpus |
| 0 | AIME24 and AIME25: no chance correction |
+| 0.1055 | JEEBench: shared 10.55% baseline; the paper reports approximately 10.5% |
| ≈1/4 | ARC Challenge: initial approximation, including translated variants |
-| ≈0.2501613 | ARC Easy: average 1/choice_count across the published 2,376-question test split |
+| 1/4 | ARC Easy: conventional approximation despite a few questions with different option counts |
+
+The choice-based baselines model uniform *valid* guesses; JEEBench uses the paper's mixed-format guessing policy described below. These are not measured random-language-model or majority-class baselines. AIME24 and AIME25 use a zero floor without a uniform-integer guessing correction. ARC Easy uses 0.25 for consistency; the full published split has mean random accuracy approximately 0.2501613. Sources describe task definitions, but the CSV does not pin the exact run's dataset revision.
-These baselines model uniform *valid* guesses, not a random language model or a majority-class predictor. AIME24 and AIME25 use a zero floor without a uniform-integer guessing correction. ARC Easy assumes the full published test split, whose size matches this export. Sources describe task definitions, but the CSV does not pin the exact run's dataset revision.
+ARC Challenge uses an approximate 25% baseline. In its published test split, 1,165 of 1,172 questions have four options, four have three, and three have five, giving an exact mean of about 25.0156%. The approximation is also applied to translated variants, whose individual choice counts have not all been audited. JEEBench uses a 10.55% overall random baseline to match the shared scoring policy; [Table 2 of its paper](https://aclanthology.org/2023.emnlp-main.468.pdf#page=5) reports approximately 10.5%. This combines single-choice guessing and random option subsets with partial credit, assigning zero expected score to integer and numeric answers. It assumes the full 515-question benchmark with those scoring rules.
-ARC Challenge uses an approximate 25% baseline. In its published test split, 1,165 of 1,172 questions have four options, four have three, and three have five, giving an exact mean of about 25.0156%. The approximation is also applied to translated variants, whose individual choice counts have not all been audited. JEEBench remains unresolved and uncorrected because it mixes single-choice, multiple-answer, integer, and numeric questions. AMC23 is open-ended in the selected evaluator: the original contest's answer options are removed. Code generation, translation chrF, overlap F1, and other open-ended exact-match tasks do not receive an invented chance baseline.
+AMC23 is open-ended in the selected evaluator: the original contest's answer options are removed. Code generation, translation chrF, overlap F1, and other open-ended exact-match tasks do not receive an invented chance baseline.
-`normalize.basis` may be `uniform_choice`, `uniform_integer`, `not_applicable`, or `unresolved`. It documents the rationale; `min`, `max`, and `clip` control the actual calculation. Optional `sources` is a list of HTTP(S) URLs, and `note` is free text. Set `min: 0` and `max: 1` to disable correction. An optional eval-level `warning` string appears in the Warnings tab once for all models and in the eval configuration details. Remove it when the concern is resolved; it does not change selection or arithmetic. The supplied config uses it for translation calibration and the SIB-200 metric exception. Exported YAML preserves config values and notes; YAML comments are not retained.
+`normalize.basis` may be `uniform_choice`, `uniform_integer`, `not_applicable`, or `unresolved`. It documents the rationale; `min`, `max`, and `clip` control the actual calculation. Optional `sources` is a list of HTTP(S) URLs, and `note` is free text. Set `min: 0` and `max: 1` to disable correction. An optional eval-level `warning` string appears in the Warnings tab once for all models and in the eval configuration details. Remove it when the concern is resolved; it does not change selection or arithmetic. The supplied config uses it for translation calibration and the Croatian/Serbian language-grouping approximation. Exported YAML preserves config values and notes; YAML comments are not retained.
`acc_norm` in lm-eval refers to choosing answers using length-normalized likelihoods; it does **not** remove chance accuracy. Chance correction here is applied to each selected variant's aggregate score before averaging evals. Clipping after aggregation is not equivalent to clipping individual items, and a mixture of corrected and uncorrected metrics is still a provisional composite.
@@ -147,11 +220,19 @@ The final `eng_Latn → spa_Latn` is a single pair label. Each pair expands into
## Scoring choices and consistency warnings
-The supplied config prefers `acc_norm` over `acc` when both exist for the selected protocol. The metric remains explicit in `metric`; missing fields never silently fall back to another metric. SIB-200 is an explicit exception using `acc`: 34 of its 36 exported `acc_norm` values are exactly 0.25, so an eval-level `warning` requests investigation before changing that choice. Length-normalized option scoring and chance normalization are separate operations.
+The supplied config prefers `acc_norm` over `acc` when both exist for the selected protocol. The metric remains explicit in `metric`; missing fields never silently fall back to another metric. SIB-200 explicitly uses `acc`. A YAML comment explains that `acc_norm` is not reliable/useful in this export: 34 of its 36 values are exactly 0.25. This selection does not generate a config warning. Length-normalized option scoring and chance normalization are separate operations.
+
+For each selected real model, the dashboard compares the sets of selected settings per task within each eval: `n_shot`, `metric`, `filter`, `harness`, and `backend`. Different sets generate one warning naming the settings, with expandable lists of affected tasks and their explicit language assignments (source → target for translation). Identical sets across tasks are consistent even if each task has multiple settings. Excluded alternate metrics, summary children, and protocols do not trigger this check. The current data has mismatches for MGSM (0/5 shots), ARC Challenge (0/10), and PIQA (0/10). These warnings do not exclude scores; use the YAML selection rules to choose comparable protocols after reviewing coverage. Fields absent from the CSV, such as prompt templates or dataset revisions, cannot be compared.
+
+Translation metric preference is **chrF++ > chrF > BLEU**. The catalogue records an explicit selection based on the available, identified fields: FLORES200 uses `chrf++` with filter `rescored`; OpenSubtitles uses `chrf` with filter `none`. FLORES200's export has separate `chrf` and `chrf++` rows, but does not record a rescoring signature. Do not infer the variant from a generic “chrF” label alone.
+
+The referenced [OpenSubtitles task](https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/opensubtitles_multi40/_opensubtitles_multi40_common.yaml) selects the harness's `chrf` aggregation. The [harness implementation](https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/api/metrics.py) uses SacreBLEU defaults: character order 6, word order 0, beta 2. That is plain chrF; chrF++ adds word n-grams through word order 2. Update the catalogue if a better identified metric becomes available. A model missing the configured metric warns and is excluded; the browser does not silently compare different metrics or substitute BLEU.
+
+Both translation evals retain native 0–100 points (`score.scale: 100`, `normalize: {min: 0, max: 1}`), without chance correction. This accepted policy is documented in their notes rather than flagged as an unresolved caveat. A shared numerical range does not imply equal difficulty across metrics or language pairs.
-For each real model, the dashboard compares the sets of selected settings per task within each eval: `n_shot`, `metric`, `filter`, `harness`, and `backend`. Different sets generate one warning naming the settings, with expandable lists of affected tasks and their explicit language assignments (source → target for translation). Identical sets across tasks are consistent even if each task has multiple settings. Excluded alternate metrics, summary children, and protocols do not trigger this check. The current data has mismatches for MGSM (0/5 shots), ARC Challenge (0/10), and PIQA (0/10). These warnings do not exclude scores; use the YAML selection rules to choose comparable protocols after reviewing coverage. Fields absent from the CSV, such as prompt templates or dataset revisions, cannot be compared.
+Completion-based PIQA retains `acc_norm` and a 0.5 baseline. **Global PIQA (prompted)** has a separate interpretation rule for `exact_match` / `strict_match`, a provisional zero floor, and a warning that normalization and metric selection have not been validated. It is excluded by the supplied Any available set and omitted from flagship-1. Its data generates a **Not used** notice while excluded. Including it in a custom set surfaces its scoring caveat; configuration inspection always shows the attached warning.
-Both translation evals use chrF, which is bounded by 0 and 100 in [SacreBLEU](https://github.com/mjpost/sacrebleu/blob/master/sacrebleu/metrics/chrf.py). `score.scale: 100` and `normalize: {min: 0, max: 1}` preserve native chrF points. The open question is calibration against accuracy metrics and between language pairs, including a meaningful chance floor; a shared numerical range does not settle those questions. Each translation config carries an editable `warning` while this remains unresolved. The composite continues to include translation at the configured weight.
+With these defaults, MultiBlimp's Croatian/Serbian pooling is the only configured caveat for included evals. Runtime warnings for coverage, inconsistent settings, missing fields, and unused data still apply.
MMLU and Global MMLU have separate eval configs and aggregates. MMLU selects only `mmlu`; Global MMLU selects `global_mmlu_full_[a-z]+` language summaries. Subject-level rows stay available for inspection but are excluded from both composites. Both evals retain the four-choice 25% floor and each gets one equal share of Knowledge.
@@ -167,7 +248,7 @@ The prominent **Score calculation** panel offers three modes:
All modes apply the configured category weights last. The selector affects both model score cards, category/eval contributions, effective weights, and weighted delta bars. Raw comparison columns and descriptive language/category breakdowns retain their meanings.
-Optional top-level YAML fields persist the choice and shares:
+Optional top-level fields in the **weighting profile** persist the choice and shares:
```yaml
aggregate: english_eval
@@ -204,10 +285,10 @@ Per-eval balancing preserves equal eval weights, but does not guarantee English
## Missing comparison data
-Model comparisons use the intersection of selected measurements, matching task, metric, extraction filter, shot count, harness, and backend. A measurement present only in A or only in B is excluded from **both** score calculations and listed in a comparison-coverage warning. If no measurements remain for an eval, that eval is excluded from both scores and the remaining evals share its category weight equally. An empty category is excluded and remaining category weights are rescaled proportionally to sum to 1. No shared scores leaves the composite unavailable. Config coverage warnings remain visible for every loaded model.
+Model comparisons use the intersection of selected measurements, matching task, metric, extraction filter, shot count, harness, and backend. A measurement present only in A or only in B is excluded from **both** score calculations and listed in a comparison-coverage warning. If no measurements remain for an eval, that eval is excluded from both scores and the remaining evals share its category weight equally. An empty category is excluded and remaining category weights are rescaled proportionally to sum to 1. No shared scores leaves the composite unavailable. Data warnings apply to the selected models; the configuration audit retains all loaded real exports.
-The weights table shows effective weights after exclusions; the editor preserves the configured weights. Configured-but-missing evals and unconfigured tasks remain visible in Warnings. The build's model summaries are per-model summaries of available data; the interactive dashboard applies the common comparison coverage when two models are selected.
+The weights table shows effective weights after exclusions; the editor preserves the configured weights. Missing named-set requirements and unconfigured tasks remain visible in Warnings. Unused global catalogue rules are allowed. The build's model summaries are per-model summaries of available data; the interactive dashboard applies the common comparison coverage when two models are selected.
## Data validation and failure behavior
@@ -220,12 +301,16 @@ Builds and browser imports use the same CSV parser. Required columns are `checkp
| Selected score is blank, nonnumeric, nonfinite, or outside `0..score.scale` | Reject with CSV row, model, task, and metric in the error. Decimal and scientific notation are accepted; booleans and hexadecimal values are not scores. |
| New model uses an already loaded checkpoint name or the reserved synthetic-demo name | Reject the import. Give the model a distinct checkpoint label. |
| Task has no eval config, or lacks the configured metric/filter/shots | Warn and exclude from scoring. Alternate metrics remain inspectable and never silently substitute for the configured metric. |
-| Eval is absent, or measurements match only one of the compared models | Warn; use only shared measurements and redistribute weights as described above. |
+| Measurements match only one of the compared models | Warn; use only shared measurements and redistribute weights. |
+| Named set requirement is missing from either or both models | Warn and mark the set incomplete; compare the shared subset. |
+| Eval/task data is not selected by a named set or freeform exclusion | Show **Not used** and exclude it, even if only an alternate metric exists; keep it in the audit. |
+| Catalogue rule has no results in either model | No warning unless required by the selected named set. |
+| Shared category has no profile weight | Warn; zero contribution until a weight is assigned. |
| Selected task has no explicit language assignment | Warn and retain the score. Language views show Unknown; English-balance modes use the English fallback. Known mixed-language pools use the documented fallback without claiming a resolved single language. |
| Selected variants use inconsistent scoring settings | Warn and retain scores so the reviewer can inspect the protocols. |
| Matched A/B measurements report different positive `n_samples` | Warn and retain scores; sample count does not determine score weights. Review whether dataset coverage is comparable. |
| Supplied `n_samples` is not a positive integer | Warn and retain scores; omit it from sample-count comparisons. Absent/blank sample counts are allowed. |
-| Eval has a YAML `warning` | Show the caveat once for all models and alongside that eval's configuration. |
+| Eval has a YAML `warning` | Show the caveat in Warnings when the eval has shared comparison data. Always show it in its configuration details. Excluded evals get **Not used**, not an active scoring caveat. |
| No shared data with positive category weight | Show an unavailable composite (`—`), never an invented zero. |
Invalid model/config imports leave the active models, settings, and scores unchanged, including multi-model files where a later model is invalid. Build input validation completes before existing output files are replaced. Warnings are calculated from the current config and loaded results; fixing or removing the underlying issue removes its warning. Fields not used for scoring, such as source paths and standard errors, remain audit information; their presence is not a guarantee that dataset revisions or prompts match. Invalid values in excluded alternate metrics remain visible but are not normalized using the selected metric's scale.
diff --git a/docs/development.md b/docs/development.md
index 56e120c..a4684b9 100644
--- a/docs/development.md
+++ b/docs/development.md
@@ -8,7 +8,7 @@ Run commands from the repository root. Python 3.8+ and Node.js 18+ build the das
| --- | --- |
| `app/` | Python builder, browser application, HTML template, and bundled YAML parser. |
| `tests/` | Public contract tests, browser checks, and optional private-export regressions. |
-| `configs/` | Selectable scoring YAML files and `default.txt`. |
+| `configs/` | Global catalogue, weighting profiles, optional named sets, and fictional examples. |
| `results/` | Public CSV exports contributed to the shared dashboard. |
| `examples/` | Fictional input data for the demo and tests. |
| `docs/` | Configuration and contributor documentation. |
@@ -18,10 +18,11 @@ Run commands from the repository root. Python 3.8+ and Node.js 18+ build the das
```sh
python3 -m app.build --results-dir results --output output/shared
-python3 -m app.build examples/scores.csv --config configs/example.yaml --output output/demo
+python3 -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \
+ --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml --output output/demo
```
-Open each generated `index.html` in a browser. The shared build embeds every YAML in `configs/` and selects the filename in `configs/default.txt`. To try another directory, pass `--configs-dir PATH`. To build with just one scoring config, pass `--config PATH`. Supplying both selects that explicit config and also includes choices from the directory.
+Open each generated `index.html` in a browser. The shared build embeds the global catalogue, profiles from `configs/weights/`, and sets from `configs/sets/`. Each selector uses its directory's `default.txt`. See the [build flags](configuration.md#build-defaults-and-browser-imports) to supply other inputs.
Check the views affected by your change, including their warnings and failed-input behavior. Generated output is self-contained; do not commit it. Private exports belong in `data/`, never in the public `results/` directory.
@@ -31,12 +32,12 @@ These tests use small fixtures and run from a fresh checkout without private eva
```sh
python3 -m unittest tests.test_data
-node --test tests/test_data.cjs tests/test_yaml.cjs
+node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs
```
They cover input validation, normalization, warning/exclusion behavior, failed-build preservation, Python/JavaScript parity, hierarchy sorting, and deterministic randomized scoring comparisons against an independent calculation.
-The public browser suite also checks empty startup, shared models, config selection, temporary uploads, rollback, and that file imports make no network requests. It requires Node.js 22+ and Chrome. Start an isolated browser session, then run the suite in another terminal:
+The public browser suite also checks empty startup, shared models, independent profile/set selection, missing requirements, temporary uploads, rollback, and that file imports make no network requests. It requires Node.js 22+ and Chrome. Start an isolated browser session, then run the suite in another terminal:
```sh
"/Applications/Google Chrome.app/Contents/MacOS/Google Chrome" \
@@ -55,9 +56,9 @@ Adjust the Chrome executable path for your platform. Stop that isolated Chrome p
The full-export regression tests require the original private CSV at `data/v2zloss_86k.flag-evals-436.tasks.csv` and its freshly built output:
```sh
-python3 -m app.build data/v2zloss_86k.flag-evals-436.tasks.csv --config configs/oellm.yaml
+python3 -m app.build data/v2zloss_86k.flag-evals-436.tasks.csv
python3 -m unittest tests.test_analysis tests.test_data
-node --test tests/test_app.cjs tests/test_english.cjs tests/test_data.cjs tests/test_yaml.cjs
+node --test tests/test_app.cjs tests/test_english.cjs tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs
```
With the isolated Chrome session above running, use `node tests/test_browser.mjs` for the full-export browser checks: filtering, sortable hierarchies, scroll preservation, warnings, model swapping, header alignment, and mobile layouts. Screenshots go into the ignored `output/` directory.
diff --git a/results/README.md b/results/README.md
index ad732c3..8a7d8ea 100644
--- a/results/README.md
+++ b/results/README.md
@@ -11,13 +11,13 @@ checkpoint,task,metric,filter,n_shot,harness,backend,value
method-a-100k,my_eval_en,acc_norm,none,0,lm-eval,vllm,0.72
```
-The row above illustrates the format; `my_eval_en` needs a matching eval entry and an explicit language assignment in the chosen YAML config. See [CSV requirements](../docs/configuration.md#data-validation-and-failure-behavior) and [adding configurations](../configs/README.md).
+The row above illustrates the format; `my_eval_en` needs a matching eval entry and an explicit language assignment in the global catalogue. See [CSV requirements](../docs/configuration.md#data-validation-and-failure-behavior) and [adding configurations](../configs/README.md).
- Use a distinct `checkpoint` label for each model/run. A file may contain several models, but a label cannot occur in two files. Replace a model’s existing file when updating it, or give a new run a new label.
- Keep raw metric values in their original scale. The config chooses the metric and applies normalization.
- Only CSV files directly in this directory are loaded; subdirectories are not scanned.
-- Malformed files, duplicate labels, and invalid selected scores stop the build. Missing eval coverage or an unconfigured task appears as a warning in the dashboard; it is excluded from the relevant score.
-- Review each configuration’s warnings when comparing results. Different configs can select different metrics, evals, and normalizations from the same CSVs.
+- Malformed files, duplicate labels, and invalid selected scores stop the build. One-sided coverage, missing named-set requirements, or an unconfigured task appears as a warning in the dashboard; it is excluded from the relevant score.
+- Review the selected comparison’s warnings. Any available needs no expected-eval list; a named set checks missing and extra measurements. Weights can be changed independently of that choice.
To check a contribution locally:
diff --git a/tests/test_analysis.py b/tests/test_analysis.py
index 5f8b5ef..69e2b92 100644
--- a/tests/test_analysis.py
+++ b/tests/test_analysis.py
@@ -2,12 +2,12 @@
import unittest
from pathlib import Path
from app.build import classify, summarize
-from app.config_engine import task_language, validate_config, normalize_score, match_task, load_config
+from app.config_engine import task_language, validate_config, normalize_score, match_task, shared_config, load_catalogue
from copy import deepcopy
import subprocess
import tempfile
ROOT=Path(__file__).resolve().parent.parent
-CONFIG=load_config(ROOT/'configs/oellm.yaml')
+CONFIG=shared_config('resolve',value=[load_catalogue(ROOT/'configs/catalogue.yaml'),shared_config('suite',ROOT/'configs/sets/any-available.yaml'),shared_config('weights',ROOT/'configs/weights/oellm.yaml')])
def resolve_languages(tasks):return [task_language(t,CONFIG) for t in tasks]
DATA=json.loads((ROOT/'output/analysis.json').read_text())
class AnalysisTests(unittest.TestCase):
@@ -25,7 +25,7 @@ def test_english_weighting_language_fallback(self):
def test_complete_source_coverage(self):
self.assertEqual(len(DATA['rows']),2124)
self.assertTrue(all(r['eval'] for r in DATA['rows']))
- self.assertEqual(len(DATA['models'][0]['evals']),45)
+ self.assertEqual(len([e for e in DATA['models'][0]['evals'] if not e['excluded']]),45)
def test_selected_measurements_unique(self):
rr=[r for r in DATA['rows'] if r['selected']]
keys=[tuple(r[k] for k in ['task','metric','filter','n_shot','harness','backend']) for r in rr]
@@ -103,8 +103,9 @@ def test_explicit_mapping_has_no_suffix_inference(self):
def test_alternate_config_build_and_javascript_parity(self):
c=deepcopy(CONFIG);c['evals'][0]['normalize']={'min':.25,'max':1};c['aggregate']='english_category'
with tempfile.TemporaryDirectory() as tmp:
- path=Path(tmp);yaml=subprocess.check_output(['node','-e',"process.stdout.write(require('./app/eval_config.js').serializeConfig(JSON.parse(require('fs').readFileSync(0,'utf8'))))"],input=json.dumps(c),text=True,cwd=ROOT);(path/'config.yaml').write_text(yaml)
- subprocess.run(['python3','-m','app.build',str(ROOT/'data/v2zloss_86k.flag-evals-436.tasks.csv'),'--config',str(path/'config.yaml'),'--output',str(path/'result')],check=True,capture_output=True)
+ from tests.test_data import inputs
+ path=Path(tmp);kw=inputs(path,c)
+ subprocess.run(['python3','-m','app.build',str(ROOT/'data/v2zloss_86k.flag-evals-436.tasks.csv'),'--catalogue',str(kw['catalogue_path']),'--weights',str(kw['weights_path']),'--eval-set',str(kw['suite_path']),'--output',str(path/'result')],check=True,capture_output=True)
data=json.loads((path/'result/analysis.json').read_text())
self.assertNotEqual(data['models'][0]['score'],DATA['models'][0]['score'])
script="const fs=require('fs'),e=require('./app/eval_config.js');const d=JSON.parse(fs.readFileSync(process.argv[1]));console.log(JSON.stringify({rows:e.auditRows(d.rows,d.scheme),metadata:d.metadata.map(m=>e.taskLanguage(m.task,d.scheme))}));"
@@ -131,19 +132,22 @@ def test_unconfigured_rows_are_excluded(self):
self.assertIsNone(result['score_100']);self.assertIn('No eval config',result['decision'])
def test_chance_baselines_and_raw_score_preservation(self):
evals={e['name']:e for e in CONFIG['evals']}
- expected={'SIB-200':1/7,'Language ID':1/11,'Social IQa':1/3,'HellaSwag':.25,'PIQA':.5,'CommonsenseQA':.2,'AIME24':0,'AIME25':0,'ARC Easy':(2365/4+7/3+4/5)/2376}
+ expected={'SIB-200':1/7,'Language ID':1/11,'Social IQa':1/3,'HellaSwag':.25,'PIQA':.5,'CommonsenseQA':.2,'AIME24':0,'AIME25':0,'JEEBench':.1055,'ARC Easy':.25}
for name,chance in expected.items():
e=evals[name];self.assertEqual(e['normalize']['min'],chance)
self.assertAlmostEqual(normalize_score(chance*e['score']['scale'],e)[1],0)
self.assertTrue(e['normalize']['sources'])
self.assertEqual(evals['ARC Challenge']['normalize']['min'],.25)
self.assertIn('approximation',evals['ARC Challenge']['normalize']['note'])
- self.assertEqual(evals['JEEBench']['normalize']['basis'],'unresolved')
+ self.assertNotEqual(evals['JEEBench']['normalize'].get('basis'),'unresolved')
self.assertEqual(evals['AMC23']['normalize']['min'],0)
c=deepcopy(CONFIG)
for e in c['evals']:e['normalize']={'min':0,'max':1}
raw=classify(DATA['rows'],c)
- self.assertAlmostEqual(summarize(raw,c)[0]['score'],50.04555748071083)
- self.assertLess(DATA['models'][0]['score'],50.04555748071083)
+ scoped=shared_config('scope',value=[raw,DATA['suite']])['rows']
+ from statistics import mean
+ expected_raw=sum(weight*mean(mean(r['raw_score_100'] for r in scoped if r['eval']==e) for e in {r['eval'] for r in scoped if r['category']==category}) for category,weight in c['weights'].items())
+ self.assertAlmostEqual(summarize(scoped,c)[0]['score'],expected_raw)
+ self.assertLess(DATA['models'][0]['score'],expected_raw)
self.assertEqual([r['raw_score_100'] for r in raw],[r['raw_score_100'] for r in DATA['rows']])
if __name__=='__main__': unittest.main()
diff --git a/tests/test_app.cjs b/tests/test_app.cjs
index 1c7e428..9e4dbe1 100644
--- a/tests/test_app.cjs
+++ b/tests/test_app.cjs
@@ -1,7 +1,8 @@
const assert=require('node:assert/strict'),fs=require('node:fs');
const {parseCSV,selectRows,totals,pairRows,synthetic}=require('../app/app.js');
const data=JSON.parse(fs.readFileSync(__dirname+'/../output/analysis.json')),scheme=data.scheme;
-const selected=selectRows(data.rows,scheme);
+const {scopeRows,inSuite}=require('../app/suite_config.js');
+const selected=scopeRows(selectRows(data.rows,scheme),data.suite).rows;
assert.equal(selected.length,403);
assert.ok(Math.abs(totals(selected,scheme,scheme.weights).score-data.models[0].score)<1e-10);
const withoutIF=totals(selected.filter(r=>r.eval!=='IFEval'),scheme,scheme.weights);
@@ -25,9 +26,10 @@ console.log('JS checks passed: scoring, missing data, protocol matching, CSV par
const {auditRows,buildCatalogue}=require('../app/app.js');
const audited=auditRows(data.rows,scheme);
assert.equal(audited.length,2124);
-assert.equal(audited.filter(r=>r.selected).length,403);
+assert.equal(scopeRows(audited,data.suite).rows.length,403);
+assert.equal(audited.filter(r=>r.selected).length,435);
const catalogue=buildCatalogue(audited,scheme);
-assert.equal(catalogue.length,45);
+assert.equal(catalogue.length,46);
assert.equal(catalogue.reduce((n,f)=>n+f.tasks.length,0),1556);
assert.equal(catalogue.flatMap(f=>f.tasks).reduce((n,t)=>n+t.rows.length,0),2124);
const he=catalogue.find(f=>f.name==='HumanEval').tasks[0];
@@ -75,7 +77,7 @@ assert.deepEqual(languageRoles({task:'not-a-known-task'},metadata),[{language:'U
// Raw deltas stay independent of chance normalization; weighted deltas reconcile.
const {normalizeScore,validateConfig,taskLanguage,matchTask}=require('../app/eval_config.js');
const adjusted=structuredClone(scheme);adjusted.evals[0].normalize={min:.25,max:1};
-const ar=selectRows(data.rows,adjusted),br=synthetic(ar,adjusted),pp=pairRows(ar,br);
+const ar=scopeRows(selectRows(data.rows,adjusted),data.suite).rows,br=synthetic(ar,adjusted),pp=pairRows(ar,br);
const rawPairs=pairRows(selected,fake);
assert.deepEqual(pp.map(r=>r.delta),rawPairs.map(r=>r.delta));
const contributions=comparisonRows(pp,ar,adjusted,adjusted.weights,'variant','weighted','descending');
@@ -116,16 +118,16 @@ assert.equal(breakdownAggregate([...pairs,pairs[0]]).a,breakdownAggregate(pairs)
console.log('Hierarchy checks passed: category/eval/language and language/category/eval ordering, translation directions/pairs, filtered branches, unique aggregates.');
const {collectWarnings}=require('../app/app.js');
-const baselineWarnings=collectWarnings(new Map([['real',audited]]),scheme);
+const baselineWarnings=collectWarnings(new Map([['real',audited]]),scheme,'standard',{},data.suite);
assert.deepEqual(baselineWarnings.filter(w=>w.type==='Inconsistent scoring settings').map(w=>w.name).sort(),['ARC Challenge','MGSM','PIQA']);
-assert.deepEqual(baselineWarnings.filter(w=>w.type==='Config caveat').map(w=>w.name).sort(),['FLORES200','MultiBlimp','OpenSubtitles','SIB-200']);
+assert.deepEqual(baselineWarnings.filter(w=>w.type==='Config caveat').map(w=>w.name).sort(),['MultiBlimp']);
const unknown=auditRows([{...data.rows[0],task:'new_task_without_config'}],scheme)[0];
assert.equal(unknown.selected,false);assert.equal(unknown.eval,'');assert.equal(unknown.score_100,null);
const warningRows=audited.filter(r=>r.eval!=='HumanEval').concat(unknown);
-const warnings=collectWarnings(new Map([['real',warningRows]]),scheme);
-assert.equal(warnings.length,baselineWarnings.length+2);
+const warnings=collectWarnings(new Map([['real',warningRows]]),scheme,'standard',{},data.suite);
+assert.equal(warnings.length,baselineWarnings.length+1);
assert.ok(warnings.some(w=>w.type==='No config'&&w.name==='new_task_without_config'));
-assert.ok(warnings.some(w=>w.type==='No eval data'&&w.name==='HumanEval'));
+assert.ok(!warnings.some(w=>w.type==='No eval data'));
const alternate=structuredClone(scheme);alternate.evals.find(e=>e.name==='HumanEval').metric='nonexistent';
assert.ok(collectWarnings(new Map([['real',auditRows(data.rows,alternate)]]),alternate).some(w=>w.type==='No selected score'));
console.log('Warning checks passed: unconfigured exclusion, absent evals, selected metric gaps.');
@@ -156,7 +158,7 @@ console.log('Column sort and per-task missing metric/setting checks passed.');
// Grouped multilingual scores must expose differences in the selected protocol.
const protocolConfig=structuredClone(scheme);
for(const e of protocolConfig.evals)delete e.warning;
-const consistent=audited.map(r=>({...r,n_shot:'0'}));
+const consistent=audited.filter(r=>inSuite(r,data.suite)).map(r=>({...r,n_shot:'0'}));
assert.deepEqual(collectWarnings(new Map([['real',consistent]]),protocolConfig),[]);
for(const field of ['n_shot','filter','metric','harness','backend']){
const changed=consistent.map(r=>r.task==='arc_challenge_mt_cs'&&r.selected?{...r,[field]:field==='n_shot'?'5':'different'}:r);
@@ -176,4 +178,4 @@ console.log('Protocol consistency and editable normalization warning checks pass
for(const name of ['LSAT AR','X-CSQA','Belebele','MultiBlimp'])assert.equal(scheme.evals.find(e=>e.name===name).metric,'acc_norm');
assert.equal(scheme.evals.find(e=>e.name==='SIB-200').metric,'acc');
-assert.match(scheme.evals.find(e=>e.name==='SIB-200').warning,/34 of 36/);
+assert.equal(scheme.evals.find(e=>e.name==='SIB-200').warning,undefined);
diff --git a/tests/test_browser.mjs b/tests/test_browser.mjs
index 392b8e4..4e830a3 100644
--- a/tests/test_browser.mjs
+++ b/tests/test_browser.mjs
@@ -19,7 +19,7 @@ await send('Page.navigate',{url:new URL('../output/index.html',import.meta.url).
for(let i=0;i<50;i++){if(await evaluate("!!document.querySelector('#cards strong')"))break;await new Promise(r=>setTimeout(r,100));}
assert.equal(await evaluate("document.querySelector('#cards strong').textContent"),initialScore);
assert.equal(await evaluate("document.querySelectorAll('[data-view]').length"),6);
-assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'7');
+assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'5');
assert.equal(await evaluate("document.querySelector('#filters').hidden"),true);
assert.match(await evaluate("document.querySelector('#weightEditor .notice').textContent"),/saved but inactive/);
const englishScore=JSON.parse(fs.readFileSync(root+'output/analysis.json','utf8')).aggregates.english_category[0].score.toFixed(2);
@@ -40,7 +40,7 @@ assert.notEqual(await evaluate("document.querySelector('#cards strong').textCont
await change('[data-english-weight="Math"]','0.5');
await change('[data-english-weight="Code"]','0.5');
assert.equal(await evaluate("document.querySelector('#cards strong').textContent"),englishScore);
-assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'7');
+assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'5');
assert.match(await evaluate("document.querySelector('#view').textContent"),/English weighting group only · full weight/);
await click('#weightEditor > summary');
assert.equal(await evaluate("document.querySelectorAll('#weights tbody tr').length"),9);
@@ -198,7 +198,7 @@ assert.equal(await evaluate("document.querySelectorAll('.catalogue-task').length
assert.match(await evaluate("document.querySelector('#view').textContent"),/Excluded/);
await click('#clear');
// Config import changes metric/category/normalization/languages atomically.
-await evaluate(`window.loadTestConfig=async config=>{const dt=new DataTransfer();dt.items.add(new File(['# Browser YAML import\\n'+jsyaml.dump(config,{schema:jsyaml.CORE_SCHEMA})],'config.yaml'));const e=document.querySelector('#configFile');e.files=dt.files;await document.querySelector('#view').onchange({target:e});};`);
+await evaluate(`window.loadTestConfig=async config=>{for(const [selector,object] of [['#weightsFile',Object.fromEntries(Object.entries(config).filter(([k])=>['version','name','weights','english_weights','aggregate'].includes(k)))],['#configFile',Object.fromEntries(Object.entries(config).filter(([k])=>!['weights','english_weights','aggregate'].includes(k)))]]){const dt=new DataTransfer();dt.items.add(new File([jsyaml.dump(object,{schema:jsyaml.CORE_SCHEMA})],'config.yaml'));const e=document.querySelector(selector);e.files=dt.files;await document.querySelector('#view').onchange({target:e});if(document.querySelector('#error').textContent)break;}};`);
await evaluate(`(async()=>{const c=structuredClone(DATA.scheme);c.name='Browser custom config';const he=c.evals.find(e=>e.name==='HumanEval');he.metric='sh_pass@1';he.category='Math';c.evals.find(e=>e.name==='HellaSwag').normalize={min:.25,max:1};const group=c.languages.find(g=>g.tasks.includes('AIME24'));group.tasks=group.tasks.filter(t=>t!=='AIME24');c.languages=c.languages.filter(g=>g.tasks.length);c.languages.push({tasks:['AIME24'],scope:'single',language:'fin_Latn'});await loadTestConfig(c);})()`);
assert.equal(await evaluate("document.querySelector('#error').textContent"),'');
assert.notEqual(await evaluate("document.querySelector('#cards strong').textContent"),initialScore);
@@ -219,7 +219,7 @@ assert.equal(await evaluate("document.querySelector('#error').textContent"),'');
// Removing one task's configured metric warns even when other languages are selected.
await evaluate(`(async()=>{const rows=DATA.rows.filter(r=>!(r.task==='arc_challenge_mt_cs'&&r.metric==='acc_norm')).map(r=>({...r,checkpoint:'Missing Czech metric'}));const fields=Object.keys(rows[0]);const csv=[fields,...rows.map(r=>fields.map(f=>r[f]))].map(row=>row.map(v=>JSON.stringify(String(v??''))).join(',')).join('\\n');const dt=new DataTransfer();dt.items.add(new File([csv],'missing.csv'));const e=document.querySelector('#modelFile');e.files=dt.files;await e.onchange({target:e});})()`);
assert.equal(await evaluate("document.querySelector('#error').textContent"),'');
-assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'12');
+assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'11');
assert.match(await evaluate("document.querySelector('[data-eval=\"ARC Challenge\"] > summary').textContent"),/1 missing scoring/);
await click('[data-eval="ARC Challenge"] > summary');
await click('[data-task=arc_challenge_mt_cs] > summary');
@@ -232,20 +232,20 @@ assert.match(await evaluate("document.querySelector('#view').textContent"),/Comp
await evaluate(`(async()=>{const fields=['checkpoint','task','metric','filter','n_shot','harness','backend','value'];const rows=DATA.rows.filter(r=>r.eval!=='IFEval').map(r=>({...r,checkpoint:'Missing IFEval'}));const csv=[fields.join(','),...rows.map(r=>fields.map(f=>r[f]).join(','))].join('\\n');const dt=new DataTransfer();dt.items.add(new File([csv],'missing-eval.csv'));const e=document.querySelector('#modelFile');e.files=dt.files;await e.onchange({target:e});})()`);
assert.notEqual(await evaluate("document.querySelector('#cards strong').textContent"),'—');
assert.equal(await evaluate("document.querySelector('#cards .score-card:last-child strong').textContent"),'0.00');
-assert.match(await evaluate("document.querySelector('#view').textContent"),/No eval data.*IFEval/s);
+assert.doesNotMatch(await evaluate("document.querySelector('#view').textContent"),/No eval data/);
assert.match(await evaluate("document.querySelector('#view').textContent"),/Comparison coverage.*IFEval/s);
for(const mode of ['standard','english_eval','english_category']){await click('[data-aggregate='+mode+']');assert.equal(await evaluate("document.querySelector('#cards .score-card:last-child strong').textContent"),'0.00');}
// Reload restores the original export before the remaining import checks.
await send('Page.reload');
-for(let i=0;i<50;i++){await new Promise(r=>setTimeout(r,100));if(await evaluate("document.querySelector('#warningCount')?.textContent==='7'"))break;}
+for(let i=0;i<50;i++){await new Promise(r=>setTimeout(r,100));if(await evaluate("document.querySelector('#warningCount')?.textContent==='5'"))break;}
await click('[data-view=config]');
-await evaluate(`window.loadTestConfig=async config=>{const dt=new DataTransfer();dt.items.add(new File([jsyaml.dump(config,{schema:jsyaml.CORE_SCHEMA})],'config.yaml'));const e=document.querySelector('#configFile');e.files=dt.files;await document.querySelector('#view').onchange({target:e});};`);
+await evaluate(`window.loadTestConfig=async config=>{for(const [selector,object] of [['#weightsFile',Object.fromEntries(Object.entries(config).filter(([k])=>['version','name','weights','english_weights','aggregate'].includes(k)))],['#configFile',Object.fromEntries(Object.entries(config).filter(([k])=>!['weights','english_weights','aggregate'].includes(k)))]]){const dt=new DataTransfer();dt.items.add(new File([jsyaml.dump(object,{schema:jsyaml.CORE_SCHEMA})],'config.yaml'));const e=document.querySelector(selector);e.files=dt.files;await document.querySelector('#view').onchange({target:e});if(document.querySelector('#error').textContent)break;}};`);
// English shares and the active aggregate are editable and portable in YAML.
await click('[data-view=score]');await click('[data-aggregate=english_category]');await change('[data-english-weight="Reading"]','0.7');
await click('[data-view=config]');
await evaluate(`window.originalCreate=URL.createObjectURL;window.originalClick=HTMLAnchorElement.prototype.click;URL.createObjectURL=blob=>{window.exportBlob=blob;return 'blob:test'};HTMLAnchorElement.prototype.click=function(){};`);
-await click('#exportConfig');
-const englishExport=await evaluate(`exportBlob.text().then(parseConfig)`);
+await click('#exportWeights');
+const englishExport=await evaluate(`exportBlob.text().then(parseWeightProfile)`);
assert.equal(englishExport.english_weights.Reading,.7);assert.equal(englishExport.aggregate,'english_category');
await evaluate(`URL.createObjectURL=originalCreate;HTMLAnchorElement.prototype.click=originalClick;loadTestConfig(DATA.scheme)`);
assert.equal(await evaluate("document.querySelector('[data-aggregate][aria-pressed=true]').dataset.aggregate"),'standard');
@@ -254,33 +254,36 @@ await click('[data-view=score]');
await change('[data-weight="Code"]','.16');await change('[data-weight="Math"]','.14');
await click('[data-view=config]');
await evaluate(`window.originalCreate=URL.createObjectURL;window.originalClick=HTMLAnchorElement.prototype.click;URL.createObjectURL=blob=>{window.exportBlob=blob;return 'blob:test'};HTMLAnchorElement.prototype.click=function(){};`);
-await click('#exportConfig');
-const exported=await evaluate(`exportBlob.text().then(parseConfig)`);
+await click('#exportWeights');
+const exported=await evaluate(`exportBlob.text().then(parseWeightProfile)`);
assert.equal(exported.weights.Code,.16);assert.equal(exported.weights.Math,.14);
-assert.equal(exported.languages.flatMap(g=>g.tasks).length,1556);
+assert.equal(exported.languages,undefined);
+await click('#exportConfig');assert.equal(await evaluate('exportBlob.text().then(parseCatalogue).then(c=>c.languages.flatMap(g=>g.tasks).length)'),1556);
await evaluate(`URL.createObjectURL=originalCreate;HTMLAnchorElement.prototype.click=originalClick;loadTestConfig(DATA.scheme)`);
// Missing configuration is a warning and exclusion, not a failed import.
await evaluate(`(async()=>{const c=structuredClone(DATA.scheme);c.evals.find(e=>e.name==='HumanEval').match={name:'absent_eval_task'};await loadTestConfig(c);})()`);
assert.equal(await evaluate("document.querySelector('#error').textContent"),'');
-assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'9');
+assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'6');
assert.equal(await evaluate("document.querySelector('#warningCount').classList.contains('has-warnings')"),true);
await click('[data-view=warnings]');
assert.match(await evaluate("document.querySelector('#view').textContent"),/No config.*HumanEval/);
-assert.match(await evaluate("document.querySelector('#view').textContent"),/No eval data.*HumanEval/);
+assert.doesNotMatch(await evaluate("document.querySelector('#view').textContent"),/No eval data/);
assert.equal(await evaluate("document.querySelector('#filters').hidden"),true);
await screenshot('warnings-preview');
await click('[data-view=config]');await evaluate(`loadTestConfig(DATA.scheme)`);
-assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'7');
+assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'5');
assert.equal(await evaluate("document.querySelector('#warningCount').classList.contains('has-warnings')"),true);
// Warnings expose concrete settings and editable config caveats.
await click('[data-view=warnings]');
assert.match(await evaluate("document.querySelector('#view').textContent"),/Inconsistent scoring settings.*MGSM.*0 shots.*5 shots/s);
-assert.match(await evaluate("document.querySelector('#view').textContent"),/Translation calibration is unresolved/);
+assert.match(await evaluate("document.querySelector('#view').textContent"),/Not used.*Global PIQA \(prompted\)/s);
await click('[data-view=config]');
assert.match(await evaluate(`document.querySelector('[data-eval="ARC Challenge"] > summary').textContent`),/Inconsistent scoring settings/);
-assert.match(await evaluate(`document.querySelector('[data-eval="SIB-200"] .normalization-info').textContent`),/34 of 36/);
-assert.match(await evaluate(`document.querySelector('[data-eval="FLORES200"] .normalization-info').textContent`),/bounded by 0 and 100/);
-assert.match(await evaluate(`document.querySelector('[data-eval="OpenSubtitles"] .normalization-info').textContent`),/bounded by 0 and 100/);
+assert.doesNotMatch(await evaluate(`document.querySelector('[data-eval="SIB-200"] .normalization-info').textContent`),/Metric-selection exception/);
+assert.match(await evaluate(`document.querySelector('[data-eval="JEEBench"] > summary').textContent`),/10.55% baseline/);
+assert.match(await evaluate(`document.querySelector('[data-eval="JEEBench"] .normalization-info').textContent`),/Table 2/);
+assert.match(await evaluate(`document.querySelector('[data-eval="FLORES200"] .normalization-info').textContent`),/0–100 points/g);
+assert.match(await evaluate(`document.querySelector('[data-eval="OpenSubtitles"] .normalization-info').textContent`),/0–100 points/g);
await click('[data-eval="Language ID"] > summary');
await click('[data-task="bigbench_language_identification_multiple_choice"] > summary');
assert.match(await evaluate(`document.querySelector('[data-task="bigbench_language_identification_multiple_choice"] .task-details').textContent`),/English-balance group: English \(fallback/);
@@ -296,8 +299,8 @@ await click('[data-eval="MultiBlimp"] > summary');
for(const name of ['AIME24','AIME25'])assert.match(await evaluate(`document.querySelector('[data-eval="${name}"] .normalization-info').textContent`),/random_score = 0/);
// Custom caveats survive import/export and can be removed when resolved.
-await evaluate(`(async()=>{const c=structuredClone(DATA.scheme);delete c.evals.find(e=>e.name==='FLORES200').warning;await loadTestConfig(c);})()`);
-assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'6');
+await evaluate(`(async()=>{const c=structuredClone(DATA.scheme);delete c.evals.find(e=>e.name==='MultiBlimp').warning;await loadTestConfig(c);})()`);
+assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'4');
await evaluate(`loadTestConfig(DATA.scheme)`);
// Normalization is visible without opening individual language variants.
assert.match(await evaluate(`document.querySelector('[data-eval="SIB-200"] > summary').textContent`),/14.29% baseline/);
diff --git a/tests/test_data.py b/tests/test_data.py
index 63086f5..180ff41 100644
--- a/tests/test_data.py
+++ b/tests/test_data.py
@@ -32,111 +32,130 @@ def javascript(cases, expression):
return json.loads(subprocess.check_output(['node', '-e', script], input=json.dumps(cases), text=True, cwd=ROOT))
+def inputs(folder, c=None):
+ c = c or config()
+ catalogue = {k:v for k,v in c.items() if k not in ['weights','english_weights','aggregate']}
+ profile = {k:v for k,v in c.items() if k in ['version','name','weights','english_weights','aggregate']}
+ (folder/'catalogue.yaml').write_text(json.dumps(catalogue))
+ (folder/'weights.yaml').write_text(json.dumps(profile))
+ return dict(catalogue_path=folder/'catalogue.yaml', weights_path=folder/'weights.yaml', suite_path=ROOT/'configs/sets/any-available.yaml')
+
+
class DataContracts(unittest.TestCase):
- def test_directory_default_controls_selection_and_explicit_config_overrides_it(self):
+ def test_independent_directory_defaults_and_explicit_overrides(self):
with tempfile.TemporaryDirectory() as tmp:
- folder=Path(tmp);configs=folder/'configs';configs.mkdir()
- first=config();first['name']='First'
- second=config();second['name']='Second'
- (configs/'first.yaml').write_text(json.dumps(first))
- (configs/'second.yml').write_text(json.dumps(second))
- (configs/'default.txt').write_text('second.yml\n')
- with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',configs_dir=configs)
+ folder=Path(tmp);kw=inputs(folder);profiles=folder/'profiles';profiles.mkdir()
+ p=dict(version=1,name='First',weights={'C':1})
+ (profiles/'first.yaml').write_text(json.dumps(p));p['name']='Second'
+ (profiles/'second.yml').write_text(json.dumps(p));(profiles/'default.txt').write_text('second.yml\n')
+ kw.pop('weights_path')
+ with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',**kw,weights_dir=profiles)
data=json.loads((folder/'out/analysis.json').read_text())
- self.assertEqual(data['config_file'],'second.yml')
- self.assertEqual([p['config']['name'] for p in data['configurations']],['Second','First'])
- with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',configs/'first.yaml',configs_dir=configs)
- self.assertEqual(json.loads((folder/'out/analysis.json').read_text())['scheme']['name'],'First')
+ self.assertEqual(data['profile_file'],'second.yml')
+ self.assertEqual([p['config']['name'] for p in data['profiles']],['Second','First'])
+ self.assertEqual(data['suite']['mode'],'available')
+ with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',**kw,weights_path=profiles/'first.yaml',weights_dir=profiles)
+ self.assertEqual(json.loads((folder/'out/analysis.json').read_text())['profile']['name'],'First')
def test_invalid_directory_defaults_preserve_existing_output(self):
with tempfile.TemporaryDirectory() as tmp:
- folder=Path(tmp);configs=folder/'configs';configs.mkdir();out=folder/'out';out.mkdir()
+ folder=Path(tmp);kw=inputs(folder);kw.pop('weights_path');profiles=folder/'profiles';profiles.mkdir();out=folder/'out';out.mkdir()
(out/'index.html').write_text('keep')
- (configs/'valid.yaml').write_text(json.dumps(config()))
for value in [None,'','valid.yaml\nother.yaml','../valid.yaml','/valid.yaml','sub/valid.yaml','sub\\valid.yaml','default.txt','missing.yaml']:
with self.subTest(default=value):
- if value is not None:(configs/'default.txt').write_text(value)
- with self.assertRaisesRegex(ValueError,'default.txt'):build(None,out,configs_dir=configs)
+ if value is not None:(profiles/'default.txt').write_text(value)
+ with self.assertRaisesRegex(ValueError,'default.txt'):build(None,out,**kw,weights_dir=profiles)
self.assertEqual((out/'index.html').read_text(),'keep')
- def test_cli_uses_repository_default_without_config_flags(self):
+ def test_cli_defaults_to_any_available_and_separate_weight_profile(self):
with tempfile.TemporaryDirectory() as tmp:
subprocess.run(['python3','-m','app.build','--output',tmp],cwd=ROOT,check=True,stdout=subprocess.DEVNULL)
data=json.loads((Path(tmp)/'analysis.json').read_text())
- self.assertEqual(data['config_file'],(ROOT/'configs/default.txt').read_text().strip())
- self.assertEqual(set(p['file'] for p in data['configurations']),{'oellm.yaml','example.yaml'})
-
- def test_empty_dashboard_contains_config_but_no_models(self):
- with tempfile.TemporaryDirectory() as tmp:
- folder=Path(tmp);cfg=folder/'config.yaml';cfg.write_text(json.dumps(config()))
- with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',cfg)
- data=json.loads((folder/'out/analysis.json').read_text())
- self.assertEqual(data['rows'],[]);self.assertEqual(data['models'],[])
- self.assertEqual(data['scheme'],config());self.assertTrue((folder/'out/index.html').is_file())
+ self.assertEqual(data['suite']['mode'],'available')
+ self.assertEqual(data['profile']['name'],'Original')
+ self.assertEqual([p['config']['name'] for p in data['profiles']],['Original','Code & math emphasis'])
+ self.assertEqual(data['profiles'][1]['config']['weights'],{'Code':.2,'Math':.2,'Reasoning':.1,'Knowledge':.125,'Commonsense':.125,'Reading':.15,'Translation':.1/3,'Language':.1/3,'Instruction following':.1/3})
+ self.assertEqual({p['file'] for p in data['suites']},{'any-available.yaml','flagship-1.yaml'})
+ self.assertNotIn('weights',data['catalogue']);self.assertNotIn('weights',data['suite']);self.assertNotIn('evals',data['profile'])
def test_empty_rebuild_removes_stale_generated_scores(self):
with tempfile.TemporaryDirectory() as tmp:
- folder=Path(tmp);cfg=folder/'config.yaml';cfg.write_text(json.dumps(config()))
+ folder=Path(tmp);kw=inputs(folder);out=folder/'out'
source=folder/'scores.csv';r=row();source.write_text(','.join(r)+'\n'+','.join(r.values()))
- out=folder/'out'
- with contextlib.redirect_stdout(io.StringIO()):build(source,out,cfg)
+ with contextlib.redirect_stdout(io.StringIO()):build(source,out,**kw)
self.assertEqual(len(list(out.glob('*.csv'))),4)
(out/'personal-note.txt').write_text('keep')
- with contextlib.redirect_stdout(io.StringIO()):build(None,out,cfg)
+ with contextlib.redirect_stdout(io.StringIO()):build(None,out,**kw)
self.assertEqual(list(out.glob('*.csv')),[])
self.assertEqual((out/'personal-note.txt').read_text(),'keep')
+ data=json.loads((out/'analysis.json').read_text());self.assertEqual(data['rows'],[]);self.assertEqual(data['models'],[])
+
+ def test_all_choices_validated_before_output_is_replaced(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ folder=Path(tmp);kw=inputs(folder);sets=folder/'sets';sets.mkdir();out=folder/'out'
+ s=dict(version=1,name='Named',mode='fixed',evals=[dict(name='Eval')])
+ (sets/'named.yaml').write_text(json.dumps(s))
+ with contextlib.redirect_stdout(io.StringIO()):build(None,out,**kw,sets_dir=sets)
+ data=json.loads((out/'analysis.json').read_text());self.assertEqual(len(data['suites']),2)
+ previous=(out/'index.html').read_bytes()
+ (sets/'duplicate.yaml').write_text(json.dumps(s))
+ with self.assertRaisesRegex(ValueError,'duplicate config name'):build(None,out,**kw,sets_dir=sets)
+ self.assertEqual((out/'index.html').read_bytes(),previous)
+ (sets/'duplicate.yaml').unlink();s['evals'][0]['name']='Unknown';(sets/'named.yaml').write_text(json.dumps(s))
+ with self.assertRaisesRegex(ValueError,'no catalogue rule'):build(None,out,**kw,sets_dir=sets)
+ self.assertEqual((out/'index.html').read_bytes(),previous)
- def test_config_choices_are_embedded_and_validated_before_writing(self):
+ def test_available_set_exclusions_apply_in_build_summaries(self):
with tempfile.TemporaryDirectory() as tmp:
- folder=Path(tmp);cfg=folder/'default.yaml';cfg.write_text(json.dumps(config()));configs=folder/'configs';configs.mkdir()
- alternate=config();alternate['name']='Alternate';alternate['english_weights']={'C':.5};alternate['aggregate']='english_eval'
- (configs/'alternate.yml').write_text(json.dumps(alternate))
- with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',cfg,configs_dir=configs)
+ folder=Path(tmp);kw=inputs(folder)
+ excluded=dict(version=1,name='Excluded',mode='available',exclude=['Eval'])
+ (folder/'set.yaml').write_text(json.dumps(excluded));kw['suite_path']=folder/'set.yaml'
+ source=folder/'scores.csv';r=row();source.write_text(','.join(r)+'\n'+','.join(r.values()))
+ with contextlib.redirect_stdout(io.StringIO()):build(source,folder/'out',**kw)
+ data=json.loads((folder/'out/analysis.json').read_text())
+ self.assertEqual(len(data['rows']),1);self.assertTrue(data['rows'][0]['selected'])
+ self.assertIsNone(data['models'][0]['score']);self.assertEqual(data['models'][0]['evals'][0]['count'],0)
+
+ def test_named_set_scopes_score_but_keeps_full_audit(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ folder=Path(tmp);kw=inputs(folder)
+ s=dict(version=1,name='Named',mode='fixed',evals=[dict(name='Eval',variants=[dict(task='task_en',n_shot=0)])])
+ (folder/'set.yaml').write_text(json.dumps(s));kw['suite_path']=folder/'set.yaml'
+ rows=[row(),row(task='task_fr',value='.25')];source=folder/'scores.csv'
+ source.write_text(','.join(rows[0])+'\n'+'\n'.join(','.join(r.values()) for r in rows))
+ with contextlib.redirect_stdout(io.StringIO()):build(source,folder/'out',**kw)
data=json.loads((folder/'out/analysis.json').read_text())
- self.assertEqual([p['config']['name'] for p in data['configurations']],['Fixture','Alternate'])
- self.assertEqual(data['configurations'][1]['config']['aggregate'],'english_eval')
- (configs/'duplicate.yaml').write_text(json.dumps(alternate))
- previous=(folder/'out/index.html').read_bytes()
- with self.assertRaisesRegex(ValueError,'duplicate config name'):build(None,folder/'out',cfg,configs_dir=configs)
- self.assertEqual((folder/'out/index.html').read_bytes(),previous)
- (configs/'duplicate.yaml').unlink();alternate['evals'][0]['score']['scale']=.1
- (configs/'alternate.yml').write_text(json.dumps(alternate));r=row();source=folder/'scores.csv';source.write_text(','.join(r)+'\n'+','.join(r.values()))
- with self.assertRaisesRegex(ValueError,'alternate.yml.*Invalid score'):build(source,folder/'out',cfg,configs_dir=configs)
- self.assertEqual((folder/'out/index.html').read_bytes(),previous)
+ self.assertEqual(len(data['rows']),2);self.assertEqual(data['models'][0]['score'],50)
+ self.assertEqual(data['models'][0]['evals'][0]['count'],1)
def test_cli_rejects_both_csv_and_results_directory(self):
result=subprocess.run(['python3','-m','app.build','unused.csv','--results-dir','unused'],capture_output=True,text=True)
self.assertEqual(result.returncode,2);self.assertIn('not both',result.stderr)
- def test_shared_results_directory_combines_models(self):
+ def test_shared_results_directory_combines_models_and_rejects_duplicates(self):
with tempfile.TemporaryDirectory() as tmp:
- folder=Path(tmp);results=folder/'results';results.mkdir();cfg=folder/'config.yaml';cfg.write_text(json.dumps(config()))
+ folder=Path(tmp);kw=inputs(folder);results=folder/'results';results.mkdir();out=folder/'out'
(results/'README.md').write_text('Not a CSV')
for name in ['B','A']:
r=row(checkpoint=name);(results/(name+'.csv')).write_text(','.join(r)+'\n'+','.join(r.values()))
- with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',cfg,results_dir=results)
- data=json.loads((folder/'out/analysis.json').read_text())
+ with contextlib.redirect_stdout(io.StringIO()):build(None,out,**kw,results_dir=results)
+ data=json.loads((out/'analysis.json').read_text())
self.assertEqual([m['model'] for m in data['models']],['A','B'])
self.assertEqual([s['file'] for s in data['sources']],['A.csv','B.csv'])
self.assertTrue(all(len(s['sha256'])==64 for s in data['sources']))
-
- def test_shared_results_reject_duplicate_model_names_without_replacing_output(self):
- with tempfile.TemporaryDirectory() as tmp:
- folder=Path(tmp);results=folder/'results';results.mkdir();out=folder/'out';out.mkdir();(out/'index.html').write_text('keep')
- cfg=folder/'config.yaml';cfg.write_text(json.dumps(config()))
- for name in ['one','two']:
- r=row(task='task_'+name);(results/(name+'.csv')).write_text(','.join(r)+'\n'+','.join(r.values()))
- with self.assertRaisesRegex(ValueError,'Model A.*one.csv.*two.csv'):build(None,out,cfg,results_dir=results)
- self.assertEqual((out/'index.html').read_text(),'keep')
+ previous=(out/'index.html').read_bytes()
+ (results/'duplicate.csv').write_text((results/'A.csv').read_text())
+ with self.assertRaisesRegex(ValueError,'Duplicate model'):build(None,out,**kw,results_dir=results)
+ self.assertEqual((out/'index.html').read_bytes(),previous)
def test_shared_results_report_bad_filename_and_reject_missing_directory(self):
with tempfile.TemporaryDirectory() as tmp:
- folder=Path(tmp);cfg=folder/'config.yaml';cfg.write_text(json.dumps(config()))
- with self.assertRaisesRegex(ValueError,'directory'):build(None,folder/'out',cfg,results_dir=folder/'missing')
+ folder=Path(tmp);kw=inputs(folder)
+ with self.assertRaisesRegex(ValueError,'directory'):build(None,folder/'out',**kw,results_dir=folder/'missing')
results=folder/'results';results.mkdir();r=row(value='NaN');(results/'bad.csv').write_text(','.join(r)+'\n'+','.join(r.values()))
- with self.assertRaisesRegex(ValueError,'bad.csv.*Model A.*task_en'):build(None,folder/'out',cfg,results_dir=results)
+ with self.assertRaisesRegex(ValueError,'bad.csv.*Model A.*task_en'):build(None,folder/'out',**kw,results_dir=results)
(results/'bad.csv').unlink()
- with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',cfg,results_dir=results)
+ with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',**kw,results_dir=results)
self.assertEqual(json.loads((folder/'out/analysis.json').read_text())['models'],[])
def assert_nested_close(self, a, b):
@@ -234,20 +253,20 @@ def test_all_modes_python_browser_parity_with_incomplete_data(self):
def test_invalid_builds_do_not_replace_existing_output(self):
with tempfile.TemporaryDirectory() as tmp:
folder=Path(tmp);target=folder/'out';target.mkdir();(target/'index.html').write_text('keep me')
- cfg=folder/'config.yaml';cfg.write_text(json.dumps(config()))
+ kw=inputs(folder)
header=','.join(row())+'\n'
for source in ['', header, 'a,a\n1,2', header+'too,few', header+','.join(row(value='NaN').values())]:
path=folder/'input.csv';path.write_text(source)
- with self.subTest(source=source), self.assertRaises(ValueError):build(path,target,cfg)
+ with self.subTest(source=source), self.assertRaises(ValueError):build(path,target,**kw)
self.assertEqual((target/'index.html').read_text(),'keep me')
self.assertEqual(len(list(target.iterdir())),1)
def test_build_embeds_data_safely_and_uses_custom_categories(self):
with tempfile.TemporaryDirectory() as tmp:
- folder=Path(tmp);cfg=folder/'config.yaml';c=config();c['name']='';cfg.write_text(json.dumps(c))
+ folder=Path(tmp);c=config();c['name']='';kw=inputs(folder,c)
path=folder/'input.csv';path.write_text('\ufeff'+','.join(row())+'\r\n'+','.join(row().values()))
self.assertEqual(load_csv(path),[row()])
- with contextlib.redirect_stdout(io.StringIO()):build(path,folder/'out',cfg)
+ with contextlib.redirect_stdout(io.StringIO()):build(path,folder/'out',**kw)
html=(folder/'out/index.html').read_text();self.assertNotIn(c['name'],html)
self.assertIn('\\u003c/script>',html)
payload=json.loads((folder/'out/analysis.json').read_text())
diff --git a/tests/test_english.cjs b/tests/test_english.cjs
index 46c00ef..84980f4 100644
--- a/tests/test_english.cjs
+++ b/tests/test_english.cjs
@@ -40,13 +40,13 @@ close(totals(rows,scheme,scheme.weights,'english_category',scheme.english_weight
const translation=new Map(rows.map(r=>[r.task,{scope:'translation',source_language:metadata.get(r.task).language==='eng_Latn'?'fra_Latn':'eng_Latn',target_language:metadata.get(r.task).language}]));
close(totals(rows,scheme,scheme.weights,'english_category',scheme.english_weights,translation).score,60);
for(const invalid of [-.1,1.1,NaN])assert.equal(totals(rows,scheme,scheme.weights,'english_category',{C:invalid},metadata).score,null);
-const config=require('../app/eval_config.js').parseConfig(require('node:fs').readFileSync('configs/oellm.yaml','utf8'));
+const config=JSON.parse(require('node:fs').readFileSync('output/analysis.json','utf8')).scheme;
for(const english_weights of [{Code:-1},{Code:1.1},{Missing:.5},{Code:'0.5'}])assert.throws(()=>validateConfig({...config,english_weights}));
assert.throws(()=>validateConfig({...config,aggregate:'oops'}));
console.log('English-share arithmetic, coverage, translation, contribution reconciliation, filters, and validation passed.');
const data=require('../output/analysis.json');
for(const mode of ['standard','english_eval','english_category']){
- const actual=totals(data.rows.filter(r=>r.selected),data.scheme,data.scheme.weights,mode),expected=data.aggregates[mode][0];
+ const actual=totals(require('../app/suite_config.js').scopeRows(data.rows,data.suite).rows,data.scheme,data.scheme.weights,mode),expected=data.aggregates[mode][0];
close(actual.score,expected.score);
for(const c of actual.categories)close(c.score,expected.categories.find(x=>x.name===c.name).score);
for(const e of actual.evals){const x=expected.evals.find(x=>x.name===e.name);close(e.weight,x.weight);close(e.contribution,x.contribution);}
diff --git a/tests/test_public_browser.mjs b/tests/test_public_browser.mjs
index 59b78fa..f90fc55 100644
--- a/tests/test_public_browser.mjs
+++ b/tests/test_public_browser.mjs
@@ -6,20 +6,21 @@ import assert from 'node:assert/strict';
import {execFileSync} from 'node:child_process';
import {fileURLToPath,pathToFileURL} from 'node:url';
import {createRequire} from 'node:module';
-const require=createRequire(import.meta.url),{parseConfig,serializeConfig}=require('../app/eval_config.js');
+const require=createRequire(import.meta.url),{parseCatalogue,serializeCatalogue}=require('../app/eval_config.js'),{parseWeightProfile,serializeWeightProfile,serializeSuite}=require('../app/suite_config.js');
const root=fileURLToPath(new URL('../',import.meta.url));
const temporary=fs.mkdtempSync(path.join(os.tmpdir(),'quickdash-browser-'));
let ws;
try{
- const fixture=parseConfig(fs.readFileSync(path.join(root,'configs/example.yaml'),'utf8'));
- const configs=path.join(temporary,'configs');fs.mkdirSync(configs);
- fs.copyFileSync(path.join(root,'configs/oellm.yaml'),path.join(configs,'oellm.yaml'));
- fs.writeFileSync(path.join(configs,'default.txt'),'oellm.yaml\n');
- fs.writeFileSync(path.join(configs,'example.yaml'),serializeConfig(fixture));
- const invalidForScores=structuredClone(fixture);invalidForScores.name='Small scale fixture';invalidForScores.evals[0].score.scale=.1;
- fs.writeFileSync(path.join(configs,'small.yaml'),serializeConfig(invalidForScores));
+ const fixture=parseCatalogue(fs.readFileSync(path.join(root,'configs/examples/catalogue.yaml'),'utf8'));
+ fixture.evals[0].match.regex='example_reasoning_(en|fr|de)';
+ fixture.languages[1].tasks.push('example_reasoning_de');
+ const weighting=parseWeightProfile(fs.readFileSync(path.join(root,'configs/examples/weights.yaml'),'utf8'));
+ const profiles=path.join(temporary,'weights');fs.mkdirSync(profiles);
+ fs.copyFileSync(path.join(root,'configs/weights/oellm.yaml'),path.join(profiles,'oellm.yaml'));
+ fs.writeFileSync(path.join(profiles,'default.txt'),'oellm.yaml\n');
+ fs.writeFileSync(path.join(profiles,'example.yaml'),serializeWeightProfile(weighting));
const empty=path.join(temporary,'empty');
- execFileSync('python3',['-m','app.build','--configs-dir',configs,'--output',empty],{cwd:root,stdio:'pipe'});
+ execFileSync('python3',['-m','app.build','--weights-dir',profiles,'--output',empty],{cwd:root,stdio:'pipe'});
let tabs;
// A cold CI runner can take longer than five seconds to launch Chrome.
const startupDeadline=Date.now()+30_000;
@@ -42,16 +43,18 @@ try{
const click=selector=>evaluate(`document.querySelector(${JSON.stringify(selector)}).click()`);
const change=(selector,value)=>evaluate(`{const e=document.querySelector(${JSON.stringify(selector)});e.value=${JSON.stringify(value)};e.dispatchEvent(new Event('change',{bubbles:true}));}`);
const navigate=async url=>{await send('Page.navigate',{url});for(let i=0;i<50;i++){await new Promise(r=>setTimeout(r,50));if(await evaluate("document.querySelector('#view')?.textContent.length>0"))return;}throw Error('Dashboard did not render');};
+ await send('Runtime.discardConsoleEntries');
await send('Runtime.enable');await send('Network.enable');await send('Page.enable');
await navigate(pathToFileURL(path.join(empty,'index.html')).href);
assert.equal(await evaluate("document.querySelector('#modelA').options.length"),0);
assert.equal(await evaluate("document.querySelector('#cards').hidden"),true);
assert.match(await evaluate("document.querySelector('#view').textContent"),/Compare your evaluation results/);
- assert.equal(await evaluate("document.querySelector('#configPreset').options.length"),3);
+ assert.equal(await evaluate("document.querySelector('#suitePreset').options.length"),2);
+ assert.equal(await evaluate("document.querySelector('#suitePreset').selectedOptions[0].textContent"),'Any available');
for(const view of ['categories','languages','comparisons','config','warnings','score'])await click('[data-view='+view+']');
- await change('#configPreset','1');
- assert.equal(await evaluate("document.querySelector('#error').textContent"),'');
const upload=async(selector,content,name)=>evaluate(`(async()=>{const dt=new DataTransfer();dt.items.add(new File([${JSON.stringify(content)}],${JSON.stringify(name)}));const e=document.querySelector(${JSON.stringify(selector)});e.files=dt.files;if(e.id==='modelFile')await e.onchange({target:e});else await document.querySelector('#view').onchange({target:e});})()`);
+ await click('[data-view=config]');await upload('#configFile',serializeCatalogue(fixture),'catalogue.yaml');
+ await change('#weightPreset','1');
await upload('#modelFile',fs.readFileSync(path.join(root,'examples/scores.csv'),'utf8'),'scores.csv');
assert.equal(await evaluate("document.querySelector('#error').textContent"),'');
assert.equal(await evaluate("document.querySelector('#modelA').value"),'Example A');
@@ -59,18 +62,45 @@ try{
assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'0');
assert.equal(await evaluate("document.querySelector('#cards .score-card:last-child strong').textContent"),'-2.50');
const original=await evaluate("document.querySelector('#cards').textContent");
- await change('#configPreset','2');
+ // A catalogue without loaded data is not a coverage requirement.
+ const extended=structuredClone(fixture);extended.evals.push({...extended.evals[0],name:'Unused eval',match:{name:'unused'}});
+ await upload('#configFile',serializeCatalogue(extended),'extended.yaml');
+ assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'0');
+ const invalid=structuredClone(fixture);invalid.evals[0].score.scale=.1;
+ await upload('#configFile',serializeCatalogue(invalid),'invalid.yaml');
assert.match(await evaluate("document.querySelector('#error').textContent"),/Invalid score/);
- assert.equal(await evaluate("document.querySelector('#configPreset').value"),'1');
assert.equal(await evaluate("document.querySelector('#cards').textContent"),original);
+ // Named sets check requirements missing from both models and retain extras for inspection.
+ const fixed={version:1,name:'Example required set',mode:'fixed',evals:[{name:'Example reasoning',variants:[{task:'example_reasoning_en',n_shot:0},{task:'example_reasoning_de',n_shot:0}]}]};
+ await upload('#suiteFile',serializeSuite(fixed),'set.yaml');
+ assert.equal(await evaluate("document.querySelector('#error').textContent"),'');
+ assert.match(await evaluate("document.querySelector('#coverage').textContent"),/INCOMPLETE.*1\/2 requirements shared.*4 extra/);
+ assert.equal(await evaluate("document.querySelector('#weightPreset').value"),'1');
+ await click('[data-view=warnings]');
+ assert.match(await evaluate("document.querySelector('#view').textContent"),/Missing suite data.*example_reasoning_de/s);
+ assert.match(await evaluate("document.querySelector('#view').textContent"),/Not used/);
await click('[data-view=config]');
- const uploaded=structuredClone(fixture);uploaded.name='Temporary config';uploaded.evals[0].normalize.min=0;
- await upload('#configFile',serializeConfig(uploaded),'temporary.yaml');
- assert.equal(await evaluate("document.querySelector('#configPreset').value"),'custom');
- assert.notEqual(await evaluate("document.querySelector('#cards').textContent"),original);
- await change('#configPreset','1');assert.equal(await evaluate("document.querySelector('#cards').textContent"),original);
- await click('#clearModels');assert.equal(await evaluate("document.querySelector('#modelA').options.length"),0);
- await change('#configPreset','2');assert.equal(await evaluate("document.querySelector('#error').textContent"),'');
+ assert.match(await evaluate("document.querySelector('#view').textContent"),/Example math/);
+ await click('[data-view=score]');await change('[data-weight=Reasoning]','.8');await change('[data-weight=Math]','.2');
+ await change('#suitePreset','0');
+ assert.equal(await evaluate("document.querySelector('[data-weight=Reasoning]').value"),'0.8');
+ await click('[data-view=config]');
+ const unavailable={...fixed,evals:[{name:'Absent eval'}]};
+ await upload('#suiteFile',serializeSuite(unavailable),'invalid-set.yaml');
+ assert.match(await evaluate("document.querySelector('#error').textContent"),/no catalogue rule/);
+ assert.equal(await evaluate("document.querySelector('#suitePreset').value"),'0');
+ await upload('#suiteFile',serializeSuite(fixed),'set.yaml');
+ await change('#weightPreset','0');
+ assert.equal(await evaluate("document.querySelector('#suitePreset').value"),'custom');
+ await change('#weightPreset','1');
+ await evaluate(`window.originalCreate=URL.createObjectURL;window.originalClick=HTMLAnchorElement.prototype.click;URL.createObjectURL=b=>{window.exportBlob=b;return 'blob:test'};HTMLAnchorElement.prototype.click=function(){};`);
+ await click('#exportConfig');assert.deepEqual(await evaluate('exportBlob.text().then(parseCatalogue)'),extended);
+ await click('#exportSuite');assert.deepEqual(await evaluate('exportBlob.text().then(parseSuite)'),fixed);
+ await click('#exportWeights');assert.deepEqual(await evaluate('exportBlob.text().then(parseWeightProfile)'),{...weighting,english_weights:weighting.english_weights,aggregate:'standard'});
+ await evaluate('URL.createObjectURL=originalCreate;HTMLAnchorElement.prototype.click=originalClick');
+ await change('#suitePreset','0');await click('#clearModels');
+ assert.equal(await evaluate("document.querySelector('#modelA').options.length"),0);
+ await click('[data-view=config]');await upload('#configFile',serializeCatalogue(invalid),'small.yaml');
await upload('#modelFile',fs.readFileSync(path.join(root,'examples/scores.csv'),'utf8'),'bad-for-config.csv');
assert.match(await evaluate("document.querySelector('#error').textContent"),/Invalid score/);
assert.equal(await evaluate("document.querySelector('#modelA').options.length"),0);
@@ -78,13 +108,32 @@ try{
const results=path.join(temporary,'results');fs.mkdirSync(results);
fs.copyFileSync(path.join(root,'examples/scores.csv'),path.join(results,'example.csv'));
const shared=path.join(temporary,'shared');
- execFileSync('python3',['-m','app.build','--results-dir',results,'--config',path.join(root,'configs/example.yaml'),'--output',shared],{cwd:root,stdio:'pipe'});
+ execFileSync('python3',['-m','app.build','--results-dir',results,'--catalogue',path.join(root,'configs/examples/catalogue.yaml'),'--weights',path.join(root,'configs/examples/weights.yaml'),'--eval-set',path.join(root,'configs/sets/any-available.yaml'),'--output',shared],{cwd:root,stdio:'pipe'});
await navigate(pathToFileURL(path.join(shared,'index.html')).href);
assert.equal(await evaluate("document.querySelector('#modelB').value"),'Example B');
for(const view of ['score','categories','languages','comparisons','config','warnings'])await click('[data-view='+view+']');
assert.equal(await evaluate("document.querySelector('#error').textContent"),'');
+ // Starting directly on a named subset must retain out-of-set data, including the demo.
+ const subset=path.join(temporary,'subset.yaml');
+ fs.writeFileSync(subset,serializeSuite({version:1,name:'Reasoning only',mode:'fixed',evals:[{name:'Example reasoning'}]}));
+ const fixedBuild=path.join(temporary,'fixed-build');
+ execFileSync('python3',['-m','app.build','--results-dir',results,'--catalogue',path.join(root,'configs/examples/catalogue.yaml'),'--weights',path.join(root,'configs/examples/weights.yaml'),'--eval-set',subset,'--output',fixedBuild],{cwd:root,stdio:'pipe'});
+ await navigate(pathToFileURL(path.join(fixedBuild,'index.html')).href);
+ assert.match(await evaluate("document.querySelector('#coverage').textContent"),/Reasoning only.*Complete.*2 extra measurements excluded/);
+ await click('[data-view=config]');assert.match(await evaluate("document.querySelector('#view').textContent"),/Example math/);
+ // Caveats follow comparison membership; excluded data remains inspectable with its caveat.
+ const caveated=structuredClone(fixture);caveated.evals[1].warning='Example math requires review.';
+ await upload('#configFile',serializeCatalogue(caveated),'caveat.yaml');
+ await click('[data-view=warnings]');
+ assert.match(await evaluate("document.querySelector('#view').textContent"),/Not used.*Example math/s);
+ assert.doesNotMatch(await evaluate("document.querySelector('#view').textContent"),/Example math requires review/);
+ await click('[data-view=config]');
+ assert.match(await evaluate("document.querySelector('[data-eval=\"Example math\"] .normalization-info').textContent"),/Example math requires review/);
+ await upload('#suiteFile',serializeSuite({version:1,name:'Freeform',mode:'available'}),'freeform.yaml');
+ await click('[data-view=warnings]');
+ assert.match(await evaluate("document.querySelector('#view').textContent"),/Config caveat.*Example math requires review/s);
assert.deepEqual(errors,[]);assert.deepEqual(network,[],'Loading and comparing local files must not send HTTP requests');
- console.log('Public browser checks passed: empty start, shared models, config choices, temporary uploads, rollback, clear models and no uploads.');
+ console.log('Public browser checks passed: empty start, shared models, independent weights and eval sets, required coverage, temporary uploads, rollback, clear models and no uploads.');
}finally{
ws?.close();fs.rmSync(temporary,{recursive:true,force:true});
}
diff --git a/tests/test_suites.cjs b/tests/test_suites.cjs
new file mode 100644
index 0000000..7937c89
--- /dev/null
+++ b/tests/test_suites.cjs
@@ -0,0 +1,83 @@
+'use strict';
+const test=require('node:test'),assert=require('node:assert/strict');
+const {validateCatalogue}=require('../app/eval_config.js');
+const {validateWeightProfile,parseWeightProfile,serializeWeightProfile,validateSuite,resolveConfig,scopeRows,suiteCoverage,parseSuite,serializeSuite}=require('../app/suite_config.js');
+const catalogue=()=>({version:1,name:'Catalogue',evals:[{name:'E',category:'C',match:{regex:'e_.+'},metric:'acc',filter:'none',score:{scale:1}},{name:'Unused',category:'D',match:{name:'unused'},metric:'acc',filter:'none',score:{scale:1}}],languages:[{tasks:['e_en'],scope:'single',language:'eng_Latn'},{tasks:['e_fr'],scope:'single',language:'fra_Latn'}]});
+const suite=()=>({version:1,name:'Required',mode:'fixed',evals:[{name:'E',variants:[{task:'e_en',n_shot:0},{task:'e_fr',n_shot:5}]}]});
+const profile=()=>({version:1,name:'Weights',weights:{C:.5,D:.5}});
+const row=(task,n_shot='0')=>({task,n_shot,eval:'E',selected:true});
+test('catalogue, weight profile and optional eval set are independent',()=>{
+ assert.equal(validateCatalogue(catalogue()).evals.length,2);
+ const c=resolveConfig(catalogue(),suite(),profile());assert.deepEqual(c.evals.map(e=>e.name),['E']);assert.deepEqual(c.weights,{C:.5,D:.5});
+ assert.deepEqual(parseSuite(serializeSuite(suite())),suite());
+ assert.throws(()=>validateCatalogue({...catalogue(),weights:{C:1}}));
+});
+test('fixed membership respects tasks and shots, ignores unselected alternate metrics',()=>{
+ const s=suite(),rows=[row('e_en'),row('e_fr','5'),row('e_fr','0'),row('e_de'),{...row('e_en'),selected:false}];
+ const scope=scopeRows(rows,s);assert.equal(scope.rows.length,2);assert.equal(scope.extras.length,2);assert.equal(scope.missing.length,0);
+ const c=suiteCoverage([row('e_en')],[row('e_en')],s);
+ assert.equal(c.complete,false);assert.equal(c.required,2);assert.equal(c.presentA,1);assert.equal(c.presentB,1);
+ assert.equal(c.warnings.length,2);assert.ok(c.warnings.every(w=>w.type==='Missing suite data'));
+});
+test('any available creates no missing or extra suite requirements',()=>{
+ const s={version:1,name:'Any available',mode:'available'};
+ assert.equal(resolveConfig(catalogue(),s,profile()).evals.length,2);
+ const c=suiteCoverage([row('e_en')],[],s);assert.deepEqual(c.warnings,[]);assert.equal(c.required,null);
+});
+test('eval-only requirement permits all its recognized variants',()=>{
+ const s={...suite(),evals:[{name:'E'}]};assert.equal(scopeRows([row('e_de')],s).missing.length,0);
+ assert.equal(scopeRows([],s).missing.length,1);
+});
+test('invalid suite definitions and catalogue references fail clearly',()=>{
+ for(const change of [s=>s.mode='wrong',s=>s.evals.push(s.evals[0]),s=>s.evals[0].variants.push(s.evals[0].variants[0]),s=>s.evals[0].variants[0].n_shot=-1,s=>s.evals[0].variants=[],s=>s.evals=[]]){const s=suite();change(s);assert.throws(()=>validateSuite(s));}
+ for(const change of [s=>s.evals[0].name='Absent',s=>s.evals[0].variants[0].task='unmapped']){const s=suite();change(s);assert.throws(()=>resolveConfig(catalogue(),s,profile()));}
+ assert.throws(()=>validateSuite({...suite(),mode:'available'}));
+});
+
+test('weight profiles validate independently and do not require an eval set',()=>{
+ assert.deepEqual(parseWeightProfile(serializeWeightProfile(profile())),profile());
+ for(const patch of [{weights:{C:-1}},{english_weights:{Z:.5}},{aggregate:'wrong'},{notes:null},{evals:[]}])assert.throws(()=>validateWeightProfile({...profile(),...patch}));
+ assert.throws(()=>validateSuite({...suite(),weights:{C:1}}));
+ assert.equal(resolveConfig(catalogue(),suite(),{...profile(),weights:{D:1}}).weights.C,0);
+});
+test('a fixed set is incomplete when required measurements have incompatible protocols',()=>{
+ const s={...suite(),evals:[{name:'E',variants:[{task:'e_en'}]}]};
+ const c=suiteCoverage([row('e_en','0')],[row('e_en','5')],s);
+ assert.equal(c.presentA,1);assert.equal(c.presentB,1);assert.equal(c.sharedRequired,0);assert.equal(c.complete,false);
+});
+test('fixed-set scores use the shared subset and contributions reconcile in every aggregate',()=>{
+ const {auditRows}=require('../app/eval_config.js'),{comparisonCoverage,totals,comparisonRows}=require('../app/app.js');
+ const c=catalogue(),s=suite(),p={...profile(),english_weights:{C:.5,D:.5}},config=resolveConfig(c,s,p);
+ const measurement=(task,shot,value,checkpoint='A')=>({checkpoint,task,n_shot:String(shot),value:String(value),metric:'acc',filter:'none',harness:'test',backend:'cpu'});
+ const a=auditRows([measurement('e_en',0,.8),measurement('e_fr',5,.6),measurement('e_de',0,1)],c);
+ const b=auditRows([measurement('e_en',0,.4,'B'),measurement('e_fr',0,.2,'B')],c);
+ const scope=suiteCoverage(a,b,s),coverage=comparisonCoverage(scope.a,scope.b,config);
+ assert.equal(scope.complete,false);assert.equal(scope.presentB,1);assert.equal(scope.sharedRequired,1);
+ assert.equal(scope.extrasA,1);assert.equal(scope.extrasB,1);
+ assert.ok(scope.warnings.some(w=>w.type==='Missing suite data'&&w.model==='B'));
+ assert.ok(coverage.warnings.some(w=>w.type==='Comparison coverage'));
+ const metadata=new Map(c.languages.flatMap(g=>g.tasks.map(t=>[t,g])));
+ for(const mode of ['standard','english_eval','english_category']){
+ const left=totals(coverage.a,config,config.weights,mode),right=totals(coverage.b,config,config.weights,mode);
+ assert.equal(left.score,80);assert.equal(right.score,40);
+ const bars=comparisonRows(coverage.pairs,coverage.a,config,config.weights,'eval','weighted','descending','delta',metadata,mode);
+ assert.equal(bars.reduce((n,b)=>n+b.weightedDelta,0),left.score-right.score);
+ }
+});
+test('an alternate metric cannot satisfy a required measurement',()=>{
+ const {auditRows}=require('../app/eval_config.js'),{collectWarnings}=require('../app/app.js');
+ const rows=auditRows([{checkpoint:'A',task:'e_en',n_shot:'0',value:'.8',metric:'acc_norm',filter:'none',harness:'test',backend:'cpu'}],catalogue());
+ assert.equal(scopeRows(rows,suite()).missing.length,2);
+ assert.equal(scopeRows(rows,suite()).extras.length,0);
+ assert.ok(collectWarnings(new Map([['A',rows]]),catalogue()).some(w=>w.type==='Missing scoring field'));
+});
+test('all shipped sets and profiles are independent and resolve against the global catalogue',()=>{
+ const fs=require('node:fs'),{parseCatalogue}=require('../app/eval_config.js');
+ const c=parseCatalogue(fs.readFileSync('configs/catalogue.yaml','utf8'));
+ const profiles=fs.readdirSync('configs/weights').filter(f=>f.endsWith('.yaml')).map(f=>parseWeightProfile(fs.readFileSync('configs/weights/'+f,'utf8')));
+ for(const p of profiles)for(const filename of fs.readdirSync('configs/sets').filter(f=>f.endsWith('.yaml'))){
+ const s=parseSuite(fs.readFileSync('configs/sets/'+filename,'utf8'));
+ resolveConfig(c,s,p);
+ assert.equal(s.weights,undefined);assert.equal(c.weights,undefined);assert.equal(p.evals,undefined);
+ }
+});
diff --git a/tests/test_warning_policy.cjs b/tests/test_warning_policy.cjs
new file mode 100644
index 0000000..bd5ccfe
--- /dev/null
+++ b/tests/test_warning_policy.cjs
@@ -0,0 +1,64 @@
+'use strict';
+const test=require('node:test'),assert=require('node:assert/strict');
+const {auditRows}=require('../app/eval_config.js');
+const {collectWarnings,comparisonCoverage,totals}=require('../app/app.js');
+const {resolveConfig,suiteCoverage}=require('../app/suite_config.js');
+const catalogue=()=>({version:1,name:'Rules',evals:['E','F'].map(name=>({name,category:'C',match:{name:name.toLowerCase()},metric:'acc_norm',filter:'none',score:{scale:1},warning:name+' caveat'})),languages:[{tasks:['e','f'],scope:'single',language:'eng_Latn'}]});
+const profile={version:1,name:'Weights',weights:{C:1}};
+const available={version:1,name:'Any available',mode:'available'};
+const fixed={version:1,name:'Required',mode:'fixed',evals:[{name:'E'},{name:'F'}]};
+const row=(task='e',metric='acc_norm')=>({checkpoint:'A',task,metric,filter:'none',n_shot:'0',harness:'test',backend:'cpu',value:'.8'});
+function run(left,right,suite=available){
+ const c=catalogue(),a=auditRows(left,c),b=auditRows(right.map(r=>({...r,checkpoint:'B'})),c),config=resolveConfig(c,suite,profile);
+ const scope=suiteCoverage(a,b,suite),coverage=comparisonCoverage(scope.a,scope.b,config);
+ const warnings=collectWarnings(new Map([['A',a],['B',b]]),c,'standard',{},suite,coverage.a).concat(scope.warnings,coverage.warnings);
+ return {warnings,scope,coverage,score:totals(coverage.a,config,config.weights).score};
+}
+test('displayed evals emit their caveats; unused catalogue rules and excluded evals do not',()=>{
+ const r=run([row()],[row()]);assert.deepEqual(r.warnings.filter(w=>w.type==='Config caveat').map(w=>w.name),['E']);
+ const excluded=run([row(),row('f')],[row(),row('f')],{...available,exclude:['F']});
+ assert.deepEqual(excluded.warnings.filter(w=>w.type==='Config caveat').map(w=>w.name),['E']);
+ assert.equal(excluded.warnings.filter(w=>w.type==='Not used'&&w.name==='F').length,2);
+ assert.equal(excluded.coverage.pairs.length,1);
+});
+test('unselected eval data warns even if it only contains the wrong metric',()=>{
+ const r=run([row(),row('f','acc')],[row()],{...fixed,evals:[{name:'E'}]});
+ assert.ok(r.warnings.some(w=>w.type==='Not used'&&w.name==='F'&&w.model==='A'));
+ assert.ok(!r.warnings.some(w=>w.type==='Missing scoring field'&&w.name==='f'));
+ assert.equal(r.coverage.pairs.length,1);assert.equal(r.score,80);
+});
+test('required eval absent from one or both models warns and is excluded from both scores',()=>{
+ for(const b of [[row()],[row(),row('f')]]){
+ const r=run([row()],b,fixed);assert.ok(r.warnings.some(w=>w.type==='Missing suite data'&&w.name==='F'));
+ assert.equal(r.scope.complete,false);assert.equal(r.coverage.pairs.length,1);assert.equal(r.score,80);
+ }
+});
+test('missing configured metric warns, excludes, and never substitutes an alternate metric',()=>{
+ const r=run([row(),row('f','acc')],[row(),row('f')],fixed);
+ assert.ok(r.warnings.some(w=>w.type==='Missing scoring field'&&w.name==='f'&&w.model==='A'));
+ assert.ok(r.warnings.some(w=>w.type==='Missing suite data'&&w.name==='F'));
+ assert.equal(r.coverage.pairs.length,1);assert.equal(r.score,80);
+});
+test('freeform mismatches warn and use only common measurements',()=>{
+ const r=run([row(),row('f')],[row()]);assert.ok(r.warnings.some(w=>w.type==='Comparison coverage'&&w.name==='F'));
+ assert.equal(r.coverage.pairs.length,1);assert.equal(r.score,80);
+ assert.ok(!r.warnings.some(w=>w.type==='Missing suite data'));
+});
+test('unconfigured eval data warns and cannot affect the score',()=>{
+ const r=run([row(),row('unknown')],[row(),row('unknown')]);
+ assert.equal(r.warnings.filter(w=>w.type==='No config'&&w.name==='unknown').length,2);
+ assert.equal(r.coverage.pairs.length,1);assert.equal(r.score,80);
+});
+test('a caveat is not advertised for an eval excluded by A/B coverage',()=>{
+ const r=run([row(),row('f')],[row()]);
+ assert.deepEqual(r.warnings.filter(w=>w.type==='Config caveat').map(w=>w.name),['E']);
+});
+test('available-mode exclusions are explicit, validated, and portable across catalogues',()=>{
+ const {validateSuite}=require('../app/suite_config.js');
+ for(const exclude of [null,'E',['E','E'],[''],[1]])assert.throws(()=>validateSuite({...available,exclude}));
+ assert.throws(()=>validateSuite({...fixed,exclude:['E']}));
+ assert.doesNotThrow(()=>resolveConfig(catalogue(),{...available,exclude:['Absent from this catalogue']},profile));
+ const r=run([row()],[row()],{...available,exclude:['E','F']});
+ assert.equal(r.score,null);assert.equal(r.coverage.pairs.length,0);
+ assert.equal(r.warnings.filter(w=>w.type==='Not used').length,2);
+});
diff --git a/tests/test_yaml.cjs b/tests/test_yaml.cjs
index ecee1bb..11170df 100644
--- a/tests/test_yaml.cjs
+++ b/tests/test_yaml.cjs
@@ -1,12 +1,28 @@
const assert=require('node:assert/strict'),fs=require('node:fs');
-const {parseConfig,serializeConfig}=require('../app/eval_config.js');
-const config=parseConfig(fs.readFileSync('configs/oellm.yaml','utf8'));
-assert.equal(config.evals.length,45);
-assert.deepEqual(parseConfig(serializeConfig(config)),config);
+const {parseCatalogue,serializeCatalogue,normalizeScore,auditRows}=require('../app/eval_config.js');
+const config=parseCatalogue(fs.readFileSync('configs/catalogue.yaml','utf8'));
+assert.equal(config.evals.length,46);
+assert.deepEqual(parseCatalogue(serializeCatalogue(config)),config);
+// The published OELLM config selects reliable SIB-200 accuracy and the paper's
+// approximate JEEBench baseline; neither is an unresolved config caveat.
+const sib=config.evals.find(e=>e.name==='SIB-200'),jee=config.evals.find(e=>e.name==='JEEBench');
+assert.equal(sib.metric,'acc');assert.equal(sib.warning,undefined);
+assert.equal(jee.normalize.min,.1055);assert.equal(jee.normalize.max,1);
+assert.equal(jee.normalize.clip,true);assert.notEqual(jee.normalize.basis,'unresolved');
+assert.equal(jee.warning,undefined);
+for(const [value,expected] of [[0,0],[.105,0],[.1055,0],[.55275,50],[1,100]]){
+ const score=normalizeScore(value,jee);
+ assert.ok(Math.abs(score.score_100-expected)<1e-10);
+ assert.ok(Math.abs(score.raw_score_100-value*100)<1e-10);
+}
+const sibRows=auditRows(['acc','acc_norm'].map(metric=>({checkpoint:'Fixture',task:'sib200_eng_Latn',metric,filter:'none',n_shot:'0',harness:'test',backend:'cpu',value:'.5'})),config);
+assert.deepEqual(sibRows.filter(r=>r.selected).map(r=>r.metric),['acc']);
+const {collectWarnings}=require('../app/app.js');
+assert.deepEqual(collectWarnings(new Map(),config),[]);
+assert.deepEqual(config.evals.filter(e=>e.warning).map(e=>e.name).sort(),['Global PIQA (prompted)','MultiBlimp']);
const source=`# A four-choice example with quoted regex and a flow mapping
version: 1
name: Example
-weights: {Reasoning: 1}
evals:
- name: Example eval
category: Reasoning
@@ -25,9 +41,20 @@ notes:
A folded explanation
for people editing the config.
`;
-const example=parseConfig(source);
+const example=parseCatalogue(source);
assert.equal(example.evals[0].normalize.min,.25);
assert.equal(example.evals[0].filter,'');
assert.equal(example.notes[0],'A folded explanation for people editing the config.');
-for(const bad of [source+'version: 1\n',source+'---\nversion: 1',source.replace('min: 0.25','min: .nan'),source.replace('filter: \'\'','filter: [oops'),source.replace('name: Example\n','name: !!js/function function(){}\n')])assert.throws(()=>parseConfig(bad));
+for(const bad of [source+'version: 1\n',source+'---\nversion: 1',source.replace('min: 0.25','min: .nan'),source.replace('filter: \'\'','filter: [oops'),source.replace('name: Example\n','name: !!js/function function(){}\n')])assert.throws(()=>parseCatalogue(bad));
console.log('YAML checks passed: source config, round-trip, comments, quoted regex, flow syntax, multiline notes, invalid syntax and duplicate keys.');
+
+assert.equal(config.evals.find(e=>e.name==='AMC23').normalize.min,0);
+assert.equal(config.evals.find(e=>e.name==='ARC Easy').normalize.min,.25);
+assert.equal(config.evals.find(e=>e.name==='FLORES200').metric,'chrf++');
+assert.equal(config.evals.find(e=>e.name==='OpenSubtitles').metric,'chrf');
+for(const e of config.evals.filter(e=>e.category==='Translation')){assert.equal(e.warning,undefined);assert.equal(e.normalize.min,0);assert.equal(e.score.scale,100);}
+const {parseSuite,inSuite}=require('../app/suite_config.js');
+const freeform=parseSuite(fs.readFileSync('configs/sets/any-available.yaml','utf8'));
+const prompted=config.evals.find(e=>e.name==='Global PIQA (prompted)');
+assert.match(prompted.warning,/Normalization and metric selection have not been validated/);
+assert.equal(inSuite({eval:prompted.name},freeform),false);