diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index af9cc0f..010e6dc 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -24,10 +24,24 @@ jobs: - uses: actions/setup-node@v7 with: node-version: '22' + - name: Install the native Python library + run: python3 -m pip install -e . - name: Check scoring, imports, and configuration run: | - python3 -m unittest tests.test_data - node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs + python3 -m unittest tests.test_data tests.test_engines + node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs tests/test_components.cjs + - name: Check the installed package without Node on PATH + run: | + python3 -m pip wheel . --wheel-dir "$RUNNER_TEMP/quickdash-wheels" + python3 -m venv "$RUNNER_TEMP/quickdash-package" + "$RUNNER_TEMP/quickdash-package/bin/python" -m pip install --no-index \ + --find-links "$RUNNER_TEMP/quickdash-wheels" oellm-quickdash + cd "$RUNNER_TEMP" + PATH="$RUNNER_TEMP/quickdash-package/bin" \ + "$RUNNER_TEMP/quickdash-package/bin/quickdash" "$GITHUB_WORKSPACE/examples/scores.csv" \ + --catalogue "$GITHUB_WORKSPACE/configs/examples/catalogue.yaml" \ + --weights "$GITHUB_WORKSPACE/configs/examples/weights.yaml" \ + --compare 'Example A' 'Example B' --format json > quickdash-installed.json - name: Check the browser with public fixtures run: | google-chrome --headless --no-sandbox --disable-gpu \ @@ -40,7 +54,7 @@ jobs: node tests/test_public_browser.mjs || { cat "$RUNNER_TEMP/quickdash-chrome.log"; exit 1; } - name: Build shared results and the fictional demo run: | - python3 -m app.build --results-dir results \ + python3 -m app.build --results-dir results --sample-csv examples/sample-evals.csv \ --output output/shared > "$RUNNER_TEMP/quickdash-shared.json" python3 -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \ --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml \ diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..80ecd1d --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,36 @@ +# Working on Quickdash + +Quickdash has two implementations of the scoring contract: the native Python +library in `quickdash/` and the browser engine in `app/analysis.js` with its +configuration parsers. Neither implementation is the reference for the other. + +## Keep Python and JavaScript in sync + +- Every change to scoring, configuration interpretation, input validation, + coverage/exclusion policy, or diagnostic conditions must include a shared + regression case in `tests/test_engines.py` that runs through both engines. + A bug fix should fail before the fix and pass in both implementations afterward. +- Compare scores, trees, effective weights, contributions, included/excluded + measurement identities, and diagnostic codes and context. Human-facing warning + wording does not need to match. Invalid inputs must be rejected by both engines. +- Include independently calculated expectations or invariants: agreement alone + does not prove correctness if both implementations share the same mistake. +- Do not assume fixed eval, category, language, component, profile, or set counts. + Exercise changing contents and sizes. The full sample test discovers shipped + profiles and sets automatically and covers every supported aggregation mode. +- Keep these checks in CI. Do not skip parity tests when changing only one engine, + and do not weaken comparisons merely to accommodate a disagreement. +- For dashboard behavior changes, also extend the public browser tests. Node + exercises the actual browser calculation module; browser tests check that the + UI supplies and renders those calculations correctly. + +Run the shared contract and scoring checks before publishing: + +```sh +python -m unittest tests.test_data tests.test_engines +node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs tests/test_components.cjs +``` + +See [development and publishing](docs/development.md) for browser checks and +standalone builds. Keep README setup and common commands accurate, and update the +relevant guide when public interfaces or scoring policies change. diff --git a/README.md b/README.md index 7310b2f..2ae2912 100644 --- a/README.md +++ b/README.md @@ -4,9 +4,11 @@ A standalone, offline dashboard for comparing model evaluation scores. Explore c **[Open the dashboard](https://openeurollm.github.io/quickdash/)** or **[try the fictional example](https://openeurollm.github.io/quickdash/demo.html)**. No installation is needed to use either page. +While no shared CSVs have been added to `results/`, the main page opens with our [sample eval export](examples/README.md) and synthetic comparison choices. Shared CSVs replace that fallback automatically on the next successful deployment. + ## Compare models -1. Select shared models as A and B, or use **Add model CSV** to open your exports. With one real model, a labelled synthetic comparison is supplied for exploring the interface. +1. Select shared models as A and B, or use **Add model CSV** to open your exports. Three labelled synthetic comparisons (perturbed, higher, and lower scores) are supplied for exploring the interface. 2. Choose a **Weighting profile**. Leave **Eval set** on **Any available** to compare the measurements both models have, or select **flagship-1** to check an expected set. Weights and eval sets are independent. The supplied sets exclude prompted Global PIQA pending scoring validation. 3. Review **Warnings**, then explore the scores and breakdowns. The global catalogue determines how to interpret each eval: category, scoring field, normalization, and language assignments. @@ -26,12 +28,15 @@ The [Pages workflow](.github/workflows/pages.yml) publishes only the generated d ## Build a standalone file -Building requires Python 3.8+ and Node.js 18+. The YAML parser is [bundled with its license](app/vendor/README.md); no package installation is needed. +Building requires Python 3.8+ and the Python package below. Node.js is needed only for development tests; the generated dashboard remains standalone and offline. ```sh git clone https://github.com/OpenEuroLLM/quickdash.git cd quickdash -python3 -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \ +python3 -m venv .venv +source .venv/bin/activate +python -m pip install -e . +python -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \ --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml --output output/example open output/example/index.html # macOS; elsewhere, open it in your browser ``` @@ -41,10 +46,10 @@ The example compares two fictional models with a multilingual reasoning eval and To build the shared dashboard, including all shared results and config choices: ```sh -python3 -m app.build --results-dir results --output output/shared +python3 -m app.build --results-dir results --sample-csv examples/sample-evals.csv --output output/shared ``` -An empty `results/` directory produces a page ready for local CSV imports. To start without embedded models regardless of the directory’s contents, omit `--results-dir`. +The sample is used only when `results/` contains no CSVs. Omit `--sample-csv` to leave an empty results directory upload-ready; omit both options to start without embedded models regardless of shared data. For private exports, put your CSV in the ignored `data/` directory: @@ -67,15 +72,29 @@ The builder replaces the files it generates in the chosen output directory, incl Filters affect inspection views, while composite scores and contribution weights use shared coverage within the selected eval set. A named set with missing requirements is labelled incomplete; it uses the shared subset with redistributed weights. Unmatched measurements are excluded from both compared scores with warnings. Malformed input and invalid selected scores are rejected; failed imports preserve the active dashboard. See the [data-handling policy](docs/configuration.md#data-validation-and-failure-behavior). -Language/category breakdowns show descriptive raw averages. Weighted scores use configured normalization, whose baselines and limitations are visible per eval. Shared numerical scales do not establish comparable difficulty across benchmarks. Unknown/mixed-language scores use the documented English fallback for balancing; this does not change their language labels. +Language/category breakdowns show descriptive raw averages for ordinary evals. Evals with configured components, including PolyMath, show calculated scores when collapsed; expand them to see individual raw scores, relative component weights, and contributions. PolyMath stores `relative_weight` values of 1, 2, 4, and 8, divided by their total of 15 when scoring; incomplete language/protocol groups are excluded with warnings. Incompatible component configurations are rejected before taking effect. See [component aggregation](docs/configuration.md#weighted-components-within-an-eval). Weighted scores use configured normalization, whose baselines and limitations are visible per eval. Shared numerical scales do not establish comparable difficulty across benchmarks. Unknown/mixed-language scores use the documented English fallback for balancing; this does not change their language labels. + +## Analyze from Python or the command line + +The native Python library uses the same CSVs, YAML rules, weighting profiles, and optional eval sets as the dashboard. It returns calculated trees, effective weights, contributions, coverage, and structured diagnostics. Python emits warnings by default; applications can explicitly collect them or reject results with warnings. See the [Python API and CLI guide](docs/python-api.md). + +After installing the package as above: + +```sh +quickdash examples/scores.csv --catalogue configs/examples/catalogue.yaml \ + --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml \ + --compare 'Example A' 'Example B' +``` + +Add `--format json` for a complete report. Diagnostics go to stderr. Python and browser engines run against the same behavioral test cases. ## Development -Application code lives in `app/`, tests in `tests/`, and contributor documentation in `docs/`. Common checks: +Application code lives in `app/` and `quickdash/`, tests in `tests/`, and contributor documentation in `docs/`. Common checks: ```sh -python3 -m unittest tests.test_data -node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs +python -m unittest tests.test_data tests.test_engines +node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs tests/test_components.cjs ``` See [development and publishing](docs/development.md) for the source layout, browser tests, and GitHub Pages workflow. diff --git a/app/analysis.js b/app/analysis.js new file mode 100644 index 0000000..fd60e97 --- /dev/null +++ b/app/analysis.js @@ -0,0 +1,293 @@ +'use strict'; +const QuickdashAnalysis=(()=>{ +'use strict'; +const key = r => JSON.stringify(['task','metric','filter','n_shot','harness','backend'].map(k=>r[k])); +const avg = xs => xs.length ? xs.reduce((a,b)=>a+b,0)/xs.length : null; +const fmt = (x,digits=2) => x === null || !Number.isFinite(x) ? '—' : x.toFixed(digits); +const esc = x => String(x??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); +const {parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel,isDemoModel}=typeof module!=='undefined'?require('./eval_config.js'):EvalConfig; +const {parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveConfig,inSuite,suiteCoverage,scopeRows}=typeof module!=='undefined'?require('./suite_config.js'):SuiteConfig; +function selectRows(rows,scheme){const selected=auditRows(rows,scheme).filter(r=>r.selected);if(!selected.length)throw Error('No selected measurements');return selected;} +function buildCatalogue(rows,scheme){ + return scheme.evals.map(f=>{const tasks=new Map();for(const r of rows.filter(r=>r.eval===f.name)){if(!tasks.has(r.task))tasks.set(r.task,[]);tasks.get(r.task).push(r);}return {...f,tasks:[...tasks].map(([name,rows])=>({name,rows})).sort((a,b)=>a.name.localeCompare(b.name))};}).filter(f=>f.tasks.length); +} +function comparisonRows(pairs,reference,scheme,weights,group,measure,sort,sortBy='delta',metadata=new Map(),aggregate='standard',englishWeights=scheme.english_weights||{}){ + const allocation=totals(reference,scheme,weights,aggregate,englishWeights,metadata),coefficients=new Map(reference.map(r=>[key(r),allocation.rowWeights.get(r)])); + const groups=new Map();for(const r of pairs){const k=group==='category'?r.category:group==='eval'?r.eval:key(r);if(!groups.has(k))groups.set(k,[]);groups.get(k).push(r);} + const items=[...groups].map(([k,rows])=>{ + const raw=group==='category'?avg([...new Set(rows.map(r=>r.eval))].map(f=>avg(rows.filter(r=>r.eval===f).map(r=>r.delta)))):avg(rows.map(r=>r.delta)); + const weighted=rows.reduce((sum,r)=>sum+r.score_delta*coefficients.get(key(r)),0); + return {a:group==='category'?avg([...new Set(rows.map(r=>r.eval))].map(f=>avg(rows.filter(r=>r.eval===f).map(r=>r.a)))):avg(rows.map(r=>r.a)),b:group==='category'?avg([...new Set(rows.map(r=>r.eval))].map(f=>avg(rows.filter(r=>r.eval===f).map(r=>r.b)))):avg(rows.map(r=>r.b)),tasks:[...new Set(rows.map(r=>r.task))],languageCount:languageCoverage(rows,metadata).count,weightedDelta:weighted,task:group==='variant'?rows[0].task:'',label:group==='variant'?rows[0].task+' · '+rows[0].n_shot+' shot':k,category:rows[0].category,eval:group==='category'?'':rows[0].eval,count:rows.length,rawDelta:raw,delta:measure==='weighted'?weighted:raw}; + }); + const field=sort==='name'?'label':sortBy,direction=sort==='ascending'||sort==='name'?1:-1; + items.sort((a,b)=>{const x=a[field],y=b[field],difference=typeof x==='string'?x.localeCompare(y):sort==='absolute'?Math.abs(x)-Math.abs(y):x-y;return direction*difference||a.label.localeCompare(b.label);}); + return items; +} +function weightingLanguage(row,metadata){ + const m=metadata.get(row.task); + return m?.scope==='translation'?m.target_language:['single','pooled'].includes(m?.scope)?m.language:null; +} +function scoreLanguage(row,metadata){ + const language=weightingLanguage(row,metadata); + return !language||language==='mul'||language==='eng_Latn'?'english':'other'; +} +function englishAssignment(row,metadata){ + const language=weightingLanguage(row,metadata); + return !language||language==='mul'?'English (fallback: unknown or mixed language; weighting only)':language==='eng_Latn'?'English':'Other languages'; +} +// Component completeness is checked before scoring, within an explicit language and protocol. +function componentCoverage(rows,config){ + const accepted=new Set(),groups=[],warnings=[]; + const metadata=new Map((config.languages||[]).flatMap(g=>g.tasks.map(t=>[t,g]))); + for(const e of config.evals){ + const rr=rows.filter(r=>r.eval===e.name); + if(!e.aggregation){for(const r of rr)accepted.add(r);continue;} + const buckets=new Map(); + for(const r of rr){ + const m=metadata.get(r.task),language=m?.scope==='translation'?m.source_language+' → '+m.target_language:m?.language; + const id=JSON.stringify([r.checkpoint,language||'Unknown',m?.scope,...['metric','filter','n_shot','harness','backend'].map(k=>r[k])]); + if(!buckets.has(id))buckets.set(id,{language:language||'Unknown',known:!!language,rows:[]});buckets.get(id).rows.push(r); + } + for(const group of buckets.values()){ + const matches=new Map(e.aggregation.components.map(c=>[c,[]])),problems=[]; + if(!group.known)problems.push('explicit language assignment missing'); + for(const r of group.rows){const cc=e.aggregation.components.filter(c=>matchTask(c.match,r.task));if(cc.length!==1)problems.push(r.task+': '+(cc.length?'ambiguous component matches':'no component match'));else matches.get(cc[0]).push(r);} + for(const [c,rr] of matches)if(rr.length!==1)problems.push(c.name+': '+(rr.length?'multiple results':'missing')); + if(problems.length){warnings.push({type:'Incomplete components',name:e.name,eval:e.name,detail:group.language+' · '+group.rows[0].n_shot+' shots: '+problems.join('; ')+'. This language/protocol group is excluded from the calculation. All configured components are required; raw results remain in Eval configuration.',variants:[{settings:'Excluded component group',tasks:group.rows.map(r=>r.task)}]});continue;} + const total=e.aggregation.components.reduce((sum,c)=>sum+c.relative_weight,0),parts=[...matches].map(([c,rr])=>({row:rr[0],component:c,share:c.relative_weight/total})); + for(const r of group.rows)accepted.add(r);groups.push({...group,eval:e.name,parts}); + } + } + return {rows:rows.filter(r=>accepted.has(r)),excluded:rows.filter(r=>!accepted.has(r)),groups,warnings}; +} +// Each complete language/protocol group has equal influence within its eval. +function evalDistribution(rows,config){ + const e=config.evals.find(e=>e.name===rows[0]?.eval),coefficients=new Map(); + const coverage=e?.aggregation?componentCoverage(rows,{...config,evals:[e]}):{rows,groups:[]}; + if(e?.aggregation){for(const g of coverage.groups)for(const p of g.parts)coefficients.set(p.row,p.share/coverage.groups.length);} + else for(const r of coverage.rows)coefficients.set(r,1/coverage.rows.length); + return {rows:coverage.rows,coefficients,score:coverage.rows.length?coverage.rows.reduce((sum,r)=>sum+r.score_100*coefficients.get(r),0):null}; +} +function totals(rows,scheme,weights,aggregate='standard',englishWeights=scheme.english_weights||{},metadata=new Map((scheme.languages||[]).flatMap(g=>g.tasks.map(task=>[task,g])))){ + const rowWeights=new Map(rows.map(r=>[r,0])); + const fs=scheme.evals.map(f=>{const distribution=evalDistribution(rows.filter(r=>r.eval===f.name),scheme),rr=distribution.rows;return {...f,score:distribution.score,rows:rr,weight:0,contribution:rr.length?null:0,aggregateScore:null,excluded:!rr.length,englishShare:0,effectiveEnglishShare:null,englishScore:null,otherScore:null,issue:''};}); + const availableWeight=Object.entries(weights).filter(([name])=>fs.some(f=>f.category===name&&f.rows.length)).reduce((sum,[,weight])=>sum+weight,0); + const cats=Object.keys(weights).map(name=>{ + const configured=fs.filter(f=>f.category===name),ff=configured.filter(f=>f.rows.length),share=aggregate!=='standard'?(Object.hasOwn(englishWeights,name)?englishWeights[name]:0):0,split=share!==0; + const c={name,weight:ff.length&&availableWeight?weights[name]/availableWeight:0,excluded:!ff.length,excludedEvals:configured.filter(f=>!f.rows.length).map(f=>f.name),score:null,englishShare:share,effectiveEnglishShare:null,englishScore:null,otherScore:null,issue:''}; + if(c.excluded)return c; + if(!Number.isFinite(share)||share<0||share>1)c.issue='English share must be between 0 and 1.'; + else if(!split){c.score=avg(ff.map(f=>f.score));for(const f of ff)for(const [r,w] of evalDistribution(f.rows,scheme).coefficients)rowWeights.set(r,c.weight/ff.length*w);} + else if(aggregate==='english_eval'){ + for(const f of ff){ + f.englishShare=share; + const groups=['english','other'].map(side=>f.rows.filter(r=>scoreLanguage(r,metadata)===side)); + [f.englishScore,f.otherScore]=groups.map(group=>evalDistribution(group,scheme).score); + f.effectiveEnglishShare=f.englishScore===null?0:f.otherScore===null?1:share; + f.aggregateScore=f.effectiveEnglishShare*(f.englishScore??0)+(1-f.effectiveEnglishShare)*(f.otherScore??0); + groups.forEach((group,i)=>{for(const [r,w] of evalDistribution(group,scheme).coefficients)rowWeights.set(r,c.weight/ff.length*(i===0?f.effectiveEnglishShare:1-f.effectiveEnglishShare)*w);}); + } + if(!c.issue)c.score=avg(ff.map(f=>f.aggregateScore)); + } + else { + const groups=['english','other'].map(side=>ff.map(f=>({eval:f,rows:f.rows.filter(r=>scoreLanguage(r,metadata)===side)})).filter(f=>f.rows.length)); + [c.englishScore,c.otherScore]=groups.map(group=>avg(group.map(f=>evalDistribution(f.rows,scheme).score))); + if(!c.issue){ + c.effectiveEnglishShare=c.englishScore===null?0:c.otherScore===null?1:share; + const parts=[c.effectiveEnglishShare,1-c.effectiveEnglishShare]; + c.score=parts[0]*(c.englishScore??0)+parts[1]*(c.otherScore??0); + groups.forEach((group,i)=>{for(const f of group)for(const [r,w] of evalDistribution(f.rows,scheme).coefficients)rowWeights.set(r,c.weight*parts[i]/group.length*w);}); + } + } + for(const f of ff){f.weight=f.rows.reduce((sum,r)=>sum+rowWeights.get(r),0);f.contribution=c.score===null?null:f.rows.reduce((sum,r)=>sum+r.score_100*rowWeights.get(r),0);f.aggregateScore=c.score!==null&&f.weight?f.contribution/f.weight:null;} + return c; + }); + const valid=Object.values(weights).every(w=>Number.isFinite(w)&&w>=0)&&Math.abs(Object.values(weights).reduce((s,w)=>s+w,0)-1)<1e-8; + return {evals:fs,categories:cats,rowWeights,score:valid&&availableWeight>0&&cats.filter(c=>!c.excluded).every(c=>c.score!==null)?cats.reduce((s,c)=>s+(c.excluded?0:c.score*c.weight),0):null}; +} +function pairRows(a,b){const bm=new Map(b.map(r=>[key(r),r]));return a.filter(r=>bm.has(key(r))).map(r=>({...r,a:r.raw_score_100,b:bm.get(key(r)).raw_score_100,delta:r.raw_score_100-bm.get(key(r)).raw_score_100,score_delta:r.score_100-bm.get(key(r)).score_100}));} +function sampleCount(row){const value=String(row.n_samples??'');return /^[1-9][0-9]*$(?![\s\S])/.test(value)&&Number.isSafeInteger(Number(value))?Number(value):null;} +function comparisonCoverage(a,b,config){ + const left=componentCoverage(a,config),right=componentCoverage(b,config),initial=pairRows(left.rows,right.rows),shared=new Set(initial.map(key)); + const sharedLeft=componentCoverage(left.rows.filter(r=>shared.has(key(r))),config),sharedRight=componentCoverage(right.rows.filter(r=>shared.has(key(r))),config); + const aa=sharedLeft.rows,bb=sharedRight.rows,pairs=pairRows(aa,bb),matched=new Set(pairRows(a,b).map(key)); + const onlyA=a.filter(r=>!matched.has(key(r))),onlyB=b.filter(r=>!matched.has(key(r))),warnings=[]; + for(const [model,coverage] of [['A',left],['B',right],['Shared A',sharedLeft],['Shared B',sharedRight]])for(const w of coverage.warnings)warnings.push({...w,detail:model+': '+w.detail}); + const excludedA=a.filter(r=>!aa.includes(r)),excludedB=b.filter(r=>!bb.includes(r)); + for(const e of config.evals){const left=onlyA.filter(r=>r.eval===e.name),right=onlyB.filter(r=>r.eval===e.name);if(!left.length&&!right.length)continue; + const hasShared=pairs.some(r=>r.eval===e.name); + warnings.push({type:'Comparison coverage',name:e.name,eval:e.name,detail:(hasShared?'Unmatched variants are excluded from both scores.':'No matching scores: this eval is excluded from both scores.')+' '+left.length+' measurement(s) available only in A; '+right.length+' only in B. Remaining evals share their category weight; empty categories are excluded and remaining category weights are rescaled.',variants:[[left,'Available only in A (missing from B)'],[right,'Available only in B (missing from A)']].filter(([rr])=>rr.length).map(([rr,settings])=>({settings,tasks:rr.map(r=>r.task+' · '+r.metric+' / '+(r.filter||'blank filter')+' / '+r.n_shot+' shots / '+r.harness+' / '+r.backend)}))}); + } + const bm=new Map(bb.map(r=>[key(r),r])); + for(const e of config.evals){ + const mismatch=aa.filter(r=>r.eval===e.name&&sampleCount(r)!==null&&sampleCount(bm.get(key(r)))!==null&&sampleCount(r)!==sampleCount(bm.get(key(r)))); + if(mismatch.length)warnings.push({type:'Sample-count mismatch',name:e.name,eval:e.name,detail:'Matched measurements report different n_samples for A and B. Scores remain included; review dataset coverage before comparing.',variants:[{settings:'Reported sample counts',tasks:mismatch.map(r=>r.task+' · '+r.metric+' · A: '+r.n_samples+' / B: '+bm.get(key(r)).n_samples)}]}); + } + return {a:aa,b:bb,pairs,warnings,onlyA,onlyB,excludedA,excludedB}; +} +const syntheticOptions=Object.freeze([ + Object.freeze({name:demoModel,offset:0}), + Object.freeze({name:'SYNTHETIC demo — higher scores',offset:3}), + Object.freeze({name:'SYNTHETIC demo — lower scores',offset:-3}) +]); +function synthetic(rows,config,option=syntheticOptions[0]){ + let seed=20260930; + const rand=()=>{seed=(Math.imul(1664525,seed)+1013904223)>>>0;return (seed+.5)/4294967296;}; + const evals=new Map(config.evals.map(e=>[e.name,e])); + return rows.map(r=>{ + const perturb=2*Math.sqrt(-2*Math.log(rand()))*Math.cos(2*Math.PI*rand()),e=evals.get(r.eval); + const value=Math.max(0,Math.min(100,r.raw_score_100+option.offset+perturb))/100*e.score.scale; + return {...r,checkpoint:option.name,value:String(value),...normalizeScore(value,e),stderr:'',result_time:'',results_file:`synthetic: seed 20260930; mean shift ${option.offset}; normal sd 2 score points; clipped to [0,100]`}; + }); +} +function languageRoles(row,metadata){const m=metadata.get(row.task);return m?.scope==='translation'?[{language:m.source_language,role:'from'},{language:m.target_language,role:'to'}]:[{language:m?.language||'Unknown',role:'eval'}];} +function matchesLanguage(row,metadata,language='',role=''){return languageRoles(row,metadata).some(m=>(!language||m.language===language)&&(!role||m.role===role));} +function languageCoverage(rows,metadata){ + const codes=new Set(),pooled=new Set(),unknown=new Set(); + for(const r of rows){const m=metadata.get(r.task);if(!m||m.scope==='unknown')unknown.add(r.task);else if(m.scope==='pooled')pooled.add(m.language);else for(const role of languageRoles(r,metadata))codes.add(role.language);} + return {count:codes.size,pooled:pooled.size,unknown:unknown.size}; +} +function languageCountLabel(coverage){return [coverage.count?String(coverage.count):'',coverage.pooled?coverage.pooled+' pooled':'',coverage.unknown?'unknown':''].filter(Boolean).join(' + ')||'0';} +function languageLabel(code){return code==='mul'?'Multilingual (pooled)':code;} +function sortBreakdownTree(tree,field='label',order='ascending'){ + const label=n=>n.kind==='language'?languageLabel(n.label):n.label; + const value=n=>field==='label'?label(n):n[field]; + return tree.map(n=>({...n,children:sortBreakdownTree(n.children,field,order)})).sort((a,b)=>{ + const x=value(a),y=value(b),missing=v=>v===null||v===undefined||typeof v==='number'&&!Number.isFinite(v); + if(missing(x)!==missing(y))return missing(x)?1:-1; + const difference=missing(x)?0:typeof x==='string'?x.localeCompare(y):x-y; + return (order==='ascending'?1:-1)*difference||label(a).localeCompare(label(b))||(a.detail||'').localeCompare(b.detail||''); + }); +} +// Summaries use distinct measurements, independent of repeated translation branches. +function breakdownAggregate(rows,config=null){ + const unique=[...new Map(rows.map(r=>[key(r),r])).values()],evals=[...new Set(unique.map(r=>r.eval))]; + const scores=evals.map(name=>{ + const rr=unique.filter(r=>r.eval===name),e=config?.evals.find(e=>e.name===name); + if(!e?.aggregation)return {a:avg(rr.map(r=>r.a)),b:avg(rr.map(r=>r.b))}; + const dist=evalDistribution(rr,config); + if(dist.rows.length!==rr.length||!rr.length)return {a:null,b:null}; + return {a:dist.score,b:rr.reduce((sum,r)=>sum+(r.score_100-r.score_delta)*dist.coefficients.get(r),0)}; + }); + const a=scores.some(s=>s.a===null)?null:avg(scores.map(s=>s.a)),b=scores.some(s=>s.b===null)?null:avg(scores.map(s=>s.b)); + return {a,b,delta:a===null||b===null?null:a-b,count:unique.length,evals:evals.length}; +} +function buildBreakdownTree(rows,metadata,view,languageFilter='',roleFilter='',config=null,reference=rows){ + const components=new Map(); + if(config)for(const group of componentCoverage(reference,config).groups)for(const p of group.parts)components.set(key(p.row),{componentName:p.component.name,componentRelativeWeight:p.component.relative_weight,componentShare:p.share,componentNormalizedA:p.row.score_100,componentNormalizedB:p.row.score_100-p.row.score_delta,componentA:p.row.score_100*p.share,componentB:(p.row.score_100-p.row.score_delta)*p.share}); + const groups=(rr,values)=>{const map=new Map();for(const r of rr)for(const value of new Set(values(r))){if(!map.has(value))map.set(value,[]);map.get(value).push(r);}return [...map].sort(([a],[b])=>a.localeCompare(b));}; + const node=(kind,label,rr,children=[])=>({kind,label,...breakdownAggregate(rr,config),componentAggregate:rr.length>0&&rr.every(r=>config?.evals.find(e=>e.name===r.eval)?.aggregation),children}); + const leaves=rr=>rr.slice().sort((a,b)=>a.task.localeCompare(b.task)||key(a).localeCompare(key(b))).map(r=>({...node('variant',r.task,[r]),...breakdownAggregate([r]),...components.get(key(r)),detail:r.metric+' · '+(r.filter||'no filter')+' · '+r.n_shot+' shot'})); + const languageGroups=rr=>groups(rr,r=>languageRoles(r,metadata).filter(m=>(!languageFilter||m.language===languageFilter)&&(!roleFilter||m.role===roleFilter)).map(m=>m.language)); + function details(rr,language){ + const ordinary=rr.filter(r=>metadata.get(r.task)?.scope!=='translation'),translation=rr.filter(r=>metadata.get(r.task)?.scope==='translation'); + const children=leaves(ordinary); + for(const role of ['from','to']){ + if(roleFilter&&roleFilter!==role)continue; + const directed=translation.filter(r=>matchesLanguage(r,metadata,language,role));if(!directed.length)continue; + children.push(node('direction',(role==='from'?'From ':'To ')+language,directed,groups(directed,r=>{const m=metadata.get(r.task);return [m.source_language+' → '+m.target_language];}).map(([pair,rr])=>node('pair',pair,rr,leaves(rr))))); + } + return children; + } + if(view==='category')return groups(rows,r=>[r.category]).map(([category,rr])=>node('category',category,rr,groups(rr,r=>[r.eval]).map(([name,ee])=>node('eval',name,ee,languageGroups(ee).map(([language,ll])=>node('language',language,ll,details(ll,language))))))); + return languageGroups(rows).map(([language,ll])=>node('language',language,ll,groups(ll,r=>[r.category]).map(([category,cc])=>node('category',category,cc,groups(cc,r=>[r.eval]).map(([name,ee])=>node('eval',name,ee,details(ee,language))))))); +} +function protocolWarning(rows,evalConfig,config,model){ + // Compare sets per task: identical alternate settings in every language are consistent. + const fields=['n_shot','metric','filter','harness','backend'],tasks=new Map(); + for(const r of rows.filter(r=>r.selected)){if(!tasks.has(r.task))tasks.set(r.task,new Set());tasks.get(r.task).add(JSON.stringify(fields.map(f=>String(r[f]))));} + const groups=new Map();for(const [task,settings] of tasks){const signature=JSON.stringify([...settings].sort());if(!groups.has(signature))groups.set(signature,[]);groups.get(signature).push(task);} + if(groups.size<2)return null; + const variants=[...groups].map(([signature,names])=>{ + const settings=JSON.parse(signature).map(k=>{const [shots,metric,filter,harness,backend]=JSON.parse(k);return shots+' shots / '+metric+' / filter '+(filter||'(empty)')+' / '+harness+' / '+backend;}).join('; '); + const labels=names.sort().map(task=>{const m=taskLanguage(task,config),language=m.scope==='translation'?m.source_language+' → '+m.target_language:m.language||'Unknown';return task+' ['+language+']';}); + return {settings,tasks:labels}; + }); + return {type:'Inconsistent scoring settings',name:evalConfig.name,eval:evalConfig.name,model,detail:'Selected variants use different settings: '+variants.map(v=>v.settings+' ('+v.tasks.length+' tasks; '+v.tasks.slice(0,2).join(', ')+(v.tasks.length>2?', …':'')+')').join(' versus ')+'. '+(evalConfig.aggregation?'Only complete component groups within each protocol are included.':'These variants are still included in the aggregate.')+' Review the protocol before comparing languages.',variants}; +} +function normalizationLabel(e){const n=e.normalize;if(!n||n.min===0&&n.max===1)return n?.basis==='unresolved'?'Unresolved · no correction':'No chance correction';return fmt(n.min*100,2)+'% baseline';} +function sameCoverage(a,b){return a.length===b.length&&pairRows(a,b).length===a.length;} +// Public, serializable analysis boundary used by Python parity tests and the UI. +function compareText(a,b){const aa=Array.from(a,c=>c.codePointAt(0)),bb=Array.from(b,c=>c.codePointAt(0));for(let i=0;ir[k])]);} +const diagnosticTitles={config_caveat:'Config caveat',no_config:'No config',not_used:'Not used',unknown_language:'Unknown language',invalid_sample_count:'Invalid sample count',inconsistent_scoring_settings:'Inconsistent scoring settings',missing_scoring_field:'Missing scoring field',missing_scoring_setting:'Missing scoring setting',no_selected_score:'No selected score',missing_suite_data:'Missing suite data',incomplete_components:'Incomplete components',comparison_coverage:'Comparison coverage',sample_count_mismatch:'Sample-count mismatch',no_category_weight:'No category weight'}; +function diagnostic(code,model,evalName,rows,detail,effect='included',tasks=null){ + const names=[...new Set(tasks??rows.map(r=>r.task))].sort(compareText); + return {code,type:diagnosticTitles[code],model,eval:evalName,name:evalName||names[0]||model,tasks:names,measurement_ids:rows.map(measurementId).sort(compareText),effect,detail,variants:names.length?[{settings:detail,tasks:names}]:[]}; +} +function reportDiagnostics(audits,config,included,comparison=false){ + const {catalogue,suite,profile}=config,scheme=resolveConfig(catalogue,suite,profile),out=[]; + const used=new Set([...included.values()].flat().map(r=>r.eval)); + for(const e of catalogue.evals)if(e.warning&&used.has(e.name))out.push(diagnostic('config_caveat','Selected comparison',e.name,[],e.warning)); + for(const [model,rows] of audits){ + const scope=scopeRows(rows,suite),accepted=included.get(model)||[],acceptedIds=new Set(accepted.map(measurementId)); + for(const task of [...new Set(rows.filter(r=>!r.eval).map(r=>r.task))].sort(compareText))out.push(diagnostic('no_config',model,null,rows.filter(r=>r.task===task),'No eval configuration; excluded from scoring.','excluded')); + for(const e of catalogue.evals){ + const all=rows.filter(r=>r.eval===e.name),outside=all.filter(r=>(!e.select||matchTask(e.select,r.task))&&!inSuite(r,suite)),matching=all.filter(r=>inSuite(r,suite)),selected=matching.filter(r=>r.selected); + const add=(code,rr,detail,effect='included',tasks=null)=>out.push(diagnostic(code,model,e.name,rr,detail,effect,tasks)); + if(outside.length)add('not_used',outside,'Eval data is present but not selected by '+suite.name+'. Excluded from the calculation.','excluded'); + if(!matching.length)continue; + const unknown=selected.filter(r=>taskLanguage(r.task,catalogue).status==='unknown'); + if(unknown.length)add('unknown_language',unknown,e.aggregation?'Explicit language assignment missing; component groups are excluded.':'Unknown language; English fallback applies for weighting only.',e.aggregation?'excluded':'included'); + const bad=selected.filter(r=>r.n_samples!==undefined&&r.n_samples!==null&&String(r.n_samples)!==''&&sampleCount(r)===null); + if(bad.length)add('invalid_sample_count',bad,'Invalid sample count; retained scores are not weighted by sample count.'); + const protocol=protocolWarning(matching,e,catalogue,model); + if(protocol)add('inconsistent_scoring_settings',selected,protocol.detail); + for(const task of [...new Set(matching.filter(r=>!e.select||matchTask(e.select,r.task)).map(r=>r.task))].sort(compareText)){ + const rr=matching.filter(r=>r.task===task);if(rr.some(r=>r.selected))continue; + add(rr.some(r=>r.metric===e.metric)?'missing_scoring_setting':'missing_scoring_field',rr,'Excluded: expected '+e.metric+' / '+(e.filter||'(empty)')+('shots'in e?' / '+e.shots+' shots':'')+'.','excluded'); + } + if(!selected.length)add('no_selected_score',matching,'No score matches the configured metric, filter, shots and selection. Excluded.','excluded'); + } + for(const name of [...new Set(scope.missing.map(r=>r.eval))])out.push(diagnostic('missing_suite_data',model,name,[],'Required results are missing from '+suite.name+'. Excluded; remaining weights are redistributed.','excluded',scope.missing.filter(r=>r.eval===name&&r.task).map(r=>r.task))); + const component=componentCoverage(scope.rows,scheme); + for(const name of [...new Set(component.excluded.map(r=>r.eval))])out.push(diagnostic('incomplete_components',model,name,component.excluded.filter(r=>r.eval===name),component.warnings.filter(w=>w.eval===name).map(w=>w.detail).join(' '),'excluded')); + if(comparison)for(const name of [...new Set(component.rows.filter(r=>!acceptedIds.has(measurementId(r))).map(r=>r.eval))])out.push(diagnostic('comparison_coverage',model,name,component.rows.filter(r=>r.eval===name&&!acceptedIds.has(measurementId(r))),'Unmatched results are excluded from both scores; weights use shared data only.','excluded')); + } + for(const category of [...new Set([...included.values()].flat().map(r=>r.category))])if(!Object.hasOwn(profile.weights,category))out.push({...diagnostic('no_category_weight','Selected comparison',null,[...included.values()].flat().filter(r=>r.category===category),category+' has no category weight and contributes zero.','zero_weight'),name:category,category}); + if(comparison){const entries=[...included],a=entries[0]?.[1]||[],b=entries[1]?.[1]||a,bm=new Map(b.map(r=>[key(r),r])); + for(const e of catalogue.evals){const rr=a.filter(r=>r.eval===e.name&&bm.has(key(r))&&sampleCount(r)!==null&&sampleCount(bm.get(key(r)))!==null&&sampleCount(r)!==sampleCount(bm.get(key(r))));if(rr.length)out.push(diagnostic('sample_count_mismatch','Selected comparison',e.name,rr.concat(rr.map(r=>bm.get(key(r)))),'Matched results have different sample counts. Scores remain included.'));} + } + return out; +} +function allocationTree(rows,config,allocation,model){ + const metadata=new Map(config.languages.flatMap(g=>g.tasks.map(t=>[t,g]))); + const grouped=(rr,fn)=>{const groups=new Map();for(const r of rr){const k=fn(r);if(!groups.has(k))groups.set(k,[]);groups.get(k).push(r);}return [...groups].sort(([a],[b])=>compareText(a,b));}; + const language=r=>{const m=metadata.get(r.task);return m?.scope==='translation'?m.source_language+' → '+m.target_language:m?.language||'Unknown';}; + function node(kind,label,rr,children=[],relative=null,measurement=null){const weight=rr.reduce((s,r)=>s+(allocation.rowWeights.get(r)||0),0),contribution=rr.reduce((s,r)=>s+r.score_100*(allocation.rowWeights.get(r)||0),0);return {kind,label,score:weight?contribution/weight:measurement?rr[0].score_100:null,weight:0,effective_weight:weight,contribution,relative_weight:relative,measurement_id:measurement,children};} + const leaves=(rr,e)=>rr.slice().sort((a,b)=>compareText(measurementId(a),measurementId(b))).map(r=>{const c=e.aggregation?.components.find(c=>matchTask(c.match,r.task));return node(c?'component':'measurement',c?.name||r.task,[r],[],c?.relative_weight??null,measurementId(r));}); + const langs=(rr,e)=>grouped(rr,language).map(([l,ll])=>node('language',l,ll,e.aggregation?grouped(ll,r=>JSON.stringify(['metric','filter','n_shot','harness','backend'].map(k=>r[k]))).map(([p,pp])=>node('protocol',p,pp,leaves(pp,e))):leaves(ll,e))); + const balance=(rr,children)=>grouped(rr,r=>scoreLanguage(r,metadata)).map(([side,ss])=>node('language_group',side,ss,children(ss))); + const evalNodes=(rr,split)=>config.evals.filter(e=>rr.some(r=>r.eval===e.name)).map(e=>{const ee=rr.filter(r=>r.eval===e.name);return node('eval',e.name,ee,split?balance(ee,ss=>langs(ss,e)):langs(ee,e));}); + const cats=Object.keys(config.weights).sort(compareText).filter(c=>rows.some(r=>r.category===c)).map(c=>{const rr=rows.filter(r=>r.category===c),split=config.aggregate!=='standard'&&(Object.hasOwn(config.english_weights,c)?config.english_weights[c]:0)!==0;return node('category',c,rr,split&&config.aggregate==='english_category'?balance(rr,ss=>evalNodes(ss,false)):evalNodes(rr,split));}); + const root=node('model',model,rows,cats);root.score=allocation.score;root.weight=allocation.score===null?0:1; + function assign(n,path){n.id=JSON.stringify(path);for(const child of n.children){child.weight=n.effective_weight?child.effective_weight/n.effective_weight:0;assign(child,[...path,[child.kind,child.measurement_id||child.label]]);}} + assign(root,[['model',model]]);return root; +} +function modelReport(model,audit,rows,config,suite){ + const allocation=totals(rows,config,config.weights,config.aggregate,config.english_weights),ids=new Set(rows.map(measurementId)); + return {model,score:allocation.score,tree:allocationTree(rows,config,allocation,model),measurements:audit.map(r=>({...r,id:measurementId(r),language:taskLanguage(r.task,config),included:ids.has(measurementId(r)),exclusion:r.selected?inSuite(r,suite)?ids.has(measurementId(r))?null:'coverage':'eval_set':'interpretation',effective_weight:allocation.rowWeights.get(r)||0,contribution:ids.has(measurementId(r))?r.score_100*(allocation.rowWeights.get(r)||0):0}))}; +} +function prepareAnalysis(rows,config){ + const scheme=resolveConfig(config.catalogue,config.suite,config.profile),audit=auditRows(rows,config.catalogue),models=[...new Set(audit.map(r=>r.checkpoint))].sort(compareText); + return {scheme,audits:new Map(models.map(m=>[m,audit.filter(r=>r.checkpoint===m)]))}; +} +function analyze(rows,config){ + const {scheme,audits}=prepareAnalysis(rows,config),included=new Map([...audits].map(([m,rr])=>[m,componentCoverage(scopeRows(rr,config.suite).rows,scheme).rows])); + return {models:[...audits].map(([m,rr])=>modelReport(m,rr,included.get(m),scheme,config.suite)),diagnostics:reportDiagnostics(audits,config,included),coverage:[...audits].map(([model,rr])=>{const scope=scopeRows(rr,config.suite);return {model,missing:scope.missing,complete:!scope.missing.length&&scope.rows.length===included.get(model).length};})}; +} +function compareAudits(auditA,auditB,config,a,b){ + const scheme=resolveConfig(config.catalogue,config.suite,config.profile),scope=suiteCoverage(auditA,auditB,config.suite),coverage=comparisonCoverage(scope.a,scope.b,scheme),effective=suiteCoverage(coverage.a,coverage.b,config.suite); + const audits=new Map([[a,auditA],[b,auditB]]),included=new Map([[a,coverage.a],[b,coverage.b]]),left=modelReport(a,auditA,coverage.a,scheme,config.suite),right=modelReport(b,auditB,coverage.b,scheme,config.suite); + const aw=new Map(left.measurements.map(r=>[r.id,r.effective_weight])); + const result={a:left,b:right,delta:left.score===null||right.score===null?null:left.score-right.score,diagnostics:reportDiagnostics(audits,config,included,true),coverage:{complete:config.suite.mode==='fixed'?effective.complete:!coverage.excludedA.length&&!coverage.excludedB.length,required:scope.required,sharedRequired:effective.sharedRequired,presentA:scope.presentA,presentB:scope.presentB,extrasA:scope.extrasA,extrasB:scope.extrasB},deltas:coverage.pairs.map(r=>({task:r.task,measurement_a:measurementId(r),measurement_b:measurementId({...r,checkpoint:b}),raw_delta:r.delta,score_delta:r.score_delta,effective_weight:aw.get(measurementId(r)),contribution_delta:r.score_delta*aw.get(measurementId(r))}))}; + return {result,scope:{...scope,complete:effective.complete,sharedRequired:effective.sharedRequired},...coverage}; +} +function compare(rows,config,a,b){ + const {audits}=prepareAnalysis(rows,config);if(!audits.has(a)||!audits.has(b))throw Error('Unknown comparison model'); + return compareAudits(audits.get(a),audits.get(b),config,a,b).result; +} + +return {analyze,compare,compareAudits,allocationTree,measurementId,selectRows,buildCatalogue,comparisonRows,weightingLanguage,scoreLanguage,englishAssignment,componentCoverage,evalDistribution,totals,pairRows,sampleCount,comparisonCoverage,synthetic,syntheticOptions,isDemoModel,languageRoles,matchesLanguage,languageCoverage,languageCountLabel,languageLabel,sortBreakdownTree,breakdownAggregate,buildBreakdownTree,protocolWarning,reportDiagnostics,normalizationLabel,sameCoverage,key,avg,fmt,esc,parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel,parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveConfig,inSuite,suiteCoverage,scopeRows}; +})(); +if(typeof module!=='undefined')module.exports=QuickdashAnalysis; diff --git a/app/app.js b/app/app.js index ba4ef4a..3f02072 100644 --- a/app/app.js +++ b/app/app.js @@ -1,177 +1,5 @@ 'use strict'; -const key = r => JSON.stringify(['task','metric','filter','n_shot','harness','backend'].map(k=>r[k])); -const avg = xs => xs.length ? xs.reduce((a,b)=>a+b,0)/xs.length : null; -const fmt = (x,digits=2) => x === null || !Number.isFinite(x) ? '—' : x.toFixed(digits); -const esc = x => String(x??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); -const {parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel}=typeof module!=='undefined'?require('./eval_config.js'):EvalConfig; -const {parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveConfig,inSuite,suiteCoverage}=typeof module!=='undefined'?require('./suite_config.js'):SuiteConfig; -function selectRows(rows,scheme){const selected=auditRows(rows,scheme).filter(r=>r.selected);if(!selected.length)throw Error('No selected measurements');return selected;} -function buildCatalogue(rows,scheme){ - return scheme.evals.map(f=>{const tasks=new Map();for(const r of rows.filter(r=>r.eval===f.name)){if(!tasks.has(r.task))tasks.set(r.task,[]);tasks.get(r.task).push(r);}return {...f,tasks:[...tasks].map(([name,rows])=>({name,rows})).sort((a,b)=>a.name.localeCompare(b.name))};}).filter(f=>f.tasks.length); -} -function comparisonRows(pairs,reference,scheme,weights,group,measure,sort,sortBy='delta',metadata=new Map(),aggregate='standard',englishWeights=scheme.english_weights||{}){ - const allocation=totals(reference,scheme,weights,aggregate,englishWeights,metadata),coefficients=new Map(reference.map(r=>[key(r),allocation.rowWeights.get(r)])); - const groups=new Map();for(const r of pairs){const k=group==='category'?r.category:group==='eval'?r.eval:key(r);if(!groups.has(k))groups.set(k,[]);groups.get(k).push(r);} - const items=[...groups].map(([k,rows])=>{ - const raw=group==='category'?avg([...new Set(rows.map(r=>r.eval))].map(f=>avg(rows.filter(r=>r.eval===f).map(r=>r.delta)))):avg(rows.map(r=>r.delta)); - const weighted=rows.reduce((sum,r)=>sum+r.score_delta*coefficients.get(key(r)),0); - return {a:group==='category'?avg([...new Set(rows.map(r=>r.eval))].map(f=>avg(rows.filter(r=>r.eval===f).map(r=>r.a)))):avg(rows.map(r=>r.a)),b:group==='category'?avg([...new Set(rows.map(r=>r.eval))].map(f=>avg(rows.filter(r=>r.eval===f).map(r=>r.b)))):avg(rows.map(r=>r.b)),tasks:[...new Set(rows.map(r=>r.task))],languageCount:languageCoverage(rows,metadata).count,weightedDelta:weighted,task:group==='variant'?rows[0].task:'',label:group==='variant'?rows[0].task+' · '+rows[0].n_shot+' shot':k,category:rows[0].category,eval:group==='category'?'':rows[0].eval,count:rows.length,rawDelta:raw,delta:measure==='weighted'?weighted:raw}; - }); - const field=sort==='name'?'label':sortBy,direction=sort==='ascending'||sort==='name'?1:-1; - items.sort((a,b)=>{const x=a[field],y=b[field],difference=typeof x==='string'?x.localeCompare(y):sort==='absolute'?Math.abs(x)-Math.abs(y):x-y;return direction*difference||a.label.localeCompare(b.label);}); - return items; -} -function weightingLanguage(row,metadata){ - const m=metadata.get(row.task); - return m?.scope==='translation'?m.target_language:['single','pooled'].includes(m?.scope)?m.language:null; -} -function scoreLanguage(row,metadata){ - const language=weightingLanguage(row,metadata); - return !language||language==='mul'||language==='eng_Latn'?'english':'other'; -} -function englishAssignment(row,metadata){ - const language=weightingLanguage(row,metadata); - return !language||language==='mul'?'English (fallback: unknown or mixed language; weighting only)':language==='eng_Latn'?'English':'Other languages'; -} -function totals(rows,scheme,weights,aggregate='standard',englishWeights=scheme.english_weights||{},metadata=new Map((scheme.languages||[]).flatMap(g=>g.tasks.map(task=>[task,g])))){ - const rowWeights=new Map(rows.map(r=>[r,0])); - const fs=scheme.evals.map(f=>{const rr=rows.filter(r=>r.eval===f.name);return {...f,score:avg(rr.map(r=>r.score_100)),rows:rr,weight:0,contribution:rr.length?null:0,aggregateScore:null,excluded:!rr.length,englishShare:0,effectiveEnglishShare:null,englishScore:null,otherScore:null,issue:''};}); - const availableWeight=Object.entries(weights).filter(([name])=>fs.some(f=>f.category===name&&f.rows.length)).reduce((sum,[,weight])=>sum+weight,0); - const cats=Object.keys(weights).map(name=>{ - const configured=fs.filter(f=>f.category===name),ff=configured.filter(f=>f.rows.length),share=aggregate!=='standard'?(englishWeights[name]??0):0,split=share!==0; - const c={name,weight:ff.length&&availableWeight?weights[name]/availableWeight:0,excluded:!ff.length,excludedEvals:configured.filter(f=>!f.rows.length).map(f=>f.name),score:null,englishShare:share,effectiveEnglishShare:null,englishScore:null,otherScore:null,issue:''}; - if(c.excluded)return c; - if(!Number.isFinite(share)||share<0||share>1)c.issue='English share must be between 0 and 1.'; - else if(!split){c.score=avg(ff.map(f=>f.score));for(const f of ff)for(const r of f.rows)rowWeights.set(r,c.weight/ff.length/f.rows.length);} - else if(aggregate==='english_eval'){ - for(const f of ff){ - f.englishShare=share; - const groups=['english','other'].map(side=>f.rows.filter(r=>scoreLanguage(r,metadata)===side)); - [f.englishScore,f.otherScore]=groups.map(group=>avg(group.map(r=>r.score_100))); - f.effectiveEnglishShare=f.englishScore===null?0:f.otherScore===null?1:share; - f.aggregateScore=f.effectiveEnglishShare*(f.englishScore??0)+(1-f.effectiveEnglishShare)*(f.otherScore??0); - groups.forEach((group,i)=>{for(const r of group)rowWeights.set(r,c.weight/ff.length*(i===0?f.effectiveEnglishShare:1-f.effectiveEnglishShare)/group.length);}); - } - if(!c.issue)c.score=avg(ff.map(f=>f.aggregateScore)); - } - else { - const groups=['english','other'].map(side=>ff.map(f=>({eval:f,rows:f.rows.filter(r=>scoreLanguage(r,metadata)===side)})).filter(f=>f.rows.length)); - [c.englishScore,c.otherScore]=groups.map(group=>avg(group.map(f=>avg(f.rows.map(r=>r.score_100))))); - if(!c.issue){ - c.effectiveEnglishShare=c.englishScore===null?0:c.otherScore===null?1:share; - const parts=[c.effectiveEnglishShare,1-c.effectiveEnglishShare]; - c.score=parts[0]*(c.englishScore??0)+parts[1]*(c.otherScore??0); - groups.forEach((group,i)=>{for(const f of group)for(const r of f.rows)rowWeights.set(r,c.weight*parts[i]/group.length/f.rows.length);}); - } - } - for(const f of ff){f.weight=f.rows.reduce((sum,r)=>sum+rowWeights.get(r),0);f.contribution=c.score===null?null:f.rows.reduce((sum,r)=>sum+r.score_100*rowWeights.get(r),0);f.aggregateScore=c.score!==null&&f.weight?f.contribution/f.weight:null;} - return c; - }); - const valid=Object.values(weights).every(w=>Number.isFinite(w)&&w>=0)&&Math.abs(Object.values(weights).reduce((s,w)=>s+w,0)-1)<1e-8; - return {evals:fs,categories:cats,rowWeights,score:valid&&availableWeight>0&&cats.filter(c=>!c.excluded).every(c=>c.score!==null)?cats.reduce((s,c)=>s+(c.excluded?0:c.score*c.weight),0):null}; -} -function pairRows(a,b){const bm=new Map(b.map(r=>[key(r),r]));return a.filter(r=>bm.has(key(r))).map(r=>({...r,a:r.raw_score_100,b:bm.get(key(r)).raw_score_100,delta:r.raw_score_100-bm.get(key(r)).raw_score_100,score_delta:r.score_100-bm.get(key(r)).score_100}));} -function sampleCount(row){const value=String(row.n_samples??'');return /^[1-9][0-9]*$/.test(value)&&Number.isSafeInteger(Number(value))?Number(value):null;} -function comparisonCoverage(a,b,config){ - const pairs=pairRows(a,b),matched=new Set(pairs.map(key)),aa=a.filter(r=>matched.has(key(r))),bb=b.filter(r=>matched.has(key(r))); - const onlyA=a.filter(r=>!matched.has(key(r))),onlyB=b.filter(r=>!matched.has(key(r))),warnings=[]; - for(const e of config.evals){const left=onlyA.filter(r=>r.eval===e.name),right=onlyB.filter(r=>r.eval===e.name);if(!left.length&&!right.length)continue; - const hasShared=pairs.some(r=>r.eval===e.name); - warnings.push({type:'Comparison coverage',name:e.name,eval:e.name,detail:(hasShared?'Unmatched variants are excluded from both scores.':'No matching scores: this eval is excluded from both scores.')+' '+left.length+' measurement(s) available only in A; '+right.length+' only in B. Remaining evals share their category weight; empty categories are excluded and remaining category weights are rescaled.',variants:[[left,'Available only in A (missing from B)'],[right,'Available only in B (missing from A)']].filter(([rr])=>rr.length).map(([rr,settings])=>({settings,tasks:rr.map(r=>r.task+' · '+r.metric+' / '+(r.filter||'blank filter')+' / '+r.n_shot+' shots / '+r.harness+' / '+r.backend)}))}); - } - const bm=new Map(bb.map(r=>[key(r),r])); - for(const e of config.evals){ - const mismatch=aa.filter(r=>r.eval===e.name&&sampleCount(r)!==null&&sampleCount(bm.get(key(r)))!==null&&sampleCount(r)!==sampleCount(bm.get(key(r)))); - if(mismatch.length)warnings.push({type:'Sample-count mismatch',name:e.name,eval:e.name,detail:'Matched measurements report different n_samples for A and B. Scores remain included; review dataset coverage before comparing.',variants:[{settings:'Reported sample counts',tasks:mismatch.map(r=>r.task+' · '+r.metric+' · A: '+r.n_samples+' / B: '+bm.get(key(r)).n_samples)}]}); - } - return {a:aa,b:bb,pairs,warnings,onlyA,onlyB}; -} -function synthetic(rows,config){let seed=20260930;const rand=()=>{seed=(Math.imul(1664525,seed)+1013904223)>>>0;return (seed+.5)/4294967296;};return rows.map(r=>{const perturb=2*Math.sqrt(-2*Math.log(rand()))*Math.cos(2*Math.PI*rand());return {...r,checkpoint:'SYNTHETIC demo — perturbed',...normalizeScore(Math.max(0,Math.min(100,r.raw_score_100+perturb))/100*config.evals.find(e=>e.name===r.eval).score.scale,config.evals.find(e=>e.name===r.eval)),stderr:'',result_time:'',results_file:'synthetic: seed 20260930; normal sd 2 score points; clipped to [0,100]'};});} -function languageRoles(row,metadata){const m=metadata.get(row.task);return m?.scope==='translation'?[{language:m.source_language,role:'from'},{language:m.target_language,role:'to'}]:[{language:m?.language||'Unknown',role:'eval'}];} -function matchesLanguage(row,metadata,language='',role=''){return languageRoles(row,metadata).some(m=>(!language||m.language===language)&&(!role||m.role===role));} -function languageCoverage(rows,metadata){ - const codes=new Set(),pooled=new Set(),unknown=new Set(); - for(const r of rows){const m=metadata.get(r.task);if(!m||m.scope==='unknown')unknown.add(r.task);else if(m.scope==='pooled')pooled.add(m.language);else for(const role of languageRoles(r,metadata))codes.add(role.language);} - return {count:codes.size,pooled:pooled.size,unknown:unknown.size}; -} -function languageCountLabel(coverage){return [coverage.count?String(coverage.count):'',coverage.pooled?coverage.pooled+' pooled':'',coverage.unknown?'unknown':''].filter(Boolean).join(' + ')||'0';} -function languageLabel(code){return code==='mul'?'Multilingual (pooled)':code;} -function sortBreakdownTree(tree,field='label',order='ascending'){ - const label=n=>n.kind==='language'?languageLabel(n.label):n.label; - const value=n=>field==='label'?label(n):n[field]; - return tree.map(n=>({...n,children:sortBreakdownTree(n.children,field,order)})).sort((a,b)=>{ - const x=value(a),y=value(b),missing=v=>v===null||v===undefined||typeof v==='number'&&!Number.isFinite(v); - if(missing(x)!==missing(y))return missing(x)?1:-1; - const difference=missing(x)?0:typeof x==='string'?x.localeCompare(y):x-y; - return (order==='ascending'?1:-1)*difference||label(a).localeCompare(label(b))||(a.detail||'').localeCompare(b.detail||''); - }); -} -// Summaries use distinct measurements, independent of repeated translation branches. -function breakdownAggregate(rows){ - const unique=[...new Map(rows.map(r=>[key(r),r])).values()],evals=[...new Set(unique.map(r=>r.eval))]; - const a=avg(evals.map(e=>avg(unique.filter(r=>r.eval===e).map(r=>r.a)))),b=avg(evals.map(e=>avg(unique.filter(r=>r.eval===e).map(r=>r.b)))); - return {a,b,delta:a===null||b===null?null:a-b,count:unique.length,evals:evals.length}; -} -function buildBreakdownTree(rows,metadata,view,languageFilter='',roleFilter=''){ - const groups=(rr,values)=>{const map=new Map();for(const r of rr)for(const value of new Set(values(r))){if(!map.has(value))map.set(value,[]);map.get(value).push(r);}return [...map].sort(([a],[b])=>a.localeCompare(b));}; - const node=(kind,label,rr,children=[])=>({kind,label,...breakdownAggregate(rr),children}); - const leaves=rr=>rr.slice().sort((a,b)=>a.task.localeCompare(b.task)||key(a).localeCompare(key(b))).map(r=>({...node('variant',r.task,[r]),detail:r.metric+' · '+(r.filter||'no filter')+' · '+r.n_shot+' shot'})); - const languageGroups=rr=>groups(rr,r=>languageRoles(r,metadata).filter(m=>(!languageFilter||m.language===languageFilter)&&(!roleFilter||m.role===roleFilter)).map(m=>m.language)); - function details(rr,language){ - const ordinary=rr.filter(r=>metadata.get(r.task)?.scope!=='translation'),translation=rr.filter(r=>metadata.get(r.task)?.scope==='translation'); - const children=leaves(ordinary); - for(const role of ['from','to']){ - if(roleFilter&&roleFilter!==role)continue; - const directed=translation.filter(r=>matchesLanguage(r,metadata,language,role));if(!directed.length)continue; - children.push(node('direction',(role==='from'?'From ':'To ')+language,directed,groups(directed,r=>{const m=metadata.get(r.task);return [m.source_language+' → '+m.target_language];}).map(([pair,rr])=>node('pair',pair,rr,leaves(rr))))); - } - return children; - } - if(view==='category')return groups(rows,r=>[r.category]).map(([category,rr])=>node('category',category,rr,groups(rr,r=>[r.eval]).map(([name,ee])=>node('eval',name,ee,languageGroups(ee).map(([language,ll])=>node('language',language,ll,details(ll,language))))))); - return languageGroups(rows).map(([language,ll])=>node('language',language,ll,groups(ll,r=>[r.category]).map(([category,cc])=>node('category',category,cc,groups(cc,r=>[r.eval]).map(([name,ee])=>node('eval',name,ee,details(ee,language))))))); -} -function protocolWarning(rows,evalConfig,config,model){ - // Compare sets per task: identical alternate settings in every language are consistent. - const fields=['n_shot','metric','filter','harness','backend'],tasks=new Map(); - for(const r of rows.filter(r=>r.selected)){if(!tasks.has(r.task))tasks.set(r.task,new Set());tasks.get(r.task).add(JSON.stringify(fields.map(f=>String(r[f]))));} - const groups=new Map();for(const [task,settings] of tasks){const signature=JSON.stringify([...settings].sort());if(!groups.has(signature))groups.set(signature,[]);groups.get(signature).push(task);} - if(groups.size<2)return null; - const variants=[...groups].map(([signature,names])=>{ - const settings=JSON.parse(signature).map(k=>{const [shots,metric,filter,harness,backend]=JSON.parse(k);return shots+' shots / '+metric+' / filter '+(filter||'(empty)')+' / '+harness+' / '+backend;}).join('; '); - const labels=names.sort().map(task=>{const m=taskLanguage(task,config),language=m.scope==='translation'?m.source_language+' → '+m.target_language:m.language||'Unknown';return task+' ['+language+']';}); - return {settings,tasks:labels}; - }); - return {type:'Inconsistent scoring settings',name:evalConfig.name,eval:evalConfig.name,model,detail:'Selected variants use different settings: '+variants.map(v=>v.settings+' ('+v.tasks.length+' tasks; '+v.tasks.slice(0,2).join(', ')+(v.tasks.length>2?', …':'')+')').join(' versus ')+'. These variants are still included in the aggregate. Review the protocol before comparing languages.',variants}; -} -function collectWarnings(audits,config,aggregate='standard',englishWeights=config.english_weights||{},suite={mode:'available'},displayedRows=null){ - const displayed=new Set((displayedRows??[...audits.values()].flat().filter(r=>r.selected&&inSuite(r,suite))).map(r=>r.eval)); - const warnings=config.evals.filter(e=>e.warning&&displayed.has(e.name)).map(e=>({type:'Config caveat',name:e.name,eval:e.name,model:'Selected comparison',detail:e.warning})); - for(const [model,rows] of audits){ - if(aggregate!=='standard')for(const c of totals(rows.filter(r=>r.selected),config,config.weights,aggregate,englishWeights).categories)if(c.englishShare!==0&&c.issue)warnings.push({type:'English split unavailable',name:c.name,model,detail:c.issue}); - for(const task of [...new Set(rows.filter(r=>!r.eval).map(r=>r.task))].sort())warnings.push({type:'No config',name:task,model,detail:'Excluded from scoring. Add an eval match to the YAML config.'}); - for(const e of config.evals){ - const all=rows.filter(r=>r.eval===e.name),outside=all.filter(r=>(!e.select||matchTask(e.select,r.task))&&!inSuite(r,suite)); - if(outside.length)warnings.push({type:'Not used',name:e.name,eval:e.name,model,detail:'Eval data is present but not selected by '+suite.name+'. These results are excluded from the calculation; inspect them in Eval configuration.',variants:[{settings:'Not used by the selected eval set',tasks:[...new Set(outside.map(r=>r.task+' · '+r.n_shot+' shots'))]}]}); - const matching=all.filter(r=>inSuite(r,suite)); - if(!matching.length)continue; - const selected=matching.filter(r=>r.selected),unknown=[...new Set(selected.filter(r=>taskLanguage(r.task,config).status==='unknown').map(r=>r.task))]; - if(unknown.length)warnings.push({type:'Unknown language',name:e.name,eval:e.name,model,detail:'No explicit language assignment for '+unknown.length+' selected task(s). Included in scoring; both English-balance modes use the English fallback. Language views retain Unknown.',variants:[{settings:'Add an explicit language assignment in YAML',tasks:unknown}]}); - const badSamples=selected.filter(r=>r.n_samples!==undefined&&r.n_samples!==null&&String(r.n_samples)!==''&&sampleCount(r)===null); - if(badSamples.length)warnings.push({type:'Invalid sample count',name:e.name,eval:e.name,model,detail:'n_samples must be a positive integer when supplied. Scores remain included; sample counts do not determine score weights. Check the export.',variants:[{settings:'Invalid reported n_samples',tasks:badSamples.map(r=>r.task+' · '+String(r.n_samples))}]}); - const protocol=protocolWarning(matching,e,config,model);if(protocol)warnings.push(protocol); - const tasks=[...new Set(matching.filter(r=>!e.select||matchTask(e.select,r.task)).map(r=>r.task))]; - for(const task of tasks){const variants=matching.filter(r=>r.task===task);if(variants.some(r=>r.selected))continue; - const hasMetric=variants.some(r=>r.metric===e.metric); - warnings.push({type:hasMetric?'Missing scoring setting':'Missing scoring field',name:task,eval:e.name,model,detail:'Excluded: expected '+e.metric+' with filter '+(e.filter||'(empty)')+('shots'in e?', '+e.shots+' shots':'')+'. Available: '+[...new Set(variants.map(r=>r.metric+' / '+(r.filter||'(empty)')+' / '+r.n_shot+' shots'))].join('; ')+'.'}); - } - if(!matching.some(r=>r.selected))warnings.push({type:'No selected score',name:e.name,model,detail:'Tasks exist, but none match the configured metric, filter, shots and selection rule. Excluded from scoring; remaining evals share the category weight.'}); - } - } - return warnings; -} -function normalizationLabel(e){const n=e.normalize;if(!n||n.min===0&&n.max===1)return n?.basis==='unresolved'?'Unresolved · no correction':'No chance correction';return fmt(n.min*100,2)+'% baseline';} -function sameCoverage(a,b){return a.length===b.length&&pairRows(a,b).length===a.length;} -if(typeof module!=='undefined')module.exports={parseCSV,auditRows,selectRows,buildCatalogue,totals,pairRows,comparisonCoverage,synthetic,comparisonRows,languageRoles,matchesLanguage,languageCoverage,breakdownAggregate,buildBreakdownTree,sortBreakdownTree,collectWarnings,sameCoverage}; +const {compareAudits,selectRows,buildCatalogue,comparisonRows,weightingLanguage,scoreLanguage,englishAssignment,componentCoverage,evalDistribution,totals,pairRows,sampleCount,comparisonCoverage,synthetic,syntheticOptions,isDemoModel,languageRoles,matchesLanguage,languageCoverage,languageCountLabel,languageLabel,sortBreakdownTree,breakdownAggregate,buildBreakdownTree,protocolWarning,normalizationLabel,sameCoverage,key,avg,fmt,esc,parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel,parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveConfig,inSuite,suiteCoverage}=QuickdashAnalysis; if(typeof document!=='undefined')start(); function start(){ @@ -180,7 +8,8 @@ function start(){ let catalogueGroups=new Map(); let models=new Map(),sourceAudits=new Map(),metadata=new Map(DATA.metadata.map(r=>[r.task,r])); for(const m of DATA.models){const audit=auditRows(DATA.rows.filter(r=>r.checkpoint===m.model),catalogue);sourceAudits.set(m.model,audit);models.set(m.model,audit.filter(r=>r.selected));} - if(models.size)models.set(demoModel,synthetic([...models.values()][0],catalogue)); + function addSynthetic(target,config){if(!target.size)return;const rows=[...target.values()][0];for(const option of syntheticOptions)target.set(option.name,synthetic(rows,config,option));} + addSynthetic(models,catalogue); const state={aggregate:scheme.aggregate||'standard',view:'score',scoreCategory:Object.keys(weights)[0],group:'eval',expandedComparisons:new Set(),measure:'raw',sort:'descending',sortBy:'delta',languageSort:'label',languageOrder:'ascending'}; const languages=r=>languageRoles(r,metadata).map(m=>m.language); const td=x=>''+esc(x)+''; @@ -198,16 +27,17 @@ function start(){ } function languageOptions(){const previous=$('language').value;const all=[...new Set([...sourceAudits.values()].flat().flatMap(languages))].sort();$('language').innerHTML=options([['','All languages'],...all.map(k=>[k,languageLabel(k)])],previous);} function filters(r){const q=$('search').value.trim().toLowerCase();return (!$('category').value||r.category===$('category').value)&&(!$('eval').value||r.eval===$('eval').value)&&matchesLanguage(r,metadata,$('language').value,$('direction').value)&&(!q||(r.task+' '+r.eval+' '+r.metric).toLowerCase().includes(q));} + let comparisonCache=null; function selected(){ - const scope=suiteCoverage(models.get($('modelA').value)||[],models.get($('modelB').value)||[],suite); - const coverage=comparisonCoverage(scope.a,scope.b,scheme); - return {...coverage,scope,shown:coverage.pairs.filter(filters)}; + if(!comparisonCache){ + const a=$('modelA').value,b=$('modelB').value; + comparisonCache=compareAudits(sourceAudits.get(a)||models.get(a)||[],sourceAudits.get(b)||models.get(b)||[],{catalogue,suite,profile},a,b); + } + return {...comparisonCache,shown:comparisonCache.pairs.filter(filters)}; } function activeWarnings(){ - const coverage=selected(),chosen=new Map([...sourceAudits].filter(([name])=>[$('modelA').value,$('modelB').value].includes(name))); - const warnings=collectWarnings(chosen,catalogue,'standard',{},suite,coverage.a).concat(coverage.warnings.map(w=>({...w,model:'A: '+$('modelA').value+' · B: '+$('modelB').value}))); - if(models.size)warnings.push(...coverage.scope.warnings.map(w=>({...w,model:w.model+': '+$(w.model==='A'?'modelA':'modelB').value}))); - for(const category of new Set(coverage.a.map(r=>r.category)))if(!Object.hasOwn(profile.weights,category)&&weights[category]===0)warnings.push({type:'No category weight',name:category,model:'Selected comparison',detail:'This category is not in the weighting profile and contributes zero. Add it to the profile to include it in the weighted score.'}); + if(!models.size)return []; + const coverage=selected(),warnings=coverage.result.diagnostics.filter(w=>!isDemoModel(w.model)&&(w.code!=='no_category_weight'||weights[w.category]===0)); if(state.aggregate!=='standard')for(const c of totals(coverage.a,scheme,weights,state.aggregate,englishWeights,metadata).categories)if(c.issue)warnings.push({type:'English split unavailable',name:c.name,model:'Selected comparison',detail:c.issue}); return warnings; } @@ -223,9 +53,9 @@ function start(){ function renderScore(a,b,ta,tb,valid){ if(!models.size)return '

Compare your evaluation results

Use Add model CSV to open your results. Add a second model to compare training methods.

Active config: '+esc(scheme.name)+'. To use your own YAML, open and choose Load catalogue.

CSV and YAML files opened here stay in your browser; they are not uploaded. Reloading restores the published models and settings.

'; - let html='

Weighted score

Follow selected variants through eval means and category weights.

Normalize variant scores to 0–100→'+(state.aggregate==='english_eval'?'Balance languages within each eval':'Mean within each eval')+'→'+(state.aggregate==='english_category'?'Balance languages across each category':'Average evals equally within each category')+'→Apply category weights
'; + let html='

Weighted score

Follow selected variants through eval means and category weights.

Normalize variant scores to 0–100→Combine configured components within each language / protocol→'+(state.aggregate==='english_eval'?'Balance languages within each eval':'Mean within each eval')+'→'+(state.aggregate==='english_category'?'Balance languages across each category':'Average evals equally within each category')+'→Apply category weights
'; html+=table(['Category','Weight','A','B','A contribution','B contribution','Contribution Δ','Explore'],ta.categories.map((c,i)=>{const d=tb.categories[i],w=c.weight,ca=c.excluded?0:c.score===null?null:c.score*w,cb=d.excluded?0:d.score===null?null:d.score*w;return ''+esc(c.name)+(state.aggregate!=='standard'?''+esc(state.aggregate==='english_eval'?(c.englishShare?fmt(c.englishShare*100,0)+'% English within each eval':'Original · split off'):englishShareLabel(c))+'':'')+(c.excluded?'Excluded · no shared scores':c.excludedEvals.length?''+c.excludedEvals.length+' eval(s) excluded':'')+(c.issue?''+esc(c.issue)+'':'')+''+fmt(w*100,3)+'%'+num(c.score)+num(d.score)+num(ca,4)+num(cb,4)+delta(valid&&ca!==null&&cb!==null?ca-cb:null,4)+''+button('Evals','data-score-category',c.name)+'';}),[1,2,3,4,5,6]); - html+='
Adjust category weights and English shares

English shares apply only when either English-balance option is selected in Score calculation above. '+(state.aggregate!=='standard'?'They are active now: '+(state.aggregate==='english_eval'?'inside each eval':'across each category')+'.':'They are saved but inactive while Original weighted score is selected.')+'

Category weights sum to 1. English share gives English and other languages 0.5 each at a 50/50 setting. It is applied inside each eval or across the category, according to the selected calculation. A group with only one language side keeps its full weight automatically. Set 0 to disable the split.

'+table(['Category','Category weight','English share'],Object.entries(weights).map(([c,w])=>''+esc(c)+''),[1,2])+'

Total category weight: '+fmt(Object.values(weights).reduce((s,w)=>s+w,0),6)+' · must equal 1.

'; + html+='
Adjust category weights and English shares

English shares apply only when either English-balance option is selected in Score calculation above. '+(state.aggregate!=='standard'?'They are active now: '+(state.aggregate==='english_eval'?'inside each eval':'across each category')+'.':'They are saved but inactive while Original weighted score is selected.')+'

Category weights sum to 1. English share gives English and other languages 0.5 each at a 50/50 setting. It is applied inside each eval or across the category, according to the selected calculation. A group with only one language side keeps its full weight automatically. Set 0 to disable the split.

'+table(['Category','Category weight','English share'],Object.entries(weights).map(([c,w])=>''+esc(c)+''),[1,2])+'

Total category weight: '+fmt(Object.values(weights).reduce((s,w)=>s+w,0),6)+' · must equal 1.

'; const category=state.scoreCategory; html+='

Inside a category

'+control('scoreCategory','Category',Object.keys(weights).map(k=>[k,k]),category)+'
'; @@ -238,26 +68,29 @@ function start(){ html+='

Effective weights and eval scores reflect the selected aggregate. Contributions sum to the category contribution. An eval represented on both sides can have different weights for its English and other-language variants.

'; return html; } - function renderBreakdown(shown){ + function renderBreakdown(shown,reference){ const isLanguage=state.view==='languages'; let html='

'+(isLanguage?'Languages':'Categories')+'

'+(isLanguage?'Languages → categories → evals':'Categories → evals → languages')+'. Expand a row to inspect its components.

'; - let tree=buildBreakdownTree(shown,metadata,isLanguage?'language':'category',$('language').value,$('direction').value); + const hasComponents=shown.some(r=>scheme.evals.find(e=>e.name===r.eval)?.aggregation); + let tree=buildBreakdownTree(shown,metadata,isLanguage?'language':'category',$('language').value,$('direction').value,scheme,reference); if(isLanguage)tree=sortBreakdownTree(tree,state.languageSort,state.languageOrder); function branch(n,depth,parent=[]){ const path=[...parent,[n.kind,n.label,n.detail||'']]; const label=n.kind==='language'?languageLabel(n.label):n.label; - const cells=''+(n.children.length?'▸':'·')+''+esc(label)+(n.detail?''+esc(n.detail)+'':'')+''+n.count+''+fmt(n.a)+''+fmt(n.b)+''+fmt(n.delta)+''; + let cells=''+(n.children.length?'▸':'·')+''+esc(label)+(n.detail?''+esc(n.detail)+'':'')+(n.componentAggregate&&n.children.length?'Calculated component score':'')+(n.componentShare!==undefined?'Normalized A '+fmt(n.componentNormalizedA)+' · B '+fmt(n.componentNormalizedB)+'':'')+''+n.count+''+fmt(n.a)+''+fmt(n.b)+''+fmt(n.delta)+''; + if(hasComponents)cells+=''+(n.componentShare!==undefined?esc(n.componentRelativeWeight)+' · '+fmt(n.componentShare*100,2)+'%':'—')+''+fmt(n.componentA,3)+''+fmt(n.componentB,3)+''; const attr=' data-kind="'+n.kind+'" data-label="'+esc(n.label)+'" data-node-key="'+esc(JSON.stringify(path))+'"'; return n.children.length?'
'+cells+''+n.children.map(c=>branch(c,depth+1,path)).join('')+'
':'
'+cells+'
'; } - const headers=[[(isLanguage?'Language':'Category')+' / details','label'],['Variants','count'],['Raw A','a'],['Raw B','b'],['A − B','delta']].map(([label,field])=>{ + const headers=[[(isLanguage?'Language':'Category')+' / details','label'],['Variants','count'],[hasComponents?'A score':'Raw A','a'],[hasComponents?'B score':'Raw B','b'],['A − B','delta'],...(hasComponents?[['Relative weight / share','componentShare'],['Group contribution A','componentA'],['Group contribution B','componentB']]:[])].map(([label,field])=>{ if(!isLanguage)return ''+label+''; const active=state.languageSort===field,order=active?state.languageOrder:'none'; return ''; }).join(''); - html+='
'+headers+'
'+tree.map(n=>branch(n,0)).join('')+'
'; + html+='
'+headers+'
'+tree.map(n=>branch(n,0)).join('')+'
'; if(isLanguage)html+='

Click a column heading to sort; click again to reverse. Sorting applies within each level of the hierarchy and keeps expanded sections open.

'; - html+='

Each aggregate averages variants within an eval, then represented evals equally. Counts and means use distinct measurements. Translation is expandable under either endpoint, then From / To, then language pairs; repeated branches do not add weight. Language scores are descriptive because eval coverage differs.

'; + if(hasComponents)html+='

Configured component evals show their calculated, normalized score when collapsed. Leaves show raw scores; component weights apply to normalized scores. Group contributions are points toward one language/protocol score, before language and category weights. Other evals retain raw averages. A filtered, incomplete component set shows — instead of a partial score.

'; + html+='

For evals without component rules, each aggregate averages variants within an eval, then represented evals equally. Counts and means use distinct measurements. Translation is expandable under either endpoint, then From / To, then language pairs; repeated branches do not add weight. Language scores are descriptive because eval coverage differs.

'; return html; } function renderComparisons(shown,reference,valid){ @@ -289,24 +122,33 @@ function start(){ const explanation=n.min===0?(n.basis==='unresolved'?'Baseline unresolved; no chance correction is currently applied.':'No chance correction is applied. Zero is a configured floor, not a measured random-model score.'):'Chance baseline used for this eval.'; return '
Score calculation

'+esc(formula)+'

random_score '+(baseline===n.min?'=':'≈')+' '+baseline+' ('+fmt(n.min*100,2)+'%). '+esc(explanation)+'

'+(n.clip===false?'Scores are not clipped.':'The result is clipped to 0–100.')+' The source value is shown in the scoring details below.

'+(e.warning?'

Warning: '+esc(e.warning)+'

':'')+'
Baseline rationale and sources

'+esc(n.note||'The baseline is set by normalize.min in the YAML config.')+'

random_score is normalize.min in the YAML config; the upper bound is normalize.max.

'+((n.sources||[]).length?'

'+n.sources.map((url,i)=>'Source '+(i+1)+'').join(' · ')+'

':'')+'
'; } + function aggregationInfo(e){ + const a=e.aggregation;if(!a)return ''; + const total=a.components.reduce((sum,c)=>sum+c.relative_weight,0); + return '
Component aggregation

group score = sum(relative_weight × normalized score) / '+esc(total)+'

'+table(['Component','Task match','Relative weight','Share of group'],a.components.map(c=>''+td(c.name)+td(c.match.name??c.match.regex)+num(c.relative_weight,3)+''+fmt(c.relative_weight/total*100,2)+'%'),[2,3])+'

All components are required within the same language and scoring protocol. Missing, ambiguous, or duplicate components exclude that group from both comparison scores. Complete groups are averaged within the eval, with the selected English balance applied afterward.

'+(a.note?'

'+esc(a.note)+'

':'')+(a.sources?.length?'

'+a.sources.map((u,i)=>'Aggregation source '+(i+1)+'').join(' · ')+'

':'')+'
'; + } function renderWarnings(){const warnings=activeWarnings();return '

Warnings

Coverage for the selected comparison, scoring consistency, and config caveats. A named eval set checks required measurements; unused global catalogue rules are allowed.

'+(warnings.length?table(['Warning','Eval / task','Model','Details'],warnings.map(w=>''+td(w.type)+td(w.name)+td(w.model)+''+esc(w.detail)+(w.variants?'
All affected variants'+w.variants.map(v=>'

'+esc(v.settings)+'
'+v.tasks.map(esc).join('
')+'

').join('')+'
':'')+'')):'

No warnings. The selected comparison has no detected data or configuration issues.

');} function scoringOptions(e,rows){ const used=rows.filter(r=>r.selected),values=(rr,key)=>[...new Set(rr.map(r=>String(r[key])))].sort((a,b)=>key==='n_shot'?Number(a)-Number(b):a.localeCompare(b)); const shots=values(used,'n_shot'),otherMetrics=values(rows,'metric').filter(m=>m!==e.metric),otherShots=values(rows,'n_shot').filter(n=>!shots.includes(n)),otherFilters=values(rows,'filter').filter(f=>f!==e.filter); - return 'Selected: '+esc(e.metric)+''+(e.metric==='python_pass@1'?'Python solutions passing tests on one attempt.':'')+''+(shots.length===1&&shots[0]==='0'?'0-shot · no examples in the prompt':shots.length?'Shots used: '+esc(shots.join(', '))+' · examples in the prompt':'No selected scores')+''+(e.filter&&e.filter!=='none'?'Answer extraction: '+esc(e.filter)+'':'')+(otherMetrics.length?'Other available metrics: '+esc(otherMetrics.join(', '))+'':'')+(otherShots.length?'Other available shot settings: '+esc(otherShots.join(', '))+'':'')+(otherFilters.length?'Other answer extraction settings: '+esc(otherFilters.map(f=>f||'not specified').join(', '))+'':''); + return 'Selected: '+esc(e.metric)+''+(e.aggregation?'Weighted components · expand for weights':'')+(e.metric==='python_pass@1'?'Python solutions passing tests on one attempt.':'')+''+(shots.length===1&&shots[0]==='0'?'0-shot · no examples in the prompt':shots.length?'Shots used: '+esc(shots.join(', '))+' · examples in the prompt':'No selected scores')+''+(e.filter&&e.filter!=='none'?'Answer extraction: '+esc(e.filter)+'':'')+(otherMetrics.length?'Other available metrics: '+esc(otherMetrics.join(', '))+'':'')+(otherShots.length?'Other available shot settings: '+esc(otherShots.join(', '))+'':'')+(otherFilters.length?'Other answer extraction settings: '+esc(otherFilters.map(f=>f||'not specified').join(', '))+'':''); } function selectionInfo(e){ return '

Selection rule: use metric '+esc(e.metric)+'; '+(e.filter?'answer extraction setting '+esc(e.filter)+'':'the CSV extraction-filter field must be blank')+'; '+('shots'in e?'require '+e.shots+' examples in the prompt':'accept any shot count found in the export')+'. A shot is one example provided in the prompt. Other metrics and shot settings remain available for inspection below.

'; } function taskDetails(f,t,missing){ const m=metadata.get(t.name)||{}; - const warning=missing.filter(w=>w.name===t.name).map(w=>'

'+esc(w.model+': '+w.detail)+'

').join(''); + const warning=missing.filter(w=>w.tasks.includes(t.name)).map(w=>'

'+esc(w.model+': '+w.detail)+'

').join(''); const info='

'+esc(m.provenance||'No language assignment available.')+(m.evidence?' Source':'')+'

'+esc(f.category)+' · language assignment: '+esc(m.status||'unknown')+'. Selected metrics use the eval’s score calculation above.

'; - return warning+info+'

English-balance group: '+esc(englishAssignment({task:t.name},metadata))+'. Used in both English-balance modes when this category’s English share is non-zero.

Raw source scores are shown for every metric. The 0–100 columns apply to selected scores only.

'+table(['Model','Metric','Filter','Shots','Raw source score','Raw / 100','Normalized / 100','Use','Reason','Harness / backend'],t.rows.map(r=>''+td(r.checkpoint)+td(r.metric)+td(r.filter||'Not specified')+''+esc(r.n_shot)+''+esc(r.value)+''+num(r.raw_score_100)+num(r.score_100)+''+(r.selected?(inSuite(r,suite)?'Selected':'Outside eval set'):'Excluded')+''+td(r.selected&&!inSuite(r,suite)?'Excluded by '+suite.name:r.decision)+td(r.harness+' / '+r.backend)+''),[3,4,5,6]); + const cc=f.aggregation?.components.filter(c=>matchTask(c.match,t.name))||[],component=cc.length===1?cc[0]:null; + const componentInfo=component?'

Component: '+esc(component.name)+' · relative weight '+esc(component.relative_weight)+' / '+f.aggregation.components.reduce((sum,c)=>sum+c.relative_weight,0)+'. Applied after normalization, before language and category weights.

':''; + const coverage=selected(),used=new Map([[$('modelA').value,new Set(coverage.a.map(key))],[$('modelB').value,new Set(coverage.b.map(key))]]); + const use=r=>!r.selected?['Excluded',r.decision]:!inSuite(r,suite)?['Outside eval set','Excluded by '+suite.name]:used.has(r.checkpoint)&&!used.get(r.checkpoint).has(key(r))?['Excluded from comparison','No complete shared component group or matching A/B measurement; see Warnings.']:['Selected',r.decision]; + return warning+info+componentInfo+'

English-balance group: '+esc(englishAssignment({task:t.name},metadata))+'. Used in both English-balance modes when this category’s English share is non-zero.

Raw source scores are shown for every metric. The 0–100 columns apply to selected scores only.

'+table(['Model','Metric','Filter','Shots','Raw source score','Raw / 100','Normalized / 100','Use','Reason','Harness / backend'],t.rows.map(r=>''+td(r.checkpoint)+td(r.metric)+td(r.filter||'Not specified')+''+esc(r.n_shot)+''+esc(r.value)+''+num(r.raw_score_100)+num(r.score_100)+''+esc(use(r)[0])+''+td(use(r)[1])+td(r.harness+' / '+r.backend)+''),[3,4,5,6]); } function catalogueTasks(f,missing){ return table(['Variant / scoring details','Language(s)','Selected metric','Other metrics'],f.tasks.map(t=>{ - const m=metadata.get(t.name)||{},selected=[...new Set(t.rows.filter(r=>r.selected).map(r=>r.metric))],other=[...new Set(t.rows.filter(r=>!r.selected).map(r=>r.metric))],hasMissing=missing.some(w=>w.name===t.name); + const m=metadata.get(t.name)||{},selected=[...new Set(t.rows.filter(r=>r.selected).map(r=>r.metric))],other=[...new Set(t.rows.filter(r=>!r.selected).map(r=>r.metric))],hasMissing=missing.some(w=>w.tasks.includes(t.name)); const language=m.source_language?m.source_language+' → '+m.target_language:languageLabel(m.language||'Unknown'); return '
'+esc(t.name)+'
'+td(language)+''+esc(selected.join(', ')||'Excluded')+(hasMissing?'Missing configured field/settings':'')+''+td(other.join(', ')||'—')+''; })); @@ -322,30 +164,32 @@ function start(){ $('filterStatus').textContent=new Set(filtered.map(r=>r.task)).size+' of '+new Set(all.map(r=>r.task)).size+' task names · '+filtered.length+' metric rows · all loaded real exports; comparison exclusions appear in Warnings'; let html='

Eval configuration

Category, language, and scoring field for every eval. Expand an eval to inspect its variants.

EvalCategoryScoring optionsLanguagesNormalization
'; catalogueGroups=new Map(); - html+=groups.map(f=>{const allRows=all.filter(r=>r.eval===f.name),missing=warnings.filter(w=>w.eval===f.name&&['Missing scoring field','Missing scoring setting'].includes(w.type)),inconsistent=warnings.filter(w=>w.eval===f.name&&w.type==='Inconsistent scoring settings');catalogueGroups.set(f.name,{f,missing});const langs=[...new Set(f.tasks.flatMap(t=>languages(t.rows[0]).map(languageLabel)))].sort();return '
'+esc(f.name)+' ('+f.tasks.length+')'+esc(f.category)+''+scoringOptions(f,allRows)+(missing.length?''+missing.length+' missing scoring field/settings':'')+(inconsistent.length?'Inconsistent scoring settings':'')+''+esc(langs.length>4?langs.slice(0,3).join(', ')+' + '+(langs.length-3)+' more':langs.join(', '))+''+esc(normalizationLabel(f))+(f.warning?'Config warning':'')+''+selectionInfo(f)+normalizationInfo(f)+inconsistent.map(w=>'

'+esc(w.model+': '+w.detail)+'

').join('')+'
'+(groups.length===1?catalogueTasks(f,missing):'')+'
'+'
';}).join(''); + html+=groups.map(f=>{const allRows=all.filter(r=>r.eval===f.name),missing=warnings.filter(w=>w.eval===f.name&&['Missing scoring field','Missing scoring setting'].includes(w.type)),inconsistent=warnings.filter(w=>w.eval===f.name&&w.type==='Inconsistent scoring settings');catalogueGroups.set(f.name,{f,missing});const langs=[...new Set(f.tasks.flatMap(t=>languages(t.rows[0]).map(languageLabel)))].sort();return '
'+esc(f.name)+' ('+f.tasks.length+')'+esc(f.category)+''+scoringOptions(f,allRows)+(missing.length?''+missing.length+' missing scoring field/settings':'')+(inconsistent.length?'Inconsistent scoring settings':'')+''+esc(langs.length>4?langs.slice(0,3).join(', ')+' + '+(langs.length-3)+' more':langs.join(', '))+''+esc(normalizationLabel(f))+(f.warning?'Config warning':'')+''+selectionInfo(f)+normalizationInfo(f)+aggregationInfo(f)+inconsistent.map(w=>'

'+esc(w.model+': '+w.detail)+'

').join('')+'
'+(groups.length===1?catalogueTasks(f,missing):'')+'
'+'
';}).join(''); if(!groups.length)html+='

No evals match the filters.

'; html+='
Scoring assumptions and source files
    '+(scheme.notes||[]).map(n=>'
  1. '+esc(n)+'
  2. ').join('')+'

Source row audit · Language assignments · Analysis JSON

'+esc(DATA.source)+' · SHA-256 '+esc(DATA.sha256)+'

'; - const configControls='
Global eval catalogue

The catalogue defines matching, categories, scoring fields, normalization, and language assignments for every known eval. It does not require models to run all those evals.

Normalization: each eval lists its baseline, formula, and sources below. Chance correction affects weighted scores; raw scores stay unchanged. acc_norm is length-normalized answer scoring, not chance correction.

Active catalogue: '+esc(catalogue.name)+' · '+catalogue.languages.reduce((n,g)=>n+g.tasks.length,0)+' explicit task language assignments.

Weighting profile and optional eval set

Weights apply independently of the eval set. Any available compares shared recognized measurements and warns on differences. A named set also warns about missing requirements and excludes extra measurements. An incomplete named-set score uses the shared subset with redistributed weights.

Active eval set: '+esc(suite.name)+(suite.exclude?.length?' · Explicitly excluded: '+suite.exclude.map(esc).join(', '):'')+'.

All imports are temporary and stay in your browser. Exports save the three inputs separately. Weight exports include your edits and active score calculation.

'; + const configControls='
Global eval catalogue

The catalogue defines matching, categories, scoring fields, normalization, and language assignments for every known eval. It does not require models to run all those evals.

Normalization: each eval lists its baseline, formula, and sources below. Chance correction and component aggregation affect calculated scores; individual raw scores stay unchanged. acc_norm is length-normalized answer scoring, not chance correction.

Active catalogue: '+esc(catalogue.name)+' · '+catalogue.languages.reduce((n,g)=>n+g.tasks.length,0)+' explicit task language assignments.

Weighting profile and optional eval set

Weights apply independently of the eval set. Any available compares shared recognized measurements and warns on differences. A named set also warns about missing requirements and excludes extra measurements. An incomplete named-set score uses the shared subset with redistributed weights.

Active eval set: '+esc(suite.name)+(suite.exclude?.length?' · Explicitly excluded: '+suite.exclude.map(esc).join(', '):'')+'.

All imports are temporary and stay in your browser. Exports save the three inputs separately. Weight exports include your edits and active score calculation.

'; html=html.replace('
',configControls+'
'); return html; } function render(){ + comparisonCache=null; const weightOpen=$('weightEditor')?.open,englishOpen=$('englishComponents')?.open; - const {a,b,pairs,shown,onlyA,onlyB,scope}=selected(),ta=totals(a,scheme,weights,state.aggregate,englishWeights,metadata),tb=totals(b,scheme,weights,state.aggregate,englishWeights,metadata),valid=sameCoverage(a,b)&&ta.score!==null&&tb.score!==null; - const demo=[$('modelA').value,$('modelB').value].some(n=>n===demoModel); + const {a,b,pairs,shown,excludedA,excludedB,scope}=selected(),ta=totals(a,scheme,weights,state.aggregate,englishWeights,metadata),tb=totals(b,scheme,weights,state.aggregate,englishWeights,metadata),valid=sameCoverage(a,b)&&ta.score!==null&&tb.score!==null; + const demo=[$('modelA').value,$('modelB').value].some(isDemoModel); configOptions();$('cards').hidden=!models.size;document.querySelector('.aggregate-controls').hidden=!models.size; - $('demo').textContent=!models.size?'Ready for your eval results. Load a CSV to begin.':demo?'Demo comparison: one model has synthetic scores. Replace it with a real eval CSV to compare training methods.':'Real-model comparison · Check evaluation settings and training budgets before drawing a conclusion.'; + const sample=[$('modelA').value,$('modelB').value].some(n=>(DATA.sample_models||[]).includes(n)); + $('demo').textContent=!models.size?'Ready for your eval results. Load a CSV to begin.':demo?'Demo comparison: synthetic scores are seeded perturbations of the first loaded model (2-point standard deviation; higher/lower options add/subtract 3 raw score points, clipped to 0–100). For exploration only.':sample?'Sample dataset for exploring Quickdash. Add your own CSVs to compare training methods.':'Real-model comparison · Check evaluation settings and training budgets before drawing a conclusion.'; document.querySelectorAll('[data-aggregate]').forEach(button=>button.setAttribute('aria-pressed',String(button.dataset.aggregate===state.aggregate))); - $('aggregateNote').textContent=state.aggregate!=='standard'?(state.aggregate==='english_eval'?'Balance English and other languages inside each eval, then average evals equally within categories.':'Balance English and other languages across each category. Evals with non-English coverage can receive more weight.')+' Shares are set in the weight editor. Unknown or mixed-language scores count as English for weighting; known non-English pools count as other. Raw views stay unchanged.':'Original aggregate: average variants within each eval, then evals within each category.'; + $('aggregateNote').textContent=state.aggregate!=='standard'?(state.aggregate==='english_eval'?'Balance English and other languages inside each eval, then average evals equally within categories.':'Balance English and other languages across each category. Evals with non-English coverage can receive more weight.')+' Shares are set in the weight editor. Unknown or mixed-language scores count as English for weighting; known non-English pools count as other. Raw views stay unchanged.':'Original aggregate: combine configured components within each language/protocol; average those groups within the eval. Other evals average variants. Then average evals within each category.'; $('cards').innerHTML=[[$('modelA').value,ta.score,'A · '+(state.aggregate==='english_eval'?'English balance per eval':state.aggregate==='english_category'?'English balance per category':'original weighted score')],[$('modelB').value,tb.score,'B · '+(state.aggregate==='english_eval'?'English balance per eval':state.aggregate==='english_category'?'English balance per category':'original weighted score')],['A − B',valid?ta.score-tb.score:null,'Weighted difference'+(suite.mode==='fixed'&&!scope.complete?' · incomplete set':'')]].map(([name,value,label])=>'
'+esc(name)+''+fmt(value)+''+label+'
').join(''); $('coverage').classList.toggle('notice',models.size>0&&suite.mode==='fixed'&&!scope.complete); - $('coverage').textContent=!models.size?'No models loaded · add your CSV.':(suite.mode==='fixed'?suite.name+' · '+(scope.complete?'Complete':'INCOMPLETE')+' · '+scope.sharedRequired+'/'+scope.required+' requirements shared (A '+scope.presentA+', B '+scope.presentB+') · '+(scope.extrasA+scope.extrasB)+' extra measurements excluded · ':suite.name+' · ')+pairs.length+' matched variants · excluded: '+onlyA.length+' from A, '+onlyB.length+' from B · weights use shared data only'+(valid?'':' · Score unavailable: check weights and language assignments.'); + $('coverage').textContent=!models.size?'No models loaded · add your CSV.':(suite.mode==='fixed'?suite.name+' · '+(scope.complete?'Complete':'INCOMPLETE')+' · '+scope.sharedRequired+'/'+scope.required+' requirements shared (A '+scope.presentA+', B '+scope.presentB+') · '+(scope.extrasA+scope.extrasB)+' extra measurements excluded · ':suite.name+' · ')+pairs.length+' matched variants · excluded: '+excludedA.length+' from A, '+excludedB.length+' from B · weights use shared data only'+(valid?'':' · Score unavailable: check weights and language assignments.'); $('filters').hidden=['score','warnings'].includes(state.view); $('filterStatus').textContent=shown.length+' of '+pairs.length+' matched variants · filters do not change the full weighted score'; const warnings=activeWarnings();$('warningCount').textContent=warnings.length;$('warningCount').classList.toggle('has-warnings',warnings.length>0); document.querySelectorAll('[data-view]').forEach(e=>e.classList.toggle('active',e.dataset.view===state.view)); - $('view').innerHTML=state.view==='score'?renderScore(a,b,ta,tb,valid):['categories','languages'].includes(state.view)?renderBreakdown(shown,a,valid):state.view==='comparisons'?renderComparisons(shown,a,valid):state.view==='warnings'?renderWarnings():renderConfig(); + $('view').innerHTML=state.view==='score'?renderScore(a,b,ta,tb,valid):['categories','languages'].includes(state.view)?renderBreakdown(shown,pairs):state.view==='comparisons'?renderComparisons(shown,a,valid):state.view==='warnings'?renderWarnings():renderConfig(); if(weightOpen&&$('weightEditor'))$('weightEditor').open=true;if(englishOpen&&$('englishComponents'))$('englishComponents').open=true; markHorizontalScroll(); if(!shown.length&&['categories','languages','comparisons'].includes(state.view))$('view').insertAdjacentHTML('beforeend','

No matched variants pass these filters.

'); @@ -355,7 +199,7 @@ function start(){ validateCatalogue(config);const nextScheme=resolveConfig(config,suite,profile); const audits=new Map(),nextModels=new Map(),nextMetadata=new Map(); for(const [name,rows] of sourceAudits){const audit=auditRows(rows,config);audits.set(name,audit);nextModels.set(name,audit.filter(r=>r.selected));for(const row of audit)if(!nextMetadata.has(row.task))nextMetadata.set(row.task,taskLanguage(row.task,config));} - if(nextModels.size)nextModels.set(demoModel,synthetic([...nextModels.values()][0],config)); + addSynthetic(nextModels,config); catalogue=config;scheme=nextScheme;sourceAudits=audits;models=nextModels;metadata=nextMetadata; weights={...nextScheme.weights,...weights};refreshConfig(); } @@ -441,7 +285,7 @@ function start(){ const nextModels=new Map(models),nextAudits=new Map(sourceAudits),nextMetadata=new Map(metadata); for(const name of names){nextModels.set(name,rows.filter(r=>r.checkpoint===name));nextAudits.set(name,audit.filter(r=>r.checkpoint===name));} for(const r of audit)nextMetadata.set(r.task,taskLanguage(r.task,catalogue)); - if(!nextModels.has(demoModel))nextModels.set(demoModel,synthetic([...nextModels.values()][0],catalogue)); + if(!nextModels.has(demoModel))addSynthetic(nextModels,catalogue); models=nextModels;sourceAudits=nextAudits;metadata=nextMetadata; modelOptions(sourceAudits.size>1?names.at(-1):demoModel);languageOptions();$('error').textContent='';render(); }catch(err){$('error').textContent=err.message;}finally{e.target.value='';}}; diff --git a/app/build.py b/app/build.py index 8618654..442f5ef 100644 --- a/app/build.py +++ b/app/build.py @@ -4,8 +4,9 @@ import hashlib import json from pathlib import Path -from statistics import mean -from .config_engine import classify, task_language, load_catalogue, shared_config, load_csv +from quickdash.io import load_csv +from quickdash.config import classify, task_language, load_catalogue, load_profile, load_suite, resolve_config, scope_rows +from quickdash.analysis import totals, component_coverage, diagnostic, analyze APP = Path(__file__).resolve().parent ROOT = APP.parent @@ -13,64 +14,15 @@ def summarize(audit, config, aggregate=None): - aggregate = aggregate or config.get('aggregate', 'standard') - metadata = {task: group for group in config['languages'] for task in group['tasks']} - def side(row): - language = metadata.get(row['task'], {}) - code = language.get('target_language') if language.get('scope') == 'translation' else language.get('language') if language.get('scope') in ['single', 'pooled'] else None - return 'english' if not code or code in ['eng_Latn', 'mul'] else 'other' - result = [] + """Summarize each model independently using the native Python engine.""" + config={**config,'aggregate':aggregate or config.get('aggregate','standard')} + result=[] for model in sorted({r['checkpoint'] for r in audit}): - selected = [r for r in audit if r['checkpoint'] == model and r['selected']] - eval_rows = {e['name']: [r for r in selected if r['eval'] == e['name']] for e in config['evals']} - evals = [dict(name=e['name'], category=e['category'], metric=e['metric'], count=len(eval_rows[e['name']]), - score=mean(r['score_100'] for r in eval_rows[e['name']]) if eval_rows[e['name']] else None, - weight=0, contribution=None if eval_rows[e['name']] else 0, aggregateScore=None, excluded=not eval_rows[e['name']], englishShare=0, effectiveEnglishShare=None, englishScore=None, otherScore=None, issue='') for e in config['evals']] - available_weight = sum(weight for category, weight in config['weights'].items() if any(e['category'] == category and e['count'] for e in evals)) - categories = [] - for category, weight in config['weights'].items(): - configured = [e for e in evals if e['category'] == category] - ff = [e for e in configured if e['count']] - effective_weight = weight/available_weight if ff and available_weight else 0 - share = config.get('english_weights', {}).get(category, 0) if aggregate != 'standard' else 0 - c = dict(name=category, weight=effective_weight, excluded=not ff, excludedEvals=[e['name'] for e in configured if not e['count']], score=None, evals=len(ff), englishShare=share, effectiveEnglishShare=None, englishScore=None, otherScore=None, issue='') - coefficients = {} - if not ff: - categories.append(c) - continue - if not share: - c['score'] = mean(e['score'] for e in ff) - for e in ff: - for r in eval_rows[e['name']]: coefficients[id(r)] = effective_weight / len(ff) / len(eval_rows[e['name']]) - elif aggregate == 'english_eval': - for e in ff: - rr = eval_rows[e['name']] - e['englishShare'] = share - groups = [[r for r in rr if side(r) == group] for group in ['english', 'other']] - e['englishScore'], e['otherScore'] = [mean(r['score_100'] for r in group) if group else None for group in groups] - e['effectiveEnglishShare'] = 0 if e['englishScore'] is None else 1 if e['otherScore'] is None else share - parts = [e['effectiveEnglishShare'], 1-e['effectiveEnglishShare']] - e['aggregateScore'] = parts[0]*(e['englishScore'] or 0)+parts[1]*(e['otherScore'] or 0) - for part, group in zip(parts, groups): - for r in group: coefficients[id(r)] = effective_weight/len(ff)*part/len(group) - if not c['issue']: c['score'] = mean(e['aggregateScore'] for e in ff) - else: - groups = [[variants for e in ff if (variants := [r for r in eval_rows[e['name']] if side(r) == group])] for group in ['english', 'other']] - c['englishScore'], c['otherScore'] = [mean(mean(r['score_100'] for r in variants) for variants in group) if group else None for group in groups] - if not c['issue']: - c['effectiveEnglishShare'] = 0 if c['englishScore'] is None else 1 if c['otherScore'] is None else share - parts = [c['effectiveEnglishShare'], 1-c['effectiveEnglishShare']] - c['score'] = parts[0]*(c['englishScore'] or 0) + parts[1]*(c['otherScore'] or 0) - for part, group in zip(parts, groups): - for variants in group: - for r in variants: coefficients[id(r)] = effective_weight * part / len(group) / len(variants) - for e in ff: - e['weight'] = sum(coefficients.get(id(r), 0) for r in eval_rows[e['name']]) - e['contribution'] = sum(r['score_100']*coefficients.get(id(r), 0) for r in eval_rows[e['name']]) if c['score'] is not None else None - e['aggregateScore'] = e['contribution']/e['weight'] if c['score'] is not None and e['weight'] else None - categories.append(c) - complete = available_weight > 0 and all(c['score'] is not None for c in categories if not c['excluded']) - result.append(dict(model=model, evals=evals, categories=categories, score=sum(c['score']*c['weight'] for c in categories if not c['excluded']) if complete else None)) + rows=[r for r in audit if r['checkpoint']==model and r['selected']] + t=totals(rows,config) + excluded=component_coverage(rows,config)['excluded'] + diagnostics=[diagnostic('incomplete_components',model,name,[r for r in excluded if r['eval']==name],'Incomplete component group; excluded from scoring.','excluded') for name in dict.fromkeys(r['eval'] for r in excluded)] + result.append(dict(model=model,score=t['score'],warnings=diagnostics,evals=t['evals'],categories=[{**c,'evals':sum(e['category']==c['name'] and not e['excluded'] for e in t['evals'])} for c in t['categories']])) return result @@ -105,11 +57,11 @@ def config_choices(path, directory, kind, default_directory): if path is None: directory = directory or default_directory path = default_config(directory) - choices = [dict(file=path.name, config=shared_config(kind, path))] + choices = [dict(file=path.name, config={"weights":load_profile,"suite":load_suite}[kind](path))] if directory is not None: for candidate in directory_files(directory, {'.yaml', '.yml'}): if candidate.resolve() == path.resolve(): continue - try: config = shared_config(kind, candidate) + try: config = {"weights":load_profile,"suite":load_suite}[kind](candidate) except ValueError as error: raise ValueError(f'{candidate.name}: {error}') from error if any(p['config']['name'] == config['name'] for p in choices): raise ValueError(f'{candidate.name}: duplicate config name {config["name"]!r}; use distinct names') @@ -117,8 +69,9 @@ def config_choices(path, directory, kind, default_directory): return path, choices -def build(source, output, catalogue_path=None, results_dir=None, *, weights_path=None, weights_dir=None, suite_path=None, sets_dir=None): +def build(source, output, catalogue_path=None, results_dir=None, *, weights_path=None, weights_dir=None, suite_path=None, sets_dir=None, sample_csv=None): if source is not None and results_dir is not None: raise ValueError('Choose a CSV or --results-dir, not both') + if sample_csv is not None and results_dir is None: raise ValueError('--sample-csv requires --results-dir') catalogue_path = catalogue_path or ROOT/'configs/catalogue.yaml' catalogue = load_catalogue(catalogue_path) weights_path, profiles = config_choices(weights_path, weights_dir, 'weights', ROOT/'configs/weights') @@ -126,11 +79,13 @@ def build(source, output, catalogue_path=None, results_dir=None, *, weights_path # Every offered combination must resolve before any output is replaced. for profile in profiles: for entry in suites: - try: shared_config('resolve', value=[catalogue, entry['config'], profile['config']]) + try: resolve_config(catalogue, entry['config'], profile['config']) except ValueError as error: raise ValueError(f'{entry["file"]} / {profile["file"]}: {error}') from error suite, profile = suites[0]['config'], profiles[0]['config'] - config = shared_config('resolve', value=[catalogue, suite, profile]) + config = resolve_config(catalogue, suite, profile) paths=[source] if source is not None else directory_files(results_dir,{'.csv'}) if results_dir is not None else [] + using_sample = not paths and sample_csv is not None + if using_sample: paths = [sample_csv] rows=[];audit=[];sources=[];owners={} for path in paths: try: @@ -142,12 +97,13 @@ def build(source, output, catalogue_path=None, results_dir=None, *, weights_path owners[model]=path.name rows.extend(rr);audit.extend(classified) sources.append(dict(file=path.name,sha256=hashlib.sha256(path.read_bytes()).hexdigest())) - scoped = shared_config('scope', value=[audit, suite])['rows'] + scoped = scope_rows(audit, suite)['rows'] identities = {tuple(r[k] for k in ['checkpoint','task','metric','filter','n_shot','harness','backend']) for r in scoped} included = {id(r) for r in audit if tuple(r[k] for k in ['checkpoint','task','metric','filter','n_shot','harness','backend']) in identities} scoped_audit = [dict(r, selected=r['selected'] and id(r) in included) for r in audit] aggregates = {mode: summarize(scoped_audit, config, mode) for mode in ['standard', 'english_eval', 'english_category']} summary = aggregates[config.get('aggregate', 'standard')] + diagnostics = analyze(rows,dict(catalogue=catalogue,suite=suite,profile=profile)).diagnostics output.mkdir(parents=True, exist_ok=True) if audit: write_csv(output/'row-audit.csv', audit) @@ -158,16 +114,16 @@ def build(source, output, catalogue_path=None, results_dir=None, *, weights_path (output/name).unlink(missing_ok=True) metadata = [task_language(task, catalogue) for task in sorted({r['task'] for r in rows})] if metadata:write_csv(output/'language-metadata.csv', metadata) - payload = dict(catalogue=catalogue, catalogue_file=catalogue_path.name, suite=suite, suite_file=suite_path.name, + payload = dict(diagnostics=diagnostics,catalogue=catalogue, catalogue_file=catalogue_path.name, suite=suite, suite_file=suite_path.name, profile=profile, profile_file=weights_path.name, suites=suites, profiles=profiles, - metadata=metadata, scheme=config, models=summary, aggregates=aggregates, rows=audit, sources=sources, + sample_models=sorted(owners) if using_sample else [], metadata=metadata, scheme=config, models=summary, aggregates=aggregates, rows=audit, sources=sources, source=source.name if source else results_dir.name if results_dir else '', sha256=sources[0]['sha256'] if len(sources)==1 else None) (output/'catalogue.yaml').write_text(catalogue_path.read_text()) (output/'weights.yaml').write_text(weights_path.read_text()) (output/'eval-set.yaml').write_text(suite_path.read_text()) (output/'analysis.json').write_text(json.dumps(payload, indent=2)) template = (APP/'template.html').read_text() - (output/'index.html').write_text(template.replace('__APP__', (APP/'vendor/js-yaml.js').read_text()+'\n'+(APP/'eval_config.js').read_text()+'\n'+(APP/'suite_config.js').read_text()+'\n'+(APP/'app.js').read_text()).replace('__PAYLOAD__', json.dumps(payload).replace('<', '\\u003c'))) + (output/'index.html').write_text(template.replace('__APP__', (APP/'vendor/js-yaml.js').read_text()+'\n'+(APP/'eval_config.js').read_text()+'\n'+(APP/'suite_config.js').read_text()+'\n'+(APP/'analysis.js').read_text()+'\n'+(APP/'app.js').read_text()).replace('__PAYLOAD__', json.dumps(payload).replace('<', '\\u003c'))) print(json.dumps(dict(models=summary, rows=len(audit), selected=sum(r['selected'] for r in scoped_audit)), indent=2)) @@ -175,6 +131,7 @@ def build(source, output, catalogue_path=None, results_dir=None, *, weights_path parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('csv', type=Path, nargs='?', help='CSV to embed; omit to start without results') parser.add_argument('--results-dir', type=Path, help='Embed all CSV files directly inside this directory') + parser.add_argument('--sample-csv', type=Path, help='Fallback CSV when --results-dir contains no CSVs') parser.add_argument('--catalogue', type=Path, help='Global eval interpretation YAML; default: configs/catalogue.yaml') parser.add_argument('--weights', type=Path, help='Default weighting profile YAML; used alone, embed only this profile') parser.add_argument('--weights-dir', type=Path, help='Offer weighting profiles from this directory (default: configs/weights)') @@ -183,4 +140,4 @@ def build(source, output, catalogue_path=None, results_dir=None, *, weights_path parser.add_argument('--output', type=Path, default=ROOT/'output') args = parser.parse_args() if args.csv is not None and args.results_dir is not None: parser.error('Choose a CSV or --results-dir, not both') - build(args.csv, args.output, args.catalogue, args.results_dir, weights_path=args.weights, weights_dir=args.weights_dir, suite_path=args.eval_set, sets_dir=args.sets_dir) + build(args.csv, args.output, args.catalogue, args.results_dir, weights_path=args.weights, weights_dir=args.weights_dir, suite_path=args.eval_set, sets_dir=args.sets_dir, sample_csv=args.sample_csv) diff --git a/app/config_engine.py b/app/config_engine.py deleted file mode 100644 index fa226a8..0000000 --- a/app/config_engine.py +++ /dev/null @@ -1,184 +0,0 @@ -"""Declarative eval matching, explicit language assignments and score normalization.""" -import math -import re -import json -import subprocess -from pathlib import Path - - -def shared_config(mode, path=None, value=None): - """Use bundled browser parsers and set validation during builds.""" - result = subprocess.run(["node", str(Path(__file__).with_name("config_io.cjs")), str(path) if path else "-", mode], - input=json.dumps(value) if path is None else None, capture_output=True, text=True) - if result.returncode: raise ValueError(result.stderr.strip()) - return json.loads(result.stdout) - - -def load_catalogue(path): - return validate_catalogue(shared_config('catalogue', path)) - - -def load_csv(path): - return shared_config('csv', path) - - -LANGUAGE_CODE = re.compile(r'(?:[a-z]{3}_[A-Z][a-z]{3}|mul)') -DECIMAL = re.compile(r'[+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+)(?:[eE][+-]?[0-9]+)?') - - -def object_keys(value, allowed, required=()): - if not isinstance(value, dict) or set(value)-set(allowed) or set(required)-set(value): - raise ValueError(f'Invalid config fields; allowed {sorted(allowed)}, required {sorted(required)}') - - -def number(value): - return isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value) - - -def validate_match(rule): - object_keys(rule, {'name','regex'}) - if len(rule)!=1 or not isinstance(next(iter(rule.values())),str) or not next(iter(rule.values())): - raise ValueError('A match needs exactly one nonempty name or regex') - if 'regex' in rule: - # Numeric captures use the shared Python/JavaScript regex subset. - if '(?P' in rule['regex'] or '(?<' in rule['regex']: - raise ValueError('Use numeric capture groups in portable regexes') - re.compile(rule['regex']) - - -def match_task(rule, task): - if 'name' in rule:return [task] if task==rule['name'] else None - match=re.fullmatch(rule['regex'],task) - return [match.group(0),*match.groups()] if match else None - - -def validate_config(config): - object_keys(config,{'version','name','weights','evals','languages','notes','aggregate','english_weights'}, {'version','name','weights','evals','languages'}) - if config['version']!=1 or isinstance(config['version'],bool):raise ValueError('Unsupported config version') - if not isinstance(config['name'],str) or not config['name']:raise ValueError('Config name is required') - weights=config['weights'] - if not isinstance(weights,dict) or not weights or any(not k or not number(v) or v<0 for k,v in weights.items()) or abs(sum(weights.values())-1)>1e-8: - raise ValueError('Category weights must be nonnegative and sum to 1') - if config.get('aggregate','standard') not in ['standard','english_eval','english_category']:raise ValueError('Aggregate must be standard, english_eval, or english_category') - if 'english_weights' in config: - object_keys(config['english_weights'],set(weights)) - if any(not number(v) or v<0 or v>1 for v in config['english_weights'].values()):raise ValueError('English weights must be between 0 and 1') - validate_rules(config) - if any(e['category'] not in weights for e in config['evals']):raise ValueError('Eval category has no weight') - return config - - -def validate_catalogue(config): - object_keys(config, {'version','name','evals','languages','notes'}, {'version','name','evals','languages'}) - return validate_rules(config) - - -def validate_rules(config): - if config['version'] != 1 or isinstance(config['version'], bool):raise ValueError('Unsupported config version') - if not isinstance(config['name'], str) or not config['name'].strip():raise ValueError('Config name is required') - if not isinstance(config.get('notes',[]),list) or any(not isinstance(n,str) for n in config.get('notes',[])):raise ValueError('Notes must be strings') - if not isinstance(config['evals'],list) or not config['evals']:raise ValueError('At least one eval is required') - names=set();categories=set() - for e in config['evals']: - object_keys(e,{'name','category','match','metric','filter','shots','select','score','normalize','warning'}, {'name','category','match','metric','filter','score'}) - if not isinstance(e['name'],str) or not e['name'] or e['name'] in names:raise ValueError('Eval names must be unique and nonempty') - names.add(e['name']) - if not isinstance(e['category'],str) or not e['category'].strip():raise ValueError('Eval category must be text') - categories.add(e['category']) - if not isinstance(e['metric'],str) or not e['metric'] or not isinstance(e['filter'],str):raise ValueError('Metric and filter must be strings') - validate_match(e['match']) - if 'select' in e:validate_match(e['select']) - if 'shots' in e and (not isinstance(e['shots'],int) or isinstance(e['shots'],bool) or e['shots']<0):raise ValueError('shots must be a nonnegative integer') - object_keys(e['score'],{'scale'},{'scale'}) - if not number(e['score']['scale']) or e['score']['scale']<=0:raise ValueError('Score scale must be positive') - if 'warning' in e and (not isinstance(e['warning'],str) or not e['warning'].strip()):raise ValueError('Eval warning must be nonempty text') - if 'normalize' in e: - n=e['normalize'];object_keys(n,{'min','max','clip','basis','note','sources'},{'min','max'}) - if not number(n['min']) or not number(n['max']) or not 0<=n['min']1:raise ValueError('Ambiguous eval config for task: '+task) - return matches[0] if matches else (None,None) - - -def task_language(task,config): - result=dict(task=task,language='',source_language='',target_language='',scope='unknown',status='unknown',evidence='',provenance='No explicit language assignment for this task.') - for group in config['languages']: - if task in group['tasks']: - result.update({k:group.get(k,'') for k in ['language','source_language','target_language','scope','evidence']}) - result.update(status='resolved',provenance=group.get('note','Explicit language assignment in eval config.')) - break - return result - - -def classify(rows,config): - (validate_config if 'weights' in config else validate_catalogue)(config);audit=[];seen=set() - for index,source in enumerate(rows): - r=dict(source) - for field in ['checkpoint','task','metric','filter','n_shot','harness','backend','value']: - if field not in r:raise ValueError('Missing CSV column: '+field) - for field in ['checkpoint','task','metric','harness','backend']: - if not isinstance(r[field],str) or not r[field].strip():raise ValueError(f'CSV row {index+2}: {field} must be nonempty text') - if r['checkpoint']=='SYNTHETIC demo — perturbed':raise ValueError('Checkpoint name is reserved for the synthetic demo') - if not isinstance(r['filter'],str):raise ValueError(f'CSV row {index+2}: filter must be text (blank is allowed)') - shots=str(r['n_shot']) - if not re.fullmatch(r'(?:0|[1-9][0-9]*)',shots) or int(shots)>2**53-1:raise ValueError(f'CSV row {index+2}: n_shot must be a nonnegative integer') - r['n_shot']=shots - e,_=eval_for_task(r['task'],config) - if e is None: - r.update(category='',eval='',selected=False,raw_score_100=None,score_100=None,decision='No eval config; excluded from scoring') - audit.append(r);continue - decision='Selected for the weighted score' - if 'select' in e and match_task(e['select'],r['task']) is None:decision='Excluded summary level or alternate protocol; see eval selection rule' - elif r['metric']!=e['metric']:decision='Alternate metric; using '+e['metric'] - elif r['filter']!=e['filter']:decision='Alternate extraction filter; using '+(e['filter'] or '(empty)') - elif 'shots' in e and str(r['n_shot'])!=str(e['shots']):decision='Alternate shot setting; using '+str(e['shots'])+' shots' - selected=decision=='Selected for the weighted score';raw=adjusted=None - if selected: - try:raw,adjusted=normalize_score(r['value'],e) - except (ValueError,OverflowError) as error:raise ValueError(f"CSV row {index+2} · {r['checkpoint']} · {r['task']} · {r['metric']}: Invalid score: {error}") from error - key=tuple(r[k] for k in ['checkpoint','task','metric','filter','n_shot','harness','backend']) - if key in seen:raise ValueError('Duplicate selected measurement: '+str(key)) - seen.add(key) - r.update(category=e['category'],eval=e['name'],selected=selected,raw_score_100=raw,score_100=adjusted,decision=decision) - audit.append(r) - return audit diff --git a/app/config_io.cjs b/app/config_io.cjs deleted file mode 100644 index b595c5c..0000000 --- a/app/config_io.cjs +++ /dev/null @@ -1,8 +0,0 @@ -// Python builds share the browser's parsers, validation and set membership. -const fs=require('node:fs'),evals=require('./eval_config.js'),sets=require('./suite_config.js'); -try{ - const source=fs.readFileSync(process.argv[2]==='-'?0:process.argv[2],'utf8'),mode=process.argv[3]; - const parsers={csv:evals.parseCSV,catalogue:evals.parseCatalogue,weights:sets.parseWeightProfile,suite:sets.parseSuite}; - const value=mode==='resolve'?sets.resolveConfig(...JSON.parse(source)):mode==='scope'?sets.scopeRows(...JSON.parse(source)):parsers[mode](source); - process.stdout.write(JSON.stringify(value)); -}catch(error){process.stderr.write(error.message+'\n');process.exitCode=1;} diff --git a/app/eval_config.js b/app/eval_config.js index 0772495..52fd09b 100644 --- a/app/eval_config.js +++ b/app/eval_config.js @@ -1,11 +1,12 @@ 'use strict'; const EvalConfig=(()=>{ const yaml=typeof module!=='undefined'?require('./vendor/js-yaml.js'):jsyaml; - const canonical=/^(?:[a-z]{3}_[A-Z][a-z]{3}|mul)$/; + const canonical=/^(?:[a-z]{3}_[A-Z][a-z]{3}|mul)$(?![\s\S])/; const objectKeys=(value,allowed,required=[])=>{if(!value||typeof value!=='object'||Array.isArray(value)||Object.keys(value).some(k=>!allowed.includes(k))||required.some(k=>!Object.hasOwn(value,k)))throw Error('Invalid config fields; allowed '+allowed.join(', ')+'; required '+required.join(', '));}; const number=v=>typeof v==='number'&&Number.isFinite(v); const decimal=/^[+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+)(?:[eE][+-]?[0-9]+)?$/; const demoModel='SYNTHETIC demo — perturbed'; + const isDemoModel=name=>typeof name==='string'&&name.startsWith('SYNTHETIC demo — '); function parseCSV(source){ const text=source.replace(/^\uFEFF/,''),records=[];let row=[],value='',state='start',touched=false; const field=()=>{row.push(value);value='';state='start';}; @@ -34,8 +35,30 @@ const EvalConfig=(()=>{ return Object.fromEntries(header.map((h,j)=>[h,r[j]])); }); } - function validateMatch(rule){objectKeys(rule,['name','regex']);const values=Object.values(rule);if(values.length!==1||typeof values[0]!=='string'||!values[0])throw Error('A match needs exactly one nonempty name or regex');if('regex'in rule){if(rule.regex.includes('(?P')||rule.regex.includes('(?<'))throw Error('Use portable regexes');new RegExp(rule.regex);}} - function matchTask(rule,task){return 'name'in rule?rule.name===task:new RegExp('^(?:'+rule.regex+')$(?![\\s\\S])').test(task);} + + function portableRegex(pattern){ + let inside=false,quantifier=false; + for(let i=0;i!k||!number(v)||v<0)||Math.abs(Object.values(w).reduce((a,b)=>a+b,0)-1)>1e-8)throw Error('Category weights must be nonnegative and sum to 1'); @@ -57,19 +80,28 @@ const EvalConfig=(()=>{ function serializeCatalogue(config){return yaml.dump(validateCatalogue(config),{schema:yaml.CORE_SCHEMA,lineWidth:110,noRefs:true});} function validateRules(config){ if(config.version!==1)throw Error('Unsupported config version'); - if(typeof config.name!=='string'||!config.name)throw Error('Config name is required'); + if(typeof config.name!=='string'||!config.name.trim())throw Error('Config name is required'); if('notes'in config&&(!Array.isArray(config.notes)||config.notes.some(n=>typeof n!=='string')))throw Error('Notes must be strings'); if(!Array.isArray(config.evals)||!config.evals.length)throw Error('At least one eval is required'); const names=new Set(); for(const e of config.evals){ - objectKeys(e,['name','category','match','metric','filter','shots','select','score','normalize','warning'],['name','category','match','metric','filter','score']); + objectKeys(e,['name','category','match','metric','filter','shots','select','score','normalize','warning','aggregation'],['name','category','match','metric','filter','score']); if(typeof e.name!=='string'||!e.name||names.has(e.name))throw Error('Eval names must be unique and nonempty');names.add(e.name); - if(typeof e.category!=='string'||!e.category)throw Error('Eval category must be nonempty text'); + if(typeof e.category!=='string'||!e.category.trim())throw Error('Eval category must be nonempty text'); if(typeof e.metric!=='string'||!e.metric||typeof e.filter!=='string')throw Error('Metric and filter must be strings'); validateMatch(e.match);if('select'in e)validateMatch(e.select); - if('shots'in e&&(!Number.isInteger(e.shots)||e.shots<0))throw Error('shots must be a nonnegative integer'); + if('shots'in e&&(!Number.isSafeInteger(e.shots)||e.shots<0))throw Error('shots must be a nonnegative integer'); objectKeys(e.score,['scale'],['scale']);if(!number(e.score.scale)||e.score.scale<=0)throw Error('Score scale must be positive'); if('warning'in e&&(typeof e.warning!=='string'||!e.warning.trim()))throw Error('Eval warning must be nonempty text'); + if('aggregation'in e){ + const a=e.aggregation;objectKeys(a,['components','note','sources'],['components']); + if(!Array.isArray(a.components)||!a.components.length)throw Error('Aggregation needs components'); + const names=new Set();let total=0; + for(const c of a.components){objectKeys(c,['name','match','relative_weight'],['name','match','relative_weight']);if(typeof c.name!=='string'||!c.name.trim()||names.has(c.name))throw Error('Component names must be unique and nonempty');names.add(c.name);validateMatch(c.match);if(!number(c.relative_weight)||c.relative_weight<=0)throw Error('Component weights must be positive finite numbers');total+=c.relative_weight;} + if(!Number.isFinite(total))throw Error('Component weight sum must be finite'); + if('note'in a&&typeof a.note!=='string')throw Error('Aggregation note must be text'); + if('sources'in a&&(!Array.isArray(a.sources)||a.sources.some(u=>typeof u!=='string'||!/^https?:\/\//.test(u))))throw Error('Aggregation sources must be HTTP(S) URLs'); + } if('normalize'in e){const n=e.normalize;objectKeys(n,['min','max','clip','basis','note','sources'],['min','max']);if(!number(n.min)||!number(n.max)||!(0<=n.min&&n.mintypeof u!=='string'||!/^https?:\/\//.test(u))))throw Error('Normalization sources must be HTTP(S) URLs');} } if(!Array.isArray(config.languages))throw Error('languages must be a list');const seen=new Set(); @@ -84,6 +116,39 @@ const EvalConfig=(()=>{ for(const f of ['note','evidence'])if(f in g&&typeof g[f]!=='string')throw Error(f+' must be a string'); if(g.evidence&&!/^https?:\/\//.test(g.evidence))throw Error('Evidence links must use HTTP or HTTPS'); } + return validateAggregationConfig(config); + } + // Validate concrete task selections without attempting to infer languages from regexes. + function validateAggregationSelection(e,variants,config,unique=true){ + if(!e.aggregation)return; + const metadata=new Map(config.languages.flatMap(g=>g.tasks.map(task=>[task,g]))),groups=new Map(); + const fail=detail=>{throw Error('Incompatible aggregation config for '+e.name+': '+detail);}; + for(const v of variants){ + const g=metadata.get(v.task);if(!g)fail(v.task+' needs an explicit language assignment'); + const language=g.scope==='translation'?g.source_language+' → '+g.target_language:g.language; + const shot=v.n_shot??e.shots??'*',id=JSON.stringify([g.scope,language,shot]); + if(!groups.has(id))groups.set(id,{label:language+' / shots '+shot,parts:new Map(e.aggregation.components.map(c=>[c.name,[]]))}); + const matches=e.aggregation.components.filter(c=>matchTask(c.match,v.task)); + if(matches.length!==1)fail(v.task+' matches '+matches.length+' components; expected exactly one'); + groups.get(id).parts.get(matches[0].name).push(v.task); + } + for(const group of groups.values())for(const [name,tasks] of group.parts){ + if(!tasks.length)fail(group.label+' is missing required component '+name+'; select every component with compatible shot settings'); + if(unique&&tasks.length>1)fail(group.label+' has multiple tasks for component '+name+': '+tasks.join(', ')); + } + } + function validateAggregationConfig(config){ + const knownTasks=config.languages.flatMap(g=>g.tasks); + for(const e of config.evals.filter(e=>e.aggregation)){ + const eligible=task=>matchTask(e.match,task)&&(!e.select||matchTask(e.select,task)); + const tasks=new Set(knownTasks.filter(eligible)); + for(const c of e.aggregation.components)if('name'in c.match){ + if(!eligible(c.match.name))throw Error('Incompatible aggregation config for '+e.name+': component '+c.name+' task '+c.match.name+' is excluded by the eval match/selection rule'); + tasks.add(c.match.name); + } + for(const task of tasks)if(config.evals.filter(rule=>matchTask(rule.match,task)).length!==1)throw Error('Incompatible aggregation config for '+e.name+': '+task+' has ambiguous eval rules'); + validateAggregationSelection(e,[...tasks].map(task=>({task})),config,false); + } return config; } function normalizeScore(value,e){if(!number(value)&&(typeof value!=='string'||!decimal.test(value.trim())))throw Error('Invalid score: expected a finite decimal number');const raw=Number(value)/e.score.scale;if(!Number.isFinite(raw)||raw<0||raw>1)throw Error('Invalid score: outside the configured source scale');const n=e.normalize??{min:0,max:1};let adjusted=(raw-n.min)/(n.max-n.min);if(n.clip!==false)adjusted=Math.max(0,Math.min(1,adjusted));if(!Number.isFinite(adjusted*100))throw Error('Invalid score: normalization overflow');return {raw_score_100:raw*100,score_100:adjusted*100};} @@ -93,9 +158,9 @@ const EvalConfig=(()=>{ const r={...source}; for(const field of ['checkpoint','task','metric','filter','n_shot','harness','backend','value'])if(!Object.hasOwn(r,field))throw Error('Missing CSV column: '+field); for(const field of ['checkpoint','task','metric','harness','backend'])if(typeof r[field]!=='string'||!r[field].trim())throw Error('CSV row '+(index+2)+': '+field+' must be nonempty text'); - if(r.checkpoint===demoModel)throw Error('Checkpoint name is reserved for the synthetic demo: '+demoModel); + if(isDemoModel(r.checkpoint))throw Error('Checkpoint name is reserved for the synthetic demo: '+r.checkpoint); if(typeof r.filter!=='string')throw Error('CSV row '+(index+2)+': filter must be text (blank is allowed)'); - if(!/^(?:0|[1-9][0-9]*)$/.test(String(r.n_shot))||!Number.isSafeInteger(Number(r.n_shot)))throw Error('CSV row '+(index+2)+': n_shot must be a nonnegative integer'); + if(!/^(?:0|[1-9][0-9]*)$(?![\s\S])/.test(String(r.n_shot))||!Number.isSafeInteger(Number(r.n_shot)))throw Error('CSV row '+(index+2)+': n_shot must be a nonnegative integer'); r.n_shot=String(r.n_shot); const matches=config.evals.filter(e=>matchTask(e.match,r.task));if(matches.length>1)throw Error('Ambiguous eval config for task: '+r.task);if(!matches.length)return {...r,eval:'',category:'',selected:false,decision:'No eval config; excluded from scoring',raw_score_100:null,score_100:null};const e=matches[0];let decision='Selected for the weighted score'; if(e.select&&!matchTask(e.select,r.task))decision='Excluded summary level or alternate protocol; see eval selection rule'; @@ -103,10 +168,10 @@ const EvalConfig=(()=>{ else if(r.filter!==e.filter)decision='Alternate extraction filter; using '+(e.filter||'(empty)'); else if('shots'in e&&String(r.n_shot)!==String(e.shots))decision='Alternate shot setting; using '+e.shots+' shots'; const selected=decision==='Selected for the weighted score';let scores={raw_score_100:null,score_100:null}; - if(selected){try{scores=normalizeScore(r.value,e);}catch(error){throw Error('CSV row '+(index+2)+' · '+r.checkpoint+' · '+r.task+' · '+r.metric+': '+error.message);}const key=JSON.stringify(['checkpoint','task','metric','filter','n_shot','harness','backend'].map(k=>r[k]));if(seen.has(key))throw Error('Duplicate selected measurement: '+r.checkpoint+' · '+r.task+' · '+r.metric);seen.add(key);} + if(selected){if(e.aggregation&&e.aggregation.components.filter(c=>matchTask(c.match,r.task)).length!==1)throw Error('Incompatible aggregation config for '+e.name+': '+r.task+' must match exactly one component');try{scores=normalizeScore(r.value,e);}catch(error){throw Error('CSV row '+(index+2)+' · '+r.checkpoint+' · '+r.task+' · '+r.metric+': '+error.message);}const key=JSON.stringify(['checkpoint','task','metric','filter','n_shot','harness','backend'].map(k=>r[k]));if(seen.has(key))throw Error('Duplicate selected measurement: '+r.checkpoint+' · '+r.task+' · '+r.metric);seen.add(key);} return {...r,eval:e.name,category:e.category,selected,decision,...scores}; }); } - return {parseCatalogue,serializeCatalogue,validateCatalogue,validateWeights,parseCSV,validateConfig,matchTask,normalizeScore,taskLanguage,auditRows,demoModel}; + return {validateAggregationConfig,validateAggregationSelection,parseCatalogue,serializeCatalogue,validateCatalogue,validateWeights,parseCSV,validateConfig,matchTask,normalizeScore,taskLanguage,auditRows,demoModel,isDemoModel}; })(); if(typeof module!=='undefined')module.exports=EvalConfig; diff --git a/app/suite_config.js b/app/suite_config.js index 7979ed5..ee63137 100644 --- a/app/suite_config.js +++ b/app/suite_config.js @@ -46,6 +46,7 @@ const SuiteConfig=(()=>{ const matches=catalogue.evals.filter(rule=>api.matchTask(rule.match,v.task)); if(matches.length!==1||matches[0].name!==e.name||e.select&&!api.matchTask(e.select,v.task)||'shots'in e&&'n_shot'in v&&e.shots!==v.n_shot)throw Error('Required variant is not selected by its catalogue rule: '+v.task); } + if(required.variants)api.validateAggregationSelection(e,required.variants,catalogue); return e; }); return api.validateConfig({version:1,name:catalogue.name,evals,languages:catalogue.languages,weights:{...profile.weights,...Object.fromEntries(evals.filter(e=>!Object.hasOwn(profile.weights,e.category)).map(e=>[e.category,0]))},english_weights:{...profile.english_weights},aggregate:profile.aggregate||'standard',notes:[...(catalogue.notes||[]),...(profile.notes||[])]}); diff --git a/app/template.html b/app/template.html index 8e2230e..a973430 100644 --- a/app/template.html +++ b/app/template.html @@ -3,6 +3,7 @@ :root{font:14px system-ui;color:#1e2e44;background:#f5f7fa}*{box-sizing:border-box}body{margin:0}header{padding:23px 4vw 18px;background:#173149;color:white;display:flex;justify-content:space-between;align-items:center}h1{font-size:25px;margin:0}header span{font-size:12px;letter-spacing:.1em;color:#bdd0e2}main{max-width:1540px;margin:auto;padding:20px 4vw}h2{margin:0;font-size:21px}h3{margin:0;font-size:17px}p{line-height:1.5;margin:8px 0 16px}small,.caption,.muted{color:#63758b}.caption{font-size:12px}.notice{padding:10px 14px;background:#fff4d9;border-left:3px solid #bf892d;margin:0 0 16px;font-size:13px}.model-controls,.view-controls{display:flex;gap:12px;align-items:end;flex-wrap:wrap}.model-controls{margin-bottom:16px}label{display:flex;flex-direction:column;gap:5px;font-size:12px;color:#52677f}button,select,input{font:inherit;color:#1e2e44;border:1px solid #bcc8d5;border-radius:5px;background:#fff;padding:8px}select{max-width:310px}button{cursor:pointer}button:hover{background:#edf3f9}#cards{display:grid;grid-template-columns:repeat(3,1fr);gap:14px}.score-card{background:#fff;border:1px solid #dce4ed;border-radius:8px;padding:16px 20px}.score-card strong{display:block;font-size:32px;margin:6px 0}.score-card span{font-size:12px;color:#63758b}#coverage{margin:9px 0 20px}.tabs{display:flex;gap:3px;border-bottom:1px solid #cdd8e5;margin:0 0 18px}.tabs button{border:0;border-radius:4px 4px 0 0;background:transparent;padding:12px 18px;font-size:14px}.tabs button.active{color:#1a6089;border-bottom:3px solid #267ba6;background:#eaf2f8;font-weight:600}.filters{padding:14px 16px;background:#edf2f7;border-radius:7px;margin-bottom:16px}.filters[hidden]{display:none}#filterStatus{margin:10px 0 0}#view{padding:22px;background:white;border:1px solid #dce4ed;border-radius:8px}.section-heading{display:flex;justify-content:space-between;align-items:center;gap:14px;flex-wrap:wrap;margin-bottom:18px}.section-heading p{color:#63758b;margin:6px 0 0}.formula{display:flex;gap:12px;flex-wrap:wrap;align-items:center;margin:8px 0 22px;font-size:13px}.formula span{padding:9px 12px;background:#eff5fa;border-radius:5px}.formula b{color:#8395a8}.table-scroll{overflow:auto;margin-bottom:12px}table{width:100%;border-collapse:collapse;font-size:13px}th{text-align:left;background:#f0f4f8;position:sticky;top:0;z-index:1}td,th{padding:10px 12px;border-bottom:1px solid #e7edf3}td.num,th.num{text-align:right;font-variant-numeric:tabular-nums;white-space:nowrap}.positive{color:#128276}.negative{color:#b44660}.text-button{border:none;background:transparent;padding:3px 5px;color:#206b94;font-size:12px}.delta-track{height:19px;width:260px;position:relative;background:linear-gradient(to right,transparent 49.7%,#8496a9 49.7%,#8496a9 50.3%,transparent 50.3%)}.delta-track span{position:absolute;top:2px;height:15px;border-radius:2px}.chart-cell{min-width:284px}details{margin:16px 0}summary{cursor:pointer;font-weight:600}#weights{margin:14px 0}#weights table{max-width:720px}#weights input{width:100px;text-align:right}#weights td:first-child{font-weight:600}#weightEditor{padding:15px;background:#f5f8fb;border-radius:7px;margin-bottom:26px}.catalogue-head,.catalogue-summary{display:grid;grid-template-columns:1.1fr .8fr 1fr 1.4fr 1fr;padding-left:12px;padding-right:12px;gap:14px;align-items:center}.catalogue-head{padding:12px;background:#f0f4f8;font-weight:600;font-size:12px}.catalogue-eval{margin:0;border-bottom:1px solid #e0e7ef;padding:13px 0}.catalogue-summary{list-style:none}.catalogue-summary::-webkit-details-marker{display:none}.catalogue-summary>span:first-child:before{content:'▸ ';color:#7890a6}.catalogue-eval[open]>.catalogue-summary>span:first-child:before{content:'▾ '}.catalogue-eval>.table-scroll{margin-top:16px}.catalogue-variant{margin:0;min-width:220px}.catalogue-variant[open]{min-width:670px}.catalogue-variant p{font-size:12px;font-weight:400}.catalogue-variant summary{font-size:12px}.catalogue-metric td{font-size:12px}code{overflow-wrap:anywhere}a{color:#206b94}#error{color:#a82643;margin:10px 0}footer{font-size:12px;color:#728399;margin:18px 0}li{line-height:1.5;margin:8px 0}.import-label{cursor:pointer;background:white;border:1px solid #bcc8d5;border-radius:5px;padding:8px;color:#1e2e44;font-size:14px}.import-label input{display:none}@media(max-width:700px){main{padding:14px}header{padding:18px}.score-card{padding:12px}.score-card strong{font-size:25px}#cards{gap:8px}.score-card small{word-break:break-word}.tabs{flex-wrap:wrap}.tabs button{padding:10px}#view{padding:15px}.catalogue-head{display:none}.catalogue-summary{grid-template-columns:1fr 1fr}.model-controls label{max-width:100%}select{max-width:100%}.formula{gap:7px}.formula span{padding:7px}} .breakdown-scroll{overflow:auto;margin-bottom:16px}.breakdown-tree{min-width:650px}.tree-row{display:grid;grid-template-columns:minmax(330px,1fr) 65px 80px 80px 85px;align-items:center;gap:12px;min-height:46px;padding:10px 12px;border-bottom:1px solid #e7edf3;font-size:13px;font-weight:400;list-style:none}.tree-row::-webkit-details-marker{display:none}.tree-head{background:#f0f4f8;font-weight:600;color:#52677f}.tree-head>span:not(:first-child),.tree-number,.tree-count{text-align:right;font-variant-numeric:tabular-nums}.tree-count{color:#63758b}.tree-label{display:flex;align-items:center;gap:8px;min-width:0;overflow-wrap:anywhere}.tree-label small{display:block;margin-top:3px;font-size:11px}.tree-chevron{display:inline-block;color:#7890a6;width:12px;flex-shrink:0}.breakdown-node{margin:0}.breakdown-node>summary:hover{background:#f5f8fb}.breakdown-node[open]>summary{background:#f5f8fb}.breakdown-node[open]>summary .tree-chevron{transform:rotate(90deg)}.breakdown-tree>.breakdown-node>summary{font-weight:600}.tree-leaf{color:#52677f} +.breakdown-tree.with-components{min-width:1120px}.breakdown-tree.with-components .tree-row{grid-template-columns:minmax(330px,1fr) 65px 80px 80px 85px 110px 120px 120px} .variant-language{display:block;margin-top:4px;font-size:11px}.comparison-child{padding-left:30px;background:#f8fafc}.comparison-detail{font-size:12px} .chart-heading{width:260px;text-align:center}.warning-count{display:inline-block;border-radius:10px;padding:1px 6px;background:#e2e9ef;color:#52677f}.warning-count.has-warnings{background:#b3263e;color:white}.normalization-info{margin:12px;padding:12px;background:#f5f8fb;border-radius:5px;font-size:12px}.normalization-info p{margin:6px 0} .sort-header{border:0;background:transparent;padding:0;color:inherit;font:inherit;font-weight:600;text-align:inherit;white-space:nowrap}.sort-header:hover{color:#206b94;text-decoration:underline} diff --git a/configs/README.md b/configs/README.md index b4d0f60..c10f405 100644 --- a/configs/README.md +++ b/configs/README.md @@ -3,6 +3,7 @@ Choose the file to edit based on what you want to change: - **Interpret a new eval:** add its task matching, category, scoring metric, normalization, and explicit language assignments to [catalogue.yaml](catalogue.yaml). Adding a rule does not require every model to run it. +- **Combine component results:** add an `aggregation` rule to the eval in the catalogue. Each component stores a `relative_weight`; PolyMath uses 1, 2, 4, and 8, divided by their sum when scoring. Named sets must select complete component groups with compatible shot settings; incompatible configurations are errors. Missing results within a valid selection warn and exclude the group; see [component aggregation](../docs/configuration.md#weighted-components-within-an-eval). - **Try different weighting:** add a YAML profile to [weights/](weights/). It can be used with any eval set. Category weights, English shares, and the default calculation live here. - **Require a standard comparison set:** add a YAML file to [sets/](sets/). [flagship-1.yaml](sets/flagship-1.yaml) pins expected tasks and shot counts; [any-available.yaml](sets/any-available.yaml) needs no required-eval list and compares shared data, with an explicit exclusion for unvalidated prompted Global PIQA. diff --git a/configs/catalogue.yaml b/configs/catalogue.yaml index ea3f25b..0a98cd5 100644 --- a/configs/catalogue.yaml +++ b/configs/catalogue.yaml @@ -273,6 +273,27 @@ evals: filter: none match: regex: polymath_.+ + aggregation: + components: + - name: low + match: {regex: 'polymath_.+_low'} + relative_weight: 1 + - name: medium + match: {regex: 'polymath_.+_medium'} + relative_weight: 2 + - name: high + match: {regex: 'polymath_.+_high'} + relative_weight: 4 + - name: top + match: {regex: 'polymath_.+_top'} + relative_weight: 8 + note: >- + PolyMath difficulty-weighted accuracy combines all four levels within each language: + (low + 2 × medium + 4 × high + 8 × top) / 15. Exported exact_match values are + unweighted per-level accuracies. Each level is required; incomplete groups are excluded. + sources: + - https://qwen-polymath.github.io/#benchmark-score + - https://github.com/QwenLM/PolyMath/blob/main/eval/run_eval.py score: scale: 1 normalize: diff --git a/docs/configuration.md b/docs/configuration.md index a65acae..6b4927f 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -80,6 +80,7 @@ The builder embeds YAML profiles directly in `configs/weights/` and sets directl - `--weights PATH` chooses a profile; used alone, it embeds only that profile. `--weights-dir DIR` offers the profiles in another directory and uses its `default.txt` unless an explicit profile is supplied. - `--eval-set PATH` and `--sets-dir DIR` work the same way for eval sets. - `--results-dir DIR` embeds CSVs directly in that directory. A checkpoint label may occur in only one file; one file can contain multiple models. +- `--sample-csv FILE`, used with `--results-dir`, supplies a fallback only if that directory has no CSVs. Invalid shared files stop the build; they never trigger the fallback. Pages uses `examples/sample-evals.csv` for this option. Every offered set is validated against the catalogue and every profile before writing output. Raw results are classified once by the global catalogue; changing sets only changes comparison membership. An empty results directory, or no CSV input, starts without models. @@ -134,9 +135,12 @@ For an input row with `value=0.625`, the raw score is 62.5 and the normalized sc | `select` | Optional name/regex rule restricting which matched tasks contribute. Useful for selecting summaries while retaining child-task audits. | | `score.scale` | Raw metric's upper scale: 1 for fractional accuracy; 100 for percentage or chrF scores. Selected values must be finite and within 0..scale. | | `warning` | Optional nonempty text describing an unresolved scoring assumption. Appears once in Warnings when present in the selected comparison and in this eval’s configuration details; it does not change scores. | +| `aggregation` | Optional component rules with positive relative weights; see [weighted components](#weighted-components-within-an-eval). | | `normalize` | Optional object with `min`, `max`, and optional `clip`, `basis`, `note`, and `sources`. Thresholds are fractions after division by `score.scale`. | -Use the common Python/JavaScript regex subset: literal text, character classes, alternatives, groups, and ordinary quantifiers. Patterns match the entire task name. Named groups and lookbehind are rejected. Language extraction does not use these patterns. +Use the shared Python/JavaScript regex subset: literal text, character classes, alternatives, capturing/noncapturing groups, and ordinary quantifiers. Patterns match the entire task name. Flags, lookarounds, named groups, backreferences and possessive quantifiers are rejected. `\d` and `\w` use ASCII character classes; `\s` uses ECMAScript whitespace and must not appear inside a character class. Dot excludes line terminators and matches one Unicode code point. Language extraction does not use these patterns. Exact-name rules are available when regexes are unnecessary. + +The [Python API](python-api.md) consumes these same configurations and returns score trees, audits, coverage, and structured diagnostics. All configuration counts are derived from the supplied inputs. The Original weighting profile gives Code, Math, Reasoning, Knowledge, Commonsense, and Reading a weight of 0.15 each; Translation, Language, and Instruction following each receive 0.1/3. Category weights must be nonnegative and sum to 1. Names, metrics, and task strings are case-sensitive. Unknown config fields are rejected to catch typos. `version` must be 1; `name` labels the active config. Optional top-level `notes` is a list of strings. @@ -152,7 +156,60 @@ normalized_score = 100 × clamp((raw_fraction − lo) / (hi − lo), 0, 1) `0 ≤ lo < hi ≤ 1` is required. Omitting normalization gives `lo=0`, `hi=1`. `clip` defaults to true; setting it false allows normalized scores below 0 or above 100. All metrics are treated as higher-is-better. -Under the original aggregate, one variant's weighted contribution is its normalized score multiplied by the effective category weight, divided by the number of available evals in that category and by the number of shared selected variants for that eval. Filtering does not change those denominators. Weighted A−B contributions sum to the composite difference over shared coverage. The English-balance modes redistribute contributions as described in [Choosing an aggregate](#choosing-an-aggregate). +For evals without component rules, under the original aggregate, one variant's weighted contribution is its normalized score multiplied by the effective category weight, divided by the number of available evals in that category and by the number of shared selected variants for that eval. Filtering does not change those denominators. Weighted A−B contributions sum to the composite difference over shared coverage. The English-balance modes redistribute contributions as described in [Choosing an aggregate](#choosing-an-aggregate). + +## Weighted components within an eval + +Use `aggregation` when several task results form one eval score and the exporter does not supply the intended summary. It belongs in the global catalogue. Category weights and English shares remain in the weighting profile; expected task coverage remains in the eval set. + +For PolyMath, add this block to its eval entry (an excerpt, not a complete catalogue): + +```yaml +aggregation: + components: + - name: low + match: {regex: 'polymath_.+_low'} + relative_weight: 1 + - name: medium + match: {regex: 'polymath_.+_medium'} + relative_weight: 2 + - name: high + match: {regex: 'polymath_.+_high'} + relative_weight: 4 + - name: top + match: {regex: 'polymath_.+_top'} + relative_weight: 8 + note: Difficulty-weighted accuracy; each level is required. + sources: + - https://qwen-polymath.github.io/#benchmark-score +``` + +Each component requires a unique nonempty `name`, a full-task `match` (exact `name` or `regex`, as for eval matching), and a positive finite numeric `relative_weight`. The weight sum must be finite. The list must be nonempty. Optional `note` is text and `sources` is a list of HTTP(S) URLs. Unknown fields are rejected. Multiplying all component weights by the same positive constant leaves the result unchanged. + +The supplied PolyMath rule implements the authors' [Difficulty-Weighted Accuracy](https://qwen-polymath.github.io/#benchmark-score): `(low + 2×medium + 4×high + 8×top)/15`. The [oellm-eval template](https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/polymath/_default_template_yaml) emits a mean accuracy for each difficulty split; those input values are not already difficulty-weighted. + +Calculation order: + +1. Select the configured metric/filter/shot results and normalize each score. +2. Within each model, eval, explicit language assignment, and protocol, require exactly one result for every component. Protocol means metric, filter, shot count, harness, and backend. Translation uses the full source/target pair; known pooled languages use their explicit pooled assignment. Unknown languages cannot form component groups. +3. Calculate `sum(relative_weight × normalized score) / sum(relative_weights)` for each complete group. +4. Average complete groups equally within the eval, or within its English/other side when balancing is enabled. Apply the selected eval/category aggregation and category weights afterward. Evals without component rules retain their ordinary variant means. + +For example, fictional component scores of 60, 30, 15, and 0 produce `(60 + 60 + 60 + 0)/15 = 12`. Their contributions to that language/protocol score are 4, 4, 4, and 0 points. Each component's contribution to the full composite also includes its group's share within the eval, any English balance, the eval's share of its category, and the category weight. These full contributions drive the weighted delta bars and sum to the score difference. + +**Incompatible aggregation configurations are errors.** Components inherit the parent eval's metric, filter, score scale, normalization, and any fixed shot setting; they cannot override these fields. Catalogue validation checks declared task/language assignments against the component rules and eval selection. Each represented language must have every component available in the configuration, and each task must match exactly one component. An exact component task must be eligible under its parent eval. Entire languages can be omitted; alternate task aliases are permitted in the catalogue. + +A named set that lists component tasks must select exactly one task for each component at every chosen language/shot setting. For example, selecting low/medium/high at 0-shot and top at 5-shot is a configuration error, as is omitting top entirely. Omitting shots for every component is allowed; mixing unrestricted and fixed shots is rejected unless the parent eval pins the same shot count. Multiple complete shot settings are allowed. An eval-only requirement or Any available leaves task selection to the catalogue. Missing task-language assignments or duplicate component selections in a named set are errors. + +These checks run before applying a browser config or replacing build output. Rejected imports preserve active models, settings, and scores. Regex compatibility is checked against declared task names, plus newly observed selected tasks during CSV classification; the validator does not attempt to prove arbitrary regex relationships. An observed selected task matching zero or multiple component rules is also an error. + +**Missing result data still warns and excludes groups, never renormalizing over the remaining levels.** A valid selection with missing metrics/components, multiple exported task results for a component, unknown languages in newly encountered results, or incomplete scoring protocols produces warnings. Check completeness per model and again after taking the exact A/B measurement intersection; a missing component on either side excludes the entire corresponding group from both calculations. Distinct protocols cannot supply each other's missing components. Named-set completion counts reflect these exclusions. Freeform mode does not require entirely absent languages, but it does require every component for each represented group. Raw data remains in Eval configuration, including excluded results and the reason they are unused. + +In **Categories** and **Languages**, a collapsed component eval/group shows the calculated normalized score. Its leaves show individual raw scores, relative weights, effective weight percentages, and contributions to the language/protocol group. Ordinary evals still show raw averages; mixed category summaries average the displayed eval scores and are descriptive, not the full composite. These breakdowns do not apply the chosen English balance; use Weighted score for that calculation. Inspection filters that hide required components leave the affected calculated summary unavailable (`—`), rather than inventing a partial benchmark score. Leaf contributions retain the full group's weights. + +**Delta comparisons** keeps raw differences as raw differences, including simple raw averages on grouped rows. Its weighted contribution differences include component weights. **Eval configuration** exposes the component matching rules, weights, formula, and sources; catalogue YAML import/export preserves them. Group contributions are explicitly separate from contributions to the overall composite. + +The builder and browser share the aggregation implementation. Build summaries apply completeness per model; the browser additionally enforces shared A/B coverage. `analysis.json` model summaries record component warnings alongside scores. Row audits retain metric eligibility even when component coverage excludes a result from the final calculation. ## Initial chance baselines @@ -177,7 +234,7 @@ ARC Challenge uses an approximate 25% baseline. In its published test split, 1,1 AMC23 is open-ended in the selected evaluator: the original contest's answer options are removed. Code generation, translation chrF, overlap F1, and other open-ended exact-match tasks do not receive an invented chance baseline. -`normalize.basis` may be `uniform_choice`, `uniform_integer`, `not_applicable`, or `unresolved`. It documents the rationale; `min`, `max`, and `clip` control the actual calculation. Optional `sources` is a list of HTTP(S) URLs, and `note` is free text. Set `min: 0` and `max: 1` to disable correction. An optional eval-level `warning` string appears in the Warnings tab once for all models and in the eval configuration details. Remove it when the concern is resolved; it does not change selection or arithmetic. The supplied config uses it for translation calibration and the Croatian/Serbian language-grouping approximation. Exported YAML preserves config values and notes; YAML comments are not retained. +`normalize.basis` may be `uniform_choice`, `uniform_integer`, `not_applicable`, or `unresolved`. It documents the rationale; `min`, `max`, and `clip` control the actual calculation. Optional `sources` is a list of HTTP(S) URLs, and `note` is free text. Set `min: 0` and `max: 1` to disable correction. An optional eval-level `warning` string appears in the Warnings tab once for all models and in the eval configuration details. Remove it when the concern is resolved; it does not change selection or arithmetic. The supplied config uses it for unvalidated prompted Global PIQA scoring and the Croatian/Serbian language-grouping approximation. Exported YAML preserves config values and notes; YAML comments are not retained. `acc_norm` in lm-eval refers to choosing answers using length-normalized likelihoods; it does **not** remove chance accuracy. Chance correction here is applied to each selected variant's aggregate score before averaging evals. Clipping after aggregation is not equivalent to clipping individual items, and a mixture of corrected and uncorrected metrics is still a provisional composite. @@ -246,7 +303,7 @@ The prominent **Score calculation** panel offers three modes: | `english_eval` | English balance per eval | Combine English and other-language means within each eval, then average the eval scores equally. | | `english_category` | English balance per category | On each language side, average variants within evals and then represented evals equally; combine the two category means. | -All modes apply the configured category weights last. The selector affects both model score cards, category/eval contributions, effective weights, and weighted delta bars. Raw comparison columns and descriptive language/category breakdowns retain their meanings. +For evals with component rules, combine complete components first and use complete language/protocol groups in place of variants in the table above. All modes apply the configured category weights last. The selector affects both model score cards, category/eval contributions, effective weights, and weighted delta bars. Raw comparison columns and descriptive language/category breakdowns retain their meanings. Optional top-level fields in the **weighting profile** persist the choice and shares: @@ -299,15 +356,17 @@ Builds and browser imports use the same CSV parser. Required columns are `checkp | Empty CSV, missing columns, invalid identity fields, malformed quotes, inconsistent row widths | Reject the file with an error. | | Invalid YAML/schema, ambiguous eval matches, duplicate selected measurements within a model | Reject the config or file; never silently pick a rule or duplicate. | | Selected score is blank, nonnumeric, nonfinite, or outside `0..score.scale` | Reject with CSV row, model, task, and metric in the error. Decimal and scientific notation are accepted; booleans and hexadecimal values are not scores. | -| New model uses an already loaded checkpoint name or the reserved synthetic-demo name | Reject the import. Give the model a distinct checkpoint label. | +| New model uses an already loaded checkpoint name or a name starting with the reserved `SYNTHETIC demo — ` prefix | Reject the import. Give the model a distinct checkpoint label. | | Task has no eval config, or lacks the configured metric/filter/shots | Warn and exclude from scoring. Alternate metrics remain inspectable and never silently substitute for the configured metric. | | Measurements match only one of the compared models | Warn; use only shared measurements and redistribute weights. | | Named set requirement is missing from either or both models | Warn and mark the set incomplete; compare the shared subset. | | Eval/task data is not selected by a named set or freeform exclusion | Show **Not used** and exclude it, even if only an alternate metric exists; keep it in the audit. | | Catalogue rule has no results in either model | No warning unless required by the selected named set. | | Shared category has no profile weight | Warn; zero contribution until a weight is assigned. | -| Selected task has no explicit language assignment | Warn and retain the score. Language views show Unknown; English-balance modes use the English fallback. Known mixed-language pools use the documented fallback without claiming a resolved single language. | -| Selected variants use inconsistent scoring settings | Warn and retain scores so the reviewer can inspect the protocols. | +| Selected task has no explicit language assignment | For component evals, warn and exclude the group. Otherwise warn and retain the score. Language views show Unknown; English-balance modes use the English fallback. Known mixed-language pools use the documented fallback without claiming a resolved single language. | +| Selected variants use inconsistent scoring settings | Warn; component evals require completeness independently within each protocol. Ordinary evals retain scores for review. | +| Aggregation rules or named-set component selections are incompatible | Reject the config; preserve the active dashboard or previous build output. Components share parent scoring settings and must be selected completely at compatible shot settings. | +| Valid component selection has missing data or multiple exported results for a component | Warn and exclude the whole language/protocol group from both scores. Keep raw data inspectable; never average only the remaining components. | | Matched A/B measurements report different positive `n_samples` | Warn and retain scores; sample count does not determine score weights. Review whether dataset coverage is comparable. | | Supplied `n_samples` is not a positive integer | Warn and retain scores; omit it from sample-count comparisons. Absent/blank sample counts are allowed. | | Eval has a YAML `warning` | Show the caveat in Warnings when the eval has shared comparison data. Always show it in its configuration details. Excluded evals get **Not used**, not an active scoring caveat. | diff --git a/docs/development.md b/docs/development.md index a4684b9..3a014a1 100644 --- a/docs/development.md +++ b/docs/development.md @@ -1,23 +1,24 @@ # Develop and publish Quickdash -Run commands from the repository root. Python 3.8+ and Node.js 18+ build the dashboard without installing packages; browser tests require Node.js 22+ and Chrome. The scoring config format is documented in the [configuration reference](configuration.md). +Run commands from the repository root. Install the Python package with `python -m pip install -e .` in a virtual environment (Python 3.8+). Building and using the Python library need no Node runtime. Cross-language tests require Node.js 18+; browser tests require Node.js 22+ and Chrome. Activate the environment before running browser tests so their builder subprocess uses the installed dependencies. The scoring config format is documented in the [configuration reference](configuration.md). ## Source layout | Directory | Contents | | --- | --- | -| `app/` | Python builder, browser application, HTML template, and bundled YAML parser. | +| `quickdash/` | Native Python interpretation, analysis, diagnostics, and CLI. | +| `app/` | Python builder, DOM-independent JavaScript engine (`analysis.js`), browser renderer (`app.js`), HTML, and bundled YAML parser. | | `tests/` | Public contract tests, browser checks, and optional private-export regressions. | | `configs/` | Global catalogue, weighting profiles, optional named sets, and fictional examples. | | `results/` | Public CSV exports contributed to the shared dashboard. | -| `examples/` | Fictional input data for the demo and tests. | +| `examples/` | Public sample export for Pages and parity tests, plus small fictional quickstart data. | | `docs/` | Configuration and contributor documentation. | | `data/`, `output/` | Ignored local inputs and generated files. | ## Build and inspect a change ```sh -python3 -m app.build --results-dir results --output output/shared +python3 -m app.build --results-dir results --sample-csv examples/sample-evals.csv --output output/shared python3 -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \ --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml --output output/demo ``` @@ -31,10 +32,12 @@ Check the views affected by your change, including their warnings and failed-inp These tests use small fixtures and run from a fresh checkout without private evaluation data: ```sh -python3 -m unittest tests.test_data -node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs +python -m unittest tests.test_data tests.test_engines +node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs tests/test_components.cjs ``` +The shared suite in `tests/test_engines.py` sends the same input cases to native Python and the real browser engine through `tests/engine_adapter.cjs`. Both must satisfy independently specified expectations, then their complete semantic reports are compared. Diagnostic codes, contexts and actions are checked; presentation text is not. Numerical comparison uses absolute tolerance `1e-9` and relative tolerance `1e-12`. Fixtures vary configuration sizes, contents and ordering. Python tests also exercise warning emission and CLI stdout/stderr without Node on PATH. + They cover input validation, normalization, warning/exclusion behavior, failed-build preservation, Python/JavaScript parity, hierarchy sorting, and deterministic randomized scoring comparisons against an independent calculation. The public browser suite also checks empty startup, shared models, independent profile/set selection, missing requirements, temporary uploads, rollback, and that file imports make no network requests. It requires Node.js 22+ and Chrome. Start an isolated browser session, then run the suite in another terminal: @@ -58,7 +61,7 @@ The full-export regression tests require the original private CSV at `data/v2zlo ```sh python3 -m app.build data/v2zloss_86k.flag-evals-436.tasks.csv python3 -m unittest tests.test_analysis tests.test_data -node --test tests/test_app.cjs tests/test_english.cjs tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs +node --test tests/test_app.cjs tests/test_english.cjs tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs tests/test_components.cjs ``` With the isolated Chrome session above running, use `node tests/test_browser.mjs` for the full-export browser checks: filtering, sortable hierarchies, scroll preservation, warnings, model swapping, header alignment, and mobile layouts. Screenshots go into the ignored `output/` directory. @@ -72,9 +75,20 @@ node --test --experimental-test-coverage --test-coverage-include=app/eval_config +## Compare a dashboard refactor against a baseline + +Preserve the revision being reviewed in a temporary checkout or directory. Build an example or private dashboard with the candidate revision, then compare the calculation paths: + +```sh +QUICKDASH_BASELINE=/path/to/baseline-checkout \ + node tests/compare_baseline.cjs output/example/analysis.json +``` + +This optional check compares included rows, weights, contributions, scores and descriptive trees across profiles, sets, all aggregation modes, and missing coverage. Run the browser suites against both builds as well; arithmetic agreement alone does not establish UI behavior. Record the comparison and any intentional fixes in the PR. Keep private inputs and generated evidence outside tracked files. + ## Publish through GitHub Pages -The [workflow](../.github/workflows/pages.yml) runs public tests on pull requests and pushes to `main`. After tests pass, it builds the shared dashboard and a separate fictional demo. Only `output/site/` is uploaded as the Pages artifact: `index.html`, `demo.html`, and license files. The repository root and private local output are not published as the site. +The [workflow](../.github/workflows/pages.yml) runs public tests on pull requests and pushes to `main`. After tests pass, it builds the shared dashboard and a separate fictional demo. When `results/` has no CSVs, the shared page embeds `examples/sample-evals.csv`; real shared CSVs take precedence. The sample also runs through both engines in CI for all shipped weighting profiles, eval sets, and aggregation modes, with complete and mismatched coverage. See [contributor requirements](../AGENTS.md). Only `output/site/` is uploaded as the Pages artifact: `index.html`, `demo.html`, and license files. The repository root and private local output are not published as the site. In repository **Settings → Pages**, select **GitHub Actions** as the source. Publishing uses the generated artifact rather than a checked-in root or `docs/` folder. A successful push to `main` deploys automatically; a failed build leaves the last successful site available. Review build or deployment failures in the repository’s **Actions** tab. diff --git a/docs/python-api.md b/docs/python-api.md new file mode 100644 index 0000000..ce55b4d --- /dev/null +++ b/docs/python-api.md @@ -0,0 +1,123 @@ +# Calculate and explain evaluation scores in Python + +Use the Python library when you want to analyze CSV exports in a notebook, render a score tree, or automate a comparison. It uses the same inputs and scoring rules as the dashboard. No browser, Node.js, pandas, or network access is needed at runtime. + +## Install and calculate a score + +From a checkout of Quickdash, create an environment and install the package: + +```sh +python3 -m venv .venv +source .venv/bin/activate +python -m pip install -e . +``` + +The package requires Python 3.8+ and PyYAML 6. The installable distribution is called `oellm-quickdash`; import it as `quickdash`. Node.js is needed only for development tests that exercise the JavaScript implementation. + +Start with the fictional example included in the repository: + +```python +from quickdash import analyze, compare, load_config, read_results + +config = load_config( + catalogue="configs/examples/catalogue.yaml", + weights="configs/examples/weights.yaml", + eval_set="configs/sets/any-available.yaml", +) +results = read_results("examples/scores.csv") +report = analyze(results, config) + +for model in report.models: + print(model["model"], model["score"]) + tree = model["tree"] # Calculated nodes; rendering requires no scoring logic. + +comparison = compare(results, config, a="Example A", b="Example B") +print(comparison.delta) # -2.5 score points +``` + +`load_config()` accepts paths or configuration dictionaries, validates their combination, and returns a bundle containing `catalogue`, `profile`, and `suite`. It copies supplied dictionaries. Change configurations by loading a new bundle. The analysis functions do not mutate results or configuration inputs. + +Omitting `eval_set` means all recognized evals are eligible. To apply the project's explicit exclusions, pass its `configs/sets/any-available.yaml` file, as above. No eval names, language lists, category counts, or component counts are hardcoded into the library. + +`read_results()` accepts a path or a list of paths. It preserves CSV fields as strings and rejects model names repeated across files. Put all measurements for a model in one CSV, or deliberately combine row dictionaries before analysis. The analysis functions also accept lists of row dictionaries directly, with the same required fields and validation as CSV imports. + +## Independent scores and fair comparisons + +`analyze(results, config)` scores each model independently on its available, selected measurements. Its top-level fields are `models`, `diagnostics`, and `coverage`. + +`compare(results, config, a=..., b=...)` first restricts both models to matching valid measurements. It excludes incomplete component groups, including groups made incomplete by intersecting coverage, before redistributing weights. Its fields are `a`, `b`, `delta`, `deltas`, `diagnostics`, and `coverage`. `a` and `b` have the same model-result shape used by `analyze()`. + +Do not subtract independent scores to reproduce an A/B comparison: the models may have different coverage. A missing comparison model raises `ValueError`; comparing a model with itself is permitted. + +Both operations support original averaging, English balance per eval, and English balance per category, selected by the weighting profile. The [scoring reference](configuration.md#choosing-an-aggregate) describes their differences. Scores and contributions use score points on a **0–100 scale**; weights are fractions. Explicit `clip: false` configurations can produce normalized scores outside 0–100. + +## Render the calculated tree + +Each model result has `model`, `score`, `tree`, and `measurements`. Traverse `tree.children` recursively using dictionary access. Each node contains: + +| Field | Meaning | +| --- | --- | +| `id`, `kind`, `label` | Stable identity within the model's tree, node kind, and display label. | +| `score` | Calculated score, or `null` when unavailable. | +| `weight` | Effective weight relative to the parent. | +| `effective_weight` | Share of the final model score. | +| `contribution` | Contribution in final score points. | +| `relative_weight` | Configured component weight, otherwise `null`. | +| `measurement_id` | Source measurement identity for a leaf, otherwise `null`. | +| `children` | Child nodes in calculation order. | + +The tree follows the actual aggregation: English groups appear within evals or within categories according to the chosen mode. Component evals include language and protocol groups before component leaves. Translation groups use source/target pairs once; they are not duplicated into both language directions as in the dashboard's descriptive language view. + +Child contributions sum to the parent's contribution. For nodes with positive effective weight, child weights sum to one and reproduce the parent score. Zero-weight branches contribute zero; their internal aggregate score can be unavailable even though individual leaf measurements retain scores. An unavailable overall score is `null`, never a manufactured zero. + +`measurements` retains every input row, including alternate metrics and excluded data. Added fields include `id`, `language`, `included`, `exclusion`, `effective_weight`, and `contribution`, alongside interpretation fields such as `raw_score_100`, `score_100`, and `decision`. `exclusion` is `interpretation`, `eval_set`, `coverage`, or `null`; diagnostics explain coverage failures. Measurement IDs encode the model, task, metric, filter, shots, harness, and backend. Treat IDs as opaque strings. + +`comparison.deltas` contains raw and normalized differences, effective weights, and contribution differences linked to both source measurement IDs. These contribution differences sum to the overall delta. Inspection filters should filter the returned rows without recalculating their weights. + +Reports are dictionaries with attribute access for top-level fields. Nested records are ordinary dictionaries and lists. Use `json.dumps(report, allow_nan=False)` to serialize one; no custom encoder is needed. Renderers can format or reorder nodes without reconstructing the calculation. + +## Surface warnings + +By default, `analyze()` and `compare()` emit a `QuickdashWarning` through Python's standard `warnings` mechanism for each grouped diagnostic. Warnings normally appear on stderr. Each warning object has a `.diagnostic` attribute containing its structured record. The same diagnostics are retained in `report.diagnostics`, separately from the tree. + +Applications that display diagnostics themselves can explicitly collect them: + +```python +report = analyze(results, config, diagnostics="collect") +for diagnostic in report.diagnostics: + print(diagnostic["code"], diagnostic["model"], diagnostic["tasks"]) +``` + +For strict automation, use `diagnostics="error"`. If any diagnostic occurs, it raises `DiagnosticError`, whose `.diagnostics` contains all warnings; it returns no report. Invalid configurations, ambiguous selected matches, duplicate selected measurements, and invalid selected scores raise `ValueError` regardless of diagnostic policy. File access failures raise the usual `OSError` subclasses. + +The diagnostic contract is `code`, `model`, `eval`, `tasks`, `measurement_ids`, and `effect`. `effect` describes the warning's policy (`excluded`, `included`, or `zero_weight`); the measurement's `included` field is authoritative when several conditions apply. Config-wide caveats have no individual measurement IDs. `type`, `name`, `detail`, and `variants` support dashboard presentation and are not wording contracts. + +| Codes | Condition and action | +| --- | --- | +| `config_caveat` | An included eval has a configured warning. | +| `no_config`, `not_used` | Unknown eval or data outside the selected set; excluded. | +| `missing_scoring_field`, `missing_scoring_setting`, `no_selected_score` | Required metric/protocol unavailable; no alternate substitution. | +| `missing_suite_data` | A named set has missing requirements; available shared results are reweighted. | +| `incomplete_components` | A language/protocol group is incomplete or incompatible; the whole group is excluded. | +| `comparison_coverage` | Measurements cannot participate in shared coverage; excluded from both scores. | +| `unknown_language` | Ordinary evals use the English weighting fallback; component groups require explicit language assignments. | +| `inconsistent_scoring_settings` | Selected variants use different protocols; ordinary results or complete component groups remain eligible. | +| `invalid_sample_count`, `sample_count_mismatch` | Sample-count metadata needs review; it does not determine weights. | +| `no_category_weight` | An included category has no profile weight and contributes zero. | + +Unused catalogue entries do not warn merely because no data exists for them. Declare an expected eval set when absence should warn. Alternate metrics remain auditable without creating warnings when the selected metric is present. + +## Command line + +Print calculated trees for the fictional models: + +```sh +quickdash examples/scores.csv \ + --catalogue configs/examples/catalogue.yaml \ + --weights configs/examples/weights.yaml \ + --eval-set configs/sets/any-available.yaml +``` + +Add `--compare 'Example A' 'Example B'` for shared-coverage scores, or `--format json` for the complete machine-readable report. `python -m quickdash` provides the same command. Warnings go to stderr; stdout contains only the selected result format. + +Exit status is 0 for a successful calculation, including recoverable warnings. `--strict` prints warnings and exits 1 without writing a result. Invalid input exits 2. A score can be unavailable when no valid weighted data remains; inspect diagnostics and coverage or use strict mode when warnings must block a workflow. diff --git a/examples/README.md b/examples/README.md new file mode 100644 index 0000000..d40211a --- /dev/null +++ b/examples/README.md @@ -0,0 +1,28 @@ +# Example results + +`sample-evals.csv` contains the evaluation scores from the `v2zloss_86k` export +used to develop Quickdash. The checkpoint is labelled **SAMPLE — v2zloss_86k**. +Scores, task names, scoring settings, sample counts, and reported standard errors +are preserved. Original filesystem paths, result timestamps, and file-count +metadata are omitted. This is sample data for exploring the dashboard, not a +claim about a current model or a comparison of training methods. + +The main Pages dashboard embeds this file only while `results/` has no CSVs. +Adding a shared result replaces the fallback on the next successful deployment. +To reproduce the Pages build: + +```sh +python -m app.build --results-dir results --sample-csv examples/sample-evals.csv --output output/shared +``` + +The browser offers three **SYNTHETIC demo** choices derived from the first loaded +model. Each uses deterministic Gaussian noise with a standard deviation of +2 raw percentage points (seed `20260930`). The higher/lower choices also shift +scores by +3/−3 points. Values are clipped to 0–100 and converted to each eval's +configured source scale before normalization. These are visual examples, not +additional model runs. They preserve the source's selected measurement coverage; +reported errors and result timestamps are cleared. + +`scores.csv` is a separate, small fictional dataset for the Python quickstart +and the minimal `demo.html` page. It uses `configs/examples/` rather than the +full eval catalogue. diff --git a/examples/sample-evals.csv b/examples/sample-evals.csv new file mode 100644 index 0000000..3c6cd5f --- /dev/null +++ b/examples/sample-evals.csv @@ -0,0 +1,2125 @@ +checkpoint,task,n_shot,harness,backend,is_group,metric,filter,value,stderr,n_samples +SAMPLE — v2zloss_86k,AIME24,0,Evalchemy,vllm,,accuracy_avg,,0.10666666666666666,0.004216370213557838,30 +SAMPLE — v2zloss_86k,AIME24,0,Evalchemy,vllm,,solved_avg,,3.2,,30 +SAMPLE — v2zloss_86k,AIME25,0,Evalchemy,vllm,,accuracy_avg,,0.1,0.0,30 +SAMPLE — v2zloss_86k,AIME25,0,Evalchemy,vllm,,solved_avg,,3.0,,30 +SAMPLE — v2zloss_86k,AMC23,0,Evalchemy,vllm,,accuracy_avg,,0.31500000000000006,0.00387298334620742,40 +SAMPLE — v2zloss_86k,AMC23,0,Evalchemy,vllm,,solved_avg,,12.6,,40 +SAMPLE — v2zloss_86k,GPQADiamond,0,Evalchemy,vllm,,accuracy_avg,,0.2946127946127946,0.008361203381454177,198 +SAMPLE — v2zloss_86k,GPQADiamond,0,Evalchemy,vllm,,solved_avg,,58.333333333333336,,198 +SAMPLE — v2zloss_86k,HumanEval,0,Evalchemy,vllm,,python_pass@1,,0.17073170731707318,, +SAMPLE — v2zloss_86k,HumanEval,0,Evalchemy,vllm,,sh_pass@1,,0.012658227848101266,, +SAMPLE — v2zloss_86k,JEEBench,0,Evalchemy,vllm,,accuracy_avg,,0.14012944983818773,0.0001321191878523813,515 +SAMPLE — v2zloss_86k,JEEBench,0,Evalchemy,vllm,,solved_avg,,72.16666666666667,,515 +SAMPLE — v2zloss_86k,LiveCodeBench,0,Evalchemy,vllm,,accuracy_avg,,0.16536203522504891,0.0006684247074989953,511 +SAMPLE — v2zloss_86k,LiveCodeBench,0,Evalchemy,vllm,,accuracy_easy_avg,,0.3974358974358974,0.0011583434652631503,511 +SAMPLE — v2zloss_86k,LiveCodeBench,0,Evalchemy,vllm,,accuracy_hard_avg,,0.006775067750677508,0.0013550135501355014,511 +SAMPLE — v2zloss_86k,LiveCodeBench,0,Evalchemy,vllm,,accuracy_medium_avg,,0.055016181229773455,0.0016181229773462793,511 +SAMPLE — v2zloss_86k,LiveCodeBench,0,Evalchemy,vllm,,solved_avg,,84.5,,511 +SAMPLE — v2zloss_86k,MATH500,0,Evalchemy,vllm,,accuracy,,0.444,,500 +SAMPLE — v2zloss_86k,MATH500,0,Evalchemy,vllm,,num_solved,,222,,500 +SAMPLE — v2zloss_86k,agieval_lsat_ar,3,lm-eval,vllm,,acc,none,0.2565217391304348,0.028858814315305684,230 +SAMPLE — v2zloss_86k,agieval_lsat_ar,3,lm-eval,vllm,,acc_norm,none,0.23043478260869565,0.02782780752227618,230 +SAMPLE — v2zloss_86k,arc_challenge,10,lm-eval,vllm,,acc,none,0.5955631399317406,0.014342036483436283,1172 +SAMPLE — v2zloss_86k,arc_challenge,10,lm-eval,vllm,,acc_norm,none,0.6339590443686007,0.014077223108470144,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_bg,0,lm-eval,vllm,,acc,none,0.4035836177474403,0.014337158914268453,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_bg,0,lm-eval,vllm,,acc_norm,none,0.4539249146757679,0.014549221105171829,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_cs,0,lm-eval,vllm,,acc,none,0.40017064846416384,0.014317197787809148,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_cs,0,lm-eval,vllm,,acc_norm,none,0.447098976109215,0.01452938016052677,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_da,0,lm-eval,vllm,,acc,none,0.42662116040955633,0.014453185592920293,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_da,0,lm-eval,vllm,,acc_norm,none,0.46245733788395904,0.014570144495075503,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_de,0,lm-eval,vllm,,acc,none,0.40102389078498296,0.014322255790719744,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_de,0,lm-eval,vllm,,acc_norm,none,0.45307167235494883,0.014546892052005678,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_el,0,lm-eval,vllm,,acc,none,0.42662116040955633,0.014453185592920293,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_el,0,lm-eval,vllm,,acc_norm,none,0.4564846416382253,0.014555949760496541,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_es,0,lm-eval,vllm,,acc,none,0.46245733788395904,0.014570144495075503,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_es,0,lm-eval,vllm,,acc_norm,none,0.4709897610921502,0.014586776355294413,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_et,0,lm-eval,vllm,,acc,none,0.363481228668942,0.014056207319068465,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_et,0,lm-eval,vllm,,acc_norm,none,0.4069965870307167,0.014356399418009202,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_fi,0,lm-eval,vllm,,acc,none,0.38054607508532423,0.014188277712349937,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_fi,0,lm-eval,vllm,,acc_norm,none,0.4189419795221843,0.01441810695363902,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_fr,0,lm-eval,vllm,,acc,none,0.4300341296928328,0.014467631559137991,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_fr,0,lm-eval,vllm,,acc_norm,none,0.4906143344709898,0.014608816322065041,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_hu,0,lm-eval,vllm,,acc,none,0.38310580204778155,0.01420647266167275,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_hu,0,lm-eval,vllm,,acc_norm,none,0.44112627986348124,0.014509747749064826,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_is,0,lm-eval,vllm,,acc,none,0.3703071672354949,0.014111298751675015,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_is,0,lm-eval,vllm,,acc_norm,none,0.41467576791808874,0.014397070564409184,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_it,0,lm-eval,vllm,,acc,none,0.45563139931740615,0.014553749939306847,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_it,0,lm-eval,vllm,,acc_norm,none,0.47696245733788395,0.014595873205358168,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_lt,0,lm-eval,vllm,,acc,none,0.3916382252559727,0.014264122124938267,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_lt,0,lm-eval,vllm,,acc_norm,none,0.4112627986348123,0.014379441068522219,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_lv,0,lm-eval,vllm,,acc,none,0.3677474402730375,0.01409099561816849,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_lv,0,lm-eval,vllm,,acc_norm,none,0.4121160409556314,0.014383915302225433,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_nb,0,lm-eval,vllm,,acc,none,0.42662116040955633,0.014453185592920293,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_nb,0,lm-eval,vllm,,acc_norm,none,0.48293515358361777,0.014602878388536585,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_nl,0,lm-eval,vllm,,acc,none,0.4274744027303754,0.014456862944650666,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_nl,0,lm-eval,vllm,,acc_norm,none,0.4667235494880546,0.0145789958596058,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_pl,0,lm-eval,vllm,,acc,none,0.41638225255972694,0.014405618279436053,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_pl,0,lm-eval,vllm,,acc_norm,none,0.45563139931740615,0.014553749939306847,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_pt,0,lm-eval,vllm,,acc,none,0.43430034129692835,0.014484703048857371,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_pt,0,lm-eval,vllm,,acc_norm,none,0.4684300341296928,0.014582236460866985,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_ro,0,lm-eval,vllm,,acc,none,0.4206484641638225,0.014426211252508337,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_ro,0,lm-eval,vllm,,acc_norm,none,0.44368600682593856,0.014518421825670504,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_sk,0,lm-eval,vllm,,acc,none,0.4104095563139932,0.01437492219264264,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_sk,0,lm-eval,vllm,,acc_norm,none,0.447098976109215,0.01452938016052677,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_sl,0,lm-eval,vllm,,acc,none,0.39078498293515357,0.014258563880513803,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_sl,0,lm-eval,vllm,,acc_norm,none,0.43856655290102387,0.014500682618212822,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_sv,0,lm-eval,vllm,,acc,none,0.4197952218430034,0.014422181226303012,1172 +SAMPLE — v2zloss_86k,arc_challenge_mt_sv,0,lm-eval,vllm,,acc_norm,none,0.4709897610921502,0.014586776355294413,1172 +SAMPLE — v2zloss_86k,arc_easy,10,lm-eval,vllm,,acc,none,0.8602693602693603,0.007114284756773912,2376 +SAMPLE — v2zloss_86k,arc_easy,10,lm-eval,vllm,,acc_norm,none,0.8661616161616161,0.006986473271895545,2376 +SAMPLE — v2zloss_86k,belebele_bul_Cyrl,5,lm-eval,vllm,,acc,none,0.8511111111111112,0.011872561521396034,900 +SAMPLE — v2zloss_86k,belebele_bul_Cyrl,5,lm-eval,vllm,,acc_norm,none,0.8511111111111112,0.011872561521396034,900 +SAMPLE — v2zloss_86k,belebele_ces_Latn,5,lm-eval,vllm,,acc,none,0.8488888888888889,0.011945209697456726,900 +SAMPLE — v2zloss_86k,belebele_ces_Latn,5,lm-eval,vllm,,acc_norm,none,0.8488888888888889,0.011945209697456726,900 +SAMPLE — v2zloss_86k,belebele_dan_Latn,5,lm-eval,vllm,,acc,none,0.8566666666666667,0.011686909711653044,900 +SAMPLE — v2zloss_86k,belebele_dan_Latn,5,lm-eval,vllm,,acc_norm,none,0.8566666666666667,0.011686909711653044,900 +SAMPLE — v2zloss_86k,belebele_deu_Latn,5,lm-eval,vllm,,acc,none,0.8622222222222222,0.011495274539524399,900 +SAMPLE — v2zloss_86k,belebele_deu_Latn,5,lm-eval,vllm,,acc_norm,none,0.8622222222222222,0.011495274539524399,900 +SAMPLE — v2zloss_86k,belebele_ell_Grek,5,lm-eval,vllm,,acc,none,0.8455555555555555,0.012052506467621092,900 +SAMPLE — v2zloss_86k,belebele_ell_Grek,5,lm-eval,vllm,,acc_norm,none,0.8455555555555555,0.012052506467621092,900 +SAMPLE — v2zloss_86k,belebele_eng_Latn,5,lm-eval,vllm,,acc,none,0.8688888888888889,0.011256983379669072,900 +SAMPLE — v2zloss_86k,belebele_eng_Latn,5,lm-eval,vllm,,acc_norm,none,0.8688888888888889,0.011256983379669072,900 +SAMPLE — v2zloss_86k,belebele_est_Latn,5,lm-eval,vllm,,acc,none,0.8244444444444444,0.012688437383538158,900 +SAMPLE — v2zloss_86k,belebele_est_Latn,5,lm-eval,vllm,,acc_norm,none,0.8244444444444444,0.012688437383538158,900 +SAMPLE — v2zloss_86k,belebele_fin_Latn,5,lm-eval,vllm,,acc,none,0.8333333333333334,0.012429507075907847,900 +SAMPLE — v2zloss_86k,belebele_fin_Latn,5,lm-eval,vllm,,acc_norm,none,0.8333333333333334,0.012429507075907847,900 +SAMPLE — v2zloss_86k,belebele_fra_Latn,5,lm-eval,vllm,,acc,none,0.8766666666666667,0.010966742231624015,900 +SAMPLE — v2zloss_86k,belebele_fra_Latn,5,lm-eval,vllm,,acc_norm,none,0.8766666666666667,0.010966742231624015,900 +SAMPLE — v2zloss_86k,belebele_hrv_Latn,5,lm-eval,vllm,,acc,none,0.8466666666666667,0.012016961604722237,900 +SAMPLE — v2zloss_86k,belebele_hrv_Latn,5,lm-eval,vllm,,acc_norm,none,0.8466666666666667,0.012016961604722237,900 +SAMPLE — v2zloss_86k,belebele_hun_Latn,5,lm-eval,vllm,,acc,none,0.8366666666666667,0.012329168844652528,900 +SAMPLE — v2zloss_86k,belebele_hun_Latn,5,lm-eval,vllm,,acc_norm,none,0.8366666666666667,0.012329168844652528,900 +SAMPLE — v2zloss_86k,belebele_ita_Latn,5,lm-eval,vllm,,acc,none,0.8377777777777777,0.012295317431637264,900 +SAMPLE — v2zloss_86k,belebele_ita_Latn,5,lm-eval,vllm,,acc_norm,none,0.8377777777777777,0.012295317431637264,900 +SAMPLE — v2zloss_86k,belebele_lit_Latn,5,lm-eval,vllm,,acc,none,0.84,0.01222699651698952,900 +SAMPLE — v2zloss_86k,belebele_lit_Latn,5,lm-eval,vllm,,acc_norm,none,0.84,0.01222699651698952,900 +SAMPLE — v2zloss_86k,belebele_lvs_Latn,5,lm-eval,vllm,,acc,none,0.8388888888888889,0.012261260561360088,900 +SAMPLE — v2zloss_86k,belebele_lvs_Latn,5,lm-eval,vllm,,acc_norm,none,0.8388888888888889,0.012261260561360088,900 +SAMPLE — v2zloss_86k,belebele_mlt_Latn,5,lm-eval,vllm,,acc,none,0.8122222222222222,0.013025058588131893,900 +SAMPLE — v2zloss_86k,belebele_mlt_Latn,5,lm-eval,vllm,,acc_norm,none,0.8122222222222222,0.013025058588131893,900 +SAMPLE — v2zloss_86k,belebele_nld_Latn,5,lm-eval,vllm,,acc,none,0.8477777777777777,0.011981196673569677,900 +SAMPLE — v2zloss_86k,belebele_nld_Latn,5,lm-eval,vllm,,acc_norm,none,0.8477777777777777,0.011981196673569677,900 +SAMPLE — v2zloss_86k,belebele_nob_Latn,5,lm-eval,vllm,,acc,none,0.8555555555555555,0.011724509515301441,900 +SAMPLE — v2zloss_86k,belebele_nob_Latn,5,lm-eval,vllm,,acc_norm,none,0.8555555555555555,0.011724509515301441,900 +SAMPLE — v2zloss_86k,belebele_pol_Latn,5,lm-eval,vllm,,acc,none,0.8444444444444444,0.012087833203630684,900 +SAMPLE — v2zloss_86k,belebele_pol_Latn,5,lm-eval,vllm,,acc_norm,none,0.8444444444444444,0.012087833203630684,900 +SAMPLE — v2zloss_86k,belebele_por_Latn,5,lm-eval,vllm,,acc,none,0.8611111111111112,0.011534094424130552,900 +SAMPLE — v2zloss_86k,belebele_por_Latn,5,lm-eval,vllm,,acc_norm,none,0.8611111111111112,0.011534094424130552,900 +SAMPLE — v2zloss_86k,belebele_ron_Latn,5,lm-eval,vllm,,acc,none,0.8533333333333334,0.011799000521176654,900 +SAMPLE — v2zloss_86k,belebele_ron_Latn,5,lm-eval,vllm,,acc_norm,none,0.8533333333333334,0.011799000521176654,900 +SAMPLE — v2zloss_86k,belebele_slk_Latn,5,lm-eval,vllm,,acc,none,0.8411111111111111,0.01219252355189244,900 +SAMPLE — v2zloss_86k,belebele_slk_Latn,5,lm-eval,vllm,,acc_norm,none,0.8411111111111111,0.01219252355189244,900 +SAMPLE — v2zloss_86k,belebele_slv_Latn,5,lm-eval,vllm,,acc,none,0.8355555555555556,0.012362816488132724,900 +SAMPLE — v2zloss_86k,belebele_slv_Latn,5,lm-eval,vllm,,acc_norm,none,0.8355555555555556,0.012362816488132724,900 +SAMPLE — v2zloss_86k,belebele_spa_Latn,5,lm-eval,vllm,,acc,none,0.8633333333333333,0.011456203243543855,900 +SAMPLE — v2zloss_86k,belebele_spa_Latn,5,lm-eval,vllm,,acc_norm,none,0.8633333333333333,0.011456203243543855,900 +SAMPLE — v2zloss_86k,belebele_swe_Latn,5,lm-eval,vllm,,acc,none,0.8611111111111112,0.011534094424130552,900 +SAMPLE — v2zloss_86k,belebele_swe_Latn,5,lm-eval,vllm,,acc_norm,none,0.8611111111111112,0.011534094424130552,900 +SAMPLE — v2zloss_86k,bigbench_cs_algorithms_generate_until,10,lm-eval,vllm,,exact_match,strict-match,0.7537878787878788,0.0118619719270762,1320 +SAMPLE — v2zloss_86k,bigbench_dyck_languages_generate_until,10,lm-eval,vllm,,exact_match,strict-match,0.522,0.015803979428161946,1000 +SAMPLE — v2zloss_86k,bigbench_language_identification_multiple_choice,10,lm-eval,vllm,,acc,none,0.2497,0.004328610017831203,10000 +SAMPLE — v2zloss_86k,bigbench_operators_generate_until,10,lm-eval,vllm,,exact_match,strict-match,0.7619047619047619,0.029461344042368887,210 +SAMPLE — v2zloss_86k,bigbench_qa_wikidata_generate_until,10,lm-eval,vllm,,exact_match,strict-match,0.7010481767629546,0.003211535178044664,20321 +SAMPLE — v2zloss_86k,bigbench_repeat_copy_logic_generate_until,10,lm-eval,vllm,,exact_match,strict-match,0.5,0.08980265101338744,32 +SAMPLE — v2zloss_86k,boolq,10,lm-eval,vllm,,acc,none,0.863914373088685,0.005996999582703271,3270 +SAMPLE — v2zloss_86k,commonsense_qa,10,lm-eval,vllm,,acc,none,0.7862407862407862,0.01173708611212719,1221 +SAMPLE — v2zloss_86k,copa,0,lm-eval,vllm,,acc,none,0.88,0.032659863237109045,100 +SAMPLE — v2zloss_86k,coqa,0,lm-eval,vllm,,em,none,0.6893333333333334,0.017170902772605603,500 +SAMPLE — v2zloss_86k,coqa,0,lm-eval,vllm,,f1,none,0.8130025286567438,0.012694351593796397,500 +SAMPLE — v2zloss_86k,flores200:als_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,43.15484357740711,0.017939582735682903,1012 +SAMPLE — v2zloss_86k,flores200:als_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7016398150107399,0.003928774177574626,1012 +SAMPLE — v2zloss_86k,flores200:als_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.26070986207612373,0.006068773122371049,1012 +SAMPLE — v2zloss_86k,flores200:als_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,67.9082,,1012 +SAMPLE — v2zloss_86k,flores200:als_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,66.3796,,1012 +SAMPLE — v2zloss_86k,flores200:bos_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,45.26800412709635,0.018388421015858556,1012 +SAMPLE — v2zloss_86k,flores200:bos_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7103528657615289,0.0037940013483790088,1012 +SAMPLE — v2zloss_86k,flores200:bos_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.2770834756045526,0.006201431597852179,1012 +SAMPLE — v2zloss_86k,flores200:bos_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,69.2449,,1012 +SAMPLE — v2zloss_86k,flores200:bos_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,67.6195,,1012 +SAMPLE — v2zloss_86k,flores200:bul_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,43.550206044926256,0.017798261667984417,1012 +SAMPLE — v2zloss_86k,flores200:bul_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.698149376737196,0.003883807912133704,1012 +SAMPLE — v2zloss_86k,flores200:bul_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.2564895443978894,0.005882879564446523,1012 +SAMPLE — v2zloss_86k,flores200:bul_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,68.4733,,1012 +SAMPLE — v2zloss_86k,flores200:bul_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,66.6918,,1012 +SAMPLE — v2zloss_86k,flores200:cat_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,48.76119389089646,0.019462036989439338,1012 +SAMPLE — v2zloss_86k,flores200:cat_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7298787223940586,0.003985986597985141,1012 +SAMPLE — v2zloss_86k,flores200:cat_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.31786561059517116,0.006892960950108528,1012 +SAMPLE — v2zloss_86k,flores200:cat_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,71.5745,,1012 +SAMPLE — v2zloss_86k,flores200:cat_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,70.0741,,1012 +SAMPLE — v2zloss_86k,flores200:ces_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,42.3121633916169,0.019007890927171513,1012 +SAMPLE — v2zloss_86k,flores200:ces_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6888920951595143,0.004088195843553164,1012 +SAMPLE — v2zloss_86k,flores200:ces_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.25412341045794007,0.006259876583712118,1012 +SAMPLE — v2zloss_86k,flores200:ces_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,67.0804,,1012 +SAMPLE — v2zloss_86k,flores200:ces_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,65.2597,,1012 +SAMPLE — v2zloss_86k,flores200:dan_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,51.298363425282716,0.019462559405425597,1012 +SAMPLE — v2zloss_86k,flores200:dan_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7479330683060331,0.0038909565902536325,1012 +SAMPLE — v2zloss_86k,flores200:dan_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.34268821384195847,0.006885934096055711,1012 +SAMPLE — v2zloss_86k,flores200:dan_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,73.3215,,1012 +SAMPLE — v2zloss_86k,flores200:dan_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,71.9241,,1012 +SAMPLE — v2zloss_86k,flores200:deu_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,45.99277042752986,0.019327541331427202,1012 +SAMPLE — v2zloss_86k,flores200:deu_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7175212145516533,0.003950692266783763,1012 +SAMPLE — v2zloss_86k,flores200:deu_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.2871437514733195,0.00655039757287239,1012 +SAMPLE — v2zloss_86k,flores200:deu_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,69.8267,,1012 +SAMPLE — v2zloss_86k,flores200:deu_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,68.1997,,1012 +SAMPLE — v2zloss_86k,flores200:ell_Grek-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,38.38772719856739,0.018432909134369246,1012 +SAMPLE — v2zloss_86k,flores200:ell_Grek-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6562251989184927,0.00420064560255897,1012 +SAMPLE — v2zloss_86k,flores200:ell_Grek-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.21566590867790425,0.005589467303406059,1012 +SAMPLE — v2zloss_86k,flores200:ell_Grek-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,63.6226,,1012 +SAMPLE — v2zloss_86k,flores200:ell_Grek-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,61.9782,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-als_Latn,4,lighteval,hf/accelerate,,bleu,none,32.46839339768063,0.015008056145177823,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-als_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6132245557830914,0.00383761991922053,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-als_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.166243744488621,0.0046019676182615194,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-als_Latn,4,lighteval,hf/accelerate,,chrf,rescored,59.9068,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-als_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,57.6852,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bos_Latn,4,lighteval,hf/accelerate,,bleu,none,30.259506735912083,0.017134851964932623,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bos_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5819622722108273,0.004484923667468899,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bos_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.14964054585535175,0.004631760970337812,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bos_Latn,4,lighteval,hf/accelerate,,chrf,rescored,59.5742,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bos_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,56.75,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bul_Cyrl,4,lighteval,hf/accelerate,,bleu,none,39.4783697094754,0.01718689300899459,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bul_Cyrl,4,lighteval,hf/accelerate,,bleu_1,none,0.6503314554634276,0.003982275711190928,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bul_Cyrl,4,lighteval,hf/accelerate,,bleu_4,none,0.2166797605081167,0.005454440842039369,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bul_Cyrl,4,lighteval,hf/accelerate,,chrf,rescored,65.8023,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-bul_Cyrl,4,lighteval,hf/accelerate,,chrf++,rescored,63.4463,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-cat_Latn,4,lighteval,hf/accelerate,,bleu,none,43.28778476576059,0.01793376424195155,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-cat_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6763633281659894,0.004069707919196525,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-cat_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.2630116375631319,0.005925509207736194,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-cat_Latn,4,lighteval,hf/accelerate,,chrf,rescored,66.921,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-cat_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,65.1371,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ces_Latn,4,lighteval,hf/accelerate,,bleu,none,31.79834828951852,0.018617501204324517,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ces_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5809244089825549,0.005024090823401042,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ces_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.16302828490183777,0.005451366895856798,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ces_Latn,4,lighteval,hf/accelerate,,chrf,rescored,58.5993,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ces_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,56.1423,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-dan_Latn,4,lighteval,hf/accelerate,,bleu,none,47.14966839074961,0.019928461466258856,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-dan_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7026469560676726,0.004351878366766089,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-dan_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.3060392260281265,0.006976290351907474,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-dan_Latn,4,lighteval,hf/accelerate,,chrf,rescored,70.1772,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-dan_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,68.3041,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-deu_Latn,4,lighteval,hf/accelerate,,bleu,none,38.77408634740993,0.021604342612178407,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-deu_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6440324577920763,0.004639316493605525,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-deu_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.22786421586025526,0.006396930823134115,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-deu_Latn,4,lighteval,hf/accelerate,,chrf,rescored,65.101,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-deu_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,62.6667,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ell_Grek,4,lighteval,hf/accelerate,,bleu,none,25.19706434720217,0.014691550447362088,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ell_Grek,4,lighteval,hf/accelerate,,bleu_1,none,0.5222809878696243,0.004367040777251122,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ell_Grek,4,lighteval,hf/accelerate,,bleu_4,none,0.11471316572790595,0.003919741776828707,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ell_Grek,4,lighteval,hf/accelerate,,chrf,rescored,52.1776,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ell_Grek,4,lighteval,hf/accelerate,,chrf++,rescored,49.8258,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-est_Latn,4,lighteval,hf/accelerate,,bleu,none,25.658755267745242,0.017876792736864468,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-est_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5413757097773075,0.005091981762639796,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-est_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.11921908298618301,0.005150562762854398,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-est_Latn,4,lighteval,hf/accelerate,,chrf,rescored,58.4753,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-est_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,54.7326,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-eus_Latn,4,lighteval,hf/accelerate,,bleu,none,16.329277574976444,0.015285842389502978,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-eus_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.45711975278626904,0.004584383393488051,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-eus_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.059460935758665914,0.003435337412665108,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-eus_Latn,4,lighteval,hf/accelerate,,chrf,rescored,54.536,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-eus_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,49.4724,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fin_Latn,4,lighteval,hf/accelerate,,bleu,none,23.713652104338358,0.01821909264103113,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fin_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5130173089243595,0.005077220666594102,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fin_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.10225428229015586,0.004674825060733169,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fin_Latn,4,lighteval,hf/accelerate,,chrf,rescored,58.2195,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fin_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,53.9887,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fra_Latn,4,lighteval,hf/accelerate,,bleu,none,50.39735969212533,0.022673186767757283,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fra_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6969449833136956,0.0050150531078441765,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fra_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.35562942137483583,0.007811512820111758,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fra_Latn,4,lighteval,hf/accelerate,,chrf,rescored,71.3959,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-fra_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,69.5294,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-gle_Latn,4,lighteval,hf/accelerate,,bleu,none,25.166503032457186,0.017530459333468244,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-gle_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5282746167798673,0.004523910460807955,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-gle_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.11494984399768038,0.004120667610731716,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-gle_Latn,4,lighteval,hf/accelerate,,chrf,rescored,52.4116,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-gle_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,49.9747,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-glg_Latn,4,lighteval,hf/accelerate,,bleu,none,36.533186358122144,0.015966765022370048,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-glg_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6347708721104863,0.003821364715956283,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-glg_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.20264847411918663,0.0052724828975282256,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-glg_Latn,4,lighteval,hf/accelerate,,chrf,rescored,62.503,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-glg_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,60.4355,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hrv_Latn,4,lighteval,hf/accelerate,,bleu,none,30.37873245692972,0.018565229237391397,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hrv_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5681705989787643,0.0049847103544917,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hrv_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.15371363167923632,0.005271824403331943,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hrv_Latn,4,lighteval,hf/accelerate,,chrf,rescored,59.5726,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hrv_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,56.4901,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hun_Latn,4,lighteval,hf/accelerate,,bleu,none,25.204867178270682,0.01770565553464993,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hun_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5473130744396111,0.004481553522947644,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hun_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.10408013352936916,0.004219922197130141,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hun_Latn,4,lighteval,hf/accelerate,,chrf,rescored,56.8627,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-hun_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,53.6175,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-isl_Latn,4,lighteval,hf/accelerate,,bleu,none,25.82073694044935,0.017848541126692746,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-isl_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.520174542282388,0.004934567256462505,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-isl_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.12183356466519976,0.005002829356825339,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-isl_Latn,4,lighteval,hf/accelerate,,chrf,rescored,53.4857,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-isl_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,50.7322,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ita_Latn,4,lighteval,hf/accelerate,,bleu,none,31.27525878567498,0.014784617707220338,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ita_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5891500102720587,0.003782081158533112,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ita_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.1549534373888949,0.0041281870692816665,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ita_Latn,4,lighteval,hf/accelerate,,chrf,rescored,59.8926,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ita_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,57.3065,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-kat_Geor,4,lighteval,hf/accelerate,,bleu,none,3.573560803030421,0.008198151025502583,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-kat_Geor,4,lighteval,hf/accelerate,,bleu_1,none,0.20564255820036081,0.004154561304829092,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-kat_Geor,4,lighteval,hf/accelerate,,bleu_4,none,0.005856853207845286,0.0009599110196054988,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-kat_Geor,4,lighteval,hf/accelerate,,chrf,rescored,30.2564,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-kat_Geor,4,lighteval,hf/accelerate,,chrf++,rescored,26.2106,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lit_Latn,4,lighteval,hf/accelerate,,bleu,none,25.010824565180215,0.018627436744201722,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lit_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5305690535858375,0.0049725161894998435,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lit_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.11433740063091617,0.004856009910407719,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lit_Latn,4,lighteval,hf/accelerate,,chrf,rescored,56.6502,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lit_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,53.1458,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lvs_Latn,4,lighteval,hf/accelerate,,bleu,none,29.903179898772134,0.018673081207676245,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lvs_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5789355495228588,0.0051903053873248225,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lvs_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.15650459712433817,0.005885951815794233,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lvs_Latn,4,lighteval,hf/accelerate,,chrf,rescored,59.3496,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-lvs_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,56.3446,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mkd_Cyrl,4,lighteval,hf/accelerate,,bleu,none,33.99253708724857,0.01657299441932838,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mkd_Cyrl,4,lighteval,hf/accelerate,,bleu_1,none,0.6106083514771972,0.0040385393952270185,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mkd_Cyrl,4,lighteval,hf/accelerate,,bleu_4,none,0.1766549623126907,0.004981284657625351,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mkd_Cyrl,4,lighteval,hf/accelerate,,chrf,rescored,62.5925,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mkd_Cyrl,4,lighteval,hf/accelerate,,chrf++,rescored,59.9119,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mlt_Latn,4,lighteval,hf/accelerate,,bleu,none,33.788798816500865,0.01915314299599444,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mlt_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6032979123114204,0.0046240150562844,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mlt_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.18317520104237345,0.00579666457076755,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mlt_Latn,4,lighteval,hf/accelerate,,chrf,rescored,65.4875,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-mlt_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,61.9397,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nld_Latn,4,lighteval,hf/accelerate,,bleu,none,27.73019855521587,0.01602267064942607,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nld_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.573096265063972,0.0044248628735436245,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nld_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.1281687676010653,0.004478670667305216,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nld_Latn,4,lighteval,hf/accelerate,,chrf,rescored,58.6028,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nld_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,55.7049,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nob_Latn,4,lighteval,hf/accelerate,,bleu,none,33.921352296717785,0.015755745965914912,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nob_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6117227692395014,0.00381629611078369,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nob_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.16598423537106177,0.004527757206170532,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nob_Latn,4,lighteval,hf/accelerate,,chrf,rescored,62.0817,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-nob_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,59.6167,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-pol_Latn,4,lighteval,hf/accelerate,,bleu,none,21.82848176887518,0.015581699311942886,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-pol_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.4908700961457207,0.004505996969952573,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-pol_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.09044638767284063,0.003829878819260267,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-pol_Latn,4,lighteval,hf/accelerate,,chrf,rescored,51.8318,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-pol_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,48.6807,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-por_Latn,4,lighteval,hf/accelerate,,bleu,none,48.40099426779949,0.020645497512573135,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-por_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7146327482033148,0.004594631131248135,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-por_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.32402890820750185,0.007498203287606576,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-por_Latn,4,lighteval,hf/accelerate,,chrf,rescored,70.4469,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-por_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,68.7136,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ron_Latn,4,lighteval,hf/accelerate,,bleu,none,39.62556758876426,0.018433516832299397,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ron_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6390948170275935,0.004522951018139962,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ron_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.24108561712555887,0.006101988181598187,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ron_Latn,4,lighteval,hf/accelerate,,chrf,rescored,64.3,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ron_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,62.1267,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slk_Latn,4,lighteval,hf/accelerate,,bleu,none,32.682261715412984,0.020470376447817126,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slk_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5861280399671898,0.005124823374947697,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slk_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.17591686409455712,0.005934956443314791,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slk_Latn,4,lighteval,hf/accelerate,,chrf,rescored,59.5298,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slk_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,56.9499,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slv_Latn,4,lighteval,hf/accelerate,,bleu,none,29.630412445393045,0.01860552849377903,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slv_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5715666903700491,0.00496635964998798,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slv_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.15284010897061898,0.00536381944223259,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slv_Latn,4,lighteval,hf/accelerate,,chrf,rescored,57.5688,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-slv_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,54.9216,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-spa_Latn,4,lighteval,hf/accelerate,,bleu,none,28.661774091199923,0.013332610378918835,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-spa_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.571590822349121,0.00361404784586061,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-spa_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.13535120270481318,0.003674758913929313,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-spa_Latn,4,lighteval,hf/accelerate,,chrf,rescored,56.3516,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-spa_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,54.1636,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-srp_Cyrl,4,lighteval,hf/accelerate,,bleu,none,32.89715811482724,0.01826329496232478,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-srp_Cyrl,4,lighteval,hf/accelerate,,bleu_1,none,0.5957259083103951,0.00473685565446639,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-srp_Cyrl,4,lighteval,hf/accelerate,,bleu_4,none,0.17377514565973415,0.005459385661574218,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-srp_Cyrl,4,lighteval,hf/accelerate,,chrf,rescored,60.0796,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-srp_Cyrl,4,lighteval,hf/accelerate,,chrf++,rescored,57.5619,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-swe_Latn,4,lighteval,hf/accelerate,,bleu,none,44.594903923715044,0.020335230520540273,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-swe_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.681172564915298,0.004529381466519326,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-swe_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.2782333540908908,0.006965749289154411,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-swe_Latn,4,lighteval,hf/accelerate,,chrf,rescored,68.6997,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-swe_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,66.5718,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-tur_Latn,4,lighteval,hf/accelerate,,bleu,none,23.60763219178395,0.018346164659821425,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-tur_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.5211662887486086,0.004975521965020394,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-tur_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.1044246701267615,0.004760008181504263,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-tur_Latn,4,lighteval,hf/accelerate,,chrf,rescored,57.4347,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-tur_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,53.5511,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ukr_Cyrl,4,lighteval,hf/accelerate,,bleu,none,28.318763585338143,0.018157786733603947,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ukr_Cyrl,4,lighteval,hf/accelerate,,bleu_1,none,0.5442049166087083,0.004805411719360678,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ukr_Cyrl,4,lighteval,hf/accelerate,,bleu_4,none,0.1341098355749339,0.004751565776125876,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ukr_Cyrl,4,lighteval,hf/accelerate,,chrf,rescored,56.9172,,1012 +SAMPLE — v2zloss_86k,flores200:eng_Latn-ukr_Cyrl,4,lighteval,hf/accelerate,,chrf++,rescored,53.9879,,1012 +SAMPLE — v2zloss_86k,flores200:est_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,39.07187895703809,0.01873952570879084,1012 +SAMPLE — v2zloss_86k,flores200:est_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6669668376612574,0.004263456528126067,1012 +SAMPLE — v2zloss_86k,flores200:est_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.2240550185282144,0.006145724329720707,1012 +SAMPLE — v2zloss_86k,flores200:est_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,64.5804,,1012 +SAMPLE — v2zloss_86k,flores200:est_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,62.7229,,1012 +SAMPLE — v2zloss_86k,flores200:eus_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,32.977239136520154,0.018265117186369823,1012 +SAMPLE — v2zloss_86k,flores200:eus_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6152178865980656,0.00440373104346501,1012 +SAMPLE — v2zloss_86k,flores200:eus_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.17178776712911928,0.005415831856262811,1012 +SAMPLE — v2zloss_86k,flores200:eus_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,59.7224,,1012 +SAMPLE — v2zloss_86k,flores200:eus_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,57.7023,,1012 +SAMPLE — v2zloss_86k,flores200:fin_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,36.17403024476525,0.01793146948009941,1012 +SAMPLE — v2zloss_86k,flores200:fin_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6474431366256146,0.0042614339956196946,1012 +SAMPLE — v2zloss_86k,flores200:fin_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.1978345901371916,0.005756009510078358,1012 +SAMPLE — v2zloss_86k,flores200:fin_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,62.7081,,1012 +SAMPLE — v2zloss_86k,flores200:fin_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,60.7471,,1012 +SAMPLE — v2zloss_86k,flores200:fra_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,47.64231270125276,0.020189478353379905,1012 +SAMPLE — v2zloss_86k,flores200:fra_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7251439755695396,0.004303338372947135,1012 +SAMPLE — v2zloss_86k,flores200:fra_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.31268101911971546,0.00705920701971619,1012 +SAMPLE — v2zloss_86k,flores200:fra_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,70.4759,,1012 +SAMPLE — v2zloss_86k,flores200:fra_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,69.0873,,1012 +SAMPLE — v2zloss_86k,flores200:gle_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,41.364754613117704,0.02058338169764184,1012 +SAMPLE — v2zloss_86k,flores200:gle_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6771625218420935,0.004538929567482198,1012 +SAMPLE — v2zloss_86k,flores200:gle_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.25534773239427816,0.006798873019790954,1012 +SAMPLE — v2zloss_86k,flores200:gle_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,65.4949,,1012 +SAMPLE — v2zloss_86k,flores200:gle_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,64.031,,1012 +SAMPLE — v2zloss_86k,flores200:glg_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,44.056506983260405,0.01858671981027671,1012 +SAMPLE — v2zloss_86k,flores200:glg_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7045959575632684,0.004070347752197927,1012 +SAMPLE — v2zloss_86k,flores200:glg_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.2709658425334686,0.006377331154233219,1012 +SAMPLE — v2zloss_86k,flores200:glg_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,69.0982,,1012 +SAMPLE — v2zloss_86k,flores200:glg_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,67.4022,,1012 +SAMPLE — v2zloss_86k,flores200:hrv_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,40.23955869626295,0.01904200840423673,1012 +SAMPLE — v2zloss_86k,flores200:hrv_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.669596573816148,0.004278709939624304,1012 +SAMPLE — v2zloss_86k,flores200:hrv_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.23005288628940407,0.006058710323252195,1012 +SAMPLE — v2zloss_86k,flores200:hrv_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,65.2713,,1012 +SAMPLE — v2zloss_86k,flores200:hrv_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,63.415,,1012 +SAMPLE — v2zloss_86k,flores200:hun_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,37.43883973191005,0.01795642897213757,1012 +SAMPLE — v2zloss_86k,flores200:hun_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6504431098234527,0.004352165152880017,1012 +SAMPLE — v2zloss_86k,flores200:hun_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.21085022247685692,0.0058853845769528575,1012 +SAMPLE — v2zloss_86k,flores200:hun_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,63.3295,,1012 +SAMPLE — v2zloss_86k,flores200:hun_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,61.3877,,1012 +SAMPLE — v2zloss_86k,flores200:isl_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,37.71724892123652,0.01891715092241016,1012 +SAMPLE — v2zloss_86k,flores200:isl_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6453007437193008,0.004340721399115146,1012 +SAMPLE — v2zloss_86k,flores200:isl_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.21124603641775402,0.005941275041057287,1012 +SAMPLE — v2zloss_86k,flores200:isl_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,62.2385,,1012 +SAMPLE — v2zloss_86k,flores200:isl_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,60.592,,1012 +SAMPLE — v2zloss_86k,flores200:ita_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,36.350811150140494,0.017401199125080136,1012 +SAMPLE — v2zloss_86k,flores200:ita_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6516568754739583,0.003853750105943939,1012 +SAMPLE — v2zloss_86k,flores200:ita_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.19912005603472258,0.005408855082502472,1012 +SAMPLE — v2zloss_86k,flores200:ita_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,63.8595,,1012 +SAMPLE — v2zloss_86k,flores200:ita_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,61.9357,,1012 +SAMPLE — v2zloss_86k,flores200:kat_Geor-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,20.265423195588852,0.015108030640879875,1012 +SAMPLE — v2zloss_86k,flores200:kat_Geor-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.49948041380117997,0.004451731578627528,1012 +SAMPLE — v2zloss_86k,flores200:kat_Geor-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.07834403094467393,0.003776569120022487,1012 +SAMPLE — v2zloss_86k,flores200:kat_Geor-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,48.3665,,1012 +SAMPLE — v2zloss_86k,flores200:kat_Geor-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,46.144,,1012 +SAMPLE — v2zloss_86k,flores200:lit_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,36.02169296707567,0.01883689362339753,1012 +SAMPLE — v2zloss_86k,flores200:lit_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6312266029101086,0.004530299198600238,1012 +SAMPLE — v2zloss_86k,flores200:lit_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.1978897041216494,0.005852463849795031,1012 +SAMPLE — v2zloss_86k,flores200:lit_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,61.4546,,1012 +SAMPLE — v2zloss_86k,flores200:lit_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,59.5395,,1012 +SAMPLE — v2zloss_86k,flores200:lvs_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,38.38332964881752,0.020041901293705363,1012 +SAMPLE — v2zloss_86k,flores200:lvs_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6567882739389189,0.004539247117510753,1012 +SAMPLE — v2zloss_86k,flores200:lvs_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.21980334440981714,0.0061876403232037,1012 +SAMPLE — v2zloss_86k,flores200:lvs_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,63.9706,,1012 +SAMPLE — v2zloss_86k,flores200:lvs_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,62.0813,,1012 +SAMPLE — v2zloss_86k,flores200:mkd_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,43.40269042521689,0.020332864311967278,1012 +SAMPLE — v2zloss_86k,flores200:mkd_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6852446991007115,0.004949223658420087,1012 +SAMPLE — v2zloss_86k,flores200:mkd_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.2628625008283234,0.006058447413793952,1012 +SAMPLE — v2zloss_86k,flores200:mkd_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,66.1748,,1012 +SAMPLE — v2zloss_86k,flores200:mkd_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,64.7465,,1012 +SAMPLE — v2zloss_86k,flores200:mlt_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,56.005917936821334,0.020274966121612076,1012 +SAMPLE — v2zloss_86k,flores200:mlt_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7727925897868655,0.003870920218253921,1012 +SAMPLE — v2zloss_86k,flores200:mlt_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.3993635772511284,0.0075436734010463,1012 +SAMPLE — v2zloss_86k,flores200:mlt_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,76.5943,,1012 +SAMPLE — v2zloss_86k,flores200:mlt_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,75.3534,,1012 +SAMPLE — v2zloss_86k,flores200:nld_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,33.63691139703907,0.018285951166248482,1012 +SAMPLE — v2zloss_86k,flores200:nld_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6146562451093543,0.004374568847045918,1012 +SAMPLE — v2zloss_86k,flores200:nld_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.17494820293560417,0.005263149545924782,1012 +SAMPLE — v2zloss_86k,flores200:nld_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,60.7298,,1012 +SAMPLE — v2zloss_86k,flores200:nld_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,58.5523,,1012 +SAMPLE — v2zloss_86k,flores200:nob_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,46.30904840805819,0.01866738130584222,1012 +SAMPLE — v2zloss_86k,flores200:nob_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7051966790654006,0.004001215446818904,1012 +SAMPLE — v2zloss_86k,flores200:nob_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.289310662192357,0.006313157806972042,1012 +SAMPLE — v2zloss_86k,flores200:nob_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,69.3614,,1012 +SAMPLE — v2zloss_86k,flores200:nob_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,67.7577,,1012 +SAMPLE — v2zloss_86k,flores200:pol_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,32.22471322459842,0.017484554285783438,1012 +SAMPLE — v2zloss_86k,flores200:pol_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6100597471670259,0.004363798981443197,1012 +SAMPLE — v2zloss_86k,flores200:pol_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.16457460578592126,0.005025181859580228,1012 +SAMPLE — v2zloss_86k,flores200:pol_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,59.4459,,1012 +SAMPLE — v2zloss_86k,flores200:pol_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,57.3631,,1012 +SAMPLE — v2zloss_86k,flores200:por_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,51.88661547575336,0.01941702217411646,1012 +SAMPLE — v2zloss_86k,flores200:por_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.748107281539116,0.004086296771384581,1012 +SAMPLE — v2zloss_86k,flores200:por_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.34983896865398495,0.00714865451774383,1012 +SAMPLE — v2zloss_86k,flores200:por_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,73.3752,,1012 +SAMPLE — v2zloss_86k,flores200:por_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,72.0524,,1012 +SAMPLE — v2zloss_86k,flores200:ron_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,45.77985100269046,0.018872769929196852,1012 +SAMPLE — v2zloss_86k,flores200:ron_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7172791642960482,0.004057228835309759,1012 +SAMPLE — v2zloss_86k,flores200:ron_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.29039394254569834,0.006587206746967689,1012 +SAMPLE — v2zloss_86k,flores200:ron_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,70.1614,,1012 +SAMPLE — v2zloss_86k,flores200:ron_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,68.5194,,1012 +SAMPLE — v2zloss_86k,flores200:slk_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,42.10608144276334,0.019872332615263954,1012 +SAMPLE — v2zloss_86k,flores200:slk_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.683108905758397,0.004261310887335916,1012 +SAMPLE — v2zloss_86k,flores200:slk_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.249470166608758,0.00639593131770323,1012 +SAMPLE — v2zloss_86k,flores200:slk_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,66.9033,,1012 +SAMPLE — v2zloss_86k,flores200:slk_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,65.0481,,1012 +SAMPLE — v2zloss_86k,flores200:slv_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,38.08315706032328,0.01888904357828561,1012 +SAMPLE — v2zloss_86k,flores200:slv_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6548240830614656,0.004163289737139833,1012 +SAMPLE — v2zloss_86k,flores200:slv_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.21204838834807846,0.005822530273603072,1012 +SAMPLE — v2zloss_86k,flores200:slv_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,63.6138,,1012 +SAMPLE — v2zloss_86k,flores200:slv_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,61.7616,,1012 +SAMPLE — v2zloss_86k,flores200:spa_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,33.34916087144877,0.02109416656173379,1012 +SAMPLE — v2zloss_86k,flores200:spa_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6304270994467864,0.0040727944653932095,1012 +SAMPLE — v2zloss_86k,flores200:spa_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.17943509088668366,0.005150086556110622,1012 +SAMPLE — v2zloss_86k,flores200:spa_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,61.7717,,1012 +SAMPLE — v2zloss_86k,flores200:spa_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,59.8581,,1012 +SAMPLE — v2zloss_86k,flores200:srp_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,46.102582725188796,0.01862818220171685,1012 +SAMPLE — v2zloss_86k,flores200:srp_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7156158619793758,0.004073870093011215,1012 +SAMPLE — v2zloss_86k,flores200:srp_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.29415454656306905,0.006570894360884576,1012 +SAMPLE — v2zloss_86k,flores200:srp_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,69.7286,,1012 +SAMPLE — v2zloss_86k,flores200:srp_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,68.1542,,1012 +SAMPLE — v2zloss_86k,flores200:swe_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,50.43425323362832,0.019184093488803632,1012 +SAMPLE — v2zloss_86k,flores200:swe_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.7385501614213487,0.003963386256096678,1012 +SAMPLE — v2zloss_86k,flores200:swe_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.33115754358201066,0.006815101931363007,1012 +SAMPLE — v2zloss_86k,flores200:swe_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,72.0367,,1012 +SAMPLE — v2zloss_86k,flores200:swe_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,70.6011,,1012 +SAMPLE — v2zloss_86k,flores200:tur_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,38.63651905454886,0.018725923734772637,1012 +SAMPLE — v2zloss_86k,flores200:tur_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6630482473669941,0.004395866843760001,1012 +SAMPLE — v2zloss_86k,flores200:tur_Latn-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.2206211265030549,0.006091205336688021,1012 +SAMPLE — v2zloss_86k,flores200:tur_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,63.9879,,1012 +SAMPLE — v2zloss_86k,flores200:tur_Latn-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,62.1983,,1012 +SAMPLE — v2zloss_86k,flores200:ukr_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu,none,41.61111991264253,0.022250789588599353,1012 +SAMPLE — v2zloss_86k,flores200:ukr_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu_1,none,0.6814199067452068,0.0042840508776382695,1012 +SAMPLE — v2zloss_86k,flores200:ukr_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,bleu_4,none,0.25063887267395896,0.006381069605743584,1012 +SAMPLE — v2zloss_86k,flores200:ukr_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,chrf,rescored,66.0521,,1012 +SAMPLE — v2zloss_86k,flores200:ukr_Cyrl-eng_Latn,4,lighteval,hf/accelerate,,chrf++,rescored,64.444,,1012 +SAMPLE — v2zloss_86k,global_mgsm_ca,0,lm-eval,vllm,,exact_match,flexible-extract,0.132,0.021450980824038093,250 +SAMPLE — v2zloss_86k,global_mgsm_cs,0,lm-eval,vllm,,exact_match,flexible-extract,0.02,0.008872139507342683,250 +SAMPLE — v2zloss_86k,global_mgsm_de,0,lm-eval,vllm,,exact_match,flexible-extract,0.052,0.01407039102564164,250 +SAMPLE — v2zloss_86k,global_mgsm_el,0,lm-eval,vllm,,exact_match,flexible-extract,0.16,0.023232714782060664,250 +SAMPLE — v2zloss_86k,global_mgsm_en,0,lm-eval,vllm,,exact_match,flexible-extract,0.64,0.030418764025174985,250 +SAMPLE — v2zloss_86k,global_mgsm_es,0,lm-eval,vllm,,exact_match,flexible-extract,0.168,0.023692813205492592,250 +SAMPLE — v2zloss_86k,global_mgsm_eu,0,lm-eval,vllm,,exact_match,flexible-extract,0.012,0.00690032302369429,250 +SAMPLE — v2zloss_86k,global_mgsm_fr,0,lm-eval,vllm,,exact_match,flexible-extract,0.104,0.01934510097484389,250 +SAMPLE — v2zloss_86k,global_mgsm_gl,0,lm-eval,vllm,,exact_match,flexible-extract,0.088,0.0179530847770529,250 +SAMPLE — v2zloss_86k,global_mgsm_hu,0,lm-eval,vllm,,exact_match,flexible-extract,0.068,0.015953748410747023,250 +SAMPLE — v2zloss_86k,global_mgsm_sr,0,lm-eval,vllm,,exact_match,flexible-extract,0.016,0.00795166118887433,250 +SAMPLE — v2zloss_86k,global_mmlu_full_cs,5,lm-eval,vllm,yes,acc,none,0.6001994017946162,0.003953637555353729, +SAMPLE — v2zloss_86k,global_mmlu_full_cs_abstract_algebra,5,lm-eval,vllm,,acc,none,0.3,0.04605661864718382,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_anatomy,5,lm-eval,vllm,,acc,none,0.5481481481481482,0.04299268905480864,135 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_astronomy,5,lm-eval,vllm,,acc,none,0.6973684210526315,0.03738520676119667,152 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_business_ethics,5,lm-eval,vllm,,acc,none,0.56,0.049888765156985884,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6415094339622641,0.02951470358398169,265 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_college_biology,5,lm-eval,vllm,,acc,none,0.6319444444444444,0.04032999053960717,144 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_college_chemistry,5,lm-eval,vllm,,acc,none,0.49,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_college_computer_science,5,lm-eval,vllm,,acc,none,0.49,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_college_mathematics,5,lm-eval,vllm,,acc,none,0.44,0.049888765156985884,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_college_medicine,5,lm-eval,vllm,,acc,none,0.6069364161849711,0.0372424959581773,173 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_college_physics,5,lm-eval,vllm,,acc,none,0.38235294117647056,0.048355036961072254,102 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_computer_security,5,lm-eval,vllm,,acc,none,0.68,0.046882617226215076,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6638297872340425,0.03088161852067694,235 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_econometrics,5,lm-eval,vllm,,acc,none,0.45614035087719296,0.04685473041907789,114 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6344827586206897,0.04013124195424389,145 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5767195767195767,0.025446365634406765,378 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_formal_logic,5,lm-eval,vllm,,acc,none,0.5238095238095238,0.04467062628403266,126 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_global_facts,5,lm-eval,vllm,,acc,none,0.39,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_biology,5,lm-eval,vllm,,acc,none,0.7258064516129032,0.025378139970885245,310 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5369458128078818,0.03508370520442665,203 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.73,0.04461960433384737,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7636363636363637,0.03317505930009174,165 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_geography,5,lm-eval,vllm,,acc,none,0.7575757575757576,0.030532892233932022,198 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.8082901554404145,0.02840895362624522,193 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6256410256410256,0.024537591572830485,390 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.3925925925925926,0.02977384701253303,270 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6512605042016807,0.03095663632856658,238 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_physics,5,lm-eval,vllm,,acc,none,0.41721854304635764,0.040261414976346124,151 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_psychology,5,lm-eval,vllm,,acc,none,0.7889908256880734,0.017493922404112613,545 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_statistics,5,lm-eval,vllm,,acc,none,0.5972222222222222,0.0334488738299786,216 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7450980392156863,0.030587591351604302,204 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7890295358649789,0.0265583725026619,237 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_human_aging,5,lm-eval,vllm,,acc,none,0.7040358744394619,0.03063659134869978,223 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_human_sexuality,5,lm-eval,vllm,,acc,none,0.7175572519083969,0.039484061257683584,131 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_humanities,5,lm-eval,vllm,yes,acc,none,0.5432518597236982,0.006840997409617259, +SAMPLE — v2zloss_86k,global_mmlu_full_cs_international_law,5,lm-eval,vllm,,acc,none,0.7851239669421488,0.037494924487096966,121 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_jurisprudence,5,lm-eval,vllm,,acc,none,0.6481481481481481,0.04616631111801715,108 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_logical_fallacies,5,lm-eval,vllm,,acc,none,0.6441717791411042,0.03761521380046734,163 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_machine_learning,5,lm-eval,vllm,,acc,none,0.4642857142857143,0.04733667890053756,112 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_management,5,lm-eval,vllm,,acc,none,0.7281553398058253,0.04405268024140923,103 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_marketing,5,lm-eval,vllm,,acc,none,0.8247863247863247,0.02490443909891819,234 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_medical_genetics,5,lm-eval,vllm,,acc,none,0.6,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_miscellaneous,5,lm-eval,vllm,,acc,none,0.70242656449553,0.01634911191290953,783 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_moral_disputes,5,lm-eval,vllm,,acc,none,0.6705202312138728,0.02530525813187972,346 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_moral_scenarios,5,lm-eval,vllm,,acc,none,0.3139664804469274,0.015521923933523658,895 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_nutrition,5,lm-eval,vllm,,acc,none,0.6928104575163399,0.02641560191438905,306 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_other,5,lm-eval,vllm,yes,acc,none,0.6311554554232378,0.00838589252654852, +SAMPLE — v2zloss_86k,global_mmlu_full_cs_philosophy,5,lm-eval,vllm,,acc,none,0.684887459807074,0.026385273703464482,311 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_prehistory,5,lm-eval,vllm,,acc,none,0.6820987654320988,0.02591006352824088,324 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_professional_accounting,5,lm-eval,vllm,,acc,none,0.3546099290780142,0.0285386500288787,282 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_professional_law,5,lm-eval,vllm,,acc,none,0.43741851368970014,0.012669813464935648,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_professional_medicine,5,lm-eval,vllm,,acc,none,0.5882352941176471,0.02989616303312549,272 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_professional_psychology,5,lm-eval,vllm,,acc,none,0.6127450980392157,0.019706875804085634,612 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_public_relations,5,lm-eval,vllm,,acc,none,0.6545454545454545,0.045546196175410524,110 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_security_studies,5,lm-eval,vllm,,acc,none,0.7142857142857143,0.028920583220675568,245 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_social_sciences,5,lm-eval,vllm,yes,acc,none,0.6948326291842704,0.00816574921828659, +SAMPLE — v2zloss_86k,global_mmlu_full_cs_sociology,5,lm-eval,vllm,,acc,none,0.7810945273631841,0.02923917463664699,201 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_stem,5,lm-eval,vllm,yes,acc,none,0.5623215984776403,0.008604582202165931, +SAMPLE — v2zloss_86k,global_mmlu_full_cs_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.78,0.041633319989322654,100 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_virology,5,lm-eval,vllm,,acc,none,0.5060240963855421,0.03892212195333041,166 +SAMPLE — v2zloss_86k,global_mmlu_full_cs_world_religions,5,lm-eval,vllm,,acc,none,0.8011695906432749,0.030611116557432514,171 +SAMPLE — v2zloss_86k,global_mmlu_full_de,5,lm-eval,vllm,yes,acc,none,0.6061814556331007,0.003930969350956732, +SAMPLE — v2zloss_86k,global_mmlu_full_de_abstract_algebra,5,lm-eval,vllm,,acc,none,0.37,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_anatomy,5,lm-eval,vllm,,acc,none,0.5555555555555556,0.04292596718256977,135 +SAMPLE — v2zloss_86k,global_mmlu_full_de_astronomy,5,lm-eval,vllm,,acc,none,0.6776315789473685,0.03803510248351587,152 +SAMPLE — v2zloss_86k,global_mmlu_full_de_business_ethics,5,lm-eval,vllm,,acc,none,0.6,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6528301886792452,0.02930010170554966,265 +SAMPLE — v2zloss_86k,global_mmlu_full_de_college_biology,5,lm-eval,vllm,,acc,none,0.6319444444444444,0.04032999053960717,144 +SAMPLE — v2zloss_86k,global_mmlu_full_de_college_chemistry,5,lm-eval,vllm,,acc,none,0.51,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_college_computer_science,5,lm-eval,vllm,,acc,none,0.55,0.05,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_college_mathematics,5,lm-eval,vllm,,acc,none,0.42,0.04960449637488583,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_college_medicine,5,lm-eval,vllm,,acc,none,0.6011560693641619,0.03733626655383514,173 +SAMPLE — v2zloss_86k,global_mmlu_full_de_college_physics,5,lm-eval,vllm,,acc,none,0.3627450980392157,0.047840607041056527,102 +SAMPLE — v2zloss_86k,global_mmlu_full_de_computer_security,5,lm-eval,vllm,,acc,none,0.7,0.04605661864718383,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_conceptual_physics,5,lm-eval,vllm,,acc,none,0.676595744680851,0.03057944277361037,235 +SAMPLE — v2zloss_86k,global_mmlu_full_de_econometrics,5,lm-eval,vllm,,acc,none,0.4824561403508772,0.04700708033551044,114 +SAMPLE — v2zloss_86k,global_mmlu_full_de_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6689655172413793,0.03921545312467122,145 +SAMPLE — v2zloss_86k,global_mmlu_full_de_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5317460317460317,0.02569935283213174,378 +SAMPLE — v2zloss_86k,global_mmlu_full_de_formal_logic,5,lm-eval,vllm,,acc,none,0.5158730158730159,0.044698818540726104,126 +SAMPLE — v2zloss_86k,global_mmlu_full_de_global_facts,5,lm-eval,vllm,,acc,none,0.37,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_biology,5,lm-eval,vllm,,acc,none,0.7483870967741936,0.02468597928624002,310 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5467980295566502,0.035025446508458756,203 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.73,0.04461960433384737,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_european_history,5,lm-eval,vllm,,acc,none,0.793939393939394,0.03158415324047706,165 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_geography,5,lm-eval,vllm,,acc,none,0.7575757575757576,0.030532892233932022,198 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.844559585492228,0.026148483469153303,193 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6641025641025641,0.023946724741564056,390 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.37407407407407406,0.02950286112895523,270 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6890756302521008,0.030066761582977983,238 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_physics,5,lm-eval,vllm,,acc,none,0.4370860927152318,0.040500357222306375,151 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_psychology,5,lm-eval,vllm,,acc,none,0.7834862385321101,0.017658710594443204,545 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_statistics,5,lm-eval,vllm,,acc,none,0.5925925925925926,0.03350991604696042,216 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7107843137254902,0.031822318676475524,204 +SAMPLE — v2zloss_86k,global_mmlu_full_de_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7426160337552743,0.028458820991460302,237 +SAMPLE — v2zloss_86k,global_mmlu_full_de_human_aging,5,lm-eval,vllm,,acc,none,0.6591928251121076,0.03181149747055356,223 +SAMPLE — v2zloss_86k,global_mmlu_full_de_human_sexuality,5,lm-eval,vllm,,acc,none,0.7404580152671756,0.03844876139785267,131 +SAMPLE — v2zloss_86k,global_mmlu_full_de_humanities,5,lm-eval,vllm,yes,acc,none,0.5434643995749203,0.006839687004527041, +SAMPLE — v2zloss_86k,global_mmlu_full_de_international_law,5,lm-eval,vllm,,acc,none,0.7851239669421488,0.037494924487096966,121 +SAMPLE — v2zloss_86k,global_mmlu_full_de_jurisprudence,5,lm-eval,vllm,,acc,none,0.6481481481481481,0.04616631111801715,108 +SAMPLE — v2zloss_86k,global_mmlu_full_de_logical_fallacies,5,lm-eval,vllm,,acc,none,0.6748466257668712,0.036803503712864595,163 +SAMPLE — v2zloss_86k,global_mmlu_full_de_machine_learning,5,lm-eval,vllm,,acc,none,0.36607142857142855,0.04572372358737434,112 +SAMPLE — v2zloss_86k,global_mmlu_full_de_management,5,lm-eval,vllm,,acc,none,0.7378640776699029,0.04354631077260595,103 +SAMPLE — v2zloss_86k,global_mmlu_full_de_marketing,5,lm-eval,vllm,,acc,none,0.8333333333333334,0.02441494730454365,234 +SAMPLE — v2zloss_86k,global_mmlu_full_de_medical_genetics,5,lm-eval,vllm,,acc,none,0.69,0.046482319871173176,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_miscellaneous,5,lm-eval,vllm,,acc,none,0.7394636015325671,0.015696008563807068,783 +SAMPLE — v2zloss_86k,global_mmlu_full_de_moral_disputes,5,lm-eval,vllm,,acc,none,0.6734104046242775,0.02524826477424287,346 +SAMPLE — v2zloss_86k,global_mmlu_full_de_moral_scenarios,5,lm-eval,vllm,,acc,none,0.29832402234636873,0.015301840045129371,895 +SAMPLE — v2zloss_86k,global_mmlu_full_de_nutrition,5,lm-eval,vllm,,acc,none,0.6928104575163399,0.02641560191438905,306 +SAMPLE — v2zloss_86k,global_mmlu_full_de_other,5,lm-eval,vllm,yes,acc,none,0.643707756678468,0.008304145380510951, +SAMPLE — v2zloss_86k,global_mmlu_full_de_philosophy,5,lm-eval,vllm,,acc,none,0.684887459807074,0.026385273703464482,311 +SAMPLE — v2zloss_86k,global_mmlu_full_de_prehistory,5,lm-eval,vllm,,acc,none,0.7253086419753086,0.024836057868294684,324 +SAMPLE — v2zloss_86k,global_mmlu_full_de_professional_accounting,5,lm-eval,vllm,,acc,none,0.3829787234042553,0.0289990809048062,282 +SAMPLE — v2zloss_86k,global_mmlu_full_de_professional_law,5,lm-eval,vllm,,acc,none,0.4485006518904824,0.012702317490559806,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_de_professional_medicine,5,lm-eval,vllm,,acc,none,0.5845588235294118,0.029935342707877722,272 +SAMPLE — v2zloss_86k,global_mmlu_full_de_professional_psychology,5,lm-eval,vllm,,acc,none,0.6290849673202614,0.01954210156485416,612 +SAMPLE — v2zloss_86k,global_mmlu_full_de_public_relations,5,lm-eval,vllm,,acc,none,0.6545454545454545,0.045546196175410524,110 +SAMPLE — v2zloss_86k,global_mmlu_full_de_security_studies,5,lm-eval,vllm,,acc,none,0.6938775510204082,0.029504896454595975,245 +SAMPLE — v2zloss_86k,global_mmlu_full_de_social_sciences,5,lm-eval,vllm,yes,acc,none,0.710107247318817,0.008057638458083773, +SAMPLE — v2zloss_86k,global_mmlu_full_de_sociology,5,lm-eval,vllm,,acc,none,0.8109452736318408,0.02768691358801299,201 +SAMPLE — v2zloss_86k,global_mmlu_full_de_stem,5,lm-eval,vllm,yes,acc,none,0.5613701236917221,0.008574858891863943, +SAMPLE — v2zloss_86k,global_mmlu_full_de_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_mmlu_full_de_virology,5,lm-eval,vllm,,acc,none,0.4879518072289157,0.038913644958358196,166 +SAMPLE — v2zloss_86k,global_mmlu_full_de_world_religions,5,lm-eval,vllm,,acc,none,0.7543859649122807,0.033014059469872556,171 +SAMPLE — v2zloss_86k,global_mmlu_full_el,5,lm-eval,vllm,yes,acc,none,0.5729952998148412,0.004027900095236198, +SAMPLE — v2zloss_86k,global_mmlu_full_el_abstract_algebra,5,lm-eval,vllm,,acc,none,0.34,0.04760952285695233,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_anatomy,5,lm-eval,vllm,,acc,none,0.48148148148148145,0.043163785995113245,135 +SAMPLE — v2zloss_86k,global_mmlu_full_el_astronomy,5,lm-eval,vllm,,acc,none,0.6381578947368421,0.03910525752849724,152 +SAMPLE — v2zloss_86k,global_mmlu_full_el_business_ethics,5,lm-eval,vllm,,acc,none,0.59,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6,0.03015113445777636,265 +SAMPLE — v2zloss_86k,global_mmlu_full_el_college_biology,5,lm-eval,vllm,,acc,none,0.5208333333333334,0.04177578950739996,144 +SAMPLE — v2zloss_86k,global_mmlu_full_el_college_chemistry,5,lm-eval,vllm,,acc,none,0.5,0.050251890762960605,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_college_computer_science,5,lm-eval,vllm,,acc,none,0.51,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_college_mathematics,5,lm-eval,vllm,,acc,none,0.46,0.05009082659620332,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_college_medicine,5,lm-eval,vllm,,acc,none,0.5375722543352601,0.03801685104524462,173 +SAMPLE — v2zloss_86k,global_mmlu_full_el_college_physics,5,lm-eval,vllm,,acc,none,0.29411764705882354,0.04533838195929776,102 +SAMPLE — v2zloss_86k,global_mmlu_full_el_computer_security,5,lm-eval,vllm,,acc,none,0.63,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6297872340425532,0.031565646822367815,235 +SAMPLE — v2zloss_86k,global_mmlu_full_el_econometrics,5,lm-eval,vllm,,acc,none,0.4298245614035088,0.04657047260594963,114 +SAMPLE — v2zloss_86k,global_mmlu_full_el_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6137931034482759,0.04057324734419032,145 +SAMPLE — v2zloss_86k,global_mmlu_full_el_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5211640211640212,0.025728230952130723,378 +SAMPLE — v2zloss_86k,global_mmlu_full_el_formal_logic,5,lm-eval,vllm,,acc,none,0.5158730158730159,0.044698818540726104,126 +SAMPLE — v2zloss_86k,global_mmlu_full_el_global_facts,5,lm-eval,vllm,,acc,none,0.38,0.04878317312145634,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_biology,5,lm-eval,vllm,,acc,none,0.7,0.026069362295335054,310 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5073891625615764,0.03517603540361012,203 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.68,0.046882617226215076,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7333333333333333,0.034531318018854146,165 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_geography,5,lm-eval,vllm,,acc,none,0.7929292929292929,0.02886977846026699,198 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.7668393782383419,0.03051611137147603,193 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6153846153846154,0.02466674491518715,390 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.4,0.029869605095316977,270 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6596638655462185,0.030778057422931663,238 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_physics,5,lm-eval,vllm,,acc,none,0.4105960264900662,0.04016689594849927,151 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_psychology,5,lm-eval,vllm,,acc,none,0.728440366972477,0.01906909836319152,545 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_statistics,5,lm-eval,vllm,,acc,none,0.625,0.033016908987210894,216 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7009803921568627,0.032133257173736156,204 +SAMPLE — v2zloss_86k,global_mmlu_full_el_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7763713080168776,0.027123298205229966,237 +SAMPLE — v2zloss_86k,global_mmlu_full_el_human_aging,5,lm-eval,vllm,,acc,none,0.6502242152466368,0.03200736719484501,223 +SAMPLE — v2zloss_86k,global_mmlu_full_el_human_sexuality,5,lm-eval,vllm,,acc,none,0.6412213740458015,0.04206739313864908,131 +SAMPLE — v2zloss_86k,global_mmlu_full_el_humanities,5,lm-eval,vllm,yes,acc,none,0.5249734325185972,0.006939919069721227, +SAMPLE — v2zloss_86k,global_mmlu_full_el_international_law,5,lm-eval,vllm,,acc,none,0.7768595041322314,0.038007544752287334,121 +SAMPLE — v2zloss_86k,global_mmlu_full_el_jurisprudence,5,lm-eval,vllm,,acc,none,0.6481481481481481,0.04616631111801715,108 +SAMPLE — v2zloss_86k,global_mmlu_full_el_logical_fallacies,5,lm-eval,vllm,,acc,none,0.6503067484662577,0.0374666832547002,163 +SAMPLE — v2zloss_86k,global_mmlu_full_el_machine_learning,5,lm-eval,vllm,,acc,none,0.35714285714285715,0.045479609997643805,112 +SAMPLE — v2zloss_86k,global_mmlu_full_el_management,5,lm-eval,vllm,,acc,none,0.6893203883495146,0.04582124160161549,103 +SAMPLE — v2zloss_86k,global_mmlu_full_el_marketing,5,lm-eval,vllm,,acc,none,0.782051282051282,0.027046857630716698,234 +SAMPLE — v2zloss_86k,global_mmlu_full_el_medical_genetics,5,lm-eval,vllm,,acc,none,0.61,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_miscellaneous,5,lm-eval,vllm,,acc,none,0.6296296296296297,0.01726860756000581,783 +SAMPLE — v2zloss_86k,global_mmlu_full_el_moral_disputes,5,lm-eval,vllm,,acc,none,0.6647398843930635,0.025416003773165545,346 +SAMPLE — v2zloss_86k,global_mmlu_full_el_moral_scenarios,5,lm-eval,vllm,,acc,none,0.3106145251396648,0.01547651543800557,895 +SAMPLE — v2zloss_86k,global_mmlu_full_el_nutrition,5,lm-eval,vllm,,acc,none,0.6372549019607843,0.02753007844711037,306 +SAMPLE — v2zloss_86k,global_mmlu_full_el_other,5,lm-eval,vllm,yes,acc,none,0.592854843900869,0.008646598939700245, +SAMPLE — v2zloss_86k,global_mmlu_full_el_philosophy,5,lm-eval,vllm,,acc,none,0.6463022508038585,0.027155208103200844,311 +SAMPLE — v2zloss_86k,global_mmlu_full_el_prehistory,5,lm-eval,vllm,,acc,none,0.595679012345679,0.02730662529732765,324 +SAMPLE — v2zloss_86k,global_mmlu_full_el_professional_accounting,5,lm-eval,vllm,,acc,none,0.37943262411347517,0.028947338851614133,282 +SAMPLE — v2zloss_86k,global_mmlu_full_el_professional_law,5,lm-eval,vllm,,acc,none,0.4367666232073012,0.012667701919603853,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_el_professional_medicine,5,lm-eval,vllm,,acc,none,0.5698529411764706,0.030074971917302858,272 +SAMPLE — v2zloss_86k,global_mmlu_full_el_professional_psychology,5,lm-eval,vllm,,acc,none,0.5735294117647058,0.02000791273935941,612 +SAMPLE — v2zloss_86k,global_mmlu_full_el_public_relations,5,lm-eval,vllm,,acc,none,0.6545454545454545,0.045546196175410524,110 +SAMPLE — v2zloss_86k,global_mmlu_full_el_security_studies,5,lm-eval,vllm,,acc,none,0.689795918367347,0.029613459872484392,245 +SAMPLE — v2zloss_86k,global_mmlu_full_el_social_sciences,5,lm-eval,vllm,yes,acc,none,0.6681832954176146,0.00835800451777467, +SAMPLE — v2zloss_86k,global_mmlu_full_el_sociology,5,lm-eval,vllm,,acc,none,0.7711442786069652,0.0297052840567725,201 +SAMPLE — v2zloss_86k,global_mmlu_full_el_stem,5,lm-eval,vllm,yes,acc,none,0.5321915635902316,0.008684523180372529, +SAMPLE — v2zloss_86k,global_mmlu_full_el_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.77,0.042295258468165065,100 +SAMPLE — v2zloss_86k,global_mmlu_full_el_virology,5,lm-eval,vllm,,acc,none,0.5,0.03892494720807614,166 +SAMPLE — v2zloss_86k,global_mmlu_full_el_world_religions,5,lm-eval,vllm,,acc,none,0.672514619883041,0.03599335771456024,171 +SAMPLE — v2zloss_86k,global_mmlu_full_en,5,lm-eval,vllm,yes,acc,none,0.6611593790058397,0.0038114285053581073, +SAMPLE — v2zloss_86k,global_mmlu_full_en_abstract_algebra,5,lm-eval,vllm,,acc,none,0.39,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_anatomy,5,lm-eval,vllm,,acc,none,0.562962962962963,0.042849586397534056,135 +SAMPLE — v2zloss_86k,global_mmlu_full_en_astronomy,5,lm-eval,vllm,,acc,none,0.7171052631578947,0.03665349695640767,152 +SAMPLE — v2zloss_86k,global_mmlu_full_en_business_ethics,5,lm-eval,vllm,,acc,none,0.65,0.04793724854411023,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6943396226415094,0.028353298073322628,265 +SAMPLE — v2zloss_86k,global_mmlu_full_en_college_biology,5,lm-eval,vllm,,acc,none,0.75,0.03621034121889507,144 +SAMPLE — v2zloss_86k,global_mmlu_full_en_college_chemistry,5,lm-eval,vllm,,acc,none,0.52,0.05021167315686783,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_college_computer_science,5,lm-eval,vllm,,acc,none,0.6,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_college_mathematics,5,lm-eval,vllm,,acc,none,0.41,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_college_medicine,5,lm-eval,vllm,,acc,none,0.6358381502890174,0.03669072477416912,173 +SAMPLE — v2zloss_86k,global_mmlu_full_en_college_physics,5,lm-eval,vllm,,acc,none,0.4117647058823529,0.048971049527263624,102 +SAMPLE — v2zloss_86k,global_mmlu_full_en_computer_security,5,lm-eval,vllm,,acc,none,0.73,0.04461960433384737,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_conceptual_physics,5,lm-eval,vllm,,acc,none,0.7489361702127659,0.02834696377716252,235 +SAMPLE — v2zloss_86k,global_mmlu_full_en_econometrics,5,lm-eval,vllm,,acc,none,0.543859649122807,0.04685473041907789,114 +SAMPLE — v2zloss_86k,global_mmlu_full_en_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6827586206896552,0.03878352372138618,145 +SAMPLE — v2zloss_86k,global_mmlu_full_en_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5740740740740741,0.02546714904546955,378 +SAMPLE — v2zloss_86k,global_mmlu_full_en_formal_logic,5,lm-eval,vllm,,acc,none,0.5873015873015873,0.04403438954768178,126 +SAMPLE — v2zloss_86k,global_mmlu_full_en_global_facts,5,lm-eval,vllm,,acc,none,0.37,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_biology,5,lm-eval,vllm,,acc,none,0.8193548387096774,0.021886178567172513,310 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.6157635467980296,0.03422398565657553,203 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.75,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7818181818181819,0.032250781083062896,165 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_geography,5,lm-eval,vllm,,acc,none,0.8232323232323232,0.027178752639044908,198 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.9015544041450777,0.021500249576033463,193 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.7333333333333333,0.022421273612923794,390 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.45925925925925926,0.030384169232350797,270 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.7941176470588235,0.026265024608275907,238 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_physics,5,lm-eval,vllm,,acc,none,0.46357615894039733,0.04071636065944211,151 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_psychology,5,lm-eval,vllm,,acc,none,0.8642201834862385,0.014686907556339952,545 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_statistics,5,lm-eval,vllm,,acc,none,0.6435185185185185,0.03266478331527272,216 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7941176470588235,0.028379449451588684,204 +SAMPLE — v2zloss_86k,global_mmlu_full_en_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7974683544303798,0.02616056824660147,237 +SAMPLE — v2zloss_86k,global_mmlu_full_en_human_aging,5,lm-eval,vllm,,acc,none,0.7130044843049327,0.030360379710291947,223 +SAMPLE — v2zloss_86k,global_mmlu_full_en_human_sexuality,5,lm-eval,vllm,,acc,none,0.7862595419847328,0.03595461611774691,131 +SAMPLE — v2zloss_86k,global_mmlu_full_en_humanities,5,lm-eval,vllm,yes,acc,none,0.6055260361317747,0.006774229995444949, +SAMPLE — v2zloss_86k,global_mmlu_full_en_international_law,5,lm-eval,vllm,,acc,none,0.7933884297520661,0.03695980128098826,121 +SAMPLE — v2zloss_86k,global_mmlu_full_en_jurisprudence,5,lm-eval,vllm,,acc,none,0.7592592592592593,0.041331194402438355,108 +SAMPLE — v2zloss_86k,global_mmlu_full_en_logical_fallacies,5,lm-eval,vllm,,acc,none,0.7852760736196319,0.032262193772867716,163 +SAMPLE — v2zloss_86k,global_mmlu_full_en_machine_learning,5,lm-eval,vllm,,acc,none,0.42857142857142855,0.04697113923010208,112 +SAMPLE — v2zloss_86k,global_mmlu_full_en_management,5,lm-eval,vllm,,acc,none,0.7961165048543689,0.0398913985953177,103 +SAMPLE — v2zloss_86k,global_mmlu_full_en_marketing,5,lm-eval,vllm,,acc,none,0.8846153846153846,0.0209301931851793,234 +SAMPLE — v2zloss_86k,global_mmlu_full_en_medical_genetics,5,lm-eval,vllm,,acc,none,0.68,0.046882617226215076,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_miscellaneous,5,lm-eval,vllm,,acc,none,0.776500638569604,0.014897235229450705,783 +SAMPLE — v2zloss_86k,global_mmlu_full_en_moral_disputes,5,lm-eval,vllm,,acc,none,0.7398843930635838,0.02361867831006933,346 +SAMPLE — v2zloss_86k,global_mmlu_full_en_moral_scenarios,5,lm-eval,vllm,,acc,none,0.394413407821229,0.016345386762104078,895 +SAMPLE — v2zloss_86k,global_mmlu_full_en_nutrition,5,lm-eval,vllm,,acc,none,0.7287581699346405,0.025457756696667815,306 +SAMPLE — v2zloss_86k,global_mmlu_full_en_other,5,lm-eval,vllm,yes,acc,none,0.6881235918892823,0.008039794937220505, +SAMPLE — v2zloss_86k,global_mmlu_full_en_philosophy,5,lm-eval,vllm,,acc,none,0.7459807073954984,0.024723861504771655,311 +SAMPLE — v2zloss_86k,global_mmlu_full_en_prehistory,5,lm-eval,vllm,,acc,none,0.7253086419753086,0.024836057868294684,324 +SAMPLE — v2zloss_86k,global_mmlu_full_en_professional_accounting,5,lm-eval,vllm,,acc,none,0.475177304964539,0.0297907192438297,282 +SAMPLE — v2zloss_86k,global_mmlu_full_en_professional_law,5,lm-eval,vllm,,acc,none,0.5071707953063885,0.012768922739553434,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_en_professional_medicine,5,lm-eval,vllm,,acc,none,0.6360294117647058,0.029227192460031966,272 +SAMPLE — v2zloss_86k,global_mmlu_full_en_professional_psychology,5,lm-eval,vllm,,acc,none,0.6944444444444444,0.01863559403442406,612 +SAMPLE — v2zloss_86k,global_mmlu_full_en_public_relations,5,lm-eval,vllm,,acc,none,0.6909090909090909,0.04426294648200096,110 +SAMPLE — v2zloss_86k,global_mmlu_full_en_security_studies,5,lm-eval,vllm,,acc,none,0.7142857142857143,0.028920583220675568,245 +SAMPLE — v2zloss_86k,global_mmlu_full_en_social_sciences,5,lm-eval,vllm,yes,acc,none,0.7702307442313943,0.007451560182398421, +SAMPLE — v2zloss_86k,global_mmlu_full_en_sociology,5,lm-eval,vllm,,acc,none,0.8208955223880597,0.027113286753111865,201 +SAMPLE — v2zloss_86k,global_mmlu_full_en_stem,5,lm-eval,vllm,yes,acc,none,0.6111639708214399,0.008384516376716625, +SAMPLE — v2zloss_86k,global_mmlu_full_en_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.81,0.039427724440366255,100 +SAMPLE — v2zloss_86k,global_mmlu_full_en_virology,5,lm-eval,vllm,,acc,none,0.5301204819277109,0.03885425420866771,166 +SAMPLE — v2zloss_86k,global_mmlu_full_en_world_religions,5,lm-eval,vllm,,acc,none,0.7894736842105263,0.03126781714663182,171 +SAMPLE — v2zloss_86k,global_mmlu_full_es,5,lm-eval,vllm,yes,acc,none,0.6168636946303945,0.003904489772341443, +SAMPLE — v2zloss_86k,global_mmlu_full_es_abstract_algebra,5,lm-eval,vllm,,acc,none,0.41,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_anatomy,5,lm-eval,vllm,,acc,none,0.5037037037037037,0.04319223625811333,135 +SAMPLE — v2zloss_86k,global_mmlu_full_es_astronomy,5,lm-eval,vllm,,acc,none,0.6973684210526315,0.03738520676119667,152 +SAMPLE — v2zloss_86k,global_mmlu_full_es_business_ethics,5,lm-eval,vllm,,acc,none,0.65,0.04793724854411023,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6264150943396226,0.029773082713319812,265 +SAMPLE — v2zloss_86k,global_mmlu_full_es_college_biology,5,lm-eval,vllm,,acc,none,0.6319444444444444,0.04032999053960717,144 +SAMPLE — v2zloss_86k,global_mmlu_full_es_college_chemistry,5,lm-eval,vllm,,acc,none,0.54,0.05009082659620331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_college_computer_science,5,lm-eval,vllm,,acc,none,0.59,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_college_mathematics,5,lm-eval,vllm,,acc,none,0.4,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_college_medicine,5,lm-eval,vllm,,acc,none,0.5953757225433526,0.037424611938872455,173 +SAMPLE — v2zloss_86k,global_mmlu_full_es_college_physics,5,lm-eval,vllm,,acc,none,0.4117647058823529,0.048971049527263624,102 +SAMPLE — v2zloss_86k,global_mmlu_full_es_computer_security,5,lm-eval,vllm,,acc,none,0.74,0.0440844002276808,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6723404255319149,0.030683020843230994,235 +SAMPLE — v2zloss_86k,global_mmlu_full_es_econometrics,5,lm-eval,vllm,,acc,none,0.49122807017543857,0.0470288043204961,114 +SAMPLE — v2zloss_86k,global_mmlu_full_es_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6275862068965518,0.0402873153294756,145 +SAMPLE — v2zloss_86k,global_mmlu_full_es_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5661375661375662,0.02552503438247484,378 +SAMPLE — v2zloss_86k,global_mmlu_full_es_formal_logic,5,lm-eval,vllm,,acc,none,0.6031746031746031,0.04375888492727054,126 +SAMPLE — v2zloss_86k,global_mmlu_full_es_global_facts,5,lm-eval,vllm,,acc,none,0.39,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_biology,5,lm-eval,vllm,,acc,none,0.7774193548387097,0.02366421667164245,310 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5615763546798029,0.034912078574865175,203 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_european_history,5,lm-eval,vllm,,acc,none,0.793939393939394,0.03158415324047706,165 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_geography,5,lm-eval,vllm,,acc,none,0.7878787878787878,0.029126522834586777,198 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.8134715025906736,0.0281120912101175,193 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6538461538461539,0.024121125416941242,390 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.3851851851851852,0.029670906124630803,270 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.7058823529411765,0.029597329730978096,238 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_physics,5,lm-eval,vllm,,acc,none,0.44370860927152317,0.04056527902281736,151 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_psychology,5,lm-eval,vllm,,acc,none,0.8055045871559633,0.016970289090458102,545 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_statistics,5,lm-eval,vllm,,acc,none,0.5694444444444444,0.03376922151252338,216 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7598039215686274,0.029983733055913633,204 +SAMPLE — v2zloss_86k,global_mmlu_full_es_high_school_world_history,5,lm-eval,vllm,,acc,none,0.810126582278481,0.025530100460233456,237 +SAMPLE — v2zloss_86k,global_mmlu_full_es_human_aging,5,lm-eval,vllm,,acc,none,0.6636771300448431,0.031708824268455046,223 +SAMPLE — v2zloss_86k,global_mmlu_full_es_human_sexuality,5,lm-eval,vllm,,acc,none,0.7404580152671756,0.03844876139785267,131 +SAMPLE — v2zloss_86k,global_mmlu_full_es_humanities,5,lm-eval,vllm,yes,acc,none,0.5566418703506908,0.006711927682249389, +SAMPLE — v2zloss_86k,global_mmlu_full_es_international_law,5,lm-eval,vllm,,acc,none,0.8099173553719008,0.03581796951709283,121 +SAMPLE — v2zloss_86k,global_mmlu_full_es_jurisprudence,5,lm-eval,vllm,,acc,none,0.6574074074074074,0.04587904741301815,108 +SAMPLE — v2zloss_86k,global_mmlu_full_es_logical_fallacies,5,lm-eval,vllm,,acc,none,0.7361963190184049,0.03462419931615622,163 +SAMPLE — v2zloss_86k,global_mmlu_full_es_machine_learning,5,lm-eval,vllm,,acc,none,0.4375,0.04708567521880525,112 +SAMPLE — v2zloss_86k,global_mmlu_full_es_management,5,lm-eval,vllm,,acc,none,0.7766990291262136,0.041235531898914324,103 +SAMPLE — v2zloss_86k,global_mmlu_full_es_marketing,5,lm-eval,vllm,,acc,none,0.8034188034188035,0.026035386098951282,234 +SAMPLE — v2zloss_86k,global_mmlu_full_es_medical_genetics,5,lm-eval,vllm,,acc,none,0.63,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_miscellaneous,5,lm-eval,vllm,,acc,none,0.722860791826309,0.01600563629412252,783 +SAMPLE — v2zloss_86k,global_mmlu_full_es_moral_disputes,5,lm-eval,vllm,,acc,none,0.708092485549133,0.02447699407624737,346 +SAMPLE — v2zloss_86k,global_mmlu_full_es_moral_scenarios,5,lm-eval,vllm,,acc,none,0.2759776536312849,0.01495010300247546,895 +SAMPLE — v2zloss_86k,global_mmlu_full_es_nutrition,5,lm-eval,vllm,,acc,none,0.6928104575163399,0.02641560191438905,306 +SAMPLE — v2zloss_86k,global_mmlu_full_es_other,5,lm-eval,vllm,yes,acc,none,0.6485355648535565,0.008383886513954076, +SAMPLE — v2zloss_86k,global_mmlu_full_es_philosophy,5,lm-eval,vllm,,acc,none,0.7234726688102894,0.025403832978179643,311 +SAMPLE — v2zloss_86k,global_mmlu_full_es_prehistory,5,lm-eval,vllm,,acc,none,0.6975308641975309,0.025557653981868003,324 +SAMPLE — v2zloss_86k,global_mmlu_full_es_professional_accounting,5,lm-eval,vllm,,acc,none,0.4574468085106383,0.029719281272236827,282 +SAMPLE — v2zloss_86k,global_mmlu_full_es_professional_law,5,lm-eval,vllm,,acc,none,0.45436766623207303,0.012716941720734844,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_es_professional_medicine,5,lm-eval,vllm,,acc,none,0.6176470588235294,0.029520095697687675,272 +SAMPLE — v2zloss_86k,global_mmlu_full_es_professional_psychology,5,lm-eval,vllm,,acc,none,0.6519607843137255,0.019270998708223946,612 +SAMPLE — v2zloss_86k,global_mmlu_full_es_public_relations,5,lm-eval,vllm,,acc,none,0.7,0.04389311454644288,110 +SAMPLE — v2zloss_86k,global_mmlu_full_es_security_studies,5,lm-eval,vllm,,acc,none,0.7142857142857143,0.028920583220675568,245 +SAMPLE — v2zloss_86k,global_mmlu_full_es_social_sciences,5,lm-eval,vllm,yes,acc,none,0.7214819629509263,0.007969951890023, +SAMPLE — v2zloss_86k,global_mmlu_full_es_sociology,5,lm-eval,vllm,,acc,none,0.7910447761194029,0.02874829893172869,201 +SAMPLE — v2zloss_86k,global_mmlu_full_es_stem,5,lm-eval,vllm,yes,acc,none,0.5734221376466857,0.008562686591669748, +SAMPLE — v2zloss_86k,global_mmlu_full_es_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.82,0.03861229196653691,100 +SAMPLE — v2zloss_86k,global_mmlu_full_es_virology,5,lm-eval,vllm,,acc,none,0.5301204819277109,0.03885425420866771,166 +SAMPLE — v2zloss_86k,global_mmlu_full_es_world_religions,5,lm-eval,vllm,,acc,none,0.7953216374269005,0.03094445977853322,171 +SAMPLE — v2zloss_86k,global_mmlu_full_fr,5,lm-eval,vllm,yes,acc,none,0.6172197692636376,0.003931419318600634, +SAMPLE — v2zloss_86k,global_mmlu_full_fr_abstract_algebra,5,lm-eval,vllm,,acc,none,0.36,0.048241815132442176,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_anatomy,5,lm-eval,vllm,,acc,none,0.5851851851851851,0.04256193767901412,135 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_astronomy,5,lm-eval,vllm,,acc,none,0.7105263157894737,0.03690677986137281,152 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_business_ethics,5,lm-eval,vllm,,acc,none,0.59,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6754716981132075,0.02881561571343209,265 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_college_biology,5,lm-eval,vllm,,acc,none,0.625,0.04048439222695598,144 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_college_chemistry,5,lm-eval,vllm,,acc,none,0.51,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_college_computer_science,5,lm-eval,vllm,,acc,none,0.49,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_college_mathematics,5,lm-eval,vllm,,acc,none,0.45,0.05,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_college_medicine,5,lm-eval,vllm,,acc,none,0.630057803468208,0.036812296333943256,173 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_college_physics,5,lm-eval,vllm,,acc,none,0.4117647058823529,0.048971049527263624,102 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_computer_security,5,lm-eval,vllm,,acc,none,0.76,0.04292346959909278,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6893617021276596,0.03025123757921315,235 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_econometrics,5,lm-eval,vllm,,acc,none,0.47368421052631576,0.046970851366478654,114 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6206896551724138,0.040434618619167494,145 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.544973544973545,0.025646928361049343,378 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_formal_logic,5,lm-eval,vllm,,acc,none,0.49206349206349204,0.04471572536294351,126 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_global_facts,5,lm-eval,vllm,,acc,none,0.34,0.04760952285695233,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_biology,5,lm-eval,vllm,,acc,none,0.8032258064516129,0.02261640942074208,310 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5467980295566502,0.035025446508458756,203 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7636363636363637,0.03317505930009174,165 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_geography,5,lm-eval,vllm,,acc,none,0.7676767676767676,0.030088629490217438,198 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.8134715025906736,0.0281120912101175,193 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6384615384615384,0.02435958146539706,390 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.4,0.029869605095316977,270 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6974789915966386,0.029837962388291957,238 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_physics,5,lm-eval,vllm,,acc,none,0.4503311258278146,0.040622900186837764,151 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_psychology,5,lm-eval,vllm,,acc,none,0.8055045871559633,0.016970289090458102,545 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_statistics,5,lm-eval,vllm,,acc,none,0.6111111111111112,0.03324708911809121,216 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7647058823529411,0.02977177522814563,204 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7510548523206751,0.028146970599422647,237 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_human_aging,5,lm-eval,vllm,,acc,none,0.6771300448430493,0.031381476375754995,223 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_human_sexuality,5,lm-eval,vllm,,acc,none,0.7251908396946565,0.03915345408847834,131 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_humanities,5,lm-eval,vllm,yes,acc,none,0.5617428267800213,0.0068815044125228614, +SAMPLE — v2zloss_86k,global_mmlu_full_fr_international_law,5,lm-eval,vllm,,acc,none,0.8347107438016529,0.03390780612972774,121 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_jurisprudence,5,lm-eval,vllm,,acc,none,0.7129629629629629,0.0437331304091476,108 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_logical_fallacies,5,lm-eval,vllm,,acc,none,0.6441717791411042,0.03761521380046734,163 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_machine_learning,5,lm-eval,vllm,,acc,none,0.45535714285714285,0.04726835553719097,112 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_management,5,lm-eval,vllm,,acc,none,0.7669902912621359,0.041858325989283136,103 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_marketing,5,lm-eval,vllm,,acc,none,0.8034188034188035,0.026035386098951282,234 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_medical_genetics,5,lm-eval,vllm,,acc,none,0.61,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_miscellaneous,5,lm-eval,vllm,,acc,none,0.7343550446998723,0.015794302487888694,783 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_moral_disputes,5,lm-eval,vllm,,acc,none,0.6936416184971098,0.024818350129436558,346 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_moral_scenarios,5,lm-eval,vllm,,acc,none,0.36312849162011174,0.0160837499868536,895 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_nutrition,5,lm-eval,vllm,,acc,none,0.7156862745098039,0.025829163272757385,306 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_other,5,lm-eval,vllm,yes,acc,none,0.6475700032185387,0.008306332295353705, +SAMPLE — v2zloss_86k,global_mmlu_full_fr_philosophy,5,lm-eval,vllm,,acc,none,0.6816720257234726,0.026457225067810987,311 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_prehistory,5,lm-eval,vllm,,acc,none,0.691358024691358,0.025702640260603798,324 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_professional_accounting,5,lm-eval,vllm,,acc,none,0.40425531914893614,0.029275532159704753,282 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_professional_law,5,lm-eval,vllm,,acc,none,0.45827900912646674,0.012725701656953702,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_professional_medicine,5,lm-eval,vllm,,acc,none,0.5882352941176471,0.02989616303312549,272 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_professional_psychology,5,lm-eval,vllm,,acc,none,0.6241830065359477,0.019594021136577423,612 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_public_relations,5,lm-eval,vllm,,acc,none,0.6636363636363637,0.04525393596302509,110 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_security_studies,5,lm-eval,vllm,,acc,none,0.7306122448979592,0.028401252029022956,245 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_social_sciences,5,lm-eval,vllm,yes,acc,none,0.7107572310692233,0.008036568268468743, +SAMPLE — v2zloss_86k,global_mmlu_full_fr_sociology,5,lm-eval,vllm,,acc,none,0.7960199004975125,0.02849317624532607,201 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_stem,5,lm-eval,vllm,yes,acc,none,0.578813828100222,0.008520854022932062, +SAMPLE — v2zloss_86k,global_mmlu_full_fr_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.81,0.039427724440366255,100 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_virology,5,lm-eval,vllm,,acc,none,0.5060240963855421,0.03892212195333041,166 +SAMPLE — v2zloss_86k,global_mmlu_full_fr_world_religions,5,lm-eval,vllm,,acc,none,0.783625730994152,0.03158149539338735,171 +SAMPLE — v2zloss_86k,global_mmlu_full_it,5,lm-eval,vllm,yes,acc,none,0.6120922945449366,0.003922384617500552, +SAMPLE — v2zloss_86k,global_mmlu_full_it_abstract_algebra,5,lm-eval,vllm,,acc,none,0.41,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_anatomy,5,lm-eval,vllm,,acc,none,0.5777777777777777,0.042667634040995855,135 +SAMPLE — v2zloss_86k,global_mmlu_full_it_astronomy,5,lm-eval,vllm,,acc,none,0.6578947368421053,0.0386073159931609,152 +SAMPLE — v2zloss_86k,global_mmlu_full_it_business_ethics,5,lm-eval,vllm,,acc,none,0.63,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6679245283018868,0.028985455652334374,265 +SAMPLE — v2zloss_86k,global_mmlu_full_it_college_biology,5,lm-eval,vllm,,acc,none,0.6458333333333334,0.03999411135753542,144 +SAMPLE — v2zloss_86k,global_mmlu_full_it_college_chemistry,5,lm-eval,vllm,,acc,none,0.49,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_college_computer_science,5,lm-eval,vllm,,acc,none,0.45,0.05,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_college_mathematics,5,lm-eval,vllm,,acc,none,0.39,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_college_medicine,5,lm-eval,vllm,,acc,none,0.6127167630057804,0.03714325906302067,173 +SAMPLE — v2zloss_86k,global_mmlu_full_it_college_physics,5,lm-eval,vllm,,acc,none,0.4117647058823529,0.048971049527263624,102 +SAMPLE — v2zloss_86k,global_mmlu_full_it_computer_security,5,lm-eval,vllm,,acc,none,0.74,0.0440844002276808,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6638297872340425,0.03088161852067694,235 +SAMPLE — v2zloss_86k,global_mmlu_full_it_econometrics,5,lm-eval,vllm,,acc,none,0.5087719298245614,0.0470288043204961,114 +SAMPLE — v2zloss_86k,global_mmlu_full_it_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6551724137931034,0.03960933549451213,145 +SAMPLE — v2zloss_86k,global_mmlu_full_it_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.544973544973545,0.025646928361049343,378 +SAMPLE — v2zloss_86k,global_mmlu_full_it_formal_logic,5,lm-eval,vllm,,acc,none,0.5396825396825397,0.04458029125470974,126 +SAMPLE — v2zloss_86k,global_mmlu_full_it_global_facts,5,lm-eval,vllm,,acc,none,0.36,0.048241815132442176,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_biology,5,lm-eval,vllm,,acc,none,0.7838709677419354,0.023415293433568577,310 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.541871921182266,0.035056301407857406,203 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7636363636363637,0.03317505930009174,165 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_geography,5,lm-eval,vllm,,acc,none,0.8131313131313131,0.02777253333421894,198 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.8290155440414507,0.02717121368316455,193 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6333333333333333,0.024433016466052476,390 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.4074074074074074,0.02995824925008211,270 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6890756302521008,0.030066761582977983,238 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_physics,5,lm-eval,vllm,,acc,none,0.4370860927152318,0.040500357222306375,151 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_psychology,5,lm-eval,vllm,,acc,none,0.8036697247706422,0.017030719339154426,545 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_statistics,5,lm-eval,vllm,,acc,none,0.5972222222222222,0.0334488738299786,216 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7696078431372549,0.029554292605695046,204 +SAMPLE — v2zloss_86k,global_mmlu_full_it_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7890295358649789,0.0265583725026619,237 +SAMPLE — v2zloss_86k,global_mmlu_full_it_human_aging,5,lm-eval,vllm,,acc,none,0.6502242152466368,0.03200736719484501,223 +SAMPLE — v2zloss_86k,global_mmlu_full_it_human_sexuality,5,lm-eval,vllm,,acc,none,0.7099236641221374,0.03980066246467765,131 +SAMPLE — v2zloss_86k,global_mmlu_full_it_humanities,5,lm-eval,vllm,yes,acc,none,0.5515409139213603,0.006779539286821161, +SAMPLE — v2zloss_86k,global_mmlu_full_it_international_law,5,lm-eval,vllm,,acc,none,0.8512396694214877,0.03248470083807198,121 +SAMPLE — v2zloss_86k,global_mmlu_full_it_jurisprudence,5,lm-eval,vllm,,acc,none,0.6759259259259259,0.04524596007030053,108 +SAMPLE — v2zloss_86k,global_mmlu_full_it_logical_fallacies,5,lm-eval,vllm,,acc,none,0.7177914110429447,0.03536117886664741,163 +SAMPLE — v2zloss_86k,global_mmlu_full_it_machine_learning,5,lm-eval,vllm,,acc,none,0.4732142857142857,0.047389751192741525,112 +SAMPLE — v2zloss_86k,global_mmlu_full_it_management,5,lm-eval,vllm,,acc,none,0.7572815533980582,0.04245022486384496,103 +SAMPLE — v2zloss_86k,global_mmlu_full_it_marketing,5,lm-eval,vllm,,acc,none,0.8034188034188035,0.026035386098951282,234 +SAMPLE — v2zloss_86k,global_mmlu_full_it_medical_genetics,5,lm-eval,vllm,,acc,none,0.67,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_miscellaneous,5,lm-eval,vllm,,acc,none,0.7266922094508301,0.015936681062628598,783 +SAMPLE — v2zloss_86k,global_mmlu_full_it_moral_disputes,5,lm-eval,vllm,,acc,none,0.6907514450867052,0.02488314057007179,346 +SAMPLE — v2zloss_86k,global_mmlu_full_it_moral_scenarios,5,lm-eval,vllm,,acc,none,0.3139664804469274,0.015521923933523658,895 +SAMPLE — v2zloss_86k,global_mmlu_full_it_nutrition,5,lm-eval,vllm,,acc,none,0.6895424836601307,0.026493033225145922,306 +SAMPLE — v2zloss_86k,global_mmlu_full_it_other,5,lm-eval,vllm,yes,acc,none,0.646604441583521,0.008369082859576006, +SAMPLE — v2zloss_86k,global_mmlu_full_it_philosophy,5,lm-eval,vllm,,acc,none,0.7138263665594855,0.02567025924218898,311 +SAMPLE — v2zloss_86k,global_mmlu_full_it_prehistory,5,lm-eval,vllm,,acc,none,0.6851851851851852,0.025842248700902248,324 +SAMPLE — v2zloss_86k,global_mmlu_full_it_professional_accounting,5,lm-eval,vllm,,acc,none,0.44680851063829785,0.02965823509766695,282 +SAMPLE — v2zloss_86k,global_mmlu_full_it_professional_law,5,lm-eval,vllm,,acc,none,0.43415906127770537,0.012659033237067258,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_it_professional_medicine,5,lm-eval,vllm,,acc,none,0.5808823529411765,0.02997280717046463,272 +SAMPLE — v2zloss_86k,global_mmlu_full_it_professional_psychology,5,lm-eval,vllm,,acc,none,0.6339869281045751,0.019488025745529686,612 +SAMPLE — v2zloss_86k,global_mmlu_full_it_public_relations,5,lm-eval,vllm,,acc,none,0.6090909090909091,0.046737523336702363,110 +SAMPLE — v2zloss_86k,global_mmlu_full_it_security_studies,5,lm-eval,vllm,,acc,none,0.7020408163265306,0.029279567411065615,245 +SAMPLE — v2zloss_86k,global_mmlu_full_it_social_sciences,5,lm-eval,vllm,yes,acc,none,0.7117322066948326,0.008024886156470137, +SAMPLE — v2zloss_86k,global_mmlu_full_it_sociology,5,lm-eval,vllm,,acc,none,0.8009950248756219,0.02823136509275842,201 +SAMPLE — v2zloss_86k,global_mmlu_full_it_stem,5,lm-eval,vllm,yes,acc,none,0.5712020298128766,0.008571339350722792, +SAMPLE — v2zloss_86k,global_mmlu_full_it_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.81,0.039427724440366255,100 +SAMPLE — v2zloss_86k,global_mmlu_full_it_virology,5,lm-eval,vllm,,acc,none,0.5120481927710844,0.038913644958358196,166 +SAMPLE — v2zloss_86k,global_mmlu_full_it_world_religions,5,lm-eval,vllm,,acc,none,0.783625730994152,0.03158149539338735,171 +SAMPLE — v2zloss_86k,global_mmlu_full_lt,5,lm-eval,vllm,yes,acc,none,0.5689360489958696,0.004026738712559823, +SAMPLE — v2zloss_86k,global_mmlu_full_lt_abstract_algebra,5,lm-eval,vllm,,acc,none,0.33,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_anatomy,5,lm-eval,vllm,,acc,none,0.5259259259259259,0.043135316967505756,135 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_astronomy,5,lm-eval,vllm,,acc,none,0.6973684210526315,0.03738520676119667,152 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_business_ethics,5,lm-eval,vllm,,acc,none,0.57,0.049756985195624305,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6452830188679245,0.029445175328199655,265 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_college_biology,5,lm-eval,vllm,,acc,none,0.5763888888888888,0.041321250197233685,144 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_college_chemistry,5,lm-eval,vllm,,acc,none,0.51,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_college_computer_science,5,lm-eval,vllm,,acc,none,0.49,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_college_mathematics,5,lm-eval,vllm,,acc,none,0.42,0.04960449637488583,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_college_medicine,5,lm-eval,vllm,,acc,none,0.5491329479768786,0.03794012674697029,173 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_college_physics,5,lm-eval,vllm,,acc,none,0.3333333333333333,0.04690650298201946,102 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_computer_security,5,lm-eval,vllm,,acc,none,0.63,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6127659574468085,0.031843892653395246,235 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_econometrics,5,lm-eval,vllm,,acc,none,0.41228070175438597,0.046306532033665936,114 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_electrical_engineering,5,lm-eval,vllm,,acc,none,0.5655172413793104,0.041307408795554966,145 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5608465608465608,0.025559920550531027,378 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_formal_logic,5,lm-eval,vllm,,acc,none,0.5317460317460317,0.04463112720677173,126 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_global_facts,5,lm-eval,vllm,,acc,none,0.41,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_biology,5,lm-eval,vllm,,acc,none,0.6838709677419355,0.026450874489042778,310 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5073891625615764,0.03517603540361012,203 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.69,0.046482319871173176,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7272727272727273,0.03477691162163663,165 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_geography,5,lm-eval,vllm,,acc,none,0.7727272727272727,0.029857515673386438,198 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.7409326424870466,0.03161877917935408,193 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.5974358974358974,0.02486499515976774,390 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.37777777777777777,0.029560707392465774,270 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6428571428571429,0.031124619309328222,238 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_physics,5,lm-eval,vllm,,acc,none,0.423841059602649,0.04034846678603395,151 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_psychology,5,lm-eval,vllm,,acc,none,0.726605504587156,0.01910929984609821,545 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_statistics,5,lm-eval,vllm,,acc,none,0.5879629629629629,0.03356787758160834,216 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_us_history,5,lm-eval,vllm,,acc,none,0.6813725490196079,0.032702871814820796,204 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_high_school_world_history,5,lm-eval,vllm,,acc,none,0.759493670886076,0.02782078198114972,237 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_human_aging,5,lm-eval,vllm,,acc,none,0.6233183856502242,0.03252113489929187,223 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_human_sexuality,5,lm-eval,vllm,,acc,none,0.7251908396946565,0.03915345408847834,131 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_humanities,5,lm-eval,vllm,yes,acc,none,0.5190223166843784,0.006900248150434389, +SAMPLE — v2zloss_86k,global_mmlu_full_lt_international_law,5,lm-eval,vllm,,acc,none,0.7851239669421488,0.037494924487096966,121 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_jurisprudence,5,lm-eval,vllm,,acc,none,0.6944444444444444,0.044531975073749855,108 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_logical_fallacies,5,lm-eval,vllm,,acc,none,0.588957055214724,0.03865697853785365,163 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_machine_learning,5,lm-eval,vllm,,acc,none,0.38392857142857145,0.04616143075028549,112 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_management,5,lm-eval,vllm,,acc,none,0.6990291262135923,0.045416094465039504,103 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_marketing,5,lm-eval,vllm,,acc,none,0.7692307692307693,0.0276019213814176,234 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_medical_genetics,5,lm-eval,vllm,,acc,none,0.54,0.05009082659620331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_miscellaneous,5,lm-eval,vllm,,acc,none,0.6245210727969349,0.017316613197182775,783 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_moral_disputes,5,lm-eval,vllm,,acc,none,0.6358381502890174,0.025906632631016134,346 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_moral_scenarios,5,lm-eval,vllm,,acc,none,0.30614525139664805,0.015414494487903146,895 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_nutrition,5,lm-eval,vllm,,acc,none,0.6633986928104575,0.0270579746244944,306 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_other,5,lm-eval,vllm,yes,acc,none,0.5896363051174767,0.008640956347153313, +SAMPLE — v2zloss_86k,global_mmlu_full_lt_philosophy,5,lm-eval,vllm,,acc,none,0.6752411575562701,0.02659678228769707,311 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_prehistory,5,lm-eval,vllm,,acc,none,0.6203703703703703,0.027002521034516468,324 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_professional_accounting,5,lm-eval,vllm,,acc,none,0.35815602836879434,0.028602085862759398,282 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_professional_law,5,lm-eval,vllm,,acc,none,0.41264667535853977,0.012573836633799015,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_professional_medicine,5,lm-eval,vllm,,acc,none,0.5588235294117647,0.030161911930767053,272 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_professional_psychology,5,lm-eval,vllm,,acc,none,0.565359477124183,0.020054269200726567,612 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_public_relations,5,lm-eval,vllm,,acc,none,0.6272727272727273,0.04631381319425461,110 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_security_studies,5,lm-eval,vllm,,acc,none,0.6816326530612244,0.029822533793982028,245 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_social_sciences,5,lm-eval,vllm,yes,acc,none,0.6581085472863178,0.008423529370086509, +SAMPLE — v2zloss_86k,global_mmlu_full_lt_sociology,5,lm-eval,vllm,,acc,none,0.7263681592039801,0.03152439186555401,201 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_stem,5,lm-eval,vllm,yes,acc,none,0.5359974627339043,0.008689480624463198, +SAMPLE — v2zloss_86k,global_mmlu_full_lt_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.77,0.042295258468165065,100 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_virology,5,lm-eval,vllm,,acc,none,0.46987951807228917,0.038854254208667706,166 +SAMPLE — v2zloss_86k,global_mmlu_full_lt_world_religions,5,lm-eval,vllm,,acc,none,0.7719298245614035,0.03218093795602355,171 +SAMPLE — v2zloss_86k,global_mmlu_full_nl,5,lm-eval,vllm,yes,acc,none,0.6073208944594787,0.003950045514570262, +SAMPLE — v2zloss_86k,global_mmlu_full_nl_abstract_algebra,5,lm-eval,vllm,,acc,none,0.37,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_anatomy,5,lm-eval,vllm,,acc,none,0.4962962962962963,0.043192236258113345,135 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_astronomy,5,lm-eval,vllm,,acc,none,0.6578947368421053,0.0386073159931609,152 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_business_ethics,5,lm-eval,vllm,,acc,none,0.6,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6490566037735849,0.029373646253234693,265 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_college_biology,5,lm-eval,vllm,,acc,none,0.6597222222222222,0.03962135573486219,144 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_college_chemistry,5,lm-eval,vllm,,acc,none,0.54,0.05009082659620331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_college_computer_science,5,lm-eval,vllm,,acc,none,0.5,0.050251890762960605,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_college_mathematics,5,lm-eval,vllm,,acc,none,0.48,0.05021167315686783,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_college_medicine,5,lm-eval,vllm,,acc,none,0.6127167630057804,0.03714325906302067,173 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_college_physics,5,lm-eval,vllm,,acc,none,0.37254901960784315,0.048108401480826374,102 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_computer_security,5,lm-eval,vllm,,acc,none,0.71,0.045604802157206865,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6638297872340425,0.03088161852067694,235 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_econometrics,5,lm-eval,vllm,,acc,none,0.47368421052631576,0.046970851366478654,114 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_electrical_engineering,5,lm-eval,vllm,,acc,none,0.5655172413793104,0.041307408795554966,145 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5476190476190477,0.025634258115554923,378 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_formal_logic,5,lm-eval,vllm,,acc,none,0.5238095238095238,0.04467062628403266,126 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_global_facts,5,lm-eval,vllm,,acc,none,0.37,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_biology,5,lm-eval,vllm,,acc,none,0.7451612903225806,0.02479011845933226,310 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5714285714285714,0.03481904844438804,203 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.71,0.045604802157206865,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7696969696969697,0.03287666758603482,165 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_geography,5,lm-eval,vllm,,acc,none,0.7676767676767676,0.030088629490217438,198 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.8031088082901554,0.028697873971860723,193 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.676923076923077,0.023710888501970607,390 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.40370370370370373,0.02991481234222757,270 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.7058823529411765,0.029597329730978096,238 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_physics,5,lm-eval,vllm,,acc,none,0.4370860927152318,0.040500357222306375,151 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_psychology,5,lm-eval,vllm,,acc,none,0.8091743119266055,0.01684767640009113,545 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_statistics,5,lm-eval,vllm,,acc,none,0.625,0.033016908987210894,216 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7843137254901961,0.028867431449849337,204 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7932489451476793,0.026361651668389073,237 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_human_aging,5,lm-eval,vllm,,acc,none,0.6412556053811659,0.032190792004199886,223 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_human_sexuality,5,lm-eval,vllm,,acc,none,0.7480916030534351,0.03807387116306089,131 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_humanities,5,lm-eval,vllm,yes,acc,none,0.5494155154091392,0.006863589263241966, +SAMPLE — v2zloss_86k,global_mmlu_full_nl_international_law,5,lm-eval,vllm,,acc,none,0.7933884297520661,0.03695980128098826,121 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_jurisprudence,5,lm-eval,vllm,,acc,none,0.7222222222222222,0.04330043749650743,108 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_logical_fallacies,5,lm-eval,vllm,,acc,none,0.656441717791411,0.037311335196738946,163 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_machine_learning,5,lm-eval,vllm,,acc,none,0.44642857142857145,0.04718471485219583,112 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_management,5,lm-eval,vllm,,acc,none,0.7475728155339806,0.04301250399690879,103 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_marketing,5,lm-eval,vllm,,acc,none,0.8205128205128205,0.025140935950335442,234 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_medical_genetics,5,lm-eval,vllm,,acc,none,0.63,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_miscellaneous,5,lm-eval,vllm,,acc,none,0.70242656449553,0.01634911191290953,783 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_moral_disputes,5,lm-eval,vllm,,acc,none,0.6763005780346821,0.02519018132760835,346 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_moral_scenarios,5,lm-eval,vllm,,acc,none,0.33631284916201115,0.01580100372914583,895 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_nutrition,5,lm-eval,vllm,,acc,none,0.6993464052287581,0.026256053835718912,306 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_other,5,lm-eval,vllm,yes,acc,none,0.6317991631799164,0.008421380268017687, +SAMPLE — v2zloss_86k,global_mmlu_full_nl_philosophy,5,lm-eval,vllm,,acc,none,0.6881028938906752,0.026311858071854183,311 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_prehistory,5,lm-eval,vllm,,acc,none,0.6388888888888888,0.026725868809100838,324 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_professional_accounting,5,lm-eval,vllm,,acc,none,0.4078014184397163,0.02931601177634362,282 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_professional_law,5,lm-eval,vllm,,acc,none,0.4426336375488918,0.012685906538206263,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_professional_medicine,5,lm-eval,vllm,,acc,none,0.5625,0.030134614954403924,272 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_professional_psychology,5,lm-eval,vllm,,acc,none,0.6356209150326797,0.0194695182215738,612 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_public_relations,5,lm-eval,vllm,,acc,none,0.6363636363636364,0.04607582090719978,110 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_security_studies,5,lm-eval,vllm,,acc,none,0.6979591836734694,0.029393609319879853,245 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_social_sciences,5,lm-eval,vllm,yes,acc,none,0.7140071498212545,0.008030265503417322, +SAMPLE — v2zloss_86k,global_mmlu_full_nl_sociology,5,lm-eval,vllm,,acc,none,0.7761194029850746,0.029475250236017162,201 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_stem,5,lm-eval,vllm,yes,acc,none,0.5654931810973676,0.008619567563344066, +SAMPLE — v2zloss_86k,global_mmlu_full_nl_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.79,0.040936018074033236,100 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_virology,5,lm-eval,vllm,,acc,none,0.4879518072289157,0.038913644958358196,166 +SAMPLE — v2zloss_86k,global_mmlu_full_nl_world_religions,5,lm-eval,vllm,,acc,none,0.7485380116959064,0.03327504423846844,171 +SAMPLE — v2zloss_86k,global_mmlu_full_pl,5,lm-eval,vllm,yes,acc,none,0.5898732374305654,0.003985669593663225, +SAMPLE — v2zloss_86k,global_mmlu_full_pl_abstract_algebra,5,lm-eval,vllm,,acc,none,0.31,0.04648231987117317,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_anatomy,5,lm-eval,vllm,,acc,none,0.4740740740740741,0.043135316967505756,135 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_astronomy,5,lm-eval,vllm,,acc,none,0.618421052631579,0.039531733777491924,152 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_business_ethics,5,lm-eval,vllm,,acc,none,0.62,0.04878317312145634,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6339622641509434,0.02964781353936531,265 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_college_biology,5,lm-eval,vllm,,acc,none,0.5694444444444444,0.04140685639111502,144 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_college_chemistry,5,lm-eval,vllm,,acc,none,0.47,0.05016135580465919,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_college_computer_science,5,lm-eval,vllm,,acc,none,0.55,0.05,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_college_mathematics,5,lm-eval,vllm,,acc,none,0.46,0.05009082659620332,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_college_medicine,5,lm-eval,vllm,,acc,none,0.5953757225433526,0.037424611938872455,173 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_college_physics,5,lm-eval,vllm,,acc,none,0.43137254901960786,0.049280995972875316,102 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_computer_security,5,lm-eval,vllm,,acc,none,0.67,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6680851063829787,0.030783736757745685,235 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_econometrics,5,lm-eval,vllm,,acc,none,0.49122807017543857,0.0470288043204961,114 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6206896551724138,0.040434618619167494,145 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5767195767195767,0.025446365634406765,378 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_formal_logic,5,lm-eval,vllm,,acc,none,0.5317460317460317,0.04463112720677173,126 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_global_facts,5,lm-eval,vllm,,acc,none,0.41,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_biology,5,lm-eval,vllm,,acc,none,0.7290322580645161,0.02528441611490011,310 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5369458128078818,0.03508370520442665,203 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.68,0.046882617226215076,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7212121212121212,0.03501438706296781,165 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_geography,5,lm-eval,vllm,,acc,none,0.7171717171717171,0.03208779558786752,198 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.8134715025906736,0.0281120912101175,193 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6538461538461539,0.024121125416941242,390 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.3592592592592593,0.02925290592725196,270 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6890756302521008,0.030066761582977983,238 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_physics,5,lm-eval,vllm,,acc,none,0.3973509933774834,0.039955240076816834,151 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_psychology,5,lm-eval,vllm,,acc,none,0.7798165137614679,0.0177659786523276,545 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_statistics,5,lm-eval,vllm,,acc,none,0.6064814814814815,0.033317478763703126,216 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_us_history,5,lm-eval,vllm,,acc,none,0.6911764705882353,0.03242661719827215,204 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7637130801687764,0.027652153144159246,237 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_human_aging,5,lm-eval,vllm,,acc,none,0.6412556053811659,0.032190792004199886,223 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_human_sexuality,5,lm-eval,vllm,,acc,none,0.7022900763358778,0.04010358942462197,131 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_humanities,5,lm-eval,vllm,yes,acc,none,0.5317747077577045,0.006912112394402239, +SAMPLE — v2zloss_86k,global_mmlu_full_pl_international_law,5,lm-eval,vllm,,acc,none,0.7851239669421488,0.037494924487096966,121 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_jurisprudence,5,lm-eval,vllm,,acc,none,0.6666666666666666,0.04557239513497755,108 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_logical_fallacies,5,lm-eval,vllm,,acc,none,0.656441717791411,0.037311335196738946,163 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_machine_learning,5,lm-eval,vllm,,acc,none,0.44642857142857145,0.04718471485219583,112 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_management,5,lm-eval,vllm,,acc,none,0.7087378640776699,0.044986763205729245,103 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_marketing,5,lm-eval,vllm,,acc,none,0.8205128205128205,0.025140935950335442,234 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_medical_genetics,5,lm-eval,vllm,,acc,none,0.61,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_miscellaneous,5,lm-eval,vllm,,acc,none,0.685823754789272,0.016599291735884803,783 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_moral_disputes,5,lm-eval,vllm,,acc,none,0.6676300578034682,0.025361168749688214,346 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_moral_scenarios,5,lm-eval,vllm,,acc,none,0.3307262569832402,0.015735026258966122,895 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_nutrition,5,lm-eval,vllm,,acc,none,0.6699346405228758,0.026925654653615645,306 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_other,5,lm-eval,vllm,yes,acc,none,0.6160283231412939,0.008491667040520026, +SAMPLE — v2zloss_86k,global_mmlu_full_pl_philosophy,5,lm-eval,vllm,,acc,none,0.6559485530546624,0.02698147804364805,311 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_prehistory,5,lm-eval,vllm,,acc,none,0.6604938271604939,0.026348564412011603,324 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_professional_accounting,5,lm-eval,vllm,,acc,none,0.375886524822695,0.028893955412115868,282 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_professional_law,5,lm-eval,vllm,,acc,none,0.42046936114732725,0.012607654553832918,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_professional_medicine,5,lm-eval,vllm,,acc,none,0.5367647058823529,0.03029061918048574,272 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_professional_psychology,5,lm-eval,vllm,,acc,none,0.5947712418300654,0.019861155193829163,612 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_public_relations,5,lm-eval,vllm,,acc,none,0.6272727272727273,0.04631381319425461,110 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_security_studies,5,lm-eval,vllm,,acc,none,0.7061224489795919,0.02916273841024971,245 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_social_sciences,5,lm-eval,vllm,yes,acc,none,0.6925576860578485,0.008200830284242342, +SAMPLE — v2zloss_86k,global_mmlu_full_pl_sociology,5,lm-eval,vllm,,acc,none,0.7860696517412935,0.028996909693328965,201 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_stem,5,lm-eval,vllm,yes,acc,none,0.5505867427846496,0.008638467113875519, +SAMPLE — v2zloss_86k,global_mmlu_full_pl_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.76,0.04292346959909278,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_virology,5,lm-eval,vllm,,acc,none,0.463855421686747,0.038823108508905954,166 +SAMPLE — v2zloss_86k,global_mmlu_full_pl_world_religions,5,lm-eval,vllm,,acc,none,0.7602339181286549,0.03274485211946959,171 +SAMPLE — v2zloss_86k,global_mmlu_full_pt,5,lm-eval,vllm,yes,acc,none,0.612021079618288,0.003922782368822873, +SAMPLE — v2zloss_86k,global_mmlu_full_pt_abstract_algebra,5,lm-eval,vllm,,acc,none,0.33,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_anatomy,5,lm-eval,vllm,,acc,none,0.5777777777777777,0.042667634040995855,135 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_astronomy,5,lm-eval,vllm,,acc,none,0.7105263157894737,0.03690677986137281,152 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_business_ethics,5,lm-eval,vllm,,acc,none,0.62,0.04878317312145634,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6566037735849056,0.02922452646912477,265 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_college_biology,5,lm-eval,vllm,,acc,none,0.6319444444444444,0.04032999053960717,144 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_college_chemistry,5,lm-eval,vllm,,acc,none,0.47,0.05016135580465919,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_college_computer_science,5,lm-eval,vllm,,acc,none,0.62,0.04878317312145634,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_college_mathematics,5,lm-eval,vllm,,acc,none,0.39,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_college_medicine,5,lm-eval,vllm,,acc,none,0.5953757225433526,0.037424611938872455,173 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_college_physics,5,lm-eval,vllm,,acc,none,0.38235294117647056,0.048355036961072254,102 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_computer_security,5,lm-eval,vllm,,acc,none,0.76,0.04292346959909278,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_conceptual_physics,5,lm-eval,vllm,,acc,none,0.676595744680851,0.03057944277361037,235 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_econometrics,5,lm-eval,vllm,,acc,none,0.5,0.047036043419179864,114 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6551724137931034,0.03960933549451213,145 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.544973544973545,0.025646928361049343,378 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_formal_logic,5,lm-eval,vllm,,acc,none,0.5952380952380952,0.043902592653775656,126 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_global_facts,5,lm-eval,vllm,,acc,none,0.37,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_biology,5,lm-eval,vllm,,acc,none,0.7677419354838709,0.024022256130308235,310 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5812807881773399,0.03471192860518467,203 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7454545454545455,0.03401506715249038,165 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_geography,5,lm-eval,vllm,,acc,none,0.8080808080808081,0.028057791672989062,198 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.8549222797927462,0.025416343096306457,193 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6153846153846154,0.02466674491518715,390 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.3814814814814815,0.029616718927497565,270 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6764705882352942,0.03038835355188678,238 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_physics,5,lm-eval,vllm,,acc,none,0.45695364238410596,0.0406732517424744,151 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_psychology,5,lm-eval,vllm,,acc,none,0.8220183486238533,0.016399436366612896,545 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_statistics,5,lm-eval,vllm,,acc,none,0.5925925925925926,0.03350991604696042,216 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7745098039215687,0.029331162294251714,204 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7848101265822784,0.026750826994676173,237 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_human_aging,5,lm-eval,vllm,,acc,none,0.6591928251121076,0.03181149747055356,223 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_human_sexuality,5,lm-eval,vllm,,acc,none,0.7862595419847328,0.03595461611774691,131 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_humanities,5,lm-eval,vllm,yes,acc,none,0.5526036131774708,0.006817739127136887, +SAMPLE — v2zloss_86k,global_mmlu_full_pt_international_law,5,lm-eval,vllm,,acc,none,0.8099173553719008,0.03581796951709283,121 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_jurisprudence,5,lm-eval,vllm,,acc,none,0.6481481481481481,0.04616631111801715,108 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_logical_fallacies,5,lm-eval,vllm,,acc,none,0.656441717791411,0.037311335196738946,163 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_machine_learning,5,lm-eval,vllm,,acc,none,0.4375,0.04708567521880525,112 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_management,5,lm-eval,vllm,,acc,none,0.7961165048543689,0.0398913985953177,103 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_marketing,5,lm-eval,vllm,,acc,none,0.7735042735042735,0.027421007295392878,234 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_medical_genetics,5,lm-eval,vllm,,acc,none,0.6,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_miscellaneous,5,lm-eval,vllm,,acc,none,0.7254150702426565,0.015959829933084018,783 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_moral_disputes,5,lm-eval,vllm,,acc,none,0.7109826589595376,0.024405173935783293,346 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_moral_scenarios,5,lm-eval,vllm,,acc,none,0.3039106145251397,0.015382845587584475,895 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_nutrition,5,lm-eval,vllm,,acc,none,0.6797385620915033,0.02671611838015683,306 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_other,5,lm-eval,vllm,yes,acc,none,0.6392018023817188,0.008401198318298529, +SAMPLE — v2zloss_86k,global_mmlu_full_pt_philosophy,5,lm-eval,vllm,,acc,none,0.6913183279742765,0.026236965881153203,311 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_prehistory,5,lm-eval,vllm,,acc,none,0.7006172839506173,0.025483115601195486,324 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_professional_accounting,5,lm-eval,vllm,,acc,none,0.41843971631205673,0.02942799403941995,282 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_professional_law,5,lm-eval,vllm,,acc,none,0.455019556714472,0.012718456618701763,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_professional_medicine,5,lm-eval,vllm,,acc,none,0.5919117647058824,0.02985526139348388,272 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_professional_psychology,5,lm-eval,vllm,,acc,none,0.6258169934640523,0.019576953122088805,612 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_public_relations,5,lm-eval,vllm,,acc,none,0.6454545454545455,0.04582004841505413,110 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_security_studies,5,lm-eval,vllm,,acc,none,0.7142857142857143,0.028920583220675568,245 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_social_sciences,5,lm-eval,vllm,yes,acc,none,0.7143321416964575,0.0079712198823932, +SAMPLE — v2zloss_86k,global_mmlu_full_pt_sociology,5,lm-eval,vllm,,acc,none,0.7711442786069652,0.0297052840567725,201 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_stem,5,lm-eval,vllm,yes,acc,none,0.5740564541706311,0.008524366878456735, +SAMPLE — v2zloss_86k,global_mmlu_full_pt_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_virology,5,lm-eval,vllm,,acc,none,0.5120481927710844,0.038913644958358196,166 +SAMPLE — v2zloss_86k,global_mmlu_full_pt_world_religions,5,lm-eval,vllm,,acc,none,0.7309941520467836,0.03401052620104092,171 +SAMPLE — v2zloss_86k,global_mmlu_full_ro,5,lm-eval,vllm,yes,acc,none,0.6031904287138584,0.00396795043769042, +SAMPLE — v2zloss_86k,global_mmlu_full_ro_abstract_algebra,5,lm-eval,vllm,,acc,none,0.32,0.04688261722621507,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_anatomy,5,lm-eval,vllm,,acc,none,0.562962962962963,0.042849586397534056,135 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_astronomy,5,lm-eval,vllm,,acc,none,0.6578947368421053,0.0386073159931609,152 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_business_ethics,5,lm-eval,vllm,,acc,none,0.61,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6566037735849056,0.02922452646912477,265 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_college_biology,5,lm-eval,vllm,,acc,none,0.6180555555555556,0.04062990784146671,144 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_college_chemistry,5,lm-eval,vllm,,acc,none,0.54,0.05009082659620331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_college_computer_science,5,lm-eval,vllm,,acc,none,0.53,0.05016135580465919,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_college_mathematics,5,lm-eval,vllm,,acc,none,0.46,0.05009082659620332,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_college_medicine,5,lm-eval,vllm,,acc,none,0.6473988439306358,0.03643037168958545,173 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_college_physics,5,lm-eval,vllm,,acc,none,0.35294117647058826,0.04755129616062946,102 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_computer_security,5,lm-eval,vllm,,acc,none,0.68,0.046882617226215076,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6382978723404256,0.03141082197596243,235 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_econometrics,5,lm-eval,vllm,,acc,none,0.42105263157894735,0.04644602091222323,114 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6413793103448275,0.03996629574876715,145 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5343915343915344,0.025690321762493876,378 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_formal_logic,5,lm-eval,vllm,,acc,none,0.5238095238095238,0.04467062628403266,126 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_global_facts,5,lm-eval,vllm,,acc,none,0.4,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_biology,5,lm-eval,vllm,,acc,none,0.7354838709677419,0.02509189237885932,310 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5270935960591133,0.03512819077876112,203 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.71,0.045604802157206865,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7575757575757576,0.03346409881055956,165 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_geography,5,lm-eval,vllm,,acc,none,0.7222222222222222,0.03191178226713548,198 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.8186528497409327,0.027807032360686077,193 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6333333333333333,0.024433016466052476,390 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.43703703703703706,0.030242862397654047,270 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6722689075630253,0.030489911417673234,238 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_physics,5,lm-eval,vllm,,acc,none,0.4105960264900662,0.04016689594849927,151 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_psychology,5,lm-eval,vllm,,acc,none,0.7889908256880734,0.017493922404112613,545 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_statistics,5,lm-eval,vllm,,acc,none,0.6064814814814815,0.033317478763703126,216 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7450980392156863,0.030587591351604302,204 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7932489451476793,0.026361651668389073,237 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_human_aging,5,lm-eval,vllm,,acc,none,0.6457399103139013,0.032100621541349836,223 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_human_sexuality,5,lm-eval,vllm,,acc,none,0.7175572519083969,0.039484061257683584,131 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_humanities,5,lm-eval,vllm,yes,acc,none,0.5540913921360255,0.0068735020845344715, +SAMPLE — v2zloss_86k,global_mmlu_full_ro_international_law,5,lm-eval,vllm,,acc,none,0.8016528925619835,0.03640118271990949,121 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_jurisprudence,5,lm-eval,vllm,,acc,none,0.6851851851851852,0.0448993107359131,108 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_logical_fallacies,5,lm-eval,vllm,,acc,none,0.6441717791411042,0.03761521380046734,163 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_machine_learning,5,lm-eval,vllm,,acc,none,0.4017857142857143,0.04653333146973646,112 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_management,5,lm-eval,vllm,,acc,none,0.7378640776699029,0.04354631077260595,103 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_marketing,5,lm-eval,vllm,,acc,none,0.8247863247863247,0.02490443909891819,234 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_medical_genetics,5,lm-eval,vllm,,acc,none,0.58,0.04960449637488582,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_miscellaneous,5,lm-eval,vllm,,acc,none,0.6973180076628352,0.01642878158174946,783 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_moral_disputes,5,lm-eval,vllm,,acc,none,0.6878612716763006,0.024946792225272324,346 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_moral_scenarios,5,lm-eval,vllm,,acc,none,0.34301675977653634,0.015876912673057682,895 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_nutrition,5,lm-eval,vllm,,acc,none,0.6633986928104575,0.0270579746244944,306 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_other,5,lm-eval,vllm,yes,acc,none,0.6321210170582555,0.008437845686463103, +SAMPLE — v2zloss_86k,global_mmlu_full_ro_philosophy,5,lm-eval,vllm,,acc,none,0.6816720257234726,0.026457225067810987,311 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_prehistory,5,lm-eval,vllm,,acc,none,0.6759259259259259,0.026041766202717188,324 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_professional_accounting,5,lm-eval,vllm,,acc,none,0.41134751773049644,0.02935491115994096,282 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_professional_law,5,lm-eval,vllm,,acc,none,0.45045632333767927,0.012707390438502405,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_professional_medicine,5,lm-eval,vllm,,acc,none,0.6029411764705882,0.02972215209928001,272 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_professional_psychology,5,lm-eval,vllm,,acc,none,0.6062091503267973,0.01976621199107302,612 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_public_relations,5,lm-eval,vllm,,acc,none,0.6727272727272727,0.04494290866252085,110 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_security_studies,5,lm-eval,vllm,,acc,none,0.7306122448979592,0.028401252029022956,245 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_social_sciences,5,lm-eval,vllm,yes,acc,none,0.6948326291842704,0.008154123033247546, +SAMPLE — v2zloss_86k,global_mmlu_full_ro_sociology,5,lm-eval,vllm,,acc,none,0.7810945273631841,0.02923917463664699,201 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_stem,5,lm-eval,vllm,yes,acc,none,0.5585156993339676,0.008638689250332761, +SAMPLE — v2zloss_86k,global_mmlu_full_ro_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.77,0.042295258468165065,100 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_virology,5,lm-eval,vllm,,acc,none,0.463855421686747,0.038823108508905954,166 +SAMPLE — v2zloss_86k,global_mmlu_full_ro_world_religions,5,lm-eval,vllm,,acc,none,0.7777777777777778,0.03188578017686397,171 +SAMPLE — v2zloss_86k,global_mmlu_full_sr,5,lm-eval,vllm,yes,acc,none,0.566372311636519,0.004016105038536441, +SAMPLE — v2zloss_86k,global_mmlu_full_sr_abstract_algebra,5,lm-eval,vllm,,acc,none,0.36,0.048241815132442176,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_anatomy,5,lm-eval,vllm,,acc,none,0.4740740740740741,0.043135316967505756,135 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_astronomy,5,lm-eval,vllm,,acc,none,0.625,0.039397364351956274,152 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_business_ethics,5,lm-eval,vllm,,acc,none,0.63,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6226415094339622,0.029832808114796033,265 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_college_biology,5,lm-eval,vllm,,acc,none,0.5347222222222222,0.04171115858181617,144 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_college_chemistry,5,lm-eval,vllm,,acc,none,0.53,0.05016135580465919,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_college_computer_science,5,lm-eval,vllm,,acc,none,0.48,0.05021167315686783,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_college_mathematics,5,lm-eval,vllm,,acc,none,0.4,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_college_medicine,5,lm-eval,vllm,,acc,none,0.5433526011560693,0.03798106566014504,173 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_college_physics,5,lm-eval,vllm,,acc,none,0.35294117647058826,0.04755129616062946,102 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_computer_security,5,lm-eval,vllm,,acc,none,0.64,0.048241815132442176,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6212765957446809,0.03170995606040651,235 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_econometrics,5,lm-eval,vllm,,acc,none,0.35964912280701755,0.04514496132873635,114 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6137931034482759,0.04057324734419032,145 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5317460317460317,0.02569935283213174,378 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_formal_logic,5,lm-eval,vllm,,acc,none,0.5238095238095238,0.04467062628403266,126 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_global_facts,5,lm-eval,vllm,,acc,none,0.33,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_biology,5,lm-eval,vllm,,acc,none,0.7290322580645161,0.02528441611490011,310 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5123152709359606,0.035169204442208966,203 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.69,0.046482319871173176,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7272727272727273,0.03477691162163663,165 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_geography,5,lm-eval,vllm,,acc,none,0.7525252525252525,0.030746300742124488,198 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.7357512953367875,0.03182155050916644,193 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6205128205128205,0.024603626924097316,390 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.3592592592592593,0.02925290592725196,270 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.634453781512605,0.031282177063684614,238 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_physics,5,lm-eval,vllm,,acc,none,0.4503311258278146,0.040622900186837764,151 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_psychology,5,lm-eval,vllm,,acc,none,0.7577981651376147,0.018368176306598663,545 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_statistics,5,lm-eval,vllm,,acc,none,0.6018518518518519,0.033384734032073975,216 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_us_history,5,lm-eval,vllm,,acc,none,0.6813725490196079,0.032702871814820796,204 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_high_school_world_history,5,lm-eval,vllm,,acc,none,0.759493670886076,0.02782078198114972,237 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_human_aging,5,lm-eval,vllm,,acc,none,0.6188340807174888,0.0325962511841683,223 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_human_sexuality,5,lm-eval,vllm,,acc,none,0.6870229007633588,0.040669629056776964,131 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_humanities,5,lm-eval,vllm,yes,acc,none,0.5069075451647184,0.006878772724973893, +SAMPLE — v2zloss_86k,global_mmlu_full_sr_international_law,5,lm-eval,vllm,,acc,none,0.7933884297520661,0.03695980128098826,121 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_jurisprudence,5,lm-eval,vllm,,acc,none,0.6481481481481481,0.04616631111801715,108 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_logical_fallacies,5,lm-eval,vllm,,acc,none,0.6134969325153374,0.03825825548848611,163 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_machine_learning,5,lm-eval,vllm,,acc,none,0.4642857142857143,0.04733667890053756,112 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_management,5,lm-eval,vllm,,acc,none,0.6310679611650486,0.0477761518115674,103 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_marketing,5,lm-eval,vllm,,acc,none,0.8076923076923077,0.025819233256483692,234 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_medical_genetics,5,lm-eval,vllm,,acc,none,0.54,0.05009082659620331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_miscellaneous,5,lm-eval,vllm,,acc,none,0.5964240102171137,0.017544332237926417,783 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_moral_disputes,5,lm-eval,vllm,,acc,none,0.6416184971098265,0.025816756791584197,346 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_moral_scenarios,5,lm-eval,vllm,,acc,none,0.2558659217877095,0.014593620923210763,895 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_nutrition,5,lm-eval,vllm,,acc,none,0.6633986928104575,0.0270579746244944,306 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_other,5,lm-eval,vllm,yes,acc,none,0.5841647891857097,0.008674348480010103, +SAMPLE — v2zloss_86k,global_mmlu_full_sr_philosophy,5,lm-eval,vllm,,acc,none,0.6720257234726688,0.02666441088693758,311 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_prehistory,5,lm-eval,vllm,,acc,none,0.5987654320987654,0.027272582849839785,324 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_professional_accounting,5,lm-eval,vllm,,acc,none,0.4219858156028369,0.029462189233370586,282 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_professional_law,5,lm-eval,vllm,,acc,none,0.4230769230769231,0.012618204066588503,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_professional_medicine,5,lm-eval,vllm,,acc,none,0.5220588235294118,0.030343264224213528,272 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_professional_psychology,5,lm-eval,vllm,,acc,none,0.5800653594771242,0.0199668111782565,612 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_public_relations,5,lm-eval,vllm,,acc,none,0.6454545454545455,0.04582004841505413,110 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_security_studies,5,lm-eval,vllm,,acc,none,0.7142857142857143,0.028920583220675568,245 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_social_sciences,5,lm-eval,vllm,yes,acc,none,0.6688332791680208,0.008332842958608035, +SAMPLE — v2zloss_86k,global_mmlu_full_sr_sociology,5,lm-eval,vllm,,acc,none,0.7512437810945274,0.030567675938916686,201 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_stem,5,lm-eval,vllm,yes,acc,none,0.5375832540437678,0.008680801085528999, +SAMPLE — v2zloss_86k,global_mmlu_full_sr_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.78,0.041633319989322654,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_virology,5,lm-eval,vllm,,acc,none,0.5,0.03892494720807614,166 +SAMPLE — v2zloss_86k,global_mmlu_full_sr_world_religions,5,lm-eval,vllm,,acc,none,0.6491228070175439,0.03660298834049164,171 +SAMPLE — v2zloss_86k,global_mmlu_full_sv,5,lm-eval,vllm,yes,acc,none,0.604970801880074,0.0039527774626804965, +SAMPLE — v2zloss_86k,global_mmlu_full_sv_abstract_algebra,5,lm-eval,vllm,,acc,none,0.38,0.04878317312145634,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_anatomy,5,lm-eval,vllm,,acc,none,0.5333333333333333,0.043097329010363616,135 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_astronomy,5,lm-eval,vllm,,acc,none,0.6776315789473685,0.03803510248351587,152 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_business_ethics,5,lm-eval,vllm,,acc,none,0.59,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6415094339622641,0.02951470358398169,265 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_college_biology,5,lm-eval,vllm,,acc,none,0.6319444444444444,0.04032999053960717,144 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_college_chemistry,5,lm-eval,vllm,,acc,none,0.46,0.05009082659620332,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_college_computer_science,5,lm-eval,vllm,,acc,none,0.52,0.05021167315686783,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_college_mathematics,5,lm-eval,vllm,,acc,none,0.43,0.049756985195624305,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_college_medicine,5,lm-eval,vllm,,acc,none,0.6127167630057804,0.03714325906302067,173 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_college_physics,5,lm-eval,vllm,,acc,none,0.43137254901960786,0.049280995972875316,102 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_computer_security,5,lm-eval,vllm,,acc,none,0.67,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6680851063829787,0.030783736757745685,235 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_econometrics,5,lm-eval,vllm,,acc,none,0.4473684210526316,0.04677473004491205,114 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6413793103448275,0.03996629574876715,145 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.582010582010582,0.025402555503260853,378 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_formal_logic,5,lm-eval,vllm,,acc,none,0.49206349206349204,0.04471572536294351,126 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_global_facts,5,lm-eval,vllm,,acc,none,0.4,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_biology,5,lm-eval,vllm,,acc,none,0.7516129032258064,0.024580028921481045,310 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5369458128078818,0.03508370520442665,203 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.68,0.046882617226215076,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7575757575757576,0.03346409881055956,165 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_geography,5,lm-eval,vllm,,acc,none,0.7828282828282829,0.02937661648494561,198 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.8031088082901554,0.028697873971860723,193 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6256410256410256,0.024537591572830485,390 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.40370370370370373,0.02991481234222757,270 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6596638655462185,0.030778057422931663,238 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_physics,5,lm-eval,vllm,,acc,none,0.4768211920529801,0.04078093859163086,151 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_psychology,5,lm-eval,vllm,,acc,none,0.8036697247706422,0.017030719339154426,545 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_statistics,5,lm-eval,vllm,,acc,none,0.6064814814814815,0.033317478763703126,216 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7352941176470589,0.030964517926923372,204 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7637130801687764,0.027652153144159246,237 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_human_aging,5,lm-eval,vllm,,acc,none,0.7174887892376681,0.030216831011508755,223 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_human_sexuality,5,lm-eval,vllm,,acc,none,0.7709923664122137,0.036853466317118506,131 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_humanities,5,lm-eval,vllm,yes,acc,none,0.5419766206163655,0.006883387099015404, +SAMPLE — v2zloss_86k,global_mmlu_full_sv_international_law,5,lm-eval,vllm,,acc,none,0.7768595041322314,0.038007544752287334,121 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_jurisprudence,5,lm-eval,vllm,,acc,none,0.6574074074074074,0.04587904741301815,108 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_logical_fallacies,5,lm-eval,vllm,,acc,none,0.6257668711656442,0.03802068102899616,163 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_machine_learning,5,lm-eval,vllm,,acc,none,0.39285714285714285,0.04635550135609972,112 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_management,5,lm-eval,vllm,,acc,none,0.7378640776699029,0.04354631077260595,103 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_marketing,5,lm-eval,vllm,,acc,none,0.811965811965812,0.025598193686652254,234 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_medical_genetics,5,lm-eval,vllm,,acc,none,0.61,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_miscellaneous,5,lm-eval,vllm,,acc,none,0.7100893997445722,0.016225017944770968,783 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_moral_disputes,5,lm-eval,vllm,,acc,none,0.6878612716763006,0.024946792225272324,346 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_moral_scenarios,5,lm-eval,vllm,,acc,none,0.3318435754189944,0.015748421208187313,895 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_nutrition,5,lm-eval,vllm,,acc,none,0.696078431372549,0.026336613469046647,306 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_other,5,lm-eval,vllm,yes,acc,none,0.6359832635983264,0.008389478002266963, +SAMPLE — v2zloss_86k,global_mmlu_full_sv_philosophy,5,lm-eval,vllm,,acc,none,0.6881028938906752,0.026311858071854183,311 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_prehistory,5,lm-eval,vllm,,acc,none,0.6604938271604939,0.026348564412011603,324 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_professional_accounting,5,lm-eval,vllm,,acc,none,0.3900709219858156,0.02909767559946399,282 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_professional_law,5,lm-eval,vllm,,acc,none,0.43546284224250326,0.012663412101248344,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_professional_medicine,5,lm-eval,vllm,,acc,none,0.5514705882352942,0.030211479609121628,272 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_professional_psychology,5,lm-eval,vllm,,acc,none,0.6339869281045751,0.019488025745529686,612 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_public_relations,5,lm-eval,vllm,,acc,none,0.6545454545454545,0.045546196175410524,110 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_security_studies,5,lm-eval,vllm,,acc,none,0.7387755102040816,0.028123429335142832,245 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_social_sciences,5,lm-eval,vllm,yes,acc,none,0.7075073123171921,0.008055304164414303, +SAMPLE — v2zloss_86k,global_mmlu_full_sv_sociology,5,lm-eval,vllm,,acc,none,0.7810945273631841,0.02923917463664699,201 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_stem,5,lm-eval,vllm,yes,acc,none,0.5683476054551221,0.008617152862403369, +SAMPLE — v2zloss_86k,global_mmlu_full_sv_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.78,0.041633319989322654,100 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_virology,5,lm-eval,vllm,,acc,none,0.5120481927710844,0.038913644958358196,166 +SAMPLE — v2zloss_86k,global_mmlu_full_sv_world_religions,5,lm-eval,vllm,,acc,none,0.783625730994152,0.03158149539338735,171 +SAMPLE — v2zloss_86k,global_mmlu_full_tr,5,lm-eval,vllm,yes,acc,none,0.5630252100840336,0.004036992716674874, +SAMPLE — v2zloss_86k,global_mmlu_full_tr_abstract_algebra,5,lm-eval,vllm,,acc,none,0.35,0.04793724854411023,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_anatomy,5,lm-eval,vllm,,acc,none,0.4666666666666667,0.043097329010363616,135 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_astronomy,5,lm-eval,vllm,,acc,none,0.5986842105263158,0.03988903703336284,152 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_business_ethics,5,lm-eval,vllm,,acc,none,0.52,0.05021167315686783,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.5962264150943396,0.03019761160019793,265 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_college_biology,5,lm-eval,vllm,,acc,none,0.5763888888888888,0.041321250197233685,144 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_college_chemistry,5,lm-eval,vllm,,acc,none,0.49,0.05024183937956913,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_college_computer_science,5,lm-eval,vllm,,acc,none,0.46,0.05009082659620332,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_college_mathematics,5,lm-eval,vllm,,acc,none,0.3,0.04605661864718382,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_college_medicine,5,lm-eval,vllm,,acc,none,0.5375722543352601,0.03801685104524462,173 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_college_physics,5,lm-eval,vllm,,acc,none,0.30392156862745096,0.04576665403207762,102 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_computer_security,5,lm-eval,vllm,,acc,none,0.62,0.04878317312145634,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6170212765957447,0.03177821250236923,235 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_econometrics,5,lm-eval,vllm,,acc,none,0.42105263157894735,0.04644602091222323,114 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6068965517241379,0.04070329013707074,145 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5052910052910053,0.025749868288556587,378 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_formal_logic,5,lm-eval,vllm,,acc,none,0.5634920634920635,0.044359328928514664,126 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_global_facts,5,lm-eval,vllm,,acc,none,0.34,0.04760952285695233,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_biology,5,lm-eval,vllm,,acc,none,0.6774193548387096,0.026593084516572225,310 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.4876847290640394,0.035169204442208966,203 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.7,0.04605661864718383,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7212121212121212,0.03501438706296781,165 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_geography,5,lm-eval,vllm,,acc,none,0.7474747474747475,0.030954055470365872,198 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.694300518134715,0.03324837939758158,193 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.5794871794871795,0.02502861027671089,390 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.3888888888888889,0.0297232789614767,270 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6722689075630253,0.030489911417673234,238 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_physics,5,lm-eval,vllm,,acc,none,0.3708609271523179,0.039439666991836285,151 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_psychology,5,lm-eval,vllm,,acc,none,0.7504587155963303,0.018553897629501544,545 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_statistics,5,lm-eval,vllm,,acc,none,0.5416666666666666,0.03398110890294638,216 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_us_history,5,lm-eval,vllm,,acc,none,0.7549019607843137,0.030190282453501943,204 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7341772151898734,0.02875679962965839,237 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_human_aging,5,lm-eval,vllm,,acc,none,0.6367713004484304,0.03227790442850494,223 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_human_sexuality,5,lm-eval,vllm,,acc,none,0.6793893129770993,0.04093329229834282,131 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_humanities,5,lm-eval,vllm,yes,acc,none,0.5228480340063762,0.0069314851098040255, +SAMPLE — v2zloss_86k,global_mmlu_full_tr_international_law,5,lm-eval,vllm,,acc,none,0.7520661157024794,0.039418975265163046,121 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_jurisprudence,5,lm-eval,vllm,,acc,none,0.6481481481481481,0.04616631111801715,108 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_logical_fallacies,5,lm-eval,vllm,,acc,none,0.5828220858895705,0.038741028598180835,163 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_machine_learning,5,lm-eval,vllm,,acc,none,0.4107142857142857,0.04669510663875191,112 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_management,5,lm-eval,vllm,,acc,none,0.6893203883495146,0.04582124160161549,103 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_marketing,5,lm-eval,vllm,,acc,none,0.811965811965812,0.025598193686652254,234 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_medical_genetics,5,lm-eval,vllm,,acc,none,0.61,0.04902071300001973,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_miscellaneous,5,lm-eval,vllm,,acc,none,0.632183908045977,0.017243828891846193,783 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_moral_disputes,5,lm-eval,vllm,,acc,none,0.6271676300578035,0.026033890613576312,346 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_moral_scenarios,5,lm-eval,vllm,,acc,none,0.31508379888268156,0.015536850852473702,895 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_nutrition,5,lm-eval,vllm,,acc,none,0.6470588235294118,0.027363593284684976,306 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_other,5,lm-eval,vllm,yes,acc,none,0.588670743482459,0.008613348518581261, +SAMPLE — v2zloss_86k,global_mmlu_full_tr_philosophy,5,lm-eval,vllm,,acc,none,0.6688102893890675,0.02673062072800498,311 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_prehistory,5,lm-eval,vllm,,acc,none,0.6574074074074074,0.0264061459736257,324 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_professional_accounting,5,lm-eval,vllm,,acc,none,0.36524822695035464,0.028723863853281288,282 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_professional_law,5,lm-eval,vllm,,acc,none,0.4211212516297262,0.01261032573349008,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_professional_medicine,5,lm-eval,vllm,,acc,none,0.5588235294117647,0.030161911930767053,272 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_professional_psychology,5,lm-eval,vllm,,acc,none,0.5686274509803921,0.020036393768352655,612 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_public_relations,5,lm-eval,vllm,,acc,none,0.5636363636363636,0.04750185058907295,110 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_security_studies,5,lm-eval,vllm,,acc,none,0.6775510204081633,0.029923100563683976,245 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_social_sciences,5,lm-eval,vllm,yes,acc,none,0.6499837504062398,0.008481222130578388, +SAMPLE — v2zloss_86k,global_mmlu_full_tr_sociology,5,lm-eval,vllm,,acc,none,0.6965174129353234,0.03251006816458615,201 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_stem,5,lm-eval,vllm,yes,acc,none,0.5128449096098954,0.00870158871304097, +SAMPLE — v2zloss_86k,global_mmlu_full_tr_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.7,0.04605661864718383,100 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_virology,5,lm-eval,vllm,,acc,none,0.4819277108433735,0.03889951252827222,166 +SAMPLE — v2zloss_86k,global_mmlu_full_tr_world_religions,5,lm-eval,vllm,,acc,none,0.7017543859649122,0.03508771929824561,171 +SAMPLE — v2zloss_86k,global_mmlu_full_uk,5,lm-eval,vllm,yes,acc,none,0.5676541803161943,0.004018285417420435, +SAMPLE — v2zloss_86k,global_mmlu_full_uk_abstract_algebra,5,lm-eval,vllm,,acc,none,0.37,0.048523658709390974,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_anatomy,5,lm-eval,vllm,,acc,none,0.43703703703703706,0.042849586397534056,135 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_astronomy,5,lm-eval,vllm,,acc,none,0.618421052631579,0.039531733777491924,152 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_business_ethics,5,lm-eval,vllm,,acc,none,0.59,0.04943110704237104,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6113207547169811,0.03000048544867597,265 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_college_biology,5,lm-eval,vllm,,acc,none,0.6180555555555556,0.04062990784146671,144 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_college_chemistry,5,lm-eval,vllm,,acc,none,0.47,0.05016135580465919,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_college_computer_science,5,lm-eval,vllm,,acc,none,0.5,0.050251890762960605,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_college_mathematics,5,lm-eval,vllm,,acc,none,0.4,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_college_medicine,5,lm-eval,vllm,,acc,none,0.5317919075144508,0.03804749744364767,173 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_college_physics,5,lm-eval,vllm,,acc,none,0.3627450980392157,0.047840607041056527,102 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_computer_security,5,lm-eval,vllm,,acc,none,0.65,0.04793724854411023,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_conceptual_physics,5,lm-eval,vllm,,acc,none,0.6127659574468085,0.031843892653395246,235 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_econometrics,5,lm-eval,vllm,,acc,none,0.4298245614035088,0.04657047260594963,114 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6137931034482759,0.04057324734419032,145 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5396825396825397,0.025670080636909207,378 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_formal_logic,5,lm-eval,vllm,,acc,none,0.5238095238095238,0.04467062628403266,126 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_global_facts,5,lm-eval,vllm,,acc,none,0.38,0.04878317312145634,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_biology,5,lm-eval,vllm,,acc,none,0.6967741935483871,0.026148685930671777,310 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.5320197044334976,0.03510766597959214,203 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7515151515151515,0.03374402644139407,165 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_geography,5,lm-eval,vllm,,acc,none,0.7373737373737373,0.031353050095330834,198 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.7772020725388601,0.030031147977641528,193 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.6,0.02483881198803324,390 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.4074074074074074,0.02995824925008211,270 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.6638655462184874,0.030684737115135356,238 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_physics,5,lm-eval,vllm,,acc,none,0.4370860927152318,0.040500357222306375,151 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_psychology,5,lm-eval,vllm,,acc,none,0.7596330275229358,0.01832060732096402,545 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_statistics,5,lm-eval,vllm,,acc,none,0.6018518518518519,0.033384734032073975,216 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_us_history,5,lm-eval,vllm,,acc,none,0.6911764705882353,0.03242661719827215,204 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7510548523206751,0.028146970599422647,237 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_human_aging,5,lm-eval,vllm,,acc,none,0.6457399103139013,0.032100621541349836,223 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_human_sexuality,5,lm-eval,vllm,,acc,none,0.6412213740458015,0.04206739313864908,131 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_humanities,5,lm-eval,vllm,yes,acc,none,0.5064824654622742,0.0068727371225628265, +SAMPLE — v2zloss_86k,global_mmlu_full_uk_international_law,5,lm-eval,vllm,,acc,none,0.7768595041322314,0.038007544752287334,121 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_jurisprudence,5,lm-eval,vllm,,acc,none,0.6666666666666666,0.04557239513497755,108 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_logical_fallacies,5,lm-eval,vllm,,acc,none,0.6073619631901841,0.038367409078310266,163 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_machine_learning,5,lm-eval,vllm,,acc,none,0.375,0.04595091388086298,112 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_management,5,lm-eval,vllm,,acc,none,0.6601941747572816,0.046897659372781356,103 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_marketing,5,lm-eval,vllm,,acc,none,0.7777777777777778,0.02723601394619673,234 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_medical_genetics,5,lm-eval,vllm,,acc,none,0.64,0.048241815132442176,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_miscellaneous,5,lm-eval,vllm,,acc,none,0.6181353767560664,0.017373732736677562,783 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_moral_disputes,5,lm-eval,vllm,,acc,none,0.6502890173410405,0.02567428145653101,346 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_moral_scenarios,5,lm-eval,vllm,,acc,none,0.25027932960893856,0.014487500852850398,895 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_nutrition,5,lm-eval,vllm,,acc,none,0.6797385620915033,0.02671611838015683,306 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_other,5,lm-eval,vllm,yes,acc,none,0.5934985516575475,0.008646994118005126, +SAMPLE — v2zloss_86k,global_mmlu_full_uk_philosophy,5,lm-eval,vllm,,acc,none,0.639871382636656,0.027264297599803998,311 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_prehistory,5,lm-eval,vllm,,acc,none,0.6172839506172839,0.027044538138402588,324 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_professional_accounting,5,lm-eval,vllm,,acc,none,0.40070921985815605,0.029233465745573072,282 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_professional_law,5,lm-eval,vllm,,acc,none,0.424380704041721,0.012623343757430041,1534 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_professional_medicine,5,lm-eval,vllm,,acc,none,0.5625,0.030134614954403924,272 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_professional_psychology,5,lm-eval,vllm,,acc,none,0.5735294117647058,0.02000791273935941,612 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_public_relations,5,lm-eval,vllm,,acc,none,0.6454545454545455,0.04582004841505413,110 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_security_studies,5,lm-eval,vllm,,acc,none,0.6612244897959184,0.030299506562154174,245 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_social_sciences,5,lm-eval,vllm,yes,acc,none,0.6646083847903802,0.00837885396047753, +SAMPLE — v2zloss_86k,global_mmlu_full_uk_sociology,5,lm-eval,vllm,,acc,none,0.746268656716418,0.030769444967295976,201 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_stem,5,lm-eval,vllm,yes,acc,none,0.5388518870916588,0.008695800515186878, +SAMPLE — v2zloss_86k,global_mmlu_full_uk_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.76,0.04292346959909278,100 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_virology,5,lm-eval,vllm,,acc,none,0.463855421686747,0.038823108508905954,166 +SAMPLE — v2zloss_86k,global_mmlu_full_uk_world_religions,5,lm-eval,vllm,,acc,none,0.6432748538011696,0.03674013002860956,171 +SAMPLE — v2zloss_86k,global_piqa_completions_als_latn,0,lm-eval,vllm,,acc,none,0.67,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_piqa_completions_als_latn,0,lm-eval,vllm,,acc_bytes,none,0.67,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_piqa_completions_als_latn,0,lm-eval,vllm,,acc_norm,none,0.68,0.046882617226215076,100 +SAMPLE — v2zloss_86k,global_piqa_completions_bos_latn,0,lm-eval,vllm,,acc,none,0.75,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_piqa_completions_bos_latn,0,lm-eval,vllm,,acc_bytes,none,0.78,0.041633319989322654,100 +SAMPLE — v2zloss_86k,global_piqa_completions_bos_latn,0,lm-eval,vllm,,acc_norm,none,0.77,0.042295258468165065,100 +SAMPLE — v2zloss_86k,global_piqa_completions_bul_cyrl,0,lm-eval,vllm,,acc,none,0.76,0.04292346959909278,100 +SAMPLE — v2zloss_86k,global_piqa_completions_bul_cyrl,0,lm-eval,vllm,,acc_bytes,none,0.75,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_piqa_completions_bul_cyrl,0,lm-eval,vllm,,acc_norm,none,0.75,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_piqa_completions_cat_latn,0,lm-eval,vllm,,acc,none,0.77,0.042295258468165065,100 +SAMPLE — v2zloss_86k,global_piqa_completions_cat_latn,0,lm-eval,vllm,,acc_bytes,none,0.83,0.03775251680686369,100 +SAMPLE — v2zloss_86k,global_piqa_completions_cat_latn,0,lm-eval,vllm,,acc_norm,none,0.82,0.03861229196653691,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ces_latn,0,lm-eval,vllm,,acc,none,0.66,0.04760952285695234,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ces_latn,0,lm-eval,vllm,,acc_bytes,none,0.77,0.042295258468165065,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ces_latn,0,lm-eval,vllm,,acc_norm,none,0.75,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_piqa_completions_deu_latn,0,lm-eval,vllm,,acc,none,0.87,0.033799766898963114,100 +SAMPLE — v2zloss_86k,global_piqa_completions_deu_latn,0,lm-eval,vllm,,acc_bytes,none,0.83,0.03775251680686369,100 +SAMPLE — v2zloss_86k,global_piqa_completions_deu_latn,0,lm-eval,vllm,,acc_norm,none,0.83,0.03775251680686369,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ekk_latn,0,lm-eval,vllm,,acc,none,0.64,0.048241815132442176,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ekk_latn,0,lm-eval,vllm,,acc_bytes,none,0.64,0.048241815132442176,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ekk_latn,0,lm-eval,vllm,,acc_norm,none,0.64,0.048241815132442176,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ell_grek,0,lm-eval,vllm,,acc,none,0.54,0.05009082659620331,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ell_grek,0,lm-eval,vllm,,acc_bytes,none,0.56,0.049888765156985884,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ell_grek,0,lm-eval,vllm,,acc_norm,none,0.56,0.049888765156985884,100 +SAMPLE — v2zloss_86k,global_piqa_completions_eng_latn,0,lm-eval,vllm,,acc,none,0.85,0.03588702812826367,100 +SAMPLE — v2zloss_86k,global_piqa_completions_eng_latn,0,lm-eval,vllm,,acc_bytes,none,0.81,0.039427724440366255,100 +SAMPLE — v2zloss_86k,global_piqa_completions_eng_latn,0,lm-eval,vllm,,acc_norm,none,0.81,0.039427724440366255,100 +SAMPLE — v2zloss_86k,global_piqa_completions_fin_latn,0,lm-eval,vllm,,acc,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_piqa_completions_fin_latn,0,lm-eval,vllm,,acc_bytes,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_piqa_completions_fin_latn,0,lm-eval,vllm,,acc_norm,none,0.82,0.03861229196653691,100 +SAMPLE — v2zloss_86k,global_piqa_completions_fra_latn_fran,0,lm-eval,vllm,,acc,none,0.69,0.046482319871173176,100 +SAMPLE — v2zloss_86k,global_piqa_completions_fra_latn_fran,0,lm-eval,vllm,,acc_bytes,none,0.66,0.04760952285695234,100 +SAMPLE — v2zloss_86k,global_piqa_completions_fra_latn_fran,0,lm-eval,vllm,,acc_norm,none,0.67,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_piqa_completions_glg_latn,0,lm-eval,vllm,,acc,none,0.82,0.03861229196653691,100 +SAMPLE — v2zloss_86k,global_piqa_completions_glg_latn,0,lm-eval,vllm,,acc_bytes,none,0.83,0.03775251680686369,100 +SAMPLE — v2zloss_86k,global_piqa_completions_glg_latn,0,lm-eval,vllm,,acc_norm,none,0.82,0.03861229196653691,100 +SAMPLE — v2zloss_86k,global_piqa_completions_hrv_latn,0,lm-eval,vllm,,acc,none,0.81,0.039427724440366255,100 +SAMPLE — v2zloss_86k,global_piqa_completions_hrv_latn,0,lm-eval,vllm,,acc_bytes,none,0.78,0.041633319989322654,100 +SAMPLE — v2zloss_86k,global_piqa_completions_hrv_latn,0,lm-eval,vllm,,acc_norm,none,0.79,0.040936018074033236,100 +SAMPLE — v2zloss_86k,global_piqa_completions_hun_latn,0,lm-eval,vllm,,acc,none,0.6,0.0492365963917331,100 +SAMPLE — v2zloss_86k,global_piqa_completions_hun_latn,0,lm-eval,vllm,,acc_bytes,none,0.66,0.04760952285695234,100 +SAMPLE — v2zloss_86k,global_piqa_completions_hun_latn,0,lm-eval,vllm,,acc_norm,none,0.65,0.04793724854411023,100 +SAMPLE — v2zloss_86k,global_piqa_completions_isl_latn,0,lm-eval,vllm,,acc,none,0.58,0.04960449637488582,100 +SAMPLE — v2zloss_86k,global_piqa_completions_isl_latn,0,lm-eval,vllm,,acc_bytes,none,0.56,0.049888765156985884,100 +SAMPLE — v2zloss_86k,global_piqa_completions_isl_latn,0,lm-eval,vllm,,acc_norm,none,0.55,0.05,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ita_latn,0,lm-eval,vllm,,acc,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ita_latn,0,lm-eval,vllm,,acc_bytes,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ita_latn,0,lm-eval,vllm,,acc_norm,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_piqa_completions_kat_geor,0,lm-eval,vllm,,acc,none,0.62,0.04878317312145634,100 +SAMPLE — v2zloss_86k,global_piqa_completions_kat_geor,0,lm-eval,vllm,,acc_bytes,none,0.55,0.05,100 +SAMPLE — v2zloss_86k,global_piqa_completions_kat_geor,0,lm-eval,vllm,,acc_norm,none,0.58,0.04960449637488582,100 +SAMPLE — v2zloss_86k,global_piqa_completions_lit_latn,0,lm-eval,vllm,,acc,none,0.67,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_piqa_completions_lit_latn,0,lm-eval,vllm,,acc_bytes,none,0.74,0.0440844002276808,100 +SAMPLE — v2zloss_86k,global_piqa_completions_lit_latn,0,lm-eval,vllm,,acc_norm,none,0.74,0.0440844002276808,100 +SAMPLE — v2zloss_86k,global_piqa_completions_mkd_cyrl,0,lm-eval,vllm,,acc,none,0.82,0.03861229196653691,100 +SAMPLE — v2zloss_86k,global_piqa_completions_mkd_cyrl,0,lm-eval,vllm,,acc_bytes,none,0.79,0.040936018074033236,100 +SAMPLE — v2zloss_86k,global_piqa_completions_mkd_cyrl,0,lm-eval,vllm,,acc_norm,none,0.81,0.039427724440366255,100 +SAMPLE — v2zloss_86k,global_piqa_completions_nld_latn,0,lm-eval,vllm,,acc,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_piqa_completions_nld_latn,0,lm-eval,vllm,,acc_bytes,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_piqa_completions_nld_latn,0,lm-eval,vllm,,acc_norm,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_piqa_completions_nno_latn,0,lm-eval,vllm,,acc,none,0.76,0.04292346959909278,100 +SAMPLE — v2zloss_86k,global_piqa_completions_nno_latn,0,lm-eval,vllm,,acc_bytes,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_piqa_completions_nno_latn,0,lm-eval,vllm,,acc_norm,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_piqa_completions_nob_latn,0,lm-eval,vllm,,acc,none,0.68,0.046882617226215076,100 +SAMPLE — v2zloss_86k,global_piqa_completions_nob_latn,0,lm-eval,vllm,,acc_bytes,none,0.67,0.04725815626252609,100 +SAMPLE — v2zloss_86k,global_piqa_completions_nob_latn,0,lm-eval,vllm,,acc_norm,none,0.66,0.04760952285695234,100 +SAMPLE — v2zloss_86k,global_piqa_completions_pol_latn,0,lm-eval,vllm,,acc,none,0.77,0.042295258468165065,100 +SAMPLE — v2zloss_86k,global_piqa_completions_pol_latn,0,lm-eval,vllm,,acc_bytes,none,0.75,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_piqa_completions_pol_latn,0,lm-eval,vllm,,acc_norm,none,0.75,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_piqa_completions_por_latn_port,0,lm-eval,vllm,,acc,none,0.76,0.04292346959909278,100 +SAMPLE — v2zloss_86k,global_piqa_completions_por_latn_port,0,lm-eval,vllm,,acc_bytes,none,0.7,0.04605661864718383,100 +SAMPLE — v2zloss_86k,global_piqa_completions_por_latn_port,0,lm-eval,vllm,,acc_norm,none,0.72,0.045126085985421296,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ron_latn,0,lm-eval,vllm,,acc,none,0.74,0.0440844002276808,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ron_latn,0,lm-eval,vllm,,acc_bytes,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ron_latn,0,lm-eval,vllm,,acc_norm,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_piqa_completions_slk_latn,0,lm-eval,vllm,,acc,none,0.83,0.03775251680686369,100 +SAMPLE — v2zloss_86k,global_piqa_completions_slk_latn,0,lm-eval,vllm,,acc_bytes,none,0.85,0.03588702812826367,100 +SAMPLE — v2zloss_86k,global_piqa_completions_slk_latn,0,lm-eval,vllm,,acc_norm,none,0.84,0.03684529491774706,100 +SAMPLE — v2zloss_86k,global_piqa_completions_slv_latn,0,lm-eval,vllm,,acc,none,0.65,0.04793724854411023,100 +SAMPLE — v2zloss_86k,global_piqa_completions_slv_latn,0,lm-eval,vllm,,acc_bytes,none,0.69,0.046482319871173176,100 +SAMPLE — v2zloss_86k,global_piqa_completions_slv_latn,0,lm-eval,vllm,,acc_norm,none,0.69,0.046482319871173176,100 +SAMPLE — v2zloss_86k,global_piqa_completions_spa_latn_spai,0,lm-eval,vllm,,acc,none,0.82,0.03861229196653691,100 +SAMPLE — v2zloss_86k,global_piqa_completions_spa_latn_spai,0,lm-eval,vllm,,acc_bytes,none,0.82,0.03861229196653691,100 +SAMPLE — v2zloss_86k,global_piqa_completions_spa_latn_spai,0,lm-eval,vllm,,acc_norm,none,0.83,0.03775251680686369,100 +SAMPLE — v2zloss_86k,global_piqa_completions_srp_cyrl,0,lm-eval,vllm,,acc,none,0.89,0.031446603773522014,100 +SAMPLE — v2zloss_86k,global_piqa_completions_srp_cyrl,0,lm-eval,vllm,,acc_bytes,none,0.87,0.033799766898963114,100 +SAMPLE — v2zloss_86k,global_piqa_completions_srp_cyrl,0,lm-eval,vllm,,acc_norm,none,0.88,0.032659863237109045,100 +SAMPLE — v2zloss_86k,global_piqa_completions_swe_latn,0,lm-eval,vllm,,acc,none,0.85,0.03588702812826367,100 +SAMPLE — v2zloss_86k,global_piqa_completions_swe_latn,0,lm-eval,vllm,,acc_bytes,none,0.83,0.03775251680686369,100 +SAMPLE — v2zloss_86k,global_piqa_completions_swe_latn,0,lm-eval,vllm,,acc_norm,none,0.8,0.04020151261036849,100 +SAMPLE — v2zloss_86k,global_piqa_completions_tur_latn,0,lm-eval,vllm,,acc,none,0.57,0.049756985195624305,100 +SAMPLE — v2zloss_86k,global_piqa_completions_tur_latn,0,lm-eval,vllm,,acc_bytes,none,0.73,0.04461960433384737,100 +SAMPLE — v2zloss_86k,global_piqa_completions_tur_latn,0,lm-eval,vllm,,acc_norm,none,0.75,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ukr_cyrl,0,lm-eval,vllm,,acc,none,0.79,0.040936018074033236,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ukr_cyrl,0,lm-eval,vllm,,acc_bytes,none,0.81,0.039427724440366255,100 +SAMPLE — v2zloss_86k,global_piqa_completions_ukr_cyrl,0,lm-eval,vllm,,acc_norm,none,0.82,0.03861229196653691,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_als_latn,0,lm-eval,vllm,,exact_match,strict_match,0.23,0.04229525846816505,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_bos_latn,0,lm-eval,vllm,,exact_match,strict_match,0.33,0.04725815626252604,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_bul_cyrl,0,lm-eval,vllm,,exact_match,strict_match,0.28,0.04512608598542128,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_cat_latn,0,lm-eval,vllm,,exact_match,strict_match,0.31,0.04648231987117316,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_ces_latn,0,lm-eval,vllm,,exact_match,strict_match,0.33,0.047258156262526045,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_deu_latn,0,lm-eval,vllm,,exact_match,strict_match,0.34,0.04760952285695235,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_ekk_latn,0,lm-eval,vllm,,exact_match,strict_match,0.21,0.040936018074033256,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_ell_grek,0,lm-eval,vllm,,exact_match,strict_match,0.24,0.04292346959909283,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_eng_latn,0,lm-eval,vllm,,exact_match,strict_match,0.34,0.04760952285695235,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_fin_latn,0,lm-eval,vllm,,exact_match,strict_match,0.31,0.04648231987117316,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_fra_latn_fran,0,lm-eval,vllm,,exact_match,strict_match,0.35,0.04793724854411019,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_glg_latn,0,lm-eval,vllm,,exact_match,strict_match,0.28,0.04512608598542128,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_hrv_latn,0,lm-eval,vllm,,exact_match,strict_match,0.25,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_hun_latn,0,lm-eval,vllm,,exact_match,strict_match,0.34,0.04760952285695235,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_isl_latn,0,lm-eval,vllm,,exact_match,strict_match,0.33,0.04725815626252605,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_ita_latn,0,lm-eval,vllm,,exact_match,strict_match,0.24,0.04292346959909283,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_kat_geor,0,lm-eval,vllm,,exact_match,strict_match,0.24,0.04292346959909282,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_lit_latn,0,lm-eval,vllm,,exact_match,strict_match,0.34,0.04760952285695235,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_mkd_cyrl,0,lm-eval,vllm,,exact_match,strict_match,0.42,0.049604496374885836,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_nld_latn,0,lm-eval,vllm,,exact_match,strict_match,0.3,0.046056618647183814,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_nno_latn,0,lm-eval,vllm,,exact_match,strict_match,0.29,0.045604802157206845,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_nob_latn,0,lm-eval,vllm,,exact_match,strict_match,0.37,0.048523658709391,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_pol_latn,0,lm-eval,vllm,,exact_match,strict_match,0.32,0.04688261722621505,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_por_latn_port,0,lm-eval,vllm,,exact_match,strict_match,0.39,0.04902071300001975,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_ron_latn,0,lm-eval,vllm,,exact_match,strict_match,0.25,0.04351941398892446,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_slk_latn,0,lm-eval,vllm,,exact_match,strict_match,0.3,0.046056618647183814,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_slv_latn,0,lm-eval,vllm,,exact_match,strict_match,0.35,0.0479372485441102,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_spa_latn_spai,0,lm-eval,vllm,,exact_match,strict_match,0.37,0.04852365870939099,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_srp_cyrl,0,lm-eval,vllm,,exact_match,strict_match,0.29,0.04560480215720683,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_swe_latn,0,lm-eval,vllm,,exact_match,strict_match,0.38,0.048783173121456316,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_tur_latn,0,lm-eval,vllm,,exact_match,strict_match,0.32,0.046882617226215034,100 +SAMPLE — v2zloss_86k,global_piqa_prompted_ukr_cyrl,0,lm-eval,vllm,,exact_match,strict_match,0.3,0.046056618647183814,100 +SAMPLE — v2zloss_86k,gsm8k,4,lm-eval,vllm,,exact_match,flexible-extract,0.7391963608794542,0.012094252417332751,1319 +SAMPLE — v2zloss_86k,gsm8k,4,lm-eval,vllm,,exact_match,strict-match,0.7384382107657316,0.012105605733382456,1319 +SAMPLE — v2zloss_86k,hellaswag,0,lm-eval,vllm,,acc,none,0.5845449113722366,0.004917931778592766,10042 +SAMPLE — v2zloss_86k,hellaswag,0,lm-eval,vllm,,acc_norm,none,0.7772356104361681,0.0041525105563423115,10042 +SAMPLE — v2zloss_86k,hellaswag,10,lm-eval,vllm,,acc,none,0.611929894443338,0.004863147544177631,10042 +SAMPLE — v2zloss_86k,hellaswag,10,lm-eval,vllm,,acc_norm,none,0.8133837880900219,0.0038880689432921425,10042 +SAMPLE — v2zloss_86k,hellaswag_ca,0,lm-eval,vllm,,acc,none,0.48832917164260126,0.005208610091396434,9211 +SAMPLE — v2zloss_86k,hellaswag_ca,0,lm-eval,vllm,,acc_norm,none,0.6441211594832266,0.004988902901392588,9211 +SAMPLE — v2zloss_86k,hellaswag_da,0,lm-eval,vllm,,acc,none,0.5086512627619559,0.005182867840416907,9305 +SAMPLE — v2zloss_86k,hellaswag_da,0,lm-eval,vllm,,acc_norm,none,0.6662009672219237,0.0048888905605458275,9305 +SAMPLE — v2zloss_86k,hellaswag_de,0,lm-eval,vllm,,acc,none,0.5116353543979505,0.005164783503041605,9368 +SAMPLE — v2zloss_86k,hellaswag_de,0,lm-eval,vllm,,acc_norm,none,0.6609735269000854,0.0048911229337050294,9368 +SAMPLE — v2zloss_86k,hellaswag_es,0,lm-eval,vllm,,acc,none,0.541391081715383,0.005146802321210821,9374 +SAMPLE — v2zloss_86k,hellaswag_es,0,lm-eval,vllm,,acc_norm,none,0.7086622573074461,0.004693304355064702,9374 +SAMPLE — v2zloss_86k,hellaswag_eu,0,lm-eval,vllm,,acc,none,0.2902215018908698,0.004718038325618079,9255 +SAMPLE — v2zloss_86k,hellaswag_eu,0,lm-eval,vllm,,acc_norm,none,0.3462992976769314,0.004945959246768342,9255 +SAMPLE — v2zloss_86k,hellaswag_fr,0,lm-eval,vllm,,acc,none,0.527736131934033,0.005166507870466839,9338 +SAMPLE — v2zloss_86k,hellaswag_fr,0,lm-eval,vllm,,acc_norm,none,0.699400299850075,0.00474518882761414,9338 +SAMPLE — v2zloss_86k,hellaswag_hr,0,lm-eval,vllm,,acc,none,0.47143912997571535,0.005129621584807367,9471 +SAMPLE — v2zloss_86k,hellaswag_hr,0,lm-eval,vllm,,acc_norm,none,0.6218984267764756,0.004982978138075943,9471 +SAMPLE — v2zloss_86k,hellaswag_hu,0,lm-eval,vllm,,acc,none,0.42307270534049785,0.005173902649499095,9119 +SAMPLE — v2zloss_86k,hellaswag_hu,0,lm-eval,vllm,,acc_norm,none,0.5530211646013817,0.005206724060736212,9119 +SAMPLE — v2zloss_86k,hellaswag_it,0,lm-eval,vllm,,acc,none,0.5061459806374415,0.005214734293869396,9193 +SAMPLE — v2zloss_86k,hellaswag_it,0,lm-eval,vllm,,acc_norm,none,0.6738823017513326,0.004889609784915807,9193 +SAMPLE — v2zloss_86k,hellaswag_nl,0,lm-eval,vllm,,acc,none,0.5113869400971398,0.005193475397139516,9265 +SAMPLE — v2zloss_86k,hellaswag_nl,0,lm-eval,vllm,,acc_norm,none,0.666594711279007,0.004897990077337533,9265 +SAMPLE — v2zloss_86k,hellaswag_pt,0,lm-eval,vllm,,acc,none,0.5205331021779175,0.005200555050598286,9229 +SAMPLE — v2zloss_86k,hellaswag_pt,0,lm-eval,vllm,,acc_norm,none,0.6893487918517716,0.004817283881486349,9229 +SAMPLE — v2zloss_86k,hellaswag_ro,0,lm-eval,vllm,,acc,none,0.48696592752839374,0.00519867207683696,9245 +SAMPLE — v2zloss_86k,hellaswag_ro,0,lm-eval,vllm,,acc_norm,none,0.638182801514332,0.0049978956342485404,9245 +SAMPLE — v2zloss_86k,hellaswag_sk,0,lm-eval,vllm,,acc,none,0.45239852398523983,0.005110896921580549,9485 +SAMPLE — v2zloss_86k,hellaswag_sk,0,lm-eval,vllm,,acc_norm,none,0.5888244596731682,0.005052551912356063,9485 +SAMPLE — v2zloss_86k,hellaswag_sr,0,lm-eval,vllm,,acc,none,0.4709493068049529,0.005135299559821532,9449 +SAMPLE — v2zloss_86k,hellaswag_sr,0,lm-eval,vllm,,acc_norm,none,0.6171023388718383,0.005000921191926585,9449 +SAMPLE — v2zloss_86k,hellaswag_sv,0,lm-eval,vllm,,acc,none,0.5047138785354089,0.005235154161150903,9122 +SAMPLE — v2zloss_86k,hellaswag_sv,0,lm-eval,vllm,,acc_norm,none,0.6673975005481254,0.004933257835378732,9122 +SAMPLE — v2zloss_86k,hellaswag_uk,0,lm-eval,vllm,,acc,none,0.4400042512488043,0.005117668454840295,9409 +SAMPLE — v2zloss_86k,hellaswag_uk,0,lm-eval,vllm,,acc_norm,none,0.5745562759060474,0.00509728237378608,9409 +SAMPLE — v2zloss_86k,ifeval,0,lm-eval,vllm,,inst_level_loose_acc,none,0.36930455635491605,,541 +SAMPLE — v2zloss_86k,ifeval,0,lm-eval,vllm,,inst_level_strict_acc,none,0.35731414868105515,,541 +SAMPLE — v2zloss_86k,ifeval,0,lm-eval,vllm,,prompt_level_loose_acc,none,0.2476894639556377,0.0185761392851853,541 +SAMPLE — v2zloss_86k,ifeval,0,lm-eval,vllm,,prompt_level_strict_acc,none,0.23290203327171904,0.01818926607409182,541 +SAMPLE — v2zloss_86k,include_base_44_albanian,0,lm-eval,vllm,yes,acc,none,0.6279491833030852,0.020468336777727362, +SAMPLE — v2zloss_86k,include_base_44_albanian_arts_humanities,0,lm-eval,vllm,,acc,none,0.6412556053811659,0.032190792004199886,223 +SAMPLE — v2zloss_86k,include_base_44_albanian_business_commerce,0,lm-eval,vllm,,acc,none,0.6547085201793722,0.03191100192835789,223 +SAMPLE — v2zloss_86k,include_base_44_albanian_health_oriented_education,0,lm-eval,vllm,,acc,none,0.32,0.09521904571390466,25 +SAMPLE — v2zloss_86k,include_base_44_albanian_social_science,0,lm-eval,vllm,,acc,none,0.6363636363636364,0.06546202725664504,55 +SAMPLE — v2zloss_86k,include_base_44_albanian_stem,0,lm-eval,vllm,,acc,none,0.56,0.10132456102380442,25 +SAMPLE — v2zloss_86k,include_base_44_basque,0,lm-eval,vllm,yes,acc,none,0.394,0.02187429930168918, +SAMPLE — v2zloss_86k,include_base_44_basque_professional_certification,0,lm-eval,vllm,,acc,none,0.394,0.02187429930168918,500 +SAMPLE — v2zloss_86k,include_base_44_bulgarian,0,lm-eval,vllm,yes,acc,none,0.6,0.020777635923320874, +SAMPLE — v2zloss_86k,include_base_44_bulgarian_arts_humanities,0,lm-eval,vllm,,acc,none,0.668,0.029844039047465857,250 +SAMPLE — v2zloss_86k,include_base_44_bulgarian_social_science,0,lm-eval,vllm,,acc,none,0.544,0.031563285061213475,250 +SAMPLE — v2zloss_86k,include_base_44_bulgarian_stem,0,lm-eval,vllm,,acc,none,0.54,0.07119963311072636,50 +SAMPLE — v2zloss_86k,include_base_44_croatian,0,lm-eval,vllm,yes,acc,none,0.6036363636363636,0.020795545476142818, +SAMPLE — v2zloss_86k,include_base_44_croatian_arts_humanities,0,lm-eval,vllm,,acc,none,0.58,0.03127799950463661,250 +SAMPLE — v2zloss_86k,include_base_44_croatian_social_science,0,lm-eval,vllm,,acc,none,0.652,0.030186568464511673,250 +SAMPLE — v2zloss_86k,include_base_44_croatian_stem,0,lm-eval,vllm,,acc,none,0.48,0.0713714056959817,50 +SAMPLE — v2zloss_86k,include_base_44_dutch,0,lm-eval,vllm,yes,acc,none,0.5716878402903811,0.02109427667454283, +SAMPLE — v2zloss_86k,include_base_44_dutch_applied_science,0,lm-eval,vllm,,acc,none,0.7333333333333333,0.11818736805705576,15 +SAMPLE — v2zloss_86k,include_base_44_dutch_arts_humanities,0,lm-eval,vllm,,acc,none,0.5925925925925926,0.031585291311942286,243 +SAMPLE — v2zloss_86k,include_base_44_dutch_health_oriented_education,0,lm-eval,vllm,,acc,none,0.44,0.10132456102380442,25 +SAMPLE — v2zloss_86k,include_base_44_dutch_social_science,0,lm-eval,vllm,,acc,none,0.551440329218107,0.03197066660004076,243 +SAMPLE — v2zloss_86k,include_base_44_dutch_stem,0,lm-eval,vllm,,acc,none,0.6,0.1,25 +SAMPLE — v2zloss_86k,include_base_44_estonian,0,lm-eval,vllm,yes,acc,none,0.48660714285714285,0.03264826540013552, +SAMPLE — v2zloss_86k,include_base_44_estonian_applied_science,0,lm-eval,vllm,,acc,none,0.6428571428571429,0.13289435848843764,14 +SAMPLE — v2zloss_86k,include_base_44_estonian_arts_humanities,0,lm-eval,vllm,,acc,none,0.422360248447205,0.039049013190061876,161 +SAMPLE — v2zloss_86k,include_base_44_estonian_health_oriented_education,0,lm-eval,vllm,,acc,none,0.5,0.5,2 +SAMPLE — v2zloss_86k,include_base_44_estonian_social_science,0,lm-eval,vllm,,acc,none,0.36363636363636365,0.15212000482437738,11 +SAMPLE — v2zloss_86k,include_base_44_estonian_stem,0,lm-eval,vllm,,acc,none,0.75,0.07319250547113999,36 +SAMPLE — v2zloss_86k,include_base_44_finnish,0,lm-eval,vllm,yes,acc,none,0.47186932849364793,0.021312090945304524, +SAMPLE — v2zloss_86k,include_base_44_finnish_applied_science,0,lm-eval,vllm,,acc,none,0.5172413793103449,0.09443492370778726,29 +SAMPLE — v2zloss_86k,include_base_44_finnish_arts_humanities,0,lm-eval,vllm,,acc,none,0.4336283185840708,0.03303834808259587,226 +SAMPLE — v2zloss_86k,include_base_44_finnish_health_oriented_education,0,lm-eval,vllm,,acc,none,0.4888888888888889,0.0753592220347252,45 +SAMPLE — v2zloss_86k,include_base_44_finnish_social_science,0,lm-eval,vllm,,acc,none,0.504424778761062,0.0333320280633051,226 +SAMPLE — v2zloss_86k,include_base_44_finnish_stem,0,lm-eval,vllm,,acc,none,0.44,0.10132456102380442,25 +SAMPLE — v2zloss_86k,include_base_44_french,0,lm-eval,vllm,yes,acc,none,0.5536992840095465,0.023904078770187496, +SAMPLE — v2zloss_86k,include_base_44_french_arts_humanities,0,lm-eval,vllm,,acc,none,0.6052631578947368,0.03002638258737012,266 +SAMPLE — v2zloss_86k,include_base_44_french_driving_license,0,lm-eval,vllm,,acc,none,0.3404255319148936,0.06986570800554748,47 +SAMPLE — v2zloss_86k,include_base_44_french_health_oriented_education,0,lm-eval,vllm,,acc,none,0.5,0.12909944487358055,16 +SAMPLE — v2zloss_86k,include_base_44_french_social_science,0,lm-eval,vllm,,acc,none,0.581081081081081,0.057746002446083286,74 +SAMPLE — v2zloss_86k,include_base_44_french_stem,0,lm-eval,vllm,,acc,none,0.25,0.11180339887498948,16 +SAMPLE — v2zloss_86k,include_base_44_georgian,0,lm-eval,vllm,yes,acc,none,0.412,0.02203367799374094, +SAMPLE — v2zloss_86k,include_base_44_georgian_arts_humanities,0,lm-eval,vllm,,acc,none,0.412,0.02203367799374094,500 +SAMPLE — v2zloss_86k,include_base_44_german,0,lm-eval,vllm,yes,acc,none,0.38848920863309355,0.04173896375123696, +SAMPLE — v2zloss_86k,include_base_44_german_driving_license,0,lm-eval,vllm,,acc,none,0.34782608695652173,0.10154334054280735,23 +SAMPLE — v2zloss_86k,include_base_44_german_social_science,0,lm-eval,vllm,,acc,none,0.4065934065934066,0.05177678676654834,91 +SAMPLE — v2zloss_86k,include_base_44_german_stem,0,lm-eval,vllm,,acc,none,0.36,0.09797958971132711,25 +SAMPLE — v2zloss_86k,include_base_44_greek,0,lm-eval,vllm,yes,acc,none,0.46557971014492755,0.021126546348312698, +SAMPLE — v2zloss_86k,include_base_44_greek_arts_humanities,0,lm-eval,vllm,,acc,none,0.32432432432432434,0.07802030664724674,37 +SAMPLE — v2zloss_86k,include_base_44_greek_business_commerce,0,lm-eval,vllm,,acc,none,0.6190476190476191,0.06167391674311745,63 +SAMPLE — v2zloss_86k,include_base_44_greek_health_oriented_education,0,lm-eval,vllm,,acc,none,0.32,0.09521904571390466,25 +SAMPLE — v2zloss_86k,include_base_44_greek_medical_license,0,lm-eval,vllm,,acc,none,0.5,0.043355498476206004,134 +SAMPLE — v2zloss_86k,include_base_44_greek_professional_certification,0,lm-eval,vllm,,acc,none,0.43283582089552236,0.042962562216652206,134 +SAMPLE — v2zloss_86k,include_base_44_greek_social_science,0,lm-eval,vllm,,acc,none,0.4626865671641791,0.043234602868397205,134 +SAMPLE — v2zloss_86k,include_base_44_greek_stem,0,lm-eval,vllm,,acc,none,0.44,0.10132456102380442,25 +SAMPLE — v2zloss_86k,include_base_44_hungarian,0,lm-eval,vllm,yes,acc,none,0.41454545454545455,0.021026637229283336, +SAMPLE — v2zloss_86k,include_base_44_hungarian_applied_science,0,lm-eval,vllm,,acc,none,0.39589442815249265,0.026522023584755073,341 +SAMPLE — v2zloss_86k,include_base_44_hungarian_social_science,0,lm-eval,vllm,,acc,none,0.43478260869565216,0.03664530116751979,184 +SAMPLE — v2zloss_86k,include_base_44_hungarian_stem,0,lm-eval,vllm,,acc,none,0.52,0.10198039027185571,25 +SAMPLE — v2zloss_86k,include_base_44_italian,0,lm-eval,vllm,yes,acc,none,0.6167883211678832,0.020413234199736018, +SAMPLE — v2zloss_86k,include_base_44_italian_applied_science,0,lm-eval,vllm,,acc,none,0.5714285714285714,0.08486978939800065,35 +SAMPLE — v2zloss_86k,include_base_44_italian_arts_humanities,0,lm-eval,vllm,,acc,none,0.5149700598802395,0.0387901286432998,167 +SAMPLE — v2zloss_86k,include_base_44_italian_health_oriented_education,0,lm-eval,vllm,,acc,none,0.75,0.1305582419667734,12 +SAMPLE — v2zloss_86k,include_base_44_italian_professional_certification,0,lm-eval,vllm,,acc,none,0.5741935483870968,0.03984509920961712,155 +SAMPLE — v2zloss_86k,include_base_44_italian_social_science,0,lm-eval,vllm,,acc,none,0.7604790419161677,0.03312541599851859,167 +SAMPLE — v2zloss_86k,include_base_44_italian_stem,0,lm-eval,vllm,,acc,none,0.5833333333333334,0.1486470975026408,12 +SAMPLE — v2zloss_86k,include_base_44_lithuanian,0,lm-eval,vllm,yes,acc,none,0.50187265917603,0.021716755375490372, +SAMPLE — v2zloss_86k,include_base_44_lithuanian_arts_humanities,0,lm-eval,vllm,,acc,none,0.4955223880597015,0.027357685703287414,335 +SAMPLE — v2zloss_86k,include_base_44_lithuanian_business_commerce,0,lm-eval,vllm,,acc,none,0.5,0.08006407690254357,40 +SAMPLE — v2zloss_86k,include_base_44_lithuanian_professional_certification,0,lm-eval,vllm,,acc,none,0.5,0.07293249574894728,48 +SAMPLE — v2zloss_86k,include_base_44_lithuanian_social_science,0,lm-eval,vllm,,acc,none,0.4935064935064935,0.05734909653459641,77 +SAMPLE — v2zloss_86k,include_base_44_lithuanian_stem,0,lm-eval,vllm,,acc,none,0.5882352941176471,0.08567283308875519,34 +SAMPLE — v2zloss_86k,include_base_44_north macedonian,0,lm-eval,vllm,yes,acc,none,0.6860254083484574,0.019672360562804077, +SAMPLE — v2zloss_86k,include_base_44_north macedonian_arts_humanities,0,lm-eval,vllm,,acc,none,0.7142857142857143,0.030251682134830482,224 +SAMPLE — v2zloss_86k,include_base_44_north macedonian_business_commerce,0,lm-eval,vllm,,acc,none,0.6875,0.03103908645389514,224 +SAMPLE — v2zloss_86k,include_base_44_north macedonian_social_science,0,lm-eval,vllm,,acc,none,0.7358490566037735,0.06113906319252698,53 +SAMPLE — v2zloss_86k,include_base_44_north macedonian_stem,0,lm-eval,vllm,,acc,none,0.5,0.07142857142857142,50 +SAMPLE — v2zloss_86k,include_base_44_polish,0,lm-eval,vllm,yes,acc,none,0.39233576642335766,0.02079135191460003, +SAMPLE — v2zloss_86k,include_base_44_polish_professional_certification,0,lm-eval,vllm,,acc,none,0.4032258064516129,0.022048374523245283,496 +SAMPLE — v2zloss_86k,include_base_44_polish_social_science,0,lm-eval,vllm,,acc,none,0.75,0.25,4 +SAMPLE — v2zloss_86k,include_base_44_polish_stem,0,lm-eval,vllm,,acc,none,0.25,0.06316139407998893,48 +SAMPLE — v2zloss_86k,include_base_44_portuguese,0,lm-eval,vllm,yes,acc,none,0.5263157894736842,0.021193348739692124, +SAMPLE — v2zloss_86k,include_base_44_portuguese_applied_science,0,lm-eval,vllm,,acc,none,0.47619047619047616,0.05481987004287438,84 +SAMPLE — v2zloss_86k,include_base_44_portuguese_arts_humanities,0,lm-eval,vllm,,acc,none,0.564935064935065,0.0400802656558681,154 +SAMPLE — v2zloss_86k,include_base_44_portuguese_business_commerce,0,lm-eval,vllm,,acc,none,0.4027777777777778,0.05820650942569535,72 +SAMPLE — v2zloss_86k,include_base_44_portuguese_health_oriented_education,0,lm-eval,vllm,,acc,none,0.5483870967741935,0.06371796062332931,62 +SAMPLE — v2zloss_86k,include_base_44_portuguese_social_science,0,lm-eval,vllm,,acc,none,0.5844155844155844,0.03984233708298028,154 +SAMPLE — v2zloss_86k,include_base_44_portuguese_stem,0,lm-eval,vllm,,acc,none,0.4,0.1,25 +SAMPLE — v2zloss_86k,include_base_44_serbian,0,lm-eval,vllm,yes,acc,none,0.5581818181818182,0.021069492449930012, +SAMPLE — v2zloss_86k,include_base_44_serbian_arts_humanities,0,lm-eval,vllm,,acc,none,0.5303514376996805,0.02825472448749027,313 +SAMPLE — v2zloss_86k,include_base_44_serbian_social_science,0,lm-eval,vllm,,acc,none,0.6363636363636364,0.03527198153014412,187 +SAMPLE — v2zloss_86k,include_base_44_serbian_stem,0,lm-eval,vllm,,acc,none,0.44,0.07091242083423346,50 +SAMPLE — v2zloss_86k,include_base_44_spanish,0,lm-eval,vllm,yes,acc,none,0.5472727272727272,0.021113839200728176, +SAMPLE — v2zloss_86k,include_base_44_spanish_arts_humanities,0,lm-eval,vllm,,acc,none,0.496,0.0316851985511992,250 +SAMPLE — v2zloss_86k,include_base_44_spanish_health_oriented_education,0,lm-eval,vllm,,acc,none,0.52,0.10198039027185571,25 +SAMPLE — v2zloss_86k,include_base_44_spanish_social_science,0,lm-eval,vllm,,acc,none,0.616,0.030821679117375447,250 +SAMPLE — v2zloss_86k,include_base_44_spanish_stem,0,lm-eval,vllm,,acc,none,0.4,0.1,25 +SAMPLE — v2zloss_86k,include_base_44_turkish,0,lm-eval,vllm,yes,acc,none,0.46167883211678834,0.020367652206262448, +SAMPLE — v2zloss_86k,include_base_44_turkish_arts_humanities,0,lm-eval,vllm,,acc,none,0.4819277108433735,0.03889951252827222,166 +SAMPLE — v2zloss_86k,include_base_44_turkish_business_commerce,0,lm-eval,vllm,,acc,none,0.6506024096385542,0.0371172519074075,166 +SAMPLE — v2zloss_86k,include_base_44_turkish_social_science,0,lm-eval,vllm,,acc,none,0.3373493975903614,0.03680783690727582,166 +SAMPLE — v2zloss_86k,include_base_44_turkish_stem,0,lm-eval,vllm,,acc,none,0.18,0.054883922035138706,50 +SAMPLE — v2zloss_86k,include_base_44_ukrainian,0,lm-eval,vllm,yes,acc,none,0.5381818181818182,0.02128964836422063, +SAMPLE — v2zloss_86k,include_base_44_ukrainian_arts_humanities,0,lm-eval,vllm,,acc,none,0.516,0.03166998503010743,250 +SAMPLE — v2zloss_86k,include_base_44_ukrainian_social_science,0,lm-eval,vllm,,acc,none,0.548,0.03153986449255664,250 +SAMPLE — v2zloss_86k,include_base_44_ukrainian_stem,0,lm-eval,vllm,,acc,none,0.6,0.06998542122237651,50 +SAMPLE — v2zloss_86k,jeopardy,10,lm-eval,vllm,,exact_match,strict-match,0.5134624468587624,0.010865624557948508,2117 +SAMPLE — v2zloss_86k,lambada_openai,0,lm-eval,vllm,,acc,none,0.7071608771589365,0.006339948563069273,5153 +SAMPLE — v2zloss_86k,lambada_openai,0,lm-eval,vllm,,perplexity,none,3.8544005693753536,0.0821824603682731,5153 +SAMPLE — v2zloss_86k,mbpp,3,lm-eval,vllm,,pass_at_1,none,0.538,0.022318338119870527,500 +SAMPLE — v2zloss_86k,mgsm_native_cot_de,5,lm-eval,vllm,,exact_match,flexible-extract,0.644,0.030343680657153215,250 +SAMPLE — v2zloss_86k,mgsm_native_cot_de,5,lm-eval,vllm,,exact_match,strict-match,0.676,0.02965829492454558,250 +SAMPLE — v2zloss_86k,mgsm_native_cot_en,5,lm-eval,vllm,,exact_match,flexible-extract,0.8,0.025348970020979078,250 +SAMPLE — v2zloss_86k,mgsm_native_cot_en,5,lm-eval,vllm,,exact_match,strict-match,0.732,0.028068762382526688,250 +SAMPLE — v2zloss_86k,mgsm_native_cot_es,5,lm-eval,vllm,,exact_match,flexible-extract,0.76,0.02706529365223901,250 +SAMPLE — v2zloss_86k,mgsm_native_cot_es,5,lm-eval,vllm,,exact_match,strict-match,0.7,0.029040893477575852,250 +SAMPLE — v2zloss_86k,mgsm_native_cot_fr,5,lm-eval,vllm,,exact_match,flexible-extract,0.644,0.03034368065715321,250 +SAMPLE — v2zloss_86k,mgsm_native_cot_fr,5,lm-eval,vllm,,exact_match,strict-match,0.68,0.029561724955241037,250 +SAMPLE — v2zloss_86k,mmlu,5,lm-eval,vllm,yes,acc,none,0.6600199401794616,0.003797925020866869, +SAMPLE — v2zloss_86k,mmlu_abstract_algebra,5,lm-eval,vllm,,acc,none,0.41,0.04943110704237104,100 +SAMPLE — v2zloss_86k,mmlu_anatomy,5,lm-eval,vllm,,acc,none,0.5777777777777777,0.042667634040995855,135 +SAMPLE — v2zloss_86k,mmlu_astronomy,5,lm-eval,vllm,,acc,none,0.7302631578947368,0.036117805602848975,152 +SAMPLE — v2zloss_86k,mmlu_business_ethics,5,lm-eval,vllm,,acc,none,0.66,0.04760952285695234,100 +SAMPLE — v2zloss_86k,mmlu_clinical_knowledge,5,lm-eval,vllm,,acc,none,0.6981132075471698,0.028254200344438735,265 +SAMPLE — v2zloss_86k,mmlu_college_biology,5,lm-eval,vllm,,acc,none,0.7569444444444444,0.03586879280080337,144 +SAMPLE — v2zloss_86k,mmlu_college_chemistry,5,lm-eval,vllm,,acc,none,0.54,0.05009082659620331,100 +SAMPLE — v2zloss_86k,mmlu_college_computer_science,5,lm-eval,vllm,,acc,none,0.62,0.04878317312145634,100 +SAMPLE — v2zloss_86k,mmlu_college_mathematics,5,lm-eval,vllm,,acc,none,0.43,0.049756985195624305,100 +SAMPLE — v2zloss_86k,mmlu_college_medicine,5,lm-eval,vllm,,acc,none,0.6358381502890174,0.03669072477416912,173 +SAMPLE — v2zloss_86k,mmlu_college_physics,5,lm-eval,vllm,,acc,none,0.38235294117647056,0.048355036961072254,102 +SAMPLE — v2zloss_86k,mmlu_computer_security,5,lm-eval,vllm,,acc,none,0.75,0.04351941398892446,100 +SAMPLE — v2zloss_86k,mmlu_conceptual_physics,5,lm-eval,vllm,,acc,none,0.7489361702127659,0.02834696377716252,235 +SAMPLE — v2zloss_86k,mmlu_econometrics,5,lm-eval,vllm,,acc,none,0.5614035087719298,0.04668000738510451,114 +SAMPLE — v2zloss_86k,mmlu_electrical_engineering,5,lm-eval,vllm,,acc,none,0.6896551724137931,0.03855289616378948,145 +SAMPLE — v2zloss_86k,mmlu_elementary_mathematics,5,lm-eval,vllm,,acc,none,0.5714285714285714,0.025487187147859396,378 +SAMPLE — v2zloss_86k,mmlu_formal_logic,5,lm-eval,vllm,,acc,none,0.5952380952380952,0.043902592653775656,126 +SAMPLE — v2zloss_86k,mmlu_global_facts,5,lm-eval,vllm,,acc,none,0.37,0.048523658709390974,100 +SAMPLE — v2zloss_86k,mmlu_high_school_biology,5,lm-eval,vllm,,acc,none,0.8193548387096774,0.021886178567172513,310 +SAMPLE — v2zloss_86k,mmlu_high_school_chemistry,5,lm-eval,vllm,,acc,none,0.6157635467980296,0.03422398565657553,203 +SAMPLE — v2zloss_86k,mmlu_high_school_computer_science,5,lm-eval,vllm,,acc,none,0.73,0.04461960433384737,100 +SAMPLE — v2zloss_86k,mmlu_high_school_european_history,5,lm-eval,vllm,,acc,none,0.7818181818181819,0.032250781083062896,165 +SAMPLE — v2zloss_86k,mmlu_high_school_geography,5,lm-eval,vllm,,acc,none,0.8383838383838383,0.02622591986362927,198 +SAMPLE — v2zloss_86k,mmlu_high_school_government_and_politics,5,lm-eval,vllm,,acc,none,0.9067357512953368,0.020986854593289705,193 +SAMPLE — v2zloss_86k,mmlu_high_school_macroeconomics,5,lm-eval,vllm,,acc,none,0.717948717948718,0.02281581309889655,390 +SAMPLE — v2zloss_86k,mmlu_high_school_mathematics,5,lm-eval,vllm,,acc,none,0.45555555555555555,0.03036486250482447,270 +SAMPLE — v2zloss_86k,mmlu_high_school_microeconomics,5,lm-eval,vllm,,acc,none,0.7941176470588235,0.026265024608275907,238 +SAMPLE — v2zloss_86k,mmlu_high_school_physics,5,lm-eval,vllm,,acc,none,0.45695364238410596,0.0406732517424744,151 +SAMPLE — v2zloss_86k,mmlu_high_school_psychology,5,lm-eval,vllm,,acc,none,0.8605504587155963,0.014852421490033118,545 +SAMPLE — v2zloss_86k,mmlu_high_school_statistics,5,lm-eval,vllm,,acc,none,0.6435185185185185,0.03266478331527272,216 +SAMPLE — v2zloss_86k,mmlu_high_school_us_history,5,lm-eval,vllm,,acc,none,0.803921568627451,0.027865942286639318,204 +SAMPLE — v2zloss_86k,mmlu_high_school_world_history,5,lm-eval,vllm,,acc,none,0.7974683544303798,0.02616056824660147,237 +SAMPLE — v2zloss_86k,mmlu_human_aging,5,lm-eval,vllm,,acc,none,0.7399103139013453,0.029442495585857424,223 +SAMPLE — v2zloss_86k,mmlu_human_sexuality,5,lm-eval,vllm,,acc,none,0.8091603053435115,0.03446513350752599,131 +SAMPLE — v2zloss_86k,mmlu_humanities,5,lm-eval,vllm,yes,acc,none,0.5978746014877789,0.0067319816887166945, +SAMPLE — v2zloss_86k,mmlu_international_law,5,lm-eval,vllm,,acc,none,0.7933884297520661,0.03695980128098826,121 +SAMPLE — v2zloss_86k,mmlu_jurisprudence,5,lm-eval,vllm,,acc,none,0.7407407407407407,0.04236511258094632,108 +SAMPLE — v2zloss_86k,mmlu_logical_fallacies,5,lm-eval,vllm,,acc,none,0.7852760736196319,0.032262193772867716,163 +SAMPLE — v2zloss_86k,mmlu_machine_learning,5,lm-eval,vllm,,acc,none,0.4375,0.04708567521880525,112 +SAMPLE — v2zloss_86k,mmlu_management,5,lm-eval,vllm,,acc,none,0.7961165048543689,0.0398913985953177,103 +SAMPLE — v2zloss_86k,mmlu_marketing,5,lm-eval,vllm,,acc,none,0.8931623931623932,0.02023714900899092,234 +SAMPLE — v2zloss_86k,mmlu_medical_genetics,5,lm-eval,vllm,,acc,none,0.68,0.046882617226215076,100 +SAMPLE — v2zloss_86k,mmlu_miscellaneous,5,lm-eval,vllm,,acc,none,0.7624521072796935,0.015218733046150228,783 +SAMPLE — v2zloss_86k,mmlu_moral_disputes,5,lm-eval,vllm,,acc,none,0.7485549132947977,0.023357365785874006,346 +SAMPLE — v2zloss_86k,mmlu_moral_scenarios,5,lm-eval,vllm,,acc,none,0.3474860335195531,0.015925564060208213,895 +SAMPLE — v2zloss_86k,mmlu_nutrition,5,lm-eval,vllm,,acc,none,0.7222222222222222,0.025646863097137932,306 +SAMPLE — v2zloss_86k,mmlu_other,5,lm-eval,vllm,yes,acc,none,0.6871580302542646,0.008050355270879028, +SAMPLE — v2zloss_86k,mmlu_philosophy,5,lm-eval,vllm,,acc,none,0.729903536977492,0.02521804037341055,311 +SAMPLE — v2zloss_86k,mmlu_prehistory,5,lm-eval,vllm,,acc,none,0.7283950617283951,0.024748624490537455,324 +SAMPLE — v2zloss_86k,mmlu_professional_accounting,5,lm-eval,vllm,,acc,none,0.475177304964539,0.0297907192438297,282 +SAMPLE — v2zloss_86k,mmlu_professional_law,5,lm-eval,vllm,,acc,none,0.5091264667535854,0.01276810860163994,1534 +SAMPLE — v2zloss_86k,mmlu_professional_medicine,5,lm-eval,vllm,,acc,none,0.6397058823529411,0.02916312857067069,272 +SAMPLE — v2zloss_86k,mmlu_professional_psychology,5,lm-eval,vllm,,acc,none,0.696078431372549,0.018607552131279893,612 +SAMPLE — v2zloss_86k,mmlu_public_relations,5,lm-eval,vllm,,acc,none,0.7090909090909091,0.04350271442923247,110 +SAMPLE — v2zloss_86k,mmlu_security_studies,5,lm-eval,vllm,,acc,none,0.746938775510204,0.027833023871399704,245 +SAMPLE — v2zloss_86k,mmlu_social_sciences,5,lm-eval,vllm,yes,acc,none,0.7747806304842378,0.007402903106299554, +SAMPLE — v2zloss_86k,mmlu_sociology,5,lm-eval,vllm,,acc,none,0.8308457711442786,0.026508590656233278,201 +SAMPLE — v2zloss_86k,mmlu_stem,5,lm-eval,vllm,yes,acc,none,0.6140183951791944,0.00837002950399499, +SAMPLE — v2zloss_86k,mmlu_us_foreign_policy,5,lm-eval,vllm,,acc,none,0.81,0.039427724440366255,100 +SAMPLE — v2zloss_86k,mmlu_virology,5,lm-eval,vllm,,acc,none,0.5240963855421686,0.03887971849597267,166 +SAMPLE — v2zloss_86k,mmlu_world_religions,5,lm-eval,vllm,,acc,none,0.8070175438596491,0.030267457554898448,171 +SAMPLE — v2zloss_86k,multiblimp_bul,0,lm-eval,vllm,,acc,none,0.9580960130187144,0.004042309956300658,2458 +SAMPLE — v2zloss_86k,multiblimp_bul,0,lm-eval,vllm,,acc_norm,none,0.951586655817738,0.004330161903584474,2458 +SAMPLE — v2zloss_86k,multiblimp_cat,0,lm-eval,vllm,,acc,none,0.9737302977232924,0.0033472947883197695,2284 +SAMPLE — v2zloss_86k,multiblimp_cat,0,lm-eval,vllm,,acc_norm,none,0.973292469352014,0.0033743147747588515,2284 +SAMPLE — v2zloss_86k,multiblimp_ces,0,lm-eval,vllm,,acc,none,0.9292763157894737,0.003930113476658942,4256 +SAMPLE — v2zloss_86k,multiblimp_ces,0,lm-eval,vllm,,acc_norm,none,0.9283364661654135,0.0039541399195811575,4256 +SAMPLE — v2zloss_86k,multiblimp_dan,0,lm-eval,vllm,,acc,none,1.0,0.0,50 +SAMPLE — v2zloss_86k,multiblimp_dan,0,lm-eval,vllm,,acc_norm,none,1.0,0.0,50 +SAMPLE — v2zloss_86k,multiblimp_deu,0,lm-eval,vllm,,acc,none,0.9830287206266318,0.0026950069966784774,2298 +SAMPLE — v2zloss_86k,multiblimp_deu,0,lm-eval,vllm,,acc_norm,none,0.97911227154047,0.00298388001559491,2298 +SAMPLE — v2zloss_86k,multiblimp_ell,0,lm-eval,vllm,,acc,none,0.9917883211678832,0.002727208947237988,1096 +SAMPLE — v2zloss_86k,multiblimp_ell,0,lm-eval,vllm,,acc_norm,none,0.9854014598540146,0.003624551336944677,1096 +SAMPLE — v2zloss_86k,multiblimp_eng,0,lm-eval,vllm,,acc,none,0.987012987012987,0.004082751067819237,770 +SAMPLE — v2zloss_86k,multiblimp_eng,0,lm-eval,vllm,,acc_norm,none,0.9818181818181818,0.00481804687014778,770 +SAMPLE — v2zloss_86k,multiblimp_est,0,lm-eval,vllm,,acc,none,0.8866019417475728,0.006249753332723915,2575 +SAMPLE — v2zloss_86k,multiblimp_est,0,lm-eval,vllm,,acc_norm,none,0.8854368932038835,0.006277647516083091,2575 +SAMPLE — v2zloss_86k,multiblimp_eus,0,lm-eval,vllm,,acc,none,0.9633699633699634,0.011390184926989774,273 +SAMPLE — v2zloss_86k,multiblimp_eus,0,lm-eval,vllm,,acc_norm,none,0.63003663003663,0.02927371304052674,273 +SAMPLE — v2zloss_86k,multiblimp_fin,0,lm-eval,vllm,,acc,none,0.9089494163424124,0.005675827298368331,2570 +SAMPLE — v2zloss_86k,multiblimp_fin,0,lm-eval,vllm,,acc_norm,none,0.8964980544747082,0.006009894941848468,2570 +SAMPLE — v2zloss_86k,multiblimp_fra,0,lm-eval,vllm,,acc,none,0.9878335949764521,0.0021722437610319687,2548 +SAMPLE — v2zloss_86k,multiblimp_fra,0,lm-eval,vllm,,acc_norm,none,0.9823390894819466,0.0026098934986873044,2548 +SAMPLE — v2zloss_86k,multiblimp_gle,0,lm-eval,vllm,,acc,none,0.7857142857142857,0.0789672569132238,28 +SAMPLE — v2zloss_86k,multiblimp_gle,0,lm-eval,vllm,,acc_norm,none,0.6428571428571429,0.09221388919541469,28 +SAMPLE — v2zloss_86k,multiblimp_glg,0,lm-eval,vllm,,acc,none,0.9415670650730412,0.008553533474378566,753 +SAMPLE — v2zloss_86k,multiblimp_glg,0,lm-eval,vllm,,acc_norm,none,0.9256308100929614,0.009567677016053744,753 +SAMPLE — v2zloss_86k,multiblimp_hbs,0,lm-eval,vllm,,acc,none,0.965611685940353,0.0031793549174727034,3286 +SAMPLE — v2zloss_86k,multiblimp_hbs,0,lm-eval,vllm,,acc_norm,none,0.9643944004869142,0.003233097519828991,3286 +SAMPLE — v2zloss_86k,multiblimp_hun,0,lm-eval,vllm,,acc,none,0.9763313609467456,0.005232557866188476,845 +SAMPLE — v2zloss_86k,multiblimp_hun,0,lm-eval,vllm,,acc_norm,none,0.9715976331360947,0.005718067360306842,845 +SAMPLE — v2zloss_86k,multiblimp_isl,0,lm-eval,vllm,,acc,none,0.9446626204926812,0.004320844574211562,2801 +SAMPLE — v2zloss_86k,multiblimp_isl,0,lm-eval,vllm,,acc_norm,none,0.938236344162799,0.004549289844691021,2801 +SAMPLE — v2zloss_86k,multiblimp_ita,0,lm-eval,vllm,,acc,none,0.9533177725908636,0.003852820850159222,2999 +SAMPLE — v2zloss_86k,multiblimp_ita,0,lm-eval,vllm,,acc_norm,none,0.9399799933311104,0.00433801960563242,2999 +SAMPLE — v2zloss_86k,multiblimp_kat,0,lm-eval,vllm,,acc,none,0.9166666666666666,0.019398452135813888,204 +SAMPLE — v2zloss_86k,multiblimp_kat,0,lm-eval,vllm,,acc_norm,none,0.8774509803921569,0.023015389732458234,204 +SAMPLE — v2zloss_86k,multiblimp_lav,0,lm-eval,vllm,,acc,none,0.941952506596306,0.004247303253458211,3032 +SAMPLE — v2zloss_86k,multiblimp_lav,0,lm-eval,vllm,,acc_norm,none,0.9403034300791556,0.004303439788744742,3032 +SAMPLE — v2zloss_86k,multiblimp_lit,0,lm-eval,vllm,,acc,none,0.9703389830508474,0.004940806613681997,1180 +SAMPLE — v2zloss_86k,multiblimp_lit,0,lm-eval,vllm,,acc_norm,none,0.9601694915254237,0.005695409745635608,1180 +SAMPLE — v2zloss_86k,multiblimp_mkd,0,lm-eval,vllm,,acc,none,0.8717948717948718,0.05423355275954148,39 +SAMPLE — v2zloss_86k,multiblimp_mkd,0,lm-eval,vllm,,acc_norm,none,0.8717948717948718,0.05423355275954148,39 +SAMPLE — v2zloss_86k,multiblimp_nld,0,lm-eval,vllm,,acc,none,0.9377949377949378,0.005003672146861063,2331 +SAMPLE — v2zloss_86k,multiblimp_nld,0,lm-eval,vllm,,acc_norm,none,0.9086229086229086,0.005969425625260505,2331 +SAMPLE — v2zloss_86k,multiblimp_pol,0,lm-eval,vllm,,acc,none,0.9437652811735942,0.004028042000218542,3272 +SAMPLE — v2zloss_86k,multiblimp_pol,0,lm-eval,vllm,,acc_norm,none,0.9238997555012225,0.004636232212090755,3272 +SAMPLE — v2zloss_86k,multiblimp_por,0,lm-eval,vllm,,acc,none,0.9645669291338582,0.0033491481047593233,3048 +SAMPLE — v2zloss_86k,multiblimp_por,0,lm-eval,vllm,,acc_norm,none,0.9603018372703412,0.0035371449442911295,3048 +SAMPLE — v2zloss_86k,multiblimp_ron,0,lm-eval,vllm,,acc,none,0.958171206225681,0.004416246591731525,2056 +SAMPLE — v2zloss_86k,multiblimp_ron,0,lm-eval,vllm,,acc_norm,none,0.9464980544747081,0.004964079602690654,2056 +SAMPLE — v2zloss_86k,multiblimp_slk,0,lm-eval,vllm,,acc,none,0.9256936067551267,0.004074148452129984,4145 +SAMPLE — v2zloss_86k,multiblimp_slk,0,lm-eval,vllm,,acc_norm,none,0.9266586248492159,0.004049715712940963,4145 +SAMPLE — v2zloss_86k,multiblimp_slv,0,lm-eval,vllm,,acc,none,0.9121124247155923,0.004229139406999552,4483 +SAMPLE — v2zloss_86k,multiblimp_slv,0,lm-eval,vllm,,acc_norm,none,0.8956056212357796,0.0045673157485991,4483 +SAMPLE — v2zloss_86k,multiblimp_spa,0,lm-eval,vllm,,acc,none,0.9763872491145218,0.0030127804460631574,2541 +SAMPLE — v2zloss_86k,multiblimp_spa,0,lm-eval,vllm,,acc_norm,none,0.9799291617473436,0.002782679818137993,2541 +SAMPLE — v2zloss_86k,multiblimp_sqi,0,lm-eval,vllm,,acc,none,0.9423868312757202,0.0149784820195792,243 +SAMPLE — v2zloss_86k,multiblimp_sqi,0,lm-eval,vllm,,acc_norm,none,0.897119341563786,0.01952919287113833,243 +SAMPLE — v2zloss_86k,multiblimp_swe,0,lm-eval,vllm,,acc,none,1.0,0.0,201 +SAMPLE — v2zloss_86k,multiblimp_swe,0,lm-eval,vllm,,acc_norm,none,0.9900497512437811,0.007018276606798947,201 +SAMPLE — v2zloss_86k,multiblimp_tur,0,lm-eval,vllm,,acc,none,0.8823191733639495,0.00772264956475455,1742 +SAMPLE — v2zloss_86k,multiblimp_tur,0,lm-eval,vllm,,acc_norm,none,0.8019517795637199,0.009551250036773418,1742 +SAMPLE — v2zloss_86k,multiblimp_ukr,0,lm-eval,vllm,,acc,none,0.9486151603498543,0.004215505201817796,2744 +SAMPLE — v2zloss_86k,multiblimp_ukr,0,lm-eval,vllm,,acc_norm,none,0.9358600583090378,0.004677963541819617,2744 +SAMPLE — v2zloss_86k,openbookqa,0,lm-eval,vllm,,acc,none,0.342,0.021236147199899316,500 +SAMPLE — v2zloss_86k,openbookqa,0,lm-eval,vllm,,acc_norm,none,0.466,0.022331264423258324,500 +SAMPLE — v2zloss_86k,opensubtitles_multi40_bg_to_en,0,lm-eval,vllm,,bleu,none,15.041273434568684,0.27865713479939225,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_bg_to_en,0,lm-eval,vllm,,chrf,none,33.51868944740595,0.2895441938589838,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_cs_to_en,0,lm-eval,vllm,,bleu,none,18.737924303742602,0.4524105518166843,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_cs_to_en,0,lm-eval,vllm,,chrf,none,37.02886034675586,0.29243497981692995,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_da_to_en,0,lm-eval,vllm,,bleu,none,23.639958175209976,0.4654460326268294,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_da_to_en,0,lm-eval,vllm,,chrf,none,43.80502587324379,0.30657585870620785,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_de_to_en,0,lm-eval,vllm,,bleu,none,21.55101840952253,0.44087220870504135,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_de_to_en,0,lm-eval,vllm,,chrf,none,41.993981257006816,0.3313497866484437,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_el_to_en,0,lm-eval,vllm,,bleu,none,26.242700146152767,0.32607949812934667,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_el_to_en,0,lm-eval,vllm,,chrf,none,44.07801529008939,0.2971020313957815,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_bg,0,lm-eval,vllm,,bleu,none,11.45825428369587,0.26223128555463104,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_bg,0,lm-eval,vllm,,chrf,none,33.80321381669062,0.3393769341160704,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_cs,0,lm-eval,vllm,,bleu,none,15.388623600607179,0.30598375016690577,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_cs,0,lm-eval,vllm,,chrf,none,35.937512134792236,0.28860066423257724,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_da,0,lm-eval,vllm,,bleu,none,27.865276921296413,0.3206208173956248,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_da,0,lm-eval,vllm,,chrf,none,49.04345816997705,0.2725122212503656,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_de,0,lm-eval,vllm,,bleu,none,22.332752089109807,0.29598150858132105,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_de,0,lm-eval,vllm,,chrf,none,45.253470554612974,0.2895526917224016,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_el,0,lm-eval,vllm,,bleu,none,18.972915105790012,0.28315574768422447,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_el,0,lm-eval,vllm,,chrf,none,36.4036746838506,0.36380908670672574,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_es,0,lm-eval,vllm,,bleu,none,26.270659675516406,0.36229598326326923,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_es,0,lm-eval,vllm,,chrf,none,44.43848638163801,0.3863608532627852,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_et,0,lm-eval,vllm,,bleu,none,20.45515294412305,0.3935412401907301,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_et,0,lm-eval,vllm,,chrf,none,42.42732153086645,0.2985317071676002,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_fi,0,lm-eval,vllm,,bleu,none,16.050508089732176,0.3202115622308699,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_fi,0,lm-eval,vllm,,chrf,none,43.488675773058034,0.2998974218101569,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_fr,0,lm-eval,vllm,,bleu,none,15.48987306505655,0.24815455635921638,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_fr,0,lm-eval,vllm,,chrf,none,40.52559027336866,0.29525448282944683,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_hr,0,lm-eval,vllm,,bleu,none,23.34566244990506,0.35369612609324,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_hr,0,lm-eval,vllm,,chrf,none,49.52529905057605,0.3065548330472121,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_hu,0,lm-eval,vllm,,bleu,none,15.879618838121337,0.2938035310029843,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_hu,0,lm-eval,vllm,,chrf,none,37.85484109393435,0.25948405684252707,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_it,0,lm-eval,vllm,,bleu,none,27.71613576568587,0.30029379210737595,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_it,0,lm-eval,vllm,,chrf,none,49.55012685991665,0.2538350972644054,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_lt,0,lm-eval,vllm,,bleu,none,17.26368216558132,0.3004101354690009,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_lt,0,lm-eval,vllm,,chrf,none,40.77688918966899,0.24163862622114504,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_lv,0,lm-eval,vllm,,bleu,none,18.361601082387303,0.36724219519875134,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_lv,0,lm-eval,vllm,,chrf,none,41.043438668384994,0.27663699162295924,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_nl,0,lm-eval,vllm,,bleu,none,19.21216767767043,0.274534605232773,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_nl,0,lm-eval,vllm,,chrf,none,43.67321042083741,0.27738651394101466,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_no,0,lm-eval,vllm,,bleu,none,29.887928642307223,0.3243870581671605,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_no,0,lm-eval,vllm,,chrf,none,52.533570716132616,0.33612165061641847,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_pl,0,lm-eval,vllm,,bleu,none,18.171419954259015,0.3150694218958703,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_pl,0,lm-eval,vllm,,chrf,none,41.630809743839116,0.3037334462190814,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_pt,0,lm-eval,vllm,,bleu,none,23.52042615075562,0.29423551692688815,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_pt,0,lm-eval,vllm,,chrf,none,47.6956243424614,0.32919466382074763,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_ro,0,lm-eval,vllm,,bleu,none,24.952444698026646,0.3917161194852725,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_ro,0,lm-eval,vllm,,chrf,none,47.63639636953957,0.3057480612245929,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_sk,0,lm-eval,vllm,,bleu,none,21.82976272129786,0.37656640839709066,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_sk,0,lm-eval,vllm,,chrf,none,42.77146158389196,0.2940690705731782,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_sl,0,lm-eval,vllm,,bleu,none,13.309549614571724,0.2619165443960259,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_sl,0,lm-eval,vllm,,chrf,none,37.94538626486992,0.28440756234666,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_sr,0,lm-eval,vllm,,bleu,none,6.406004085151879,0.2024505412662407,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_sr,0,lm-eval,vllm,,chrf,none,11.877465263554694,0.3082244973029031,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_sv,0,lm-eval,vllm,,bleu,none,31.079824655425746,0.35486087538285,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_sv,0,lm-eval,vllm,,chrf,none,49.60308450963951,0.3100203204532906,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_tr,0,lm-eval,vllm,,bleu,none,19.096769510347496,0.38462901691767803,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_tr,0,lm-eval,vllm,,chrf,none,44.93306656135502,0.32890151821682073,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_uk,0,lm-eval,vllm,,bleu,none,15.870078083769606,0.2810088929248753,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_en_to_uk,0,lm-eval,vllm,,chrf,none,36.627906499698135,0.258224523913718,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_es_to_en,0,lm-eval,vllm,,bleu,none,26.783004781823227,0.34686472489582937,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_es_to_en,0,lm-eval,vllm,,chrf,none,45.34924314434098,0.3038715149114251,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_et_to_en,0,lm-eval,vllm,,bleu,none,22.356708296550355,0.39736871681720287,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_et_to_en,0,lm-eval,vllm,,chrf,none,41.66371896819983,0.27743880439186924,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_fi_to_en,0,lm-eval,vllm,,bleu,none,19.870260725911685,0.29564762375652726,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_fi_to_en,0,lm-eval,vllm,,chrf,none,38.16021220963815,0.3001572534675677,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_fr_to_en,0,lm-eval,vllm,,bleu,none,16.689098942901623,0.24880907401307473,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_fr_to_en,0,lm-eval,vllm,,chrf,none,34.94340216460491,0.25360880701045513,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_hr_to_en,0,lm-eval,vllm,,bleu,none,25.650323686890875,0.3615387184202445,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_hr_to_en,0,lm-eval,vllm,,chrf,none,44.25743789299229,0.3480963724971564,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_hu_to_en,0,lm-eval,vllm,,bleu,none,20.00092615399439,0.29401566963516496,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_hu_to_en,0,lm-eval,vllm,,chrf,none,37.89938171874787,0.25464484880820765,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_it_to_en,0,lm-eval,vllm,,bleu,none,27.26278816425581,0.28601657844399686,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_it_to_en,0,lm-eval,vllm,,chrf,none,45.34353605132681,0.2703506374322918,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_lt_to_en,0,lm-eval,vllm,,bleu,none,19.283174862481467,0.4307200604255094,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_lt_to_en,0,lm-eval,vllm,,chrf,none,38.09781077474902,0.3142289861263512,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_lv_to_en,0,lm-eval,vllm,,bleu,none,21.042614035975962,0.42270400127945285,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_lv_to_en,0,lm-eval,vllm,,chrf,none,38.71471415192964,0.28851464406100213,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_nl_to_en,0,lm-eval,vllm,,bleu,none,21.08484290576437,0.2842626502968993,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_nl_to_en,0,lm-eval,vllm,,chrf,none,39.52740446889072,0.2641192748124852,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_no_to_en,0,lm-eval,vllm,,bleu,none,22.040142168727026,0.5035596750788149,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_no_to_en,0,lm-eval,vllm,,chrf,none,42.91040368362307,0.32322016044191704,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_pl_to_en,0,lm-eval,vllm,,bleu,none,23.780325286968008,0.30443031623913913,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_pl_to_en,0,lm-eval,vllm,,chrf,none,42.52734950228035,0.285064487398411,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_pt_to_en,0,lm-eval,vllm,,bleu,none,25.700692797895112,0.5121049105979106,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_pt_to_en,0,lm-eval,vllm,,chrf,none,45.13199563831567,0.34904728903179816,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_ro_to_en,0,lm-eval,vllm,,bleu,none,28.27351428488073,0.39928720427720543,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_ro_to_en,0,lm-eval,vllm,,chrf,none,45.3358209173333,0.3535215057708006,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_sk_to_en,0,lm-eval,vllm,,bleu,none,26.042350933177207,0.34072379475248754,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_sk_to_en,0,lm-eval,vllm,,chrf,none,42.90567337440091,0.32745888423382974,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_sl_to_en,0,lm-eval,vllm,,bleu,none,16.706022227733964,0.22164207737575928,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_sl_to_en,0,lm-eval,vllm,,chrf,none,34.6497128911344,0.25981129470089287,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_sr_to_en,0,lm-eval,vllm,,bleu,none,22.32255710572662,0.5503111630724108,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_sr_to_en,0,lm-eval,vllm,,chrf,none,43.180325873313876,0.3356989359073047,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_sv_to_en,0,lm-eval,vllm,,bleu,none,30.54059367624084,0.5630066300841026,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_sv_to_en,0,lm-eval,vllm,,chrf,none,48.23802924832672,0.3421074047642783,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_tr_to_en,0,lm-eval,vllm,,bleu,none,23.938063095109175,0.2879205393031149,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_tr_to_en,0,lm-eval,vllm,,chrf,none,43.24814338296646,0.29586205103157215,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_uk_to_en,0,lm-eval,vllm,,bleu,none,20.921486430943617,0.296425574990372,6005 +SAMPLE — v2zloss_86k,opensubtitles_multi40_uk_to_en,0,lm-eval,vllm,,chrf,none,39.067433313670115,0.2701228381819381,6005 +SAMPLE — v2zloss_86k,piqa,10,lm-eval,vllm,,acc,none,0.8150163220892275,0.009059313316508697,1838 +SAMPLE — v2zloss_86k,piqa,10,lm-eval,vllm,,acc_norm,none,0.8220892274211099,0.008922899948085554,1838 +SAMPLE — v2zloss_86k,polymath_de_high,0,lm-eval,vllm,,exact_match,none,0.0,0.0,125 +SAMPLE — v2zloss_86k,polymath_de_low,0,lm-eval,vllm,,exact_match,none,0.344,0.04265994570720801,125 +SAMPLE — v2zloss_86k,polymath_de_medium,0,lm-eval,vllm,,exact_match,none,0.016,0.01126799635851396,125 +SAMPLE — v2zloss_86k,polymath_de_top,0,lm-eval,vllm,,exact_match,none,0.0,0.0,125 +SAMPLE — v2zloss_86k,polymath_en_high,0,lm-eval,vllm,,exact_match,none,0.0,0.0,125 +SAMPLE — v2zloss_86k,polymath_en_low,0,lm-eval,vllm,,exact_match,none,0.48,0.04486539006635796,125 +SAMPLE — v2zloss_86k,polymath_en_medium,0,lm-eval,vllm,,exact_match,none,0.064,0.02197946255470202,125 +SAMPLE — v2zloss_86k,polymath_en_top,0,lm-eval,vllm,,exact_match,none,0.008,0.008,125 +SAMPLE — v2zloss_86k,polymath_es_high,0,lm-eval,vllm,,exact_match,none,0.016,0.01126799635851396,125 +SAMPLE — v2zloss_86k,polymath_es_low,0,lm-eval,vllm,,exact_match,none,0.384,0.043676228124985866,125 +SAMPLE — v2zloss_86k,polymath_es_medium,0,lm-eval,vllm,,exact_match,none,0.016,0.01126799635851396,125 +SAMPLE — v2zloss_86k,polymath_es_top,0,lm-eval,vllm,,exact_match,none,0.008,0.008,125 +SAMPLE — v2zloss_86k,polymath_fr_high,0,lm-eval,vllm,,exact_match,none,0.0,0.0,125 +SAMPLE — v2zloss_86k,polymath_fr_low,0,lm-eval,vllm,,exact_match,none,0.376,0.04349860954396202,125 +SAMPLE — v2zloss_86k,polymath_fr_medium,0,lm-eval,vllm,,exact_match,none,0.032,0.01580526657835619,125 +SAMPLE — v2zloss_86k,polymath_fr_top,0,lm-eval,vllm,,exact_match,none,0.0,0.0,125 +SAMPLE — v2zloss_86k,polymath_it_high,0,lm-eval,vllm,,exact_match,none,0.016,0.01126799635851396,125 +SAMPLE — v2zloss_86k,polymath_it_low,0,lm-eval,vllm,,exact_match,none,0.392,0.04384135623049351,125 +SAMPLE — v2zloss_86k,polymath_it_medium,0,lm-eval,vllm,,exact_match,none,0.032,0.01580526657835619,125 +SAMPLE — v2zloss_86k,polymath_it_top,0,lm-eval,vllm,,exact_match,none,0.0,0.0,125 +SAMPLE — v2zloss_86k,polymath_pt_high,0,lm-eval,vllm,,exact_match,none,0.024,0.013744206990818044,125 +SAMPLE — v2zloss_86k,polymath_pt_low,0,lm-eval,vllm,,exact_match,none,0.448,0.04465783896075986,125 +SAMPLE — v2zloss_86k,polymath_pt_medium,0,lm-eval,vllm,,exact_match,none,0.024,0.013744206990818044,125 +SAMPLE — v2zloss_86k,polymath_pt_top,0,lm-eval,vllm,,exact_match,none,0.0,0.0,125 +SAMPLE — v2zloss_86k,sib200_als_Latn,0,lm-eval,vllm,,acc,none,0.3872549019607843,0.0341893123383334,204 +SAMPLE — v2zloss_86k,sib200_als_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_bos_Latn,0,lm-eval,vllm,,acc,none,0.43137254901960786,0.034760990605016404,204 +SAMPLE — v2zloss_86k,sib200_bos_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_bul_Cyrl,0,lm-eval,vllm,,acc,none,0.47549019607843135,0.035050931943487976,204 +SAMPLE — v2zloss_86k,sib200_bul_Cyrl,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_cat_Latn,0,lm-eval,vllm,,acc,none,0.46078431372549017,0.03498501649369533,204 +SAMPLE — v2zloss_86k,sib200_cat_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_ces_Latn,0,lm-eval,vllm,,acc,none,0.4852941176470588,0.035077938347913305,204 +SAMPLE — v2zloss_86k,sib200_ces_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_dan_Latn,0,lm-eval,vllm,,acc,none,0.46568627450980393,0.03501038327635894,204 +SAMPLE — v2zloss_86k,sib200_dan_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_deu_Latn,0,lm-eval,vllm,,acc,none,0.47549019607843135,0.035050931943487976,204 +SAMPLE — v2zloss_86k,sib200_deu_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_ell_Grek,0,lm-eval,vllm,,acc,none,0.47549019607843135,0.035050931943487976,204 +SAMPLE — v2zloss_86k,sib200_ell_Grek,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_eng_Latn,0,lm-eval,vllm,,acc,none,0.5098039215686274,0.035086373586305744,204 +SAMPLE — v2zloss_86k,sib200_eng_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_est_Latn,0,lm-eval,vllm,,acc,none,0.5196078431372549,0.035066125605248695,204 +SAMPLE — v2zloss_86k,sib200_est_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_eus_Latn,0,lm-eval,vllm,,acc,none,0.5490196078431373,0.034924061041636124,204 +SAMPLE — v2zloss_86k,sib200_eus_Latn,0,lm-eval,vllm,,acc_norm,none,0.2549019607843137,0.030587591351604302,204 +SAMPLE — v2zloss_86k,sib200_fin_Latn,0,lm-eval,vllm,,acc,none,0.4950980392156863,0.03509143375606784,204 +SAMPLE — v2zloss_86k,sib200_fin_Latn,0,lm-eval,vllm,,acc_norm,none,0.2549019607843137,0.030587591351604302,204 +SAMPLE — v2zloss_86k,sib200_fra_Latn,0,lm-eval,vllm,,acc,none,0.4803921568627451,0.03506612560524869,204 +SAMPLE — v2zloss_86k,sib200_fra_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_gle_Latn,0,lm-eval,vllm,,acc,none,0.5245098039215687,0.035050931943487976,204 +SAMPLE — v2zloss_86k,sib200_gle_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_glg_Latn,0,lm-eval,vllm,,acc,none,0.45588235294117646,0.034956245220154746,204 +SAMPLE — v2zloss_86k,sib200_glg_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_hrv_Latn,0,lm-eval,vllm,,acc,none,0.47549019607843135,0.035050931943487976,204 +SAMPLE — v2zloss_86k,sib200_hrv_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_hun_Latn,0,lm-eval,vllm,,acc,none,0.3480392156862745,0.03343311240488421,204 +SAMPLE — v2zloss_86k,sib200_hun_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_isl_Latn,0,lm-eval,vllm,,acc,none,0.45588235294117646,0.034956245220154746,204 +SAMPLE — v2zloss_86k,sib200_isl_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_ita_Latn,0,lm-eval,vllm,,acc,none,0.4950980392156863,0.03509143375606784,204 +SAMPLE — v2zloss_86k,sib200_ita_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_kat_Geor,0,lm-eval,vllm,,acc,none,0.29411764705882354,0.031980016601150726,204 +SAMPLE — v2zloss_86k,sib200_kat_Geor,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_lit_Latn,0,lm-eval,vllm,,acc,none,0.5588235294117647,0.034849415144292274,204 +SAMPLE — v2zloss_86k,sib200_lit_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_lvs_Latn,0,lm-eval,vllm,,acc,none,0.47058823529411764,0.03503235296367989,204 +SAMPLE — v2zloss_86k,sib200_lvs_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_mkd_Cyrl,0,lm-eval,vllm,,acc,none,0.4264705882352941,0.034711579079534254,204 +SAMPLE — v2zloss_86k,sib200_mkd_Cyrl,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_mlt_Latn,0,lm-eval,vllm,,acc,none,0.5245098039215687,0.035050931943487976,204 +SAMPLE — v2zloss_86k,sib200_mlt_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_nld_Latn,0,lm-eval,vllm,,acc,none,0.45588235294117646,0.034956245220154746,204 +SAMPLE — v2zloss_86k,sib200_nld_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_nob_Latn,0,lm-eval,vllm,,acc,none,0.4117647058823529,0.03454236585380609,204 +SAMPLE — v2zloss_86k,sib200_nob_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_pol_Latn,0,lm-eval,vllm,,acc,none,0.4803921568627451,0.03506612560524869,204 +SAMPLE — v2zloss_86k,sib200_pol_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_por_Latn,0,lm-eval,vllm,,acc,none,0.47549019607843135,0.035050931943487976,204 +SAMPLE — v2zloss_86k,sib200_por_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_ron_Latn,0,lm-eval,vllm,,acc,none,0.4068627450980392,0.03447891136353379,204 +SAMPLE — v2zloss_86k,sib200_ron_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_slk_Latn,0,lm-eval,vllm,,acc,none,0.5147058823529411,0.035077938347913305,204 +SAMPLE — v2zloss_86k,sib200_slk_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_slv_Latn,0,lm-eval,vllm,,acc,none,0.5,0.03509312031717982,204 +SAMPLE — v2zloss_86k,sib200_slv_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_spa_Latn,0,lm-eval,vllm,,acc,none,0.4166666666666667,0.03460228327239165,204 +SAMPLE — v2zloss_86k,sib200_spa_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_srp_Cyrl,0,lm-eval,vllm,,acc,none,0.4411764705882353,0.034849415144292274,204 +SAMPLE — v2zloss_86k,sib200_srp_Cyrl,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_swe_Latn,0,lm-eval,vllm,,acc,none,0.46078431372549017,0.03498501649369533,204 +SAMPLE — v2zloss_86k,sib200_swe_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_tur_Latn,0,lm-eval,vllm,,acc,none,0.5049019607843137,0.03509143375606784,204 +SAMPLE — v2zloss_86k,sib200_tur_Latn,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,sib200_ukr_Cyrl,0,lm-eval,vllm,,acc,none,0.47058823529411764,0.03503235296367989,204 +SAMPLE — v2zloss_86k,sib200_ukr_Cyrl,0,lm-eval,vllm,,acc_norm,none,0.25,0.03039153369274154,204 +SAMPLE — v2zloss_86k,social_iqa,0,lm-eval,vllm,,acc,none,0.49641760491299897,0.011313780151825046,1954 +SAMPLE — v2zloss_86k,squadv2,10,lm-eval,vllm,,HasAns_exact,none,74.07219973009447,,11873 +SAMPLE — v2zloss_86k,squadv2,10,lm-eval,vllm,,HasAns_f1,none,80.32304587007627,,11873 +SAMPLE — v2zloss_86k,squadv2,10,lm-eval,vllm,,NoAns_exact,none,0.01682085786375105,,11873 +SAMPLE — v2zloss_86k,squadv2,10,lm-eval,vllm,,NoAns_f1,none,0.01682085786375105,,11873 +SAMPLE — v2zloss_86k,squadv2,10,lm-eval,vllm,,best_exact,none,59.37842162890592,,11873 +SAMPLE — v2zloss_86k,squadv2,10,lm-eval,vllm,,best_f1,none,61.18500551579533,,11873 +SAMPLE — v2zloss_86k,squadv2,10,lm-eval,vllm,,exact,none,36.99149330413543,,11873 +SAMPLE — v2zloss_86k,squadv2,10,lm-eval,vllm,,f1,none,40.11244133056617,,11873 +SAMPLE — v2zloss_86k,winogrande,0,lm-eval,vllm,,acc,none,0.7016574585635359,0.012858885010030283,1267 +SAMPLE — v2zloss_86k,wsc273,0,lm-eval,vllm,,acc,none,0.8058608058608059,0.02398292648198635,273 +SAMPLE — v2zloss_86k,xcopa:et,0,lighteval,hf/accelerate,,acc,none,0.604,0.021893529941665817, +SAMPLE — v2zloss_86k,xcopa:it,0,lighteval,hf/accelerate,,acc,none,0.624,0.021683827539286132, +SAMPLE — v2zloss_86k,xcopa:tr,0,lighteval,hf/accelerate,,acc,none,0.618,0.021750820591250844, +SAMPLE — v2zloss_86k,xcsqa_deu_Latn,0,lm-eval,vllm,,acc,none,0.453,0.015749255189977683,1000 +SAMPLE — v2zloss_86k,xcsqa_deu_Latn,0,lm-eval,vllm,,acc_norm,none,0.44,0.015704987954361718,1000 +SAMPLE — v2zloss_86k,xcsqa_eng_Latn,0,lm-eval,vllm,,acc,none,0.626,0.015308767369006505,1000 +SAMPLE — v2zloss_86k,xcsqa_eng_Latn,0,lm-eval,vllm,,acc_norm,none,0.562,0.01569721001969466,1000 +SAMPLE — v2zloss_86k,xcsqa_fra_Latn,0,lm-eval,vllm,,acc,none,0.456,0.01575792855397918,1000 +SAMPLE — v2zloss_86k,xcsqa_fra_Latn,0,lm-eval,vllm,,acc_norm,none,0.394,0.015459721957493382,1000 +SAMPLE — v2zloss_86k,xcsqa_ita_Latn,0,lm-eval,vllm,,acc,none,0.428,0.015654426245029267,1000 +SAMPLE — v2zloss_86k,xcsqa_ita_Latn,0,lm-eval,vllm,,acc_norm,none,0.388,0.015417317979911216,1000 +SAMPLE — v2zloss_86k,xcsqa_nld_Latn,0,lm-eval,vllm,,acc,none,0.408,0.015549205052920803,1000 +SAMPLE — v2zloss_86k,xcsqa_nld_Latn,0,lm-eval,vllm,,acc_norm,none,0.402,0.015512467135714959,1000 +SAMPLE — v2zloss_86k,xcsqa_pol_Latn,0,lm-eval,vllm,,acc,none,0.357,0.01515852172148659,1000 +SAMPLE — v2zloss_86k,xcsqa_pol_Latn,0,lm-eval,vllm,,acc_norm,none,0.333,0.014910846164230029,1000 +SAMPLE — v2zloss_86k,xcsqa_por_Latn,0,lm-eval,vllm,,acc,none,0.443,0.015716169953204184,1000 +SAMPLE — v2zloss_86k,xcsqa_por_Latn,0,lm-eval,vllm,,acc_norm,none,0.412,0.015572363292015074,1000 +SAMPLE — v2zloss_86k,xcsqa_spa_Latn,0,lm-eval,vllm,,acc,none,0.425,0.01564032031704017,1000 +SAMPLE — v2zloss_86k,xcsqa_spa_Latn,0,lm-eval,vllm,,acc_norm,none,0.364,0.015222868840522005,1000 diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..5b2de71 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,17 @@ +[build-system] +requires = ["setuptools>=61"] +build-backend = "setuptools.build_meta" + +[project] +name = "oellm-quickdash" +version = "0.1.0" +description = "Configurable model evaluation analysis for Python and the Quickdash dashboard" +requires-python = ">=3.8" +dependencies = ["PyYAML>=6,<7"] +license = {text = "Apache-2.0"} + +[project.scripts] +quickdash = "quickdash.cli:main" + +[tool.setuptools] +packages = ["quickdash"] diff --git a/quickdash/__init__.py b/quickdash/__init__.py new file mode 100644 index 0000000..14660c0 --- /dev/null +++ b/quickdash/__init__.py @@ -0,0 +1,21 @@ +"""Quickdash's native, configuration-driven evaluation analysis API.""" + +from .analysis import ( + analyze, + compare, + read_results, + QuickdashWarning, + DiagnosticError, + Report, +) +from .config import load_config + +__all__ = [ + "analyze", + "compare", + "read_results", + "load_config", + "QuickdashWarning", + "DiagnosticError", + "Report", +] diff --git a/quickdash/__main__.py b/quickdash/__main__.py new file mode 100644 index 0000000..eb53e2f --- /dev/null +++ b/quickdash/__main__.py @@ -0,0 +1,3 @@ +from .cli import main + +raise SystemExit(main()) diff --git a/quickdash/analysis.py b/quickdash/analysis.py new file mode 100644 index 0000000..78a1682 --- /dev/null +++ b/quickdash/analysis.py @@ -0,0 +1,848 @@ +"""Native evaluation allocation, coverage and serializable analysis reports.""" + +import json +import re +import warnings +from .config import ( + classify, + in_suite, + match_task, + resolve_config, + scope_rows, + task_language, +) +from .io import load_csv + +IDENTITY = ("task", "metric", "filter", "n_shot", "harness", "backend") + + +def encoded(value): + return json.dumps(value, ensure_ascii=False, separators=(",", ":")) + + +def key(row): + return tuple(row[k] for k in IDENTITY) + + +def measurement_id(row): + return encoded([row["checkpoint"], *key(row)]) + + +class Report(dict): + """JSON-serializable mapping with attribute access to top-level fields.""" + + def __getattr__(self, name): + try: + return self[name] + except KeyError as error: + raise AttributeError(name) from error + + +class QuickdashWarning(UserWarning): + """A recoverable diagnostic; the original structured record is available.""" + + def __init__(self, diagnostic): + self.diagnostic = diagnostic + super().__init__( + f"{diagnostic['code']} / {diagnostic['model']} / {diagnostic['name']}: {diagnostic['detail']}" + ) + + +class DiagnosticError(ValueError): + """Strict mode rejected the result. All diagnostics remain inspectable.""" + + def __init__(self, diagnostics): + self.diagnostics = diagnostics + super().__init__(f"Analysis produced {len(diagnostics)} warning(s)") + + +def finish(report, policy): + if policy not in ("warn", "collect", "error"): + raise ValueError("diagnostics must be warn, collect, or error") + if policy == "error" and report["diagnostics"]: + raise DiagnosticError(report["diagnostics"]) + if policy == "warn": + for diagnostic in report["diagnostics"]: + warnings.warn(QuickdashWarning(diagnostic), stacklevel=3) + return Report(report) + + +def read_results(paths): + """Read CSV paths, preserving source strings and rejecting duplicate models across files.""" + if isinstance(paths, (str, bytes)) or hasattr(paths, "read_text"): + paths = [paths] + rows, owners = [], {} + for path in paths: + rr = load_csv(path) + if any("checkpoint" not in r for r in rr): + raise ValueError("Missing CSV column: checkpoint") + for model in {r["checkpoint"] for r in rr}: + if model in owners: + raise ValueError("Duplicate model name across CSV files: " + model) + owners[model] = path + rows.extend(rr) + return rows + + +def component_coverage(rows, config): + metadata = {t: g for g in config["languages"] for t in g["tasks"]} + accepted, groups = set(), [] + for e in config["evals"]: + rr = [r for r in rows if r["eval"] == e["name"]] + if "aggregation" not in e: + accepted.update(map(measurement_id, rr)) + continue + buckets = {} + for r in rr: + m = metadata.get(r["task"], {}) + language = ( + (m["source_language"] + " → " + m["target_language"]) + if m.get("scope") == "translation" + else m.get("language") + ) + identity = ( + r["checkpoint"], + language or "Unknown", + m.get("scope"), + *[r[k] for k in IDENTITY if k != "task"], + ) + buckets.setdefault(identity, {"language": language, "rows": []})[ + "rows" + ].append(r) + for group in buckets.values(): + parts = {c["name"]: [] for c in e["aggregation"]["components"]} + valid = bool(group["language"]) + for r in group["rows"]: + matches = [ + c + for c in e["aggregation"]["components"] + if match_task(c["match"], r["task"]) is not None + ] + if len(matches) != 1: + valid = False + else: + parts[matches[0]["name"]].append(r) + if not valid or any(len(rr) != 1 for rr in parts.values()): + continue + total = sum(c["relative_weight"] for c in e["aggregation"]["components"]) + groups.append( + { + "eval": e["name"], + "rows": group["rows"], + "parts": [ + (parts[c["name"]][0], c["relative_weight"] / total) + for c in e["aggregation"]["components"] + ], + } + ) + accepted.update(map(measurement_id, group["rows"])) + return dict( + rows=[r for r in rows if measurement_id(r) in accepted], + excluded=[r for r in rows if measurement_id(r) not in accepted], + groups=groups, + ) + + +def distribution(rows, config): + if not rows: + return {}, None + e = next(e for e in config["evals"] if e["name"] == rows[0]["eval"]) + if "aggregation" in e: + coverage = component_coverage(rows, {**config, "evals": [e]}) + coefficients = { + measurement_id(r): share / len(coverage["groups"]) + for g in coverage["groups"] + for r, share in g["parts"] + } + rows = coverage["rows"] + else: + coefficients = {measurement_id(r): 1 / len(rows) for r in rows} + return coefficients, sum( + r["score_100"] * coefficients[measurement_id(r)] for r in rows + ) if rows else None + + +def side(row, metadata): + m = metadata.get(row["task"], {}) + language = ( + m.get("target_language") + if m.get("scope") == "translation" + else m.get("language") + ) + return "english" if language in (None, "", "mul", "eng_Latn") else "other" + + +def totals(rows, config): + """Allocate every effective score coefficient before computing contributions.""" + rows = component_coverage(rows, config)["rows"] + weights = config["weights"] + mode = config.get("aggregate", "standard") + metadata = {t: g for g in config["languages"] for t in g["tasks"]} + coefficients = {measurement_id(r): 0 for r in rows} + evals = [] + for e in config["evals"]: + rr = [r for r in rows if r["eval"] == e["name"]] + _, score = distribution(rr, config) + evals.append( + dict( + name=e["name"], + category=e["category"], + metric=e["metric"], + score=score, + weight=0, + contribution=None if rr else 0, + aggregateScore=None, + excluded=not rr, + englishShare=0, + effectiveEnglishShare=None, + englishScore=None, + otherScore=None, + issue="", + count=len(rr), + ) + ) + available = sum( + w + for name, w in weights.items() + if any(e["category"] == name and e["count"] for e in evals) + ) + categories = [] + for name, w in weights.items(): + ee = [e for e in evals if e["category"] == name and e["count"]] + share = ( + config.get("english_weights", {}).get(name, 0) if mode != "standard" else 0 + ) + cat = dict( + name=name, + weight=w / available if ee and available else 0, + excluded=not ee, + excludedEvals=[ + e["name"] for e in evals if e["category"] == name and not e["count"] + ], + score=None, + englishShare=share, + effectiveEnglishShare=None, + englishScore=None, + otherScore=None, + issue="", + ) + categories.append(cat) + if not ee: + continue + + def allocate(rr, factor): + cc, _ = distribution(rr, config) + for rid, coefficient in cc.items(): + coefficients[rid] = factor * coefficient + + if not share: + cat["score"] = sum(e["score"] for e in ee) / len(ee) + for e in ee: + allocate( + [r for r in rows if r["eval"] == e["name"]], cat["weight"] / len(ee) + ) + elif mode == "english_eval": + for e in ee: + rr = [r for r in rows if r["eval"] == e["name"]] + groups = [ + [r for r in rr if side(r, metadata) == s] + for s in ("english", "other") + ] + e["englishShare"] = share + e["englishScore"], e["otherScore"] = [ + distribution(g, config)[1] for g in groups + ] + effective = ( + 0 + if e["englishScore"] is None + else 1 + if e["otherScore"] is None + else share + ) + e["effectiveEnglishShare"] = effective + e["aggregateScore"] = effective * (e["englishScore"] or 0) + ( + 1 - effective + ) * (e["otherScore"] or 0) + for group, part in zip(groups, (effective, 1 - effective)): + allocate(group, cat["weight"] / len(ee) * part) + cat["score"] = sum(e["aggregateScore"] for e in ee) / len(ee) + else: + groups = [ + [ + [ + r + for r in rows + if r["eval"] == e["name"] and side(r, metadata) == s + ] + for e in ee + ] + for s in ("english", "other") + ] + groups = [[rr for rr in g if rr] for g in groups] + cat["englishScore"], cat["otherScore"] = [ + sum(distribution(rr, config)[1] for rr in g) / len(g) if g else None + for g in groups + ] + effective = ( + 0 + if cat["englishScore"] is None + else 1 + if cat["otherScore"] is None + else share + ) + cat["effectiveEnglishShare"] = effective + cat["score"] = effective * (cat["englishScore"] or 0) + (1 - effective) * ( + cat["otherScore"] or 0 + ) + for group, part in zip(groups, (effective, 1 - effective)): + for rr in group: + allocate(rr, cat["weight"] * part / len(group)) + for e in ee: + rr = [r for r in rows if r["eval"] == e["name"]] + e["weight"] = sum(coefficients[measurement_id(r)] for r in rr) + e["contribution"] = sum( + r["score_100"] * coefficients[measurement_id(r)] for r in rr + ) + e["aggregateScore"] = ( + e["contribution"] / e["weight"] if e["weight"] else None + ) + score = ( + sum(c["score"] * c["weight"] for c in categories if not c["excluded"]) + if available + else None + ) + return dict( + score=score, evals=evals, categories=categories, rowWeights=coefficients + ) + + +def comparison_coverage(a, b, config): + left = component_coverage(a, config)["rows"] + right = component_coverage(b, config)["rows"] + shared = set(map(key, left)) & set(map(key, right)) + aa = component_coverage([r for r in left if key(r) in shared], config)["rows"] + bb = component_coverage([r for r in right if key(r) in shared], config)["rows"] + return aa, bb + + +def sample_count(row): + value = str(row.get("n_samples", "")) + return ( + int(value) + if re.fullmatch("[1-9][0-9]*", value) and int(value) <= 2**53 - 1 + else None + ) + + +def protocol_inconsistent(rows): + tasks = {} + for r in rows: + if r["selected"]: + tasks.setdefault(r["task"], set()).add( + tuple(r[k] for k in IDENTITY if k != "task") + ) + return len({frozenset(s) for s in tasks.values()}) > 1 + + +TITLES = dict( + config_caveat="Config caveat", + no_config="No config", + not_used="Not used", + unknown_language="Unknown language", + invalid_sample_count="Invalid sample count", + inconsistent_scoring_settings="Inconsistent scoring settings", + missing_scoring_field="Missing scoring field", + missing_scoring_setting="Missing scoring setting", + no_selected_score="No selected score", + missing_suite_data="Missing suite data", + incomplete_components="Incomplete components", + comparison_coverage="Comparison coverage", + sample_count_mismatch="Sample-count mismatch", + no_category_weight="No category weight", +) + + +def diagnostic(code, model, eval_name, rows, detail, effect="included", tasks=None): + names = sorted(set(tasks if tasks is not None else [r["task"] for r in rows])) + return dict( + code=code, + type=TITLES[code], + model=model, + eval=eval_name, + name=eval_name or (names[0] if names else model), + tasks=names, + measurement_ids=sorted(map(measurement_id, rows)), + effect=effect, + detail=detail, + variants=[dict(settings=detail, tasks=names)] if names else [], + ) + + +def report_diagnostics(audits, config, included, comparison=False): + catalogue, suite, profile = config["catalogue"], config["suite"], config["profile"] + scheme = resolve_config(catalogue, suite, profile) + out = [] + used = {r["eval"] for rr in included.values() for r in rr} + for e in catalogue["evals"]: + if e.get("warning") and e["name"] in used: + out.append( + diagnostic( + "config_caveat", "Selected comparison", e["name"], [], e["warning"] + ) + ) + for model, rows in audits.items(): + scope = scope_rows(rows, suite) + accepted = {measurement_id(r) for r in included.get(model, [])} + for task in sorted({r["task"] for r in rows if not r["eval"]}): + out.append( + diagnostic( + "no_config", + model, + None, + [r for r in rows if r["task"] == task], + "No eval configuration; excluded from scoring.", + "excluded", + ) + ) + for e in catalogue["evals"]: + all_rows = [r for r in rows if r["eval"] == e["name"]] + outside = [ + r + for r in all_rows + if ("select" not in e or match_task(e["select"], r["task"]) is not None) + and not in_suite(r, suite) + ] + matching = [r for r in all_rows if in_suite(r, suite)] + selected = [r for r in matching if r["selected"]] + + def add(code, rr, detail, effect="included"): + out.append(diagnostic(code, model, e["name"], rr, detail, effect)) + + if outside: + add( + "not_used", + outside, + "Eval data is not selected by " + suite["name"] + ". Excluded.", + "excluded", + ) + if not matching: + continue + unknown = [ + r + for r in selected + if task_language(r["task"], catalogue)["status"] == "unknown" + ] + if unknown: + add( + "unknown_language", + unknown, + "Explicit language assignment missing; component groups excluded." + if "aggregation" in e + else "Unknown language; English fallback applies for weighting only.", + "excluded" if "aggregation" in e else "included", + ) + bad = [ + r + for r in selected + if r.get("n_samples") not in (None, "") and sample_count(r) is None + ] + if bad: + add( + "invalid_sample_count", + bad, + "Invalid sample count; scores are not weighted by sample count.", + ) + if protocol_inconsistent(matching): + add( + "inconsistent_scoring_settings", + selected, + "Selected variants use inconsistent scoring settings; complete protocols remain included.", + ) + for task in sorted( + { + r["task"] + for r in matching + if "select" not in e + or match_task(e["select"], r["task"]) is not None + } + ): + rr = [r for r in matching if r["task"] == task] + if any(r["selected"] for r in rr): + continue + add( + "missing_scoring_setting" + if any(r["metric"] == e["metric"] for r in rr) + else "missing_scoring_field", + rr, + "Excluded: expected " + + e["metric"] + + " / " + + (e["filter"] or "(empty)") + + ".", + "excluded", + ) + if not selected: + add( + "no_selected_score", + matching, + "No score matches the configured metric, filter, shots and selection. Excluded.", + "excluded", + ) + for name in dict.fromkeys(r["eval"] for r in scope["missing"]): + out.append( + diagnostic( + "missing_suite_data", + model, + name, + [], + "Required results missing; remaining weights are redistributed.", + "excluded", + [ + r["task"] + for r in scope["missing"] + if r["eval"] == name and "task" in r + ], + ) + ) + component = component_coverage(scope["rows"], scheme) + for name in dict.fromkeys(r["eval"] for r in component["excluded"]): + out.append( + diagnostic( + "incomplete_components", + model, + name, + [r for r in component["excluded"] if r["eval"] == name], + "Incomplete component group; entire language/protocol group excluded.", + "excluded", + ) + ) + unmatched = [r for r in component["rows"] if measurement_id(r) not in accepted] + if comparison: + for name in dict.fromkeys(r["eval"] for r in unmatched): + out.append( + diagnostic( + "comparison_coverage", + model, + name, + [r for r in unmatched if r["eval"] == name], + "Unmatched results excluded from both scores; weights use shared data only.", + "excluded", + ) + ) + for category in dict.fromkeys( + r["category"] for rr in included.values() for r in rr + ): + if category not in profile["weights"]: + affected = [ + r for rr in included.values() for r in rr if r["category"] == category + ] + item = diagnostic( + "no_category_weight", + "Selected comparison", + None, + affected, + category + " has no category weight and contributes zero.", + "zero_weight", + ) + item.update(name=category, category=category) + out.append(item) + if comparison: + entries = list(included.values()) + a = entries[0] if entries else [] + b = entries[1] if len(entries) > 1 else a + bm = {key(r): r for r in b} + for e in catalogue["evals"]: + rr = [ + r + for r in a + if r["eval"] == e["name"] + and key(r) in bm + and sample_count(r) is not None + and sample_count(bm[key(r)]) is not None + and sample_count(r) != sample_count(bm[key(r)]) + ] + if rr: + out.append( + diagnostic( + "sample_count_mismatch", + "Selected comparison", + e["name"], + rr + [bm[key(r)] for r in rr], + "Matched results have different sample counts. Scores remain included.", + ) + ) + return out + + +def allocation_tree(rows, config, allocation, model): + metadata = {t: g for g in config["languages"] for t in g["tasks"]} + + def grouped(rr, fn): + groups = {} + for r in rr: + groups.setdefault(fn(r), []).append(r) + return sorted(groups.items()) + + def language(r): + m = metadata.get(r["task"], {}) + return ( + m["source_language"] + " → " + m["target_language"] + if m.get("scope") == "translation" + else m.get("language", "Unknown") + ) + + def node(kind, label, rr, children=None, relative=None, measurement=None): + weight = sum(allocation["rowWeights"].get(measurement_id(r), 0) for r in rr) + contribution = sum( + r["score_100"] * allocation["rowWeights"].get(measurement_id(r), 0) + for r in rr + ) + return dict( + kind=kind, + label=label, + score=contribution / weight + if weight + else rr[0]["score_100"] + if measurement + else None, + weight=0, + effective_weight=weight, + contribution=contribution, + relative_weight=relative, + measurement_id=measurement, + children=children or [], + ) + + def leaves(rr, e): + result = [] + for r in sorted(rr, key=measurement_id): + c = next( + ( + c + for c in e.get("aggregation", {}).get("components", []) + if match_task(c["match"], r["task"]) is not None + ), + None, + ) + result.append( + node( + "component" if c else "measurement", + c["name"] if c else r["task"], + [r], + relative=c["relative_weight"] if c else None, + measurement=measurement_id(r), + ) + ) + return result + + def langs(rr, e): + return [ + node( + "language", + label, + ll, + [ + node("protocol", p, pp, leaves(pp, e)) + for p, pp in grouped( + ll, lambda r: encoded([r[k] for k in IDENTITY if k != "task"]) + ) + ] + if "aggregation" in e + else leaves(ll, e), + ) + for label, ll in grouped(rr, language) + ] + + def balance(rr, children): + return [ + node("language_group", s, ss, children(ss)) + for s, ss in grouped(rr, lambda r: side(r, metadata)) + ] + + def eval_nodes(rr, split): + result = [] + for e in config["evals"]: + ee = [r for r in rr if r["eval"] == e["name"]] + if ee: + result.append( + node( + "eval", + e["name"], + ee, + balance(ee, lambda ss: langs(ss, e)) if split else langs(ee, e), + ) + ) + return result + + cats = [] + for category in sorted(config["weights"]): + rr = [r for r in rows if r["category"] == category] + if not rr: + continue + split = ( + config["aggregate"] != "standard" + and config.get("english_weights", {}).get(category, 0) != 0 + ) + cats.append( + node( + "category", + category, + rr, + balance(rr, lambda ss: eval_nodes(ss, False)) + if split and config["aggregate"] == "english_category" + else eval_nodes(rr, split), + ) + ) + root = node("model", model, rows, cats) + root["score"] = allocation["score"] + root["weight"] = 0 if allocation["score"] is None else 1 + + def assign(n, path): + n["id"] = encoded(path) + for child in n["children"]: + child["weight"] = ( + child["effective_weight"] / n["effective_weight"] + if n["effective_weight"] + else 0 + ) + assign( + child, + path + [[child["kind"], child["measurement_id"] or child["label"]]], + ) + + assign(root, [["model", model]]) + return root + + +def model_report(model, audit, rows, config, suite): + allocation = totals(rows, config) + ids = {measurement_id(r) for r in rows} + measurements = [] + for r in audit: + rid = measurement_id(r) + included = rid in ids + weight = allocation["rowWeights"].get(rid, 0) + measurements.append( + { + **r, + "id": rid, + "language": task_language(r["task"], config), + "included": included, + "exclusion": None + if included + else "interpretation" + if not r["selected"] + else "eval_set" + if not in_suite(r, suite) + else "coverage", + "effective_weight": weight, + "contribution": r["score_100"] * weight if included else 0, + } + ) + return dict( + model=model, + score=allocation["score"], + tree=allocation_tree(rows, config, allocation, model), + measurements=measurements, + ) + + +def prepare(rows, config): + scheme = resolve_config(config["catalogue"], config["suite"], config["profile"]) + audit = classify(rows, config["catalogue"]) + return scheme, { + m: [r for r in audit if r["checkpoint"] == m] + for m in sorted({r["checkpoint"] for r in audit}) + } + + +def analyze(results, config, *, diagnostics="warn"): + """Score each model on its own available data; emit recoverable warnings by default.""" + scheme, audits = prepare(results, config) + included = { + m: component_coverage(scope_rows(rr, config["suite"])["rows"], scheme)["rows"] + for m, rr in audits.items() + } + coverage = [] + for model, rr in audits.items(): + scope = scope_rows(rr, config["suite"]) + coverage.append( + dict( + model=model, + missing=scope["missing"], + complete=not scope["missing"] + and len(scope["rows"]) == len(included[model]), + ) + ) + return finish( + dict( + models=[ + model_report(m, rr, included[m], scheme, config["suite"]) + for m, rr in audits.items() + ], + diagnostics=report_diagnostics(audits, config, included), + coverage=coverage, + ), + diagnostics, + ) + + +def compare(results, config, *, a, b, diagnostics="warn"): + """Score both models on the same valid measurements, with symmetric exclusions.""" + scheme, all_audits = prepare(results, config) + if a not in all_audits or b not in all_audits: + raise ValueError("Unknown comparison model") + audits = {a: all_audits[a], b: all_audits[b]} + suite = config["suite"] + sa = scope_rows(audits[a], suite) + sb = scope_rows(audits[b], suite) + aa, bb = comparison_coverage(sa["rows"], sb["rows"], scheme) + effective = scope_rows(aa, suite) + left = model_report(a, audits[a], aa, scheme, suite) + right = model_report(b, audits[b], bb, scheme, suite) + required = ( + sum(len(e["variants"]) if "variants" in e else 1 for e in suite["evals"]) + if suite["mode"] == "fixed" + else None + ) + coverage = dict( + complete=not effective["missing"] + if required is not None + else len(aa) == len(sa["rows"]) and len(bb) == len(sb["rows"]), + required=required, + sharedRequired=required - len(effective["missing"]) + if required is not None + else None, + presentA=required - len(sa["missing"]) if required is not None else None, + presentB=required - len(sb["missing"]) if required is not None else None, + extrasA=len(sa["extras"]), + extrasB=len(sb["extras"]), + ) + bm = {key(r): r for r in bb} + aw = {r["id"]: r["effective_weight"] for r in left["measurements"]} + deltas = [ + dict( + task=r["task"], + measurement_a=measurement_id(r), + measurement_b=measurement_id(bm[key(r)]), + raw_delta=r["raw_score_100"] - bm[key(r)]["raw_score_100"], + score_delta=r["score_100"] - bm[key(r)]["score_100"], + effective_weight=aw[measurement_id(r)], + contribution_delta=(r["score_100"] - bm[key(r)]["score_100"]) + * aw[measurement_id(r)], + ) + for r in aa + ] + return finish( + dict( + a=left, + b=right, + delta=left["score"] - right["score"] + if left["score"] is not None and right["score"] is not None + else None, + diagnostics=report_diagnostics(audits, config, {a: aa, b: bb}, True), + coverage=coverage, + deltas=deltas, + ), + diagnostics, + ) diff --git a/quickdash/cli.py b/quickdash/cli.py new file mode 100644 index 0000000..1a231d0 --- /dev/null +++ b/quickdash/cli.py @@ -0,0 +1,93 @@ +"""Render calculated trees or JSON; warnings always go to stderr.""" + +import argparse +import json +import sys +from . import analyze, compare, load_config, read_results + + +def render_tree(node, prefix="", last=True, root=True): + score = "unavailable" if node["score"] is None else f"{node['score']:.3f}" + branch = "" if root else "└── " if last else "├── " + yield f"{prefix}{branch}{node['label']} = {score} (weight {node['weight']:.2%}; contribution {node['contribution']:.3f})" + children = node["children"] + for i, child in enumerate(children): + yield from render_tree( + child, + prefix + ("" if root else " " if last else "│ "), + i == len(children) - 1, + False, + ) + + +def main(argv=None): + parser = argparse.ArgumentParser( + description="Calculate Quickdash scores and explain their weighted contributions. Scores use a 0–100 scale." + ) + parser.add_argument( + "csv", + nargs="+", + help="Result CSV files (model labels must be unique across files)", + ) + parser.add_argument( + "--catalogue", required=True, help="Global eval interpretation YAML" + ) + parser.add_argument("--weights", required=True, help="Weighting profile YAML") + parser.add_argument( + "--eval-set", help="Optional named eval set; otherwise compare any available" + ) + parser.add_argument( + "--compare", + nargs=2, + metavar=("A", "B"), + help="Model names to compare on shared coverage", + ) + parser.add_argument("--format", choices=["tree", "json"], default="tree") + parser.add_argument( + "--strict", + action="store_true", + help="Emit diagnostics and exit unsuccessfully without a result if any warning occurs", + ) + args = parser.parse_args(argv) + try: + config = load_config( + catalogue=args.catalogue, weights=args.weights, eval_set=args.eval_set + ) + rows = read_results(args.csv) + report = ( + compare( + rows, + config, + a=args.compare[0], + b=args.compare[1], + diagnostics="collect", + ) + if args.compare + else analyze(rows, config, diagnostics="collect") + ) + for d in report.diagnostics: + print( + f"warning [{d['code']}] {d['model']} / {d['name']}: {d['detail']}" + + (f" Tasks: {', '.join(d['tasks'])}" if d["tasks"] else ""), + file=sys.stderr, + ) + if args.strict and report.diagnostics: + return 1 + if args.format == "json": + print(json.dumps(report, ensure_ascii=False, indent=2, allow_nan=False)) + else: + for model in [report["a"], report["b"]] if args.compare else report.models: + print("\n".join(render_tree(model["tree"]))) + if args.compare: + print( + "A − B: " + + ("unavailable" if report.delta is None else f"{report.delta:.3f}") + ) + return 0 + except (ValueError, OSError) as error: + print("error: " + str(error), file=sys.stderr) + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/quickdash/config.py b/quickdash/config.py new file mode 100644 index 0000000..4c6cae4 --- /dev/null +++ b/quickdash/config.py @@ -0,0 +1,827 @@ +"""Declarative eval matching, explicit language assignments and score normalization.""" + +import math +import re +from pathlib import Path +from copy import deepcopy +from .io import parse_yaml + + +def load_catalogue(path): + return validate_catalogue(parse_yaml(Path(path).read_text(encoding="utf-8"))) + + +LANGUAGE_CODE = re.compile(r"(?:[a-z]{3}_[A-Z][a-z]{3}|mul)") +DECIMAL = re.compile(r"[+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+)(?:[eE][+-]?[0-9]+)?") + + +def object_keys(value, allowed, required=()): + if ( + not isinstance(value, dict) + or any(not isinstance(k, str) for k in value) + or set(value) - set(allowed) + or set(required) - set(value) + ): + raise ValueError( + f"Invalid config fields; allowed {sorted(allowed)}, required {sorted(required)}" + ) + + +def number(value): + return ( + isinstance(value, (int, float)) + and not isinstance(value, bool) + and math.isfinite(value) + ) + + +def validate_match(rule): + object_keys(rule, {"name", "regex"}) + if ( + len(rule) != 1 + or not isinstance(next(iter(rule.values())), str) + or not next(iter(rule.values())) + ): + raise ValueError("A match needs exactly one nonempty name or regex") + if "regex" in rule: + # Numeric captures use the shared Python/JavaScript regex subset. + if "(?P" in rule["regex"] or "(?<" in rule["regex"]: + raise ValueError("Use numeric capture groups in portable regexes") + try: + re.compile(portable_regex(rule["regex"]), re.ASCII) + except re.error as error: + raise ValueError("Invalid portable regex: " + str(error)) from error + + +def match_task(rule, task): + if "name" in rule: + return [task] if task == rule["name"] else None + match = re.fullmatch(portable_regex(rule["regex"]), task, re.ASCII) + return [match.group(0), *match.groups()] if match else None + + +def validate_config(config): + object_keys( + config, + { + "version", + "name", + "weights", + "evals", + "languages", + "notes", + "aggregate", + "english_weights", + }, + {"version", "name", "weights", "evals", "languages"}, + ) + if config["version"] != 1 or isinstance(config["version"], bool): + raise ValueError("Unsupported config version") + if not isinstance(config["name"], str) or not config["name"]: + raise ValueError("Config name is required") + validate_weights(config) + validate_rules(config) + if any(e["category"] not in config["weights"] for e in config["evals"]): + raise ValueError("Eval category has no weight") + return config + + +def validate_catalogue(config): + object_keys( + config, + {"version", "name", "evals", "languages", "notes"}, + {"version", "name", "evals", "languages"}, + ) + return validate_rules(config) + + +def validate_rules(config): + if config["version"] != 1 or isinstance(config["version"], bool): + raise ValueError("Unsupported config version") + if not isinstance(config["name"], str) or not config["name"].strip(): + raise ValueError("Config name is required") + if not isinstance(config.get("notes", []), list) or any( + not isinstance(n, str) for n in config.get("notes", []) + ): + raise ValueError("Notes must be strings") + if not isinstance(config["evals"], list) or not config["evals"]: + raise ValueError("At least one eval is required") + names = set() + categories = set() + for e in config["evals"]: + object_keys( + e, + { + "name", + "category", + "match", + "metric", + "filter", + "shots", + "select", + "score", + "normalize", + "warning", + "aggregation", + }, + {"name", "category", "match", "metric", "filter", "score"}, + ) + if not isinstance(e["name"], str) or not e["name"] or e["name"] in names: + raise ValueError("Eval names must be unique and nonempty") + names.add(e["name"]) + if not isinstance(e["category"], str) or not e["category"].strip(): + raise ValueError("Eval category must be text") + categories.add(e["category"]) + if ( + not isinstance(e["metric"], str) + or not e["metric"] + or not isinstance(e["filter"], str) + ): + raise ValueError("Metric and filter must be strings") + validate_match(e["match"]) + if "select" in e: + validate_match(e["select"]) + if "shots" in e and ( + not isinstance(e["shots"], (int, float)) + or not float(e["shots"]).is_integer() + or e["shots"] > 2**53 - 1 + or isinstance(e["shots"], bool) + or e["shots"] < 0 + ): + raise ValueError("shots must be a nonnegative integer") + object_keys(e["score"], {"scale"}, {"scale"}) + if not number(e["score"]["scale"]) or e["score"]["scale"] <= 0: + raise ValueError("Score scale must be positive") + if "warning" in e and ( + not isinstance(e["warning"], str) or not e["warning"].strip() + ): + raise ValueError("Eval warning must be nonempty text") + if "aggregation" in e: + a = e["aggregation"] + object_keys(a, {"components", "note", "sources"}, {"components"}) + if not isinstance(a["components"], list) or not a["components"]: + raise ValueError("Aggregation needs components") + names_seen = set() + total = 0 + for c in a["components"]: + object_keys( + c, + {"name", "match", "relative_weight"}, + {"name", "match", "relative_weight"}, + ) + if ( + not isinstance(c["name"], str) + or not c["name"].strip() + or c["name"] in names_seen + ): + raise ValueError("Component names must be unique and nonempty") + names_seen.add(c["name"]) + validate_match(c["match"]) + if not number(c["relative_weight"]) or c["relative_weight"] <= 0: + raise ValueError( + "Component weights must be positive finite numbers" + ) + total += c["relative_weight"] + if not math.isfinite(total): + raise ValueError("Component weight sum must be finite") + if "note" in a and not isinstance(a["note"], str): + raise ValueError("Aggregation note must be text") + if "sources" in a and ( + not isinstance(a["sources"], list) + or any( + not isinstance(u, str) or not re.match(r"^https?://", u) + for u in a["sources"] + ) + ): + raise ValueError("Aggregation sources must be HTTP(S) URLs") + if "normalize" in e: + n = e["normalize"] + object_keys( + n, {"min", "max", "clip", "basis", "note", "sources"}, {"min", "max"} + ) + if ( + not number(n["min"]) + or not number(n["max"]) + or not 0 <= n["min"] < n["max"] <= 1 + ): + raise ValueError("Normalization needs 0 <= min < max <= 1") + if "clip" in n and not isinstance(n["clip"], bool): + raise ValueError("Normalization clip must be boolean") + if "basis" in n and n["basis"] not in [ + "uniform_choice", + "uniform_integer", + "not_applicable", + "unresolved", + ]: + raise ValueError("Invalid normalization basis") + if "note" in n and not isinstance(n["note"], str): + raise ValueError("Normalization note must be text") + if "sources" in n and ( + not isinstance(n["sources"], list) + or any( + not isinstance(u, str) or not re.match(r"^https?://", u) + for u in n["sources"] + ) + ): + raise ValueError("Normalization sources must be HTTP(S) URLs") + seen = set() + if not isinstance(config["languages"], list): + raise ValueError("languages must be a list") + for group in config["languages"]: + object_keys( + group, + { + "tasks", + "scope", + "language", + "source_language", + "target_language", + "evidence", + "note", + }, + {"tasks", "scope"}, + ) + tasks = group["tasks"] + if ( + not isinstance(tasks, list) + or not tasks + or any(not isinstance(t, str) or not t for t in tasks) + ): + raise ValueError("Language groups need exact task names") + for task in tasks: + if task in seen: + raise ValueError("Duplicate language assignment: " + task) + seen.add(task) + scope = group["scope"] + if scope not in ["single", "pooled", "translation"]: + raise ValueError("Invalid language scope") + fields = ( + ["source_language", "target_language"] + if scope == "translation" + else ["language"] + ) + forbidden = ( + ["language"] + if scope == "translation" + else ["source_language", "target_language"] + ) + if any(k in group for k in forbidden): + raise ValueError( + "Use language for single/pooled; source and target for translation" + ) + for field in fields: + value = group.get(field) + if ( + not isinstance(value, str) + or not LANGUAGE_CODE.fullmatch(value) + or (value == "mul" and scope != "pooled") + ): + raise ValueError("Use canonical language codes, such as eng_Latn") + for field in ["note", "evidence"]: + if field in group and not isinstance(group[field], str): + raise ValueError(field + " must be a string") + if group.get("evidence") and not re.match(r"^https?://", group["evidence"]): + raise ValueError("Evidence links must use HTTP or HTTPS") + validate_aggregation_config(config) + return config + + +def normalize_score(value, e): + if not number(value) and ( + not isinstance(value, str) or not DECIMAL.fullmatch(value.strip()) + ): + raise ValueError("Invalid score: expected a finite decimal number") + raw = float(value) / e["score"]["scale"] + if not math.isfinite(raw) or not 0 <= raw <= 1: + raise ValueError("Score outside the configured source scale") + n = e.get("normalize", {"min": 0, "max": 1}) + adjusted = (raw - n["min"]) / (n["max"] - n["min"]) + if n.get("clip", True): + adjusted = max(0, min(1, adjusted)) + if not math.isfinite(adjusted * 100): + raise ValueError("Invalid score: normalization overflow") + return raw * 100, adjusted * 100 + + +def eval_for_task(task, config): + matches = [(e, match_task(e["match"], task)) for e in config["evals"]] + matches = [(e, m) for e, m in matches if m is not None] + if len(matches) > 1: + raise ValueError("Ambiguous eval config for task: " + task) + return matches[0] if matches else (None, None) + + +def task_language(task, config): + result = dict( + task=task, + language="", + source_language="", + target_language="", + scope="unknown", + status="unknown", + evidence="", + provenance="No explicit language assignment for this task.", + ) + for group in config["languages"]: + if task in group["tasks"]: + result.update( + { + k: group.get(k, "") + for k in [ + "language", + "source_language", + "target_language", + "scope", + "evidence", + ] + } + ) + result.update( + status="resolved", + provenance=group.get( + "note", "Explicit language assignment in eval config." + ), + ) + break + return result + + +def classify(rows, config): + (validate_config if "weights" in config else validate_catalogue)(config) + audit = [] + seen = set() + for index, source in enumerate(rows): + r = dict(source) + for field in [ + "checkpoint", + "task", + "metric", + "filter", + "n_shot", + "harness", + "backend", + "value", + ]: + if field not in r: + raise ValueError("Missing CSV column: " + field) + for field in ["checkpoint", "task", "metric", "harness", "backend"]: + if not isinstance(r[field], str) or not r[field].strip(): + raise ValueError(f"CSV row {index + 2}: {field} must be nonempty text") + if r["checkpoint"].startswith("SYNTHETIC demo — "): + raise ValueError("Checkpoint name is reserved for the synthetic demo") + if not isinstance(r["filter"], str): + raise ValueError( + f"CSV row {index + 2}: filter must be text (blank is allowed)" + ) + shots = ( + str(int(r["n_shot"])) + if isinstance(r["n_shot"], float) and r["n_shot"].is_integer() + else str(r["n_shot"]) + ) + if not re.fullmatch(r"(?:0|[1-9][0-9]*)", shots) or int(shots) > 2**53 - 1: + raise ValueError( + f"CSV row {index + 2}: n_shot must be a nonnegative integer" + ) + r["n_shot"] = shots + e, _ = eval_for_task(r["task"], config) + if e is None: + r.update( + category="", + eval="", + selected=False, + raw_score_100=None, + score_100=None, + decision="No eval config; excluded from scoring", + ) + audit.append(r) + continue + decision = "Selected for the weighted score" + if "select" in e and match_task(e["select"], r["task"]) is None: + decision = ( + "Excluded summary level or alternate protocol; see eval selection rule" + ) + elif r["metric"] != e["metric"]: + decision = "Alternate metric; using " + e["metric"] + elif r["filter"] != e["filter"]: + decision = "Alternate extraction filter; using " + ( + e["filter"] or "(empty)" + ) + elif "shots" in e and str(r["n_shot"]) != str(int(e["shots"])): + decision = ( + "Alternate shot setting; using " + str(int(e["shots"])) + " shots" + ) + selected = decision == "Selected for the weighted score" + raw = adjusted = None + if selected: + if ( + "aggregation" in e + and sum( + match_task(c["match"], r["task"]) is not None + for c in e["aggregation"]["components"] + ) + != 1 + ): + raise ValueError( + "Incompatible aggregation config for " + + e["name"] + + ": " + + r["task"] + + " must match exactly one component" + ) + try: + raw, adjusted = normalize_score(r["value"], e) + except (ValueError, OverflowError) as error: + raise ValueError( + f"CSV row {index + 2} · {r['checkpoint']} · {r['task']} · {r['metric']}: Invalid score: {error}" + ) from error + key = tuple( + r[k] + for k in [ + "checkpoint", + "task", + "metric", + "filter", + "n_shot", + "harness", + "backend", + ] + ) + if key in seen: + raise ValueError("Duplicate selected measurement: " + str(key)) + seen.add(key) + r.update( + category=e["category"], + eval=e["name"], + selected=selected, + raw_score_100=raw, + score_100=adjusted, + decision=decision, + ) + audit.append(r) + return audit + + +def portable_regex(pattern): + """Validate the shared regex syntax and use ECMAScript character semantics.""" + whitespace = ( + r"\t\n\v\f\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff" + ) + out, inside, quantifier, i = "", False, False, 0 + while i < len(pattern): + c = pattern[i] + if c == "\\": + if i + 1 == len(pattern): + raise ValueError("Incomplete portable regex escape") + nxt = pattern[i + 1] + if nxt not in "dDwWsSnrt.^$*+?{}[]()|/\\-" or nxt == "-" and not inside: + raise ValueError("Unsupported portable regex escape") + if nxt in "sS": + if inside: + raise ValueError( + "Use an explicit whitespace class inside character classes" + ) + out += "[" + ("^" if nxt == "S" else "") + whitespace + "]" + else: + out += c + nxt + quantifier = False + i += 2 + continue + if c == "[": + if inside or pattern[i : i + 2] == "[]" or pattern[i : i + 3] == "[^]": + raise ValueError("Invalid portable regex character class") + inside = True + elif c == "]": + if not inside: + raise ValueError("Unmatched portable regex bracket") + inside = False + elif not inside: + if ( + c == "(" + and pattern[i + 1 : i + 2] == "?" + and pattern[i + 2 : i + 3] != ":" + ): + raise ValueError("Use portable regexes: no flags or lookarounds") + if c == "{": + match = re.match(r"\{[0-9]+(?:,[0-9]*)?\}", pattern[i:]) + if not match: + raise ValueError("Invalid portable regex quantifier") + out += match[0] + i += len(match[0]) + quantifier = True + continue + if c == "}" or c == "+" and quantifier: + raise ValueError("Invalid portable regex quantifier") + out += r"[^\r\n\u2028\u2029]" if c == "." and not inside else c + quantifier = not inside and c in "*+?" + i += 1 + return out + + +def validate_aggregation_selection(e, variants, config, unique=True): + if "aggregation" not in e: + return + metadata = {task: g for g in config["languages"] for task in g["tasks"]} + groups = {} + + def fail(detail): + raise ValueError( + "Incompatible aggregation config for " + e["name"] + ": " + detail + ) + + for v in variants: + g = metadata.get(v["task"]) + if not g: + fail(v["task"] + " needs an explicit language assignment") + language = ( + (g["source_language"] + " → " + g["target_language"]) + if g["scope"] == "translation" + else g["language"] + ) + shot = v.get("n_shot", e.get("shots", "*")) + key = (g["scope"], language, shot) + parts = groups.setdefault( + key, {c["name"]: [] for c in e["aggregation"]["components"]} + ) + matches = [ + c + for c in e["aggregation"]["components"] + if match_task(c["match"], v["task"]) is not None + ] + if len(matches) != 1: + fail(v["task"] + " must match exactly one component") + parts[matches[0]["name"]].append(v["task"]) + for (_, language, shot), parts in groups.items(): + for name, tasks in parts.items(): + if not tasks: + fail(f"{language} / shots {shot} is missing required component {name}") + if unique and len(tasks) > 1: + fail(f"{language} has multiple tasks for component {name}") + + +def validate_aggregation_config(config): + known = [task for g in config["languages"] for task in g["tasks"]] + for e in config["evals"]: + if "aggregation" not in e: + continue + + def eligible(task): + return match_task(e["match"], task) is not None and ( + "select" not in e or match_task(e["select"], task) is not None + ) + + tasks = list(dict.fromkeys(t for t in known if eligible(t))) + for c in e["aggregation"]["components"]: + if "name" in c["match"]: + task = c["match"]["name"] + if not eligible(task): + raise ValueError( + "Incompatible aggregation config for " + + e["name"] + + ": component task is excluded" + ) + if task not in tasks: + tasks.append(task) + for task in tasks: + if ( + sum( + match_task(rule["match"], task) is not None + for rule in config["evals"] + ) + != 1 + ): + raise ValueError( + "Incompatible aggregation config for " + + e["name"] + + ": ambiguous eval rules" + ) + validate_aggregation_selection(e, [{"task": t} for t in tasks], config, False) + return config + + +def validate_suite(s): + object_keys( + s, + {"version", "name", "mode", "evals", "exclude", "notes"}, + {"version", "name", "mode"}, + ) + if ( + s["version"] != 1 + or isinstance(s["version"], bool) + or not isinstance(s["name"], str) + or not s["name"].strip() + ): + raise ValueError("Suite requires version 1 and a name") + if s["mode"] not in ("available", "fixed"): + raise ValueError("Invalid suite mode") + if "notes" in s and ( + not isinstance(s["notes"], list) + or any(not isinstance(n, str) for n in s["notes"]) + ): + raise ValueError("Notes must be strings") + if "exclude" in s and ( + s["mode"] != "available" + or not isinstance(s["exclude"], list) + or any(not isinstance(n, str) or not n.strip() for n in s["exclude"]) + or len(set(s["exclude"])) != len(s["exclude"]) + ): + raise ValueError("Invalid exclusions") + if s["mode"] == "available": + if "evals" in s: + raise ValueError("Available mode does not declare required evals") + return s + if not isinstance(s.get("evals"), list) or not s["evals"]: + raise ValueError("Fixed suite needs required evals") + names = set() + for e in s["evals"]: + object_keys(e, {"name", "variants"}, {"name"}) + if ( + not isinstance(e["name"], str) + or not e["name"].strip() + or e["name"] in names + ): + raise ValueError("Suite eval names must be unique") + names.add(e["name"]) + if "variants" not in e: + continue + if not isinstance(e["variants"], list) or not e["variants"]: + raise ValueError("Required variants must be nonempty") + seen = {} + for v in e["variants"]: + object_keys(v, {"task", "n_shot"}, {"task"}) + if not isinstance(v["task"], str) or not v["task"].strip(): + raise ValueError("Required variant needs a task name") + shot = v.get("n_shot", "*") + if "n_shot" in v and ( + not number(shot) or int(shot) != shot or not 0 <= shot <= 2**53 - 1 + ): + raise ValueError("Invalid n_shot") + shots = seen.setdefault(v["task"], set()) + if shot in shots or "*" in shots or shot == "*" and shots: + raise ValueError("Duplicate or overlapping required variant") + shots.add(shot) + return s + + +def validate_weights(config): + weights = config["weights"] + if ( + not isinstance(weights, dict) + or not weights + or any(not k or not number(v) or v < 0 for k, v in weights.items()) + or abs(sum(weights.values()) - 1) > 1e-8 + ): + raise ValueError("Category weights must be nonnegative and sum to 1") + if config.get("aggregate", "standard") not in [ + "standard", + "english_eval", + "english_category", + ]: + raise ValueError( + "Aggregate must be standard, english_eval, or english_category" + ) + if "english_weights" in config: + object_keys(config["english_weights"], set(weights)) + if any( + not number(v) or v < 0 or v > 1 for v in config["english_weights"].values() + ): + raise ValueError("English weights must be between 0 and 1") + + +def validate_profile(p): + object_keys( + p, + {"version", "name", "weights", "english_weights", "aggregate", "notes"}, + {"version", "name", "weights"}, + ) + if ( + p["version"] != 1 + or isinstance(p["version"], bool) + or not isinstance(p["name"], str) + or not p["name"].strip() + ): + raise ValueError("Weight profile requires version 1 and a name") + if "notes" in p and ( + not isinstance(p["notes"], list) + or any(not isinstance(n, str) for n in p["notes"]) + ): + raise ValueError("Notes must be strings") + validate_weights(p) + return p + + +def resolve_config(catalogue, suite, profile): + validate_catalogue(catalogue) + validate_suite(suite) + validate_profile(profile) + evals = catalogue["evals"] + if suite["mode"] == "fixed": + evals = [] + for required in suite["evals"]: + e = next( + (e for e in catalogue["evals"] if e["name"] == required["name"]), None + ) + if e is None: + raise ValueError( + "Suite eval has no catalogue rule: " + required["name"] + ) + for v in required.get("variants", []): + matches = [ + r + for r in catalogue["evals"] + if match_task(r["match"], v["task"]) is not None + ] + if ( + len(matches) != 1 + or matches[0]["name"] != e["name"] + or ("select" in e and match_task(e["select"], v["task"]) is None) + or ("shots" in e and "n_shot" in v and e["shots"] != v["n_shot"]) + ): + raise ValueError( + "Required variant is not selected by its catalogue rule: " + + v["task"] + ) + if "variants" in required: + validate_aggregation_selection(e, required["variants"], catalogue) + evals.append(e) + weights = dict(profile["weights"]) + for e in evals: + weights.setdefault(e["category"], 0) + return validate_config( + dict( + version=1, + name=catalogue["name"], + evals=evals, + languages=catalogue["languages"], + weights=weights, + english_weights=profile.get("english_weights", {}), + aggregate=profile.get("aggregate", "standard"), + notes=catalogue.get("notes", []) + profile.get("notes", []), + ) + ) + + +def in_suite(row, suite): + if suite["mode"] == "available": + return row["eval"] not in suite.get("exclude", []) + e = next((e for e in suite["evals"] if e["name"] == row["eval"]), None) + return e is not None and ( + "variants" not in e + or any( + v["task"] == row["task"] + and ("n_shot" not in v or str(int(v["n_shot"])) == row["n_shot"]) + for v in e["variants"] + ) + ) + + +def scope_rows(rows, suite): + selected = [r for r in rows if r["selected"]] + included = [r for r in selected if in_suite(r, suite)] + missing = [] + if suite["mode"] == "fixed": + for e in suite["evals"]: + if "variants" in e: + for v in e["variants"]: + if not any( + r["eval"] == e["name"] + and r["task"] == v["task"] + and ("n_shot" not in v or str(int(v["n_shot"])) == r["n_shot"]) + for r in included + ): + missing.append({"eval": e["name"], **v}) + elif not any(r["eval"] == e["name"] for r in included): + missing.append({"eval": e["name"]}) + return dict( + rows=included, + extras=[r for r in selected if not in_suite(r, suite)], + missing=missing, + ) + + +def load_config(*, catalogue, weights, eval_set=None): + """Load paths or configuration mappings. No eval set means any available.""" + + def load(value): + return ( + deepcopy(value) + if isinstance(value, dict) + else parse_yaml(Path(value).read_text(encoding="utf-8")) + ) + + bundle = dict( + catalogue=load(catalogue), + profile=load(weights), + suite=load(eval_set) + if eval_set is not None + else dict(version=1, name="Any available", mode="available"), + ) + resolve_config(bundle["catalogue"], bundle["suite"], bundle["profile"]) + return bundle + + +def load_profile(path): + return validate_profile(parse_yaml(Path(path).read_text(encoding="utf-8"))) + + +def load_suite(path): + return validate_suite(parse_yaml(Path(path).read_text(encoding="utf-8"))) diff --git a/quickdash/io.py b/quickdash/io.py new file mode 100644 index 0000000..0b4b102 --- /dev/null +++ b/quickdash/io.py @@ -0,0 +1,157 @@ +"""Strict CSV and YAML 1.2 Core inputs, independent of a JavaScript runtime.""" + +import json +import math +import re +from pathlib import Path +import yaml + + +class CoreLoader(yaml.SafeLoader): + # YAML 1.2 Core does not turn yes/no into booleans or dates into objects. + yaml_implicit_resolvers = {} + yaml_constructors = { + k: v + for k, v in yaml.SafeLoader.yaml_constructors.items() + if k is None + or k.rsplit(":", 1)[-1] in {"str", "seq", "map", "null", "bool", "int", "float"} + } + + +TAG_PATTERNS = {} + + +def scalar(loader, node): + value = loader.construct_scalar(node) + if not TAG_PATTERNS[node.tag].fullmatch(value): + raise ValueError("Invalid YAML scalar: " + value) + if node.tag.endswith(":null"): + return None + if node.tag.endswith(":bool"): + return value.lower() == "true" + cleaned = value.replace("_", "") + sign = -1 if cleaned.startswith("-") else 1 + unsigned = cleaned.lstrip("+-") + if node.tag.endswith(":int"): + return sign * int( + unsigned, 0 if unsigned.startswith(("0x", "0o", "0b")) else 10 + ) + if unsigned.lower() in (".inf", ".nan"): + return sign * (math.inf if unsigned.lower() == ".inf" else math.nan) + return float(cleaned) + + +for kind, pattern, chars in [ + ("null", r"^(?:~|null|Null|NULL|)$", ["~", "n", "N", ""]), + ("bool", r"^(?:true|True|TRUE|false|False|FALSE)$", list("tTfF")), + ( + "int", + r"^(?!.*_$)[+-]?(?:0b[01_]*[01]|0o[0-7_]*[0-7]|0x[0-9a-fA-F_]*[0-9a-fA-F]|[0-9][0-9_]*)$", + list("-+0123456789"), + ), + ( + "float", + r"^(?!.*_$)(?:[-+]?[0-9][0-9_]*(?:\.[0-9_]*)?(?:[eE][-+]?[0-9]+)?|\.[0-9_]+(?:[eE][-+]?[0-9]+)?|[-+]?\.(?:inf|Inf|INF)|\.(?:nan|NaN|NAN))$", + list("-+0123456789."), + ), +]: + tag = "tag:yaml.org,2002:" + kind + TAG_PATTERNS[tag] = re.compile(pattern) + CoreLoader.add_implicit_resolver(tag, TAG_PATTERNS[tag], chars) + CoreLoader.add_constructor(tag, scalar) + + +def mapping(loader, node): + result = {} + for key_node, value_node in node.value: + if not isinstance(key_node, yaml.ScalarNode): + raise ValueError("YAML mapping keys must be strings") + parsed = loader.construct_object(key_node) + key = ( + "null" + if parsed is None + else str(parsed).lower() + if isinstance(parsed, bool) + else str(int(parsed)) + if isinstance(parsed, (int, float)) + and math.isfinite(parsed) + and int(parsed) == parsed + else str(parsed) + ) + if key in result: + raise ValueError("Duplicate YAML key: " + key) + result[key] = loader.construct_object(value_node) + return result + + +CoreLoader.add_constructor("tag:yaml.org,2002:map", mapping) + + +def parse_yaml(source): + try: + result = yaml.load(source, Loader=CoreLoader) + # Configs must be finite trees; reject recursive aliases before validation. + json.dumps(result, check_circular=True) + return result + except (yaml.YAMLError, TypeError, RecursionError) as error: + raise ValueError("Invalid YAML: " + str(error)) from error + + +def parse_csv(source): + """Parse the browser's strict CSV dialect; preserve source strings verbatim.""" + source = source[1:] if source.startswith("\ufeff") else source + records, row, value = [], [], "" + state, touched, i = "start", False, 0 + + def record(): + nonlocal row, value, state, touched + if touched or row or value: + records.append(row + [value]) + row, value, state, touched = [], "", "start", False + + while i < len(source): + c = source[i] + if state == "quoted": + if c == '"': + if i + 1 < len(source) and source[i + 1] == '"': + value += '"' + i += 1 + else: + state = "closed" + else: + value += c + elif c == ",": + row.append(value) + value, state, touched = "", "start", True + elif c in "\r\n": + record() + if c == "\r" and i + 1 < len(source) and source[i + 1] == "\n": + i += 1 + elif c == '"' and state == "start": + state, touched = "quoted", True + elif c == '"' or state == "closed": + raise ValueError("Malformed CSV quote") + else: + state, value, touched = "plain", value + c, True + i += 1 + if state == "quoted": + raise ValueError("Unclosed quoted field in CSV") + record() + if not records: + raise ValueError("Empty CSV") + header, *records = records + if any(not h.strip() for h in header) or len(set(header)) != len(header): + raise ValueError("CSV headers must be nonempty and unique") + if not records: + raise ValueError("No measurements in CSV") + for index, row in enumerate(records, 2): + if len(row) != len(header): + raise ValueError(f"CSV row {index} width mismatch") + if all(not v.strip() for v in row): + raise ValueError(f"Empty CSV row {index}") + return [dict(zip(header, row)) for row in records] + + +def load_csv(path): + with Path(path).open(encoding="utf-8", newline="") as stream: + return parse_csv(stream.read()) diff --git a/results/README.md b/results/README.md index 8a7d8ea..e921868 100644 --- a/results/README.md +++ b/results/README.md @@ -2,6 +2,8 @@ Add a CSV directly to this directory to make its models available on the [shared dashboard](https://openeurollm.github.io/quickdash/). You can use GitHub’s **Add file → Upload files** or submit a pull request. After the change reaches `main`, the Pages workflow validates the inputs and rebuilds the site. +Until this directory contains a CSV, Pages shows the [sample dataset and synthetic comparisons](../examples/README.md). Adding the first shared CSV replaces the sample; removing all shared CSVs restores it on the next successful build. + **This repository and its dashboard are public.** Commit only results you intend to share publicly, including any source paths or metadata in the CSV. For private comparisons, use the dashboard’s **Add model CSV** button instead; those files stay in your browser. The ignored `data/` directory is for local exports. Use a descriptive filename such as `method-a-100k.csv`. Each row needs these columns: diff --git a/tests/compare_baseline.cjs b/tests/compare_baseline.cjs new file mode 100644 index 0000000..fbaaf53 --- /dev/null +++ b/tests/compare_baseline.cjs @@ -0,0 +1,28 @@ +// Run against an explicitly preserved checkout; never embeds or publishes result data. +// QUICKDASH_BASELINE=/path/to/checkout node tests/compare_baseline.cjs output/analysis.json +const assert=require('node:assert/strict'),fs=require('node:fs'),path=require('node:path'); +const baseline=process.env.QUICKDASH_BASELINE; +if(!baseline)throw Error('Set QUICKDASH_BASELINE to the checkout being compared'); +const before=require(path.resolve(baseline,'app/app.js')); +const after=require('../app/analysis.js'),S=require('../app/suite_config.js'); +const data=JSON.parse(fs.readFileSync(process.argv[2]||'output/example/analysis.json','utf8')); +const normalized=t=>({score:t.score,evals:t.evals.map(e=>({name:e.name,score:e.aggregateScore,weight:e.weight,contribution:e.contribution,excluded:e.excluded})),categories:t.categories,rowWeights:[...t.rowWeights].map(([r,w])=>[before.key?before.key(r):JSON.stringify(['task','metric','filter','n_shot','harness','backend'].map(k=>r[k])),w])}); +function close(a,b){if(typeof a==='number'){assert.ok(Math.abs(a-b)<1e-9);}else if(Array.isArray(a)){assert.equal(a.length,b.length);a.forEach((v,i)=>close(v,b[i]));}else if(a&&typeof a==='object'){assert.deepEqual(Object.keys(a),Object.keys(b));for(const k of Object.keys(a))close(a[k],b[k]);}else assert.equal(a,b);} +let cases=0; +for(const {config:profile} of data.profiles)for(const {config:suite} of data.suites)for(const aggregate of ['standard','english_eval','english_category']){ + const scheme=S.resolveConfig(data.catalogue,suite,{...profile,aggregate}); + const original=before.auditRows(data.rows,data.catalogue),model=original[0]?.checkpoint; + const a=S.scopeRows(original.filter(r=>r.checkpoint===model),suite).rows; + const full=before.synthetic(a,scheme); + for(const b of [full,full.slice(1),full.filter(r=>r.eval!==full[0]?.eval),full.map((r,i)=>i? r:{...r,backend:'different'})]){ + const old=before.comparisonCoverage(a,b,scheme),now=after.comparisonCoverage(a,b,scheme); + assert.deepEqual(now.a,old.a);assert.deepEqual(now.b,old.b); + for(const side of ['a','b'])close(normalized(before.totals(old[side],scheme,scheme.weights,aggregate)),normalized(after.totals(now[side],scheme,scheme.weights,aggregate))); + for(const view of ['category','language']){ + const metadata=new Map(data.metadata.map(r=>[r.task,r])); + close(before.buildBreakdownTree(old.pairs,metadata,view,'','',scheme),after.buildBreakdownTree(now.pairs,metadata,view,'','',scheme)); + } + cases++; + } +} +console.log(`Before/after comparison passed: ${cases} combinations of profiles, sets, modes and missing coverage; identical included rows, scores, weights, contributions and descriptive trees.`); diff --git a/tests/diagnostic_fixture.cjs b/tests/diagnostic_fixture.cjs new file mode 100644 index 0000000..0b82c22 --- /dev/null +++ b/tests/diagnostic_fixture.cjs @@ -0,0 +1,12 @@ +// Turn the small audited-row fixtures into inputs for the production diagnostic pass. +const {reportDiagnostics,componentCoverage}=require('../app/analysis.js'); +const {resolveConfig,scopeRows}=require('../app/suite_config.js'); +function diagnosticsFor(audits,config,aggregate='standard',english_weights={},suite={version:1,name:'Available',mode:'available'}){ + const catalogue=Object.fromEntries(Object.entries(config).filter(([k])=>['version','name','evals','languages','notes'].includes(k))); + const categories=[...new Set(catalogue.evals.map(e=>e.category))]; + const profile={version:1,name:'Fixture weights',weights:config.weights||Object.fromEntries(categories.map(c=>[c,1/categories.length])),aggregate,english_weights}; + const bundle={catalogue,profile,suite},scheme=resolveConfig(catalogue,suite,profile); + const included=new Map([...audits].map(([m,rr])=>[m,componentCoverage(scopeRows(rr,suite).rows,scheme).rows])); + return reportDiagnostics(audits,bundle,included); +} +module.exports={diagnosticsFor}; diff --git a/tests/engine_adapter.cjs b/tests/engine_adapter.cjs new file mode 100644 index 0000000..3d93ba2 --- /dev/null +++ b/tests/engine_adapter.cjs @@ -0,0 +1,9 @@ +// One process accepts a batch so differential tests don't pay startup per case. +const E=require('../app/eval_config.js'),S=require('../app/suite_config.js'),A=require('../app/analysis.js'); +const fs=require('node:fs'); +function run(c){try{ + const config=c.yaml?{catalogue:E.parseCatalogue(c.yaml.catalogue),profile:S.parseWeightProfile(c.yaml.profile),suite:S.parseSuite(c.yaml.suite)}:c.config; + const rows=c.csv!==undefined?E.parseCSV(c.csv):c.rows; + return {value:c.operation==='compare'?A.compare(rows,config,c.a||'A',c.b||'B'):A.analyze(rows,config)}; +}catch(error){return {error:true,message:error.message};}} +process.stdout.write(JSON.stringify(JSON.parse(fs.readFileSync(0,'utf8')).map(run))); diff --git a/tests/test_analysis.py b/tests/test_analysis.py index 69e2b92..5746135 100644 --- a/tests/test_analysis.py +++ b/tests/test_analysis.py @@ -2,18 +2,19 @@ import unittest from pathlib import Path from app.build import classify, summarize -from app.config_engine import task_language, validate_config, normalize_score, match_task, shared_config, load_catalogue +from quickdash.config import task_language, validate_config, normalize_score, match_task, load_catalogue, load_profile, load_suite, resolve_config, scope_rows from copy import deepcopy import subprocess +import sys import tempfile ROOT=Path(__file__).resolve().parent.parent -CONFIG=shared_config('resolve',value=[load_catalogue(ROOT/'configs/catalogue.yaml'),shared_config('suite',ROOT/'configs/sets/any-available.yaml'),shared_config('weights',ROOT/'configs/weights/oellm.yaml')]) +CONFIG=resolve_config(load_catalogue(ROOT/'configs/catalogue.yaml'),load_suite(ROOT/'configs/sets/any-available.yaml'),load_profile(ROOT/'configs/weights/oellm.yaml')) def resolve_languages(tasks):return [task_language(t,CONFIG) for t in tasks] DATA=json.loads((ROOT/'output/analysis.json').read_text()) class AnalysisTests(unittest.TestCase): def test_english_weighting_language_fallback(self): config={'evals':[{'name':'mixed','category':'C','metric':'acc'},{'name':'english','category':'C','metric':'acc'}], 'weights':{'C':1}, 'english_weights':{'C':.5}} - rows=[dict(task=task,eval=ev,score_100=score,selected=True,checkpoint='A') for task,ev,score in [('en','mixed',80),('fr','mixed',20),('de','mixed',40),('english','english',100)]] + rows=[dict(task=task,eval=ev,score_100=score,selected=True,checkpoint='A',metric='acc',filter='none',n_shot='0',harness='fixture',backend='cpu') for task,ev,score in [('en','mixed',80),('fr','mixed',20),('de','mixed',40),('english','english',100)]] for code in [None,'mul','hbs_Latn']: config['languages']=[{'tasks':['en','english'],'scope':'single','language':'eng_Latn'},{'tasks':['de'],'scope':'single','language':'deu_Latn'}] if code:config['languages'].append({'tasks':['fr'],'scope':'pooled','language':code}) @@ -25,7 +26,7 @@ def test_english_weighting_language_fallback(self): def test_complete_source_coverage(self): self.assertEqual(len(DATA['rows']),2124) self.assertTrue(all(r['eval'] for r in DATA['rows'])) - self.assertEqual(len([e for e in DATA['models'][0]['evals'] if not e['excluded']]),45) + self.assertEqual({e['name'] for e in DATA['models'][0]['evals'] if not e['excluded']},{r['eval'] for r in scope_rows(DATA['rows'],DATA['suite'])['rows']}) def test_selected_measurements_unique(self): rr=[r for r in DATA['rows'] if r['selected']] keys=[tuple(r[k] for k in ['task','metric','filter','n_shot','harness','backend']) for r in rr] @@ -105,7 +106,7 @@ def test_alternate_config_build_and_javascript_parity(self): with tempfile.TemporaryDirectory() as tmp: from tests.test_data import inputs path=Path(tmp);kw=inputs(path,c) - subprocess.run(['python3','-m','app.build',str(ROOT/'data/v2zloss_86k.flag-evals-436.tasks.csv'),'--catalogue',str(kw['catalogue_path']),'--weights',str(kw['weights_path']),'--eval-set',str(kw['suite_path']),'--output',str(path/'result')],check=True,capture_output=True) + subprocess.run([sys.executable,'-m','app.build',str(ROOT/'data/v2zloss_86k.flag-evals-436.tasks.csv'),'--catalogue',str(kw['catalogue_path']),'--weights',str(kw['weights_path']),'--eval-set',str(kw['suite_path']),'--output',str(path/'result')],check=True,capture_output=True) data=json.loads((path/'result/analysis.json').read_text()) self.assertNotEqual(data['models'][0]['score'],DATA['models'][0]['score']) script="const fs=require('fs'),e=require('./app/eval_config.js');const d=JSON.parse(fs.readFileSync(process.argv[1]));console.log(JSON.stringify({rows:e.auditRows(d.rows,d.scheme),metadata:d.metadata.map(m=>e.taskLanguage(m.task,d.scheme))}));" @@ -144,9 +145,15 @@ def test_chance_baselines_and_raw_score_preservation(self): c=deepcopy(CONFIG) for e in c['evals']:e['normalize']={'min':0,'max':1} raw=classify(DATA['rows'],c) - scoped=shared_config('scope',value=[raw,DATA['suite']])['rows'] + scoped=scope_rows(raw,DATA['suite'])['rows'] from statistics import mean - expected_raw=sum(weight*mean(mean(r['raw_score_100'] for r in scoped if r['eval']==e) for e in {r['eval'] for r in scoped if r['category']==category}) for category,weight in c['weights'].items()) + def raw_eval(name): + rr=[r for r in scoped if r['eval']==name] + if name!='PolyMath':return mean(r['raw_score_100'] for r in rr) + levels={'low':1,'medium':2,'high':4,'top':8} + languages={r['task'].split('_')[1] for r in rr} + return mean(sum(r['raw_score_100']*levels[r['task'].split('_')[-1]] for r in rr if r['task'].split('_')[1]==lang)/15 for lang in languages) + expected_raw=sum(weight*mean(raw_eval(e) for e in {r['eval'] for r in scoped if r['category']==category}) for category,weight in c['weights'].items()) self.assertAlmostEqual(summarize(scoped,c)[0]['score'],expected_raw) self.assertLess(DATA['models'][0]['score'],expected_raw) self.assertEqual([r['raw_score_100'] for r in raw],[r['raw_score_100'] for r in DATA['rows']]) diff --git a/tests/test_app.cjs b/tests/test_app.cjs index 9e4dbe1..15bc669 100644 --- a/tests/test_app.cjs +++ b/tests/test_app.cjs @@ -1,9 +1,9 @@ const assert=require('node:assert/strict'),fs=require('node:fs'); -const {parseCSV,selectRows,totals,pairRows,synthetic}=require('../app/app.js'); +const {parseCSV,selectRows,totals,pairRows,synthetic}=require('../app/analysis.js'); const data=JSON.parse(fs.readFileSync(__dirname+'/../output/analysis.json')),scheme=data.scheme; const {scopeRows,inSuite}=require('../app/suite_config.js'); const selected=scopeRows(selectRows(data.rows,scheme),data.suite).rows; -assert.equal(selected.length,403); +assert.ok(selected.length); assert.ok(Math.abs(totals(selected,scheme,scheme.weights).score-data.models[0].score)<1e-10); const withoutIF=totals(selected.filter(r=>r.eval!=='IFEval'),scheme,scheme.weights); assert.ok(Number.isFinite(withoutIF.score));assert.equal(withoutIF.categories.find(c=>c.name==='Instruction following').excluded,true); @@ -16,22 +16,22 @@ const mockScheme={evals:[{name:'f1',category:'C'},{name:'f2',category:'C'},{name const mockRows=[{eval:'f1',score_100:0},{eval:'f1',score_100:100},{eval:'f2',score_100:100},{eval:'f3',score_100:0}]; assert.equal(totals(mockRows,mockScheme,{C:.8,D:.2}).score,60); // f1=50, f2=100; C=75. No row-count weighting. const fake=synthetic(selected,scheme);assert.deepEqual(fake,synthetic(selected,scheme));assert.ok(fake.every(r=>r.score_100>=0&&r.score_100<=100)); -assert.equal(pairRows(selected,fake).length,403); +assert.equal(pairRows(selected,fake).length,selected.length); assert.equal(pairRows([selected[0]],[{...fake[0],n_shot:'999'}]).length,0); assert.ok(pairRows(selected,selected).every(r=>r.delta===0)); -assert.equal(pairRows(selected,fake.slice(1)).length,402); +assert.equal(pairRows(selected,fake.slice(1)).length,selected.length-1); console.log('JS checks passed: scoring, missing data, protocol matching, CSV parsing, synthetic reproducibility, Python parity.'); // The catalogue must retain excluded tasks, summaries and alternate metrics. -const {auditRows,buildCatalogue}=require('../app/app.js'); +const {auditRows,buildCatalogue}=require('../app/analysis.js'); const audited=auditRows(data.rows,scheme); -assert.equal(audited.length,2124); -assert.equal(scopeRows(audited,data.suite).rows.length,403); -assert.equal(audited.filter(r=>r.selected).length,435); +assert.equal(audited.length,data.rows.length); +assert.equal(scopeRows(audited,data.suite).rows.length,selected.length); +assert.ok(audited.filter(r=>r.selected).length>=selected.length); const catalogue=buildCatalogue(audited,scheme); -assert.equal(catalogue.length,46); -assert.equal(catalogue.reduce((n,f)=>n+f.tasks.length,0),1556); -assert.equal(catalogue.flatMap(f=>f.tasks).reduce((n,t)=>n+t.rows.length,0),2124); +assert.equal(catalogue.length,new Set(audited.filter(r=>r.eval).map(r=>r.eval)).size); +assert.equal(catalogue.reduce((n,f)=>n+f.tasks.length,0),new Set(audited.filter(r=>r.eval).map(r=>r.task)).size); +assert.equal(catalogue.flatMap(f=>f.tasks).reduce((n,t)=>n+t.rows.length,0),audited.filter(r=>r.eval).length); const he=catalogue.find(f=>f.name==='HumanEval').tasks[0]; assert.equal(he.rows.find(r=>r.metric==='python_pass@1').selected,true); assert.equal(he.rows.find(r=>r.metric==='sh_pass@1').selected,false); @@ -50,7 +50,7 @@ assert.equal(union.find(f=>f.name==='HumanEval').tasks.length,1); assert.equal(union.find(f=>f.name==='HumanEval').tasks[0].rows.length,3); console.log('Catalogue checks passed: all tasks and metrics retained, category mapping, protocols, model union.'); -const {comparisonRows,languageRoles,matchesLanguage}=require('../app/app.js'); +const {comparisonRows,languageRoles,matchesLanguage}=require('../app/analysis.js'); const toy=[{task:'a',eval:'f1',category:'C',delta:10,score_delta:10,a:30,b:20},{task:'b',eval:'f1',category:'C',delta:30,score_delta:30,a:50,b:20},{task:'c',eval:'f2',category:'C',delta:-10,score_delta:-10,a:10,b:20},{task:'d',eval:'f3',category:'D',delta:5,score_delta:5,a:25,b:20}]; const toyWeights={C:.8,D:.2}; for(const grouping of ['category','eval','variant']){ @@ -61,7 +61,7 @@ for(const grouping of ['category','eval','variant']){ assert.equal(comparisonRows(toy.slice(0,1),toy,mockScheme,toyWeights,'variant','weighted','name')[0].weightedDelta,2); const actual=comparisonRows(pairRows(selected,fake),selected,scheme,scheme.weights,'variant','weighted','descending'); assert.ok(Math.abs(actual.reduce((s,r)=>s+r.weightedDelta,0)-(totals(selected,scheme,scheme.weights).score-totals(fake,scheme,scheme.weights).score))<1e-10); -assert.equal(actual.length,403); +assert.equal(actual.length,selected.length); const asc=comparisonRows(toy,toy,mockScheme,toyWeights,'variant','raw','ascending'); assert.deepEqual(asc.map(r=>r.rawDelta),[-10,5,10,30]); const weighted=comparisonRows(toy,toy,mockScheme,toyWeights,'variant','weighted','descending'); @@ -93,10 +93,10 @@ assert.equal(matchTask({regex:'eval_(fr|en)'},'eval_fr\n'),false); for(const mutate of [c=>c.languages.push(c.languages[0]),c=>c.languages[0].language='English',c=>c.evals[0].normalize={min:1,max:1},c=>c.weights.Code=.2,c=>c.evals[0].shots='0',c=>c.evals[0].extra='typo']){const c=structuredClone(scheme);mutate(c);assert.throws(()=>validateConfig(c));} console.log('Config and comparison checks passed: explicit languages, translation roles, normalization, raw preservation, weighted reconciliation, validation.'); -const {buildBreakdownTree,breakdownAggregate}=require('../app/app.js'); +const {buildBreakdownTree,breakdownAggregate}=require('../app/analysis.js'); const pairs=pairRows(selected,fake),ct=buildBreakdownTree(pairs,metadata,'category'),lt=buildBreakdownTree(pairs,metadata,'language'); -assert.equal(ct.length,9); -assert.equal(ct.reduce((n,c)=>n+c.count,0),403); +assert.equal(ct.length,new Set(selected.map(r=>r.category)).size); +assert.equal(ct.reduce((n,c)=>n+c.count,0),selected.length); const translate=ct.find(n=>n.label==='Translation'); assert.ok(translate.children.every(e=>e.kind==='eval')); const english=translate.children[0].children.find(n=>n.label==='eng_Latn'); @@ -113,25 +113,25 @@ for(const view of ['category','language']){ function check(nodes){for(const n of nodes){if(n.kind==='language')assert.equal(n.label,'fin_Latn');if(n.kind==='direction')assert.equal(n.label,'To fin_Latn');check(n.children);}} check(tree); } -assert.equal(breakdownAggregate([...pairs,pairs[0]]).count,403); +assert.equal(breakdownAggregate([...pairs,pairs[0]]).count,selected.length); assert.equal(breakdownAggregate([...pairs,pairs[0]]).a,breakdownAggregate(pairs).a); console.log('Hierarchy checks passed: category/eval/language and language/category/eval ordering, translation directions/pairs, filtered branches, unique aggregates.'); -const {collectWarnings}=require('../app/app.js'); -const baselineWarnings=collectWarnings(new Map([['real',audited]]),scheme,'standard',{},data.suite); +const {diagnosticsFor}=require('./diagnostic_fixture.cjs'); +const baselineWarnings=diagnosticsFor(new Map([['real',audited]]),scheme,'standard',{},data.suite); assert.deepEqual(baselineWarnings.filter(w=>w.type==='Inconsistent scoring settings').map(w=>w.name).sort(),['ARC Challenge','MGSM','PIQA']); assert.deepEqual(baselineWarnings.filter(w=>w.type==='Config caveat').map(w=>w.name).sort(),['MultiBlimp']); const unknown=auditRows([{...data.rows[0],task:'new_task_without_config'}],scheme)[0]; assert.equal(unknown.selected,false);assert.equal(unknown.eval,'');assert.equal(unknown.score_100,null); const warningRows=audited.filter(r=>r.eval!=='HumanEval').concat(unknown); -const warnings=collectWarnings(new Map([['real',warningRows]]),scheme,'standard',{},data.suite); +const warnings=diagnosticsFor(new Map([['real',warningRows]]),scheme,'standard',{},data.suite); assert.equal(warnings.length,baselineWarnings.length+1); assert.ok(warnings.some(w=>w.type==='No config'&&w.name==='new_task_without_config')); assert.ok(!warnings.some(w=>w.type==='No eval data')); const alternate=structuredClone(scheme);alternate.evals.find(e=>e.name==='HumanEval').metric='nonexistent'; -assert.ok(collectWarnings(new Map([['real',auditRows(data.rows,alternate)]]),alternate).some(w=>w.type==='No selected score')); +assert.ok(diagnosticsFor(new Map([['real',auditRows(data.rows,alternate)]]),alternate).some(w=>w.type==='No selected score')); console.log('Warning checks passed: unconfigured exclusion, absent evals, selected metric gaps.'); -const {languageCoverage}=require('../app/app.js'); +const {languageCoverage}=require('../app/analysis.js'); assert.deepEqual(languageCoverage([{task:'AIME24'},{task:'AIME25'}],metadata),{count:1,pooled:0,unknown:0}); assert.deepEqual(languageCoverage([{task:'flores200:eng_Latn-fin_Latn'},{task:'flores200:fin_Latn-eng_Latn'}],metadata),{count:2,pooled:0,unknown:0}); assert.deepEqual(languageCoverage([{task:'bigbench_language_identification_multiple_choice'},{task:'multiblimp_hbs'},{task:'not-known'}],metadata),{count:1,pooled:1,unknown:1}); @@ -145,33 +145,33 @@ for(const field of ['label','category','a','b','rawDelta','weightedDelta','langu } } const missingArc=audited.filter(r=>!(r.task==='arc_challenge_mt_cs'&&r.metric==='acc_norm')); -const arcWarnings=collectWarnings(new Map([['real',missingArc]]),scheme).filter(w=>w.type==='Missing scoring field'); +const arcWarnings=diagnosticsFor(new Map([['real',missingArc]]),scheme).filter(w=>w.type==='Missing scoring field'); assert.equal(arcWarnings.length,1); assert.equal(arcWarnings[0].type,'Missing scoring field'); -assert.equal(arcWarnings[0].name,'arc_challenge_mt_cs'); +assert.deepEqual(arcWarnings[0].tasks,['arc_challenge_mt_cs']); assert.match(arcWarnings[0].detail,/expected acc_norm/); const wrongFilter=data.rows.map(r=>r.task==='arc_challenge_mt_cs'&&r.metric==='acc_norm'?{...r,filter:'wrong'}:r); -assert.ok(collectWarnings(new Map([['real',auditRows(wrongFilter,scheme)]]),scheme).some(w=>w.type==='Missing scoring setting')); +assert.ok(diagnosticsFor(new Map([['real',auditRows(wrongFilter,scheme)]]),scheme).some(w=>w.type==='Missing scoring setting')); // MMLU child tasks are intentionally excluded, so missing summary metrics there are not warnings. -assert.ok(!collectWarnings(new Map([['real',audited]]),scheme).some(w=>w.name==='mmlu_abstract_algebra')); +assert.ok(!diagnosticsFor(new Map([['real',audited]]),scheme).some(w=>w.name==='mmlu_abstract_algebra')); console.log('Column sort and per-task missing metric/setting checks passed.'); // Grouped multilingual scores must expose differences in the selected protocol. const protocolConfig=structuredClone(scheme); for(const e of protocolConfig.evals)delete e.warning; const consistent=audited.filter(r=>inSuite(r,data.suite)).map(r=>({...r,n_shot:'0'})); -assert.deepEqual(collectWarnings(new Map([['real',consistent]]),protocolConfig),[]); +assert.deepEqual(diagnosticsFor(new Map([['real',consistent]]),protocolConfig),[]); for(const field of ['n_shot','filter','metric','harness','backend']){ const changed=consistent.map(r=>r.task==='arc_challenge_mt_cs'&&r.selected?{...r,[field]:field==='n_shot'?'5':'different'}:r); - const issues=collectWarnings(new Map([['real',changed]]),protocolConfig).filter(w=>w.type==='Inconsistent scoring settings'); + const issues=diagnosticsFor(new Map([['real',changed]]),protocolConfig).filter(w=>w.type==='Inconsistent scoring settings'); assert.equal(issues.length,1,field);assert.equal(issues[0].name,'ARC Challenge');assert.match(issues[0].detail,/arc_challenge_mt_cs/);assert.match(issues[0].detail,/ces_Latn/); } // Alternate, excluded measurements do not change the protocol used in aggregates. const excludedChange=consistent.map(r=>!r.selected?{...r,n_shot:'99'}:r); -assert.deepEqual(collectWarnings(new Map([['real',excludedChange]]),protocolConfig),[]); +assert.deepEqual(diagnosticsFor(new Map([['real',excludedChange]]),protocolConfig),[]); const repeated=consistent.concat(consistent.filter(r=>r.eval==='ARC Challenge'&&r.selected).map(r=>({...r,n_shot:'5'}))); -assert.deepEqual(collectWarnings(new Map([['real',repeated]]),protocolConfig),[],'same settings set for every task is consistent'); +assert.deepEqual(diagnosticsFor(new Map([['real',repeated]]),protocolConfig),[],'same settings set for every task is consistent'); const noteConfig=structuredClone(protocolConfig);noteConfig.evals.find(e=>e.name==='FLORES200').warning='Review language-pair calibration.'; -const noteWarnings=collectWarnings(new Map([['first',consistent],['second',consistent]]),noteConfig).filter(w=>w.type==='Config caveat'); +const noteWarnings=diagnosticsFor(new Map([['first',consistent],['second',consistent]]),noteConfig).filter(w=>w.type==='Config caveat'); assert.equal(noteWarnings.length,1,'config notes are not duplicated for each model');assert.match(noteWarnings[0].detail,/language-pair/); assert.throws(()=>validateConfig({...noteConfig,evals:noteConfig.evals.map(e=>({...e,warning:17}))}),/warning/i); console.log('Protocol consistency and editable normalization warning checks passed.'); diff --git a/tests/test_browser.mjs b/tests/test_browser.mjs index 4e830a3..86a94cd 100644 --- a/tests/test_browser.mjs +++ b/tests/test_browser.mjs @@ -2,7 +2,11 @@ import fs from 'node:fs'; import assert from 'node:assert/strict'; import {fileURLToPath} from 'node:url'; const root=fileURLToPath(new URL('../',import.meta.url)); -const initialScore=JSON.parse(fs.readFileSync(root+'output/analysis.json','utf8')).models[0].score.toFixed(2); +const payload=JSON.parse(fs.readFileSync(root+'output/analysis.json','utf8')); +const initialScore=payload.models[0].score.toFixed(2); +const configuredTasks=payload.catalogue.languages.reduce((n,g)=>n+g.tasks.length,0); +const usedEvals=payload.models[0].evals.filter(e=>!e.excluded).length; +const usedRows=payload.models[0].evals.reduce((n,e)=>n+e.count,0); const tabs=await(await fetch('http://127.0.0.1:9227/json/list')).json(); const ws=new WebSocket(tabs.find(t=>t.type==='page').webSocketDebuggerUrl); await new Promise(r=>ws.addEventListener('open',r,{once:true})); @@ -43,7 +47,7 @@ assert.equal(await evaluate("document.querySelector('#cards strong').textContent assert.equal(await evaluate("document.querySelector('#warningCount').textContent"),'5'); assert.match(await evaluate("document.querySelector('#view').textContent"),/English weighting group only · full weight/); await click('#weightEditor > summary'); -assert.equal(await evaluate("document.querySelectorAll('#weights tbody tr').length"),9); +assert.equal(await evaluate("document.querySelectorAll('#weights tbody tr').length"),Object.keys(payload.profile.weights).length); assert.ok(await evaluate("[...document.querySelectorAll('#weights tbody tr')].every(r=>r.cells.length===3&&r.querySelectorAll('input').length===2)")); await evaluate("document.querySelector('#weightEditor').scrollIntoView()");await screenshot('weight-editor-preview'); await click('#englishComponents > summary'); @@ -67,7 +71,7 @@ await screenshot('score-preview'); assert.equal(await evaluate("document.querySelector('#breakdownBy')"),null); // Collapsed hierarchies have aggregates and expand through the requested orders. await click('[data-view=categories]'); -assert.equal(await evaluate("document.querySelectorAll('.breakdown-tree > .breakdown-node').length"),9); +assert.equal(await evaluate("document.querySelectorAll('.breakdown-tree > .breakdown-node').length"),Object.keys(payload.profile.weights).length); assert.equal(await evaluate("document.querySelectorAll('.breakdown-node[open]').length"),0); const categoryPath='.breakdown-tree > [data-label="Translation"]'; await click(categoryPath+' > summary'); @@ -111,7 +115,7 @@ assert.ok(await evaluate("[...document.querySelectorAll('[data-kind=direction]') await click('#clear'); await click('[data-view=comparisons]'); assert.equal(await evaluate("document.querySelector('#compareGroup').value"),'eval'); -assert.equal(await evaluate("document.querySelectorAll('.comparison-row').length"),45); +assert.equal(await evaluate("document.querySelectorAll('.comparison-row').length"),usedEvals); assert.ok(await evaluate("!!document.querySelector('[data-chart-eval=\"Global MMLU\"]')&&!!document.querySelector('[data-chart-eval=MMLU]')")); // Clickable headers sort every displayed value and expose their active direction. for(const [field,index] of [['label',0],['category',1],['a',2],['b',3],['rawDelta',4],['weightedDelta',5],['languageCount',6]]){ @@ -134,7 +138,7 @@ assert.ok(await evaluate(`(()=>{const button=document.querySelector('[data-expan assert.ok(await evaluate("document.querySelectorAll('.comparison-detail').length>1")); assert.match(await evaluate("document.querySelector('.comparison-detail').textContent"),/Latn|Cyrl/); await click('[data-expand-eval="HellaSwag"]'); -assert.equal(await evaluate("document.querySelectorAll('.comparison-row').length"),45); +assert.equal(await evaluate("document.querySelectorAll('.comparison-row').length"),usedEvals); assert.ok(await evaluate(`Math.abs(document.querySelector('[data-expand-eval="HellaSwag"]').getBoundingClientRect().top-scrollBefore.button)<2`),'collapse moved the clicked row'); assert.equal(await evaluate(`document.querySelector('[data-chart-eval="HumanEval"] .language-count').textContent`),'1'); assert.ok(await evaluate(`Number(document.querySelector('[data-chart-eval="FLORES200"] .language-count').textContent)>30`)); @@ -144,8 +148,8 @@ assert.equal(await evaluate(`document.querySelector('[data-chart-eval="Belebele" assert.equal(await evaluate(`document.querySelector('[data-chart-eval="FLORES200"] .language-count').textContent`),'2'); await click('#clear');await change('#compareSort','descending'); await change('#compareGroup','variant'); -assert.equal(await evaluate("document.querySelectorAll('.comparison-row').length"),403); -assert.equal(await evaluate("document.querySelectorAll('.delta-track').length"),403); +assert.equal(await evaluate("document.querySelectorAll('.comparison-row').length"),usedRows); +assert.equal(await evaluate("document.querySelectorAll('.delta-track').length"),usedRows); const original=await evaluate("document.querySelector('#cards').textContent"); await change('#category','Code'); assert.equal(await evaluate("document.querySelectorAll('.comparison-row').length"),4); @@ -160,7 +164,7 @@ await change('#compareMeasure','raw');await change('#compareSort','descending'); values=await evaluate("[...document.querySelectorAll('.comparison-row')].map(r=>Number(r.cells[4].textContent))"); assert.deepEqual(values,[...values].sort((a,b)=>b-a)); await change('#compareGroup','eval'); -assert.equal(await evaluate("document.querySelectorAll('.comparison-row').length"),45); +assert.equal(await evaluate("document.querySelectorAll('.comparison-row').length"),usedEvals); await click('[data-expand-eval="HumanEval"]'); assert.equal(await evaluate("document.querySelectorAll('.comparison-detail').length"),1); await click('[data-expand-eval="HumanEval"]'); @@ -173,7 +177,7 @@ await click('#clear'); await click('[data-view=config]'); assert.equal(await evaluate("document.querySelectorAll('.catalogue-task').length"),0); -assert.match(await evaluate("document.querySelector('#filterStatus').textContent"),/1556 of 1556 task names · 2124 metric rows/); +assert.ok((await evaluate("document.querySelector('#filterStatus').textContent")).startsWith(`${new Set(payload.rows.map(r=>r.task)).size} of ${new Set(payload.rows.map(r=>r.task)).size} task names · ${payload.rows.length} metric rows`)); assert.ok(await evaluate("document.querySelectorAll('#view *').length<2500"),'collapsed catalogue rendered too much content'); assert.equal(await evaluate("document.querySelectorAll('.catalogue-metric').length"),0); assert.match(await evaluate("document.querySelector('[data-eval=\"ARC Challenge\"] > summary .scoring-options').textContent"),/Selected: acc_norm.*Other available metrics: acc/); @@ -258,7 +262,7 @@ await click('#exportWeights'); const exported=await evaluate(`exportBlob.text().then(parseWeightProfile)`); assert.equal(exported.weights.Code,.16);assert.equal(exported.weights.Math,.14); assert.equal(exported.languages,undefined); -await click('#exportConfig');assert.equal(await evaluate('exportBlob.text().then(parseCatalogue).then(c=>c.languages.flatMap(g=>g.tasks).length)'),1556); +await click('#exportConfig');assert.equal(await evaluate('exportBlob.text().then(parseCatalogue).then(c=>c.languages.flatMap(g=>g.tasks).length)'),configuredTasks); await evaluate(`URL.createObjectURL=originalCreate;HTMLAnchorElement.prototype.click=originalClick;loadTestConfig(DATA.scheme)`); // Missing configuration is a warning and exclusion, not a failed import. await evaluate(`(async()=>{const c=structuredClone(DATA.scheme);c.evals.find(e=>e.name==='HumanEval').match={name:'absent_eval_task'};await loadTestConfig(c);})()`); diff --git a/tests/test_components.cjs b/tests/test_components.cjs new file mode 100644 index 0000000..2910ba3 --- /dev/null +++ b/tests/test_components.cjs @@ -0,0 +1,119 @@ +'use strict'; +const test=require('node:test'),assert=require('node:assert/strict'); +const E=require('../app/eval_config.js'),A=require('../app/analysis.js'); +const levels=['low','medium','high','top']; +const config=()=>({version:1,name:'Components',weights:{Reasoning:1},english_weights:{Reasoning:.5},evals:[{name:'Poly',category:'Reasoning',match:{regex:'poly_.+'},metric:'acc',filter:'none',score:{scale:1},aggregation:{components:levels.map((name,i)=>({name,match:{regex:'poly_.+_'+name},relative_weight:2**i}))}}],languages:['en','de','fr'].map((lang,i)=>({tasks:levels.map(l=>'poly_'+lang+'_'+l),scope:'single',language:['eng_Latn','deu_Latn','fra_Latn'][i]}))}); +const rows=(cfg=config(),langs=['en'],values=[.6,.3,.15,0],checkpoint='A')=>E.auditRows(langs.flatMap(lang=>levels.map((l,i)=>({checkpoint,task:'poly_'+lang+'_'+l,metric:'acc',filter:'none',n_shot:'0',harness:'test',backend:'cpu',value:values[i]}))),cfg); +const close=(a,b)=>assert.ok(Math.abs(a-b)<1e-9,`${a} != ${b}`); +test('relative weights round trip; invalid component rules reject',()=>{ + const c=config();E.validateConfig(c);const catalogue={version:1,name:c.name,evals:c.evals,languages:c.languages};assert.deepEqual(E.parseCatalogue(E.serializeCatalogue(catalogue)),catalogue); + for(const mutate of [a=>a.components=[],a=>{a.components[0].weight=1;delete a.components[0].relative_weight;},a=>a.components[0].relative_weight=0,a=>a.components[0].relative_weight=-1,a=>a.components[0].relative_weight=Infinity,a=>a.components[0].relative_weight='1',a=>a.components[0].name='medium',a=>a.components[0].extra=true,a=>a.components[0].match={regex:'['},a=>a.extra=true,a=>a.components.forEach(c=>c.relative_weight=1e308)]){const bad=config();mutate(bad.evals[0].aggregation);assert.throws(()=>E.validateConfig(bad));} +}); +test('PolyMath formula, scale invariance, coefficients and unchanged source scores',()=>{ + const c=config(),r=rows(c),result=A.totals(r,c,c.weights);close(result.score,12);close([...result.rowWeights.values()].reduce((s,v)=>s+v,0),1);close(result.rowWeights.get(r[3]),8/15);close(r[0].raw_score_100,60); + c.evals[0].aggregation.components.forEach(x=>x.relative_weight*=10);close(A.totals(r,c,c.weights).score,12); +}); +test('components normalize before weighting and balance languages after weighting',()=>{ + const c=config();c.evals[0].normalize={min:.25,max:1}; + const r=[...rows(c,['en'],[.25,.25,.25,1]),...rows(c,['de','fr'],[.25,.25,.25,.25])]; + close(A.totals(r,c,c.weights).score,100*8/45); + for(const mode of ['english_eval','english_category'])close(A.totals(r,c,c.weights,mode).score,100*4/15); +}); +test('incomplete language excluded from both models, preserving other languages and raw audit',()=>{ + const c=config(),a=rows(c,['en','de']),b=rows(c,['en','de'],[.2,.3,.4,.5],'B').filter(r=>r.task!=='poly_en_top'); + const coverage=A.comparisonCoverage(a,b,c);assert.equal(coverage.a.length,4);assert.equal(coverage.b.length,4);assert.ok(coverage.a.every(r=>r.task.includes('_de_')));assert.equal(a.length,8);assert.ok(coverage.warnings.some(w=>w.type==='Incomplete components'&&w.detail.includes('top'))); + close(A.totals(coverage.a,c,c.weights).score,12); +}); +test('incomplete on both sides, missing metric, unknown language and unknown component never score',()=>{ + for(const kind of ['missing','metric','language','unmatched','ambiguous','duplicate']){ + const c=config();let raw=rows(c); + if(kind==='missing')raw.pop(); + if(kind==='metric')raw=E.auditRows(raw.map((r,i)=>i===3?{...r,metric:'wrong'}:r),c).filter(r=>r.selected); + if(kind==='language')c.languages=[]; + if(kind==='unmatched')c.evals[0].aggregation.components[3].match={name:'absent'}; + if(kind==='ambiguous')c.evals[0].aggregation.components[3].match={regex:'poly_.+'}; + if(kind==='duplicate'){raw.push({...raw[0],task:'poly_en_extra_low'});c.languages[0].tasks.push('poly_en_extra_low');} + const result=A.comparisonCoverage(raw,raw,c);assert.equal(result.pairs.length,0,kind);assert.ok(result.warnings.length,kind);assert.equal(A.totals(raw,c,c.weights).score,null,kind); + } +}); +test('different protocols cannot fill missing components; complete protocols average equally',()=>{ + const c=config(),a=rows(c),mixed=a.map((r,i)=>({...r,n_shot:i===3?'5':'0'}));assert.equal(A.totals(mixed,c,c.weights).score,null); + const extra=rows(c,['en'],[1,1,1,1]).map(r=>({...r,n_shot:'5'}));close(A.totals(a.concat(extra),c,c.weights).score,56); +}); +test('delta contributions reconcile and raw comparisons remain raw',()=>{ + const c=config(),a=rows(c,['en','de']),b=rows(c,['en','de'],[.2,.3,.4,.5],'B'),coverage=A.comparisonCoverage(a,b,c); + for(const mode of ['standard','english_eval','english_category']){ + const score=A.totals(coverage.a,c,c.weights,mode).score-A.totals(coverage.b,c,c.weights,mode).score; + const items=A.comparisonRows(coverage.pairs,coverage.a,c,c.weights,'variant','weighted','descending','delta',new Map(),mode); + close(items.reduce((s,r)=>s+r.weightedDelta,0),score);close(items.find(r=>r.task==='poly_en_low').rawDelta,40); + } +}); +test('breakdowns expose calculated parents and individual weights/contributions',()=>{ + const c=config(),r=rows(c),p=A.pairRows(r,r),m=new Map(c.languages.flatMap(g=>g.tasks.map(t=>[t,g]))); + const tree=A.buildBreakdownTree(p,m,'category','','',c),language=tree[0].children[0].children[0];close(language.a,12);const low=language.children.find(n=>n.label==='poly_en_low');close(low.a,60);close(low.componentShare,1/15);close(low.componentA,4); + const filtered=A.buildBreakdownTree(p.slice(0,1),m,'category','','',c);assert.equal(filtered[0].a,null); +}); +test('ordinary evals retain simple averages',()=>{const c=config();delete c.evals[0].aggregation;close(A.totals(rows(c),c,c.weights).score,26.25);}); + +test('200 independent component calculations reconcile modes, groups, and category shares',()=>{ + let seed=811;const rand=()=>((seed=(Math.imul(seed,1664525)+1013904223)>>>0)/2**32); + for(let trial=0;trial<200;trial++){ + const c=config(),share=rand(),weights=levels.map(()=>1+Math.floor(rand()*8));c.english_weights.Reasoning=share; + c.evals[0].aggregation.components.forEach((x,i)=>x.relative_weight=weights[i]); + c.evals.push({name:'Ordinary',category:'Reasoning',match:{name:'ordinary'},metric:'acc',filter:'none',score:{scale:1}}); + c.languages.push({tasks:['ordinary'],scope:'single',language:'eng_Latn'}); + const inputs=['en','de','fr'].map(()=>levels.map(()=>rand())); + let r=inputs.flatMap((values,i)=>rows(c,[['en','de','fr'][i]],values)); + const ordinary=rand()*100;r.push({...r[0],task:'ordinary',eval:'Ordinary',score_100:ordinary,raw_score_100:ordinary}); + const means=inputs.map(xs=>xs.reduce((sum,x,i)=>sum+x*100*weights[i],0)/weights.reduce((a,b)=>a+b,0)),other=(means[1]+means[2])/2; + const expected={standard:((means[0]+means[1]+means[2])/3+ordinary)/2,english_eval:(share*means[0]+(1-share)*other+ordinary)/2,english_category:share*(means[0]+ordinary)/2+(1-share)*other}; + for(const [mode,want] of Object.entries(expected)){const t=A.totals(r,c,c.weights,mode);close(t.score,want);close(r.reduce((sum,x)=>sum+x.score_100*t.rowWeights.get(x),0),want);close([...t.rowWeights.values()].reduce((sum,x)=>sum+x,0),1);} + } +}); +test('missing groups redistribute eval/category weights',()=>{ + const c=config();c.weights={Reasoning:.4,Other:.6};c.evals.push({name:'Other',category:'Other',metric:'acc',filter:'none',match:{name:'other'},score:{scale:1}}); + const incomplete=rows(c).slice(0,3),ordinary={...incomplete[0],task:'other',eval:'Other',category:'Other',score_100:70,raw_score_100:70}; + const t=A.totals([...incomplete,ordinary],c,c.weights);close(t.score,70);assert.ok(t.evals.find(e=>e.name==='Poly').excluded);close(t.rowWeights.get(ordinary),1); +}); +test('explicit translation pairs are independent component groups',()=>{ + const c=config(),r=rows(c,['en','de']);c.languages=c.languages.slice(0,2).map((g,i)=>({tasks:g.tasks,scope:'translation',source_language:i?'deu_Latn':'eng_Latn',target_language:'fra_Latn'})); + assert.equal(A.componentCoverage(r,c).groups.length,2);close(A.totals(r,c,c.weights).score,12); +}); + +test('catalogue rejects incompatible component matching and selection',()=>{ + for(const mutate of [ + c=>c.evals[0].aggregation.components[0].match={regex:'poly_.+'}, + c=>c.evals[0].aggregation.components[0].match={name:'other_eval'}, + c=>c.evals[0].select={regex:'poly_.+_(low|medium|high)'}, + c=>c.languages[0].tasks.pop(), + c=>c.evals[0].aggregation.components[0].metric='other', + c=>c.evals[0].aggregation.components[0].filter='other', + c=>c.evals[0].aggregation.components[0].score={scale:100}, + c=>c.evals[0].aggregation.components[0].normalize={min:.25,max:1}, + c=>c.evals[0].aggregation.components[0].shots=5 + ]){const c=config();mutate(c);assert.throws(()=>E.validateConfig(c));} + // A wholly absent language is not an implicit requirement. + const c=config();c.languages.pop();assert.doesNotThrow(()=>E.validateConfig(c)); +}); +test('named sets require compatible complete component selections per language and shots',()=>{ + const S=require('../app/suite_config.js'),c=config(),catalogue={version:1,name:c.name,evals:c.evals,languages:c.languages},profile={version:1,name:'P',weights:c.weights}; + const suite=()=>({version:1,name:'Components',mode:'fixed',evals:[{name:'Poly',variants:levels.map(l=>({task:'poly_en_'+l,n_shot:0}))}]}); + for(const mutate of [s=>s.evals[0].variants.pop(),s=>s.evals[0].variants[3].n_shot=5,s=>delete s.evals[0].variants[3].n_shot,s=>s.evals[0].variants[3].task='poly_de_top']){ + const s=suite();mutate(s);assert.throws(()=>S.resolveConfig(catalogue,s,profile),/aggregation.*Poly|Poly.*aggregation/i); + } + for(const s of [suite(),{...suite(),evals:[{name:'Poly'}]},{version:1,name:'Available',mode:'available'}])assert.doesNotThrow(()=>S.resolveConfig(catalogue,s,profile)); + const allShots=suite();allShots.evals[0].variants.forEach(v=>delete v.n_shot);assert.doesNotThrow(()=>S.resolveConfig(catalogue,allShots,profile)); + const pinned=structuredClone(catalogue);pinned.evals[0].shots=0;const mixed=suite();delete mixed.evals[0].variants[0].n_shot;assert.doesNotThrow(()=>S.resolveConfig(pinned,mixed,profile)); + const both=suite();both.evals[0].variants.push(...both.evals[0].variants.map(v=>({...v,n_shot:5})));assert.doesNotThrow(()=>S.resolveConfig(catalogue,both,profile)); + const alias=structuredClone(catalogue);alias.languages[0].tasks.push('poly_en_alias_low');const duplicate=suite();duplicate.evals[0].variants.push({task:'poly_en_alias_low',n_shot:0});assert.throws(()=>S.resolveConfig(alias,duplicate,profile),/multiple/i); +}); + +test('newly observed tasks must match exactly one component even beyond declared metadata',()=>{ + const c=config(),raw={...rows(c)[0],task:'poly_en_future'}; + assert.throws(()=>E.auditRows([raw],c),/Incompatible aggregation config.*exactly one/); + c.evals[0].aggregation.components[0].match.regex='poly_.+_(low|future)'; + c.evals[0].aggregation.components[1].match.regex='poly_.+_(medium|future)'; + assert.doesNotThrow(()=>E.validateConfig(c)); + assert.throws(()=>E.auditRows([raw],c),/Incompatible aggregation config.*exactly one/); + assert.equal(E.auditRows([{...raw,metric:'alternate'}],c)[0].selected,false); +}); diff --git a/tests/test_data.cjs b/tests/test_data.cjs index 4ede5a6..59eb179 100644 --- a/tests/test_data.cjs +++ b/tests/test_data.cjs @@ -1,13 +1,14 @@ +const {diagnosticsFor}=require('./diagnostic_fixture.cjs'); 'use strict'; const test=require('node:test'),assert=require('node:assert/strict'); -const {parseCSV,totals,comparisonRows,comparisonCoverage,collectWarnings,languageRoles}=require('../app/app.js'); +const {parseCSV,totals,comparisonRows,comparisonCoverage,languageRoles}=require('../app/analysis.js'); const {auditRows,normalizeScore,validateConfig}=require('../app/eval_config.js'); const config=()=>({version:1,name:'Fixture',weights:{C:1},evals:[{name:'Eval',category:'C',match:{regex:'task_.+'},metric:'acc',filter:'',score:{scale:1},normalize:{min:.25,max:1}}],languages:[{tasks:['task_en'],scope:'single',language:'eng_Latn'}]}); const row=(patch={})=>({checkpoint:'Model A',task:'task_en',metric:'acc',filter:'',n_shot:'0',harness:'test',backend:'cpu',value:'.625',...patch}); const close=(a,b)=>{assert.ok(Number.isFinite(a)&&Number.isFinite(b));assert.ok(Math.abs(a-b)<1e-9,`${a} != ${b}`);}; test('language columns sort siblings recursively without changing scores or hierarchy',()=>{ - const {sortBreakdownTree}=require('../app/app.js'); + const {sortBreakdownTree}=require('../app/analysis.js'); const node=(label,a,children=[])=>({kind:'language',label,a,b:a===null?null:100-a,delta:a===null?null:2*a-100,count:a,children}); const tree=[node('fra_Latn',30,[node('z',2),node('a',8)]),node('eng_Latn',80),node('deu_Latn',80),node('mul',null)]; const original=structuredClone(tree); @@ -66,11 +67,11 @@ test('duplicates are rejected within a model; distinct models and protocols rema test('invalid alternate scores are retained for inspection but never normalized or substituted',()=>{ const cfg=config(),rows=auditRows([row({metric:'acc_norm',value:'not numeric'})],cfg); assert.equal(rows[0].selected,false);assert.equal(rows[0].value,'not numeric');assert.equal(rows[0].score_100,null); - assert.ok(collectWarnings(new Map([['A',rows]]),cfg).some(w=>w.type==='Missing scoring field')); + assert.ok(diagnosticsFor(new Map([['A',rows]]),cfg).some(w=>w.type==='Missing scoring field')); }); test('unknown language warns without exclusion or invented language labels',()=>{ const cfg=config(),rows=auditRows([row({task:'task_unknown'})],cfg); - const warnings=collectWarnings(new Map([['A',rows]]),cfg); + const warnings=diagnosticsFor(new Map([['A',rows]]),cfg); assert.ok(warnings.some(w=>w.type==='Unknown language'&&w.detail.includes('English'))); assert.equal(rows[0].selected,true);assert.equal(languageRoles(rows[0],new Map())[0].language,'Unknown'); close(totals(rows,cfg,cfg.weights,'english_eval',{C:.5}).score,50); @@ -85,12 +86,12 @@ test('invalid optional sample counts warn without becoming weights or dropping s const cfg=config(); for(const n_samples of ['0','-1','1.5','many']){ const rows=auditRows([row({n_samples})],cfg); - assert.ok(collectWarnings(new Map([['A',rows]]),cfg).some(w=>w.type==='Invalid sample count')); + assert.ok(diagnosticsFor(new Map([['A',rows]]),cfg).some(w=>w.type==='Invalid sample count')); close(totals(rows,cfg,cfg.weights).score,50); } for(const n_samples of ['',undefined,'10']){ const rows=auditRows([row({n_samples})],cfg); - assert.ok(!collectWarnings(new Map([['A',rows]]),cfg).some(w=>w.type==='Invalid sample count')); + assert.ok(!diagnosticsFor(new Map([['A',rows]]),cfg).some(w=>w.type==='Invalid sample count')); } }); test('no shared data produces unavailable totals, not a zero score',()=>{ @@ -153,3 +154,27 @@ test('300 varied comparisons reconcile all aggregates, row weights, filters and } } }); + + +test('synthetic choices preserve coverage and scale, are deterministic, and never mutate input',()=>{ + const {synthetic,syntheticOptions,isDemoModel}=require('../app/analysis.js'); + const c=config();c.evals[0].score.scale=100; + const rows=auditRows([row({value:'0'}),row({task:'task_fr',value:'60'}),row({task:'task_de',value:'100'})],c),before=structuredClone(rows); + const results=syntheticOptions.map(option=>{ + assert.ok(isDemoModel(option.name)); + assert.throws(()=>auditRows([row({checkpoint:option.name})],c),/reserved/i); + const actual=synthetic(rows,c,option); + assert.deepEqual(actual,synthetic(rows,c,option)); + assert.deepEqual(actual.map(r=>r.task),rows.map(r=>r.task)); + for(const r of actual){ + assert.equal(r.checkpoint,option.name); + assert.ok(r.raw_score_100>=0&&r.raw_score_100<=100); + close(r.raw_score_100,Number(r.value)); + close(r.score_100,normalizeScore(r.value,c.evals[0]).score_100); + assert.equal(r.stderr,''); + } + return actual; + }); + assert.equal(new Set(results.map(rows=>JSON.stringify(rows.map(r=>r.value)))).size,syntheticOptions.length); + assert.deepEqual(rows,before); +}); diff --git a/tests/test_data.py b/tests/test_data.py index 180ff41..37126f6 100644 --- a/tests/test_data.py +++ b/tests/test_data.py @@ -4,13 +4,15 @@ import json import random import subprocess +import sys import tempfile import unittest from copy import deepcopy from pathlib import Path from app.build import build, summarize -from app.config_engine import classify, load_csv, normalize_score, validate_config +from quickdash.io import load_csv +from quickdash.config import classify, normalize_score, validate_config ROOT = Path(__file__).resolve().parent.parent @@ -28,7 +30,7 @@ def row(**patch): def javascript(cases, expression): - script = "const api=require('./app/eval_config.js'),app=require('./app/app.js');const cases=JSON.parse(require('fs').readFileSync(0,'utf8'));process.stdout.write(JSON.stringify(cases.map(c=>{try{return {value:" + expression + "}}catch(e){return {error:e.message}}})));" + script = "const api=require('./app/eval_config.js'),app=require('./app/analysis.js');const cases=JSON.parse(require('fs').readFileSync(0,'utf8'));process.stdout.write(JSON.stringify(cases.map(c=>{try{return {value:" + expression + "}}catch(e){return {error:e.message}}})));" return json.loads(subprocess.check_output(['node', '-e', script], input=json.dumps(cases), text=True, cwd=ROOT)) @@ -42,6 +44,50 @@ def inputs(folder, c=None): class DataContracts(unittest.TestCase): + def test_component_config_validation_and_standalone_build(self): + c=config();e=c['evals'][0];e.pop('normalize') + levels=['low','medium','high','top'] + e['aggregation']={'components':[dict(name=name,match={'regex':'task_.+_'+name},relative_weight=2**i) for i,name in enumerate(levels)]} + c['languages']=[dict(tasks=['task_en_'+name for name in levels],scope='single',language='eng_Latn')] + validate_config(c) + bad_values=[{'components':[dict(name='low',match={'name':'x'},weight=1)]},None,{}, {'components':[]}, {'components':[dict(name='low',match={'name':'x'},relative_weight=True)]}] + for value in bad_values: + bad=deepcopy(c);bad['evals'][0]['aggregation']=value + with self.assertRaises(ValueError):validate_config(bad) + self.assertIn('error',javascript([bad],'api.validateConfig(c)')[0]) + for mutation in [lambda c:c['languages'][0]['tasks'].pop(), + lambda c:c['evals'][0].update(select={'regex':'task_.+_(low|medium|high)'}), + lambda c:c['evals'][0]['aggregation']['components'][0].update(match={'regex':'task_.+'}), + lambda c:c['evals'][0]['aggregation']['components'][0].update(metric='other')]: + bad=deepcopy(c);mutation(bad) + with self.assertRaises(ValueError):validate_config(bad) + self.assertIn('error',javascript([bad],'api.validateConfig(c)')[0]) + rr=[row(task='task_en_'+name,value=str(value)) for name,value in zip(levels,[.6,.3,.15,0])] + for mode in ['standard','english_eval','english_category']: + c['english_weights']={'C':.5} + self.assertAlmostEqual(summarize(classify(rr,c),c,mode)[0]['score'],12) + result=summarize(classify(rr[:-1],c),c,mode)[0] + self.assertIsNone(result['score']);self.assertTrue(result['warnings']) + with tempfile.TemporaryDirectory() as tmp: + folder=Path(tmp);kw=inputs(folder,c);source=folder/'scores.csv' + source.write_text(','.join(rr[0])+'\n'+'\n'.join(','.join(r.values()) for r in rr)) + with contextlib.redirect_stdout(io.StringIO()):build(source,folder/'out',**kw) + data=json.loads((folder/'out/analysis.json').read_text()) + self.assertAlmostEqual(data['models'][0]['score'],12) + self.assertEqual(data['models'][0]['evals'][0]['count'],4) + self.assertIn('Component aggregation',(folder/'out/index.html').read_text()) + self.assertTrue((folder/'out/eval-scores.csv').exists()) + previous=(folder/'out/index.html').read_bytes() + suite=dict(version=1,name='Partial components',mode='fixed',evals=[dict(name='Eval',variants=[dict(task=r['task'],n_shot=0) for r in rr[:-1]])]) + (folder/'set.yaml').write_text(json.dumps(suite));kw['suite_path']=folder/'set.yaml' + with self.assertRaisesRegex(ValueError,'Incompatible aggregation config.*top'):build(source,folder/'out',**kw) + self.assertEqual((folder/'out/index.html').read_bytes(),previous) + suite['evals'][0]['variants'].append(dict(task=rr[-1]['task'],n_shot=5)) + (folder/'set.yaml').write_text(json.dumps(suite)) + with self.assertRaisesRegex(ValueError,'Incompatible aggregation config'):build(source,folder/'out',**kw) + self.assertEqual((folder/'out/index.html').read_bytes(),previous) + + def test_independent_directory_defaults_and_explicit_overrides(self): with tempfile.TemporaryDirectory() as tmp: folder=Path(tmp);kw=inputs(folder);profiles=folder/'profiles';profiles.mkdir() @@ -69,7 +115,7 @@ def test_invalid_directory_defaults_preserve_existing_output(self): def test_cli_defaults_to_any_available_and_separate_weight_profile(self): with tempfile.TemporaryDirectory() as tmp: - subprocess.run(['python3','-m','app.build','--output',tmp],cwd=ROOT,check=True,stdout=subprocess.DEVNULL) + subprocess.run([sys.executable,'-m','app.build','--output',tmp],cwd=ROOT,check=True,stdout=subprocess.DEVNULL) data=json.loads((Path(tmp)/'analysis.json').read_text()) self.assertEqual(data['suite']['mode'],'available') self.assertEqual(data['profile']['name'],'Original') @@ -129,7 +175,7 @@ def test_named_set_scopes_score_but_keeps_full_audit(self): self.assertEqual(data['models'][0]['evals'][0]['count'],1) def test_cli_rejects_both_csv_and_results_directory(self): - result=subprocess.run(['python3','-m','app.build','unused.csv','--results-dir','unused'],capture_output=True,text=True) + result=subprocess.run([sys.executable,'-m','app.build','unused.csv','--results-dir','unused'],capture_output=True,text=True) self.assertEqual(result.returncode,2);self.assertIn('not both',result.stderr) def test_shared_results_directory_combines_models_and_rejects_duplicates(self): @@ -148,6 +194,30 @@ def test_shared_results_directory_combines_models_and_rejects_duplicates(self): with self.assertRaisesRegex(ValueError,'Duplicate model'):build(None,out,**kw,results_dir=results) self.assertEqual((out/'index.html').read_bytes(),previous) + def test_sample_is_used_only_for_an_empty_results_directory(self): + with tempfile.TemporaryDirectory() as tmp: + folder=Path(tmp);kw=inputs(folder);results=folder/'results';results.mkdir();out=folder/'out' + sample=folder/'sample.csv';r=row(checkpoint='Sample') + sample.write_text(','.join(r)+'\n'+','.join(r.values())) + with contextlib.redirect_stdout(io.StringIO()): + build(None,out,**kw,results_dir=results,sample_csv=sample) + data=json.loads((out/'analysis.json').read_text()) + self.assertEqual(data['sample_models'],['Sample']) + self.assertEqual([m['model'] for m in data['models']],['Sample']) + self.assertEqual(data['sources'][0]['file'],'sample.csv') + r=row(checkpoint='Shared');(results/'real.csv').write_text(','.join(r)+'\n'+','.join(r.values())) + # A real dataset wins even if the fallback path is unavailable. + with contextlib.redirect_stdout(io.StringIO()): + build(None,out,**kw,results_dir=results,sample_csv=folder/'absent.csv') + data=json.loads((out/'analysis.json').read_text()) + self.assertEqual(data['sample_models'],[]) + self.assertEqual([m['model'] for m in data['models']],['Shared']) + (results/'real.csv').write_text('invalid') + with self.assertRaises(ValueError):build(None,out,**kw,results_dir=results,sample_csv=sample) + with self.assertRaisesRegex(ValueError,'sample.*results-dir'): + build(None,out,**kw,sample_csv=sample) + with self.assertRaises(ValueError):build(None,out,**kw,results_dir=folder/'missing',sample_csv=sample) + def test_shared_results_report_bad_filename_and_reject_missing_directory(self): with tempfile.TemporaryDirectory() as tmp: folder=Path(tmp);kw=inputs(folder) @@ -222,7 +292,7 @@ def test_row_validation_parity_and_errors(self): for field in ['checkpoint', 'task', 'metric', 'harness', 'backend']: invalid.extend(row(**{field: value}) for value in ['', ' ', None, 1]) invalid.extend(row(n_shot=value) for value in ['', '-1', '1.5', 'NaN', '00', '1e1', True, None]) - invalid.extend([row(filter=None), row(value=True), row(checkpoint='SYNTHETIC demo — perturbed')]) + invalid.extend([row(filter=None), row(value=True), row(checkpoint='SYNTHETIC demo — perturbed'), row(checkpoint='SYNTHETIC demo — higher scores')]) cases = [dict(rows=[r], config=config()) for r in invalid] for c, js in zip(cases, javascript(cases, 'api.auditRows(c.rows,c.config)')): with self.subTest(row=c['rows']): diff --git a/tests/test_engines.py b/tests/test_engines.py new file mode 100644 index 0000000..2b691ac --- /dev/null +++ b/tests/test_engines.py @@ -0,0 +1,922 @@ +"""Identical behavioral cases exercise native Python and the actual browser engine.""" + +import csv +import io +import json +import math +import random +import subprocess +import unittest +import warnings +from copy import deepcopy +from pathlib import Path +from quickdash import analyze, compare, load_config, QuickdashWarning, DiagnosticError +from quickdash.io import parse_csv, parse_yaml + +ROOT = Path(__file__).resolve().parent.parent + + +def fixture(): + catalogue = dict( + version=1, + name="Fixture", + evals=[ + dict( + name="E", + category="C", + match={"regex": "e_.+"}, + metric="acc", + filter="none", + score={"scale": 1}, + normalize={"min": 0.25, "max": 1}, + ) + ], + languages=[ + dict(tasks=["e_en"], scope="single", language="eng_Latn"), + dict(tasks=["e_fr"], scope="single", language="fra_Latn"), + ], + ) + return dict( + catalogue=catalogue, + profile=dict(version=1, name="Weights", weights={"C": 1}), + suite=dict(version=1, name="Available", mode="available"), + ) + + +def row(task="e_en", value=".625", **kwargs): + return { + **dict( + checkpoint="A", + task=task, + metric="acc", + filter="none", + n_shot="0", + harness="fixture", + backend="cpu", + value=value, + ), + **kwargs, + } + + +def paired(rows): + return rows + [{**r, "checkpoint": "B", "value": ".4"} for r in rows] + + +def native(c): + try: + cfg = ( + load_config( + catalogue=parse_yaml(c["yaml"]["catalogue"]), + weights=parse_yaml(c["yaml"]["profile"]), + eval_set=parse_yaml(c["yaml"]["suite"]), + ) + if "yaml" in c + else c["config"] + ) + rows = parse_csv(c["csv"]) if "csv" in c else c["rows"] + value = ( + compare( + rows, cfg, a=c.get("a", "A"), b=c.get("b", "B"), diagnostics="collect" + ) + if c.get("operation") == "compare" + else analyze(rows, cfg, diagnostics="collect") + ) + return {"value": value} + except ValueError as error: + return {"error": True, "message": str(error)} + + +def semantics(value): + # Human-facing wording is deliberately not a cross-language API contract. + if isinstance(value, dict): + return { + k: semantics(v) for k, v in value.items() if k not in ("detail", "variants") + } + if isinstance(value, list): + return [semantics(v) for v in value] + return value + + +class Engines(unittest.TestCase): + def close(self, a, b, path="result"): + if isinstance(a, dict): + self.assertEqual(a.keys(), b.keys(), path) + for k in a: + self.close(a[k], b[k], path + "." + k) + elif isinstance(a, list): + self.assertEqual(len(a), len(b), path) + for i, (x, y) in enumerate(zip(a, b)): + self.close(x, y, f"{path}[{i}]") + elif isinstance(a, (int, float)) and not isinstance(a, bool): + self.assertIsInstance(b, (int, float), path) + self.assertTrue( + math.isclose(a, b, rel_tol=1e-12, abs_tol=1e-9), f"{path}: {a} != {b}" + ) + else: + self.assertEqual(a, b, path) + + def both(self, cases): + js = json.loads( + subprocess.check_output( + ["node", "tests/engine_adapter.cjs"], + input=json.dumps(cases), + text=True, + cwd=ROOT, + ) + ) + self.assertEqual(len(js), len(cases), "Every case must produce a JS result") + results = [] + for i, (case, other) in enumerate(zip(cases, js)): + py = native(case) + with self.subTest(case=i): + self.assertEqual("error" in py, "error" in other, (py, other, case)) + if "error" not in py: + self.close(semantics(py["value"]), semantics(other["value"])) + results.append((py, other)) + return results + + def test_published_sample_across_all_shipped_configs(self): + # Discover files so adding a profile or set automatically extends parity coverage. + rows = parse_csv((ROOT / "examples/sample-evals.csv").read_text()) + a = rows[0]["checkpoint"] + b = "Parity comparison" + copy = [{**r, "checkpoint": b} for r in rows] + profiles = sorted( + p + for p in (ROOT / "configs/weights").iterdir() + if p.suffix in {".yaml", ".yml"} + ) + suites = sorted( + p + for p in (ROOT / "configs/sets").iterdir() + if p.suffix in {".yaml", ".yml"} + ) + self.assertTrue(profiles) + self.assertTrue(suites) + for weights in profiles: + for suite in suites: + config = load_config( + catalogue=ROOT / "configs/catalogue.yaml", + weights=weights, + eval_set=suite, + ) + for mode in ("standard", "english_eval", "english_category"): + config["profile"]["aggregate"] = mode + with self.subTest( + weights=weights.name, suite=suite.name, mode=mode + ): + cases = [ + dict(config=config, rows=rows), + dict( + config=config, + rows=rows + copy, + operation="compare", + a=a, + b=b, + ), + dict( + config=config, + rows=rows + [r for i, r in enumerate(copy) if i % 7], + operation="compare", + a=a, + b=b, + ), + ] + for outputs in self.both(cases): + for result in outputs: + self.assertNotIn("error", result) + report = result["value"] + models = report.get( + "models", [report.get("a"), report.get("b")] + ) + for model in models: + self.assertIsNotNone(model["score"]) + self.check_tree(model["tree"]) + # Matching data has identical aggregate scores and contributions. + self.assertEqual(native(cases[1])["value"]["delta"], 0) + + def test_hand_calculated_modes_and_tree(self): + c = fixture() + c["catalogue"]["evals"].append( + dict( + name="F", + category="C", + match={"name": "f_en"}, + metric="acc", + filter="none", + score={"scale": 1}, + ) + ) + c["catalogue"]["languages"][0]["tasks"].append("f_en") + c["profile"]["english_weights"] = {"C": 0.5} + rr = [row(value="1"), row("e_fr", ".25"), row("f_en", ".6")] + cases = [] + for mode in ("standard", "english_eval", "english_category"): + cc = deepcopy(c) + cc["profile"]["aggregate"] = mode + cases.append(dict(config=cc, rows=rr)) + for outputs, expected in zip(self.both(cases), (55, 55, 40)): + for result in outputs: + self.assertAlmostEqual(result["value"]["models"][0]["score"], expected) + self.check_tree(result["value"]["models"][0]["tree"]) + # Unequal language counts distinguish per-eval balancing from ordinary averaging. + c["catalogue"]["languages"].append( + dict(tasks=["e_de"], scope="single", language="deu_Latn") + ) + rr.append(row("e_de", ".25")) + cases = [] + for mode in ("standard", "english_eval", "english_category"): + cc = deepcopy(c) + cc["profile"]["aggregate"] = mode + cases.append(dict(config=cc, rows=rr)) + for outputs, expected in zip(self.both(cases), (140 / 3, 55, 40)): + for result in outputs: + self.assertAlmostEqual(result["value"]["models"][0]["score"], expected) + + def check_tree(self, node): + if node["children"]: + self.assertAlmostEqual( + sum(c["contribution"] for c in node["children"]), node["contribution"] + ) + if node["effective_weight"]: + self.assertAlmostEqual(sum(c["weight"] for c in node["children"]), 1) + self.assertAlmostEqual( + sum(c["weight"] * (c["score"] or 0) for c in node["children"]), + node["score"], + ) + for child in node["children"]: + self.check_tree(child) + + def test_warning_and_exclusion_contract(self): + cases = [] + expected = [] + + def add(config, rows, codes, operation="compare"): + cases.append(dict(config=config, rows=rows, operation=operation)) + expected.append(set(codes)) + + c = fixture() + c["catalogue"]["evals"][0]["warning"] = "Caveat" + add(c, paired([row()]), ["config_caveat"]) + c = fixture() + c["suite"]["exclude"] = ["E"] + add(c, paired([row()]), ["not_used"]) + c = fixture() + c["suite"] = dict( + version=1, + name="Required", + mode="fixed", + evals=[dict(name="E", variants=[dict(task="e_en"), dict(task="e_fr")])], + ) + add(c, paired([row()]), ["missing_suite_data"]) + c = fixture() + rr = paired([row()]) + rr[0]["metric"] = "acc_norm" + add( + c, rr, ["missing_scoring_field", "no_selected_score", "comparison_coverage"] + ) + c = fixture() + rr = paired([row()]) + rr[0]["filter"] = "other" + add( + c, + rr, + ["missing_scoring_setting", "no_selected_score", "comparison_coverage"], + ) + add(fixture(), paired([row()]) + [row("e_fr")], ["comparison_coverage"]) + add(fixture(), paired([row(), row("unknown")]), ["no_config"]) + add(fixture(), paired([row("e_unknown")]), ["unknown_language"]) + add(fixture(), paired([row(n_samples="many")]), ["invalid_sample_count"]) + rr = paired([row(n_samples="100")]) + rr[1]["n_samples"] = "90" + add(fixture(), rr, ["sample_count_mismatch"]) + add( + fixture(), + paired([row(), row("e_fr", n_shot="5")]), + ["inconsistent_scoring_settings"], + ) + for outputs, codes in zip(self.both(cases), expected): + for result in outputs: + self.assertNotIn("error", result) + report = result["value"] + self.assertEqual({d["code"] for d in report["diagnostics"]}, codes) + self.assertEqual( + {r["task"] for r in report["a"]["measurements"] if r["included"]}, + {r["task"] for r in report["b"]["measurements"] if r["included"]}, + ) + + def test_dynamic_components(self): + cases = [] + expected = [] + for n in (1, 2, 3, 4, 7): + c = fixture() + e = c["catalogue"]["evals"][0] + e.pop("normalize") + e["aggregation"] = { + "components": [ + dict( + name=str(i), match={"regex": f"e_.+_{i}"}, relative_weight=i + 1 + ) + for i in range(n) + ] + } + c["catalogue"]["languages"] = [ + dict( + tasks=[f"e_en_{i}" for i in range(n)], + scope="single", + language="eng_Latn", + ) + ] + rows = [row(f"e_en_{i}", str((i + 1) / (n + 1))) for i in range(n)] + cases.append(dict(config=c, rows=paired(rows))) + expected.append( + 100 + * sum((i + 1) ** 2 / (n + 1) for i in range(n)) + / sum(range(1, n + 1)) + ) + for outputs, score in zip(self.both(cases), expected): + for result in outputs: + self.assertAlmostEqual(result["value"]["models"][0]["score"], score) + self.check_tree(result["value"]["models"][0]["tree"]) + c = cases[-1]["config"] + rr = cases[-1]["rows"][:-1] + for result in self.both([dict(config=c, rows=rr, operation="compare")])[0]: + self.assertIsNone(result["value"]["a"]["score"]) + self.assertIn( + "incomplete_components", + {d["code"] for d in result["value"]["diagnostics"]}, + ) + c = deepcopy(c) + c["suite"] = dict( + version=1, + name="Incomplete", + mode="fixed", + evals=[dict(name="E", variants=[dict(task="e_en_0")])], + ) + for result in self.both([dict(config=c, rows=rr)])[0]: + self.assertIn("error", result) + + def test_input_and_config_edge_cases(self): + cases = [] + mutations = [ + lambda c: c["catalogue"].update(name=" "), + lambda c: c["catalogue"]["evals"][0].update(category=" "), + lambda c: c["profile"].update(weights={"C": 0}), + lambda c: c["catalogue"]["evals"][0].update(match={"regex": "(?=e)e.*"}), + lambda c: c["catalogue"]["evals"][0].update(shots=True), + ] + for mutate in mutations: + c = fixture() + mutate(c) + cases.append(dict(config=c, rows=[row()])) + for field in ( + "checkpoint", + "task", + "metric", + "filter", + "n_shot", + "harness", + "backend", + "value", + ): + r = row() + del r[field] + cases.append(dict(config=fixture(), rows=[r])) + for value in ("NaN", "Infinity", "", "0x1", None, True, -0.1, 1.1): + cases.append(dict(config=fixture(), rows=[row(value=value)])) + for name in ( + "SYNTHETIC demo — perturbed", + "SYNTHETIC demo — higher scores", + "SYNTHETIC demo — future option", + ): + cases.append(dict(config=fixture(), rows=[row(checkpoint=name)])) + cases.append(dict(config=fixture(), rows=[row(), row()])) + for outputs in self.both(cases): + for result in outputs: + self.assertIn("error", result) + c = fixture() + c["catalogue"]["evals"][0]["match"] = {"regex": r"e_\d+"} + for result in self.both([dict(config=c, rows=[row("e_١")])])[0]: + self.assertFalse( + result["value"]["models"][0]["measurements"][0]["selected"] + ) + + def test_yaml_csv_and_diagnostics_delivery(self): + c = fixture() + stream = io.StringIO() + w = csv.DictWriter(stream, fieldnames=row()) + w.writeheader() + w.writerow(row()) + case = dict( + yaml={k: json.dumps(v) for k, v in c.items()}, + csv="\ufeff" + stream.getvalue(), + ) + for result in self.both([case])[0]: + self.assertEqual(result["value"]["models"][0]["score"], 50) + with warnings.catch_warnings(record=True) as seen: + warnings.simplefilter("always") + r = analyze([row("unknown")], c) + self.assertEqual(len(seen), 1) + self.assertIsInstance(seen[0].message, QuickdashWarning) + self.assertEqual(seen[0].message.diagnostic, r.diagnostics[0]) + with self.assertRaises(DiagnosticError): + analyze([row("unknown")], c, diagnostics="error") + with warnings.catch_warnings(record=True) as seen: + analyze([row("unknown")], c, diagnostics="collect") + self.assertFalse(seen) + + def test_randomized_sizes_and_order(self): + rng = random.Random(91403) + cases = [] + for _ in range(60): + c = fixture() + c["catalogue"]["evals"] = [] + c["catalogue"]["languages"] = [] + c["profile"]["weights"] = {} + rows = [] + nc = rng.randint(1, 5) + for cat in range(nc): + category = f"C{cat}" + c["profile"]["weights"][category] = 1 / nc + for ev in range(rng.randint(1, 6)): + name = f"e{cat}_{ev}" + c["catalogue"]["evals"].append( + dict( + name=name, + category=category, + match={"regex": name + "_.+"}, + metric="acc", + filter="none", + score={"scale": 1}, + ) + ) + for lang in ("eng_Latn", "fra_Latn", "deu_Latn")[ + : rng.randint(1, 3) + ]: + task = name + "_" + lang + c["catalogue"]["languages"].append( + dict(tasks=[task], scope="single", language=lang) + ) + rows.append(row(task, str(rng.random()))) + c["profile"]["aggregate"] = rng.choice( + ["standard", "english_eval", "english_category"] + ) + c["profile"]["english_weights"] = { + k: rng.choice([0, 0.3, 0.5, 1]) for k in c["profile"]["weights"] + } + rng.shuffle(rows) + rng.shuffle(c["catalogue"]["evals"]) + cases.append(dict(config=c, rows=paired(rows), operation="compare")) + for outputs in self.both(cases): + for result in outputs: + report = result["value"] + self.check_tree(report["a"]["tree"]) + self.assertAlmostEqual( + sum(r["contribution_delta"] for r in report["deltas"]), + report["delta"], + ) + + def test_translation_pooling_and_unknown_language_balance(self): + c = fixture() + c["catalogue"]["evals"][0].pop("normalize") + c["profile"].update(aggregate="english_eval", english_weights={"C": 0.5}) + c["catalogue"]["languages"] = [ + dict( + tasks=["e_to_en"], + scope="translation", + source_language="fra_Latn", + target_language="eng_Latn", + ), + dict( + tasks=["e_from_en"], + scope="translation", + source_language="eng_Latn", + target_language="fra_Latn", + ), + dict(tasks=["e_pool"], scope="pooled", language="mul"), + ] + rr = [ + row("e_to_en", "1"), + row("e_from_en", "0"), + row("e_pool", ".5"), + row("e_unknown", "0"), + ] + for result in self.both([dict(config=c, rows=rr)])[0]: + report = result["value"] + self.assertAlmostEqual(report["models"][0]["score"], 25) + self.assertEqual( + {d["code"] for d in report["diagnostics"]}, {"unknown_language"} + ) + tree = report["models"][0]["tree"] + self.check_tree(tree) + self.assertEqual( + tree["children"][0]["children"][0]["children"][0]["kind"], + "language_group", + ) + self.assertEqual(len(report["models"][0]["measurements"]), 4) + + def test_empty_zero_weight_missing_category_and_self_comparison(self): + cases = [dict(config=fixture(), rows=[])] + c = fixture() + c["profile"]["weights"] = {"Other": 1} + cases.append(dict(config=c, rows=paired([row()]), operation="compare")) + c = fixture() + c["profile"]["weights"] = {"C": 0.3, "Other": 0.7} + cases.append(dict(config=c, rows=[row()])) + cases.append( + dict(config=fixture(), rows=[row()], operation="compare", a="A", b="A") + ) + results = self.both(cases) + for result in results[0]: + self.assertEqual(result["value"]["models"], []) + for result in results[1]: + self.assertIsNone(result["value"]["delta"]) + self.assertIn( + "no_category_weight", + {d["code"] for d in result["value"]["diagnostics"]}, + ) + for result in results[2]: + self.assertEqual(result["value"]["models"][0]["score"], 50) + for result in results[3]: + self.assertEqual(result["value"]["delta"], 0) + + def test_shots_protocols_and_named_requirements(self): + c = fixture() + c["suite"] = dict( + version=1, + name="Named", + mode="fixed", + evals=[ + dict( + name="E", + variants=[dict(task="e_en", n_shot=0), dict(task="e_fr", n_shot=5)], + ) + ], + ) + rr = paired([row(), row("e_fr", n_shot="5")]) + rr[-1]["backend"] = "different" + for result in self.both([dict(config=c, rows=rr, operation="compare")])[0]: + report = result["value"] + self.assertFalse(report["coverage"]["complete"]) + self.assertEqual(report["coverage"]["sharedRequired"], 1) + self.assertEqual(report["a"]["score"], 50) + self.assertEqual( + {r["task"] for r in report["a"]["measurements"] if r["included"]}, + {"e_en"}, + ) + c["catalogue"]["evals"][0]["shots"] = 0 + for result in self.both([dict(config=c, rows=rr)])[0]: + self.assertIn("error", result) + + def test_component_protocol_compatibility_and_scaling(self): + c = fixture() + e = c["catalogue"]["evals"][0] + e["aggregation"] = { + "components": [ + dict(name="low", match={"regex": "e_.+_low"}, relative_weight=1), + dict(name="top", match={"regex": "e_.+_top"}, relative_weight=8), + ] + } + c["catalogue"]["languages"] = [ + dict(tasks=["e_en_low", "e_en_top"], scope="single", language="eng_Latn") + ] + rr = paired([row("e_en_low", "1"), row("e_en_top", ".25")]) + cases = [dict(config=c, rows=rr, operation="compare")] + for field in ("n_shot", "harness", "backend"): + mutated = deepcopy(rr) + mutated[1][field] = "5" if field == "n_shot" else "different" + cases.append(dict(config=c, rows=mutated, operation="compare")) + scaled = deepcopy(c) + for component in scaled["catalogue"]["evals"][0]["aggregation"]["components"]: + component["relative_weight"] *= 11 + cases.append(dict(config=scaled, rows=rr, operation="compare")) + results = self.both(cases) + for result in results[0] + results[-1]: + self.assertAlmostEqual(result["value"]["a"]["score"], 100 / 9) + for outputs in results[1:-1]: + for result in outputs: + self.assertIsNone(result["value"]["delta"]) + self.assertIn( + "incomplete_components", + {d["code"] for d in result["value"]["diagnostics"]}, + ) + + def test_csv_and_yaml_failures_in_both_engines(self): + cases = [] + for source in ( + "", + "a,b\n", + "a,a\n1,2", + "a,b\n1", + "a,b\n1,2,3", + 'a,b\n"unclosed', + 'a,b\na"b,1', + 'a,b\n"a"b,1', + "a,b\n,", + ): + cases.append(dict(config=fixture(), csv=source)) + base = {k: json.dumps(v) for k, v in fixture().items()} + for source in ( + "version: 1\nversion: 1", + "!!python/object:evil {}", + "[unclosed", + "---\n{}\n---\n{}", + "version: 1\nname: &x [*x]\nevals: []\nlanguages: []", + ): + cases.append(dict(yaml={**base, "catalogue": source}, rows=[row()])) + for outputs in self.both(cases): + for result in outputs: + self.assertIn("error", result) + # These YAML words must remain strings; aliases and folded notes are supported. + source = """version: 1 +name: yes +evals: + - name: on + category: C + match: {name: e_en} + metric: acc + filter: none + score: {scale: 1} +languages: [] +notes: + - ¬e >- + Folded text + across lines. + - *note +""" + for result in self.both( + [dict(yaml={**base, "catalogue": source}, rows=[row()])] + )[0]: + self.assertEqual(result["value"]["models"][0]["score"], 62.5) + + def test_portable_regex_character_semantics(self): + cases = [] + expected = [] + for pattern, task, selected in [ + (r"e_\d+", "e_123", True), + (r"e_\d+", "e_١", False), + (r"e_\w+", "e_é", False), + (r"e_\s+", "e_\u00a0", True), + ("e_.+", "e_\r", False), + ("e_.", "e_😀", True), + ("e_[a-z]+", "e_en\n", False), + ]: + c = fixture() + c["catalogue"]["evals"][0]["match"] = {"regex": pattern} + cases.append(dict(config=c, rows=[row(task)])) + expected.append(selected) + for outputs, selected in zip(self.both(cases), expected): + for result in outputs: + self.assertEqual( + result["value"]["models"][0]["measurements"][0]["selected"], + selected, + ) + for field, value in [("n_shot", "0\n"), ("n_shot", 2**53), ("n_shot", 1.0)]: + case = dict(config=fixture(), rows=[row(**{field: value})]) + for result in self.both([case])[0]: + self.assertEqual("error" in result, value != 1.0) + + def test_python_runtime_never_calls_node_and_cli_streams(self): + import os + import sys + import tempfile + from unittest.mock import patch + + with patch( + "subprocess.run", side_effect=AssertionError("No subprocess allowed") + ): + c = load_config( + catalogue=fixture()["catalogue"], weights=fixture()["profile"] + ) + self.assertEqual( + analyze([row()], c, diagnostics="collect").models[0]["score"], 50 + ) + with tempfile.TemporaryDirectory() as tmp: + folder = Path(tmp) + for key, value in fixture().items(): + (folder / (key + ".yaml")).write_text(json.dumps(value)) + stream = io.StringIO() + writer = csv.DictWriter(stream, fieldnames=row()) + writer.writeheader() + writer.writerow(row("unknown")) + (folder / "scores.csv").write_text(stream.getvalue()) + command = [ + sys.executable, + "-m", + "quickdash", + str(folder / "scores.csv"), + "--catalogue", + str(folder / "catalogue.yaml"), + "--weights", + str(folder / "profile.yaml"), + "--format", + "json", + ] + env = {**os.environ, "PATH": tmp} + result = subprocess.run( + command, capture_output=True, text=True, env=env, cwd=ROOT + ) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertIn("warning [no_config]", result.stderr) + self.assertIsNone(json.loads(result.stdout)["models"][0]["score"]) + strict = subprocess.run( + command + ["--strict"], + capture_output=True, + text=True, + env=env, + cwd=ROOT, + ) + self.assertEqual(strict.returncode, 1) + self.assertEqual(strict.stdout, "") + self.assertIn("no_config", strict.stderr) + + def test_diagnostic_context_and_suppression(self): + c = fixture() + c["catalogue"]["evals"][0]["warning"] = "Review this eval" + rr = paired([row()]) + rr[0]["metric"] = "alternate" + for output in self.both([dict(config=c, rows=rr, operation="compare")])[0]: + report = output["value"] + self.assertNotIn( + "config_caveat", {d["code"] for d in report["diagnostics"]} + ) + d = next( + d for d in report["diagnostics"] if d["code"] == "missing_scoring_field" + ) + self.assertEqual( + (d["model"], d["eval"], d["tasks"], d["effect"]), + ("A", "E", ["e_en"], "excluded"), + ) + self.assertEqual( + d["measurement_ids"], [report["a"]["measurements"][0]["id"]] + ) + self.assertIsNone(report["delta"]) + c["suite"] = dict( + version=1, + name="Other set", + mode="fixed", + evals=[dict(name="E", variants=[dict(task="e_fr")])], + ) + for output in self.both([dict(config=c, rows=rr, operation="compare")])[0]: + self.assertEqual( + {d["code"] for d in output["value"]["diagnostics"]}, + {"not_used", "missing_suite_data"}, + ) + # An unused catalogue rule neither requires data nor advertises its caveat. + c = fixture() + c["catalogue"]["evals"].append( + dict( + name="Unused", + category="Other", + match={"name": "unused"}, + metric="acc", + filter="none", + score={"scale": 1}, + warning="Unused caveat", + ) + ) + for output in self.both( + [dict(config=c, rows=paired([row()]), operation="compare")] + )[0]: + self.assertFalse(output["value"]["diagnostics"]) + + def test_swaps_and_permutations_preserve_allocation(self): + c = fixture() + rr = paired([row(), row("e_fr", ".4")]) + cases = [ + dict(config=c, rows=rr, operation="compare"), + dict(config=c, rows=list(reversed(rr)), operation="compare"), + dict(config=c, rows=rr, operation="compare", a="B", b="A"), + ] + results = self.both(cases) + for engine in (0, 1): + reports = [r[engine]["value"] for r in results] + self.assertAlmostEqual(reports[0]["delta"], reports[1]["delta"]) + self.assertAlmostEqual(reports[0]["delta"], -reports[2]["delta"]) + self.close(reports[0]["a"]["tree"], reports[1]["a"]["tree"]) + self.assertEqual( + { + r["id"]: r["effective_weight"] + for r in reports[0]["a"]["measurements"] + }, + { + r["id"]: r["effective_weight"] + for r in reports[1]["a"]["measurements"] + }, + ) + + def test_yaml_scalar_contract(self): + base = {k: json.dumps(v) for k, v in fixture().items()} + cases = [] + for token in ( + "0", + "0.0", + "0_0", + "0_", + "+.0", + "0x0", + "0b0", + "0o0", + ".nan", + ".inf", + "!!float 0", + "!!bool yes", + ): + catalogue = """version: 1 +name: Scalar test +evals: + - name: E + category: C + match: {name: e_en} + metric: acc + filter: none + score: {scale: 1} + normalize: {min: TOKEN, max: 1} +languages: [] +""".replace("TOKEN", token) + cases.append(dict(yaml={**base, "catalogue": catalogue}, rows=[row()])) + self.both(cases) + + def test_regex_rejection_and_literal_escapes(self): + cases = [] + for pattern in ( + r"e_[\s]", + r"e_\_", + r"e_\-", + r"e_[]", + r"e_[^]", + r"e_}", + r"e_{x}", + r"e_(", + r"(e)_\1", + r"(?i)e_en", + r"e_.++", + ): + c = fixture() + c["catalogue"]["evals"][0]["match"] = {"regex": pattern} + cases.append(dict(config=c, rows=[row()])) + for outputs in self.both(cases): + for result in outputs: + self.assertIn("error", result) + for pattern, task in [ + (r"e_\++", "e_++"), + (r"e_[a-z]{2,3}", "e_en"), + (r"e_\(\?", "e_(?"), + (r"e_[\d]+", "e_12"), + ]: + c = fixture() + c["catalogue"]["evals"][0]["match"] = {"regex": pattern} + for result in self.both([dict(config=c, rows=[row(task)])])[0]: + self.assertTrue( + result["value"]["models"][0]["measurements"][0]["selected"] + ) + + def test_csv_file_preserves_quoted_line_endings(self): + import tempfile + from quickdash import read_results + + stream = io.StringIO(newline="") + writer = csv.DictWriter(stream, fieldnames=row()) + writer.writeheader() + writer.writerow(row(harness="first\r\nsecond")) + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "scores.csv" + path.write_bytes(stream.getvalue().encode()) + rows = read_results(path) + self.assertEqual(rows[0]["harness"], "first\r\nsecond") + self.both([dict(config=fixture(), csv=stream.getvalue())]) + + def test_category_names_do_not_inherit_object_properties(self): + cases = [] + for category in ("constructor", "toString", "__proto__", "hasOwnProperty"): + c = fixture() + c["catalogue"]["evals"][0]["category"] = category + c["profile"].update( + weights={category: 1}, english_weights={}, aggregate="english_eval" + ) + cases.append(dict(config=c, rows=paired([row()]), operation="compare")) + for outputs in self.both(cases): + for result in outputs: + self.assertEqual(result["value"]["a"]["score"], 50) + + def test_non_ascii_names_and_numeric_category_order(self): + c = fixture() + c["catalogue"]["evals"][0]["category"] = "2" + c["profile"]["weights"] = {"2": 0.5, "1": 0.5} + c["catalogue"]["evals"].append( + dict( + name="Other", + category="1", + match={"name": "other"}, + metric="acc", + filter="none", + score={"scale": 1}, + ) + ) + rr = [ + row(checkpoint="😀"), + row(checkpoint="\ue000"), + row("other", checkpoint="😀"), + ] + self.both([dict(config=c, rows=rr)]) diff --git a/tests/test_english.cjs b/tests/test_english.cjs index 84980f4..4655548 100644 --- a/tests/test_english.cjs +++ b/tests/test_english.cjs @@ -1,5 +1,5 @@ const assert=require('node:assert/strict'); -const {totals,comparisonRows,pairRows}=require('../app/app.js'); +const {totals,comparisonRows,pairRows}=require('../app/analysis.js'); const {validateConfig}=require('../app/eval_config.js'); const close=(a,b)=>assert.ok(Math.abs(a-b)<1e-10,`${a} != ${b}`); const scheme={evals:[{name:'mixed',category:'C'},{name:'english',category:'C'}],weights:{C:1},english_weights:{C:.5},languages:[]}; @@ -58,7 +58,7 @@ close(totals([kept],missingScheme,missingScheme.weights).score,80); close(totals([kept],missingScheme,missingScheme.weights).categories.find(c=>c.name==='C').weight,1); assert.equal(totals([kept],missingScheme,missingScheme.weights).categories.find(c=>c.name==='D').excluded,true); assert.equal(totals([],missingScheme,missingScheme.weights).score,null); -const {comparisonCoverage}=require('../app/app.js'); +const {comparisonCoverage}=require('../app/analysis.js'); const overlap=comparisonCoverage(rows,b.filter(r=>r.task!=='english'),scheme); assert.equal(overlap.a.length,3);assert.equal(overlap.b.length,3); assert.ok(overlap.warnings.some(w=>w.name==='english')); diff --git a/tests/test_public_browser.mjs b/tests/test_public_browser.mjs index f90fc55..6031b84 100644 --- a/tests/test_public_browser.mjs +++ b/tests/test_public_browser.mjs @@ -132,8 +132,91 @@ try{ await upload('#suiteFile',serializeSuite({version:1,name:'Freeform',mode:'available'}),'freeform.yaml'); await click('[data-view=warnings]'); assert.match(await evaluate("document.querySelector('#view').textContent"),/Config caveat.*Example math requires review/s); + // Weighted components use fictional scores and remain inspectable after exclusion. + await click('#clearModels');await click('[data-view=config]'); + const levels=['low','medium','high','top']; + const componentCatalogue={version:1,name:'Component example',evals:[{name:'Poly example',category:'Reasoning',match:{regex:'poly_.+'},metric:'acc',filter:'none',score:{scale:1},aggregation:{components:levels.map((name,i)=>({name,match:{regex:'poly_.+_'+name},relative_weight:2**i})),note:'Fictional component fixture.'}}],languages:['en','de'].map((lang,i)=>({tasks:levels.map(l=>'poly_'+lang+'_'+l),scope:'single',language:i?'deu_Latn':'eng_Latn'}))}; + await upload('#configFile',serializeCatalogue(componentCatalogue),'components.yaml'); + await upload('#weightsFile',serializeWeightProfile({version:1,name:'Component weights',weights:{Reasoning:1},english_weights:{Reasoning:.5}}),'weights.yaml'); + const componentCSV=(model,omit=false)=>['checkpoint,task,metric,filter,n_shot,harness,backend,value',...['en','de'].flatMap(lang=>levels.flatMap((l,i)=>omit&&lang==='en'&&l==='top'?[]:[`${model},poly_${lang}_${l},acc,none,0,test,cpu,${model==='Component A'?(lang==='en'?[.6,.3,.15,0][i]:.2):(lang==='en'?.3:.1)}`]))].join('\n'); + await upload('#modelFile',componentCSV('Component A'),'a.csv'); + await upload('#modelFile',componentCSV('Component B'),'b.csv'); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + assert.deepEqual(await evaluate("[...document.querySelectorAll('.score-card strong')].map(e=>e.textContent)"),['16.00','20.00','-4.00']); + for(const view of ['categories','languages']){ + await click('[data-view='+view+']'); + const parent= view==='categories'?'[data-kind=eval][data-label="Poly example"]':'[data-kind=language][data-label=eng_Latn]'; + const expected=view==='categories'?'16.00':'12.00'; + assert.equal(await evaluate(`document.querySelector(${JSON.stringify(parent)}+' > summary').children[2].textContent`),expected); + const leaf='[data-kind=variant][data-label=poly_en_low]'; + assert.deepEqual(await evaluate(`Array.from(document.querySelector(${JSON.stringify(leaf)}).children).slice(2).map(e=>e.textContent)`),['60.00','30.00','30.00','1 · 6.67%','4.000','2.000']); + await evaluate("document.querySelectorAll('.breakdown-node').forEach(e=>e.open=true)"); + const aligned=await evaluate(`{const header=[...document.querySelector('.tree-head').children].map(e=>e.getBoundingClientRect().right),row=[...document.querySelector(${JSON.stringify(leaf)}).children].map(e=>e.getBoundingClientRect().right);header.every((x,i)=>i===0||Math.abs(x-row[i])<2)}`); + assert.equal(aligned,true); + if(view==='languages'){await click('[data-language-sort=componentA]');assert.match(await evaluate("document.querySelector('#view').textContent"),/Group contribution/);} + } + await click('[data-view=config]'); + assert.match(await evaluate("document.querySelector('.component-info').textContent"),/sum\(relative_weight × normalized score\) \/ 15/); + await evaluate(`window.originalCreate=URL.createObjectURL;window.originalClick=HTMLAnchorElement.prototype.click;URL.createObjectURL=b=>{window.exportBlob=b;return 'blob:test'};HTMLAnchorElement.prototype.click=function(){};`); + await click('#exportConfig');assert.deepEqual(await evaluate('exportBlob.text().then(parseCatalogue)'),componentCatalogue); + await evaluate('URL.createObjectURL=originalCreate;HTMLAnchorElement.prototype.click=originalClick'); + await upload('#configFile',serializeCatalogue(componentCatalogue).replace('relative_weight: 1','relative_weight: 0'),'invalid-components.yaml'); + assert.match(await evaluate("document.querySelector('#error').textContent"),/positive/); + assert.equal(await evaluate("document.querySelector('.score-card strong').textContent"),'16.00'); + await upload('#modelFile',componentCSV('Component C',true),'incomplete.csv'); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + assert.deepEqual(await evaluate("[...document.querySelectorAll('.score-card strong')].map(e=>e.textContent)"),['20.00','10.00','10.00']); + await click('[data-view=warnings]');assert.match(await evaluate("document.querySelector('#view').textContent"),/Incomplete components.*top: missing/s); + await click('[data-view=config]'); + await click('[data-task=poly_en_low] > summary'); + assert.match(await evaluate("document.querySelector('[data-task=poly_en_low] .task-details').textContent"),/Excluded from comparison/); + const componentSet={version:1,name:'All component tasks',mode:'fixed',evals:[{name:'Poly example',variants:componentCatalogue.languages.flatMap(g=>g.tasks.map(task=>({task,n_shot:0})))}]}; + const cardsBeforeRejectedConfig=await evaluate("document.querySelector('#cards').textContent"); + for(const modify of [s=>s.evals[0].variants.pop(),s=>s.evals[0].variants[0].n_shot=5,s=>delete s.evals[0].variants[0].n_shot]){ + const bad=structuredClone(componentSet);modify(bad); + await upload('#suiteFile',serializeSuite(bad),'incompatible-set.yaml'); + assert.match(await evaluate("document.querySelector('#error').textContent"),/Incompatible aggregation config/); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),cardsBeforeRejectedConfig); + } + const incompatibleCatalogue=structuredClone(componentCatalogue);incompatibleCatalogue.evals[0].select={regex:'poly_.+_(low|medium|high)'}; + await upload('#configFile',JSON.stringify(incompatibleCatalogue),'incompatible-catalogue.yaml'); + assert.match(await evaluate("document.querySelector('#error').textContent"),/Incompatible aggregation config.*top/); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),cardsBeforeRejectedConfig); + await upload('#suiteFile',serializeSuite(componentSet),'components-set.yaml'); + assert.match(await evaluate("document.querySelector('#coverage').textContent"),/INCOMPLETE.*4\/8 requirements shared/); + await click('[data-view=comparisons]');await change('#compareGroup','variant');await change('#compareMeasure','weighted'); + assert.match(await evaluate("document.querySelector('#view').textContent"),/10.0000 index points/); + // Editable weights resolve a category missing from the loaded profile. + const beforeCategoryEdit=await evaluate("document.querySelector('#cards').textContent"); + const customCategory=structuredClone(componentCatalogue);customCategory.evals[0].category='Custom'; + await click('[data-view=config]');await upload('#configFile',serializeCatalogue(customCategory),'custom-category.yaml'); + await click('[data-view=warnings]');assert.match(await evaluate("document.querySelector('#view').textContent"),/No category weight/); + await click('[data-view=score]'); + await evaluate(`{for(const input of document.querySelectorAll('[data-weight]')){input.value=input.dataset.weight==='Custom'?'1':'0';document.querySelector('#view').onchange({target:input});}}`); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),beforeCategoryEdit); + await click('[data-view=warnings]');assert.doesNotMatch(await evaluate("document.querySelector('#view').textContent"),/No category weight/); + // Exercise the actual Pages fallback with the full public sample and every synthetic choice. + const sample=path.join(temporary,'sample'),resultsDir=path.join(temporary,'empty-results');fs.mkdirSync(resultsDir); + execFileSync('python3',['-m','app.build','--results-dir',resultsDir,'--sample-csv','examples/sample-evals.csv','--output',sample],{cwd:root,stdio:'pipe'}); + await navigate(pathToFileURL(path.join(sample,'index.html')).href); + assert.match(await evaluate("document.querySelector('#modelA').value"),/^SAMPLE/); + assert.equal(await evaluate("document.querySelector('#cards').hidden"),false); + const syntheticNames=await evaluate("syntheticOptions.map(o=>o.name)"); + const scores=[]; + for(const name of syntheticNames){ + await change('#modelB',name); + assert.match(await evaluate("document.querySelector('#demo').textContent"),/synthetic scores/); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + scores.push(await evaluate("document.querySelector('#cards').textContent")); + for(const view of ['categories','languages','comparisons','config','warnings','score'])await click('[data-view='+view+']'); + } + assert.equal(new Set(scores).size,syntheticNames.length); + await change('#modelB',await evaluate("document.querySelector('#modelA').value")); + assert.match(await evaluate("document.querySelector('#demo').textContent"),/Sample dataset/); + await click('#clearModels'); + assert.equal(await evaluate("document.querySelector('#modelA').options.length"),0); assert.deepEqual(errors,[]);assert.deepEqual(network,[],'Loading and comparing local files must not send HTTP requests'); - console.log('Public browser checks passed: empty start, shared models, independent weights and eval sets, required coverage, temporary uploads, rollback, clear models and no uploads.'); + console.log('Public browser checks passed: empty start, shared models, independent weights and eval sets, required coverage, temporary uploads, rollback, Pages sample, synthetic choices, clear models and no uploads.'); }finally{ ws?.close();fs.rmSync(temporary,{recursive:true,force:true}); } diff --git a/tests/test_suites.cjs b/tests/test_suites.cjs index 7937c89..403c071 100644 --- a/tests/test_suites.cjs +++ b/tests/test_suites.cjs @@ -46,7 +46,7 @@ test('a fixed set is incomplete when required measurements have incompatible pro assert.equal(c.presentA,1);assert.equal(c.presentB,1);assert.equal(c.sharedRequired,0);assert.equal(c.complete,false); }); test('fixed-set scores use the shared subset and contributions reconcile in every aggregate',()=>{ - const {auditRows}=require('../app/eval_config.js'),{comparisonCoverage,totals,comparisonRows}=require('../app/app.js'); + const {auditRows}=require('../app/eval_config.js'),{comparisonCoverage,totals,comparisonRows}=require('../app/analysis.js'); const c=catalogue(),s=suite(),p={...profile(),english_weights:{C:.5,D:.5}},config=resolveConfig(c,s,p); const measurement=(task,shot,value,checkpoint='A')=>({checkpoint,task,n_shot:String(shot),value:String(value),metric:'acc',filter:'none',harness:'test',backend:'cpu'}); const a=auditRows([measurement('e_en',0,.8),measurement('e_fr',5,.6),measurement('e_de',0,1)],c); @@ -65,11 +65,11 @@ test('fixed-set scores use the shared subset and contributions reconcile in ever } }); test('an alternate metric cannot satisfy a required measurement',()=>{ - const {auditRows}=require('../app/eval_config.js'),{collectWarnings}=require('../app/app.js'); + const {auditRows}=require('../app/eval_config.js'),{diagnosticsFor}=require('./diagnostic_fixture.cjs'); const rows=auditRows([{checkpoint:'A',task:'e_en',n_shot:'0',value:'.8',metric:'acc_norm',filter:'none',harness:'test',backend:'cpu'}],catalogue()); assert.equal(scopeRows(rows,suite()).missing.length,2); assert.equal(scopeRows(rows,suite()).extras.length,0); - assert.ok(collectWarnings(new Map([['A',rows]]),catalogue()).some(w=>w.type==='Missing scoring field')); + assert.ok(diagnosticsFor(new Map([['A',rows]]),catalogue()).some(w=>w.type==='Missing scoring field')); }); test('all shipped sets and profiles are independent and resolve against the global catalogue',()=>{ const fs=require('node:fs'),{parseCatalogue}=require('../app/eval_config.js'); diff --git a/tests/test_warning_policy.cjs b/tests/test_warning_policy.cjs index bd5ccfe..3927b55 100644 --- a/tests/test_warning_policy.cjs +++ b/tests/test_warning_policy.cjs @@ -1,7 +1,7 @@ 'use strict'; const test=require('node:test'),assert=require('node:assert/strict'); const {auditRows}=require('../app/eval_config.js'); -const {collectWarnings,comparisonCoverage,totals}=require('../app/app.js'); +const {compare,comparisonCoverage,totals}=require('../app/analysis.js'); const {resolveConfig,suiteCoverage}=require('../app/suite_config.js'); const catalogue=()=>({version:1,name:'Rules',evals:['E','F'].map(name=>({name,category:'C',match:{name:name.toLowerCase()},metric:'acc_norm',filter:'none',score:{scale:1},warning:name+' caveat'})),languages:[{tasks:['e','f'],scope:'single',language:'eng_Latn'}]}); const profile={version:1,name:'Weights',weights:{C:1}}; @@ -11,7 +11,7 @@ const row=(task='e',metric='acc_norm')=>({checkpoint:'A',task,metric,filter:'non function run(left,right,suite=available){ const c=catalogue(),a=auditRows(left,c),b=auditRows(right.map(r=>({...r,checkpoint:'B'})),c),config=resolveConfig(c,suite,profile); const scope=suiteCoverage(a,b,suite),coverage=comparisonCoverage(scope.a,scope.b,config); - const warnings=collectWarnings(new Map([['A',a],['B',b]]),c,'standard',{},suite,coverage.a).concat(scope.warnings,coverage.warnings); + const warnings=compare([...a,...b],{catalogue:c,suite,profile},'A','B').diagnostics; return {warnings,scope,coverage,score:totals(coverage.a,config,config.weights).score}; } test('displayed evals emit their caveats; unused catalogue rules and excluded evals do not',()=>{ @@ -24,7 +24,7 @@ test('displayed evals emit their caveats; unused catalogue rules and excluded ev test('unselected eval data warns even if it only contains the wrong metric',()=>{ const r=run([row(),row('f','acc')],[row()],{...fixed,evals:[{name:'E'}]}); assert.ok(r.warnings.some(w=>w.type==='Not used'&&w.name==='F'&&w.model==='A')); - assert.ok(!r.warnings.some(w=>w.type==='Missing scoring field'&&w.name==='f')); + assert.ok(!r.warnings.some(w=>w.type==='Missing scoring field'&&w.tasks.includes('f'))); assert.equal(r.coverage.pairs.length,1);assert.equal(r.score,80); }); test('required eval absent from one or both models warns and is excluded from both scores',()=>{ @@ -35,7 +35,7 @@ test('required eval absent from one or both models warns and is excluded from bo }); test('missing configured metric warns, excludes, and never substitutes an alternate metric',()=>{ const r=run([row(),row('f','acc')],[row(),row('f')],fixed); - assert.ok(r.warnings.some(w=>w.type==='Missing scoring field'&&w.name==='f'&&w.model==='A')); + assert.ok(r.warnings.some(w=>w.type==='Missing scoring field'&&w.tasks.includes('f')&&w.model==='A')); assert.ok(r.warnings.some(w=>w.type==='Missing suite data'&&w.name==='F')); assert.equal(r.coverage.pairs.length,1);assert.equal(r.score,80); }); diff --git a/tests/test_yaml.cjs b/tests/test_yaml.cjs index 11170df..7841c3b 100644 --- a/tests/test_yaml.cjs +++ b/tests/test_yaml.cjs @@ -1,25 +1,9 @@ const assert=require('node:assert/strict'),fs=require('node:fs'); const {parseCatalogue,serializeCatalogue,normalizeScore,auditRows}=require('../app/eval_config.js'); const config=parseCatalogue(fs.readFileSync('configs/catalogue.yaml','utf8')); -assert.equal(config.evals.length,46); assert.deepEqual(parseCatalogue(serializeCatalogue(config)),config); -// The published OELLM config selects reliable SIB-200 accuracy and the paper's -// approximate JEEBench baseline; neither is an unresolved config caveat. -const sib=config.evals.find(e=>e.name==='SIB-200'),jee=config.evals.find(e=>e.name==='JEEBench'); -assert.equal(sib.metric,'acc');assert.equal(sib.warning,undefined); -assert.equal(jee.normalize.min,.1055);assert.equal(jee.normalize.max,1); -assert.equal(jee.normalize.clip,true);assert.notEqual(jee.normalize.basis,'unresolved'); -assert.equal(jee.warning,undefined); -for(const [value,expected] of [[0,0],[.105,0],[.1055,0],[.55275,50],[1,100]]){ - const score=normalizeScore(value,jee); - assert.ok(Math.abs(score.score_100-expected)<1e-10); - assert.ok(Math.abs(score.raw_score_100-value*100)<1e-10); -} -const sibRows=auditRows(['acc','acc_norm'].map(metric=>({checkpoint:'Fixture',task:'sib200_eng_Latn',metric,filter:'none',n_shot:'0',harness:'test',backend:'cpu',value:'.5'})),config); -assert.deepEqual(sibRows.filter(r=>r.selected).map(r=>r.metric),['acc']); -const {collectWarnings}=require('../app/app.js'); -assert.deepEqual(collectWarnings(new Map(),config),[]); -assert.deepEqual(config.evals.filter(e=>e.warning).map(e=>e.name).sort(),['Global PIQA (prompted)','MultiBlimp']); +const {diagnosticsFor}=require('./diagnostic_fixture.cjs'); +assert.deepEqual(diagnosticsFor(new Map(),config),[]); const source=`# A four-choice example with quoted regex and a flow mapping version: 1 name: Example @@ -47,14 +31,3 @@ assert.equal(example.evals[0].filter,''); assert.equal(example.notes[0],'A folded explanation for people editing the config.'); for(const bad of [source+'version: 1\n',source+'---\nversion: 1',source.replace('min: 0.25','min: .nan'),source.replace('filter: \'\'','filter: [oops'),source.replace('name: Example\n','name: !!js/function function(){}\n')])assert.throws(()=>parseCatalogue(bad)); console.log('YAML checks passed: source config, round-trip, comments, quoted regex, flow syntax, multiline notes, invalid syntax and duplicate keys.'); - -assert.equal(config.evals.find(e=>e.name==='AMC23').normalize.min,0); -assert.equal(config.evals.find(e=>e.name==='ARC Easy').normalize.min,.25); -assert.equal(config.evals.find(e=>e.name==='FLORES200').metric,'chrf++'); -assert.equal(config.evals.find(e=>e.name==='OpenSubtitles').metric,'chrf'); -for(const e of config.evals.filter(e=>e.category==='Translation')){assert.equal(e.warning,undefined);assert.equal(e.normalize.min,0);assert.equal(e.score.scale,100);} -const {parseSuite,inSuite}=require('../app/suite_config.js'); -const freeform=parseSuite(fs.readFileSync('configs/sets/any-available.yaml','utf8')); -const prompted=config.evals.find(e=>e.name==='Global PIQA (prompted)'); -assert.match(prompted.warning,/Normalization and metric selection have not been validated/); -assert.equal(inSuite({eval:prompted.name},freeform),false);