From 837ffe3832ba3ce36a7032bded1be6155dea0f38 Mon Sep 17 00:00:00 2001 From: Jonathan Burdge Date: Fri, 2 Oct 2026 09:44:04 +0300 Subject: [PATCH 1/4] Keep each eval's interpretation and language mappings in one YAML file --- .github/workflows/pages.yml | 6 + README.md | 2 +- app/app.js | 2 +- app/build.py | 6 +- app/catalogue_io.cjs | 22 + app/eval_config.js | 19 +- configs/README.md | 10 +- configs/catalogue.yaml | 4701 +---------------------- configs/evals/aime24.yaml | 24 + configs/evals/aime25.yaml | 24 + configs/evals/amc23.yaml | 23 + configs/evals/arc_challenge.yaml | 138 + configs/evals/arc_easy.yaml | 25 + configs/evals/belebele.yaml | 137 + configs/evals/boolq.yaml | 22 + configs/evals/commonsenseqa.yaml | 22 + configs/evals/copa.yaml | 22 + configs/evals/coqa.yaml | 22 + configs/evals/cs_algorithms.yaml | 22 + configs/evals/dyck_languages.yaml | 23 + configs/evals/flores200.yaml | 510 +++ configs/evals/global_mmlu.yaml | 404 ++ configs/evals/global_piqa_prompted.yaml | 172 + configs/evals/gpqa_diamond.yaml | 22 + configs/evals/gsm8k.yaml | 23 + configs/evals/hellaswag.yaml | 105 + configs/evals/humaneval.yaml | 23 + configs/evals/ifeval.yaml | 22 + configs/evals/include.yaml | 149 + configs/evals/jeebench.yaml | 26 + configs/evals/jeopardy.yaml | 22 + configs/evals/lambada.yaml | 23 + configs/evals/language_id.yaml | 24 + configs/evals/livecodebench.yaml | 22 + configs/evals/lsat_ar.yaml | 24 + configs/evals/math_500.yaml | 22 + configs/evals/mbpp.yaml | 22 + configs/evals/mgsm.yaml | 92 + configs/evals/mmlu.yaml | 37 + configs/evals/multiblimp.yaml | 184 + configs/evals/openbookqa.yaml | 22 + configs/evals/opensubtitles.yaml | 421 ++ configs/evals/operators.yaml | 22 + configs/evals/piqa.yaml | 179 + configs/evals/polymath.yaml | 66 + configs/evals/qa_wikidata.yaml | 22 + configs/evals/repeat_copy_logic.yaml | 22 + configs/evals/sib_200.yaml | 200 + configs/evals/social_iqa.yaml | 22 + configs/evals/squad_v2.yaml | 23 + configs/evals/winogrande.yaml | 22 + configs/evals/wsc273.yaml | 22 + configs/evals/x_csqa.yaml | 58 + configs/evals/xcopa.yaml | 32 + docs/configuration.md | 86 +- docs/development.md | 6 +- docs/python-api.md | 20 + quickdash/cli.py | 2 +- quickdash/config.py | 72 +- tests/engine_adapter.cjs | 1 + tests/test_data.py | 68 +- tests/test_engines.py | 72 + tests/test_public_browser.mjs | 13 + tests/test_suites.cjs | 2 +- tests/test_yaml.cjs | 2 +- 65 files changed, 3934 insertions(+), 4743 deletions(-) create mode 100644 app/catalogue_io.cjs create mode 100644 configs/evals/aime24.yaml create mode 100644 configs/evals/aime25.yaml create mode 100644 configs/evals/amc23.yaml create mode 100644 configs/evals/arc_challenge.yaml create mode 100644 configs/evals/arc_easy.yaml create mode 100644 configs/evals/belebele.yaml create mode 100644 configs/evals/boolq.yaml create mode 100644 configs/evals/commonsenseqa.yaml create mode 100644 configs/evals/copa.yaml create mode 100644 configs/evals/coqa.yaml create mode 100644 configs/evals/cs_algorithms.yaml create mode 100644 configs/evals/dyck_languages.yaml create mode 100644 configs/evals/flores200.yaml create mode 100644 configs/evals/global_mmlu.yaml create mode 100644 configs/evals/global_piqa_prompted.yaml create mode 100644 configs/evals/gpqa_diamond.yaml create mode 100644 configs/evals/gsm8k.yaml create mode 100644 configs/evals/hellaswag.yaml create mode 100644 configs/evals/humaneval.yaml create mode 100644 configs/evals/ifeval.yaml create mode 100644 configs/evals/include.yaml create mode 100644 configs/evals/jeebench.yaml create mode 100644 configs/evals/jeopardy.yaml create mode 100644 configs/evals/lambada.yaml create mode 100644 configs/evals/language_id.yaml create mode 100644 configs/evals/livecodebench.yaml create mode 100644 configs/evals/lsat_ar.yaml create mode 100644 configs/evals/math_500.yaml create mode 100644 configs/evals/mbpp.yaml create mode 100644 configs/evals/mgsm.yaml create mode 100644 configs/evals/mmlu.yaml create mode 100644 configs/evals/multiblimp.yaml create mode 100644 configs/evals/openbookqa.yaml create mode 100644 configs/evals/opensubtitles.yaml create mode 100644 configs/evals/operators.yaml create mode 100644 configs/evals/piqa.yaml create mode 100644 configs/evals/polymath.yaml create mode 100644 configs/evals/qa_wikidata.yaml create mode 100644 configs/evals/repeat_copy_logic.yaml create mode 100644 configs/evals/sib_200.yaml create mode 100644 configs/evals/social_iqa.yaml create mode 100644 configs/evals/squad_v2.yaml create mode 100644 configs/evals/winogrande.yaml create mode 100644 configs/evals/wsc273.yaml create mode 100644 configs/evals/x_csqa.yaml create mode 100644 configs/evals/xcopa.yaml diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index 010e6dc..67af5df 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -42,6 +42,12 @@ jobs: --catalogue "$GITHUB_WORKSPACE/configs/examples/catalogue.yaml" \ --weights "$GITHUB_WORKSPACE/configs/examples/weights.yaml" \ --compare 'Example A' 'Example B' --format json > quickdash-installed.json + PATH="$RUNNER_TEMP/quickdash-package/bin" \ + "$RUNNER_TEMP/quickdash-package/bin/quickdash" "$GITHUB_WORKSPACE/examples/sample-evals.csv" \ + --catalogue "$GITHUB_WORKSPACE/configs/catalogue.yaml" \ + --weights "$GITHUB_WORKSPACE/configs/weights/oellm.yaml" \ + --eval-set "$GITHUB_WORKSPACE/configs/sets/any-available.yaml" \ + --format json > quickdash-installed-sample.json - name: Check the browser with public fixtures run: | google-chrome --headless --no-sandbox --disable-gpu \ diff --git a/README.md b/README.md index 2ae2912..caf80a2 100644 --- a/README.md +++ b/README.md @@ -19,7 +19,7 @@ Files opened here stay in your browser; they are not uploaded. Changes last unti ## Share results and scoring configs - Add public CSV exports to [results/](results/README.md) to offer their models in the shared dashboard. -- Add interpretation rules to [configs/catalogue.yaml](configs/catalogue.yaml). +- Edit or add a self-contained eval file in [configs/evals/](configs/evals/), such as [polymath.yaml](configs/evals/polymath.yaml). Each file holds its scoring rules and language assignments together; the catalogue combines them automatically. - Add weighting profiles to [configs/weights/](configs/weights/), or optional named eval sets to [configs/sets/](configs/sets/). Each directory has a `default.txt` choosing its startup selection. See [contributing configs](configs/README.md). Use a pull request or GitHub’s **Add file → Upload files**. Changes on `main` trigger tests and a GitHub Pages rebuild; pull requests are checked without publishing. Invalid inputs stop the update and leave the last successful site online. The repository and dashboard are public, so use browser imports for private comparisons. diff --git a/app/app.js b/app/app.js index 3f02072..4ac1ba1 100644 --- a/app/app.js +++ b/app/app.js @@ -167,7 +167,7 @@ function start(){ html+=groups.map(f=>{const allRows=all.filter(r=>r.eval===f.name),missing=warnings.filter(w=>w.eval===f.name&&['Missing scoring field','Missing scoring setting'].includes(w.type)),inconsistent=warnings.filter(w=>w.eval===f.name&&w.type==='Inconsistent scoring settings');catalogueGroups.set(f.name,{f,missing});const langs=[...new Set(f.tasks.flatMap(t=>languages(t.rows[0]).map(languageLabel)))].sort();return '
'+esc(f.name)+' ('+f.tasks.length+')'+esc(f.category)+''+scoringOptions(f,allRows)+(missing.length?''+missing.length+' missing scoring field/settings':'')+(inconsistent.length?'Inconsistent scoring settings':'')+''+esc(langs.length>4?langs.slice(0,3).join(', ')+' + '+(langs.length-3)+' more':langs.join(', '))+''+esc(normalizationLabel(f))+(f.warning?'Config warning':'')+''+selectionInfo(f)+normalizationInfo(f)+aggregationInfo(f)+inconsistent.map(w=>'

'+esc(w.model+': '+w.detail)+'

').join('')+'
'+(groups.length===1?catalogueTasks(f,missing):'')+'
'+'
';}).join(''); if(!groups.length)html+='

No evals match the filters.

'; html+='
Scoring assumptions and source files
    '+(scheme.notes||[]).map(n=>'
  1. '+esc(n)+'
  2. ').join('')+'

Source row audit · Language assignments · Analysis JSON

'+esc(DATA.source)+' · SHA-256 '+esc(DATA.sha256)+'

'; - const configControls='
Global eval catalogue

The catalogue defines matching, categories, scoring fields, normalization, and language assignments for every known eval. It does not require models to run all those evals.

Normalization: each eval lists its baseline, formula, and sources below. Chance correction and component aggregation affect calculated scores; individual raw scores stay unchanged. acc_norm is length-normalized answer scoring, not chance correction.

Active catalogue: '+esc(catalogue.name)+' · '+catalogue.languages.reduce((n,g)=>n+g.tasks.length,0)+' explicit task language assignments.

Weighting profile and optional eval set

Weights apply independently of the eval set. Any available compares shared recognized measurements and warns on differences. A named set also warns about missing requirements and excludes extra measurements. An incomplete named-set score uses the shared subset with redistributed weights.

Active eval set: '+esc(suite.name)+(suite.exclude?.length?' · Explicitly excluded: '+suite.exclude.map(esc).join(', '):'')+'.

All imports are temporary and stay in your browser. Exports save the three inputs separately. Weight exports include your edits and active score calculation.

'; + const configControls='
Global eval catalogue

The catalogue defines matching, categories, scoring fields, normalization, and language assignments for every known eval. It does not require models to run all those evals.

Normalization: each eval lists its baseline, formula, and sources below. Chance correction and component aggregation affect calculated scores; individual raw scores stay unchanged. acc_norm is length-normalized answer scoring, not chance correction.

Catalogue export includes every eval and language in one portable YAML file.

Active catalogue: '+esc(catalogue.name)+' · '+catalogue.languages.reduce((n,g)=>n+g.tasks.length,0)+' explicit task language assignments.

Weighting profile and optional eval set

Weights apply independently of the eval set. Any available compares shared recognized measurements and warns on differences. A named set also warns about missing requirements and excludes extra measurements. An incomplete named-set score uses the shared subset with redistributed weights.

Active eval set: '+esc(suite.name)+(suite.exclude?.length?' · Explicitly excluded: '+suite.exclude.map(esc).join(', '):'')+'.

All imports are temporary and stay in your browser. Exports save the three inputs separately. Weight exports include your edits and active score calculation.

'; html=html.replace('
',configControls+'
'); return html; diff --git a/app/build.py b/app/build.py index 442f5ef..60abbb9 100644 --- a/app/build.py +++ b/app/build.py @@ -5,7 +5,7 @@ import json from pathlib import Path from quickdash.io import load_csv -from quickdash.config import classify, task_language, load_catalogue, load_profile, load_suite, resolve_config, scope_rows +from quickdash.config import classify, task_language, load_catalogue, serialize_catalogue, load_profile, load_suite, resolve_config, scope_rows from quickdash.analysis import totals, component_coverage, diagnostic, analyze APP = Path(__file__).resolve().parent @@ -118,7 +118,7 @@ def build(source, output, catalogue_path=None, results_dir=None, *, weights_path profile=profile, profile_file=weights_path.name, suites=suites, profiles=profiles, sample_models=sorted(owners) if using_sample else [], metadata=metadata, scheme=config, models=summary, aggregates=aggregates, rows=audit, sources=sources, source=source.name if source else results_dir.name if results_dir else '', sha256=sources[0]['sha256'] if len(sources)==1 else None) - (output/'catalogue.yaml').write_text(catalogue_path.read_text()) + (output/'catalogue.yaml').write_text(serialize_catalogue(catalogue)) (output/'weights.yaml').write_text(weights_path.read_text()) (output/'eval-set.yaml').write_text(suite_path.read_text()) (output/'analysis.json').write_text(json.dumps(payload, indent=2)) @@ -132,7 +132,7 @@ def build(source, output, catalogue_path=None, results_dir=None, *, weights_path parser.add_argument('csv', type=Path, nargs='?', help='CSV to embed; omit to start without results') parser.add_argument('--results-dir', type=Path, help='Embed all CSV files directly inside this directory') parser.add_argument('--sample-csv', type=Path, help='Fallback CSV when --results-dir contains no CSVs') - parser.add_argument('--catalogue', type=Path, help='Global eval interpretation YAML; default: configs/catalogue.yaml') + parser.add_argument('--catalogue', type=Path, help='Catalogue YAML or per-eval manifest; default: configs/catalogue.yaml') parser.add_argument('--weights', type=Path, help='Default weighting profile YAML; used alone, embed only this profile') parser.add_argument('--weights-dir', type=Path, help='Offer weighting profiles from this directory (default: configs/weights)') parser.add_argument('--eval-set', type=Path, help='Default named eval set YAML; used alone, embed only this set') diff --git a/app/catalogue_io.cjs b/app/catalogue_io.cjs new file mode 100644 index 0000000..bb50b81 --- /dev/null +++ b/app/catalogue_io.cjs @@ -0,0 +1,22 @@ +// Filesystem loading for Node consumers; the browser receives the assembled catalogue. +const fs=require('node:fs'),path=require('node:path'); +const yaml=require('./vendor/js-yaml.js'); +const {parseCatalogue,assembleCatalogue}=require('./eval_config.js'); +function loadCatalogue(filename){ + try{ + const text=fs.readFileSync(filename,'utf8'),config=yaml.load(text,{schema:yaml.CORE_SCHEMA}); + if(!config||typeof config!=='object'||!Object.hasOwn(config,'evals_dir'))return parseCatalogue(text); + if(Object.keys(config).some(k=>!['version','name','notes','evals_dir'].includes(k)))throw Error('Invalid catalogue manifest fields'); + const {evals_dir,...metadata}=config; + if(typeof evals_dir!=='string'||!evals_dir.trim())throw Error('evals_dir must be a nonempty directory path'); + const directory=path.resolve(path.dirname(filename),evals_dir); + const files=fs.readdirSync(directory).filter(name=>/\.ya?ml$/i.test(name)&&fs.statSync(path.join(directory,name)).isFile()).sort((a,b)=>Buffer.compare(Buffer.from(a),Buffer.from(b))); + const definitions=files.map(name=>{ + const source=path.join(directory,name); + try{const definition=yaml.load(fs.readFileSync(source,'utf8'),{schema:yaml.CORE_SCHEMA});assembleCatalogue(metadata,[definition]);return definition;} + catch(error){throw Error(source+': '+error.message);} + }); + return assembleCatalogue(metadata,definitions); + }catch(error){throw Error(filename+': '+error.message);} +} +module.exports={loadCatalogue}; diff --git a/app/eval_config.js b/app/eval_config.js index 52fd09b..3dd2c90 100644 --- a/app/eval_config.js +++ b/app/eval_config.js @@ -65,6 +65,21 @@ const EvalConfig=(()=>{ if('aggregate'in config&&!['standard','english_eval','english_category'].includes(config.aggregate))throw Error('Aggregate must be standard, english_eval, or english_category'); if('english_weights'in config){objectKeys(config.english_weights,Object.keys(w));if(Object.values(config.english_weights).some(v=>!number(v)||v<0||v>1))throw Error('English weights must be between 0 and 1');} } + function assembleCatalogue(metadata,definitions){ + objectKeys(metadata,['version','name','notes'],['version','name']); + if(!Array.isArray(definitions)||!definitions.length)throw Error('At least one eval definition is required'); + const catalogue={...structuredClone(metadata),evals:[],languages:[]}; + for(const definition of definitions){ + if(!definition||typeof definition!=='object'||Array.isArray(definition)||!Object.hasOwn(definition,'languages'))throw Error('Each eval definition needs its own languages list'); + const {languages,...e}=structuredClone(definition); + validateCatalogue({...metadata,evals:[e],languages}); + for(const group of languages)for(const task of group.tasks)if(!matchTask(e.match,task))throw Error(`Language task ${task} does not belong to eval ${e.name}`); + catalogue.evals.push(e);catalogue.languages.push(...languages); + } + validateCatalogue(catalogue); + for(const group of catalogue.languages)for(const task of group.tasks)if(catalogue.evals.filter(e=>matchTask(e.match,task)).length!==1)throw Error('Ambiguous eval config for task: '+task); + return catalogue; + } function validateCatalogue(config){ objectKeys(config,['version','name','evals','languages','notes'],['version','name','evals','languages']); return validateRules(config); @@ -76,7 +91,7 @@ const EvalConfig=(()=>{ for(const e of config.evals)if(!Object.hasOwn(config.weights,e.category))throw Error('Eval category has no weight: '+e.category); return config; } - function parseCatalogue(source){return validateCatalogue(yaml.load(source,{schema:yaml.CORE_SCHEMA}));} + function parseCatalogue(source){const config=yaml.load(source,{schema:yaml.CORE_SCHEMA});if(config&&Object.hasOwn(config,'evals_dir'))throw Error('This catalogue manifest needs files on disk. Build it first, then import the generated catalogue.yaml or export the complete catalogue from a dashboard.');return validateCatalogue(config);} function serializeCatalogue(config){return yaml.dump(validateCatalogue(config),{schema:yaml.CORE_SCHEMA,lineWidth:110,noRefs:true});} function validateRules(config){ if(config.version!==1)throw Error('Unsupported config version'); @@ -172,6 +187,6 @@ const EvalConfig=(()=>{ return {...r,eval:e.name,category:e.category,selected,decision,...scores}; }); } - return {validateAggregationConfig,validateAggregationSelection,parseCatalogue,serializeCatalogue,validateCatalogue,validateWeights,parseCSV,validateConfig,matchTask,normalizeScore,taskLanguage,auditRows,demoModel,isDemoModel}; + return {assembleCatalogue,validateAggregationConfig,validateAggregationSelection,parseCatalogue,serializeCatalogue,validateCatalogue,validateWeights,parseCSV,validateConfig,matchTask,normalizeScore,taskLanguage,auditRows,demoModel,isDemoModel}; })(); if(typeof module!=='undefined')module.exports=EvalConfig; diff --git a/configs/README.md b/configs/README.md index c10f405..b57da44 100644 --- a/configs/README.md +++ b/configs/README.md @@ -2,11 +2,13 @@ Choose the file to edit based on what you want to change: -- **Interpret a new eval:** add its task matching, category, scoring metric, normalization, and explicit language assignments to [catalogue.yaml](catalogue.yaml). Adding a rule does not require every model to run it. -- **Combine component results:** add an `aggregation` rule to the eval in the catalogue. Each component stores a `relative_weight`; PolyMath uses 1, 2, 4, and 8, divided by their sum when scoring. Named sets must select complete component groups with compatible shot settings; incompatible configurations are errors. Missing results within a valid selection warn and exclude the group; see [component aggregation](../docs/configuration.md#weighted-components-within-an-eval). +- **Interpret a new eval:** add one YAML file to [evals/](evals/), containing its task matching, category, scoring metric, normalization, and language assignments. [polymath.yaml](evals/polymath.yaml) shows an eval with weighted components; [boolq.yaml](evals/boolq.yaml) is a simpler example. Adding a rule does not require every model to run it. +- **Combine component results:** add an `aggregation` rule in that eval’s file. Each component stores a `relative_weight`; PolyMath uses 1, 2, 4, and 8, divided by their sum when scoring. Named sets must select complete component groups with compatible shot settings; incompatible configurations are errors. Missing results within a valid selection warn and exclude the group; see [component aggregation](../docs/configuration.md#weighted-components-within-an-eval). - **Try different weighting:** add a YAML profile to [weights/](weights/). It can be used with any eval set. Category weights, English shares, and the default calculation live here. - **Require a standard comparison set:** add a YAML file to [sets/](sets/). [flagship-1.yaml](sets/flagship-1.yaml) pins expected tasks and shot counts; [any-available.yaml](sets/any-available.yaml) needs no required-eval list and compares shared data, with an explicit exclusion for unvalidated prompted Global PIQA. +[catalogue.yaml](catalogue.yaml) is a small manifest pointing to `evals/`. Every `.yaml` or `.yml` file directly in that directory is loaded in filename order; adding a file needs no registration elsewhere. Keep language tasks in the file for the eval they match. The loader rejects misplaced tasks, duplicate names or assignments, and incompatible component configurations before changing the dashboard. An empty `languages: []` is allowed for an eval without known language metadata; unknown-language warnings still apply to its results. + Each profile or set needs a distinct `name` within its directory. To change a selector's startup choice, edit that directory's `default.txt` to name one YAML file. The catalogue is selected at build time with `--catalogue`, or temporarily loaded in the browser. The [configuration reference](../docs/configuration.md) describes all three formats with small examples. [examples/](examples/) contains the fictional catalogue and weights used by [examples/scores.csv](../examples/scores.csv). @@ -14,9 +16,11 @@ The [configuration reference](../docs/configuration.md) describes all three form Submit changes as a PR, then check the generated dashboard: ```sh -python3 -m app.build --results-dir results --output output/shared +python3 -m app.build --results-dir results --sample-csv examples/sample-evals.csv --output output/shared ``` The build validates every offered combination before replacing output. In the dashboard, inspect **Warnings** for the comparison you intend to use. Named-set scores with missing requirements are explicitly incomplete; extras are excluded but remain inspectable. +The generated `output/shared/catalogue.yaml` contains the complete catalogue, including every language assignment, with no file references. Browser catalogue export produces the same portable format. Import that complete file when moving settings between dashboards; the repository manifest alone is not a browser import. + For temporary changes, use the separate load/export controls under **Eval configuration**. Files stay in the browser. Weight exports save edited weights and the active calculation; catalogue and eval-set exports are independent. None of these controls modify the repository. diff --git a/configs/catalogue.yaml b/configs/catalogue.yaml index 0a98cd5..1343366 100644 --- a/configs/catalogue.yaml +++ b/configs/catalogue.yaml @@ -1,4700 +1,5 @@ +# Each file in evals/ contains one eval and its language assignments. +# Builds and the Python API assemble these into one portable catalogue. version: 1 name: OELLM eval catalogue -evals: - - category: Code - name: CS Algorithms - metric: exact_match - filter: strict-match - match: - name: bigbench_cs_algorithms_generate_until - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: Free-response exact match over algorithm outputs; no defined uniform answer space. - sources: - - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/cs_algorithms - - category: Code - name: HumanEval - metric: python_pass@1 - filter: '' - match: - name: HumanEval - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: Executable code pass@1 has no benchmark-defined uniform random-program baseline. - sources: - - https://github.com/openai/human-eval - - category: Code - name: LiveCodeBench - metric: accuracy_avg - filter: '' - match: - name: LiveCodeBench - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: Generated code is graded by tests; there is no finite choice set to support a 1/K correction. - sources: - - https://github.com/LiveCodeBench/LiveCodeBench - - category: Code - name: MBPP - metric: pass_at_1 - filter: none - match: - name: mbpp - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: Generated code is graded by tests; there is no benchmark-defined random-program baseline. - sources: - - https://github.com/google-research/google-research/tree/master/mbpp - - category: Math - name: Dyck languages - metric: exact_match - filter: strict-match - match: - name: bigbench_dyck_languages_generate_until - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - Exact-match generated bracket completions have length-dependent spaces; no single fixed baseline is - assigned. - sources: - - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/dyck_languages - - category: Math - name: Operators - metric: exact_match - filter: strict-match - match: - name: bigbench_operators_generate_until - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: The selected task uses generated exact-match answers, with no specified random-answer distribution. - sources: - - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/operators - - category: Math - name: Repeat Copy Logic - metric: exact_match - filter: strict-match - match: - name: bigbench_repeat_copy_logic_generate_until - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: The selected task generates output strings; no defined uniform answer space. - sources: - - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/repeat_copy_logic - - category: Math - name: GSM8K - metric: exact_match - filter: flexible-extract - match: - name: gsm8k - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - Generated numerical answers are not a fixed multiple-choice task; no uniform finite answer space is - specified. - sources: - - https://github.com/openai/grade-school-math - - category: Math - name: MGSM - metric: exact_match - filter: flexible-extract - match: - regex: (?:mgsm_native_cot_.+|global_mgsm_.+) - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: Generated numerical answers follow GSM-style scoring; no fixed uniform answer space is specified. - sources: - - https://github.com/google-research/url-nlp/tree/main/mgsm - - category: Math - name: MATH-500 - metric: accuracy - filter: '' - match: - name: MATH500 - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: Generated mathematical answers have varying domains; no single uniform guess distribution is defined. - sources: - - https://huggingface.co/datasets/HuggingFaceH4/MATH-500 - - category: Reasoning - name: AIME24 - metric: accuracy_avg - filter: '' - match: - name: AIME24 - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - Open-ended integer-answer accuracy. Use a zero minimum with no chance correction; - a uniform random-integer guessing model is not used for this dashboard. - sources: - - https://maa.org/maa-invitational-competitions/ - - >- - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/AIME24/eval_instruct.py - - category: Reasoning - name: AIME25 - metric: accuracy_avg - filter: '' - match: - name: AIME25 - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - Open-ended integer-answer accuracy. Use a zero minimum with no chance correction; - a uniform random-integer guessing model is not used for this dashboard. - sources: - - https://maa.org/maa-invitational-competitions/ - - >- - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/AIME25/eval_instruct.py - - category: Reasoning - name: AMC23 - metric: accuracy_avg - filter: '' - match: - name: AMC23 - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - This evaluator supplies open-ended questions with answer options removed. The original contest - five-choice baseline does not apply to this prompt. - sources: - - >- - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/AMC23/data/amc23.json - - category: Reasoning - name: JEEBench - metric: accuracy_avg - filter: '' - match: - name: JEEBench - score: - scale: 1 - normalize: - # Shared 10.55% baseline; Table 2 reports approximately 10.5%. - min: 0.1055 - max: 1 - clip: true - note: >- - Configured 10.55% overall random baseline, aligned with the shared scoring policy. Table 2 of the - JEEBench paper reports an approximate 10.5% baseline. It combines - uniform single-choice guessing and random option subsets with partial credit for multi-answer - questions, assigning zero expected score to integer and numeric answers. This assumes the full - 515-question benchmark and the paper's scoring rules. - sources: - - https://aclanthology.org/2023.emnlp-main.468.pdf#page=5 - - >- - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/JEEBench/eval_instruct.py - - category: Reasoning - name: LSAT AR - metric: acc_norm - filter: none - match: - name: agieval_lsat_ar - score: - scale: 1 - normalize: - min: 0.2 - max: 1 - clip: true - basis: uniform_choice - note: >- - One correct choice among five LSAT analytical reasoning options; uniform valid-choice guessing gives - 1/5. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/agieval/lsat-ar.yaml - - https://huggingface.co/datasets/hails/agieval-lsat-ar - - category: Reasoning - name: PolyMath - metric: exact_match - filter: none - match: - regex: polymath_.+ - aggregation: - components: - - name: low - match: {regex: 'polymath_.+_low'} - relative_weight: 1 - - name: medium - match: {regex: 'polymath_.+_medium'} - relative_weight: 2 - - name: high - match: {regex: 'polymath_.+_high'} - relative_weight: 4 - - name: top - match: {regex: 'polymath_.+_top'} - relative_weight: 8 - note: >- - PolyMath difficulty-weighted accuracy combines all four levels within each language: - (low + 2 × medium + 4 × high + 8 × top) / 15. Exported exact_match values are - unweighted per-level accuracies. Each level is required; incomplete groups are excluded. - sources: - - https://qwen-polymath.github.io/#benchmark-score - - https://github.com/QwenLM/PolyMath/blob/main/eval/run_eval.py - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: Generated mathematical answers are scored by exact match; no common finite answer space is defined. - sources: - - https://github.com/QwenLM/PolyMath - - category: Knowledge - name: ARC Challenge - metric: acc_norm - filter: none - match: - regex: arc_challenge(?:_mt_.+)? - score: - scale: 1 - normalize: - min: 0.25 - max: 1 - clip: true - basis: uniform_choice - note: >- - Initial approximation: 25% chance. In the published 1,172-question Challenge test split, 1,165 - questions have four choices, four have three and three have five. Mean uniform-guess accuracy is - 25.0156428%, so 25% is a close approximation. The same initial floor is applied to translated - variants; preservation of every choice count has not been audited. This is an approximate baseline, - not a claim that every item has exactly four options. - sources: - - https://ai2-public-datasets.s3.amazonaws.com/arc/ARC-V1-Feb2018.zip - - https://huggingface.co/datasets/allenai/ai2_arc - - https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - - category: Knowledge - name: ARC Easy - metric: acc_norm - filter: none - match: - name: arc_easy - score: - scale: 1 - normalize: - min: 0.25 - max: 1 - clip: true - basis: uniform_choice - note: >- - Use the conventional four-choice approximation of 0.25. The published test split includes a few - three- and five-choice questions (mean guessing accuracy approximately 0.2501613); this small - difference is intentionally ignored for consistency. - sources: - - https://ai2-public-datasets.s3.amazonaws.com/arc/ARC-V1-Feb2018.zip - - https://huggingface.co/datasets/allenai/ai2_arc - - category: Knowledge - name: GPQA Diamond - metric: accuracy_avg - filter: '' - match: - name: GPQADiamond - score: - scale: 1 - normalize: - min: 0.25 - max: 1 - clip: true - basis: uniform_choice - note: The evaluator constructs four options and extracts A/B/C/D; uniform valid-letter guessing gives 1/4. - sources: - - >- - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/GPQADiamond/eval_instruct.py - - category: Knowledge - name: INCLUDE - metric: acc - filter: none - match: - regex: include_base_44_.+ - score: - scale: 1 - select: - regex: include_base_44_[^_]+ - normalize: - min: 0.25 - max: 1 - clip: true - basis: uniform_choice - note: >- - The INCLUDE task templates score A/B/C/D against one answer; 1/4 also holds for weighted aggregates of - four-choice subjects. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/include/default/Albanian/_albanian_template_yaml - - https://huggingface.co/datasets/CohereLabs/include-base-44 - - category: Knowledge - name: Jeopardy - metric: exact_match - filter: strict-match - match: - name: jeopardy - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: Generated exact-match answers have no defined uniform choice set. - sources: - - >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/jeopardy.yaml - - category: Knowledge - name: MMLU - metric: acc - filter: none - match: - regex: mmlu(?:_.+)? - score: - scale: 1 - select: - name: mmlu - normalize: - min: 0.25 - max: 1 - clip: true - basis: uniform_choice - note: >- - Original MMLU uses four choices and one correct label. Uniform guessing gives 1/4, including - subject summaries. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/mmlu/default/_default_template_yaml - - category: Knowledge - name: Global MMLU - metric: acc - filter: none - match: - regex: global_mmlu_.+ - score: - scale: 1 - select: - regex: global_mmlu_full_[a-z]+ - normalize: - min: 0.25 - max: 1 - clip: true - basis: uniform_choice - note: >- - Global MMLU uses four choices and one correct label. Uniform guessing gives 1/4, including - subject summaries. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/global_mmlu/default/en/_en_template_yaml - - category: Knowledge - name: OpenBookQA - metric: acc_norm - filter: none - match: - name: openbookqa - score: - scale: 1 - normalize: - min: 0.25 - max: 1 - clip: true - basis: uniform_choice - note: The benchmark has exactly four choices per question. The paper reports a 25% uniform-guess baseline. - sources: - - https://arxiv.org/html/1809.02789v1#S3.SS3 - - category: Knowledge - name: QA Wikidata - metric: exact_match - filter: strict-match - match: - name: bigbench_qa_wikidata_generate_until - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: The selected generate-until task uses open-ended exact match; no fixed choice count. - sources: - - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/qa_wikidata - - category: Commonsense - name: CommonsenseQA - metric: acc - filter: none - match: - name: commonsense_qa - score: - scale: 1 - normalize: - min: 0.2 - max: 1 - clip: true - basis: uniform_choice - note: The task scores five labels A through E; uniform guessing gives 1/5. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/commonsense_qa/default.yaml - - category: Commonsense - name: COPA - metric: acc - filter: none - match: - name: copa - score: - scale: 1 - normalize: - min: 0.5 - max: 1 - clip: true - basis: uniform_choice - note: Two possible causes or effects are scored; uniform guessing gives 1/2. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/super_glue/copa/utils.py - - category: Commonsense - name: HellaSwag - metric: acc_norm - filter: none - shots: 0 - match: - regex: hellaswag(?:_.+)? - score: - scale: 1 - normalize: - min: 0.25 - max: 1 - clip: true - basis: uniform_choice - note: >- - Four candidate endings; the benchmark reports random performance of 25%. Applied to the translated - variants of the same task. - sources: - - https://rowanzellers.com/hellaswag/ - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/hellaswag/hellaswag.yaml - - category: Commonsense - name: PIQA - metric: acc_norm - filter: none - match: - regex: (?:piqa|global_piqa_completions_.+) - score: - scale: 1 - normalize: - min: 0.5 - max: 1 - clip: true - basis: uniform_choice - note: >- - The completion tasks compare two solutions. Uniform guessing gives 1/2. Prompted Global PIQA is - configured separately and excluded from the supplied comparison sets. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/piqa/piqa.yaml - - >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/global_piqa/completions/_template - - category: Commonsense - name: Global PIQA (prompted) - match: - regex: global_piqa_prompted_.+ - metric: exact_match - filter: strict_match - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: unresolved - note: Provisional scale conversion only; no validated chance correction for the prompted protocol. - warning: >- - Normalization and metric selection have not been validated for prompted Global PIQA. - The supplied comparison sets exclude this eval. Review the protocol before including its scores. - - category: Commonsense - name: Social IQa - metric: acc - filter: none - match: - name: social_iqa - score: - scale: 1 - normalize: - min: 0.3333333333333333 - max: 1 - clip: true - basis: uniform_choice - note: The task scores answerA, answerB and answerC; uniform guessing gives 1/3. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/siqa/siqa.yaml - - category: Commonsense - name: WSC273 - metric: acc - filter: none - match: - name: wsc273 - score: - scale: 1 - normalize: - min: 0.5 - max: 1 - clip: true - basis: uniform_choice - note: The evaluator compares two candidate antecedents; uniform guessing gives 1/2. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/wsc273/default.yaml - - category: Commonsense - name: WinoGrande - metric: acc - filter: none - match: - name: winogrande - score: - scale: 1 - normalize: - min: 0.5 - max: 1 - clip: true - basis: uniform_choice - note: The evaluator compares option1 and option2; uniform guessing gives 1/2. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/winogrande/preprocess_winogrande.py - - category: Commonsense - name: X-CSQA - metric: acc_norm - filter: none - match: - regex: xcsqa_.+ - score: - scale: 1 - normalize: - min: 0.2 - max: 1 - clip: true - basis: uniform_choice - note: Translated CommonsenseQA retains five answer options; uniform guessing gives 1/5. - sources: - - https://inklab.usc.edu/XCSR/xcsr_datasets#x-csqa - - >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/xcsqa/_default_template_yaml - - category: Commonsense - name: XCOPA - metric: acc - filter: none - match: - regex: xcopa:.+ - score: - scale: 1 - normalize: - min: 0.5 - max: 1 - clip: true - basis: uniform_choice - note: Translated COPA retains choice1 and choice2; uniform guessing gives 1/2. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/xcopa/utils.py - - category: Reading - name: Belebele - metric: acc_norm - filter: none - match: - regex: belebele_.+ - score: - scale: 1 - normalize: - min: 0.25 - max: 1 - clip: true - basis: uniform_choice - note: All language variants score four labels A/B/C/D; uniform guessing gives 1/4. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/belebele/_default_template_yaml - - category: Reading - name: BoolQ - metric: acc - filter: none - match: - name: boolq - score: - scale: 1 - normalize: - min: 0.5 - max: 1 - clip: true - basis: uniform_choice - note: The task scores no/yes. This is a uniform-guess baseline of 1/2, not a majority-class baseline. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/super_glue/boolq/default.yaml - - category: Reading - name: CoQA - metric: f1 - filter: none - match: - name: coqa - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: Token-overlap F1 for generated conversational answers has no universal chance floor. - sources: - - https://stanfordnlp.github.io/coqa/ - - category: Reading - name: LAMBADA - metric: acc - filter: none - match: - name: lambada_openai - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - Next-word prediction accuracy is not fixed-choice accuracy. Vocabulary and a guessing distribution - would be needed. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/lambada/lambada_openai.yaml - - category: Reading - name: SIB-200 - # Use acc: acc_norm is not a reliable/useful metric for this eval in our export - # (exactly 0.25 for 34 of 36 languages). - metric: acc - filter: none - match: - regex: sib200_.+ - score: - scale: 1 - normalize: - min: 0.14285714285714285 - max: 1 - clip: true - basis: uniform_choice - note: >- - The OELLM template lists seven topic choices; uniform guessing gives 1/7. The selected field is raw - acc, not acc_norm. - sources: - - >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/sib200/_default_template_yaml - - category: Reading - name: SQuAD v2 - metric: f1 - filter: none - match: - name: squadv2 - score: - scale: 100 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - F1 mixes span overlap and unanswerable questions; neither 0 nor 50% is an established random baseline. - Keep only scale conversion. - sources: - - https://rajpurkar.github.io/SQuAD-explorer/ - - category: Translation - name: FLORES200 - metric: chrf++ - filter: rescored - match: - regex: flores200:.+ - score: - scale: 100 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - Select the explicit chrf++ field from the rescored export (preferred over chrf, then bleu). - Retain native 0–100 points with no chance correction. The CSV does not record the rescoring - implementation signature; the separate chrf++ label identifies the intended metric. - sources: - - https://github.com/mjpost/sacrebleu/blob/master/sacrebleu/metrics/chrf.py - - https://github.com/facebookresearch/flores - - category: Translation - name: OpenSubtitles - metric: chrf - filter: none - match: - regex: opensubtitles_.+ - score: - scale: 100 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - Select chrf: the referenced harness uses SacreBLEU defaults (char_order=6, word_order=0, beta=2), - so this is plain chrF, not chrF++. Prefer an explicitly identified chrF++ result when available, - then chrF, then BLEU. Retain native 0–100 points with no chance correction. - sources: - - >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/opensubtitles_multi40/utils.py - - https://github.com/mjpost/sacrebleu/blob/master/sacrebleu/metrics/chrf.py - - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/api/metrics.py - - category: Language - name: Language ID - metric: acc - filter: none - match: - name: bigbench_language_identification_multiple_choice - score: - scale: 1 - normalize: - min: 0.09090909090909091 - max: 1 - clip: true - basis: uniform_choice - note: >- - Each question offers 11 language names with one correct answer. The corpus covers 1,000 languages, but - the per-question guess baseline is 1/11. - sources: - - >- - https://github.com/google/BIG-bench/blob/main/bigbench/benchmark_tasks/language_identification/README.md - - category: Language - name: MultiBlimp - warning: >- - Language-grouping approximation: multiblimp_hbs pools Croatian and Serbian. Assign its combined - score to Serbian (srp_Latn) because Serbian has more speakers, not because of the dataset's - language proportions. The score still includes both languages and is not a Serbian-only result. - metric: acc_norm - filter: none - match: - regex: multiblimp_.+ - score: - scale: 1 - normalize: - min: 0.5 - max: 1 - clip: true - basis: uniform_choice - note: >- - A minimal pair compares a grammatical sentence with an ungrammatical one; chance under random - preference is 1/2. - sources: - - >- - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/multiblimp/_template_yaml - - category: Instruction following - name: IFEval - metric: prompt_level_strict_acc - filter: none - match: - name: ifeval - score: - scale: 1 - normalize: - min: 0 - max: 1 - clip: true - basis: not_applicable - note: >- - Strict prompt-level instruction success is not a finite-choice task; no defined uniform random - baseline. - sources: - - https://github.com/google-research/google-research/tree/master/instruction_following_eval -languages: - - tasks: - - AIME24 - - AIME25 - - AMC23 - - GPQADiamond - - LiveCodeBench - - MATH500 - - agieval_lsat_ar - - arc_challenge - - arc_easy - - boolq - - commonsense_qa - - copa - - gsm8k - - hellaswag - - ifeval - - jeopardy - - lambada_openai - - mbpp - - mmlu - - mmlu_abstract_algebra - - mmlu_anatomy - - mmlu_astronomy - - mmlu_business_ethics - - mmlu_clinical_knowledge - - mmlu_college_biology - - mmlu_college_chemistry - - mmlu_college_computer_science - - mmlu_college_mathematics - - mmlu_college_medicine - - mmlu_college_physics - - mmlu_computer_security - - mmlu_conceptual_physics - - mmlu_econometrics - - mmlu_electrical_engineering - - mmlu_elementary_mathematics - - mmlu_formal_logic - - mmlu_global_facts - - mmlu_high_school_biology - - mmlu_high_school_chemistry - - mmlu_high_school_computer_science - - mmlu_high_school_european_history - - mmlu_high_school_geography - - mmlu_high_school_government_and_politics - - mmlu_high_school_macroeconomics - - mmlu_high_school_mathematics - - mmlu_high_school_microeconomics - - mmlu_high_school_physics - - mmlu_high_school_psychology - - mmlu_high_school_statistics - - mmlu_high_school_us_history - - mmlu_high_school_world_history - - mmlu_human_aging - - mmlu_human_sexuality - - mmlu_humanities - - mmlu_international_law - - mmlu_jurisprudence - - mmlu_logical_fallacies - - mmlu_machine_learning - - mmlu_management - - mmlu_marketing - - mmlu_medical_genetics - - mmlu_miscellaneous - - mmlu_moral_disputes - - mmlu_moral_scenarios - - mmlu_nutrition - - mmlu_other - - mmlu_philosophy - - mmlu_prehistory - - mmlu_professional_accounting - - mmlu_professional_law - - mmlu_professional_medicine - - mmlu_professional_psychology - - mmlu_public_relations - - mmlu_security_studies - - mmlu_social_sciences - - mmlu_sociology - - mmlu_stem - - mmlu_us_foreign_policy - - mmlu_virology - - mmlu_world_religions - - openbookqa - - piqa - - social_iqa - - winogrande - - wsc273 - scope: single - language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Original English benchmark variant; language is the prompt/question language. - - tasks: - - HumanEval - scope: single - language: eng_Latn - evidence: https://github.com/openai/human-eval - note: >- - Original English benchmark variant; language is the prompt/question language. Programming-language - metrics are separate from natural-language prompts. - - tasks: - - JEEBench - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/daman1209arora/jeebench - note: Original English benchmark variant; language is the prompt/question language. - - tasks: - - arc_challenge_mt_bg - scope: single - language: bul_Cyrl - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_cs - scope: single - language: ces_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_da - scope: single - language: dan_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_de - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_el - scope: single - language: ell_Grek - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_es - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_et - scope: single - language: est_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_fi - scope: single - language: fin_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_fr - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_hu - scope: single - language: hun_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_is - scope: single - language: isl_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_it - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_lt - scope: single - language: lit_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_lv - scope: single - language: lav_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_nb - scope: single - language: nor_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_nl - scope: single - language: nld_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_pl - scope: single - language: pol_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_pt - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_ro - scope: single - language: ron_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_sk - scope: single - language: slk_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_sl - scope: single - language: slv_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - arc_challenge_mt_sv - scope: single - language: swe_Latn - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_bul_Cyrl - scope: single - language: bul_Cyrl - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_ces_Latn - scope: single - language: ces_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_dan_Latn - scope: single - language: dan_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_deu_Latn - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_ell_Grek - scope: single - language: ell_Grek - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_eng_Latn - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_est_Latn - scope: single - language: est_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_fin_Latn - scope: single - language: fin_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_fra_Latn - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_hrv_Latn - scope: single - language: hrv_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_hun_Latn - scope: single - language: hun_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_ita_Latn - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_lit_Latn - scope: single - language: lit_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_lvs_Latn - scope: single - language: lav_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_mlt_Latn - scope: single - language: mlt_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_nld_Latn - scope: single - language: nld_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_nob_Latn - scope: single - language: nor_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_pol_Latn - scope: single - language: pol_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_por_Latn - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_ron_Latn - scope: single - language: ron_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_slk_Latn - scope: single - language: slk_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_slv_Latn - scope: single - language: slv_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_spa_Latn - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - belebele_swe_Latn - scope: single - language: swe_Latn - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - bigbench_cs_algorithms_generate_until - scope: single - language: eng_Latn - evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/cs_algorithms - note: English instructions/prompts; symbolic or numerical content is not a separate natural language. - - tasks: - - bigbench_dyck_languages_generate_until - scope: single - language: eng_Latn - evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/dyck_languages - note: English instructions/prompts; symbolic or numerical content is not a separate natural language. - - tasks: - - bigbench_language_identification_multiple_choice - scope: pooled - language: mul - evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/language_identification - note: >- - One aggregate over 1,000 languages; this CSV has no per-language scores. Kept as a pooled multilingual - result. - - tasks: - - bigbench_operators_generate_until - scope: single - language: eng_Latn - evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/operators - note: English instructions/prompts; symbolic or numerical content is not a separate natural language. - - tasks: - - bigbench_qa_wikidata_generate_until - scope: single - language: eng_Latn - evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/qa_wikidata - note: English instructions/prompts; symbolic or numerical content is not a separate natural language. - - tasks: - - bigbench_repeat_copy_logic_generate_until - scope: single - language: eng_Latn - evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/repeat_copy_logic - note: English instructions/prompts; symbolic or numerical content is not a separate natural language. - - tasks: - - coqa - scope: single - language: eng_Latn - evidence: https://stanfordnlp.github.io/coqa/ - note: Original English benchmark variant; language is the prompt/question language. - - tasks: - - flores200:als_Latn-eng_Latn - scope: translation - source_language: sqi_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:bos_Latn-eng_Latn - scope: translation - source_language: bos_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:bul_Cyrl-eng_Latn - scope: translation - source_language: bul_Cyrl - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:cat_Latn-eng_Latn - scope: translation - source_language: cat_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:ces_Latn-eng_Latn - scope: translation - source_language: ces_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:dan_Latn-eng_Latn - scope: translation - source_language: dan_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:deu_Latn-eng_Latn - scope: translation - source_language: deu_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:ell_Grek-eng_Latn - scope: translation - source_language: ell_Grek - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-als_Latn - scope: translation - source_language: eng_Latn - target_language: sqi_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-bos_Latn - scope: translation - source_language: eng_Latn - target_language: bos_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-bul_Cyrl - scope: translation - source_language: eng_Latn - target_language: bul_Cyrl - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-cat_Latn - scope: translation - source_language: eng_Latn - target_language: cat_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-ces_Latn - scope: translation - source_language: eng_Latn - target_language: ces_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-dan_Latn - scope: translation - source_language: eng_Latn - target_language: dan_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-deu_Latn - scope: translation - source_language: eng_Latn - target_language: deu_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-ell_Grek - scope: translation - source_language: eng_Latn - target_language: ell_Grek - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-est_Latn - scope: translation - source_language: eng_Latn - target_language: est_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-eus_Latn - scope: translation - source_language: eng_Latn - target_language: eus_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-fin_Latn - scope: translation - source_language: eng_Latn - target_language: fin_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-fra_Latn - scope: translation - source_language: eng_Latn - target_language: fra_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-gle_Latn - scope: translation - source_language: eng_Latn - target_language: gle_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-glg_Latn - scope: translation - source_language: eng_Latn - target_language: glg_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-hrv_Latn - scope: translation - source_language: eng_Latn - target_language: hrv_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-hun_Latn - scope: translation - source_language: eng_Latn - target_language: hun_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-isl_Latn - scope: translation - source_language: eng_Latn - target_language: isl_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-ita_Latn - scope: translation - source_language: eng_Latn - target_language: ita_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-kat_Geor - scope: translation - source_language: eng_Latn - target_language: kat_Geor - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-lit_Latn - scope: translation - source_language: eng_Latn - target_language: lit_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-lvs_Latn - scope: translation - source_language: eng_Latn - target_language: lav_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-mkd_Cyrl - scope: translation - source_language: eng_Latn - target_language: mkd_Cyrl - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-mlt_Latn - scope: translation - source_language: eng_Latn - target_language: mlt_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-nld_Latn - scope: translation - source_language: eng_Latn - target_language: nld_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-nob_Latn - scope: translation - source_language: eng_Latn - target_language: nor_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-pol_Latn - scope: translation - source_language: eng_Latn - target_language: pol_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-por_Latn - scope: translation - source_language: eng_Latn - target_language: por_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-ron_Latn - scope: translation - source_language: eng_Latn - target_language: ron_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-slk_Latn - scope: translation - source_language: eng_Latn - target_language: slk_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-slv_Latn - scope: translation - source_language: eng_Latn - target_language: slv_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-spa_Latn - scope: translation - source_language: eng_Latn - target_language: spa_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-srp_Cyrl - scope: translation - source_language: eng_Latn - target_language: srp_Cyrl - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-swe_Latn - scope: translation - source_language: eng_Latn - target_language: swe_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-tur_Latn - scope: translation - source_language: eng_Latn - target_language: tur_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eng_Latn-ukr_Cyrl - scope: translation - source_language: eng_Latn - target_language: ukr_Cyrl - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:est_Latn-eng_Latn - scope: translation - source_language: est_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:eus_Latn-eng_Latn - scope: translation - source_language: eus_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:fin_Latn-eng_Latn - scope: translation - source_language: fin_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:fra_Latn-eng_Latn - scope: translation - source_language: fra_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:gle_Latn-eng_Latn - scope: translation - source_language: gle_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:glg_Latn-eng_Latn - scope: translation - source_language: glg_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:hrv_Latn-eng_Latn - scope: translation - source_language: hrv_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:hun_Latn-eng_Latn - scope: translation - source_language: hun_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:isl_Latn-eng_Latn - scope: translation - source_language: isl_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:ita_Latn-eng_Latn - scope: translation - source_language: ita_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:kat_Geor-eng_Latn - scope: translation - source_language: kat_Geor - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:lit_Latn-eng_Latn - scope: translation - source_language: lit_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:lvs_Latn-eng_Latn - scope: translation - source_language: lav_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:mkd_Cyrl-eng_Latn - scope: translation - source_language: mkd_Cyrl - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:mlt_Latn-eng_Latn - scope: translation - source_language: mlt_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:nld_Latn-eng_Latn - scope: translation - source_language: nld_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:nob_Latn-eng_Latn - scope: translation - source_language: nor_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:pol_Latn-eng_Latn - scope: translation - source_language: pol_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:por_Latn-eng_Latn - scope: translation - source_language: por_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:ron_Latn-eng_Latn - scope: translation - source_language: ron_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:slk_Latn-eng_Latn - scope: translation - source_language: slk_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:slv_Latn-eng_Latn - scope: translation - source_language: slv_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:spa_Latn-eng_Latn - scope: translation - source_language: spa_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:srp_Cyrl-eng_Latn - scope: translation - source_language: srp_Cyrl - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:swe_Latn-eng_Latn - scope: translation - source_language: swe_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:tur_Latn-eng_Latn - scope: translation - source_language: tur_Latn - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - flores200:ukr_Cyrl-eng_Latn - scope: translation - source_language: ukr_Cyrl - target_language: eng_Latn - evidence: https://huggingface.co/datasets/facebook/flores - note: >- - Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - - tasks: - - global_mgsm_ca - scope: single - language: cat_Latn - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_cs - scope: single - language: ces_Latn - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_de - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_el - scope: single - language: ell_Grek - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_en - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_es - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_eu - scope: single - language: eus_Latn - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_fr - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_gl - scope: single - language: glg_Latn - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_hu - scope: single - language: hun_Latn - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mgsm_sr - scope: single - language: srp_Cyrl - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_cs - - global_mmlu_full_cs_abstract_algebra - - global_mmlu_full_cs_anatomy - - global_mmlu_full_cs_astronomy - - global_mmlu_full_cs_business_ethics - - global_mmlu_full_cs_clinical_knowledge - - global_mmlu_full_cs_college_biology - - global_mmlu_full_cs_college_chemistry - - global_mmlu_full_cs_college_computer_science - - global_mmlu_full_cs_college_mathematics - - global_mmlu_full_cs_college_medicine - - global_mmlu_full_cs_college_physics - - global_mmlu_full_cs_computer_security - - global_mmlu_full_cs_conceptual_physics - - global_mmlu_full_cs_econometrics - - global_mmlu_full_cs_electrical_engineering - - global_mmlu_full_cs_elementary_mathematics - - global_mmlu_full_cs_formal_logic - - global_mmlu_full_cs_global_facts - - global_mmlu_full_cs_high_school_biology - - global_mmlu_full_cs_high_school_chemistry - - global_mmlu_full_cs_high_school_computer_science - - global_mmlu_full_cs_high_school_european_history - - global_mmlu_full_cs_high_school_geography - - global_mmlu_full_cs_high_school_government_and_politics - - global_mmlu_full_cs_high_school_macroeconomics - - global_mmlu_full_cs_high_school_mathematics - - global_mmlu_full_cs_high_school_microeconomics - - global_mmlu_full_cs_high_school_physics - - global_mmlu_full_cs_high_school_psychology - - global_mmlu_full_cs_high_school_statistics - - global_mmlu_full_cs_high_school_us_history - - global_mmlu_full_cs_high_school_world_history - - global_mmlu_full_cs_human_aging - - global_mmlu_full_cs_human_sexuality - - global_mmlu_full_cs_humanities - - global_mmlu_full_cs_international_law - - global_mmlu_full_cs_jurisprudence - - global_mmlu_full_cs_logical_fallacies - - global_mmlu_full_cs_machine_learning - - global_mmlu_full_cs_management - - global_mmlu_full_cs_marketing - - global_mmlu_full_cs_medical_genetics - - global_mmlu_full_cs_miscellaneous - - global_mmlu_full_cs_moral_disputes - - global_mmlu_full_cs_moral_scenarios - - global_mmlu_full_cs_nutrition - - global_mmlu_full_cs_other - - global_mmlu_full_cs_philosophy - - global_mmlu_full_cs_prehistory - - global_mmlu_full_cs_professional_accounting - - global_mmlu_full_cs_professional_law - - global_mmlu_full_cs_professional_medicine - - global_mmlu_full_cs_professional_psychology - - global_mmlu_full_cs_public_relations - - global_mmlu_full_cs_security_studies - - global_mmlu_full_cs_social_sciences - - global_mmlu_full_cs_sociology - - global_mmlu_full_cs_stem - - global_mmlu_full_cs_us_foreign_policy - - global_mmlu_full_cs_virology - - global_mmlu_full_cs_world_religions - scope: single - language: ces_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_de - - global_mmlu_full_de_abstract_algebra - - global_mmlu_full_de_anatomy - - global_mmlu_full_de_astronomy - - global_mmlu_full_de_business_ethics - - global_mmlu_full_de_clinical_knowledge - - global_mmlu_full_de_college_biology - - global_mmlu_full_de_college_chemistry - - global_mmlu_full_de_college_computer_science - - global_mmlu_full_de_college_mathematics - - global_mmlu_full_de_college_medicine - - global_mmlu_full_de_college_physics - - global_mmlu_full_de_computer_security - - global_mmlu_full_de_conceptual_physics - - global_mmlu_full_de_econometrics - - global_mmlu_full_de_electrical_engineering - - global_mmlu_full_de_elementary_mathematics - - global_mmlu_full_de_formal_logic - - global_mmlu_full_de_global_facts - - global_mmlu_full_de_high_school_biology - - global_mmlu_full_de_high_school_chemistry - - global_mmlu_full_de_high_school_computer_science - - global_mmlu_full_de_high_school_european_history - - global_mmlu_full_de_high_school_geography - - global_mmlu_full_de_high_school_government_and_politics - - global_mmlu_full_de_high_school_macroeconomics - - global_mmlu_full_de_high_school_mathematics - - global_mmlu_full_de_high_school_microeconomics - - global_mmlu_full_de_high_school_physics - - global_mmlu_full_de_high_school_psychology - - global_mmlu_full_de_high_school_statistics - - global_mmlu_full_de_high_school_us_history - - global_mmlu_full_de_high_school_world_history - - global_mmlu_full_de_human_aging - - global_mmlu_full_de_human_sexuality - - global_mmlu_full_de_humanities - - global_mmlu_full_de_international_law - - global_mmlu_full_de_jurisprudence - - global_mmlu_full_de_logical_fallacies - - global_mmlu_full_de_machine_learning - - global_mmlu_full_de_management - - global_mmlu_full_de_marketing - - global_mmlu_full_de_medical_genetics - - global_mmlu_full_de_miscellaneous - - global_mmlu_full_de_moral_disputes - - global_mmlu_full_de_moral_scenarios - - global_mmlu_full_de_nutrition - - global_mmlu_full_de_other - - global_mmlu_full_de_philosophy - - global_mmlu_full_de_prehistory - - global_mmlu_full_de_professional_accounting - - global_mmlu_full_de_professional_law - - global_mmlu_full_de_professional_medicine - - global_mmlu_full_de_professional_psychology - - global_mmlu_full_de_public_relations - - global_mmlu_full_de_security_studies - - global_mmlu_full_de_social_sciences - - global_mmlu_full_de_sociology - - global_mmlu_full_de_stem - - global_mmlu_full_de_us_foreign_policy - - global_mmlu_full_de_virology - - global_mmlu_full_de_world_religions - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_el - - global_mmlu_full_el_abstract_algebra - - global_mmlu_full_el_anatomy - - global_mmlu_full_el_astronomy - - global_mmlu_full_el_business_ethics - - global_mmlu_full_el_clinical_knowledge - - global_mmlu_full_el_college_biology - - global_mmlu_full_el_college_chemistry - - global_mmlu_full_el_college_computer_science - - global_mmlu_full_el_college_mathematics - - global_mmlu_full_el_college_medicine - - global_mmlu_full_el_college_physics - - global_mmlu_full_el_computer_security - - global_mmlu_full_el_conceptual_physics - - global_mmlu_full_el_econometrics - - global_mmlu_full_el_electrical_engineering - - global_mmlu_full_el_elementary_mathematics - - global_mmlu_full_el_formal_logic - - global_mmlu_full_el_global_facts - - global_mmlu_full_el_high_school_biology - - global_mmlu_full_el_high_school_chemistry - - global_mmlu_full_el_high_school_computer_science - - global_mmlu_full_el_high_school_european_history - - global_mmlu_full_el_high_school_geography - - global_mmlu_full_el_high_school_government_and_politics - - global_mmlu_full_el_high_school_macroeconomics - - global_mmlu_full_el_high_school_mathematics - - global_mmlu_full_el_high_school_microeconomics - - global_mmlu_full_el_high_school_physics - - global_mmlu_full_el_high_school_psychology - - global_mmlu_full_el_high_school_statistics - - global_mmlu_full_el_high_school_us_history - - global_mmlu_full_el_high_school_world_history - - global_mmlu_full_el_human_aging - - global_mmlu_full_el_human_sexuality - - global_mmlu_full_el_humanities - - global_mmlu_full_el_international_law - - global_mmlu_full_el_jurisprudence - - global_mmlu_full_el_logical_fallacies - - global_mmlu_full_el_machine_learning - - global_mmlu_full_el_management - - global_mmlu_full_el_marketing - - global_mmlu_full_el_medical_genetics - - global_mmlu_full_el_miscellaneous - - global_mmlu_full_el_moral_disputes - - global_mmlu_full_el_moral_scenarios - - global_mmlu_full_el_nutrition - - global_mmlu_full_el_other - - global_mmlu_full_el_philosophy - - global_mmlu_full_el_prehistory - - global_mmlu_full_el_professional_accounting - - global_mmlu_full_el_professional_law - - global_mmlu_full_el_professional_medicine - - global_mmlu_full_el_professional_psychology - - global_mmlu_full_el_public_relations - - global_mmlu_full_el_security_studies - - global_mmlu_full_el_social_sciences - - global_mmlu_full_el_sociology - - global_mmlu_full_el_stem - - global_mmlu_full_el_us_foreign_policy - - global_mmlu_full_el_virology - - global_mmlu_full_el_world_religions - scope: single - language: ell_Grek - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_en - - global_mmlu_full_en_abstract_algebra - - global_mmlu_full_en_anatomy - - global_mmlu_full_en_astronomy - - global_mmlu_full_en_business_ethics - - global_mmlu_full_en_clinical_knowledge - - global_mmlu_full_en_college_biology - - global_mmlu_full_en_college_chemistry - - global_mmlu_full_en_college_computer_science - - global_mmlu_full_en_college_mathematics - - global_mmlu_full_en_college_medicine - - global_mmlu_full_en_college_physics - - global_mmlu_full_en_computer_security - - global_mmlu_full_en_conceptual_physics - - global_mmlu_full_en_econometrics - - global_mmlu_full_en_electrical_engineering - - global_mmlu_full_en_elementary_mathematics - - global_mmlu_full_en_formal_logic - - global_mmlu_full_en_global_facts - - global_mmlu_full_en_high_school_biology - - global_mmlu_full_en_high_school_chemistry - - global_mmlu_full_en_high_school_computer_science - - global_mmlu_full_en_high_school_european_history - - global_mmlu_full_en_high_school_geography - - global_mmlu_full_en_high_school_government_and_politics - - global_mmlu_full_en_high_school_macroeconomics - - global_mmlu_full_en_high_school_mathematics - - global_mmlu_full_en_high_school_microeconomics - - global_mmlu_full_en_high_school_physics - - global_mmlu_full_en_high_school_psychology - - global_mmlu_full_en_high_school_statistics - - global_mmlu_full_en_high_school_us_history - - global_mmlu_full_en_high_school_world_history - - global_mmlu_full_en_human_aging - - global_mmlu_full_en_human_sexuality - - global_mmlu_full_en_humanities - - global_mmlu_full_en_international_law - - global_mmlu_full_en_jurisprudence - - global_mmlu_full_en_logical_fallacies - - global_mmlu_full_en_machine_learning - - global_mmlu_full_en_management - - global_mmlu_full_en_marketing - - global_mmlu_full_en_medical_genetics - - global_mmlu_full_en_miscellaneous - - global_mmlu_full_en_moral_disputes - - global_mmlu_full_en_moral_scenarios - - global_mmlu_full_en_nutrition - - global_mmlu_full_en_other - - global_mmlu_full_en_philosophy - - global_mmlu_full_en_prehistory - - global_mmlu_full_en_professional_accounting - - global_mmlu_full_en_professional_law - - global_mmlu_full_en_professional_medicine - - global_mmlu_full_en_professional_psychology - - global_mmlu_full_en_public_relations - - global_mmlu_full_en_security_studies - - global_mmlu_full_en_social_sciences - - global_mmlu_full_en_sociology - - global_mmlu_full_en_stem - - global_mmlu_full_en_us_foreign_policy - - global_mmlu_full_en_virology - - global_mmlu_full_en_world_religions - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_es - - global_mmlu_full_es_abstract_algebra - - global_mmlu_full_es_anatomy - - global_mmlu_full_es_astronomy - - global_mmlu_full_es_business_ethics - - global_mmlu_full_es_clinical_knowledge - - global_mmlu_full_es_college_biology - - global_mmlu_full_es_college_chemistry - - global_mmlu_full_es_college_computer_science - - global_mmlu_full_es_college_mathematics - - global_mmlu_full_es_college_medicine - - global_mmlu_full_es_college_physics - - global_mmlu_full_es_computer_security - - global_mmlu_full_es_conceptual_physics - - global_mmlu_full_es_econometrics - - global_mmlu_full_es_electrical_engineering - - global_mmlu_full_es_elementary_mathematics - - global_mmlu_full_es_formal_logic - - global_mmlu_full_es_global_facts - - global_mmlu_full_es_high_school_biology - - global_mmlu_full_es_high_school_chemistry - - global_mmlu_full_es_high_school_computer_science - - global_mmlu_full_es_high_school_european_history - - global_mmlu_full_es_high_school_geography - - global_mmlu_full_es_high_school_government_and_politics - - global_mmlu_full_es_high_school_macroeconomics - - global_mmlu_full_es_high_school_mathematics - - global_mmlu_full_es_high_school_microeconomics - - global_mmlu_full_es_high_school_physics - - global_mmlu_full_es_high_school_psychology - - global_mmlu_full_es_high_school_statistics - - global_mmlu_full_es_high_school_us_history - - global_mmlu_full_es_high_school_world_history - - global_mmlu_full_es_human_aging - - global_mmlu_full_es_human_sexuality - - global_mmlu_full_es_humanities - - global_mmlu_full_es_international_law - - global_mmlu_full_es_jurisprudence - - global_mmlu_full_es_logical_fallacies - - global_mmlu_full_es_machine_learning - - global_mmlu_full_es_management - - global_mmlu_full_es_marketing - - global_mmlu_full_es_medical_genetics - - global_mmlu_full_es_miscellaneous - - global_mmlu_full_es_moral_disputes - - global_mmlu_full_es_moral_scenarios - - global_mmlu_full_es_nutrition - - global_mmlu_full_es_other - - global_mmlu_full_es_philosophy - - global_mmlu_full_es_prehistory - - global_mmlu_full_es_professional_accounting - - global_mmlu_full_es_professional_law - - global_mmlu_full_es_professional_medicine - - global_mmlu_full_es_professional_psychology - - global_mmlu_full_es_public_relations - - global_mmlu_full_es_security_studies - - global_mmlu_full_es_social_sciences - - global_mmlu_full_es_sociology - - global_mmlu_full_es_stem - - global_mmlu_full_es_us_foreign_policy - - global_mmlu_full_es_virology - - global_mmlu_full_es_world_religions - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_fr - - global_mmlu_full_fr_abstract_algebra - - global_mmlu_full_fr_anatomy - - global_mmlu_full_fr_astronomy - - global_mmlu_full_fr_business_ethics - - global_mmlu_full_fr_clinical_knowledge - - global_mmlu_full_fr_college_biology - - global_mmlu_full_fr_college_chemistry - - global_mmlu_full_fr_college_computer_science - - global_mmlu_full_fr_college_mathematics - - global_mmlu_full_fr_college_medicine - - global_mmlu_full_fr_college_physics - - global_mmlu_full_fr_computer_security - - global_mmlu_full_fr_conceptual_physics - - global_mmlu_full_fr_econometrics - - global_mmlu_full_fr_electrical_engineering - - global_mmlu_full_fr_elementary_mathematics - - global_mmlu_full_fr_formal_logic - - global_mmlu_full_fr_global_facts - - global_mmlu_full_fr_high_school_biology - - global_mmlu_full_fr_high_school_chemistry - - global_mmlu_full_fr_high_school_computer_science - - global_mmlu_full_fr_high_school_european_history - - global_mmlu_full_fr_high_school_geography - - global_mmlu_full_fr_high_school_government_and_politics - - global_mmlu_full_fr_high_school_macroeconomics - - global_mmlu_full_fr_high_school_mathematics - - global_mmlu_full_fr_high_school_microeconomics - - global_mmlu_full_fr_high_school_physics - - global_mmlu_full_fr_high_school_psychology - - global_mmlu_full_fr_high_school_statistics - - global_mmlu_full_fr_high_school_us_history - - global_mmlu_full_fr_high_school_world_history - - global_mmlu_full_fr_human_aging - - global_mmlu_full_fr_human_sexuality - - global_mmlu_full_fr_humanities - - global_mmlu_full_fr_international_law - - global_mmlu_full_fr_jurisprudence - - global_mmlu_full_fr_logical_fallacies - - global_mmlu_full_fr_machine_learning - - global_mmlu_full_fr_management - - global_mmlu_full_fr_marketing - - global_mmlu_full_fr_medical_genetics - - global_mmlu_full_fr_miscellaneous - - global_mmlu_full_fr_moral_disputes - - global_mmlu_full_fr_moral_scenarios - - global_mmlu_full_fr_nutrition - - global_mmlu_full_fr_other - - global_mmlu_full_fr_philosophy - - global_mmlu_full_fr_prehistory - - global_mmlu_full_fr_professional_accounting - - global_mmlu_full_fr_professional_law - - global_mmlu_full_fr_professional_medicine - - global_mmlu_full_fr_professional_psychology - - global_mmlu_full_fr_public_relations - - global_mmlu_full_fr_security_studies - - global_mmlu_full_fr_social_sciences - - global_mmlu_full_fr_sociology - - global_mmlu_full_fr_stem - - global_mmlu_full_fr_us_foreign_policy - - global_mmlu_full_fr_virology - - global_mmlu_full_fr_world_religions - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_it - - global_mmlu_full_it_abstract_algebra - - global_mmlu_full_it_anatomy - - global_mmlu_full_it_astronomy - - global_mmlu_full_it_business_ethics - - global_mmlu_full_it_clinical_knowledge - - global_mmlu_full_it_college_biology - - global_mmlu_full_it_college_chemistry - - global_mmlu_full_it_college_computer_science - - global_mmlu_full_it_college_mathematics - - global_mmlu_full_it_college_medicine - - global_mmlu_full_it_college_physics - - global_mmlu_full_it_computer_security - - global_mmlu_full_it_conceptual_physics - - global_mmlu_full_it_econometrics - - global_mmlu_full_it_electrical_engineering - - global_mmlu_full_it_elementary_mathematics - - global_mmlu_full_it_formal_logic - - global_mmlu_full_it_global_facts - - global_mmlu_full_it_high_school_biology - - global_mmlu_full_it_high_school_chemistry - - global_mmlu_full_it_high_school_computer_science - - global_mmlu_full_it_high_school_european_history - - global_mmlu_full_it_high_school_geography - - global_mmlu_full_it_high_school_government_and_politics - - global_mmlu_full_it_high_school_macroeconomics - - global_mmlu_full_it_high_school_mathematics - - global_mmlu_full_it_high_school_microeconomics - - global_mmlu_full_it_high_school_physics - - global_mmlu_full_it_high_school_psychology - - global_mmlu_full_it_high_school_statistics - - global_mmlu_full_it_high_school_us_history - - global_mmlu_full_it_high_school_world_history - - global_mmlu_full_it_human_aging - - global_mmlu_full_it_human_sexuality - - global_mmlu_full_it_humanities - - global_mmlu_full_it_international_law - - global_mmlu_full_it_jurisprudence - - global_mmlu_full_it_logical_fallacies - - global_mmlu_full_it_machine_learning - - global_mmlu_full_it_management - - global_mmlu_full_it_marketing - - global_mmlu_full_it_medical_genetics - - global_mmlu_full_it_miscellaneous - - global_mmlu_full_it_moral_disputes - - global_mmlu_full_it_moral_scenarios - - global_mmlu_full_it_nutrition - - global_mmlu_full_it_other - - global_mmlu_full_it_philosophy - - global_mmlu_full_it_prehistory - - global_mmlu_full_it_professional_accounting - - global_mmlu_full_it_professional_law - - global_mmlu_full_it_professional_medicine - - global_mmlu_full_it_professional_psychology - - global_mmlu_full_it_public_relations - - global_mmlu_full_it_security_studies - - global_mmlu_full_it_social_sciences - - global_mmlu_full_it_sociology - - global_mmlu_full_it_stem - - global_mmlu_full_it_us_foreign_policy - - global_mmlu_full_it_virology - - global_mmlu_full_it_world_religions - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_lt - - global_mmlu_full_lt_abstract_algebra - - global_mmlu_full_lt_anatomy - - global_mmlu_full_lt_astronomy - - global_mmlu_full_lt_business_ethics - - global_mmlu_full_lt_clinical_knowledge - - global_mmlu_full_lt_college_biology - - global_mmlu_full_lt_college_chemistry - - global_mmlu_full_lt_college_computer_science - - global_mmlu_full_lt_college_mathematics - - global_mmlu_full_lt_college_medicine - - global_mmlu_full_lt_college_physics - - global_mmlu_full_lt_computer_security - - global_mmlu_full_lt_conceptual_physics - - global_mmlu_full_lt_econometrics - - global_mmlu_full_lt_electrical_engineering - - global_mmlu_full_lt_elementary_mathematics - - global_mmlu_full_lt_formal_logic - - global_mmlu_full_lt_global_facts - - global_mmlu_full_lt_high_school_biology - - global_mmlu_full_lt_high_school_chemistry - - global_mmlu_full_lt_high_school_computer_science - - global_mmlu_full_lt_high_school_european_history - - global_mmlu_full_lt_high_school_geography - - global_mmlu_full_lt_high_school_government_and_politics - - global_mmlu_full_lt_high_school_macroeconomics - - global_mmlu_full_lt_high_school_mathematics - - global_mmlu_full_lt_high_school_microeconomics - - global_mmlu_full_lt_high_school_physics - - global_mmlu_full_lt_high_school_psychology - - global_mmlu_full_lt_high_school_statistics - - global_mmlu_full_lt_high_school_us_history - - global_mmlu_full_lt_high_school_world_history - - global_mmlu_full_lt_human_aging - - global_mmlu_full_lt_human_sexuality - - global_mmlu_full_lt_humanities - - global_mmlu_full_lt_international_law - - global_mmlu_full_lt_jurisprudence - - global_mmlu_full_lt_logical_fallacies - - global_mmlu_full_lt_machine_learning - - global_mmlu_full_lt_management - - global_mmlu_full_lt_marketing - - global_mmlu_full_lt_medical_genetics - - global_mmlu_full_lt_miscellaneous - - global_mmlu_full_lt_moral_disputes - - global_mmlu_full_lt_moral_scenarios - - global_mmlu_full_lt_nutrition - - global_mmlu_full_lt_other - - global_mmlu_full_lt_philosophy - - global_mmlu_full_lt_prehistory - - global_mmlu_full_lt_professional_accounting - - global_mmlu_full_lt_professional_law - - global_mmlu_full_lt_professional_medicine - - global_mmlu_full_lt_professional_psychology - - global_mmlu_full_lt_public_relations - - global_mmlu_full_lt_security_studies - - global_mmlu_full_lt_social_sciences - - global_mmlu_full_lt_sociology - - global_mmlu_full_lt_stem - - global_mmlu_full_lt_us_foreign_policy - - global_mmlu_full_lt_virology - - global_mmlu_full_lt_world_religions - scope: single - language: lit_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_nl - - global_mmlu_full_nl_abstract_algebra - - global_mmlu_full_nl_anatomy - - global_mmlu_full_nl_astronomy - - global_mmlu_full_nl_business_ethics - - global_mmlu_full_nl_clinical_knowledge - - global_mmlu_full_nl_college_biology - - global_mmlu_full_nl_college_chemistry - - global_mmlu_full_nl_college_computer_science - - global_mmlu_full_nl_college_mathematics - - global_mmlu_full_nl_college_medicine - - global_mmlu_full_nl_college_physics - - global_mmlu_full_nl_computer_security - - global_mmlu_full_nl_conceptual_physics - - global_mmlu_full_nl_econometrics - - global_mmlu_full_nl_electrical_engineering - - global_mmlu_full_nl_elementary_mathematics - - global_mmlu_full_nl_formal_logic - - global_mmlu_full_nl_global_facts - - global_mmlu_full_nl_high_school_biology - - global_mmlu_full_nl_high_school_chemistry - - global_mmlu_full_nl_high_school_computer_science - - global_mmlu_full_nl_high_school_european_history - - global_mmlu_full_nl_high_school_geography - - global_mmlu_full_nl_high_school_government_and_politics - - global_mmlu_full_nl_high_school_macroeconomics - - global_mmlu_full_nl_high_school_mathematics - - global_mmlu_full_nl_high_school_microeconomics - - global_mmlu_full_nl_high_school_physics - - global_mmlu_full_nl_high_school_psychology - - global_mmlu_full_nl_high_school_statistics - - global_mmlu_full_nl_high_school_us_history - - global_mmlu_full_nl_high_school_world_history - - global_mmlu_full_nl_human_aging - - global_mmlu_full_nl_human_sexuality - - global_mmlu_full_nl_humanities - - global_mmlu_full_nl_international_law - - global_mmlu_full_nl_jurisprudence - - global_mmlu_full_nl_logical_fallacies - - global_mmlu_full_nl_machine_learning - - global_mmlu_full_nl_management - - global_mmlu_full_nl_marketing - - global_mmlu_full_nl_medical_genetics - - global_mmlu_full_nl_miscellaneous - - global_mmlu_full_nl_moral_disputes - - global_mmlu_full_nl_moral_scenarios - - global_mmlu_full_nl_nutrition - - global_mmlu_full_nl_other - - global_mmlu_full_nl_philosophy - - global_mmlu_full_nl_prehistory - - global_mmlu_full_nl_professional_accounting - - global_mmlu_full_nl_professional_law - - global_mmlu_full_nl_professional_medicine - - global_mmlu_full_nl_professional_psychology - - global_mmlu_full_nl_public_relations - - global_mmlu_full_nl_security_studies - - global_mmlu_full_nl_social_sciences - - global_mmlu_full_nl_sociology - - global_mmlu_full_nl_stem - - global_mmlu_full_nl_us_foreign_policy - - global_mmlu_full_nl_virology - - global_mmlu_full_nl_world_religions - scope: single - language: nld_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_pl - - global_mmlu_full_pl_abstract_algebra - - global_mmlu_full_pl_anatomy - - global_mmlu_full_pl_astronomy - - global_mmlu_full_pl_business_ethics - - global_mmlu_full_pl_clinical_knowledge - - global_mmlu_full_pl_college_biology - - global_mmlu_full_pl_college_chemistry - - global_mmlu_full_pl_college_computer_science - - global_mmlu_full_pl_college_mathematics - - global_mmlu_full_pl_college_medicine - - global_mmlu_full_pl_college_physics - - global_mmlu_full_pl_computer_security - - global_mmlu_full_pl_conceptual_physics - - global_mmlu_full_pl_econometrics - - global_mmlu_full_pl_electrical_engineering - - global_mmlu_full_pl_elementary_mathematics - - global_mmlu_full_pl_formal_logic - - global_mmlu_full_pl_global_facts - - global_mmlu_full_pl_high_school_biology - - global_mmlu_full_pl_high_school_chemistry - - global_mmlu_full_pl_high_school_computer_science - - global_mmlu_full_pl_high_school_european_history - - global_mmlu_full_pl_high_school_geography - - global_mmlu_full_pl_high_school_government_and_politics - - global_mmlu_full_pl_high_school_macroeconomics - - global_mmlu_full_pl_high_school_mathematics - - global_mmlu_full_pl_high_school_microeconomics - - global_mmlu_full_pl_high_school_physics - - global_mmlu_full_pl_high_school_psychology - - global_mmlu_full_pl_high_school_statistics - - global_mmlu_full_pl_high_school_us_history - - global_mmlu_full_pl_high_school_world_history - - global_mmlu_full_pl_human_aging - - global_mmlu_full_pl_human_sexuality - - global_mmlu_full_pl_humanities - - global_mmlu_full_pl_international_law - - global_mmlu_full_pl_jurisprudence - - global_mmlu_full_pl_logical_fallacies - - global_mmlu_full_pl_machine_learning - - global_mmlu_full_pl_management - - global_mmlu_full_pl_marketing - - global_mmlu_full_pl_medical_genetics - - global_mmlu_full_pl_miscellaneous - - global_mmlu_full_pl_moral_disputes - - global_mmlu_full_pl_moral_scenarios - - global_mmlu_full_pl_nutrition - - global_mmlu_full_pl_other - - global_mmlu_full_pl_philosophy - - global_mmlu_full_pl_prehistory - - global_mmlu_full_pl_professional_accounting - - global_mmlu_full_pl_professional_law - - global_mmlu_full_pl_professional_medicine - - global_mmlu_full_pl_professional_psychology - - global_mmlu_full_pl_public_relations - - global_mmlu_full_pl_security_studies - - global_mmlu_full_pl_social_sciences - - global_mmlu_full_pl_sociology - - global_mmlu_full_pl_stem - - global_mmlu_full_pl_us_foreign_policy - - global_mmlu_full_pl_virology - - global_mmlu_full_pl_world_religions - scope: single - language: pol_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_pt - - global_mmlu_full_pt_abstract_algebra - - global_mmlu_full_pt_anatomy - - global_mmlu_full_pt_astronomy - - global_mmlu_full_pt_business_ethics - - global_mmlu_full_pt_clinical_knowledge - - global_mmlu_full_pt_college_biology - - global_mmlu_full_pt_college_chemistry - - global_mmlu_full_pt_college_computer_science - - global_mmlu_full_pt_college_mathematics - - global_mmlu_full_pt_college_medicine - - global_mmlu_full_pt_college_physics - - global_mmlu_full_pt_computer_security - - global_mmlu_full_pt_conceptual_physics - - global_mmlu_full_pt_econometrics - - global_mmlu_full_pt_electrical_engineering - - global_mmlu_full_pt_elementary_mathematics - - global_mmlu_full_pt_formal_logic - - global_mmlu_full_pt_global_facts - - global_mmlu_full_pt_high_school_biology - - global_mmlu_full_pt_high_school_chemistry - - global_mmlu_full_pt_high_school_computer_science - - global_mmlu_full_pt_high_school_european_history - - global_mmlu_full_pt_high_school_geography - - global_mmlu_full_pt_high_school_government_and_politics - - global_mmlu_full_pt_high_school_macroeconomics - - global_mmlu_full_pt_high_school_mathematics - - global_mmlu_full_pt_high_school_microeconomics - - global_mmlu_full_pt_high_school_physics - - global_mmlu_full_pt_high_school_psychology - - global_mmlu_full_pt_high_school_statistics - - global_mmlu_full_pt_high_school_us_history - - global_mmlu_full_pt_high_school_world_history - - global_mmlu_full_pt_human_aging - - global_mmlu_full_pt_human_sexuality - - global_mmlu_full_pt_humanities - - global_mmlu_full_pt_international_law - - global_mmlu_full_pt_jurisprudence - - global_mmlu_full_pt_logical_fallacies - - global_mmlu_full_pt_machine_learning - - global_mmlu_full_pt_management - - global_mmlu_full_pt_marketing - - global_mmlu_full_pt_medical_genetics - - global_mmlu_full_pt_miscellaneous - - global_mmlu_full_pt_moral_disputes - - global_mmlu_full_pt_moral_scenarios - - global_mmlu_full_pt_nutrition - - global_mmlu_full_pt_other - - global_mmlu_full_pt_philosophy - - global_mmlu_full_pt_prehistory - - global_mmlu_full_pt_professional_accounting - - global_mmlu_full_pt_professional_law - - global_mmlu_full_pt_professional_medicine - - global_mmlu_full_pt_professional_psychology - - global_mmlu_full_pt_public_relations - - global_mmlu_full_pt_security_studies - - global_mmlu_full_pt_social_sciences - - global_mmlu_full_pt_sociology - - global_mmlu_full_pt_stem - - global_mmlu_full_pt_us_foreign_policy - - global_mmlu_full_pt_virology - - global_mmlu_full_pt_world_religions - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_ro - - global_mmlu_full_ro_abstract_algebra - - global_mmlu_full_ro_anatomy - - global_mmlu_full_ro_astronomy - - global_mmlu_full_ro_business_ethics - - global_mmlu_full_ro_clinical_knowledge - - global_mmlu_full_ro_college_biology - - global_mmlu_full_ro_college_chemistry - - global_mmlu_full_ro_college_computer_science - - global_mmlu_full_ro_college_mathematics - - global_mmlu_full_ro_college_medicine - - global_mmlu_full_ro_college_physics - - global_mmlu_full_ro_computer_security - - global_mmlu_full_ro_conceptual_physics - - global_mmlu_full_ro_econometrics - - global_mmlu_full_ro_electrical_engineering - - global_mmlu_full_ro_elementary_mathematics - - global_mmlu_full_ro_formal_logic - - global_mmlu_full_ro_global_facts - - global_mmlu_full_ro_high_school_biology - - global_mmlu_full_ro_high_school_chemistry - - global_mmlu_full_ro_high_school_computer_science - - global_mmlu_full_ro_high_school_european_history - - global_mmlu_full_ro_high_school_geography - - global_mmlu_full_ro_high_school_government_and_politics - - global_mmlu_full_ro_high_school_macroeconomics - - global_mmlu_full_ro_high_school_mathematics - - global_mmlu_full_ro_high_school_microeconomics - - global_mmlu_full_ro_high_school_physics - - global_mmlu_full_ro_high_school_psychology - - global_mmlu_full_ro_high_school_statistics - - global_mmlu_full_ro_high_school_us_history - - global_mmlu_full_ro_high_school_world_history - - global_mmlu_full_ro_human_aging - - global_mmlu_full_ro_human_sexuality - - global_mmlu_full_ro_humanities - - global_mmlu_full_ro_international_law - - global_mmlu_full_ro_jurisprudence - - global_mmlu_full_ro_logical_fallacies - - global_mmlu_full_ro_machine_learning - - global_mmlu_full_ro_management - - global_mmlu_full_ro_marketing - - global_mmlu_full_ro_medical_genetics - - global_mmlu_full_ro_miscellaneous - - global_mmlu_full_ro_moral_disputes - - global_mmlu_full_ro_moral_scenarios - - global_mmlu_full_ro_nutrition - - global_mmlu_full_ro_other - - global_mmlu_full_ro_philosophy - - global_mmlu_full_ro_prehistory - - global_mmlu_full_ro_professional_accounting - - global_mmlu_full_ro_professional_law - - global_mmlu_full_ro_professional_medicine - - global_mmlu_full_ro_professional_psychology - - global_mmlu_full_ro_public_relations - - global_mmlu_full_ro_security_studies - - global_mmlu_full_ro_social_sciences - - global_mmlu_full_ro_sociology - - global_mmlu_full_ro_stem - - global_mmlu_full_ro_us_foreign_policy - - global_mmlu_full_ro_virology - - global_mmlu_full_ro_world_religions - scope: single - language: ron_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_sr - - global_mmlu_full_sr_abstract_algebra - - global_mmlu_full_sr_anatomy - - global_mmlu_full_sr_astronomy - - global_mmlu_full_sr_business_ethics - - global_mmlu_full_sr_clinical_knowledge - - global_mmlu_full_sr_college_biology - - global_mmlu_full_sr_college_chemistry - - global_mmlu_full_sr_college_computer_science - - global_mmlu_full_sr_college_mathematics - - global_mmlu_full_sr_college_medicine - - global_mmlu_full_sr_college_physics - - global_mmlu_full_sr_computer_security - - global_mmlu_full_sr_conceptual_physics - - global_mmlu_full_sr_econometrics - - global_mmlu_full_sr_electrical_engineering - - global_mmlu_full_sr_elementary_mathematics - - global_mmlu_full_sr_formal_logic - - global_mmlu_full_sr_global_facts - - global_mmlu_full_sr_high_school_biology - - global_mmlu_full_sr_high_school_chemistry - - global_mmlu_full_sr_high_school_computer_science - - global_mmlu_full_sr_high_school_european_history - - global_mmlu_full_sr_high_school_geography - - global_mmlu_full_sr_high_school_government_and_politics - - global_mmlu_full_sr_high_school_macroeconomics - - global_mmlu_full_sr_high_school_mathematics - - global_mmlu_full_sr_high_school_microeconomics - - global_mmlu_full_sr_high_school_physics - - global_mmlu_full_sr_high_school_psychology - - global_mmlu_full_sr_high_school_statistics - - global_mmlu_full_sr_high_school_us_history - - global_mmlu_full_sr_high_school_world_history - - global_mmlu_full_sr_human_aging - - global_mmlu_full_sr_human_sexuality - - global_mmlu_full_sr_humanities - - global_mmlu_full_sr_international_law - - global_mmlu_full_sr_jurisprudence - - global_mmlu_full_sr_logical_fallacies - - global_mmlu_full_sr_machine_learning - - global_mmlu_full_sr_management - - global_mmlu_full_sr_marketing - - global_mmlu_full_sr_medical_genetics - - global_mmlu_full_sr_miscellaneous - - global_mmlu_full_sr_moral_disputes - - global_mmlu_full_sr_moral_scenarios - - global_mmlu_full_sr_nutrition - - global_mmlu_full_sr_other - - global_mmlu_full_sr_philosophy - - global_mmlu_full_sr_prehistory - - global_mmlu_full_sr_professional_accounting - - global_mmlu_full_sr_professional_law - - global_mmlu_full_sr_professional_medicine - - global_mmlu_full_sr_professional_psychology - - global_mmlu_full_sr_public_relations - - global_mmlu_full_sr_security_studies - - global_mmlu_full_sr_social_sciences - - global_mmlu_full_sr_sociology - - global_mmlu_full_sr_stem - - global_mmlu_full_sr_us_foreign_policy - - global_mmlu_full_sr_virology - - global_mmlu_full_sr_world_religions - scope: single - language: srp_Cyrl - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_sv - - global_mmlu_full_sv_abstract_algebra - - global_mmlu_full_sv_anatomy - - global_mmlu_full_sv_astronomy - - global_mmlu_full_sv_business_ethics - - global_mmlu_full_sv_clinical_knowledge - - global_mmlu_full_sv_college_biology - - global_mmlu_full_sv_college_chemistry - - global_mmlu_full_sv_college_computer_science - - global_mmlu_full_sv_college_mathematics - - global_mmlu_full_sv_college_medicine - - global_mmlu_full_sv_college_physics - - global_mmlu_full_sv_computer_security - - global_mmlu_full_sv_conceptual_physics - - global_mmlu_full_sv_econometrics - - global_mmlu_full_sv_electrical_engineering - - global_mmlu_full_sv_elementary_mathematics - - global_mmlu_full_sv_formal_logic - - global_mmlu_full_sv_global_facts - - global_mmlu_full_sv_high_school_biology - - global_mmlu_full_sv_high_school_chemistry - - global_mmlu_full_sv_high_school_computer_science - - global_mmlu_full_sv_high_school_european_history - - global_mmlu_full_sv_high_school_geography - - global_mmlu_full_sv_high_school_government_and_politics - - global_mmlu_full_sv_high_school_macroeconomics - - global_mmlu_full_sv_high_school_mathematics - - global_mmlu_full_sv_high_school_microeconomics - - global_mmlu_full_sv_high_school_physics - - global_mmlu_full_sv_high_school_psychology - - global_mmlu_full_sv_high_school_statistics - - global_mmlu_full_sv_high_school_us_history - - global_mmlu_full_sv_high_school_world_history - - global_mmlu_full_sv_human_aging - - global_mmlu_full_sv_human_sexuality - - global_mmlu_full_sv_humanities - - global_mmlu_full_sv_international_law - - global_mmlu_full_sv_jurisprudence - - global_mmlu_full_sv_logical_fallacies - - global_mmlu_full_sv_machine_learning - - global_mmlu_full_sv_management - - global_mmlu_full_sv_marketing - - global_mmlu_full_sv_medical_genetics - - global_mmlu_full_sv_miscellaneous - - global_mmlu_full_sv_moral_disputes - - global_mmlu_full_sv_moral_scenarios - - global_mmlu_full_sv_nutrition - - global_mmlu_full_sv_other - - global_mmlu_full_sv_philosophy - - global_mmlu_full_sv_prehistory - - global_mmlu_full_sv_professional_accounting - - global_mmlu_full_sv_professional_law - - global_mmlu_full_sv_professional_medicine - - global_mmlu_full_sv_professional_psychology - - global_mmlu_full_sv_public_relations - - global_mmlu_full_sv_security_studies - - global_mmlu_full_sv_social_sciences - - global_mmlu_full_sv_sociology - - global_mmlu_full_sv_stem - - global_mmlu_full_sv_us_foreign_policy - - global_mmlu_full_sv_virology - - global_mmlu_full_sv_world_religions - scope: single - language: swe_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_tr - - global_mmlu_full_tr_abstract_algebra - - global_mmlu_full_tr_anatomy - - global_mmlu_full_tr_astronomy - - global_mmlu_full_tr_business_ethics - - global_mmlu_full_tr_clinical_knowledge - - global_mmlu_full_tr_college_biology - - global_mmlu_full_tr_college_chemistry - - global_mmlu_full_tr_college_computer_science - - global_mmlu_full_tr_college_mathematics - - global_mmlu_full_tr_college_medicine - - global_mmlu_full_tr_college_physics - - global_mmlu_full_tr_computer_security - - global_mmlu_full_tr_conceptual_physics - - global_mmlu_full_tr_econometrics - - global_mmlu_full_tr_electrical_engineering - - global_mmlu_full_tr_elementary_mathematics - - global_mmlu_full_tr_formal_logic - - global_mmlu_full_tr_global_facts - - global_mmlu_full_tr_high_school_biology - - global_mmlu_full_tr_high_school_chemistry - - global_mmlu_full_tr_high_school_computer_science - - global_mmlu_full_tr_high_school_european_history - - global_mmlu_full_tr_high_school_geography - - global_mmlu_full_tr_high_school_government_and_politics - - global_mmlu_full_tr_high_school_macroeconomics - - global_mmlu_full_tr_high_school_mathematics - - global_mmlu_full_tr_high_school_microeconomics - - global_mmlu_full_tr_high_school_physics - - global_mmlu_full_tr_high_school_psychology - - global_mmlu_full_tr_high_school_statistics - - global_mmlu_full_tr_high_school_us_history - - global_mmlu_full_tr_high_school_world_history - - global_mmlu_full_tr_human_aging - - global_mmlu_full_tr_human_sexuality - - global_mmlu_full_tr_humanities - - global_mmlu_full_tr_international_law - - global_mmlu_full_tr_jurisprudence - - global_mmlu_full_tr_logical_fallacies - - global_mmlu_full_tr_machine_learning - - global_mmlu_full_tr_management - - global_mmlu_full_tr_marketing - - global_mmlu_full_tr_medical_genetics - - global_mmlu_full_tr_miscellaneous - - global_mmlu_full_tr_moral_disputes - - global_mmlu_full_tr_moral_scenarios - - global_mmlu_full_tr_nutrition - - global_mmlu_full_tr_other - - global_mmlu_full_tr_philosophy - - global_mmlu_full_tr_prehistory - - global_mmlu_full_tr_professional_accounting - - global_mmlu_full_tr_professional_law - - global_mmlu_full_tr_professional_medicine - - global_mmlu_full_tr_professional_psychology - - global_mmlu_full_tr_public_relations - - global_mmlu_full_tr_security_studies - - global_mmlu_full_tr_social_sciences - - global_mmlu_full_tr_sociology - - global_mmlu_full_tr_stem - - global_mmlu_full_tr_us_foreign_policy - - global_mmlu_full_tr_virology - - global_mmlu_full_tr_world_religions - scope: single - language: tur_Latn - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_mmlu_full_uk - - global_mmlu_full_uk_abstract_algebra - - global_mmlu_full_uk_anatomy - - global_mmlu_full_uk_astronomy - - global_mmlu_full_uk_business_ethics - - global_mmlu_full_uk_clinical_knowledge - - global_mmlu_full_uk_college_biology - - global_mmlu_full_uk_college_chemistry - - global_mmlu_full_uk_college_computer_science - - global_mmlu_full_uk_college_mathematics - - global_mmlu_full_uk_college_medicine - - global_mmlu_full_uk_college_physics - - global_mmlu_full_uk_computer_security - - global_mmlu_full_uk_conceptual_physics - - global_mmlu_full_uk_econometrics - - global_mmlu_full_uk_electrical_engineering - - global_mmlu_full_uk_elementary_mathematics - - global_mmlu_full_uk_formal_logic - - global_mmlu_full_uk_global_facts - - global_mmlu_full_uk_high_school_biology - - global_mmlu_full_uk_high_school_chemistry - - global_mmlu_full_uk_high_school_computer_science - - global_mmlu_full_uk_high_school_european_history - - global_mmlu_full_uk_high_school_geography - - global_mmlu_full_uk_high_school_government_and_politics - - global_mmlu_full_uk_high_school_macroeconomics - - global_mmlu_full_uk_high_school_mathematics - - global_mmlu_full_uk_high_school_microeconomics - - global_mmlu_full_uk_high_school_physics - - global_mmlu_full_uk_high_school_psychology - - global_mmlu_full_uk_high_school_statistics - - global_mmlu_full_uk_high_school_us_history - - global_mmlu_full_uk_high_school_world_history - - global_mmlu_full_uk_human_aging - - global_mmlu_full_uk_human_sexuality - - global_mmlu_full_uk_humanities - - global_mmlu_full_uk_international_law - - global_mmlu_full_uk_jurisprudence - - global_mmlu_full_uk_logical_fallacies - - global_mmlu_full_uk_machine_learning - - global_mmlu_full_uk_management - - global_mmlu_full_uk_marketing - - global_mmlu_full_uk_medical_genetics - - global_mmlu_full_uk_miscellaneous - - global_mmlu_full_uk_moral_disputes - - global_mmlu_full_uk_moral_scenarios - - global_mmlu_full_uk_nutrition - - global_mmlu_full_uk_other - - global_mmlu_full_uk_philosophy - - global_mmlu_full_uk_prehistory - - global_mmlu_full_uk_professional_accounting - - global_mmlu_full_uk_professional_law - - global_mmlu_full_uk_professional_medicine - - global_mmlu_full_uk_professional_psychology - - global_mmlu_full_uk_public_relations - - global_mmlu_full_uk_security_studies - - global_mmlu_full_uk_social_sciences - - global_mmlu_full_uk_sociology - - global_mmlu_full_uk_stem - - global_mmlu_full_uk_us_foreign_policy - - global_mmlu_full_uk_virology - - global_mmlu_full_uk_world_religions - scope: single - language: ukr_Cyrl - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_als_latn - - global_piqa_prompted_als_latn - scope: single - language: sqi_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_bos_latn - - global_piqa_prompted_bos_latn - scope: single - language: bos_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_bul_cyrl - - global_piqa_prompted_bul_cyrl - scope: single - language: bul_Cyrl - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_cat_latn - - global_piqa_prompted_cat_latn - scope: single - language: cat_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_ces_latn - - global_piqa_prompted_ces_latn - scope: single - language: ces_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_deu_latn - - global_piqa_prompted_deu_latn - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_ekk_latn - - global_piqa_prompted_ekk_latn - scope: single - language: est_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_ell_grek - - global_piqa_prompted_ell_grek - scope: single - language: ell_Grek - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_eng_latn - - global_piqa_prompted_eng_latn - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_fin_latn - - global_piqa_prompted_fin_latn - scope: single - language: fin_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_fra_latn_fran - - global_piqa_prompted_fra_latn_fran - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_glg_latn - - global_piqa_prompted_glg_latn - scope: single - language: glg_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_hrv_latn - - global_piqa_prompted_hrv_latn - scope: single - language: hrv_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_hun_latn - - global_piqa_prompted_hun_latn - scope: single - language: hun_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_isl_latn - - global_piqa_prompted_isl_latn - scope: single - language: isl_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_ita_latn - - global_piqa_prompted_ita_latn - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_kat_geor - - global_piqa_prompted_kat_geor - scope: single - language: kat_Geor - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_lit_latn - - global_piqa_prompted_lit_latn - scope: single - language: lit_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_mkd_cyrl - - global_piqa_prompted_mkd_cyrl - scope: single - language: mkd_Cyrl - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_nld_latn - - global_piqa_prompted_nld_latn - scope: single - language: nld_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_nno_latn - - global_piqa_completions_nob_latn - - global_piqa_prompted_nno_latn - - global_piqa_prompted_nob_latn - scope: single - language: nor_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_pol_latn - - global_piqa_prompted_pol_latn - scope: single - language: pol_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_por_latn_port - - global_piqa_prompted_por_latn_port - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_ron_latn - - global_piqa_prompted_ron_latn - scope: single - language: ron_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_slk_latn - - global_piqa_prompted_slk_latn - scope: single - language: slk_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_slv_latn - - global_piqa_prompted_slv_latn - scope: single - language: slv_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_spa_latn_spai - - global_piqa_prompted_spa_latn_spai - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_srp_cyrl - - global_piqa_prompted_srp_cyrl - scope: single - language: srp_Cyrl - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_swe_latn - - global_piqa_prompted_swe_latn - scope: single - language: swe_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_tur_latn - - global_piqa_prompted_tur_latn - scope: single - language: tur_Latn - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - global_piqa_completions_ukr_cyrl - - global_piqa_prompted_ukr_cyrl - scope: single - language: ukr_Cyrl - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_ca - scope: single - language: cat_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_da - scope: single - language: dan_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_de - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_es - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_eu - scope: single - language: eus_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_fr - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_hr - scope: single - language: hrv_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_hu - scope: single - language: hun_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_it - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_nl - scope: single - language: nld_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_pt - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_ro - scope: single - language: ron_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_sk - scope: single - language: slk_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_sr - scope: single - language: srp_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: The sr dataset examples use Latin script; this overrides the generic Serbian Cyrillic default. - - tasks: - - hellaswag_sv - scope: single - language: swe_Latn - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - hellaswag_uk - scope: single - language: ukr_Cyrl - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_albanian - - include_base_44_albanian_arts_humanities - - include_base_44_albanian_business_commerce - - include_base_44_albanian_health_oriented_education - - include_base_44_albanian_social_science - - include_base_44_albanian_stem - scope: single - language: sqi_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_basque - - include_base_44_basque_professional_certification - scope: single - language: eus_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_bulgarian - - include_base_44_bulgarian_arts_humanities - - include_base_44_bulgarian_social_science - - include_base_44_bulgarian_stem - scope: single - language: bul_Cyrl - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_croatian - - include_base_44_croatian_arts_humanities - - include_base_44_croatian_social_science - - include_base_44_croatian_stem - scope: single - language: hrv_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_dutch - - include_base_44_dutch_applied_science - - include_base_44_dutch_arts_humanities - - include_base_44_dutch_health_oriented_education - - include_base_44_dutch_social_science - - include_base_44_dutch_stem - scope: single - language: nld_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_estonian - - include_base_44_estonian_applied_science - - include_base_44_estonian_arts_humanities - - include_base_44_estonian_health_oriented_education - - include_base_44_estonian_social_science - - include_base_44_estonian_stem - scope: single - language: est_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_finnish - - include_base_44_finnish_applied_science - - include_base_44_finnish_arts_humanities - - include_base_44_finnish_health_oriented_education - - include_base_44_finnish_social_science - - include_base_44_finnish_stem - scope: single - language: fin_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_french - - include_base_44_french_arts_humanities - - include_base_44_french_driving_license - - include_base_44_french_health_oriented_education - - include_base_44_french_social_science - - include_base_44_french_stem - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_georgian - - include_base_44_georgian_arts_humanities - scope: single - language: kat_Geor - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_german - - include_base_44_german_driving_license - - include_base_44_german_social_science - - include_base_44_german_stem - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_greek - - include_base_44_greek_arts_humanities - - include_base_44_greek_business_commerce - - include_base_44_greek_health_oriented_education - - include_base_44_greek_medical_license - - include_base_44_greek_professional_certification - - include_base_44_greek_social_science - - include_base_44_greek_stem - scope: single - language: ell_Grek - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_hungarian - - include_base_44_hungarian_applied_science - - include_base_44_hungarian_social_science - - include_base_44_hungarian_stem - scope: single - language: hun_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_italian - - include_base_44_italian_applied_science - - include_base_44_italian_arts_humanities - - include_base_44_italian_health_oriented_education - - include_base_44_italian_professional_certification - - include_base_44_italian_social_science - - include_base_44_italian_stem - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_lithuanian - - include_base_44_lithuanian_arts_humanities - - include_base_44_lithuanian_business_commerce - - include_base_44_lithuanian_professional_certification - - include_base_44_lithuanian_social_science - - include_base_44_lithuanian_stem - scope: single - language: lit_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_north macedonian - - include_base_44_north macedonian_arts_humanities - - include_base_44_north macedonian_business_commerce - - include_base_44_north macedonian_social_science - - include_base_44_north macedonian_stem - scope: single - language: mkd_Cyrl - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_polish - - include_base_44_polish_professional_certification - - include_base_44_polish_social_science - - include_base_44_polish_stem - scope: single - language: pol_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_portuguese - - include_base_44_portuguese_applied_science - - include_base_44_portuguese_arts_humanities - - include_base_44_portuguese_business_commerce - - include_base_44_portuguese_health_oriented_education - - include_base_44_portuguese_social_science - - include_base_44_portuguese_stem - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_serbian - - include_base_44_serbian_arts_humanities - - include_base_44_serbian_social_science - - include_base_44_serbian_stem - scope: single - language: srp_Cyrl - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_spanish - - include_base_44_spanish_arts_humanities - - include_base_44_spanish_health_oriented_education - - include_base_44_spanish_social_science - - include_base_44_spanish_stem - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_turkish - - include_base_44_turkish_arts_humanities - - include_base_44_turkish_business_commerce - - include_base_44_turkish_social_science - - include_base_44_turkish_stem - scope: single - language: tur_Latn - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - include_base_44_ukrainian - - include_base_44_ukrainian_arts_humanities - - include_base_44_ukrainian_social_science - - include_base_44_ukrainian_stem - scope: single - language: ukr_Cyrl - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - mgsm_native_cot_de - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - mgsm_native_cot_en - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - mgsm_native_cot_es - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - mgsm_native_cot_fr - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_bul - scope: single - language: bul_Cyrl - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_cat - scope: single - language: cat_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_ces - scope: single - language: ces_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_dan - scope: single - language: dan_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_deu - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_ell - scope: single - language: ell_Grek - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_eng - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_est - scope: single - language: est_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_eus - scope: single - language: eus_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_fin - scope: single - language: fin_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_fra - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_gle - scope: single - language: gle_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_glg - scope: single - language: glg_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_hbs - scope: single - language: srp_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp/blob/main/hbs/data.tsv - note: >- - Grouping convention: assign the pooled Croatian/Serbian score to Serbian, the language with more - speakers; retain Latin script. This does not separate the data or produce a Serbian-only score. - UCLA estimates approximately 11 million Serbian speakers and 6 million Croatian speakers: - https://slavic.ucla.edu/languages/bcs/serbian-background-info/ and - https://slavic.ucla.edu/languages/bcs/croatian-background-info/. - - tasks: - - multiblimp_hun - scope: single - language: hun_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_isl - scope: single - language: isl_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_ita - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_kat - scope: single - language: kat_Geor - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_lav - scope: single - language: lav_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_lit - scope: single - language: lit_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_mkd - scope: single - language: mkd_Cyrl - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_nld - scope: single - language: nld_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_pol - scope: single - language: pol_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_por - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_ron - scope: single - language: ron_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_slk - scope: single - language: slk_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_slv - scope: single - language: slv_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_spa - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_sqi - scope: single - language: sqi_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_swe - scope: single - language: swe_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_tur - scope: single - language: tur_Latn - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - multiblimp_ukr - scope: single - language: ukr_Cyrl - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - opensubtitles_multi40_bg_to_en - scope: translation - source_language: bul_Cyrl - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_cs_to_en - scope: translation - source_language: ces_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_da_to_en - scope: translation - source_language: dan_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_de_to_en - scope: translation - source_language: deu_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_el_to_en - scope: translation - source_language: ell_Grek - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_bg - scope: translation - source_language: eng_Latn - target_language: bul_Cyrl - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_cs - scope: translation - source_language: eng_Latn - target_language: ces_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_da - scope: translation - source_language: eng_Latn - target_language: dan_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_de - scope: translation - source_language: eng_Latn - target_language: deu_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_el - scope: translation - source_language: eng_Latn - target_language: ell_Grek - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_es - scope: translation - source_language: eng_Latn - target_language: spa_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_et - scope: translation - source_language: eng_Latn - target_language: est_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_fi - scope: translation - source_language: eng_Latn - target_language: fin_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_fr - scope: translation - source_language: eng_Latn - target_language: fra_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_hr - scope: translation - source_language: eng_Latn - target_language: hrv_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_hu - scope: translation - source_language: eng_Latn - target_language: hun_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_it - scope: translation - source_language: eng_Latn - target_language: ita_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_lt - scope: translation - source_language: eng_Latn - target_language: lit_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_lv - scope: translation - source_language: eng_Latn - target_language: lav_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_nl - scope: translation - source_language: eng_Latn - target_language: nld_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_no - scope: translation - source_language: eng_Latn - target_language: nor_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_pl - scope: translation - source_language: eng_Latn - target_language: pol_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_pt - scope: translation - source_language: eng_Latn - target_language: por_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_ro - scope: translation - source_language: eng_Latn - target_language: ron_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_sk - scope: translation - source_language: eng_Latn - target_language: slk_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_sl - scope: translation - source_language: eng_Latn - target_language: slv_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_sr - scope: translation - source_language: eng_Latn - target_language: srp_Cyrl - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_sv - scope: translation - source_language: eng_Latn - target_language: swe_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_tr - scope: translation - source_language: eng_Latn - target_language: tur_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_en_to_uk - scope: translation - source_language: eng_Latn - target_language: ukr_Cyrl - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_es_to_en - scope: translation - source_language: spa_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_et_to_en - scope: translation - source_language: est_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_fi_to_en - scope: translation - source_language: fin_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_fr_to_en - scope: translation - source_language: fra_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_hr_to_en - scope: translation - source_language: hrv_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_hu_to_en - scope: translation - source_language: hun_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_it_to_en - scope: translation - source_language: ita_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_lt_to_en - scope: translation - source_language: lit_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_lv_to_en - scope: translation - source_language: lav_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_nl_to_en - scope: translation - source_language: nld_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_no_to_en - scope: translation - source_language: nor_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_pl_to_en - scope: translation - source_language: pol_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_pt_to_en - scope: translation - source_language: por_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_ro_to_en - scope: translation - source_language: ron_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_sk_to_en - scope: translation - source_language: slk_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_sl_to_en - scope: translation - source_language: slv_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_sr_to_en - scope: translation - source_language: srp_Cyrl - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_sv_to_en - scope: translation - source_language: swe_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_tr_to_en - scope: translation - source_language: tur_Latn - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - opensubtitles_multi40_uk_to_en - scope: translation - source_language: ukr_Cyrl - target_language: eng_Latn - evidence: >- - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: >- - Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, not a - sample-level script audit. Source and target are explicit for separate translation-into and - translation-from views. - - tasks: - - polymath_de_high - - polymath_de_low - - polymath_de_medium - - polymath_de_top - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - polymath_en_high - - polymath_en_low - - polymath_en_medium - - polymath_en_top - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - polymath_es_high - - polymath_es_low - - polymath_es_medium - - polymath_es_top - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - polymath_fr_high - - polymath_fr_low - - polymath_fr_medium - - polymath_fr_top - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - polymath_it_high - - polymath_it_low - - polymath_it_medium - - polymath_it_top - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - polymath_pt_high - - polymath_pt_low - - polymath_pt_medium - - polymath_pt_top - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_als_Latn - scope: single - language: sqi_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_bos_Latn - scope: single - language: bos_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_bul_Cyrl - scope: single - language: bul_Cyrl - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_cat_Latn - scope: single - language: cat_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_ces_Latn - scope: single - language: ces_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_dan_Latn - scope: single - language: dan_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_deu_Latn - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_ell_Grek - scope: single - language: ell_Grek - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_eng_Latn - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_est_Latn - scope: single - language: est_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_eus_Latn - scope: single - language: eus_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_fin_Latn - scope: single - language: fin_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_fra_Latn - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_gle_Latn - scope: single - language: gle_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_glg_Latn - scope: single - language: glg_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_hrv_Latn - scope: single - language: hrv_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_hun_Latn - scope: single - language: hun_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_isl_Latn - scope: single - language: isl_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_ita_Latn - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_kat_Geor - scope: single - language: kat_Geor - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_lit_Latn - scope: single - language: lit_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_lvs_Latn - scope: single - language: lav_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_mkd_Cyrl - scope: single - language: mkd_Cyrl - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_mlt_Latn - scope: single - language: mlt_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_nld_Latn - scope: single - language: nld_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_nob_Latn - scope: single - language: nor_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_pol_Latn - scope: single - language: pol_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_por_Latn - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_ron_Latn - scope: single - language: ron_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_slk_Latn - scope: single - language: slk_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_slv_Latn - scope: single - language: slv_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_spa_Latn - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_srp_Cyrl - scope: single - language: srp_Cyrl - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_swe_Latn - scope: single - language: swe_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_tur_Latn - scope: single - language: tur_Latn - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - sib200_ukr_Cyrl - scope: single - language: ukr_Cyrl - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - squadv2 - scope: single - language: eng_Latn - evidence: https://rajpurkar.github.io/SQuAD-explorer/ - note: Original English benchmark variant; language is the prompt/question language. - - tasks: - - xcopa:et - scope: single - language: est_Latn - evidence: https://huggingface.co/datasets/cambridgeltl/xcopa - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcopa:it - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/cambridgeltl/xcopa - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcopa:tr - scope: single - language: tur_Latn - evidence: https://huggingface.co/datasets/cambridgeltl/xcopa - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcsqa_deu_Latn - scope: single - language: deu_Latn - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcsqa_eng_Latn - scope: single - language: eng_Latn - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcsqa_fra_Latn - scope: single - language: fra_Latn - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcsqa_ita_Latn - scope: single - language: ita_Latn - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcsqa_nld_Latn - scope: single - language: nld_Latn - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcsqa_pol_Latn - scope: single - language: pol_Latn - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcsqa_por_Latn - scope: single - language: por_Latn - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - - tasks: - - xcsqa_spa_Latn - scope: single - language: spa_Latn - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. +evals_dir: evals diff --git a/configs/evals/aime24.yaml b/configs/evals/aime24.yaml new file mode 100644 index 0000000..d2b8c07 --- /dev/null +++ b/configs/evals/aime24.yaml @@ -0,0 +1,24 @@ +name: AIME24 +category: Reasoning +match: + name: AIME24 +metric: accuracy_avg +filter: '' +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Open-ended integer-answer accuracy. Use a zero minimum with no chance correction; a uniform random-integer + guessing model is not used for this dashboard. + sources: + - https://maa.org/maa-invitational-competitions/ + - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/AIME24/eval_instruct.py +languages: + - language: eng_Latn + scope: single + tasks: [AIME24] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/aime25.yaml b/configs/evals/aime25.yaml new file mode 100644 index 0000000..58d4cea --- /dev/null +++ b/configs/evals/aime25.yaml @@ -0,0 +1,24 @@ +name: AIME25 +category: Reasoning +match: + name: AIME25 +metric: accuracy_avg +filter: '' +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Open-ended integer-answer accuracy. Use a zero minimum with no chance correction; a uniform random-integer + guessing model is not used for this dashboard. + sources: + - https://maa.org/maa-invitational-competitions/ + - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/AIME25/eval_instruct.py +languages: + - language: eng_Latn + scope: single + tasks: [AIME25] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/amc23.yaml b/configs/evals/amc23.yaml new file mode 100644 index 0000000..3640935 --- /dev/null +++ b/configs/evals/amc23.yaml @@ -0,0 +1,23 @@ +name: AMC23 +category: Reasoning +match: + name: AMC23 +metric: accuracy_avg +filter: '' +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: This evaluator supplies open-ended questions with answer options removed. The original contest five-choice + baseline does not apply to this prompt. + sources: + - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/AMC23/data/amc23.json +languages: + - language: eng_Latn + scope: single + tasks: [AMC23] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/arc_challenge.yaml b/configs/evals/arc_challenge.yaml new file mode 100644 index 0000000..e196786 --- /dev/null +++ b/configs/evals/arc_challenge.yaml @@ -0,0 +1,138 @@ +name: ARC Challenge +category: Knowledge +match: + regex: arc_challenge(?:_mt_.+)? +metric: acc_norm +filter: none +score: + scale: 1 +normalize: + min: 0.25 + max: 1 + clip: true + basis: uniform_choice + note: 'Initial approximation: 25% chance. In the published 1,172-question Challenge test split, 1,165 questions + have four choices, four have three and three have five. Mean uniform-guess accuracy is 25.0156428%, so + 25% is a close approximation. The same initial floor is applied to translated variants; preservation of + every choice count has not been audited. This is an approximate baseline, not a claim that every item has + exactly four options.' + sources: + - https://ai2-public-datasets.s3.amazonaws.com/arc/ARC-V1-Feb2018.zip + - https://huggingface.co/datasets/allenai/ai2_arc + - https://huggingface.co/datasets/LumiOpen/arc_challenge_mt +languages: + - language: eng_Latn + scope: single + tasks: [arc_challenge] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. + - language: bul_Cyrl + scope: single + tasks: [arc_challenge_mt_bg] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ces_Latn + scope: single + tasks: [arc_challenge_mt_cs] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: dan_Latn + scope: single + tasks: [arc_challenge_mt_da] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [arc_challenge_mt_de] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ell_Grek + scope: single + tasks: [arc_challenge_mt_el] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [arc_challenge_mt_es] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: est_Latn + scope: single + tasks: [arc_challenge_mt_et] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fin_Latn + scope: single + tasks: [arc_challenge_mt_fi] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [arc_challenge_mt_fr] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hun_Latn + scope: single + tasks: [arc_challenge_mt_hu] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: isl_Latn + scope: single + tasks: [arc_challenge_mt_is] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [arc_challenge_mt_it] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lit_Latn + scope: single + tasks: [arc_challenge_mt_lt] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lav_Latn + scope: single + tasks: [arc_challenge_mt_lv] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nor_Latn + scope: single + tasks: [arc_challenge_mt_nb] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [arc_challenge_mt_nl] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: pol_Latn + scope: single + tasks: [arc_challenge_mt_pl] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [arc_challenge_mt_pt] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ron_Latn + scope: single + tasks: [arc_challenge_mt_ro] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slk_Latn + scope: single + tasks: [arc_challenge_mt_sk] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slv_Latn + scope: single + tasks: [arc_challenge_mt_sl] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: swe_Latn + scope: single + tasks: [arc_challenge_mt_sv] + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/arc_easy.yaml b/configs/evals/arc_easy.yaml new file mode 100644 index 0000000..a61a2fe --- /dev/null +++ b/configs/evals/arc_easy.yaml @@ -0,0 +1,25 @@ +name: ARC Easy +category: Knowledge +match: + name: arc_easy +metric: acc_norm +filter: none +score: + scale: 1 +normalize: + min: 0.25 + max: 1 + clip: true + basis: uniform_choice + note: Use the conventional four-choice approximation of 0.25. The published test split includes a few three- + and five-choice questions (mean guessing accuracy approximately 0.2501613); this small difference is intentionally + ignored for consistency. + sources: + - https://ai2-public-datasets.s3.amazonaws.com/arc/ARC-V1-Feb2018.zip + - https://huggingface.co/datasets/allenai/ai2_arc +languages: + - language: eng_Latn + scope: single + tasks: [arc_easy] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/belebele.yaml b/configs/evals/belebele.yaml new file mode 100644 index 0000000..904712c --- /dev/null +++ b/configs/evals/belebele.yaml @@ -0,0 +1,137 @@ +name: Belebele +category: Reading +match: + regex: belebele_.+ +metric: acc_norm +filter: none +score: + scale: 1 +normalize: + min: 0.25 + max: 1 + clip: true + basis: uniform_choice + note: All language variants score four labels A/B/C/D; uniform guessing gives 1/4. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/belebele/_default_template_yaml +languages: + - language: bul_Cyrl + scope: single + tasks: [belebele_bul_Cyrl] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ces_Latn + scope: single + tasks: [belebele_ces_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: dan_Latn + scope: single + tasks: [belebele_dan_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [belebele_deu_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ell_Grek + scope: single + tasks: [belebele_ell_Grek] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [belebele_eng_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: est_Latn + scope: single + tasks: [belebele_est_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fin_Latn + scope: single + tasks: [belebele_fin_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [belebele_fra_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hrv_Latn + scope: single + tasks: [belebele_hrv_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hun_Latn + scope: single + tasks: [belebele_hun_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [belebele_ita_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lit_Latn + scope: single + tasks: [belebele_lit_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lav_Latn + scope: single + tasks: [belebele_lvs_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: mlt_Latn + scope: single + tasks: [belebele_mlt_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [belebele_nld_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nor_Latn + scope: single + tasks: [belebele_nob_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: pol_Latn + scope: single + tasks: [belebele_pol_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [belebele_por_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ron_Latn + scope: single + tasks: [belebele_ron_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slk_Latn + scope: single + tasks: [belebele_slk_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slv_Latn + scope: single + tasks: [belebele_slv_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [belebele_spa_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: swe_Latn + scope: single + tasks: [belebele_swe_Latn] + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/boolq.yaml b/configs/evals/boolq.yaml new file mode 100644 index 0000000..1687e42 --- /dev/null +++ b/configs/evals/boolq.yaml @@ -0,0 +1,22 @@ +name: BoolQ +category: Reading +match: + name: boolq +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.5 + max: 1 + clip: true + basis: uniform_choice + note: The task scores no/yes. This is a uniform-guess baseline of 1/2, not a majority-class baseline. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/super_glue/boolq/default.yaml +languages: + - language: eng_Latn + scope: single + tasks: [boolq] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/commonsenseqa.yaml b/configs/evals/commonsenseqa.yaml new file mode 100644 index 0000000..35f982b --- /dev/null +++ b/configs/evals/commonsenseqa.yaml @@ -0,0 +1,22 @@ +name: CommonsenseQA +category: Commonsense +match: + name: commonsense_qa +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.2 + max: 1 + clip: true + basis: uniform_choice + note: The task scores five labels A through E; uniform guessing gives 1/5. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/commonsense_qa/default.yaml +languages: + - language: eng_Latn + scope: single + tasks: [commonsense_qa] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/copa.yaml b/configs/evals/copa.yaml new file mode 100644 index 0000000..4cafe65 --- /dev/null +++ b/configs/evals/copa.yaml @@ -0,0 +1,22 @@ +name: COPA +category: Commonsense +match: + name: copa +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.5 + max: 1 + clip: true + basis: uniform_choice + note: Two possible causes or effects are scored; uniform guessing gives 1/2. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/super_glue/copa/utils.py +languages: + - language: eng_Latn + scope: single + tasks: [copa] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/coqa.yaml b/configs/evals/coqa.yaml new file mode 100644 index 0000000..ffab5e7 --- /dev/null +++ b/configs/evals/coqa.yaml @@ -0,0 +1,22 @@ +name: CoQA +category: Reading +match: + name: coqa +metric: f1 +filter: none +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Token-overlap F1 for generated conversational answers has no universal chance floor. + sources: + - https://stanfordnlp.github.io/coqa/ +languages: + - language: eng_Latn + scope: single + tasks: [coqa] + evidence: https://stanfordnlp.github.io/coqa/ + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/cs_algorithms.yaml b/configs/evals/cs_algorithms.yaml new file mode 100644 index 0000000..4a39533 --- /dev/null +++ b/configs/evals/cs_algorithms.yaml @@ -0,0 +1,22 @@ +name: CS Algorithms +category: Code +match: + name: bigbench_cs_algorithms_generate_until +metric: exact_match +filter: strict-match +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Free-response exact match over algorithm outputs; no defined uniform answer space. + sources: + - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/cs_algorithms +languages: + - language: eng_Latn + scope: single + tasks: [bigbench_cs_algorithms_generate_until] + evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/cs_algorithms + note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/dyck_languages.yaml b/configs/evals/dyck_languages.yaml new file mode 100644 index 0000000..93dd3f2 --- /dev/null +++ b/configs/evals/dyck_languages.yaml @@ -0,0 +1,23 @@ +name: Dyck languages +category: Math +match: + name: bigbench_dyck_languages_generate_until +metric: exact_match +filter: strict-match +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Exact-match generated bracket completions have length-dependent spaces; no single fixed baseline is + assigned. + sources: + - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/dyck_languages +languages: + - language: eng_Latn + scope: single + tasks: [bigbench_dyck_languages_generate_until] + evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/dyck_languages + note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/flores200.yaml b/configs/evals/flores200.yaml new file mode 100644 index 0000000..0b9a3f8 --- /dev/null +++ b/configs/evals/flores200.yaml @@ -0,0 +1,510 @@ +name: FLORES200 +category: Translation +match: + regex: flores200:.+ +metric: chrf++ +filter: rescored +score: + scale: 100 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Select the explicit chrf++ field from the rescored export (preferred over chrf, then bleu). Retain + native 0–100 points with no chance correction. The CSV does not record the rescoring implementation signature; + the separate chrf++ label identifies the intended metric. + sources: + - https://github.com/mjpost/sacrebleu/blob/master/sacrebleu/metrics/chrf.py + - https://github.com/facebookresearch/flores +languages: + - source_language: sqi_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:als_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: bos_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:bos_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: bul_Cyrl + target_language: eng_Latn + scope: translation + tasks: ['flores200:bul_Cyrl-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: cat_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:cat_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: ces_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:ces_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: dan_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:dan_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: deu_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:deu_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: ell_Grek + target_language: eng_Latn + scope: translation + tasks: ['flores200:ell_Grek-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: sqi_Latn + scope: translation + tasks: ['flores200:eng_Latn-als_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: bos_Latn + scope: translation + tasks: ['flores200:eng_Latn-bos_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: bul_Cyrl + scope: translation + tasks: ['flores200:eng_Latn-bul_Cyrl'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: cat_Latn + scope: translation + tasks: ['flores200:eng_Latn-cat_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: ces_Latn + scope: translation + tasks: ['flores200:eng_Latn-ces_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: dan_Latn + scope: translation + tasks: ['flores200:eng_Latn-dan_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: deu_Latn + scope: translation + tasks: ['flores200:eng_Latn-deu_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: ell_Grek + scope: translation + tasks: ['flores200:eng_Latn-ell_Grek'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: est_Latn + scope: translation + tasks: ['flores200:eng_Latn-est_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: eus_Latn + scope: translation + tasks: ['flores200:eng_Latn-eus_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: fin_Latn + scope: translation + tasks: ['flores200:eng_Latn-fin_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: fra_Latn + scope: translation + tasks: ['flores200:eng_Latn-fra_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: gle_Latn + scope: translation + tasks: ['flores200:eng_Latn-gle_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: glg_Latn + scope: translation + tasks: ['flores200:eng_Latn-glg_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: hrv_Latn + scope: translation + tasks: ['flores200:eng_Latn-hrv_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: hun_Latn + scope: translation + tasks: ['flores200:eng_Latn-hun_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: isl_Latn + scope: translation + tasks: ['flores200:eng_Latn-isl_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: ita_Latn + scope: translation + tasks: ['flores200:eng_Latn-ita_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: kat_Geor + scope: translation + tasks: ['flores200:eng_Latn-kat_Geor'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: lit_Latn + scope: translation + tasks: ['flores200:eng_Latn-lit_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: lav_Latn + scope: translation + tasks: ['flores200:eng_Latn-lvs_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: mkd_Cyrl + scope: translation + tasks: ['flores200:eng_Latn-mkd_Cyrl'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: mlt_Latn + scope: translation + tasks: ['flores200:eng_Latn-mlt_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: nld_Latn + scope: translation + tasks: ['flores200:eng_Latn-nld_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: nor_Latn + scope: translation + tasks: ['flores200:eng_Latn-nob_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: pol_Latn + scope: translation + tasks: ['flores200:eng_Latn-pol_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: por_Latn + scope: translation + tasks: ['flores200:eng_Latn-por_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: ron_Latn + scope: translation + tasks: ['flores200:eng_Latn-ron_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: slk_Latn + scope: translation + tasks: ['flores200:eng_Latn-slk_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: slv_Latn + scope: translation + tasks: ['flores200:eng_Latn-slv_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: spa_Latn + scope: translation + tasks: ['flores200:eng_Latn-spa_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: srp_Cyrl + scope: translation + tasks: ['flores200:eng_Latn-srp_Cyrl'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: swe_Latn + scope: translation + tasks: ['flores200:eng_Latn-swe_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: tur_Latn + scope: translation + tasks: ['flores200:eng_Latn-tur_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eng_Latn + target_language: ukr_Cyrl + scope: translation + tasks: ['flores200:eng_Latn-ukr_Cyrl'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: est_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:est_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: eus_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:eus_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: fin_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:fin_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: fra_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:fra_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: gle_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:gle_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: glg_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:glg_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: hrv_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:hrv_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: hun_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:hun_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: isl_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:isl_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: ita_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:ita_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: kat_Geor + target_language: eng_Latn + scope: translation + tasks: ['flores200:kat_Geor-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: lit_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:lit_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: lav_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:lvs_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: mkd_Cyrl + target_language: eng_Latn + scope: translation + tasks: ['flores200:mkd_Cyrl-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: mlt_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:mlt_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: nld_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:nld_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: nor_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:nob_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: pol_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:pol_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: por_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:por_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: ron_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:ron_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: slk_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:slk_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: slv_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:slv_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: spa_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:spa_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: srp_Cyrl + target_language: eng_Latn + scope: translation + tasks: ['flores200:srp_Cyrl-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: swe_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:swe_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: tur_Latn + target_language: eng_Latn + scope: translation + tasks: ['flores200:tur_Latn-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. + - source_language: ukr_Cyrl + target_language: eng_Latn + scope: translation + tasks: ['flores200:ukr_Cyrl-eng_Latn'] + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. diff --git a/configs/evals/global_mmlu.yaml b/configs/evals/global_mmlu.yaml new file mode 100644 index 0000000..4b09e43 --- /dev/null +++ b/configs/evals/global_mmlu.yaml @@ -0,0 +1,404 @@ +name: Global MMLU +category: Knowledge +match: + regex: global_mmlu_.+ +select: + regex: global_mmlu_full_[a-z]+ +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.25 + max: 1 + clip: true + basis: uniform_choice + note: Global MMLU uses four choices and one correct label. Uniform guessing gives 1/4, including subject + summaries. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/global_mmlu/default/en/_en_template_yaml +languages: + - language: ces_Latn + scope: single + tasks: [global_mmlu_full_cs, global_mmlu_full_cs_abstract_algebra, global_mmlu_full_cs_anatomy, global_mmlu_full_cs_astronomy, + global_mmlu_full_cs_business_ethics, global_mmlu_full_cs_clinical_knowledge, global_mmlu_full_cs_college_biology, + global_mmlu_full_cs_college_chemistry, global_mmlu_full_cs_college_computer_science, global_mmlu_full_cs_college_mathematics, + global_mmlu_full_cs_college_medicine, global_mmlu_full_cs_college_physics, global_mmlu_full_cs_computer_security, + global_mmlu_full_cs_conceptual_physics, global_mmlu_full_cs_econometrics, global_mmlu_full_cs_electrical_engineering, + global_mmlu_full_cs_elementary_mathematics, global_mmlu_full_cs_formal_logic, global_mmlu_full_cs_global_facts, + global_mmlu_full_cs_high_school_biology, global_mmlu_full_cs_high_school_chemistry, global_mmlu_full_cs_high_school_computer_science, + global_mmlu_full_cs_high_school_european_history, global_mmlu_full_cs_high_school_geography, global_mmlu_full_cs_high_school_government_and_politics, + global_mmlu_full_cs_high_school_macroeconomics, global_mmlu_full_cs_high_school_mathematics, global_mmlu_full_cs_high_school_microeconomics, + global_mmlu_full_cs_high_school_physics, global_mmlu_full_cs_high_school_psychology, global_mmlu_full_cs_high_school_statistics, + global_mmlu_full_cs_high_school_us_history, global_mmlu_full_cs_high_school_world_history, global_mmlu_full_cs_human_aging, + global_mmlu_full_cs_human_sexuality, global_mmlu_full_cs_humanities, global_mmlu_full_cs_international_law, + global_mmlu_full_cs_jurisprudence, global_mmlu_full_cs_logical_fallacies, global_mmlu_full_cs_machine_learning, + global_mmlu_full_cs_management, global_mmlu_full_cs_marketing, global_mmlu_full_cs_medical_genetics, + global_mmlu_full_cs_miscellaneous, global_mmlu_full_cs_moral_disputes, global_mmlu_full_cs_moral_scenarios, + global_mmlu_full_cs_nutrition, global_mmlu_full_cs_other, global_mmlu_full_cs_philosophy, global_mmlu_full_cs_prehistory, + global_mmlu_full_cs_professional_accounting, global_mmlu_full_cs_professional_law, global_mmlu_full_cs_professional_medicine, + global_mmlu_full_cs_professional_psychology, global_mmlu_full_cs_public_relations, global_mmlu_full_cs_security_studies, + global_mmlu_full_cs_social_sciences, global_mmlu_full_cs_sociology, global_mmlu_full_cs_stem, global_mmlu_full_cs_us_foreign_policy, + global_mmlu_full_cs_virology, global_mmlu_full_cs_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [global_mmlu_full_de, global_mmlu_full_de_abstract_algebra, global_mmlu_full_de_anatomy, global_mmlu_full_de_astronomy, + global_mmlu_full_de_business_ethics, global_mmlu_full_de_clinical_knowledge, global_mmlu_full_de_college_biology, + global_mmlu_full_de_college_chemistry, global_mmlu_full_de_college_computer_science, global_mmlu_full_de_college_mathematics, + global_mmlu_full_de_college_medicine, global_mmlu_full_de_college_physics, global_mmlu_full_de_computer_security, + global_mmlu_full_de_conceptual_physics, global_mmlu_full_de_econometrics, global_mmlu_full_de_electrical_engineering, + global_mmlu_full_de_elementary_mathematics, global_mmlu_full_de_formal_logic, global_mmlu_full_de_global_facts, + global_mmlu_full_de_high_school_biology, global_mmlu_full_de_high_school_chemistry, global_mmlu_full_de_high_school_computer_science, + global_mmlu_full_de_high_school_european_history, global_mmlu_full_de_high_school_geography, global_mmlu_full_de_high_school_government_and_politics, + global_mmlu_full_de_high_school_macroeconomics, global_mmlu_full_de_high_school_mathematics, global_mmlu_full_de_high_school_microeconomics, + global_mmlu_full_de_high_school_physics, global_mmlu_full_de_high_school_psychology, global_mmlu_full_de_high_school_statistics, + global_mmlu_full_de_high_school_us_history, global_mmlu_full_de_high_school_world_history, global_mmlu_full_de_human_aging, + global_mmlu_full_de_human_sexuality, global_mmlu_full_de_humanities, global_mmlu_full_de_international_law, + global_mmlu_full_de_jurisprudence, global_mmlu_full_de_logical_fallacies, global_mmlu_full_de_machine_learning, + global_mmlu_full_de_management, global_mmlu_full_de_marketing, global_mmlu_full_de_medical_genetics, + global_mmlu_full_de_miscellaneous, global_mmlu_full_de_moral_disputes, global_mmlu_full_de_moral_scenarios, + global_mmlu_full_de_nutrition, global_mmlu_full_de_other, global_mmlu_full_de_philosophy, global_mmlu_full_de_prehistory, + global_mmlu_full_de_professional_accounting, global_mmlu_full_de_professional_law, global_mmlu_full_de_professional_medicine, + global_mmlu_full_de_professional_psychology, global_mmlu_full_de_public_relations, global_mmlu_full_de_security_studies, + global_mmlu_full_de_social_sciences, global_mmlu_full_de_sociology, global_mmlu_full_de_stem, global_mmlu_full_de_us_foreign_policy, + global_mmlu_full_de_virology, global_mmlu_full_de_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ell_Grek + scope: single + tasks: [global_mmlu_full_el, global_mmlu_full_el_abstract_algebra, global_mmlu_full_el_anatomy, global_mmlu_full_el_astronomy, + global_mmlu_full_el_business_ethics, global_mmlu_full_el_clinical_knowledge, global_mmlu_full_el_college_biology, + global_mmlu_full_el_college_chemistry, global_mmlu_full_el_college_computer_science, global_mmlu_full_el_college_mathematics, + global_mmlu_full_el_college_medicine, global_mmlu_full_el_college_physics, global_mmlu_full_el_computer_security, + global_mmlu_full_el_conceptual_physics, global_mmlu_full_el_econometrics, global_mmlu_full_el_electrical_engineering, + global_mmlu_full_el_elementary_mathematics, global_mmlu_full_el_formal_logic, global_mmlu_full_el_global_facts, + global_mmlu_full_el_high_school_biology, global_mmlu_full_el_high_school_chemistry, global_mmlu_full_el_high_school_computer_science, + global_mmlu_full_el_high_school_european_history, global_mmlu_full_el_high_school_geography, global_mmlu_full_el_high_school_government_and_politics, + global_mmlu_full_el_high_school_macroeconomics, global_mmlu_full_el_high_school_mathematics, global_mmlu_full_el_high_school_microeconomics, + global_mmlu_full_el_high_school_physics, global_mmlu_full_el_high_school_psychology, global_mmlu_full_el_high_school_statistics, + global_mmlu_full_el_high_school_us_history, global_mmlu_full_el_high_school_world_history, global_mmlu_full_el_human_aging, + global_mmlu_full_el_human_sexuality, global_mmlu_full_el_humanities, global_mmlu_full_el_international_law, + global_mmlu_full_el_jurisprudence, global_mmlu_full_el_logical_fallacies, global_mmlu_full_el_machine_learning, + global_mmlu_full_el_management, global_mmlu_full_el_marketing, global_mmlu_full_el_medical_genetics, + global_mmlu_full_el_miscellaneous, global_mmlu_full_el_moral_disputes, global_mmlu_full_el_moral_scenarios, + global_mmlu_full_el_nutrition, global_mmlu_full_el_other, global_mmlu_full_el_philosophy, global_mmlu_full_el_prehistory, + global_mmlu_full_el_professional_accounting, global_mmlu_full_el_professional_law, global_mmlu_full_el_professional_medicine, + global_mmlu_full_el_professional_psychology, global_mmlu_full_el_public_relations, global_mmlu_full_el_security_studies, + global_mmlu_full_el_social_sciences, global_mmlu_full_el_sociology, global_mmlu_full_el_stem, global_mmlu_full_el_us_foreign_policy, + global_mmlu_full_el_virology, global_mmlu_full_el_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [global_mmlu_full_en, global_mmlu_full_en_abstract_algebra, global_mmlu_full_en_anatomy, global_mmlu_full_en_astronomy, + global_mmlu_full_en_business_ethics, global_mmlu_full_en_clinical_knowledge, global_mmlu_full_en_college_biology, + global_mmlu_full_en_college_chemistry, global_mmlu_full_en_college_computer_science, global_mmlu_full_en_college_mathematics, + global_mmlu_full_en_college_medicine, global_mmlu_full_en_college_physics, global_mmlu_full_en_computer_security, + global_mmlu_full_en_conceptual_physics, global_mmlu_full_en_econometrics, global_mmlu_full_en_electrical_engineering, + global_mmlu_full_en_elementary_mathematics, global_mmlu_full_en_formal_logic, global_mmlu_full_en_global_facts, + global_mmlu_full_en_high_school_biology, global_mmlu_full_en_high_school_chemistry, global_mmlu_full_en_high_school_computer_science, + global_mmlu_full_en_high_school_european_history, global_mmlu_full_en_high_school_geography, global_mmlu_full_en_high_school_government_and_politics, + global_mmlu_full_en_high_school_macroeconomics, global_mmlu_full_en_high_school_mathematics, global_mmlu_full_en_high_school_microeconomics, + global_mmlu_full_en_high_school_physics, global_mmlu_full_en_high_school_psychology, global_mmlu_full_en_high_school_statistics, + global_mmlu_full_en_high_school_us_history, global_mmlu_full_en_high_school_world_history, global_mmlu_full_en_human_aging, + global_mmlu_full_en_human_sexuality, global_mmlu_full_en_humanities, global_mmlu_full_en_international_law, + global_mmlu_full_en_jurisprudence, global_mmlu_full_en_logical_fallacies, global_mmlu_full_en_machine_learning, + global_mmlu_full_en_management, global_mmlu_full_en_marketing, global_mmlu_full_en_medical_genetics, + global_mmlu_full_en_miscellaneous, global_mmlu_full_en_moral_disputes, global_mmlu_full_en_moral_scenarios, + global_mmlu_full_en_nutrition, global_mmlu_full_en_other, global_mmlu_full_en_philosophy, global_mmlu_full_en_prehistory, + global_mmlu_full_en_professional_accounting, global_mmlu_full_en_professional_law, global_mmlu_full_en_professional_medicine, + global_mmlu_full_en_professional_psychology, global_mmlu_full_en_public_relations, global_mmlu_full_en_security_studies, + global_mmlu_full_en_social_sciences, global_mmlu_full_en_sociology, global_mmlu_full_en_stem, global_mmlu_full_en_us_foreign_policy, + global_mmlu_full_en_virology, global_mmlu_full_en_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [global_mmlu_full_es, global_mmlu_full_es_abstract_algebra, global_mmlu_full_es_anatomy, global_mmlu_full_es_astronomy, + global_mmlu_full_es_business_ethics, global_mmlu_full_es_clinical_knowledge, global_mmlu_full_es_college_biology, + global_mmlu_full_es_college_chemistry, global_mmlu_full_es_college_computer_science, global_mmlu_full_es_college_mathematics, + global_mmlu_full_es_college_medicine, global_mmlu_full_es_college_physics, global_mmlu_full_es_computer_security, + global_mmlu_full_es_conceptual_physics, global_mmlu_full_es_econometrics, global_mmlu_full_es_electrical_engineering, + global_mmlu_full_es_elementary_mathematics, global_mmlu_full_es_formal_logic, global_mmlu_full_es_global_facts, + global_mmlu_full_es_high_school_biology, global_mmlu_full_es_high_school_chemistry, global_mmlu_full_es_high_school_computer_science, + global_mmlu_full_es_high_school_european_history, global_mmlu_full_es_high_school_geography, global_mmlu_full_es_high_school_government_and_politics, + global_mmlu_full_es_high_school_macroeconomics, global_mmlu_full_es_high_school_mathematics, global_mmlu_full_es_high_school_microeconomics, + global_mmlu_full_es_high_school_physics, global_mmlu_full_es_high_school_psychology, global_mmlu_full_es_high_school_statistics, + global_mmlu_full_es_high_school_us_history, global_mmlu_full_es_high_school_world_history, global_mmlu_full_es_human_aging, + global_mmlu_full_es_human_sexuality, global_mmlu_full_es_humanities, global_mmlu_full_es_international_law, + global_mmlu_full_es_jurisprudence, global_mmlu_full_es_logical_fallacies, global_mmlu_full_es_machine_learning, + global_mmlu_full_es_management, global_mmlu_full_es_marketing, global_mmlu_full_es_medical_genetics, + global_mmlu_full_es_miscellaneous, global_mmlu_full_es_moral_disputes, global_mmlu_full_es_moral_scenarios, + global_mmlu_full_es_nutrition, global_mmlu_full_es_other, global_mmlu_full_es_philosophy, global_mmlu_full_es_prehistory, + global_mmlu_full_es_professional_accounting, global_mmlu_full_es_professional_law, global_mmlu_full_es_professional_medicine, + global_mmlu_full_es_professional_psychology, global_mmlu_full_es_public_relations, global_mmlu_full_es_security_studies, + global_mmlu_full_es_social_sciences, global_mmlu_full_es_sociology, global_mmlu_full_es_stem, global_mmlu_full_es_us_foreign_policy, + global_mmlu_full_es_virology, global_mmlu_full_es_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [global_mmlu_full_fr, global_mmlu_full_fr_abstract_algebra, global_mmlu_full_fr_anatomy, global_mmlu_full_fr_astronomy, + global_mmlu_full_fr_business_ethics, global_mmlu_full_fr_clinical_knowledge, global_mmlu_full_fr_college_biology, + global_mmlu_full_fr_college_chemistry, global_mmlu_full_fr_college_computer_science, global_mmlu_full_fr_college_mathematics, + global_mmlu_full_fr_college_medicine, global_mmlu_full_fr_college_physics, global_mmlu_full_fr_computer_security, + global_mmlu_full_fr_conceptual_physics, global_mmlu_full_fr_econometrics, global_mmlu_full_fr_electrical_engineering, + global_mmlu_full_fr_elementary_mathematics, global_mmlu_full_fr_formal_logic, global_mmlu_full_fr_global_facts, + global_mmlu_full_fr_high_school_biology, global_mmlu_full_fr_high_school_chemistry, global_mmlu_full_fr_high_school_computer_science, + global_mmlu_full_fr_high_school_european_history, global_mmlu_full_fr_high_school_geography, global_mmlu_full_fr_high_school_government_and_politics, + global_mmlu_full_fr_high_school_macroeconomics, global_mmlu_full_fr_high_school_mathematics, global_mmlu_full_fr_high_school_microeconomics, + global_mmlu_full_fr_high_school_physics, global_mmlu_full_fr_high_school_psychology, global_mmlu_full_fr_high_school_statistics, + global_mmlu_full_fr_high_school_us_history, global_mmlu_full_fr_high_school_world_history, global_mmlu_full_fr_human_aging, + global_mmlu_full_fr_human_sexuality, global_mmlu_full_fr_humanities, global_mmlu_full_fr_international_law, + global_mmlu_full_fr_jurisprudence, global_mmlu_full_fr_logical_fallacies, global_mmlu_full_fr_machine_learning, + global_mmlu_full_fr_management, global_mmlu_full_fr_marketing, global_mmlu_full_fr_medical_genetics, + global_mmlu_full_fr_miscellaneous, global_mmlu_full_fr_moral_disputes, global_mmlu_full_fr_moral_scenarios, + global_mmlu_full_fr_nutrition, global_mmlu_full_fr_other, global_mmlu_full_fr_philosophy, global_mmlu_full_fr_prehistory, + global_mmlu_full_fr_professional_accounting, global_mmlu_full_fr_professional_law, global_mmlu_full_fr_professional_medicine, + global_mmlu_full_fr_professional_psychology, global_mmlu_full_fr_public_relations, global_mmlu_full_fr_security_studies, + global_mmlu_full_fr_social_sciences, global_mmlu_full_fr_sociology, global_mmlu_full_fr_stem, global_mmlu_full_fr_us_foreign_policy, + global_mmlu_full_fr_virology, global_mmlu_full_fr_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [global_mmlu_full_it, global_mmlu_full_it_abstract_algebra, global_mmlu_full_it_anatomy, global_mmlu_full_it_astronomy, + global_mmlu_full_it_business_ethics, global_mmlu_full_it_clinical_knowledge, global_mmlu_full_it_college_biology, + global_mmlu_full_it_college_chemistry, global_mmlu_full_it_college_computer_science, global_mmlu_full_it_college_mathematics, + global_mmlu_full_it_college_medicine, global_mmlu_full_it_college_physics, global_mmlu_full_it_computer_security, + global_mmlu_full_it_conceptual_physics, global_mmlu_full_it_econometrics, global_mmlu_full_it_electrical_engineering, + global_mmlu_full_it_elementary_mathematics, global_mmlu_full_it_formal_logic, global_mmlu_full_it_global_facts, + global_mmlu_full_it_high_school_biology, global_mmlu_full_it_high_school_chemistry, global_mmlu_full_it_high_school_computer_science, + global_mmlu_full_it_high_school_european_history, global_mmlu_full_it_high_school_geography, global_mmlu_full_it_high_school_government_and_politics, + global_mmlu_full_it_high_school_macroeconomics, global_mmlu_full_it_high_school_mathematics, global_mmlu_full_it_high_school_microeconomics, + global_mmlu_full_it_high_school_physics, global_mmlu_full_it_high_school_psychology, global_mmlu_full_it_high_school_statistics, + global_mmlu_full_it_high_school_us_history, global_mmlu_full_it_high_school_world_history, global_mmlu_full_it_human_aging, + global_mmlu_full_it_human_sexuality, global_mmlu_full_it_humanities, global_mmlu_full_it_international_law, + global_mmlu_full_it_jurisprudence, global_mmlu_full_it_logical_fallacies, global_mmlu_full_it_machine_learning, + global_mmlu_full_it_management, global_mmlu_full_it_marketing, global_mmlu_full_it_medical_genetics, + global_mmlu_full_it_miscellaneous, global_mmlu_full_it_moral_disputes, global_mmlu_full_it_moral_scenarios, + global_mmlu_full_it_nutrition, global_mmlu_full_it_other, global_mmlu_full_it_philosophy, global_mmlu_full_it_prehistory, + global_mmlu_full_it_professional_accounting, global_mmlu_full_it_professional_law, global_mmlu_full_it_professional_medicine, + global_mmlu_full_it_professional_psychology, global_mmlu_full_it_public_relations, global_mmlu_full_it_security_studies, + global_mmlu_full_it_social_sciences, global_mmlu_full_it_sociology, global_mmlu_full_it_stem, global_mmlu_full_it_us_foreign_policy, + global_mmlu_full_it_virology, global_mmlu_full_it_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lit_Latn + scope: single + tasks: [global_mmlu_full_lt, global_mmlu_full_lt_abstract_algebra, global_mmlu_full_lt_anatomy, global_mmlu_full_lt_astronomy, + global_mmlu_full_lt_business_ethics, global_mmlu_full_lt_clinical_knowledge, global_mmlu_full_lt_college_biology, + global_mmlu_full_lt_college_chemistry, global_mmlu_full_lt_college_computer_science, global_mmlu_full_lt_college_mathematics, + global_mmlu_full_lt_college_medicine, global_mmlu_full_lt_college_physics, global_mmlu_full_lt_computer_security, + global_mmlu_full_lt_conceptual_physics, global_mmlu_full_lt_econometrics, global_mmlu_full_lt_electrical_engineering, + global_mmlu_full_lt_elementary_mathematics, global_mmlu_full_lt_formal_logic, global_mmlu_full_lt_global_facts, + global_mmlu_full_lt_high_school_biology, global_mmlu_full_lt_high_school_chemistry, global_mmlu_full_lt_high_school_computer_science, + global_mmlu_full_lt_high_school_european_history, global_mmlu_full_lt_high_school_geography, global_mmlu_full_lt_high_school_government_and_politics, + global_mmlu_full_lt_high_school_macroeconomics, global_mmlu_full_lt_high_school_mathematics, global_mmlu_full_lt_high_school_microeconomics, + global_mmlu_full_lt_high_school_physics, global_mmlu_full_lt_high_school_psychology, global_mmlu_full_lt_high_school_statistics, + global_mmlu_full_lt_high_school_us_history, global_mmlu_full_lt_high_school_world_history, global_mmlu_full_lt_human_aging, + global_mmlu_full_lt_human_sexuality, global_mmlu_full_lt_humanities, global_mmlu_full_lt_international_law, + global_mmlu_full_lt_jurisprudence, global_mmlu_full_lt_logical_fallacies, global_mmlu_full_lt_machine_learning, + global_mmlu_full_lt_management, global_mmlu_full_lt_marketing, global_mmlu_full_lt_medical_genetics, + global_mmlu_full_lt_miscellaneous, global_mmlu_full_lt_moral_disputes, global_mmlu_full_lt_moral_scenarios, + global_mmlu_full_lt_nutrition, global_mmlu_full_lt_other, global_mmlu_full_lt_philosophy, global_mmlu_full_lt_prehistory, + global_mmlu_full_lt_professional_accounting, global_mmlu_full_lt_professional_law, global_mmlu_full_lt_professional_medicine, + global_mmlu_full_lt_professional_psychology, global_mmlu_full_lt_public_relations, global_mmlu_full_lt_security_studies, + global_mmlu_full_lt_social_sciences, global_mmlu_full_lt_sociology, global_mmlu_full_lt_stem, global_mmlu_full_lt_us_foreign_policy, + global_mmlu_full_lt_virology, global_mmlu_full_lt_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [global_mmlu_full_nl, global_mmlu_full_nl_abstract_algebra, global_mmlu_full_nl_anatomy, global_mmlu_full_nl_astronomy, + global_mmlu_full_nl_business_ethics, global_mmlu_full_nl_clinical_knowledge, global_mmlu_full_nl_college_biology, + global_mmlu_full_nl_college_chemistry, global_mmlu_full_nl_college_computer_science, global_mmlu_full_nl_college_mathematics, + global_mmlu_full_nl_college_medicine, global_mmlu_full_nl_college_physics, global_mmlu_full_nl_computer_security, + global_mmlu_full_nl_conceptual_physics, global_mmlu_full_nl_econometrics, global_mmlu_full_nl_electrical_engineering, + global_mmlu_full_nl_elementary_mathematics, global_mmlu_full_nl_formal_logic, global_mmlu_full_nl_global_facts, + global_mmlu_full_nl_high_school_biology, global_mmlu_full_nl_high_school_chemistry, global_mmlu_full_nl_high_school_computer_science, + global_mmlu_full_nl_high_school_european_history, global_mmlu_full_nl_high_school_geography, global_mmlu_full_nl_high_school_government_and_politics, + global_mmlu_full_nl_high_school_macroeconomics, global_mmlu_full_nl_high_school_mathematics, global_mmlu_full_nl_high_school_microeconomics, + global_mmlu_full_nl_high_school_physics, global_mmlu_full_nl_high_school_psychology, global_mmlu_full_nl_high_school_statistics, + global_mmlu_full_nl_high_school_us_history, global_mmlu_full_nl_high_school_world_history, global_mmlu_full_nl_human_aging, + global_mmlu_full_nl_human_sexuality, global_mmlu_full_nl_humanities, global_mmlu_full_nl_international_law, + global_mmlu_full_nl_jurisprudence, global_mmlu_full_nl_logical_fallacies, global_mmlu_full_nl_machine_learning, + global_mmlu_full_nl_management, global_mmlu_full_nl_marketing, global_mmlu_full_nl_medical_genetics, + global_mmlu_full_nl_miscellaneous, global_mmlu_full_nl_moral_disputes, global_mmlu_full_nl_moral_scenarios, + global_mmlu_full_nl_nutrition, global_mmlu_full_nl_other, global_mmlu_full_nl_philosophy, global_mmlu_full_nl_prehistory, + global_mmlu_full_nl_professional_accounting, global_mmlu_full_nl_professional_law, global_mmlu_full_nl_professional_medicine, + global_mmlu_full_nl_professional_psychology, global_mmlu_full_nl_public_relations, global_mmlu_full_nl_security_studies, + global_mmlu_full_nl_social_sciences, global_mmlu_full_nl_sociology, global_mmlu_full_nl_stem, global_mmlu_full_nl_us_foreign_policy, + global_mmlu_full_nl_virology, global_mmlu_full_nl_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: pol_Latn + scope: single + tasks: [global_mmlu_full_pl, global_mmlu_full_pl_abstract_algebra, global_mmlu_full_pl_anatomy, global_mmlu_full_pl_astronomy, + global_mmlu_full_pl_business_ethics, global_mmlu_full_pl_clinical_knowledge, global_mmlu_full_pl_college_biology, + global_mmlu_full_pl_college_chemistry, global_mmlu_full_pl_college_computer_science, global_mmlu_full_pl_college_mathematics, + global_mmlu_full_pl_college_medicine, global_mmlu_full_pl_college_physics, global_mmlu_full_pl_computer_security, + global_mmlu_full_pl_conceptual_physics, global_mmlu_full_pl_econometrics, global_mmlu_full_pl_electrical_engineering, + global_mmlu_full_pl_elementary_mathematics, global_mmlu_full_pl_formal_logic, global_mmlu_full_pl_global_facts, + global_mmlu_full_pl_high_school_biology, global_mmlu_full_pl_high_school_chemistry, global_mmlu_full_pl_high_school_computer_science, + global_mmlu_full_pl_high_school_european_history, global_mmlu_full_pl_high_school_geography, global_mmlu_full_pl_high_school_government_and_politics, + global_mmlu_full_pl_high_school_macroeconomics, global_mmlu_full_pl_high_school_mathematics, global_mmlu_full_pl_high_school_microeconomics, + global_mmlu_full_pl_high_school_physics, global_mmlu_full_pl_high_school_psychology, global_mmlu_full_pl_high_school_statistics, + global_mmlu_full_pl_high_school_us_history, global_mmlu_full_pl_high_school_world_history, global_mmlu_full_pl_human_aging, + global_mmlu_full_pl_human_sexuality, global_mmlu_full_pl_humanities, global_mmlu_full_pl_international_law, + global_mmlu_full_pl_jurisprudence, global_mmlu_full_pl_logical_fallacies, global_mmlu_full_pl_machine_learning, + global_mmlu_full_pl_management, global_mmlu_full_pl_marketing, global_mmlu_full_pl_medical_genetics, + global_mmlu_full_pl_miscellaneous, global_mmlu_full_pl_moral_disputes, global_mmlu_full_pl_moral_scenarios, + global_mmlu_full_pl_nutrition, global_mmlu_full_pl_other, global_mmlu_full_pl_philosophy, global_mmlu_full_pl_prehistory, + global_mmlu_full_pl_professional_accounting, global_mmlu_full_pl_professional_law, global_mmlu_full_pl_professional_medicine, + global_mmlu_full_pl_professional_psychology, global_mmlu_full_pl_public_relations, global_mmlu_full_pl_security_studies, + global_mmlu_full_pl_social_sciences, global_mmlu_full_pl_sociology, global_mmlu_full_pl_stem, global_mmlu_full_pl_us_foreign_policy, + global_mmlu_full_pl_virology, global_mmlu_full_pl_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [global_mmlu_full_pt, global_mmlu_full_pt_abstract_algebra, global_mmlu_full_pt_anatomy, global_mmlu_full_pt_astronomy, + global_mmlu_full_pt_business_ethics, global_mmlu_full_pt_clinical_knowledge, global_mmlu_full_pt_college_biology, + global_mmlu_full_pt_college_chemistry, global_mmlu_full_pt_college_computer_science, global_mmlu_full_pt_college_mathematics, + global_mmlu_full_pt_college_medicine, global_mmlu_full_pt_college_physics, global_mmlu_full_pt_computer_security, + global_mmlu_full_pt_conceptual_physics, global_mmlu_full_pt_econometrics, global_mmlu_full_pt_electrical_engineering, + global_mmlu_full_pt_elementary_mathematics, global_mmlu_full_pt_formal_logic, global_mmlu_full_pt_global_facts, + global_mmlu_full_pt_high_school_biology, global_mmlu_full_pt_high_school_chemistry, global_mmlu_full_pt_high_school_computer_science, + global_mmlu_full_pt_high_school_european_history, global_mmlu_full_pt_high_school_geography, global_mmlu_full_pt_high_school_government_and_politics, + global_mmlu_full_pt_high_school_macroeconomics, global_mmlu_full_pt_high_school_mathematics, global_mmlu_full_pt_high_school_microeconomics, + global_mmlu_full_pt_high_school_physics, global_mmlu_full_pt_high_school_psychology, global_mmlu_full_pt_high_school_statistics, + global_mmlu_full_pt_high_school_us_history, global_mmlu_full_pt_high_school_world_history, global_mmlu_full_pt_human_aging, + global_mmlu_full_pt_human_sexuality, global_mmlu_full_pt_humanities, global_mmlu_full_pt_international_law, + global_mmlu_full_pt_jurisprudence, global_mmlu_full_pt_logical_fallacies, global_mmlu_full_pt_machine_learning, + global_mmlu_full_pt_management, global_mmlu_full_pt_marketing, global_mmlu_full_pt_medical_genetics, + global_mmlu_full_pt_miscellaneous, global_mmlu_full_pt_moral_disputes, global_mmlu_full_pt_moral_scenarios, + global_mmlu_full_pt_nutrition, global_mmlu_full_pt_other, global_mmlu_full_pt_philosophy, global_mmlu_full_pt_prehistory, + global_mmlu_full_pt_professional_accounting, global_mmlu_full_pt_professional_law, global_mmlu_full_pt_professional_medicine, + global_mmlu_full_pt_professional_psychology, global_mmlu_full_pt_public_relations, global_mmlu_full_pt_security_studies, + global_mmlu_full_pt_social_sciences, global_mmlu_full_pt_sociology, global_mmlu_full_pt_stem, global_mmlu_full_pt_us_foreign_policy, + global_mmlu_full_pt_virology, global_mmlu_full_pt_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ron_Latn + scope: single + tasks: [global_mmlu_full_ro, global_mmlu_full_ro_abstract_algebra, global_mmlu_full_ro_anatomy, global_mmlu_full_ro_astronomy, + global_mmlu_full_ro_business_ethics, global_mmlu_full_ro_clinical_knowledge, global_mmlu_full_ro_college_biology, + global_mmlu_full_ro_college_chemistry, global_mmlu_full_ro_college_computer_science, global_mmlu_full_ro_college_mathematics, + global_mmlu_full_ro_college_medicine, global_mmlu_full_ro_college_physics, global_mmlu_full_ro_computer_security, + global_mmlu_full_ro_conceptual_physics, global_mmlu_full_ro_econometrics, global_mmlu_full_ro_electrical_engineering, + global_mmlu_full_ro_elementary_mathematics, global_mmlu_full_ro_formal_logic, global_mmlu_full_ro_global_facts, + global_mmlu_full_ro_high_school_biology, global_mmlu_full_ro_high_school_chemistry, global_mmlu_full_ro_high_school_computer_science, + global_mmlu_full_ro_high_school_european_history, global_mmlu_full_ro_high_school_geography, global_mmlu_full_ro_high_school_government_and_politics, + global_mmlu_full_ro_high_school_macroeconomics, global_mmlu_full_ro_high_school_mathematics, global_mmlu_full_ro_high_school_microeconomics, + global_mmlu_full_ro_high_school_physics, global_mmlu_full_ro_high_school_psychology, global_mmlu_full_ro_high_school_statistics, + global_mmlu_full_ro_high_school_us_history, global_mmlu_full_ro_high_school_world_history, global_mmlu_full_ro_human_aging, + global_mmlu_full_ro_human_sexuality, global_mmlu_full_ro_humanities, global_mmlu_full_ro_international_law, + global_mmlu_full_ro_jurisprudence, global_mmlu_full_ro_logical_fallacies, global_mmlu_full_ro_machine_learning, + global_mmlu_full_ro_management, global_mmlu_full_ro_marketing, global_mmlu_full_ro_medical_genetics, + global_mmlu_full_ro_miscellaneous, global_mmlu_full_ro_moral_disputes, global_mmlu_full_ro_moral_scenarios, + global_mmlu_full_ro_nutrition, global_mmlu_full_ro_other, global_mmlu_full_ro_philosophy, global_mmlu_full_ro_prehistory, + global_mmlu_full_ro_professional_accounting, global_mmlu_full_ro_professional_law, global_mmlu_full_ro_professional_medicine, + global_mmlu_full_ro_professional_psychology, global_mmlu_full_ro_public_relations, global_mmlu_full_ro_security_studies, + global_mmlu_full_ro_social_sciences, global_mmlu_full_ro_sociology, global_mmlu_full_ro_stem, global_mmlu_full_ro_us_foreign_policy, + global_mmlu_full_ro_virology, global_mmlu_full_ro_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: srp_Cyrl + scope: single + tasks: [global_mmlu_full_sr, global_mmlu_full_sr_abstract_algebra, global_mmlu_full_sr_anatomy, global_mmlu_full_sr_astronomy, + global_mmlu_full_sr_business_ethics, global_mmlu_full_sr_clinical_knowledge, global_mmlu_full_sr_college_biology, + global_mmlu_full_sr_college_chemistry, global_mmlu_full_sr_college_computer_science, global_mmlu_full_sr_college_mathematics, + global_mmlu_full_sr_college_medicine, global_mmlu_full_sr_college_physics, global_mmlu_full_sr_computer_security, + global_mmlu_full_sr_conceptual_physics, global_mmlu_full_sr_econometrics, global_mmlu_full_sr_electrical_engineering, + global_mmlu_full_sr_elementary_mathematics, global_mmlu_full_sr_formal_logic, global_mmlu_full_sr_global_facts, + global_mmlu_full_sr_high_school_biology, global_mmlu_full_sr_high_school_chemistry, global_mmlu_full_sr_high_school_computer_science, + global_mmlu_full_sr_high_school_european_history, global_mmlu_full_sr_high_school_geography, global_mmlu_full_sr_high_school_government_and_politics, + global_mmlu_full_sr_high_school_macroeconomics, global_mmlu_full_sr_high_school_mathematics, global_mmlu_full_sr_high_school_microeconomics, + global_mmlu_full_sr_high_school_physics, global_mmlu_full_sr_high_school_psychology, global_mmlu_full_sr_high_school_statistics, + global_mmlu_full_sr_high_school_us_history, global_mmlu_full_sr_high_school_world_history, global_mmlu_full_sr_human_aging, + global_mmlu_full_sr_human_sexuality, global_mmlu_full_sr_humanities, global_mmlu_full_sr_international_law, + global_mmlu_full_sr_jurisprudence, global_mmlu_full_sr_logical_fallacies, global_mmlu_full_sr_machine_learning, + global_mmlu_full_sr_management, global_mmlu_full_sr_marketing, global_mmlu_full_sr_medical_genetics, + global_mmlu_full_sr_miscellaneous, global_mmlu_full_sr_moral_disputes, global_mmlu_full_sr_moral_scenarios, + global_mmlu_full_sr_nutrition, global_mmlu_full_sr_other, global_mmlu_full_sr_philosophy, global_mmlu_full_sr_prehistory, + global_mmlu_full_sr_professional_accounting, global_mmlu_full_sr_professional_law, global_mmlu_full_sr_professional_medicine, + global_mmlu_full_sr_professional_psychology, global_mmlu_full_sr_public_relations, global_mmlu_full_sr_security_studies, + global_mmlu_full_sr_social_sciences, global_mmlu_full_sr_sociology, global_mmlu_full_sr_stem, global_mmlu_full_sr_us_foreign_policy, + global_mmlu_full_sr_virology, global_mmlu_full_sr_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: swe_Latn + scope: single + tasks: [global_mmlu_full_sv, global_mmlu_full_sv_abstract_algebra, global_mmlu_full_sv_anatomy, global_mmlu_full_sv_astronomy, + global_mmlu_full_sv_business_ethics, global_mmlu_full_sv_clinical_knowledge, global_mmlu_full_sv_college_biology, + global_mmlu_full_sv_college_chemistry, global_mmlu_full_sv_college_computer_science, global_mmlu_full_sv_college_mathematics, + global_mmlu_full_sv_college_medicine, global_mmlu_full_sv_college_physics, global_mmlu_full_sv_computer_security, + global_mmlu_full_sv_conceptual_physics, global_mmlu_full_sv_econometrics, global_mmlu_full_sv_electrical_engineering, + global_mmlu_full_sv_elementary_mathematics, global_mmlu_full_sv_formal_logic, global_mmlu_full_sv_global_facts, + global_mmlu_full_sv_high_school_biology, global_mmlu_full_sv_high_school_chemistry, global_mmlu_full_sv_high_school_computer_science, + global_mmlu_full_sv_high_school_european_history, global_mmlu_full_sv_high_school_geography, global_mmlu_full_sv_high_school_government_and_politics, + global_mmlu_full_sv_high_school_macroeconomics, global_mmlu_full_sv_high_school_mathematics, global_mmlu_full_sv_high_school_microeconomics, + global_mmlu_full_sv_high_school_physics, global_mmlu_full_sv_high_school_psychology, global_mmlu_full_sv_high_school_statistics, + global_mmlu_full_sv_high_school_us_history, global_mmlu_full_sv_high_school_world_history, global_mmlu_full_sv_human_aging, + global_mmlu_full_sv_human_sexuality, global_mmlu_full_sv_humanities, global_mmlu_full_sv_international_law, + global_mmlu_full_sv_jurisprudence, global_mmlu_full_sv_logical_fallacies, global_mmlu_full_sv_machine_learning, + global_mmlu_full_sv_management, global_mmlu_full_sv_marketing, global_mmlu_full_sv_medical_genetics, + global_mmlu_full_sv_miscellaneous, global_mmlu_full_sv_moral_disputes, global_mmlu_full_sv_moral_scenarios, + global_mmlu_full_sv_nutrition, global_mmlu_full_sv_other, global_mmlu_full_sv_philosophy, global_mmlu_full_sv_prehistory, + global_mmlu_full_sv_professional_accounting, global_mmlu_full_sv_professional_law, global_mmlu_full_sv_professional_medicine, + global_mmlu_full_sv_professional_psychology, global_mmlu_full_sv_public_relations, global_mmlu_full_sv_security_studies, + global_mmlu_full_sv_social_sciences, global_mmlu_full_sv_sociology, global_mmlu_full_sv_stem, global_mmlu_full_sv_us_foreign_policy, + global_mmlu_full_sv_virology, global_mmlu_full_sv_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: tur_Latn + scope: single + tasks: [global_mmlu_full_tr, global_mmlu_full_tr_abstract_algebra, global_mmlu_full_tr_anatomy, global_mmlu_full_tr_astronomy, + global_mmlu_full_tr_business_ethics, global_mmlu_full_tr_clinical_knowledge, global_mmlu_full_tr_college_biology, + global_mmlu_full_tr_college_chemistry, global_mmlu_full_tr_college_computer_science, global_mmlu_full_tr_college_mathematics, + global_mmlu_full_tr_college_medicine, global_mmlu_full_tr_college_physics, global_mmlu_full_tr_computer_security, + global_mmlu_full_tr_conceptual_physics, global_mmlu_full_tr_econometrics, global_mmlu_full_tr_electrical_engineering, + global_mmlu_full_tr_elementary_mathematics, global_mmlu_full_tr_formal_logic, global_mmlu_full_tr_global_facts, + global_mmlu_full_tr_high_school_biology, global_mmlu_full_tr_high_school_chemistry, global_mmlu_full_tr_high_school_computer_science, + global_mmlu_full_tr_high_school_european_history, global_mmlu_full_tr_high_school_geography, global_mmlu_full_tr_high_school_government_and_politics, + global_mmlu_full_tr_high_school_macroeconomics, global_mmlu_full_tr_high_school_mathematics, global_mmlu_full_tr_high_school_microeconomics, + global_mmlu_full_tr_high_school_physics, global_mmlu_full_tr_high_school_psychology, global_mmlu_full_tr_high_school_statistics, + global_mmlu_full_tr_high_school_us_history, global_mmlu_full_tr_high_school_world_history, global_mmlu_full_tr_human_aging, + global_mmlu_full_tr_human_sexuality, global_mmlu_full_tr_humanities, global_mmlu_full_tr_international_law, + global_mmlu_full_tr_jurisprudence, global_mmlu_full_tr_logical_fallacies, global_mmlu_full_tr_machine_learning, + global_mmlu_full_tr_management, global_mmlu_full_tr_marketing, global_mmlu_full_tr_medical_genetics, + global_mmlu_full_tr_miscellaneous, global_mmlu_full_tr_moral_disputes, global_mmlu_full_tr_moral_scenarios, + global_mmlu_full_tr_nutrition, global_mmlu_full_tr_other, global_mmlu_full_tr_philosophy, global_mmlu_full_tr_prehistory, + global_mmlu_full_tr_professional_accounting, global_mmlu_full_tr_professional_law, global_mmlu_full_tr_professional_medicine, + global_mmlu_full_tr_professional_psychology, global_mmlu_full_tr_public_relations, global_mmlu_full_tr_security_studies, + global_mmlu_full_tr_social_sciences, global_mmlu_full_tr_sociology, global_mmlu_full_tr_stem, global_mmlu_full_tr_us_foreign_policy, + global_mmlu_full_tr_virology, global_mmlu_full_tr_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ukr_Cyrl + scope: single + tasks: [global_mmlu_full_uk, global_mmlu_full_uk_abstract_algebra, global_mmlu_full_uk_anatomy, global_mmlu_full_uk_astronomy, + global_mmlu_full_uk_business_ethics, global_mmlu_full_uk_clinical_knowledge, global_mmlu_full_uk_college_biology, + global_mmlu_full_uk_college_chemistry, global_mmlu_full_uk_college_computer_science, global_mmlu_full_uk_college_mathematics, + global_mmlu_full_uk_college_medicine, global_mmlu_full_uk_college_physics, global_mmlu_full_uk_computer_security, + global_mmlu_full_uk_conceptual_physics, global_mmlu_full_uk_econometrics, global_mmlu_full_uk_electrical_engineering, + global_mmlu_full_uk_elementary_mathematics, global_mmlu_full_uk_formal_logic, global_mmlu_full_uk_global_facts, + global_mmlu_full_uk_high_school_biology, global_mmlu_full_uk_high_school_chemistry, global_mmlu_full_uk_high_school_computer_science, + global_mmlu_full_uk_high_school_european_history, global_mmlu_full_uk_high_school_geography, global_mmlu_full_uk_high_school_government_and_politics, + global_mmlu_full_uk_high_school_macroeconomics, global_mmlu_full_uk_high_school_mathematics, global_mmlu_full_uk_high_school_microeconomics, + global_mmlu_full_uk_high_school_physics, global_mmlu_full_uk_high_school_psychology, global_mmlu_full_uk_high_school_statistics, + global_mmlu_full_uk_high_school_us_history, global_mmlu_full_uk_high_school_world_history, global_mmlu_full_uk_human_aging, + global_mmlu_full_uk_human_sexuality, global_mmlu_full_uk_humanities, global_mmlu_full_uk_international_law, + global_mmlu_full_uk_jurisprudence, global_mmlu_full_uk_logical_fallacies, global_mmlu_full_uk_machine_learning, + global_mmlu_full_uk_management, global_mmlu_full_uk_marketing, global_mmlu_full_uk_medical_genetics, + global_mmlu_full_uk_miscellaneous, global_mmlu_full_uk_moral_disputes, global_mmlu_full_uk_moral_scenarios, + global_mmlu_full_uk_nutrition, global_mmlu_full_uk_other, global_mmlu_full_uk_philosophy, global_mmlu_full_uk_prehistory, + global_mmlu_full_uk_professional_accounting, global_mmlu_full_uk_professional_law, global_mmlu_full_uk_professional_medicine, + global_mmlu_full_uk_professional_psychology, global_mmlu_full_uk_public_relations, global_mmlu_full_uk_security_studies, + global_mmlu_full_uk_social_sciences, global_mmlu_full_uk_sociology, global_mmlu_full_uk_stem, global_mmlu_full_uk_us_foreign_policy, + global_mmlu_full_uk_virology, global_mmlu_full_uk_world_religions] + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/global_piqa_prompted.yaml b/configs/evals/global_piqa_prompted.yaml new file mode 100644 index 0000000..dcc753d --- /dev/null +++ b/configs/evals/global_piqa_prompted.yaml @@ -0,0 +1,172 @@ +name: Global PIQA (prompted) +category: Commonsense +match: + regex: global_piqa_prompted_.+ +metric: exact_match +filter: strict_match +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: unresolved + note: Provisional scale conversion only; no validated chance correction for the prompted protocol. +warning: Normalization and metric selection have not been validated for prompted Global PIQA. The supplied + comparison sets exclude this eval. Review the protocol before including its scores. +languages: + - language: sqi_Latn + scope: single + tasks: [global_piqa_prompted_als_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: bos_Latn + scope: single + tasks: [global_piqa_prompted_bos_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: bul_Cyrl + scope: single + tasks: [global_piqa_prompted_bul_cyrl] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: cat_Latn + scope: single + tasks: [global_piqa_prompted_cat_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ces_Latn + scope: single + tasks: [global_piqa_prompted_ces_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [global_piqa_prompted_deu_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: est_Latn + scope: single + tasks: [global_piqa_prompted_ekk_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ell_Grek + scope: single + tasks: [global_piqa_prompted_ell_grek] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [global_piqa_prompted_eng_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fin_Latn + scope: single + tasks: [global_piqa_prompted_fin_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [global_piqa_prompted_fra_latn_fran] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: glg_Latn + scope: single + tasks: [global_piqa_prompted_glg_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hrv_Latn + scope: single + tasks: [global_piqa_prompted_hrv_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hun_Latn + scope: single + tasks: [global_piqa_prompted_hun_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: isl_Latn + scope: single + tasks: [global_piqa_prompted_isl_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [global_piqa_prompted_ita_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: kat_Geor + scope: single + tasks: [global_piqa_prompted_kat_geor] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lit_Latn + scope: single + tasks: [global_piqa_prompted_lit_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: mkd_Cyrl + scope: single + tasks: [global_piqa_prompted_mkd_cyrl] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [global_piqa_prompted_nld_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nor_Latn + scope: single + tasks: [global_piqa_prompted_nno_latn, global_piqa_prompted_nob_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: pol_Latn + scope: single + tasks: [global_piqa_prompted_pol_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [global_piqa_prompted_por_latn_port] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ron_Latn + scope: single + tasks: [global_piqa_prompted_ron_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slk_Latn + scope: single + tasks: [global_piqa_prompted_slk_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slv_Latn + scope: single + tasks: [global_piqa_prompted_slv_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [global_piqa_prompted_spa_latn_spai] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: srp_Cyrl + scope: single + tasks: [global_piqa_prompted_srp_cyrl] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: swe_Latn + scope: single + tasks: [global_piqa_prompted_swe_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: tur_Latn + scope: single + tasks: [global_piqa_prompted_tur_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ukr_Cyrl + scope: single + tasks: [global_piqa_prompted_ukr_cyrl] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/gpqa_diamond.yaml b/configs/evals/gpqa_diamond.yaml new file mode 100644 index 0000000..ea15fe2 --- /dev/null +++ b/configs/evals/gpqa_diamond.yaml @@ -0,0 +1,22 @@ +name: GPQA Diamond +category: Knowledge +match: + name: GPQADiamond +metric: accuracy_avg +filter: '' +score: + scale: 1 +normalize: + min: 0.25 + max: 1 + clip: true + basis: uniform_choice + note: The evaluator constructs four options and extracts A/B/C/D; uniform valid-letter guessing gives 1/4. + sources: + - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/GPQADiamond/eval_instruct.py +languages: + - language: eng_Latn + scope: single + tasks: [GPQADiamond] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/gsm8k.yaml b/configs/evals/gsm8k.yaml new file mode 100644 index 0000000..17d646a --- /dev/null +++ b/configs/evals/gsm8k.yaml @@ -0,0 +1,23 @@ +name: GSM8K +category: Math +match: + name: gsm8k +metric: exact_match +filter: flexible-extract +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Generated numerical answers are not a fixed multiple-choice task; no uniform finite answer space is + specified. + sources: + - https://github.com/openai/grade-school-math +languages: + - language: eng_Latn + scope: single + tasks: [gsm8k] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/hellaswag.yaml b/configs/evals/hellaswag.yaml new file mode 100644 index 0000000..acda37c --- /dev/null +++ b/configs/evals/hellaswag.yaml @@ -0,0 +1,105 @@ +name: HellaSwag +category: Commonsense +match: + regex: hellaswag(?:_.+)? +metric: acc_norm +filter: none +shots: 0 +score: + scale: 1 +normalize: + min: 0.25 + max: 1 + clip: true + basis: uniform_choice + note: Four candidate endings; the benchmark reports random performance of 25%. Applied to the translated + variants of the same task. + sources: + - https://rowanzellers.com/hellaswag/ + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/hellaswag/hellaswag.yaml +languages: + - language: eng_Latn + scope: single + tasks: [hellaswag] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. + - language: cat_Latn + scope: single + tasks: [hellaswag_ca] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: dan_Latn + scope: single + tasks: [hellaswag_da] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [hellaswag_de] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [hellaswag_es] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eus_Latn + scope: single + tasks: [hellaswag_eu] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [hellaswag_fr] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hrv_Latn + scope: single + tasks: [hellaswag_hr] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hun_Latn + scope: single + tasks: [hellaswag_hu] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [hellaswag_it] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [hellaswag_nl] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [hellaswag_pt] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ron_Latn + scope: single + tasks: [hellaswag_ro] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slk_Latn + scope: single + tasks: [hellaswag_sk] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: srp_Latn + scope: single + tasks: [hellaswag_sr] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: The sr dataset examples use Latin script; this overrides the generic Serbian Cyrillic default. + - language: swe_Latn + scope: single + tasks: [hellaswag_sv] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ukr_Cyrl + scope: single + tasks: [hellaswag_uk] + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/humaneval.yaml b/configs/evals/humaneval.yaml new file mode 100644 index 0000000..866fdcc --- /dev/null +++ b/configs/evals/humaneval.yaml @@ -0,0 +1,23 @@ +name: HumanEval +category: Code +match: + name: HumanEval +metric: python_pass@1 +filter: '' +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Executable code pass@1 has no benchmark-defined uniform random-program baseline. + sources: + - https://github.com/openai/human-eval +languages: + - language: eng_Latn + scope: single + tasks: [HumanEval] + evidence: https://github.com/openai/human-eval + note: Original English benchmark variant; language is the prompt/question language. Programming-language + metrics are separate from natural-language prompts. diff --git a/configs/evals/ifeval.yaml b/configs/evals/ifeval.yaml new file mode 100644 index 0000000..bfa37b9 --- /dev/null +++ b/configs/evals/ifeval.yaml @@ -0,0 +1,22 @@ +name: IFEval +category: Instruction following +match: + name: ifeval +metric: prompt_level_strict_acc +filter: none +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Strict prompt-level instruction success is not a finite-choice task; no defined uniform random baseline. + sources: + - https://github.com/google-research/google-research/tree/master/instruction_following_eval +languages: + - language: eng_Latn + scope: single + tasks: [ifeval] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/include.yaml b/configs/evals/include.yaml new file mode 100644 index 0000000..d716cb6 --- /dev/null +++ b/configs/evals/include.yaml @@ -0,0 +1,149 @@ +name: INCLUDE +category: Knowledge +match: + regex: include_base_44_.+ +select: + regex: include_base_44_[^_]+ +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.25 + max: 1 + clip: true + basis: uniform_choice + note: The INCLUDE task templates score A/B/C/D against one answer; 1/4 also holds for weighted aggregates + of four-choice subjects. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/include/default/Albanian/_albanian_template_yaml + - https://huggingface.co/datasets/CohereLabs/include-base-44 +languages: + - language: sqi_Latn + scope: single + tasks: [include_base_44_albanian, include_base_44_albanian_arts_humanities, include_base_44_albanian_business_commerce, + include_base_44_albanian_health_oriented_education, include_base_44_albanian_social_science, include_base_44_albanian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eus_Latn + scope: single + tasks: [include_base_44_basque, include_base_44_basque_professional_certification] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: bul_Cyrl + scope: single + tasks: [include_base_44_bulgarian, include_base_44_bulgarian_arts_humanities, include_base_44_bulgarian_social_science, + include_base_44_bulgarian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hrv_Latn + scope: single + tasks: [include_base_44_croatian, include_base_44_croatian_arts_humanities, include_base_44_croatian_social_science, + include_base_44_croatian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [include_base_44_dutch, include_base_44_dutch_applied_science, include_base_44_dutch_arts_humanities, + include_base_44_dutch_health_oriented_education, include_base_44_dutch_social_science, include_base_44_dutch_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: est_Latn + scope: single + tasks: [include_base_44_estonian, include_base_44_estonian_applied_science, include_base_44_estonian_arts_humanities, + include_base_44_estonian_health_oriented_education, include_base_44_estonian_social_science, include_base_44_estonian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fin_Latn + scope: single + tasks: [include_base_44_finnish, include_base_44_finnish_applied_science, include_base_44_finnish_arts_humanities, + include_base_44_finnish_health_oriented_education, include_base_44_finnish_social_science, include_base_44_finnish_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [include_base_44_french, include_base_44_french_arts_humanities, include_base_44_french_driving_license, + include_base_44_french_health_oriented_education, include_base_44_french_social_science, include_base_44_french_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: kat_Geor + scope: single + tasks: [include_base_44_georgian, include_base_44_georgian_arts_humanities] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [include_base_44_german, include_base_44_german_driving_license, include_base_44_german_social_science, + include_base_44_german_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ell_Grek + scope: single + tasks: [include_base_44_greek, include_base_44_greek_arts_humanities, include_base_44_greek_business_commerce, + include_base_44_greek_health_oriented_education, include_base_44_greek_medical_license, include_base_44_greek_professional_certification, + include_base_44_greek_social_science, include_base_44_greek_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hun_Latn + scope: single + tasks: [include_base_44_hungarian, include_base_44_hungarian_applied_science, include_base_44_hungarian_social_science, + include_base_44_hungarian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [include_base_44_italian, include_base_44_italian_applied_science, include_base_44_italian_arts_humanities, + include_base_44_italian_health_oriented_education, include_base_44_italian_professional_certification, + include_base_44_italian_social_science, include_base_44_italian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lit_Latn + scope: single + tasks: [include_base_44_lithuanian, include_base_44_lithuanian_arts_humanities, include_base_44_lithuanian_business_commerce, + include_base_44_lithuanian_professional_certification, include_base_44_lithuanian_social_science, include_base_44_lithuanian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: mkd_Cyrl + scope: single + tasks: [include_base_44_north macedonian, include_base_44_north macedonian_arts_humanities, include_base_44_north + macedonian_business_commerce, include_base_44_north macedonian_social_science, include_base_44_north + macedonian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: pol_Latn + scope: single + tasks: [include_base_44_polish, include_base_44_polish_professional_certification, include_base_44_polish_social_science, + include_base_44_polish_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [include_base_44_portuguese, include_base_44_portuguese_applied_science, include_base_44_portuguese_arts_humanities, + include_base_44_portuguese_business_commerce, include_base_44_portuguese_health_oriented_education, include_base_44_portuguese_social_science, + include_base_44_portuguese_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: srp_Cyrl + scope: single + tasks: [include_base_44_serbian, include_base_44_serbian_arts_humanities, include_base_44_serbian_social_science, + include_base_44_serbian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [include_base_44_spanish, include_base_44_spanish_arts_humanities, include_base_44_spanish_health_oriented_education, + include_base_44_spanish_social_science, include_base_44_spanish_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: tur_Latn + scope: single + tasks: [include_base_44_turkish, include_base_44_turkish_arts_humanities, include_base_44_turkish_business_commerce, + include_base_44_turkish_social_science, include_base_44_turkish_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ukr_Cyrl + scope: single + tasks: [include_base_44_ukrainian, include_base_44_ukrainian_arts_humanities, include_base_44_ukrainian_social_science, + include_base_44_ukrainian_stem] + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/jeebench.yaml b/configs/evals/jeebench.yaml new file mode 100644 index 0000000..c62af73 --- /dev/null +++ b/configs/evals/jeebench.yaml @@ -0,0 +1,26 @@ +name: JEEBench +category: Reasoning +match: + name: JEEBench +metric: accuracy_avg +filter: '' +score: + scale: 1 +normalize: + # Shared 10.55% baseline; Table 2 reports approximately 10.5%. + min: 0.1055 + max: 1 + clip: true + note: Configured 10.55% overall random baseline, aligned with the shared scoring policy. Table 2 of the JEEBench + paper reports an approximate 10.5% baseline. It combines uniform single-choice guessing and random option + subsets with partial credit for multi-answer questions, assigning zero expected score to integer and numeric + answers. This assumes the full 515-question benchmark and the paper's scoring rules. + sources: + - https://aclanthology.org/2023.emnlp-main.468.pdf#page=5 + - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/JEEBench/eval_instruct.py +languages: + - language: eng_Latn + scope: single + tasks: [JEEBench] + evidence: https://huggingface.co/datasets/daman1209arora/jeebench + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/jeopardy.yaml b/configs/evals/jeopardy.yaml new file mode 100644 index 0000000..8e1cfc0 --- /dev/null +++ b/configs/evals/jeopardy.yaml @@ -0,0 +1,22 @@ +name: Jeopardy +category: Knowledge +match: + name: jeopardy +metric: exact_match +filter: strict-match +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Generated exact-match answers have no defined uniform choice set. + sources: + - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/jeopardy.yaml +languages: + - language: eng_Latn + scope: single + tasks: [jeopardy] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/lambada.yaml b/configs/evals/lambada.yaml new file mode 100644 index 0000000..8f378a5 --- /dev/null +++ b/configs/evals/lambada.yaml @@ -0,0 +1,23 @@ +name: LAMBADA +category: Reading +match: + name: lambada_openai +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Next-word prediction accuracy is not fixed-choice accuracy. Vocabulary and a guessing distribution + would be needed. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/lambada/lambada_openai.yaml +languages: + - language: eng_Latn + scope: single + tasks: [lambada_openai] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/language_id.yaml b/configs/evals/language_id.yaml new file mode 100644 index 0000000..adbf063 --- /dev/null +++ b/configs/evals/language_id.yaml @@ -0,0 +1,24 @@ +name: Language ID +category: Language +match: + name: bigbench_language_identification_multiple_choice +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.09090909090909091 + max: 1 + clip: true + basis: uniform_choice + note: Each question offers 11 language names with one correct answer. The corpus covers 1,000 languages, + but the per-question guess baseline is 1/11. + sources: + - https://github.com/google/BIG-bench/blob/main/bigbench/benchmark_tasks/language_identification/README.md +languages: + - language: mul + scope: pooled + tasks: [bigbench_language_identification_multiple_choice] + evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/language_identification + note: One aggregate over 1,000 languages; this CSV has no per-language scores. Kept as a pooled multilingual + result. diff --git a/configs/evals/livecodebench.yaml b/configs/evals/livecodebench.yaml new file mode 100644 index 0000000..d6c4b06 --- /dev/null +++ b/configs/evals/livecodebench.yaml @@ -0,0 +1,22 @@ +name: LiveCodeBench +category: Code +match: + name: LiveCodeBench +metric: accuracy_avg +filter: '' +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Generated code is graded by tests; there is no finite choice set to support a 1/K correction. + sources: + - https://github.com/LiveCodeBench/LiveCodeBench +languages: + - language: eng_Latn + scope: single + tasks: [LiveCodeBench] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/lsat_ar.yaml b/configs/evals/lsat_ar.yaml new file mode 100644 index 0000000..652f28d --- /dev/null +++ b/configs/evals/lsat_ar.yaml @@ -0,0 +1,24 @@ +name: LSAT AR +category: Reasoning +match: + name: agieval_lsat_ar +metric: acc_norm +filter: none +score: + scale: 1 +normalize: + min: 0.2 + max: 1 + clip: true + basis: uniform_choice + note: One correct choice among five LSAT analytical reasoning options; uniform valid-choice guessing gives + 1/5. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/agieval/lsat-ar.yaml + - https://huggingface.co/datasets/hails/agieval-lsat-ar +languages: + - language: eng_Latn + scope: single + tasks: [agieval_lsat_ar] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/math_500.yaml b/configs/evals/math_500.yaml new file mode 100644 index 0000000..228acce --- /dev/null +++ b/configs/evals/math_500.yaml @@ -0,0 +1,22 @@ +name: MATH-500 +category: Math +match: + name: MATH500 +metric: accuracy +filter: '' +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Generated mathematical answers have varying domains; no single uniform guess distribution is defined. + sources: + - https://huggingface.co/datasets/HuggingFaceH4/MATH-500 +languages: + - language: eng_Latn + scope: single + tasks: [MATH500] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/mbpp.yaml b/configs/evals/mbpp.yaml new file mode 100644 index 0000000..a91bbc3 --- /dev/null +++ b/configs/evals/mbpp.yaml @@ -0,0 +1,22 @@ +name: MBPP +category: Code +match: + name: mbpp +metric: pass_at_1 +filter: none +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Generated code is graded by tests; there is no benchmark-defined random-program baseline. + sources: + - https://github.com/google-research/google-research/tree/master/mbpp +languages: + - language: eng_Latn + scope: single + tasks: [mbpp] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/mgsm.yaml b/configs/evals/mgsm.yaml new file mode 100644 index 0000000..adde01f --- /dev/null +++ b/configs/evals/mgsm.yaml @@ -0,0 +1,92 @@ +name: MGSM +category: Math +match: + regex: (?:mgsm_native_cot_.+|global_mgsm_.+) +metric: exact_match +filter: flexible-extract +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Generated numerical answers follow GSM-style scoring; no fixed uniform answer space is specified. + sources: + - https://github.com/google-research/url-nlp/tree/main/mgsm +languages: + - language: cat_Latn + scope: single + tasks: [global_mgsm_ca] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ces_Latn + scope: single + tasks: [global_mgsm_cs] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [global_mgsm_de] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ell_Grek + scope: single + tasks: [global_mgsm_el] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [global_mgsm_en] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [global_mgsm_es] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eus_Latn + scope: single + tasks: [global_mgsm_eu] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [global_mgsm_fr] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: glg_Latn + scope: single + tasks: [global_mgsm_gl] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hun_Latn + scope: single + tasks: [global_mgsm_hu] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: srp_Cyrl + scope: single + tasks: [global_mgsm_sr] + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [mgsm_native_cot_de] + evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [mgsm_native_cot_en] + evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [mgsm_native_cot_es] + evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [mgsm_native_cot_fr] + evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/mmlu.yaml b/configs/evals/mmlu.yaml new file mode 100644 index 0000000..69c8824 --- /dev/null +++ b/configs/evals/mmlu.yaml @@ -0,0 +1,37 @@ +name: MMLU +category: Knowledge +match: + regex: mmlu(?:_.+)? +select: + name: mmlu +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.25 + max: 1 + clip: true + basis: uniform_choice + note: Original MMLU uses four choices and one correct label. Uniform guessing gives 1/4, including subject + summaries. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/mmlu/default/_default_template_yaml +languages: + - language: eng_Latn + scope: single + tasks: [mmlu, mmlu_abstract_algebra, mmlu_anatomy, mmlu_astronomy, mmlu_business_ethics, mmlu_clinical_knowledge, + mmlu_college_biology, mmlu_college_chemistry, mmlu_college_computer_science, mmlu_college_mathematics, + mmlu_college_medicine, mmlu_college_physics, mmlu_computer_security, mmlu_conceptual_physics, mmlu_econometrics, + mmlu_electrical_engineering, mmlu_elementary_mathematics, mmlu_formal_logic, mmlu_global_facts, mmlu_high_school_biology, + mmlu_high_school_chemistry, mmlu_high_school_computer_science, mmlu_high_school_european_history, mmlu_high_school_geography, + mmlu_high_school_government_and_politics, mmlu_high_school_macroeconomics, mmlu_high_school_mathematics, + mmlu_high_school_microeconomics, mmlu_high_school_physics, mmlu_high_school_psychology, mmlu_high_school_statistics, + mmlu_high_school_us_history, mmlu_high_school_world_history, mmlu_human_aging, mmlu_human_sexuality, + mmlu_humanities, mmlu_international_law, mmlu_jurisprudence, mmlu_logical_fallacies, mmlu_machine_learning, + mmlu_management, mmlu_marketing, mmlu_medical_genetics, mmlu_miscellaneous, mmlu_moral_disputes, mmlu_moral_scenarios, + mmlu_nutrition, mmlu_other, mmlu_philosophy, mmlu_prehistory, mmlu_professional_accounting, mmlu_professional_law, + mmlu_professional_medicine, mmlu_professional_psychology, mmlu_public_relations, mmlu_security_studies, + mmlu_social_sciences, mmlu_sociology, mmlu_stem, mmlu_us_foreign_policy, mmlu_virology, mmlu_world_religions] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/multiblimp.yaml b/configs/evals/multiblimp.yaml new file mode 100644 index 0000000..20342e6 --- /dev/null +++ b/configs/evals/multiblimp.yaml @@ -0,0 +1,184 @@ +name: MultiBlimp +category: Language +match: + regex: multiblimp_.+ +metric: acc_norm +filter: none +score: + scale: 1 +normalize: + min: 0.5 + max: 1 + clip: true + basis: uniform_choice + note: A minimal pair compares a grammatical sentence with an ungrammatical one; chance under random preference + is 1/2. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/multiblimp/_template_yaml +warning: 'Language-grouping approximation: multiblimp_hbs pools Croatian and Serbian. Assign its combined score + to Serbian (srp_Latn) because Serbian has more speakers, not because of the dataset''s language proportions. + The score still includes both languages and is not a Serbian-only result.' +languages: + - language: bul_Cyrl + scope: single + tasks: [multiblimp_bul] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: cat_Latn + scope: single + tasks: [multiblimp_cat] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ces_Latn + scope: single + tasks: [multiblimp_ces] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: dan_Latn + scope: single + tasks: [multiblimp_dan] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [multiblimp_deu] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ell_Grek + scope: single + tasks: [multiblimp_ell] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [multiblimp_eng] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: est_Latn + scope: single + tasks: [multiblimp_est] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eus_Latn + scope: single + tasks: [multiblimp_eus] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fin_Latn + scope: single + tasks: [multiblimp_fin] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [multiblimp_fra] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: gle_Latn + scope: single + tasks: [multiblimp_gle] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: glg_Latn + scope: single + tasks: [multiblimp_glg] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: srp_Latn + scope: single + tasks: [multiblimp_hbs] + evidence: https://huggingface.co/datasets/jumelet/multiblimp/blob/main/hbs/data.tsv + note: 'Grouping convention: assign the pooled Croatian/Serbian score to Serbian, the language with more + speakers; retain Latin script. This does not separate the data or produce a Serbian-only score. UCLA + estimates approximately 11 million Serbian speakers and 6 million Croatian speakers: https://slavic.ucla.edu/languages/bcs/serbian-background-info/ + and https://slavic.ucla.edu/languages/bcs/croatian-background-info/.' + - language: hun_Latn + scope: single + tasks: [multiblimp_hun] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: isl_Latn + scope: single + tasks: [multiblimp_isl] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [multiblimp_ita] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: kat_Geor + scope: single + tasks: [multiblimp_kat] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lav_Latn + scope: single + tasks: [multiblimp_lav] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lit_Latn + scope: single + tasks: [multiblimp_lit] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: mkd_Cyrl + scope: single + tasks: [multiblimp_mkd] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [multiblimp_nld] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: pol_Latn + scope: single + tasks: [multiblimp_pol] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [multiblimp_por] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ron_Latn + scope: single + tasks: [multiblimp_ron] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slk_Latn + scope: single + tasks: [multiblimp_slk] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slv_Latn + scope: single + tasks: [multiblimp_slv] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [multiblimp_spa] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: sqi_Latn + scope: single + tasks: [multiblimp_sqi] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: swe_Latn + scope: single + tasks: [multiblimp_swe] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: tur_Latn + scope: single + tasks: [multiblimp_tur] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ukr_Cyrl + scope: single + tasks: [multiblimp_ukr] + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/openbookqa.yaml b/configs/evals/openbookqa.yaml new file mode 100644 index 0000000..e15b1ba --- /dev/null +++ b/configs/evals/openbookqa.yaml @@ -0,0 +1,22 @@ +name: OpenBookQA +category: Knowledge +match: + name: openbookqa +metric: acc_norm +filter: none +score: + scale: 1 +normalize: + min: 0.25 + max: 1 + clip: true + basis: uniform_choice + note: The benchmark has exactly four choices per question. The paper reports a 25% uniform-guess baseline. + sources: + - https://arxiv.org/html/1809.02789v1#S3.SS3 +languages: + - language: eng_Latn + scope: single + tasks: [openbookqa] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/opensubtitles.yaml b/configs/evals/opensubtitles.yaml new file mode 100644 index 0000000..07c17ab --- /dev/null +++ b/configs/evals/opensubtitles.yaml @@ -0,0 +1,421 @@ +name: OpenSubtitles +category: Translation +match: + regex: opensubtitles_.+ +metric: chrf +filter: none +score: + scale: 100 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: 'Select chrf: the referenced harness uses SacreBLEU defaults (char_order=6, word_order=0, beta=2), + so this is plain chrF, not chrF++. Prefer an explicitly identified chrF++ result when available, then chrF, + then BLEU. Retain native 0–100 points with no chance correction.' + sources: + - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/opensubtitles_multi40/utils.py + - https://github.com/mjpost/sacrebleu/blob/master/sacrebleu/metrics/chrf.py + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/api/metrics.py +languages: + - source_language: bul_Cyrl + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_bg_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: ces_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_cs_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: dan_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_da_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: deu_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_de_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: ell_Grek + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_el_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: bul_Cyrl + scope: translation + tasks: [opensubtitles_multi40_en_to_bg] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: ces_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_cs] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: dan_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_da] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: deu_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_de] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: ell_Grek + scope: translation + tasks: [opensubtitles_multi40_en_to_el] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: spa_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_es] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: est_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_et] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: fin_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_fi] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: fra_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_fr] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: hrv_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_hr] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: hun_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_hu] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: ita_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_it] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: lit_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_lt] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: lav_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_lv] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: nld_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_nl] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: nor_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_no] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: pol_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_pl] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: por_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_pt] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: ron_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_ro] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: slk_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_sk] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: slv_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_sl] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: srp_Cyrl + scope: translation + tasks: [opensubtitles_multi40_en_to_sr] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: swe_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_sv] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: tur_Latn + scope: translation + tasks: [opensubtitles_multi40_en_to_tr] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: eng_Latn + target_language: ukr_Cyrl + scope: translation + tasks: [opensubtitles_multi40_en_to_uk] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: spa_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_es_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: est_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_et_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: fin_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_fi_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: fra_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_fr_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: hrv_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_hr_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: hun_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_hu_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: ita_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_it_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: lit_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_lt_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: lav_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_lv_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: nld_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_nl_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: nor_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_no_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: pol_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_pl_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: por_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_pt_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: ron_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_ro_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: slk_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_sk_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: slv_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_sl_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: srp_Cyrl + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_sr_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: swe_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_sv_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: tur_Latn + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_tr_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. + - source_language: ukr_Cyrl + target_language: eng_Latn + scope: translation + tasks: [opensubtitles_multi40_uk_to_en] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. diff --git a/configs/evals/operators.yaml b/configs/evals/operators.yaml new file mode 100644 index 0000000..5e47196 --- /dev/null +++ b/configs/evals/operators.yaml @@ -0,0 +1,22 @@ +name: Operators +category: Math +match: + name: bigbench_operators_generate_until +metric: exact_match +filter: strict-match +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: The selected task uses generated exact-match answers, with no specified random-answer distribution. + sources: + - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/operators +languages: + - language: eng_Latn + scope: single + tasks: [bigbench_operators_generate_until] + evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/operators + note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/piqa.yaml b/configs/evals/piqa.yaml new file mode 100644 index 0000000..cf763f3 --- /dev/null +++ b/configs/evals/piqa.yaml @@ -0,0 +1,179 @@ +name: PIQA +category: Commonsense +match: + regex: (?:piqa|global_piqa_completions_.+) +metric: acc_norm +filter: none +score: + scale: 1 +normalize: + min: 0.5 + max: 1 + clip: true + basis: uniform_choice + note: The completion tasks compare two solutions. Uniform guessing gives 1/2. Prompted Global PIQA is configured + separately and excluded from the supplied comparison sets. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/piqa/piqa.yaml + - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/global_piqa/completions/_template +languages: + - language: eng_Latn + scope: single + tasks: [piqa] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. + - language: sqi_Latn + scope: single + tasks: [global_piqa_completions_als_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: bos_Latn + scope: single + tasks: [global_piqa_completions_bos_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: bul_Cyrl + scope: single + tasks: [global_piqa_completions_bul_cyrl] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: cat_Latn + scope: single + tasks: [global_piqa_completions_cat_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ces_Latn + scope: single + tasks: [global_piqa_completions_ces_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [global_piqa_completions_deu_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: est_Latn + scope: single + tasks: [global_piqa_completions_ekk_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ell_Grek + scope: single + tasks: [global_piqa_completions_ell_grek] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [global_piqa_completions_eng_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fin_Latn + scope: single + tasks: [global_piqa_completions_fin_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [global_piqa_completions_fra_latn_fran] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: glg_Latn + scope: single + tasks: [global_piqa_completions_glg_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hrv_Latn + scope: single + tasks: [global_piqa_completions_hrv_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hun_Latn + scope: single + tasks: [global_piqa_completions_hun_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: isl_Latn + scope: single + tasks: [global_piqa_completions_isl_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [global_piqa_completions_ita_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: kat_Geor + scope: single + tasks: [global_piqa_completions_kat_geor] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lit_Latn + scope: single + tasks: [global_piqa_completions_lit_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: mkd_Cyrl + scope: single + tasks: [global_piqa_completions_mkd_cyrl] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [global_piqa_completions_nld_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nor_Latn + scope: single + tasks: [global_piqa_completions_nno_latn, global_piqa_completions_nob_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: pol_Latn + scope: single + tasks: [global_piqa_completions_pol_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [global_piqa_completions_por_latn_port] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ron_Latn + scope: single + tasks: [global_piqa_completions_ron_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slk_Latn + scope: single + tasks: [global_piqa_completions_slk_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slv_Latn + scope: single + tasks: [global_piqa_completions_slv_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [global_piqa_completions_spa_latn_spai] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: srp_Cyrl + scope: single + tasks: [global_piqa_completions_srp_cyrl] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: swe_Latn + scope: single + tasks: [global_piqa_completions_swe_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: tur_Latn + scope: single + tasks: [global_piqa_completions_tur_latn] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ukr_Cyrl + scope: single + tasks: [global_piqa_completions_ukr_cyrl] + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/polymath.yaml b/configs/evals/polymath.yaml new file mode 100644 index 0000000..55062f3 --- /dev/null +++ b/configs/evals/polymath.yaml @@ -0,0 +1,66 @@ +name: PolyMath +category: Reasoning +match: {regex: 'polymath_.+'} +metric: exact_match +filter: none +score: {scale: 1} +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: Generated mathematical answers are scored by exact match; no common finite answer space is defined. + sources: + - https://github.com/QwenLM/PolyMath +aggregation: + components: + - name: low + match: {regex: 'polymath_.+_low'} + relative_weight: 1 + - name: medium + match: {regex: 'polymath_.+_medium'} + relative_weight: 2 + - name: high + match: {regex: 'polymath_.+_high'} + relative_weight: 4 + - name: top + match: {regex: 'polymath_.+_top'} + relative_weight: 8 + note: >- + PolyMath difficulty-weighted accuracy combines all four levels within each language: + (low + 2 × medium + 4 × high + 8 × top) / 15. Exported exact_match values are + unweighted per-level accuracies. Each level is required; incomplete groups are excluded. + sources: + - https://qwen-polymath.github.io/#benchmark-score + - https://github.com/QwenLM/PolyMath/blob/main/eval/run_eval.py +languages: + - language: deu_Latn + scope: single + tasks: [polymath_de_high, polymath_de_low, polymath_de_medium, polymath_de_top] + evidence: https://huggingface.co/datasets/Qwen/PolyMath + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [polymath_en_high, polymath_en_low, polymath_en_medium, polymath_en_top] + evidence: https://huggingface.co/datasets/Qwen/PolyMath + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [polymath_es_high, polymath_es_low, polymath_es_medium, polymath_es_top] + evidence: https://huggingface.co/datasets/Qwen/PolyMath + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [polymath_fr_high, polymath_fr_low, polymath_fr_medium, polymath_fr_top] + evidence: https://huggingface.co/datasets/Qwen/PolyMath + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [polymath_it_high, polymath_it_low, polymath_it_medium, polymath_it_top] + evidence: https://huggingface.co/datasets/Qwen/PolyMath + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [polymath_pt_high, polymath_pt_low, polymath_pt_medium, polymath_pt_top] + evidence: https://huggingface.co/datasets/Qwen/PolyMath + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/qa_wikidata.yaml b/configs/evals/qa_wikidata.yaml new file mode 100644 index 0000000..1f97d1d --- /dev/null +++ b/configs/evals/qa_wikidata.yaml @@ -0,0 +1,22 @@ +name: QA Wikidata +category: Knowledge +match: + name: bigbench_qa_wikidata_generate_until +metric: exact_match +filter: strict-match +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: The selected generate-until task uses open-ended exact match; no fixed choice count. + sources: + - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/qa_wikidata +languages: + - language: eng_Latn + scope: single + tasks: [bigbench_qa_wikidata_generate_until] + evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/qa_wikidata + note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/repeat_copy_logic.yaml b/configs/evals/repeat_copy_logic.yaml new file mode 100644 index 0000000..3c47725 --- /dev/null +++ b/configs/evals/repeat_copy_logic.yaml @@ -0,0 +1,22 @@ +name: Repeat Copy Logic +category: Math +match: + name: bigbench_repeat_copy_logic_generate_until +metric: exact_match +filter: strict-match +score: + scale: 1 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: The selected task generates output strings; no defined uniform answer space. + sources: + - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/repeat_copy_logic +languages: + - language: eng_Latn + scope: single + tasks: [bigbench_repeat_copy_logic_generate_until] + evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/repeat_copy_logic + note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/sib_200.yaml b/configs/evals/sib_200.yaml new file mode 100644 index 0000000..aa3be4c --- /dev/null +++ b/configs/evals/sib_200.yaml @@ -0,0 +1,200 @@ +name: SIB-200 +category: Reading +match: + regex: sib200_.+ +# Use acc: acc_norm is not a reliable/useful metric for this eval in our export +# (exactly 0.25 for 34 of 36 languages). +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.14285714285714285 + max: 1 + clip: true + basis: uniform_choice + note: The OELLM template lists seven topic choices; uniform guessing gives 1/7. The selected field is raw + acc, not acc_norm. + sources: + - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/sib200/_default_template_yaml +languages: + - language: sqi_Latn + scope: single + tasks: [sib200_als_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: bos_Latn + scope: single + tasks: [sib200_bos_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: bul_Cyrl + scope: single + tasks: [sib200_bul_Cyrl] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: cat_Latn + scope: single + tasks: [sib200_cat_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ces_Latn + scope: single + tasks: [sib200_ces_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: dan_Latn + scope: single + tasks: [sib200_dan_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: deu_Latn + scope: single + tasks: [sib200_deu_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ell_Grek + scope: single + tasks: [sib200_ell_Grek] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [sib200_eng_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: est_Latn + scope: single + tasks: [sib200_est_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eus_Latn + scope: single + tasks: [sib200_eus_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fin_Latn + scope: single + tasks: [sib200_fin_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [sib200_fra_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: gle_Latn + scope: single + tasks: [sib200_gle_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: glg_Latn + scope: single + tasks: [sib200_glg_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hrv_Latn + scope: single + tasks: [sib200_hrv_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: hun_Latn + scope: single + tasks: [sib200_hun_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: isl_Latn + scope: single + tasks: [sib200_isl_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [sib200_ita_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: kat_Geor + scope: single + tasks: [sib200_kat_Geor] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lit_Latn + scope: single + tasks: [sib200_lit_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: lav_Latn + scope: single + tasks: [sib200_lvs_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: mkd_Cyrl + scope: single + tasks: [sib200_mkd_Cyrl] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: mlt_Latn + scope: single + tasks: [sib200_mlt_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [sib200_nld_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nor_Latn + scope: single + tasks: [sib200_nob_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: pol_Latn + scope: single + tasks: [sib200_pol_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [sib200_por_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ron_Latn + scope: single + tasks: [sib200_ron_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slk_Latn + scope: single + tasks: [sib200_slk_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: slv_Latn + scope: single + tasks: [sib200_slv_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [sib200_spa_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: srp_Cyrl + scope: single + tasks: [sib200_srp_Cyrl] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: swe_Latn + scope: single + tasks: [sib200_swe_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: tur_Latn + scope: single + tasks: [sib200_tur_Latn] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ukr_Cyrl + scope: single + tasks: [sib200_ukr_Cyrl] + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/social_iqa.yaml b/configs/evals/social_iqa.yaml new file mode 100644 index 0000000..85bcb06 --- /dev/null +++ b/configs/evals/social_iqa.yaml @@ -0,0 +1,22 @@ +name: Social IQa +category: Commonsense +match: + name: social_iqa +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.3333333333333333 + max: 1 + clip: true + basis: uniform_choice + note: The task scores answerA, answerB and answerC; uniform guessing gives 1/3. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/siqa/siqa.yaml +languages: + - language: eng_Latn + scope: single + tasks: [social_iqa] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/squad_v2.yaml b/configs/evals/squad_v2.yaml new file mode 100644 index 0000000..94b9070 --- /dev/null +++ b/configs/evals/squad_v2.yaml @@ -0,0 +1,23 @@ +name: SQuAD v2 +category: Reading +match: + name: squadv2 +metric: f1 +filter: none +score: + scale: 100 +normalize: + min: 0 + max: 1 + clip: true + basis: not_applicable + note: F1 mixes span overlap and unanswerable questions; neither 0 nor 50% is an established random baseline. + Keep only scale conversion. + sources: + - https://rajpurkar.github.io/SQuAD-explorer/ +languages: + - language: eng_Latn + scope: single + tasks: [squadv2] + evidence: https://rajpurkar.github.io/SQuAD-explorer/ + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/winogrande.yaml b/configs/evals/winogrande.yaml new file mode 100644 index 0000000..8d404b0 --- /dev/null +++ b/configs/evals/winogrande.yaml @@ -0,0 +1,22 @@ +name: WinoGrande +category: Commonsense +match: + name: winogrande +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.5 + max: 1 + clip: true + basis: uniform_choice + note: The evaluator compares option1 and option2; uniform guessing gives 1/2. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/winogrande/preprocess_winogrande.py +languages: + - language: eng_Latn + scope: single + tasks: [winogrande] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/wsc273.yaml b/configs/evals/wsc273.yaml new file mode 100644 index 0000000..19ec50f --- /dev/null +++ b/configs/evals/wsc273.yaml @@ -0,0 +1,22 @@ +name: WSC273 +category: Commonsense +match: + name: wsc273 +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.5 + max: 1 + clip: true + basis: uniform_choice + note: The evaluator compares two candidate antecedents; uniform guessing gives 1/2. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/wsc273/default.yaml +languages: + - language: eng_Latn + scope: single + tasks: [wsc273] + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/x_csqa.yaml b/configs/evals/x_csqa.yaml new file mode 100644 index 0000000..13ea7af --- /dev/null +++ b/configs/evals/x_csqa.yaml @@ -0,0 +1,58 @@ +name: X-CSQA +category: Commonsense +match: + regex: xcsqa_.+ +metric: acc_norm +filter: none +score: + scale: 1 +normalize: + min: 0.2 + max: 1 + clip: true + basis: uniform_choice + note: Translated CommonsenseQA retains five answer options; uniform guessing gives 1/5. + sources: + - https://inklab.usc.edu/XCSR/xcsr_datasets#x-csqa + - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/xcsqa/_default_template_yaml +languages: + - language: deu_Latn + scope: single + tasks: [xcsqa_deu_Latn] + evidence: https://huggingface.co/datasets/INK-USC/xcsr + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: eng_Latn + scope: single + tasks: [xcsqa_eng_Latn] + evidence: https://huggingface.co/datasets/INK-USC/xcsr + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: fra_Latn + scope: single + tasks: [xcsqa_fra_Latn] + evidence: https://huggingface.co/datasets/INK-USC/xcsr + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: [xcsqa_ita_Latn] + evidence: https://huggingface.co/datasets/INK-USC/xcsr + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: nld_Latn + scope: single + tasks: [xcsqa_nld_Latn] + evidence: https://huggingface.co/datasets/INK-USC/xcsr + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: pol_Latn + scope: single + tasks: [xcsqa_pol_Latn] + evidence: https://huggingface.co/datasets/INK-USC/xcsr + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: por_Latn + scope: single + tasks: [xcsqa_por_Latn] + evidence: https://huggingface.co/datasets/INK-USC/xcsr + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: spa_Latn + scope: single + tasks: [xcsqa_spa_Latn] + evidence: https://huggingface.co/datasets/INK-USC/xcsr + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/xcopa.yaml b/configs/evals/xcopa.yaml new file mode 100644 index 0000000..d63d6dc --- /dev/null +++ b/configs/evals/xcopa.yaml @@ -0,0 +1,32 @@ +name: XCOPA +category: Commonsense +match: + regex: xcopa:.+ +metric: acc +filter: none +score: + scale: 1 +normalize: + min: 0.5 + max: 1 + clip: true + basis: uniform_choice + note: Translated COPA retains choice1 and choice2; uniform guessing gives 1/2. + sources: + - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/xcopa/utils.py +languages: + - language: est_Latn + scope: single + tasks: ['xcopa:et'] + evidence: https://huggingface.co/datasets/cambridgeltl/xcopa + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: ita_Latn + scope: single + tasks: ['xcopa:it'] + evidence: https://huggingface.co/datasets/cambridgeltl/xcopa + note: Dataset language resolved from the OELLM task registry and benchmark configuration. + - language: tur_Latn + scope: single + tasks: ['xcopa:tr'] + evidence: https://huggingface.co/datasets/cambridgeltl/xcopa + note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/docs/configuration.md b/docs/configuration.md index 6b4927f..002241e 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -7,7 +7,7 @@ Quickdash separates four inputs: model results, a weighting profile, an optional | CSV results | Raw measurements for models A and B | Shared files in `results/`, or browser imports | | Weighting profile | Category weights, English shares, default calculation | `configs/weights/oellm.yaml` | | Eval set | Optional expected evals and variants | `configs/sets/any-available.yaml` | -| Global catalogue | Match tasks to evals, categories, metrics, normalization and languages | `configs/catalogue.yaml` | +| Global catalogue | Match tasks to evals, categories, metrics, normalization and languages | `configs/catalogue.yaml` → `configs/evals/` | ## Choose weights and expected coverage @@ -76,56 +76,88 @@ Weights must be nonnegative and sum to 1. Empty categories have their weight red The builder embeds YAML profiles directly in `configs/weights/` and sets directly in `configs/sets/`. Each directory's `default.txt` contains one YAML filename selected on startup. Missing or invalid defaults stop the build before replacing output. Names must be unique within each selector. -- `--catalogue PATH` chooses the global interpretation file. +- `--catalogue PATH` chooses a complete catalogue YAML or a manifest pointing to per-eval files. Relative `evals_dir` paths resolve against the manifest’s directory, independent of the working directory. - `--weights PATH` chooses a profile; used alone, it embeds only that profile. `--weights-dir DIR` offers the profiles in another directory and uses its `default.txt` unless an explicit profile is supplied. - `--eval-set PATH` and `--sets-dir DIR` work the same way for eval sets. - `--results-dir DIR` embeds CSVs directly in that directory. A checkpoint label may occur in only one file; one file can contain multiple models. - `--sample-csv FILE`, used with `--results-dir`, supplies a fallback only if that directory has no CSVs. Invalid shared files stop the build; they never trigger the fallback. Pages uses `examples/sample-evals.csv` for this option. -Every offered set is validated against the catalogue and every profile before writing output. Raw results are classified once by the global catalogue; changing sets only changes comparison membership. An empty results directory, or no CSV input, starts without models. +Every offered set is validated against the catalogue and every profile before writing output. Raw results are classified once by the global catalogue; changing sets only changes comparison membership. An empty results directory starts without models unless `--sample-csv` supplies a fallback. Omitting both input options starts without models. Under **Eval configuration**, load or export the catalogue, weights, and eval set separately. Uploaded choices are temporary; reload restores published defaults. Catalogue imports reinterpret all loaded real models and regenerate the synthetic comparison. Invalid imports preserve the previous models and settings. If a new catalogue does not contain the active named set's evals, switch to Any available before loading it. **Clear models** retains settings. -`analysis.json` records the catalogue, selected profile and set, available `profiles` and `suites`, and source filenames/hashes. It also contains the resolved internal `scheme` used for arithmetic; that combined object is not a YAML input format. Generated `catalogue.yaml`, `weights.yaml`, and `eval-set.yaml` record the build inputs. +`analysis.json` records the catalogue, selected profile and set, available `profiles` and `suites`, and source filenames/hashes. It also contains the resolved internal `scheme` used for arithmetic; that combined object is not a YAML input format. Generated `catalogue.yaml` contains the assembled, portable catalogue, with no `evals_dir` reference. `weights.yaml` and `eval-set.yaml` record the other configuration inputs. Browser export also saves the complete catalogue in one file; browser import accepts complete catalogues, not filesystem manifests. -## Complete small example +## Edit one eval -This catalogue selects `acc_norm` for two explicit language variants of a four-choice eval. Additional categories and evals follow the same structure. +The repository keeps each eval’s interpretation and language mappings in one file. +For example, this `evals/example.yaml` selects `acc_norm` for two language variants +of a four-choice eval: ```yaml -version: 1 -name: Example scoring -evals: - - name: Example eval - category: Reasoning - match: {regex: 'example_(en|fr)'} - metric: acc_norm - filter: none - shots: 0 - score: {scale: 1} - normalize: - min: 0.25 # Four choices: uniform guessing gets 1/4 correct. - max: 1 - clip: true - basis: uniform_choice - note: Four-choice chance correction is enabled for this example. - +name: Example eval +category: Reasoning +match: {regex: 'example_(en|fr)'} +metric: acc_norm +filter: none +shots: 0 +score: {scale: 1} +normalize: + min: 0.25 # Four choices: uniform guessing gets 1/4 correct. + max: 1 + clip: true + basis: uniform_choice + note: Four-choice chance correction is enabled for this example. languages: - - tasks: [example_en] + - language: eng_Latn scope: single - language: eng_Latn - - tasks: [example_fr] + tasks: [example_en] + - language: fra_Latn scope: single - language: fra_Latn + tasks: [example_fr] +``` + +A `catalogue.yaml` beside the `evals/` directory gathers those files: +```yaml +version: 1 +name: Example scoring +evals_dir: evals ``` +`evals_dir` is a directory path. The loader reads its direct `.yaml` and `.yml` +files in filename order; subdirectories and other extensions are ignored. Each +file is one eval definition with its own `languages` list. There is no per-file +`version` field: the manifest specifies the format version. Add a new eval by +adding a file; no separate list needs updating. Metadata `notes` may be supplied +on the manifest as a list of strings. + +Every explicit language task must match the eval in its file. Duplicate eval +names, duplicate task-language assignments, known tasks matching multiple evals, +and incompatible component configurations are errors. Missing or empty eval +directories and malformed files stop loading. Errors identify the source file +when a single file is invalid. All validation finishes before the builder replaces +output or a browser import takes effect. + +## Complete small example + +For a portable single file, the format remains `version`, `name`, `evals`, and +`languages`, with optional `notes`. The loader collects each per-eval definition +under `evals` and its language groups under `languages`. This assembled format is +what the browser and scoring engines use. It contains no directory references. +See the complete [fictional catalogue](../configs/examples/catalogue.yaml), or +build a dashboard and use its generated `catalogue.yaml`. Both formats work with +`--catalogue` and Python’s `load_config`; only the complete format can be imported +into a standalone browser page. A manifest cannot also contain `evals` or +`languages`. + For an input row with `value=0.625`, the raw score is 62.5 and the normalized score is 50. Raw comparisons show the former; the composite uses the latter. A row using `acc` is retained in the configuration audit but excluded because this config selects `acc_norm`. The UI shows the original source value for every metric, including excluded metrics; the scaled and normalized columns apply to selected scores. Eval summaries label the selected metric and shot counts actually used, then list other available metrics and settings. Expanded selection rules explain blank extraction-filter fields and unrestricted shot counts separately from the data currently selected. A shot is one example in the prompt; 0-shot means no examples. Tasks eligible under `select` that lack the configured metric/filter/shot combination are highlighted and listed in Warnings for each affected model, even when other tasks in the eval have selected scores. ## Eval fields | Field | Meaning | |---|---| +| `languages` | In a per-eval file, the explicit language groups for that eval. Required; use `[]` when unknown. In a complete catalogue, these groups live in the top-level `languages` list. | | `name` | Unique display name of the eval, grouping its variants. | | `category` | Category label used by weighting profiles and breakdowns. | | `match` | Exactly one of `{name: exact task name}` or `{regex: 'full-match pattern'}`. Unmatched tasks are excluded and listed in Warnings; overlapping matches are an error. | diff --git a/docs/development.md b/docs/development.md index 3a014a1..71661eb 100644 --- a/docs/development.md +++ b/docs/development.md @@ -9,7 +9,7 @@ Run commands from the repository root. Install the Python package with `python - | `quickdash/` | Native Python interpretation, analysis, diagnostics, and CLI. | | `app/` | Python builder, DOM-independent JavaScript engine (`analysis.js`), browser renderer (`app.js`), HTML, and bundled YAML parser. | | `tests/` | Public contract tests, browser checks, and optional private-export regressions. | -| `configs/` | Global catalogue, weighting profiles, optional named sets, and fictional examples. | +| `configs/` | Catalogue manifest, self-contained `evals/` files, weighting profiles, optional named sets, and fictional examples. | | `results/` | Public CSV exports contributed to the shared dashboard. | | `examples/` | Public sample export for Pages and parity tests, plus small fictional quickstart data. | | `docs/` | Configuration and contributor documentation. | @@ -88,7 +88,9 @@ This optional check compares included rows, weights, contributions, scores and d ## Publish through GitHub Pages -The [workflow](../.github/workflows/pages.yml) runs public tests on pull requests and pushes to `main`. After tests pass, it builds the shared dashboard and a separate fictional demo. When `results/` has no CSVs, the shared page embeds `examples/sample-evals.csv`; real shared CSVs take precedence. The sample also runs through both engines in CI for all shipped weighting profiles, eval sets, and aggregation modes, with complete and mismatched coverage. See [contributor requirements](../AGENTS.md). Only `output/site/` is uploaded as the Pages artifact: `index.html`, `demo.html`, and license files. The repository root and private local output are not published as the site. +The [workflow](../.github/workflows/pages.yml) runs public tests on pull requests and pushes to `main`. The Python loader and Node filesystem helper both assemble the per-eval files declared by `configs/catalogue.yaml`. Shared tests check assembly, language ownership, file loading, and portable export/import as well as scoring. + +After tests pass, it builds the shared dashboard and a separate fictional demo. When `results/` has no CSVs, the shared page embeds `examples/sample-evals.csv`; real shared CSVs take precedence. The sample also runs through both engines in CI for all shipped weighting profiles, eval sets, and aggregation modes, with complete and mismatched coverage. See [contributor requirements](../AGENTS.md). Only `output/site/` is uploaded as the Pages artifact: `index.html`, `demo.html`, and license files. The repository root and private local output are not published as the site. In repository **Settings → Pages**, select **GitHub Actions** as the source. Publishing uses the generated artifact rather than a checked-in root or `docs/` folder. A successful push to `main` deploys automatically; a failed build leaves the last successful site available. Review build or deployment failures in the repository’s **Actions** tab. diff --git a/docs/python-api.md b/docs/python-api.md index ce55b4d..f31b445 100644 --- a/docs/python-api.md +++ b/docs/python-api.md @@ -76,6 +76,26 @@ Child contributions sum to the parent's contribution. For nodes with positive ef Reports are dictionaries with attribute access for top-level fields. Nested records are ordinary dictionaries and lists. Use `json.dumps(report, allow_nan=False)` to serialize one; no custom encoder is needed. Renderers can format or reorder nodes without reconstructing the calculation. +## Load the repository’s per-eval files + +Use the same manifest as the dashboard build: + +```python +config = load_config( + catalogue="configs/catalogue.yaml", + weights="configs/weights/oellm.yaml", + eval_set="configs/sets/any-available.yaml", +) +``` + +The loader reads each eval file from the manifest’s `evals_dir`, resolves relative +paths beside that manifest, and validates the combined catalogue. Each eval file +contains its own language assignments. The returned `config["catalogue"]` is a +complete in-memory catalogue, so later analysis does not access those files. +Existing single-file catalogues remain supported. When passing a dictionary +instead of a filename, supply the complete catalogue; a filesystem manifest needs +a filename to resolve its directory. See [editing an eval](configuration.md#edit-one-eval). + ## Surface warnings By default, `analyze()` and `compare()` emit a `QuickdashWarning` through Python's standard `warnings` mechanism for each grouped diagnostic. Warnings normally appear on stderr. Each warning object has a `.diagnostic` attribute containing its structured record. The same diagnostics are retained in `report.diagnostics`, separately from the tree. diff --git a/quickdash/cli.py b/quickdash/cli.py index 1a231d0..e1fdadb 100644 --- a/quickdash/cli.py +++ b/quickdash/cli.py @@ -30,7 +30,7 @@ def main(argv=None): help="Result CSV files (model labels must be unique across files)", ) parser.add_argument( - "--catalogue", required=True, help="Global eval interpretation YAML" + "--catalogue", required=True, help="Catalogue YAML or manifest pointing to per-eval files" ) parser.add_argument("--weights", required=True, help="Weighting profile YAML") parser.add_argument( diff --git a/quickdash/config.py b/quickdash/config.py index 4c6cae4..11b320e 100644 --- a/quickdash/config.py +++ b/quickdash/config.py @@ -4,11 +4,77 @@ import re from pathlib import Path from copy import deepcopy + +import yaml + from .io import parse_yaml +def assemble_catalogue(metadata, definitions): + """Combine self-contained eval definitions into a portable runtime catalogue.""" + object_keys(metadata, {"version", "name", "notes"}, {"version", "name"}) + if not isinstance(definitions, list) or not definitions: + raise ValueError("At least one eval definition is required") + catalogue = dict(deepcopy(metadata), evals=[], languages=[]) + for definition in definitions: + if not isinstance(definition, dict) or "languages" not in definition: + raise ValueError("Each eval definition needs its own languages list") + e = {k: deepcopy(v) for k, v in definition.items() if k != "languages"} + groups = deepcopy(definition["languages"]) + validate_catalogue(dict(metadata, evals=[e], languages=groups)) + for group in groups: + for task in group["tasks"]: + if match_task(e["match"], task) is None: + raise ValueError( + f"Language task {task!r} does not belong to eval {e['name']!r}" + ) + catalogue["evals"].append(e) + catalogue["languages"].extend(groups) + validate_catalogue(catalogue) + for group in catalogue["languages"]: + for task in group["tasks"]: + eval_for_task(task, catalogue) # Reject overlap for known task assignments. + return catalogue + + def load_catalogue(path): - return validate_catalogue(parse_yaml(Path(path).read_text(encoding="utf-8"))) + """Read a portable catalogue, or a manifest pointing to per-eval YAML files.""" + path = Path(path) + try: + config = parse_yaml(path.read_text(encoding="utf-8")) + if not isinstance(config, dict) or "evals_dir" not in config: + return validate_catalogue(config) + object_keys( + config, + {"version", "name", "notes", "evals_dir"}, + {"version", "name", "evals_dir"}, + ) + directory = config["evals_dir"] + if not isinstance(directory, str) or not directory.strip(): + raise ValueError("evals_dir must be a nonempty directory path") + directory = path.parent / directory + if not directory.is_dir(): + raise ValueError(f"Eval directory does not exist: {directory}") + metadata = {k: v for k, v in config.items() if k != "evals_dir"} + definitions = [] + for source in sorted(directory.iterdir()): + if not source.is_file() or source.suffix.lower() not in {".yaml", ".yml"}: + continue + try: + definition = parse_yaml(source.read_text(encoding="utf-8")) + assemble_catalogue(metadata, [definition]) + except (ValueError, OSError) as error: + raise ValueError(f"{source}: {error}") from error + definitions.append(definition) + return assemble_catalogue(metadata, definitions) + except (ValueError, OSError) as error: + raise ValueError(f"{path}: {error}") from error + + +def serialize_catalogue(catalogue): + """Export a complete catalogue with no filesystem references.""" + validate_catalogue(catalogue) + return yaml.safe_dump(catalogue, sort_keys=False, allow_unicode=True, width=100) LANGUAGE_CODE = re.compile(r"(?:[a-z]{3}_[A-Z][a-z]{3}|mul)") @@ -809,7 +875,9 @@ def load(value): ) bundle = dict( - catalogue=load(catalogue), + catalogue=deepcopy(catalogue) + if isinstance(catalogue, dict) + else load_catalogue(catalogue), profile=load(weights), suite=load(eval_set) if eval_set is not None diff --git a/tests/engine_adapter.cjs b/tests/engine_adapter.cjs index 3d93ba2..5e45e35 100644 --- a/tests/engine_adapter.cjs +++ b/tests/engine_adapter.cjs @@ -3,6 +3,7 @@ const E=require('../app/eval_config.js'),S=require('../app/suite_config.js'),A=r const fs=require('node:fs'); function run(c){try{ const config=c.yaml?{catalogue:E.parseCatalogue(c.yaml.catalogue),profile:S.parseWeightProfile(c.yaml.profile),suite:S.parseSuite(c.yaml.suite)}:c.config; + if(Object.hasOwn(c,'eval_definitions'))config.catalogue=E.assembleCatalogue(config.catalogue,c.eval_definitions); const rows=c.csv!==undefined?E.parseCSV(c.csv):c.rows; return {value:c.operation==='compare'?A.compare(rows,config,c.a||'A',c.b||'B'):A.analyze(rows,config)}; }catch(error){return {error:true,message:error.message};}} diff --git a/tests/test_data.py b/tests/test_data.py index 37126f6..15dc47f 100644 --- a/tests/test_data.py +++ b/tests/test_data.py @@ -12,7 +12,7 @@ from app.build import build, summarize from quickdash.io import load_csv -from quickdash.config import classify, normalize_score, validate_config +from quickdash.config import classify, normalize_score, validate_config, load_catalogue ROOT = Path(__file__).resolve().parent.parent @@ -44,6 +44,72 @@ def inputs(folder, c=None): class DataContracts(unittest.TestCase): + def test_python_and_node_load_the_same_catalogue_files(self): + def node(path): + return json.loads(subprocess.check_output(['node','-e', + "const {loadCatalogue}=require('./app/catalogue_io.cjs');try{console.log(JSON.stringify({value:loadCatalogue(process.argv[1])}));}catch(e){console.log(JSON.stringify({error:e.message}));}",str(path)],cwd=ROOT,text=True)) + self.assertEqual(node(ROOT/'configs/catalogue.yaml')['value'],load_catalogue(ROOT/'configs/catalogue.yaml')) + with tempfile.TemporaryDirectory() as tmp: + folder=Path(tmp);kw=inputs(folder);path=kw['catalogue_path'] + original=load_catalogue(path) + self.assertEqual(node(path)['value'],original) + modules=folder/'definitions';modules.mkdir() + (modules/'nested').mkdir();(modules/'nested/ignored.yaml').write_text('not a definition') + manifest=dict(version=1,name='Fixture',evals_dir='definitions') + path.write_text(json.dumps(manifest)) + e={**original['evals'][0],'languages':original['languages']} + (modules/'first.YML').write_text(json.dumps(e)) + self.assertEqual(node(path)['value'],load_catalogue(path)) + (modules/'second.yaml').write_text(json.dumps(e)) + self.assertIn('error',node(path)) + with self.assertRaises(ValueError):load_catalogue(path) + (modules/'second.yaml').unlink() + for source in ['name: duplicate\nname: again','name: incomplete']: + (modules/'first.YML').write_text(source) + self.assertIn('first.YML',node(path)['error']) + with self.assertRaisesRegex(ValueError,'first.YML'):load_catalogue(path) + for patch in [dict(evals=[],languages=[]),dict(evals_dir=None),dict(evals_dir='missing'),dict(evals_dir='')]: + path.write_text(json.dumps({**manifest,**patch})) + self.assertIn('error',node(path)) + with self.assertRaises(ValueError):load_catalogue(path) + + def test_modular_catalogue_loading_and_portable_build(self): + with tempfile.TemporaryDirectory() as tmp: + folder=Path(tmp);kw=inputs(folder);modules=folder/'evals';modules.mkdir() + original=json.loads((folder/'catalogue.yaml').read_text()) + definition={**original['evals'][0],'languages':original['languages']} + (modules/'eval.yml').write_text(json.dumps(definition)) + (modules/'README.md').write_text('Not YAML') + (folder/'catalogue.yaml').write_text('version: 1\nname: Fixture\nevals_dir: evals\n') + compiled=load_catalogue(folder/'catalogue.yaml') + self.assertEqual(compiled,original) + with contextlib.redirect_stdout(io.StringIO()):build(None,folder/'out',**kw) + self.assertEqual(load_catalogue(folder/'out/catalogue.yaml'),original) + previous=(folder/'out/index.html').read_bytes() + bad=deepcopy(definition);bad['languages'][0]['tasks']=['different_eval'] + (modules/'eval.yml').write_text(json.dumps(bad)) + with self.assertRaisesRegex(ValueError,'does not belong'): + build(None,folder/'out',**kw) + self.assertEqual((folder/'out/index.html').read_bytes(),previous) + # Export is portable: remove the entire source directory and import it. + (modules/'eval.yml').unlink();(modules/'README.md').unlink();modules.rmdir() + self.assertEqual(load_catalogue(folder/'out/catalogue.yaml'),original) + with self.assertRaisesRegex(ValueError,'evals'):load_catalogue(folder/'catalogue.yaml') + + def test_modular_catalogue_rejects_bad_files_and_mixed_formats(self): + with tempfile.TemporaryDirectory() as tmp: + folder=Path(tmp);modules=folder/'evals';modules.mkdir();path=folder/'catalogue.yaml' + manifest=dict(version=1,name='Test',evals_dir='evals') + path.write_text(json.dumps(manifest)) + with self.assertRaisesRegex(ValueError,'eval'):load_catalogue(path) + (modules/'bad.yaml').write_text('name: Missing everything else') + with self.assertRaisesRegex(ValueError,'bad.yaml'):load_catalogue(path) + (modules/'bad.yaml').write_text('name: x\nname: duplicate') + with self.assertRaisesRegex(ValueError,'bad.yaml'):load_catalogue(path) + for patch in [dict(evals=[],languages=[]),dict(evals_dir=''),dict(evals_dir=None)]: + path.write_text(json.dumps({**manifest,**patch})) + with self.assertRaises(ValueError):load_catalogue(path) + def test_component_config_validation_and_standalone_build(self): c=config();e=c['evals'][0];e.pop('normalize') levels=['low','medium','high','top'] diff --git a/tests/test_engines.py b/tests/test_engines.py index 2b691ac..1e24338 100644 --- a/tests/test_engines.py +++ b/tests/test_engines.py @@ -12,6 +12,7 @@ from pathlib import Path from quickdash import analyze, compare, load_config, QuickdashWarning, DiagnosticError from quickdash.io import parse_csv, parse_yaml +from quickdash.config import assemble_catalogue ROOT = Path(__file__).resolve().parent.parent @@ -74,6 +75,11 @@ def native(c): if "yaml" in c else c["config"] ) + if "eval_definitions" in c: + cfg = deepcopy(cfg) + cfg["catalogue"] = assemble_catalogue( + cfg["catalogue"], c["eval_definitions"] + ) rows = parse_csv(c["csv"]) if "csv" in c else c["rows"] value = ( compare( @@ -136,6 +142,72 @@ def both(self, cases): results.append((py, other)) return results + def test_per_eval_catalogue_assembly(self): + original = fixture() + metadata = {"version": 1, "name": "Fixture", "notes": ["Shared catalogue"]} + definition = { + **original["catalogue"]["evals"][0], + "languages": original["catalogue"]["languages"], + } + c = {**original, "catalogue": metadata} + good = dict( + config=c, + eval_definitions=[definition], + rows=paired([row(), row("e_fr")]), + operation="compare", + ) + reference_case = {k: v for k, v in good.items() if k != "eval_definitions"} + reference_case["config"] = { + **original, + "catalogue": {**original["catalogue"], "notes": metadata["notes"]}, + } + reference = native(reference_case) + for result in self.both([good])[0]: + self.assertNotIn("error", result) + self.close(semantics(result), semantics(reference)) + self.assertEqual(result["value"]["a"]["score"], 50) + invalid = [] + for definitions in ([], None, [None], [definition, definition]): + invalid.append({**good, "eval_definitions": definitions}) + for mutate in ( + lambda e: e.pop("languages"), + lambda e: e.update(languages={}), + lambda e: e.update(unexpected=True), + lambda e: e["languages"][0].update(tasks=["foreign_task"]), + lambda e: e["languages"][0].update(language="de"), + lambda e: e["languages"].append(deepcopy(e["languages"][0])), + ): + e = deepcopy(definition) + mutate(e) + invalid.append({**good, "eval_definitions": [e]}) + # Overlap across files is rejected for explicitly assigned tasks. + overlap = deepcopy(definition) + overlap.update(name="Other", languages=[]) + invalid.append({**good, "eval_definitions": [definition, overlap]}) + for outputs in self.both(invalid): + for result in outputs: + self.assertIn("error", result) + # Adding entries changes the result according to their data, not file count. + for size in (1, 3, 7): + definitions = [] + rows = [] + for i in range(size): + e = deepcopy(definition) + e.update( + name=f"Eval {i}", + match={"name": f"task{i}"}, + languages=[ + dict(tasks=[f"task{i}"], scope="single", language="eng_Latn") + ], + ) + definitions.append(e) + rows.append(row(f"task{i}", ".625")) + for result in self.both( + [dict(config=c, eval_definitions=definitions, rows=rows)] + )[0]: + self.assertNotIn("error", result) + self.assertEqual(result["value"]["models"][0]["score"], 50) + def test_published_sample_across_all_shipped_configs(self): # Discover files so adding a profile or set automatically extends parity coverage. rows = parse_csv((ROOT / "examples/sample-evals.csv").read_text()) diff --git a/tests/test_public_browser.mjs b/tests/test_public_browser.mjs index 6031b84..1872a08 100644 --- a/tests/test_public_browser.mjs +++ b/tests/test_public_browser.mjs @@ -200,6 +200,19 @@ try{ execFileSync('python3',['-m','app.build','--results-dir',resultsDir,'--sample-csv','examples/sample-evals.csv','--output',sample],{cwd:root,stdio:'pipe'}); await navigate(pathToFileURL(path.join(sample,'index.html')).href); assert.match(await evaluate("document.querySelector('#modelA').value"),/^SAMPLE/); + // The generated catalogue contains every eval and language, with no filesystem dependency. + const exportedCatalogue=fs.readFileSync(path.join(sample,'catalogue.yaml'),'utf8'); + assert.deepEqual(parseCatalogue(exportedCatalogue),require('../app/catalogue_io.cjs').loadCatalogue(path.join(root,'configs/catalogue.yaml'))); + const sampleScore=await evaluate("document.querySelector('#cards').textContent"); + await click('[data-view=config]');await upload('#configFile',exportedCatalogue,'portable-catalogue.yaml'); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),sampleScore); + // A manifest needs files on disk; importing it must preserve the current dashboard. + await upload('#configFile',fs.readFileSync(path.join(root,'configs/catalogue.yaml'),'utf8'),'catalogue-manifest.yaml'); + assert.match(await evaluate("document.querySelector('#error').textContent"),/manifest.*build/i); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),sampleScore); + await upload('#configFile',exportedCatalogue,'portable-catalogue.yaml'); + assert.equal(await evaluate("document.querySelector('#cards').hidden"),false); const syntheticNames=await evaluate("syntheticOptions.map(o=>o.name)"); const scores=[]; diff --git a/tests/test_suites.cjs b/tests/test_suites.cjs index 403c071..cafce4a 100644 --- a/tests/test_suites.cjs +++ b/tests/test_suites.cjs @@ -73,7 +73,7 @@ test('an alternate metric cannot satisfy a required measurement',()=>{ }); test('all shipped sets and profiles are independent and resolve against the global catalogue',()=>{ const fs=require('node:fs'),{parseCatalogue}=require('../app/eval_config.js'); - const c=parseCatalogue(fs.readFileSync('configs/catalogue.yaml','utf8')); + const c=require('../app/catalogue_io.cjs').loadCatalogue('configs/catalogue.yaml'); const profiles=fs.readdirSync('configs/weights').filter(f=>f.endsWith('.yaml')).map(f=>parseWeightProfile(fs.readFileSync('configs/weights/'+f,'utf8'))); for(const p of profiles)for(const filename of fs.readdirSync('configs/sets').filter(f=>f.endsWith('.yaml'))){ const s=parseSuite(fs.readFileSync('configs/sets/'+filename,'utf8')); diff --git a/tests/test_yaml.cjs b/tests/test_yaml.cjs index 7841c3b..f58e7f8 100644 --- a/tests/test_yaml.cjs +++ b/tests/test_yaml.cjs @@ -1,6 +1,6 @@ const assert=require('node:assert/strict'),fs=require('node:fs'); const {parseCatalogue,serializeCatalogue,normalizeScore,auditRows}=require('../app/eval_config.js'); -const config=parseCatalogue(fs.readFileSync('configs/catalogue.yaml','utf8')); +const config=require('../app/catalogue_io.cjs').loadCatalogue('configs/catalogue.yaml'); assert.deepEqual(parseCatalogue(serializeCatalogue(config)),config); const {diagnosticsFor}=require('./diagnostic_fixture.cjs'); assert.deepEqual(diagnosticsFor(new Map(),config),[]); From 2b2b2f6054b758555ed5f640dd6b0db5daa12c51 Mon Sep 17 00:00:00 2001 From: Jonathan Burdge Date: Fri, 2 Oct 2026 10:05:50 +0300 Subject: [PATCH 2/4] Share language metadata and infer routine scope defaults --- app/eval_config.js | 17 +- configs/README.md | 2 + configs/evals/aime24.yaml | 2 - configs/evals/aime25.yaml | 2 - configs/evals/amc23.yaml | 2 - configs/evals/arc_challenge.yaml | 71 +----- configs/evals/arc_easy.yaml | 2 - configs/evals/belebele.yaml | 76 +------ configs/evals/boolq.yaml | 2 - configs/evals/commonsenseqa.yaml | 2 - configs/evals/copa.yaml | 2 - configs/evals/coqa.yaml | 2 - configs/evals/cs_algorithms.yaml | 2 - configs/evals/dyck_languages.yaml | 2 - configs/evals/flores200.yaml | 285 +----------------------- configs/evals/global_mmlu.yaml | 52 +---- configs/evals/global_piqa_prompted.yaml | 97 +------- configs/evals/gpqa_diamond.yaml | 2 - configs/evals/gsm8k.yaml | 2 - configs/evals/hellaswag.yaml | 52 +---- configs/evals/humaneval.yaml | 2 - configs/evals/ifeval.yaml | 2 - configs/evals/include.yaml | 67 +----- configs/evals/jeebench.yaml | 2 - configs/evals/jeopardy.yaml | 2 - configs/evals/lambada.yaml | 2 - configs/evals/language_id.yaml | 2 - configs/evals/livecodebench.yaml | 2 - configs/evals/lsat_ar.yaml | 2 - configs/evals/math_500.yaml | 2 - configs/evals/mbpp.yaml | 2 - configs/evals/mgsm.yaml | 45 +--- configs/evals/mmlu.yaml | 2 - configs/evals/multiblimp.yaml | 98 +------- configs/evals/openbookqa.yaml | 2 - configs/evals/opensubtitles.yaml | 256 +-------------------- configs/evals/operators.yaml | 2 - configs/evals/piqa.yaml | 98 +------- configs/evals/polymath.yaml | 22 +- configs/evals/qa_wikidata.yaml | 2 - configs/evals/repeat_copy_logic.yaml | 2 - configs/evals/sib_200.yaml | 112 +--------- configs/evals/social_iqa.yaml | 2 - configs/evals/squad_v2.yaml | 2 - configs/evals/winogrande.yaml | 2 - configs/evals/wsc273.yaml | 2 - configs/evals/x_csqa.yaml | 28 +-- configs/evals/xcopa.yaml | 13 +- docs/configuration.md | 36 ++- docs/python-api.md | 3 +- quickdash/config.py | 38 +++- tests/test_engines.py | 146 ++++++++++++ 52 files changed, 277 insertions(+), 1399 deletions(-) diff --git a/app/eval_config.js b/app/eval_config.js index 3dd2c90..3be2229 100644 --- a/app/eval_config.js +++ b/app/eval_config.js @@ -71,7 +71,15 @@ const EvalConfig=(()=>{ const catalogue={...structuredClone(metadata),evals:[],languages:[]}; for(const definition of definitions){ if(!definition||typeof definition!=='object'||Array.isArray(definition)||!Object.hasOwn(definition,'languages'))throw Error('Each eval definition needs its own languages list'); - const {languages,...e}=structuredClone(definition); + const {languages:groups,language_defaults:defaults={},...e}=structuredClone(definition); + objectKeys(defaults,['evidence','note']);validateLanguageMetadata(defaults); + if(!Array.isArray(groups))throw Error('languages must be a list'); + const languages=groups.map(group=>{ + if(!group||typeof group!=='object'||Array.isArray(group))return group; + const g={...defaults,...group}; + if(!Object.hasOwn(g,'scope'))g.scope=Object.hasOwn(g,'source_language')||Object.hasOwn(g,'target_language')?'translation':g.language==='mul'?'pooled':'single'; + return g; + }); validateCatalogue({...metadata,evals:[e],languages}); for(const group of languages)for(const task of group.tasks)if(!matchTask(e.match,task))throw Error(`Language task ${task} does not belong to eval ${e.name}`); catalogue.evals.push(e);catalogue.languages.push(...languages); @@ -128,11 +136,14 @@ const EvalConfig=(()=>{ const fields=g.scope==='translation'?['source_language','target_language']:['language'],forbidden=g.scope==='translation'?['language']:['source_language','target_language']; if(forbidden.some(k=>k in g))throw Error('Use language for single/pooled; source and target for translation'); for(const f of fields)if(typeof g[f]!=='string'||!canonical.test(g[f])||(g[f]==='mul'&&g.scope!=='pooled'))throw Error('Use canonical language codes, such as eng_Latn'); - for(const f of ['note','evidence'])if(f in g&&typeof g[f]!=='string')throw Error(f+' must be a string'); - if(g.evidence&&!/^https?:\/\//.test(g.evidence))throw Error('Evidence links must use HTTP or HTTPS'); + validateLanguageMetadata(g); } return validateAggregationConfig(config); } + function validateLanguageMetadata(g){ + for(const f of ['note','evidence'])if(f in g&&typeof g[f]!=='string')throw Error(f+' must be a string'); + if(g.evidence&&!/^https?:\/\//.test(g.evidence))throw Error('Evidence links must use HTTP or HTTPS'); + } // Validate concrete task selections without attempting to infer languages from regexes. function validateAggregationSelection(e,variants,config,unique=true){ if(!e.aggregation)return; diff --git a/configs/README.md b/configs/README.md index b57da44..b118706 100644 --- a/configs/README.md +++ b/configs/README.md @@ -9,6 +9,8 @@ Choose the file to edit based on what you want to change: [catalogue.yaml](catalogue.yaml) is a small manifest pointing to `evals/`. Every `.yaml` or `.yml` file directly in that directory is loaded in filename order; adding a file needs no registration elsewhere. Keep language tasks in the file for the eval they match. The loader rejects misplaced tasks, duplicate names or assignments, and incompatible component configurations before changing the dashboard. An empty `languages: []` is allowed for an eval without known language metadata; unknown-language warnings still apply to its results. +Put repeated language `evidence` and `note` under `language_defaults` in the eval file; individual language entries can override either field. Ordinary entries need only a canonical `language` and `tasks`: scope is inferred from the declared language fields. Keep an explicit `scope: pooled` when a pooled result is assigned to a specific language label. See [shared language metadata](../docs/configuration.md#shared-language-metadata). + Each profile or set needs a distinct `name` within its directory. To change a selector's startup choice, edit that directory's `default.txt` to name one YAML file. The catalogue is selected at build time with `--catalogue`, or temporarily loaded in the browser. The [configuration reference](../docs/configuration.md) describes all three formats with small examples. [examples/](examples/) contains the fictional catalogue and weights used by [examples/scores.csv](../examples/scores.csv). diff --git a/configs/evals/aime24.yaml b/configs/evals/aime24.yaml index d2b8c07..c1ba499 100644 --- a/configs/evals/aime24.yaml +++ b/configs/evals/aime24.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Open-ended integer-answer accuracy. Use a zero minimum with no chance correction; a uniform random-integer guessing model is not used for this dashboard. @@ -18,7 +17,6 @@ normalize: - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/AIME24/eval_instruct.py languages: - language: eng_Latn - scope: single tasks: [AIME24] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/aime25.yaml b/configs/evals/aime25.yaml index 58d4cea..f30d9ed 100644 --- a/configs/evals/aime25.yaml +++ b/configs/evals/aime25.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Open-ended integer-answer accuracy. Use a zero minimum with no chance correction; a uniform random-integer guessing model is not used for this dashboard. @@ -18,7 +17,6 @@ normalize: - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/AIME25/eval_instruct.py languages: - language: eng_Latn - scope: single tasks: [AIME25] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/amc23.yaml b/configs/evals/amc23.yaml index 3640935..ddfc87e 100644 --- a/configs/evals/amc23.yaml +++ b/configs/evals/amc23.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: This evaluator supplies open-ended questions with answer options removed. The original contest five-choice baseline does not apply to this prompt. @@ -17,7 +16,6 @@ normalize: - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/AMC23/data/amc23.json languages: - language: eng_Latn - scope: single tasks: [AMC23] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/arc_challenge.yaml b/configs/evals/arc_challenge.yaml index e196786..d655d0f 100644 --- a/configs/evals/arc_challenge.yaml +++ b/configs/evals/arc_challenge.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0.25 max: 1 - clip: true basis: uniform_choice note: 'Initial approximation: 25% chance. In the published 1,172-question Challenge test split, 1,165 questions have four choices, four have three and three have five. Mean uniform-guess accuracy is 25.0156428%, so @@ -20,119 +19,55 @@ normalize: - https://ai2-public-datasets.s3.amazonaws.com/arc/ARC-V1-Feb2018.zip - https://huggingface.co/datasets/allenai/ai2_arc - https://huggingface.co/datasets/LumiOpen/arc_challenge_mt +language_defaults: + evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: eng_Latn - scope: single tasks: [arc_challenge] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. - language: bul_Cyrl - scope: single tasks: [arc_challenge_mt_bg] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ces_Latn - scope: single tasks: [arc_challenge_mt_cs] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: dan_Latn - scope: single tasks: [arc_challenge_mt_da] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [arc_challenge_mt_de] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ell_Grek - scope: single tasks: [arc_challenge_mt_el] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [arc_challenge_mt_es] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: est_Latn - scope: single tasks: [arc_challenge_mt_et] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fin_Latn - scope: single tasks: [arc_challenge_mt_fi] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [arc_challenge_mt_fr] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hun_Latn - scope: single tasks: [arc_challenge_mt_hu] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: isl_Latn - scope: single tasks: [arc_challenge_mt_is] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [arc_challenge_mt_it] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lit_Latn - scope: single tasks: [arc_challenge_mt_lt] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lav_Latn - scope: single tasks: [arc_challenge_mt_lv] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nor_Latn - scope: single tasks: [arc_challenge_mt_nb] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [arc_challenge_mt_nl] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: pol_Latn - scope: single tasks: [arc_challenge_mt_pl] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [arc_challenge_mt_pt] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ron_Latn - scope: single tasks: [arc_challenge_mt_ro] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slk_Latn - scope: single tasks: [arc_challenge_mt_sk] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slv_Latn - scope: single tasks: [arc_challenge_mt_sl] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: swe_Latn - scope: single tasks: [arc_challenge_mt_sv] - evidence: https://huggingface.co/datasets/LumiOpen/arc_challenge_mt - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/arc_easy.yaml b/configs/evals/arc_easy.yaml index a61a2fe..7c65352 100644 --- a/configs/evals/arc_easy.yaml +++ b/configs/evals/arc_easy.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0.25 max: 1 - clip: true basis: uniform_choice note: Use the conventional four-choice approximation of 0.25. The published test split includes a few three- and five-choice questions (mean guessing accuracy approximately 0.2501613); this small difference is intentionally @@ -19,7 +18,6 @@ normalize: - https://huggingface.co/datasets/allenai/ai2_arc languages: - language: eng_Latn - scope: single tasks: [arc_easy] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/belebele.yaml b/configs/evals/belebele.yaml index 904712c..68c21a4 100644 --- a/configs/evals/belebele.yaml +++ b/configs/evals/belebele.yaml @@ -9,129 +9,59 @@ score: normalize: min: 0.25 max: 1 - clip: true basis: uniform_choice note: All language variants score four labels A/B/C/D; uniform guessing gives 1/4. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/belebele/_default_template_yaml +language_defaults: + evidence: https://huggingface.co/datasets/facebook/belebele + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: bul_Cyrl - scope: single tasks: [belebele_bul_Cyrl] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ces_Latn - scope: single tasks: [belebele_ces_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: dan_Latn - scope: single tasks: [belebele_dan_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [belebele_deu_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ell_Grek - scope: single tasks: [belebele_ell_Grek] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [belebele_eng_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: est_Latn - scope: single tasks: [belebele_est_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fin_Latn - scope: single tasks: [belebele_fin_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [belebele_fra_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hrv_Latn - scope: single tasks: [belebele_hrv_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hun_Latn - scope: single tasks: [belebele_hun_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [belebele_ita_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lit_Latn - scope: single tasks: [belebele_lit_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lav_Latn - scope: single tasks: [belebele_lvs_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: mlt_Latn - scope: single tasks: [belebele_mlt_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [belebele_nld_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nor_Latn - scope: single tasks: [belebele_nob_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: pol_Latn - scope: single tasks: [belebele_pol_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [belebele_por_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ron_Latn - scope: single tasks: [belebele_ron_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slk_Latn - scope: single tasks: [belebele_slk_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slv_Latn - scope: single tasks: [belebele_slv_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [belebele_spa_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: swe_Latn - scope: single tasks: [belebele_swe_Latn] - evidence: https://huggingface.co/datasets/facebook/belebele - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/boolq.yaml b/configs/evals/boolq.yaml index 1687e42..24c8f4c 100644 --- a/configs/evals/boolq.yaml +++ b/configs/evals/boolq.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0.5 max: 1 - clip: true basis: uniform_choice note: The task scores no/yes. This is a uniform-guess baseline of 1/2, not a majority-class baseline. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/super_glue/boolq/default.yaml languages: - language: eng_Latn - scope: single tasks: [boolq] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/commonsenseqa.yaml b/configs/evals/commonsenseqa.yaml index 35f982b..2bb5fe4 100644 --- a/configs/evals/commonsenseqa.yaml +++ b/configs/evals/commonsenseqa.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0.2 max: 1 - clip: true basis: uniform_choice note: The task scores five labels A through E; uniform guessing gives 1/5. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/commonsense_qa/default.yaml languages: - language: eng_Latn - scope: single tasks: [commonsense_qa] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/copa.yaml b/configs/evals/copa.yaml index 4cafe65..f424b15 100644 --- a/configs/evals/copa.yaml +++ b/configs/evals/copa.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0.5 max: 1 - clip: true basis: uniform_choice note: Two possible causes or effects are scored; uniform guessing gives 1/2. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/super_glue/copa/utils.py languages: - language: eng_Latn - scope: single tasks: [copa] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/coqa.yaml b/configs/evals/coqa.yaml index ffab5e7..38cca1b 100644 --- a/configs/evals/coqa.yaml +++ b/configs/evals/coqa.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Token-overlap F1 for generated conversational answers has no universal chance floor. sources: - https://stanfordnlp.github.io/coqa/ languages: - language: eng_Latn - scope: single tasks: [coqa] evidence: https://stanfordnlp.github.io/coqa/ note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/cs_algorithms.yaml b/configs/evals/cs_algorithms.yaml index 4a39533..1d45dbc 100644 --- a/configs/evals/cs_algorithms.yaml +++ b/configs/evals/cs_algorithms.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Free-response exact match over algorithm outputs; no defined uniform answer space. sources: - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/cs_algorithms languages: - language: eng_Latn - scope: single tasks: [bigbench_cs_algorithms_generate_until] evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/cs_algorithms note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/dyck_languages.yaml b/configs/evals/dyck_languages.yaml index 93dd3f2..34e9999 100644 --- a/configs/evals/dyck_languages.yaml +++ b/configs/evals/dyck_languages.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Exact-match generated bracket completions have length-dependent spaces; no single fixed baseline is assigned. @@ -17,7 +16,6 @@ normalize: - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/dyck_languages languages: - language: eng_Latn - scope: single tasks: [bigbench_dyck_languages_generate_until] evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/dyck_languages note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/flores200.yaml b/configs/evals/flores200.yaml index 0b9a3f8..ad9d038 100644 --- a/configs/evals/flores200.yaml +++ b/configs/evals/flores200.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Select the explicit chrf++ field from the rescored export (preferred over chrf, then bleu). Retain native 0–100 points with no chance correction. The CSV does not record the rescoring implementation signature; @@ -17,494 +16,218 @@ normalize: sources: - https://github.com/mjpost/sacrebleu/blob/master/sacrebleu/metrics/chrf.py - https://github.com/facebookresearch/flores +language_defaults: + evidence: https://huggingface.co/datasets/facebook/flores + note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target + are explicit for separate translation-into and translation-from views. languages: - source_language: sqi_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:als_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: bos_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:bos_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: bul_Cyrl target_language: eng_Latn - scope: translation tasks: ['flores200:bul_Cyrl-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: cat_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:cat_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: ces_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:ces_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: dan_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:dan_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: deu_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:deu_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: ell_Grek target_language: eng_Latn - scope: translation tasks: ['flores200:ell_Grek-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: sqi_Latn - scope: translation tasks: ['flores200:eng_Latn-als_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: bos_Latn - scope: translation tasks: ['flores200:eng_Latn-bos_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: bul_Cyrl - scope: translation tasks: ['flores200:eng_Latn-bul_Cyrl'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: cat_Latn - scope: translation tasks: ['flores200:eng_Latn-cat_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: ces_Latn - scope: translation tasks: ['flores200:eng_Latn-ces_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: dan_Latn - scope: translation tasks: ['flores200:eng_Latn-dan_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: deu_Latn - scope: translation tasks: ['flores200:eng_Latn-deu_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: ell_Grek - scope: translation tasks: ['flores200:eng_Latn-ell_Grek'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: est_Latn - scope: translation tasks: ['flores200:eng_Latn-est_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: eus_Latn - scope: translation tasks: ['flores200:eng_Latn-eus_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: fin_Latn - scope: translation tasks: ['flores200:eng_Latn-fin_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: fra_Latn - scope: translation tasks: ['flores200:eng_Latn-fra_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: gle_Latn - scope: translation tasks: ['flores200:eng_Latn-gle_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: glg_Latn - scope: translation tasks: ['flores200:eng_Latn-glg_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: hrv_Latn - scope: translation tasks: ['flores200:eng_Latn-hrv_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: hun_Latn - scope: translation tasks: ['flores200:eng_Latn-hun_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: isl_Latn - scope: translation tasks: ['flores200:eng_Latn-isl_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: ita_Latn - scope: translation tasks: ['flores200:eng_Latn-ita_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: kat_Geor - scope: translation tasks: ['flores200:eng_Latn-kat_Geor'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: lit_Latn - scope: translation tasks: ['flores200:eng_Latn-lit_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: lav_Latn - scope: translation tasks: ['flores200:eng_Latn-lvs_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: mkd_Cyrl - scope: translation tasks: ['flores200:eng_Latn-mkd_Cyrl'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: mlt_Latn - scope: translation tasks: ['flores200:eng_Latn-mlt_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: nld_Latn - scope: translation tasks: ['flores200:eng_Latn-nld_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: nor_Latn - scope: translation tasks: ['flores200:eng_Latn-nob_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: pol_Latn - scope: translation tasks: ['flores200:eng_Latn-pol_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: por_Latn - scope: translation tasks: ['flores200:eng_Latn-por_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: ron_Latn - scope: translation tasks: ['flores200:eng_Latn-ron_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: slk_Latn - scope: translation tasks: ['flores200:eng_Latn-slk_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: slv_Latn - scope: translation tasks: ['flores200:eng_Latn-slv_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: spa_Latn - scope: translation tasks: ['flores200:eng_Latn-spa_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: srp_Cyrl - scope: translation tasks: ['flores200:eng_Latn-srp_Cyrl'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: swe_Latn - scope: translation tasks: ['flores200:eng_Latn-swe_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: tur_Latn - scope: translation tasks: ['flores200:eng_Latn-tur_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eng_Latn target_language: ukr_Cyrl - scope: translation tasks: ['flores200:eng_Latn-ukr_Cyrl'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: est_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:est_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: eus_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:eus_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: fin_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:fin_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: fra_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:fra_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: gle_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:gle_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: glg_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:glg_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: hrv_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:hrv_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: hun_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:hun_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: isl_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:isl_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: ita_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:ita_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: kat_Geor target_language: eng_Latn - scope: translation tasks: ['flores200:kat_Geor-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: lit_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:lit_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: lav_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:lvs_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: mkd_Cyrl target_language: eng_Latn - scope: translation tasks: ['flores200:mkd_Cyrl-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: mlt_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:mlt_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: nld_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:nld_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: nor_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:nob_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: pol_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:pol_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: por_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:por_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: ron_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:ron_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: slk_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:slk_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: slv_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:slv_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: spa_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:spa_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: srp_Cyrl target_language: eng_Latn - scope: translation tasks: ['flores200:srp_Cyrl-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: swe_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:swe_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: tur_Latn target_language: eng_Latn - scope: translation tasks: ['flores200:tur_Latn-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. - source_language: ukr_Cyrl target_language: eng_Latn - scope: translation tasks: ['flores200:ukr_Cyrl-eng_Latn'] - evidence: https://huggingface.co/datasets/facebook/flores - note: Dataset language resolved from the OELLM task registry and benchmark configuration. Source and target - are explicit for separate translation-into and translation-from views. diff --git a/configs/evals/global_mmlu.yaml b/configs/evals/global_mmlu.yaml index 4b09e43..0ffb4e4 100644 --- a/configs/evals/global_mmlu.yaml +++ b/configs/evals/global_mmlu.yaml @@ -11,15 +11,16 @@ score: normalize: min: 0.25 max: 1 - clip: true basis: uniform_choice note: Global MMLU uses four choices and one correct label. Uniform guessing gives 1/4, including subject summaries. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/global_mmlu/default/en/_en_template_yaml +language_defaults: + evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: ces_Latn - scope: single tasks: [global_mmlu_full_cs, global_mmlu_full_cs_abstract_algebra, global_mmlu_full_cs_anatomy, global_mmlu_full_cs_astronomy, global_mmlu_full_cs_business_ethics, global_mmlu_full_cs_clinical_knowledge, global_mmlu_full_cs_college_biology, global_mmlu_full_cs_college_chemistry, global_mmlu_full_cs_college_computer_science, global_mmlu_full_cs_college_mathematics, @@ -40,10 +41,7 @@ languages: global_mmlu_full_cs_professional_psychology, global_mmlu_full_cs_public_relations, global_mmlu_full_cs_security_studies, global_mmlu_full_cs_social_sciences, global_mmlu_full_cs_sociology, global_mmlu_full_cs_stem, global_mmlu_full_cs_us_foreign_policy, global_mmlu_full_cs_virology, global_mmlu_full_cs_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [global_mmlu_full_de, global_mmlu_full_de_abstract_algebra, global_mmlu_full_de_anatomy, global_mmlu_full_de_astronomy, global_mmlu_full_de_business_ethics, global_mmlu_full_de_clinical_knowledge, global_mmlu_full_de_college_biology, global_mmlu_full_de_college_chemistry, global_mmlu_full_de_college_computer_science, global_mmlu_full_de_college_mathematics, @@ -64,10 +62,7 @@ languages: global_mmlu_full_de_professional_psychology, global_mmlu_full_de_public_relations, global_mmlu_full_de_security_studies, global_mmlu_full_de_social_sciences, global_mmlu_full_de_sociology, global_mmlu_full_de_stem, global_mmlu_full_de_us_foreign_policy, global_mmlu_full_de_virology, global_mmlu_full_de_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ell_Grek - scope: single tasks: [global_mmlu_full_el, global_mmlu_full_el_abstract_algebra, global_mmlu_full_el_anatomy, global_mmlu_full_el_astronomy, global_mmlu_full_el_business_ethics, global_mmlu_full_el_clinical_knowledge, global_mmlu_full_el_college_biology, global_mmlu_full_el_college_chemistry, global_mmlu_full_el_college_computer_science, global_mmlu_full_el_college_mathematics, @@ -88,10 +83,7 @@ languages: global_mmlu_full_el_professional_psychology, global_mmlu_full_el_public_relations, global_mmlu_full_el_security_studies, global_mmlu_full_el_social_sciences, global_mmlu_full_el_sociology, global_mmlu_full_el_stem, global_mmlu_full_el_us_foreign_policy, global_mmlu_full_el_virology, global_mmlu_full_el_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [global_mmlu_full_en, global_mmlu_full_en_abstract_algebra, global_mmlu_full_en_anatomy, global_mmlu_full_en_astronomy, global_mmlu_full_en_business_ethics, global_mmlu_full_en_clinical_knowledge, global_mmlu_full_en_college_biology, global_mmlu_full_en_college_chemistry, global_mmlu_full_en_college_computer_science, global_mmlu_full_en_college_mathematics, @@ -112,10 +104,7 @@ languages: global_mmlu_full_en_professional_psychology, global_mmlu_full_en_public_relations, global_mmlu_full_en_security_studies, global_mmlu_full_en_social_sciences, global_mmlu_full_en_sociology, global_mmlu_full_en_stem, global_mmlu_full_en_us_foreign_policy, global_mmlu_full_en_virology, global_mmlu_full_en_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [global_mmlu_full_es, global_mmlu_full_es_abstract_algebra, global_mmlu_full_es_anatomy, global_mmlu_full_es_astronomy, global_mmlu_full_es_business_ethics, global_mmlu_full_es_clinical_knowledge, global_mmlu_full_es_college_biology, global_mmlu_full_es_college_chemistry, global_mmlu_full_es_college_computer_science, global_mmlu_full_es_college_mathematics, @@ -136,10 +125,7 @@ languages: global_mmlu_full_es_professional_psychology, global_mmlu_full_es_public_relations, global_mmlu_full_es_security_studies, global_mmlu_full_es_social_sciences, global_mmlu_full_es_sociology, global_mmlu_full_es_stem, global_mmlu_full_es_us_foreign_policy, global_mmlu_full_es_virology, global_mmlu_full_es_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [global_mmlu_full_fr, global_mmlu_full_fr_abstract_algebra, global_mmlu_full_fr_anatomy, global_mmlu_full_fr_astronomy, global_mmlu_full_fr_business_ethics, global_mmlu_full_fr_clinical_knowledge, global_mmlu_full_fr_college_biology, global_mmlu_full_fr_college_chemistry, global_mmlu_full_fr_college_computer_science, global_mmlu_full_fr_college_mathematics, @@ -160,10 +146,7 @@ languages: global_mmlu_full_fr_professional_psychology, global_mmlu_full_fr_public_relations, global_mmlu_full_fr_security_studies, global_mmlu_full_fr_social_sciences, global_mmlu_full_fr_sociology, global_mmlu_full_fr_stem, global_mmlu_full_fr_us_foreign_policy, global_mmlu_full_fr_virology, global_mmlu_full_fr_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [global_mmlu_full_it, global_mmlu_full_it_abstract_algebra, global_mmlu_full_it_anatomy, global_mmlu_full_it_astronomy, global_mmlu_full_it_business_ethics, global_mmlu_full_it_clinical_knowledge, global_mmlu_full_it_college_biology, global_mmlu_full_it_college_chemistry, global_mmlu_full_it_college_computer_science, global_mmlu_full_it_college_mathematics, @@ -184,10 +167,7 @@ languages: global_mmlu_full_it_professional_psychology, global_mmlu_full_it_public_relations, global_mmlu_full_it_security_studies, global_mmlu_full_it_social_sciences, global_mmlu_full_it_sociology, global_mmlu_full_it_stem, global_mmlu_full_it_us_foreign_policy, global_mmlu_full_it_virology, global_mmlu_full_it_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lit_Latn - scope: single tasks: [global_mmlu_full_lt, global_mmlu_full_lt_abstract_algebra, global_mmlu_full_lt_anatomy, global_mmlu_full_lt_astronomy, global_mmlu_full_lt_business_ethics, global_mmlu_full_lt_clinical_knowledge, global_mmlu_full_lt_college_biology, global_mmlu_full_lt_college_chemistry, global_mmlu_full_lt_college_computer_science, global_mmlu_full_lt_college_mathematics, @@ -208,10 +188,7 @@ languages: global_mmlu_full_lt_professional_psychology, global_mmlu_full_lt_public_relations, global_mmlu_full_lt_security_studies, global_mmlu_full_lt_social_sciences, global_mmlu_full_lt_sociology, global_mmlu_full_lt_stem, global_mmlu_full_lt_us_foreign_policy, global_mmlu_full_lt_virology, global_mmlu_full_lt_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [global_mmlu_full_nl, global_mmlu_full_nl_abstract_algebra, global_mmlu_full_nl_anatomy, global_mmlu_full_nl_astronomy, global_mmlu_full_nl_business_ethics, global_mmlu_full_nl_clinical_knowledge, global_mmlu_full_nl_college_biology, global_mmlu_full_nl_college_chemistry, global_mmlu_full_nl_college_computer_science, global_mmlu_full_nl_college_mathematics, @@ -232,10 +209,7 @@ languages: global_mmlu_full_nl_professional_psychology, global_mmlu_full_nl_public_relations, global_mmlu_full_nl_security_studies, global_mmlu_full_nl_social_sciences, global_mmlu_full_nl_sociology, global_mmlu_full_nl_stem, global_mmlu_full_nl_us_foreign_policy, global_mmlu_full_nl_virology, global_mmlu_full_nl_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: pol_Latn - scope: single tasks: [global_mmlu_full_pl, global_mmlu_full_pl_abstract_algebra, global_mmlu_full_pl_anatomy, global_mmlu_full_pl_astronomy, global_mmlu_full_pl_business_ethics, global_mmlu_full_pl_clinical_knowledge, global_mmlu_full_pl_college_biology, global_mmlu_full_pl_college_chemistry, global_mmlu_full_pl_college_computer_science, global_mmlu_full_pl_college_mathematics, @@ -256,10 +230,7 @@ languages: global_mmlu_full_pl_professional_psychology, global_mmlu_full_pl_public_relations, global_mmlu_full_pl_security_studies, global_mmlu_full_pl_social_sciences, global_mmlu_full_pl_sociology, global_mmlu_full_pl_stem, global_mmlu_full_pl_us_foreign_policy, global_mmlu_full_pl_virology, global_mmlu_full_pl_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [global_mmlu_full_pt, global_mmlu_full_pt_abstract_algebra, global_mmlu_full_pt_anatomy, global_mmlu_full_pt_astronomy, global_mmlu_full_pt_business_ethics, global_mmlu_full_pt_clinical_knowledge, global_mmlu_full_pt_college_biology, global_mmlu_full_pt_college_chemistry, global_mmlu_full_pt_college_computer_science, global_mmlu_full_pt_college_mathematics, @@ -280,10 +251,7 @@ languages: global_mmlu_full_pt_professional_psychology, global_mmlu_full_pt_public_relations, global_mmlu_full_pt_security_studies, global_mmlu_full_pt_social_sciences, global_mmlu_full_pt_sociology, global_mmlu_full_pt_stem, global_mmlu_full_pt_us_foreign_policy, global_mmlu_full_pt_virology, global_mmlu_full_pt_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ron_Latn - scope: single tasks: [global_mmlu_full_ro, global_mmlu_full_ro_abstract_algebra, global_mmlu_full_ro_anatomy, global_mmlu_full_ro_astronomy, global_mmlu_full_ro_business_ethics, global_mmlu_full_ro_clinical_knowledge, global_mmlu_full_ro_college_biology, global_mmlu_full_ro_college_chemistry, global_mmlu_full_ro_college_computer_science, global_mmlu_full_ro_college_mathematics, @@ -304,10 +272,7 @@ languages: global_mmlu_full_ro_professional_psychology, global_mmlu_full_ro_public_relations, global_mmlu_full_ro_security_studies, global_mmlu_full_ro_social_sciences, global_mmlu_full_ro_sociology, global_mmlu_full_ro_stem, global_mmlu_full_ro_us_foreign_policy, global_mmlu_full_ro_virology, global_mmlu_full_ro_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: srp_Cyrl - scope: single tasks: [global_mmlu_full_sr, global_mmlu_full_sr_abstract_algebra, global_mmlu_full_sr_anatomy, global_mmlu_full_sr_astronomy, global_mmlu_full_sr_business_ethics, global_mmlu_full_sr_clinical_knowledge, global_mmlu_full_sr_college_biology, global_mmlu_full_sr_college_chemistry, global_mmlu_full_sr_college_computer_science, global_mmlu_full_sr_college_mathematics, @@ -328,10 +293,7 @@ languages: global_mmlu_full_sr_professional_psychology, global_mmlu_full_sr_public_relations, global_mmlu_full_sr_security_studies, global_mmlu_full_sr_social_sciences, global_mmlu_full_sr_sociology, global_mmlu_full_sr_stem, global_mmlu_full_sr_us_foreign_policy, global_mmlu_full_sr_virology, global_mmlu_full_sr_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: swe_Latn - scope: single tasks: [global_mmlu_full_sv, global_mmlu_full_sv_abstract_algebra, global_mmlu_full_sv_anatomy, global_mmlu_full_sv_astronomy, global_mmlu_full_sv_business_ethics, global_mmlu_full_sv_clinical_knowledge, global_mmlu_full_sv_college_biology, global_mmlu_full_sv_college_chemistry, global_mmlu_full_sv_college_computer_science, global_mmlu_full_sv_college_mathematics, @@ -352,10 +314,7 @@ languages: global_mmlu_full_sv_professional_psychology, global_mmlu_full_sv_public_relations, global_mmlu_full_sv_security_studies, global_mmlu_full_sv_social_sciences, global_mmlu_full_sv_sociology, global_mmlu_full_sv_stem, global_mmlu_full_sv_us_foreign_policy, global_mmlu_full_sv_virology, global_mmlu_full_sv_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: tur_Latn - scope: single tasks: [global_mmlu_full_tr, global_mmlu_full_tr_abstract_algebra, global_mmlu_full_tr_anatomy, global_mmlu_full_tr_astronomy, global_mmlu_full_tr_business_ethics, global_mmlu_full_tr_clinical_knowledge, global_mmlu_full_tr_college_biology, global_mmlu_full_tr_college_chemistry, global_mmlu_full_tr_college_computer_science, global_mmlu_full_tr_college_mathematics, @@ -376,10 +335,7 @@ languages: global_mmlu_full_tr_professional_psychology, global_mmlu_full_tr_public_relations, global_mmlu_full_tr_security_studies, global_mmlu_full_tr_social_sciences, global_mmlu_full_tr_sociology, global_mmlu_full_tr_stem, global_mmlu_full_tr_us_foreign_policy, global_mmlu_full_tr_virology, global_mmlu_full_tr_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ukr_Cyrl - scope: single tasks: [global_mmlu_full_uk, global_mmlu_full_uk_abstract_algebra, global_mmlu_full_uk_anatomy, global_mmlu_full_uk_astronomy, global_mmlu_full_uk_business_ethics, global_mmlu_full_uk_clinical_knowledge, global_mmlu_full_uk_college_biology, global_mmlu_full_uk_college_chemistry, global_mmlu_full_uk_college_computer_science, global_mmlu_full_uk_college_mathematics, @@ -400,5 +356,3 @@ languages: global_mmlu_full_uk_professional_psychology, global_mmlu_full_uk_public_relations, global_mmlu_full_uk_security_studies, global_mmlu_full_uk_social_sciences, global_mmlu_full_uk_sociology, global_mmlu_full_uk_stem, global_mmlu_full_uk_us_foreign_policy, global_mmlu_full_uk_virology, global_mmlu_full_uk_world_religions] - evidence: https://huggingface.co/datasets/CohereLabs/Global-MMLU - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/global_piqa_prompted.yaml b/configs/evals/global_piqa_prompted.yaml index dcc753d..2b59d38 100644 --- a/configs/evals/global_piqa_prompted.yaml +++ b/configs/evals/global_piqa_prompted.yaml @@ -9,164 +9,73 @@ score: normalize: min: 0 max: 1 - clip: true basis: unresolved note: Provisional scale conversion only; no validated chance correction for the prompted protocol. warning: Normalization and metric selection have not been validated for prompted Global PIQA. The supplied comparison sets exclude this eval. Review the protocol before including its scores. +language_defaults: + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: sqi_Latn - scope: single tasks: [global_piqa_prompted_als_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: bos_Latn - scope: single tasks: [global_piqa_prompted_bos_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: bul_Cyrl - scope: single tasks: [global_piqa_prompted_bul_cyrl] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: cat_Latn - scope: single tasks: [global_piqa_prompted_cat_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ces_Latn - scope: single tasks: [global_piqa_prompted_ces_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [global_piqa_prompted_deu_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: est_Latn - scope: single tasks: [global_piqa_prompted_ekk_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ell_Grek - scope: single tasks: [global_piqa_prompted_ell_grek] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [global_piqa_prompted_eng_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fin_Latn - scope: single tasks: [global_piqa_prompted_fin_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [global_piqa_prompted_fra_latn_fran] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: glg_Latn - scope: single tasks: [global_piqa_prompted_glg_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hrv_Latn - scope: single tasks: [global_piqa_prompted_hrv_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hun_Latn - scope: single tasks: [global_piqa_prompted_hun_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: isl_Latn - scope: single tasks: [global_piqa_prompted_isl_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [global_piqa_prompted_ita_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: kat_Geor - scope: single tasks: [global_piqa_prompted_kat_geor] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lit_Latn - scope: single tasks: [global_piqa_prompted_lit_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: mkd_Cyrl - scope: single tasks: [global_piqa_prompted_mkd_cyrl] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [global_piqa_prompted_nld_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nor_Latn - scope: single tasks: [global_piqa_prompted_nno_latn, global_piqa_prompted_nob_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: pol_Latn - scope: single tasks: [global_piqa_prompted_pol_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [global_piqa_prompted_por_latn_port] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ron_Latn - scope: single tasks: [global_piqa_prompted_ron_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slk_Latn - scope: single tasks: [global_piqa_prompted_slk_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slv_Latn - scope: single tasks: [global_piqa_prompted_slv_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [global_piqa_prompted_spa_latn_spai] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: srp_Cyrl - scope: single tasks: [global_piqa_prompted_srp_cyrl] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: swe_Latn - scope: single tasks: [global_piqa_prompted_swe_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: tur_Latn - scope: single tasks: [global_piqa_prompted_tur_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ukr_Cyrl - scope: single tasks: [global_piqa_prompted_ukr_cyrl] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/gpqa_diamond.yaml b/configs/evals/gpqa_diamond.yaml index ea15fe2..a22297f 100644 --- a/configs/evals/gpqa_diamond.yaml +++ b/configs/evals/gpqa_diamond.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0.25 max: 1 - clip: true basis: uniform_choice note: The evaluator constructs four options and extracts A/B/C/D; uniform valid-letter guessing gives 1/4. sources: - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/GPQADiamond/eval_instruct.py languages: - language: eng_Latn - scope: single tasks: [GPQADiamond] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/gsm8k.yaml b/configs/evals/gsm8k.yaml index 17d646a..e165e67 100644 --- a/configs/evals/gsm8k.yaml +++ b/configs/evals/gsm8k.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Generated numerical answers are not a fixed multiple-choice task; no uniform finite answer space is specified. @@ -17,7 +16,6 @@ normalize: - https://github.com/openai/grade-school-math languages: - language: eng_Latn - scope: single tasks: [gsm8k] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/hellaswag.yaml b/configs/evals/hellaswag.yaml index acda37c..e82c605 100644 --- a/configs/evals/hellaswag.yaml +++ b/configs/evals/hellaswag.yaml @@ -10,96 +10,50 @@ score: normalize: min: 0.25 max: 1 - clip: true basis: uniform_choice note: Four candidate endings; the benchmark reports random performance of 25%. Applied to the translated variants of the same task. sources: - https://rowanzellers.com/hellaswag/ - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/hellaswag/hellaswag.yaml +language_defaults: + evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: eng_Latn - scope: single tasks: [hellaswag] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. - language: cat_Latn - scope: single tasks: [hellaswag_ca] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: dan_Latn - scope: single tasks: [hellaswag_da] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [hellaswag_de] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [hellaswag_es] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eus_Latn - scope: single tasks: [hellaswag_eu] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [hellaswag_fr] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hrv_Latn - scope: single tasks: [hellaswag_hr] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hun_Latn - scope: single tasks: [hellaswag_hu] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [hellaswag_it] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [hellaswag_nl] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [hellaswag_pt] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ron_Latn - scope: single tasks: [hellaswag_ro] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slk_Latn - scope: single tasks: [hellaswag_sk] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: srp_Latn - scope: single tasks: [hellaswag_sr] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag note: The sr dataset examples use Latin script; this overrides the generic Serbian Cyrillic default. - language: swe_Latn - scope: single tasks: [hellaswag_sv] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ukr_Cyrl - scope: single tasks: [hellaswag_uk] - evidence: https://huggingface.co/datasets/alexandrainst/m_hellaswag - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/humaneval.yaml b/configs/evals/humaneval.yaml index 866fdcc..9d65016 100644 --- a/configs/evals/humaneval.yaml +++ b/configs/evals/humaneval.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Executable code pass@1 has no benchmark-defined uniform random-program baseline. sources: - https://github.com/openai/human-eval languages: - language: eng_Latn - scope: single tasks: [HumanEval] evidence: https://github.com/openai/human-eval note: Original English benchmark variant; language is the prompt/question language. Programming-language diff --git a/configs/evals/ifeval.yaml b/configs/evals/ifeval.yaml index bfa37b9..1656e89 100644 --- a/configs/evals/ifeval.yaml +++ b/configs/evals/ifeval.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Strict prompt-level instruction success is not a finite-choice task; no defined uniform random baseline. sources: - https://github.com/google-research/google-research/tree/master/instruction_following_eval languages: - language: eng_Latn - scope: single tasks: [ifeval] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/include.yaml b/configs/evals/include.yaml index d716cb6..f8ce2b1 100644 --- a/configs/evals/include.yaml +++ b/configs/evals/include.yaml @@ -11,139 +11,78 @@ score: normalize: min: 0.25 max: 1 - clip: true basis: uniform_choice note: The INCLUDE task templates score A/B/C/D against one answer; 1/4 also holds for weighted aggregates of four-choice subjects. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/include/default/Albanian/_albanian_template_yaml - https://huggingface.co/datasets/CohereLabs/include-base-44 +language_defaults: + evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: sqi_Latn - scope: single tasks: [include_base_44_albanian, include_base_44_albanian_arts_humanities, include_base_44_albanian_business_commerce, include_base_44_albanian_health_oriented_education, include_base_44_albanian_social_science, include_base_44_albanian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eus_Latn - scope: single tasks: [include_base_44_basque, include_base_44_basque_professional_certification] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: bul_Cyrl - scope: single tasks: [include_base_44_bulgarian, include_base_44_bulgarian_arts_humanities, include_base_44_bulgarian_social_science, include_base_44_bulgarian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hrv_Latn - scope: single tasks: [include_base_44_croatian, include_base_44_croatian_arts_humanities, include_base_44_croatian_social_science, include_base_44_croatian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [include_base_44_dutch, include_base_44_dutch_applied_science, include_base_44_dutch_arts_humanities, include_base_44_dutch_health_oriented_education, include_base_44_dutch_social_science, include_base_44_dutch_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: est_Latn - scope: single tasks: [include_base_44_estonian, include_base_44_estonian_applied_science, include_base_44_estonian_arts_humanities, include_base_44_estonian_health_oriented_education, include_base_44_estonian_social_science, include_base_44_estonian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fin_Latn - scope: single tasks: [include_base_44_finnish, include_base_44_finnish_applied_science, include_base_44_finnish_arts_humanities, include_base_44_finnish_health_oriented_education, include_base_44_finnish_social_science, include_base_44_finnish_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [include_base_44_french, include_base_44_french_arts_humanities, include_base_44_french_driving_license, include_base_44_french_health_oriented_education, include_base_44_french_social_science, include_base_44_french_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: kat_Geor - scope: single tasks: [include_base_44_georgian, include_base_44_georgian_arts_humanities] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [include_base_44_german, include_base_44_german_driving_license, include_base_44_german_social_science, include_base_44_german_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ell_Grek - scope: single tasks: [include_base_44_greek, include_base_44_greek_arts_humanities, include_base_44_greek_business_commerce, include_base_44_greek_health_oriented_education, include_base_44_greek_medical_license, include_base_44_greek_professional_certification, include_base_44_greek_social_science, include_base_44_greek_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hun_Latn - scope: single tasks: [include_base_44_hungarian, include_base_44_hungarian_applied_science, include_base_44_hungarian_social_science, include_base_44_hungarian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [include_base_44_italian, include_base_44_italian_applied_science, include_base_44_italian_arts_humanities, include_base_44_italian_health_oriented_education, include_base_44_italian_professional_certification, include_base_44_italian_social_science, include_base_44_italian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lit_Latn - scope: single tasks: [include_base_44_lithuanian, include_base_44_lithuanian_arts_humanities, include_base_44_lithuanian_business_commerce, include_base_44_lithuanian_professional_certification, include_base_44_lithuanian_social_science, include_base_44_lithuanian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: mkd_Cyrl - scope: single tasks: [include_base_44_north macedonian, include_base_44_north macedonian_arts_humanities, include_base_44_north macedonian_business_commerce, include_base_44_north macedonian_social_science, include_base_44_north macedonian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: pol_Latn - scope: single tasks: [include_base_44_polish, include_base_44_polish_professional_certification, include_base_44_polish_social_science, include_base_44_polish_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [include_base_44_portuguese, include_base_44_portuguese_applied_science, include_base_44_portuguese_arts_humanities, include_base_44_portuguese_business_commerce, include_base_44_portuguese_health_oriented_education, include_base_44_portuguese_social_science, include_base_44_portuguese_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: srp_Cyrl - scope: single tasks: [include_base_44_serbian, include_base_44_serbian_arts_humanities, include_base_44_serbian_social_science, include_base_44_serbian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [include_base_44_spanish, include_base_44_spanish_arts_humanities, include_base_44_spanish_health_oriented_education, include_base_44_spanish_social_science, include_base_44_spanish_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: tur_Latn - scope: single tasks: [include_base_44_turkish, include_base_44_turkish_arts_humanities, include_base_44_turkish_business_commerce, include_base_44_turkish_social_science, include_base_44_turkish_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ukr_Cyrl - scope: single tasks: [include_base_44_ukrainian, include_base_44_ukrainian_arts_humanities, include_base_44_ukrainian_social_science, include_base_44_ukrainian_stem] - evidence: https://huggingface.co/datasets/CohereLabs/include-base-44 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/jeebench.yaml b/configs/evals/jeebench.yaml index c62af73..9234d01 100644 --- a/configs/evals/jeebench.yaml +++ b/configs/evals/jeebench.yaml @@ -10,7 +10,6 @@ normalize: # Shared 10.55% baseline; Table 2 reports approximately 10.5%. min: 0.1055 max: 1 - clip: true note: Configured 10.55% overall random baseline, aligned with the shared scoring policy. Table 2 of the JEEBench paper reports an approximate 10.5% baseline. It combines uniform single-choice guessing and random option subsets with partial credit for multi-answer questions, assigning zero expected score to integer and numeric @@ -20,7 +19,6 @@ normalize: - https://github.com/mlfoundations/evalchemy/blob/6ed674159b37f740f2353a86f596f49f6ac13c19/eval/chat_benchmarks/JEEBench/eval_instruct.py languages: - language: eng_Latn - scope: single tasks: [JEEBench] evidence: https://huggingface.co/datasets/daman1209arora/jeebench note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/jeopardy.yaml b/configs/evals/jeopardy.yaml index 8e1cfc0..a82c5ff 100644 --- a/configs/evals/jeopardy.yaml +++ b/configs/evals/jeopardy.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Generated exact-match answers have no defined uniform choice set. sources: - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/jeopardy.yaml languages: - language: eng_Latn - scope: single tasks: [jeopardy] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/lambada.yaml b/configs/evals/lambada.yaml index 8f378a5..ee9b46f 100644 --- a/configs/evals/lambada.yaml +++ b/configs/evals/lambada.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Next-word prediction accuracy is not fixed-choice accuracy. Vocabulary and a guessing distribution would be needed. @@ -17,7 +16,6 @@ normalize: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/lambada/lambada_openai.yaml languages: - language: eng_Latn - scope: single tasks: [lambada_openai] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/language_id.yaml b/configs/evals/language_id.yaml index adbf063..7657b26 100644 --- a/configs/evals/language_id.yaml +++ b/configs/evals/language_id.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0.09090909090909091 max: 1 - clip: true basis: uniform_choice note: Each question offers 11 language names with one correct answer. The corpus covers 1,000 languages, but the per-question guess baseline is 1/11. @@ -17,7 +16,6 @@ normalize: - https://github.com/google/BIG-bench/blob/main/bigbench/benchmark_tasks/language_identification/README.md languages: - language: mul - scope: pooled tasks: [bigbench_language_identification_multiple_choice] evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/language_identification note: One aggregate over 1,000 languages; this CSV has no per-language scores. Kept as a pooled multilingual diff --git a/configs/evals/livecodebench.yaml b/configs/evals/livecodebench.yaml index d6c4b06..93699d4 100644 --- a/configs/evals/livecodebench.yaml +++ b/configs/evals/livecodebench.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Generated code is graded by tests; there is no finite choice set to support a 1/K correction. sources: - https://github.com/LiveCodeBench/LiveCodeBench languages: - language: eng_Latn - scope: single tasks: [LiveCodeBench] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/lsat_ar.yaml b/configs/evals/lsat_ar.yaml index 652f28d..fc44e1a 100644 --- a/configs/evals/lsat_ar.yaml +++ b/configs/evals/lsat_ar.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0.2 max: 1 - clip: true basis: uniform_choice note: One correct choice among five LSAT analytical reasoning options; uniform valid-choice guessing gives 1/5. @@ -18,7 +17,6 @@ normalize: - https://huggingface.co/datasets/hails/agieval-lsat-ar languages: - language: eng_Latn - scope: single tasks: [agieval_lsat_ar] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/math_500.yaml b/configs/evals/math_500.yaml index 228acce..9db4585 100644 --- a/configs/evals/math_500.yaml +++ b/configs/evals/math_500.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Generated mathematical answers have varying domains; no single uniform guess distribution is defined. sources: - https://huggingface.co/datasets/HuggingFaceH4/MATH-500 languages: - language: eng_Latn - scope: single tasks: [MATH500] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/mbpp.yaml b/configs/evals/mbpp.yaml index a91bbc3..1d2f696 100644 --- a/configs/evals/mbpp.yaml +++ b/configs/evals/mbpp.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Generated code is graded by tests; there is no benchmark-defined random-program baseline. sources: - https://github.com/google-research/google-research/tree/master/mbpp languages: - language: eng_Latn - scope: single tasks: [mbpp] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/mgsm.yaml b/configs/evals/mgsm.yaml index adde01f..e6c0631 100644 --- a/configs/evals/mgsm.yaml +++ b/configs/evals/mgsm.yaml @@ -9,84 +9,45 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Generated numerical answers follow GSM-style scoring; no fixed uniform answer space is specified. sources: - https://github.com/google-research/url-nlp/tree/main/mgsm +language_defaults: + evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: cat_Latn - scope: single tasks: [global_mgsm_ca] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ces_Latn - scope: single tasks: [global_mgsm_cs] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [global_mgsm_de] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ell_Grek - scope: single tasks: [global_mgsm_el] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [global_mgsm_en] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [global_mgsm_es] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eus_Latn - scope: single tasks: [global_mgsm_eu] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [global_mgsm_fr] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: glg_Latn - scope: single tasks: [global_mgsm_gl] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hun_Latn - scope: single tasks: [global_mgsm_hu] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: srp_Cyrl - scope: single tasks: [global_mgsm_sr] - evidence: https://huggingface.co/datasets/CohereLabs/global-mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [mgsm_native_cot_de] evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [mgsm_native_cot_en] evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [mgsm_native_cot_es] evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [mgsm_native_cot_fr] evidence: https://huggingface.co/datasets/jbross-ibm-research/mgsm - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/mmlu.yaml b/configs/evals/mmlu.yaml index 69c8824..28f8f2a 100644 --- a/configs/evals/mmlu.yaml +++ b/configs/evals/mmlu.yaml @@ -11,7 +11,6 @@ score: normalize: min: 0.25 max: 1 - clip: true basis: uniform_choice note: Original MMLU uses four choices and one correct label. Uniform guessing gives 1/4, including subject summaries. @@ -19,7 +18,6 @@ normalize: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/mmlu/default/_default_template_yaml languages: - language: eng_Latn - scope: single tasks: [mmlu, mmlu_abstract_algebra, mmlu_anatomy, mmlu_astronomy, mmlu_business_ethics, mmlu_clinical_knowledge, mmlu_college_biology, mmlu_college_chemistry, mmlu_college_computer_science, mmlu_college_mathematics, mmlu_college_medicine, mmlu_college_physics, mmlu_computer_security, mmlu_conceptual_physics, mmlu_econometrics, diff --git a/configs/evals/multiblimp.yaml b/configs/evals/multiblimp.yaml index 20342e6..f1b8274 100644 --- a/configs/evals/multiblimp.yaml +++ b/configs/evals/multiblimp.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0.5 max: 1 - clip: true basis: uniform_choice note: A minimal pair compares a grammatical sentence with an ungrammatical one; chance under random preference is 1/2. @@ -18,74 +17,37 @@ normalize: warning: 'Language-grouping approximation: multiblimp_hbs pools Croatian and Serbian. Assign its combined score to Serbian (srp_Latn) because Serbian has more speakers, not because of the dataset''s language proportions. The score still includes both languages and is not a Serbian-only result.' +language_defaults: + evidence: https://huggingface.co/datasets/jumelet/multiblimp + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: bul_Cyrl - scope: single tasks: [multiblimp_bul] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: cat_Latn - scope: single tasks: [multiblimp_cat] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ces_Latn - scope: single tasks: [multiblimp_ces] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: dan_Latn - scope: single tasks: [multiblimp_dan] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [multiblimp_deu] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ell_Grek - scope: single tasks: [multiblimp_ell] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [multiblimp_eng] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: est_Latn - scope: single tasks: [multiblimp_est] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eus_Latn - scope: single tasks: [multiblimp_eus] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fin_Latn - scope: single tasks: [multiblimp_fin] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [multiblimp_fra] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: gle_Latn - scope: single tasks: [multiblimp_gle] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: glg_Latn - scope: single tasks: [multiblimp_glg] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: srp_Latn - scope: single tasks: [multiblimp_hbs] evidence: https://huggingface.co/datasets/jumelet/multiblimp/blob/main/hbs/data.tsv note: 'Grouping convention: assign the pooled Croatian/Serbian score to Serbian, the language with more @@ -93,92 +55,38 @@ languages: estimates approximately 11 million Serbian speakers and 6 million Croatian speakers: https://slavic.ucla.edu/languages/bcs/serbian-background-info/ and https://slavic.ucla.edu/languages/bcs/croatian-background-info/.' - language: hun_Latn - scope: single tasks: [multiblimp_hun] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: isl_Latn - scope: single tasks: [multiblimp_isl] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [multiblimp_ita] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: kat_Geor - scope: single tasks: [multiblimp_kat] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lav_Latn - scope: single tasks: [multiblimp_lav] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lit_Latn - scope: single tasks: [multiblimp_lit] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: mkd_Cyrl - scope: single tasks: [multiblimp_mkd] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [multiblimp_nld] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: pol_Latn - scope: single tasks: [multiblimp_pol] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [multiblimp_por] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ron_Latn - scope: single tasks: [multiblimp_ron] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slk_Latn - scope: single tasks: [multiblimp_slk] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slv_Latn - scope: single tasks: [multiblimp_slv] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [multiblimp_spa] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: sqi_Latn - scope: single tasks: [multiblimp_sqi] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: swe_Latn - scope: single tasks: [multiblimp_swe] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: tur_Latn - scope: single tasks: [multiblimp_tur] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ukr_Cyrl - scope: single tasks: [multiblimp_ukr] - evidence: https://huggingface.co/datasets/jumelet/multiblimp - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/openbookqa.yaml b/configs/evals/openbookqa.yaml index e15b1ba..d182247 100644 --- a/configs/evals/openbookqa.yaml +++ b/configs/evals/openbookqa.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0.25 max: 1 - clip: true basis: uniform_choice note: The benchmark has exactly four choices per question. The paper reports a 25% uniform-guess baseline. sources: - https://arxiv.org/html/1809.02789v1#S3.SS3 languages: - language: eng_Latn - scope: single tasks: [openbookqa] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/opensubtitles.yaml b/configs/evals/opensubtitles.yaml index 07c17ab..f62f544 100644 --- a/configs/evals/opensubtitles.yaml +++ b/configs/evals/opensubtitles.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: 'Select chrf: the referenced harness uses SacreBLEU defaults (char_order=6, word_order=0, beta=2), so this is plain chrF, not chrF++. Prefer an explicitly identified chrF++ result when available, then chrF, @@ -18,404 +17,159 @@ normalize: - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/opensubtitles_multi40/utils.py - https://github.com/mjpost/sacrebleu/blob/master/sacrebleu/metrics/chrf.py - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/api/metrics.py +language_defaults: + evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml + note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, + not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from + views. languages: - source_language: bul_Cyrl target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_bg_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: ces_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_cs_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: dan_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_da_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: deu_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_de_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: ell_Grek target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_el_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: bul_Cyrl - scope: translation tasks: [opensubtitles_multi40_en_to_bg] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: ces_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_cs] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: dan_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_da] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: deu_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_de] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: ell_Grek - scope: translation tasks: [opensubtitles_multi40_en_to_el] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: spa_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_es] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: est_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_et] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: fin_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_fi] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: fra_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_fr] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: hrv_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_hr] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: hun_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_hu] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: ita_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_it] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: lit_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_lt] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: lav_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_lv] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: nld_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_nl] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: nor_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_no] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: pol_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_pl] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: por_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_pt] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: ron_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_ro] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: slk_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_sk] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: slv_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_sl] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: srp_Cyrl - scope: translation tasks: [opensubtitles_multi40_en_to_sr] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: swe_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_sv] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: tur_Latn - scope: translation tasks: [opensubtitles_multi40_en_to_tr] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: eng_Latn target_language: ukr_Cyrl - scope: translation tasks: [opensubtitles_multi40_en_to_uk] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: spa_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_es_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: est_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_et_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: fin_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_fi_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: fra_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_fr_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: hrv_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_hr_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: hun_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_hu_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: ita_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_it_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: lit_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_lt_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: lav_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_lv_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: nld_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_nl_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: nor_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_no_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: pol_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_pl_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: por_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_pt_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: ron_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_ro_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: slk_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_sk_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: slv_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_sl_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: srp_Cyrl target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_sr_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: swe_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_sv_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: tur_Latn target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_tr_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. - source_language: ukr_Cyrl target_language: eng_Latn - scope: translation tasks: [opensubtitles_multi40_uk_to_en] - evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml - note: Source and target from the OpenSubtitles task metadata; bare sr uses the catalogue default script, - not a sample-level script audit. Source and target are explicit for separate translation-into and translation-from - views. diff --git a/configs/evals/operators.yaml b/configs/evals/operators.yaml index 5e47196..664a18d 100644 --- a/configs/evals/operators.yaml +++ b/configs/evals/operators.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: The selected task uses generated exact-match answers, with no specified random-answer distribution. sources: - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/operators languages: - language: eng_Latn - scope: single tasks: [bigbench_operators_generate_until] evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/operators note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/piqa.yaml b/configs/evals/piqa.yaml index cf763f3..e37a607 100644 --- a/configs/evals/piqa.yaml +++ b/configs/evals/piqa.yaml @@ -9,171 +9,79 @@ score: normalize: min: 0.5 max: 1 - clip: true basis: uniform_choice note: The completion tasks compare two solutions. Uniform guessing gives 1/2. Prompted Global PIQA is configured separately and excluded from the supplied comparison sets. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/piqa/piqa.yaml - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/global_piqa/completions/_template +language_defaults: + evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: eng_Latn - scope: single tasks: [piqa] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. - language: sqi_Latn - scope: single tasks: [global_piqa_completions_als_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: bos_Latn - scope: single tasks: [global_piqa_completions_bos_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: bul_Cyrl - scope: single tasks: [global_piqa_completions_bul_cyrl] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: cat_Latn - scope: single tasks: [global_piqa_completions_cat_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ces_Latn - scope: single tasks: [global_piqa_completions_ces_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [global_piqa_completions_deu_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: est_Latn - scope: single tasks: [global_piqa_completions_ekk_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ell_Grek - scope: single tasks: [global_piqa_completions_ell_grek] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [global_piqa_completions_eng_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fin_Latn - scope: single tasks: [global_piqa_completions_fin_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [global_piqa_completions_fra_latn_fran] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: glg_Latn - scope: single tasks: [global_piqa_completions_glg_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hrv_Latn - scope: single tasks: [global_piqa_completions_hrv_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hun_Latn - scope: single tasks: [global_piqa_completions_hun_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: isl_Latn - scope: single tasks: [global_piqa_completions_isl_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [global_piqa_completions_ita_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: kat_Geor - scope: single tasks: [global_piqa_completions_kat_geor] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lit_Latn - scope: single tasks: [global_piqa_completions_lit_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: mkd_Cyrl - scope: single tasks: [global_piqa_completions_mkd_cyrl] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [global_piqa_completions_nld_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nor_Latn - scope: single tasks: [global_piqa_completions_nno_latn, global_piqa_completions_nob_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: pol_Latn - scope: single tasks: [global_piqa_completions_pol_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [global_piqa_completions_por_latn_port] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ron_Latn - scope: single tasks: [global_piqa_completions_ron_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slk_Latn - scope: single tasks: [global_piqa_completions_slk_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slv_Latn - scope: single tasks: [global_piqa_completions_slv_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [global_piqa_completions_spa_latn_spai] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: srp_Cyrl - scope: single tasks: [global_piqa_completions_srp_cyrl] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: swe_Latn - scope: single tasks: [global_piqa_completions_swe_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: tur_Latn - scope: single tasks: [global_piqa_completions_tur_latn] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ukr_Cyrl - scope: single tasks: [global_piqa_completions_ukr_cyrl] - evidence: https://huggingface.co/datasets/mrlbenchmarks/global-piqa-nonparallel - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/polymath.yaml b/configs/evals/polymath.yaml index 55062f3..298c505 100644 --- a/configs/evals/polymath.yaml +++ b/configs/evals/polymath.yaml @@ -7,7 +7,6 @@ score: {scale: 1} normalize: min: 0 max: 1 - clip: true basis: not_applicable note: Generated mathematical answers are scored by exact match; no common finite answer space is defined. sources: @@ -33,34 +32,19 @@ aggregation: sources: - https://qwen-polymath.github.io/#benchmark-score - https://github.com/QwenLM/PolyMath/blob/main/eval/run_eval.py +language_defaults: + evidence: https://huggingface.co/datasets/Qwen/PolyMath + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: deu_Latn - scope: single tasks: [polymath_de_high, polymath_de_low, polymath_de_medium, polymath_de_top] - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [polymath_en_high, polymath_en_low, polymath_en_medium, polymath_en_top] - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [polymath_es_high, polymath_es_low, polymath_es_medium, polymath_es_top] - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [polymath_fr_high, polymath_fr_low, polymath_fr_medium, polymath_fr_top] - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [polymath_it_high, polymath_it_low, polymath_it_medium, polymath_it_top] - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [polymath_pt_high, polymath_pt_low, polymath_pt_medium, polymath_pt_top] - evidence: https://huggingface.co/datasets/Qwen/PolyMath - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/qa_wikidata.yaml b/configs/evals/qa_wikidata.yaml index 1f97d1d..5f460c6 100644 --- a/configs/evals/qa_wikidata.yaml +++ b/configs/evals/qa_wikidata.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: The selected generate-until task uses open-ended exact match; no fixed choice count. sources: - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/qa_wikidata languages: - language: eng_Latn - scope: single tasks: [bigbench_qa_wikidata_generate_until] evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/qa_wikidata note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/repeat_copy_logic.yaml b/configs/evals/repeat_copy_logic.yaml index 3c47725..0b5b381 100644 --- a/configs/evals/repeat_copy_logic.yaml +++ b/configs/evals/repeat_copy_logic.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: The selected task generates output strings; no defined uniform answer space. sources: - https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/repeat_copy_logic languages: - language: eng_Latn - scope: single tasks: [bigbench_repeat_copy_logic_generate_until] evidence: https://github.com/google/BIG-bench/tree/main/bigbench/benchmark_tasks/repeat_copy_logic note: English instructions/prompts; symbolic or numerical content is not a separate natural language. diff --git a/configs/evals/sib_200.yaml b/configs/evals/sib_200.yaml index aa3be4c..c463f5e 100644 --- a/configs/evals/sib_200.yaml +++ b/configs/evals/sib_200.yaml @@ -11,190 +11,84 @@ score: normalize: min: 0.14285714285714285 max: 1 - clip: true basis: uniform_choice note: The OELLM template lists seven topic choices; uniform guessing gives 1/7. The selected field is raw acc, not acc_norm. sources: - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/sib200/_default_template_yaml +language_defaults: + evidence: https://huggingface.co/datasets/Davlan/sib200 + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: sqi_Latn - scope: single tasks: [sib200_als_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: bos_Latn - scope: single tasks: [sib200_bos_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: bul_Cyrl - scope: single tasks: [sib200_bul_Cyrl] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: cat_Latn - scope: single tasks: [sib200_cat_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ces_Latn - scope: single tasks: [sib200_ces_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: dan_Latn - scope: single tasks: [sib200_dan_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: deu_Latn - scope: single tasks: [sib200_deu_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ell_Grek - scope: single tasks: [sib200_ell_Grek] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [sib200_eng_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: est_Latn - scope: single tasks: [sib200_est_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eus_Latn - scope: single tasks: [sib200_eus_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fin_Latn - scope: single tasks: [sib200_fin_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [sib200_fra_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: gle_Latn - scope: single tasks: [sib200_gle_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: glg_Latn - scope: single tasks: [sib200_glg_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hrv_Latn - scope: single tasks: [sib200_hrv_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: hun_Latn - scope: single tasks: [sib200_hun_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: isl_Latn - scope: single tasks: [sib200_isl_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [sib200_ita_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: kat_Geor - scope: single tasks: [sib200_kat_Geor] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lit_Latn - scope: single tasks: [sib200_lit_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: lav_Latn - scope: single tasks: [sib200_lvs_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: mkd_Cyrl - scope: single tasks: [sib200_mkd_Cyrl] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: mlt_Latn - scope: single tasks: [sib200_mlt_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [sib200_nld_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nor_Latn - scope: single tasks: [sib200_nob_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: pol_Latn - scope: single tasks: [sib200_pol_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [sib200_por_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ron_Latn - scope: single tasks: [sib200_ron_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slk_Latn - scope: single tasks: [sib200_slk_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: slv_Latn - scope: single tasks: [sib200_slv_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [sib200_spa_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: srp_Cyrl - scope: single tasks: [sib200_srp_Cyrl] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: swe_Latn - scope: single tasks: [sib200_swe_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: tur_Latn - scope: single tasks: [sib200_tur_Latn] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ukr_Cyrl - scope: single tasks: [sib200_ukr_Cyrl] - evidence: https://huggingface.co/datasets/Davlan/sib200 - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/social_iqa.yaml b/configs/evals/social_iqa.yaml index 85bcb06..3b27824 100644 --- a/configs/evals/social_iqa.yaml +++ b/configs/evals/social_iqa.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0.3333333333333333 max: 1 - clip: true basis: uniform_choice note: The task scores answerA, answerB and answerC; uniform guessing gives 1/3. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/siqa/siqa.yaml languages: - language: eng_Latn - scope: single tasks: [social_iqa] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/squad_v2.yaml b/configs/evals/squad_v2.yaml index 94b9070..547183a 100644 --- a/configs/evals/squad_v2.yaml +++ b/configs/evals/squad_v2.yaml @@ -9,7 +9,6 @@ score: normalize: min: 0 max: 1 - clip: true basis: not_applicable note: F1 mixes span overlap and unanswerable questions; neither 0 nor 50% is an established random baseline. Keep only scale conversion. @@ -17,7 +16,6 @@ normalize: - https://rajpurkar.github.io/SQuAD-explorer/ languages: - language: eng_Latn - scope: single tasks: [squadv2] evidence: https://rajpurkar.github.io/SQuAD-explorer/ note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/winogrande.yaml b/configs/evals/winogrande.yaml index 8d404b0..33cbed6 100644 --- a/configs/evals/winogrande.yaml +++ b/configs/evals/winogrande.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0.5 max: 1 - clip: true basis: uniform_choice note: The evaluator compares option1 and option2; uniform guessing gives 1/2. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/winogrande/preprocess_winogrande.py languages: - language: eng_Latn - scope: single tasks: [winogrande] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/wsc273.yaml b/configs/evals/wsc273.yaml index 19ec50f..a32cc72 100644 --- a/configs/evals/wsc273.yaml +++ b/configs/evals/wsc273.yaml @@ -9,14 +9,12 @@ score: normalize: min: 0.5 max: 1 - clip: true basis: uniform_choice note: The evaluator compares two candidate antecedents; uniform guessing gives 1/2. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/wsc273/default.yaml languages: - language: eng_Latn - scope: single tasks: [wsc273] evidence: https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/task-groups.yaml note: Original English benchmark variant; language is the prompt/question language. diff --git a/configs/evals/x_csqa.yaml b/configs/evals/x_csqa.yaml index 13ea7af..1825fdf 100644 --- a/configs/evals/x_csqa.yaml +++ b/configs/evals/x_csqa.yaml @@ -9,50 +9,28 @@ score: normalize: min: 0.2 max: 1 - clip: true basis: uniform_choice note: Translated CommonsenseQA retains five answer options; uniform guessing gives 1/5. sources: - https://inklab.usc.edu/XCSR/xcsr_datasets#x-csqa - https://github.com/OpenEuroLLM/oellm-eval/blob/8a4b2412a8e8f7f0d95e3845e2164c792add6a79/oellm/resources/custom_lm_eval_tasks/xcsqa/_default_template_yaml +language_defaults: + evidence: https://huggingface.co/datasets/INK-USC/xcsr + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: deu_Latn - scope: single tasks: [xcsqa_deu_Latn] - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: eng_Latn - scope: single tasks: [xcsqa_eng_Latn] - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: fra_Latn - scope: single tasks: [xcsqa_fra_Latn] - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: [xcsqa_ita_Latn] - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: nld_Latn - scope: single tasks: [xcsqa_nld_Latn] - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: pol_Latn - scope: single tasks: [xcsqa_pol_Latn] - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: por_Latn - scope: single tasks: [xcsqa_por_Latn] - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: spa_Latn - scope: single tasks: [xcsqa_spa_Latn] - evidence: https://huggingface.co/datasets/INK-USC/xcsr - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/configs/evals/xcopa.yaml b/configs/evals/xcopa.yaml index d63d6dc..4466d4e 100644 --- a/configs/evals/xcopa.yaml +++ b/configs/evals/xcopa.yaml @@ -9,24 +9,17 @@ score: normalize: min: 0.5 max: 1 - clip: true basis: uniform_choice note: Translated COPA retains choice1 and choice2; uniform guessing gives 1/2. sources: - https://github.com/EleutherAI/lm-evaluation-harness/blob/d6de81643928d653435c431bae19945d41d32520/lm_eval/tasks/xcopa/utils.py +language_defaults: + evidence: https://huggingface.co/datasets/cambridgeltl/xcopa + note: Dataset language resolved from the OELLM task registry and benchmark configuration. languages: - language: est_Latn - scope: single tasks: ['xcopa:et'] - evidence: https://huggingface.co/datasets/cambridgeltl/xcopa - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: ita_Latn - scope: single tasks: ['xcopa:it'] - evidence: https://huggingface.co/datasets/cambridgeltl/xcopa - note: Dataset language resolved from the OELLM task registry and benchmark configuration. - language: tur_Latn - scope: single tasks: ['xcopa:tr'] - evidence: https://huggingface.co/datasets/cambridgeltl/xcopa - note: Dataset language resolved from the OELLM task registry and benchmark configuration. diff --git a/docs/configuration.md b/docs/configuration.md index 002241e..c5d4657 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -105,15 +105,12 @@ score: {scale: 1} normalize: min: 0.25 # Four choices: uniform guessing gets 1/4 correct. max: 1 - clip: true basis: uniform_choice note: Four-choice chance correction is enabled for this example. languages: - language: eng_Latn - scope: single tasks: [example_en] - language: fra_Latn - scope: single tasks: [example_fr] ``` @@ -139,6 +136,38 @@ directories and malformed files stop loading. Errors identify the source file when a single file is invalid. All validation finishes before the builder replaces output or a browser import takes effect. +## Shared language metadata + +Put repeated language evidence and notes in `language_defaults` within the eval +file. A language entry inherits each omitted field and can override either field +independently: + +```yaml +language_defaults: + evidence: https://huggingface.co/datasets/Qwen/PolyMath + note: Language assignments follow the benchmark configuration. +languages: + - language: deu_Latn + tasks: [polymath_de_low, polymath_de_medium, polymath_de_high, polymath_de_top] + - language: eng_Latn + tasks: [polymath_en_low, polymath_en_medium, polymath_en_high, polymath_en_top] + note: A language-specific explanation can replace the shared note. +``` + +`language_defaults` accepts only `evidence` and `note`. Both must be strings; +nonempty evidence must be an HTTP(S) URL. An explicit empty string clears an +inherited field; `null` is invalid. Invalid defaults are rejected even if every +language overrides them. The assembled catalogue and single-file exports contain +the resolved metadata, so they remain self-contained. + +In per-eval files, `scope` is optional: `language` implies `single`, `language: mul` +implies `pooled`, and source/target fields imply `translation`. Both translation +languages are required. Use an explicit `scope: pooled` when a score pools languages +but is grouped under one specific language code. Explicit overrides must agree +with the fields: translation cannot also have `language`, and single/pooled cannot +have source/target fields. Assembly records an explicit scope for every entry. +This uses the declared fields; it does not infer language identity from task names. + ## Complete small example For a portable single file, the format remains `version`, `name`, `evals`, and @@ -157,6 +186,7 @@ For an input row with `value=0.625`, the raw score is 62.5 and the normalized sc | Field | Meaning | |---|---| +| `language_defaults` | Optional per-eval defaults for language `evidence` and `note`; individual language entries override them. Expanded during assembly. | | `languages` | In a per-eval file, the explicit language groups for that eval. Required; use `[]` when unknown. In a complete catalogue, these groups live in the top-level `languages` list. | | `name` | Unique display name of the eval, grouping its variants. | | `category` | Category label used by weighting profiles and breakdowns. | diff --git a/docs/python-api.md b/docs/python-api.md index f31b445..c8ad095 100644 --- a/docs/python-api.md +++ b/docs/python-api.md @@ -90,7 +90,8 @@ config = load_config( The loader reads each eval file from the manifest’s `evals_dir`, resolves relative paths beside that manifest, and validates the combined catalogue. Each eval file -contains its own language assignments. The returned `config["catalogue"]` is a +contains its own language assignments. Shared language evidence/notes and omitted +scopes are resolved during loading; local metadata overrides are preserved. The returned `config["catalogue"]` is a complete in-memory catalogue, so later analysis does not access those files. Existing single-file catalogues remain supported. When passing a dictionary instead of a filename, supply the complete catalogue; a filesystem manifest needs diff --git a/quickdash/config.py b/quickdash/config.py index 11b320e..740bb58 100644 --- a/quickdash/config.py +++ b/quickdash/config.py @@ -19,8 +19,30 @@ def assemble_catalogue(metadata, definitions): for definition in definitions: if not isinstance(definition, dict) or "languages" not in definition: raise ValueError("Each eval definition needs its own languages list") - e = {k: deepcopy(v) for k, v in definition.items() if k != "languages"} + e = { + k: deepcopy(v) + for k, v in definition.items() + if k not in {"languages", "language_defaults"} + } + defaults = definition.get("language_defaults", {}) + object_keys(defaults, {"evidence", "note"}) + validate_language_metadata(defaults) groups = deepcopy(definition["languages"]) + if not isinstance(groups, list): + raise ValueError("languages must be a list") + for index, group in enumerate(groups): + if not isinstance(group, dict): + continue # The ordinary language-group validator rejects this value. + group = {**defaults, **group} + if "scope" not in group: + group["scope"] = ( + "translation" + if "source_language" in group or "target_language" in group + else "pooled" + if group.get("language") == "mul" + else "single" + ) + groups[index] = group validate_catalogue(dict(metadata, evals=[e], languages=groups)) for group in groups: for task in group["tasks"]: @@ -343,15 +365,19 @@ def validate_rules(config): or (value == "mul" and scope != "pooled") ): raise ValueError("Use canonical language codes, such as eng_Latn") - for field in ["note", "evidence"]: - if field in group and not isinstance(group[field], str): - raise ValueError(field + " must be a string") - if group.get("evidence") and not re.match(r"^https?://", group["evidence"]): - raise ValueError("Evidence links must use HTTP or HTTPS") + validate_language_metadata(group) validate_aggregation_config(config) return config +def validate_language_metadata(group): + for field in ["note", "evidence"]: + if field in group and not isinstance(group[field], str): + raise ValueError(field + " must be a string") + if group.get("evidence") and not re.match(r"^https?://", group["evidence"]): + raise ValueError("Evidence links must use HTTP or HTTPS") + + def normalize_score(value, e): if not number(value) and ( not isinstance(value, str) or not DECIMAL.fullmatch(value.strip()) diff --git a/tests/test_engines.py b/tests/test_engines.py index 1e24338..5ace88e 100644 --- a/tests/test_engines.py +++ b/tests/test_engines.py @@ -208,6 +208,152 @@ def test_per_eval_catalogue_assembly(self): self.assertNotIn("error", result) self.assertEqual(result["value"]["models"][0]["score"], 50) + def test_language_metadata_defaults_and_local_overrides(self): + original = fixture() + metadata = {"version": 1, "name": "Fixture"} + definition = { + **original["catalogue"]["evals"][0], + "language_defaults": { + "evidence": "https://example.org/shared", + "note": "Shared explanation", + }, + "languages": deepcopy(original["catalogue"]["languages"]), + } + definition["languages"][1]["note"] = "French-specific explanation" + reference = deepcopy(original) + reference["catalogue"]["languages"][0].update(definition["language_defaults"]) + reference["catalogue"]["languages"][1].update( + evidence="https://example.org/shared", note="French-specific explanation" + ) + case = dict( + config={**original, "catalogue": metadata}, + eval_definitions=[definition], + rows=paired([row(), row("e_fr")]), + operation="compare", + ) + before = deepcopy(case) + expected = native( + dict(config=reference, rows=case["rows"], operation="compare") + ) + for result in self.both([case])[0]: + self.assertNotIn("error", result) + self.close(semantics(result), semantics(expected)) + self.assertEqual(case, before) + compiled = assemble_catalogue(metadata, [definition]) + self.assertEqual(compiled, reference["catalogue"]) + # An explicit empty string clears an inherited value; omitted fields inherit independently. + definition["languages"][1].update(evidence="", note="") + reference["catalogue"]["languages"][1].update(evidence="", note="") + expected = native( + dict(config=reference, rows=case["rows"], operation="compare") + ) + for result in self.both([case])[0]: + self.assertNotIn("error", result) + self.close(semantics(result), semantics(expected)) + invalid = [] + for defaults in ( + None, + [], + "note", + {"note": False}, + {"evidence": None}, + {"evidence": "file:///tmp/source"}, + {"scope": "single"}, + {"language": "eng_Latn"}, + ): + bad = deepcopy(case) + bad["eval_definitions"][0]["language_defaults"] = defaults + invalid.append(bad) + # Bad defaults remain invalid even when all local fields override them or there are no groups. + empty = deepcopy(bad) + empty["eval_definitions"][0]["languages"] = [] + invalid.append(empty) + for outputs in self.both(invalid): + for result in outputs: + self.assertIn("error", result) + + def test_language_scope_inference_and_explicit_overrides(self): + c = fixture() + metadata = {"version": 1, "name": "Fixture"} + assignments = [ + dict(language="eng_Latn"), + dict(language="fra_Latn"), + dict(source_language="eng_Latn", target_language="fra_Latn"), + dict(language="mul"), + dict(language="srp_Latn", scope="pooled"), + dict(language="srp_Latn", scope="single"), + ] + cases = [] + for fields, scope in zip( + assignments, + ["single", "single", "translation", "pooled", "pooled", "single"], + ): + group = dict(tasks=["e_en"], **fields) + definition = {**c["catalogue"]["evals"][0], "languages": [group]} + expected = {**group, "scope": scope} + self.assertEqual( + assemble_catalogue(metadata, [definition])["languages"], [expected] + ) + reference = deepcopy(c) + reference["catalogue"]["languages"] = [expected] + case = dict( + config={**c, "catalogue": metadata}, + eval_definitions=[definition], + rows=[row()], + ) + cases.append(case) + for result in self.both([case])[0]: + self.assertNotIn("error", result) + self.close( + semantics(result), + semantics(native(dict(config=reference, rows=[row()]))), + ) + bad_fields = [ + {}, + dict(language="mul", scope="single"), + dict(source_language="eng_Latn"), + dict(target_language="fra_Latn"), + dict( + language="eng_Latn", + source_language="eng_Latn", + target_language="fra_Latn", + ), + dict(language="eng_Latn", scope="translation"), + dict( + source_language="eng_Latn", target_language="fra_Latn", scope="single" + ), + dict(language="eng_Latn", scope=None), + dict(language="eng_Latn", scope="guess"), + ] + invalid = [] + for fields in bad_fields: + definition = { + **c["catalogue"]["evals"][0], + "languages": [dict(tasks=["e_en"], **fields)], + } + invalid.append( + dict( + config={**c, "catalogue": metadata}, + eval_definitions=[definition], + rows=[row()], + ) + ) + for outputs in self.both(invalid): + for result in outputs: + self.assertIn("error", result) + + def test_clipping_defaults_to_true(self): + cases = [] + for clip in (None, True, False): + c = fixture() + if clip is not None: + c["catalogue"]["evals"][0]["normalize"]["clip"] = clip + cases.append(dict(config=c, rows=[row(value=".1")])) + for outputs, score in zip(self.both(cases), (0, 0, -20)): + for result in outputs: + self.assertNotIn("error", result) + self.assertAlmostEqual(result["value"]["models"][0]["score"], score) + def test_published_sample_across_all_shipped_configs(self): # Discover files so adding a profile or set automatically extends parity coverage. rows = parse_csv((ROOT / "examples/sample-evals.csv").read_text()) From 33bcc11a56d0adfd11e2026bcb219938e24b52a1 Mon Sep 17 00:00:00 2001 From: Jonathan Burdge Date: Fri, 2 Oct 2026 11:00:25 +0300 Subject: [PATCH 3/4] Resolve catalogue defaults, eval-set overrides and language exclusions with explicit relaxed matching --- .github/workflows/pages.yml | 2 +- README.md | 6 +- app/analysis.js | 107 ++- app/app.js | 62 +- app/build.py | 9 +- app/eval_config.js | 9 +- app/suite_config.js | 77 ++- app/template.html | 4 +- configs/README.md | 4 +- configs/evals/aime24.yaml | 3 +- configs/evals/aime25.yaml | 3 +- configs/evals/amc23.yaml | 3 +- configs/evals/arc_challenge.yaml | 3 +- configs/evals/arc_easy.yaml | 3 +- configs/evals/belebele.yaml | 3 +- configs/evals/boolq.yaml | 3 +- configs/evals/commonsenseqa.yaml | 3 +- configs/evals/copa.yaml | 3 +- configs/evals/coqa.yaml | 3 +- configs/evals/cs_algorithms.yaml | 3 +- configs/evals/dyck_languages.yaml | 3 +- configs/evals/flores200.yaml | 3 +- configs/evals/global_mmlu.yaml | 3 +- configs/evals/global_piqa_prompted.yaml | 2 +- configs/evals/gpqa_diamond.yaml | 3 +- configs/evals/gsm8k.yaml | 3 +- configs/evals/hellaswag.yaml | 2 +- configs/evals/humaneval.yaml | 3 +- configs/evals/ifeval.yaml | 3 +- configs/evals/include.yaml | 3 +- configs/evals/jeebench.yaml | 3 +- configs/evals/jeopardy.yaml | 3 +- configs/evals/lambada.yaml | 3 +- configs/evals/language_id.yaml | 3 +- configs/evals/livecodebench.yaml | 3 +- configs/evals/lsat_ar.yaml | 3 +- configs/evals/math_500.yaml | 3 +- configs/evals/mbpp.yaml | 3 +- configs/evals/mgsm.yaml | 3 +- configs/evals/mmlu.yaml | 3 +- configs/evals/multiblimp.yaml | 3 +- configs/evals/openbookqa.yaml | 3 +- configs/evals/opensubtitles.yaml | 3 +- configs/evals/operators.yaml | 3 +- configs/evals/piqa.yaml | 3 +- configs/evals/polymath.yaml | 3 +- configs/evals/qa_wikidata.yaml | 3 +- configs/evals/repeat_copy_logic.yaml | 3 +- configs/evals/sib_200.yaml | 3 +- configs/evals/social_iqa.yaml | 3 +- configs/evals/squad_v2.yaml | 3 +- configs/evals/winogrande.yaml | 3 +- configs/evals/wsc273.yaml | 3 +- configs/evals/x_csqa.yaml | 3 +- configs/evals/xcopa.yaml | 3 +- configs/examples/catalogue.yaml | 4 +- configs/examples/eval-set.yaml | 3 + configs/sets/flagship-1.yaml | 859 +----------------------- docs/configuration.md | 59 +- docs/development.md | 4 +- docs/python-api.md | 39 +- quickdash/analysis.py | 213 +++++- quickdash/cli.py | 13 +- quickdash/config.py | 207 ++++-- tests/engine_adapter.cjs | 2 +- tests/test_components.cjs | 8 +- tests/test_data.cjs | 2 +- tests/test_data.py | 4 +- tests/test_engines.py | 437 +++++++++++- tests/test_public_browser.mjs | 77 ++- tests/test_suites.cjs | 13 +- tests/test_warning_policy.cjs | 6 +- tests/test_yaml.cjs | 6 +- 73 files changed, 1248 insertions(+), 1124 deletions(-) create mode 100644 configs/examples/eval-set.yaml diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index 67af5df..4dab1ff 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -63,7 +63,7 @@ jobs: python3 -m app.build --results-dir results --sample-csv examples/sample-evals.csv \ --output output/shared > "$RUNNER_TEMP/quickdash-shared.json" python3 -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \ - --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml \ + --weights configs/examples/weights.yaml --eval-set configs/examples/eval-set.yaml \ --output output/demo > "$RUNNER_TEMP/quickdash-demo.json" mkdir -p output/site cp output/shared/index.html output/site/index.html diff --git a/README.md b/README.md index caf80a2..07ef3cd 100644 --- a/README.md +++ b/README.md @@ -37,7 +37,7 @@ python3 -m venv .venv source .venv/bin/activate python -m pip install -e . python -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \ - --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml --output output/example + --weights configs/examples/weights.yaml --eval-set configs/examples/eval-set.yaml --output output/example open output/example/index.html # macOS; elsewhere, open it in your browser ``` @@ -70,6 +70,8 @@ The builder replaces the files it generates in the chosen output directory, incl - **Eval configuration:** inspect every task's category, languages, selected field, raw alternate fields, normalization, and source notes. Load or export YAML here. - **Warnings:** review missing coverage, scoring inconsistencies, language fallbacks, sample-count issues, and caveats stored in the config. +Catalogue files provide scoring defaults; named eval sets can override the metric, metric filter, or shot count for a whole eval and exclude languages. **Strict matching** is the default. **Relaxed — allow few-shot differences** accepts differing shot counts with explicit warnings and an inconsistency notice; metric, filter, harness, and backend matching remain strict. See [configuration and matching](docs/configuration.md#strict-and-relaxed-matching). + Filters affect inspection views, while composite scores and contribution weights use shared coverage within the selected eval set. A named set with missing requirements is labelled incomplete; it uses the shared subset with redistributed weights. Unmatched measurements are excluded from both compared scores with warnings. Malformed input and invalid selected scores are rejected; failed imports preserve the active dashboard. See the [data-handling policy](docs/configuration.md#data-validation-and-failure-behavior). Language/category breakdowns show descriptive raw averages for ordinary evals. Evals with configured components, including PolyMath, show calculated scores when collapsed; expand them to see individual raw scores, relative component weights, and contributions. PolyMath stores `relative_weight` values of 1, 2, 4, and 8, divided by their total of 15 when scoring; incomplete language/protocol groups are excluded with warnings. Incompatible component configurations are rejected before taking effect. See [component aggregation](docs/configuration.md#weighted-components-within-an-eval). Weighted scores use configured normalization, whose baselines and limitations are visible per eval. Shared numerical scales do not establish comparable difficulty across benchmarks. Unknown/mixed-language scores use the documented English fallback for balancing; this does not change their language labels. @@ -82,7 +84,7 @@ After installing the package as above: ```sh quickdash examples/scores.csv --catalogue configs/examples/catalogue.yaml \ - --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml \ + --weights configs/examples/weights.yaml --eval-set configs/examples/eval-set.yaml \ --compare 'Example A' 'Example B' ``` diff --git a/app/analysis.js b/app/analysis.js index fd60e97..e26889c 100644 --- a/app/analysis.js +++ b/app/analysis.js @@ -5,8 +5,35 @@ const key = r => JSON.stringify(['task','metric','filter','n_shot','harness','ba const avg = xs => xs.length ? xs.reduce((a,b)=>a+b,0)/xs.length : null; const fmt = (x,digits=2) => x === null || !Number.isFinite(x) ? '—' : x.toFixed(digits); const esc = x => String(x??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); -const {parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel,isDemoModel}=typeof module!=='undefined'?require('./eval_config.js'):EvalConfig; -const {parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveConfig,inSuite,suiteCoverage,scopeRows}=typeof module!=='undefined'?require('./suite_config.js'):SuiteConfig; +const {compareText,parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel,isDemoModel}=typeof module!=='undefined'?require('./eval_config.js'):EvalConfig; +const {parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveInputs,resolveConfig,effectiveCatalogue,resolveSuite,inSuite,suiteCoverage,scopeRows}=typeof module!=='undefined'?require('./suite_config.js'):SuiteConfig; +function matchingKey(row,config,matching='strict'){ + const fields=['task','metric','filter','n_shot','harness','backend'],e=matching==='relaxed'?config.evals.find(e=>e.name===row.eval):null; + return JSON.stringify(fields.map(k=>k==='n_shot'&&e&&'shots'in e?String(e.shots):row[k])); +} +function matchingAudit(audit,catalogue,matching='strict'){ + if(!['strict','relaxed'].includes(matching))throw Error('Matching must be strict or relaxed'); + if(matching==='strict')return audit; + const result=audit.map(r=>({...r})),evals=new Map(catalogue.evals.map(e=>[e.name,e])),groups=new Map(); + for(const r of result){ + const e=evals.get(r.eval); + if(!e||!('shots'in e)||r.metric!==e.metric||r.filter!==e.metric_filter||e.select&&!matchTask(e.select,r.task))continue; + const id=JSON.stringify(['checkpoint','task','metric','filter','harness','backend'].map(k=>r[k])); + if(!groups.has(id))groups.set(id,[]);groups.get(id).push(r); + } + for(const rr of groups.values()){ + const e=evals.get(rr[0].eval),distance=Math.min(...rr.map(r=>Math.abs(Number(r.n_shot)-e.shots))); + const nearest=[...new Set(rr.filter(r=>Math.abs(Number(r.n_shot)-e.shots)===distance).map(r=>Number(r.n_shot)))].sort((a,b)=>a-b); + for(const r of rr){ + Object.assign(r,{selected:false,raw_score_100:null,score_100:null});delete r.matching_ambiguous_shots; + if(nearest.length!==1)Object.assign(r,{decision:'Equally close shot settings; excluded',matching_ambiguous_shots:nearest}); + else if(Number(r.n_shot)!==nearest[0])r.decision='Alternate shot setting; a closer setting is available'; + else Object.assign(r,normalizeScore(r.value,e),{selected:true,decision:nearest[0]===e.shots?'Selected for the weighted score':`Using ${nearest[0]} shots despite expected ${e.shots}; relaxed matching`}); + } + } + const seen=new Set();for(const r of result)if(r.selected){const id=measurementId(r);if(seen.has(id))throw Error('Duplicate selected measurement: '+r.task);seen.add(id);} + return result; +} function selectRows(rows,scheme){const selected=auditRows(rows,scheme).filter(r=>r.selected);if(!selected.length)throw Error('No selected measurements');return selected;} function buildCatalogue(rows,scheme){ return scheme.evals.map(f=>{const tasks=new Map();for(const r of rows.filter(r=>r.eval===f.name)){if(!tasks.has(r.task))tasks.set(r.task,[]);tasks.get(r.task).push(r);}return {...f,tasks:[...tasks].map(([name,rows])=>({name,rows})).sort((a,b)=>a.name.localeCompare(b.name))};}).filter(f=>f.tasks.length); @@ -105,23 +132,27 @@ function totals(rows,scheme,weights,aggregate='standard',englishWeights=scheme.e const valid=Object.values(weights).every(w=>Number.isFinite(w)&&w>=0)&&Math.abs(Object.values(weights).reduce((s,w)=>s+w,0)-1)<1e-8; return {evals:fs,categories:cats,rowWeights,score:valid&&availableWeight>0&&cats.filter(c=>!c.excluded).every(c=>c.score!==null)?cats.reduce((s,c)=>s+(c.excluded?0:c.score*c.weight),0):null}; } -function pairRows(a,b){const bm=new Map(b.map(r=>[key(r),r]));return a.filter(r=>bm.has(key(r))).map(r=>({...r,a:r.raw_score_100,b:bm.get(key(r)).raw_score_100,delta:r.raw_score_100-bm.get(key(r)).raw_score_100,score_delta:r.score_100-bm.get(key(r)).score_100}));} +function pairRows(a,b,config=null,matching='strict'){ + const match=r=>matchingKey(r,config,matching),bm=new Map(b.map(r=>[match(r),r])); + return a.filter(r=>bm.has(match(r))).map(r=>{const other=bm.get(match(r));return {...r,a:r.raw_score_100,b:other.raw_score_100,delta:r.raw_score_100-other.raw_score_100,score_delta:r.score_100-other.score_100,n_shot_b:other.n_shot,measurement_b:measurementId(other)};}); +} function sampleCount(row){const value=String(row.n_samples??'');return /^[1-9][0-9]*$(?![\s\S])/.test(value)&&Number.isSafeInteger(Number(value))?Number(value):null;} -function comparisonCoverage(a,b,config){ - const left=componentCoverage(a,config),right=componentCoverage(b,config),initial=pairRows(left.rows,right.rows),shared=new Set(initial.map(key)); - const sharedLeft=componentCoverage(left.rows.filter(r=>shared.has(key(r))),config),sharedRight=componentCoverage(right.rows.filter(r=>shared.has(key(r))),config); - const aa=sharedLeft.rows,bb=sharedRight.rows,pairs=pairRows(aa,bb),matched=new Set(pairRows(a,b).map(key)); - const onlyA=a.filter(r=>!matched.has(key(r))),onlyB=b.filter(r=>!matched.has(key(r))),warnings=[]; +function comparisonCoverage(a,b,config,matching='strict'){ + const match=r=>matchingKey(r,config,matching); + const left=componentCoverage(a,config),right=componentCoverage(b,config),initial=pairRows(left.rows,right.rows,config,matching),shared=new Set(initial.map(match)); + const sharedLeft=componentCoverage(left.rows.filter(r=>shared.has(match(r))),config),sharedRight=componentCoverage(right.rows.filter(r=>shared.has(match(r))),config); + const aa=sharedLeft.rows,bb=sharedRight.rows,pairs=pairRows(aa,bb,config,matching),matched=new Set(pairRows(a,b,config,matching).map(match)); + const onlyA=a.filter(r=>!matched.has(match(r))),onlyB=b.filter(r=>!matched.has(match(r))),warnings=[]; for(const [model,coverage] of [['A',left],['B',right],['Shared A',sharedLeft],['Shared B',sharedRight]])for(const w of coverage.warnings)warnings.push({...w,detail:model+': '+w.detail}); const excludedA=a.filter(r=>!aa.includes(r)),excludedB=b.filter(r=>!bb.includes(r)); for(const e of config.evals){const left=onlyA.filter(r=>r.eval===e.name),right=onlyB.filter(r=>r.eval===e.name);if(!left.length&&!right.length)continue; const hasShared=pairs.some(r=>r.eval===e.name); warnings.push({type:'Comparison coverage',name:e.name,eval:e.name,detail:(hasShared?'Unmatched variants are excluded from both scores.':'No matching scores: this eval is excluded from both scores.')+' '+left.length+' measurement(s) available only in A; '+right.length+' only in B. Remaining evals share their category weight; empty categories are excluded and remaining category weights are rescaled.',variants:[[left,'Available only in A (missing from B)'],[right,'Available only in B (missing from A)']].filter(([rr])=>rr.length).map(([rr,settings])=>({settings,tasks:rr.map(r=>r.task+' · '+r.metric+' / '+(r.filter||'blank filter')+' / '+r.n_shot+' shots / '+r.harness+' / '+r.backend)}))}); } - const bm=new Map(bb.map(r=>[key(r),r])); + const bm=new Map(bb.map(r=>[match(r),r])); for(const e of config.evals){ - const mismatch=aa.filter(r=>r.eval===e.name&&sampleCount(r)!==null&&sampleCount(bm.get(key(r)))!==null&&sampleCount(r)!==sampleCount(bm.get(key(r)))); - if(mismatch.length)warnings.push({type:'Sample-count mismatch',name:e.name,eval:e.name,detail:'Matched measurements report different n_samples for A and B. Scores remain included; review dataset coverage before comparing.',variants:[{settings:'Reported sample counts',tasks:mismatch.map(r=>r.task+' · '+r.metric+' · A: '+r.n_samples+' / B: '+bm.get(key(r)).n_samples)}]}); + const mismatch=aa.filter(r=>r.eval===e.name&&sampleCount(r)!==null&&sampleCount(bm.get(match(r)))!==null&&sampleCount(r)!==sampleCount(bm.get(match(r)))); + if(mismatch.length)warnings.push({type:'Sample-count mismatch',name:e.name,eval:e.name,detail:'Matched measurements report different n_samples for A and B. Scores remain included; review dataset coverage before comparing.',variants:[{settings:'Reported sample counts',tasks:mismatch.map(r=>r.task+' · '+r.metric+' · A: '+r.n_samples+' / B: '+bm.get(match(r)).n_samples)}]}); } return {a:aa,b:bb,pairs,warnings,onlyA,onlyB,excludedA,excludedB}; } @@ -206,17 +237,16 @@ function protocolWarning(rows,evalConfig,config,model){ return {type:'Inconsistent scoring settings',name:evalConfig.name,eval:evalConfig.name,model,detail:'Selected variants use different settings: '+variants.map(v=>v.settings+' ('+v.tasks.length+' tasks; '+v.tasks.slice(0,2).join(', ')+(v.tasks.length>2?', …':'')+')').join(' versus ')+'. '+(evalConfig.aggregation?'Only complete component groups within each protocol are included.':'These variants are still included in the aggregate.')+' Review the protocol before comparing languages.',variants}; } function normalizationLabel(e){const n=e.normalize;if(!n||n.min===0&&n.max===1)return n?.basis==='unresolved'?'Unresolved · no correction':'No chance correction';return fmt(n.min*100,2)+'% baseline';} -function sameCoverage(a,b){return a.length===b.length&&pairRows(a,b).length===a.length;} +function sameCoverage(a,b,config=null,matching='strict'){return a.length===b.length&&pairRows(a,b,config,matching).length===a.length;} // Public, serializable analysis boundary used by Python parity tests and the UI. -function compareText(a,b){const aa=Array.from(a,c=>c.codePointAt(0)),bb=Array.from(b,c=>c.codePointAt(0));for(let i=0;ir[k])]);} -const diagnosticTitles={config_caveat:'Config caveat',no_config:'No config',not_used:'Not used',unknown_language:'Unknown language',invalid_sample_count:'Invalid sample count',inconsistent_scoring_settings:'Inconsistent scoring settings',missing_scoring_field:'Missing scoring field',missing_scoring_setting:'Missing scoring setting',no_selected_score:'No selected score',missing_suite_data:'Missing suite data',incomplete_components:'Incomplete components',comparison_coverage:'Comparison coverage',sample_count_mismatch:'Sample-count mismatch',no_category_weight:'No category weight'}; +const diagnosticTitles={config_caveat:'Config caveat',no_config:'No config',not_used:'Not used',unknown_language:'Unknown language',invalid_sample_count:'Invalid sample count',relaxed_shot_setting:'Few-shot mismatch allowed',ambiguous_shot_setting:'Ambiguous few-shot setting',inconsistent_scoring_settings:'Inconsistent scoring settings',missing_scoring_field:'Missing scoring field',missing_scoring_setting:'Missing scoring setting',no_selected_score:'No selected score',missing_suite_data:'Missing suite data',incomplete_components:'Incomplete components',comparison_coverage:'Comparison coverage',sample_count_mismatch:'Sample-count mismatch',no_category_weight:'No category weight'}; function diagnostic(code,model,evalName,rows,detail,effect='included',tasks=null){ const names=[...new Set(tasks??rows.map(r=>r.task))].sort(compareText); return {code,type:diagnosticTitles[code],model,eval:evalName,name:evalName||names[0]||model,tasks:names,measurement_ids:rows.map(measurementId).sort(compareText),effect,detail,variants:names.length?[{settings:detail,tasks:names}]:[]}; } -function reportDiagnostics(audits,config,included,comparison=false){ - const {catalogue,suite,profile}=config,scheme=resolveConfig(catalogue,suite,profile),out=[]; +function reportDiagnostics(audits,config,included,comparison=false,matchingMode='strict',resolved=null){ + const {profile,catalogue,suite,scheme}=resolved||resolveInputs(config.catalogue,config.suite,config.profile),out=[]; const used=new Set([...included.values()].flat().map(r=>r.eval)); for(const e of catalogue.evals)if(e.warning&&used.has(e.name))out.push(diagnostic('config_caveat','Selected comparison',e.name,[],e.warning)); for(const [model,rows] of audits){ @@ -225,6 +255,12 @@ function reportDiagnostics(audits,config,included,comparison=false){ for(const e of catalogue.evals){ const all=rows.filter(r=>r.eval===e.name),outside=all.filter(r=>(!e.select||matchTask(e.select,r.task))&&!inSuite(r,suite)),matching=all.filter(r=>inSuite(r,suite)),selected=matching.filter(r=>r.selected); const add=(code,rr,detail,effect='included',tasks=null)=>out.push(diagnostic(code,model,e.name,rr,detail,effect,tasks)); + if(matchingMode==='relaxed'&&'shots'in e){ + const actuals=[...new Set(selected.filter(r=>acceptedIds.has(measurementId(r))&&Number(r.n_shot)!==e.shots).map(r=>Number(r.n_shot)))].sort((a,b)=>a-b); + for(const actual of actuals){const rr=selected.filter(r=>acceptedIds.has(measurementId(r))&&Number(r.n_shot)===actual);out.push({...diagnostic('relaxed_shot_setting',model,e.name,rr,`Using ${actual} shots despite expected ${e.shots}; allowed by relaxed matching.`),expected_shots:e.shots,actual_shots:actual});} + const ambiguous=matching.filter(r=>r.matching_ambiguous_shots); + if(ambiguous.length)out.push({...diagnostic('ambiguous_shot_setting',model,e.name,ambiguous,`Multiple equally close shot settings for expected ${e.shots}; excluded.`,'excluded'),expected_shots:e.shots,actual_shots:[...new Set(ambiguous.map(r=>Number(r.n_shot)))].sort((a,b)=>a-b)}); + } if(outside.length)add('not_used',outside,'Eval data is present but not selected by '+suite.name+'. Excluded from the calculation.','excluded'); if(!matching.length)continue; const unknown=selected.filter(r=>taskLanguage(r.task,catalogue).status==='unknown'); @@ -235,7 +271,7 @@ function reportDiagnostics(audits,config,included,comparison=false){ if(protocol)add('inconsistent_scoring_settings',selected,protocol.detail); for(const task of [...new Set(matching.filter(r=>!e.select||matchTask(e.select,r.task)).map(r=>r.task))].sort(compareText)){ const rr=matching.filter(r=>r.task===task);if(rr.some(r=>r.selected))continue; - add(rr.some(r=>r.metric===e.metric)?'missing_scoring_setting':'missing_scoring_field',rr,'Excluded: expected '+e.metric+' / '+(e.filter||'(empty)')+('shots'in e?' / '+e.shots+' shots':'')+'.','excluded'); + add(rr.some(r=>r.metric===e.metric)?'missing_scoring_setting':'missing_scoring_field',rr,'Excluded: expected '+e.metric+' / '+(e.metric_filter||'(empty)')+('shots'in e?' / '+e.shots+' shots':'')+'.','excluded'); } if(!selected.length)add('no_selected_score',matching,'No score matches the configured metric, filter, shots and selection. Excluded.','excluded'); } @@ -245,8 +281,8 @@ function reportDiagnostics(audits,config,included,comparison=false){ if(comparison)for(const name of [...new Set(component.rows.filter(r=>!acceptedIds.has(measurementId(r))).map(r=>r.eval))])out.push(diagnostic('comparison_coverage',model,name,component.rows.filter(r=>r.eval===name&&!acceptedIds.has(measurementId(r))),'Unmatched results are excluded from both scores; weights use shared data only.','excluded')); } for(const category of [...new Set([...included.values()].flat().map(r=>r.category))])if(!Object.hasOwn(profile.weights,category))out.push({...diagnostic('no_category_weight','Selected comparison',null,[...included.values()].flat().filter(r=>r.category===category),category+' has no category weight and contributes zero.','zero_weight'),name:category,category}); - if(comparison){const entries=[...included],a=entries[0]?.[1]||[],b=entries[1]?.[1]||a,bm=new Map(b.map(r=>[key(r),r])); - for(const e of catalogue.evals){const rr=a.filter(r=>r.eval===e.name&&bm.has(key(r))&&sampleCount(r)!==null&&sampleCount(bm.get(key(r)))!==null&&sampleCount(r)!==sampleCount(bm.get(key(r))));if(rr.length)out.push(diagnostic('sample_count_mismatch','Selected comparison',e.name,rr.concat(rr.map(r=>bm.get(key(r)))),'Matched results have different sample counts. Scores remain included.'));} + if(comparison){const match=r=>matchingKey(r,scheme,matchingMode),entries=[...included],a=entries[0]?.[1]||[],b=entries[1]?.[1]||a,bm=new Map(b.map(r=>[match(r),r])); + for(const e of catalogue.evals){const rr=a.filter(r=>r.eval===e.name&&bm.has(match(r))&&sampleCount(r)!==null&&sampleCount(bm.get(match(r)))!==null&&sampleCount(r)!==sampleCount(bm.get(match(r))));if(rr.length)out.push(diagnostic('sample_count_mismatch','Selected comparison',e.name,rr.concat(rr.map(r=>bm.get(match(r)))),'Matched results have different sample counts. Scores remain included.'));} } return out; } @@ -268,26 +304,31 @@ function modelReport(model,audit,rows,config,suite){ const allocation=totals(rows,config,config.weights,config.aggregate,config.english_weights),ids=new Set(rows.map(measurementId)); return {model,score:allocation.score,tree:allocationTree(rows,config,allocation,model),measurements:audit.map(r=>({...r,id:measurementId(r),language:taskLanguage(r.task,config),included:ids.has(measurementId(r)),exclusion:r.selected?inSuite(r,suite)?ids.has(measurementId(r))?null:'coverage':'eval_set':'interpretation',effective_weight:allocation.rowWeights.get(r)||0,contribution:ids.has(measurementId(r))?r.score_100*(allocation.rowWeights.get(r)||0):0}))}; } -function prepareAnalysis(rows,config){ - const scheme=resolveConfig(config.catalogue,config.suite,config.profile),audit=auditRows(rows,config.catalogue),models=[...new Set(audit.map(r=>r.checkpoint))].sort(compareText); - return {scheme,audits:new Map(models.map(m=>[m,audit.filter(r=>r.checkpoint===m)]))}; +function prepareAnalysis(rows,config,matching='strict'){ + const resolved=resolveInputs(config.catalogue,config.suite,config.profile),{catalogue,scheme}=resolved,audit=matchingAudit(auditRows(rows,catalogue),catalogue,matching),models=[...new Set(audit.map(r=>r.checkpoint))].sort(compareText); + return {resolved,scheme,audits:new Map(models.map(m=>[m,audit.filter(r=>r.checkpoint===m)]))}; } -function analyze(rows,config){ - const {scheme,audits}=prepareAnalysis(rows,config),included=new Map([...audits].map(([m,rr])=>[m,componentCoverage(scopeRows(rr,config.suite).rows,scheme).rows])); - return {models:[...audits].map(([m,rr])=>modelReport(m,rr,included.get(m),scheme,config.suite)),diagnostics:reportDiagnostics(audits,config,included),coverage:[...audits].map(([model,rr])=>{const scope=scopeRows(rr,config.suite);return {model,missing:scope.missing,complete:!scope.missing.length&&scope.rows.length===included.get(model).length};})}; +function analyze(rows,config,matching='strict'){ + const {resolved,scheme,audits}=prepareAnalysis(rows,config,matching),{suite}=resolved,included=new Map([...audits].map(([m,rr])=>[m,componentCoverage(scopeRows(rr,suite).rows,scheme).rows])); + const diagnostics=reportDiagnostics(audits,config,included,false,matching,resolved); + return {matching,inconsistent:diagnostics.some(d=>d.code==='relaxed_shot_setting'),models:[...audits].map(([m,rr])=>modelReport(m,rr,included.get(m),scheme,suite)),diagnostics,coverage:[...audits].map(([model,rr])=>{const scope=scopeRows(rr,suite);return {model,missing:scope.missing,complete:!scope.missing.length&&scope.rows.length===included.get(model).length};})}; } -function compareAudits(auditA,auditB,config,a,b){ - const scheme=resolveConfig(config.catalogue,config.suite,config.profile),scope=suiteCoverage(auditA,auditB,config.suite),coverage=comparisonCoverage(scope.a,scope.b,scheme),effective=suiteCoverage(coverage.a,coverage.b,config.suite); - const audits=new Map([[a,auditA],[b,auditB]]),included=new Map([[a,coverage.a],[b,coverage.b]]),left=modelReport(a,auditA,coverage.a,scheme,config.suite),right=modelReport(b,auditB,coverage.b,scheme,config.suite); +function compareAudits(auditA,auditB,config,a,b,matching='strict',resolved=null){ + resolved=resolved||resolveInputs(config.catalogue,config.suite,config.profile); + const {catalogue,suite,scheme}=resolved; + auditA=matchingAudit(auditA,catalogue,matching);auditB=matchingAudit(auditB,catalogue,matching); + const scope=suiteCoverage(auditA,auditB,suite,r=>matchingKey(r,scheme,matching)),coverage=comparisonCoverage(scope.a,scope.b,scheme,matching),effective=suiteCoverage(coverage.a,coverage.b,suite,r=>matchingKey(r,scheme,matching)); + const audits=new Map([[a,auditA],[b,auditB]]),included=new Map([[a,coverage.a],[b,coverage.b]]),left=modelReport(a,auditA,coverage.a,scheme,suite),right=modelReport(b,auditB,coverage.b,scheme,suite); const aw=new Map(left.measurements.map(r=>[r.id,r.effective_weight])); - const result={a:left,b:right,delta:left.score===null||right.score===null?null:left.score-right.score,diagnostics:reportDiagnostics(audits,config,included,true),coverage:{complete:config.suite.mode==='fixed'?effective.complete:!coverage.excludedA.length&&!coverage.excludedB.length,required:scope.required,sharedRequired:effective.sharedRequired,presentA:scope.presentA,presentB:scope.presentB,extrasA:scope.extrasA,extrasB:scope.extrasB},deltas:coverage.pairs.map(r=>({task:r.task,measurement_a:measurementId(r),measurement_b:measurementId({...r,checkpoint:b}),raw_delta:r.delta,score_delta:r.score_delta,effective_weight:aw.get(measurementId(r)),contribution_delta:r.score_delta*aw.get(measurementId(r))}))}; + const diagnostics=reportDiagnostics(audits,config,included,true,matching,resolved); + const result={matching,inconsistent:diagnostics.some(d=>d.code==='relaxed_shot_setting'),a:left,b:right,delta:left.score===null||right.score===null?null:left.score-right.score,diagnostics,coverage:{complete:config.suite.mode==='fixed'?effective.complete:!coverage.excludedA.length&&!coverage.excludedB.length,required:scope.required,sharedRequired:effective.sharedRequired,presentA:scope.presentA,presentB:scope.presentB,extrasA:scope.extrasA,extrasB:scope.extrasB},deltas:coverage.pairs.map(r=>({task:r.task,measurement_a:measurementId(r),measurement_b:r.measurement_b,raw_delta:r.delta,score_delta:r.score_delta,effective_weight:aw.get(measurementId(r)),contribution_delta:r.score_delta*aw.get(measurementId(r))}))}; return {result,scope:{...scope,complete:effective.complete,sharedRequired:effective.sharedRequired},...coverage}; } -function compare(rows,config,a,b){ - const {audits}=prepareAnalysis(rows,config);if(!audits.has(a)||!audits.has(b))throw Error('Unknown comparison model'); - return compareAudits(audits.get(a),audits.get(b),config,a,b).result; +function compare(rows,config,a,b,matching='strict'){ + const {audits,resolved}=prepareAnalysis(rows,config);if(!audits.has(a)||!audits.has(b))throw Error('Unknown comparison model'); + return compareAudits(audits.get(a),audits.get(b),config,a,b,matching,resolved).result; } -return {analyze,compare,compareAudits,allocationTree,measurementId,selectRows,buildCatalogue,comparisonRows,weightingLanguage,scoreLanguage,englishAssignment,componentCoverage,evalDistribution,totals,pairRows,sampleCount,comparisonCoverage,synthetic,syntheticOptions,isDemoModel,languageRoles,matchesLanguage,languageCoverage,languageCountLabel,languageLabel,sortBreakdownTree,breakdownAggregate,buildBreakdownTree,protocolWarning,reportDiagnostics,normalizationLabel,sameCoverage,key,avg,fmt,esc,parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel,parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveConfig,inSuite,suiteCoverage,scopeRows}; +return {analyze,compare,compareAudits,matchingAudit,matchingKey,allocationTree,measurementId,selectRows,buildCatalogue,comparisonRows,weightingLanguage,scoreLanguage,englishAssignment,componentCoverage,evalDistribution,totals,pairRows,sampleCount,comparisonCoverage,synthetic,syntheticOptions,isDemoModel,languageRoles,matchesLanguage,languageCoverage,languageCountLabel,languageLabel,sortBreakdownTree,breakdownAggregate,buildBreakdownTree,protocolWarning,reportDiagnostics,normalizationLabel,sameCoverage,key,avg,fmt,esc,parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel,parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveInputs,resolveConfig,effectiveCatalogue,resolveSuite,inSuite,suiteCoverage,scopeRows}; })(); if(typeof module!=='undefined')module.exports=QuickdashAnalysis; diff --git a/app/app.js b/app/app.js index 4ac1ba1..53e39f4 100644 --- a/app/app.js +++ b/app/app.js @@ -1,16 +1,17 @@ 'use strict'; -const {compareAudits,selectRows,buildCatalogue,comparisonRows,weightingLanguage,scoreLanguage,englishAssignment,componentCoverage,evalDistribution,totals,pairRows,sampleCount,comparisonCoverage,synthetic,syntheticOptions,isDemoModel,languageRoles,matchesLanguage,languageCoverage,languageCountLabel,languageLabel,sortBreakdownTree,breakdownAggregate,buildBreakdownTree,protocolWarning,normalizationLabel,sameCoverage,key,avg,fmt,esc,parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel,parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveConfig,inSuite,suiteCoverage}=QuickdashAnalysis; +const {compareAudits,matchingAudit,selectRows,buildCatalogue,comparisonRows,weightingLanguage,scoreLanguage,englishAssignment,componentCoverage,evalDistribution,totals,pairRows,sampleCount,comparisonCoverage,synthetic,syntheticOptions,isDemoModel,languageRoles,matchesLanguage,languageCoverage,languageCountLabel,languageLabel,sortBreakdownTree,breakdownAggregate,buildBreakdownTree,protocolWarning,normalizationLabel,sameCoverage,key,avg,fmt,esc,parseCSV,parseCatalogue,serializeCatalogue,validateCatalogue,normalizeScore,taskLanguage,auditRows,matchTask,demoModel,parseSuite,serializeSuite,parseWeightProfile,serializeWeightProfile,resolveInputs,resolveConfig,effectiveCatalogue,resolveSuite,inSuite,suiteCoverage}=QuickdashAnalysis; if(typeof document!=='undefined')start(); function start(){ const $=id=>document.getElementById(id);let catalogue=DATA.catalogue,suite=DATA.suite,profile=DATA.profile,scheme=DATA.scheme,weights={...scheme.weights},englishWeights={...scheme.english_weights}; + let interpretation=effectiveCatalogue(catalogue,suite),selection=resolveSuite(interpretation,suite); const suites=DATA.suites,profiles=DATA.profiles;let activeSuite='0',activeProfile='0'; let catalogueGroups=new Map(); let models=new Map(),sourceAudits=new Map(),metadata=new Map(DATA.metadata.map(r=>[r.task,r])); - for(const m of DATA.models){const audit=auditRows(DATA.rows.filter(r=>r.checkpoint===m.model),catalogue);sourceAudits.set(m.model,audit);models.set(m.model,audit.filter(r=>r.selected));} - function addSynthetic(target,config){if(!target.size)return;const rows=[...target.values()][0];for(const option of syntheticOptions)target.set(option.name,synthetic(rows,config,option));} - addSynthetic(models,catalogue); - const state={aggregate:scheme.aggregate||'standard',view:'score',scoreCategory:Object.keys(weights)[0],group:'eval',expandedComparisons:new Set(),measure:'raw',sort:'descending',sortBy:'delta',languageSort:'label',languageOrder:'ascending'}; + for(const m of DATA.models){const audit=auditRows(DATA.rows.filter(r=>r.checkpoint===m.model),interpretation);sourceAudits.set(m.model,audit);models.set(m.model,audit.filter(r=>r.selected));} + function addSynthetic(target,config,audits=sourceAudits,matching='strict'){if(!audits.size)return;const rows=matchingAudit([...audits.values()][0],config,matching).filter(r=>r.selected);for(const option of syntheticOptions)target.set(option.name,synthetic(rows,config,option));} + addSynthetic(models,interpretation); + const state={matching:'strict',aggregate:scheme.aggregate||'standard',view:'score',scoreCategory:Object.keys(weights)[0],group:'eval',expandedComparisons:new Set(),measure:'raw',sort:'descending',sortBy:'delta',languageSort:'label',languageOrder:'ascending'}; const languages=r=>languageRoles(r,metadata).map(m=>m.language); const td=x=>''+esc(x)+''; const num=(x,digits=2)=>''+fmt(x,digits)+''; @@ -31,13 +32,13 @@ function start(){ function selected(){ if(!comparisonCache){ const a=$('modelA').value,b=$('modelB').value; - comparisonCache=compareAudits(sourceAudits.get(a)||models.get(a)||[],sourceAudits.get(b)||models.get(b)||[],{catalogue,suite,profile},a,b); + comparisonCache=compareAudits(sourceAudits.get(a)||models.get(a)||[],sourceAudits.get(b)||models.get(b)||[],{catalogue,suite,profile},a,b,state.matching); } return {...comparisonCache,shown:comparisonCache.pairs.filter(filters)}; } function activeWarnings(){ if(!models.size)return []; - const coverage=selected(),warnings=coverage.result.diagnostics.filter(w=>!isDemoModel(w.model)&&(w.code!=='no_category_weight'||weights[w.category]===0)); + const coverage=selected(),warnings=coverage.result.diagnostics.filter(w=>(!isDemoModel(w.model)||w.code==='relaxed_shot_setting')&&(w.code!=='no_category_weight'||weights[w.category]===0)); if(state.aggregate!=='standard')for(const c of totals(coverage.a,scheme,weights,state.aggregate,englishWeights,metadata).categories)if(c.issue)warnings.push({type:'English split unavailable',name:c.name,model:'Selected comparison',detail:c.issue}); return warnings; } @@ -130,11 +131,13 @@ function start(){ function renderWarnings(){const warnings=activeWarnings();return '

Warnings

Coverage for the selected comparison, scoring consistency, and config caveats. A named eval set checks required measurements; unused global catalogue rules are allowed.

'+(warnings.length?table(['Warning','Eval / task','Model','Details'],warnings.map(w=>''+td(w.type)+td(w.name)+td(w.model)+''+esc(w.detail)+(w.variants?'
All affected variants'+w.variants.map(v=>'

'+esc(v.settings)+'
'+v.tasks.map(esc).join('
')+'

').join('')+'
':'')+'')):'

No warnings. The selected comparison has no detected data or configuration issues.

');} function scoringOptions(e,rows){ const used=rows.filter(r=>r.selected),values=(rr,key)=>[...new Set(rr.map(r=>String(r[key])))].sort((a,b)=>key==='n_shot'?Number(a)-Number(b):a.localeCompare(b)); - const shots=values(used,'n_shot'),otherMetrics=values(rows,'metric').filter(m=>m!==e.metric),otherShots=values(rows,'n_shot').filter(n=>!shots.includes(n)),otherFilters=values(rows,'filter').filter(f=>f!==e.filter); - return 'Selected: '+esc(e.metric)+''+(e.aggregation?'Weighted components · expand for weights':'')+(e.metric==='python_pass@1'?'Python solutions passing tests on one attempt.':'')+''+(shots.length===1&&shots[0]==='0'?'0-shot · no examples in the prompt':shots.length?'Shots used: '+esc(shots.join(', '))+' · examples in the prompt':'No selected scores')+''+(e.filter&&e.filter!=='none'?'Answer extraction: '+esc(e.filter)+'':'')+(otherMetrics.length?'Other available metrics: '+esc(otherMetrics.join(', '))+'':'')+(otherShots.length?'Other available shot settings: '+esc(otherShots.join(', '))+'':'')+(otherFilters.length?'Other answer extraction settings: '+esc(otherFilters.map(f=>f||'not specified').join(', '))+'':''); + const shots=values(used,'n_shot'),otherMetrics=values(rows,'metric').filter(m=>m!==e.metric),otherShots=values(rows,'n_shot').filter(n=>!shots.includes(n)),otherFilters=values(rows,'filter').filter(f=>f!==e.metric_filter); + return 'Selected: '+esc(e.metric)+''+(e.aggregation?'Weighted components · expand for weights':'')+(e.metric==='python_pass@1'?'Python solutions passing tests on one attempt.':'')+''+(shots.length===1&&shots[0]==='0'?'0-shot · no examples in the prompt':shots.length?'Shots used: '+esc(shots.join(', '))+' · examples in the prompt':'No selected scores')+''+(e.metric_filter&&e.metric_filter!=='none'?'Metric filter: '+esc(e.metric_filter)+'':'')+(otherMetrics.length?'Other available metrics: '+esc(otherMetrics.join(', '))+'':'')+(otherShots.length?'Other available shot settings: '+esc(otherShots.join(', '))+'':'')+(otherFilters.length?'Other metric filters: '+esc(otherFilters.map(f=>f||'not specified').join(', '))+'':''); } function selectionInfo(e){ - return '

Selection rule: use metric '+esc(e.metric)+'; '+(e.filter?'answer extraction setting '+esc(e.filter)+'':'the CSV extraction-filter field must be blank')+'; '+('shots'in e?'require '+e.shots+' examples in the prompt':'accept any shot count found in the export')+'. A shot is one example provided in the prompt. Other metrics and shot settings remain available for inspection below.

'; + const override=suite.evals?.find(x=>x.name===e.name),fields=['metric','metric_filter','shots'].filter(k=>override&&Object.hasOwn(override,k)); + const source=fields.length?'

Eval-set overrides: '+fields.map(k=>esc(k)+' = '+esc(JSON.stringify(override[k]))+'').join(', ')+'. Other settings inherit from the catalogue.

':'

Scoring settings inherited from the global catalogue.

'; + return source+ '

Selection rule: use metric '+esc(e.metric)+'; '+(e.metric_filter?'metric filter '+esc(e.metric_filter)+'':'the CSV filter column must be blank')+'; '+('shots'in e?'expected '+e.shots+' examples in the prompt'+(state.matching==='relaxed'?' (few-shot differences may be allowed with warnings)':''):'accept any shot count found in the export')+'. A shot is one example provided in the prompt. Other metrics and shot settings remain available for inspection below.

'; } function taskDetails(f,t,missing){ const m=metadata.get(t.name)||{}; @@ -143,7 +146,7 @@ function start(){ const cc=f.aggregation?.components.filter(c=>matchTask(c.match,t.name))||[],component=cc.length===1?cc[0]:null; const componentInfo=component?'

Component: '+esc(component.name)+' · relative weight '+esc(component.relative_weight)+' / '+f.aggregation.components.reduce((sum,c)=>sum+c.relative_weight,0)+'. Applied after normalization, before language and category weights.

':''; const coverage=selected(),used=new Map([[$('modelA').value,new Set(coverage.a.map(key))],[$('modelB').value,new Set(coverage.b.map(key))]]); - const use=r=>!r.selected?['Excluded',r.decision]:!inSuite(r,suite)?['Outside eval set','Excluded by '+suite.name]:used.has(r.checkpoint)&&!used.get(r.checkpoint).has(key(r))?['Excluded from comparison','No complete shared component group or matching A/B measurement; see Warnings.']:['Selected',r.decision]; + const use=r=>!r.selected?['Excluded',r.decision]:!inSuite(r,selection)?['Outside eval set','Excluded by '+suite.name]:used.has(r.checkpoint)&&!used.get(r.checkpoint).has(key(r))?['Excluded from comparison','No complete shared component group or matching A/B measurement; see Warnings.']:['Selected',r.decision]; return warning+info+componentInfo+'

English-balance group: '+esc(englishAssignment({task:t.name},metadata))+'. Used in both English-balance modes when this category’s English share is non-zero.

Raw source scores are shown for every metric. The 0–100 columns apply to selected scores only.

'+table(['Model','Metric','Filter','Shots','Raw source score','Raw / 100','Normalized / 100','Use','Reason','Harness / backend'],t.rows.map(r=>''+td(r.checkpoint)+td(r.metric)+td(r.filter||'Not specified')+''+esc(r.n_shot)+''+esc(r.value)+''+num(r.raw_score_100)+num(r.score_100)+''+esc(use(r)[0])+''+td(use(r)[1])+td(r.harness+' / '+r.backend)+''),[3,4,5,6]); } function catalogueTasks(f,missing){ @@ -160,14 +163,14 @@ function start(){ else if(details.matches('.catalogue-variant')){const box=details.querySelector('.task-details');if(!box.dataset.loaded){const task=entry.f.tasks.find(t=>t.name===details.dataset.task);box.innerHTML=taskDetails(entry.f,task,entry.missing);box.dataset.loaded='true';}} } function renderConfig(){ - const all=[...sourceAudits.values()].flat(),filtered=all.filter(filters),groups=buildCatalogue(filtered,catalogue),warnings=activeWarnings(); + const all=[...sourceAudits.values()].flatMap(rows=>matchingAudit(rows,interpretation,state.matching)),filtered=all.filter(filters),groups=buildCatalogue(filtered,interpretation),warnings=activeWarnings(); $('filterStatus').textContent=new Set(filtered.map(r=>r.task)).size+' of '+new Set(all.map(r=>r.task)).size+' task names · '+filtered.length+' metric rows · all loaded real exports; comparison exclusions appear in Warnings'; let html='

Eval configuration

Category, language, and scoring field for every eval. Expand an eval to inspect its variants.

EvalCategoryScoring optionsLanguagesNormalization
'; catalogueGroups=new Map(); html+=groups.map(f=>{const allRows=all.filter(r=>r.eval===f.name),missing=warnings.filter(w=>w.eval===f.name&&['Missing scoring field','Missing scoring setting'].includes(w.type)),inconsistent=warnings.filter(w=>w.eval===f.name&&w.type==='Inconsistent scoring settings');catalogueGroups.set(f.name,{f,missing});const langs=[...new Set(f.tasks.flatMap(t=>languages(t.rows[0]).map(languageLabel)))].sort();return '
'+esc(f.name)+' ('+f.tasks.length+')'+esc(f.category)+''+scoringOptions(f,allRows)+(missing.length?''+missing.length+' missing scoring field/settings':'')+(inconsistent.length?'Inconsistent scoring settings':'')+''+esc(langs.length>4?langs.slice(0,3).join(', ')+' + '+(langs.length-3)+' more':langs.join(', '))+''+esc(normalizationLabel(f))+(f.warning?'Config warning':'')+''+selectionInfo(f)+normalizationInfo(f)+aggregationInfo(f)+inconsistent.map(w=>'

'+esc(w.model+': '+w.detail)+'

').join('')+'
'+(groups.length===1?catalogueTasks(f,missing):'')+'
'+'
';}).join(''); if(!groups.length)html+='

No evals match the filters.

'; html+='
Scoring assumptions and source files
    '+(scheme.notes||[]).map(n=>'
  1. '+esc(n)+'
  2. ').join('')+'

Source row audit · Language assignments · Analysis JSON

'+esc(DATA.source)+' · SHA-256 '+esc(DATA.sha256)+'

'; - const configControls='
Global eval catalogue

The catalogue defines matching, categories, scoring fields, normalization, and language assignments for every known eval. It does not require models to run all those evals.

Normalization: each eval lists its baseline, formula, and sources below. Chance correction and component aggregation affect calculated scores; individual raw scores stay unchanged. acc_norm is length-normalized answer scoring, not chance correction.

Catalogue export includes every eval and language in one portable YAML file.

Active catalogue: '+esc(catalogue.name)+' · '+catalogue.languages.reduce((n,g)=>n+g.tasks.length,0)+' explicit task language assignments.

Weighting profile and optional eval set

Weights apply independently of the eval set. Any available compares shared recognized measurements and warns on differences. A named set also warns about missing requirements and excludes extra measurements. An incomplete named-set score uses the shared subset with redistributed weights.

Active eval set: '+esc(suite.name)+(suite.exclude?.length?' · Explicitly excluded: '+suite.exclude.map(esc).join(', '):'')+'.

All imports are temporary and stay in your browser. Exports save the three inputs separately. Weight exports include your edits and active score calculation.

'; + const configControls='
Global eval catalogue

The catalogue defines matching, categories, scoring fields, normalization, and language assignments for every known eval. It does not require models to run all those evals.

Normalization: each eval lists its baseline, formula, and sources below. Chance correction and component aggregation affect calculated scores; individual raw scores stay unchanged. acc_norm is length-normalized answer scoring, not chance correction.

Catalogue export includes every eval and language in one portable YAML file.

Active catalogue: '+esc(catalogue.name)+' · '+catalogue.languages.reduce((n,g)=>n+g.tasks.length,0)+' explicit task language assignments.

Weighting profile and optional eval set

Weights apply independently of the eval set. Any available compares shared recognized measurements and warns on differences. A named set requires all known catalogue tasks for each listed eval, except excluded languages. It can override metric, metric_filter, and shots for a whole eval. Missing requirements warn; extra measurements are excluded. An incomplete named-set score uses the shared subset with redistributed weights.

Active eval set: '+esc(suite.name)+(suite.exclude?.length?' · Excluded evals: '+suite.exclude.map(esc).join(', '):'')+(suite.exclude_languages?.length?' · Excluded languages across the set: '+suite.exclude_languages.map(esc).join(', '):'')+(suite.evals?.some(e=>e.exclude_languages?.length)?' · Per-eval language exclusions: '+suite.evals.filter(e=>e.exclude_languages?.length).map(e=>esc(e.name)+': '+e.exclude_languages.map(esc).join(', ')).join('; '):'')+'.

All imports are temporary and stay in your browser. Exports save the three inputs separately. Weight exports include your edits and active score calculation.

'; html=html.replace('
',configControls+'
'); return html; @@ -175,13 +178,17 @@ function start(){ function render(){ comparisonCache=null; const weightOpen=$('weightEditor')?.open,englishOpen=$('englishComponents')?.open; - const {a,b,pairs,shown,excludedA,excludedB,scope}=selected(),ta=totals(a,scheme,weights,state.aggregate,englishWeights,metadata),tb=totals(b,scheme,weights,state.aggregate,englishWeights,metadata),valid=sameCoverage(a,b)&&ta.score!==null&&tb.score!==null; + const {a,b,pairs,shown,excludedA,excludedB,scope}=selected(),ta=totals(a,scheme,weights,state.aggregate,englishWeights,metadata),tb=totals(b,scheme,weights,state.aggregate,englishWeights,metadata),valid=sameCoverage(a,b,scheme,state.matching)&&ta.score!==null&&tb.score!==null; const demo=[$('modelA').value,$('modelB').value].some(isDemoModel); configOptions();$('cards').hidden=!models.size;document.querySelector('.aggregate-controls').hidden=!models.size; const sample=[$('modelA').value,$('modelB').value].some(n=>(DATA.sample_models||[]).includes(n)); $('demo').textContent=!models.size?'Ready for your eval results. Load a CSV to begin.':demo?'Demo comparison: synthetic scores are seeded perturbations of the first loaded model (2-point standard deviation; higher/lower options add/subtract 3 raw score points, clipped to 0–100). For exploration only.':sample?'Sample dataset for exploring Quickdash. Add your own CSVs to compare training methods.':'Real-model comparison · Check evaluation settings and training budgets before drawing a conclusion.'; document.querySelectorAll('[data-aggregate]').forEach(button=>button.setAttribute('aria-pressed',String(button.dataset.aggregate===state.aggregate))); $('aggregateNote').textContent=state.aggregate!=='standard'?(state.aggregate==='english_eval'?'Balance English and other languages inside each eval, then average evals equally within categories.':'Balance English and other languages across each category. Evals with non-English coverage can receive more weight.')+' Shares are set in the weight editor. Unknown or mixed-language scores count as English for weighting; known non-English pools count as other. Raw views stay unchanged.':'Original aggregate: combine configured components within each language/protocol; average those groups within the eval. Other evals average variants. Then average evals within each category.'; + const mismatchWarnings=selected().result.diagnostics.filter(w=>w.code==='relaxed_shot_setting'); + $('matchingNotice').hidden=state.matching!=='relaxed'||!models.size; + $('matchingNotice').className=mismatchWarnings.length?'notice':'caption'; + $('matchingNotice').textContent=mismatchWarnings.length?'INCONSISTENT EVALUATION SETTINGS — Relaxed matching includes '+mismatchWarnings.reduce((n,w)=>n+w.tasks.length,0)+' measurement(s) with unexpected few-shot settings. See Warnings for expected and actual counts.':'Relaxed matching enabled. No few-shot mismatches are included in this comparison.'; $('cards').innerHTML=[[$('modelA').value,ta.score,'A · '+(state.aggregate==='english_eval'?'English balance per eval':state.aggregate==='english_category'?'English balance per category':'original weighted score')],[$('modelB').value,tb.score,'B · '+(state.aggregate==='english_eval'?'English balance per eval':state.aggregate==='english_category'?'English balance per category':'original weighted score')],['A − B',valid?ta.score-tb.score:null,'Weighted difference'+(suite.mode==='fixed'&&!scope.complete?' · incomplete set':'')]].map(([name,value,label])=>'
'+esc(name)+''+fmt(value)+''+label+'
').join(''); $('coverage').classList.toggle('notice',models.size>0&&suite.mode==='fixed'&&!scope.complete); $('coverage').textContent=!models.size?'No models loaded · add your CSV.':(suite.mode==='fixed'?suite.name+' · '+(scope.complete?'Complete':'INCOMPLETE')+' · '+scope.sharedRequired+'/'+scope.required+' requirements shared (A '+scope.presentA+', B '+scope.presentB+') · '+(scope.extrasA+scope.extrasB)+' extra measurements excluded · ':suite.name+' · ')+pairs.length+' matched variants · excluded: '+excludedA.length+' from A, '+excludedB.length+' from B · weights use shared data only'+(valid?'':' · Score unavailable: check weights and language assignments.'); @@ -195,19 +202,16 @@ function start(){ if(!shown.length&&['categories','languages','comparisons'].includes(state.view))$('view').insertAdjacentHTML('beforeend','

No matched variants pass these filters.

'); } function refreshConfig(){state.scoreCategory=Object.keys(weights)[0];filterOptions();clearFilters();languageOptions();} - function importCatalogue(config){ - validateCatalogue(config);const nextScheme=resolveConfig(config,suite,profile); + function reinterpret(nextCatalogue,nextSuite){ + const {scheme:nextScheme,catalogue:nextInterpretation,suite:nextSelection}=resolveInputs(nextCatalogue,nextSuite,profile); const audits=new Map(),nextModels=new Map(),nextMetadata=new Map(); - for(const [name,rows] of sourceAudits){const audit=auditRows(rows,config);audits.set(name,audit);nextModels.set(name,audit.filter(r=>r.selected));for(const row of audit)if(!nextMetadata.has(row.task))nextMetadata.set(row.task,taskLanguage(row.task,config));} - addSynthetic(nextModels,config); - catalogue=config;scheme=nextScheme;sourceAudits=audits;models=nextModels;metadata=nextMetadata; + for(const [name,rows] of sourceAudits){const audit=auditRows(rows,nextInterpretation);matchingAudit(audit,nextInterpretation,state.matching);audits.set(name,audit);nextModels.set(name,audit.filter(r=>r.selected));for(const row of audit)if(!nextMetadata.has(row.task))nextMetadata.set(row.task,taskLanguage(row.task,nextInterpretation));} + addSynthetic(nextModels,nextInterpretation,audits,state.matching); + catalogue=nextCatalogue;suite=nextSuite;scheme=nextScheme;interpretation=nextInterpretation;selection=nextSelection;sourceAudits=audits;models=nextModels;metadata=nextMetadata; weights={...nextScheme.weights,...weights};refreshConfig(); } - function importSuite(config,preset='custom'){ - const next=resolveConfig(catalogue,config,profile);suite=config;scheme=next;activeSuite=preset; - // Changing membership preserves the user's weighting choices and calculation. - weights={...next.weights,...weights};refreshConfig(); - } + function importCatalogue(config){reinterpret(config,suite);} + function importSuite(config,preset='custom'){reinterpret(catalogue,config);activeSuite=preset;} function importWeights(config,preset='custom'){ const next=resolveConfig(catalogue,suite,config);profile=config;scheme=next;activeProfile=preset; weights={...next.weights};englishWeights={...next.english_weights};state.aggregate=next.aggregate;refreshConfig(); @@ -277,15 +281,21 @@ function start(){ $('search').oninput=render;$('clear').onclick=()=>{clearFilters();render();}; $('swap').onclick=()=>{const value=$('modelA').value;$('modelA').value=$('modelB').value;$('modelB').value=value;render();}; for(const [id,presets,apply] of [['suitePreset',suites,importSuite],['weightPreset',profiles,importWeights]])$(id).onchange=()=>{const chosen=$(id).value;if(chosen==='custom')return;try{apply(structuredClone(presets[Number(chosen)].config),chosen);$('error').textContent='';render();}catch(err){$('error').textContent=err.message;configOptions();}}; + $('matching').onchange=()=>{const previous=state.matching,next=$('matching').value;try{ + for(const audit of sourceAudits.values())matchingAudit(audit,interpretation,next); + const nextModels=new Map(models);addSynthetic(nextModels,interpretation,sourceAudits,next); + state.matching=next;models=nextModels;render();$('error').textContent=''; + }catch(error){state.matching=previous;$('matching').value=previous;render();$('error').textContent=error.message;}}; $('clearModels').onclick=()=>{models=new Map();sourceAudits=new Map();metadata=new Map();modelOptions();clearFilters();languageOptions();state.view='score';$('error').textContent='';render();}; $('modelFile').onchange=async e=>{try{ const file=e.target.files[0];if(!file)return; - const audit=auditRows(parseCSV(await file.text()),catalogue),rows=audit.filter(r=>r.selected),names=[...new Set(audit.map(r=>r.checkpoint))]; + const audit=auditRows(parseCSV(await file.text()),interpretation),rows=audit.filter(r=>r.selected),names=[...new Set(audit.map(r=>r.checkpoint))]; + matchingAudit(audit,interpretation,state.matching); if(!audit.length)throw Error('No measurements in CSV');if(names.some(n=>models.has(n)))throw Error('Checkpoint name already loaded; use a distinct model label.'); const nextModels=new Map(models),nextAudits=new Map(sourceAudits),nextMetadata=new Map(metadata); for(const name of names){nextModels.set(name,rows.filter(r=>r.checkpoint===name));nextAudits.set(name,audit.filter(r=>r.checkpoint===name));} for(const r of audit)nextMetadata.set(r.task,taskLanguage(r.task,catalogue)); - if(!nextModels.has(demoModel))addSynthetic(nextModels,catalogue); + addSynthetic(nextModels,interpretation,nextAudits,state.matching); models=nextModels;sourceAudits=nextAudits;metadata=nextMetadata; modelOptions(sourceAudits.size>1?names.at(-1):demoModel);languageOptions();$('error').textContent='';render(); }catch(err){$('error').textContent=err.message;}finally{e.target.value='';}}; diff --git a/app/build.py b/app/build.py index 60abbb9..2f76c26 100644 --- a/app/build.py +++ b/app/build.py @@ -5,7 +5,7 @@ import json from pathlib import Path from quickdash.io import load_csv -from quickdash.config import classify, task_language, load_catalogue, serialize_catalogue, load_profile, load_suite, resolve_config, scope_rows +from quickdash.config import classify, task_language, load_catalogue, serialize_catalogue, load_profile, load_suite, resolve_config, resolve_inputs, scope_rows from quickdash.analysis import totals, component_coverage, diagnostic, analyze APP = Path(__file__).resolve().parent @@ -82,7 +82,8 @@ def build(source, output, catalogue_path=None, results_dir=None, *, weights_path try: resolve_config(catalogue, entry['config'], profile['config']) except ValueError as error: raise ValueError(f'{entry["file"]} / {profile["file"]}: {error}') from error suite, profile = suites[0]['config'], profiles[0]['config'] - config = resolve_config(catalogue, suite, profile) + resolved = resolve_inputs(catalogue, suite, profile) + config, interpretation, selection = (resolved[k] for k in ("scheme", "catalogue", "suite")) paths=[source] if source is not None else directory_files(results_dir,{'.csv'}) if results_dir is not None else [] using_sample = not paths and sample_csv is not None if using_sample: paths = [sample_csv] @@ -90,14 +91,14 @@ def build(source, output, catalogue_path=None, results_dir=None, *, weights_path for path in paths: try: rr=load_csv(path) - classified=classify(rr,catalogue) + classified=classify(rr,interpretation) except ValueError as error:raise ValueError(f'{path.name}: {error}') from error for model in {r['checkpoint'] for r in rr}: if model in owners:raise ValueError(f'Duplicate model name {model!r} in {owners[model]} and {path.name}; combine its results in one file or rename the checkpoint') owners[model]=path.name rows.extend(rr);audit.extend(classified) sources.append(dict(file=path.name,sha256=hashlib.sha256(path.read_bytes()).hexdigest())) - scoped = scope_rows(audit, suite)['rows'] + scoped = scope_rows(audit, selection)['rows'] identities = {tuple(r[k] for k in ['checkpoint','task','metric','filter','n_shot','harness','backend']) for r in scoped} included = {id(r) for r in audit if tuple(r[k] for k in ['checkpoint','task','metric','filter','n_shot','harness','backend']) in identities} scoped_audit = [dict(r, selected=r['selected'] and id(r) in included) for r in audit] diff --git a/app/eval_config.js b/app/eval_config.js index 3be2229..072e922 100644 --- a/app/eval_config.js +++ b/app/eval_config.js @@ -108,10 +108,10 @@ const EvalConfig=(()=>{ if(!Array.isArray(config.evals)||!config.evals.length)throw Error('At least one eval is required'); const names=new Set(); for(const e of config.evals){ - objectKeys(e,['name','category','match','metric','filter','shots','select','score','normalize','warning','aggregation'],['name','category','match','metric','filter','score']); + objectKeys(e,['name','category','match','metric','metric_filter','shots','select','score','normalize','warning','aggregation'],['name','category','match','metric','metric_filter','score']); if(typeof e.name!=='string'||!e.name||names.has(e.name))throw Error('Eval names must be unique and nonempty');names.add(e.name); if(typeof e.category!=='string'||!e.category.trim())throw Error('Eval category must be nonempty text'); - if(typeof e.metric!=='string'||!e.metric||typeof e.filter!=='string')throw Error('Metric and filter must be strings'); + if(typeof e.metric!=='string'||!e.metric||typeof e.metric_filter!=='string')throw Error('Metric and filter must be strings'); validateMatch(e.match);if('select'in e)validateMatch(e.select); if('shots'in e&&(!Number.isSafeInteger(e.shots)||e.shots<0))throw Error('shots must be a nonnegative integer'); objectKeys(e.score,['scale'],['scale']);if(!number(e.score.scale)||e.score.scale<=0)throw Error('Score scale must be positive'); @@ -191,13 +191,14 @@ const EvalConfig=(()=>{ const matches=config.evals.filter(e=>matchTask(e.match,r.task));if(matches.length>1)throw Error('Ambiguous eval config for task: '+r.task);if(!matches.length)return {...r,eval:'',category:'',selected:false,decision:'No eval config; excluded from scoring',raw_score_100:null,score_100:null};const e=matches[0];let decision='Selected for the weighted score'; if(e.select&&!matchTask(e.select,r.task))decision='Excluded summary level or alternate protocol; see eval selection rule'; else if(r.metric!==e.metric)decision='Alternate metric; using '+e.metric; - else if(r.filter!==e.filter)decision='Alternate extraction filter; using '+(e.filter||'(empty)'); + else if(r.filter!==e.metric_filter)decision='Alternate extraction filter; using '+(e.metric_filter||'(empty)'); else if('shots'in e&&String(r.n_shot)!==String(e.shots))decision='Alternate shot setting; using '+e.shots+' shots'; const selected=decision==='Selected for the weighted score';let scores={raw_score_100:null,score_100:null}; if(selected){if(e.aggregation&&e.aggregation.components.filter(c=>matchTask(c.match,r.task)).length!==1)throw Error('Incompatible aggregation config for '+e.name+': '+r.task+' must match exactly one component');try{scores=normalizeScore(r.value,e);}catch(error){throw Error('CSV row '+(index+2)+' · '+r.checkpoint+' · '+r.task+' · '+r.metric+': '+error.message);}const key=JSON.stringify(['checkpoint','task','metric','filter','n_shot','harness','backend'].map(k=>r[k]));if(seen.has(key))throw Error('Duplicate selected measurement: '+r.checkpoint+' · '+r.task+' · '+r.metric);seen.add(key);} return {...r,eval:e.name,category:e.category,selected,decision,...scores}; }); } - return {assembleCatalogue,validateAggregationConfig,validateAggregationSelection,parseCatalogue,serializeCatalogue,validateCatalogue,validateWeights,parseCSV,validateConfig,matchTask,normalizeScore,taskLanguage,auditRows,demoModel,isDemoModel}; +function compareText(a,b){const aa=Array.from(a,c=>c.codePointAt(0)),bb=Array.from(b,c=>c.codePointAt(0));for(let i=0;i{ const yaml=typeof module!=='undefined'?require('./vendor/js-yaml.js'):jsyaml; const keys=(o,allowed,required=[])=>{if(!o||typeof o!=='object'||Array.isArray(o)||Object.keys(o).some(k=>!allowed.includes(k))||required.some(k=>!Object.hasOwn(o,k)))throw Error('Invalid config fields; allowed '+allowed.join(', '));}; const text=v=>typeof v==='string'&&v.trim().length>0; + function validateLanguageExclusions(o){ + if('exclude_languages'in o&&(!Array.isArray(o.exclude_languages)||o.exclude_languages.some(x=>typeof x!=='string'||!/^(?:[a-z]{3}_[A-Z][a-z]{3}|mul)$/.test(x))||new Set(o.exclude_languages).size!==o.exclude_languages.length))throw Error('exclude_languages must be unique canonical language codes'); + } function validateSuite(s){ - keys(s,['version','name','mode','evals','exclude','notes'],['version','name','mode']); + keys(s,['version','name','mode','evals','exclude','exclude_languages','notes'],['version','name','mode']); if(s.version!==1||!text(s.name))throw Error('Suite requires version 1 and a name'); if(!['available','fixed'].includes(s.mode))throw Error('Suite mode must be available or fixed'); if('notes'in s&&(!Array.isArray(s.notes)||s.notes.some(n=>typeof n!=='string')))throw Error('Suite notes must be strings'); if('exclude'in s&&(s.mode!=='available'||!Array.isArray(s.exclude)||s.exclude.some(n=>!text(n))||new Set(s.exclude).size!==s.exclude.length))throw Error('exclude must be a unique list of eval names in available mode'); + validateLanguageExclusions(s); if(s.mode==='available'){if('evals'in s)throw Error('Available mode does not declare required evals');return s;} if(!Array.isArray(s.evals)||!s.evals.length)throw Error('Fixed suite needs required evals'); const names=new Set(); for(const e of s.evals){ - keys(e,['name','variants'],['name']);if(!text(e.name)||names.has(e.name))throw Error('Suite eval names must be unique');names.add(e.name); + keys(e,['name','variants','metric','metric_filter','shots','exclude_languages'],['name']);validateLanguageExclusions(e); + if('metric'in e&&!text(e.metric))throw Error('metric must be nonempty text'); + if('metric_filter'in e&&typeof e.metric_filter!=='string')throw Error('metric_filter must be text'); + if('shots'in e&&(!Number.isSafeInteger(e.shots)||e.shots<0))throw Error('shots must be a nonnegative integer'); + if(!text(e.name)||names.has(e.name))throw Error('Suite eval names must be unique');names.add(e.name); if(!('variants'in e))continue; if(!Array.isArray(e.variants)||!e.variants.length)throw Error('Required variants must be a nonempty list'); const seen=new Map(); @@ -38,21 +46,57 @@ const SuiteConfig=(()=>{ } const parseWeightProfile=source=>validateWeightProfile(yaml.load(source,{schema:yaml.CORE_SCHEMA})); const serializeWeightProfile=p=>yaml.dump(validateWeightProfile(p),{schema:yaml.CORE_SCHEMA,lineWidth:110,noRefs:true}); - function resolveConfig(catalogue,suite,profile){ - api.validateCatalogue(catalogue);validateSuite(suite);validateWeightProfile(profile); - const evals=suite.mode==='available'?catalogue.evals:suite.evals.map(required=>{ - const e=catalogue.evals.find(e=>e.name===required.name);if(!e)throw Error('Suite eval has no catalogue rule: '+required.name); - for(const v of required.variants||[]){ + function catalogueTasks(catalogue,e){ + const tasks=new Set(catalogue.languages.flatMap(g=>g.tasks));if('name'in e.match)tasks.add(e.match.name); + return [...tasks].filter(t=>api.matchTask(e.match,t)&&(!e.select||api.matchTask(e.select,t))).sort(api.compareText); + } + function effectiveCatalogue(catalogue,suite){ + api.validateCatalogue(catalogue);validateSuite(suite); + const result=structuredClone(catalogue),byName=new Map(result.evals.map(e=>[e.name,e])); + for(const name of suite.exclude||[])if(!byName.has(name))throw Error('Excluded eval has no catalogue rule: '+name); + for(const setting of suite.evals||[]){ + const e=byName.get(setting.name);if(!e)throw Error('Suite eval has no catalogue rule: '+setting.name); + for(const k of ['metric','metric_filter','shots'])if(Object.hasOwn(setting,k))e[k]=setting[k]; + } + return api.validateCatalogue(result); + } + function resolveSuite(catalogue,suite){ + validateSuite(suite); + const byName=new Map(catalogue.evals.map(e=>[e.name,e])),taskSets=new Map(catalogue.evals.map(e=>[e.name,catalogueTasks(catalogue,e)])); + const metadata=new Map(catalogue.languages.flatMap(g=>g.tasks.map(t=>[t,g]))); + function excluded(languages,tasks,context){ + const codes=t=>['language','source_language','target_language'].map(k=>metadata.get(t)?.[k]); + const known=new Set(tasks.flatMap(codes)); + for(const language of languages)if(!known.has(language))throw Error('Excluded language has no catalogue assignment in '+context+': '+language); + return new Set(tasks.filter(t=>codes(t).some(c=>languages.includes(c)))); + } + const removed=excluded(suite.exclude_languages||[],[...new Set([...taskSets.values()].flat())],'catalogue'),result=structuredClone(suite); + if(suite.mode==='available'){ + for(const name of suite.exclude||[])if(!byName.has(name))throw Error('Excluded eval has no catalogue rule: '+name); + result._excluded_tasks=[...removed].sort(api.compareText);return result; + } + for(const required of result.evals){ + const e=byName.get(required.name);if(!e)throw Error('Suite eval has no catalogue rule: '+required.name); + const tasks=taskSets.get(e.name),local=excluded(required.exclude_languages||[],tasks,e.name),variants=required.variants||tasks.map(task=>({task})); + if(!variants.length)throw Error('Required eval needs known tasks in the catalogue: '+e.name); + for(const v of variants){ const matches=catalogue.evals.filter(rule=>api.matchTask(rule.match,v.task)); - if(matches.length!==1||matches[0].name!==e.name||e.select&&!api.matchTask(e.select,v.task)||'shots'in e&&'n_shot'in v&&e.shots!==v.n_shot)throw Error('Required variant is not selected by its catalogue rule: '+v.task); + if(!tasks.includes(v.task)||matches.length!==1||matches[0].name!==e.name||'shots'in e&&'n_shot'in v&&e.shots!==v.n_shot)throw Error('Required variant is not selected by its catalogue rule: '+v.task); } - if(required.variants)api.validateAggregationSelection(e,required.variants,catalogue); - return e; - }); - return api.validateConfig({version:1,name:catalogue.name,evals,languages:catalogue.languages,weights:{...profile.weights,...Object.fromEntries(evals.filter(e=>!Object.hasOwn(profile.weights,e.category)).map(e=>[e.category,0]))},english_weights:{...profile.english_weights},aggregate:profile.aggregate||'standard',notes:[...(catalogue.notes||[]),...(profile.notes||[])]}); + required.variants=variants.filter(v=>!removed.has(v.task)&&!local.has(v.task)); + api.validateAggregationSelection(e,required.variants,catalogue); + } + return result; + } + function resolveInputs(catalogue,suite,profile){ + catalogue=effectiveCatalogue(catalogue,suite);const resolved=resolveSuite(catalogue,suite);validateWeightProfile(profile); + const names=new Set((resolved.evals||[]).map(e=>e.name)),evals=catalogue.evals.filter(e=>suite.mode==='available'||names.has(e.name)); + const scheme=api.validateConfig({version:1,name:catalogue.name,evals,languages:catalogue.languages,weights:{...profile.weights,...Object.fromEntries(evals.filter(e=>!Object.hasOwn(profile.weights,e.category)).map(e=>[e.category,0]))},english_weights:{...profile.english_weights},aggregate:profile.aggregate||'standard',notes:[...(catalogue.notes||[]),...(profile.notes||[])]}); + return {catalogue,suite:resolved,profile,scheme}; } + const resolveConfig=(catalogue,suite,profile)=>resolveInputs(catalogue,suite,profile).scheme; function inSuite(row,suite){ - if(suite.mode==='available')return !(suite.exclude||[]).includes(row.eval); + if(suite.mode==='available')return !(suite.exclude||[]).includes(row.eval)&&!(suite._excluded_tasks||[]).includes(row.task); const e=suite.evals.find(e=>e.name===row.eval); return !!e&&(!e.variants||e.variants.some(v=>v.task===row.task&&(!('n_shot'in v)||String(v.n_shot)===String(row.n_shot)))); } @@ -64,16 +108,15 @@ const SuiteConfig=(()=>{ } return {rows:included,extras,missing}; } - function suiteCoverage(a,b,suite){ + function suiteCoverage(a,b,suite,identity=r=>JSON.stringify(['task','metric','filter','n_shot','harness','backend'].map(k=>String(r[k]??'')))){ const left=scopeRows(a,suite),right=scopeRows(b,suite),warnings=[]; for(const [side,scope] of [['A',left],['B',right]]){ for(const name of new Set(scope.missing.map(r=>r.eval))){const rr=scope.missing.filter(r=>r.eval===name);warnings.push({type:'Missing suite data',name,eval:name,model:side,detail:rr.length+' requirement(s) missing from '+suite.name+'. Missing results are excluded from both calculations. Comparison is incomplete; shared scores are used with redistributed weights.',variants:[{settings:'Required in '+side,tasks:rr.map(r=>r.task?(r.task+('n_shot'in r&&r.n_shot!==undefined?' · '+r.n_shot+' shots':'')):'Any selected measurement for '+r.eval)}]});} } - const identity=r=>JSON.stringify(['task','metric','filter','n_shot','harness','backend'].map(k=>String(r[k]??''))); const rightKeys=new Set(right.rows.map(identity)),shared=scopeRows(left.rows.filter(r=>rightKeys.has(identity(r))),suite); - const required=suite.mode==='fixed'?suite.evals.reduce((n,e)=>n+(e.variants?.length||1),0):null; + const required=suite.mode==='fixed'?suite.evals.reduce((n,e)=>n+('variants'in e?e.variants.length:1),0):null; return {a:left.rows,b:right.rows,warnings,complete:!shared.missing.length,required,sharedRequired:required===null?null:required-shared.missing.length,presentA:required===null?null:required-left.missing.length,presentB:required===null?null:required-right.missing.length,extrasA:left.extras.length,extrasB:right.extras.length}; } - return {validateWeightProfile,parseWeightProfile,serializeWeightProfile,validateSuite,parseSuite,serializeSuite,resolveConfig,inSuite,scopeRows,suiteCoverage}; + return {resolveInputs,effectiveCatalogue,resolveSuite,validateWeightProfile,parseWeightProfile,serializeWeightProfile,validateSuite,parseSuite,serializeSuite,resolveConfig,inSuite,scopeRows,suiteCoverage}; })(); if(typeof module!=='undefined')module.exports=SuiteConfig; diff --git a/app/template.html b/app/template.html index a973430..f2affaf 100644 --- a/app/template.html +++ b/app/template.html @@ -14,9 +14,9 @@

Quickdash · OELLM evals

DRAFT SCORING SCHEME

-
Weights and eval sets are independent. “Any available” needs no list of evals.
+
Weights and eval sets are independent. “Any available” needs no list of evals.
-

Score calculation

+

Score calculation

Local, offline analysis · Synthetic results are for exploring the interface · Imports and adjustments last until reload
diff --git a/configs/README.md b/configs/README.md index b118706..5c74d16 100644 --- a/configs/README.md +++ b/configs/README.md @@ -5,10 +5,12 @@ Choose the file to edit based on what you want to change: - **Interpret a new eval:** add one YAML file to [evals/](evals/), containing its task matching, category, scoring metric, normalization, and language assignments. [polymath.yaml](evals/polymath.yaml) shows an eval with weighted components; [boolq.yaml](evals/boolq.yaml) is a simpler example. Adding a rule does not require every model to run it. - **Combine component results:** add an `aggregation` rule in that eval’s file. Each component stores a `relative_weight`; PolyMath uses 1, 2, 4, and 8, divided by their sum when scoring. Named sets must select complete component groups with compatible shot settings; incompatible configurations are errors. Missing results within a valid selection warn and exclude the group; see [component aggregation](../docs/configuration.md#weighted-components-within-an-eval). - **Try different weighting:** add a YAML profile to [weights/](weights/). It can be used with any eval set. Category weights, English shares, and the default calculation live here. -- **Require a standard comparison set:** add a YAML file to [sets/](sets/). [flagship-1.yaml](sets/flagship-1.yaml) pins expected tasks and shot counts; [any-available.yaml](sets/any-available.yaml) needs no required-eval list and compares shared data, with an explicit exclusion for unvalidated prompted Global PIQA. +- **Require a standard comparison set:** add a YAML file to [sets/](sets/). [flagship-1.yaml](sets/flagship-1.yaml) names whole eval groups with set-wide and per-eval language exclusions; [any-available.yaml](sets/any-available.yaml) needs no required-eval list and compares shared data, with an explicit exclusion for unvalidated prompted Global PIQA. [catalogue.yaml](catalogue.yaml) is a small manifest pointing to `evals/`. Every `.yaml` or `.yml` file directly in that directory is loaded in filename order; adding a file needs no registration elsewhere. Keep language tasks in the file for the eval they match. The loader rejects misplaced tasks, duplicate names or assignments, and incompatible component configurations before changing the dashboard. An empty `languages: []` is allowed for an eval without known language metadata; unknown-language warnings still apply to its results. +Catalogue `metric`, `metric_filter`, and `shots` are defaults. A named set can override these for a whole eval; omitted values inherit. The dashboard can explicitly relax few-shot matching with warnings. Language exclusions use the catalogue’s assignments, including both translation endpoints. Unknown names are configuration errors. See [set configuration and matching](../docs/configuration.md#strict-and-relaxed-matching). + Put repeated language `evidence` and `note` under `language_defaults` in the eval file; individual language entries can override either field. Ordinary entries need only a canonical `language` and `tasks`: scope is inferred from the declared language fields. Keep an explicit `scope: pooled` when a pooled result is assigned to a specific language label. See [shared language metadata](../docs/configuration.md#shared-language-metadata). Each profile or set needs a distinct `name` within its directory. To change a selector's startup choice, edit that directory's `default.txt` to name one YAML file. The catalogue is selected at build time with `--catalogue`, or temporarily loaded in the browser. diff --git a/configs/evals/aime24.yaml b/configs/evals/aime24.yaml index c1ba499..842fdfc 100644 --- a/configs/evals/aime24.yaml +++ b/configs/evals/aime24.yaml @@ -3,7 +3,8 @@ category: Reasoning match: name: AIME24 metric: accuracy_avg -filter: '' +metric_filter: '' +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/aime25.yaml b/configs/evals/aime25.yaml index f30d9ed..ed340ed 100644 --- a/configs/evals/aime25.yaml +++ b/configs/evals/aime25.yaml @@ -3,7 +3,8 @@ category: Reasoning match: name: AIME25 metric: accuracy_avg -filter: '' +metric_filter: '' +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/amc23.yaml b/configs/evals/amc23.yaml index ddfc87e..e3861c8 100644 --- a/configs/evals/amc23.yaml +++ b/configs/evals/amc23.yaml @@ -3,7 +3,8 @@ category: Reasoning match: name: AMC23 metric: accuracy_avg -filter: '' +metric_filter: '' +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/arc_challenge.yaml b/configs/evals/arc_challenge.yaml index d655d0f..5204329 100644 --- a/configs/evals/arc_challenge.yaml +++ b/configs/evals/arc_challenge.yaml @@ -3,7 +3,8 @@ category: Knowledge match: regex: arc_challenge(?:_mt_.+)? metric: acc_norm -filter: none +metric_filter: none +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/arc_easy.yaml b/configs/evals/arc_easy.yaml index 7c65352..afd8f44 100644 --- a/configs/evals/arc_easy.yaml +++ b/configs/evals/arc_easy.yaml @@ -3,7 +3,8 @@ category: Knowledge match: name: arc_easy metric: acc_norm -filter: none +metric_filter: none +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/belebele.yaml b/configs/evals/belebele.yaml index 68c21a4..eef3a9c 100644 --- a/configs/evals/belebele.yaml +++ b/configs/evals/belebele.yaml @@ -3,7 +3,8 @@ category: Reading match: regex: belebele_.+ metric: acc_norm -filter: none +metric_filter: none +shots: 5 score: scale: 1 normalize: diff --git a/configs/evals/boolq.yaml b/configs/evals/boolq.yaml index 24c8f4c..280b51f 100644 --- a/configs/evals/boolq.yaml +++ b/configs/evals/boolq.yaml @@ -3,7 +3,8 @@ category: Reading match: name: boolq metric: acc -filter: none +metric_filter: none +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/commonsenseqa.yaml b/configs/evals/commonsenseqa.yaml index 2bb5fe4..1ad0330 100644 --- a/configs/evals/commonsenseqa.yaml +++ b/configs/evals/commonsenseqa.yaml @@ -3,7 +3,8 @@ category: Commonsense match: name: commonsense_qa metric: acc -filter: none +metric_filter: none +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/copa.yaml b/configs/evals/copa.yaml index f424b15..01e34f5 100644 --- a/configs/evals/copa.yaml +++ b/configs/evals/copa.yaml @@ -3,7 +3,8 @@ category: Commonsense match: name: copa metric: acc -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/coqa.yaml b/configs/evals/coqa.yaml index 38cca1b..b97ef4b 100644 --- a/configs/evals/coqa.yaml +++ b/configs/evals/coqa.yaml @@ -3,7 +3,8 @@ category: Reading match: name: coqa metric: f1 -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/cs_algorithms.yaml b/configs/evals/cs_algorithms.yaml index 1d45dbc..477ada0 100644 --- a/configs/evals/cs_algorithms.yaml +++ b/configs/evals/cs_algorithms.yaml @@ -3,7 +3,8 @@ category: Code match: name: bigbench_cs_algorithms_generate_until metric: exact_match -filter: strict-match +metric_filter: strict-match +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/dyck_languages.yaml b/configs/evals/dyck_languages.yaml index 34e9999..e925f33 100644 --- a/configs/evals/dyck_languages.yaml +++ b/configs/evals/dyck_languages.yaml @@ -3,7 +3,8 @@ category: Math match: name: bigbench_dyck_languages_generate_until metric: exact_match -filter: strict-match +metric_filter: strict-match +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/flores200.yaml b/configs/evals/flores200.yaml index ad9d038..0ea1d56 100644 --- a/configs/evals/flores200.yaml +++ b/configs/evals/flores200.yaml @@ -3,7 +3,8 @@ category: Translation match: regex: flores200:.+ metric: chrf++ -filter: rescored +metric_filter: rescored +shots: 4 score: scale: 100 normalize: diff --git a/configs/evals/global_mmlu.yaml b/configs/evals/global_mmlu.yaml index 0ffb4e4..29a2ba4 100644 --- a/configs/evals/global_mmlu.yaml +++ b/configs/evals/global_mmlu.yaml @@ -5,7 +5,8 @@ match: select: regex: global_mmlu_full_[a-z]+ metric: acc -filter: none +metric_filter: none +shots: 5 score: scale: 1 normalize: diff --git a/configs/evals/global_piqa_prompted.yaml b/configs/evals/global_piqa_prompted.yaml index 2b59d38..a2a5652 100644 --- a/configs/evals/global_piqa_prompted.yaml +++ b/configs/evals/global_piqa_prompted.yaml @@ -3,7 +3,7 @@ category: Commonsense match: regex: global_piqa_prompted_.+ metric: exact_match -filter: strict_match +metric_filter: strict_match score: scale: 1 normalize: diff --git a/configs/evals/gpqa_diamond.yaml b/configs/evals/gpqa_diamond.yaml index a22297f..5db1482 100644 --- a/configs/evals/gpqa_diamond.yaml +++ b/configs/evals/gpqa_diamond.yaml @@ -3,7 +3,8 @@ category: Knowledge match: name: GPQADiamond metric: accuracy_avg -filter: '' +metric_filter: '' +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/gsm8k.yaml b/configs/evals/gsm8k.yaml index e165e67..b0cd554 100644 --- a/configs/evals/gsm8k.yaml +++ b/configs/evals/gsm8k.yaml @@ -3,7 +3,8 @@ category: Math match: name: gsm8k metric: exact_match -filter: flexible-extract +metric_filter: flexible-extract +shots: 4 score: scale: 1 normalize: diff --git a/configs/evals/hellaswag.yaml b/configs/evals/hellaswag.yaml index e82c605..acd22f3 100644 --- a/configs/evals/hellaswag.yaml +++ b/configs/evals/hellaswag.yaml @@ -3,7 +3,7 @@ category: Commonsense match: regex: hellaswag(?:_.+)? metric: acc_norm -filter: none +metric_filter: none shots: 0 score: scale: 1 diff --git a/configs/evals/humaneval.yaml b/configs/evals/humaneval.yaml index 9d65016..f3d9021 100644 --- a/configs/evals/humaneval.yaml +++ b/configs/evals/humaneval.yaml @@ -3,7 +3,8 @@ category: Code match: name: HumanEval metric: python_pass@1 -filter: '' +metric_filter: '' +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/ifeval.yaml b/configs/evals/ifeval.yaml index 1656e89..74d77c9 100644 --- a/configs/evals/ifeval.yaml +++ b/configs/evals/ifeval.yaml @@ -3,7 +3,8 @@ category: Instruction following match: name: ifeval metric: prompt_level_strict_acc -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/include.yaml b/configs/evals/include.yaml index f8ce2b1..6cb8359 100644 --- a/configs/evals/include.yaml +++ b/configs/evals/include.yaml @@ -5,7 +5,8 @@ match: select: regex: include_base_44_[^_]+ metric: acc -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/jeebench.yaml b/configs/evals/jeebench.yaml index 9234d01..17db105 100644 --- a/configs/evals/jeebench.yaml +++ b/configs/evals/jeebench.yaml @@ -3,7 +3,8 @@ category: Reasoning match: name: JEEBench metric: accuracy_avg -filter: '' +metric_filter: '' +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/jeopardy.yaml b/configs/evals/jeopardy.yaml index a82c5ff..108c91c 100644 --- a/configs/evals/jeopardy.yaml +++ b/configs/evals/jeopardy.yaml @@ -3,7 +3,8 @@ category: Knowledge match: name: jeopardy metric: exact_match -filter: strict-match +metric_filter: strict-match +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/lambada.yaml b/configs/evals/lambada.yaml index ee9b46f..4027348 100644 --- a/configs/evals/lambada.yaml +++ b/configs/evals/lambada.yaml @@ -3,7 +3,8 @@ category: Reading match: name: lambada_openai metric: acc -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/language_id.yaml b/configs/evals/language_id.yaml index 7657b26..de559cc 100644 --- a/configs/evals/language_id.yaml +++ b/configs/evals/language_id.yaml @@ -3,7 +3,8 @@ category: Language match: name: bigbench_language_identification_multiple_choice metric: acc -filter: none +metric_filter: none +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/livecodebench.yaml b/configs/evals/livecodebench.yaml index 93699d4..14fe7a8 100644 --- a/configs/evals/livecodebench.yaml +++ b/configs/evals/livecodebench.yaml @@ -3,7 +3,8 @@ category: Code match: name: LiveCodeBench metric: accuracy_avg -filter: '' +metric_filter: '' +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/lsat_ar.yaml b/configs/evals/lsat_ar.yaml index fc44e1a..8d606e0 100644 --- a/configs/evals/lsat_ar.yaml +++ b/configs/evals/lsat_ar.yaml @@ -3,7 +3,8 @@ category: Reasoning match: name: agieval_lsat_ar metric: acc_norm -filter: none +metric_filter: none +shots: 3 score: scale: 1 normalize: diff --git a/configs/evals/math_500.yaml b/configs/evals/math_500.yaml index 9db4585..d8bc355 100644 --- a/configs/evals/math_500.yaml +++ b/configs/evals/math_500.yaml @@ -3,7 +3,8 @@ category: Math match: name: MATH500 metric: accuracy -filter: '' +metric_filter: '' +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/mbpp.yaml b/configs/evals/mbpp.yaml index 1d2f696..c60d1ee 100644 --- a/configs/evals/mbpp.yaml +++ b/configs/evals/mbpp.yaml @@ -3,7 +3,8 @@ category: Code match: name: mbpp metric: pass_at_1 -filter: none +metric_filter: none +shots: 3 score: scale: 1 normalize: diff --git a/configs/evals/mgsm.yaml b/configs/evals/mgsm.yaml index e6c0631..206e1fa 100644 --- a/configs/evals/mgsm.yaml +++ b/configs/evals/mgsm.yaml @@ -3,7 +3,8 @@ category: Math match: regex: (?:mgsm_native_cot_.+|global_mgsm_.+) metric: exact_match -filter: flexible-extract +metric_filter: flexible-extract +shots: 5 score: scale: 1 normalize: diff --git a/configs/evals/mmlu.yaml b/configs/evals/mmlu.yaml index 28f8f2a..38b974d 100644 --- a/configs/evals/mmlu.yaml +++ b/configs/evals/mmlu.yaml @@ -5,7 +5,8 @@ match: select: name: mmlu metric: acc -filter: none +metric_filter: none +shots: 5 score: scale: 1 normalize: diff --git a/configs/evals/multiblimp.yaml b/configs/evals/multiblimp.yaml index f1b8274..c1f6c43 100644 --- a/configs/evals/multiblimp.yaml +++ b/configs/evals/multiblimp.yaml @@ -3,7 +3,8 @@ category: Language match: regex: multiblimp_.+ metric: acc_norm -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/openbookqa.yaml b/configs/evals/openbookqa.yaml index d182247..cf64317 100644 --- a/configs/evals/openbookqa.yaml +++ b/configs/evals/openbookqa.yaml @@ -3,7 +3,8 @@ category: Knowledge match: name: openbookqa metric: acc_norm -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/opensubtitles.yaml b/configs/evals/opensubtitles.yaml index f62f544..29888b7 100644 --- a/configs/evals/opensubtitles.yaml +++ b/configs/evals/opensubtitles.yaml @@ -3,7 +3,8 @@ category: Translation match: regex: opensubtitles_.+ metric: chrf -filter: none +metric_filter: none +shots: 0 score: scale: 100 normalize: diff --git a/configs/evals/operators.yaml b/configs/evals/operators.yaml index 664a18d..cca1fbe 100644 --- a/configs/evals/operators.yaml +++ b/configs/evals/operators.yaml @@ -3,7 +3,8 @@ category: Math match: name: bigbench_operators_generate_until metric: exact_match -filter: strict-match +metric_filter: strict-match +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/piqa.yaml b/configs/evals/piqa.yaml index e37a607..48b1f74 100644 --- a/configs/evals/piqa.yaml +++ b/configs/evals/piqa.yaml @@ -3,7 +3,8 @@ category: Commonsense match: regex: (?:piqa|global_piqa_completions_.+) metric: acc_norm -filter: none +metric_filter: none +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/polymath.yaml b/configs/evals/polymath.yaml index 298c505..aa9add2 100644 --- a/configs/evals/polymath.yaml +++ b/configs/evals/polymath.yaml @@ -2,7 +2,8 @@ name: PolyMath category: Reasoning match: {regex: 'polymath_.+'} metric: exact_match -filter: none +metric_filter: none +shots: 0 score: {scale: 1} normalize: min: 0 diff --git a/configs/evals/qa_wikidata.yaml b/configs/evals/qa_wikidata.yaml index 5f460c6..63a39e4 100644 --- a/configs/evals/qa_wikidata.yaml +++ b/configs/evals/qa_wikidata.yaml @@ -3,7 +3,8 @@ category: Knowledge match: name: bigbench_qa_wikidata_generate_until metric: exact_match -filter: strict-match +metric_filter: strict-match +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/repeat_copy_logic.yaml b/configs/evals/repeat_copy_logic.yaml index 0b5b381..8cbe247 100644 --- a/configs/evals/repeat_copy_logic.yaml +++ b/configs/evals/repeat_copy_logic.yaml @@ -3,7 +3,8 @@ category: Math match: name: bigbench_repeat_copy_logic_generate_until metric: exact_match -filter: strict-match +metric_filter: strict-match +shots: 10 score: scale: 1 normalize: diff --git a/configs/evals/sib_200.yaml b/configs/evals/sib_200.yaml index c463f5e..7f99834 100644 --- a/configs/evals/sib_200.yaml +++ b/configs/evals/sib_200.yaml @@ -5,7 +5,8 @@ match: # Use acc: acc_norm is not a reliable/useful metric for this eval in our export # (exactly 0.25 for 34 of 36 languages). metric: acc -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/social_iqa.yaml b/configs/evals/social_iqa.yaml index 3b27824..f214391 100644 --- a/configs/evals/social_iqa.yaml +++ b/configs/evals/social_iqa.yaml @@ -3,7 +3,8 @@ category: Commonsense match: name: social_iqa metric: acc -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/squad_v2.yaml b/configs/evals/squad_v2.yaml index 547183a..57bbe2c 100644 --- a/configs/evals/squad_v2.yaml +++ b/configs/evals/squad_v2.yaml @@ -3,7 +3,8 @@ category: Reading match: name: squadv2 metric: f1 -filter: none +metric_filter: none +shots: 10 score: scale: 100 normalize: diff --git a/configs/evals/winogrande.yaml b/configs/evals/winogrande.yaml index 33cbed6..26ed694 100644 --- a/configs/evals/winogrande.yaml +++ b/configs/evals/winogrande.yaml @@ -3,7 +3,8 @@ category: Commonsense match: name: winogrande metric: acc -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/wsc273.yaml b/configs/evals/wsc273.yaml index a32cc72..907ad53 100644 --- a/configs/evals/wsc273.yaml +++ b/configs/evals/wsc273.yaml @@ -3,7 +3,8 @@ category: Commonsense match: name: wsc273 metric: acc -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/x_csqa.yaml b/configs/evals/x_csqa.yaml index 1825fdf..4dce847 100644 --- a/configs/evals/x_csqa.yaml +++ b/configs/evals/x_csqa.yaml @@ -3,7 +3,8 @@ category: Commonsense match: regex: xcsqa_.+ metric: acc_norm -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/evals/xcopa.yaml b/configs/evals/xcopa.yaml index 4466d4e..895430a 100644 --- a/configs/evals/xcopa.yaml +++ b/configs/evals/xcopa.yaml @@ -3,7 +3,8 @@ category: Commonsense match: regex: xcopa:.+ metric: acc -filter: none +metric_filter: none +shots: 0 score: scale: 1 normalize: diff --git a/configs/examples/catalogue.yaml b/configs/examples/catalogue.yaml index fcce2f3..3629744 100644 --- a/configs/examples/catalogue.yaml +++ b/configs/examples/catalogue.yaml @@ -5,7 +5,7 @@ evals: category: Reasoning match: {regex: 'example_reasoning_(en|fr)'} metric: acc_norm - filter: none + metric_filter: none shots: 0 score: {scale: 1} normalize: @@ -17,7 +17,7 @@ evals: category: Math match: {name: example_math_en} metric: exact_match - filter: none + metric_filter: none shots: 0 score: {scale: 1} normalize: diff --git a/configs/examples/eval-set.yaml b/configs/examples/eval-set.yaml new file mode 100644 index 0000000..fc244b6 --- /dev/null +++ b/configs/examples/eval-set.yaml @@ -0,0 +1,3 @@ +version: 1 +name: Any available example evals +mode: available diff --git a/configs/sets/flagship-1.yaml b/configs/sets/flagship-1.yaml index 280926f..c78aa06 100644 --- a/configs/sets/flagship-1.yaml +++ b/configs/sets/flagship-1.yaml @@ -1,904 +1,55 @@ version: 1 name: flagship-1 mode: fixed +exclude_languages: [kat_Geor] notes: - - >- - Expected evals, tasks and shot settings for the flagship-1 comparison. Missing requirements make the score - incomplete; extra measurements are excluded. + - Require the catalogue tasks for each listed eval, inheriting its scoring defaults. + - Exclude Georgian results because of a vocabulary error. Exclude English X-CSQA to avoid overlap with CommonsenseQA. evals: - name: CS Algorithms - variants: - - task: bigbench_cs_algorithms_generate_until - n_shot: 10 - name: HumanEval - variants: - - task: HumanEval - n_shot: 0 - name: LiveCodeBench - variants: - - task: LiveCodeBench - n_shot: 0 - name: MBPP - variants: - - task: mbpp - n_shot: 3 - name: Dyck languages - variants: - - task: bigbench_dyck_languages_generate_until - n_shot: 10 - name: Operators - variants: - - task: bigbench_operators_generate_until - n_shot: 10 - name: Repeat Copy Logic - variants: - - task: bigbench_repeat_copy_logic_generate_until - n_shot: 10 - name: GSM8K - variants: - - task: gsm8k - n_shot: 4 - name: MGSM - variants: - - task: global_mgsm_ca - n_shot: 0 - - task: global_mgsm_cs - n_shot: 0 - - task: global_mgsm_de - n_shot: 0 - - task: global_mgsm_el - n_shot: 0 - - task: global_mgsm_en - n_shot: 0 - - task: global_mgsm_es - n_shot: 0 - - task: global_mgsm_eu - n_shot: 0 - - task: global_mgsm_fr - n_shot: 0 - - task: global_mgsm_gl - n_shot: 0 - - task: global_mgsm_hu - n_shot: 0 - - task: global_mgsm_sr - n_shot: 0 - - task: mgsm_native_cot_de - n_shot: 5 - - task: mgsm_native_cot_en - n_shot: 5 - - task: mgsm_native_cot_es - n_shot: 5 - - task: mgsm_native_cot_fr - n_shot: 5 - name: MATH-500 - variants: - - task: MATH500 - n_shot: 0 - name: AIME24 - variants: - - task: AIME24 - n_shot: 0 - name: AIME25 - variants: - - task: AIME25 - n_shot: 0 - name: AMC23 - variants: - - task: AMC23 - n_shot: 0 - name: JEEBench - variants: - - task: JEEBench - n_shot: 0 - name: LSAT AR - variants: - - task: agieval_lsat_ar - n_shot: 3 - name: PolyMath - variants: - - task: polymath_de_high - n_shot: 0 - - task: polymath_de_low - n_shot: 0 - - task: polymath_de_medium - n_shot: 0 - - task: polymath_de_top - n_shot: 0 - - task: polymath_en_high - n_shot: 0 - - task: polymath_en_low - n_shot: 0 - - task: polymath_en_medium - n_shot: 0 - - task: polymath_en_top - n_shot: 0 - - task: polymath_es_high - n_shot: 0 - - task: polymath_es_low - n_shot: 0 - - task: polymath_es_medium - n_shot: 0 - - task: polymath_es_top - n_shot: 0 - - task: polymath_fr_high - n_shot: 0 - - task: polymath_fr_low - n_shot: 0 - - task: polymath_fr_medium - n_shot: 0 - - task: polymath_fr_top - n_shot: 0 - - task: polymath_it_high - n_shot: 0 - - task: polymath_it_low - n_shot: 0 - - task: polymath_it_medium - n_shot: 0 - - task: polymath_it_top - n_shot: 0 - - task: polymath_pt_high - n_shot: 0 - - task: polymath_pt_low - n_shot: 0 - - task: polymath_pt_medium - n_shot: 0 - - task: polymath_pt_top - n_shot: 0 - name: ARC Challenge - variants: - - task: arc_challenge - n_shot: 10 - - task: arc_challenge_mt_bg - n_shot: 0 - - task: arc_challenge_mt_cs - n_shot: 0 - - task: arc_challenge_mt_da - n_shot: 0 - - task: arc_challenge_mt_de - n_shot: 0 - - task: arc_challenge_mt_el - n_shot: 0 - - task: arc_challenge_mt_es - n_shot: 0 - - task: arc_challenge_mt_et - n_shot: 0 - - task: arc_challenge_mt_fi - n_shot: 0 - - task: arc_challenge_mt_fr - n_shot: 0 - - task: arc_challenge_mt_hu - n_shot: 0 - - task: arc_challenge_mt_is - n_shot: 0 - - task: arc_challenge_mt_it - n_shot: 0 - - task: arc_challenge_mt_lt - n_shot: 0 - - task: arc_challenge_mt_lv - n_shot: 0 - - task: arc_challenge_mt_nb - n_shot: 0 - - task: arc_challenge_mt_nl - n_shot: 0 - - task: arc_challenge_mt_pl - n_shot: 0 - - task: arc_challenge_mt_pt - n_shot: 0 - - task: arc_challenge_mt_ro - n_shot: 0 - - task: arc_challenge_mt_sk - n_shot: 0 - - task: arc_challenge_mt_sl - n_shot: 0 - - task: arc_challenge_mt_sv - n_shot: 0 - name: ARC Easy - variants: - - task: arc_easy - n_shot: 10 - name: GPQA Diamond - variants: - - task: GPQADiamond - n_shot: 0 - name: INCLUDE - variants: - - task: include_base_44_albanian - n_shot: 0 - - task: include_base_44_basque - n_shot: 0 - - task: include_base_44_bulgarian - n_shot: 0 - - task: include_base_44_croatian - n_shot: 0 - - task: include_base_44_dutch - n_shot: 0 - - task: include_base_44_estonian - n_shot: 0 - - task: include_base_44_finnish - n_shot: 0 - - task: include_base_44_french - n_shot: 0 - - task: include_base_44_georgian - n_shot: 0 - - task: include_base_44_german - n_shot: 0 - - task: include_base_44_greek - n_shot: 0 - - task: include_base_44_hungarian - n_shot: 0 - - task: include_base_44_italian - n_shot: 0 - - task: include_base_44_lithuanian - n_shot: 0 - - task: include_base_44_north macedonian - n_shot: 0 - - task: include_base_44_polish - n_shot: 0 - - task: include_base_44_portuguese - n_shot: 0 - - task: include_base_44_serbian - n_shot: 0 - - task: include_base_44_spanish - n_shot: 0 - - task: include_base_44_turkish - n_shot: 0 - - task: include_base_44_ukrainian - n_shot: 0 - name: Jeopardy - variants: - - task: jeopardy - n_shot: 10 - name: MMLU - variants: - - task: mmlu - n_shot: 5 - name: Global MMLU - variants: - - task: global_mmlu_full_cs - n_shot: 5 - - task: global_mmlu_full_de - n_shot: 5 - - task: global_mmlu_full_el - n_shot: 5 - - task: global_mmlu_full_en - n_shot: 5 - - task: global_mmlu_full_es - n_shot: 5 - - task: global_mmlu_full_fr - n_shot: 5 - - task: global_mmlu_full_it - n_shot: 5 - - task: global_mmlu_full_lt - n_shot: 5 - - task: global_mmlu_full_nl - n_shot: 5 - - task: global_mmlu_full_pl - n_shot: 5 - - task: global_mmlu_full_pt - n_shot: 5 - - task: global_mmlu_full_ro - n_shot: 5 - - task: global_mmlu_full_sr - n_shot: 5 - - task: global_mmlu_full_sv - n_shot: 5 - - task: global_mmlu_full_tr - n_shot: 5 - - task: global_mmlu_full_uk - n_shot: 5 - name: OpenBookQA - variants: - - task: openbookqa - n_shot: 0 - name: QA Wikidata - variants: - - task: bigbench_qa_wikidata_generate_until - n_shot: 10 - name: CommonsenseQA - variants: - - task: commonsense_qa - n_shot: 10 - name: COPA - variants: - - task: copa - n_shot: 0 - name: HellaSwag - variants: - - task: hellaswag - n_shot: 0 - - task: hellaswag_ca - n_shot: 0 - - task: hellaswag_da - n_shot: 0 - - task: hellaswag_de - n_shot: 0 - - task: hellaswag_es - n_shot: 0 - - task: hellaswag_eu - n_shot: 0 - - task: hellaswag_fr - n_shot: 0 - - task: hellaswag_hr - n_shot: 0 - - task: hellaswag_hu - n_shot: 0 - - task: hellaswag_it - n_shot: 0 - - task: hellaswag_nl - n_shot: 0 - - task: hellaswag_pt - n_shot: 0 - - task: hellaswag_ro - n_shot: 0 - - task: hellaswag_sk - n_shot: 0 - - task: hellaswag_sr - n_shot: 0 - - task: hellaswag_sv - n_shot: 0 - - task: hellaswag_uk - n_shot: 0 - name: PIQA - variants: - - task: global_piqa_completions_als_latn - n_shot: 0 - - task: global_piqa_completions_bos_latn - n_shot: 0 - - task: global_piqa_completions_bul_cyrl - n_shot: 0 - - task: global_piqa_completions_cat_latn - n_shot: 0 - - task: global_piqa_completions_ces_latn - n_shot: 0 - - task: global_piqa_completions_deu_latn - n_shot: 0 - - task: global_piqa_completions_ekk_latn - n_shot: 0 - - task: global_piqa_completions_ell_grek - n_shot: 0 - - task: global_piqa_completions_eng_latn - n_shot: 0 - - task: global_piqa_completions_fin_latn - n_shot: 0 - - task: global_piqa_completions_fra_latn_fran - n_shot: 0 - - task: global_piqa_completions_glg_latn - n_shot: 0 - - task: global_piqa_completions_hrv_latn - n_shot: 0 - - task: global_piqa_completions_hun_latn - n_shot: 0 - - task: global_piqa_completions_isl_latn - n_shot: 0 - - task: global_piqa_completions_ita_latn - n_shot: 0 - - task: global_piqa_completions_kat_geor - n_shot: 0 - - task: global_piqa_completions_lit_latn - n_shot: 0 - - task: global_piqa_completions_mkd_cyrl - n_shot: 0 - - task: global_piqa_completions_nld_latn - n_shot: 0 - - task: global_piqa_completions_nno_latn - n_shot: 0 - - task: global_piqa_completions_nob_latn - n_shot: 0 - - task: global_piqa_completions_pol_latn - n_shot: 0 - - task: global_piqa_completions_por_latn_port - n_shot: 0 - - task: global_piqa_completions_ron_latn - n_shot: 0 - - task: global_piqa_completions_slk_latn - n_shot: 0 - - task: global_piqa_completions_slv_latn - n_shot: 0 - - task: global_piqa_completions_spa_latn_spai - n_shot: 0 - - task: global_piqa_completions_srp_cyrl - n_shot: 0 - - task: global_piqa_completions_swe_latn - n_shot: 0 - - task: global_piqa_completions_tur_latn - n_shot: 0 - - task: global_piqa_completions_ukr_cyrl - n_shot: 0 - - task: piqa - n_shot: 10 - name: Social IQa - variants: - - task: social_iqa - n_shot: 0 - name: WSC273 - variants: - - task: wsc273 - n_shot: 0 - name: WinoGrande - variants: - - task: winogrande - n_shot: 0 - name: X-CSQA - variants: - - task: xcsqa_deu_Latn - n_shot: 0 - - task: xcsqa_eng_Latn - n_shot: 0 - - task: xcsqa_fra_Latn - n_shot: 0 - - task: xcsqa_ita_Latn - n_shot: 0 - - task: xcsqa_nld_Latn - n_shot: 0 - - task: xcsqa_pol_Latn - n_shot: 0 - - task: xcsqa_por_Latn - n_shot: 0 - - task: xcsqa_spa_Latn - n_shot: 0 + exclude_languages: + - eng_Latn - name: XCOPA - variants: - - task: xcopa:et - n_shot: 0 - - task: xcopa:it - n_shot: 0 - - task: xcopa:tr - n_shot: 0 - name: Belebele - variants: - - task: belebele_bul_Cyrl - n_shot: 5 - - task: belebele_ces_Latn - n_shot: 5 - - task: belebele_dan_Latn - n_shot: 5 - - task: belebele_deu_Latn - n_shot: 5 - - task: belebele_ell_Grek - n_shot: 5 - - task: belebele_eng_Latn - n_shot: 5 - - task: belebele_est_Latn - n_shot: 5 - - task: belebele_fin_Latn - n_shot: 5 - - task: belebele_fra_Latn - n_shot: 5 - - task: belebele_hrv_Latn - n_shot: 5 - - task: belebele_hun_Latn - n_shot: 5 - - task: belebele_ita_Latn - n_shot: 5 - - task: belebele_lit_Latn - n_shot: 5 - - task: belebele_lvs_Latn - n_shot: 5 - - task: belebele_mlt_Latn - n_shot: 5 - - task: belebele_nld_Latn - n_shot: 5 - - task: belebele_nob_Latn - n_shot: 5 - - task: belebele_pol_Latn - n_shot: 5 - - task: belebele_por_Latn - n_shot: 5 - - task: belebele_ron_Latn - n_shot: 5 - - task: belebele_slk_Latn - n_shot: 5 - - task: belebele_slv_Latn - n_shot: 5 - - task: belebele_spa_Latn - n_shot: 5 - - task: belebele_swe_Latn - n_shot: 5 - name: BoolQ - variants: - - task: boolq - n_shot: 10 - name: CoQA - variants: - - task: coqa - n_shot: 0 - name: LAMBADA - variants: - - task: lambada_openai - n_shot: 0 - name: SIB-200 - variants: - - task: sib200_als_Latn - n_shot: 0 - - task: sib200_bos_Latn - n_shot: 0 - - task: sib200_bul_Cyrl - n_shot: 0 - - task: sib200_cat_Latn - n_shot: 0 - - task: sib200_ces_Latn - n_shot: 0 - - task: sib200_dan_Latn - n_shot: 0 - - task: sib200_deu_Latn - n_shot: 0 - - task: sib200_ell_Grek - n_shot: 0 - - task: sib200_eng_Latn - n_shot: 0 - - task: sib200_est_Latn - n_shot: 0 - - task: sib200_eus_Latn - n_shot: 0 - - task: sib200_fin_Latn - n_shot: 0 - - task: sib200_fra_Latn - n_shot: 0 - - task: sib200_gle_Latn - n_shot: 0 - - task: sib200_glg_Latn - n_shot: 0 - - task: sib200_hrv_Latn - n_shot: 0 - - task: sib200_hun_Latn - n_shot: 0 - - task: sib200_isl_Latn - n_shot: 0 - - task: sib200_ita_Latn - n_shot: 0 - - task: sib200_kat_Geor - n_shot: 0 - - task: sib200_lit_Latn - n_shot: 0 - - task: sib200_lvs_Latn - n_shot: 0 - - task: sib200_mkd_Cyrl - n_shot: 0 - - task: sib200_mlt_Latn - n_shot: 0 - - task: sib200_nld_Latn - n_shot: 0 - - task: sib200_nob_Latn - n_shot: 0 - - task: sib200_pol_Latn - n_shot: 0 - - task: sib200_por_Latn - n_shot: 0 - - task: sib200_ron_Latn - n_shot: 0 - - task: sib200_slk_Latn - n_shot: 0 - - task: sib200_slv_Latn - n_shot: 0 - - task: sib200_spa_Latn - n_shot: 0 - - task: sib200_srp_Cyrl - n_shot: 0 - - task: sib200_swe_Latn - n_shot: 0 - - task: sib200_tur_Latn - n_shot: 0 - - task: sib200_ukr_Cyrl - n_shot: 0 - name: SQuAD v2 - variants: - - task: squadv2 - n_shot: 10 - name: FLORES200 - variants: - - task: flores200:als_Latn-eng_Latn - n_shot: 4 - - task: flores200:bos_Latn-eng_Latn - n_shot: 4 - - task: flores200:bul_Cyrl-eng_Latn - n_shot: 4 - - task: flores200:cat_Latn-eng_Latn - n_shot: 4 - - task: flores200:ces_Latn-eng_Latn - n_shot: 4 - - task: flores200:dan_Latn-eng_Latn - n_shot: 4 - - task: flores200:deu_Latn-eng_Latn - n_shot: 4 - - task: flores200:ell_Grek-eng_Latn - n_shot: 4 - - task: flores200:eng_Latn-als_Latn - n_shot: 4 - - task: flores200:eng_Latn-bos_Latn - n_shot: 4 - - task: flores200:eng_Latn-bul_Cyrl - n_shot: 4 - - task: flores200:eng_Latn-cat_Latn - n_shot: 4 - - task: flores200:eng_Latn-ces_Latn - n_shot: 4 - - task: flores200:eng_Latn-dan_Latn - n_shot: 4 - - task: flores200:eng_Latn-deu_Latn - n_shot: 4 - - task: flores200:eng_Latn-ell_Grek - n_shot: 4 - - task: flores200:eng_Latn-est_Latn - n_shot: 4 - - task: flores200:eng_Latn-eus_Latn - n_shot: 4 - - task: flores200:eng_Latn-fin_Latn - n_shot: 4 - - task: flores200:eng_Latn-fra_Latn - n_shot: 4 - - task: flores200:eng_Latn-gle_Latn - n_shot: 4 - - task: flores200:eng_Latn-glg_Latn - n_shot: 4 - - task: flores200:eng_Latn-hrv_Latn - n_shot: 4 - - task: flores200:eng_Latn-hun_Latn - n_shot: 4 - - task: flores200:eng_Latn-isl_Latn - n_shot: 4 - - task: flores200:eng_Latn-ita_Latn - n_shot: 4 - - task: flores200:eng_Latn-kat_Geor - n_shot: 4 - - task: flores200:eng_Latn-lit_Latn - n_shot: 4 - - task: flores200:eng_Latn-lvs_Latn - n_shot: 4 - - task: flores200:eng_Latn-mkd_Cyrl - n_shot: 4 - - task: flores200:eng_Latn-mlt_Latn - n_shot: 4 - - task: flores200:eng_Latn-nld_Latn - n_shot: 4 - - task: flores200:eng_Latn-nob_Latn - n_shot: 4 - - task: flores200:eng_Latn-pol_Latn - n_shot: 4 - - task: flores200:eng_Latn-por_Latn - n_shot: 4 - - task: flores200:eng_Latn-ron_Latn - n_shot: 4 - - task: flores200:eng_Latn-slk_Latn - n_shot: 4 - - task: flores200:eng_Latn-slv_Latn - n_shot: 4 - - task: flores200:eng_Latn-spa_Latn - n_shot: 4 - - task: flores200:eng_Latn-srp_Cyrl - n_shot: 4 - - task: flores200:eng_Latn-swe_Latn - n_shot: 4 - - task: flores200:eng_Latn-tur_Latn - n_shot: 4 - - task: flores200:eng_Latn-ukr_Cyrl - n_shot: 4 - - task: flores200:est_Latn-eng_Latn - n_shot: 4 - - task: flores200:eus_Latn-eng_Latn - n_shot: 4 - - task: flores200:fin_Latn-eng_Latn - n_shot: 4 - - task: flores200:fra_Latn-eng_Latn - n_shot: 4 - - task: flores200:gle_Latn-eng_Latn - n_shot: 4 - - task: flores200:glg_Latn-eng_Latn - n_shot: 4 - - task: flores200:hrv_Latn-eng_Latn - n_shot: 4 - - task: flores200:hun_Latn-eng_Latn - n_shot: 4 - - task: flores200:isl_Latn-eng_Latn - n_shot: 4 - - task: flores200:ita_Latn-eng_Latn - n_shot: 4 - - task: flores200:kat_Geor-eng_Latn - n_shot: 4 - - task: flores200:lit_Latn-eng_Latn - n_shot: 4 - - task: flores200:lvs_Latn-eng_Latn - n_shot: 4 - - task: flores200:mkd_Cyrl-eng_Latn - n_shot: 4 - - task: flores200:mlt_Latn-eng_Latn - n_shot: 4 - - task: flores200:nld_Latn-eng_Latn - n_shot: 4 - - task: flores200:nob_Latn-eng_Latn - n_shot: 4 - - task: flores200:pol_Latn-eng_Latn - n_shot: 4 - - task: flores200:por_Latn-eng_Latn - n_shot: 4 - - task: flores200:ron_Latn-eng_Latn - n_shot: 4 - - task: flores200:slk_Latn-eng_Latn - n_shot: 4 - - task: flores200:slv_Latn-eng_Latn - n_shot: 4 - - task: flores200:spa_Latn-eng_Latn - n_shot: 4 - - task: flores200:srp_Cyrl-eng_Latn - n_shot: 4 - - task: flores200:swe_Latn-eng_Latn - n_shot: 4 - - task: flores200:tur_Latn-eng_Latn - n_shot: 4 - - task: flores200:ukr_Cyrl-eng_Latn - n_shot: 4 - name: OpenSubtitles - variants: - - task: opensubtitles_multi40_bg_to_en - n_shot: 0 - - task: opensubtitles_multi40_cs_to_en - n_shot: 0 - - task: opensubtitles_multi40_da_to_en - n_shot: 0 - - task: opensubtitles_multi40_de_to_en - n_shot: 0 - - task: opensubtitles_multi40_el_to_en - n_shot: 0 - - task: opensubtitles_multi40_en_to_bg - n_shot: 0 - - task: opensubtitles_multi40_en_to_cs - n_shot: 0 - - task: opensubtitles_multi40_en_to_da - n_shot: 0 - - task: opensubtitles_multi40_en_to_de - n_shot: 0 - - task: opensubtitles_multi40_en_to_el - n_shot: 0 - - task: opensubtitles_multi40_en_to_es - n_shot: 0 - - task: opensubtitles_multi40_en_to_et - n_shot: 0 - - task: opensubtitles_multi40_en_to_fi - n_shot: 0 - - task: opensubtitles_multi40_en_to_fr - n_shot: 0 - - task: opensubtitles_multi40_en_to_hr - n_shot: 0 - - task: opensubtitles_multi40_en_to_hu - n_shot: 0 - - task: opensubtitles_multi40_en_to_it - n_shot: 0 - - task: opensubtitles_multi40_en_to_lt - n_shot: 0 - - task: opensubtitles_multi40_en_to_lv - n_shot: 0 - - task: opensubtitles_multi40_en_to_nl - n_shot: 0 - - task: opensubtitles_multi40_en_to_no - n_shot: 0 - - task: opensubtitles_multi40_en_to_pl - n_shot: 0 - - task: opensubtitles_multi40_en_to_pt - n_shot: 0 - - task: opensubtitles_multi40_en_to_ro - n_shot: 0 - - task: opensubtitles_multi40_en_to_sk - n_shot: 0 - - task: opensubtitles_multi40_en_to_sl - n_shot: 0 - - task: opensubtitles_multi40_en_to_sr - n_shot: 0 - - task: opensubtitles_multi40_en_to_sv - n_shot: 0 - - task: opensubtitles_multi40_en_to_tr - n_shot: 0 - - task: opensubtitles_multi40_en_to_uk - n_shot: 0 - - task: opensubtitles_multi40_es_to_en - n_shot: 0 - - task: opensubtitles_multi40_et_to_en - n_shot: 0 - - task: opensubtitles_multi40_fi_to_en - n_shot: 0 - - task: opensubtitles_multi40_fr_to_en - n_shot: 0 - - task: opensubtitles_multi40_hr_to_en - n_shot: 0 - - task: opensubtitles_multi40_hu_to_en - n_shot: 0 - - task: opensubtitles_multi40_it_to_en - n_shot: 0 - - task: opensubtitles_multi40_lt_to_en - n_shot: 0 - - task: opensubtitles_multi40_lv_to_en - n_shot: 0 - - task: opensubtitles_multi40_nl_to_en - n_shot: 0 - - task: opensubtitles_multi40_no_to_en - n_shot: 0 - - task: opensubtitles_multi40_pl_to_en - n_shot: 0 - - task: opensubtitles_multi40_pt_to_en - n_shot: 0 - - task: opensubtitles_multi40_ro_to_en - n_shot: 0 - - task: opensubtitles_multi40_sk_to_en - n_shot: 0 - - task: opensubtitles_multi40_sl_to_en - n_shot: 0 - - task: opensubtitles_multi40_sr_to_en - n_shot: 0 - - task: opensubtitles_multi40_sv_to_en - n_shot: 0 - - task: opensubtitles_multi40_tr_to_en - n_shot: 0 - - task: opensubtitles_multi40_uk_to_en - n_shot: 0 - name: Language ID - variants: - - task: bigbench_language_identification_multiple_choice - n_shot: 10 - name: MultiBlimp - variants: - - task: multiblimp_bul - n_shot: 0 - - task: multiblimp_cat - n_shot: 0 - - task: multiblimp_ces - n_shot: 0 - - task: multiblimp_dan - n_shot: 0 - - task: multiblimp_deu - n_shot: 0 - - task: multiblimp_ell - n_shot: 0 - - task: multiblimp_eng - n_shot: 0 - - task: multiblimp_est - n_shot: 0 - - task: multiblimp_eus - n_shot: 0 - - task: multiblimp_fin - n_shot: 0 - - task: multiblimp_fra - n_shot: 0 - - task: multiblimp_gle - n_shot: 0 - - task: multiblimp_glg - n_shot: 0 - - task: multiblimp_hbs - n_shot: 0 - - task: multiblimp_hun - n_shot: 0 - - task: multiblimp_isl - n_shot: 0 - - task: multiblimp_ita - n_shot: 0 - - task: multiblimp_kat - n_shot: 0 - - task: multiblimp_lav - n_shot: 0 - - task: multiblimp_lit - n_shot: 0 - - task: multiblimp_mkd - n_shot: 0 - - task: multiblimp_nld - n_shot: 0 - - task: multiblimp_pol - n_shot: 0 - - task: multiblimp_por - n_shot: 0 - - task: multiblimp_ron - n_shot: 0 - - task: multiblimp_slk - n_shot: 0 - - task: multiblimp_slv - n_shot: 0 - - task: multiblimp_spa - n_shot: 0 - - task: multiblimp_sqi - n_shot: 0 - - task: multiblimp_swe - n_shot: 0 - - task: multiblimp_tur - n_shot: 0 - - task: multiblimp_ukr - n_shot: 0 - name: IFEval - variants: - - task: ifeval - n_shot: 0 diff --git a/docs/configuration.md b/docs/configuration.md index c5d4657..a590bfc 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -31,34 +31,57 @@ The **Weighting profile** and **Eval set** selectors operate independently. Swit **Any available** uses recognized selected measurements shared by A and B. Measurements present on only one side generate comparison warnings and are excluded from both scores. Catalogue entries absent from both models do not generate warnings. The supplied freeform set explicitly excludes prompted Global PIQA pending validation; present data for it generates a **Not used** warning. -**flagship-1** names the expected 45 evals and 403 task/shot requirements for the flagship comparison. A required measurement missing from either or both models generates a warning. Extra selected measurements are excluded with warnings, but remain inspectable in **Eval configuration**. The dashboard labels the score **INCOMPLETE**, shows shared/required coverage, and redistributes weights across the shared subset. Do not interpret an incomplete score as covering the full named set. Exact comparison identity still includes metric, filter, shots, harness, and backend; incompatible protocols cannot satisfy shared coverage just by sharing a task name. +**flagship-1** requires the catalogue's known tasks for each listed eval, excluding Georgian throughout the set because of a vocabulary error and English specifically for X-CSQA. Original and translated tasks stay under their existing eval group (for example ARC Challenge). A missing required result generates a warning even if other languages for that eval are present. The score is labelled **INCOMPLETE** and uses the shared subset with redistributed weights. Excluded tasks are not requirements; present excluded data produces **Not used** warnings and remains inspectable. -A named set can require a whole eval, or exact tasks with optional shot counts: +A named set can be concise: ```yaml version: 1 name: Example required set mode: fixed +exclude_languages: [kat_Geor] evals: - - name: Example eval - variants: - - {task: example_en, n_shot: 0} - - {task: example_fr, n_shot: 0} + - name: ARC Challenge + shots: 10 + - name: SIB-200 + metric: acc + metric_filter: none + - name: X-CSQA + exclude_languages: [eng_Latn] ``` -Omitting `variants` requires at least one shared selected measurement for that eval and permits all its variants. Omitting `n_shot` accepts any shot count, but A and B must still match each other's protocol. Required tasks must match the named catalogue eval and its `select`/`shots` restrictions. Duplicate or overlapping requirements, unknown eval names, and unknown fields are errors. Metric and normalization choices belong only in the catalogue. Eval sets never contain weights. +This example requires a catalogue containing those evals and language assignments. `mode: fixed` means a declared set of required evals; `mode: available` means whatever recognized results are shared by the models. This choice is independent of strict/relaxed matching. -Freeform mode needs no required-eval list. It can optionally exclude evals by their exact catalogue names: +For each named eval, omitted `metric`, `metric_filter`, and `shots` inherit from the catalogue. Set overrides apply to the **whole eval group**, including its translated tasks. They do not alter the global catalogue. Score scale and normalization remain the catalogue's interpretation of that eval; an alternative metric must be compatible with that interpretation. There are no variant-specific setting overrides. **Eval configuration** displays the effective settings and identifies fields overridden by the active set. Changing sets reinterprets loaded results and regenerates synthetic comparisons while preserving weight edits. + +Known tasks come from the catalogue's explicit task-language assignments and literal `match.name`, limited by the eval's `match` and optional `select` rules. Required coverage is independent of the CSVs: absence from both models still warns. Updating the catalogue can therefore expand a named set's requirements. An eval using only a regex and no known concrete tasks cannot be required until its task inventory is supplied. A recognized task outside the known inventory is excluded from a fixed set, with a warning. Freeform mode can still use it and report missing language metadata. + +`exclude_languages` applies across the entire set or just one eval. Set-wide and local exclusions combine. They use explicit canonical language assignments, not substrings of task names. For translation, either endpoint excludes the pair. Pooled results use their declared language label; exclusions cannot split pooled scores. Excluding a whole component language group is valid; partial component selections are configuration errors. If every task is excluded, the score is unavailable rather than zero. + +Unknown eval names and language exclusions with no matching catalogue assignment are configuration errors. A local language exclusion must exist within that eval's eligible catalogue tasks; a set-wide exclusion must exist somewhere in the catalogue's eligible tasks. Duplicate/invalid language codes and misspelled fields also fail. A known selected task missing from the CSV is instead a recoverable warning. Failed imports preserve the active dashboard. + +Available-mode sets can exclude entire evals or languages without creating required coverage: ```yaml version: 1 name: Any available mode: available -exclude: - - Global PIQA (prompted) +exclude: [Global PIQA (prompted)] ``` -`exclude` is allowed only in available mode and must contain unique names. An excluded name absent from the active catalogue is harmless, so a set can be reused with another catalogue. Fixed sets already exclude everything outside their membership. Exclusions warn whenever matching eval data is present, including rows containing only an alternate metric. Alternate fields or summary children of a selected eval do not generate unused-eval warnings merely because a preferred field or summary is selected. +`exclude` is allowed only in available mode and contains unique, exact catalogue eval names. All names must exist even if no corresponding model results are loaded. Fixed sets exclude unlisted evals automatically. Exclusions warn when eligible task data is present, including rows with only the wrong metric. Alternate metric rows or summary children of a selected eval do not warn merely because another field or summary was selected. + +For an explicitly pinned task inventory, a fixed entry may still use `variants: [{task: example_en}]`, optionally with a hard `n_shot` requirement. Each task must be known and selected by its catalogue rule. These are membership constraints, not setting overrides; a pinned `n_shot` must agree with the effective eval settings and is not loosened by relaxed matching. Ordinary named sets need no `variants` list. + +## Strict and relaxed matching + +The dashboard starts with **Strict matching**. Expected settings are resolved in this order: catalogue defaults, then optional whole-eval set overrides. Strict matching requires the configured metric, metric filter, and (when specified) shot count. Shared comparisons also require the same harness and backend. + +**Relaxed — allow few-shot differences** may select a different shot count for each concrete task. It prefers the expected count; otherwise it uses the uniquely closest available count, independently for each model and task with the same metric/filter/harness/backend. Equally close alternatives are ambiguous and excluded with a warning. Scores never influence that choice. Without a configured shot expectation, shot counts still have to match across models. + +Relaxed matching never substitutes a different metric or metric filter, or pairs different harnesses/backends. Component groups must still contain every component at one consistent actual shot count. If per-task selection leaves a mixed-shot or incomplete component group, the group is excluded; the engine does not search for a different combination to rescue it. + +When a differing shot count actually contributes, the page prominently reports **INCONSISTENT EVALUATION SETTINGS**. Warnings give the eval, model, tasks, expected count, and actual count. Real measurement identities retain their actual settings. Enabling relaxed mode alone does not label a comparison inconsistent if no mismatched measurements contribute. The shipped working expectations are 10 shots for ARC Challenge and PIQA, and 5 for MGSM; their 0-shot translated/global variants need relaxed matching unless the set overrides the expectation. A weighting profile works with either mode: @@ -82,16 +105,16 @@ The builder embeds YAML profiles directly in `configs/weights/` and sets directl - `--results-dir DIR` embeds CSVs directly in that directory. A checkpoint label may occur in only one file; one file can contain multiple models. - `--sample-csv FILE`, used with `--results-dir`, supplies a fallback only if that directory has no CSVs. Invalid shared files stop the build; they never trigger the fallback. Pages uses `examples/sample-evals.csv` for this option. -Every offered set is validated against the catalogue and every profile before writing output. Raw results are classified once by the global catalogue; changing sets only changes comparison membership. An empty results directory starts without models unless `--sample-csv` supplies a fallback. Omitting both input options starts without models. +Every offered set is validated against the catalogue and every profile before writing output. Raw results are interpreted using the catalogue with the active set’s overrides; changing sets can change both membership and scoring-field selection. An empty results directory starts without models unless `--sample-csv` supplies a fallback. Omitting both input options starts without models. -Under **Eval configuration**, load or export the catalogue, weights, and eval set separately. Uploaded choices are temporary; reload restores published defaults. Catalogue imports reinterpret all loaded real models and regenerate the synthetic comparison. Invalid imports preserve the previous models and settings. If a new catalogue does not contain the active named set's evals, switch to Any available before loading it. **Clear models** retains settings. +Under **Eval configuration**, load or export the catalogue, weights, and eval set separately. Uploaded choices are temporary; reload restores published defaults. Catalogue and eval-set imports reinterpret all loaded real models and regenerate synthetic comparisons. Invalid imports preserve the previous models and settings. When replacing a catalogue, first load a compatible set. A neutral set with `mode: available` and no exclusions works with any catalogue; the project's supplied Any available set references prompted Global PIQA and requires that catalogue entry. `configs/examples/eval-set.yaml` is a neutral set for the fictional example. **Clear models** retains settings. `analysis.json` records the catalogue, selected profile and set, available `profiles` and `suites`, and source filenames/hashes. It also contains the resolved internal `scheme` used for arithmetic; that combined object is not a YAML input format. Generated `catalogue.yaml` contains the assembled, portable catalogue, with no `evals_dir` reference. `weights.yaml` and `eval-set.yaml` record the other configuration inputs. Browser export also saves the complete catalogue in one file; browser import accepts complete catalogues, not filesystem manifests. ## Edit one eval The repository keeps each eval’s interpretation and language mappings in one file. -For example, this `evals/example.yaml` selects `acc_norm` for two language variants +For example, this `evals/example.yaml` selects `acc_norm` with the `none` metric filter for two language variants of a four-choice eval: ```yaml @@ -99,7 +122,7 @@ name: Example eval category: Reasoning match: {regex: 'example_(en|fr)'} metric: acc_norm -filter: none +metric_filter: none shots: 0 score: {scale: 1} normalize: @@ -192,7 +215,7 @@ For an input row with `value=0.625`, the raw score is 62.5 and the normalized sc | `category` | Category label used by weighting profiles and breakdowns. | | `match` | Exactly one of `{name: exact task name}` or `{regex: 'full-match pattern'}`. Unmatched tasks are excluded and listed in Warnings; overlapping matches are an error. | | `metric` | Exact CSV scoring field, such as `acc`, `acc_norm`, or `python_pass@1`. | -| `filter` | Exact CSV filter string, including `""` if empty. | +| `metric_filter` | Exact value of the CSV `filter` column, including `""` if empty. | | `shots` | Optional nonnegative integer selecting the CSV `n_shot`. Omit to include all shot settings. | | `select` | Optional name/regex rule restricting which matched tasks contribute. Useful for selecting summaries while retaining child-task audits. | | `score.scale` | Raw metric's upper scale: 1 for fractional accuracy; 100 for percentage or chrF scores. Selected values must be finite and within 0..scale. | @@ -200,6 +223,8 @@ For an input row with `value=0.625`, the raw score is 62.5 and the normalized sc | `aggregation` | Optional component rules with positive relative weights; see [weighted components](#weighted-components-within-an-eval). | | `normalize` | Optional object with `min`, `max`, and optional `clip`, `basis`, `note`, and `sources`. Thresholds are fractions after division by `score.scale`. | +`metric_filter` selects the evaluator’s response-processing label recorded beside a metric; Quickdash does not execute that processing. `metric_filter: ''` matches a blank CSV field, while `metric_filter: none` matches the literal text `none`. These are distinct values, and neither means “any filter.” + Use the shared Python/JavaScript regex subset: literal text, character classes, alternatives, capturing/noncapturing groups, and ordinary quantifiers. Patterns match the entire task name. Flags, lookarounds, named groups, backreferences and possessive quantifiers are rejected. `\d` and `\w` use ASCII character classes; `\s` uses ECMAScript whitespace and must not appear inside a character class. Dot excludes line terminators and matches one Unicode code point. Language extraction does not use these patterns. Exact-name rules are available when regexes are unnecessary. The [Python API](python-api.md) consumes these same configurations and returns score trees, audits, coverage, and structured diagnostics. All configuration counts are derived from the supplied inputs. @@ -261,7 +286,7 @@ For example, fictional component scores of 60, 30, 15, and 0 produce `(60 + 60 + **Incompatible aggregation configurations are errors.** Components inherit the parent eval's metric, filter, score scale, normalization, and any fixed shot setting; they cannot override these fields. Catalogue validation checks declared task/language assignments against the component rules and eval selection. Each represented language must have every component available in the configuration, and each task must match exactly one component. An exact component task must be eligible under its parent eval. Entire languages can be omitted; alternate task aliases are permitted in the catalogue. -A named set that lists component tasks must select exactly one task for each component at every chosen language/shot setting. For example, selecting low/medium/high at 0-shot and top at 5-shot is a configuration error, as is omitting top entirely. Omitting shots for every component is allowed; mixing unrestricted and fixed shots is rejected unless the parent eval pins the same shot count. Multiple complete shot settings are allowed. An eval-only requirement or Any available leaves task selection to the catalogue. Missing task-language assignments or duplicate component selections in a named set are errors. +A named set that lists component tasks must select exactly one task for each component at every chosen language/shot setting. For example, selecting low/medium/high at 0-shot and top at 5-shot is a configuration error, as is omitting top entirely. Omitting shots for every component is allowed; mixing unrestricted and fixed shots is rejected unless the parent eval pins the same shot count. Multiple complete shot settings are allowed. A whole-eval requirement expands to all eligible known tasks, after language exclusions. Any available has no required task inventory. Missing task-language assignments or duplicate component selections in a named set are errors. These checks run before applying a browser config or replacing build output. Rejected imports preserve active models, settings, and scores. Regex compatibility is checked against declared task names, plus newly observed selected tasks during CSV classification; the validator does not attempt to prove arbitrary regex relationships. An observed selected task matching zero or multiple component rules is also an error. diff --git a/docs/development.md b/docs/development.md index 71661eb..a14aaf9 100644 --- a/docs/development.md +++ b/docs/development.md @@ -20,7 +20,7 @@ Run commands from the repository root. Install the Python package with `python - ```sh python3 -m app.build --results-dir results --sample-csv examples/sample-evals.csv --output output/shared python3 -m app.build examples/scores.csv --catalogue configs/examples/catalogue.yaml \ - --weights configs/examples/weights.yaml --eval-set configs/sets/any-available.yaml --output output/demo + --weights configs/examples/weights.yaml --eval-set configs/examples/eval-set.yaml --output output/demo ``` Open each generated `index.html` in a browser. The shared build embeds the global catalogue, profiles from `configs/weights/`, and sets from `configs/sets/`. Each selector uses its directory's `default.txt`. See the [build flags](configuration.md#build-defaults-and-browser-imports) to supply other inputs. @@ -36,7 +36,7 @@ python -m unittest tests.test_data tests.test_engines node --test tests/test_data.cjs tests/test_yaml.cjs tests/test_suites.cjs tests/test_warning_policy.cjs tests/test_components.cjs ``` -The shared suite in `tests/test_engines.py` sends the same input cases to native Python and the real browser engine through `tests/engine_adapter.cjs`. Both must satisfy independently specified expectations, then their complete semantic reports are compared. Diagnostic codes, contexts and actions are checked; presentation text is not. Numerical comparison uses absolute tolerance `1e-9` and relative tolerance `1e-12`. Fixtures vary configuration sizes, contents and ordering. Python tests also exercise warning emission and CLI stdout/stderr without Node on PATH. +The shared suite in `tests/test_engines.py` sends the same input cases to native Python and the real browser engine through `tests/engine_adapter.cjs`. Both must satisfy independently specified expectations, then their complete semantic reports are compared. Diagnostic codes, contexts and actions are checked; presentation text is not. Numerical comparison uses absolute tolerance `1e-9` and relative tolerance `1e-12`. Fixtures vary configuration sizes, contents and ordering. The sample matrix discovers all shipped sets and profiles and exercises every aggregation in both strict and relaxed matching. Shared cases cover override inheritance, language exclusions (including both translation endpoints), invalid references, missing requirements, component completeness, and warning conditions. Browser checks additionally verify effective settings, unchanged catalogue exports, set switching, and failed-import rollback. Python tests also exercise warning emission and CLI stdout/stderr without Node on PATH. They cover input validation, normalization, warning/exclusion behavior, failed-build preservation, Python/JavaScript parity, hierarchy sorting, and deterministic randomized scoring comparisons against an independent calculation. diff --git a/docs/python-api.md b/docs/python-api.md index c8ad095..0a91819 100644 --- a/docs/python-api.md +++ b/docs/python-api.md @@ -22,7 +22,7 @@ from quickdash import analyze, compare, load_config, read_results config = load_config( catalogue="configs/examples/catalogue.yaml", weights="configs/examples/weights.yaml", - eval_set="configs/sets/any-available.yaml", + eval_set="configs/examples/eval-set.yaml", ) results = read_results("examples/scores.csv") report = analyze(results, config) @@ -37,15 +37,15 @@ print(comparison.delta) # -2.5 score points `load_config()` accepts paths or configuration dictionaries, validates their combination, and returns a bundle containing `catalogue`, `profile`, and `suite`. It copies supplied dictionaries. Change configurations by loading a new bundle. The analysis functions do not mutate results or configuration inputs. -Omitting `eval_set` means all recognized evals are eligible. To apply the project's explicit exclusions, pass its `configs/sets/any-available.yaml` file, as above. No eval names, language lists, category counts, or component counts are hardcoded into the library. +Omitting `eval_set` means all recognized evals are eligible. To apply the project's explicit exclusions, pass its `configs/sets/any-available.yaml` file with the project catalogue. Names in exclusions must exist in the active catalogue; the fictional example uses its own neutral set. No eval names, language lists, category counts, or component counts are hardcoded into the library. `read_results()` accepts a path or a list of paths. It preserves CSV fields as strings and rejects model names repeated across files. Put all measurements for a model in one CSV, or deliberately combine row dictionaries before analysis. The analysis functions also accept lists of row dictionaries directly, with the same required fields and validation as CSV imports. ## Independent scores and fair comparisons -`analyze(results, config)` scores each model independently on its available, selected measurements. Its top-level fields are `models`, `diagnostics`, and `coverage`. +`analyze(results, config)` scores each model independently on its available, selected measurements. Its top-level fields are `models`, `diagnostics`, `coverage`, `matching`, and `inconsistent`. -`compare(results, config, a=..., b=...)` first restricts both models to matching valid measurements. It excludes incomplete component groups, including groups made incomplete by intersecting coverage, before redistributing weights. Its fields are `a`, `b`, `delta`, `deltas`, `diagnostics`, and `coverage`. `a` and `b` have the same model-result shape used by `analyze()`. +`compare(results, config, a=..., b=...)` first restricts both models to matching valid measurements. It excludes incomplete component groups, including groups made incomplete by intersecting coverage, before redistributing weights. Its fields are `a`, `b`, `delta`, `deltas`, `diagnostics`, `coverage`, `matching`, and `inconsistent`. `a` and `b` have the same model-result shape used by `analyze()`. Do not subtract independent scores to reproduce an A/B comparison: the models may have different coverage. A missing comparison model raises `ValueError`; comparing a model with itself is permitted. @@ -97,6 +97,31 @@ Existing single-file catalogues remain supported. When passing a dictionary instead of a filename, supply the complete catalogue; a filesystem manifest needs a filename to resolve its directory. See [editing an eval](configuration.md#edit-one-eval). +## Expected settings and relaxed comparisons + +Both `analyze()` and `compare()` accept `matching="strict"` (the default) or +`matching="relaxed"`. They first apply the eval set's optional whole-eval `metric`, +`metric_filter`, and `shots` overrides to the catalogue defaults. Set-wide and +per-eval `exclude_languages` remove results from membership and required coverage. +See the [configuration contract](configuration.md#strict-and-relaxed-matching) +for selection, ambiguity, and component-group rules. + +```python +comparison = compare( + results, config, a="Example A", b="Example B", + matching="relaxed", diagnostics="collect", +) +print(comparison.matching, comparison.inconsistent) +``` + +`matching` records the requested mode. `inconsistent` is true when included +measurements have an allowed few-shot mismatch. Actual shot counts remain in +measurement records and IDs. Structured `relaxed_shot_setting` diagnostics contain +`expected_shots` and `actual_shots`; `ambiguous_shot_setting` has `expected_shots` +and an `actual_shots` list. Excluded alternatives do not emit an allowed-mismatch +diagnostic. Configuration errors always raise, including unknown names in set +requirements or exclusions; missing data for a valid requirement warns instead. + ## Surface warnings By default, `analyze()` and `compare()` emit a `QuickdashWarning` through Python's standard `warnings` mechanism for each grouped diagnostic. Warnings normally appear on stderr. Each warning object has a `.diagnostic` attribute containing its structured record. The same diagnostics are retained in `report.diagnostics`, separately from the tree. @@ -119,6 +144,8 @@ The diagnostic contract is `code`, `model`, `eval`, `tasks`, `measurement_ids`, | `no_config`, `not_used` | Unknown eval or data outside the selected set; excluded. | | `missing_scoring_field`, `missing_scoring_setting`, `no_selected_score` | Required metric/protocol unavailable; no alternate substitution. | | `missing_suite_data` | A named set has missing requirements; available shared results are reweighted. | +| `relaxed_shot_setting` | A differing shot count is included under relaxed matching; records expected/actual counts. | +| `ambiguous_shot_setting` | Equally close alternative shot counts; excluded. | | `incomplete_components` | A language/protocol group is incomplete or incompatible; the whole group is excluded. | | `comparison_coverage` | Measurements cannot participate in shared coverage; excluded from both scores. | | `unknown_language` | Ordinary evals use the English weighting fallback; component groups require explicit language assignments. | @@ -136,9 +163,11 @@ Print calculated trees for the fictional models: quickdash examples/scores.csv \ --catalogue configs/examples/catalogue.yaml \ --weights configs/examples/weights.yaml \ - --eval-set configs/sets/any-available.yaml + --eval-set configs/examples/eval-set.yaml ``` Add `--compare 'Example A' 'Example B'` for shared-coverage scores, or `--format json` for the complete machine-readable report. `python -m quickdash` provides the same command. Warnings go to stderr; stdout contains only the selected result format. Exit status is 0 for a successful calculation, including recoverable warnings. `--strict` prints warnings and exits 1 without writing a result. Invalid input exits 2. A score can be unavailable when no valid weighted data remains; inspect diagnostics and coverage or use strict mode when warnings must block a workflow. + +Use `--matching relaxed` to allow few-shot differences. This is separate from `--strict`, which makes the CLI fail on diagnostics; it does not choose the matching mode. diff --git a/quickdash/analysis.py b/quickdash/analysis.py index 78a1682..e7b8212 100644 --- a/quickdash/analysis.py +++ b/quickdash/analysis.py @@ -7,7 +7,8 @@ classify, in_suite, match_task, - resolve_config, + normalize_score, + resolve_inputs, scope_rows, task_language, ) @@ -24,6 +25,75 @@ def key(row): return tuple(row[k] for k in IDENTITY) +def matching_key(row, config, matching="strict"): + identity = list(key(row)) + if matching == "relaxed": + e = next((e for e in config["evals"] if e["name"] == row["eval"]), {}) + if "shots" in e: + identity[3] = str(int(e["shots"])) + return tuple(identity) + + +def matching_audit(audit, catalogue, matching="strict"): + if matching not in {"strict", "relaxed"}: + raise ValueError("Matching must be strict or relaxed") + if matching == "strict": + return audit + result = [dict(r) for r in audit] + evals = {e["name"]: e for e in catalogue["evals"]} + groups = {} + for r in result: + e = evals.get(r["eval"], {}) + if ( + "shots" not in e + or r["metric"] != e["metric"] + or r["filter"] != e["metric_filter"] + or "select" in e + and match_task(e["select"], r["task"]) is None + ): + continue + identity = (r["checkpoint"], *[r[k] for k in IDENTITY if k != "n_shot"]) + groups.setdefault(identity, []).append(r) + for rr in groups.values(): + e = evals[rr[0]["eval"]] + distance = min(abs(int(r["n_shot"]) - e["shots"]) for r in rr) + nearest = sorted( + { + int(r["n_shot"]) + for r in rr + if abs(int(r["n_shot"]) - e["shots"]) == distance + } + ) + for r in rr: + r.update(selected=False, raw_score_100=None, score_100=None) + r.pop("matching_ambiguous_shots", None) + if len(nearest) != 1: + r.update( + decision="Equally close shot settings; excluded", + matching_ambiguous_shots=nearest, + ) + elif int(r["n_shot"]) != nearest[0]: + r["decision"] = "Alternate shot setting; a closer setting is available" + else: + raw, adjusted = normalize_score(r["value"], e) + r.update( + selected=True, + raw_score_100=raw, + score_100=adjusted, + decision="Selected for the weighted score" + if nearest[0] == e["shots"] + else f"Using {nearest[0]} shots despite expected {int(e['shots'])}; relaxed matching", + ) + seen = set() + for r in result: + if r["selected"]: + identity = (r["checkpoint"], *key(r)) + if identity in seen: + raise ValueError("Duplicate selected measurement: " + r["task"]) + seen.add(identity) + return result + + def measurement_id(row): return encoded([row["checkpoint"], *key(row)]) @@ -316,12 +386,16 @@ def allocate(rr, factor): ) -def comparison_coverage(a, b, config): +def comparison_coverage(a, b, config, matching="strict"): left = component_coverage(a, config)["rows"] right = component_coverage(b, config)["rows"] - shared = set(map(key, left)) & set(map(key, right)) - aa = component_coverage([r for r in left if key(r) in shared], config)["rows"] - bb = component_coverage([r for r in right if key(r) in shared], config)["rows"] + + def match(r): + return matching_key(r, config, matching) + + shared = set(map(match, left)) & set(map(match, right)) + aa = component_coverage([r for r in left if match(r) in shared], config)["rows"] + bb = component_coverage([r for r in right if match(r) in shared], config)["rows"] return aa, bb @@ -351,6 +425,8 @@ def protocol_inconsistent(rows): unknown_language="Unknown language", invalid_sample_count="Invalid sample count", inconsistent_scoring_settings="Inconsistent scoring settings", + relaxed_shot_setting="Few-shot mismatch allowed", + ambiguous_shot_setting="Ambiguous few-shot setting", missing_scoring_field="Missing scoring field", missing_scoring_setting="Missing scoring setting", no_selected_score="No selected score", @@ -378,9 +454,15 @@ def diagnostic(code, model, eval_name, rows, detail, effect="included", tasks=No ) -def report_diagnostics(audits, config, included, comparison=False): - catalogue, suite, profile = config["catalogue"], config["suite"], config["profile"] - scheme = resolve_config(catalogue, suite, profile) +def report_diagnostics( + audits, config, included, comparison=False, matching_mode="strict", resolved=None +): + resolved = resolved or resolve_inputs( + config["catalogue"], config["suite"], config["profile"] + ) + catalogue, suite, scheme, profile = ( + resolved[k] for k in ("catalogue", "suite", "scheme", "profile") + ) out = [] used = {r["eval"] for rr in included.values() for r in rr} for e in catalogue["evals"]: @@ -418,6 +500,44 @@ def report_diagnostics(audits, config, included, comparison=False): def add(code, rr, detail, effect="included"): out.append(diagnostic(code, model, e["name"], rr, detail, effect)) + if matching_mode == "relaxed" and "shots" in e: + for actual in sorted( + { + int(r["n_shot"]) + for r in selected + if measurement_id(r) in accepted + and int(r["n_shot"]) != e["shots"] + } + ): + rr = [ + r + for r in selected + if measurement_id(r) in accepted and int(r["n_shot"]) == actual + ] + item = diagnostic( + "relaxed_shot_setting", + model, + e["name"], + rr, + f"Using {actual} shots despite expected {e['shots']}; allowed by relaxed matching.", + ) + item.update(expected_shots=e["shots"], actual_shots=actual) + out.append(item) + ambiguous = [r for r in matching if r.get("matching_ambiguous_shots")] + if ambiguous: + item = diagnostic( + "ambiguous_shot_setting", + model, + e["name"], + ambiguous, + f"Multiple equally close shot settings for expected {e['shots']}; excluded.", + "excluded", + ) + item.update( + expected_shots=e["shots"], + actual_shots=sorted({int(r["n_shot"]) for r in ambiguous}), + ) + out.append(item) if outside: add( "not_used", @@ -477,7 +597,8 @@ def add(code, rr, detail, effect="included"): "Excluded: expected " + e["metric"] + " / " - + (e["filter"] or "(empty)") + + (e["metric_filter"] or "(empty)") + + (f" / {e['shots']} shots" if "shots" in e else "") + ".", "excluded", ) @@ -550,16 +671,20 @@ def add(code, rr, detail, effect="included"): entries = list(included.values()) a = entries[0] if entries else [] b = entries[1] if len(entries) > 1 else a - bm = {key(r): r for r in b} + + def match(r): + return matching_key(r, scheme, matching_mode) + + bm = {match(r): r for r in b} for e in catalogue["evals"]: rr = [ r for r in a if r["eval"] == e["name"] - and key(r) in bm + and match(r) in bm and sample_count(r) is not None - and sample_count(bm[key(r)]) is not None - and sample_count(r) != sample_count(bm[key(r)]) + and sample_count(bm[match(r)]) is not None + and sample_count(r) != sample_count(bm[match(r)]) ] if rr: out.append( @@ -567,7 +692,7 @@ def add(code, rr, detail, effect="included"): "sample_count_mismatch", "Selected comparison", e["name"], - rr + [bm[key(r)] for r in rr], + rr + [bm[match(r)] for r in rr], "Matched results have different sample counts. Scores remain included.", ) ) @@ -747,25 +872,27 @@ def model_report(model, audit, rows, config, suite): ) -def prepare(rows, config): - scheme = resolve_config(config["catalogue"], config["suite"], config["profile"]) - audit = classify(rows, config["catalogue"]) - return scheme, { +def prepare(rows, config, matching="strict"): + resolved = resolve_inputs(config["catalogue"], config["suite"], config["profile"]) + catalogue = resolved["catalogue"] + audit = matching_audit(classify(rows, catalogue), catalogue, matching) + return resolved, { m: [r for r in audit if r["checkpoint"] == m] for m in sorted({r["checkpoint"] for r in audit}) } -def analyze(results, config, *, diagnostics="warn"): +def analyze(results, config, *, diagnostics="warn", matching="strict"): """Score each model on its own available data; emit recoverable warnings by default.""" - scheme, audits = prepare(results, config) + resolved, audits = prepare(results, config, matching) + scheme, suite = resolved["scheme"], resolved["suite"] included = { - m: component_coverage(scope_rows(rr, config["suite"])["rows"], scheme)["rows"] + m: component_coverage(scope_rows(rr, suite)["rows"], scheme)["rows"] for m, rr in audits.items() } coverage = [] for model, rr in audits.items(): - scope = scope_rows(rr, config["suite"]) + scope = scope_rows(rr, suite) coverage.append( dict( model=model, @@ -774,29 +901,36 @@ def analyze(results, config, *, diagnostics="warn"): and len(scope["rows"]) == len(included[model]), ) ) + diagnostics_list = report_diagnostics( + audits, config, included, matching_mode=matching, resolved=resolved + ) return finish( dict( + matching=matching, + inconsistent=any( + d["code"] == "relaxed_shot_setting" for d in diagnostics_list + ), models=[ - model_report(m, rr, included[m], scheme, config["suite"]) + model_report(m, rr, included[m], scheme, suite) for m, rr in audits.items() ], - diagnostics=report_diagnostics(audits, config, included), + diagnostics=diagnostics_list, coverage=coverage, ), diagnostics, ) -def compare(results, config, *, a, b, diagnostics="warn"): +def compare(results, config, *, a, b, diagnostics="warn", matching="strict"): """Score both models on the same valid measurements, with symmetric exclusions.""" - scheme, all_audits = prepare(results, config) + resolved, all_audits = prepare(results, config, matching) + scheme, suite = resolved["scheme"], resolved["suite"] if a not in all_audits or b not in all_audits: raise ValueError("Unknown comparison model") audits = {a: all_audits[a], b: all_audits[b]} - suite = config["suite"] sa = scope_rows(audits[a], suite) sb = scope_rows(audits[b], suite) - aa, bb = comparison_coverage(sa["rows"], sb["rows"], scheme) + aa, bb = comparison_coverage(sa["rows"], sb["rows"], scheme, matching) effective = scope_rows(aa, suite) left = model_report(a, audits[a], aa, scheme, suite) right = model_report(b, audits[b], bb, scheme, suite) @@ -818,29 +952,40 @@ def compare(results, config, *, a, b, diagnostics="warn"): extrasA=len(sa["extras"]), extrasB=len(sb["extras"]), ) - bm = {key(r): r for r in bb} + + def match(r): + return matching_key(r, scheme, matching) + + bm = {match(r): r for r in bb} aw = {r["id"]: r["effective_weight"] for r in left["measurements"]} deltas = [ dict( task=r["task"], measurement_a=measurement_id(r), - measurement_b=measurement_id(bm[key(r)]), - raw_delta=r["raw_score_100"] - bm[key(r)]["raw_score_100"], - score_delta=r["score_100"] - bm[key(r)]["score_100"], + measurement_b=measurement_id(bm[match(r)]), + raw_delta=r["raw_score_100"] - bm[match(r)]["raw_score_100"], + score_delta=r["score_100"] - bm[match(r)]["score_100"], effective_weight=aw[measurement_id(r)], - contribution_delta=(r["score_100"] - bm[key(r)]["score_100"]) + contribution_delta=(r["score_100"] - bm[match(r)]["score_100"]) * aw[measurement_id(r)], ) for r in aa ] + diagnostics_list = report_diagnostics( + audits, config, {a: aa, b: bb}, True, matching, resolved + ) return finish( dict( + matching=matching, + inconsistent=any( + d["code"] == "relaxed_shot_setting" for d in diagnostics_list + ), a=left, b=right, delta=left["score"] - right["score"] if left["score"] is not None and right["score"] is not None else None, - diagnostics=report_diagnostics(audits, config, {a: aa, b: bb}, True), + diagnostics=diagnostics_list, coverage=coverage, deltas=deltas, ), diff --git a/quickdash/cli.py b/quickdash/cli.py index e1fdadb..1966e20 100644 --- a/quickdash/cli.py +++ b/quickdash/cli.py @@ -30,7 +30,9 @@ def main(argv=None): help="Result CSV files (model labels must be unique across files)", ) parser.add_argument( - "--catalogue", required=True, help="Catalogue YAML or manifest pointing to per-eval files" + "--catalogue", + required=True, + help="Catalogue YAML or manifest pointing to per-eval files", ) parser.add_argument("--weights", required=True, help="Weighting profile YAML") parser.add_argument( @@ -42,6 +44,12 @@ def main(argv=None): metavar=("A", "B"), help="Model names to compare on shared coverage", ) + parser.add_argument( + "--matching", + choices=["strict", "relaxed"], + default="strict", + help="Relaxed allows few-shot mismatches with explicit warnings", + ) parser.add_argument("--format", choices=["tree", "json"], default="tree") parser.add_argument( "--strict", @@ -61,9 +69,10 @@ def main(argv=None): a=args.compare[0], b=args.compare[1], diagnostics="collect", + matching=args.matching, ) if args.compare - else analyze(rows, config, diagnostics="collect") + else analyze(rows, config, diagnostics="collect", matching=args.matching) ) for d in report.diagnostics: print( diff --git a/quickdash/config.py b/quickdash/config.py index 740bb58..4872cf0 100644 --- a/quickdash/config.py +++ b/quickdash/config.py @@ -204,7 +204,7 @@ def validate_rules(config): "category", "match", "metric", - "filter", + "metric_filter", "shots", "select", "score", @@ -212,7 +212,7 @@ def validate_rules(config): "warning", "aggregation", }, - {"name", "category", "match", "metric", "filter", "score"}, + {"name", "category", "match", "metric", "metric_filter", "score"}, ) if not isinstance(e["name"], str) or not e["name"] or e["name"] in names: raise ValueError("Eval names must be unique and nonempty") @@ -223,7 +223,7 @@ def validate_rules(config): if ( not isinstance(e["metric"], str) or not e["metric"] - or not isinstance(e["filter"], str) + or not isinstance(e["metric_filter"], str) ): raise ValueError("Metric and filter must be strings") validate_match(e["match"]) @@ -494,9 +494,9 @@ def classify(rows, config): ) elif r["metric"] != e["metric"]: decision = "Alternate metric; using " + e["metric"] - elif r["filter"] != e["filter"]: + elif r["filter"] != e["metric_filter"]: decision = "Alternate extraction filter; using " + ( - e["filter"] or "(empty)" + e["metric_filter"] or "(empty)" ) elif "shots" in e and str(r["n_shot"]) != str(int(e["shots"])): decision = ( @@ -690,10 +690,26 @@ def eligible(task): return config +def validate_language_exclusions(value): + if "exclude_languages" in value: + languages = value["exclude_languages"] + if ( + not isinstance(languages, list) + or any( + not isinstance(x, str) or not LANGUAGE_CODE.fullmatch(x) + for x in languages + ) + or len(set(languages)) != len(languages) + ): + raise ValueError( + "exclude_languages must be unique canonical language codes" + ) + + def validate_suite(s): object_keys( s, - {"version", "name", "mode", "evals", "exclude", "notes"}, + {"version", "name", "mode", "evals", "exclude", "exclude_languages", "notes"}, {"version", "name", "mode"}, ) if ( @@ -717,6 +733,7 @@ def validate_suite(s): or len(set(s["exclude"])) != len(s["exclude"]) ): raise ValueError("Invalid exclusions") + validate_language_exclusions(s) if s["mode"] == "available": if "evals" in s: raise ValueError("Available mode does not declare required evals") @@ -725,7 +742,31 @@ def validate_suite(s): raise ValueError("Fixed suite needs required evals") names = set() for e in s["evals"]: - object_keys(e, {"name", "variants"}, {"name"}) + object_keys( + e, + { + "name", + "variants", + "metric", + "metric_filter", + "shots", + "exclude_languages", + }, + {"name"}, + ) + validate_language_exclusions(e) + if "metric" in e and ( + not isinstance(e["metric"], str) or not e["metric"].strip() + ): + raise ValueError("metric must be nonempty text") + if "metric_filter" in e and not isinstance(e["metric_filter"], str): + raise ValueError("metric_filter must be text") + if "shots" in e and ( + not number(e["shots"]) + or int(e["shots"]) != e["shots"] + or not 0 <= e["shots"] <= 2**53 - 1 + ): + raise ValueError("shots must be a nonnegative integer") if ( not isinstance(e["name"], str) or not e["name"].strip() @@ -801,44 +842,127 @@ def validate_profile(p): return p -def resolve_config(catalogue, suite, profile): +def catalogue_tasks(catalogue, e): + """Known concrete tasks, independent of which model results happen to be loaded.""" + tasks = {t for g in catalogue["languages"] for t in g["tasks"]} + if "name" in e["match"]: + tasks.add(e["match"]["name"]) + return sorted( + t + for t in tasks + if match_task(e["match"], t) is not None + and ("select" not in e or match_task(e["select"], t) is not None) + ) + + +def effective_catalogue(catalogue, suite): + """Apply whole-eval overrides without changing the global interpretation rules.""" validate_catalogue(catalogue) validate_suite(suite) - validate_profile(profile) - evals = catalogue["evals"] - if suite["mode"] == "fixed": - evals = [] - for required in suite["evals"]: - e = next( - (e for e in catalogue["evals"] if e["name"] == required["name"]), None + result = deepcopy(catalogue) + by_name = {e["name"]: e for e in result["evals"]} + for name in suite.get("exclude", []): + if name not in by_name: + raise ValueError("Excluded eval has no catalogue rule: " + name) + for setting in suite.get("evals", []): + if setting["name"] not in by_name: + raise ValueError("Suite eval has no catalogue rule: " + setting["name"]) + e = by_name[setting["name"]] + e.update( + { + k: setting[k] + for k in ("metric", "metric_filter", "shots") + if k in setting + } + ) + return validate_catalogue(result) + + +def resolve_suite(catalogue, suite): + """Compile authored membership into concrete requirements and exclusions.""" + validate_suite(suite) + by_name = {e["name"]: e for e in catalogue["evals"]} + task_sets = {name: catalogue_tasks(catalogue, e) for name, e in by_name.items()} + metadata = {t: g for g in catalogue["languages"] for t in g["tasks"]} + + def excluded(languages, tasks, context): + def codes(task): + g = metadata.get(task, {}) + return { + g.get(k) for k in ("language", "source_language", "target_language") + } + + known = set().union(*(codes(t) for t in tasks)) + for language in languages: + if language not in known: + raise ValueError( + "Excluded language has no catalogue assignment in " + + context + + ": " + + language + ) + return {t for t in tasks if codes(t).intersection(languages)} + + all_tasks = set(t for tasks in task_sets.values() for t in tasks) + removed = excluded(suite.get("exclude_languages", []), all_tasks, "catalogue") + result = deepcopy(suite) + if suite["mode"] == "available": + for name in suite.get("exclude", []): + if name not in by_name: + raise ValueError("Excluded eval has no catalogue rule: " + name) + result["_excluded_tasks"] = sorted(removed) + return result + for required in result["evals"]: + e = by_name.get(required["name"]) + if e is None: + raise ValueError("Suite eval has no catalogue rule: " + required["name"]) + tasks = task_sets[e["name"]] + local_removed = excluded( + required.get("exclude_languages", []), tasks, e["name"] + ) + variants = required.get("variants", [{"task": t} for t in tasks]) + if not variants: + raise ValueError( + "Required eval needs known tasks in the catalogue: " + e["name"] ) - if e is None: + for v in variants: + matches = [ + rule + for rule in catalogue["evals"] + if match_task(rule["match"], v["task"]) is not None + ] + if ( + v["task"] not in tasks + or len(matches) != 1 + or matches[0]["name"] != e["name"] + or ("shots" in e and "n_shot" in v and e["shots"] != v["n_shot"]) + ): raise ValueError( - "Suite eval has no catalogue rule: " + required["name"] + "Required variant is not selected by its catalogue rule: " + + v["task"] ) - for v in required.get("variants", []): - matches = [ - r - for r in catalogue["evals"] - if match_task(r["match"], v["task"]) is not None - ] - if ( - len(matches) != 1 - or matches[0]["name"] != e["name"] - or ("select" in e and match_task(e["select"], v["task"]) is None) - or ("shots" in e and "n_shot" in v and e["shots"] != v["n_shot"]) - ): - raise ValueError( - "Required variant is not selected by its catalogue rule: " - + v["task"] - ) - if "variants" in required: - validate_aggregation_selection(e, required["variants"], catalogue) - evals.append(e) + required["variants"] = [ + v for v in variants if v["task"] not in removed | local_removed + ] + validate_aggregation_selection(e, required["variants"], catalogue) + return result + + +def resolve_inputs(catalogue, suite, profile): + """Resolve once for classification, membership, scoring, and diagnostics.""" + catalogue = effective_catalogue(catalogue, suite) + resolved = resolve_suite(catalogue, suite) + validate_profile(profile) + names = {e["name"] for e in resolved.get("evals", [])} + evals = [ + e + for e in catalogue["evals"] + if suite["mode"] == "available" or e["name"] in names + ] weights = dict(profile["weights"]) for e in evals: weights.setdefault(e["category"], 0) - return validate_config( + scheme = validate_config( dict( version=1, name=catalogue["name"], @@ -850,11 +974,18 @@ def resolve_config(catalogue, suite, profile): notes=catalogue.get("notes", []) + profile.get("notes", []), ) ) + return dict(catalogue=catalogue, suite=resolved, profile=profile, scheme=scheme) + + +def resolve_config(catalogue, suite, profile): + return resolve_inputs(catalogue, suite, profile)["scheme"] def in_suite(row, suite): if suite["mode"] == "available": - return row["eval"] not in suite.get("exclude", []) + return row["eval"] not in suite.get("exclude", []) and row[ + "task" + ] not in suite.get("_excluded_tasks", []) e = next((e for e in suite["evals"] if e["name"] == row["eval"]), None) return e is not None and ( "variants" not in e diff --git a/tests/engine_adapter.cjs b/tests/engine_adapter.cjs index 5e45e35..3ecaa19 100644 --- a/tests/engine_adapter.cjs +++ b/tests/engine_adapter.cjs @@ -5,6 +5,6 @@ function run(c){try{ const config=c.yaml?{catalogue:E.parseCatalogue(c.yaml.catalogue),profile:S.parseWeightProfile(c.yaml.profile),suite:S.parseSuite(c.yaml.suite)}:c.config; if(Object.hasOwn(c,'eval_definitions'))config.catalogue=E.assembleCatalogue(config.catalogue,c.eval_definitions); const rows=c.csv!==undefined?E.parseCSV(c.csv):c.rows; - return {value:c.operation==='compare'?A.compare(rows,config,c.a||'A',c.b||'B'):A.analyze(rows,config)}; + return {value:c.operation==='compare'?A.compare(rows,config,c.a||'A',c.b||'B',c.matching||'strict'):A.analyze(rows,config,c.matching||'strict')}; }catch(error){return {error:true,message:error.message};}} process.stdout.write(JSON.stringify(JSON.parse(fs.readFileSync(0,'utf8')).map(run))); diff --git a/tests/test_components.cjs b/tests/test_components.cjs index 2910ba3..6f1516f 100644 --- a/tests/test_components.cjs +++ b/tests/test_components.cjs @@ -2,7 +2,7 @@ const test=require('node:test'),assert=require('node:assert/strict'); const E=require('../app/eval_config.js'),A=require('../app/analysis.js'); const levels=['low','medium','high','top']; -const config=()=>({version:1,name:'Components',weights:{Reasoning:1},english_weights:{Reasoning:.5},evals:[{name:'Poly',category:'Reasoning',match:{regex:'poly_.+'},metric:'acc',filter:'none',score:{scale:1},aggregation:{components:levels.map((name,i)=>({name,match:{regex:'poly_.+_'+name},relative_weight:2**i}))}}],languages:['en','de','fr'].map((lang,i)=>({tasks:levels.map(l=>'poly_'+lang+'_'+l),scope:'single',language:['eng_Latn','deu_Latn','fra_Latn'][i]}))}); +const config=()=>({version:1,name:'Components',weights:{Reasoning:1},english_weights:{Reasoning:.5},evals:[{name:'Poly',category:'Reasoning',match:{regex:'poly_.+'},metric:'acc',metric_filter:'none',score:{scale:1},aggregation:{components:levels.map((name,i)=>({name,match:{regex:'poly_.+_'+name},relative_weight:2**i}))}}],languages:['en','de','fr'].map((lang,i)=>({tasks:levels.map(l=>'poly_'+lang+'_'+l),scope:'single',language:['eng_Latn','deu_Latn','fra_Latn'][i]}))}); const rows=(cfg=config(),langs=['en'],values=[.6,.3,.15,0],checkpoint='A')=>E.auditRows(langs.flatMap(lang=>levels.map((l,i)=>({checkpoint,task:'poly_'+lang+'_'+l,metric:'acc',filter:'none',n_shot:'0',harness:'test',backend:'cpu',value:values[i]}))),cfg); const close=(a,b)=>assert.ok(Math.abs(a-b)<1e-9,`${a} != ${b}`); test('relative weights round trip; invalid component rules reject',()=>{ @@ -60,7 +60,7 @@ test('200 independent component calculations reconcile modes, groups, and catego for(let trial=0;trial<200;trial++){ const c=config(),share=rand(),weights=levels.map(()=>1+Math.floor(rand()*8));c.english_weights.Reasoning=share; c.evals[0].aggregation.components.forEach((x,i)=>x.relative_weight=weights[i]); - c.evals.push({name:'Ordinary',category:'Reasoning',match:{name:'ordinary'},metric:'acc',filter:'none',score:{scale:1}}); + c.evals.push({name:'Ordinary',category:'Reasoning',match:{name:'ordinary'},metric:'acc',metric_filter:'none',score:{scale:1}}); c.languages.push({tasks:['ordinary'],scope:'single',language:'eng_Latn'}); const inputs=['en','de','fr'].map(()=>levels.map(()=>rand())); let r=inputs.flatMap((values,i)=>rows(c,[['en','de','fr'][i]],values)); @@ -71,7 +71,7 @@ test('200 independent component calculations reconcile modes, groups, and catego } }); test('missing groups redistribute eval/category weights',()=>{ - const c=config();c.weights={Reasoning:.4,Other:.6};c.evals.push({name:'Other',category:'Other',metric:'acc',filter:'none',match:{name:'other'},score:{scale:1}}); + const c=config();c.weights={Reasoning:.4,Other:.6};c.evals.push({name:'Other',category:'Other',metric:'acc',metric_filter:'none',match:{name:'other'},score:{scale:1}}); const incomplete=rows(c).slice(0,3),ordinary={...incomplete[0],task:'other',eval:'Other',category:'Other',score_100:70,raw_score_100:70}; const t=A.totals([...incomplete,ordinary],c,c.weights);close(t.score,70);assert.ok(t.evals.find(e=>e.name==='Poly').excluded);close(t.rowWeights.get(ordinary),1); }); @@ -87,7 +87,7 @@ test('catalogue rejects incompatible component matching and selection',()=>{ c=>c.evals[0].select={regex:'poly_.+_(low|medium|high)'}, c=>c.languages[0].tasks.pop(), c=>c.evals[0].aggregation.components[0].metric='other', - c=>c.evals[0].aggregation.components[0].filter='other', + c=>c.evals[0].aggregation.components[0].metric_filter='other', c=>c.evals[0].aggregation.components[0].score={scale:100}, c=>c.evals[0].aggregation.components[0].normalize={min:.25,max:1}, c=>c.evals[0].aggregation.components[0].shots=5 diff --git a/tests/test_data.cjs b/tests/test_data.cjs index 59eb179..4c1d9ad 100644 --- a/tests/test_data.cjs +++ b/tests/test_data.cjs @@ -3,7 +3,7 @@ const {diagnosticsFor}=require('./diagnostic_fixture.cjs'); const test=require('node:test'),assert=require('node:assert/strict'); const {parseCSV,totals,comparisonRows,comparisonCoverage,languageRoles}=require('../app/analysis.js'); const {auditRows,normalizeScore,validateConfig}=require('../app/eval_config.js'); -const config=()=>({version:1,name:'Fixture',weights:{C:1},evals:[{name:'Eval',category:'C',match:{regex:'task_.+'},metric:'acc',filter:'',score:{scale:1},normalize:{min:.25,max:1}}],languages:[{tasks:['task_en'],scope:'single',language:'eng_Latn'}]}); +const config=()=>({version:1,name:'Fixture',weights:{C:1},evals:[{name:'Eval',category:'C',match:{regex:'task_.+'},metric:'acc',metric_filter:'',score:{scale:1},normalize:{min:.25,max:1}}],languages:[{tasks:['task_en'],scope:'single',language:'eng_Latn'}]}); const row=(patch={})=>({checkpoint:'Model A',task:'task_en',metric:'acc',filter:'',n_shot:'0',harness:'test',backend:'cpu',value:'.625',...patch}); const close=(a,b)=>{assert.ok(Number.isFinite(a)&&Number.isFinite(b));assert.ok(Math.abs(a-b)<1e-9,`${a} != ${b}`);}; diff --git a/tests/test_data.py b/tests/test_data.py index 15dc47f..106f64a 100644 --- a/tests/test_data.py +++ b/tests/test_data.py @@ -19,7 +19,7 @@ def config(): return dict(version=1, name='Fixture', weights={'C': 1}, evals=[dict( - name='Eval', category='C', match={'regex': 'task_.+'}, metric='acc', filter='', + name='Eval', category='C', match={'regex': 'task_.+'}, metric='acc', metric_filter='', score={'scale': 1}, normalize={'min': .25, 'max': 1})], languages=[dict(tasks=['task_en'], scope='single', language='eng_Latn')]) @@ -40,7 +40,7 @@ def inputs(folder, c=None): profile = {k:v for k,v in c.items() if k in ['version','name','weights','english_weights','aggregate']} (folder/'catalogue.yaml').write_text(json.dumps(catalogue)) (folder/'weights.yaml').write_text(json.dumps(profile)) - return dict(catalogue_path=folder/'catalogue.yaml', weights_path=folder/'weights.yaml', suite_path=ROOT/'configs/sets/any-available.yaml') + return dict(catalogue_path=folder/'catalogue.yaml', weights_path=folder/'weights.yaml', suite_path=ROOT/'configs/examples/eval-set.yaml') class DataContracts(unittest.TestCase): diff --git a/tests/test_engines.py b/tests/test_engines.py index 5ace88e..e6a8659 100644 --- a/tests/test_engines.py +++ b/tests/test_engines.py @@ -27,7 +27,7 @@ def fixture(): category="C", match={"regex": "e_.+"}, metric="acc", - filter="none", + metric_filter="none", score={"scale": 1}, normalize={"min": 0.25, "max": 1}, ) @@ -83,10 +83,17 @@ def native(c): rows = parse_csv(c["csv"]) if "csv" in c else c["rows"] value = ( compare( - rows, cfg, a=c.get("a", "A"), b=c.get("b", "B"), diagnostics="collect" + rows, + cfg, + a=c.get("a", "A"), + b=c.get("b", "B"), + diagnostics="collect", + matching=c.get("matching", "strict"), ) if c.get("operation") == "compare" - else analyze(rows, cfg, diagnostics="collect") + else analyze( + rows, cfg, diagnostics="collect", matching=c.get("matching", "strict") + ) ) return {"value": value} except ValueError as error: @@ -354,6 +361,413 @@ def test_clipping_defaults_to_true(self): self.assertNotIn("error", result) self.assertAlmostEqual(result["value"]["models"][0]["score"], score) + def test_relaxed_fewshot_selection_and_warnings(self): + c = fixture() + c["catalogue"]["evals"][0]["shots"] = 5 + rr = [row(n_shot="5"), row(checkpoint="B", n_shot="0", value=".4")] + case = dict(config=c, rows=rr, operation="compare", matching="relaxed") + for result in self.both([case])[0]: + self.assertNotIn("error", result) + r = result["value"] + self.assertAlmostEqual(r["delta"], 30) + self.assertTrue(r["inconsistent"]) + self.assertEqual(r["b"]["measurements"][0]["n_shot"], "0") + self.assertEqual( + r["deltas"][0]["measurement_b"], r["b"]["measurements"][0]["id"] + ) + warnings = [ + d for d in r["diagnostics"] if d["code"] == "relaxed_shot_setting" + ] + self.assertEqual( + [ + (d["model"], d["expected_shots"], d["actual_shots"]) + for d in warnings + ], + [("B", 5, 0)], + ) + for result in self.both([{**case, "matching": "strict"}])[0]: + self.assertIsNone(result["value"]["delta"]) + self.assertFalse(result["value"]["inconsistent"]) + # Exact settings win, irrespective of score; an unused alternative's value is not validated. + more = rr + [ + row(checkpoint="B", n_shot="5", value=".55"), + row(checkpoint="B", n_shot="1", value="NaN"), + ] + for result in self.both([{**case, "rows": more}])[0]: + self.assertNotIn("error", result) + r = result["value"] + self.assertAlmostEqual(r["delta"], 10) + self.assertFalse(r["inconsistent"]) + self.assertEqual( + [m["n_shot"] for m in r["b"]["measurements"] if m["included"]], ["5"] + ) + for result in self.both( + [{**case, "rows": rr + [row(checkpoint="B", n_shot="3", value=".7")]}] + )[0]: + self.assertEqual( + [ + m["n_shot"] + for m in result["value"]["b"]["measurements"] + if m["included"] + ], + ["3"], + ) + tie = [rr[0], row(checkpoint="B", n_shot="3"), row(checkpoint="B", n_shot="7")] + for result in self.both([{**case, "rows": tie}])[0]: + self.assertIsNone(result["value"]["delta"]) + self.assertIn( + "ambiguous_shot_setting", + {d["code"] for d in result["value"]["diagnostics"]}, + ) + for field, value in [ + ("filter", "other"), + ("metric", "other"), + ("harness", "other"), + ("backend", "other"), + ]: + bad = deepcopy(rr) + bad[1][field] = value + for result in self.both([{**case, "rows": bad}])[0]: + self.assertIsNone(result["value"]["delta"]) + self.assertNotIn( + "relaxed_shot_setting", + {d["code"] for d in result["value"]["diagnostics"]}, + ) + for result in self.both([{**case, "matching": "best_score"}])[0]: + self.assertIn("error", result) + for result in self.both([{**case, "operation": "analyze"}])[0]: + self.assertEqual(len(result["value"]["models"]), 2) + self.assertTrue(result["value"]["inconsistent"]) + # An eval without an expected shot setting retains strict A/B matching. + unpinned = fixture() + for result in self.both([{**case, "config": unpinned}])[0]: + self.assertIsNone(result["value"]["delta"]) + # Required task coverage comes from the set; expected shots come from the catalogue. + fixed = deepcopy(c) + fixed["suite"] = dict( + version=1, + name="Required", + mode="fixed", + evals=[dict(name="E", variants=[dict(task="e_en")])], + ) + for result in self.both([{**case, "config": fixed}])[0]: + self.assertTrue(result["value"]["coverage"]["complete"]) + self.assertEqual(result["value"]["coverage"]["sharedRequired"], 1) + for result in self.both([{**case, "a": "B", "b": "A"}])[0]: + self.assertAlmostEqual(result["value"]["delta"], -30) + + def test_relaxed_matching_preserves_component_protocols_and_validation(self): + c = fixture() + e = c["catalogue"]["evals"][0] + e["shots"] = 5 + e["aggregation"] = { + "components": [ + dict(name=n, match={"name": "e_" + n}, relative_weight=i + 1) + for i, n in enumerate(["low", "high"]) + ] + } + c["catalogue"]["languages"] = [ + dict(tasks=["e_low", "e_high"], language="eng_Latn", scope="single") + ] + rr = [ + row("e_" + n, n_shot=shots, checkpoint=model) + for model, shots in [("A", "5"), ("B", "0")] + for n in ["low", "high"] + ] + case = dict(config=c, rows=rr, operation="compare", matching="relaxed") + for result in self.both([case])[0]: + self.assertNotIn("error", result) + self.assertEqual(result["value"]["delta"], 0) + self.assertTrue(result["value"]["inconsistent"]) + self.check_tree(result["value"]["b"]["tree"]) + split = deepcopy(rr) + split[-1]["n_shot"] = "5" + for rows in [split, rr[:-1]]: + for result in self.both([{**case, "rows": rows}])[0]: + self.assertNotIn("error", result) + self.assertIsNone(result["value"]["delta"]) + self.assertFalse(result["value"]["inconsistent"]) + self.assertIn( + "incomplete_components", + {d["code"] for d in result["value"]["diagnostics"]}, + ) + bad = deepcopy(rr) + bad[-1]["value"] = "NaN" + for rows in [bad, rr + [dict(rr[-1])]]: + for result in self.both([{**case, "rows": rows}])[0]: + self.assertIn("error", result) + + def test_eval_set_defaults_overrides_and_language_exclusions(self): + c = fixture() + c["catalogue"]["evals"][0]["shots"] = 5 + c["catalogue"]["languages"].append( + dict(tasks=["e_ka"], scope="single", language="kat_Geor") + ) + c["suite"] = dict( + version=1, + name="Required", + mode="fixed", + exclude_languages=["kat_Geor"], + evals=[dict(name="E", shots=0, metric="acc_norm", metric_filter="")], + ) + rr = paired( + [row(t, metric="acc_norm", filter="") for t in ["e_en", "e_fr", "e_ka"]] + ) + before = deepcopy(c) + for result in self.both([dict(config=c, rows=rr, operation="compare")])[0]: + self.assertNotIn("error", result) + r = result["value"] + self.assertEqual(r["coverage"]["required"], 2) + self.assertTrue(r["coverage"]["complete"]) + self.assertEqual(r["a"]["score"], 50) + self.assertEqual( + {m["task"] for m in r["a"]["measurements"] if m["included"]}, + {"e_en", "e_fr"}, + ) + self.assertEqual({d["code"] for d in r["diagnostics"]}, {"not_used"}) + self.assertEqual(c, before, "Resolution must not mutate inputs") + c["suite"]["evals"][0]["exclude_languages"] = ["eng_Latn"] + cases = [dict(config=deepcopy(c), rows=rr, operation="compare")] + # Missing excluded data is not a missing requirement. + cases.append( + dict( + config=deepcopy(c), + rows=[r for r in rr if r["task"] == "e_fr"], + operation="compare", + ) + ) + # Relaxation applies after the effective override, preserving actual identities. + changed = [{**r, "n_shot": "3"} if r["checkpoint"] == "B" else r for r in rr] + cases.append( + dict( + config=deepcopy(c), + rows=changed, + operation="compare", + matching="relaxed", + ) + ) + for i, results in enumerate(self.both(cases)): + for result in results: + r = result["value"] + self.assertTrue(r["coverage"]["complete"]) + self.assertEqual(r["coverage"]["required"], 1) + self.assertEqual( + [m["task"] for m in r["a"]["measurements"] if m["included"]], + ["e_fr"], + ) + if i == 2: + d = next( + d + for d in r["diagnostics"] + if d["code"] == "relaxed_shot_setting" + ) + self.assertEqual((d["expected_shots"], d["actual_shots"]), (0, 3)) + # Removing an included language must warn even though another language exists. + c["suite"]["evals"][0].pop("exclude_languages") + for result in self.both( + [ + dict( + config=c, + rows=paired([row(metric="acc_norm", filter="")]), + operation="compare", + ) + ] + )[0]: + self.assertFalse(result["value"]["coverage"]["complete"]) + self.assertIn( + "missing_suite_data", + {d["code"] for d in result["value"]["diagnostics"]}, + ) + # Metric mismatches are never relaxed. + for result in self.both( + [ + dict( + config=c, + rows=paired([row()]), + operation="compare", + matching="relaxed", + ) + ] + )[0]: + self.assertIsNone(result["value"]["a"]["score"]) + self.assertIn( + "missing_scoring_field", + {d["code"] for d in result["value"]["diagnostics"]}, + ) + + def test_required_inventory_unicode_order(self): + c = fixture() + tasks = ["e_😀", "e_\ue000", "e_fr"] + c["catalogue"]["languages"] = [ + dict(tasks=tasks, scope="single", language="fra_Latn") + ] + c["suite"] = dict( + version=1, name="Required", mode="fixed", evals=[dict(name="E")] + ) + for result in self.both([dict(config=c, rows=[row("e_fr")])])[0]: + self.assertEqual( + [v["task"] for v in result["value"]["coverage"][0]["missing"]], + sorted(tasks[:2]), + ) + + def test_eval_set_reference_errors(self): + base = fixture() + base["suite"] = dict( + version=1, name="Required", mode="fixed", evals=[dict(name="E")] + ) + bad = [] + for patch in [ + dict(name="Typo"), + dict(exclude_languages=["deu_Latn"]), + dict(shots=-1), + dict(shots=True), + dict(metric=""), + dict(metric_filter=None), + dict(filter="none"), + dict(variants=[dict(task="e_typo")]), + ]: + c = deepcopy(base) + c["suite"]["evals"][0].update(patch) + bad.append(c) + c = deepcopy(base) + c["suite"]["exclude_languages"] = ["kat_Geor"] + bad.append(c) + c = deepcopy(base) + c["suite"]["exclude_languages"] = ["eng_Latn", "eng_Latn"] + bad.append(c) + c = fixture() + c["suite"]["exclude"] = ["Typo"] + bad.append(c) + for outputs in self.both([dict(config=c, rows=paired([row()])) for c in bad]): + for result in outputs: + self.assertIn("error", result) + + def test_eval_set_translation_exclusions_and_empty_selection(self): + c = fixture() + c["catalogue"]["languages"] = [ + dict( + tasks=["e_to_ka"], + scope="translation", + source_language="eng_Latn", + target_language="kat_Geor", + ), + dict( + tasks=["e_from_ka"], + scope="translation", + source_language="kat_Geor", + target_language="eng_Latn", + ), + dict(tasks=["e_fr"], scope="single", language="fra_Latn"), + ] + rr = paired([row(t) for t in ["e_to_ka", "e_from_ka", "e_fr"]]) + cases = [] + for mode in ["fixed", "available"]: + s = dict( + version=1, name="Excluded", mode=mode, exclude_languages=["kat_Geor"] + ) + if mode == "fixed": + s["evals"] = [dict(name="E")] + c["suite"] = s + cases.append(dict(config=deepcopy(c), rows=rr, operation="compare")) + c["suite"]["exclude_languages"].append("fra_Latn") + cases.append(dict(config=deepcopy(c), rows=rr, operation="compare")) + for i, results in enumerate(self.both(cases)): + for result in results: + r = result["value"] + self.assertEqual( + [m["task"] for m in r["a"]["measurements"] if m["included"]], + [] if i % 2 else ["e_fr"], + ) + self.assertNotIn( + "missing_suite_data", {d["code"] for d in r["diagnostics"]} + ) + if i % 2: + self.assertIsNone(r["a"]["score"]) + + def test_eval_set_inheritance_and_component_language_exclusions(self): + c = fixture() + e = c["catalogue"]["evals"][0] + e["shots"] = 5 + e["aggregation"] = { + "components": [ + dict(name=n, match={"regex": "e_.+_" + n}, relative_weight=i + 1) + for i, n in enumerate(["low", "high"]) + ] + } + c["catalogue"]["languages"] = [ + dict( + tasks=["e_" + lang + "_low", "e_" + lang + "_high"], + language=code, + scope="single", + ) + for lang, code in [("en", "eng_Latn"), ("fr", "fra_Latn")] + ] + c["suite"] = dict( + version=1, + name="Required", + mode="fixed", + evals=[dict(name="E", exclude_languages=["fra_Latn"])], + ) + rr = paired( + [ + row("e_" + lang + "_" + part, n_shot="5", value=value) + for lang in ["en", "fr"] + for part, value in [("low", ".25"), ("high", "1")] + ] + ) + for result in self.both([dict(config=c, rows=rr, operation="compare")])[0]: + r = result["value"] + self.assertEqual(r["coverage"]["required"], 2) + self.assertTrue(r["coverage"]["complete"]) + self.assertAlmostEqual(r["a"]["score"], 200 / 3) + self.assertEqual({d["code"] for d in r["diagnostics"]}, {"not_used"}) + # Explicit membership may not slice a required component group. + c["suite"]["evals"][0]["variants"] = [dict(task="e_en_low")] + for result in self.both([dict(config=c, rows=rr)])[0]: + self.assertIn("error", result) + + def test_flagship_exclusions_are_set_policy(self): + config = load_config( + catalogue=ROOT / "configs/catalogue.yaml", + weights=ROOT / "configs/weights/oellm.yaml", + eval_set=ROOT / "configs/sets/flagship-1.yaml", + ) + rows = parse_csv((ROOT / "examples/sample-evals.csv").read_text()) + report = analyze(rows, config, matching="relaxed", diagnostics="collect") + records = report.models[0]["measurements"] + georgian = [ + r + for r in records + if "kat_Geor" + in [ + r["language"].get(k) + for k in ("language", "source_language", "target_language") + ] + ] + self.assertTrue(georgian) + self.assertTrue(all(not r["included"] for r in georgian)) + self.assertTrue( + all(not r["included"] for r in records if r["task"] == "xcsqa_eng_Latn") + ) + config["suite"] = dict(version=1, name="Available", mode="available") + free = analyze(rows, config, matching="relaxed", diagnostics="collect") + included = {r["task"] for r in free.models[0]["measurements"] if r["included"]} + excluded_tasks = ({r["task"] for r in georgian} | {"xcsqa_eng_Latn"}) & included + self.assertTrue(excluded_tasks) + self.assertIn("xcsqa_eng_Latn", excluded_tasks) + warned = { + t for d in report.diagnostics if d["code"] == "not_used" for t in d["tasks"] + } + self.assertTrue(excluded_tasks <= warned) + self.assertFalse( + excluded_tasks.intersection( + t + for d in report.diagnostics + if d["code"] == "missing_suite_data" + for t in d["tasks"] + ) + ) + def test_published_sample_across_all_shipped_configs(self): # Discover files so adding a profile or set automatically extends parity coverage. rows = parse_csv((ROOT / "examples/sample-evals.csv").read_text()) @@ -401,6 +815,11 @@ def test_published_sample_across_all_shipped_configs(self): b=b, ), ] + cases = [ + {**case, "matching": matching} + for matching in ("strict", "relaxed") + for case in cases + ] for outputs in self.both(cases): for result in outputs: self.assertNotIn("error", result) @@ -422,7 +841,7 @@ def test_hand_calculated_modes_and_tree(self): category="C", match={"name": "f_en"}, metric="acc", - filter="none", + metric_filter="none", score={"scale": 1}, ) ) @@ -665,7 +1084,7 @@ def test_randomized_sizes_and_order(self): category=category, match={"regex": name + "_.+"}, metric="acc", - filter="none", + metric_filter="none", score={"scale": 1}, ) ) @@ -854,7 +1273,7 @@ def test_csv_and_yaml_failures_in_both_engines(self): category: C match: {name: e_en} metric: acc - filter: none + metric_filter: none score: {scale: 1} languages: [] notes: @@ -989,7 +1408,7 @@ def test_diagnostic_context_and_suppression(self): category="Other", match={"name": "unused"}, metric="acc", - filter="none", + metric_filter="none", score={"scale": 1}, warning="Unused caveat", ) @@ -1048,7 +1467,7 @@ def test_yaml_scalar_contract(self): category: C match: {name: e_en} metric: acc - filter: none + metric_filter: none score: {scale: 1} normalize: {min: TOKEN, max: 1} languages: [] @@ -1128,7 +1547,7 @@ def test_non_ascii_names_and_numeric_category_order(self): category="1", match={"name": "other"}, metric="acc", - filter="none", + metric_filter="none", score={"scale": 1}, ) ) diff --git a/tests/test_public_browser.mjs b/tests/test_public_browser.mjs index 1872a08..4665fe9 100644 --- a/tests/test_public_browser.mjs +++ b/tests/test_public_browser.mjs @@ -53,7 +53,7 @@ try{ assert.equal(await evaluate("document.querySelector('#suitePreset').selectedOptions[0].textContent"),'Any available'); for(const view of ['categories','languages','comparisons','config','warnings','score'])await click('[data-view='+view+']'); const upload=async(selector,content,name)=>evaluate(`(async()=>{const dt=new DataTransfer();dt.items.add(new File([${JSON.stringify(content)}],${JSON.stringify(name)}));const e=document.querySelector(${JSON.stringify(selector)});e.files=dt.files;if(e.id==='modelFile')await e.onchange({target:e});else await document.querySelector('#view').onchange({target:e});})()`); - await click('[data-view=config]');await upload('#configFile',serializeCatalogue(fixture),'catalogue.yaml'); + await click('[data-view=config]');await upload('#suiteFile',serializeSuite({version:1,name:'Any example',mode:'available'}),'any.yaml');await upload('#configFile',serializeCatalogue(fixture),'catalogue.yaml'); await change('#weightPreset','1'); await upload('#modelFile',fs.readFileSync(path.join(root,'examples/scores.csv'),'utf8'),'scores.csv'); assert.equal(await evaluate("document.querySelector('#error').textContent"),''); @@ -82,13 +82,13 @@ try{ await click('[data-view=config]'); assert.match(await evaluate("document.querySelector('#view').textContent"),/Example math/); await click('[data-view=score]');await change('[data-weight=Reasoning]','.8');await change('[data-weight=Math]','.2'); - await change('#suitePreset','0'); + await click('[data-view=config]');await upload('#suiteFile',serializeSuite({version:1,name:'Any example',mode:'available'}),'any.yaml');await click('[data-view=score]'); assert.equal(await evaluate("document.querySelector('[data-weight=Reasoning]').value"),'0.8'); await click('[data-view=config]'); const unavailable={...fixed,evals:[{name:'Absent eval'}]}; await upload('#suiteFile',serializeSuite(unavailable),'invalid-set.yaml'); assert.match(await evaluate("document.querySelector('#error').textContent"),/no catalogue rule/); - assert.equal(await evaluate("document.querySelector('#suitePreset').value"),'0'); + assert.equal(await evaluate("document.querySelector('#suitePreset').value"),'custom'); await upload('#suiteFile',serializeSuite(fixed),'set.yaml'); await change('#weightPreset','0'); assert.equal(await evaluate("document.querySelector('#suitePreset').value"),'custom'); @@ -98,7 +98,7 @@ try{ await click('#exportSuite');assert.deepEqual(await evaluate('exportBlob.text().then(parseSuite)'),fixed); await click('#exportWeights');assert.deepEqual(await evaluate('exportBlob.text().then(parseWeightProfile)'),{...weighting,english_weights:weighting.english_weights,aggregate:'standard'}); await evaluate('URL.createObjectURL=originalCreate;HTMLAnchorElement.prototype.click=originalClick'); - await change('#suitePreset','0');await click('#clearModels'); + await click('[data-view=config]');await upload('#suiteFile',serializeSuite({version:1,name:'Any example',mode:'available'}),'any.yaml');await click('[data-view=score]');await click('#clearModels'); assert.equal(await evaluate("document.querySelector('#modelA').options.length"),0); await click('[data-view=config]');await upload('#configFile',serializeCatalogue(invalid),'small.yaml'); await upload('#modelFile',fs.readFileSync(path.join(root,'examples/scores.csv'),'utf8'),'bad-for-config.csv'); @@ -108,7 +108,7 @@ try{ const results=path.join(temporary,'results');fs.mkdirSync(results); fs.copyFileSync(path.join(root,'examples/scores.csv'),path.join(results,'example.csv')); const shared=path.join(temporary,'shared'); - execFileSync('python3',['-m','app.build','--results-dir',results,'--catalogue',path.join(root,'configs/examples/catalogue.yaml'),'--weights',path.join(root,'configs/examples/weights.yaml'),'--eval-set',path.join(root,'configs/sets/any-available.yaml'),'--output',shared],{cwd:root,stdio:'pipe'}); + execFileSync('python3',['-m','app.build','--results-dir',results,'--catalogue',path.join(root,'configs/examples/catalogue.yaml'),'--weights',path.join(root,'configs/examples/weights.yaml'),'--eval-set',path.join(root,'configs/examples/eval-set.yaml'),'--output',shared],{cwd:root,stdio:'pipe'}); await navigate(pathToFileURL(path.join(shared,'index.html')).href); assert.equal(await evaluate("document.querySelector('#modelB').value"),'Example B'); for(const view of ['score','categories','languages','comparisons','config','warnings'])await click('[data-view='+view+']'); @@ -135,7 +135,7 @@ try{ // Weighted components use fictional scores and remain inspectable after exclusion. await click('#clearModels');await click('[data-view=config]'); const levels=['low','medium','high','top']; - const componentCatalogue={version:1,name:'Component example',evals:[{name:'Poly example',category:'Reasoning',match:{regex:'poly_.+'},metric:'acc',filter:'none',score:{scale:1},aggregation:{components:levels.map((name,i)=>({name,match:{regex:'poly_.+_'+name},relative_weight:2**i})),note:'Fictional component fixture.'}}],languages:['en','de'].map((lang,i)=>({tasks:levels.map(l=>'poly_'+lang+'_'+l),scope:'single',language:i?'deu_Latn':'eng_Latn'}))}; + const componentCatalogue={version:1,name:'Component example',evals:[{name:'Poly example',category:'Reasoning',match:{regex:'poly_.+'},metric:'acc',metric_filter:'none',score:{scale:1},aggregation:{components:levels.map((name,i)=>({name,match:{regex:'poly_.+_'+name},relative_weight:2**i})),note:'Fictional component fixture.'}}],languages:['en','de'].map((lang,i)=>({tasks:levels.map(l=>'poly_'+lang+'_'+l),scope:'single',language:i?'deu_Latn':'eng_Latn'}))}; await upload('#configFile',serializeCatalogue(componentCatalogue),'components.yaml'); await upload('#weightsFile',serializeWeightProfile({version:1,name:'Component weights',weights:{Reasoning:1},english_weights:{Reasoning:.5}}),'weights.yaml'); const componentCSV=(model,omit=false)=>['checkpoint,task,metric,filter,n_shot,harness,backend,value',...['en','de'].flatMap(lang=>levels.flatMap((l,i)=>omit&&lang==='en'&&l==='top'?[]:[`${model},poly_${lang}_${l},acc,none,0,test,cpu,${model==='Component A'?(lang==='en'?[.6,.3,.15,0][i]:.2):(lang==='en'?.3:.1)}`]))].join('\n'); @@ -214,6 +214,14 @@ try{ await upload('#configFile',exportedCatalogue,'portable-catalogue.yaml'); assert.equal(await evaluate("document.querySelector('#cards').hidden"),false); + const strictSampleCards=await evaluate("document.querySelector('#cards').textContent"); + await change('#matching','relaxed'); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + assert.match(await evaluate("document.querySelector('#matchingNotice').textContent"),/INCONSISTENT EVALUATION SETTINGS/); + await click('[data-view=warnings]'); + assert.match(await evaluate("document.querySelector('#view').textContent"),/Using 0 shots despite expected 10/); + await change('#matching','strict'); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),strictSampleCards); const syntheticNames=await evaluate("syntheticOptions.map(o=>o.name)"); const scores=[]; for(const name of syntheticNames){ @@ -228,6 +236,63 @@ try{ assert.match(await evaluate("document.querySelector('#demo').textContent"),/Sample dataset/); await click('#clearModels'); assert.equal(await evaluate("document.querySelector('#modelA').options.length"),0); + // Real A/B few-shot differences are allowed only through the explicit runtime option. + await navigate(pathToFileURL(path.join(empty,'index.html')).href); + const expectedShots=structuredClone(fixture);expectedShots.evals[0].shots=5; + await click('[data-view=config]');await upload('#suiteFile',serializeSuite({version:1,name:'Any example',mode:'available'}),'any.yaml'); + await click('[data-view=config]');await upload('#configFile',serializeCatalogue(expectedShots),'expected-shots.yaml'); + await change('#weightPreset','1'); + const shotCSV=(value='0.625')=>'checkpoint,task,metric,filter,n_shot,harness,backend,value\nShots A,example_reasoning_en,acc_norm,none,5,test,cpu,1\nShots B,example_reasoning_en,acc_norm,none,0,test,cpu,'+value; + await upload('#modelFile',shotCSV(),'shots.csv'); + const noShared=await evaluate("document.querySelector('#cards').textContent"); + await change('#matching','relaxed'); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + assert.match(await evaluate("document.querySelector('#matchingNotice').textContent"),/INCONSISTENT/); + assert.equal(await evaluate("document.querySelector('#cards .score-card:last-child strong').textContent"),'50.00'); + await click('[data-view=warnings]');assert.match(await evaluate("document.querySelector('#view').textContent"),/Using 0 shots despite expected 5/); + await change('#matching','strict');assert.equal(await evaluate("document.querySelector('#cards').textContent"),noShared); + // Set overrides reinterpret all models, while catalogue export retains defaults. + await click('[data-view=config]'); + const shotSet={version:1,name:'Override example',mode:'fixed',evals:[{name:'Example reasoning',shots:0,exclude_languages:['fra_Latn']}]}; + await upload('#suiteFile',serializeSuite(shotSet),'override.yaml'); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + assert.match(await evaluate("document.querySelector('#view').textContent"),/Eval-set overrides: shots = 0/); + await change('#matching','relaxed'); + assert.match(await evaluate("document.querySelector('#coverage').textContent"),/1\/1 requirements shared/); + await click('[data-view=warnings]'); + assert.match(await evaluate("document.querySelector('#view').textContent"),/Using 5 shots despite expected 0/); + await change('#matching','strict');await click('[data-view=config]'); + const beforeInvalidSet=await evaluate("document.querySelector('#cards').textContent"); + await upload('#suiteFile',serializeSuite({...shotSet,exclude_languages:['kat_Geor']}),'unknown-language.yaml'); + assert.match(await evaluate("document.querySelector('#error').textContent"),/no catalogue assignment/); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),beforeInvalidSet); + const alternateCSV=shotCSV()+'\nShots A,example_reasoning_en,acc,,0,test,cpu,0.25\nShots B,example_reasoning_en,acc,,0,test,cpu,1'; + await click('#clearModels');await upload('#modelFile',alternateCSV,'alternates.csv');await click('[data-view=config]'); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + const metricSet=structuredClone(shotSet);Object.assign(metricSet.evals[0],{metric:'acc',metric_filter:''}); + await upload('#suiteFile',serializeSuite(metricSet),'metric.yaml'); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + assert.equal(await evaluate("document.querySelector('#cards .score-card:last-child strong').textContent"),'-100.00'); + assert.match(await evaluate("document.querySelector('#view').textContent"),/Eval-set overrides: metric = "acc", metric_filter = "", shots = 0/); + await evaluate(`window.originalCreate=URL.createObjectURL;window.originalClick=HTMLAnchorElement.prototype.click;URL.createObjectURL=b=>{window.exportBlob=b;return 'blob:test'};HTMLAnchorElement.prototype.click=function(){};`); + await click('#exportConfig');assert.deepEqual(await evaluate('exportBlob.text().then(parseCatalogue)'),expectedShots); + await click('#exportSuite');assert.deepEqual(await evaluate('exportBlob.text().then(parseSuite)'),metricSet); + await evaluate('URL.createObjectURL=originalCreate;HTMLAnchorElement.prototype.click=originalClick'); + await upload('#suiteFile',serializeSuite({version:1,name:'Any example',mode:'available'}),'any.yaml'); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),noShared); + // A present excluded language warns; absent excluded languages are not requirements. + const frenchCSV='checkpoint,task,metric,filter,n_shot,harness,backend,value\nShots A,example_reasoning_fr,acc_norm,none,5,test,cpu,0.5\nShots B,example_reasoning_fr,acc_norm,none,5,test,cpu,0.5'; + await click('#clearModels');await upload('#modelFile',shotCSV()+'\n'+frenchCSV.split('\n').slice(1).join('\n'),'languages.csv');await click('[data-view=config]'); + assert.equal(await evaluate("document.querySelector('#error').textContent"),''); + await upload('#suiteFile',serializeSuite({...shotSet,exclude_languages:['fra_Latn']}),'exclusions.yaml'); + await click('[data-view=warnings]');assert.match(await evaluate("document.querySelector('#view').textContent"),/Not used.*example_reasoning_fr/s); + await click('[data-view=config]');await upload('#suiteFile',serializeSuite({version:1,name:'Any example',mode:'available'}),'any.yaml'); + await click('#clearModels');await upload('#modelFile',shotCSV('bad'),'invalid-alternative.csv'); + const beforeRelax=await evaluate("document.querySelector('#cards').textContent"); + await change('#matching','relaxed'); + assert.equal(await evaluate("document.querySelector('#matching').value"),'strict'); + assert.match(await evaluate("document.querySelector('#error').textContent"),/Invalid score/); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),beforeRelax); assert.deepEqual(errors,[]);assert.deepEqual(network,[],'Loading and comparing local files must not send HTTP requests'); console.log('Public browser checks passed: empty start, shared models, independent weights and eval sets, required coverage, temporary uploads, rollback, Pages sample, synthetic choices, clear models and no uploads.'); }finally{ diff --git a/tests/test_suites.cjs b/tests/test_suites.cjs index cafce4a..b1d7e7f 100644 --- a/tests/test_suites.cjs +++ b/tests/test_suites.cjs @@ -1,8 +1,8 @@ 'use strict'; const test=require('node:test'),assert=require('node:assert/strict'); const {validateCatalogue}=require('../app/eval_config.js'); -const {validateWeightProfile,parseWeightProfile,serializeWeightProfile,validateSuite,resolveConfig,scopeRows,suiteCoverage,parseSuite,serializeSuite}=require('../app/suite_config.js'); -const catalogue=()=>({version:1,name:'Catalogue',evals:[{name:'E',category:'C',match:{regex:'e_.+'},metric:'acc',filter:'none',score:{scale:1}},{name:'Unused',category:'D',match:{name:'unused'},metric:'acc',filter:'none',score:{scale:1}}],languages:[{tasks:['e_en'],scope:'single',language:'eng_Latn'},{tasks:['e_fr'],scope:'single',language:'fra_Latn'}]}); +const {validateWeightProfile,parseWeightProfile,serializeWeightProfile,validateSuite,resolveSuite,resolveConfig,scopeRows,suiteCoverage,parseSuite,serializeSuite}=require('../app/suite_config.js'); +const catalogue=()=>({version:1,name:'Catalogue',evals:[{name:'E',category:'C',match:{regex:'e_.+'},metric:'acc',metric_filter:'none',score:{scale:1}},{name:'Unused',category:'D',match:{name:'unused'},metric:'acc',metric_filter:'none',score:{scale:1}}],languages:[{tasks:['e_en'],scope:'single',language:'eng_Latn'},{tasks:['e_fr'],scope:'single',language:'fra_Latn'}]}); const suite=()=>({version:1,name:'Required',mode:'fixed',evals:[{name:'E',variants:[{task:'e_en',n_shot:0},{task:'e_fr',n_shot:5}]}]}); const profile=()=>({version:1,name:'Weights',weights:{C:.5,D:.5}}); const row=(task,n_shot='0')=>({task,n_shot,eval:'E',selected:true}); @@ -24,9 +24,12 @@ test('any available creates no missing or extra suite requirements',()=>{ assert.equal(resolveConfig(catalogue(),s,profile()).evals.length,2); const c=suiteCoverage([row('e_en')],[],s);assert.deepEqual(c.warnings,[]);assert.equal(c.required,null); }); -test('eval-only requirement permits all its recognized variants',()=>{ - const s={...suite(),evals:[{name:'E'}]};assert.equal(scopeRows([row('e_de')],s).missing.length,0); - assert.equal(scopeRows([],s).missing.length,1); +test('whole-eval requirement expands known tasks and reports missing languages',()=>{ + const s=resolveSuite(catalogue(),{...suite(),evals:[{name:'E'}]}); + assert.equal(scopeRows([row('e_en'),row('e_fr')],s).missing.length,0); + assert.equal(scopeRows([row('e_de')],s).missing.length,2); + assert.equal(scopeRows([row('e_de')],s).extras.length,1); + assert.equal(scopeRows([],s).missing.length,2); }); test('invalid suite definitions and catalogue references fail clearly',()=>{ for(const change of [s=>s.mode='wrong',s=>s.evals.push(s.evals[0]),s=>s.evals[0].variants.push(s.evals[0].variants[0]),s=>s.evals[0].variants[0].n_shot=-1,s=>s.evals[0].variants=[],s=>s.evals=[]]){const s=suite();change(s);assert.throws(()=>validateSuite(s));} diff --git a/tests/test_warning_policy.cjs b/tests/test_warning_policy.cjs index 3927b55..f4a4edc 100644 --- a/tests/test_warning_policy.cjs +++ b/tests/test_warning_policy.cjs @@ -3,7 +3,7 @@ const test=require('node:test'),assert=require('node:assert/strict'); const {auditRows}=require('../app/eval_config.js'); const {compare,comparisonCoverage,totals}=require('../app/analysis.js'); const {resolveConfig,suiteCoverage}=require('../app/suite_config.js'); -const catalogue=()=>({version:1,name:'Rules',evals:['E','F'].map(name=>({name,category:'C',match:{name:name.toLowerCase()},metric:'acc_norm',filter:'none',score:{scale:1},warning:name+' caveat'})),languages:[{tasks:['e','f'],scope:'single',language:'eng_Latn'}]}); +const catalogue=()=>({version:1,name:'Rules',evals:['E','F'].map(name=>({name,category:'C',match:{name:name.toLowerCase()},metric:'acc_norm',metric_filter:'none',score:{scale:1},warning:name+' caveat'})),languages:[{tasks:['e','f'],scope:'single',language:'eng_Latn'}]}); const profile={version:1,name:'Weights',weights:{C:1}}; const available={version:1,name:'Any available',mode:'available'}; const fixed={version:1,name:'Required',mode:'fixed',evals:[{name:'E'},{name:'F'}]}; @@ -53,11 +53,11 @@ test('a caveat is not advertised for an eval excluded by A/B coverage',()=>{ const r=run([row(),row('f')],[row()]); assert.deepEqual(r.warnings.filter(w=>w.type==='Config caveat').map(w=>w.name),['E']); }); -test('available-mode exclusions are explicit, validated, and portable across catalogues',()=>{ +test('available-mode exclusions reject unknown catalogue names',()=>{ const {validateSuite}=require('../app/suite_config.js'); for(const exclude of [null,'E',['E','E'],[''],[1]])assert.throws(()=>validateSuite({...available,exclude})); assert.throws(()=>validateSuite({...fixed,exclude:['E']})); - assert.doesNotThrow(()=>resolveConfig(catalogue(),{...available,exclude:['Absent from this catalogue']},profile)); + assert.throws(()=>resolveConfig(catalogue(),{...available,exclude:['Absent from this catalogue']},profile),/no catalogue rule/); const r=run([row()],[row()],{...available,exclude:['E','F']}); assert.equal(r.score,null);assert.equal(r.coverage.pairs.length,0); assert.equal(r.warnings.filter(w=>w.type==='Not used').length,2); diff --git a/tests/test_yaml.cjs b/tests/test_yaml.cjs index f58e7f8..317272d 100644 --- a/tests/test_yaml.cjs +++ b/tests/test_yaml.cjs @@ -12,7 +12,7 @@ evals: category: Reasoning match: {regex: 'example_(en|fr)'} metric: acc_norm - filter: '' + metric_filter: '' shots: 0 score: {scale: 1} normalize: {min: 0.25, max: 1, clip: true} @@ -27,7 +27,7 @@ notes: `; const example=parseCatalogue(source); assert.equal(example.evals[0].normalize.min,.25); -assert.equal(example.evals[0].filter,''); +assert.equal(example.evals[0].metric_filter,''); assert.equal(example.notes[0],'A folded explanation for people editing the config.'); -for(const bad of [source+'version: 1\n',source+'---\nversion: 1',source.replace('min: 0.25','min: .nan'),source.replace('filter: \'\'','filter: [oops'),source.replace('name: Example\n','name: !!js/function function(){}\n')])assert.throws(()=>parseCatalogue(bad)); +for(const bad of [source+'version: 1\n',source+'---\nversion: 1',source.replace('min: 0.25','min: .nan'),source.replace('metric_filter: \'\'','metric_filter: [oops'),source.replace('name: Example\n','name: !!js/function function(){}\n')])assert.throws(()=>parseCatalogue(bad)); console.log('YAML checks passed: source config, round-trip, comments, quoted regex, flow syntax, multiline notes, invalid syntax and duplicate keys.'); From 9eb53b4c43927a551c76d8b2f43d601deca5c307 Mon Sep 17 00:00:00 2001 From: Jonathan Burdge Date: Fri, 2 Oct 2026 11:20:25 +0300 Subject: [PATCH 4/4] Group strict few-shot exclusions and avoid duplicate diagnostics --- app/analysis.js | 21 +++++-- app/app.js | 2 +- docs/configuration.md | 2 + docs/python-api.md | 3 + quickdash/analysis.py | 55 ++++++++++++++---- tests/test_engines.py | 106 ++++++++++++++++++++++++++++++++++ tests/test_public_browser.mjs | 14 +++++ 7 files changed, 186 insertions(+), 17 deletions(-) diff --git a/app/analysis.js b/app/analysis.js index e26889c..b98bced 100644 --- a/app/analysis.js +++ b/app/analysis.js @@ -240,7 +240,7 @@ function normalizationLabel(e){const n=e.normalize;if(!n||n.min===0&&n.max===1)r function sameCoverage(a,b,config=null,matching='strict'){return a.length===b.length&&pairRows(a,b,config,matching).length===a.length;} // Public, serializable analysis boundary used by Python parity tests and the UI. function measurementId(r){return JSON.stringify([r.checkpoint,...['task','metric','filter','n_shot','harness','backend'].map(k=>r[k])]);} -const diagnosticTitles={config_caveat:'Config caveat',no_config:'No config',not_used:'Not used',unknown_language:'Unknown language',invalid_sample_count:'Invalid sample count',relaxed_shot_setting:'Few-shot mismatch allowed',ambiguous_shot_setting:'Ambiguous few-shot setting',inconsistent_scoring_settings:'Inconsistent scoring settings',missing_scoring_field:'Missing scoring field',missing_scoring_setting:'Missing scoring setting',no_selected_score:'No selected score',missing_suite_data:'Missing suite data',incomplete_components:'Incomplete components',comparison_coverage:'Comparison coverage',sample_count_mismatch:'Sample-count mismatch',no_category_weight:'No category weight'}; +const diagnosticTitles={config_caveat:'Config caveat',no_config:'No config',not_used:'Not used',unknown_language:'Unknown language',invalid_sample_count:'Invalid sample count',strict_shot_setting:'Few-shot mismatch excluded',relaxed_shot_setting:'Few-shot mismatch allowed',ambiguous_shot_setting:'Ambiguous few-shot setting',inconsistent_scoring_settings:'Inconsistent scoring settings',missing_scoring_field:'Missing scoring field',missing_scoring_setting:'Missing scoring setting',no_selected_score:'No selected score',missing_suite_data:'Missing suite data',incomplete_components:'Incomplete components',comparison_coverage:'Comparison coverage',sample_count_mismatch:'Sample-count mismatch',no_category_weight:'No category weight'}; function diagnostic(code,model,evalName,rows,detail,effect='included',tasks=null){ const names=[...new Set(tasks??rows.map(r=>r.task))].sort(compareText); return {code,type:diagnosticTitles[code],model,eval:evalName,name:evalName||names[0]||model,tasks:names,measurement_ids:rows.map(measurementId).sort(compareText),effect,detail,variants:names.length?[{settings:detail,tasks:names}]:[]}; @@ -250,7 +250,7 @@ function reportDiagnostics(audits,config,included,comparison=false,matchingMode= const used=new Set([...included.values()].flat().map(r=>r.eval)); for(const e of catalogue.evals)if(e.warning&&used.has(e.name))out.push(diagnostic('config_caveat','Selected comparison',e.name,[],e.warning)); for(const [model,rows] of audits){ - const scope=scopeRows(rows,suite),accepted=included.get(model)||[],acceptedIds=new Set(accepted.map(measurementId)); + const scope=scopeRows(rows,suite),explainedShots=new Set(),accepted=included.get(model)||[],acceptedIds=new Set(accepted.map(measurementId)); for(const task of [...new Set(rows.filter(r=>!r.eval).map(r=>r.task))].sort(compareText))out.push(diagnostic('no_config',model,null,rows.filter(r=>r.task===task),'No eval configuration; excluded from scoring.','excluded')); for(const e of catalogue.evals){ const all=rows.filter(r=>r.eval===e.name),outside=all.filter(r=>(!e.select||matchTask(e.select,r.task))&&!inSuite(r,suite)),matching=all.filter(r=>inSuite(r,suite)),selected=matching.filter(r=>r.selected); @@ -269,13 +269,24 @@ function reportDiagnostics(audits,config,included,comparison=false,matchingMode= if(bad.length)add('invalid_sample_count',bad,'Invalid sample count; retained scores are not weighted by sample count.'); const protocol=protocolWarning(matching,e,catalogue,model); if(protocol)add('inconsistent_scoring_settings',selected,protocol.detail); - for(const task of [...new Set(matching.filter(r=>!e.select||matchTask(e.select,r.task)).map(r=>r.task))].sort(compareText)){ + const eligibleTasks=[...new Set(matching.filter(r=>!e.select||matchTask(e.select,r.task)).map(r=>r.task))].sort(compareText),shotGroups=new Map(),shotTasks=new Set(); + for(const task of eligibleTasks){ const rr=matching.filter(r=>r.task===task);if(rr.some(r=>r.selected))continue; + const shotRows=rr.filter(r=>r.metric===e.metric&&r.filter===e.metric_filter&&Number(r.n_shot)!==e.shots); + if(matchingMode==='strict'&&'shots'in e&&shotRows.length){ + shotTasks.add(task);explainedShots.add(JSON.stringify([e.name,task])); + for(const r of shotRows){const actual=Number(r.n_shot);if(!shotGroups.has(actual))shotGroups.set(actual,[]);shotGroups.get(actual).push(r);}continue; + } add(rr.some(r=>r.metric===e.metric)?'missing_scoring_setting':'missing_scoring_field',rr,'Excluded: expected '+e.metric+' / '+(e.metric_filter||'(empty)')+('shots'in e?' / '+e.shots+' shots':'')+'.','excluded'); } - if(!selected.length)add('no_selected_score',matching,'No score matches the configured metric, filter, shots and selection. Excluded.','excluded'); + for(const [actual,rr] of [...shotGroups].sort((a,b)=>a[0]-b[0])){ + const count=new Set(rr.map(r=>r.task)).size; + out.push({...diagnostic('strict_shot_setting',model,e.name,rr,`${count} ${count===1?'task uses':'tasks use'} ${actual} shots; expected ${e.shots}. Excluded under strict matching; remaining weights are redistributed.`,'excluded'),expected_shots:e.shots,actual_shots:actual}); + } + if(!selected.length&&!(shotTasks.size&&eligibleTasks.every(t=>shotTasks.has(t))))add('no_selected_score',matching,'No score matches the configured metric, filter, shots and selection. Excluded.','excluded'); } - for(const name of [...new Set(scope.missing.map(r=>r.eval))])out.push(diagnostic('missing_suite_data',model,name,[],'Required results are missing from '+suite.name+'. Excluded; remaining weights are redistributed.','excluded',scope.missing.filter(r=>r.eval===name&&r.task).map(r=>r.task))); + const unexplainedMissing=scope.missing.filter(r=>!explainedShots.has(JSON.stringify([r.eval,r.task]))); + for(const name of [...new Set(unexplainedMissing.map(r=>r.eval))])out.push(diagnostic('missing_suite_data',model,name,[],'Required results are missing from '+suite.name+'. Excluded; remaining weights are redistributed.','excluded',unexplainedMissing.filter(r=>r.eval===name&&r.task).map(r=>r.task))); const component=componentCoverage(scope.rows,scheme); for(const name of [...new Set(component.excluded.map(r=>r.eval))])out.push(diagnostic('incomplete_components',model,name,component.excluded.filter(r=>r.eval===name),component.warnings.filter(w=>w.eval===name).map(w=>w.detail).join(' '),'excluded')); if(comparison)for(const name of [...new Set(component.rows.filter(r=>!acceptedIds.has(measurementId(r))).map(r=>r.eval))])out.push(diagnostic('comparison_coverage',model,name,component.rows.filter(r=>r.eval===name&&!acceptedIds.has(measurementId(r))),'Unmatched results are excluded from both scores; weights use shared data only.','excluded')); diff --git a/app/app.js b/app/app.js index 53e39f4..08422f0 100644 --- a/app/app.js +++ b/app/app.js @@ -167,7 +167,7 @@ function start(){ $('filterStatus').textContent=new Set(filtered.map(r=>r.task)).size+' of '+new Set(all.map(r=>r.task)).size+' task names · '+filtered.length+' metric rows · all loaded real exports; comparison exclusions appear in Warnings'; let html='

Eval configuration

Category, language, and scoring field for every eval. Expand an eval to inspect its variants.

EvalCategoryScoring optionsLanguagesNormalization
'; catalogueGroups=new Map(); - html+=groups.map(f=>{const allRows=all.filter(r=>r.eval===f.name),missing=warnings.filter(w=>w.eval===f.name&&['Missing scoring field','Missing scoring setting'].includes(w.type)),inconsistent=warnings.filter(w=>w.eval===f.name&&w.type==='Inconsistent scoring settings');catalogueGroups.set(f.name,{f,missing});const langs=[...new Set(f.tasks.flatMap(t=>languages(t.rows[0]).map(languageLabel)))].sort();return '
'+esc(f.name)+' ('+f.tasks.length+')'+esc(f.category)+''+scoringOptions(f,allRows)+(missing.length?''+missing.length+' missing scoring field/settings':'')+(inconsistent.length?'Inconsistent scoring settings':'')+''+esc(langs.length>4?langs.slice(0,3).join(', ')+' + '+(langs.length-3)+' more':langs.join(', '))+''+esc(normalizationLabel(f))+(f.warning?'Config warning':'')+''+selectionInfo(f)+normalizationInfo(f)+aggregationInfo(f)+inconsistent.map(w=>'

'+esc(w.model+': '+w.detail)+'

').join('')+'
'+(groups.length===1?catalogueTasks(f,missing):'')+'
'+'
';}).join(''); + html+=groups.map(f=>{const allRows=all.filter(r=>r.eval===f.name),missing=warnings.filter(w=>w.eval===f.name&&['Missing scoring field','Missing scoring setting','Few-shot mismatch excluded'].includes(w.type)),inconsistent=warnings.filter(w=>w.eval===f.name&&w.type==='Inconsistent scoring settings'),missingTasks=new Set(missing.flatMap(w=>w.tasks)).size;catalogueGroups.set(f.name,{f,missing});const langs=[...new Set(f.tasks.flatMap(t=>languages(t.rows[0]).map(languageLabel)))].sort();return '
'+esc(f.name)+' ('+f.tasks.length+')'+esc(f.category)+''+scoringOptions(f,allRows)+(missing.length?''+missingTasks+' tasks missing configured field/settings':'')+(inconsistent.length?'Inconsistent scoring settings':'')+''+esc(langs.length>4?langs.slice(0,3).join(', ')+' + '+(langs.length-3)+' more':langs.join(', '))+''+esc(normalizationLabel(f))+(f.warning?'Config warning':'')+''+selectionInfo(f)+normalizationInfo(f)+aggregationInfo(f)+inconsistent.map(w=>'

'+esc(w.model+': '+w.detail)+'

').join('')+'
'+(groups.length===1?catalogueTasks(f,missing):'')+'
'+'
';}).join(''); if(!groups.length)html+='

No evals match the filters.

'; html+='
Scoring assumptions and source files
    '+(scheme.notes||[]).map(n=>'
  1. '+esc(n)+'
  2. ').join('')+'

Source row audit · Language assignments · Analysis JSON

'+esc(DATA.source)+' · SHA-256 '+esc(DATA.sha256)+'

'; const configControls='
Global eval catalogue

The catalogue defines matching, categories, scoring fields, normalization, and language assignments for every known eval. It does not require models to run all those evals.

Normalization: each eval lists its baseline, formula, and sources below. Chance correction and component aggregation affect calculated scores; individual raw scores stay unchanged. acc_norm is length-normalized answer scoring, not chance correction.

Catalogue export includes every eval and language in one portable YAML file.

Active catalogue: '+esc(catalogue.name)+' · '+catalogue.languages.reduce((n,g)=>n+g.tasks.length,0)+' explicit task language assignments.

Weighting profile and optional eval set

Weights apply independently of the eval set. Any available compares shared recognized measurements and warns on differences. A named set requires all known catalogue tasks for each listed eval, except excluded languages. It can override metric, metric_filter, and shots for a whole eval. Missing requirements warn; extra measurements are excluded. An incomplete named-set score uses the shared subset with redistributed weights.

Active eval set: '+esc(suite.name)+(suite.exclude?.length?' · Excluded evals: '+suite.exclude.map(esc).join(', '):'')+(suite.exclude_languages?.length?' · Excluded languages across the set: '+suite.exclude_languages.map(esc).join(', '):'')+(suite.evals?.some(e=>e.exclude_languages?.length)?' · Per-eval language exclusions: '+suite.evals.filter(e=>e.exclude_languages?.length).map(e=>esc(e.name)+': '+e.exclude_languages.map(esc).join(', ')).join('; '):'')+'.

All imports are temporary and stay in your browser. Exports save the three inputs separately. Weight exports include your edits and active score calculation.

'; diff --git a/docs/configuration.md b/docs/configuration.md index a590bfc..acbdc42 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -75,6 +75,8 @@ For an explicitly pinned task inventory, a fixed entry may still use `variants: ## Strict and relaxed matching +Strict shot-mismatch warnings group tasks by eval, model, and expected/actual shot-count pair, with a count and expandable task list. They replace duplicate missing-setting/coverage warnings for those same tasks; missing requirements still count toward incomplete coverage. + The dashboard starts with **Strict matching**. Expected settings are resolved in this order: catalogue defaults, then optional whole-eval set overrides. Strict matching requires the configured metric, metric filter, and (when specified) shot count. Shared comparisons also require the same harness and backend. **Relaxed — allow few-shot differences** may select a different shot count for each concrete task. It prefers the expected count; otherwise it uses the uniquely closest available count, independently for each model and task with the same metric/filter/harness/backend. Equally close alternatives are ambiguous and excluded with a warning. Scores never influence that choice. Without a configured shot expectation, shot counts still have to match across models. diff --git a/docs/python-api.md b/docs/python-api.md index 0a91819..19e4c21 100644 --- a/docs/python-api.md +++ b/docs/python-api.md @@ -144,6 +144,7 @@ The diagnostic contract is `code`, `model`, `eval`, `tasks`, `measurement_ids`, | `no_config`, `not_used` | Unknown eval or data outside the selected set; excluded. | | `missing_scoring_field`, `missing_scoring_setting`, `no_selected_score` | Required metric/protocol unavailable; no alternate substitution. | | `missing_suite_data` | A named set has missing requirements; available shared results are reweighted. | +| `strict_shot_setting` | Tasks excluded for a shot mismatch, grouped by eval/model/expected/actual count. Includes expected/actual counts and all affected task/measurement IDs. | | `relaxed_shot_setting` | A differing shot count is included under relaxed matching; records expected/actual counts. | | `ambiguous_shot_setting` | Equally close alternative shot counts; excluded. | | `incomplete_components` | A language/protocol group is incomplete or incompatible; the whole group is excluded. | @@ -153,6 +154,8 @@ The diagnostic contract is `code`, `model`, `eval`, `tasks`, `measurement_ids`, | `invalid_sample_count`, `sample_count_mismatch` | Sample-count metadata needs review; it does not determine weights. | | `no_category_weight` | An included category has no profile weight and contributes zero. | +Strict shot mismatches produce one diagnostic per eval, model, and expected/actual shot-count pair. Covered tasks do not also generate `missing_scoring_setting`, `no_selected_score`, or `missing_suite_data` diagnostics for the same cause. Coverage still records those unsatisfied requirements and remains incomplete. Genuinely absent tasks and other missing settings retain their diagnostics. + Unused catalogue entries do not warn merely because no data exists for them. Declare an expected eval set when absence should warn. Alternate metrics remain auditable without creating warnings when the selected metric is present. ## Command line diff --git a/quickdash/analysis.py b/quickdash/analysis.py index e7b8212..739bc27 100644 --- a/quickdash/analysis.py +++ b/quickdash/analysis.py @@ -425,6 +425,7 @@ def protocol_inconsistent(rows): unknown_language="Unknown language", invalid_sample_count="Invalid sample count", inconsistent_scoring_settings="Inconsistent scoring settings", + strict_shot_setting="Few-shot mismatch excluded", relaxed_shot_setting="Few-shot mismatch allowed", ambiguous_shot_setting="Ambiguous few-shot setting", missing_scoring_field="Missing scoring field", @@ -474,6 +475,7 @@ def report_diagnostics( ) for model, rows in audits.items(): scope = scope_rows(rows, suite) + explained_shots = set() accepted = {measurement_id(r) for r in included.get(model, [])} for task in sorted({r["task"] for r in rows if not r["eval"]}): out.append( @@ -578,17 +580,30 @@ def add(code, rr, detail, effect="included"): selected, "Selected variants use inconsistent scoring settings; complete protocols remain included.", ) - for task in sorted( - { - r["task"] - for r in matching - if "select" not in e - or match_task(e["select"], r["task"]) is not None - } - ): + eligible_tasks = { + r["task"] + for r in matching + if "select" not in e or match_task(e["select"], r["task"]) is not None + } + shot_groups = {} + shot_tasks = set() + for task in sorted(eligible_tasks): rr = [r for r in matching if r["task"] == task] if any(r["selected"] for r in rr): continue + shot_rows = [ + r + for r in rr + if r["metric"] == e["metric"] + and r["filter"] == e["metric_filter"] + and int(r["n_shot"]) != e.get("shots") + ] + if matching_mode == "strict" and "shots" in e and shot_rows: + shot_tasks.add(task) + explained_shots.add((e["name"], task)) + for r in shot_rows: + shot_groups.setdefault(int(r["n_shot"]), []).append(r) + continue add( "missing_scoring_setting" if any(r["metric"] == e["metric"] for r in rr) @@ -602,14 +617,32 @@ def add(code, rr, detail, effect="included"): + ".", "excluded", ) - if not selected: + for actual, rr in sorted(shot_groups.items()): + count = len({r["task"] for r in rr}) + noun = "task uses" if count == 1 else "tasks use" + item = diagnostic( + "strict_shot_setting", + model, + e["name"], + rr, + f"{count} {noun} {actual} shots; expected {e['shots']}. Excluded under strict matching; remaining weights are redistributed.", + "excluded", + ) + item.update(expected_shots=e["shots"], actual_shots=actual) + out.append(item) + if not selected and not (shot_tasks and eligible_tasks <= shot_tasks): add( "no_selected_score", matching, "No score matches the configured metric, filter, shots and selection. Excluded.", "excluded", ) - for name in dict.fromkeys(r["eval"] for r in scope["missing"]): + unexplained_missing = [ + r + for r in scope["missing"] + if (r["eval"], r.get("task")) not in explained_shots + ] + for name in dict.fromkeys(r["eval"] for r in unexplained_missing): out.append( diagnostic( "missing_suite_data", @@ -620,7 +653,7 @@ def add(code, rr, detail, effect="included"): "excluded", [ r["task"] - for r in scope["missing"] + for r in unexplained_missing if r["eval"] == name and "task" in r ], ) diff --git a/tests/test_engines.py b/tests/test_engines.py index e6a8659..55de7b4 100644 --- a/tests/test_engines.py +++ b/tests/test_engines.py @@ -361,6 +361,112 @@ def test_clipping_defaults_to_true(self): self.assertNotIn("error", result) self.assertAlmostEqual(result["value"]["models"][0]["score"], score) + def test_strict_fewshot_warnings_are_grouped_without_duplicate_coverage_warnings( + self, + ): + c = fixture() + c["catalogue"]["evals"][0]["shots"] = 5 + rr = [row(t, n_shot="0") for t in ["e_en", "e_fr"]] + cases = [] + for mode in ["available", "fixed"]: + cfg = deepcopy(c) + if mode == "fixed": + cfg["suite"] = dict( + version=1, name="Required", mode=mode, evals=[dict(name="E")] + ) + cases.append(dict(config=cfg, rows=rr)) + for i, results in enumerate(self.both(cases)): + for result in results: + report = result["value"] + self.assertEqual( + [d["code"] for d in report["diagnostics"]], ["strict_shot_setting"] + ) + d = report["diagnostics"][0] + self.assertEqual( + ( + d["eval"], + d["model"], + d["expected_shots"], + d["actual_shots"], + d["effect"], + ), + ("E", "A", 5, 0, "excluded"), + ) + self.assertEqual(d["tasks"], ["e_en", "e_fr"]) + self.assertEqual(len(d["measurement_ids"]), 2) + self.assertIsNone(report["models"][0]["score"]) + self.assertTrue( + all( + not m["included"] and m["effective_weight"] == 0 + for m in report["models"][0]["measurements"] + ) + ) + if i == 1: + self.assertEqual(len(report["coverage"][0]["missing"]), 2) + # Each actual setting has its own group; another protocol doesn't inflate the task count. + many = rr + [row("e_en", n_shot="0", backend="other"), row("e_fr", n_shot="3")] + for result in self.both([dict(config=c, rows=many)])[0]: + ds = result["value"]["diagnostics"] + self.assertEqual([d["actual_shots"] for d in ds], [0, 3]) + self.assertEqual([len(d["tasks"]) for d in ds], [2, 1]) + # Exact selected data makes alternative shots harmless for that task. + for result in self.both([dict(config=c, rows=rr + [row(n_shot="5")])])[0]: + ds = result["value"]["diagnostics"] + self.assertEqual(len(ds), 1) + self.assertEqual(ds[0]["tasks"], ["e_fr"]) + self.assertEqual(result["value"]["models"][0]["score"], 50) + # Preserve genuinely absent requirements and other causes (wrong metric/filter). + cfg = deepcopy(c) + cfg["catalogue"]["languages"][0]["tasks"] += [ + "e_missing", + "e_wrong_metric", + "e_wrong_filter", + ] + cfg["suite"] = dict( + version=1, name="Required", mode="fixed", evals=[dict(name="E", shots=10)] + ) + mixed = rr + [ + row("e_wrong_metric", metric="other"), + row("e_wrong_filter", filter="other"), + ] + for result in self.both([dict(config=cfg, rows=mixed)])[0]: + ds = result["value"]["diagnostics"] + shot = next(d for d in ds if d["code"] == "strict_shot_setting") + self.assertEqual(shot["expected_shots"], 10) + self.assertEqual(shot["tasks"], ["e_en", "e_fr"]) + missing = next(d for d in ds if d["code"] == "missing_suite_data") + self.assertEqual( + missing["tasks"], ["e_missing", "e_wrong_filter", "e_wrong_metric"] + ) + self.assertIn("missing_scoring_field", {d["code"] for d in ds}) + self.assertIn("missing_scoring_setting", {d["code"] for d in ds}) + # Excluded languages never enter the mismatch count. + cfg["suite"]["evals"][0]["exclude_languages"] = ["fra_Latn"] + for result in self.both([dict(config=cfg, rows=rr)])[0]: + ds = result["value"]["diagnostics"] + shot = next(d for d in ds if d["code"] == "strict_shot_setting") + self.assertEqual(shot["tasks"], ["e_en"]) + self.assertIn("not_used", {d["code"] for d in ds}) + # Counts are grouped independently for each model. + for result in self.both([dict(config=c, rows=paired(rr))])[0]: + ds = result["value"]["diagnostics"] + self.assertEqual( + [d["code"] for d in ds], ["strict_shot_setting", "strict_shot_setting"] + ) + self.assertEqual([d["model"] for d in ds], ["A", "B"]) + self.assertEqual( + [d["tasks"] for d in ds], [["e_en", "e_fr"], ["e_en", "e_fr"]] + ) + # Relaxed mode retains its own warning. + for result in self.both([dict(config=c, rows=paired(rr), matching="relaxed")])[ + 0 + ]: + ds = result["value"]["diagnostics"] + self.assertEqual( + [d["code"] for d in ds], + ["relaxed_shot_setting", "relaxed_shot_setting"], + ) + def test_relaxed_fewshot_selection_and_warnings(self): c = fixture() c["catalogue"]["evals"][0]["shots"] = 5 diff --git a/tests/test_public_browser.mjs b/tests/test_public_browser.mjs index 4665fe9..310792b 100644 --- a/tests/test_public_browser.mjs +++ b/tests/test_public_browser.mjs @@ -215,6 +215,20 @@ try{ assert.equal(await evaluate("document.querySelector('#cards').hidden"),false); const strictSampleCards=await evaluate("document.querySelector('#cards').textContent"); + // Strict shot mismatches collapse per eval, including duplicate missing-set warnings. + await change('#suitePreset','1');await click('[data-view=warnings]'); + const arcWarnings=await evaluate("[...document.querySelectorAll('#view tbody tr')].filter(r=>r.cells[1]?.textContent==='ARC Challenge').map(r=>r.textContent)"); + assert.equal(arcWarnings.length,1); + assert.match(arcWarnings[0],/Few-shot mismatch excluded.*\d+ tasks use 0 shots; expected 10/s); + assert.match(arcWarnings[0],/arc_challenge_mt_cs/); + assert.match(await evaluate("document.querySelector('#coverage').textContent"),/INCOMPLETE/); + await click('[data-view=config]'); + await evaluate("document.querySelector('.catalogue-eval[data-eval=\"ARC Challenge\"]').open=true"); + await new Promise(r=>setTimeout(r,50)); + assert.ok(await evaluate("document.querySelectorAll('.catalogue-eval[data-eval=\"ARC Challenge\"] .has-missing-field').length>0")); + await change('#suitePreset','0'); + assert.equal(await evaluate("document.querySelector('#cards').textContent"),strictSampleCards); + await change('#matching','relaxed'); assert.equal(await evaluate("document.querySelector('#error').textContent"),''); assert.match(await evaluate("document.querySelector('#matchingNotice').textContent"),/INCONSISTENT EVALUATION SETTINGS/);