From bf0fff66b8fac34739eaa652f0470f8d1cc33609 Mon Sep 17 00:00:00 2001 From: Venkata Nainala Date: Mon, 28 Sep 2026 17:53:01 +0100 Subject: [PATCH 1/3] feat: add assignment validation against nmrshiftdb2 quickcheck Adds nmr-cli validate-assignments and POST /validate/assignments. The command checks 1H/13C assignments of a structure against nmrshiftdb2 quickcheck in one request and returns a quality report per nucleus (mark, per-atom deviation, HOSE spheres) plus an assignment check (per-row status, swap suggestions, equivalence, missing signals, referencing offset) and a combined verdict. --- app/main.py | 11 +- app/routers/validate.py | 290 +++++++ app/scripts/nmr-cli/package.json | 1 + app/scripts/nmr-cli/src/index.ts | 17 + app/scripts/nmr-cli/src/validation/index.ts | 47 ++ app/scripts/nmr-cli/src/validation/molfile.ts | 134 +++ .../nmr-cli/src/validation/quickcheck.ts | 49 ++ .../nmr-cli/src/validation/solvents.ts | 63 ++ .../nmr-cli/src/validation/topology.ts | 45 + app/scripts/nmr-cli/src/validation/types.ts | 165 ++++ .../nmr-cli/src/validation/validate.ts | 767 ++++++++++++++++++ .../nmr-cli/test/fixtures/nicotine.mol | 30 + .../test/fixtures/quickcheck-responses.json | 704 ++++++++++++++++ .../test/fixtures/trimethoxybenzaldehyde.mol | 46 ++ app/scripts/nmr-cli/test/validation.test.cjs | 235 ++++++ docs/src/modules/validation.md | 79 +- tests/test_validate.py | 147 ++++ 17 files changed, 2821 insertions(+), 9 deletions(-) create mode 100644 app/routers/validate.py create mode 100644 app/scripts/nmr-cli/src/validation/index.ts create mode 100644 app/scripts/nmr-cli/src/validation/molfile.ts create mode 100644 app/scripts/nmr-cli/src/validation/quickcheck.ts create mode 100644 app/scripts/nmr-cli/src/validation/solvents.ts create mode 100644 app/scripts/nmr-cli/src/validation/topology.ts create mode 100644 app/scripts/nmr-cli/src/validation/types.ts create mode 100644 app/scripts/nmr-cli/src/validation/validate.ts create mode 100644 app/scripts/nmr-cli/test/fixtures/nicotine.mol create mode 100644 app/scripts/nmr-cli/test/fixtures/quickcheck-responses.json create mode 100644 app/scripts/nmr-cli/test/fixtures/trimethoxybenzaldehyde.mol create mode 100644 app/scripts/nmr-cli/test/validation.test.cjs create mode 100644 tests/test_validate.py diff --git a/app/main.py b/app/main.py index de9406d..2515780 100644 --- a/app/main.py +++ b/app/main.py @@ -5,7 +5,7 @@ from .routers import registration from .routers import chem -from .routers import spectra, converter, predict +from .routers import spectra, converter, predict, validate from fastapi.middleware.cors import CORSMiddleware from app.core import config, tasks @@ -28,6 +28,7 @@ | **Spectra** | Parse NMR spectra from files or URLs | | **Converter** | Convert NMR raw data to NMRium JSON | | **Predict** | Predict NMR spectra using nmrdb.org or nmrshift engines | +| **Validate** | Check 1H/13C assignments against nmrshiftdb2 quickcheck predictions | | **Registration** | Register and query molecules via lwreg | ### Links @@ -61,6 +62,13 @@ "**nmrdb.org** or **nmrshift** prediction engines." ), }, + { + "name": "validate", + "description": ( + "Validate 1H/13C assignments of a structure against **nmrshiftdb2 quickcheck** " + "predictions: quality mark per nucleus, per-atom deviations and assignment checks." + ), + }, { "name": "registration", "description": "Register, query, and retrieve molecules using the lwreg registration system.", @@ -89,6 +97,7 @@ app.include_router(spectra.router) app.include_router(converter.router) app.include_router(predict.router) +app.include_router(validate.router) if hasattr(app, "add_event_handler"): app.add_event_handler("startup", tasks.create_start_app_handler(app)) diff --git a/app/routers/validate.py b/app/routers/validate.py new file mode 100644 index 0000000..61cbfac --- /dev/null +++ b/app/routers/validate.py @@ -0,0 +1,290 @@ +from fastapi import APIRouter, HTTPException, status +from app.schemas import HealthCheck +from pydantic import BaseModel, Field +from typing import Any, Dict, List, Literal, Optional +from collections import OrderedDict +import hashlib +import json +import subprocess +import threading +import time + +router = APIRouter( + prefix="/validate", + tags=["validate"], + dependencies=[], + responses={ + 404: {"description": "Not Found"}, + 408: {"description": "Validation timed out"}, + 422: {"description": "Invalid assignment set or structure"}, + 500: {"description": "Docker or nmr-converter container not available"}, + 503: {"description": "nmrshiftdb2 quickcheck is unavailable"}, + }, +) + +# Container name for nmr-cli (from docker-compose.yml) +NMR_CLI_CONTAINER = "nmr-converter" +VALIDATION_TIMEOUT_SECONDS = 120 + +# nmr-cli validate-assignments exit codes +EXIT_INVALID_INPUT = 2 +EXIT_QUICKCHECK_UNAVAILABLE = 3 + +CACHE_TTL_SECONDS = 24 * 60 * 60 +CACHE_MAX_ENTRIES = 512 + + +# ============================================================================ +# REQUEST MODELS +# ============================================================================ + + +Nucleus = Literal["13C", "1H"] + + +class Structure(BaseModel): + molfile: str = Field(..., min_length=1, + description="V2000 MOL block; atom numbers in assignments refer to it") + source: Optional[str] = Field( + default=None, description="Origin of the structure, e.g. nmrium, mnova, nmredata, manual") + + +class Conditions(BaseModel): + solvent: Optional[str] = Field( + default=None, description="Solvent name or abbreviation, e.g. CDCl3 or DMSO-d6") + temperature_k: Optional[float] = Field(default=None, description="Temperature in K") + frequency_mhz: Optional[Dict[Nucleus, float]] = Field( + default=None, description="Spectrometer frequency per nucleus") + + +class Assignment(BaseModel): + nucleus: Nucleus + atoms: List[int] = Field( + ..., + description=( + "1-based molfile atom indices. For 1H, an index may be the carrying heavy atom " + "or an explicit H atom. An empty list marks an observed but unassigned signal." + ), + ) + label: Optional[str] = Field(default=None, description="Author label, e.g. C-2, C-6 or H-5'a") + shift: float = Field(..., description="Observed chemical shift in ppm") + multiplicity: Optional[str] = None + n_h: Optional[int] = Field(default=None, ge=0, description="Integral in protons") + diastereotopic: Optional[Literal["a", "b"]] = None + + +class UnassignedPeak(BaseModel): + nucleus: Nucleus + shift: float + kind: Literal["solvent", "impurity", "unknown"] = "unknown" + + +class Tolerance(BaseModel): + ok: float = Field(..., gt=0) + fail: float = Field(..., gt=0) + + +class ValidationOptions(BaseModel): + fallback_tolerances: Optional[Dict[Nucleus, Tolerance]] = None + + +class AssignmentValidationRequest(BaseModel): + structure: Structure + conditions: Optional[Conditions] = None + assignments: List[Assignment] = Field(..., min_length=1) + unassigned_peaks: Optional[List[UnassignedPeak]] = None + options: Optional[ValidationOptions] = None + + +# ============================================================================ +# CACHE +# ============================================================================ + + +class ReportCache: + """Small in-process TTL cache; the quickcheck servlet is a shared public service.""" + + def __init__(self, ttl: int, max_entries: int): + self.ttl = ttl + self.max_entries = max_entries + self._entries: "OrderedDict[str, tuple[float, dict]]" = OrderedDict() + self._lock = threading.Lock() + + def get(self, key: str) -> Optional[dict]: + with self._lock: + entry = self._entries.get(key) + if entry is None: + return None + stored_at, report = entry + if time.time() - stored_at > self.ttl: + del self._entries[key] + return None + self._entries.move_to_end(key) + return report + + def put(self, key: str, report: dict) -> None: + with self._lock: + self._entries[key] = (time.time(), report) + self._entries.move_to_end(key) + while len(self._entries) > self.max_entries: + self._entries.popitem(last=False) + + def clear(self) -> None: + with self._lock: + self._entries.clear() + + +report_cache = ReportCache(CACHE_TTL_SECONDS, CACHE_MAX_ENTRIES) + + +def request_hash(payload: dict) -> str: + canonical = json.dumps(payload, sort_keys=True, separators=(",", ":")) + return hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +# ============================================================================ +# CLI +# ============================================================================ + + +def parse_cli_error(stderr: str) -> Dict[str, Any]: + start = stderr.find("{") + if start >= 0: + try: + return json.loads(stderr[start:]) + except json.JSONDecodeError: + pass + return {"message": stderr or "No error output from CLI"} + + +def run_validate_command(payload: dict) -> dict: + """Pipe the assignment set to `nmr-cli validate-assignments` and return the report.""" + try: + result = subprocess.run( + ["docker", "exec", "-i", NMR_CLI_CONTAINER, "nmr-cli", "validate-assignments"], + input=json.dumps(payload).encode("utf-8"), + capture_output=True, + timeout=VALIDATION_TIMEOUT_SECONDS, + ) + except subprocess.TimeoutExpired: + raise HTTPException( + status_code=408, + detail={"message": f"Validation timed out after {VALIDATION_TIMEOUT_SECONDS}s"}, + ) + except FileNotFoundError: + raise HTTPException( + status_code=500, + detail={ + "message": "Docker not found or nmr-converter container is not running", + "hint": "Run: docker compose up -d", + }, + ) + + stdout = result.stdout.decode("utf-8", errors="replace").strip() + stderr = result.stderr.decode("utf-8", errors="replace").strip() + + if result.returncode == EXIT_INVALID_INPUT: + raise HTTPException(status_code=422, detail=parse_cli_error(stderr)) + if result.returncode == EXIT_QUICKCHECK_UNAVAILABLE: + raise HTTPException(status_code=503, detail=parse_cli_error(stderr)) + if result.returncode != 0: + raise HTTPException( + status_code=500, + detail={ + "message": "NMR CLI command failed", + "exit_code": result.returncode, + "error": parse_cli_error(stderr), + }, + ) + + json_start = stdout.find("{") + try: + return json.loads(stdout[json_start:] if json_start >= 0 else stdout) + except json.JSONDecodeError as e: + raise HTTPException( + status_code=500, + detail={ + "message": "NMR CLI returned invalid JSON", + "parse_error": str(e), + "stdout_preview": stdout[:500], + }, + ) + + +# ============================================================================ +# HEALTH CHECK +# ============================================================================ + + +@router.get("/", include_in_schema=False) +@router.get( + "/health", + tags=["healthcheck"], + summary="Perform a Health Check on Validate Module", + response_description="Return HTTP Status Code 200 (OK)", + status_code=status.HTTP_200_OK, + include_in_schema=False, + response_model=HealthCheck, +) +def get_health() -> HealthCheck: + """Health check endpoint""" + return HealthCheck(status="OK") + + +# ============================================================================ +# ENDPOINTS +# ============================================================================ + + +@router.post( + "/assignments", + summary="Validate 1H/13C assignments against nmrshiftdb2 quickcheck", + description=( + "Submit a structure with assigned 1H and 13C shifts. Each nucleus is checked " + "against nmrshiftdb2 HOSE-code predictions in a single quickcheck request.\n\n" + "The response has two layers:\n\n" + "| Layer | Question it answers |\n" + "|-------|--------------------|\n" + "| `reports` | Does the shift list fit the structure? (nmrshiftdb2 quality report: mark 1–10, per-atom deviation, HOSE spheres) |\n" + "| `assignment_check` | Are the shifts on the right atoms? (per-assignment status, swap suggestions, equivalence, missing signals, referencing offset) |\n\n" + "`verdict` combines both, 13C first. Predictions with fewer than 4 HOSE spheres can " + "at most lead to `review`, never `fail`. The mark formula approximates nmrshiftdb2 " + "and is flagged with `mark_is_approximate`." + ), + response_description="Validation report", + status_code=status.HTTP_200_OK, + responses={ + 200: {"description": "Validation report"}, + 408: {"description": "Validation timed out"}, + 422: {"description": "Invalid assignment set or structure"}, + 503: {"description": "nmrshiftdb2 quickcheck is unavailable, retry later"}, + }, +) +def validate_assignments(request: AssignmentValidationRequest) -> Dict[str, Any]: + """ + ## Validate NMR assignments + + ### Example + ```json + { + "structure": {"molfile": "\\n Mnova...\\nM END", "source": "mnova"}, + "conditions": {"solvent": "CDCl3"}, + "assignments": [ + {"nucleus": "13C", "atoms": [1, 3], "label": "C-2, C-6", "shift": 106.65}, + {"nucleus": "13C", "atoms": [5], "label": "C-4", "shift": 143.52}, + {"nucleus": "1H", "atoms": [1, 3], "label": "H-2, H-6", "shift": 7.12, "n_h": 2} + ], + "unassigned_peaks": [{"nucleus": "13C", "shift": 77.16, "kind": "solvent"}] + } + ``` + """ + payload = request.model_dump(exclude_none=True) + key = request_hash(payload) + + cached = report_cache.get(key) + if cached is not None: + return {**cached, "cached": True} + + report = run_validate_command(payload) + report_cache.put(key, report) + return {**report, "cached": False} diff --git a/app/scripts/nmr-cli/package.json b/app/scripts/nmr-cli/package.json index 9180a20..d201326 100644 --- a/app/scripts/nmr-cli/package.json +++ b/app/scripts/nmr-cli/package.json @@ -5,6 +5,7 @@ "main": "./build/index.js", "scripts": { "build": "tsc", + "test": "tsc && node --test test/*.test.cjs", "start": "node build/index.js", "dev": "nmr-cli src/index.ts" }, diff --git a/app/scripts/nmr-cli/src/index.ts b/app/scripts/nmr-cli/src/index.ts index 7ca4b70..71216e2 100755 --- a/app/scripts/nmr-cli/src/index.ts +++ b/app/scripts/nmr-cli/src/index.ts @@ -7,6 +7,7 @@ import type { PeaksToNMRiumInput } from './peaks-to-nmrium' import { generateCorrelationData } from './correlation' import { hideBin } from 'yargs/helpers' import { parsePredictionCommand } from './prediction' +import { validateAssignmentsCommand } from './validation' import { readFileSync } from 'fs' import { IncludeData } from '@zakodium/nmrium-core' @@ -19,6 +20,7 @@ Commands: predict Predict spectrum from Mol peaks-to-nmrium Convert a peak list to NMRium object correlation Build correlation data from NMR spectra fetched from a URL + validate-assignments Check 1H/13C assignments against nmrshiftdb2 quickcheck (JSON on stdin) Options for 'correlation' command: -u, --url Spectra ZIP file URL @@ -93,6 +95,20 @@ Arguments for 'peaks-to-nmrium' command: } } +Arguments for 'validate-assignments' command: + Reads JSON from stdin with the following structure: + { + "structure": { "molfile": "...V2000..." }, + "conditions": { "solvent": "CDCl3" }, + "assignments": [ + { "nucleus": "13C", "atoms": [1, 3], "label": "C-2, C-6", "shift": 106.65 }, + { "nucleus": "1H", "atoms": [9], "label": "H-9a", "shift": 2.31, "n_h": 1 } + ], + "unassigned_peaks": [{ "nucleus": "13C", "shift": 77.16, "kind": "solvent" }] + } + Atom numbers are 1-based molfile indices; for 1H they may point at the carrying heavy atom. + Exit codes: 2 invalid input, 3 nmrshiftdb unavailable. + Examples: nmr-cli parse-spectra -u file-url -s // Process spectra files from a URL and capture an image for the spectra nmr-cli parse-spectra -dir directory-path -s // process a spectra files from a directory and capture an image for the spectra @@ -326,6 +342,7 @@ yargs(hideBin(process.argv)) .command(parsePredictionCommand) .command(peaksToNMRiumCommand) .command(correlationCommand) + .command(validateAssignmentsCommand) .showHelpOnFail(true) .help() .parse() diff --git a/app/scripts/nmr-cli/src/validation/index.ts b/app/scripts/nmr-cli/src/validation/index.ts new file mode 100644 index 0000000..7bed40d --- /dev/null +++ b/app/scripts/nmr-cli/src/validation/index.ts @@ -0,0 +1,47 @@ +import { readFileSync } from 'fs' +import type { CommandModule } from 'yargs' + +import { InvalidStructureError } from './molfile' +import { createQuickcheckClient, quickcheckUrl, QuickcheckUnavailableError } from './quickcheck' +import type { AssignmentSetInput } from './types' +import { InvalidInputError, validateAssignments } from './validate' + +/** Exit codes the API maps to HTTP statuses. */ +export const EXIT_INVALID_INPUT = 2 +export const EXIT_QUICKCHECK_UNAVAILABLE = 3 + +function fail(code: number, error: string, message: string): never { + console.error(JSON.stringify({ error, message })) + process.exit(code) +} + +export const validateAssignmentsCommand: CommandModule = { + command: ['validate-assignments', 'va'], + describe: 'Validate 1H/13C assignments against nmrshiftdb2 quickcheck (reads JSON from stdin)', + handler: async () => { + let input: AssignmentSetInput + try { + input = JSON.parse(readFileSync(0, 'utf-8')) + } catch (error) { + fail(EXIT_INVALID_INPUT, 'invalid_input', `Input is not valid JSON: ${String(error)}`) + } + + try { + const url = quickcheckUrl() + const report = await validateAssignments(input, { + quickcheck: createQuickcheckClient(url), + url, + }) + console.log(JSON.stringify(report)) + } catch (error) { + const message = error instanceof Error ? error.message : String(error) + if (error instanceof InvalidInputError || error instanceof InvalidStructureError) { + fail(EXIT_INVALID_INPUT, 'invalid_input', message) + } + if (error instanceof QuickcheckUnavailableError) { + fail(EXIT_QUICKCHECK_UNAVAILABLE, 'quickcheck_unavailable', message) + } + fail(1, 'validation_failed', message) + } + }, +} diff --git a/app/scripts/nmr-cli/src/validation/molfile.ts b/app/scripts/nmr-cli/src/validation/molfile.ts new file mode 100644 index 0000000..d7e1061 --- /dev/null +++ b/app/scripts/nmr-cli/src/validation/molfile.ts @@ -0,0 +1,134 @@ +export class InvalidStructureError extends Error {} + +export interface PreparedStructure { + /** V2000 molfile without explicit hydrogens, heavy atoms in original order. */ + molfile: string + heavyAtomCount: number + /** 1-based prepared index -> 1-based original index (position 0 unused). */ + toOriginal: number[] + /** 1-based original heavy-atom index -> 1-based prepared index. */ + fromOriginal: Map + /** 1-based original explicit-H index -> 1-based original parent index. */ + explicitHydrogenParents: Map + /** Element symbols by 1-based original index. */ + symbols: Map +} + +const RENUMBERED_PROPERTIES = ['M CHG', 'M ISO', 'M RAD'] + +/** + * Removes explicit hydrogens from a V2000 molfile without touching the order + * of the remaining atoms. + * + * The nmrshiftdb quickcheck servlet reports heavy atoms by their molfile + * position, but an explicit H shifts that numbering and makes predictions for + * later atoms fail, so every structure is sent H-free and mapped back. + */ +export function prepareStructure(molfile: string): PreparedStructure { + const lines = molfile.replace(/\r\n?/g, '\n').split('\n') + + if (lines.length < 4) { + throw new InvalidStructureError('Molfile is too short') + } + + const countsLine = lines[3] + if (countsLine.includes('V3000')) { + throw new InvalidStructureError('Only V2000 molfiles are supported') + } + + const atomCount = Number.parseInt(countsLine.slice(0, 3), 10) + const bondCount = Number.parseInt(countsLine.slice(3, 6), 10) + if (!Number.isFinite(atomCount) || !Number.isFinite(bondCount) || atomCount < 1) { + throw new InvalidStructureError('Molfile counts line is invalid') + } + + const atomLines = lines.slice(4, 4 + atomCount) + const bondLines = lines.slice(4 + atomCount, 4 + atomCount + bondCount) + if (atomLines.length !== atomCount || bondLines.length !== bondCount) { + throw new InvalidStructureError('Molfile atom or bond block is truncated') + } + + const symbols = new Map() + atomLines.forEach((line, index) => { + symbols.set(index + 1, line.slice(31, 34).trim()) + }) + + const isExplicitHydrogen = (index: number) => symbols.get(index) === 'H' + + const toOriginal: number[] = [0] + const fromOriginal = new Map() + for (let index = 1; index <= atomCount; index++) { + if (isExplicitHydrogen(index)) continue + fromOriginal.set(index, toOriginal.length) + toOriginal.push(index) + } + + const explicitHydrogenParents = new Map() + const keptBonds: string[] = [] + for (const line of bondLines) { + const first = Number.parseInt(line.slice(0, 3), 10) + const second = Number.parseInt(line.slice(3, 6), 10) + const firstIsH = isExplicitHydrogen(first) + const secondIsH = isExplicitHydrogen(second) + + if (firstIsH || secondIsH) { + if (firstIsH && !secondIsH) explicitHydrogenParents.set(first, second) + if (secondIsH && !firstIsH) explicitHydrogenParents.set(second, first) + continue + } + + const newFirst = fromOriginal.get(first) + const newSecond = fromOriginal.get(second) + if (newFirst === undefined || newSecond === undefined) { + throw new InvalidStructureError(`Bond references unknown atom: ${line.trim()}`) + } + keptBonds.push(pad3(newFirst) + pad3(newSecond) + line.slice(6)) + } + + const heavyAtomCount = toOriginal.length - 1 + const properties = lines + .slice(4 + atomCount + bondCount) + .filter((line) => RENUMBERED_PROPERTIES.some((prefix) => line.startsWith(prefix))) + .map((line) => renumberPropertyLine(line, fromOriginal)) + .filter((line): line is string => line !== null) + + const output = [ + ...lines.slice(0, 3), + pad3(heavyAtomCount) + pad3(keptBonds.length) + countsLine.slice(6), + ...toOriginal.slice(1).map((original) => atomLines[original - 1]), + ...keptBonds, + ...properties, + 'M END', + '', + ].join('\n') + + return { + molfile: output, + heavyAtomCount, + toOriginal, + fromOriginal, + explicitHydrogenParents, + symbols, + } +} + +function pad3(value: number): string { + return String(value).padStart(3, ' ') +} + +function renumberPropertyLine(line: string, fromOriginal: Map): string | null { + const prefix = line.slice(0, 6) + const entries = line.slice(9).trim().split(/\s+/).map(Number) + const pairs: [number, number][] = [] + for (let i = 0; i + 1 < entries.length; i += 2) { + const atom = fromOriginal.get(entries[i]) + if (atom !== undefined) pairs.push([atom, entries[i + 1]]) + } + if (pairs.length === 0) return null + + return ( + prefix + + pad3(pairs.length) + + pairs.map(([atom, value]) => ' ' + pad3(atom) + ' ' + pad3(value)).join('') + ) +} diff --git a/app/scripts/nmr-cli/src/validation/quickcheck.ts b/app/scripts/nmr-cli/src/validation/quickcheck.ts new file mode 100644 index 0000000..2c1ac2a --- /dev/null +++ b/app/scripts/nmr-cli/src/validation/quickcheck.ts @@ -0,0 +1,49 @@ +import https from 'https' +import axios from 'axios' + +import type { QuickcheckClient, QuickcheckInput, QuickcheckResult } from './types' + +export class QuickcheckUnavailableError extends Error {} + +const DEFAULT_TIMEOUT_MS = 60_000 + +export function quickcheckUrl(): string { + const url = process.env['NMR_PREDICTION_URL'] + if (!url) { + throw new Error('Environment variable NMR_PREDICTION_URL is not defined.') + } + return url +} + +/** + * Calls the nmrshiftdb2 quickcheck servlet once for all nuclei. + * + * Certificate verification is disabled to match the existing nmrshift engine; + * the servlet's certificate chain is not trusted by the container image. + */ +export function createQuickcheckClient( + url: string, + timeoutMs = DEFAULT_TIMEOUT_MS, +): QuickcheckClient { + const httpsAgent = new https.Agent({ rejectUnauthorized: false }) + + return async (molfile: string, inputs: QuickcheckInput[]): Promise => { + try { + const response = await axios.post<{ result: QuickcheckResult[] }>( + url, + { inputs, moltxt: molfile }, + { headers: { 'Content-Type': 'application/json' }, httpsAgent, timeout: timeoutMs }, + ) + if (!Array.isArray(response.data?.result)) { + throw new QuickcheckUnavailableError('nmrshiftdb returned an unexpected response') + } + return response.data.result + } catch (error) { + if (error instanceof QuickcheckUnavailableError) throw error + const message = axios.isAxiosError(error) + ? `nmrshiftdb quickcheck failed: ${error.response?.status ?? error.code ?? error.message}` + : `nmrshiftdb quickcheck failed: ${String(error)}` + throw new QuickcheckUnavailableError(message) + } + } +} diff --git a/app/scripts/nmr-cli/src/validation/solvents.ts b/app/scripts/nmr-cli/src/validation/solvents.ts new file mode 100644 index 0000000..74d0347 --- /dev/null +++ b/app/scripts/nmr-cli/src/validation/solvents.ts @@ -0,0 +1,63 @@ +import type { Nucleus } from './types' + +interface SolventInfo { + nmrshiftdb: string + residual: Record +} + +const SOLVENTS: { match: RegExp; info: SolventInfo }[] = [ + { + match: /cdcl3|chloroform/, + info: { nmrshiftdb: 'Chloroform-D1 (CDCl3)', residual: { '1H': [7.26], '13C': [77.16] } }, + }, + { + match: /dmso|dimethylsulfoxide|dimethylsulphoxide/, + info: { + nmrshiftdb: 'Dimethylsulphoxide-D6 (DMSO-D6, C2D6SO)', + residual: { '1H': [2.5], '13C': [39.52] }, + }, + }, + { + match: /cd3od|methanol/, + info: { nmrshiftdb: 'Methanol-D4 (CD3OD)', residual: { '1H': [3.31], '13C': [49.0] } }, + }, + { + match: /d2o|deuteriumoxide|deuterium oxide|water/, + info: { nmrshiftdb: 'Deuteriumoxide (D2O)', residual: { '1H': [4.79], '13C': [] } }, + }, + { + match: /acetone/, + info: { + nmrshiftdb: 'Acetone-D6 ((CD3)2CO)', + residual: { '1H': [2.05], '13C': [29.84, 206.26] }, + }, + }, + { + match: /ccl4|tetrachloro/, + info: { nmrshiftdb: 'TETRACHLORO-METHANE (CCl4)', residual: { '1H': [], '13C': [96.1] } }, + }, + { + match: /pyridin/, + info: { + nmrshiftdb: 'Pyridin-D5 (C5D5N)', + residual: { '1H': [8.74, 7.58, 7.22], '13C': [150.35, 135.91, 123.87] }, + }, + }, + { + match: /c6d6|benzene/, + info: { nmrshiftdb: 'Benzene-D6 (C6D6)', residual: { '1H': [7.16], '13C': [128.06] } }, + }, + { + match: /thf|tetrahydrofuran/, + info: { + nmrshiftdb: 'Tetrahydrofuran-D8 (THF-D8, C4D4O)', + residual: { '1H': [3.58, 1.72], '13C': [67.21, 25.31] }, + }, + }, +] + +export function resolveSolvent(solvent: string | undefined): SolventInfo { + const normalized = (solvent ?? '').toLowerCase().replace(/[\s_-]/g, '') + const found = SOLVENTS.find(({ match }) => match.test(normalized)) + return found?.info ?? { nmrshiftdb: 'Any', residual: { '1H': [], '13C': [] } } +} diff --git a/app/scripts/nmr-cli/src/validation/topology.ts b/app/scripts/nmr-cli/src/validation/topology.ts new file mode 100644 index 0000000..fb041dc --- /dev/null +++ b/app/scripts/nmr-cli/src/validation/topology.ts @@ -0,0 +1,45 @@ +import { Molecule } from 'openchemlib' + +import type { PreparedStructure } from './molfile' + +export interface Topology { + /** Implicit H count by 1-based original heavy-atom index. */ + hydrogenCounts: Map + /** Diastereotopic class ID by 1-based original heavy-atom index. */ + classes: Map + /** Servlet H atom number -> 1-based original index of the carrying heavy atom. */ + servletHydrogenOwners: Map +} + +/** + * Topology of the H-free structure sent to nmrshiftdb. + * + * openchemlib keeps the atom order of an H-free V2000 molfile, so its indices + * line up with the prepared numbering. The servlet appends implicit hydrogens + * after the heavy atoms, heavy atom by heavy atom, which is what + * `servletHydrogenOwners` reproduces. + */ +export function analyseTopology(structure: PreparedStructure): Topology { + const molecule = Molecule.fromMolfile(structure.molfile) + if (molecule.getAllAtoms() !== structure.heavyAtomCount) { + throw new Error('Unexpected atom count after parsing the H-free molfile') + } + + const diaIDs = molecule.getDiastereotopicAtomIDs() + const hydrogenCounts = new Map() + const classes = new Map() + const servletHydrogenOwners = new Map() + + let nextHydrogen = structure.heavyAtomCount + 1 + for (let index = 0; index < structure.heavyAtomCount; index++) { + const original = structure.toOriginal[index + 1] + const count = molecule.getImplicitHydrogens(index) + hydrogenCounts.set(original, count) + classes.set(original, diaIDs[index]) + for (let h = 0; h < count; h++) { + servletHydrogenOwners.set(nextHydrogen++, original) + } + } + + return { hydrogenCounts, classes, servletHydrogenOwners } +} diff --git a/app/scripts/nmr-cli/src/validation/types.ts b/app/scripts/nmr-cli/src/validation/types.ts new file mode 100644 index 0000000..9cd07bb --- /dev/null +++ b/app/scripts/nmr-cli/src/validation/types.ts @@ -0,0 +1,165 @@ +export type Nucleus = '13C' | '1H' + +export const NUCLEI: readonly Nucleus[] = ['13C', '1H'] + +export interface Tolerance { + ok: number + fail: number +} + +/** + * One observed signal. `atoms` are 1-based indices of the submitted molfile. + * For 1H, an index may point at the heavy atom carrying the protons + * (NMReDATA convention) or at an explicit H atom. An empty `atoms` list is a + * signal without assignment; it only contributes to the structure fit. + */ +export interface AssignmentInput { + nucleus: Nucleus + atoms: number[] + label?: string + shift: number + multiplicity?: string + n_h?: number + diastereotopic?: 'a' | 'b' +} + +export interface UnassignedPeakInput { + nucleus: Nucleus + shift: number + kind?: 'solvent' | 'impurity' | 'unknown' +} + +export interface AssignmentSetInput { + structure: { molfile: string; source?: string } + conditions?: { + solvent?: string + temperature_k?: number + frequency_mhz?: Partial> + } + assignments: AssignmentInput[] + unassigned_peaks?: UnassignedPeakInput[] + options?: { + fallback_tolerances?: Partial> + } +} + +export interface QuickcheckShift { + atom: number + prediction: number + real: number + diff: number + status: string + hoseCode: string + spheres: number +} + +export interface QuickcheckResult { + id: number + type: string + statistics: { + accept: number + warning: number + reject: number + missing: number + total: number + } + shifts: QuickcheckShift[] +} + +export interface QuickcheckInput { + id: number + type: string + shifts: string + solvent: string +} + +export type QuickcheckClient = ( + molfile: string, + inputs: QuickcheckInput[], +) => Promise + +export type ReportStatus = 'green' | 'yellow' | 'red' | 'missing' | 'impossible' + +export interface ReportAtomRow { + label: string + atoms: number[] + observed: number | null + predicted: number | null + deviation: number | null + status: ReportStatus + spheres: number + shift_values: number | null + hose_code: string | null + pair?: string +} + +export interface NucleusReport { + mark: number + mark_is_approximate: true + result: 'accept' | 'revise' | 'reject' + penalties: { + mean_deviation: { ppm: number; points: number } + red_or_missing: { count: number; points: number } + yellow: { count: number; points: number } + } + statistics: QuickcheckResult['statistics'] + in_database_likely: boolean + atoms: ReportAtomRow[] +} + +export type AssignmentRowStatus = 'ok' | 'review' | 'fail' | 'not_assessable' + +export interface AssignmentRowResult { + label: string + nucleus: Nucleus + atoms: number[] + observed: number + predicted: number | null + delta: number | null + spheres: number | null + status: AssignmentRowStatus + status_source: 'nmrshiftdb' | 'tolerance' | null + reasons: string[] +} + +export interface AssignmentIssue { + type: + | 'missing_signal' + | 'extra_signal' + | 'equivalence_violation' + | 'accidental_overlap' + | 'solvent_assigned' + | 'proton_count' + | 'prediction_impossible' + | 'unknown_atom' + nucleus: Nucleus + labels: string[] + atoms: number[] + message: string +} + +export interface SwapSuggestion { + type: 'swap' + nucleus: Nucleus + labels: [string, string] + error_reduction: number +} + +export interface AssignmentCheck { + result: 'consistent' | 'review' | 'inconsistent' | 'not_assessable' + rows: AssignmentRowResult[] + offset: Partial> & { + suspected_referencing_error: boolean + } + suggestions: SwapSuggestion[] + issues: AssignmentIssue[] +} + +export interface ValidationReport { + engine: { name: 'nmrshift'; source: 'nmrshiftdb2 quickcheck'; url: string } + solvent: string + verdict: 'accept' | 'review' | 'reject' | 'not_assessable' + reports: Partial> + assignment_check: AssignmentCheck + adjustments: { nucleus: Nucleus; label: string; sent: number; observed: number }[] +} diff --git a/app/scripts/nmr-cli/src/validation/validate.ts b/app/scripts/nmr-cli/src/validation/validate.ts new file mode 100644 index 0000000..97435f0 --- /dev/null +++ b/app/scripts/nmr-cli/src/validation/validate.ts @@ -0,0 +1,767 @@ +import { InvalidStructureError, prepareStructure } from './molfile' +import type { PreparedStructure } from './molfile' +import { resolveSolvent } from './solvents' +import { analyseTopology } from './topology' +import type { Topology } from './topology' +import { NUCLEI } from './types' +import type { + AssignmentCheck, + AssignmentInput, + AssignmentIssue, + AssignmentRowResult, + AssignmentRowStatus, + AssignmentSetInput, + Nucleus, + NucleusReport, + QuickcheckClient, + QuickcheckInput, + QuickcheckResult, + ReportAtomRow, + ReportStatus, + SwapSuggestion, + Tolerance, + ValidationReport, +} from './types' + +export class InvalidInputError extends Error {} + +const DEFAULT_TOLERANCES: Record = { + '13C': { ok: 3, fail: 6 }, + '1H': { ok: 0.3, fail: 0.6 }, +} + +const QUICKCHECK_TYPES: Record = { + '13C': { id: 1, type: 'nmr;13C;1d' }, + '1H': { id: 2, type: 'nmr;1H;1d' }, +} + +/** Values closer than this are treated as the same signal by the servlet. */ +const SAME_VALUE = 0.0005 +const NUDGE = 0.001 + +/** Reliable HOSE predictions start at four spheres; below that an atom can at most be flagged for review. */ +const RELIABLE_SPHERES = 4 + +const SWAP_MIN_REDUCTION: Record = { '13C': 2, '1H': 0.2 } +const SWAP_MIN_PREDICTION_GAP: Record = { '13C': 0.5, '1H': 0.05 } +const EQUIVALENCE_SPREAD: Record = { '13C': 0.5, '1H': 0.05 } +const REFERENCING_OFFSET: Record = { '13C': 1, '1H': 0.1 } +const SOLVENT_WINDOW: Record = { '13C': 0.3, '1H': 0.03 } +const IN_DATABASE_MEAN_DEVIATION: Record = { '13C': 1, '1H': 0.1 } +/** + * nmrshiftdb2 does not publish its mark formula; these weights reproduce the + * example reports and are exposed as `mark_is_approximate`. + */ +const MARK_POINTS = { perPpmMeanDeviation: 0.5, redOrMissing: 2, yellow: 1 } + +export interface ValidateOptions { + quickcheck: QuickcheckClient + url: string +} + +interface NormalizedRow { + index: number + input: AssignmentInput + label: string + /** Original indices of the heavy atoms the signal belongs to. */ + carriers: number[] + /** Carrier of each submitted atom, aligned with `input.atoms`. */ + atomCarriers: (number | null)[] + unresolved: number[] + sent: number +} + +interface AtomPrediction { + prediction: number | null + spheres: number + hoseCode: string | null + statuses: string[] + reals: number[] +} + +type Predictions = Map> + +export async function validateAssignments( + input: AssignmentSetInput, + options: ValidateOptions, +): Promise { + assertValidInput(input) + + const structure = prepareStructure(input.structure.molfile) + const topology = analyseTopology(structure) + const solvent = resolveSolvent(input.conditions?.solvent) + const tolerances = { ...DEFAULT_TOLERANCES, ...(input.options?.fallback_tolerances ?? {}) } + + const rows = normalizeRows(input.assignments, structure, topology) + const adjustments = assignSentValues(rows, topology) + const quickcheckInputs = buildQuickcheckInputs(rows, input, solvent.nmrshiftdb) + + const results = quickcheckInputs.length + ? await options.quickcheck(structure.molfile, quickcheckInputs) + : [] + const predictions = collectPredictions(results, structure, topology) + + const reports: ValidationReport['reports'] = {} + for (const nucleus of NUCLEI) { + const result = results.find((item) => item.id === QUICKCHECK_TYPES[nucleus].id) + if (result) { + reports[nucleus] = buildNucleusReport(nucleus, result, predictions.get(nucleus)!, rows) + } + } + + const assignmentCheck = checkAssignments(rows, predictions, topology, structure, input, tolerances, solvent.residual) + + return { + engine: { name: 'nmrshift', source: 'nmrshiftdb2 quickcheck', url: options.url }, + solvent: solvent.nmrshiftdb, + verdict: overallVerdict(reports, assignmentCheck), + reports, + assignment_check: assignmentCheck, + adjustments: adjustments.map((row) => ({ + nucleus: row.input.nucleus, + label: row.label, + sent: round(row.sent, 4), + observed: row.input.shift, + })), + } +} + +function assertValidInput(input: AssignmentSetInput): void { + if (!input || typeof input !== 'object') { + throw new InvalidInputError('Input must be a JSON object') + } + if (typeof input.structure?.molfile !== 'string' || input.structure.molfile.trim() === '') { + throw new InvalidInputError('structure.molfile is required') + } + if (!Array.isArray(input.assignments) || input.assignments.length === 0) { + throw new InvalidInputError('assignments must contain at least one signal') + } + input.assignments.forEach((row, index) => { + if (!NUCLEI.includes(row.nucleus)) { + throw new InvalidInputError(`assignments[${index}].nucleus must be 13C or 1H`) + } + if (typeof row.shift !== 'number' || !Number.isFinite(row.shift)) { + throw new InvalidInputError(`assignments[${index}].shift must be a number`) + } + if (!Array.isArray(row.atoms) || row.atoms.some((atom) => !Number.isInteger(atom) || atom < 1)) { + throw new InvalidInputError(`assignments[${index}].atoms must be 1-based atom indices`) + } + }) +} + +function normalizeRows( + assignments: AssignmentInput[], + structure: PreparedStructure, + topology: Topology, +): NormalizedRow[] { + return assignments.map((input, index) => { + const atomCarriers = input.atoms.map((atom) => resolveCarrier(atom, input.nucleus, structure, topology)) + const carriers = new Set(atomCarriers.filter((carrier): carrier is number => carrier !== null)) + + return { + index, + input, + label: input.label?.trim() || defaultLabel(input), + carriers: [...carriers].sort((a, b) => a - b), + atomCarriers, + unresolved: input.atoms.filter((_, position) => atomCarriers[position] === null), + sent: input.shift, + } + }) +} + +function resolveCarrier( + atom: number, + nucleus: Nucleus, + structure: PreparedStructure, + topology: Topology, +): number | null { + if (nucleus === '1H') { + const parent = structure.explicitHydrogenParents.get(atom) + if (parent !== undefined) return parent + return (topology.hydrogenCounts.get(atom) ?? 0) > 0 ? atom : null + } + + return structure.symbols.get(atom) === 'C' ? atom : null +} + +function defaultLabel(input: AssignmentInput): string { + if (input.atoms.length === 0) return `${input.shift}` + const prefix = input.nucleus === '13C' ? 'C' : 'H' + return `${prefix}-${input.atoms.join(', ')}${input.diastereotopic ?? ''}` +} + +/** + * The servlet collapses identical values, so two different environments with + * the same observed shift would leave one of them "missing". Such duplicates + * are nudged apart; identical values on equivalent atoms are sent once. + */ +function assignSentValues(rows: NormalizedRow[], topology: Topology): NormalizedRow[] { + const adjusted: NormalizedRow[] = [] + + for (const nucleus of NUCLEI) { + const taken: { value: number; classes: string }[] = [] + const nucleusRows = rows + .filter((row) => row.input.nucleus === nucleus) + .sort((a, b) => a.input.shift - b.input.shift) + + for (const row of nucleusRows) { + const classes = row.carriers.map((atom) => topology.classes.get(atom)).sort().join('|') + const duplicate = taken.find((item) => Math.abs(item.value - row.input.shift) < SAME_VALUE) + + if (duplicate && classes !== '' && duplicate.classes === classes) { + continue + } + + let value = row.input.shift + while (taken.some((item) => Math.abs(item.value - value) < SAME_VALUE)) { + value += NUDGE + } + if (value !== row.input.shift) { + row.sent = value + adjusted.push(row) + } + taken.push({ value, classes }) + } + } + + return adjusted +} + +function buildQuickcheckInputs( + rows: NormalizedRow[], + input: AssignmentSetInput, + solvent: string, +): QuickcheckInput[] { + return NUCLEI.flatMap((nucleus) => { + const values = new Set() + for (const row of rows) { + if (row.input.nucleus === nucleus) values.add(round(row.sent, 4)) + } + for (const peak of input.unassigned_peaks ?? []) { + if (peak.nucleus === nucleus && (peak.kind ?? 'unknown') === 'unknown') { + values.add(round(peak.shift, 4)) + } + } + if (values.size === 0) return [] + + return [ + { + ...QUICKCHECK_TYPES[nucleus], + shifts: [...values].sort((a, b) => a - b).map(String).join(';'), + solvent, + }, + ] + }) +} + +function collectPredictions( + results: QuickcheckResult[], + structure: PreparedStructure, + topology: Topology, +): Predictions { + const predictions: Predictions = new Map(NUCLEI.map((nucleus) => [nucleus, new Map()])) + + for (const nucleus of NUCLEI) { + const result = results.find((item) => item.id === QUICKCHECK_TYPES[nucleus].id) + if (!result) continue + const byAtom = predictions.get(nucleus)! + + for (const shift of result.shifts) { + const carrier = + nucleus === '13C' + ? structure.toOriginal[shift.atom] + : topology.servletHydrogenOwners.get(shift.atom) + if (carrier === undefined) continue + + const impossible = shift.status.startsWith('prediction impossible') + const entry = byAtom.get(carrier) ?? { + prediction: null, + spheres: Number.POSITIVE_INFINITY, + hoseCode: null, + statuses: [], + reals: [], + } + if (!impossible) { + entry.prediction = shift.prediction + entry.spheres = Math.min(entry.spheres, shift.spheres) + entry.hoseCode ??= shift.hoseCode || null + } + entry.statuses.push(impossible ? 'impossible' : shift.status) + if (!impossible && shift.status !== 'missing') entry.reals.push(shift.real) + byAtom.set(carrier, entry) + } + + for (const entry of byAtom.values()) { + if (!Number.isFinite(entry.spheres)) entry.spheres = 0 + } + } + + return predictions +} + +function buildNucleusReport( + nucleus: Nucleus, + result: QuickcheckResult, + byAtom: Map, + rows: NormalizedRow[], +): NucleusReport { + const atomRows: ReportAtomRow[] = [] + + for (const [atom, entry] of [...byAtom.entries()].sort(([a], [b]) => a - b)) { + const label = labelForAtom(atom, nucleus, rows) + const base = { + atoms: [atom], + predicted: entry.prediction === null ? null : round(entry.prediction, 2), + spheres: entry.spheres, + shift_values: null, + hose_code: entry.hoseCode, + } + + if (entry.statuses.every((status) => status === 'impossible')) { + atomRows.push({ ...base, label, observed: null, deviation: null, status: 'impossible' }) + continue + } + if (entry.reals.length === 0) { + atomRows.push({ ...base, label, observed: null, deviation: null, status: 'missing' }) + continue + } + + const status = worstStatus(entry.statuses) + const observed = [...new Set(entry.reals.map((value) => round(value, 4)))].sort((a, b) => a - b) + const prediction = entry.prediction ?? 0 + + if (nucleus === '1H' && observed.length === 2) { + const stem = label.replace(/[ab]$/, '') + const deviation = round(Math.abs(mean(observed) - prediction), 3) + atomRows.push({ ...base, label: `${stem}a`, observed: observed[0], deviation, status, pair: `${stem}b` }) + atomRows.push({ ...base, label: `${stem}b`, observed: observed[1], deviation, status, pair: `${stem}a` }) + continue + } + + const value = mean(observed) + atomRows.push({ ...base, label, observed: round(value, 4), deviation: round(Math.abs(value - prediction), 3), status }) + } + + const scored = atomRows.filter((row) => row.deviation !== null) + const meanDeviation = scored.length ? mean(scored.map((row) => row.deviation!)) : 0 + const lowConfidenceWeight = (row: ReportAtomRow) => (row.spheres <= 2 ? 0.5 : 1) + const redOrMissing = atomRows.filter((row) => row.status === 'red' || row.status === 'missing') + const yellow = atomRows.filter((row) => row.status === 'yellow') + + const deviationPoints = truncate(meanDeviation * MARK_POINTS.perPpmMeanDeviation, 2) + const redPoints = round(redOrMissing.reduce((sum, row) => sum + MARK_POINTS.redOrMissing * lowConfidenceWeight(row), 0), 2) + const yellowPoints = round(yellow.reduce((sum, row) => sum + MARK_POINTS.yellow * lowConfidenceWeight(row), 0), 2) + const mark = Math.min(10, Math.max(1, Math.round(10 - deviationPoints - redPoints - yellowPoints))) + + const sixSphereShare = scored.length ? scored.filter((row) => row.spheres >= 6).length / scored.length : 0 + + return { + mark, + mark_is_approximate: true, + result: mark >= 8 ? 'accept' : mark >= 5 ? 'revise' : 'reject', + penalties: { + mean_deviation: { ppm: round(meanDeviation, 2), points: deviationPoints }, + red_or_missing: { count: redOrMissing.length, points: redPoints }, + yellow: { count: yellow.length, points: yellowPoints }, + }, + statistics: result.statistics, + in_database_likely: + scored.length >= 3 && + sixSphereShare >= 0.8 && + meanDeviation <= IN_DATABASE_MEAN_DEVIATION[nucleus], + atoms: atomRows, + } +} + +/** + * Name for an atom that has no signal of this nucleus, borrowed from the + * carbon label when the author assigned one ("C-4'" gives "H-4'"). + */ +function atomName(atom: number, nucleus: Nucleus, rows: NormalizedRow[]): string { + const carbonRow = rows.find((row) => row.input.nucleus === '13C' && row.carriers.includes(atom) && row.input.label) + if (!carbonRow) return `atom ${atom}` + const carbonLabel = labelForAtom(atom, '13C', rows) + if (nucleus === '13C') return carbonLabel + return /^C(?=[-\d])/.test(carbonLabel) ? carbonLabel.replace(/^C/, 'H') : `H on ${carbonLabel}` +} + +/** + * Report rows are per atom, like the nmrshiftdb quality report. A label that + * lists one name per submitted atom ("C-2, C-6") is split accordingly. + */ +function labelForAtom(atom: number, nucleus: Nucleus, rows: NormalizedRow[]): string { + const row = rows.find((item) => item.input.nucleus === nucleus && item.carriers.includes(atom)) + if (!row) return String(atom) + if (row.carriers.length === 1) { + return row.input.diastereotopic ? row.label.replace(/[ab]$/, '') : row.label + } + + const parts = row.label.split(/\s*(?:,|;|\band\b)\s*/).filter(Boolean) + const position = row.atomCarriers.indexOf(atom) + if (parts.length === row.input.atoms.length && position >= 0) return parts[position] + return `${row.label} [${atom}]` +} + +function worstStatus(statuses: string[]): ReportStatus { + if (statuses.includes('reject')) return 'red' + if (statuses.includes('warning')) return 'yellow' + return 'green' +} + +function checkAssignments( + rows: NormalizedRow[], + predictions: Predictions, + topology: Topology, + structure: PreparedStructure, + input: AssignmentSetInput, + tolerances: Record, + residuals: Record, +): AssignmentCheck { + const issues: AssignmentIssue[] = [] + const pairMeans = diastereotopicPairMeans(rows, topology) + + const results: AssignmentRowResult[] = rows.map((row) => { + const { nucleus } = row.input + const reasons: string[] = [] + + if (row.unresolved.length) { + issues.push({ + type: 'unknown_atom', + nucleus, + labels: [row.label], + atoms: row.unresolved, + message: + nucleus === '13C' + ? 'Atom is not a carbon in the submitted structure.' + : 'Atom carries no hydrogen in the submitted structure.', + }) + } + + const entries = row.carriers.map((atom) => predictions.get(nucleus)?.get(atom)) + const valid = entries.filter((entry): entry is AtomPrediction => entry?.prediction != null) + const impossible = row.carriers.filter((atom, i) => entries[i] && entries[i]!.prediction == null) + if (impossible.length) { + issues.push({ + type: 'prediction_impossible', + nucleus, + labels: [row.label], + atoms: impossible, + message: 'nmrshiftdb could not predict a shift for this atom.', + }) + } + + const base = { + label: row.label, + nucleus, + atoms: row.carriers, + observed: row.input.shift, + } + + if (row.carriers.length === 0 || valid.length === 0) { + return { ...base, predicted: null, delta: null, spheres: null, status: 'not_assessable', status_source: null, reasons } + } + + const predicted = mean(valid.map((entry) => entry.prediction!)) + const spheres = Math.min(...valid.map((entry) => entry.spheres)) + const pairMean = pairMeans.get(row.index) + if (pairMean !== undefined) reasons.push('diastereotopic_pair_mean') + const delta = (pairMean ?? row.input.shift) - predicted + + const matchedByServlet = valid.every((entry) => + entry.reals.some((real) => Math.abs(real - row.sent) < SAME_VALUE), + ) + let status: AssignmentRowStatus + let source: AssignmentRowResult['status_source'] + if (matchedByServlet && pairMean === undefined) { + status = servletStatus(valid.flatMap((entry) => entry.statuses)) + source = 'nmrshiftdb' + } else { + status = toleranceStatus(Math.abs(delta), spheres, tolerances[nucleus]) + source = 'tolerance' + } + if (status === 'fail' && spheres < RELIABLE_SPHERES) { + status = 'review' + reasons.push('low_confidence_prediction') + } + + return { + ...base, + predicted: round(predicted, 2), + delta: round(delta, 3), + spheres, + status, + status_source: source, + reasons, + } + }) + + issues.push(...structuralIssues(rows, results, topology, structure, input, residuals)) + const suggestions = swapSuggestions(results, pairMeans, rows, tolerances) + const offset = referencingOffset(results) + + return { + result: assignmentResult(results, suggestions, issues), + rows: results, + offset, + suggestions, + issues, + } +} + +/** + * nmrshiftdb predicts one value per CH2, so diastereotopic protons assigned to + * the same carbon are compared through their mean, as in its quality report. + */ +function diastereotopicPairMeans(rows: NormalizedRow[], topology: Topology): Map { + const byCarrier = new Map() + for (const row of rows) { + if (row.input.nucleus !== '1H' || row.carriers.length !== 1) continue + const carrier = row.carriers[0] + if ((topology.hydrogenCounts.get(carrier) ?? 0) < 2) continue + byCarrier.set(carrier, [...(byCarrier.get(carrier) ?? []), row]) + } + + const means = new Map() + for (const group of byCarrier.values()) { + if (group.length < 2) continue + const value = mean(group.map((row) => row.input.shift)) + for (const row of group) means.set(row.index, value) + } + return means +} + +function servletStatus(statuses: string[]): AssignmentRowStatus { + if (statuses.includes('reject')) return 'fail' + if (statuses.includes('warning')) return 'review' + return 'ok' +} + +function toleranceStatus(deviation: number, spheres: number, tolerance: Tolerance): AssignmentRowStatus { + const factor = spheres >= RELIABLE_SPHERES ? 1 : spheres === 3 ? 1.5 : 2 + if (deviation <= tolerance.ok * factor) return 'ok' + if (deviation <= tolerance.fail * factor) return 'review' + return 'fail' +} + +function structuralIssues( + rows: NormalizedRow[], + results: AssignmentRowResult[], + topology: Topology, + structure: PreparedStructure, + input: AssignmentSetInput, + residuals: Record, +): AssignmentIssue[] { + const issues: AssignmentIssue[] = [] + + for (const nucleus of NUCLEI) { + const nucleusRows = rows.filter((row) => row.input.nucleus === nucleus) + const assigned = nucleusRows.filter((row) => row.carriers.length > 0) + if (assigned.length === 0) continue + + const byClass = new Map() + for (const row of assigned) { + const classes = new Set(row.carriers.map((atom) => topology.classes.get(atom)!)) + if (classes.size > 1) { + issues.push({ + type: 'accidental_overlap', + nucleus, + labels: [row.label], + atoms: row.carriers, + message: 'One shift is assigned to atoms that are not equivalent.', + }) + } + for (const cls of classes) byClass.set(cls, [...(byClass.get(cls) ?? []), row]) + } + + for (const group of byClass.values()) { + const distinctCarriers = new Set(group.flatMap((row) => row.carriers)) + if (group.length < 2 || distinctCarriers.size < 2) continue + const shifts = group.map((row) => row.input.shift) + if (Math.max(...shifts) - Math.min(...shifts) > EQUIVALENCE_SPREAD[nucleus]) { + issues.push({ + type: 'equivalence_violation', + nucleus, + labels: group.map((row) => row.label), + atoms: [...distinctCarriers].sort((a, b) => a - b), + message: 'Symmetry-equivalent atoms are assigned different shifts.', + }) + } + } + + const covered = new Set(assigned.flatMap((row) => row.carriers.map((atom) => topology.classes.get(atom)))) + const missing: number[] = [] + for (const [atom, cls] of topology.classes) { + const isCarbon = structure.symbols.get(atom) === 'C' + const relevant = nucleus === '13C' ? isCarbon : isCarbon && (topology.hydrogenCounts.get(atom) ?? 0) > 0 + if (relevant && !covered.has(cls)) missing.push(atom) + } + if (missing.length) { + issues.push({ + type: 'missing_signal', + nucleus, + labels: missing.map((atom) => atomName(atom, nucleus, rows)), + atoms: missing, + message: + nucleus === '13C' + ? 'Carbon environments without an assigned signal (often quaternary carbons).' + : 'C-H environments without an assigned signal.', + }) + } + + for (const row of nucleusRows) { + const result = results[rows.indexOf(row)] + const nearSolvent = residuals[nucleus].some((value) => Math.abs(value - row.input.shift) <= SOLVENT_WINDOW[nucleus]) + if (nearSolvent && result.status !== 'ok' && result.status !== 'not_assessable') { + issues.push({ + type: 'solvent_assigned', + nucleus, + labels: [row.label], + atoms: row.carriers, + message: 'Shift is at a residual solvent position and does not fit the prediction.', + }) + } + } + + if (nucleus === '1H') { + for (const row of assigned) { + if (row.input.n_h === undefined) continue + const carried = row.carriers.reduce((sum, atom) => sum + (topology.hydrogenCounts.get(atom) ?? 0), 0) + const shared = assigned.filter((other) => other.carriers.join() === row.carriers.join()).length + if (shared === 1 && row.input.n_h !== carried) { + issues.push({ + type: 'proton_count', + nucleus, + labels: [row.label], + atoms: row.carriers, + message: `Integral of ${row.input.n_h} H, but the assigned atoms carry ${carried} H.`, + }) + } + } + } + } + + for (const peak of input.unassigned_peaks ?? []) { + if ((peak.kind ?? 'unknown') === 'solvent') continue + issues.push({ + type: 'extra_signal', + nucleus: peak.nucleus, + labels: [String(peak.shift)], + atoms: [], + message: peak.kind === 'impurity' ? 'Signal marked as impurity.' : 'Signal without assignment.', + }) + } + + return issues +} + +function swapSuggestions( + results: AssignmentRowResult[], + pairMeans: Map, + rows: NormalizedRow[], + tolerances: Record, +): SwapSuggestion[] { + const suggestions: SwapSuggestion[] = [] + const eligible = results + .map((result, index) => ({ result, index })) + .filter( + ({ result, index }) => + result.predicted !== null && + (result.spheres ?? 0) >= RELIABLE_SPHERES && + !pairMeans.has(rows[index].index), + ) + + for (let i = 0; i < eligible.length; i++) { + for (let j = i + 1; j < eligible.length; j++) { + const a = eligible[i].result + const b = eligible[j].result + if (a.nucleus !== b.nucleus) continue + if (Math.abs(a.predicted! - b.predicted!) <= SWAP_MIN_PREDICTION_GAP[a.nucleus]) continue + + const current = Math.abs(a.observed - a.predicted!) + Math.abs(b.observed - b.predicted!) + const swappedA = Math.abs(a.observed - b.predicted!) + const swappedB = Math.abs(b.observed - a.predicted!) + const reduction = current - swappedA - swappedB + + if ( + reduction >= SWAP_MIN_REDUCTION[a.nucleus] && + Math.max(swappedA, swappedB) <= tolerances[a.nucleus].fail + ) { + suggestions.push({ type: 'swap', nucleus: a.nucleus, labels: [a.label, b.label], error_reduction: round(reduction, 2) }) + } + } + } + + return suggestions.sort((a, b) => b.error_reduction - a.error_reduction).slice(0, 10) +} + +function referencingOffset(results: AssignmentRowResult[]): AssignmentCheck['offset'] { + const offset: AssignmentCheck['offset'] = { suspected_referencing_error: false } + + for (const nucleus of NUCLEI) { + const deltas = results + .filter((row) => row.nucleus === nucleus && row.delta !== null && (row.spheres ?? 0) >= RELIABLE_SPHERES) + .map((row) => row.delta!) + if (deltas.length < 3) { + offset[nucleus] = null + continue + } + const value = round(median(deltas), 3) + offset[nucleus] = value + if (Math.abs(value) > REFERENCING_OFFSET[nucleus]) offset.suspected_referencing_error = true + } + + return offset +} + +function assignmentResult( + results: AssignmentRowResult[], + suggestions: SwapSuggestion[], + issues: AssignmentIssue[], +): AssignmentCheck['result'] { + const assessed = results.filter((row) => row.status !== 'not_assessable') + if (assessed.length === 0) return 'not_assessable' + if (assessed.every((row) => (row.spheres ?? 0) <= 2)) return 'not_assessable' + if (assessed.some((row) => row.status === 'fail')) return 'inconsistent' + if ( + assessed.some((row) => row.status === 'review') || + suggestions.length > 0 || + issues.some((issue) => issue.type === 'equivalence_violation' || issue.type === 'unknown_atom') + ) { + return 'review' + } + return 'consistent' +} + +function overallVerdict( + reports: ValidationReport['reports'], + check: AssignmentCheck, +): ValidationReport['verdict'] { + const primary = reports['13C'] ?? reports['1H'] + if (!primary) return 'not_assessable' + if (primary.result === 'reject' || check.result === 'inconsistent') return 'reject' + if (primary.result === 'revise' || check.result === 'review') return 'review' + return 'accept' +} + +function mean(values: number[]): number { + return values.reduce((sum, value) => sum + value, 0) / values.length +} + +function median(values: number[]): number { + const sorted = [...values].sort((a, b) => a - b) + const middle = Math.floor(sorted.length / 2) + return sorted.length % 2 ? sorted[middle] : (sorted[middle - 1] + sorted[middle]) / 2 +} + +function round(value: number, digits: number): number { + const factor = 10 ** digits + return Math.round(value * factor) / factor +} + +function truncate(value: number, digits: number): number { + const factor = 10 ** digits + return Math.floor(value * factor + 1e-9) / factor +} + +export { InvalidStructureError } diff --git a/app/scripts/nmr-cli/test/fixtures/nicotine.mol b/app/scripts/nmr-cli/test/fixtures/nicotine.mol new file mode 100644 index 0000000..a03df6e --- /dev/null +++ b/app/scripts/nmr-cli/test/fixtures/nicotine.mol @@ -0,0 +1,30 @@ + +OCL MolfileCreator 2D + + 12 13 0 0 1 0 0 0 0 0999 V2000 + 0.8660 0.0000 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + 0.0000 -0.5000 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0 + -0.9135 -0.0933 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -1.5827 -0.8364 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -1.0827 -1.7024 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -0.1045 -1.4945 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + 0.6386 -2.1637 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + 0.4307 -3.1418 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + 1.1738 -3.8109 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + 2.1249 -3.5019 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + 2.3328 -2.5238 0.0000 N 0 0 0 0 0 0 0 0 0 0 0 0 + 1.5897 -1.8546 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + 1 2 1 0 0 0 0 + 2 3 1 0 0 0 0 + 3 4 1 0 0 0 0 + 4 5 1 0 0 0 0 + 5 6 1 0 0 0 0 + 2 6 1 0 0 0 0 + 6 7 1 6 0 0 0 + 7 8 2 0 0 0 0 + 8 9 1 0 0 0 0 + 9 10 2 0 0 0 0 + 10 11 1 0 0 0 0 + 11 12 2 0 0 0 0 + 7 12 1 0 0 0 0 +M END diff --git a/app/scripts/nmr-cli/test/fixtures/quickcheck-responses.json b/app/scripts/nmr-cli/test/fixtures/quickcheck-responses.json new file mode 100644 index 0000000..9c91fe4 --- /dev/null +++ b/app/scripts/nmr-cli/test/fixtures/quickcheck-responses.json @@ -0,0 +1,704 @@ +{ + "trimethoxybenzaldehyde-correct": { + "result": [ + { + "id": 1, + "type": "nmr;13C;1d", + "statistics": { + "accept": 10, + "warning": 0, + "reject": 0, + "missing": 0, + "total": 10 + }, + "shifts": [ + { + "atom": 1, + "prediction": 106.67300033569336, + "real": 106.646, + "diff": 0.02700033569335858, + "status": "accept", + "hoseCode": "C-3-6;H*C*C(*CC,*CO/H*C,H=O,*&O,C/*&O,,C,HHH),C,HHH/,HHH/", + "spheres": 6 + }, + { + "atom": 3, + "prediction": 106.67300033569336, + "real": 106.646, + "diff": 0.02700033569335858, + "status": "accept", + "hoseCode": "C-3-6;H*C*C(*CC,*CO/H*C,H=O,*&O,C/*&O,,C,HHH),C,HHH/,HHH/", + "spheres": 6 + }, + { + "atom": 2, + "prediction": 131.6884994506836, + "real": 131.677, + "diff": 0.011499450683601253, + "status": "accept", + "hoseCode": "C-3-6;*C*CC(H,H,*C,*C,H=O/*C,*&,O,O,/*&O,C,C),C,HHH,HHH/,HHH/", + "spheres": 6 + }, + { + "atom": 4, + "prediction": 153.6060028076172, + "real": 153.612, + "diff": 0.00599719238280727, + "status": "accept", + "hoseCode": "C-3-6;*C*CO(*CO,H*C,C/*CO,C,*&C,HHH/H*&,C,HHH,H=O),HHH,//", + "spheres": 6 + }, + { + "atom": 6, + "prediction": 153.6060028076172, + "real": 153.612, + "diff": 0.00599719238280727, + "status": "accept", + "hoseCode": "C-3-6;*C*CO(*CO,H*C,C/*CO,C,*&C,HHH/H*&,C,HHH,H=O),HHH,//", + "spheres": 6 + }, + { + "atom": 5, + "prediction": 143.5590057373047, + "real": 143.518, + "diff": 0.04100573730468682, + "status": "accept", + "hoseCode": "C-3-6;*C*CO(*C,*C,O,O,C/H,H,*C,*&,C,C,HHH/*&C,HHH,HHH),H=O/,/", + "spheres": 6 + }, + { + "atom": 7, + "prediction": 190.99699783325195, + "real": 191.088, + "diff": 0.09100216674804074, + "status": "accept", + "hoseCode": "C-3;H=OC(,*C*C/H,H,*C,*C/*C,*&,O,O)*&O,C,C/,C,HHH,HHH/", + "spheres": 6 + }, + { + "atom": 12, + "prediction": 56.22924995422363, + "real": 56.258, + "diff": 0.02875004577636986, + "status": "accept", + "hoseCode": "C-4;HHHO(C/*C*C/*CO,H*C)*CO,C,*&C/H*&,C,HHH,H=O/", + "spheres": 6 + }, + { + "atom": 14, + "prediction": 56.22924995422363, + "real": 56.258, + "diff": 0.02875004577636986, + "status": "accept", + "hoseCode": "C-4;HHHO(C/*C*C/*CO,H*C)*CO,C,*&C/H*&,C,HHH,H=O/", + "spheres": 6 + }, + { + "atom": 13, + "prediction": 60.78940391540527, + "real": 61.009, + "diff": 0.2195960845947269, + "status": "accept", + "hoseCode": "C-4;HHHO(C/*C*C/*C,*C,O,O)H,H,*C,*&,C,C/*&C,HHH,HHH/", + "spheres": 6 + } + ] + }, + { + "id": 2, + "type": "nmr;1H;1d", + "statistics": { + "accept": 12, + "warning": 0, + "reject": 0, + "missing": 0, + "total": 12 + }, + "shifts": [ + { + "atom": 15, + "prediction": 7.123149871826172, + "real": 7.123, + "diff": 0.00014987182617165473, + "status": "accept", + "hoseCode": "H-1;C(*C*C/*CC,*CO/H*C,H=O,*&O,C)*&O,,C,HHH/,C,HHH/", + "spheres": 6 + }, + { + "atom": 16, + "prediction": 7.123149871826172, + "real": 7.123, + "diff": 0.00014987182617165473, + "status": "accept", + "hoseCode": "H-1;C(*C*C/*CC,*CO/H*C,H=O,*&O,C)*&O,,C,HHH/,C,HHH/", + "spheres": 6 + }, + { + "atom": 17, + "prediction": 9.820740222930908, + "real": 9.862, + "diff": 0.041259777069091896, + "status": "accept", + "hoseCode": "H-1;C(=OC/,*C*C/H,H,*C,*C)*C,*&,O,O/*&O,C,C/", + "spheres": 6 + }, + { + "atom": 21, + "prediction": 3.8549044330914817, + "real": 3.926, + "diff": 0.07109556690851848, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*C,*C,O,O/H,H,*C,*&,C,C/", + "spheres": 6 + }, + { + "atom": 22, + "prediction": 3.8549044330914817, + "real": 3.926, + "diff": 0.07109556690851848, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*C,*C,O,O/H,H,*C,*&,C,C/", + "spheres": 6 + }, + { + "atom": 23, + "prediction": 3.8549044330914817, + "real": 3.926, + "diff": 0.07109556690851848, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*C,*C,O,O/H,H,*C,*&,C,C/", + "spheres": 6 + }, + { + "atom": 18, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 19, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 20, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 24, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 25, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 26, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + } + ] + } + ] + }, + "trimethoxybenzaldehyde-c4-wrong": { + "result": [ + { + "id": 1, + "type": "nmr;13C;1d", + "statistics": { + "accept": 9, + "warning": 0, + "reject": 1, + "missing": 0, + "total": 10 + }, + "shifts": [ + { + "atom": 1, + "prediction": 106.67300033569336, + "real": 106.646, + "diff": 0.02700033569335858, + "status": "accept", + "hoseCode": "C-3-6;H*C*C(*CC,*CO/H*C,H=O,*&O,C/*&O,,C,HHH),C,HHH/,HHH/", + "spheres": 6 + }, + { + "atom": 3, + "prediction": 106.67300033569336, + "real": 106.646, + "diff": 0.02700033569335858, + "status": "accept", + "hoseCode": "C-3-6;H*C*C(*CC,*CO/H*C,H=O,*&O,C/*&O,,C,HHH),C,HHH/,HHH/", + "spheres": 6 + }, + { + "atom": 2, + "prediction": 131.6884994506836, + "real": 131.677, + "diff": 0.011499450683601253, + "status": "accept", + "hoseCode": "C-3-6;*C*CC(H,H,*C,*C,H=O/*C,*&,O,O,/*&O,C,C),C,HHH,HHH/,HHH/", + "spheres": 6 + }, + { + "atom": 4, + "prediction": 153.6060028076172, + "real": 153.612, + "diff": 0.00599719238280727, + "status": "accept", + "hoseCode": "C-3-6;*C*CO(*CO,H*C,C/*CO,C,*&C,HHH/H*&,C,HHH,H=O),HHH,//", + "spheres": 6 + }, + { + "atom": 6, + "prediction": 153.6060028076172, + "real": 153.612, + "diff": 0.00599719238280727, + "status": "accept", + "hoseCode": "C-3-6;*C*CO(*CO,H*C,C/*CO,C,*&C,HHH/H*&,C,HHH,H=O),HHH,//", + "spheres": 6 + }, + { + "atom": 5, + "prediction": 143.5590057373047, + "real": 130.0, + "diff": 13.559005737304688, + "status": "reject", + "hoseCode": "C-3-6;*C*CO(*C,*C,O,O,C/H,H,*C,*&,C,C,HHH/*&C,HHH,HHH),H=O/,/", + "spheres": 6 + }, + { + "atom": 7, + "prediction": 190.99699783325195, + "real": 191.088, + "diff": 0.09100216674804074, + "status": "accept", + "hoseCode": "C-3;H=OC(,*C*C/H,H,*C,*C/*C,*&,O,O)*&O,C,C/,C,HHH,HHH/", + "spheres": 6 + }, + { + "atom": 12, + "prediction": 56.22924995422363, + "real": 56.258, + "diff": 0.02875004577636986, + "status": "accept", + "hoseCode": "C-4;HHHO(C/*C*C/*CO,H*C)*CO,C,*&C/H*&,C,HHH,H=O/", + "spheres": 6 + }, + { + "atom": 14, + "prediction": 56.22924995422363, + "real": 56.258, + "diff": 0.02875004577636986, + "status": "accept", + "hoseCode": "C-4;HHHO(C/*C*C/*CO,H*C)*CO,C,*&C/H*&,C,HHH,H=O/", + "spheres": 6 + }, + { + "atom": 13, + "prediction": 60.78940391540527, + "real": 61.009, + "diff": 0.2195960845947269, + "status": "accept", + "hoseCode": "C-4;HHHO(C/*C*C/*C,*C,O,O)H,H,*C,*&,C,C/*&C,HHH,HHH/", + "spheres": 6 + } + ] + }, + { + "id": 2, + "type": "nmr;1H;1d", + "statistics": { + "accept": 12, + "warning": 0, + "reject": 0, + "missing": 0, + "total": 12 + }, + "shifts": [ + { + "atom": 15, + "prediction": 7.123149871826172, + "real": 7.123, + "diff": 0.00014987182617165473, + "status": "accept", + "hoseCode": "H-1;C(*C*C/*CC,*CO/H*C,H=O,*&O,C)*&O,,C,HHH/,C,HHH/", + "spheres": 6 + }, + { + "atom": 16, + "prediction": 7.123149871826172, + "real": 7.123, + "diff": 0.00014987182617165473, + "status": "accept", + "hoseCode": "H-1;C(*C*C/*CC,*CO/H*C,H=O,*&O,C)*&O,,C,HHH/,C,HHH/", + "spheres": 6 + }, + { + "atom": 17, + "prediction": 9.820740222930908, + "real": 9.862, + "diff": 0.041259777069091896, + "status": "accept", + "hoseCode": "H-1;C(=OC/,*C*C/H,H,*C,*C)*C,*&,O,O/*&O,C,C/", + "spheres": 6 + }, + { + "atom": 21, + "prediction": 3.8549044330914817, + "real": 3.926, + "diff": 0.07109556690851848, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*C,*C,O,O/H,H,*C,*&,C,C/", + "spheres": 6 + }, + { + "atom": 22, + "prediction": 3.8549044330914817, + "real": 3.926, + "diff": 0.07109556690851848, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*C,*C,O,O/H,H,*C,*&,C,C/", + "spheres": 6 + }, + { + "atom": 23, + "prediction": 3.8549044330914817, + "real": 3.926, + "diff": 0.07109556690851848, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*C,*C,O,O/H,H,*C,*&,C,C/", + "spheres": 6 + }, + { + "atom": 18, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 19, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 20, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 24, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 25, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + }, + { + "atom": 26, + "prediction": 4.03446577390035, + "real": 3.933, + "diff": 0.10146577390035016, + "status": "accept", + "hoseCode": "H-1;C(HHO/C/*C*C)*CO,H*C/*CO,C,*&C/", + "spheres": 6 + } + ] + } + ] + }, + "nicotine": { + "result": [ + { + "id": 1, + "type": "nmr;13C;1d", + "statistics": { + "accept": 10, + "warning": 0, + "reject": 0, + "missing": 0, + "total": 10 + }, + "shifts": [ + { + "atom": 12, + "prediction": 149.7300033569336, + "real": 149.91, + "diff": 0.17999664306640284, + "status": "accept", + "hoseCode": "C-3-6;H*C*N(*CC,*C/H*C,@NHC,H*&/H*&,CC,HHC),HH&,HHH,HH&//", + "spheres": 6 + }, + { + "atom": 7, + "prediction": 138.72000122070312, + "real": 138.64, + "diff": 0.08000122070313864, + "status": "accept", + "hoseCode": "C-3-6;*C*CC(H*C,H*N,@NHC/H*C,*&,CC,HHC/H*&,HH&,HHH,HH&)//", + "spheres": 6 + }, + { + "atom": 8, + "prediction": 135.52499389648438, + "real": 135.15, + "diff": 0.3749938964843693, + "status": "accept", + "hoseCode": "C-3-6;H*C*C(*CC,H*C/H*N,@NHC,H*&/*&,CC,HHC),HH&,HHH,HH&//", + "spheres": 6 + }, + { + "atom": 9, + "prediction": 123.79999923706055, + "real": 123.35, + "diff": 0.44999923706055256, + "status": "accept", + "hoseCode": "C-3-6;H*C*C(H*C,H*N/*CC,*&/H*&,@NHC),CC,HHC/,HH&,HHH,HH&/", + "spheres": 6 + }, + { + "atom": 10, + "prediction": 148.56400190080916, + "real": 148.46, + "diff": 0.10400190080915195, + "status": "accept", + "hoseCode": "C-3-6;H*C*N(H*C,*C/H*C,H*&/*&C),@NHC/,CC,HHC/", + "spheres": 6 + }, + { + "atom": 6, + "prediction": 68.97000122070312, + "real": 68.94, + "diff": 0.030001220703127274, + "status": "accept", + "hoseCode": "C-4-5;@HCNC(*C*C,CC,HHC/H*C,H*N,HH&,HHH,HH&/H*C,*&)H*&//", + "spheres": 6 + }, + { + "atom": 3, + "prediction": 57.0, + "real": 56.88, + "diff": 0.11999999999999744, + "status": "accept", + "hoseCode": "C-4-5;HHCN(HHC,CC/HH&,@C&H,HHH/@,*C*C),H*C,H*N/,H*C,*&/", + "spheres": 6 + }, + { + "atom": 4, + "prediction": 22.299999237060547, + "real": 22.53, + "diff": 0.23000076293945426, + "status": "accept", + "hoseCode": "C-4-5;HHCC(HHC,HHN/@H&C,C&/,*C*C,HHH),H*C,H*N/,H*C,*&/", + "spheres": 6 + }, + { + "atom": 5, + "prediction": 35.14500045776367, + "real": 35.34, + "diff": 0.19499954223633154, + "status": "accept", + "hoseCode": "C-4-5;HHCC(@HNC,HHC/C&,*C*C,HH&/HHH,H*C,H*N),H*C,*&/,H*&/", + "spheres": 6 + }, + { + "atom": 1, + "prediction": 40.400001525878906, + "real": 40.35, + "diff": 0.05000152587890483, + "status": "accept", + "hoseCode": "C-4;HHHN(CC/@HCC,HHC/*C*C,HH&,HH&)H*C,H*N/H*C,*&/", + "spheres": 6 + } + ] + }, + { + "id": 2, + "type": "nmr;1H;1d", + "statistics": { + "accept": 13, + "warning": 1, + "reject": 0, + "missing": 0, + "total": 14 + }, + "shifts": [ + { + "atom": 25, + "prediction": 8.84000015258789, + "real": 8.55, + "diff": 0.2900001525878899, + "status": "accept", + "hoseCode": "H-1;C(*C*N/H*C,*C/H*C,H*&)*&C/,@NHC/", + "spheres": 6 + }, + { + "atom": 23, + "prediction": 7.690000057220459, + "real": 7.69, + "diff": 5.7220458593576495e-08, + "status": "accept", + "hoseCode": "H-1;C(*C*C/*CC,H*C/H*N,@NHC,H*&)*&,CC,HHC/,HH&,HHH,HH&/", + "spheres": 6 + }, + { + "atom": 24, + "prediction": 7.373499870300293, + "real": 7.24, + "diff": 0.13349987030029276, + "status": "accept", + "hoseCode": "H-1;C(*C*C/H*C,H*N/*CC,*&)H*&,@NHC/,CC,HHC/", + "spheres": 6 + }, + { + "atom": 26, + "prediction": 8.550000190734863, + "real": 8.5, + "diff": 0.05000019073486328, + "status": "accept", + "hoseCode": "H-1;C(*C*N/*CC,*C/H*C,@NHC,H*&)H*&,CC,HHC/,HH&,HHH,HH&/", + "spheres": 6 + }, + { + "atom": 16, + "prediction": 2.774999976158142, + "real": 3.08, + "diff": 0.305000023841858, + "status": "accept", + "hoseCode": "H-1;C(HCN/HHC,CC/HH&,@C&H,HHH)@,*C*C/,H*C,H*N/", + "spheres": 6 + }, + { + "atom": 17, + "prediction": 2.774999976158142, + "real": 3.08, + "diff": 0.305000023841858, + "status": "accept", + "hoseCode": "H-1;C(HCN/HHC,CC/HH&,@C&H,HHH)@,*C*C/,H*C,H*N/", + "spheres": 6 + }, + { + "atom": 13, + "prediction": 2.3200000524520874, + "real": 2.31, + "diff": 0.010000052452087349, + "status": "accept", + "hoseCode": "H-1;C(HHN/CC/@HCC,HHC)*C*C,HH&,HH&/H*C,H*N/", + "spheres": 4 + }, + { + "atom": 14, + "prediction": 2.3200000524520874, + "real": 2.31, + "diff": 0.010000052452087349, + "status": "accept", + "hoseCode": "H-1;C(HHN/CC/@HCC,HHC)*C*C,HH&,HH&/H*C,H*N/", + "spheres": 4 + }, + { + "atom": 15, + "prediction": 2.3200000524520874, + "real": 2.31, + "diff": 0.010000052452087349, + "status": "accept", + "hoseCode": "H-1;C(HHN/CC/@HCC,HHC)*C*C,HH&,HH&/H*C,H*N/", + "spheres": 4 + }, + { + "atom": 22, + "prediction": 4.372307667365441, + "real": 3.24, + "diff": 1.132307667365441, + "status": "warning", + "hoseCode": "H-1;C(@CNC/*C*C,CC,HHC/H*C,H*N,HH&,HHH,HH&)H*C,*&/H*&/", + "spheres": 3 + }, + { + "atom": 20, + "prediction": 1.7687500193715096, + "real": 1.73, + "diff": 0.03875001937150957, + "status": "accept", + "hoseCode": "H-1;C(HCC/@HNC,HHC/C&,*C*C,HH&)HHH,H*C,H*N/,H*C,*&/", + "spheres": 3 + }, + { + "atom": 21, + "prediction": 1.7687500193715096, + "real": 1.73, + "diff": 0.03875001937150957, + "status": "accept", + "hoseCode": "H-1;C(HCC/@HNC,HHC/C&,*C*C,HH&)HHH,H*C,H*N/,H*C,*&/", + "spheres": 3 + }, + { + "atom": 18, + "prediction": 2.0962000131607055, + "real": 2.17, + "diff": 0.07379998683929445, + "status": "accept", + "hoseCode": "H-1;C(HCC/HHC,HHN/@H&C,C&),*C*C,HHH/,H*C,H*N/", + "spheres": 4 + }, + { + "atom": 19, + "prediction": 2.0962000131607055, + "real": 2.17, + "diff": 0.07379998683929445, + "status": "accept", + "hoseCode": "H-1;C(HCC/HHC,HHN/@H&C,C&),*C*C,HHH/,H*C,H*N/", + "spheres": 4 + } + ] + } + ] + } +} \ No newline at end of file diff --git a/app/scripts/nmr-cli/test/fixtures/trimethoxybenzaldehyde.mol b/app/scripts/nmr-cli/test/fixtures/trimethoxybenzaldehyde.mol new file mode 100644 index 0000000..d0292f0 --- /dev/null +++ b/app/scripts/nmr-cli/test/fixtures/trimethoxybenzaldehyde.mol @@ -0,0 +1,46 @@ + + Mnova 11111616152D + + 15 15 0 0 0 0 0 0 0 0999 V2000 + -93.7695 79.5086 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -91.6927 78.3096 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -91.6927 75.9114 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -93.7695 74.7124 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -95.8463 75.9114 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -95.8463 78.3096 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -89.6158 79.5086 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -87.5390 78.3096 0.0000 H 0 0 0 0 0 0 0 0 0 0 0 0 + -89.6158 81.9067 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0 + -97.9232 79.5086 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0 + -97.9232 74.7124 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0 + -93.7695 72.3143 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0 + -95.8463 71.1152 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -100.0000 75.9114 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -97.9232 81.9067 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + 1 2 2 0 0 0 0 + 2 3 1 0 0 0 0 + 3 4 2 0 0 0 0 + 4 5 1 0 0 0 0 + 5 6 2 0 0 0 0 + 1 6 1 0 0 0 0 + 2 7 1 0 0 0 0 + 7 8 1 0 0 0 0 + 7 9 2 0 0 0 0 + 6 10 1 0 0 0 0 + 5 11 1 0 0 0 0 + 4 12 1 0 0 0 0 + 12 13 1 0 0 0 0 + 11 14 1 0 0 0 0 + 10 15 1 0 0 0 0 +M ZZC 1 6 +M ZZC 2 1 +M ZZC 3 2 +M ZZC 4 3 +M ZZC 5 4 +M ZZC 6 5 +M ZZC 7 7 +M ZZC 8 7a +M ZZC 13 8 +M ZZC 14 9 +M ZZC 15 10 +M END diff --git a/app/scripts/nmr-cli/test/validation.test.cjs b/app/scripts/nmr-cli/test/validation.test.cjs new file mode 100644 index 0000000..76bc2b3 --- /dev/null +++ b/app/scripts/nmr-cli/test/validation.test.cjs @@ -0,0 +1,235 @@ +const { test } = require('node:test') +const assert = require('node:assert/strict') +const { readFileSync } = require('node:fs') +const { join } = require('node:path') + +const { validateAssignments, InvalidInputError } = require('../build/validation/validate') +const { prepareStructure, InvalidStructureError } = require('../build/validation/molfile') +const { analyseTopology } = require('../build/validation/topology') + +const fixture = (name) => readFileSync(join(__dirname, 'fixtures', name), 'utf-8') +const responses = JSON.parse(fixture('quickcheck-responses.json')) + +function fakeQuickcheck(key) { + const calls = [] + const client = async (molfile, inputs) => { + calls.push({ molfile, inputs }) + return responses[key].result + } + return { client, calls } +} + +async function run(input, key) { + const { client, calls } = fakeQuickcheck(key) + const report = await validateAssignments(input, { quickcheck: client, url: 'https://example.test/quickcheck' }) + return { report, calls } +} + +const rowByLabel = (report, label) => report.assignment_check.rows.find((row) => row.label === label) + +/** Mnova export of 3,4,5-trimethoxybenzaldehyde; atom 8 is the explicit aldehyde H. */ +function trimethoxybenzaldehyde(overrides = {}) { + const shifts = { 'C-1': 131.677, 'C-4': 143.518, ...overrides } + return { + structure: { molfile: fixture('trimethoxybenzaldehyde.mol'), source: 'mnova' }, + conditions: { solvent: 'CDCl3' }, + assignments: [ + { nucleus: '13C', atoms: [1, 3], label: 'C-6, C-2', shift: 106.646 }, + { nucleus: '13C', atoms: [2], label: 'C-1', shift: shifts['C-1'] }, + { nucleus: '13C', atoms: [4, 6], label: 'C-3, C-5', shift: 153.612 }, + { nucleus: '13C', atoms: [5], label: 'C-4', shift: shifts['C-4'] }, + { nucleus: '13C', atoms: [7], label: 'C-7', shift: 191.088 }, + { nucleus: '13C', atoms: [13, 15], label: 'C-8, C-10', shift: 56.258 }, + { nucleus: '13C', atoms: [14], label: 'C-9', shift: 61.009 }, + { nucleus: '1H', atoms: [1, 3], label: 'H-6, H-2', shift: 7.123, n_h: 2 }, + { nucleus: '1H', atoms: [8], label: 'H-7', shift: 9.862, n_h: 1 }, + { nucleus: '1H', atoms: [13, 15], label: 'H-8, H-10', shift: 3.926, n_h: 6 }, + { nucleus: '1H', atoms: [14], label: 'H-9', shift: 3.933, n_h: 3 }, + ], + } +} + +/** Nicotine, numbered as in test/fixtures/nicotine.mol (1 N-CH3, 3-6 pyrrolidine C, 7-12 pyridine). */ +function nicotine() { + const c = (atoms, label, shift) => ({ nucleus: '13C', atoms, label, shift }) + const h = (atoms, label, shift, extra = {}) => ({ nucleus: '1H', atoms, label, shift, ...extra }) + return { + structure: { molfile: fixture('nicotine.mol') }, + conditions: { solvent: 'CDCl3' }, + assignments: [ + c([12], 'C-2', 149.91), + c([7], 'C-3', 138.64), + c([8], 'C-4', 135.15), + c([9], 'C-5', 123.35), + c([10], 'C-6', 148.46), + c([6], "C-2'", 68.94), + c([3], "C-5'", 56.88), + c([4], "C-4'", 22.53), + c([5], "C-3'", 35.34), + c([1], 'N-CH3', 40.35), + h([12], 'H-2', 8.55), + h([8], 'H-4', 7.69), + h([9], 'H-5', 7.24), + h([10], 'H-6', 8.5), + h([6], "H-2'", 3.08), + h([3], "H-5'a", 2.31, { n_h: 1, diastereotopic: 'a' }), + h([3], "H-5'b", 3.24, { n_h: 1, diastereotopic: 'b' }), + h([4], "H-4'a", 1.81, { n_h: 1, diastereotopic: 'a' }), + h([4], "H-4'b", 1.95, { n_h: 1, diastereotopic: 'b' }), + h([5], "H-3'a", 1.73, { n_h: 1, diastereotopic: 'a' }), + h([5], "H-3'b", 2.19, { n_h: 1, diastereotopic: 'b' }), + h([1], 'N-CH3', 2.17, { n_h: 3 }), + ], + } +} + +test('explicit hydrogens are stripped and heavy atoms renumbered', () => { + const structure = prepareStructure(fixture('trimethoxybenzaldehyde.mol')) + + assert.equal(structure.heavyAtomCount, 14) + assert.deepEqual(structure.toOriginal.slice(1), [1, 2, 3, 4, 5, 6, 7, 9, 10, 11, 12, 13, 14, 15]) + assert.equal(structure.explicitHydrogenParents.get(8), 7) + assert.doesNotMatch(structure.molfile, /^\s+\S+\s+\S+\s+\S+\s+H\s/m) + assert.doesNotMatch(structure.molfile, /M {2}ZZC/) +}) + +test('servlet hydrogen numbers follow the heavy atoms they belong to', () => { + const topology = analyseTopology(prepareStructure(fixture('nicotine.mol'))) + const owners = Object.fromEntries(topology.servletHydrogenOwners) + + assert.deepEqual( + Object.entries(owners).map(([hydrogen, owner]) => [Number(hydrogen), owner]), + [ + [13, 1], [14, 1], [15, 1], + [16, 3], [17, 3], [18, 4], [19, 4], [20, 5], [21, 5], + [22, 6], [23, 8], [24, 9], [25, 10], [26, 12], + ], + ) +}) + +test('a correct assignment gets mark 10 and a consistent result', async () => { + const { report, calls } = await run(trimethoxybenzaldehyde(), 'trimethoxybenzaldehyde-correct') + + assert.equal(calls.length, 1) + assert.deepEqual( + calls[0].inputs.map(({ id, type, solvent }) => ({ id, type, solvent })), + [ + { id: 1, type: 'nmr;13C;1d', solvent: 'Chloroform-D1 (CDCl3)' }, + { id: 2, type: 'nmr;1H;1d', solvent: 'Chloroform-D1 (CDCl3)' }, + ], + ) + assert.equal(calls[0].inputs[0].shifts.split(';').length, 7) + assert.equal(calls[0].inputs[1].shifts.split(';').length, 4) + + assert.equal(report.reports['13C'].mark, 10) + assert.equal(report.reports['13C'].result, 'accept') + assert.equal(report.reports['1H'].mark, 10) + assert.equal(report.assignment_check.result, 'consistent') + assert.equal(report.verdict, 'accept') + assert.deepEqual(report.assignment_check.suggestions, []) + + const labels = report.reports['13C'].atoms.map((row) => row.label) + assert.ok(labels.includes('C-2') && labels.includes('C-6'), `per-atom labels expected, got ${labels}`) + assert.equal(rowByLabel(report, 'H-7').atoms[0], 7) +}) + +test('a wrong carbon shift is flagged red and fails the assignment', async () => { + const { report } = await run(trimethoxybenzaldehyde({ 'C-4': 130 }), 'trimethoxybenzaldehyde-c4-wrong') + + const c4 = report.reports['13C'].atoms.find((row) => row.label === 'C-4') + assert.equal(c4.status, 'red') + assert.equal(report.reports['13C'].penalties.red_or_missing.count, 1) + assert.ok(report.reports['13C'].mark < 8) + assert.equal(rowByLabel(report, 'C-4').status, 'fail') + assert.equal(report.assignment_check.result, 'inconsistent') + assert.equal(report.verdict, 'reject') +}) + +test('swapped carbon labels keep a good structure fit but are caught and a swap is suggested', async () => { + const { report } = await run( + trimethoxybenzaldehyde({ 'C-1': 143.518, 'C-4': 131.677 }), + 'trimethoxybenzaldehyde-correct', + ) + + assert.equal(report.reports['13C'].mark, 10) + assert.equal(rowByLabel(report, 'C-1').status, 'fail') + assert.equal(rowByLabel(report, 'C-4').status, 'fail') + assert.equal(report.assignment_check.result, 'inconsistent') + assert.deepEqual(report.assignment_check.suggestions[0].labels.slice().sort(), ['C-1', 'C-4']) +}) + +test('nicotine: diastereotopic pairs are scored by their mean and low-sphere atoms never fail', async () => { + const { report, calls } = await run(nicotine(), 'nicotine') + + assert.equal(calls[0].inputs[1].shifts.split(';').length, 12) + assert.equal(report.reports['13C'].mark, 10) + + const servletFit = report.reports['1H'].atoms.filter((row) => row.atoms[0] === 3) + assert.deepEqual(servletFit.map((row) => row.label), ["H-5'"]) + + assert.equal(rowByLabel(report, "H-5'a").status, 'ok') + assert.equal(rowByLabel(report, "H-5'b").status, 'ok') + + const h2prime = rowByLabel(report, "H-2'") + assert.equal(h2prime.spheres, 3) + assert.equal(h2prime.status, 'review') + assert.ok(h2prime.reasons.includes('low_confidence_prediction')) + + assert.ok(report.assignment_check.rows.every((row) => row.status !== 'fail')) + assert.equal(report.assignment_check.result, 'review') +}) + +test('a CH2 matched to two distinct shifts is reported as an a/b pair sharing the mean deviation', async () => { + const response = structuredClone(responses.nicotine) + const proton = response.result.find((result) => result.id === 2) + proton.shifts.find((shift) => shift.atom === 16).real = 2.31 + proton.shifts.find((shift) => shift.atom === 17).real = 3.24 + responses['nicotine-split-pair'] = response + + const { report } = await run(nicotine(), 'nicotine-split-pair') + + const pairRows = report.reports['1H'].atoms.filter((row) => row.atoms[0] === 3) + assert.deepEqual(pairRows.map((row) => row.label), ["H-5'a", "H-5'b"]) + assert.deepEqual(pairRows.map((row) => row.observed), [2.31, 3.24]) + assert.equal(pairRows[0].deviation, pairRows[1].deviation) + assert.equal(pairRows[0].pair, "H-5'b") +}) + +test('unassigned C-H environments are reported by the author\'s carbon labels', async () => { + const input = nicotine() + input.assignments = input.assignments.filter((row) => !/^H-[34]'/.test(row.label)) + + const { report } = await run(input, 'nicotine') + + const missing = report.assignment_check.issues.find((issue) => issue.type === 'missing_signal') + assert.equal(missing.nucleus, '1H') + assert.deepEqual(missing.labels, ["H-4'", "H-3'"]) + assert.deepEqual(missing.atoms, [4, 5]) +}) + +test('distinct signals with the same shift are nudged so the servlet keeps both', async () => { + const input = trimethoxybenzaldehyde() + input.assignments.find((row) => row.label === 'H-9').shift = 3.926 + + const { report, calls } = await run(input, 'trimethoxybenzaldehyde-correct') + + assert.equal(calls[0].inputs[1].shifts.split(';').length, 4) + assert.equal(report.adjustments.length, 1) + assert.equal(report.adjustments[0].observed, 3.926) + assert.ok(Math.abs(report.adjustments[0].sent - 3.926) <= 0.0011) +}) + +test('invalid input and structures are rejected before calling the servlet', async () => { + const { client, calls } = fakeQuickcheck('nicotine') + const options = { quickcheck: client, url: 'https://example.test' } + + await assert.rejects(validateAssignments({ structure: { molfile: '' }, assignments: [] }, options), InvalidInputError) + await assert.rejects( + validateAssignments( + { structure: { molfile: 'not a molfile' }, assignments: [{ nucleus: '13C', atoms: [1], shift: 10 }] }, + options, + ), + InvalidStructureError, + ) + assert.equal(calls.length, 0) +}) diff --git a/docs/src/modules/validation.md b/docs/src/modules/validation.md index 6a83869..43c6aeb 100644 --- a/docs/src/modules/validation.md +++ b/docs/src/modules/validation.md @@ -1,17 +1,80 @@ # Validation Module -::: warning Planned feature -The validation module is not yet exposed via the REST API. Spectral assignment -validation is tracked in -[#15 — Validation module](https://github.com/NFDI4Chem/nmrkit/issues/15). +The validation module checks ¹H and ¹³C assignments of a structure against +[nmrshiftdb2 quickcheck](https://nmrshiftdb.nmr.uni-koeln.de/) HOSE-code +predictions. It is executed by **nmr-cli** (`validate-assignments`) inside the +`nmr-converter` container. + +**Base path:** `/latest/validate` + +The prediction is a reference, not the truth: a poor fit flags assignments for a +closer look, and the author keeps the final word. + +## Endpoints + +### `POST /assignments` + +**Request body:** + +```json +{ + "structure": { "molfile": "\n Mnova...\nM END", "source": "mnova" }, + "conditions": { "solvent": "CDCl3" }, + "assignments": [ + { "nucleus": "13C", "atoms": [1, 3], "label": "C-2, C-6", "shift": 106.65 }, + { "nucleus": "13C", "atoms": [5], "label": "C-4", "shift": 143.52 }, + { "nucleus": "1H", "atoms": [1, 3], "label": "H-2, H-6", "shift": 7.12, "n_h": 2 }, + { "nucleus": "1H", "atoms": [9], "label": "H-5'a", "shift": 2.31, "diastereotopic": "a" } + ], + "unassigned_peaks": [{ "nucleus": "13C", "shift": 77.16, "kind": "solvent" }] +} +``` + +| Field | Notes | +|-------|-------| +| `structure.molfile` | V2000. Explicit H atoms are allowed and are stripped before the servlet call. | +| `assignments[].atoms` | 1-based molfile indices. For ¹H either the carrying heavy atom or an explicit H. Equivalent atoms share one entry. | +| `assignments[].label` | Author label. `"C-2, C-6"` with two atoms gives one report row per atom. | +| `assignments[].diastereotopic` | Marks the two protons of a CH₂; the pair is compared by its mean. | +| `unassigned_peaks` | `unknown` peaks join the structure fit; `solvent` and `impurity` peaks are ignored. | +| `options.fallback_tolerances` | Per nucleus `{ok, fail}` in ppm (defaults: ¹³C 3/6, ¹H 0.3/0.6). | + +**Response:** a report with two layers and a combined verdict. + +| Key | Question it answers | +|-----|---------------------| +| `reports["13C"]`, `reports["1H"]` | Does the shift list fit the structure? Mark 1–10, `accept`/`revise`/`reject`, penalties, per-atom deviation, HOSE spheres and codes, `in_database_likely`. Mirrors the nmrshiftdb2 quality report. | +| `assignment_check` | Are the shifts on the right atoms? Status per assignment (`ok`, `review`, `fail`, `not_assessable`), swap suggestions, equivalence violations, missing signals, solvent peaks, proton counts, referencing offset. | +| `verdict` | `accept`, `review`, `reject` or `not_assessable`; ¹³C drives it. | +| `adjustments` | Shifts nudged by ±0.001 ppm so the servlet keeps distinct signals with identical values. | +| `cached` | Identical requests are served from a 24 h in-memory cache. | + +::: info Scoring +nmrshiftdb2 does not publish its mark formula. The mark is an approximation +(0.5 points per ppm mean deviation, 2 per red or missing atom, 1 per yellow atom, +halved for predictions with at most 2 spheres) and is flagged with +`mark_is_approximate`. Predictions with fewer than 4 HOSE spheres can lead to +`review`, never to `fail`. ::: -## Planned scope +**Errors:** + +| Status | Meaning | +|--------|---------| +| 422 | Invalid request, unsupported molfile or unknown atom indices | +| 408 | Validation timed out | +| 503 | nmrshiftdb2 quickcheck unavailable, retry later | +| 500 | Docker or `nmr-converter` not available | + +## CLI + +```bash +cat assignment-set.json | nmr-cli validate-assignments +``` -Implement validation scores for spectral assignments, verifying the quality and -consistency of experimental NMR data against predicted or reference values. +Exit codes: `2` invalid input, `3` quickcheck unavailable. ## Related modules -- [Prediction](./prediction) — generate reference spectra for comparison +- [Prediction](./prediction) — nmrshift engine used as the reference - [Spectra](./spectra) — parse experimental data for validation input diff --git a/tests/test_validate.py b/tests/test_validate.py new file mode 100644 index 0000000..2eb5606 --- /dev/null +++ b/tests/test_validate.py @@ -0,0 +1,147 @@ +import json +import subprocess + +import pytest +from fastapi.testclient import TestClient + +from app.main import app +from app.routers import validate + +client = TestClient(app) + +MOLFILE = """ + CDK 08302311362D + + 4 3 0 0 0 0 0 0 0 0999 V2000 + 0.9743 0.5625 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -0.3248 1.3125 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + -0.3248 2.8125 0.0000 O 0 0 0 0 0 0 0 0 0 0 0 0 + -1.6238 0.5625 0.0000 C 0 0 0 0 0 0 0 0 0 0 0 0 + 1 2 1 0 0 0 0 + 2 3 2 0 0 0 0 + 2 4 1 0 0 0 0 +M END +""" + +REPORT = { + "engine": {"name": "nmrshift", "source": "nmrshiftdb2 quickcheck", "url": "https://example.test"}, + "solvent": "Chloroform-D1 (CDCl3)", + "verdict": "accept", + "reports": {"13C": {"mark": 10, "result": "accept"}}, + "assignment_check": {"result": "consistent", "rows": [], "suggestions": [], "issues": []}, + "adjustments": [], +} + + +def acetone_request(): + return { + "structure": {"molfile": MOLFILE, "source": "manual"}, + "conditions": {"solvent": "CDCl3"}, + "assignments": [ + {"nucleus": "13C", "atoms": [1, 4], "label": "CH3", "shift": 30.8}, + {"nucleus": "13C", "atoms": [2], "label": "C=O", "shift": 206.7}, + {"nucleus": "1H", "atoms": [1, 4], "label": "CH3", "shift": 2.17, "n_h": 6}, + ], + } + + +class FakeCli: + def __init__(self, returncode=0, stdout=b"", stderr=b""): + self.returncode = returncode + self.stdout = stdout + self.stderr = stderr + self.calls = [] + + def __call__(self, cmd, **kwargs): + self.calls.append({"cmd": cmd, **kwargs}) + return subprocess.CompletedProcess(cmd, self.returncode, self.stdout, self.stderr) + + +@pytest.fixture(autouse=True) +def empty_cache(): + validate.report_cache.clear() + yield + validate.report_cache.clear() + + +def install_cli(monkeypatch, fake): + monkeypatch.setattr(validate.subprocess, "run", fake) + return fake + + +def test_validate_index(): + response = client.get("/latest/validate/") + assert response.status_code == 200 + assert response.json() == {"status": "OK"} + + +def test_assignments_are_piped_to_nmr_cli_and_the_report_is_returned(monkeypatch): + fake = install_cli(monkeypatch, FakeCli(stdout=json.dumps(REPORT).encode())) + + response = client.post("/latest/validate/assignments", json=acetone_request()) + + assert response.status_code == 200 + body = response.json() + assert body["verdict"] == "accept" + assert body["cached"] is False + + call = fake.calls[0] + assert call["cmd"] == ["docker", "exec", "-i", "nmr-converter", "nmr-cli", "validate-assignments"] + sent = json.loads(call["input"].decode()) + assert sent["structure"]["molfile"] == MOLFILE + assert sent["assignments"][2] == {"nucleus": "1H", "atoms": [1, 4], "label": "CH3", "shift": 2.17, "n_h": 6} + + +def test_identical_requests_are_served_from_cache(monkeypatch): + fake = install_cli(monkeypatch, FakeCli(stdout=json.dumps(REPORT).encode())) + + client.post("/latest/validate/assignments", json=acetone_request()) + response = client.post("/latest/validate/assignments", json=acetone_request()) + + assert response.status_code == 200 + assert response.json()["cached"] is True + assert len(fake.calls) == 1 + + +def test_invalid_structure_from_cli_maps_to_422(monkeypatch): + error = {"error": "invalid_input", "message": "Only V2000 molfiles are supported"} + install_cli(monkeypatch, FakeCli(returncode=2, stderr=json.dumps(error).encode())) + + response = client.post("/latest/validate/assignments", json=acetone_request()) + + assert response.status_code == 422 + assert response.json()["detail"] == error + + +def test_unavailable_quickcheck_maps_to_503_and_is_not_cached(monkeypatch): + error = {"error": "quickcheck_unavailable", "message": "timeout of 60000ms exceeded"} + fake = install_cli(monkeypatch, FakeCli(returncode=3, stderr=json.dumps(error).encode())) + + first = client.post("/latest/validate/assignments", json=acetone_request()) + second = client.post("/latest/validate/assignments", json=acetone_request()) + + assert first.status_code == 503 + assert second.status_code == 503 + assert len(fake.calls) == 2 + + +def test_request_schema_is_validated_before_calling_the_cli(monkeypatch): + fake = install_cli(monkeypatch, FakeCli(stdout=json.dumps(REPORT).encode())) + payload = acetone_request() + payload["assignments"][0]["nucleus"] = "15N" + + response = client.post("/latest/validate/assignments", json=payload) + + assert response.status_code == 422 + assert fake.calls == [] + + +def test_missing_docker_maps_to_500(monkeypatch): + def missing_docker(cmd, **kwargs): + raise FileNotFoundError("docker") + + monkeypatch.setattr(validate.subprocess, "run", missing_docker) + + response = client.post("/latest/validate/assignments", json=acetone_request()) + + assert response.status_code == 500 From 387e6defb613db71a2cd6f752cb236b606d626f0 Mon Sep 17 00:00:00 2001 From: Venkata Nainala Date: Mon, 28 Sep 2026 17:57:14 +0100 Subject: [PATCH 2/3] feat: resolve NMRium implicit-hydrogen atom numbers in validate-assignments --- app/scripts/nmr-cli/src/validation/types.ts | 4 +++- app/scripts/nmr-cli/src/validation/validate.ts | 5 +++++ app/scripts/nmr-cli/test/validation.test.cjs | 15 +++++++++++++++ docs/src/modules/validation.md | 2 +- 4 files changed, 24 insertions(+), 2 deletions(-) diff --git a/app/scripts/nmr-cli/src/validation/types.ts b/app/scripts/nmr-cli/src/validation/types.ts index 9cd07bb..e956161 100644 --- a/app/scripts/nmr-cli/src/validation/types.ts +++ b/app/scripts/nmr-cli/src/validation/types.ts @@ -10,7 +10,9 @@ export interface Tolerance { /** * One observed signal. `atoms` are 1-based indices of the submitted molfile. * For 1H, an index may point at the heavy atom carrying the protons - * (NMReDATA convention) or at an explicit H atom. An empty `atoms` list is a + * (NMReDATA convention), at an explicit H atom, or past the last molfile atom + * at an implicit H numbered heavy atom by heavy atom, as openchemlib's + * `addImplicitHydrogens` does (NMRium convention). An empty `atoms` list is a * signal without assignment; it only contributes to the structure fit. */ export interface AssignmentInput { diff --git a/app/scripts/nmr-cli/src/validation/validate.ts b/app/scripts/nmr-cli/src/validation/validate.ts index 97435f0..511818a 100644 --- a/app/scripts/nmr-cli/src/validation/validate.ts +++ b/app/scripts/nmr-cli/src/validation/validate.ts @@ -179,6 +179,11 @@ function resolveCarrier( if (nucleus === '1H') { const parent = structure.explicitHydrogenParents.get(atom) if (parent !== undefined) return parent + + const atomCount = structure.symbols.size + if (atom > atomCount) { + return topology.servletHydrogenOwners.get(atom - atomCount + structure.heavyAtomCount) ?? null + } return (topology.hydrogenCounts.get(atom) ?? 0) > 0 ? atom : null } diff --git a/app/scripts/nmr-cli/test/validation.test.cjs b/app/scripts/nmr-cli/test/validation.test.cjs index 76bc2b3..fbd62ee 100644 --- a/app/scripts/nmr-cli/test/validation.test.cjs +++ b/app/scripts/nmr-cli/test/validation.test.cjs @@ -207,6 +207,21 @@ test('unassigned C-H environments are reported by the author\'s carbon labels', assert.deepEqual(missing.atoms, [4, 5]) }) +test('1H atoms past the molfile resolve as implicit hydrogens (NMRium numbering)', async () => { + const input = nicotine() + const implicitHydrogens = { "H-2'": [22], "H-5'a": [16], "H-5'b": [17], 'N-CH3': [13, 14, 15] } + input.assignments = input.assignments.map((row) => + row.nucleus === '1H' && implicitHydrogens[row.label] ? { ...row, atoms: implicitHydrogens[row.label] } : row, + ) + + const { report } = await run(input, 'nicotine') + + assert.deepEqual(rowByLabel(report, "H-2'").atoms, [6]) + assert.deepEqual(rowByLabel(report, "H-5'a").atoms, [3]) + assert.deepEqual(report.assignment_check.rows.find((row) => row.nucleus === '1H' && row.label === 'N-CH3').atoms, [1]) + assert.ok(!report.assignment_check.issues.some((issue) => issue.type === 'unknown_atom')) +}) + test('distinct signals with the same shift are nudged so the servlet keeps both', async () => { const input = trimethoxybenzaldehyde() input.assignments.find((row) => row.label === 'H-9').shift = 3.926 diff --git a/docs/src/modules/validation.md b/docs/src/modules/validation.md index 43c6aeb..e668057 100644 --- a/docs/src/modules/validation.md +++ b/docs/src/modules/validation.md @@ -33,7 +33,7 @@ closer look, and the author keeps the final word. | Field | Notes | |-------|-------| | `structure.molfile` | V2000. Explicit H atoms are allowed and are stripped before the servlet call. | -| `assignments[].atoms` | 1-based molfile indices. For ¹H either the carrying heavy atom or an explicit H. Equivalent atoms share one entry. | +| `assignments[].atoms` | 1-based molfile indices. For ¹H: the carrying heavy atom, an explicit H, or an index past the last molfile atom for an implicit H numbered heavy atom by heavy atom (openchemlib `addImplicitHydrogens`, as NMRium does). Equivalent atoms share one entry. | | `assignments[].label` | Author label. `"C-2, C-6"` with two atoms gives one report row per atom. | | `assignments[].diastereotopic` | Marks the two protons of a CH₂; the pair is compared by its mean. | | `unassigned_peaks` | `unknown` peaks join the structure fit; `solvent` and `impurity` peaks are ignored. | From 69cc3a898747317c84aa700b7ee81a23bef3686e Mon Sep 17 00:00:00 2001 From: Venkata Nainala Date: Mon, 28 Sep 2026 20:58:37 +0100 Subject: [PATCH 3/3] fix: build the per-atom quality report from the author's assignments The quickcheck servlet only receives a shift list and matches it to atoms itself, so the quality report could pair a value with another atom than the author did (2,6-dimethylphenol: the OH shift landed on the aromatic CHs) and disagree with the assignment check. Observed values and statuses now come from the assignment check rows, and a nucleus with a red atom is at best 'revise'. --- app/routers/validate.py | 2 +- .../nmr-cli/src/validation/validate.ts | 69 ++++++++++++++----- app/scripts/nmr-cli/test/validation.test.cjs | 54 +++++++++++---- docs/src/modules/validation.md | 7 +- 4 files changed, 96 insertions(+), 36 deletions(-) diff --git a/app/routers/validate.py b/app/routers/validate.py index 61cbfac..141a543 100644 --- a/app/routers/validate.py +++ b/app/routers/validate.py @@ -245,7 +245,7 @@ def get_health() -> HealthCheck: "The response has two layers:\n\n" "| Layer | Question it answers |\n" "|-------|--------------------|\n" - "| `reports` | Does the shift list fit the structure? (nmrshiftdb2 quality report: mark 1–10, per-atom deviation, HOSE spheres) |\n" + "| `reports` | How well do the assigned shifts fit, atom by atom? (quality report on the author's assignments: mark 1–10, per-atom deviation, HOSE spheres) |\n" "| `assignment_check` | Are the shifts on the right atoms? (per-assignment status, swap suggestions, equivalence, missing signals, referencing offset) |\n\n" "`verdict` combines both, 13C first. Predictions with fewer than 4 HOSE spheres can " "at most lead to `review`, never `fail`. The mark formula approximates nmrshiftdb2 " diff --git a/app/scripts/nmr-cli/src/validation/validate.ts b/app/scripts/nmr-cli/src/validation/validate.ts index 511818a..e57517c 100644 --- a/app/scripts/nmr-cli/src/validation/validate.ts +++ b/app/scripts/nmr-cli/src/validation/validate.ts @@ -54,6 +54,13 @@ const IN_DATABASE_MEAN_DEVIATION: Record = { '13C': 1, '1H': 0. */ const MARK_POINTS = { perPpmMeanDeviation: 0.5, redOrMissing: 2, yellow: 1 } +const ATOM_STATUS: Record = { + ok: 'green', + review: 'yellow', + fail: 'red', + not_assessable: 'yellow', +} + export interface ValidateOptions { quickcheck: QuickcheckClient url: string @@ -100,17 +107,15 @@ export async function validateAssignments( ? await options.quickcheck(structure.molfile, quickcheckInputs) : [] const predictions = collectPredictions(results, structure, topology) + const assignmentCheck = checkAssignments(rows, predictions, topology, structure, input, tolerances, solvent.residual) const reports: ValidationReport['reports'] = {} for (const nucleus of NUCLEI) { - const result = results.find((item) => item.id === QUICKCHECK_TYPES[nucleus].id) - if (result) { - reports[nucleus] = buildNucleusReport(nucleus, result, predictions.get(nucleus)!, rows) + if (results.some((item) => item.id === QUICKCHECK_TYPES[nucleus].id)) { + reports[nucleus] = buildNucleusReport(nucleus, predictions.get(nucleus)!, rows, assignmentCheck.rows, topology) } } - const assignmentCheck = checkAssignments(rows, predictions, topology, structure, input, tolerances, solvent.residual) - return { engine: { name: 'nmrshift', source: 'nmrshiftdb2 quickcheck', url: options.url }, solvent: solvent.nmrshiftdb, @@ -305,11 +310,19 @@ function collectPredictions( return predictions } +/** + * Per-atom quality report on the author's assignments. The servlet only gets + * a shift list and matches it to atoms itself, so its own matching can pair a + * value with a different atom than the author did; observed values and + * statuses therefore come from the assignment check, keeping both tables in + * agreement. + */ function buildNucleusReport( nucleus: Nucleus, - result: QuickcheckResult, byAtom: Map, rows: NormalizedRow[], + checked: AssignmentRowResult[], + topology: Topology, ): NucleusReport { const atomRows: ReportAtomRow[] = [] @@ -323,20 +336,21 @@ function buildNucleusReport( hose_code: entry.hoseCode, } - if (entry.statuses.every((status) => status === 'impossible')) { + if (entry.prediction === null) { atomRows.push({ ...base, label, observed: null, deviation: null, status: 'impossible' }) continue } - if (entry.reals.length === 0) { + const covering = rowsAssignedTo(atom, nucleus, rows, topology) + if (covering.length === 0) { atomRows.push({ ...base, label, observed: null, deviation: null, status: 'missing' }) continue } - const status = worstStatus(entry.statuses) - const observed = [...new Set(entry.reals.map((value) => round(value, 4)))].sort((a, b) => a - b) - const prediction = entry.prediction ?? 0 + const status = worstStatus(covering.map((row) => ATOM_STATUS[checked[row.index].status])) + const observed = [...new Set(covering.map((row) => round(row.input.shift, 4)))].sort((a, b) => a - b) + const prediction = entry.prediction - if (nucleus === '1H' && observed.length === 2) { + if (nucleus === '1H' && observed.length === 2 && (topology.hydrogenCounts.get(atom) ?? 0) >= 2) { const stem = label.replace(/[ab]$/, '') const deviation = round(Math.abs(mean(observed) - prediction), 3) atomRows.push({ ...base, label: `${stem}a`, observed: observed[0], deviation, status, pair: `${stem}b` }) @@ -360,17 +374,24 @@ function buildNucleusReport( const mark = Math.min(10, Math.max(1, Math.round(10 - deviationPoints - redPoints - yellowPoints))) const sixSphereShare = scored.length ? scored.filter((row) => row.spheres >= 6).length / scored.length : 0 + const red = atomRows.filter((row) => row.status === 'red') return { mark, mark_is_approximate: true, - result: mark >= 8 ? 'accept' : mark >= 5 ? 'revise' : 'reject', + result: mark >= 8 && red.length === 0 ? 'accept' : mark >= 5 ? 'revise' : 'reject', penalties: { mean_deviation: { ppm: round(meanDeviation, 2), points: deviationPoints }, red_or_missing: { count: redOrMissing.length, points: redPoints }, yellow: { count: yellow.length, points: yellowPoints }, }, - statistics: result.statistics, + statistics: { + accept: atomRows.filter((row) => row.status === 'green').length, + warning: yellow.length, + reject: red.length, + missing: atomRows.filter((row) => row.status === 'missing').length, + total: atomRows.length, + }, in_database_likely: scored.length >= 3 && sixSphereShare >= 0.8 && @@ -408,9 +429,23 @@ function labelForAtom(atom: number, nucleus: Nucleus, rows: NormalizedRow[]): st return `${row.label} [${atom}]` } -function worstStatus(statuses: string[]): ReportStatus { - if (statuses.includes('reject')) return 'red' - if (statuses.includes('warning')) return 'yellow' +/** + * The author's rows for an atom; an atom left out of a row that covers a + * symmetry-equivalent atom ("C-2" for both C-2 and C-6) shares that row. + */ +function rowsAssignedTo(atom: number, nucleus: Nucleus, rows: NormalizedRow[], topology: Topology): NormalizedRow[] { + const nucleusRows = rows.filter((row) => row.input.nucleus === nucleus) + const own = nucleusRows.filter((row) => row.carriers.includes(atom)) + if (own.length) return own + + const cls = topology.classes.get(atom) + if (cls === undefined) return [] + return nucleusRows.filter((row) => row.carriers.some((carrier) => topology.classes.get(carrier) === cls)) +} + +function worstStatus(statuses: ReportStatus[]): ReportStatus { + if (statuses.includes('red')) return 'red' + if (statuses.includes('yellow')) return 'yellow' return 'green' } diff --git a/app/scripts/nmr-cli/test/validation.test.cjs b/app/scripts/nmr-cli/test/validation.test.cjs index fbd62ee..3fcf93f 100644 --- a/app/scripts/nmr-cli/test/validation.test.cjs +++ b/app/scripts/nmr-cli/test/validation.test.cjs @@ -145,13 +145,18 @@ test('a wrong carbon shift is flagged red and fails the assignment', async () => assert.equal(report.verdict, 'reject') }) -test('swapped carbon labels keep a good structure fit but are caught and a swap is suggested', async () => { +test('swapped carbon labels are red in the quality report, fail the assignment and suggest a swap', async () => { const { report } = await run( trimethoxybenzaldehyde({ 'C-1': 143.518, 'C-4': 131.677 }), 'trimethoxybenzaldehyde-correct', ) - assert.equal(report.reports['13C'].mark, 10) + const carbons = Object.fromEntries(report.reports['13C'].atoms.map((row) => [row.label, row])) + assert.equal(carbons['C-1'].observed, 143.518) + assert.equal(carbons['C-1'].status, 'red') + assert.equal(carbons['C-4'].status, 'red') + assert.ok(report.reports['13C'].mark < 8) + assert.notEqual(report.reports['13C'].result, 'accept') assert.equal(rowByLabel(report, 'C-1').status, 'fail') assert.equal(rowByLabel(report, 'C-4').status, 'fail') assert.equal(report.assignment_check.result, 'inconsistent') @@ -164,8 +169,9 @@ test('nicotine: diastereotopic pairs are scored by their mean and low-sphere ato assert.equal(calls[0].inputs[1].shifts.split(';').length, 12) assert.equal(report.reports['13C'].mark, 10) - const servletFit = report.reports['1H'].atoms.filter((row) => row.atoms[0] === 3) - assert.deepEqual(servletFit.map((row) => row.label), ["H-5'"]) + const pairRows = report.reports['1H'].atoms.filter((row) => row.atoms[0] === 3) + assert.deepEqual(pairRows.map((row) => row.label), ["H-5'a", "H-5'b"]) + assert.deepEqual(pairRows.map((row) => row.observed), [2.31, 3.24]) assert.equal(rowByLabel(report, "H-5'a").status, 'ok') assert.equal(rowByLabel(report, "H-5'b").status, 'ok') @@ -179,20 +185,38 @@ test('nicotine: diastereotopic pairs are scored by their mean and low-sphere ato assert.equal(report.assignment_check.result, 'review') }) -test('a CH2 matched to two distinct shifts is reported as an a/b pair sharing the mean deviation', async () => { - const response = structuredClone(responses.nicotine) +test('the quality report follows the author\'s assignments, not the servlet\'s own matching', async () => { + const response = structuredClone(responses['trimethoxybenzaldehyde-correct']) const proton = response.result.find((result) => result.id === 2) - proton.shifts.find((shift) => shift.atom === 16).real = 2.31 - proton.shifts.find((shift) => shift.atom === 17).real = 3.24 - responses['nicotine-split-pair'] = response + for (const shift of proton.shifts) { + if (shift.atom === 15 || shift.atom === 16) shift.real = 9.862 + if (shift.atom === 17) shift.real = 7.123 + } + responses['trimethoxybenzaldehyde-rematched'] = response - const { report } = await run(nicotine(), 'nicotine-split-pair') + const { report } = await run(trimethoxybenzaldehyde(), 'trimethoxybenzaldehyde-rematched') - const pairRows = report.reports['1H'].atoms.filter((row) => row.atoms[0] === 3) - assert.deepEqual(pairRows.map((row) => row.label), ["H-5'a", "H-5'b"]) - assert.deepEqual(pairRows.map((row) => row.observed), [2.31, 3.24]) - assert.equal(pairRows[0].deviation, pairRows[1].deviation) - assert.equal(pairRows[0].pair, "H-5'b") + const protons = report.reports['1H'].atoms + assert.equal(protons.find((row) => row.atoms[0] === 7).observed, 9.862) + assert.equal(protons.find((row) => row.atoms[0] === 1).observed, 7.123) + assert.ok(protons.every((row) => row.status === 'green')) + assert.equal(report.reports['1H'].result, 'accept') + assert.equal(report.assignment_check.result, 'consistent') +}) + +test('one proton that does not match keeps the nucleus from a good fit', async () => { + const input = trimethoxybenzaldehyde() + input.assignments.find((row) => row.label === 'H-7').shift = 8.2 + + const { report } = await run(input, 'trimethoxybenzaldehyde-correct') + + const aldehyde = report.reports['1H'].atoms.find((row) => row.label === 'H-7') + assert.equal(aldehyde.observed, 8.2) + assert.equal(aldehyde.status, 'red') + assert.equal(rowByLabel(report, 'H-7').status, 'fail') + assert.equal(report.reports['1H'].mark, 8) + assert.equal(report.reports['1H'].result, 'revise') + assert.equal(report.reports['1H'].statistics.reject, 1) }) test('unassigned C-H environments are reported by the author\'s carbon labels', async () => { diff --git a/docs/src/modules/validation.md b/docs/src/modules/validation.md index e668057..7d8aada 100644 --- a/docs/src/modules/validation.md +++ b/docs/src/modules/validation.md @@ -43,7 +43,7 @@ closer look, and the author keeps the final word. | Key | Question it answers | |-----|---------------------| -| `reports["13C"]`, `reports["1H"]` | Does the shift list fit the structure? Mark 1–10, `accept`/`revise`/`reject`, penalties, per-atom deviation, HOSE spheres and codes, `in_database_likely`. Mirrors the nmrshiftdb2 quality report. | +| `reports["13C"]`, `reports["1H"]` | How well do the assigned shifts fit, atom by atom? Mark 1–10, `accept`/`revise`/`reject`, penalties, per-atom deviation, HOSE spheres and codes, `in_database_likely`. Laid out like the nmrshiftdb2 quality report, but on the author's assignments: each atom carries its assigned shift and the status of its `assignment_check` row, so both layers always agree. | | `assignment_check` | Are the shifts on the right atoms? Status per assignment (`ok`, `review`, `fail`, `not_assessable`), swap suggestions, equivalence violations, missing signals, solvent peaks, proton counts, referencing offset. | | `verdict` | `accept`, `review`, `reject` or `not_assessable`; ¹³C drives it. | | `adjustments` | Shifts nudged by ±0.001 ppm so the servlet keeps distinct signals with identical values. | @@ -53,8 +53,9 @@ closer look, and the author keeps the final word. nmrshiftdb2 does not publish its mark formula. The mark is an approximation (0.5 points per ppm mean deviation, 2 per red or missing atom, 1 per yellow atom, halved for predictions with at most 2 spheres) and is flagged with -`mark_is_approximate`. Predictions with fewer than 4 HOSE spheres can lead to -`review`, never to `fail`. +`mark_is_approximate`. A nucleus with a red atom is at best `revise`, whatever +its mark. Predictions with fewer than 4 HOSE spheres can lead to `review`, +never to `fail`. ::: **Errors:**