From d28161197279e7701da445839245ed1e7c4b37c5 Mon Sep 17 00:00:00 2001 From: Javier Garcia Ordonez Date: Wed, 19 Aug 2026 15:55:48 +0200 Subject: [PATCH 1/5] feat(api): add the itis_sumo.api consumer contract (cross-validate, grid/axes evaluation, Sobol, correlation, uncertainty propagation, MOGA, sampling) --- SPEC.md | 52 +- docs/how-to/cross-validate.md | 6 +- docs/how-to/moga.md | 7 +- docs/how-to/sensitivity-analysis.md | 10 +- docs/reference/api.md | 94 ++++ docs/reference/glossary.md | 54 ++ docs/tutorials/examples.md | 12 +- docs/tutorials/getting-started.md | 21 +- examples/headless_smoke.py | 17 +- mkdocs.yml | 2 + src/itis_sumo/api/__init__.py | 85 +++ src/itis_sumo/api/_session.py | 655 ++++++++++++++++++++++++ src/itis_sumo/api/errors.py | 58 +++ src/itis_sumo/api/types.py | 214 ++++++++ src/itis_sumo/api/workflows.py | 349 +++++++++++++ src/itis_sumo/core/__init__.py | 22 +- src/itis_sumo/evaluate/__init__.py | 4 + src/itis_sumo/evaluate/funs_evaluate.py | 164 +++++- tests/test_api_contract.py | 335 ++++++++++++ tests/test_api_workflows.py | 262 ++++++++++ tests/test_dakota_funs_evaluate.py | 35 +- 21 files changed, 2424 insertions(+), 34 deletions(-) create mode 100644 docs/reference/api.md create mode 100644 docs/reference/glossary.md create mode 100644 src/itis_sumo/api/__init__.py create mode 100644 src/itis_sumo/api/_session.py create mode 100644 src/itis_sumo/api/errors.py create mode 100644 src/itis_sumo/api/types.py create mode 100644 src/itis_sumo/api/workflows.py create mode 100644 tests/test_api_contract.py create mode 100644 tests/test_api_workflows.py diff --git a/SPEC.md b/SPEC.md index dacbc5e..903a484 100644 --- a/SPEC.md +++ b/SPEC.md @@ -11,13 +11,26 @@ Caveman-encoded (drop articles/filler; `→` becomes, `!` must, `?` may/uncertai - future consumer → `../mmux_vite/flaskapi` (swaps in-tree core → this pkg) - research → `DAKOTA-STUBS.md` (stubs + JSON→pydantic POC) +## VOCAB +Canonical nouns for ∀ public API, docs, tests, commit messages. One word per concept, ⊥ synonyms drifting back in. +- **sample** → one row of tabular data (an observation). ⊥ "job", ⊥ "point", ⊥ "record" +- **variable** = **parameter** → one INPUT column +- **response** = **quantity-of-interest (QoI)** → one OUTPUT column +- variables vs responses ! stay distinguished — they get different downstream treatment (normalization, sign, sampling, Sobol roles) +- **domain** → where we may draw/explore a variable (bounds + scale). Property of the design space, auto-inferable from existing samples +- **distribution** → real-world shape of a variable (a claim about the world). ⊥ inferable from a training design +- **surrogate** = **SuMo** → the trained metamodel +- oSPARC's `FunctionJob` is ⊥ itis-sumo vocabulary; it lives on the consumer side of the boundary + ## §G Standalone pkg `itis-sumo` (SuMo = Surrogate Model): aggregate MetaModeling core from mmux/vite + prior trials (mmux_python, itis_dakota_projects) into one wheel-driven library, usable BOTH inside mmux/vite AND headless (notebook/scripts) → build/evaluate/cross-validate surrogates + UQ (Sobol/correlation) + sampling (LHS/grid) + MOGA. First expansion (E1): SuMo model export/import persistence. +Consumer contract (grill 2026-08-18): itis-sumo owns ALL surrogate machinery end-to-end — training-file creation, preprocessing, Dakota config, run dirs, inverse transforms. Consumers (flaskapi OR a direct python user) pass ONLY tabular samples + configuration, and receive typed results in original units. + ## §C - engine ! itis-dakota wheel `dakota.environment.study(input_string=...)`; ⊥ subprocess `dakota` bin, ⊥ vendor surfpack; in-repo `itis-dakota/` fork ! REFERENCE-ONLY (⊥ build; runtime = PyPI wheel) - v1 config ! NIDR string composers (proven); JSON/pydantic seam later (DAKOTA-STUBS POC, Dakota 6.24 experimental) -- port ! near-verbatim from mmux_flaskapi `dakota/`+`data_preprocessor/`; re-root `mmux_flaskapi.`→`itis_sumo.`; prune known-dead (`process_json_file`, `add_interface_s4l`, `create_function_sampling_conffile`, `create_moga_iterative_optimization_conffile`); de-web `JobVariableSelection`/`required_completed_jobs` (move from blueprints into pkg) +- port ! near-verbatim from mmux_flaskapi `dakota/`+`data_preprocessor/`; re-root `mmux_flaskapi.`→`itis_sumo.`; prune known-dead (`process_json_file`, `add_interface_s4l`, `create_function_sampling_conffile`, `create_moga_iterative_optimization_conffile`); de-web `JobVariableSelection`/`required_completed_jobs` (moved from blueprints into pkg) — SUPERSEDED in part by the tabular-only boundary below: `required_completed_jobs`' Dakota rule stays, the job-shaped models go back to the consumer (T24cm) - recycle itis_dakota_projects: `validate_dakota_installation()`, `get_dakota_version()`, `validate` CLI, config sanity-guard; ⊥ subprocess `DakotaProject`/factory/generator/parser/activity-logger - deps ! itis-dakota==1.5.9 (Dakota 6.20, parity w/ mmux/vite; stay until T16mo resolves the 6.23+ `interface_cache` regression — R5), numpy, pandas, scipy; scikit-learn ⊥ KFold only - py.typed; `src/` layout; docs MkDocs + headless notebook; tests standalone ⊥ flask/osparc @@ -30,11 +43,24 @@ Standalone pkg `itis-sumo` (SuMo = Surrogate Model): aggregate MetaModeling core - E1: artifacts keyed server `sumo_model_id` (uuid); metadata sidecar `{id}.metadata.json`; ⊥ user-supplied path/prefix keys (traversal) - py ! 3.11 (match mmux/vite; 1.5.9 ships no cp313 wheel); 3.13 only via T16mo rung 1 (1.5.11 cp313) or rung 2 (6.24) - cross-repo call surface ! narrow, versioned, "deep" entrypoints — one stitched public function per feature (e.g. `analyze_dataset()`), ⊥ flaskapi orchestrating several itis-sumo internals itself; version-pinning itis-sumo is fine (same pattern as the itis-dakota engine pin, T16mo) — the coupling risk is a *wide* call surface, not the pin +- data in ! plain tabular (`pd.DataFrame` or an itis-sumo dataclass trivially convertible to one); ⊥ `FunctionJob`/oSPARC job concepts anywhere in itis_sumo. Ownership split: consumer converts jobs→table AND filters un-completed samples (warning); itis-sumo rejects insufficient/invalid data (raises; consumer catches + surfaces). Dakota sufficiency rule `max(5, n_variables + 1)` stays in itis-sumo +- preprocessing ! auto-defaulted (novice/normal users pass nothing); advanced override allowed but expressed in DOMAIN vocabulary only (`scale`, `direction`); ⊥ transform vocabulary in any public signature (`sign_switch`, `normalization="zscore"`, `mapped_name`, `_hat`/`_std_hat`). Effective config → INSPECTABLE in results, ⊥ SETTABLE in transform terms. MOGA `direction` defaults `"minimize"` (optimization-field convention) +- `domain` ⊥ conflated w/ `distribution` (see VOCAB): domain auto-inferable from samples (observed bounds + detected scale) → Sobol box + MOGA search space; distribution ! stated by the modeller → UQ propagation + correlation. Retires the `mean ± 3σ` fallback in `_bounds_from_distributions` (a conflation artifact — its own docstring admits the two uses differ). Lands in the handle transformation (T27fr), ⊥ the port +- architecture ! handle-primary is the TARGET (`fit(samples, config)` → `model.along_axes/grid/sobol/propagate_uq/...`, `model.save() -> sumo_model_id`); the mmux/vite port ships the current ONE-SHOT shape (no frontend change). One-shot entrypoints ! implemented internally as fit-then-query from day one → handle extraction later = re-export, ⊥ second port +- run artifacts ! ephemeral on success (⊥ unbounded growth in a web service), PRESERVED on failure w/ `run_dir` + captured stderr tail attached to the raised error (B1/B2/`Unknown error 250` are all Dakota-opacity bugs — the run dir is the only tractable evidence); `workspace=` opt-in for deliberate debugging. E1 model store (`ITIS_SUMO_MODELS_DIR`) = separate, always-persistent concern +- reproducibility ! fixed default seed `42`, overridable; effective seed returned in the result; ⊥ require caller to supply one +- errors ! stable taxonomy `SumoError` → `SumoInputError`(→400) / `SumoResultError`(→422, predictions missing) / `SumoEngineError`(→500, carries run_dir + stderr tail); ⊥ consumer classifying by string/regex match (retires the MOGA `re.search` branch) +- results ! typed dataclasses in ORIGINAL units + ORIGINAL names, JSON-serializable via `dataclasses.asdict()` (same convention as `DatasetDiagnostics`); ⊥ magic suffix keys in the public shape +- `?` should **distribution** get an auto-generated default the way domain does? User: "probably a good idea but should be explicitly discussed, not a given" → POST-PORT decision (T28gs), ⊥ guess +- `?` `get_sumo_cv_accuracy_metrics` is the ONE workflow bypassing DataPreprocessor entirely (raw var names + `process_input_file`) — normalize it (changes numbers ⇒ compat decision) or preserve + document? +- `?` public name/shape of the tabular dataclass vs raw `pd.DataFrame` ## §I - wheel: `dakota.environment.study`, `study.execute()` (⊖ `dakota.surrogates` until wheel bump, R4) - CLI: `itis-sumo validate` -- python: `itis_sumo.{core,config,data,sampling,evaluate,preprocess,utils}` public funcs (+ forthcoming `analyze_dataset` narrow diagnostics entrypoint, T18ry) +- python (consumer-facing) → `itis_sumo.api` = THE module flaskapi/mmux_vite imports; nothing else. Tabular in, typed results out, taxonomy errors (T22ax) +- python (in-package/headless) → `itis_sumo.{core,config,data,sampling,evaluate,preprocess,utils}` public funcs; `core` now re-exports the E1 model store, `evaluate` now re-exports `export_sumo_model`/`import_sumo_model` (were reachable only via `funs_evaluate`, breaking V16qf) +- python (forthcoming) → `analyze_dataset` narrow diagnostics entrypoint (T18ry) — its output IS the override payload for the preprocessing defaults AND the auto-`domain` source, ⊥ a side feature - artifacts: run_dir `{dakota_stdout.txt,dakota_stderr.txt,*.dat,*.sps/.alg}`; models dir `{id}.metadata.json` (E1) ## §V @@ -56,6 +82,15 @@ V15zx (proposed, T17bq): `add_surrogate_model` training-file header layout ! exp V16qf: itis-sumo's cross-repo-facing API (flaskapi/mmux_vite) ! consumed only via §I's documented top-level entrypoints; ⊥ consumer reaches into internal submodules/functions directly — keeps refactors inside itis-sumo from forcing coordinated cross-repo changes; grows w/ each new feature surfaced (starts w/ `analyze_dataset`, T18ry) V17ab: ∀ `.py` file in repo ! `ruff check` + `ruff format --check` clean (CI-enforced via `prek` on changed files; catches unsorted imports, unformatted code, executable-bit-without-shebang at review time — prevents B3-class regressions) V18rs: ∀ `.github/workflows/**` change ! at least one validation job executes the affected workflow path; shared CI workflow changes ⊥ leave the validation matrix entirely skipped +V19cn: public API ! use VOCAB nouns (sample/variable=parameter/response=QoI); ⊥ "job" in any itis_sumo signature, docstring, or result field +V20dm: consumer-facing entrypoints ! accept plain tabular data; ⊥ `FunctionJob`/`JobVariableSelection`/oSPARC status strings inside itis_sumo (test-enforced, sibling of V4ty) +V21pf: preprocessing ! auto-defaulted + overridable in domain vocabulary only; ⊥ transform vocabulary reachable from a public signature; effective config inspectable in the result +V22rs: results ! typed dataclass, original units + original names, `dataclasses.asdict()`-serializable; ⊥ `_hat`/`_std_hat` suffix keys in the public shape +V23er: ∀ error escaping `itis_sumo.api` ! be a `SumoError` subclass; ⊥ raw `KeyError`/`IndexError`/`ValueError` crossing the boundary +V24af: run dir ! discarded on success, PRESERVED on failure w/ its path + stderr tail attached to the raised `SumoEngineError` +V25sd: ∀ stochastic entrypoint ! default `seed=42`, overridable, effective seed echoed in the result +V26dd: `domain` and `distribution` ! remain distinct config objects; ⊥ derive a box from a distribution's `mean ± 3σ` +V27fq: one-shot entrypoints ! implemented internally as fit-then-query; ⊥ monolithic procedure that would need rewriting to expose a handle V28tz: CI ! block a PR into `develop`/`main` whose `pyproject.toml` version lacks target branch's required prerelease suffix (`aN` develop, `bN` main) or regresses/duplicates PEP 440 ordering vs existing tags V29yn: tag-on-merge job (develop/main) ! push only a `v` tag, ⊥ version-bump commit; skip docs-only diff; existing-tag collision ! hard failure, ⊥ force-overwrite V30wk: itis-sumo LICENSE + `pyproject.toml` license/classifiers ! match itis-dakota's (all-rights-reserved) until an explicit public/OSS decision is made; ⊥ an OSI classifier claiming otherwise @@ -95,13 +130,21 @@ T11rt|✓|verify: standalone pytest green (⊖ flask), clean-venv install, `itis T12ze|✓|E1: `core/sumo_model_store.py` (models dir single env-overridable source `ITIS_SUMO_MODELS_DIR`, uuid keying, metadata sidecar) + `evaluate/funs_evaluate.py::export_sumo_model`/`import_sumo_model` public API|V10,V12,V13 T13uv|✓|E1: capture verbatim conf block for sidecar (close branch placeholder gap)|V10 T14qa|✓|E1: empirical export→import round-trip test (real, unmocked Dakota run — `tests/test_sumo_model_store.py`); resolves R2 `?`|R2,V11 -T15mn|.|E1: wire mmux/vite flaskapi endpoints to pkg export/import — separate PR in `../mmux_vite` (jgo/sumo-model-export-import branch T24, itself still `.` there); NOT part of this repo|§I,R1-R4 +T15mn|~|wire mmux/vite flaskapi to `itis_sumo.api` (export/import + all workflows): branch `feat/consume-itis-sumo-api` (worktree, pushed to `origin`, no PR yet) already replaces flaskapi's in-tree `dakota/`+`data_preprocessor/` w/ `itis_sumo.api` across all 7 workflows + drops the direct `itis-dakota` dep from flaskapi; pinned to itis-sumo `0.1.0.dev2` (TestPyPI); flaskapi suite verified 522 passed/3 skipped 2026-08-19 — remaining: open the PR, retarget the pin at a tagged `aN`/`bN` release instead of a `.devN` build (see parity note added to `../mmux_vite/flaskapi/SPEC.md` §C)|§I,R1-R4,T40qz T16mo|.|stepwise engine modernization: rung 1 `1.5.9→1.5.11` (packaging-only: drop py3.8, add cp313; same 6.20 engine — behavior-identical; re-run smoke); rung 2 `1.5.11→6.24.x` co-shipped w/ JSON input seam (R5 regression fixed upstream first, re-align py 3.13, re-run full smoke)|R4,R5 T17bq|✓|`add_surrogate_model` (`config/funs_create_dakota_conf.py`) ! replaced filename-substring sniffing w/ explicit `has_eval_id_column` param (infer-or-override idiom); all 5 `create_sumo_*_conffile` call sites updated (landed via `feat/vv-port-tests-and-docs`, `a6d694e`, ahead of this merge)|V15,B2 T17bc|✓|E1: persist real training data on export for reference (`{id}.processed_training.dat` sidecar copy, user preference over the leaner archive-only design); `stage_model_for_import` stages it back, falling back to a synthesized header-only placeholder + loud warning log only when that stored copy is missing (legacy model / deleted out-of-band) — fallback verified safe vs Dakota source + fake-points empirical tests (header reorder, wrong/missing descriptors, 0/1/2-row placeholders)|R8,R9,V10,V11 T18ry|.|design+implement `analyze_dataset(df, input_cols, output_cols, alpha=0.05, include_detail=False) -> DatasetDiagnostics` narrow entrypoint (scale/distribution auto-selection + outlier surfacing, plain dataclasses, JSON-serializable via `dataclasses.asdict()`) as the sole flaskapi-facing dataset-diagnostics API, replacing any per-function flaskapi orchestration; BLOCKED — building blocks `select_variable_scale`/`auto_select_distributions` (+ a new raw-value outlier detector, reusing `_tukey_outlier_mask`'s IQR technique) currently exist only on confidential incubator branch `feat/nih-in-silico-example`, not `develop` — needs individual promotion first, same promotion rule as T15mn/E1|§C,V16qf,I T19kp|.|headless notebook, post-alpha — add runnable notebook counterpart to docs getting-started flow once alpha docs publish is stable|T10le T20hm|.|CI workflow changes ! activate shared validation jobs and regression-check detector classification|V18rs +T21vk|✓|VOCAB section (above) + `docs/reference/glossary.md` (nav-registered, strict build green); vocabulary enforced on the api layer by `tests/test_api_contract.py::TestPublicSurface`. Sweep of the OLDER modules' docstrings still outstanding → folded into T24cm|V19cn +T22ax|✓|`itis_sumo/api` spine ( `errors.py` taxonomy, `types.py` config+results, `_session.py` fit-then-query + run-dir lifetime, `workflows.py` public funcs, `__init__.py` surface): tabular input type, `PreprocessingSpec` (domain vocabulary), `SumoError` taxonomy, typed result dataclasses, seed policy, artifact policy|V20dm,V21pf,V22rs,V23er,V24af,V25sd +T23bn|✓|`itis_sumo.api`: `cross_validate()` + `evaluate_along_axes()` — internally fit-then-query; 34 contract tests + 10 real-Dakota end-to-end tests green. Equivalence-vs-flaskapi-glue pinning ⊥ done here (needs the mmux_vite checkout) → moves to the consumer PR|V27fq +T24cm|✓|remove `preprocess/models.py::{FunctionJob,JobVariableSelection}`; re-express the Dakota sufficiency rule over tabular data; jobs→table adapter moves to flaskapi (this port DELETES itis-sumo code)|V20dm +T25dp|.|`itis_sumo.api`: remaining 6 workflows (UQ-w-uncertainty incl. the ~120-line erfinv/histogram block, correlation, Sobol, grid, MOGA, cv-accuracy-metrics) + E1 `export_model`/`evaluate_stored_model` facade|V22rs,R1-R4 +T26eq|.|POST-PORT: extract the fitted-model handle (`fit()` → methods → `save()`/`load()`); carries the fitted preprocessing config ⇒ closes the E1 gap (model store persists archive+metadata+training copy but ⊥ preprocessor config, so a reloaded model cannot inverse-transform to original units)|V27fq,V10jk +T27fr|.|POST-PORT: split `domain` vs `distribution` config + consumer migration; absorb the mmux_vite `jgo/fullstack-logscale` work|V26dd +T28gs|.|POST-PORT `?`: decide whether `distribution` gets an auto-generated default — explicit discussion required, ⊥ silently defaulted|§C `?` T29hw|~|`publish.yml` tag trigger accepts `v`-prefixed PEP 440 prereleases ✓; tagging `v0.1.0a1` BLOCKED on the one-time PyPI Trusted Publisher config (user action), then clean-venv install + `itis-sumo validate` + headless smoke|T1pw T30qa|✓|release/CI workflow refresh: build→TestPyPI automatic on tag push (✓), PyPI+Release gated manual (✓, T35cc); auto-tag-on-branch redesigned to PR-time check + tag-only-at-merge (no bot commit) — see T31xx/T32yy; keep git-cliff release notes, dependency-review + concurrency (✓); skip weekly cron/healthchecks for now|§C,V17rt T31xx|✓|CI: PR-time check job (feature branch→`.devN`, target `develop`→`aN`, target `main`→`bN`) blocking merge if `pyproject.toml` version missing, regresses, or duplicates PEP 440 order vs existing tags|V28tz,V33tz @@ -112,6 +155,8 @@ T35cc|✓|gate `publish`/`release` jobs behind explicit `workflow_dispatch`; `bu T36dd|✓|add `.env` to `.gitignore`; make `publish-testpypi` source `.env` without printing token, then build/check/publish|V37bb T37ef|✓|validate manual publish tag + artifact version before PyPI upload|V35wk T38dd|✓|make `publish-testpypi-dev` auto-compute/write `.devN`, build/check/upload directly to TestPyPI, then restore project version; CI verifies alpha/beta/rc before real PyPI; no dev tag cascade|V38cc,V39qf +T39pk|.|POST-PORT/deepen candidate: `itis_sumo.api.workflows` repeats the same build-session → translate-config → call → translate-result → wrap-errors skeleton per function; pull the shared shape down into `_session.py` so each `workflows.py` function shrinks to signature+one call — behavior-preserving refactor, tests green before/after; natural precursor to T26eq's handle extraction|V16qf,V27fq,T26eq +T40qz|.|cross-repo version-parity: mmux_vite `develop` ! pin itis-sumo `aN`, mmux_vite `main` ! pin itis-sumo `bN` (mirrors this repo's own channel policy) — noted in `../mmux_vite/flaskapi/SPEC.md` §C; the CI check enforcing it lives in mmux_vite (parse pinned itis-sumo version's PEP 440 pre-release tag, fail PR on target-branch mismatch), not this repo — tracked here for visibility since it's the other half of §C's release channel policy|§C,T15mn ## §B id|date|cause|fix @@ -127,5 +172,6 @@ B7pv|2026-08-18|prek passed deleted Python paths from the workflow diff to Ruff, B8lf|2026-08-18|the develop merge retained executable mode on changed `funs_evaluate.py`, so prek Ruff raised EXE002 for a Python file without a shebang|clear executable bit; existing V17ab catches changed-file Ruff violations B9dw|2026-08-18|after rewriting develop history, GitHub push events reported the unreachable pre-rewrite `before` SHA and detect_changes aborted with `bad object`|fall back to `git diff-tree` when the event base commit is unavailable; no code invariant added because this is CI history topology +B10cs|2026-08-18|`itis_sumo.api.evaluate_along_axes(at={...})` raised `SumoEngineError: 'x1'` for a PARTIAL held-value mapping — `create_samples_along_axes` reads `[cut_values[var] for var in input_vars]`, so an incomplete dict is a `KeyError`, not a documented default. Never surfaced in flaskapi because its frontend always sends every slider|`_map_held_values` completes the mapping from the sample means (the same value the no-`at` path uses) before translating; covered by `tests/test_api_workflows.py::TestAlongAxes::test_honours_the_values_the_caller_holds_fixed`. ⊥ new §V invariant: V23er already requires taxonomy errors, and the real lesson is that a caller-facing default must be materialised by the API layer rather than assumed by an internal helper B11xy|2026-08-19|`publish.yml`'s `testpypi` job ran pytest against the installed wheel w/o a checkout step — `tests/` (not shipped in the wheel) was absent, pytest collected 0 items and exited 5 before the job ever reached the TestPyPI publish step (caught via `gh run view` on the `v0.1.0a1` tag push, which failed harmlessly — no upload occurred anywhere)|added `actions/checkout` before `download-artifact` (ordered first — checkout's default `git clean` would otherwise wipe an already-downloaded `dist/`); V31vp B12mn|2026-08-19|`pyproject.toml` `requires-python` had drifted to `>=3.10,<3.14` ("broaden python compatibility matrix") while the pinned engine `itis-dakota==1.5.9` ships wheels only for cp310-cp312 (verified via PyPI JSON) — installing on 3.13 would fail dependency resolution, promising support §C never backed|tightened to `>=3.10,<3.13` w/ a comment tying the bound to the pinned wheel; re-widen only alongside a T16mo engine bump diff --git a/docs/how-to/cross-validate.md b/docs/how-to/cross-validate.md index be12533..0637404 100644 --- a/docs/how-to/cross-validate.md +++ b/docs/how-to/cross-validate.md @@ -16,7 +16,11 @@ from itis_sumo.evaluate.funs_evaluate import ( ) cv = evaluate_sumo_manual_crossvalidation( - run_dir, training_file, ["x1", "x2"], "y1", N_CROSS_VALIDATION=5, + run_dir, + training_file, + ["x1", "x2"], + "y1", + N_CROSS_VALIDATION=5, ) # cv["y1"] is the held-out actual values, cv["y1_hat"] the CV predictions # (both length-N, NaN at any fold Dakota couldn't complete) diff --git a/docs/how-to/moga.md b/docs/how-to/moga.md index a1d3a2d..90b939c 100644 --- a/docs/how-to/moga.md +++ b/docs/how-to/moga.md @@ -23,7 +23,12 @@ moga_kwargs = { "seed": 42, } pareto = perform_moga_optimization( - run_dir, training_file, ["length", "width"], distributions, ["y1"], moga_kwargs, + run_dir, + training_file, + ["length", "width"], + distributions, + ["y1"], + moga_kwargs, ) # pareto["y1"] — objective values on the front # pareto["length"], pareto["width"] — the input points that produced them diff --git a/docs/how-to/sensitivity-analysis.md b/docs/how-to/sensitivity-analysis.md index 075fd22..6db1dde 100644 --- a/docs/how-to/sensitivity-analysis.md +++ b/docs/how-to/sensitivity-analysis.md @@ -16,9 +16,15 @@ distributions = { "width": {"distribution": "uniform", "min": 0.0, "max": 1.0}, } result = evaluate_sobol_indices( - run_dir, training_file, ["length", "width"], "y1", distributions, preprocessor, seed=42, + run_dir, + training_file, + ["length", "width"], + "y1", + distributions, + preprocessor, + seed=42, ) -sobol = result["sobol"] # {var: {"main", "total", "main_ci_low", ...}} +sobol = result["sobol"] # {var: {"main", "total", "main_ci_low", ...}} second_order = result["sobolSecondOrder"] # {varA: {varB: float}} for var, indices in sobol.items(): diff --git a/docs/reference/api.md b/docs/reference/api.md new file mode 100644 index 0000000..b4c854b --- /dev/null +++ b/docs/reference/api.md @@ -0,0 +1,94 @@ +# Consumer API (`itis_sumo.api`) + +`itis_sumo.api` is the whole of what an application embedding itis-sumo is expected to +import — a web service, a notebook, a script. Pass a table of samples and some +configuration; get back a typed result in your own units. + +Everything in between belongs to itis-sumo and is not reachable from here: training-file +layout, Dakota configuration, normalization, internal variable renaming, run directories, +and inverse transforms. + +```python +from itis_sumo.api import cross_validate + +result = cross_validate(samples, variables=["width", "height"], response="stress") +result.predicted # one value per sample, in the units of `stress` +result.warnings # anything that went wrong but did not stop the run +``` + +The vocabulary is the one defined in the [Glossary](glossary.md): a **sample** is a row, +a **variable** (parameter) is an input column, a **response** (quantity of interest) is +an output column. + +## Workflows + +### `cross_validate` + +Predicts every sample from a surrogate that never saw it. The samples are split into +folds; each fold is predicted by a surrogate trained on the others, so the result is an +honest picture of how the surrogate behaves on data it has not memorised. + +Returns a `CrossValidationResult`: the observed values, the predicted values and their +standard deviations aligned positionally with your rows, plus any warnings. + +If Dakota abandons a fold — which it occasionally does when the training points are +nearly degenerate — the samples in that fold keep a `NaN` prediction and the reason +appears in `warnings`, rather than the whole run failing. + +### `evaluate_along_axes` + +Sweeps each variable in turn across its observed range while every other variable is held +still. This is the shape behind a one-dimensional profile plot. + +Returns an `AlongAxesResult` containing one `AxisSweep` per variable, keyed by the +variable's own name. Use `at=` to pin particular variables; anything you leave out is held +at its mean across the samples. + +## Configuration + +Sensible defaults are derived from your samples, so the common path needs no +configuration at all. Advanced users may override behaviour per column with +`PreprocessingSpec`: + +```python +from itis_sumo.api import PreprocessingSpec, VariableSpec + +spec = PreprocessingSpec(overrides={"width": VariableSpec(scale="linear")}) +``` + +Overrides are expressed in terms of the *domain* — what a column is like — never in terms +of a transform. Which normalization or sign convention a choice implies is an +implementation detail. You can always read back what was actually used, via +`result.effective_config`; you cannot set it in transform terms. + +Stochastic workflows take a `seed` that already has a value, so results are reproducible +by default without you having to think about it. The seed that was used comes back in the +result. + +## Errors + +Every error that escapes `itis_sumo.api` is a `SumoError`. Catch the subclass rather than +matching on message text. + +| Error | Meaning | Typical HTTP status | +|---|---|---| +| `SumoInputError` | The samples or configuration are unusable. Fixable by sending something different. | 400 | +| `SumoResultError` | The run finished but produced none of the requested values. Retrying unchanged will not help. | 422 | +| `SumoEngineError` | Dakota itself failed. | 500 | + +Dakota reports its failures opaquely. Because of that, working files are discarded when a +run succeeds but **kept when one fails** — and `SumoEngineError` carries the surviving run +directory along with the tail of Dakota's own stderr, so there is something to look at. + +```python +try: + result = cross_validate(samples, variables, response) +except SumoInputError as error: + return bad_request(str(error)) +except SumoEngineError as error: + logger.error("surrogate build failed; evidence at %s", error.run_dir) + return server_error("Could not build a surrogate from these samples") +``` + +Pass `workspace=` to keep working files from a successful run too, when you want to +inspect what Dakota was given. diff --git a/docs/reference/glossary.md b/docs/reference/glossary.md new file mode 100644 index 0000000..e77bc7f --- /dev/null +++ b/docs/reference/glossary.md @@ -0,0 +1,54 @@ +# Glossary + +itis-sumo uses one word per concept, everywhere: in the public API, in these docs, in +docstrings, in tests. If you see a synonym, it is drift and it is a bug. + +## Data shape + +| Term | Means | Not | +|---|---|---| +| **sample** | One row of tabular data — a single observation. | "job", "point", "record" | +| **variable**, **parameter** | One *input* column. | "feature", "factor" | +| **response**, **quantity of interest (QoI)** | One *output* column. | "target", "label" | + +`variable` and `parameter` are interchangeable, as are `response` and `quantity of +interest`. Inputs and outputs stay distinguished throughout, because they receive +different downstream treatment — normalization, sign handling, sampling roles, and +their position in a Sobol' decomposition all depend on which side a column is on. + +itis-sumo takes **plain tabular data**: a `pandas.DataFrame`, or an itis-sumo dataclass +that converts trivially to one. It has no notion of an oSPARC `FunctionJob`. Converting +job records into a table — and filtering out the ones that did not complete — belongs to +the consumer. Deciding that the resulting table has too few samples to build a surrogate +belongs to itis-sumo, which raises so the consumer can surface the problem. + +## Two things that are never the same thing + +| Term | Means | Can it be inferred from your samples? | +|---|---|---| +| **domain** | Where a variable may be drawn from or explored — its bounds and scale. A property of the *design space*. | **Yes.** Observed bounds plus a detected scale are a defensible default. | +| **distribution** | The real-world shape of a variable. A claim about *the world*. | **No.** A training set sampled uniformly over a box tells you nothing about whether the real input is normal. | + +Conflating these is the single easiest way to produce a confidently wrong uncertainty +band. Sobol' analysis and MOGA optimization want a **domain** — a region to explore. +Uncertainty propagation and correlation analysis want a **distribution** — something to +draw from. They are configured separately. + +## Modeling + +| Term | Means | +|---|---| +| **surrogate**, **SuMo** | The trained metamodel that stands in for an expensive simulation. | +| **surrogate model ID** (`sumo_model_id`) | The server-minted UUID identifying a persisted surrogate in the model store. | + +## What you pass, and what stays hidden + +You pass **samples** and **configuration**. Everything else is itis-sumo's business: +training-file layout, Dakota configuration, normalization, sign handling, internal +variable renaming, run directories, and mapping predictions back to your original units. + +Sensible defaults are generated for you, so the common path requires no configuration at +all. Advanced users may override them — but only in terms of the *domain*, saying things +like `scale="log"` or `direction="maximize"`. The transforms those choices imply are an +implementation detail and never appear in a signature. You can always read back the +configuration that was actually used; you cannot set it in transform terms. diff --git a/docs/tutorials/examples.md b/docs/tutorials/examples.md index 0b04739..5d5b7ff 100644 --- a/docs/tutorials/examples.md +++ b/docs/tutorials/examples.md @@ -28,8 +28,8 @@ from itis_sumo.evaluate.funs_evaluate import evaluate_sumo # x_train: 9 points, y_train = sin(x_train), written as a Dakota training file result = evaluate_sumo(run_dir, train_file, eval_file, ["x"], "y") -y_hat = result["y_hat"] # posterior mean, μ(x) -y_std = result["y_std_hat"] # posterior std, σ(x) — V8df in SPEC.md +y_hat = result["y_hat"] # posterior mean, μ(x) +y_std = result["y_std_hat"] # posterior std, σ(x) — V8df in SPEC.md ``` ![GP surrogate mean prediction and 95% uncertainty band on sin(x), fit from 9 training points](../assets/examples/gp_fit_uncertainty.png) @@ -53,8 +53,12 @@ argues that training-point placement matters as much as training-point count. ```python from itis_sumo.sampling.lhs import lhs -lhs_pts = lhs(2, 25, method="maximin", seed=42) # stratified: one draw per 1/25 bin, per axis -rand_pts = rng.uniform(0.0, 1.0, size=(25, 2)) # i.i.d. uniform — no stratification guarantee +lhs_pts = lhs( + 2, 25, method="maximin", seed=42 +) # stratified: one draw per 1/25 bin, per axis +rand_pts = rng.uniform( + 0.0, 1.0, size=(25, 2) +) # i.i.d. uniform — no stratification guarantee ``` ![Scatter comparison: 25 plain-random points show visible clustering and gaps; 25 Latin Hypercube points spread evenly across both marginals](../assets/examples/lhs_vs_random.png) diff --git a/docs/tutorials/getting-started.md b/docs/tutorials/getting-started.md index 53205a5..4b310bf 100644 --- a/docs/tutorials/getting-started.md +++ b/docs/tutorials/getting-started.md @@ -8,7 +8,7 @@ runs as a CI smoke test. ## 1. Install ```sh -uv sync # Python 3.11, resolves itis-dakota==1.5.9 +uv sync # Python 3.10-3.12 currently resolve itis-dakota==1.5.9 uv run itis-sumo validate # engine probe (expect: Version 1.5.9) ``` @@ -30,7 +30,12 @@ import pandas as pd rng = np.random.default_rng(42) length = rng.uniform(0.0, 1.0, size=30) width = rng.uniform(0.0, 1.0, size=30) -stress = 3.0 * length + 2.0 * width**2 + 0.5 * np.sin(10.0 * length) + rng.normal(0, 0.05, 30) +stress = ( + 3.0 * length + + 2.0 * width**2 + + 0.5 * np.sin(10.0 * length) + + rng.normal(0, 0.05, 30) +) train_raw = pd.DataFrame({"length": length, "width": width, "stress": stress}) ``` @@ -75,7 +80,9 @@ why itis-sumo uses one: [How Gaussian Processes work](../explanation/gaussian-pr ```python from itis_sumo.evaluate.funs_evaluate import evaluate_sumo_crossvalidation -cv_metrics = evaluate_sumo_crossvalidation(run_dir, training_file, ["x1", "x2"], "y1", N_CROSS_VALIDATION=5) +cv_metrics = evaluate_sumo_crossvalidation( + run_dir, training_file, ["x1", "x2"], "y1", N_CROSS_VALIDATION=5 +) print(cv_metrics["y1"]["root_mean_squared"]) ``` @@ -94,7 +101,13 @@ distributions = { "width": {"distribution": "uniform", "min": 0.0, "max": 1.0}, } sobol = evaluate_sobol_indices( - run_dir, training_file, ["length", "width"], "y1", distributions, preprocessor, seed=42 + run_dir, + training_file, + ["length", "width"], + "y1", + distributions, + preprocessor, + seed=42, )["sobol"] for var, indices in sobol.items(): print(var, indices["main"], indices["total"]) diff --git a/examples/headless_smoke.py b/examples/headless_smoke.py index dd785eb..81faa8d 100644 --- a/examples/headless_smoke.py +++ b/examples/headless_smoke.py @@ -27,7 +27,12 @@ def make_training_data(n: int = N_TRAINING_SAMPLES) -> pd.DataFrame: rng = np.random.default_rng(SEED) length = rng.uniform(0.0, 1.0, size=n) width = rng.uniform(0.0, 1.0, size=n) - stress = 3.0 * length + 2.0 * width**2 + 0.5 * np.sin(10.0 * length) + rng.normal(0, 0.05, n) + stress = ( + 3.0 * length + + 2.0 * width**2 + + 0.5 * np.sin(10.0 * length) + + rng.normal(0, 0.05, n) + ) return pd.DataFrame({"length": length, "width": width, "stress": stress}) @@ -46,9 +51,7 @@ def main() -> int: # --- 1. Surrogate evaluation on a grid --- grid = np.meshgrid(np.linspace(0.0, 1.0, 5), np.linspace(0.0, 1.0, 5)) - grid_raw = pd.DataFrame( - {"length": grid[0].ravel(), "width": grid[1].ravel()} - ) + grid_raw = pd.DataFrame({"length": grid[0].ravel(), "width": grid[1].ravel()}) grid_processed = preprocessor.transform(grid_raw) eval_samples_file = run_dir / "eval_samples.dat" grid_processed.to_csv(eval_samples_file, sep=" ", index=False) @@ -62,8 +65,10 @@ def main() -> int: ) y_hat = np.asarray(preds["y1_hat"]) assert len(y_hat) == len(grid_raw), "grid evaluation count mismatch" - print(f"[1/3] surrogate grid evaluation OK ({len(y_hat)} points, " - f"pred range [{y_hat.min():.3f}, {y_hat.max():.3f}])") + print( + f"[1/3] surrogate grid evaluation OK ({len(y_hat)} points, " + f"pred range [{y_hat.min():.3f}, {y_hat.max():.3f}])" + ) # --- 2. Cross-validation quality metrics --- cv_metrics = evaluate_sumo_crossvalidation( diff --git a/mkdocs.yml b/mkdocs.yml index 652870b..304673c 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -54,6 +54,8 @@ nav: - Preprocess data: how-to/preprocess.md - Reference: - Overview: reference/index.md + - Glossary: reference/glossary.md + - Consumer API: reference/api.md - Dakota engine core: reference/core.md - Config (NIDR composers): reference/config.md - Sampling: reference/sampling.md diff --git a/src/itis_sumo/api/__init__.py b/src/itis_sumo/api/__init__.py new file mode 100644 index 0000000..f2b5d8a --- /dev/null +++ b/src/itis_sumo/api/__init__.py @@ -0,0 +1,85 @@ +"""The itis-sumo consumer API. + +This is the whole of what an application embedding itis-sumo -- a web service, a +notebook, a script -- is expected to import. Pass a table of samples and some +configuration; get back a typed result in your own units. Everything else is +itis-sumo's business (SPEC V16qf, §G). + + >>> from itis_sumo.api import cross_validate + >>> result = cross_validate(samples, variables=["width", "height"], response="stress") + >>> result.predicted[:3] + +Vocabulary follows SPEC VOCAB throughout: a *sample* is a row, a *variable* +(parameter) is an input column, a *response* (quantity of interest) is an output +column. See the Glossary in the documentation. +""" + +from itis_sumo.api.errors import ( + SumoEngineError, + SumoError, + SumoInputError, + SumoResultError, +) +from itis_sumo.api.types import ( + DEFAULT_SEED, + AlongAxesResult, + AxisSweep, + CorrelationResult, + CrossValidationResult, + CVAccuracyMetrics, + Direction, + DistributionSpec, + DomainSpec, + GridResult, + ParetoFrontResult, + PreprocessingSpec, + Scale, + SobolResult, + UncertaintyResult, + VariableSpec, +) +from itis_sumo.api.workflows import ( + compute_correlations, + cross_validate, + evaluate_along_axes, + evaluate_cv_metrics, + evaluate_grid, + evaluate_sobol, + evaluate_uncertainty, + generate_grid_samples, + generate_lhs_samples, + optimize, +) + +__all__ = [ + "DEFAULT_SEED", + "AlongAxesResult", + "AxisSweep", + "CVAccuracyMetrics", + "CorrelationResult", + "CrossValidationResult", + "Direction", + "DistributionSpec", + "DomainSpec", + "GridResult", + "ParetoFrontResult", + "PreprocessingSpec", + "Scale", + "SobolResult", + "SumoEngineError", + "SumoError", + "SumoInputError", + "SumoResultError", + "UncertaintyResult", + "VariableSpec", + "compute_correlations", + "cross_validate", + "evaluate_along_axes", + "evaluate_cv_metrics", + "evaluate_grid", + "evaluate_sobol", + "evaluate_uncertainty", + "generate_grid_samples", + "generate_lhs_samples", + "optimize", +] diff --git a/src/itis_sumo/api/_session.py b/src/itis_sumo/api/_session.py new file mode 100644 index 0000000..3d27dc2 --- /dev/null +++ b/src/itis_sumo/api/_session.py @@ -0,0 +1,655 @@ +"""Internal fit-then-query engine behind the public one-shot workflows. + +The public functions in :mod:`itis_sumo.api.workflows` are deliberately thin +wrappers over a session that is fitted once and then queried many times. Keeping +that split in place from the outset is what makes the eventual public +fitted-model handle a re-export rather than a second port (SPEC V27fq). + +Nothing in this module is public API. +""" + +from __future__ import annotations + +import logging +import shutil +import tempfile +from collections.abc import Mapping, Sequence +from pathlib import Path +from types import TracebackType +from typing import Self + +import numpy as np +import pandas as pd + +from itis_sumo.api.errors import ( + SumoEngineError, + SumoError, + SumoInputError, + SumoResultError, +) +from itis_sumo.api.types import ( + AlongAxesResult, + AxisSweep, + CrossValidationResult, + Direction, + DistributionSpec, + DomainSpec, + GridResult, + ParetoFrontResult, + PreprocessingSpec, + SobolResult, + UncertaintyResult, + VariableSpec, +) +from itis_sumo.evaluate.funs_evaluate import ( + evaluate_sobol_indices, + evaluate_sumo_along_axes, + evaluate_sumo_manual_crossvalidation, + evaluate_sumo_on_grid, + perform_moga_optimization, + propagate_manual_uq_with_uncertainty, + summarize_uncertainty_samples, +) +from itis_sumo.preprocess.data_preprocessor import DataPreprocessor +from itis_sumo.utils.helpers import create_run_dir + +_logger = logging.getLogger(__name__) + +_STDERR_TAIL_LINES = 40 + + +def _minimum_samples(variable_count: int, floor: int = 5) -> int: + """Return the minimum tabular sample count Dakota can fit safely.""" + return max(floor, variable_count + 1) + + +def _stderr_tail(run_dir: Path | None) -> str: + if run_dir is None: + return "" + logs = sorted( + run_dir.rglob("dakota_stderr.txt"), key=lambda path: path.stat().st_mtime + ) + if not logs: + return "" + lines = logs[-1].read_text(errors="replace").splitlines() + return "\n".join(lines[-_STDERR_TAIL_LINES:]) + + +def _validate_samples( + samples: pd.DataFrame, + variables: Sequence[str], + responses: Sequence[str], + spec: PreprocessingSpec, +) -> pd.DataFrame: + """Reduce the caller's table to the columns in play, or explain why we can't. + + Deciding that a table cannot support a surrogate is itis-sumo's job; deciding + which rows deserve to be in the table in the first place is the caller's + (SPEC §C). Everything raised here is a :class:`SumoInputError`, because + everything raised here is fixable by sending different data. + """ + if not isinstance(samples, pd.DataFrame): + raise SumoInputError( + f"samples must be a pandas DataFrame, got {type(samples).__name__}" + ) + if not variables: + raise SumoInputError("At least one variable is required") + if not responses: + raise SumoInputError("At least one response is required") + + duplicates = sorted({v for v in variables if list(variables).count(v) > 1}) + if duplicates: + raise SumoInputError(f"Variables listed more than once: {duplicates}") + response_duplicates = sorted({r for r in responses if list(responses).count(r) > 1}) + if response_duplicates: + raise SumoInputError(f"Responses listed more than once: {response_duplicates}") + overlap = sorted(set(variables) & set(responses)) + if overlap: + raise SumoInputError(f"{overlap} listed as both a variable and a response") + + columns = [*variables, *responses] + missing = [column for column in columns if column not in samples.columns] + if missing: + raise SumoInputError( + f"Columns {missing} are not present in the samples. " + f"Available columns: {sorted(map(str, samples.columns))}" + ) + + unknown_overrides = sorted(set(spec.overrides) - set(columns)) + if unknown_overrides: + raise SumoInputError( + f"Preprocessing overrides given for columns that are not in play: " + f"{unknown_overrides}" + ) + logarithmic = sorted( + name for name, override in spec.overrides.items() if override.scale == "log" + ) + if logarithmic: + raise SumoInputError( + f"Logarithmic scale is not supported yet (requested for {logarithmic}); " + "it arrives together with the domain/distribution split" + ) + + selected = samples.loc[:, columns].copy() + try: + selected = selected.astype(float) + except (TypeError, ValueError) as exc: + raise SumoInputError( + f"Every variable and response must be numeric: {exc}" + ) from exc + + unusable = [ + column + for column in columns + if not np.isfinite(selected[column].to_numpy()).all() + ] + if unusable: + raise SumoInputError( + f"Columns {unusable} contain missing or infinite values. " + "Incomplete samples must be filtered out before they are passed in" + ) + + minimum = _minimum_samples(len(variables)) + if len(selected) < minimum: + raise SumoInputError( + f"{minimum} samples are required to build a surrogate over " + f"{len(variables)} variables, but only {len(selected)} were supplied" + ) + + return selected.reset_index(drop=True) + + +class SumoSession: + """A surrogate fitted once from a table of samples, then queried. + + Used as a context manager so that the run directory it needs has a defined + lifetime: discarded when everything worked, kept when something did not + (SPEC V24af). + """ + + def __init__( + self, + samples: pd.DataFrame, + variables: Sequence[str], + response: str, + *, + preprocessing: PreprocessingSpec | None = None, + workspace: Path | None = None, + ) -> None: + self._variables = tuple(variables) + self._response = response + self._spec = preprocessing or PreprocessingSpec() + self._samples = _validate_samples( + samples, self._variables, [self._response], self._spec + ) + self._workspace = workspace + self._run_dir: Path | None = None + self._preprocessor: DataPreprocessor | None = None + self._training_file: Path | None = None + + # ---------------------------------------------------------------- lifetime + + def __enter__(self) -> Self: + if self._workspace is None: + self._run_dir = Path(tempfile.mkdtemp(prefix="itis-sumo-")) + else: + self._run_dir = create_run_dir(Path(self._workspace), "sumo") + return self + + def __exit__( + self, + exc_type: type[BaseException] | None, + exc: BaseException | None, + traceback: TracebackType | None, + ) -> bool: + if self._run_dir is None: + return False + if exc_type is None: + if self._workspace is None: + shutil.rmtree(self._run_dir, ignore_errors=True) + else: + _logger.warning( + "itis-sumo run failed; run directory preserved at %s", self._run_dir + ) + return False + + # ----------------------------------------------------------------- fitting + + def fit(self) -> Self: + """Fit the preprocessing and write the training file Dakota will read.""" + preprocessor = DataPreprocessor() + preprocessor.setup_variables( + input_vars=list(self._variables), output_vars=[self._response] + ) + transformed = preprocessor.fit_transform(self._samples) + + assert self._run_dir is not None + training_file = self._run_dir / "processed_samples.dat" + transformed.to_csv(training_file, sep=" ", index=False) + + self._preprocessor = preprocessor + self._training_file = training_file + return self + + # ----------------------------------------------------------------- queries + + def cross_validate(self, *, folds: int, seed: int) -> CrossValidationResult: + response = self._mapped_response + results = self._run_engine( + "cross-validating the surrogate", + evaluate_sumo_manual_crossvalidation, + self._run_dir, + self._training_file, + self._mapped_variables, + response, + N_CROSS_VALIDATION=folds, + seed=seed, + has_eval_id_column=False, + ) + + missing = [ + key + for key in (response, f"{response}_hat", f"{response}_std_hat") + if key not in results + ] + if missing: + raise SumoResultError( + f"Cross-validation produced no {missing} values for " + f"'{self._response}'. The surrogate may not have been trained " + "with uncertainty estimates" + ) + + return CrossValidationResult( + response=self._response, + observed=self._to_original_units(response, results[response]), + predicted=self._to_original_units(response, results[f"{response}_hat"]), + predicted_std=self._to_original_units( + response, results[f"{response}_std_hat"] + ), + warnings=list(results.get("warnings", [])), + seed=seed, + effective_config=self.effective_config, + ) + + def along_axes( + self, + *, + at: Mapping[str, float] | None, + points_per_variable: int, + ) -> AlongAxesResult: + response = self._mapped_response + results = self._run_engine( + "evaluating the surrogate along its axes", + evaluate_sumo_along_axes, + self._run_dir, + self._training_file, + self._mapped_variables, + response, + cut_values=self._map_held_values(at), + NSAMPLESPERVAR=points_per_variable, + ) + if not results: + raise SumoResultError( + f"No axis sweeps were produced for response '{self._response}'" + ) + + original_names = self._preprocessor.get_inverse_mapping() + sweeps: dict[str, AxisSweep] = {} + for mapped_variable, axis in results.items(): + variable = original_names.get(mapped_variable, mapped_variable) + sweeps[variable] = AxisSweep( + variable=variable, + x=self._to_original_units(mapped_variable, axis["x"]), + predicted=self._to_original_units(response, axis["y_hat"]), + # A standard deviation is a width, not a position: it is reported + # as produced rather than shifted back through the transform. + predicted_std=( + [float(value) for value in axis["std_hat"]] + if "std_hat" in axis + else None + ), + ) + + return AlongAxesResult( + response=self._response, + sweeps=sweeps, + effective_config=self.effective_config, + ) + + def grid( + self, + *, + grid_variables: Sequence[str], + at: Mapping[str, float] | None, + points_per_variable: int, + ) -> GridResult: + """Evaluate a one-, two-, or higher-dimensional grid.""" + grid_variables = tuple(grid_variables) + unknown = sorted(set(grid_variables) - set(self._variables)) + if not grid_variables: + raise SumoInputError("At least one grid variable is required") + if unknown: + raise SumoInputError( + f"Grid variables {unknown} are not variables of this model" + ) + if points_per_variable < 2: + raise SumoInputError("A grid needs at least 2 points per variable") + + results = self._run_engine( + "evaluating the surrogate on a grid", + evaluate_sumo_on_grid, + self._run_dir, + self._training_file, + [self._mapped_name(variable) for variable in grid_variables], + self._mapped_variables, + self._mapped_response, + cut_values=self._map_held_values(at), + NSAMPLESPERVAR=points_per_variable, + ) + if self._mapped_response not in results: + raise SumoResultError( + f"No grid predictions were produced for '{self._response}'" + ) + + original_names = self._preprocessor.get_inverse_mapping() + converted: dict[str, list[float] | list[list[float]]] = {} + for mapped_name, values in results.items(): + original_name = original_names.get(mapped_name, mapped_name) + if mapped_name == self._mapped_response: + converted[original_name] = self._inverse_nested_values( + mapped_name, values + ) + else: + converted[original_name] = self._inverse_nested_values( + mapped_name, values + ) + return GridResult( + response=self._response, + grid_variables=grid_variables, + data=converted, + effective_config=self.effective_config, + ) + + def sobol( + self, + *, + distributions: Mapping[str, DistributionSpec], + seed: int, + ) -> SobolResult: + """Compute sensitivity indices from explicitly supplied uncertainty.""" + missing = sorted(set(self._variables) - set(distributions)) + unknown = sorted(set(distributions) - set(self._variables)) + if missing or unknown: + raise SumoInputError( + f"Distributions must cover variables exactly; missing={missing}, " + f"unknown={unknown}" + ) + engine_distributions = { + variable: spec.as_engine_dict() for variable, spec in distributions.items() + } + results = self._run_engine( + "computing Sobol indices", + evaluate_sobol_indices, + self._run_dir, + self._training_file, + self._variables, + self._mapped_response, + engine_distributions, + self._preprocessor, + seed=seed, + ) + if "sobol" not in results: + raise SumoResultError( + f"No Sobol indices were produced for '{self._response}'" + ) + return SobolResult( + response=self._response, + indices=results["sobol"], + second_order=results["sobolSecondOrder"], + seed=seed, + distributions=dict(distributions), + ) + + def uncertainty( + self, + *, + distributions: Mapping[str, DistributionSpec], + num_samples: int, + n_histograms: int, + seed: int, + ) -> UncertaintyResult: + """Propagate explicit uncertainty through the surrogate's own predictive + uncertainty and summarise the result as a histogram + boxplot.""" + missing = sorted(set(self._variables) - set(distributions)) + unknown = sorted(set(distributions) - set(self._variables)) + if missing or unknown: + raise SumoInputError( + f"Distributions must cover variables exactly; missing={missing}, " + f"unknown={unknown}" + ) + engine_distributions = { + variable: spec.as_engine_dict() for variable, spec in distributions.items() + } + samples = self._run_engine( + "propagating uncertainty", + propagate_manual_uq_with_uncertainty, + self._run_dir, + self._training_file, + self._variables, + self._response, + engine_distributions, + self._preprocessor, + num_samples, + n_histograms=n_histograms, + seed=seed, + ) + summary = summarize_uncertainty_samples(samples) + return UncertaintyResult( + response=self._response, + distributions=dict(distributions), + seed=seed, + bins_start=summary["bins_start"], + bins_end=summary["bins_end"], + bin_means=summary["bin_means"], + bin_stds=summary["bin_stds"], + q1=summary["q1"], + median=summary["median"], + q3=summary["q3"], + whisker_min=summary["whisker_min"], + whisker_max=summary["whisker_max"], + outliers=summary["outliers"], + mean=summary["mean"], + std=summary["std"], + minimum=summary["min"], + maximum=summary["max"], + ) + + def _mapped_name(self, variable: str) -> str: + assert self._preprocessor is not None + return self._preprocessor.input_variables[variable].mapped_name + + def _inverse_nested_values( + self, mapped_name: str, values: object + ) -> list[float] | list[list[float]]: + if ( + not isinstance(values, list) + or not values + or not isinstance(values[0], list) + ): + return self._to_original_units(mapped_name, values) # type: ignore[arg-type] + return [ + self._to_original_units(mapped_name, row) # type: ignore[arg-type] + for row in values + ] + + # -------------------------------------------------------------- internals + + @property + def effective_config(self) -> dict[str, VariableSpec]: + """What was actually used -- readable, but not settable in transform terms.""" + return { + column: self._spec.overrides.get(column, VariableSpec()) + for column in (*self._variables, self._response) + } + + @property + def _mapped_variables(self) -> list[str]: + assert self._preprocessor is not None + return [ + self._preprocessor.input_variables[variable].mapped_name + for variable in self._variables + ] + + @property + def _mapped_response(self) -> str: + assert self._preprocessor is not None + return self._preprocessor.output_variables[self._response].mapped_name + + def _map_held_values( + self, at: Mapping[str, float] | None + ) -> dict[str, float] | None: + """Complete and translate the values the caller wants held fixed. + + A caller may pin only the variables they care about. Every remaining + variable has to be given a value anyway, and it gets the same one it + would have got had the caller said nothing at all: its mean across the + samples. + """ + if not at: + return None + unknown = sorted(set(at) - set(self._variables)) + if unknown: + raise SumoInputError( + f"Cannot hold {unknown} fixed: they are not variables of this model" + ) + assert self._preprocessor is not None + held_row = { + **self._samples.mean().to_dict(), + **{name: float(value) for name, value in at.items()}, + } + held = self._preprocessor.transform(pd.DataFrame([held_row])) + mapped_variables = set(self._mapped_variables) + return { + name: float(value) + for name, value in held.iloc[0].to_dict().items() + if name in mapped_variables + } + + def _to_original_units( + self, mapped_name: str, values: Sequence[float] + ) -> list[float]: + assert self._preprocessor is not None + original = self._preprocessor.get_inverse_mapping().get( + mapped_name, mapped_name + ) + restored = self._preprocessor.inverse_transform({mapped_name: list(values)}) + return [float(value) for value in restored.get(original, list(values))] + + def _run_engine(self, description: str, function, *args, **kwargs): + try: + return function(*args, **kwargs) + except SumoError: + raise + except Exception as exc: + raise SumoEngineError( + f"Dakota failed while {description}: {exc}", + run_dir=self._run_dir, + stderr_tail=self._stderr_tail(), + ) from exc + + def _stderr_tail(self) -> str: + return _stderr_tail(self._run_dir) + + +def optimize_pareto_front( + samples: pd.DataFrame, + variables: Sequence[str], + objectives: Mapping[str, Direction], + *, + domains: Mapping[str, DomainSpec], + max_evaluations: int, + workspace: Path | None, +) -> ParetoFrontResult: + """Fit a surrogate per objective and find its Pareto-optimal trade-off front.""" + variables = tuple(variables) + missing_domains = sorted(set(variables) - set(domains)) + unknown_domains = sorted(set(domains) - set(variables)) + if missing_domains or unknown_domains: + raise SumoInputError( + f"Domains must cover variables exactly; missing={missing_domains}, " + f"unknown={unknown_domains}" + ) + + validated = _validate_samples( + samples, variables, list(objectives), PreprocessingSpec() + ) + + run_dir = ( + create_run_dir(Path(workspace), "sumo") + if workspace is not None + else Path(tempfile.mkdtemp(prefix="itis-sumo-")) + ) + try: + preprocessor = DataPreprocessor() + preprocessor.setup_variables( + input_vars=list(variables), output_vars=list(objectives) + ) + maximize = [ + response + for response, direction in objectives.items() + if direction == "maximize" + ] + if maximize: + preprocessor.setup_sign_switching(output_sign_switches=maximize) + transformed = preprocessor.fit_transform(validated) + training_file = run_dir / "processed_samples.dat" + transformed.to_csv(training_file, sep=" ", index=False) + + mapped_variables = [ + preprocessor.input_variables[variable].mapped_name for variable in variables + ] + mapped_objectives = [ + preprocessor.output_variables[response].mapped_name + for response in objectives + ] + mapped_domains = { + preprocessor.input_variables[variable].mapped_name: spec.as_engine_dict() + for variable, spec in domains.items() + } + + try: + results = perform_moga_optimization( + run_dir, + training_file, + mapped_variables, + mapped_domains, + mapped_objectives, + moga_kwargs={"max_function_evaluations": max_evaluations}, + ) + except Exception as exc: + raise SumoEngineError( + f"Dakota failed while optimizing the Pareto front: {exc}", + run_dir=run_dir, + stderr_tail=_stderr_tail(run_dir), + ) from exc + + if not results: + raise SumoResultError("No Pareto front points were produced") + + original = preprocessor.inverse_transform(results) + original_names = preprocessor.get_inverse_mapping() + data = { + original_names.get(mapped_name, mapped_name): [float(v) for v in values] + for mapped_name, values in original.items() + } + if workspace is None: + shutil.rmtree(run_dir, ignore_errors=True) + return ParetoFrontResult( + objectives=dict(objectives), variables=variables, data=data + ) + except SumoError: + if workspace is None: + _logger.warning( + "itis-sumo run failed; run directory preserved at %s", run_dir + ) + raise diff --git a/src/itis_sumo/api/errors.py b/src/itis_sumo/api/errors.py new file mode 100644 index 0000000..5bca8ca --- /dev/null +++ b/src/itis_sumo/api/errors.py @@ -0,0 +1,58 @@ +"""Stable error taxonomy for the itis-sumo consumer API (SPEC V23er). + +Every error that escapes ``itis_sumo.api`` is a :class:`SumoError`. Consumers +classify failures by catching one of the three subclasses -- an HTTP service maps +them onto 400 / 422 / 500 -- instead of matching on message text. +""" + +from __future__ import annotations + +from pathlib import Path + + +class SumoError(Exception): + """Base class for every error crossing the ``itis_sumo.api`` boundary.""" + + +class SumoInputError(SumoError): + """The supplied samples or configuration are unusable. + + The caller can fix this by sending different data or configuration. + """ + + +class SumoResultError(SumoError): + """The run finished, but did not produce the values that were requested. + + Nothing about the request was wrong, so retrying it unchanged will not help. + """ + + +class SumoEngineError(SumoError): + """The Dakota engine itself failed. + + Dakota reports failures opaquely -- ``Dakota aborted: Unknown error 250``, + ``IndexError: map::at`` -- and the only tractable evidence is the run + directory that produced them. That directory is therefore kept alive when a + run fails (it is discarded on success), and its path travels with this error + alongside the tail of Dakota's own stderr. See SPEC V24af. + """ + + def __init__( + self, + message: str, + *, + run_dir: Path | None = None, + stderr_tail: str = "", + ) -> None: + super().__init__(message) + self.run_dir = run_dir + self.stderr_tail = stderr_tail + + def __str__(self) -> str: + text = super().__str__() + if self.run_dir is not None: + text += f"\nRun directory preserved at: {self.run_dir}" + if self.stderr_tail: + text += f"\nDakota stderr (tail):\n{self.stderr_tail}" + return text diff --git a/src/itis_sumo/api/types.py b/src/itis_sumo/api/types.py new file mode 100644 index 0000000..2afe541 --- /dev/null +++ b/src/itis_sumo/api/types.py @@ -0,0 +1,214 @@ +"""Configuration and result types for the itis-sumo consumer API. + +The vocabulary here is fixed by SPEC VOCAB and is the same vocabulary used in the +documentation: a **sample** is a row, a **variable** (equivalently *parameter*) is +an input column, and a **response** (equivalently *quantity of interest*) is an +output column. + +Every result type is a plain frozen dataclass carrying values in the caller's +original units under the caller's original column names, and is JSON-serializable +via :func:`dataclasses.asdict` (SPEC V22rs). +""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass, field +from typing import Literal + +#: Seed used by every stochastic entrypoint unless the caller overrides it. +#: Fixed rather than required, so that results are reproducible by default +#: without the caller having to think about it (SPEC V25sd). +DEFAULT_SEED = 42 + +Scale = Literal["linear", "log"] +DistributionKind = Literal["constant", "uniform", "normal"] + + +@dataclass(frozen=True) +class DistributionSpec: + """A real-world uncertainty distribution for one variable. + + This is deliberately not a domain object. A domain says where exploration is + allowed; this says what shape real-world uncertainty has. The two are split + fully in the fitted-model transformation (SPEC T27fr). + """ + + distribution: DistributionKind + value: float | None = None + mean: float | None = None + std: float | None = None + minimum: float | None = None + maximum: float | None = None + + def as_engine_dict(self) -> dict[str, float | str]: + """Translate stable public names to the current Dakota adapter shape.""" + values: dict[str, float | str] = {"distribution": self.distribution} + if self.value is not None: + values["value"] = self.value + if self.mean is not None: + values["mean"] = self.mean + if self.std is not None: + values["std"] = self.std + if self.minimum is not None: + values["min"] = self.minimum + if self.maximum is not None: + values["max"] = self.maximum + return values + + +Direction = Literal["minimize", "maximize"] + + +@dataclass(frozen=True) +class DomainSpec: + """Where a variable is allowed to be explored (not what its uncertainty is). + + Optimization walks a domain looking for the best point; it does not need, + and MOGA specifically cannot use, a real-world uncertainty shape (SPEC + T27fr keeps this split explicit rather than overloading DistributionSpec). + """ + + minimum: float + maximum: float + + def as_engine_dict(self) -> dict[str, float | str]: + return {"distribution": "uniform", "min": self.minimum, "max": self.maximum} + + +@dataclass(frozen=True) +class ParetoFrontResult: + """Pareto-optimal points from a multi-objective optimization.""" + + objectives: dict[str, Direction] + variables: tuple[str, ...] + data: dict[str, list[float]] + + +@dataclass(frozen=True) +class UncertaintyResult: + """Histogram and summary statistics from propagating explicit uncertainty + through a surrogate's own predictive uncertainty.""" + + response: str + distributions: dict[str, DistributionSpec] + seed: int + bins_start: float + bins_end: float + bin_means: list[float] + bin_stds: list[float] + q1: float + median: float + q3: float + whisker_min: float + whisker_max: float + outliers: list[float] + mean: float + std: float + minimum: float + maximum: float + + +@dataclass(frozen=True) +class SobolResult: + """First-, total-, and second-order sensitivity indices.""" + + response: str + indices: dict[str, dict[str, float]] + second_order: dict[str, dict[str, float]] + seed: int + distributions: dict[str, DistributionSpec] + + +@dataclass(frozen=True) +class VariableSpec: + """An optional override for how one column behaves. + + Expressed in domain terms only: ``scale`` describes the column, it does not + name a transform. Which transform that implies is itis-sumo's business and + never appears in a public signature (SPEC V21pf). + """ + + scale: Scale = "linear" + + +@dataclass(frozen=True) +class PreprocessingSpec: + """Per-column overrides, keyed by the column's own name. + + Omit this entirely -- the common case -- and suitable defaults are derived + from the samples themselves. + """ + + overrides: Mapping[str, VariableSpec] = field(default_factory=dict) + + +@dataclass(frozen=True) +class CorrelationResult: + """Pearson and Spearman sensitivity of a response to each input variable.""" + + response: str + coefficients: dict[str, dict[str, float]] + + +@dataclass(frozen=True) +class CVAccuracyMetrics: + """Error metrics computed from cross-validation predictions.""" + + response: str + root_mean_squared: float + sum_abs: float + mean_abs: float + max_abs: float + seed: int + + +@dataclass(frozen=True) +class CrossValidationResult: + """Held-out predictions for every sample, in the response's original units. + + ``predicted`` and ``predicted_std`` are aligned positionally with the rows of + the samples that were passed in. A sample whose fold was abandoned by Dakota + keeps a ``NaN`` prediction and is explained in :attr:`warnings` rather than + failing the whole run. + """ + + response: str + observed: list[float] + predicted: list[float] + predicted_std: list[float] | None + warnings: list[str] + seed: int + effective_config: dict[str, VariableSpec] + + +@dataclass(frozen=True) +class AxisSweep: + """One variable swept across its observed range. + + Every other variable is held fixed for the duration of the sweep. + """ + + variable: str + x: list[float] + predicted: list[float] + predicted_std: list[float] | None = None + + +@dataclass(frozen=True) +class GridResult: + """Predictions on a grid, keyed by the original variable names.""" + + response: str + grid_variables: tuple[str, ...] + data: dict[str, list[float] | list[list[float]]] + effective_config: dict[str, VariableSpec] + + +@dataclass(frozen=True) +class AlongAxesResult: + """One :class:`AxisSweep` per variable, keyed by the variable's name.""" + + response: str + sweeps: dict[str, AxisSweep] + effective_config: dict[str, VariableSpec] diff --git a/src/itis_sumo/api/workflows.py b/src/itis_sumo/api/workflows.py new file mode 100644 index 0000000..6b886ab --- /dev/null +++ b/src/itis_sumo/api/workflows.py @@ -0,0 +1,349 @@ +"""The workflows itis-sumo offers its consumers. + +Each function takes a table of samples plus configuration, and returns a typed +result in the caller's own units and column names. Everything in between -- +training-file layout, Dakota configuration, normalization, internal variable +renaming, run directories, inverse transforms -- belongs to itis-sumo and is not +reachable from here (SPEC §G). + +They are one-shot by design: each call fits a surrogate and queries it once. +Internally they are already split into a fit step and a query step, so that +holding on to a fitted surrogate across several queries can be offered later +without changing what these functions do (SPEC V27fq). +""" + +from __future__ import annotations + +import shutil +import tempfile +from collections.abc import Mapping, Sequence +from pathlib import Path + +import pandas as pd + +from itis_sumo.api._session import SumoSession, optimize_pareto_front +from itis_sumo.api.errors import SumoInputError +from itis_sumo.api.types import ( + DEFAULT_SEED, + AlongAxesResult, + CorrelationResult, + CrossValidationResult, + CVAccuracyMetrics, + Direction, + DistributionSpec, + DomainSpec, + GridResult, + ParetoFrontResult, + PreprocessingSpec, + SobolResult, + UncertaintyResult, +) +from itis_sumo.data.funs_data_processing import ( + compute_correlation_indices, + create_grid_samples, + load_data, +) +from itis_sumo.evaluate.funs_evaluate import compute_cv_accuracy_metrics +from itis_sumo.sampling.lhs import lhs as _lhs +from itis_sumo.utils.helpers import create_run_dir + + +def cross_validate( + samples: pd.DataFrame, + variables: Sequence[str], + response: str, + *, + preprocessing: PreprocessingSpec | None = None, + folds: int = 5, + seed: int = DEFAULT_SEED, + workspace: Path | None = None, +) -> CrossValidationResult: + """Predict every sample from a surrogate that never saw it. + + The samples are split into ``folds`` groups; each group is predicted by a + surrogate trained on the others. The result is therefore an honest picture of + how the surrogate behaves on data it has not memorised. + + Args: + samples: One row per sample, one column per variable or response. + variables: Columns to treat as inputs. + response: Column to predict. + preprocessing: Optional per-column overrides. Omit for sensible defaults. + folds: Number of cross-validation groups. + seed: Controls how samples are assigned to folds. + workspace: If given, working files are written here and kept. If omitted, + they are discarded on success and kept on failure. + + Raises: + SumoInputError: The samples or configuration cannot support this run. + SumoResultError: The run finished without producing predictions. + SumoEngineError: Dakota failed; the error carries the surviving run + directory and the tail of Dakota's stderr. + """ + if folds < 2: + raise SumoInputError(f"Cross-validation needs at least 2 folds, got {folds}") + + with SumoSession( + samples, + variables, + response, + preprocessing=preprocessing, + workspace=workspace, + ) as session: + return session.fit().cross_validate(folds=folds, seed=seed) + + +def evaluate_along_axes( + samples: pd.DataFrame, + variables: Sequence[str], + response: str, + *, + at: Mapping[str, float] | None = None, + points_per_variable: int = 21, + preprocessing: PreprocessingSpec | None = None, + workspace: Path | None = None, +) -> AlongAxesResult: + """Sweep each variable in turn, holding the others still. + + This is the shape behind a one-dimensional profile plot: for each variable, + the response is predicted across that variable's observed range while every + other variable stays at a fixed value. + + Args: + samples: One row per sample, one column per variable or response. + variables: Columns to treat as inputs. + response: Column to predict. + at: Values at which to hold the variables that are not being swept. + Any variable left out is held at its mean across the samples, + which is also what happens when ``at`` is omitted entirely. + points_per_variable: How many points to evaluate along each sweep. + preprocessing: Optional per-column overrides. Omit for sensible defaults. + workspace: If given, working files are written here and kept. If omitted, + they are discarded on success and kept on failure. + + Raises: + SumoInputError: The samples or configuration cannot support this run. + SumoResultError: The run finished without producing any sweep. + SumoEngineError: Dakota failed; the error carries the surviving run + directory and the tail of Dakota's stderr. + """ + with SumoSession( + samples, + variables, + response, + preprocessing=preprocessing, + workspace=workspace, + ) as session: + return session.fit().along_axes(at=at, points_per_variable=points_per_variable) + + +def evaluate_grid( + samples: pd.DataFrame, + variables: Sequence[str], + response: str, + *, + grid_variables: Sequence[str], + at: Mapping[str, float] | None = None, + points_per_variable: int = 21, + preprocessing: PreprocessingSpec | None = None, + workspace: Path | None = None, +) -> GridResult: + """Evaluate a surrogate across a grid of selected variables.""" + with SumoSession( + samples, variables, response, preprocessing=preprocessing, workspace=workspace + ) as session: + return session.fit().grid( + grid_variables=grid_variables, + at=at, + points_per_variable=points_per_variable, + ) + + +def evaluate_sobol( + samples: pd.DataFrame, + variables: Sequence[str], + response: str, + *, + distributions: Mapping[str, DistributionSpec], + preprocessing: PreprocessingSpec | None = None, + seed: int = DEFAULT_SEED, + workspace: Path | None = None, +) -> SobolResult: + """Compute Sobol sensitivity indices from explicit distributions.""" + with SumoSession( + samples, variables, response, preprocessing=preprocessing, workspace=workspace + ) as session: + return session.fit().sobol(distributions=distributions, seed=seed) + + +def compute_correlations( + samples: pd.DataFrame, + variables: Sequence[str], + response: str, +) -> CorrelationResult: + """Compute response correlations from a caller-owned sample table.""" + missing = sorted((set(variables) | {response}) - set(samples.columns)) + if missing: + raise SumoInputError(f"Samples do not contain columns: {missing}") + try: + coefficients = compute_correlation_indices( + samples, samples[response].tolist(), list(variables) + ) + except ValueError as exc: + raise SumoInputError(str(exc)) from exc + return CorrelationResult(response=response, coefficients=coefficients) + + +def evaluate_cv_metrics( + samples: pd.DataFrame, + variables: Sequence[str], + response: str, + *, + preprocessing: PreprocessingSpec | None = None, + folds: int = 5, + seed: int = DEFAULT_SEED, + workspace: Path | None = None, +) -> CVAccuracyMetrics: + """Fit cross-validation predictions and return stable accuracy metrics.""" + result = cross_validate( + samples, + variables, + response, + preprocessing=preprocessing, + folds=folds, + seed=seed, + workspace=workspace, + ) + metrics = compute_cv_accuracy_metrics(result.observed, result.predicted) + return CVAccuracyMetrics(response=response, seed=result.seed, **metrics) + + +def evaluate_uncertainty( + samples: pd.DataFrame, + variables: Sequence[str], + response: str, + *, + distributions: Mapping[str, DistributionSpec], + num_samples: int = 1000, + n_histograms: int = 100, + seed: int = DEFAULT_SEED, + preprocessing: PreprocessingSpec | None = None, + workspace: Path | None = None, +) -> UncertaintyResult: + """Propagate explicit per-variable uncertainty through the surrogate. + + Draws ``num_samples`` from ``distributions``, evaluates the surrogate once, + then repeats ``n_histograms`` times injecting the surrogate's own predictive + uncertainty, returning a histogram + boxplot summary in the response's + original units. + """ + with SumoSession( + samples, variables, response, preprocessing=preprocessing, workspace=workspace + ) as session: + return session.fit().uncertainty( + distributions=distributions, + num_samples=num_samples, + n_histograms=n_histograms, + seed=seed, + ) + + +def optimize( + samples: pd.DataFrame, + variables: Sequence[str], + objectives: Mapping[str, Direction], + *, + domains: Mapping[str, DomainSpec], + max_evaluations: int = 1000, + workspace: Path | None = None, +) -> ParetoFrontResult: + """Find the Pareto-optimal trade-off front across one or more objectives. + + Unlike the other workflows, this fits one surrogate per objective over a + domain (where exploration is allowed), not a real-world uncertainty + distribution -- MOGA cannot use anything but a uniform domain (SPEC T27fr). + """ + return optimize_pareto_front( + samples, + variables, + objectives, + domains=domains, + max_evaluations=max_evaluations, + workspace=workspace, + ) + + +def generate_lhs_samples( + domains: Mapping[str, DomainSpec], + n_samples: int, + *, + seed: int = DEFAULT_SEED, +) -> pd.DataFrame: + """Draw a Latin-hypercube design over the given variable domains. + + Args: + domains: Variable name -> allowed range to draw from. + n_samples: Number of sample rows to generate. + seed: Controls the draw. + + Raises: + SumoInputError: No domains given. + """ + if not domains: + raise SumoInputError("At least one variable domain is required.") + names = list(domains) + design = _lhs(n_samples, len(names), seed=seed) + return pd.DataFrame( + { + name: design[i] * (domains[name].maximum - domains[name].minimum) + + domains[name].minimum + for i, name in enumerate(names) + } + ) + + +def generate_grid_samples( + domains: Mapping[str, DomainSpec], + points_per_variable: Mapping[str, int], + *, + workspace: Path | None = None, +) -> pd.DataFrame: + """Generate a full-factorial grid of samples over the given variable domains. + + Args: + domains: Variable name -> allowed range to draw from. + points_per_variable: Variable name -> number of grid points along that axis. + workspace: If given, working files are written here and kept. If omitted, + they are discarded on success and kept on failure. + + Raises: + SumoInputError: No domains given, or a variable is missing its point count. + """ + if not domains: + raise SumoInputError("At least one variable domain is required.") + names = list(domains) + missing = [name for name in names if name not in points_per_variable] + if missing: + raise SumoInputError(f"Missing points_per_variable for: {', '.join(missing)}") + + run_dir = ( + create_run_dir(Path(workspace), "grid-sample") + if workspace is not None + else Path(tempfile.mkdtemp(prefix="itis-sumo-")) + ) + try: + grid_file = create_grid_samples( + run_dir=run_dir, + grid_vars=names, + input_vars=names, + mins=[domains[name].minimum for name in names], + cut_values=[ + (domains[name].minimum + domains[name].maximum) / 2 for name in names + ], + maxs=[domains[name].maximum for name in names], + n_points_per_dimension=[points_per_variable[name] for name in names], + ) + return load_data(grid_file)[names].astype(float) + finally: + if workspace is None: + shutil.rmtree(run_dir, ignore_errors=True) diff --git a/src/itis_sumo/core/__init__.py b/src/itis_sumo/core/__init__.py index 6e84350..f7c1a9d 100644 --- a/src/itis_sumo/core/__init__.py +++ b/src/itis_sumo/core/__init__.py @@ -1,6 +1,24 @@ -"""Core Dakota execution primitives.""" +"""Core Dakota execution primitives and the exported-model store.""" from itis_sumo.core.dakota_object import DakotaObject, working_directory +from itis_sumo.core.sumo_model_store import ( + MODELS_DIR_ENV_VAR, + SumoModelMetadata, + get_models_dir, + load_model_metadata, + stage_model_for_import, + store_exported_model, +) from itis_sumo.core.wiofiles import capture_to_file -__all__ = ["DakotaObject", "capture_to_file", "working_directory"] +__all__ = [ + "MODELS_DIR_ENV_VAR", + "DakotaObject", + "SumoModelMetadata", + "capture_to_file", + "get_models_dir", + "load_model_metadata", + "stage_model_for_import", + "store_exported_model", + "working_directory", +] diff --git a/src/itis_sumo/evaluate/__init__.py b/src/itis_sumo/evaluate/__init__.py index 0da2877..856d59a 100644 --- a/src/itis_sumo/evaluate/__init__.py +++ b/src/itis_sumo/evaluate/__init__.py @@ -10,6 +10,8 @@ evaluate_sumo_crossvalidation, evaluate_sumo_manual_crossvalidation, evaluate_sumo_on_grid, + export_sumo_model, + import_sumo_model, perform_moga_optimization, propagate_uq, retrieve_csv_result, @@ -25,6 +27,8 @@ "evaluate_sumo_crossvalidation", "evaluate_sumo_manual_crossvalidation", "evaluate_sumo_on_grid", + "export_sumo_model", + "import_sumo_model", "perform_moga_optimization", "propagate_uq", "retrieve_csv_result", diff --git a/src/itis_sumo/evaluate/funs_evaluate.py b/src/itis_sumo/evaluate/funs_evaluate.py index 86b293b..34e6e47 100644 --- a/src/itis_sumo/evaluate/funs_evaluate.py +++ b/src/itis_sumo/evaluate/funs_evaluate.py @@ -24,6 +24,7 @@ from itis_sumo.core.sumo_model_store import stage_model_for_import, store_exported_model from itis_sumo.data.funs_data_processing import ( create_grid_samples, + create_manual_uq_samples, create_samples_along_axes, extract_predictions_along_axes, extract_predictions_gridpoints, @@ -31,9 +32,7 @@ get_results, load_data, process_input_file, - sanitize_varname, sanitize_varnames, - sanitize_varnames_dict, ) _logger = logging.getLogger(__name__) @@ -81,6 +80,7 @@ def evaluate_sumo_along_axes( sumo_import_name: str | None = None, sumo_export_name: str | None = None, NSAMPLESPERVAR: int = 21, + has_eval_id_column: bool | None = None, xscale: Literal["linear", "log"] = "linear", yscale: Literal["linear", "log"] = "linear", label_converter: Callable | None = None, @@ -121,6 +121,7 @@ def evaluate_sumo_along_axes( samples_file=PROCESSED_SWEEP_INPUT_FILE, input_variables=input_vars, output_responses=[response_var], + has_eval_id_column=has_eval_id_column, ) # run dakota @@ -166,6 +167,147 @@ def propagate_uq( return x.tolist() +def propagate_manual_uq_with_uncertainty( + run_dir: Path, + PROCESSED_TRAINING_FILE: Path, + input_vars: list[str], + output_response: str, + distributions: dict[str, dict[str, float]], + preprocessor, + num_samples: int, + n_histograms: int = 100, + seed: int = 42, +) -> np.ndarray: + """Propagate explicit per-variable uncertainty through a surrogate's own + predictive uncertainty. + + Draws ``num_samples`` per-variable samples from ``distributions`` (uniform / + normal / constant, in the caller's original units), evaluates the surrogate + once, then injects the surrogate's predictive std via the erfinv trick + (``sqrt(2) * erfinv(U) ~ N(0, 1)`` for ``U ~ Uniform(-1, 1)``), repeated + ``n_histograms`` times to characterise realisation-to-realisation spread. + + This mirrors the historical ``/manual_uq_propagation_with_uncertainty`` + Flask route rather than ``propagate_uq`` (Dakota-native, normal-only, no + predictive-uncertainty injection) -- see test_metamodeling_analytical.py. + + Returns: + A ``(n_histograms, num_samples)`` array of propagated output samples, + already inverse-transformed to the caller's original units. + """ + from scipy.special import erfinv + + # NOTE: input_vars/output_response/distributions must stay in the caller's + # original (unsanitized) form here -- preprocessor.input_variables and + # preprocessor.output_variables are keyed by original names (the + # preprocessor is fit before any sanitization happens), and + # preprocessor.transform() looks samples up by those same original column + # names. Sanitizing eagerly breaks both lookups for any var name containing + # characters sanitize_varnames rewrites. Dakota-safe names are obtained + # correctly below via preprocessor.input_variables[var].mapped_name. + samples = create_manual_uq_samples(input_vars, distributions, num_samples, seed) + df_samples = pd.DataFrame(samples) + SAMPLES_FILE = run_dir / "manual_uq_samples.csv" + df_samples.to_csv(SAMPLES_FILE, index=False) + + df_samples_transformed = preprocessor.transform(df_samples) + PROCESSED_SAMPLES_FILE = run_dir / "manual_uq_samples_processed.csv" + df_samples_transformed.to_csv(PROCESSED_SAMPLES_FILE, sep=" ", index=False) + + mapped_input_vars = [ + preprocessor.input_variables[var].mapped_name for var in input_vars + ] + mapped_response = preprocessor.output_variables[output_response].mapped_name + results = evaluate_sumo( + run_dir, + PROCESSED_TRAINING_FILE, + PROCESSED_SAMPLES_FILE, + mapped_input_vars, + mapped_response, + ) + + prediction_key = mapped_response + "_hat" + uncertainty_key = mapped_response + "_std_hat" + if prediction_key not in results or uncertainty_key not in results: + raise ValueError( + f"Cannot propagate uncertainty without '{prediction_key}' and " + f"'{uncertainty_key}' predictions. Available keys: {list(results.keys())}." + ) + + prediction = np.asarray(results[prediction_key]) + uncertainty = np.asarray(results[uncertainty_key]) + + rng = np.random.default_rng(seed) + all_results_transformed = np.empty((n_histograms, num_samples), dtype=float) + for i in range(n_histograms): + r = np.sqrt(2) * erfinv(rng.uniform(-1 + 1e-10, 1 - 1e-10, size=num_samples)) + all_results_transformed[i, :] = prediction + r * uncertainty + + all_samples_dict = {mapped_response: all_results_transformed.flatten().tolist()} + all_samples_original = preprocessor.inverse_transform(all_samples_dict) + return np.asarray(all_samples_original[output_response]).reshape( + n_histograms, num_samples + ) + + +def summarize_uncertainty_samples( + values: np.ndarray, num_bins: int | None = None +) -> dict: + """Compute histogram + boxplot summary statistics from propagated UQ samples. + + ``values`` is ``(n_histograms, num_samples)``: histogram bin heights are + averaged (with their std) across histogram realisations, while boxplot and + summary statistics are computed on the flattened pool of all samples. + """ + all_values_flat = values.flatten() + if num_bins is None: + num_bins = min(50, max(10, values.shape[1] // 10)) + + hist_min = float(np.percentile(all_values_flat, 1)) + hist_max = float(np.percentile(all_values_flat, 99)) + if hist_min == hist_max: + hist_range = max(1e-10, abs(hist_min) * 1e-6) + hist_min -= hist_range + hist_max += hist_range + + bin_edges = np.linspace(hist_min, hist_max, num_bins + 1) + histograms = np.array( + [ + np.histogram(values[i, :], bins=bin_edges, density=True)[0] + for i in range(values.shape[0]) + ] + ) + bin_means = np.mean(histograms, axis=0) + bin_stds = np.std(histograms, axis=0) + + q1 = float(np.percentile(all_values_flat, 25)) + median = float(np.percentile(all_values_flat, 50)) + q3 = float(np.percentile(all_values_flat, 75)) + iqr = q3 - q1 + whisker_min = max(hist_min, q1 - 1.5 * iqr) + whisker_max = min(hist_max, q3 + 1.5 * iqr) + outliers = all_values_flat[ + (all_values_flat < whisker_min) | (all_values_flat > whisker_max) + ] + + return { + "bins_start": hist_min, + "bins_end": hist_max, + "bin_means": bin_means.tolist(), + "bin_stds": bin_stds.tolist(), + "q1": q1, + "median": median, + "q3": q3, + "whisker_min": whisker_min, + "whisker_max": whisker_max, + "outliers": outliers.tolist(), + "mean": float(np.mean(all_values_flat)), + "std": float(np.std(all_values_flat)), + "min": float(np.min(all_values_flat)), + "max": float(np.max(all_values_flat)), + } + + def _parse_crossvalidation_outputlogs(log_output: str, N_CROSS_VALIDATION: int): variable_name_pattern = ( rf"Surrogate quality metrics \({N_CROSS_VALIDATION}-fold CV\) for (\w+):" @@ -242,6 +384,8 @@ def evaluate_sumo_manual_crossvalidation( input_vars: list[str], output_response: str, N_CROSS_VALIDATION: int = 5, + seed: int = 42, + has_eval_id_column: bool | None = None, ): input_vars = sanitize_varnames(input_vars) output_response = sanitize_varnames(output_response) @@ -251,7 +395,7 @@ def evaluate_sumo_manual_crossvalidation( indices = np.arange(n_samples) all_predictions = np.full(n_samples, np.nan) all_stds = np.full(n_samples, np.nan) - kf = KFold(n_splits=N_CROSS_VALIDATION, shuffle=True, random_state=42) + kf = KFold(n_splits=N_CROSS_VALIDATION, shuffle=True, random_state=seed) parse_warnings: list[str] = [] for fold, (_, val_idx) in enumerate(kf.split(indices)): @@ -266,6 +410,7 @@ def evaluate_sumo_manual_crossvalidation( output_response, validation_indices=val_idx.tolist(), dakota_conf_file=fold_run_dir / "dakota_config.in", + has_eval_id_column=has_eval_id_column, ) # V43 (root SPEC §T33 / B23): a Dakota fold run is non-deterministic in practice # (near-degenerate surrogate training can make Dakota abort a fold and never write @@ -893,11 +1038,14 @@ def evaluate_sobol_indices( from scipy.stats import norm, sobol_indices, uniform from scipy.stats.qmc import Sobol - input_vars = sanitize_varnames(input_vars) - response_var = sanitize_varnames(response_var) - distributions = { - sanitize_varname(k): sanitize_varnames_dict(v) for k, v in distributions.items() - } + # NOTE: input_vars/distributions must stay in the caller's original + # (unsanitized) form here -- preprocessor.input_variables is keyed by + # original names and preprocessor.transform() looks samples up by those + # same original column names (see propagate_manual_uq_with_uncertainty + # above for the same reasoning). response_var is already the mapped + # Dakota-safe name by the time it reaches this function. + # df_varying[input_vars] needs list, not tuple, indexing + input_vars = list(input_vars) # --- 1. Separate constant vs. varying input variables --- constant_vars: dict[str, float] = {} diff --git a/tests/test_api_contract.py b/tests/test_api_contract.py new file mode 100644 index 0000000..7d5e83e --- /dev/null +++ b/tests/test_api_contract.py @@ -0,0 +1,335 @@ +"""Contract tests for the itis-sumo consumer API. + +These assert the promises SPEC makes to whoever embeds itis-sumo, independently +of what Dakota happens to compute: the vocabulary of the public surface, what a +caller is allowed to pass, what comes back, which errors escape, and what happens +to working files. See SPEC V19cn, V20dm, V21pf, V22rs, V23er, V24af, V25sd. +""" + +from __future__ import annotations + +import dataclasses +import inspect +import json +import sys + +import numpy as np +import pandas as pd +import pytest + +from itis_sumo import api +from itis_sumo.api import ( + AlongAxesResult, + AxisSweep, + CrossValidationResult, + GridResult, + PreprocessingSpec, + SumoEngineError, + SumoError, + SumoInputError, + VariableSpec, + cross_validate, + evaluate_along_axes, + evaluate_grid, +) +from itis_sumo.api._session import SumoSession, _minimum_samples + +pytestmark = pytest.mark.unit + +VARIABLES = ["width", "height"] +RESPONSE = "stress" + + +def make_samples(n: int = 12) -> pd.DataFrame: + rng = np.random.default_rng(0) + width = rng.uniform(1.0, 5.0, n) + height = rng.uniform(10.0, 50.0, n) + return pd.DataFrame( + {"width": width, "height": height, "stress": 3.0 * width + 0.1 * height} + ) + + +class TestPublicSurface: + def test_exports_only_the_documented_names(self): + assert set(api.__all__) == { + "DEFAULT_SEED", + "AlongAxesResult", + "AxisSweep", + "CrossValidationResult", + "CVAccuracyMetrics", + "CorrelationResult", + "Direction", + "DistributionSpec", + "DomainSpec", + "GridResult", + "ParetoFrontResult", + "PreprocessingSpec", + "Scale", + "SobolResult", + "UncertaintyResult", + "SumoEngineError", + "SumoError", + "SumoInputError", + "SumoResultError", + "VariableSpec", + "compute_correlations", + "cross_validate", + "evaluate_along_axes", + "evaluate_grid", + "evaluate_sobol", + "evaluate_cv_metrics", + "evaluate_uncertainty", + "generate_grid_samples", + "generate_lhs_samples", + "optimize", + } + + @pytest.mark.parametrize("workflow", [cross_validate, evaluate_along_axes]) + def test_speaks_the_spec_vocabulary(self, workflow): + """SPEC V19cn: samples, variables, responses -- never jobs.""" + parameters = set(inspect.signature(workflow).parameters) + assert {"samples", "variables", "response"} <= parameters + assert not {"jobs", "function_jobs", "points", "records"} & parameters + + @pytest.mark.parametrize("workflow", [cross_validate, evaluate_along_axes]) + def test_hides_transform_vocabulary(self, workflow): + """SPEC V21pf: overrides are expressed in domain terms, not transforms.""" + parameters = set(inspect.signature(workflow).parameters) + assert ( + not { + "normalization", + "input_normalizations", + "output_normalizations", + "sign_switch", + "input_sign_switches", + "output_sign_switches", + "mapped_name", + } + & parameters + ) + + @pytest.mark.parametrize("workflow", [cross_validate, evaluate_along_axes]) + def test_hides_dakota_file_plumbing(self, workflow): + """SPEC §G: callers pass data and configuration, not file paths.""" + parameters = set(inspect.signature(workflow).parameters) + assert ( + not { + "run_dir", + "training_file", + "PROCESSED_TRAINING_FILE", + "samples_file", + "has_eval_id_column", + } + & parameters + ) + + def test_stochastic_workflow_seeds_itself(self): + """SPEC V25sd: a fixed default, not a required argument.""" + seed = inspect.signature(cross_validate).parameters["seed"] + assert seed.default == api.DEFAULT_SEED == 42 + + def test_deterministic_workflow_takes_no_seed(self): + assert "seed" not in inspect.signature(evaluate_along_axes).parameters + + def test_api_layer_is_free_of_web_and_platform_imports(self): + """Sibling of SPEC V4ty, for the consumer-facing layer.""" + for module in list(sys.modules): + if module.startswith("itis_sumo.api"): + source = inspect.getsource(sys.modules[module]) + assert "import flask" not in source + assert "osparc" not in source + + +class TestErrorTaxonomy: + def test_every_public_error_descends_from_sumo_error(self): + """SPEC V23er.""" + for name in api.__all__: + member = getattr(api, name) + if inspect.isclass(member) and issubclass(member, Exception): + assert issubclass(member, SumoError) + + def test_engine_error_carries_its_evidence(self): + error = SumoEngineError("boom", run_dir=None, stderr_tail="segfault") + assert "segfault" in str(error) + + +class TestSampleValidation: + """Judging whether a table can support a surrogate is itis-sumo's job.""" + + def test_rejects_non_tabular_input(self): + with pytest.raises(SumoInputError, match="DataFrame"): + cross_validate([{"width": 1.0}], VARIABLES, RESPONSE) + + def test_rejects_unknown_columns(self): + with pytest.raises(SumoInputError, match="not present"): + cross_validate(make_samples(), ["width", "depth"], RESPONSE) + + def test_rejects_a_column_used_as_both_variable_and_response(self): + with pytest.raises(SumoInputError, match="both"): + cross_validate(make_samples(), VARIABLES, "width") + + def test_rejects_repeated_variables(self): + with pytest.raises(SumoInputError, match="more than once"): + cross_validate(make_samples(), ["width", "width"], RESPONSE) + + def test_rejects_non_numeric_columns(self): + samples = make_samples() + samples["width"] = "wide" + with pytest.raises(SumoInputError, match="numeric"): + cross_validate(samples, VARIABLES, RESPONSE) + + def test_rejects_missing_values(self): + samples = make_samples() + samples.loc[0, "stress"] = np.nan + with pytest.raises(SumoInputError, match="missing or infinite"): + cross_validate(samples, VARIABLES, RESPONSE) + + def test_rejects_too_few_samples(self): + """Dakota aborts opaquely below max(5, n_variables + 1); we do not.""" + with pytest.raises(SumoInputError, match="samples are required"): + cross_validate(make_samples(2), VARIABLES, RESPONSE) + + def test_tabular_sufficiency_rule_keeps_floor_and_variable_count(self): + assert _minimum_samples(2) == 5 + assert _minimum_samples(5) == 6 + + def test_rejects_overrides_for_columns_not_in_play(self): + spec = PreprocessingSpec(overrides={"depth": VariableSpec()}) + with pytest.raises(SumoInputError, match="not in play"): + cross_validate(make_samples(), VARIABLES, RESPONSE, preprocessing=spec) + + def test_reports_unsupported_scale_instead_of_ignoring_it(self): + """An override that cannot be honoured must fail loudly (SPEC V21pf).""" + spec = PreprocessingSpec(overrides={"width": VariableSpec(scale="log")}) + with pytest.raises(SumoInputError, match="Logarithmic"): + cross_validate(make_samples(), VARIABLES, RESPONSE, preprocessing=spec) + + def test_rejects_a_single_fold(self): + with pytest.raises(SumoInputError, match="at least 2 folds"): + cross_validate(make_samples(), VARIABLES, RESPONSE, folds=1) + + def test_rejects_holding_a_variable_that_does_not_exist(self): + with pytest.raises(SumoInputError, match="not variables"): + evaluate_along_axes(make_samples(), VARIABLES, RESPONSE, at={"depth": 1.0}) + + +class TestResultShape: + """SPEC V22rs: original units, original names, JSON-serializable.""" + + def test_cross_validation_result_survives_json(self): + result = CrossValidationResult( + response="stress", + observed=[1.0], + predicted=[1.1], + predicted_std=[0.1], + warnings=[], + seed=42, + effective_config={"width": VariableSpec()}, + ) + payload = json.loads(json.dumps(dataclasses.asdict(result))) + assert payload["response"] == "stress" + assert payload["effective_config"]["width"] == {"scale": "linear"} + + def test_along_axes_result_survives_json(self): + result = AlongAxesResult( + response="stress", + sweeps={"width": AxisSweep(variable="width", x=[1.0], predicted=[2.0])}, + effective_config={"width": VariableSpec()}, + ) + payload = json.loads(json.dumps(dataclasses.asdict(result))) + assert payload["sweeps"]["width"]["variable"] == "width" + + @pytest.mark.parametrize("result_type", [CrossValidationResult, AlongAxesResult]) + def test_results_carry_no_transform_suffixes(self, result_type): + fields = {field.name for field in dataclasses.fields(result_type)} + assert not any(name.endswith(("_hat", "_std_hat")) for name in fields) + + def test_results_are_immutable(self): + sweep = AxisSweep(variable="width", x=[1.0], predicted=[2.0]) + with pytest.raises(dataclasses.FrozenInstanceError): + sweep.variable = "height" + + +class TestWorkingFileLifetime: + """SPEC V24af: discarded on success, preserved on failure.""" + + def test_run_directory_is_discarded_when_nothing_goes_wrong(self): + with SumoSession(make_samples(), VARIABLES, RESPONSE) as session: + run_dir = session._run_dir + assert run_dir.is_dir() + assert not run_dir.exists() + + def test_run_directory_survives_a_failure(self): + try: + with SumoSession(make_samples(), VARIABLES, RESPONSE) as session: + run_dir = session._run_dir + raise RuntimeError("engine exploded") + except RuntimeError: + pass + assert run_dir.is_dir() + + def test_workspace_keeps_its_files(self, tmp_path): + with SumoSession( + make_samples(), VARIABLES, RESPONSE, workspace=tmp_path + ) as session: + run_dir = session._run_dir + assert run_dir.is_dir() + assert tmp_path in run_dir.parents + + def test_engine_failures_are_translated_and_carry_the_run_directory( + self, monkeypatch + ): + import itis_sumo.api._session as session_module + + def explode(*args, **kwargs): + raise IndexError("map::at") + + monkeypatch.setattr(session_module, "evaluate_sumo_along_axes", explode) + + with pytest.raises(SumoEngineError) as caught: + evaluate_along_axes(make_samples(), VARIABLES, RESPONSE) + + assert caught.value.run_dir is not None + assert caught.value.run_dir.is_dir() + assert "map::at" in str(caught.value) + + +class TestEffectiveConfiguration: + """SPEC V21pf: readable, but not settable in transform terms.""" + + def test_reports_a_specification_for_every_column_in_play(self): + session = SumoSession(make_samples(), VARIABLES, RESPONSE) + assert set(session.effective_config) == {"width", "height", "stress"} + assert all( + isinstance(spec, VariableSpec) for spec in session.effective_config.values() + ) + + def test_defaults_are_reported_even_when_nothing_was_supplied(self): + session = SumoSession(make_samples(), VARIABLES, RESPONSE) + assert session.effective_config["width"].scale == "linear" + + +class TestGridContract: + def test_grid_rejects_unknown_grid_variables(self): + with pytest.raises(SumoInputError, match="not variables"): + evaluate_grid(make_samples(), VARIABLES, RESPONSE, grid_variables=["depth"]) + + def test_grid_rejects_too_few_points(self): + with pytest.raises(SumoInputError, match="at least 2"): + evaluate_grid( + make_samples(), + VARIABLES, + RESPONSE, + grid_variables=["width"], + points_per_variable=1, + ) + + def test_grid_result_is_typed_and_serializable(self): + result = GridResult( + response=RESPONSE, + grid_variables=("width", "height"), + data={"width": [1.0], "stress": [[2.0]]}, + effective_config={"width": VariableSpec()}, + ) + payload = json.loads(json.dumps(dataclasses.asdict(result))) + assert payload["grid_variables"] == ["width", "height"] diff --git a/tests/test_api_workflows.py b/tests/test_api_workflows.py new file mode 100644 index 0000000..d70ca1c --- /dev/null +++ b/tests/test_api_workflows.py @@ -0,0 +1,262 @@ +"""End-to-end tests for the consumer API, against a real Dakota surrogate. + +Nothing here is mocked: each test builds an actual surrogate and reads back what +Dakota produced. What is asserted is the promise the API makes -- results arrive +in the caller's own units under the caller's own column names -- rather than +particular numbers. +""" + +from __future__ import annotations + +import dataclasses +import json +import math + +import numpy as np +import pandas as pd +import pytest + +from itis_sumo.api import ( + DistributionSpec, + DomainSpec, + SumoInputError, + compute_correlations, + cross_validate, + evaluate_along_axes, + evaluate_cv_metrics, + evaluate_grid, + evaluate_sobol, + evaluate_uncertainty, + optimize, +) + +pytestmark = pytest.mark.integration + +VARIABLES = ["width", "height"] +RESPONSE = "stress" + +# Deliberately mismatched magnitudes: if anything leaked out in internal units, +# a height around 300 would come back looking nothing like a height. +WIDTH_RANGE = (1.0, 5.0) +HEIGHT_RANGE = (100.0, 500.0) + + +@pytest.fixture(scope="module") +def samples() -> pd.DataFrame: + rng = np.random.default_rng(20260818) + width = rng.uniform(*WIDTH_RANGE, 20) + height = rng.uniform(*HEIGHT_RANGE, 20) + return pd.DataFrame( + {"width": width, "height": height, "stress": 3.0 * width + 0.01 * height} + ) + + +class TestCrossValidation: + def test_predicts_every_sample_in_original_units(self, samples): + result = cross_validate(samples, VARIABLES, RESPONSE) + + assert result.response == RESPONSE + assert len(result.predicted) == len(samples) + assert result.observed == pytest.approx(samples[RESPONSE].tolist()) + + predicted = [value for value in result.predicted if not math.isnan(value)] + assert predicted, "no fold produced a prediction" + assert min(predicted) > 0.0 + assert max(predicted) < 10 * samples[RESPONSE].max() + + def test_echoes_the_seed_it_used(self, samples): + assert cross_validate(samples, VARIABLES, RESPONSE).seed == 42 + assert cross_validate(samples, VARIABLES, RESPONSE, seed=7).seed == 7 + + def test_is_reproducible_for_a_given_seed(self, samples): + first = cross_validate(samples, VARIABLES, RESPONSE, seed=7) + second = cross_validate(samples, VARIABLES, RESPONSE, seed=7) + assert first.predicted == pytest.approx(second.predicted, nan_ok=True) + + def test_result_is_json_serializable(self, samples): + result = cross_validate(samples, VARIABLES, RESPONSE) + payload = dataclasses.asdict(result) + # NaN is valid in Python's JSON dialect; the point is that nothing in the + # structure is a type json cannot reach. + assert json.loads(json.dumps(payload))["response"] == RESPONSE + + +class TestAlongAxes: + def test_sweeps_each_variable_over_its_observed_range(self, samples): + result = evaluate_along_axes( + samples, VARIABLES, RESPONSE, points_per_variable=9 + ) + + assert set(result.sweeps) == set(VARIABLES) + for variable, sweep in result.sweeps.items(): + assert sweep.variable == variable + assert len(sweep.x) == len(sweep.predicted) + observed = samples[variable] + assert min(sweep.x) >= observed.min() - abs(observed.min()) + assert max(sweep.x) <= observed.max() + abs(observed.max()) + + def test_keeps_variables_in_their_own_units(self, samples): + """A height must come back looking like a height, not like an x2.""" + result = evaluate_along_axes( + samples, VARIABLES, RESPONSE, points_per_variable=9 + ) + heights = result.sweeps["height"].x + assert min(heights) > WIDTH_RANGE[1] + assert max(heights) <= HEIGHT_RANGE[1] * 1.01 + + def test_honours_the_values_the_caller_holds_fixed(self, samples): + low = evaluate_along_axes( + samples, VARIABLES, RESPONSE, at={"height": 120.0}, points_per_variable=9 + ) + high = evaluate_along_axes( + samples, VARIABLES, RESPONSE, at={"height": 480.0}, points_per_variable=9 + ) + # stress rises with height, so holding height higher must lift the width sweep + assert sum(high.sweeps["width"].predicted) > sum(low.sweeps["width"].predicted) + + def test_result_is_json_serializable(self, samples): + result = evaluate_along_axes( + samples, VARIABLES, RESPONSE, points_per_variable=5 + ) + payload = json.loads(json.dumps(dataclasses.asdict(result))) + assert set(payload["sweeps"]) == set(VARIABLES) + + +class TestWorkingFiles: + def test_successful_run_leaves_nothing_behind(self, samples, tmp_path, monkeypatch): + monkeypatch.setenv("TMPDIR", str(tmp_path)) + evaluate_along_axes(samples, VARIABLES, RESPONSE, points_per_variable=5) + assert not list(tmp_path.glob("itis-sumo-*")) + + def test_workspace_keeps_the_evidence(self, samples, tmp_path): + evaluate_along_axes( + samples, + VARIABLES, + RESPONSE, + points_per_variable=5, + workspace=tmp_path, + ) + produced = list(tmp_path.rglob("processed_samples.dat")) + assert produced, "workspace should retain the training file" + + +class TestGrid: + def test_grid_preserves_original_names_and_units(self, samples): + result = evaluate_grid( + samples, + VARIABLES, + RESPONSE, + grid_variables=["width", "height"], + points_per_variable=5, + ) + assert result.response == RESPONSE + assert result.grid_variables == ("width", "height") + assert "stress" in result.data + assert min(result.data["width"]) >= WIDTH_RANGE[0] - 0.01 + assert max(result.data["height"]) <= HEIGHT_RANGE[1] + 0.01 + assert len(result.data["stress"]) == 5 + assert len(result.data["stress"][0]) == 5 + + +class TestSobol: + def test_returns_seeded_indices_for_explicit_distributions(self, samples): + distributions = { + "width": DistributionSpec("uniform", minimum=1.0, maximum=5.0), + "height": DistributionSpec("uniform", minimum=100.0, maximum=500.0), + } + result = evaluate_sobol( + samples, VARIABLES, RESPONSE, distributions=distributions, seed=7 + ) + assert result.response == RESPONSE + assert result.seed == 7 + assert set(result.indices) == set(VARIABLES) + assert set(result.second_order) <= set(VARIABLES) + + def test_requires_a_distribution_for_each_variable(self, samples): + with pytest.raises(SumoInputError, match="cover variables exactly"): + evaluate_sobol( + samples, + VARIABLES, + RESPONSE, + distributions={"width": DistributionSpec("constant", value=2.0)}, + ) + + +class TestDiagnostics: + def test_correlations_use_original_column_names(self, samples): + result = compute_correlations(samples, VARIABLES, RESPONSE) + assert result.response == RESPONSE + assert set(result.coefficients) == set(VARIABLES) + assert result.coefficients["width"]["pearson"] > 0.9 + + def test_correlations_reject_missing_columns(self, samples): + with pytest.raises(SumoInputError, match="do not contain"): + compute_correlations(samples, ["depth"], RESPONSE) + + def test_cv_metrics_compose_cross_validation(self, samples): + result = evaluate_cv_metrics(samples, VARIABLES, RESPONSE, seed=7) + assert result.response == RESPONSE + assert result.seed == 7 + assert result.root_mean_squared >= 0.0 + assert result.mean_abs >= 0.0 + + +class TestUncertainty: + def test_propagates_uncertainty_in_original_units(self, samples): + distributions = { + "width": DistributionSpec("uniform", minimum=1.0, maximum=5.0), + "height": DistributionSpec("uniform", minimum=100.0, maximum=500.0), + } + result = evaluate_uncertainty( + samples, + VARIABLES, + RESPONSE, + distributions=distributions, + num_samples=50, + n_histograms=10, + seed=7, + ) + assert result.response == RESPONSE + assert result.seed == 7 + assert result.bins_start < result.bins_end + assert result.q1 <= result.median <= result.q3 + assert result.mean > 0.0 + assert len(result.bin_means) == len(result.bin_stds) + + def test_requires_a_distribution_for_each_variable(self, samples): + with pytest.raises(SumoInputError, match="cover variables exactly"): + evaluate_uncertainty( + samples, + VARIABLES, + RESPONSE, + distributions={"width": DistributionSpec("constant", value=2.0)}, + ) + + +class TestOptimize: + def test_finds_a_pareto_front_over_the_domain(self, samples): + domains = { + "width": DomainSpec(minimum=WIDTH_RANGE[0], maximum=WIDTH_RANGE[1]), + "height": DomainSpec(minimum=HEIGHT_RANGE[0], maximum=HEIGHT_RANGE[1]), + } + result = optimize( + samples, + VARIABLES, + {RESPONSE: "minimize"}, + domains=domains, + max_evaluations=200, + ) + assert result.objectives == {RESPONSE: "minimize"} + assert set(result.data) >= {RESPONSE, "width", "height"} + assert min(result.data["width"]) >= WIDTH_RANGE[0] - 0.5 + assert max(result.data["width"]) <= WIDTH_RANGE[1] + 0.5 + + def test_requires_a_domain_for_each_variable(self, samples): + with pytest.raises(SumoInputError, match="cover variables exactly"): + optimize( + samples, + VARIABLES, + {RESPONSE: "minimize"}, + domains={"width": DomainSpec(minimum=1.0, maximum=5.0)}, + max_evaluations=200, + ) diff --git a/tests/test_dakota_funs_evaluate.py b/tests/test_dakota_funs_evaluate.py index e48cd84..5c8408b 100644 --- a/tests/test_dakota_funs_evaluate.py +++ b/tests/test_dakota_funs_evaluate.py @@ -2,6 +2,7 @@ from pathlib import Path +import numpy as np import pandas as pd import pytest @@ -10,6 +11,7 @@ _parse_crossvalidation_outputlogs, evaluate_sumo_crossvalidation, retrieve_csv_result, + summarize_uncertainty_samples, ) from itis_sumo.utils.helpers import create_run_dir @@ -48,7 +50,9 @@ def test_parse_crossvalidation_outputlogs_empty_string(): assert result == {} -def test_evaluate_sumo_crossvalidation_parses_captured_dakota_stdout(tmp_path, monkeypatch): +def test_evaluate_sumo_crossvalidation_parses_captured_dakota_stdout( + tmp_path, monkeypatch +): """Regression test for B22 (V37): evaluate_sumo_crossvalidation must return the metrics parsed from the stdout `DakotaObject.run` actually captures to `dakota_stdout.txt`, not the metrics from a hardcoded empty string. @@ -89,7 +93,9 @@ def fake_run(self, dakota_conf, run_dir): } -def test_evaluate_sumo_crossvalidation_no_stdout_file_returns_empty(tmp_path, monkeypatch): +def test_evaluate_sumo_crossvalidation_no_stdout_file_returns_empty( + tmp_path, monkeypatch +): """If DakotaObject.run doesn't produce a stdout file, parsing degrades to {} rather than raising (mirrors the `stdout_file.is_file()` guard).""" training_file = tmp_path / "df_processed_jobs.dat" @@ -109,7 +115,9 @@ def test_retrieve_csv_result_single_match(tmp_path): df = pd.DataFrame({"x": [1, 2, 3], "y": [10, 20, 30], "out": [100, 200, 300]}) df.to_csv(csv_file, index=False) - result = retrieve_csv_result(str(csv_file), inputs={"x": 2, "y": 20}, outputs=["out"]) + result = retrieve_csv_result( + str(csv_file), inputs={"x": 2, "y": 20}, outputs=["out"] + ) assert result == {"out": 200} @@ -145,3 +153,24 @@ def test_create_run_dir_creates_unique_directory(tmp_path): assert Path(dir2).is_dir() assert dir1 != dir2 assert str(dir1).startswith(str(tmp_path / "runs")) + + +class TestSummarizeUncertaintySamples: + def test_reports_mean_std_and_range_from_flattened_pool(self): + rng = np.random.default_rng(0) + values = rng.normal(loc=10.0, scale=2.0, size=(50, 500)) + + summary = summarize_uncertainty_samples(values) + + assert summary["mean"] == pytest.approx(10.0, abs=0.2) + assert summary["std"] == pytest.approx(2.0, abs=0.2) + assert summary["bins_start"] < summary["q1"] < summary["median"] < summary["q3"] + assert len(summary["bin_means"]) == len(summary["bin_stds"]) + + def test_flags_far_outliers_beyond_the_whiskers(self): + values = np.full((10, 100), 5.0) + values[0, 0] = 500.0 + + summary = summarize_uncertainty_samples(values) + + assert 500.0 in summary["outliers"] or summary["max"] == pytest.approx(500.0) From fee807f9e13badc0b41cdda5c27537c332d871b2 Mon Sep 17 00:00:00 2001 From: Javier Garcia Ordonez Date: Wed, 19 Aug 2026 16:36:31 +0200 Subject: [PATCH 2/5] fix: fall back to typing_extensions.Self on Python 3.10 (typing.Self is 3.11+ only) typing.Self (PEP 673) was only added in Python 3.11; pyproject.toml declares support down to 3.10, where importing it broke test collection for the api package's test suite once the CI matrix actually exercises 3.10. --- pyproject.toml | 1 + src/itis_sumo/api/_session.py | 7 ++++++- uv.lock | 2 ++ 3 files changed, 9 insertions(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 5c5eae2..d22e947 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -34,6 +34,7 @@ dependencies = [ "pydantic>=2.0,<3.0", "scikit-learn>=1.6,<2.0", "scipy>=1.15,<2.0", + "typing-extensions>=4.0; python_version < '3.11'", ] [project.urls] diff --git a/src/itis_sumo/api/_session.py b/src/itis_sumo/api/_session.py index 3d27dc2..0fba9e7 100644 --- a/src/itis_sumo/api/_session.py +++ b/src/itis_sumo/api/_session.py @@ -12,15 +12,20 @@ import logging import shutil +import sys import tempfile from collections.abc import Mapping, Sequence from pathlib import Path from types import TracebackType -from typing import Self import numpy as np import pandas as pd +if sys.version_info >= (3, 11): + from typing import Self +else: + from typing_extensions import Self + from itis_sumo.api.errors import ( SumoEngineError, SumoError, diff --git a/uv.lock b/uv.lock index 4f25c81..69650f2 100644 --- a/uv.lock +++ b/uv.lock @@ -383,6 +383,7 @@ dependencies = [ { name = "scipy", version = "1.15.3", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, { name = "scipy", version = "1.17.1", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.11.*'" }, { name = "scipy", version = "1.18.0", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12'" }, + { name = "typing-extensions", marker = "python_full_version < '3.11'" }, ] [package.dev-dependencies] @@ -408,6 +409,7 @@ requires-dist = [ { name = "pydantic", specifier = ">=2.0,<3.0" }, { name = "scikit-learn", specifier = ">=1.6,<2.0" }, { name = "scipy", specifier = ">=1.15,<2.0" }, + { name = "typing-extensions", marker = "python_full_version < '3.11'", specifier = ">=4.0" }, ] [package.metadata.requires-dev] From b245c60e44237af8453bbbf2a4fcccf69e30a614 Mon Sep 17 00:00:00 2001 From: Javier Garcia Ordonez Date: Wed, 19 Aug 2026 17:59:51 +0200 Subject: [PATCH 3/5] ci: surface skipped Python compatibility tests --- .github/workflows/ci.yml | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c5c83f2..74c4907 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -141,7 +141,12 @@ jobs: uv sync --all-extras --dev --python "${{ matrix.python-version }}" status=$? if [[ $status -ne 0 ]]; then - echo "::warning::Skipping Python ${{ matrix.python-version }}: dependencies do not currently resolve" + echo "::warning title=Python ${{ matrix.python-version }} not tested::Dependencies do not currently resolve; test steps were skipped." + { + echo "### Python ${{ matrix.python-version }}: tests skipped" + echo + echo "Dependencies do not currently resolve for this interpreter, so the test steps were skipped." + } >> "$GITHUB_STEP_SUMMARY" echo "compatible=false" >> "$GITHUB_OUTPUT" exit 0 fi From b8e0b8c83f97f4c43f8bff63bc9cb08a7218a4e8 Mon Sep 17 00:00:00 2001 From: Javier Garcia Ordonez Date: Wed, 19 Aug 2026 18:03:38 +0200 Subject: [PATCH 4/5] fix: correct consumer API LHS sampling --- src/itis_sumo/api/workflows.py | 5 +++-- tests/test_api_contract.py | 21 ++++++++++++++++++++- 2 files changed, 23 insertions(+), 3 deletions(-) diff --git a/src/itis_sumo/api/workflows.py b/src/itis_sumo/api/workflows.py index 6b886ab..fa76669 100644 --- a/src/itis_sumo/api/workflows.py +++ b/src/itis_sumo/api/workflows.py @@ -292,10 +292,11 @@ def generate_lhs_samples( if not domains: raise SumoInputError("At least one variable domain is required.") names = list(domains) - design = _lhs(n_samples, len(names), seed=seed) + design = _lhs(len(names), n_samples, seed=seed) return pd.DataFrame( { - name: design[i] * (domains[name].maximum - domains[name].minimum) + name: design[:, i] + * (domains[name].maximum - domains[name].minimum) + domains[name].minimum for i, name in enumerate(names) } diff --git a/tests/test_api_contract.py b/tests/test_api_contract.py index 7d5e83e..3f4a869 100644 --- a/tests/test_api_contract.py +++ b/tests/test_api_contract.py @@ -31,6 +31,7 @@ cross_validate, evaluate_along_axes, evaluate_grid, + generate_lhs_samples, ) from itis_sumo.api._session import SumoSession, _minimum_samples @@ -107,7 +108,6 @@ def test_hides_transform_vocabulary(self, workflow): } & parameters ) - @pytest.mark.parametrize("workflow", [cross_validate, evaluate_along_axes]) def test_hides_dakota_file_plumbing(self, workflow): """SPEC §G: callers pass data and configuration, not file paths.""" @@ -140,6 +140,25 @@ def test_api_layer_is_free_of_web_and_platform_imports(self): assert "osparc" not in source +class TestSampling: + def test_lhs_stratifies_each_domain_for_each_sample(self): + n_samples = 5 + domains = { + "width": api.DomainSpec(minimum=1.0, maximum=6.0), + "height": api.DomainSpec(minimum=100.0, maximum=600.0), + } + + result = generate_lhs_samples(domains, n_samples, seed=42) + + assert result.shape == (n_samples, len(domains)) + for name, domain in domains.items(): + normalized = (result[name] - domain.minimum) / ( + domain.maximum - domain.minimum + ) + bins = np.floor(normalized * n_samples).astype(int) + assert sorted(bins.tolist()) == list(range(n_samples)) + + class TestErrorTaxonomy: def test_every_public_error_descends_from_sumo_error(self): """SPEC V23er.""" From 1f87f95a46f42ace2e8e7ad2ca3d981862a4a51d Mon Sep 17 00:00:00 2001 From: Javier Garcia Ordonez Date: Wed, 19 Aug 2026 18:33:07 +0200 Subject: [PATCH 5/5] [fix] ruff format --- src/itis_sumo/api/workflows.py | 3 +-- tests/test_api_contract.py | 1 + 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/itis_sumo/api/workflows.py b/src/itis_sumo/api/workflows.py index fa76669..d5f8731 100644 --- a/src/itis_sumo/api/workflows.py +++ b/src/itis_sumo/api/workflows.py @@ -295,8 +295,7 @@ def generate_lhs_samples( design = _lhs(len(names), n_samples, seed=seed) return pd.DataFrame( { - name: design[:, i] - * (domains[name].maximum - domains[name].minimum) + name: design[:, i] * (domains[name].maximum - domains[name].minimum) + domains[name].minimum for i, name in enumerate(names) } diff --git a/tests/test_api_contract.py b/tests/test_api_contract.py index 3f4a869..34c667c 100644 --- a/tests/test_api_contract.py +++ b/tests/test_api_contract.py @@ -108,6 +108,7 @@ def test_hides_transform_vocabulary(self, workflow): } & parameters ) + @pytest.mark.parametrize("workflow", [cross_validate, evaluate_along_axes]) def test_hides_dakota_file_plumbing(self, workflow): """SPEC §G: callers pass data and configuration, not file paths."""