diff --git a/CHANGELOG.md b/CHANGELOG.md index e773ac7..2e46ddf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,6 +12,7 @@ All notable changes to this project will be documented in this file. - `run_sequential()` allocates bounded extra replication to unresolved comparisons and feasibility boundaries, preserves raw replicate ids, and reports budgets, stopping reasons and simultaneous finite-horizon mean intervals under explicit bounded-score assumptions (#153). - Adaptive sessions queue known configurations and import compatible completed observations with persistent identity/provenance and duplicate protection. Versioned session schemas validate reopened journals and refuse unverifiable legacy storage (#154). - Opt-in `EvaluationCache` reuses grid evaluations by typed configuration, replicate namespace/id, model/scorer revision, fidelity, objective and annotation definitions; includes provenance, bypass, invalidation and conflicting-evidence checks (#154). +- `preference_sweep()` reports ranking/selection stability, regret, feasible Pareto alternatives, raw-unit practical equivalence and optional paired uncertainty under explicit normalization and preference assumptions, with exportable per-design summaries (#155). ## [0.3.0] — 2026-10-02 diff --git a/docs/api/decision.md b/docs/api/decision.md new file mode 100644 index 0000000..15367bb --- /dev/null +++ b/docs/api/decision.md @@ -0,0 +1,62 @@ +# Decision summaries + +Use existing results to see how choices change with preferences: + +```python +policy = PreferencePolicy( + normalization="minmax", + weights=[ + {"accuracy": weight, "cost": 1 - weight} + for weight in (0.25, 0.5, 0.75) + ], + equivalence={"accuracy": 0.01, "cost": 5.0}, +) +sweep = preference_sweep(results, observables, policy=policy, constraints=constraints) +frame = sweep.summary.to_dataframe(include_metadata=True) +``` + +Every preference vector is explicit, finite and nonnegative and is normalized +by its sum; omitted objectives have zero weight. `Observable.weight` is +replaced by these preference vectors. Minimized objectives contribute positively +and maximized objectives negatively to the weighted loss; smaller loss wins. +Session raw means in metadata override already-weighted objective columns. +Raw replicated tables are aggregated by design point, with duplicate or +incomplete replicate identities refused. The summary retains raw-unit means. + +Choose normalization explicitly. `minmax` uses the finite feasible alternatives +in this table, so changing that set can change rankings. `reference` uses +caller-provided raw-unit `reference_bounds` and allows values beyond the anchors +without clipping. `none` retains original units, making the weight ratios unit +dependent. Zero-range objectives contribute equally to every eligible design. +Outputs expose the selected policy, effective weights and actual anchors. + +`ranks` shows each design's competition rank across preferences; ties share a +rank. Summary metadata includes the best/worst/mean rank, maximum weighted-loss +regret and selection fraction. Tied winners split selection credit, so fractions +sum to one when eligible alternatives exist. These are fractions of the supplied +preference scenarios, not posterior probabilities or sampling confidence. +`pareto` identifies feasible nondominated alternatives regardless of preferences. +Infeasible/non-finite designs have NaN ranks and zero selection credit; if none +remain, the result reports no choice. Feasibility follows existing constraint +rules, including required standard errors for confidence constraints. + +`equivalent[a, b]` compares every objective's raw mean difference with the +specified practical tolerance; omitted tolerances are zero. This is a pairwise +relation and need not be transitive. It expresses practical similarity of point +estimates, not statistical evidence that the designs are equivalent. + +For raw data with matching replicate sets, supply `paired_reference` as a design +point id or config selector to attach the existing paired comparisons. Only +eligible designs are compared. The requested `paired_confidence` is split across +observables and the existing Bonferroni adjustment across alternatives. Bootstrap +coverage remains approximate; paired t assumptions remain unchanged. Choosing a +reference after observing outcomes and optional stopping are not corrected. +Unequal replication needs an explicitly chosen common replicate set before +calling this API. Without raw replicate identities, paired uncertainty cannot +be reconstructed from means and standard errors alone. + +::: trade_study.PreferencePolicy + +::: trade_study.PreferenceSweep + +::: trade_study.preference_sweep diff --git a/mkdocs.yml b/mkdocs.yml index 8f08b31..a99b7a6 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -72,6 +72,7 @@ nav: - Runner: api/runner.md - Adaptive Sessions: api/session.md - Paired Comparisons: api/paired.md + - Decision Summaries: api/decision.md - Post-hoc Sensitivity: api/sensitivity.md - Study: api/study.md - Scoring: api/scoring.md diff --git a/src/trade_study/__init__.py b/src/trade_study/__init__.py index f2f1a11..6168232 100644 --- a/src/trade_study/__init__.py +++ b/src/trade_study/__init__.py @@ -7,6 +7,7 @@ from ._scoring import coverage_curve, score from ._version import __version__ from .cache import EvaluationCache +from .decision import PreferencePolicy, PreferenceSweep, preference_sweep from .design import ( Factor, FactorConstraint, @@ -67,6 +68,8 @@ "PartialEvaluator", "Phase", "PredictionSupport", + "PreferencePolicy", + "PreferenceSweep", "RegimeSurrogate", "ReplicationPolicy", "ResultsTable", @@ -97,6 +100,7 @@ "plot_front", "plot_parallel", "plot_scores", + "preference_sweep", "recommend_bucketed_config", "recommend_per_regime", "reduce_factors", diff --git a/src/trade_study/decision.py b/src/trade_study/decision.py new file mode 100644 index 0000000..0898967 --- /dev/null +++ b/src/trade_study/decision.py @@ -0,0 +1,414 @@ +"""Preference sensitivity and practical equivalence over collected results.""" + +from __future__ import annotations + +from copy import deepcopy +from dataclasses import dataclass, field +from typing import TYPE_CHECKING, Any + +import numpy as np + +from .paired import paired_rank +from .protocols import Direction, ResultsTable + +if TYPE_CHECKING: + from numpy.typing import NDArray + + from .paired import PairedDifference + from .protocols import Constraint, Observable + + +@dataclass(frozen=True) +class PreferencePolicy: + """Explicit normalization, preferences, equivalence, and paired assumptions. + + Attributes: + weights: Preference vectors by observable name; nonnegative weights + are normalized to sum to one. Omitted objectives have zero weight. + normalization: ``minmax`` over finite feasible design means, + ``reference`` using reference_bounds, or ``none`` in raw units. + reference_bounds: Raw-unit lower/upper anchors for reference scaling. + equivalence: Pairwise practical-equivalence tolerances in raw units; + omitted observables require exact equality. + paired_confidence: Nominal joint confidence across requested comparisons + and observables. Bootstrap intervals are approximate. + paired_method: Existing paired-comparison method, ``bootstrap`` or ``t``. + n_boot: Bootstrap resample count. + seed: Bootstrap seed; preference vectors themselves are explicit. + """ + + weights: list[dict[str, float]] + normalization: str + reference_bounds: dict[str, tuple[float, float]] = field(default_factory=dict) + equivalence: dict[str, float] = field(default_factory=dict) + paired_confidence: float = 0.95 + paired_method: str = "bootstrap" + n_boot: int = 2000 + seed: int = 0 + + +@dataclass(frozen=True) +class PreferenceSweep: + """Exportable decision summaries and rankings under explicit preferences. + + Attributes: + summary: One raw-mean row per design, with decision metadata; + export using ResultsTable.to_dataframe(include_metadata=True). + weights: Effective normalized weights, preference by observable. + utilities: Direction-aware weighted losses, preference by design. + ranks: Competition ranks (ties share a rank); excluded designs are NaN. + feasible: Finite designs meeting the supplied constraints. + pareto: Feasible nondominated designs, independently of preferences. + equivalent: Pairwise raw-unit practical-equivalence matrix. + normalization_bounds: Actual anchors; empty for raw-unit normalization. + paired: Optional existing paired comparisons against a supplied reference. + policy: Copy of all requested preference and inference assumptions. + """ + + summary: ResultsTable + weights: NDArray[np.float64] + utilities: NDArray[np.float64] + ranks: NDArray[np.float64] + feasible: NDArray[np.bool_] + pareto: NDArray[np.bool_] + equivalent: NDArray[np.bool_] + normalization_bounds: dict[str, tuple[float, float]] + paired: dict[str, list[PairedDifference]] + policy: PreferencePolicy + + +def _design_table(results: ResultsTable) -> ResultsTable: + if results.metadata and len(results.metadata) != len(results.configs): + msg = "Decision results require row-aligned metadata" + raise ValueError(msg) + if not any("rep" in meta for meta in results.metadata): + return deepcopy(results) + if not all("rep" in meta and "design_point" in meta for meta in results.metadata): + msg = "Replicated decisions require complete design_point/rep identities" + raise ValueError(msg) + seen: set[tuple[int, int]] = set() + configs: dict[int, dict[str, Any]] = {} + for config, meta in zip(results.configs, results.metadata, strict=True): + identity = meta["design_point"], meta["rep"] + if identity in seen or ( + identity[0] in configs and config != configs[identity[0]] + ): + msg = "Duplicate replicate or conflicting design-point configuration" + raise ValueError(msg) + seen.add(identity) + configs[identity[0]] = config + return results.aggregate_replicates() + + +def _weights(policy: PreferencePolicy, names: list[str]) -> NDArray[np.float64]: + if not policy.weights or any(set(w) - set(names) for w in policy.weights): + msg = "Preference weights require nonempty vectors of known observables" + raise ValueError(msg) + weights = np.array( + [[w.get(name, 0) for name in names] for w in policy.weights], dtype=float + ) + if ( + not np.all(np.isfinite(weights)) + or np.any(weights < 0) + or np.any(weights.max(axis=1) <= 0) + ): + msg = "Preference weights must be finite, nonnegative and have positive totals" + raise ValueError(msg) + weights /= weights.max(axis=1, keepdims=True) + weights /= weights.sum(axis=1, keepdims=True) + if set(policy.equivalence) - set(names) or any( + not np.isfinite(v) or v < 0 for v in policy.equivalence.values() + ): + msg = "Equivalence requires nonnegative finite tolerances for known observables" + raise ValueError(msg) + return weights + + +def _normalize( + raw: NDArray[np.float64], + eligible: NDArray[np.bool_], + names: list[str], + policy: PreferencePolicy, +) -> tuple[NDArray[np.float64], dict[str, tuple[float, float]]]: + if policy.normalization == "none": + return raw.copy(), {} + if policy.normalization == "minmax": + bounds = ( + { + name: (float(raw[eligible, j].min()), float(raw[eligible, j].max())) + for j, name in enumerate(names) + } + if np.any(eligible) + else dict.fromkeys(names, (np.nan, np.nan)) + ) + elif policy.normalization == "reference": + if set(policy.reference_bounds) != set(names): + msg = "reference_bounds must specify every decision observable exactly" + raise ValueError(msg) + bounds = dict(policy.reference_bounds) + for name, anchors in bounds.items(): + if ( + len(anchors) != 2 + or not np.all(np.isfinite(anchors)) + or anchors[0] > anchors[1] + ): + msg = f"Invalid normalization bounds for {name!r}" + raise ValueError(msg) + else: + msg = "normalization must be 'minmax', 'reference', or 'none'" + raise ValueError(msg) + normalized = np.full(raw.shape, np.nan) + for j, name in enumerate(names): + lower, upper = bounds[name] + if upper == lower: + if np.any(raw[eligible, j] != lower): + msg = f"Zero-range reference bounds do not describe {name!r}" + raise ValueError(msg) + normalized[eligible, j] = 0 + else: + normalized[eligible, j] = _scaled_values(raw[eligible, j], lower, upper) + return normalized, bounds + + +def _scaled_values( + values: NDArray[np.float64], lower: float, upper: float +) -> NDArray[np.float64]: + if values.size and not np.isfinite(upper - lower): + msg = "Normalization exceeds finite floating range; rescale observable units" + raise ValueError(msg) + try: + with np.errstate(over="raise", invalid="raise", divide="raise"): + return (values - lower) / (upper - lower) + except FloatingPointError as error: + msg = "Normalization exceeds finite floating range; rescale observable units" + raise ValueError(msg) from error + + +def _ranks( + utilities: NDArray[np.float64], + eligible: NDArray[np.bool_], +) -> tuple[NDArray[np.float64], NDArray[np.float64]]: + ranks = np.full(utilities.shape, np.nan) + selection = np.zeros(utilities.shape[1]) + if np.any(eligible): + for i, losses in enumerate(utilities[:, eligible]): + ranks[i, eligible] = ( + np.searchsorted(np.sort(losses), losses, side="left") + 1 + ) + winners = ranks[i] == 1 + selection[winners] += 1 / winners.sum() + selection /= len(utilities) + return ranks, selection + + +def _alternatives( + raw: NDArray[np.float64], + eligible: NDArray[np.bool_], + observables: list[Observable], + policy: PreferencePolicy, +) -> tuple[NDArray[np.bool_], NDArray[np.bool_]]: + oriented = raw * np.array([ + 1 if o.direction == Direction.MINIMIZE else -1 for o in observables + ]) + pareto = np.zeros(len(raw), dtype=bool) + for index in np.flatnonzero(eligible): + pareto[index] = not np.any( + np.all(oriented[eligible] <= oriented[index], axis=1) + & np.any(oriented[eligible] < oriented[index], axis=1) + ) + equivalent = eligible[:, None] & eligible[None, :] + safe = np.where(np.isfinite(raw), raw, 0) + for j, observable in enumerate(observables): + equivalent &= np.abs( + safe[:, j, None] - safe[None, :, j] + ) <= policy.equivalence.get(observable.name, 0) + return pareto, equivalent + + +def _paired( + results: ResultsTable, + designs: ResultsTable, + eligible: NDArray[np.bool_], + observables: list[Observable], + reference: int | dict[str, Any], + policy: PreferencePolicy, +) -> dict[str, list[PairedDifference]]: + if not results.metadata or not all( + "rep" in m and "design_point" in m for m in results.metadata + ): + msg = "Paired decision summaries require raw design_point/rep observations" + raise ValueError(msg) + ids = { + meta["design_point"] + for meta, keep in zip(designs.metadata, eligible, strict=True) + if keep + } + rows = [i for i, meta in enumerate(results.metadata) if meta["design_point"] in ids] + filtered = ResultsTable( + [results.configs[i] for i in rows], + results.scores[rows], + results.observable_names, + metadata=[results.metadata[i] for i in rows], + ) + level = 1 - (1 - policy.paired_confidence) / len(observables) + return { + o.name: paired_rank( + filtered, + o.name, + reference, + maximize=o.direction == Direction.MAXIMIZE, + confidence=level, + method=policy.paired_method, + n_boot=policy.n_boot, + seed=policy.seed, + ) + for o in observables + } + + +def _decorate_summary(report: PreferenceSweep, selection: NDArray[np.float64]) -> None: + for i, meta in enumerate(report.summary.metadata): + valid = bool(report.feasible[i]) + regrets = ( + report.utilities[:, i] - np.nanmin(report.utilities, axis=1) + if valid + else np.array([np.nan]) + ) + meta["decision"] = { + "feasible": valid, + "pareto": bool(report.pareto[i]), + "selection_fraction": float(selection[i]), + "best_rank": float(report.ranks[:, i].min()) if valid else np.nan, + "worst_rank": float(report.ranks[:, i].max()) if valid else np.nan, + "mean_rank": float(report.ranks[:, i].mean()) if valid else np.nan, + "max_regret": float(regrets.max()), + "equivalent_to": np.flatnonzero(report.equivalent[i]).tolist(), + "normalization": report.policy.normalization, + "normalization_bounds": report.normalization_bounds, + "preferences_evaluated": len(report.weights), + } + + +def _decision_table( + results: ResultsTable, names: list[str], constraints: list[Constraint] +) -> tuple[ResultsTable, NDArray[np.bool_]]: + designs = _design_table(results) + raw = np.asarray( + designs.scores[:, [designs.observable_names.index(n) for n in names]], + dtype=float, + ).copy() + for i, meta in enumerate(designs.metadata): + for j, name in enumerate(names): + raw[i, j] = meta.get("scores", {}).get(name, raw[i, j]) + eligible = designs.feasible(constraints) & np.all(np.isfinite(raw), axis=1) + table = ResultsTable( + deepcopy(designs.configs), + raw, + names, + annotations=designs.annotations, + annotation_names=designs.annotation_names, + metadata=deepcopy(designs.metadata) + if designs.metadata + else [{} for _ in designs.configs], + ) + return table, eligible + + +def _utilities( + normalized: NDArray[np.float64], + eligible: NDArray[np.bool_], + weights: NDArray[np.float64], + observables: list[Observable], +) -> NDArray[np.float64]: + signs = np.array([ + 1 if o.direction == Direction.MINIMIZE else -1 for o in observables + ]) + utilities = np.full((len(weights), len(normalized)), np.nan) + try: + with np.errstate(over="raise", invalid="raise"): + utilities[:, eligible] = weights @ (normalized[eligible] * signs).T + except FloatingPointError as error: + msg = "Weighted loss exceeds finite floating range; rescale observable units" + raise ValueError(msg) from error + return utilities + + +def preference_sweep( + results: ResultsTable, + observables: list[Observable], + *, + policy: PreferencePolicy, + constraints: list[Constraint] | None = None, + paired_reference: int | dict[str, Any] | None = None, +) -> PreferenceSweep: + """Assess choices across explicit preferences without evaluating a simulator. + + Args: + results: Existing raw-replicate or per-design results. Session raw means + in metadata override weighted score columns. + observables: Decision objectives with directions. Observable.weight is + replaced by the explicit preference vectors, avoiding double weights. + policy: Explicit normalization, preferences and equivalence assumptions. + constraints: Feasibility restrictions, using existing ResultsTable rules. + paired_reference: Optional raw design-point id or config selector for + existing paired comparisons. Requires matching replicate sets. + + Returns: + Per-design exportable summaries, ranks and regret across preferences, + feasible Pareto alternatives, pairwise practical equivalence and optional + paired uncertainty. Tied winners share selection credit. With no finite + feasible alternatives, selection fractions are zero and ranks are NaN. + + Raises: + ValueError: If objective names, table shape, preferences, normalization, + or paired inference assumptions are invalid. + + Notes: + Minmax anchors depend on this feasible set. Reference anchors do not clip + values outside their range. Practical equivalence compares raw means, + not statistical evidence. Paired bootstrap/t assumptions and approximate + coverage remain those of existing paired comparisons; optional stopping + and selecting a reference after seeing results are not corrected here. + """ + names = [o.name for o in observables] + if ( + not names + or len(set(names)) != len(names) + or set(names) - set(results.observable_names) + ): + msg = "Decision objectives must be distinct names in the results" + raise ValueError(msg) + if results.scores.shape != (len(results.configs), len(results.observable_names)): + msg = "Decision score matrix has an incompatible row/column shape" + raise ValueError(msg) + if ( + not 0 < policy.paired_confidence < 1 + or policy.paired_method not in {"bootstrap", "t"} + or policy.n_boot < 1 + ): + msg = "Invalid paired inference assumptions" + raise ValueError(msg) + policy = deepcopy(policy) + summary, eligible = _decision_table(results, names, constraints or []) + raw = np.asarray(summary.scores, dtype=np.float64) + weights = _weights(policy, names) + normalized, bounds = _normalize(raw, eligible, names, policy) + utilities = _utilities(normalized, eligible, weights, observables) + ranks, selection = _ranks(utilities, eligible) + pareto, equivalent = _alternatives(raw, eligible, observables, policy) + report = PreferenceSweep( + summary, + weights, + utilities, + ranks, + eligible, + pareto, + equivalent, + bounds, + {} + if paired_reference is None + else _paired(results, summary, eligible, observables, paired_reference, policy), + policy, + ) + _decorate_summary(report, selection) + return report diff --git a/tests/test_decision.py b/tests/test_decision.py new file mode 100644 index 0000000..0c188b1 --- /dev/null +++ b/tests/test_decision.py @@ -0,0 +1,257 @@ +"""Explicit preferences, normalization, raw equivalence, and paired summaries.""" + +from __future__ import annotations + +import warnings + +import numpy as np +import pytest + +from trade_study import ( + Constraint, + Direction, + Observable, + PreferencePolicy, + ResultsTable, + preference_sweep, +) + +_OBS = [ + Observable("cost", Direction.MINIMIZE), + Observable("accuracy", Direction.MAXIMIZE), +] + + +def _table() -> ResultsTable: + return ResultsTable( + [{"design": "a"}, {"design": "b"}, {"design": "c"}], + np.array([[100, 0.9], [200, 0.99], [300, 0.8]]), + ["cost", "accuracy"], + ) + + +def test_mixed_units_directions_and_preference_sensitive_choices() -> None: + policy = PreferencePolicy( + [{"accuracy": 1}, {"cost": 1}, {"cost": 1, "accuracy": 1}], + "minmax", + ) + result = preference_sweep(_table(), _OBS, policy=policy) + np.testing.assert_array_equal(result.pareto, [True, True, False]) + np.testing.assert_array_equal(result.ranks, [[2, 1, 3], [1, 2, 3], [1, 2, 3]]) + selection = [m["decision"]["selection_fraction"] for m in result.summary.metadata] + np.testing.assert_allclose(selection, [2 / 3, 1 / 3, 0]) + assert result.normalization_bounds == {"cost": (100, 300), "accuracy": (0.8, 0.99)} + scaled = _table() + scaled.scores[:, 0] *= 1000 + scaled.scores[:, 1] *= 100 + other_units = preference_sweep(scaled, _OBS, policy=policy) + np.testing.assert_array_equal(result.ranks, other_units.ranks) + frame = result.summary.to_dataframe() + assert "meta.decision.selection_fraction" in frame.columns + assert result.summary.metadata[0]["decision"]["best_rank"] == 1 + assert result.summary.metadata[0]["decision"]["worst_rank"] == 2 + + +def test_normalization_is_explicit_and_reference_values_are_not_clipped() -> None: + weights = [{"cost": 0.05, "accuracy": 0.95}] + raw = preference_sweep(_table(), _OBS, policy=PreferencePolicy(weights, "none")) + normalized = preference_sweep( + _table(), _OBS, policy=PreferencePolicy(weights, "minmax") + ) + assert np.argmin(raw.utilities[0]) == 0 + assert np.argmin(normalized.utilities[0]) == 1 + assert raw.normalization_bounds == {} + reference = preference_sweep( + _table(), + _OBS, + policy=PreferencePolicy( + [{"cost": 1}], + "reference", + reference_bounds={"cost": (0, 100), "accuracy": (0, 1)}, + ), + ) + np.testing.assert_array_equal(reference.utilities[0], [1, 2, 3]) + + +def test_weight_perturbations_ties_and_shared_selection_credit() -> None: + table = ResultsTable( + [{"design": "a"}, {"design": "b"}], + np.array([[0.0, 0.0], [1.0, 1.0]]), + ["cost", "accuracy"], + ) + policy = PreferencePolicy( + [ + {"cost": 0.49, "accuracy": 0.51}, + {"cost": 0.5, "accuracy": 0.5}, + {"cost": 0.51, "accuracy": 0.49}, + ], + "minmax", + ) + result = preference_sweep(table, _OBS, policy=policy) + np.testing.assert_array_equal(result.ranks, [[2, 1], [1, 1], [1, 2]]) + assert [m["decision"]["selection_fraction"] for m in result.summary.metadata] == [ + 0.5, + 0.5, + ] + # Policy is copied so later caller mutation cannot change reported assumptions. + original = result.policy.weights[0]["cost"] + policy.weights[0]["cost"] = 999 + table.scores[0, 0] = 10 + assert result.summary.scores[0, 0] == 0 + assert result.policy.weights[0]["cost"] == original + + +def test_zero_range_objectives_and_pairwise_raw_unit_equivalence() -> None: + table = ResultsTable( + [{"design": i} for i in range(3)], + np.array([[10, 0.9], [10, 0.905], [10, 0.8]]), + ["cost", "accuracy"], + ) + result = preference_sweep( + table, + _OBS, + policy=PreferencePolicy( + [{"cost": 1}], "minmax", equivalence={"accuracy": 0.01} + ), + ) + np.testing.assert_array_equal(result.ranks, np.ones((1, 3))) + assert result.equivalent[0, 1] + assert not result.equivalent[0, 2] + assert result.summary.metadata[0]["decision"][ + "selection_fraction" + ] == pytest.approx(1 / 3) + chain = ResultsTable( + [{"x": x} for x in (0, 1, 2)], np.array([[0], [0.09], [0.18]]), ["cost"] + ) + pairwise = preference_sweep( + chain, + _OBS[:1], + policy=PreferencePolicy([{"cost": 1}], "none", equivalence={"cost": 0.1}), + ) + assert pairwise.equivalent[0, 1] + assert pairwise.equivalent[1, 2] + assert not pairwise.equivalent[0, 2] + + +def test_infeasible_and_nonfinite_designs_do_not_set_normalization_anchors() -> None: + table = _table() + table.scores[2, 1] = np.nan + result = preference_sweep( + table, + _OBS, + policy=PreferencePolicy([{"cost": 1}], "minmax"), + constraints=[Constraint("minimum_cost", "cost", ">=", 150)], + ) + np.testing.assert_array_equal(result.feasible, [False, True, False]) + assert result.normalization_bounds["cost"] == (200, 200) + assert np.isnan(result.ranks[0, 0]) + assert np.isnan(result.ranks[0, 2]) + assert result.summary.metadata[1]["decision"]["selection_fraction"] == 1 + with warnings.catch_warnings(): + warnings.simplefilter("error") + none = preference_sweep( + table, + _OBS, + policy=PreferencePolicy([{"cost": 1}], "minmax"), + constraints=[Constraint("impossible", "cost", "<", 0)], + ) + assert not np.any(none.feasible) + assert np.all(np.isnan(none.ranks)) + assert all(m["decision"]["selection_fraction"] == 0 for m in none.summary.metadata) + + +def test_adaptive_raw_means_override_weighted_columns() -> None: + table = ResultsTable( + [{"x": 0}, {"x": 1}], + np.array([[1000.0, 1.0], [2000.0, 2.0]]), + ["cost", "accuracy"], + metadata=[ + {"scores": {"cost": 1, "accuracy": 1}}, + {"scores": {"cost": 2, "accuracy": 2}}, + ], + ) + result = preference_sweep( + table, + [Observable("cost", Direction.MINIMIZE, weight=1000), _OBS[1]], + policy=PreferencePolicy([{"cost": 1}], "none"), + constraints=[Constraint("raw_cost", "cost", "<=", 1.5)], + ) + np.testing.assert_array_equal(result.summary.scores, [[1, 1], [2, 2]]) + np.testing.assert_array_equal(result.feasible, [True, False]) + + +def test_replicated_means_and_existing_paired_uncertainty() -> None: + rows = [(design, rep) for design in range(2) for rep in range(4)] + table = ResultsTable( + [{"design": d} for d, _ in rows], + np.array([[100 + 10 * d + rep, 0.8 + 0.1 * d + rep * 0.01] for d, rep in rows]), + ["cost", "accuracy"], + metadata=[{"design_point": d, "rep": rep} for d, rep in rows], + ) + result = preference_sweep( + table, + _OBS, + policy=PreferencePolicy( + [{"accuracy": 1}], "minmax", paired_confidence=0.9, n_boot=50 + ), + paired_reference=0, + ) + assert len(result.summary.configs) == 2 + assert result.summary.metadata[0]["n_reps"] == 4 + assert result.paired["cost"][0].mean == pytest.approx(10) + assert result.paired["accuracy"][0].mean == pytest.approx(0.1) + assert result.paired["accuracy"][0].confidence == pytest.approx(0.95) + with pytest.raises(ValueError, match="raw design_point/rep"): + preference_sweep( + _table(), + _OBS, + policy=PreferencePolicy([{"cost": 1}], "none"), + paired_reference=0, + ) + + +@pytest.mark.parametrize( + "policy", + [ + PreferencePolicy([], "minmax"), + PreferencePolicy([{"other": 1}], "minmax"), + PreferencePolicy([{"cost": -1}], "minmax"), + PreferencePolicy([{"cost": 0}], "minmax"), + PreferencePolicy([{"cost": np.nan}], "minmax"), + PreferencePolicy([{"cost": 1}], "bad"), + PreferencePolicy([{"cost": 1}], "reference"), + PreferencePolicy( + [{"cost": 1}], + "reference", + reference_bounds={"cost": (1, 0), "accuracy": (0, 1)}, + ), + PreferencePolicy( + [{"cost": 1}], + "reference", + reference_bounds={"cost": (1, 1), "accuracy": (0, 1)}, + ), + PreferencePolicy([{"cost": 1}], "none", equivalence={"cost": -1}), + PreferencePolicy([{"cost": 1}], "none", paired_confidence=1), + ], +) +def test_invalid_assumptions_are_rejected(policy: PreferencePolicy) -> None: + with pytest.raises( + ValueError, + match=r"Preference|normalization|reference_bounds|Zero-range|Equivalence|paired", + ): + preference_sweep(_table(), _OBS, policy=policy) + + +def test_duplicate_or_incomplete_raw_identities_are_rejected() -> None: + table = ResultsTable( + [{"x": 0}, {"x": 0}], + np.array([[1.0], [2.0]]), + ["cost"], + metadata=[{"design_point": 0, "rep": 0}, {"design_point": 0, "rep": 0}], + ) + policy = PreferencePolicy([{"cost": 1}], "none") + with pytest.raises(ValueError, match="Duplicate replicate"): + preference_sweep(table, _OBS[:1], policy=policy) + del table.metadata[1]["design_point"] + with pytest.raises(ValueError, match="complete design_point/rep"): + preference_sweep(table, _OBS[:1], policy=policy)