From fd2eabe42f39da042eaeb8ba66b01b414344c4a9 Mon Sep 17 00:00:00 2001 From: Yizheng Huang Date: Tue, 15 Sep 2026 16:14:00 -0700 Subject: [PATCH 1/2] sampler: validate inputs without restricting real-valued scores --- proeval/encoder/data.py | 3 +- proeval/sampler/baselines.py | 3 +- proeval/sampler/bq.py | 171 ++++++++++++++++++++++----- proeval/sampler/data.py | 53 +++++++-- proeval/sampler/pretrain_selector.py | 25 ++-- 5 files changed, 204 insertions(+), 51 deletions(-) diff --git a/proeval/encoder/data.py b/proeval/encoder/data.py index e0e23e3..28069bd 100644 --- a/proeval/encoder/data.py +++ b/proeval/encoder/data.py @@ -67,7 +67,7 @@ def load_benchmark_data( # Extract model names and scores model_columns = [col for col in df.columns if col.startswith("label_")] - model_names = [col.replace("label_", "") for col in model_columns] + model_names = [col[len("label_") :] for col in model_columns] labels = [] for name in model_names: @@ -279,4 +279,3 @@ def prepare_holdout_split( target_model=target_model, include_target_benchmark_in_training=True, ) - diff --git a/proeval/sampler/baselines.py b/proeval/sampler/baselines.py index aff4079..fbba1b3 100644 --- a/proeval/sampler/baselines.py +++ b/proeval/sampler/baselines.py @@ -121,7 +121,7 @@ def extract_model_predictions(df: pd.DataFrame, dataset_name: str = None) -> Tup or above 0.5 are failures, consistent with the rest of ProEval. """ model_columns = [col for col in df.columns if col.startswith('label_')] - model_names = [col.replace('label_', '') for col in model_columns] + model_names = [col[len('label_'):] for col in model_columns] model_data = {} for model_name in model_names: @@ -1059,4 +1059,3 @@ def run_rf_lure_evaluation(y_true: np.ndarray, embeddings: np.ndarray, surrogate_config = SurrogateConfig(use_rf=True) return run_incremental_lure_evaluation(y_true, embeddings, steps, seed_size, surrogate_config=surrogate_config) - diff --git a/proeval/sampler/bq.py b/proeval/sampler/bq.py index f144138..8634298 100644 --- a/proeval/sampler/bq.py +++ b/proeval/sampler/bq.py @@ -56,13 +56,14 @@ """ from dataclasses import dataclass, field -from numbers import Integral +from numbers import Integral, Real from typing import Any, List, Mapping, Optional, Sequence, Tuple, Union import numpy as np import pandas as pd from proeval.sampler.data import ( + _coerce_real_array, _prepare_score_features, extract_model_predictions, load_predictions, @@ -88,7 +89,74 @@ def _resolve_predictions( return predictions.predictions(data_dir=data_dir), predictions.name if isinstance(predictions, str): return load_predictions(predictions, data_dir=data_dir), predictions - return predictions, None + if isinstance(predictions, pd.DataFrame): + return predictions, None + raise ValueError( + "predictions must be a dataset name, pandas DataFrame, or Dataset" + ) + + +def _validate_noise_variance(noise_variance: float) -> float: + """Return a normalized positive finite observation-noise variance.""" + if isinstance(noise_variance, bool) or not isinstance(noise_variance, Real): + raise ValueError( + "noise_variance must be a positive finite real number; " + f"got {noise_variance!r}" + ) + try: + normalized = float(noise_variance) + except (OverflowError, TypeError, ValueError) as exc: + raise ValueError( + "noise_variance must be a positive finite real number; " + f"got {noise_variance!r}" + ) from exc + if ( + not np.isfinite(normalized) + or normalized <= 0 + or normalized <= 1.0 / np.finfo(float).max + ): + raise ValueError( + "noise_variance must be a positive finite real number with a " + f"finite reciprocal; got {noise_variance!r}" + ) + return normalized + + +def _validate_pretrain_indices( + pretrain_indices: Sequence[int], + *, + n_models: int, + target_index: int, +) -> List[int]: + """Validate and normalize source-model indices for prior construction.""" + if isinstance(pretrain_indices, (str, bytes)): + raise ValueError("pretrain_indices must be a sequence of model indices") + try: + normalized = list(pretrain_indices) + except TypeError as exc: + raise ValueError( + "pretrain_indices must be a sequence of model indices" + ) from exc + + if len(normalized) < 2: + raise ValueError("at least two pretrain source models are required") + if any( + isinstance(index, bool) or not isinstance(index, Integral) + for index in normalized + ): + raise ValueError("pretrain_indices must contain only integers") + + normalized = [int(index) for index in normalized] + if len(set(normalized)) != len(normalized): + raise ValueError("pretrain_indices must be unique") + invalid = [index for index in normalized if not 0 <= index < n_models] + if invalid: + raise ValueError( + f"pretrain_indices must be between 0 and {n_models - 1}; got {invalid}" + ) + if target_index in normalized: + raise ValueError("pretrain_indices must not include the target model") + return normalized # Result container @@ -104,8 +172,8 @@ class SamplingResult: posterior_var: Final posterior variance ``(n_samples,)``. prior_mean: Prior mean used ``(n_samples,)``. integral_variance: BQ integral posterior variance at each step ``(budget,)``. - This is the variance of the mean estimate, computed as - ``mean(posterior_diagonal_variance)``. + This is the posterior variance of the dataset mean under the + linear-kernel feature representation. """ estimates: np.ndarray @@ -200,10 +268,7 @@ def estimate( else: ordered_scores = scores - try: - score_array = np.asarray(ordered_scores, dtype=float) - except (TypeError, ValueError) as exc: - raise ValueError("scores must be numeric") from exc + score_array = _coerce_real_array(ordered_scores, name="scores") if score_array.ndim != 1 or len(score_array) != len(self.indices): raise ValueError( "scores must be one-dimensional with one value per selected item; " @@ -481,12 +546,15 @@ def _bq_active_sampling( budget: int, n_init: int = 0, noise_variance: float = 0.3, + rng=None, ) -> SamplingResult: """Run the core BQ active sampling loop. Returns a :class:`SamplingResult` with posterior-mean estimates at each step, the acquisition-order indices, and the final posterior. """ + if rng is None: + rng = np.random plan = _make_sampling_plan( test_x, u, @@ -495,7 +563,7 @@ def _bq_active_sampling( n_init=n_init, noise_variance=noise_variance, item_ids=list(range(test_x.shape[1])), - rng=np.random, + rng=rng, ) return _estimate_active_sampling( test_x, @@ -887,12 +955,11 @@ def _validate_plan_inputs( item_ids: Optional[Sequence[Any]], ) -> Tuple[np.ndarray, List[Any]]: """Validate and normalize source scores and row IDs for planning.""" - try: - # Match the row-major matrix produced by ``extract_model_predictions``. - # Stable memory layout also avoids tie-breaking drift in linear algebra. - score_array = np.ascontiguousarray(source_scores, dtype=float) - except (TypeError, ValueError) as exc: - raise ValueError("source_scores must contain only numeric values") from exc + # Match the row-major matrix produced by ``extract_model_predictions``. + # Stable memory layout also avoids tie-breaking drift in linear algebra. + score_array = np.ascontiguousarray( + _coerce_real_array(source_scores, name="source_scores") + ) if score_array.ndim != 2: raise ValueError( "source_scores must be a two-dimensional array with shape " @@ -943,11 +1010,13 @@ class BQPriorSampler: """Bayesian Quadrature active sampler with learned prior. Uses other models' predictions as features for a GP with a linear - kernel to efficiently estimate a target model's accuracy. + kernel to efficiently estimate a target model's mean error score. Args: - noise_variance: GP observation noise variance. - n_init: Number of random initial samples before active acquisition. + noise_variance: Positive, finite GP observation noise variance with a + finite floating-point reciprocal. + n_init: Non-negative number of random initial samples before active + acquisition. Example:: @@ -960,8 +1029,16 @@ def __init__( noise_variance: float = 0.3, n_init: int = 0, ): - self.noise_variance = noise_variance - self.n_init = n_init + self.noise_variance = _validate_noise_variance(noise_variance) + if isinstance(n_init, bool) or not isinstance(n_init, Integral): + raise ValueError( + f"n_init and budget must be integers; got n_init={n_init!r}" + ) + if n_init < 0: + raise ValueError( + f"n_init must be a non-negative integer; got {n_init!r}" + ) + self.n_init = int(n_init) def plan( self, @@ -1011,7 +1088,7 @@ def sample( target_model: Union[int, str] = "gemini25_flash", budget: int = 50, data_dir: str = None, - pretrain_indices: Optional[List[int]] = None, + pretrain_indices: Optional[Sequence[int]] = None, pretrain_mode: str = "gmm", reference_benchmarks: Optional[List[str]] = None, seed: Optional[int] = None, @@ -1029,9 +1106,10 @@ def sample( pretrain_indices: Optional explicit list of model indices to use as pre-training features. If ``None``, behaviour depends on *pretrain_mode*. - pretrain_mode: ``"all"`` (default) uses every model except the target. - ``"gmm"`` auto-selects models via GMM clustering on reference - benchmarks. Ignored when *pretrain_indices* is provided. + pretrain_mode: ``"gmm"`` (default) auto-selects models via GMM + clustering on reference benchmarks. ``"all"`` uses every + model except the target. Ignored when *pretrain_indices* is + provided. reference_benchmarks: Benchmarks for GMM clustering. Only used when ``pretrain_mode="gmm"``. ``None`` → auto-selected by benchmark category. @@ -1040,13 +1118,28 @@ def sample( Returns: :class:`SamplingResult` with estimates, selected indices, and posterior. """ - if seed is not None: - np.random.seed(seed) + if ( + isinstance(budget, bool) + or not isinstance(budget, Integral) + or budget < 1 + ): + raise ValueError(f"budget must be a positive integer; got {budget!r}") + if pretrain_indices is None and ( + not isinstance(pretrain_mode, str) + or pretrain_mode not in {"gmm", "all"} + ): + raise ValueError( + "pretrain_mode must be 'gmm' or 'all'; " + f"got {pretrain_mode!r}" + ) + use_all_sources = pretrain_indices is None and pretrain_mode == "all" # Load data (accepts a dataset name, a DataFrame, or a Dataset) df, dataset_name = _resolve_predictions(predictions, data_dir) pred_matrix, model_names = extract_model_predictions(df, dataset_name) + n_samples, n_models = pred_matrix.shape + _validate_active_sampling_budget(self.n_init, budget, n_samples) # Resolve target model if isinstance(target_model, str): @@ -1056,7 +1149,17 @@ def sample( ) target_idx = model_names.index(target_model) else: + if isinstance(target_model, bool) or not isinstance(target_model, Integral): + raise ValueError( + "target_model must be a model name or integer index; " + f"got {target_model!r}" + ) target_idx = int(target_model) + if not 0 <= target_idx < n_models: + raise ValueError( + f"target_model index must be between 0 and {n_models - 1}; " + f"got {target_idx}" + ) # Auto-select pretrain indices via GMM if requested if pretrain_indices is None and pretrain_mode == "gmm": @@ -1079,15 +1182,29 @@ def sample( reference_benchmarks=reference_benchmarks, ) + if pretrain_indices is None: + pretrain_indices = [ + index for index in range(n_models) if index != target_idx + ] + pretrain_indices = _validate_pretrain_indices( + pretrain_indices, + n_models=n_models, + target_index=target_idx, + ) + _, test_x, test_y, u, S = setup_train_test_split( - pred_matrix, target_idx, pretrain_indices + pred_matrix, + target_idx, + None if use_all_sources else pretrain_indices, ) + rng = np.random if seed is None else np.random.RandomState(seed) return _bq_active_sampling( test_x, test_y, u, S, budget=budget, n_init=self.n_init, noise_variance=self.noise_variance, + rng=rng, ) # Expose internal helpers for advanced users diff --git a/proeval/sampler/data.py b/proeval/sampler/data.py index 7f0135d..72692c5 100644 --- a/proeval/sampler/data.py +++ b/proeval/sampler/data.py @@ -25,6 +25,17 @@ import pandas as pd +def _coerce_real_array(values, *, name: str) -> np.ndarray: + """Convert numeric input to float without silently discarding imaginary parts.""" + unconverted = np.asarray(values) + if np.iscomplexobj(unconverted) or unconverted.dtype.kind in {"M", "m"}: + raise ValueError(f"{name} must contain only real numeric values") + try: + return np.asarray(values, dtype=float) + except (OverflowError, TypeError, ValueError) as exc: + raise ValueError(f"{name} must contain only numeric values") from exc + + def _prepare_score_features( source_scores: np.ndarray, ) -> Tuple[np.ndarray, np.ndarray, np.ndarray]: @@ -107,15 +118,16 @@ def load_embeddings( def extract_model_predictions( df: pd.DataFrame, dataset_name: str = None ) -> Tuple[np.ndarray, List[str]]: - """Extract a binary prediction matrix from a predictions DataFrame. + """Extract a real-valued score matrix from a predictions DataFrame. - Labels use the convention: **1=error, 0=correct** (measuring failure rate). + Binary labels conventionally use **1=error, 0=correct**. General finite + real-valued scores are also supported, as in the paper's score function. For DICES/DICES-T2I datasets, continuous error scores are binarised at 0.5: scores greater than or equal to 0.5 are failures. This matches the evaluation convention used elsewhere in ProEval. - For other datasets, ``label_`` columns are already error indicators - and are used directly. + For other datasets, ``label_`` columns are used directly and must contain + finite real scores. Args: df: Predictions DataFrame with ``label_`` columns. @@ -123,15 +135,36 @@ def extract_model_predictions( Returns: ``(prediction_matrix, model_names)`` where ``prediction_matrix`` has - shape ``(n_samples, n_models)`` with values in {0, 1} - (1=error, 0=correct). + shape ``(n_samples, n_models)``. Non-DICES scores retain their scale. """ - model_columns = [col for col in df.columns if col.startswith("label_")] - model_names = [col.replace("label_", "") for col in model_columns] + model_columns = [ + column + for column in df.columns + if isinstance(column, str) and column.startswith("label_") + ] + if not model_columns: + raise ValueError( + "predictions must contain at least one column named 'label_'" + ) + model_names = [column[len("label_") :] for column in model_columns] + if any(not name for name in model_names): + raise ValueError("prediction label columns must include a model name") + if len(set(model_columns)) != len(model_columns): + raise ValueError("prediction label column names must be unique") + if len(set(model_names)) != len(model_names): + raise ValueError("prediction label model names must be unique") + + numeric_labels = _coerce_real_array( + df[model_columns].to_numpy(), name="prediction labels" + ) + if numeric_labels.shape[0] == 0: + raise ValueError("predictions must contain at least one item") + if not np.all(np.isfinite(numeric_labels)): + raise ValueError("prediction labels must contain only finite values") model_data = {} - for model_name in model_names: - y_labels = df[f"label_{model_name}"].values + for model_index, model_name in enumerate(model_names): + y_labels = numeric_labels[:, model_index] # Use raw labels: 1=error, 0=correct. # For DICES/DICES_T2I, continuous error scores at or above 0.5 count # as failures, matching the failure-discovery threshold. diff --git a/proeval/sampler/pretrain_selector.py b/proeval/sampler/pretrain_selector.py index b9fb5be..757b5c9 100644 --- a/proeval/sampler/pretrain_selector.py +++ b/proeval/sampler/pretrain_selector.py @@ -32,10 +32,12 @@ """ import os +import warnings from typing import Dict, List, Optional, Tuple import numpy as np import pandas as pd +from sklearn.exceptions import ConvergenceWarning from sklearn.mixture import GaussianMixture from proeval.sampler.data import _default_data_dir @@ -89,7 +91,7 @@ def _load_benchmark_predictions( raise ValueError(f"No label_ columns found in {csv_path}") prediction_matrix = df[model_cols].values - model_names = [c.replace("label_", "") for c in model_cols] + model_names = [c[len("label_") :] for c in model_cols] return prediction_matrix, model_names @@ -152,24 +154,27 @@ def _find_optimal_clusters( ) -> int: """Select optimal GMM cluster count using BIC.""" n_samples = features.shape[0] - max_k = min(max_clusters, n_samples - 1) + n_distinct = np.unique(features, axis=0).shape[0] + max_k = min(max_clusters, n_samples - 1, n_distinct) if max_k < 2: - return 2 + return 1 bics: List[Tuple[int, float]] = [] for k in range(2, max_k + 1): try: - gmm = GaussianMixture( - n_components=k, random_state=random_state, - covariance_type="diag", reg_covar=1e-4, - ) - gmm.fit(features) + with warnings.catch_warnings(): + warnings.filterwarnings("error", category=ConvergenceWarning) + gmm = GaussianMixture( + n_components=k, random_state=random_state, + covariance_type="diag", reg_covar=1e-4, + ) + gmm.fit(features) bics.append((k, gmm.bic(features))) except Exception: # noqa: BLE001 continue if not bics: - return 2 + return 1 return min(bics, key=lambda x: x[1])[0] @@ -291,7 +296,7 @@ def select_pretrain_models_gmm( raise FileNotFoundError(f"Target predictions not found: {target_csv}") target_df = pd.read_csv(target_csv, nrows=0) # header only target_model_cols = [c for c in target_df.columns if c.startswith("label_")] - target_model_names = [c.replace("label_", "") for c in target_model_cols] + target_model_names = [c[len("label_") :] for c in target_model_cols] pretrain_indices: List[int] = [] pretrain_names: List[str] = [] From be8fb3e8d2aa32f23a8026b243b4083c9ecfd185 Mon Sep 17 00:00:00 2001 From: Yizheng Huang Date: Tue, 15 Sep 2026 16:17:48 -0700 Subject: [PATCH 2/2] sampler: reject non-real object scalars and separate GMM fixes --- proeval/sampler/data.py | 5 +++++ proeval/sampler/pretrain_selector.py | 25 ++++++++++--------------- 2 files changed, 15 insertions(+), 15 deletions(-) diff --git a/proeval/sampler/data.py b/proeval/sampler/data.py index 72692c5..bcd52cb 100644 --- a/proeval/sampler/data.py +++ b/proeval/sampler/data.py @@ -30,6 +30,11 @@ def _coerce_real_array(values, *, name: str) -> np.ndarray: unconverted = np.asarray(values) if np.iscomplexobj(unconverted) or unconverted.dtype.kind in {"M", "m"}: raise ValueError(f"{name} must contain only real numeric values") + if unconverted.dtype.kind == "O" and any( + isinstance(value, (complex, np.complexfloating, np.datetime64, np.timedelta64)) + for value in unconverted.flat + ): + raise ValueError(f"{name} must contain only real numeric values") try: return np.asarray(values, dtype=float) except (OverflowError, TypeError, ValueError) as exc: diff --git a/proeval/sampler/pretrain_selector.py b/proeval/sampler/pretrain_selector.py index 757b5c9..b9fb5be 100644 --- a/proeval/sampler/pretrain_selector.py +++ b/proeval/sampler/pretrain_selector.py @@ -32,12 +32,10 @@ """ import os -import warnings from typing import Dict, List, Optional, Tuple import numpy as np import pandas as pd -from sklearn.exceptions import ConvergenceWarning from sklearn.mixture import GaussianMixture from proeval.sampler.data import _default_data_dir @@ -91,7 +89,7 @@ def _load_benchmark_predictions( raise ValueError(f"No label_ columns found in {csv_path}") prediction_matrix = df[model_cols].values - model_names = [c[len("label_") :] for c in model_cols] + model_names = [c.replace("label_", "") for c in model_cols] return prediction_matrix, model_names @@ -154,27 +152,24 @@ def _find_optimal_clusters( ) -> int: """Select optimal GMM cluster count using BIC.""" n_samples = features.shape[0] - n_distinct = np.unique(features, axis=0).shape[0] - max_k = min(max_clusters, n_samples - 1, n_distinct) + max_k = min(max_clusters, n_samples - 1) if max_k < 2: - return 1 + return 2 bics: List[Tuple[int, float]] = [] for k in range(2, max_k + 1): try: - with warnings.catch_warnings(): - warnings.filterwarnings("error", category=ConvergenceWarning) - gmm = GaussianMixture( - n_components=k, random_state=random_state, - covariance_type="diag", reg_covar=1e-4, - ) - gmm.fit(features) + gmm = GaussianMixture( + n_components=k, random_state=random_state, + covariance_type="diag", reg_covar=1e-4, + ) + gmm.fit(features) bics.append((k, gmm.bic(features))) except Exception: # noqa: BLE001 continue if not bics: - return 1 + return 2 return min(bics, key=lambda x: x[1])[0] @@ -296,7 +291,7 @@ def select_pretrain_models_gmm( raise FileNotFoundError(f"Target predictions not found: {target_csv}") target_df = pd.read_csv(target_csv, nrows=0) # header only target_model_cols = [c for c in target_df.columns if c.startswith("label_")] - target_model_names = [c[len("label_") :] for c in target_model_cols] + target_model_names = [c.replace("label_", "") for c in target_model_cols] pretrain_indices: List[int] = [] pretrain_names: List[str] = []