From 784a1f6a56d6b6855fb4523032e688ee1a197b88 Mon Sep 17 00:00:00 2001 From: Jayant-kernel Date: Fri, 25 Sep 2026 20:04:41 +0530 Subject: [PATCH 1/2] [ENH] Include estimation procedure metadata Signed-off-by: Jayant-kernel --- openml/evaluations/__init__.py | 8 +++- openml/evaluations/functions.py | 38 +++++++++++++-- .../test_evaluation_functions.py | 46 +++++++++++++++++++ 3 files changed, 86 insertions(+), 6 deletions(-) diff --git a/openml/evaluations/__init__.py b/openml/evaluations/__init__.py index b56d0c2d5c..29344b03a1 100644 --- a/openml/evaluations/__init__.py +++ b/openml/evaluations/__init__.py @@ -1,10 +1,16 @@ # License: BSD 3-Clause from .evaluation import OpenMLEvaluation -from .functions import list_evaluation_measures, list_evaluations, list_evaluations_setups +from .functions import ( + list_estimation_procedures, + list_evaluation_measures, + list_evaluations, + list_evaluations_setups, +) __all__ = [ "OpenMLEvaluation", + "list_estimation_procedures", "list_evaluation_measures", "list_evaluations", "list_evaluations_setups", diff --git a/openml/evaluations/functions.py b/openml/evaluations/functions.py index f4e07c1b80..e9bfa3030f 100644 --- a/openml/evaluations/functions.py +++ b/openml/evaluations/functions.py @@ -156,18 +156,46 @@ def list_evaluation_measures() -> list[str]: return openml._backend.evaluation_measure.list() -def list_estimation_procedures() -> list[str]: - """Return list of evaluation procedures available. +@overload +def list_estimation_procedures( + output_format: Literal["dataframe"], +) -> pd.DataFrame: ... + + +@overload +def list_estimation_procedures( + output_format: Literal["dict"] = ..., +) -> dict[int, dict[str, object]]: ... + + +def list_estimation_procedures( + output_format: Literal["dict", "dataframe"] = "dict", +) -> dict[int, dict[str, object]] | pd.DataFrame: + """Return the estimation procedures available on OpenML. The function performs an API call to retrieve the entire list of - evaluation procedures' names that are available. + evaluation procedures. Each procedure includes its ID, task type, + name, and type. + + Parameters + ---------- + output_format : {"dict", "dataframe"}, default="dict" + The format of the returned procedures. The dictionary format maps + procedure IDs to their remaining metadata. The DataFrame format has + one row per procedure, including an ``id`` column. Returns ------- - list + dict or pandas.DataFrame + The available estimation procedures in the requested format. """ result = openml._backend.estimation_procedure.list() - return [i.name for i in result] + records = [procedure._to_dict() for procedure in result] + + if output_format == "dataframe": + return pd.DataFrame.from_records(records) + + return {record.pop("id"): record for record in records} def list_evaluations_setups( diff --git a/tests/test_evaluations/test_evaluation_functions.py b/tests/test_evaluations/test_evaluation_functions.py index e15556d7bf..dd4cea2757 100644 --- a/tests/test_evaluations/test_evaluation_functions.py +++ b/tests/test_evaluations/test_evaluation_functions.py @@ -1,10 +1,14 @@ # License: BSD 3-Clause from __future__ import annotations +from unittest.mock import patch + import pytest import openml import openml.evaluations +from openml.estimation_procedures import OpenMLEstimationProcedure +from openml.tasks import TaskType from openml.testing import TestBase @@ -239,6 +243,48 @@ def test_list_evaluation_measures(self): assert isinstance(measures, list) is True assert all(isinstance(s, str) for s in measures) is True + def test_list_estimation_procedures_dict(self): + procedures = [ + OpenMLEstimationProcedure( + id=5, + task_type_id=TaskType.SUPERVISED_CLASSIFICATION, + name="10-fold Crossvalidation", + type="crossvalidation", + ) + ] + with patch.object(openml._backend.estimation_procedure, "list", return_value=procedures): + result = openml.evaluations.list_estimation_procedures() + + assert result == { + 5: { + "task_type_id": TaskType.SUPERVISED_CLASSIFICATION, + "name": "10-fold Crossvalidation", + "type": "crossvalidation", + } + } + + def test_list_estimation_procedures_dataframe(self): + procedures = [ + OpenMLEstimationProcedure( + id=5, + task_type_id=TaskType.SUPERVISED_CLASSIFICATION, + name="10-fold Crossvalidation", + type="crossvalidation", + ) + ] + with patch.object(openml._backend.estimation_procedure, "list", return_value=procedures): + result = openml.evaluations.list_estimation_procedures(output_format="dataframe") + + assert list(result.columns) == ["id", "task_type_id", "name", "type"] + assert result.to_dict("records") == [ + { + "id": 5, + "task_type_id": TaskType.SUPERVISED_CLASSIFICATION, + "name": "10-fold Crossvalidation", + "type": "crossvalidation", + } + ] + @pytest.mark.production_server() def test_list_evaluations_setups_filter_flow(self): self.use_production_server() From 2e36857b4176d2769ca3d93d17c0b59458949278 Mon Sep 17 00:00:00 2001 From: Jayant-kernel Date: Thu, 1 Oct 2026 20:08:42 +0530 Subject: [PATCH 2/2] [ENH] Preserve estimation procedure list compatibility --- openml/evaluations/functions.py | 49 +++++++++++++------ .../test_evaluation_functions.py | 25 ++++++---- 2 files changed, 51 insertions(+), 23 deletions(-) diff --git a/openml/evaluations/functions.py b/openml/evaluations/functions.py index e9bfa3030f..7516cfa278 100644 --- a/openml/evaluations/functions.py +++ b/openml/evaluations/functions.py @@ -2,6 +2,7 @@ # ruff: noqa: PLR0913 from __future__ import annotations +import warnings from functools import partial from itertools import chain from typing import TYPE_CHECKING, Literal @@ -15,6 +16,7 @@ import openml.utils if TYPE_CHECKING: + from openml.estimation_procedures import OpenMLEstimationProcedure from openml.evaluations import OpenMLEvaluation @@ -164,38 +166,57 @@ def list_estimation_procedures( @overload def list_estimation_procedures( - output_format: Literal["dict"] = ..., -) -> dict[int, dict[str, object]]: ... + output_format: Literal["object"], +) -> dict[int, OpenMLEstimationProcedure]: ... + + +@overload +def list_estimation_procedures( + output_format: None = None, +) -> list[str]: ... def list_estimation_procedures( - output_format: Literal["dict", "dataframe"] = "dict", -) -> dict[int, dict[str, object]] | pd.DataFrame: + output_format: Literal["object", "dataframe"] | None = None, +) -> list[str] | dict[int, OpenMLEstimationProcedure] | pd.DataFrame: """Return the estimation procedures available on OpenML. The function performs an API call to retrieve the entire list of - evaluation procedures. Each procedure includes its ID, task type, - name, and type. + evaluation procedures. Parameters ---------- - output_format : {"dict", "dataframe"}, default="dict" - The format of the returned procedures. The dictionary format maps - procedure IDs to their remaining metadata. The DataFrame format has - one row per procedure, including an ``id`` column. + output_format : {"object", "dataframe"}, optional + The format of the returned procedures. ``"object"`` returns a + dictionary mapping procedure IDs to ``OpenMLEstimationProcedure`` + instances. ``"dataframe"`` returns one row per procedure. + If omitted, a list of procedure names is returned for backwards + compatibility and a warning is emitted. Returns ------- - dict or pandas.DataFrame + list[str], dict[int, OpenMLEstimationProcedure], or pandas.DataFrame The available estimation procedures in the requested format. """ result = openml._backend.estimation_procedure.list() - records = [procedure._to_dict() for procedure in result] + + if output_format is None: + warnings.warn( + "The default output will change from a list of names to a dictionary of " + "OpenMLEstimationProcedure objects in a future release. Set " + "output_format='object' to use the new format.", + FutureWarning, + stacklevel=2, + ) + return [procedure.name for procedure in result] if output_format == "dataframe": - return pd.DataFrame.from_records(records) + return pd.DataFrame.from_records(procedure._to_dict() for procedure in result) + + if output_format == "object": + return {procedure.id: procedure for procedure in result} - return {record.pop("id"): record for record in records} + raise ValueError("Invalid output format. Only 'object' and 'dataframe' are applicable.") def list_evaluations_setups( diff --git a/tests/test_evaluations/test_evaluation_functions.py b/tests/test_evaluations/test_evaluation_functions.py index dd4cea2757..a616ea799b 100644 --- a/tests/test_evaluations/test_evaluation_functions.py +++ b/tests/test_evaluations/test_evaluation_functions.py @@ -243,7 +243,7 @@ def test_list_evaluation_measures(self): assert isinstance(measures, list) is True assert all(isinstance(s, str) for s in measures) is True - def test_list_estimation_procedures_dict(self): + def test_list_estimation_procedures_default_warns_about_future_change(self): procedures = [ OpenMLEstimationProcedure( id=5, @@ -253,15 +253,22 @@ def test_list_estimation_procedures_dict(self): ) ] with patch.object(openml._backend.estimation_procedure, "list", return_value=procedures): - result = openml.evaluations.list_estimation_procedures() + with pytest.warns(FutureWarning, match="output will change"): + result = openml.evaluations.list_estimation_procedures() - assert result == { - 5: { - "task_type_id": TaskType.SUPERVISED_CLASSIFICATION, - "name": "10-fold Crossvalidation", - "type": "crossvalidation", - } - } + assert result == ["10-fold Crossvalidation"] + + def test_list_estimation_procedures_object(self): + procedure = OpenMLEstimationProcedure( + id=5, + task_type_id=TaskType.SUPERVISED_CLASSIFICATION, + name="10-fold Crossvalidation", + type="crossvalidation", + ) + with patch.object(openml._backend.estimation_procedure, "list", return_value=[procedure]): + result = openml.evaluations.list_estimation_procedures(output_format="object") + + assert result == {5: procedure} def test_list_estimation_procedures_dataframe(self): procedures = [