From 31f36f61e08395e149ef621b6be4cd4e34145bba Mon Sep 17 00:00:00 2001 From: Alejandro Moreo Date: Tue, 6 Oct 2026 12:29:38 +0200 Subject: [PATCH] add DFMrff as a named method, promote quadprog to a standard dependency - DFMrff exposes qunfold.KMM(kernel='rff') as a public, importable quapy.method.non_aggregative class, following Dussap et al. (2023); the import of qunfold stays lazy (inside __init__/fit) so the module remains usable without it, matching the EDx/EDy pattern. Registered in NON_AGGREGATIVE_METHODS and as optional in test_methods.py. - quadprog (required by EDx/EDy) is no longer gated behind a lazy _get_quadprog() helper; it's now a direct install_requires entry, which also bumps the minimum supported Python to 3.9. - Added a manual section for DFMrff, cross-referencing the Composable Methods section for the other (non-default) KMM kernels. --- CHANGE_LOG.txt | 8 ++++ README.md | 4 +- docs/source/manuals/methods.md | 44 +++++++++++++++++++-- quapy/__init__.py | 2 +- quapy/method/__init__.py | 3 +- quapy/method/_energy.py | 3 +- quapy/method/_helper.py | 10 ----- quapy/method/aggregative.py | 5 ++- quapy/method/non_aggregative.py | 68 ++++++++++++++++++++++++++++++++- quapy/tests/test_methods.py | 22 +++++------ setup.py | 5 +-- 11 files changed, 137 insertions(+), 37 deletions(-) diff --git a/CHANGE_LOG.txt b/CHANGE_LOG.txt index 370a868..7baa781 100644 --- a/CHANGE_LOG.txt +++ b/CHANGE_LOG.txt @@ -1,3 +1,11 @@ +Change Log 0.2.4 +----------------- + +- Making DFM-RFF method explicit, and improved documentation. + +- Promoted quadprog (required by EDx/EDy) from an optional to a standard dependency; the minimum + supported Python version is now 3.9. + Change Log 0.2.3 ----------------- diff --git a/README.md b/README.md index 0c20e90..4fe100c 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # QuaPy -## version 0.2.3 +## version 0.2.4 QuaPy is an open source framework for quantification (a.k.a. supervised prevalence estimation, or learning to quantify) written in Python. @@ -15,7 +15,7 @@ for facilitating the analysis and interpretation of the experimental results. ### Last updates: -* Version 0.2.3 is released! major changes can be consulted [here](CHANGE_LOG.txt). +* Version 0.2.4 is released! major changes can be consulted [here](CHANGE_LOG.txt). * The developer API documentation is available [here](https://hlt-isti.github.io/QuaPy/index.html) ### Installation diff --git a/docs/source/manuals/methods.md b/docs/source/manuals/methods.md index dcab373..331254c 100644 --- a/docs/source/manuals/methods.md +++ b/docs/source/manuals/methods.md @@ -476,8 +476,7 @@ problem by quadratic programming. The method is proposed in In QuaPy, `EDy` works for binary and multiclass problems and lets the user choose the pairwise distance through the `distance` parameter (`'manhattan'`, `'euclidean'`, or a custom callable). -Because the optimization relies on `quadprog`, this method requires the -optional dependency `pip install quadprog`. +The optimization relies on `quadprog`, a standard QuaPy dependency. #### SMM @@ -670,7 +669,46 @@ a classifier. In this sense, `EDx` is to `EDy` what `DMx` is to `DMy`. `EDx` works for binary and multiclass problems, accepts the same `distance` options as `EDy` (`'manhattan'`, `'euclidean'`, or a custom callable), and -requires the optional dependency `pip install quadprog`. +relies on `quadprog`, a standard QuaPy dependency. + +### Distribution Feature Matching (DFMrff) + +QuaPy exposes `qp.method.non_aggregative.DFMrff`, a covariate-space +distribution-matching quantifier proposed in: + +[_Dussap, B., Blanchard, G., & Chérief-Abdellatif, B. E. (2023). Label shift +quantification with robustness guarantees via distribution feature matching. +In Joint European Conference on Machine Learning and Knowledge Discovery in +Databases (pp. 69-85). Springer._](https://doi.org/10.1007/978-3-031-43412-9_5) + +The method matches the training and test distributions through a kernel +embedding, approximated via random Fourier features (hence "RFF") for +computational efficiency; the authors report this to be the best-performing +variant among the kernels they study (energy, Gaussian, Laplacian, and RFF), +which is why `DFMrff` is the only one of them exposed as a named, public +method. `DFMrff` accepts the following hyperparameters: `sigma` (the kernel +smoothing parameter), `n_rff` (the number of random Fourier features, +default 1000), `solver` and `solver_options` (passed to +`scipy.optimize.minimize`), and `seed`. + +```python +import quapy as qp +from quapy.method.non_aggregative import DFMrff + +dataset = qp.datasets.fetch_UCIMulticlassDataset('dry-bean') +train, test = dataset.train_test + +model = DFMrff(n_rff=1000, seed=0) +model.fit(*train.Xy) +estim_prevalence = model.predict(test.X) +``` + +Internally, `DFMrff` is a thin wrapper around `qunfold.KMM(kernel='rff')` +(see the Composable Methods section below), and therefore requires the +optional `qunfold` dependency. The other kernel choices for `KMM`, as well as +arbitrary re-combinations of losses and feature representations, remain +directly accessible through `quapy.method.composable.ComposableQuantifier` +and `quapy.method.composable.QUnfoldWrapper`. ### ReadMe diff --git a/quapy/__init__.py b/quapy/__init__.py index de4575c..8067a61 100644 --- a/quapy/__init__.py +++ b/quapy/__init__.py @@ -17,7 +17,7 @@ try: except ImportError: plot = None -__version__ = '0.2.3' +__version__ = '0.2.4' def _default_cls(): diff --git a/quapy/method/__init__.py b/quapy/method/__init__.py index d295382..250076b 100644 --- a/quapy/method/__init__.py +++ b/quapy/method/__init__.py @@ -75,7 +75,8 @@ MULTICLASS_METHODS = { NON_AGGREGATIVE_METHODS = { non_aggregative.MaximumLikelihoodPrevalenceEstimation, non_aggregative.DMx, - non_aggregative.EDx + non_aggregative.EDx, + non_aggregative.DFMrff, } META_METHODS = { diff --git a/quapy/method/_energy.py b/quapy/method/_energy.py index e0ca8ed..cfcd758 100644 --- a/quapy/method/_energy.py +++ b/quapy/method/_energy.py @@ -1,11 +1,11 @@ from typing import Callable, Union import numpy as np +import quadprog from sklearn.metrics.pairwise import euclidean_distances, manhattan_distances import quapy as qp import quapy.functional as F -from quapy.method._helper import _get_quadprog class _EnergyDistanceCore: @@ -114,7 +114,6 @@ class _EnergyDistanceCore: def _solve_ed(self, G, a, C, b): """Solve the energy-distance quadratic program.""" - quadprog = _get_quadprog() sol = quadprog.solve_qp(G=G, a=a, C=C, b=b) prevalences = sol[0] prevalences = np.append(prevalences, 1 - prevalences.sum()) diff --git a/quapy/method/_helper.py b/quapy/method/_helper.py index 48d0cb6..0e5792a 100644 --- a/quapy/method/_helper.py +++ b/quapy/method/_helper.py @@ -32,16 +32,6 @@ def _get_cvxpy(): return cp -def _get_quadprog(): - try: - import quadprog - except ImportError as exc: - raise ImportError( - "EDy requires the optional 'quadprog' package." - ) from exc - return quadprog - - def _labels_to_indices(labels, classes): encoder = LabelEncoder().fit(classes) return encoder.transform(labels) diff --git a/quapy/method/aggregative.py b/quapy/method/aggregative.py index 6dcd025..440d82e 100644 --- a/quapy/method/aggregative.py +++ b/quapy/method/aggregative.py @@ -2114,8 +2114,9 @@ class EDy(_EnergyDistanceCore, AggregativeSoftQuantifier): operates directly on posterior vectors rather than on histogram summaries. This implementation works for binary and multiclass single-label - quantification and relies on the optional ``quadprog`` dependency. It was - adapted to QuaPy's current aggregative API from the original implementation + quantification and relies on the ``quadprog`` package for solving the + underlying quadratic program. It was adapted to QuaPy's current aggregative + API from the original implementation available in `quantificationlib `_, and now shares its numerical core with the classifier-free :class:`quapy.method.non_aggregative.EDx` variant. diff --git a/quapy/method/non_aggregative.py b/quapy/method/non_aggregative.py index 0250852..8a6312d 100644 --- a/quapy/method/non_aggregative.py +++ b/quapy/method/non_aggregative.py @@ -174,8 +174,9 @@ class EDx(_EnergyDistanceCore, BaseQuantifier): energy-distance quadratic program directly in feature space. This implementation works for binary and multiclass single-label - quantification and relies on the optional ``quadprog`` dependency. The - current QuaPy adaptation shares its numerical core with EDy and keeps + quantification and relies on the ``quadprog`` package for solving the + underlying quadratic program. The current QuaPy adaptation shares its + numerical core with EDy and keeps credit to the original implementation available in `quantificationlib `_. @@ -226,6 +227,69 @@ class EDx(_EnergyDistanceCore, BaseQuantifier): return self._predict_energy(X) +class DFMrff(BaseQuantifier): + """ + Distribution Feature Matching with Random Fourier Features (DFM-RFF), a covariate-space + distribution-matching quantifier proposed by: + + `Dussap, B., Blanchard, G., & Chérief-Abdellatif, B. E. (2023). Label shift quantification + with robustness guarantees via distribution feature matching. In Joint European Conference + on Machine Learning and Knowledge Discovery in Databases (pp. 69-85). Springer. + `_ + + The method matches training and test distributions in feature space through a kernel + embedding, approximated via random Fourier features for computational efficiency; the authors + report this to be the best-performing variant among the kernels they study, which is why it is + the one exposed here as a named, public method. Other kernel choices (energy, Gaussian, + Laplacian), as well as arbitrary re-combinations of losses and feature representations, remain + accessible through the more general :class:`quapy.method.composable.ComposableQuantifier`; + this class is a thin convenience wrapper that pins the kernel of + :class:`quapy.method.composable.QUnfoldWrapper`-wrapped ``qunfold.KMM`` to ``'rff'``. + + This implementation delegates to the optional `qunfold `_ + package (the same backend used by :mod:`quapy.method.composable`); see the "Composable Methods" + manual for installation instructions. + + :param sigma: smoothing parameter of the random Fourier feature kernel approximation (default 1) + :param n_rff: number of random Fourier features (default 1000) + :param solver: the `method` argument passed to `scipy.optimize.minimize` (default 'trust-ncg') + :param solver_options: dict of options passed to `scipy.optimize.minimize`; if None (default), + `{'gtol': 1e-8, 'maxiter': 1000}` is used + :param seed: seed controlling the random Fourier features and the solver (default None) + """ + + def __init__(self, sigma=1, n_rff=1000, solver='trust-ncg', solver_options=None, seed=None): + # imported here (rather than at the top of this module) so that quapy.method.non_aggregative + # remains importable without qunfold installed; this import raises a clear, actionable + # ImportError (with installation instructions) if qunfold is missing + from quapy.method.composable import QUnfoldWrapper # noqa: F401 + self.sigma = sigma + self.n_rff = n_rff + self.solver = solver + self.solver_options = solver_options + self.seed = seed + + def _build_method(self): + import qunfold + from quapy.method.composable import QUnfoldWrapper + solver_options = self.solver_options if self.solver_options is not None else {'gtol': 1e-8, 'maxiter': 1000} + return QUnfoldWrapper(qunfold.KMM( + kernel='rff', sigma=self.sigma, n_rff=self.n_rff, + solver=self.solver, solver_options=solver_options, seed=self.seed, + )) + + def fit(self, X, y): + self._method = self._build_method() + self._method.fit(X, y) + return self + + def predict(self, X): + return self._method.predict(X) + + def __str__(self): + return f'{self.__class__.__name__}(sigma={self.sigma}, n_rff={self.n_rff})' + + class ReadMe(BaseQuantifier, WithConfidenceABC): """ ReadMe is a non-aggregative quantification system proposed by diff --git a/quapy/tests/test_methods.py b/quapy/tests/test_methods.py index 0e41ff5..73054a1 100644 --- a/quapy/tests/test_methods.py +++ b/quapy/tests/test_methods.py @@ -7,7 +7,7 @@ import numpy as np from sklearn.linear_model import LogisticRegression from quapy.method import AGGREGATIVE_METHODS, BINARY_METHODS, NON_AGGREGATIVE_METHODS -from quapy.method.non_aggregative import DMx, EDx, HDx +from quapy.method.non_aggregative import DFMrff, DMx, EDx, HDx from quapy.method.aggregative import ACC, BBSEhard, BBSEsoft, DMy, EDy, KDEyCS, LEIP, RLLS from quapy.method.meta import Ensemble from quapy.functional import check_prevalence_vector @@ -20,12 +20,11 @@ OPTIONAL_AGGREGATIVE_METHODS = { 'BayesianMAPLS', 'PQ', 'RLLS', - 'EDy', 'LEIP', } OPTIONAL_NON_AGGREGATIVE_METHODS = { - 'EDx', + 'DFMrff', } @@ -235,26 +234,27 @@ class TestMethods(unittest.TestCase): self.assertTrue(check_prevalence_vector(estim_prevalences)) def test_edy(self): - try: - import quadprog # noqa: F401 - except ImportError: - return - dataset = TestMethods.tiny_dataset_multiclass q = EDy(LogisticRegression(max_iter=2000), val_split=3) q.fit(*dataset.training.Xy) estim_prevalences = q.predict(dataset.test.X) self.assertTrue(check_prevalence_vector(estim_prevalences)) - def test_edx(self): + dataset = TestMethods.tiny_dataset_multiclass + q = EDx() + q.fit(*dataset.training.Xy) + estim_prevalences = q.predict(dataset.test.X) + self.assertTrue(check_prevalence_vector(estim_prevalences)) + + def test_dfmrff(self): try: - import quadprog # noqa: F401 + import qunfold # noqa: F401 except ImportError: return dataset = TestMethods.tiny_dataset_multiclass - q = EDx() + q = DFMrff(n_rff=50) q.fit(*dataset.training.Xy) estim_prevalences = q.predict(dataset.test.X) self.assertTrue(check_prevalence_vector(estim_prevalences)) diff --git a/setup.py b/setup.py index 3ff18da..1eb6199 100644 --- a/setup.py +++ b/setup.py @@ -89,7 +89,6 @@ setup( 'License :: OSI Approved :: BSD License', 'Programming Language :: Python :: 3', - 'Programming Language :: Python :: 3.8', 'Programming Language :: Python :: 3.9', 'Programming Language :: Python :: 3 :: Only', ], @@ -117,9 +116,9 @@ setup( 'quapy.method': ['stan/*.stan'] }, - python_requires='>=3.8, <4', + python_requires='>=3.9, <4', - install_requires=['scikit-learn', 'pandas', 'tqdm', 'matplotlib', 'joblib', 'xlrd', 'abstention', 'ucimlrepo', 'certifi'], + install_requires=['scikit-learn', 'pandas', 'tqdm', 'matplotlib', 'joblib', 'xlrd', 'abstention', 'ucimlrepo', 'certifi', 'quadprog'], # List additional groups of dependencies here (e.g. development # dependencies). Users will be able to install these using the "extras"