Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .pre-commit-config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,7 @@ repos:
hooks:
- id: pydocstyle
args: ["--config=setup.cfg"]
exclude: "^skpro/libs/"

# We use the Python version instead of the original version which seems to require Docker
# https://github.com/koalaman/shellcheck-precommit
Expand Down
1 change: 1 addition & 0 deletions setup.cfg
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@ ignore = E121, E123, E126, E226, E24, E704, W503, W504
max-line-length = 88
exclude =
skpro/_contrib/*
skpro/libs/*
extend-ignore =
# See https://github.com/PyCQA/pycodestyle/issues/373
E203
Expand Down
1 change: 1 addition & 0 deletions skpro/libs/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
"""Libraries bundled with skpro for compatibility."""
124 changes: 124 additions & 0 deletions skpro/libs/cyclic_boosting/GBSregression.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,124 @@
"""
Cyclic Boosting Regression for Generalized Background Subtraction regression.
"""


import logging
from typing import Tuple, Union

import numpy as np
import pandas as pd
from sklearn.base import RegressorMixin

from skpro.libs.cyclic_boosting.base import (
CBLinkPredictionsFactors,
CyclicBoostingBase,
Feature,
)
from skpro.libs.cyclic_boosting.link import IdentityLinkMixin

_logger = logging.getLogger(__name__)


class CBGBSRegressor(RegressorMixin, CyclicBoostingBase, IdentityLinkMixin):
r"""
Variant form of Cyclic Boosting's location regressor, that corresponds to
the regression of the outcome of a previous statistical subtraction of two
classes of observations from each other (e.g. groups A and B: A - B).

For this, the target y has to be set to positive values for group A and
negative values for group B.

Additional Parameter
--------------------
regalpha: float
A hyperparameter to steer the strength of regularization, i.e. a
shrinkage of the regression result for A _B to 0. A value of 0
corresponds to no regularization.
"""

def __init__(
self,
feature_groups=None,
hierarchical_feature_groups=None,
feature_properties=None,
weight_column=None,
minimal_loss_change=1e-10,
minimal_factor_change=1e-10,
maximal_iterations=10,
observers=None,
smoother_choice=None,
output_column=None,
learn_rate=None,
regalpha=0.0,
aggregate=True,
):
CyclicBoostingBase.__init__(
self,
feature_groups=feature_groups,
hierarchical_feature_groups=hierarchical_feature_groups,
feature_properties=feature_properties,
weight_column=weight_column,
minimal_loss_change=minimal_loss_change,
minimal_factor_change=minimal_factor_change,
maximal_iterations=maximal_iterations,
observers=observers,
smoother_choice=smoother_choice,
output_column=output_column,
learn_rate=learn_rate,
aggregate=aggregate,
)

self.regalpha = regalpha

def calc_parameters(
self,
feature: Feature,
y: np.ndarray,
pred: CBLinkPredictionsFactors,
prefit_data,
) -> Tuple[np.ndarray, np.ndarray]:
lex_binnumbers = feature.lex_binned_data
minlength = feature.n_bins
prediction = pred.predict_link()

n = (y - prediction) * self.weights
d = self.weights * (1 + self.regalpha)

sum_n, sum_d, sum_nd, sum_n2, sum_d2 = (
np.bincount(lex_binnumbers, weights=w, minlength=minlength)
for w in [n, d, n * d, n * n, d * d]
)

sum_d += 1
sum_d2 += 1**2

summand = sum_n / sum_d
variance_summand = (
sum_d**2 * sum_n2 - 2.0 * sum_n * sum_d * sum_nd + sum_n**2 * sum_d2
) / sum_d**4

return summand, np.sqrt(variance_summand)

def _check_y(self, y: np.ndarray):
pass

def _init_global_scale(
self, X: Union[pd.DataFrame, np.ndarray], y: np.ndarray
) -> None:
if self.weights is None:
raise RuntimeError("The weights have to be initialized.")
self.global_scale_link_ = (y * self.weights).sum() / self.weights.sum()

def loss(self, prediction: np.ndarray, y: np.ndarray, weights: np.ndarray) -> float:
wvisitsum = ((y != 0).astype(int) * weights).sum()
loss = (weights * (prediction - y) ** 2).sum() / wvisitsum
return loss

def precalc_parameters(
self, feature: Feature, y: np.ndarray, pred: CBLinkPredictionsFactors
) -> None:
return None


__all__ = ["CBGBSRegressor"]
103 changes: 103 additions & 0 deletions skpro/libs/cyclic_boosting/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,103 @@
"""This package contains the Cyclic Boosting family of machine learning
algorithms.

If you are looking for conceptional explanations of the Cyclic Boosting
algorithm, you might have a look at the two papers
https://arxiv.org/abs/2002.03425 and https://arxiv.org/abs/2009.07052.

API reference of the different Cyclic Boosting methods:

Multiplicative Regression

- :class:`~.CBPoissonRegressor`
- :class:`~.CBNBinomRegressor`
- :class:`~.CBExponential`
- :class:`~.CBMultiplicativeQuantileRegressor`
- :class:`~.CBMultiplicativeGenericCRegressor`

Additive Regression

- :class:`~.CBLocationRegressor`
- :class:`~.CBLocPoissonRegressor`
- :class:`~.CBAdditiveQuantileRegressor`
- :class:`~.CBAdditiveGenericCRegressor`

PDF Prediction

- :class:`~.CBNBinomC`

Classification

- :class:`~.CBClassifier`
- :class:`~.CBGenericClassifier`

Background Subtraction

- :class:`~.CBGBSRegressor`
"""


from skpro.libs.cyclic_boosting.base import CyclicBoostingBase
from skpro.libs.cyclic_boosting.classification import CBClassifier
from skpro.libs.cyclic_boosting.GBSregression import CBGBSRegressor
from skpro.libs.cyclic_boosting.generic_loss import (
CBAdditiveGenericRegressor,
CBAdditiveQuantileRegressor,
CBGenericClassifier,
CBMultiplicativeGenericRegressor,
CBMultiplicativeQuantileRegressor,
)
from skpro.libs.cyclic_boosting.location import (
CBLocationRegressor,
CBLocPoissonRegressor,
)
from skpro.libs.cyclic_boosting.nbinom import CBNBinomC
from skpro.libs.cyclic_boosting.pipelines import (
pipeline_CBAdditiveGenericRegressor,
pipeline_CBAdditiveQuantileRegressor,
pipeline_CBClassifier,
pipeline_CBExponential,
pipeline_CBGBSRegressor,
pipeline_CBGenericClassifier,
pipeline_CBLocationRegressor,
pipeline_CBLocPoissonRegressor,
pipeline_CBMultiplicativeGenericRegressor,
pipeline_CBMultiplicativeQuantileRegressor,
pipeline_CBNBinomC,
pipeline_CBNBinomRegressor,
pipeline_CBPoissonRegressor,
)
from skpro.libs.cyclic_boosting.price import CBExponential
from skpro.libs.cyclic_boosting.regression import CBNBinomRegressor, CBPoissonRegressor

__all__ = [
"CyclicBoostingBase",
"CBPoissonRegressor",
"CBNBinomRegressor",
"CBExponential",
"CBLocationRegressor",
"CBLocPoissonRegressor",
"CBNBinomC",
"CBClassifier",
"CBGBSRegressor",
"CBMultiplicativeQuantileRegressor",
"CBAdditiveQuantileRegressor",
"CBMultiplicativeGenericRegressor",
"CBAdditiveGenericRegressor",
"CBGenericClassifier",
"pipeline_CBPoissonRegressor",
"pipeline_CBNBinomRegressor",
"pipeline_CBClassifier",
"pipeline_CBLocationRegressor",
"pipeline_CBExponential",
"pipeline_CBLocPoissonRegressor",
"pipeline_CBNBinomC",
"pipeline_CBGBSRegressor",
"pipeline_CBMultiplicativeQuantileRegressor",
"pipeline_CBAdditiveQuantileRegressor",
"pipeline_CBMultiplicativeGenericRegressor",
"pipeline_CBAdditiveGenericRegressor",
"pipeline_CBGenericClassifier",
]

__version__ = "1.4.0"
Loading