Skip to content

Preprocessing

Binarizers

hgp_lib.preprocessing.base.Binarizer

Bases: ABC

Abstract base class for binarizers.

A binarizer converts a mixed-type pandas.DataFrame into a purely boolean DataFrame, where every column is a boolean feature that a rule can test. Concrete implementations must preserve this contract so they are interchangeable, for example inside GPBenchmarker, which fits a fresh copy per fold.

Contract
  • fit_transform(X, y=None) learns the encoding from X (and optional labels y) and returns a boolean DataFrame.
  • transform(X) applies the learned encoding to new data and returns a boolean DataFrame with the same columns, in the same order, as the output of fit_transform.
  • get_feature_names_out() returns the output column names, in order, so a literal's feature index maps to a readable name via feature_names[index].
  • is_fitted reports whether the binarizer has been fitted.

Examples:

>>> import numpy as np
>>> import pandas as pd
>>> from hgp_lib.preprocessing.base import Binarizer
>>> class PassThrough(Binarizer):
...     def fit_transform(self, X, y=None):
...         self._columns = list(X.columns)
...         self._is_fitted = True
...         return X.astype(bool)
...     def transform(self, X):
...         return X.astype(bool)
...     def get_feature_names_out(self):
...         return list(self._columns)
>>> b = PassThrough()
>>> b.is_fitted
False
>>> out = b.fit_transform(pd.DataFrame({"x": [True, False]}))
>>> b.is_fitted
True
>>> b.get_feature_names_out()
['x']
Source code in hgp_lib\preprocessing\base.py
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
class Binarizer(ABC):
    """
    Abstract base class for binarizers.

    A binarizer converts a mixed-type ``pandas.DataFrame`` into a purely boolean
    ``DataFrame``, where every column is a boolean feature that a rule can test.
    Concrete implementations must preserve this contract so they are interchangeable,
    for example inside ``GPBenchmarker``, which fits a fresh copy per fold.

    Contract:
        - ``fit_transform(X, y=None)`` learns the encoding from ``X`` (and optional
          labels ``y``) and returns a boolean ``DataFrame``.
        - ``transform(X)`` applies the learned encoding to new data and returns a
          boolean ``DataFrame`` with the same columns, in the same order, as the
          output of ``fit_transform``.
        - ``get_feature_names_out()`` returns the output column names, in order, so a
          literal's feature index maps to a readable name via ``feature_names[index]``.
        - ``is_fitted`` reports whether the binarizer has been fitted.

    Examples:
        >>> import numpy as np
        >>> import pandas as pd
        >>> from hgp_lib.preprocessing.base import Binarizer
        >>> class PassThrough(Binarizer):
        ...     def fit_transform(self, X, y=None):
        ...         self._columns = list(X.columns)
        ...         self._is_fitted = True
        ...         return X.astype(bool)
        ...     def transform(self, X):
        ...         return X.astype(bool)
        ...     def get_feature_names_out(self):
        ...         return list(self._columns)
        >>> b = PassThrough()
        >>> b.is_fitted
        False
        >>> out = b.fit_transform(pd.DataFrame({"x": [True, False]}))
        >>> b.is_fitted
        True
        >>> b.get_feature_names_out()
        ['x']
    """

    _is_fitted: bool = False

    @property
    def is_fitted(self) -> bool:
        """Whether the binarizer has been fitted."""
        return self._is_fitted

    @abstractmethod
    def fit_transform(
        self, X: pd.DataFrame, y: Optional[np.ndarray] = None
    ) -> pd.DataFrame:
        """
        Learn the encoding from ``X`` (and optional labels ``y``) and return the
        transformed boolean ``DataFrame``.
        """
        pass

    @abstractmethod
    def transform(self, X: pd.DataFrame) -> pd.DataFrame:
        """Apply the learned encoding to new data and return a boolean ``DataFrame``."""
        pass

    @abstractmethod
    def get_feature_names_out(self) -> List[str]:
        """
        Return the output feature (column) names in order.

        The returned list is index-aligned with the boolean columns produced by
        ``fit_transform`` / ``transform``, so ``feature_names[i]`` is the name of the
        feature a rule references with ``Literal(value=i)``. Following the scikit-learn
        convention, the names are returned as an ordered ``list[str]``. Implementations
        should raise if called before the binarizer is fitted.
        """
        pass

is_fitted property

Whether the binarizer has been fitted.

fit_transform(X, y=None) abstractmethod

Learn the encoding from X (and optional labels y) and return the transformed boolean DataFrame.

Source code in hgp_lib\preprocessing\base.py
57
58
59
60
61
62
63
64
65
@abstractmethod
def fit_transform(
    self, X: pd.DataFrame, y: Optional[np.ndarray] = None
) -> pd.DataFrame:
    """
    Learn the encoding from ``X`` (and optional labels ``y``) and return the
    transformed boolean ``DataFrame``.
    """
    pass

transform(X) abstractmethod

Apply the learned encoding to new data and return a boolean DataFrame.

Source code in hgp_lib\preprocessing\base.py
67
68
69
70
@abstractmethod
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
    """Apply the learned encoding to new data and return a boolean ``DataFrame``."""
    pass

get_feature_names_out() abstractmethod

Return the output feature (column) names in order.

The returned list is index-aligned with the boolean columns produced by fit_transform / transform, so feature_names[i] is the name of the feature a rule references with Literal(value=i). Following the scikit-learn convention, the names are returned as an ordered list[str]. Implementations should raise if called before the binarizer is fitted.

Source code in hgp_lib\preprocessing\base.py
72
73
74
75
76
77
78
79
80
81
82
83
@abstractmethod
def get_feature_names_out(self) -> List[str]:
    """
    Return the output feature (column) names in order.

    The returned list is index-aligned with the boolean columns produced by
    ``fit_transform`` / ``transform``, so ``feature_names[i]`` is the name of the
    feature a rule references with ``Literal(value=i)``. Following the scikit-learn
    convention, the names are returned as an ordered ``list[str]``. Implementations
    should raise if called before the binarizer is fitted.
    """
    pass

hgp_lib.preprocessing.binarizer.StandardBinarizer

Bases: Binarizer

Converts a mixed-type DataFrame into a purely boolean DataFrame.

Boolean columns are passed through unchanged. Categorical columns are one-hot encoded into one boolean column per unique value. Numeric columns are discretised into bins and then one-hot encoded, using a :class:BinningStrategy.

Column handling:

  • Boolean columns are kept as is.
  • Categorical, string, and object columns are one-hot encoded. String and object columns trigger a :class:StringColumnWarning, since setting a category dtype is clearer. A column whose values are all distinct is dropped with a :class:HighCardinalityWarning, because one-hot encoding it carries no generalization.
  • Numeric columns are split into bins by a :class:BinningStrategy. When y is provided and no strategy is set, :class:SupervisedTreeBinning is used, otherwise :class:QuantileBinning.
  • A column that contains missing values also gets a boolean <col>_is_NA indicator column.

To change how numeric bins are chosen, pass a numeric_binning strategy or subclass and override _fit_numeric / _transform_numeric. Categorical and boolean handling can be changed the same way through their _fit_* / _transform_* hooks.

Parameters:

Name Type Description Default
num_bins int

Default number of bins for numeric columns. Must be >= 2. Default: 5.

5
column_strategy dict[str, int] | None

Per-column override for the number of bins. Keys are column names, values are the desired bin count (each >= 2). Default: None.

None
precision int

Number of decimal places used when formatting numeric bin boundary names. Must be >= 0. Default: 3.

3
numeric_binning BinningStrategy | None

Strategy for computing numeric bin edges. When None, the binarizer uses :class:SupervisedTreeBinning if labels are provided to fit_transform and :class:QuantileBinning otherwise. Default: None.

None
progress_bar bool

Whether to show progress bar. Default: True.

True
leave_progress_bar bool

Whether to leave progress bar. Default: False.

False

Examples:

>>> import pandas as pd
>>> from hgp_lib.preprocessing import StandardBinarizer
>>> df = pd.DataFrame({"flag": [True, False, True], "val": [1.0, 2.0, 3.0]})
>>> binarizer = StandardBinarizer(num_bins=2)
>>> result = binarizer.fit_transform(df)
>>> "flag" in result.columns
True
>>> result["flag"].tolist()
[True, False, True]
>>> result
    flag  val < 2.000  2.000 <= val
0   True         True         False
1  False         True         False
2   True        False          True
Source code in hgp_lib\preprocessing\binarizer.py
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
class StandardBinarizer(Binarizer):
    """
    Converts a mixed-type DataFrame into a purely boolean DataFrame.

    Boolean columns are passed through unchanged. Categorical columns are one-hot
    encoded into one boolean column per unique value. Numeric columns are discretised
    into bins and then one-hot encoded, using a :class:`BinningStrategy`.

    Column handling:

    - Boolean columns are kept as is.
    - Categorical, string, and object columns are one-hot encoded. String and object
      columns trigger a :class:`StringColumnWarning`, since setting a ``category``
      dtype is clearer. A column whose values are all distinct is dropped with a
      :class:`HighCardinalityWarning`, because one-hot encoding it carries no
      generalization.
    - Numeric columns are split into bins by a :class:`BinningStrategy`. When ``y`` is
      provided and no strategy is set, :class:`SupervisedTreeBinning` is used, otherwise
      :class:`QuantileBinning`.
    - A column that contains missing values also gets a boolean ``<col>_is_NA``
      indicator column.

    To change how numeric bins are chosen, pass a ``numeric_binning`` strategy or
    subclass and override ``_fit_numeric`` / ``_transform_numeric``. Categorical and
    boolean handling can be changed the same way through their ``_fit_*`` / ``_transform_*``
    hooks.

    Args:
        num_bins (int):
            Default number of bins for numeric columns. Must be >= 2. Default: `5`.
        column_strategy (dict[str, int] | None):
            Per-column override for the number of bins. Keys are column names, values are
            the desired bin count (each >= 2). Default: `None`.
        precision (int):
            Number of decimal places used when formatting numeric bin boundary names.
            Must be >= 0. Default: `3`.
        numeric_binning (BinningStrategy | None):
            Strategy for computing numeric bin edges. When `None`, the binarizer uses
            :class:`SupervisedTreeBinning` if labels are provided to ``fit_transform``
            and :class:`QuantileBinning` otherwise. Default: `None`.
        progress_bar (bool):
            Whether to show progress bar. Default: `True`.
        leave_progress_bar (bool):
            Whether to leave progress bar. Default: `False`.


    Examples:
        >>> import pandas as pd
        >>> from hgp_lib.preprocessing import StandardBinarizer
        >>> df = pd.DataFrame({"flag": [True, False, True], "val": [1.0, 2.0, 3.0]})
        >>> binarizer = StandardBinarizer(num_bins=2)
        >>> result = binarizer.fit_transform(df)
        >>> "flag" in result.columns
        True
        >>> result["flag"].tolist()
        [True, False, True]
        >>> result
            flag  val < 2.000  2.000 <= val
        0   True         True         False
        1  False         True         False
        2   True        False          True
    """

    def __init__(
        self,
        num_bins: int = 5,
        column_strategy: Optional[dict[str, int]] = None,
        precision: int = 3,
        numeric_binning: Optional[BinningStrategy] = None,
        progress_bar: bool = True,
        leave_progress_bar: bool = False,
    ):
        self._validate_params(num_bins, column_strategy, precision, numeric_binning)
        self.num_bins = num_bins
        self.column_strategy = column_strategy or {}
        self.precision = precision
        self.numeric_binning = numeric_binning
        self.column_precision: dict[str, int] = {}
        self.progress_bar = progress_bar
        self.leave_progress_bar = leave_progress_bar

        self._categorical_values: dict = {}
        self._numerical_bins: dict = {}
        self._original_column_dtypes: dict = {}
        self._output_names: dict = {}
        self._na_columns: Set[str] = set()
        self._skipped_columns: Set[str] = set()
        self._original_columns = None
        self._is_fitted = False
        self._feature_names: List[str] = []

    def _validate_params(
        self,
        num_bins: int,
        column_strategy: Optional[dict[str, int]],
        precision: int,
        numeric_binning: Optional[BinningStrategy],
    ) -> None:
        check_isinstance(num_bins, int)
        if num_bins < 2:
            raise ValueError(f"num_bins must be an integer >= 2, is {num_bins}")

        if column_strategy is not None:
            check_isinstance(column_strategy, dict)
            for col, bins in column_strategy.items():
                check_isinstance(bins, int)
                if bins < 2:
                    raise ValueError(
                        f"Number of bins for column {col} must be an integer >= 2, is {bins}"
                    )

        check_isinstance(precision, int)
        if precision < 0:
            raise ValueError(f"precision must be an integer >= 0, is {precision}")

        if numeric_binning is not None:
            check_isinstance(numeric_binning, BinningStrategy)

    def fit_transform(
        self, X: pd.DataFrame, y: Optional[np.ndarray] = None
    ) -> pd.DataFrame:
        """
        Learn the binarisation mapping from ``X`` (and optionally ``y``) and return the
        transformed boolean DataFrame.

        Args:
            X (pd.DataFrame):
                Input DataFrame whose columns are boolean, categorical, string, object,
                or numeric.
            y (np.ndarray | None):
                Optional target labels used for supervised binning of numeric columns.
                Default: `None`.

        Returns:
            pd.DataFrame: A DataFrame with only boolean columns.

        Raises:
            TypeError: If ``X`` is not a DataFrame.
            ValueError: If a column has an unsupported dtype.

        Examples:
            >>> import pandas as pd
            >>> from hgp_lib.preprocessing import StandardBinarizer
            >>> df = pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0]})
            >>> result = StandardBinarizer(num_bins=2).fit_transform(df)
            >>> result.shape
            (4, 2)
            >>> all(result.dtypes == bool)
            True
        """
        check_isinstance(X, pd.DataFrame)
        self._reset_state()

        outputs: dict = {}
        used_names: Set[str] = set()

        for column in tqdm(
            X.columns,
            disable=not self.progress_bar,
            desc="Fitting binarizer",
            leave=self.leave_progress_bar,
        ):
            series = X[column]
            nan_mask = series.isna().to_numpy()

            pieces = []
            if nan_mask.any():
                self._na_columns.add(column)
                pieces.append((f"{column}_is_NA", nan_mask))

            pieces.extend(self._fit_column(column, series, y, nan_mask))

            names: List[str] = []
            for base_name, values in pieces:
                name = self._ensure_unique_column_names(used_names, base_name)
                outputs[name] = values
                names.append(name)
            self._output_names[column] = names

        if len(outputs) == 0:
            warn_once(EmptyBinarizationWarning())
            outputs["default"] = np.ones(len(X), dtype=bool)

        self._original_columns = X.columns
        self._is_fitted = True
        self._feature_names = [str(name) for name in outputs.keys()]
        return pd.DataFrame(outputs, index=X.index)

    def get_feature_names_out(self) -> List[str]:
        """
        Return the output column names in order (see :meth:`Binarizer.get_feature_names_out`).

        Returns:
            List[str]: The boolean output column names, index-aligned with the
                columns produced by ``fit_transform`` / ``transform``.

        Raises:
            ValueError: If the binarizer has not been fitted yet.

        Examples:
            >>> import pandas as pd
            >>> from hgp_lib.preprocessing import StandardBinarizer
            >>> b = StandardBinarizer(num_bins=2)
            >>> _ = b.fit_transform(pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0]}))
            >>> b.get_feature_names_out()
            ['x < 2.500', '2.500 <= x']
        """
        if not self._is_fitted:
            raise ValueError(
                "Binarizer must be fitted before calling get_feature_names_out"
            )
        return list(self._feature_names)

    def transform(self, X: pd.DataFrame) -> pd.DataFrame:
        """
        Apply the previously learned binarisation to new data.

        The input must have the same columns, in the same order and with the same
        dtypes, as the data used during fitting.

        Args:
            X (pd.DataFrame):
                Input DataFrame with the same schema as the fitting data.

        Returns:
            pd.DataFrame: A boolean DataFrame with the same column layout as the fitted output.

        Raises:
            TypeError: If ``X`` is not a DataFrame.
            ValueError: If the binarizer has not been fitted yet, or if a column dtype
                differs from the one seen during fitting.
            RuntimeError: If the columns differ from the fitting data.

        Examples:
            >>> import pandas as pd
            >>> from hgp_lib.preprocessing import StandardBinarizer
            >>> b = StandardBinarizer(num_bins=2)
            >>> _ = b.fit_transform(pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0]}))
            >>> b.transform(pd.DataFrame({"x": [1.5, 3.5]})).shape
            (2, 2)
        """
        check_isinstance(X, pd.DataFrame)
        if not self._is_fitted:
            raise ValueError("Binarizer must be fitted before calling transform")
        if not self._original_columns.equals(X.columns):
            # TODO: We should add custom Errors in the library where it makes sense.;
            raise RuntimeError(
                f"Original columns do not match current columns. "
                f"Original columns: {self._original_columns}. Current columns: {X.columns}."
            )

        outputs: dict = {}
        for column in tqdm(
            X.columns,
            disable=not self.progress_bar,
            desc="Transforming binarizer",
            leave=self.leave_progress_bar,
        ):
            series = X[column]
            names = self._output_names[column]
            values_list: List[np.ndarray] = []

            nan_mask = series.isna().to_numpy()
            if column in self._na_columns:
                values_list.append(nan_mask)
            elif nan_mask.any():
                warn_once(UnseenNaNWarning(column))

            if column not in self._skipped_columns:
                values_list.extend(self._transform_column(column, series))

            if len(values_list) != len(names):
                raise RuntimeError(
                    f"Column '{column}' produced {len(values_list)} features at transform "
                    f"but {len(names)} were produced at fit."
                )
            for name, values in zip(names, values_list):
                outputs[name] = values

        if len(outputs) == 0:
            warn_once(EmptyBinarizationWarning())
            outputs["default"] = np.ones(len(X), dtype=bool)

        return pd.DataFrame(outputs, index=X.index)

    def _fit_column(
        self,
        column: str,
        series: pd.Series,
        y: Optional[np.ndarray],
        nan_mask: np.ndarray,
    ):
        """Dispatch a single column to the matching dtype hook and record its dtype."""
        # TODO: Instead of string values, we should have an enum. And an enum-like dispatch.
        if is_bool_dtype(series):
            self._original_column_dtypes[column] = "bool"
            return self._fit_boolean(column, series)
        if is_categorical_like(series):
            self._original_column_dtypes[column] = "category"
            return self._fit_categorical(column, series)
        if is_numeric_dtype(series):
            self._original_column_dtypes[column] = "numeric"
            return self._fit_numeric(column, series, y, nan_mask)
        raise ValueError(
            f"Unsupported column type for column {column} of type {series.dtype}"
        )

    def _fit_boolean(self, column: str, series: pd.Series):
        """Pass a boolean column through as a single feature."""
        return [(column, series.to_numpy(dtype=bool))]

    def _fit_categorical(self, column: str, series: pd.Series):
        """One-hot encode a categorical, string, or object column."""
        if not isinstance(series.dtype, pd.CategoricalDtype):
            warn_once(StringColumnWarning(column))

        not_na = series.dropna()
        unique_values = not_na.unique()
        if len(unique_values) == len(not_na):
            warn_once(HighCardinalityWarning(column))
            self._skipped_columns.add(column)
            return []

        self._categorical_values[column] = unique_values
        return [
            (f"{column}={value}", (series == value).to_numpy())
            for value in unique_values
        ]

    def _fit_numeric(
        self,
        column: str,
        series: pd.Series,
        y: Optional[np.ndarray],
        nan_mask: np.ndarray,
    ):
        """Bin a numeric column and one-hot encode the bins."""
        n_bins = self.column_strategy.get(column, self.num_bins)
        values = series.to_numpy()

        fit_values = values
        fit_y = y
        if nan_mask.any():
            keep = ~nan_mask
            fit_values = values[keep]
            if fit_y is not None:
                fit_y = fit_y[keep]

        strategy = self._resolve_numeric_binning(y)
        edges = strategy.compute_edges(fit_values, fit_y, n_bins)
        self._numerical_bins[column] = edges

        binned = pd.cut(values, bins=edges, labels=False, include_lowest=True)
        return [
            (
                self._format_numeric_bin_name(column, edges[i], edges[i + 1]),
                binned == i,
            )
            for i in range(len(edges) - 1)
        ]

    def _transform_column(self, column: str, series: pd.Series) -> List[np.ndarray]:
        """Apply the learned encoding for a single column, verifying its dtype."""
        expected = self._original_column_dtypes[column]
        actual = self._infer_kind(column, series)
        if actual != expected:
            raise ValueError(
                f"Original column {column} was {expected}. "
                f"Current column is {actual}. Current column must be {expected}."
            )
        if expected == "bool":
            return self._transform_boolean(series)
        if expected == "category":
            return self._transform_categorical(column, series)
        return self._transform_numeric(column, series)

    def _transform_boolean(self, series: pd.Series) -> List[np.ndarray]:
        return [series.to_numpy(dtype=bool)]

    def _transform_categorical(
        self, column: str, series: pd.Series
    ) -> List[np.ndarray]:
        return [
            (series == value).to_numpy() for value in self._categorical_values[column]
        ]

    def _transform_numeric(self, column: str, series: pd.Series) -> List[np.ndarray]:
        edges = self._numerical_bins[column]
        binned = pd.cut(
            series.to_numpy(), bins=edges, labels=False, include_lowest=True
        )
        return [binned == i for i in range(len(edges) - 1)]

    def _resolve_numeric_binning(self, y: Optional[np.ndarray]) -> BinningStrategy:
        """Pick the numeric binning strategy: the configured one, or a default by ``y``."""
        if self.numeric_binning is not None:
            return self.numeric_binning
        return SupervisedTreeBinning() if y is not None else QuantileBinning()

    def _infer_kind(self, column: str, series: pd.Series) -> str:
        if is_bool_dtype(series):
            return "bool"
        if is_categorical_like(series):
            return "category"
        if is_numeric_dtype(series):
            return "numeric"
        raise ValueError(
            f"Unsupported column type for column {column} of type {series.dtype}"
        )

    def _reset_state(self) -> None:
        self._categorical_values = {}
        self._numerical_bins = {}
        self._original_column_dtypes = {}
        self._output_names = {}
        self._na_columns = set()
        self._skipped_columns = set()
        self._original_columns = None
        self._is_fitted = False

    def _ensure_unique_column_names(
        self, column_names: Set[str], new_column_name: str
    ) -> str:
        """
        Register ``new_column_name`` in ``column_names``, appending a numeric suffix if needed.
        The set is mutated in place.

        Args:
            column_names (Set[str]):
                Mutable set of names already in use.
            new_column_name (str):
                Desired column name.

        Returns:
            str: The original name if it was unique, otherwise a suffixed variant.

        Examples:
            >>> from hgp_lib.preprocessing import StandardBinarizer
            >>> b = StandardBinarizer()
            >>> names = set(["col", "col_0"])
            >>> b._ensure_unique_column_names(names, "col")
            'col_1'
            >>> "col_1" in names
            True
        """
        unique_name = new_column_name
        counter = 0
        while unique_name in column_names:
            unique_name = f"{new_column_name}_{counter}"
            counter += 1
        column_names.add(unique_name)
        return unique_name

    def _format_numeric_bin_name(self, column: str, left: float, right: float) -> str:
        """
        Build a human-readable label for a numeric bin.

        The format depends on whether the left or right boundary is infinite:

        - Left is ``-inf``: ``"column < right"``
        - Right is ``inf``: ``"left <= column"``
        - Both finite: ``"left <= column < right"``

        Args:
            column (str):
                Name of the original numeric column.
            left (float):
                Left boundary of the bin.
            right (float):
                Right boundary of the bin.

        Returns:
            str: Formatted bin label.

        Examples:
            >>> import numpy as np
            >>> from hgp_lib.preprocessing import StandardBinarizer
            >>> b = StandardBinarizer(precision=2)
            >>> b._format_numeric_bin_name("x", -np.inf, 3.0)
            'x < 3.00'
            >>> b._format_numeric_bin_name("x", 1.0, np.inf)
            '1.00 <= x'
            >>> b._format_numeric_bin_name("x", 1.0, 3.0)
            '1.00 <= x < 3.00'
        """
        precision = self.column_precision.get(column, self.precision)
        if np.isneginf(left):
            return f"{column} < {right:.{precision}f}"
        if np.isposinf(right):
            return f"{left:.{precision}f} <= {column}"
        return f"{left:.{precision}f} <= {column} < {right:.{precision}f}"

fit_transform(X, y=None)

Learn the binarisation mapping from X (and optionally y) and return the transformed boolean DataFrame.

Parameters:

Name Type Description Default
X DataFrame

Input DataFrame whose columns are boolean, categorical, string, object, or numeric.

required
y ndarray | None

Optional target labels used for supervised binning of numeric columns. Default: None.

None

Returns:

Type Description
DataFrame

pd.DataFrame: A DataFrame with only boolean columns.

Raises:

Type Description
TypeError

If X is not a DataFrame.

ValueError

If a column has an unsupported dtype.

Examples:

>>> import pandas as pd
>>> from hgp_lib.preprocessing import StandardBinarizer
>>> df = pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0]})
>>> result = StandardBinarizer(num_bins=2).fit_transform(df)
>>> result.shape
(4, 2)
>>> all(result.dtypes == bool)
True
Source code in hgp_lib\preprocessing\binarizer.py
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
def fit_transform(
    self, X: pd.DataFrame, y: Optional[np.ndarray] = None
) -> pd.DataFrame:
    """
    Learn the binarisation mapping from ``X`` (and optionally ``y``) and return the
    transformed boolean DataFrame.

    Args:
        X (pd.DataFrame):
            Input DataFrame whose columns are boolean, categorical, string, object,
            or numeric.
        y (np.ndarray | None):
            Optional target labels used for supervised binning of numeric columns.
            Default: `None`.

    Returns:
        pd.DataFrame: A DataFrame with only boolean columns.

    Raises:
        TypeError: If ``X`` is not a DataFrame.
        ValueError: If a column has an unsupported dtype.

    Examples:
        >>> import pandas as pd
        >>> from hgp_lib.preprocessing import StandardBinarizer
        >>> df = pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0]})
        >>> result = StandardBinarizer(num_bins=2).fit_transform(df)
        >>> result.shape
        (4, 2)
        >>> all(result.dtypes == bool)
        True
    """
    check_isinstance(X, pd.DataFrame)
    self._reset_state()

    outputs: dict = {}
    used_names: Set[str] = set()

    for column in tqdm(
        X.columns,
        disable=not self.progress_bar,
        desc="Fitting binarizer",
        leave=self.leave_progress_bar,
    ):
        series = X[column]
        nan_mask = series.isna().to_numpy()

        pieces = []
        if nan_mask.any():
            self._na_columns.add(column)
            pieces.append((f"{column}_is_NA", nan_mask))

        pieces.extend(self._fit_column(column, series, y, nan_mask))

        names: List[str] = []
        for base_name, values in pieces:
            name = self._ensure_unique_column_names(used_names, base_name)
            outputs[name] = values
            names.append(name)
        self._output_names[column] = names

    if len(outputs) == 0:
        warn_once(EmptyBinarizationWarning())
        outputs["default"] = np.ones(len(X), dtype=bool)

    self._original_columns = X.columns
    self._is_fitted = True
    self._feature_names = [str(name) for name in outputs.keys()]
    return pd.DataFrame(outputs, index=X.index)

get_feature_names_out()

Return the output column names in order (see :meth:Binarizer.get_feature_names_out).

Returns:

Type Description
List[str]

List[str]: The boolean output column names, index-aligned with the columns produced by fit_transform / transform.

Raises:

Type Description
ValueError

If the binarizer has not been fitted yet.

Examples:

>>> import pandas as pd
>>> from hgp_lib.preprocessing import StandardBinarizer
>>> b = StandardBinarizer(num_bins=2)
>>> _ = b.fit_transform(pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0]}))
>>> b.get_feature_names_out()
['x < 2.500', '2.500 <= x']
Source code in hgp_lib\preprocessing\binarizer.py
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
def get_feature_names_out(self) -> List[str]:
    """
    Return the output column names in order (see :meth:`Binarizer.get_feature_names_out`).

    Returns:
        List[str]: The boolean output column names, index-aligned with the
            columns produced by ``fit_transform`` / ``transform``.

    Raises:
        ValueError: If the binarizer has not been fitted yet.

    Examples:
        >>> import pandas as pd
        >>> from hgp_lib.preprocessing import StandardBinarizer
        >>> b = StandardBinarizer(num_bins=2)
        >>> _ = b.fit_transform(pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0]}))
        >>> b.get_feature_names_out()
        ['x < 2.500', '2.500 <= x']
    """
    if not self._is_fitted:
        raise ValueError(
            "Binarizer must be fitted before calling get_feature_names_out"
        )
    return list(self._feature_names)

transform(X)

Apply the previously learned binarisation to new data.

The input must have the same columns, in the same order and with the same dtypes, as the data used during fitting.

Parameters:

Name Type Description Default
X DataFrame

Input DataFrame with the same schema as the fitting data.

required

Returns:

Type Description
DataFrame

pd.DataFrame: A boolean DataFrame with the same column layout as the fitted output.

Raises:

Type Description
TypeError

If X is not a DataFrame.

ValueError

If the binarizer has not been fitted yet, or if a column dtype differs from the one seen during fitting.

RuntimeError

If the columns differ from the fitting data.

Examples:

>>> import pandas as pd
>>> from hgp_lib.preprocessing import StandardBinarizer
>>> b = StandardBinarizer(num_bins=2)
>>> _ = b.fit_transform(pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0]}))
>>> b.transform(pd.DataFrame({"x": [1.5, 3.5]})).shape
(2, 2)
Source code in hgp_lib\preprocessing\binarizer.py
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
    """
    Apply the previously learned binarisation to new data.

    The input must have the same columns, in the same order and with the same
    dtypes, as the data used during fitting.

    Args:
        X (pd.DataFrame):
            Input DataFrame with the same schema as the fitting data.

    Returns:
        pd.DataFrame: A boolean DataFrame with the same column layout as the fitted output.

    Raises:
        TypeError: If ``X`` is not a DataFrame.
        ValueError: If the binarizer has not been fitted yet, or if a column dtype
            differs from the one seen during fitting.
        RuntimeError: If the columns differ from the fitting data.

    Examples:
        >>> import pandas as pd
        >>> from hgp_lib.preprocessing import StandardBinarizer
        >>> b = StandardBinarizer(num_bins=2)
        >>> _ = b.fit_transform(pd.DataFrame({"x": [1.0, 2.0, 3.0, 4.0]}))
        >>> b.transform(pd.DataFrame({"x": [1.5, 3.5]})).shape
        (2, 2)
    """
    check_isinstance(X, pd.DataFrame)
    if not self._is_fitted:
        raise ValueError("Binarizer must be fitted before calling transform")
    if not self._original_columns.equals(X.columns):
        # TODO: We should add custom Errors in the library where it makes sense.;
        raise RuntimeError(
            f"Original columns do not match current columns. "
            f"Original columns: {self._original_columns}. Current columns: {X.columns}."
        )

    outputs: dict = {}
    for column in tqdm(
        X.columns,
        disable=not self.progress_bar,
        desc="Transforming binarizer",
        leave=self.leave_progress_bar,
    ):
        series = X[column]
        names = self._output_names[column]
        values_list: List[np.ndarray] = []

        nan_mask = series.isna().to_numpy()
        if column in self._na_columns:
            values_list.append(nan_mask)
        elif nan_mask.any():
            warn_once(UnseenNaNWarning(column))

        if column not in self._skipped_columns:
            values_list.extend(self._transform_column(column, series))

        if len(values_list) != len(names):
            raise RuntimeError(
                f"Column '{column}' produced {len(values_list)} features at transform "
                f"but {len(names)} were produced at fit."
            )
        for name, values in zip(names, values_list):
            outputs[name] = values

    if len(outputs) == 0:
        warn_once(EmptyBinarizationWarning())
        outputs["default"] = np.ones(len(X), dtype=bool)

    return pd.DataFrame(outputs, index=X.index)

hgp_lib.preprocessing.sklearn_binarizer.SklearnBinarizer

Bases: Binarizer

Adapter that lets a scikit-learn transformer be used as a :class:Binarizer.

It wraps a transformer that outputs a dense one-hot or otherwise binary array, for example KBinsDiscretizer(encode="onehot-dense"), and returns a boolean DataFrame with readable column names. This makes scikit-learn discretizers interchangeable with :class:StandardBinarizer, including inside GPBenchmarker.

The wrapped transformer must implement fit_transform(X, y) and transform(X). Column names are taken from get_feature_names_out when available, otherwise they are generated positionally.

Parameters:

Name Type Description Default
transformer

An unfitted scikit-learn transformer producing a binary array.

required

Examples:

>>> import pandas as pd
>>> from sklearn.preprocessing import KBinsDiscretizer
>>> from hgp_lib.preprocessing import SklearnBinarizer
>>> disc = KBinsDiscretizer(n_bins=2, encode="onehot-dense", strategy="uniform")
>>> b = SklearnBinarizer(disc)
>>> out = b.fit_transform(pd.DataFrame({"x": [0.0, 1.0, 2.0, 3.0]}))
>>> bool(out.to_numpy().dtype == bool)
True
Source code in hgp_lib\preprocessing\sklearn_binarizer.py
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
class SklearnBinarizer(Binarizer):
    """
    Adapter that lets a scikit-learn transformer be used as a :class:`Binarizer`.

    It wraps a transformer that outputs a dense one-hot or otherwise binary array, for
    example ``KBinsDiscretizer(encode="onehot-dense")``, and returns a boolean
    ``DataFrame`` with readable column names. This makes scikit-learn discretizers
    interchangeable with :class:`StandardBinarizer`, including inside ``GPBenchmarker``.

    The wrapped transformer must implement ``fit_transform(X, y)`` and ``transform(X)``.
    Column names are taken from ``get_feature_names_out`` when available, otherwise they
    are generated positionally.

    Args:
        transformer:
            An unfitted scikit-learn transformer producing a binary array.

    Examples:
        >>> import pandas as pd
        >>> from sklearn.preprocessing import KBinsDiscretizer
        >>> from hgp_lib.preprocessing import SklearnBinarizer
        >>> disc = KBinsDiscretizer(n_bins=2, encode="onehot-dense", strategy="uniform")
        >>> b = SklearnBinarizer(disc)
        >>> out = b.fit_transform(pd.DataFrame({"x": [0.0, 1.0, 2.0, 3.0]}))
        >>> bool(out.to_numpy().dtype == bool)
        True
    """

    def __init__(self, transformer):
        self.transformer = transformer
        self._columns: Optional[List[str]] = None
        self._is_fitted = False

    def fit_transform(
        self, X: pd.DataFrame, y: Optional[np.ndarray] = None
    ) -> pd.DataFrame:
        check_isinstance(X, pd.DataFrame)
        array = np.asarray(self.transformer.fit_transform(X, y))
        self._columns = self._resolve_columns(X, array.shape[1])
        self._is_fitted = True
        return pd.DataFrame(array.astype(bool), columns=self._columns, index=X.index)

    def transform(self, X: pd.DataFrame) -> pd.DataFrame:
        check_isinstance(X, pd.DataFrame)
        if not self._is_fitted:
            raise ValueError("Binarizer must be fitted before calling transform")
        array = np.asarray(self.transformer.transform(X))
        return pd.DataFrame(array.astype(bool), columns=self._columns, index=X.index)

    def get_feature_names_out(self) -> List[str]:
        """
        Return the output column names in order (see :meth:`Binarizer.get_feature_names_out`).

        Names come from the wrapped transformer's ``get_feature_names_out`` when
        available, otherwise they are generated positionally.

        Returns:
            List[str]: The boolean output column names.

        Raises:
            ValueError: If the binarizer has not been fitted yet.
        """
        if not self._is_fitted:
            raise ValueError(
                "Binarizer must be fitted before calling get_feature_names_out"
            )
        return list(self._columns)

    def _resolve_columns(self, X: pd.DataFrame, n_features: int) -> List[str]:
        if hasattr(self.transformer, "get_feature_names_out"):
            return [
                str(c) for c in self.transformer.get_feature_names_out(list(X.columns))
            ]
        return [f"feature_{i}" for i in range(n_features)]

get_feature_names_out()

Return the output column names in order (see :meth:Binarizer.get_feature_names_out).

Names come from the wrapped transformer's get_feature_names_out when available, otherwise they are generated positionally.

Returns:

Type Description
List[str]

List[str]: The boolean output column names.

Raises:

Type Description
ValueError

If the binarizer has not been fitted yet.

Source code in hgp_lib\preprocessing\sklearn_binarizer.py
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
def get_feature_names_out(self) -> List[str]:
    """
    Return the output column names in order (see :meth:`Binarizer.get_feature_names_out`).

    Names come from the wrapped transformer's ``get_feature_names_out`` when
    available, otherwise they are generated positionally.

    Returns:
        List[str]: The boolean output column names.

    Raises:
        ValueError: If the binarizer has not been fitted yet.
    """
    if not self._is_fitted:
        raise ValueError(
            "Binarizer must be fitted before calling get_feature_names_out"
        )
    return list(self._columns)

Binning Strategies

hgp_lib.preprocessing.binning.BinningStrategy

Bases: ABC

Strategy for computing bin edges of a single numeric feature.

A strategy turns a 1-D array of values (and optional labels) into a sorted array of bin edges. The edges always start with -inf and end with inf so that values outside the fitted range still fall into a bin. StandardBinarizer delegates numeric binning to a strategy, so a custom binning method only needs to subclass this class and implement compute_edges.

Examples:

>>> import numpy as np
>>> from hgp_lib.preprocessing.binning import QuantileBinning
>>> QuantileBinning().compute_edges(np.array([1.0, 2.0, 3.0, 4.0]), None, 2).tolist()
[-inf, 2.5, inf]
Source code in hgp_lib\preprocessing\binning.py
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
class BinningStrategy(ABC):
    """
    Strategy for computing bin edges of a single numeric feature.

    A strategy turns a 1-D array of values (and optional labels) into a sorted array
    of bin edges. The edges always start with ``-inf`` and end with ``inf`` so that
    values outside the fitted range still fall into a bin. ``StandardBinarizer``
    delegates numeric binning to a strategy, so a custom binning method only needs to
    subclass this class and implement ``compute_edges``.

    Examples:
        >>> import numpy as np
        >>> from hgp_lib.preprocessing.binning import QuantileBinning
        >>> QuantileBinning().compute_edges(np.array([1.0, 2.0, 3.0, 4.0]), None, 2).tolist()
        [-inf, 2.5, inf]
    """

    @abstractmethod
    def compute_edges(
        self, values: np.ndarray, y: Optional[np.ndarray], n_bins: int
    ) -> np.ndarray:
        """
        Compute sorted bin edges for a single numeric feature.

        Args:
            values (np.ndarray):
                1-D array of feature values, with missing values already removed.
            y (np.ndarray | None):
                Optional 1-D label array aligned with ``values``, for supervised strategies.
            n_bins (int):
                Desired maximum number of bins.

        Returns:
            np.ndarray: Sorted bin edges beginning with ``-inf`` and ending with ``inf``.
        """
        raise NotImplementedError()

compute_edges(values, y, n_bins) abstractmethod

Compute sorted bin edges for a single numeric feature.

Parameters:

Name Type Description Default
values ndarray

1-D array of feature values, with missing values already removed.

required
y ndarray | None

Optional 1-D label array aligned with values, for supervised strategies.

required
n_bins int

Desired maximum number of bins.

required

Returns:

Type Description
ndarray

np.ndarray: Sorted bin edges beginning with -inf and ending with inf.

Source code in hgp_lib\preprocessing\binning.py
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
@abstractmethod
def compute_edges(
    self, values: np.ndarray, y: Optional[np.ndarray], n_bins: int
) -> np.ndarray:
    """
    Compute sorted bin edges for a single numeric feature.

    Args:
        values (np.ndarray):
            1-D array of feature values, with missing values already removed.
        y (np.ndarray | None):
            Optional 1-D label array aligned with ``values``, for supervised strategies.
        n_bins (int):
            Desired maximum number of bins.

    Returns:
        np.ndarray: Sorted bin edges beginning with ``-inf`` and ending with ``inf``.
    """
    raise NotImplementedError()

hgp_lib.preprocessing.binning.QuantileBinning

Bases: BinningStrategy

Unsupervised binning that places edges at evenly spaced quantiles.

Each bin holds a similar number of samples. Duplicate edges are removed, so the actual number of bins may be fewer than n_bins when many values are identical. A feature with one or fewer unique values yields a single [-inf, inf] bin.

Examples:

>>> import numpy as np
>>> from hgp_lib.preprocessing.binning import QuantileBinning
>>> QuantileBinning().compute_edges(np.array([5.0, 5.0, 5.0]), None, 3).tolist()
[-inf, inf]
Source code in hgp_lib\preprocessing\binning.py
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
class QuantileBinning(BinningStrategy):
    """
    Unsupervised binning that places edges at evenly spaced quantiles.

    Each bin holds a similar number of samples. Duplicate edges are removed, so the
    actual number of bins may be fewer than ``n_bins`` when many values are identical.
    A feature with one or fewer unique values yields a single ``[-inf, inf]`` bin.

    Examples:
        >>> import numpy as np
        >>> from hgp_lib.preprocessing.binning import QuantileBinning
        >>> QuantileBinning().compute_edges(np.array([5.0, 5.0, 5.0]), None, 3).tolist()
        [-inf, inf]
    """

    def compute_edges(
        self, values: np.ndarray, y: Optional[np.ndarray], n_bins: int
    ) -> np.ndarray:
        if len(np.unique(values)) <= 1:
            return np.array([-np.inf, np.inf])

        quantiles = np.linspace(0, 100, n_bins + 1)
        edges = np.percentile(values, quantiles)
        edges[0] = -np.inf
        edges[-1] = np.inf
        return np.unique(edges)

hgp_lib.preprocessing.binning.SupervisedTreeBinning

Bases: BinningStrategy

Supervised binning that uses a decision tree to place class-aware edges.

A shallow decision tree is fit to predict y from the single feature, and its split thresholds become the bin edges. The edges maximize class separation, which typically produces more informative features than unsupervised binning. A feature with one or fewer unique values yields a single [-inf, inf] bin.

Examples:

>>> import numpy as np
>>> from hgp_lib.preprocessing.binning import SupervisedTreeBinning
>>> SupervisedTreeBinning().compute_edges(
...     np.array([1.0, 2.0, 3.0, 4.0]), np.array([0, 0, 1, 1]), 2
... ).tolist()
[-inf, 2.5, inf]
Source code in hgp_lib\preprocessing\binning.py
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
class SupervisedTreeBinning(BinningStrategy):
    """
    Supervised binning that uses a decision tree to place class-aware edges.

    A shallow decision tree is fit to predict ``y`` from the single feature, and its
    split thresholds become the bin edges. The edges maximize class separation, which
    typically produces more informative features than unsupervised binning. A feature
    with one or fewer unique values yields a single ``[-inf, inf]`` bin.

    Examples:
        >>> import numpy as np
        >>> from hgp_lib.preprocessing.binning import SupervisedTreeBinning
        >>> SupervisedTreeBinning().compute_edges(
        ...     np.array([1.0, 2.0, 3.0, 4.0]), np.array([0, 0, 1, 1]), 2
        ... ).tolist()
        [-inf, 2.5, inf]
    """

    def compute_edges(
        self, values: np.ndarray, y: Optional[np.ndarray], n_bins: int
    ) -> np.ndarray:
        if y is None:
            raise ValueError("SupervisedTreeBinning requires labels y")
        if len(np.unique(values)) <= 1:
            return np.array([-np.inf, np.inf])

        tree = DecisionTreeClassifier(max_leaf_nodes=n_bins)
        tree.fit(values.reshape(-1, 1), y, check_input=False)
        thresholds = np.sort(tree.tree_.threshold[tree.tree_.feature != TREE_UNDEFINED])

        return np.concatenate([[-np.inf], thresholds, [np.inf]])

Warnings

hgp_lib.preprocessing.warnings.BinarizerWarning

Bases: Warning

Base class for warnings raised by binarizers.

Source code in hgp_lib\preprocessing\warnings.py
1
2
class BinarizerWarning(Warning):
    """Base class for warnings raised by binarizers."""

hgp_lib.preprocessing.warnings.StringColumnWarning

Bases: BinarizerWarning

A string or object column is being treated as categorical.

Setting the column to a pandas category dtype makes the intent explicit and avoids this warning.

Examples:

>>> from hgp_lib.preprocessing.warnings import StringColumnWarning
>>> str(StringColumnWarning("city"))
"Column 'city' has a string or object dtype and is treated as categorical. Set it to a 'category' dtype to make this explicit."
Source code in hgp_lib\preprocessing\warnings.py
 5
 6
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
class StringColumnWarning(BinarizerWarning):
    """
    A string or object column is being treated as categorical.

    Setting the column to a pandas ``category`` dtype makes the intent explicit and
    avoids this warning.

    Examples:
        >>> from hgp_lib.preprocessing.warnings import StringColumnWarning
        >>> str(StringColumnWarning("city"))
        "Column 'city' has a string or object dtype and is treated as categorical. Set it to a 'category' dtype to make this explicit."
    """

    def __init__(self, column: str):
        super().__init__(
            f"Column '{column}' has a string or object dtype and is treated as "
            f"categorical. Set it to a 'category' dtype to make this explicit."
        )

hgp_lib.preprocessing.warnings.HighCardinalityWarning

Bases: BinarizerWarning

An all-distinct categorical column is skipped.

One-hot encoding a column whose values are all distinct would produce one boolean feature per row, which carries no generalization, so the column is dropped.

Examples:

>>> from hgp_lib.preprocessing.warnings import HighCardinalityWarning
>>> str(HighCardinalityWarning("user_id"))
"Column 'user_id' has all-distinct values and is skipped, since one-hot encoding it would produce one feature per row."
Source code in hgp_lib\preprocessing\warnings.py
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
class HighCardinalityWarning(BinarizerWarning):
    """
    An all-distinct categorical column is skipped.

    One-hot encoding a column whose values are all distinct would produce one boolean
    feature per row, which carries no generalization, so the column is dropped.

    Examples:
        >>> from hgp_lib.preprocessing.warnings import HighCardinalityWarning
        >>> str(HighCardinalityWarning("user_id"))
        "Column 'user_id' has all-distinct values and is skipped, since one-hot encoding it would produce one feature per row."
    """

    def __init__(self, column: str):
        super().__init__(
            f"Column '{column}' has all-distinct values and is skipped, since "
            f"one-hot encoding it would produce one feature per row."
        )

hgp_lib.preprocessing.warnings.UnseenNaNWarning

Bases: BinarizerWarning

A column contains NaN at transform time but did not at fit time.

No new column is created for these values, so the affected rows fall into no bin and are encoded as all-false for that column.

Examples:

>>> from hgp_lib.preprocessing.warnings import UnseenNaNWarning
>>> str(UnseenNaNWarning("amount"))
"Column 'amount' has NaN values that were not seen during fit; these rows are encoded as all-false."
Source code in hgp_lib\preprocessing\warnings.py
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
class UnseenNaNWarning(BinarizerWarning):
    """
    A column contains NaN at transform time but did not at fit time.

    No new column is created for these values, so the affected rows fall into no bin
    and are encoded as all-false for that column.

    Examples:
        >>> from hgp_lib.preprocessing.warnings import UnseenNaNWarning
        >>> str(UnseenNaNWarning("amount"))
        "Column 'amount' has NaN values that were not seen during fit; these rows are encoded as all-false."
    """

    def __init__(self, column: str):
        super().__init__(
            f"Column '{column}' has NaN values that were not seen during fit; "
            f"these rows are encoded as all-false."
        )

hgp_lib.preprocessing.warnings.EmptyBinarizationWarning

Bases: BinarizerWarning

Source code in hgp_lib\preprocessing\warnings.py
65
66
67
68
69
70
class EmptyBinarizationWarning(BinarizerWarning):
    def __init__(self):
        super().__init__(
            "Empty binarization. An 'all-true' default column is generated. "
            "Please check the input data and provide binarization-compatible columns."
        )

Utilities

hgp_lib.preprocessing.utils.is_categorical_like(column)

Return whether a column should be treated as categorical.

A column is categorical-like when it uses the pandas category dtype, or holds strings or Python objects. Numeric and boolean columns are not categorical-like.

Parameters:

Name Type Description Default
column Series

The column to inspect.

required

Returns:

Name Type Description
bool bool

True if the column is categorical, string, or object dtype.

Examples:

>>> import pandas as pd
>>> from hgp_lib.preprocessing.utils import is_categorical_like
>>> is_categorical_like(pd.Series(pd.Categorical(["a", "b"])))
True
>>> is_categorical_like(pd.Series(["a", "b"], dtype="string"))
True
>>> is_categorical_like(pd.Series([1.0, 2.0]))
False
>>> is_categorical_like(pd.Series([True, False]))
False
Source code in hgp_lib\preprocessing\utils.py
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
def is_categorical_like(column: pd.Series) -> bool:
    """
    Return whether a column should be treated as categorical.

    A column is categorical-like when it uses the pandas ``category`` dtype, or holds
    strings or Python objects. Numeric and boolean columns are not categorical-like.

    Args:
        column (pd.Series):
            The column to inspect.

    Returns:
        bool: ``True`` if the column is categorical, string, or object dtype.

    Examples:
        >>> import pandas as pd
        >>> from hgp_lib.preprocessing.utils import is_categorical_like
        >>> is_categorical_like(pd.Series(pd.Categorical(["a", "b"])))
        True
        >>> is_categorical_like(pd.Series(["a", "b"], dtype="string"))
        True
        >>> is_categorical_like(pd.Series([1.0, 2.0]))
        False
        >>> is_categorical_like(pd.Series([True, False]))
        False
    """
    return (
        isinstance(column.dtype, pd.CategoricalDtype)
        or is_string_dtype(column)
        or is_object_dtype(column)
    )

hgp_lib.preprocessing.utils.load_data(data_path)

Load features and labels from a CSV or HDF file.

The file must contain a column named "target" which is used as the label array. All other columns are returned as the feature DataFrame. Labels are cast to bool.

Parameters:

Name Type Description Default
data_path str

Path to a .csv or .hdf file.

required

Returns:

Type Description
DataFrame

Tuple[pd.DataFrame, ndarray]: (data, labels) where data is the

ndarray

feature DataFrame (without the target column) and labels is a 1-D

Tuple[DataFrame, ndarray]

boolean numpy array.

Raises:

Type Description
FileNotFoundError

If data_path does not exist.

ValueError

If the file extension is not .csv or .hdf.

RuntimeError

If no "target" column is found.

Examples:

>>> import tempfile, os
>>> import pandas as pd
>>> from hgp_lib.preprocessing.utils import load_data
>>> df = pd.DataFrame({"x": [1, 2, 3], "target": [1, 0, 1]})
>>> with tempfile.TemporaryDirectory() as d:
...     path = os.path.join(d, "tmp.csv")
...     df.to_csv(path, index=False)
...     data, labels = load_data(path)
>>> list(data.columns)
['x']
>>> labels.tolist()
[True, False, True]
Source code in hgp_lib\preprocessing\utils.py
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
def load_data(data_path: str) -> Tuple[pd.DataFrame, ndarray]:
    """
    Load features and labels from a CSV or HDF file.

    The file must contain a column named ``"target"`` which is used as the label
    array. All other columns are returned as the feature DataFrame. Labels are
    cast to ``bool``.

    Args:
        data_path (str):
            Path to a ``.csv`` or ``.hdf`` file.

    Returns:
        Tuple[pd.DataFrame, ndarray]: ``(data, labels)`` where ``data`` is the
        feature DataFrame (without the target column) and ``labels`` is a 1-D
        boolean numpy array.

    Raises:
        FileNotFoundError: If ``data_path`` does not exist.
        ValueError: If the file extension is not ``.csv`` or ``.hdf``.
        RuntimeError: If no ``"target"`` column is found.

    Examples:
        >>> import tempfile, os
        >>> import pandas as pd
        >>> from hgp_lib.preprocessing.utils import load_data
        >>> df = pd.DataFrame({"x": [1, 2, 3], "target": [1, 0, 1]})
        >>> with tempfile.TemporaryDirectory() as d:
        ...     path = os.path.join(d, "tmp.csv")
        ...     df.to_csv(path, index=False)
        ...     data, labels = load_data(path)
        >>> list(data.columns)
        ['x']
        >>> labels.tolist()
        [True, False, True]
    """
    path = Path(data_path)
    if not path.exists():
        raise FileNotFoundError(f"Data file not found: {data_path}")

    logging.getLogger(__name__).info(f"Loading data from {data_path}...")
    # TODO: Create a unified logging system

    if path.suffix == ".hdf":
        df: pd.DataFrame = pd.read_hdf(data_path)
    elif path.suffix == ".csv":
        df: pd.DataFrame = pd.read_csv(data_path)
    else:
        raise ValueError(f"Unsupported file extension: {path.suffix}")

    if "target" in df.columns:
        target_column = "target"
    else:
        raise RuntimeError(f"Unknown target column. Available: {df.columns.tolist()}")

    labels = df[target_column].to_numpy(dtype=bool, copy=True)
    data = df.drop([target_column], axis=1)

    del df
    return data, labels