diff --git a/.circleci/config.yml b/.circleci/config.yml index 127b33bef..0d31c7f44 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -11,7 +11,7 @@ orbs: defaults: &defaults docker: - - image: cimg/python:3.10.0 + - image: cimg/python:3.12.1 working_directory: ~/project prepare_tox: &prepare_tox @@ -42,58 +42,6 @@ jobs: # Test matrix # ------------------------ - test_feature_engine_py39: - docker: - - image: cimg/python:3.9.0 - working_directory: ~/project - steps: - - checkout: - path: ~/project - - *prepare_tox - - run: - name: Run tests (Python 3.9) - command: | - tox -e py39 - - test_feature_engine_py310: - docker: - - image: cimg/python:3.10.0 - working_directory: ~/project - steps: - - checkout: - path: ~/project - - *prepare_tox - - run: - name: Run tests (Python 3.10) - command: | - tox -e py310 - - test_feature_engine_py311_sklearn150: - docker: - - image: cimg/python:3.11.7 - working_directory: ~/project - steps: - - checkout: - path: ~/project - - *prepare_tox - - run: - name: Run tests (Python 3.11, scikit-learn 1.5) - command: | - tox -e py311-sklearn150 - - test_feature_engine_py311_sklearn160: - docker: - - image: cimg/python:3.11.7 - working_directory: ~/project - steps: - - checkout: - path: ~/project - - *prepare_tox - - run: - name: Run tests (Python 3.11, scikit-learn 1.6) - command: | - tox -e py311-sklearn160 - test_feature_engine_py311_sklearn170: docker: - image: cimg/python:3.11.7 @@ -166,7 +114,7 @@ jobs: test_style: docker: - - image: cimg/python:3.10.0 + - image: cimg/python:3.12.1 working_directory: ~/project steps: - checkout: @@ -179,7 +127,7 @@ jobs: test_docs: docker: - - image: cimg/python:3.10.0 + - image: cimg/python:3.12.1 working_directory: ~/project steps: - checkout: @@ -192,7 +140,7 @@ jobs: test_type: docker: - - image: cimg/python:3.10.0 + - image: cimg/python:3.12.1 working_directory: ~/project steps: - checkout: @@ -277,10 +225,6 @@ workflows: test-all: jobs: - - test_feature_engine_py39 - - test_feature_engine_py310 - - test_feature_engine_py311_sklearn150 - - test_feature_engine_py311_sklearn160 - test_feature_engine_py311_sklearn170 - test_feature_engine_py312_pandas230 - test_feature_engine_py312_pandas300 @@ -298,10 +242,6 @@ workflows: - package_and_upload_to_pypi: requires: - - test_feature_engine_py39 - - test_feature_engine_py310 - - test_feature_engine_py311_sklearn150 - - test_feature_engine_py311_sklearn160 - test_feature_engine_py311_sklearn170 - test_feature_engine_py312_pandas230 - test_feature_engine_py312_pandas300 diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 000000000..96c9a3ebc --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,205 @@ +# AGENTS.md + +Conventions for working in this repo. Optimize for readability and speed, +in that order of how you decide, but don't ship a slow default when a +fast one is free. + +## Inputs + +Feature-engine transformers take dataframes (pandas, polars, or any other +narwhals-supported backend) as input, not numpy arrays. Don't add +handling for array input. + +## Never import pandas in library code + +pandas is an optional dependency (see `pyproject.toml` — it lives under +`[project.optional-dependencies]`, not core `dependencies`), so `import +pandas` must never appear anywhere in `feature_engine/`, not at module level +and not locally/lazily inside a function either — importing the module +itself would break a polars-only install regardless of which class is used. + +Backend checks go through `narwhals.dependencies` (`nwd.is_pandas_dataframe`, +`nwd.is_pandas_series`, `nwd.is_pandas_index`, `nwd.is_into_series`, etc.). +Once a branch is confirmed pandas, call its methods/attributes directly on +the object already in hand (`.loc`, `.columns`, `.index`, `.select_dtypes`, +...) — no import needed for that, since Python only needs a module imported +to reference the module itself (`pd.something`), not to call methods on an +object that's already an instance of that module's class. + +## Transformer code + +- `check_X` and `check_X_y` return a narwhals dataframe. Bind it as + `nw_X = check_X(X)`, pass the native `X` to the variable, NaN and feature-name + helpers, compute on `nw_X`, and return `.to_native()`. +- When pandas keeps a native fast path, branch with + `if nwd.is_pandas_dataframe(X):`, comment it with + `# pandas is faster than narwhals.` and put the narwhals code in `else`. +- To use the target together with `X`, call `add_target_to_X(nw_X, y)` from + `feature_engine/encoding/_helper_functions.py` and read it with `TARGET_NAME`. + It works for series, list and array targets, and with pandas the column takes + the index of `X`. Don't write this pairing again in a transformer. +- In the narwhals path, prefer narwhals expressions over Python loops on grouped + results: aggregate with simple aggregations, then combine the columns in a + `select`. +- narwhals expressions need string column names, while pandas allows integers. + Take such columns with `get_column` and rename them before using expressions. +- Name temporary columns with double underscores (`__mean__`, `__count__`) so + they can't clash with the user's columns. +- Don't add `# type: ignore`. If mypy complains because a parameter typed + `Optional` is reassigned, store the value under a new name instead (for + example `y_pd`). + +## Init parameters + +- Validate parameters in `__init__` only. Don't check them again in `fit` or + `transform`, and don't test for errors raised by changing an attribute after + init. +- For parameters that take a set of strings, check the type before the + membership test, so lists, tuples, `None` and numbers raise the same error: + + ```python + if not isinstance(encoding_method, str) or encoding_method not in [ + "ordered", + "arbitrary", + ]: + ``` + +- Error messages follow the scikit-learn convention and end with + `f"Got {param} instead."`. + +## Booleans and control flow + +- Compare booleans explicitly: `if x is True:` / `if x is False:`, never + `if x:` / `if not x:`. +- Check container emptiness with `len(x) == 0`, never `if not x:`. +- `isinstance(...)` checks and `in`/`not in` membership tests are already + explicit — leave them as-is, this rule isn't about those. +- The explicit `is True`/`is False` comparison is for flow control + (`if`/`while` conditions) only — don't tack it onto a variable + assignment. +- Call boolean checks such as `nwd.is_pandas_dataframe(X)` directly in the + condition, `if nwd.is_pandas_dataframe(X) is True:`, instead of storing the + result in a variable (`is_pandas = ...`) and testing that later. + +## Comments + +One line, two at most, in source code and tests. Only explain a non-obvious +WHY (a hidden constraint, a subtle backend difference, a workaround) — what the +reader needs to know about the code — never describe WHAT the code does. + +## Don't anticipate errors + +Don't add error handling or validation for scenarios that can't happen. If +unsure whether something can happen, check it (grep, run a quick repro) or +ask — don't guess and defensively code around it. + +## Redundant lists/sets + +- Narwhals' `.columns` is already `list[str]` — don't wrap it in `list()`. +- pandas' `.columns` is an `Index`, not a list — `list()` is required there + (an `Index == list` comparison is elementwise, not a clean bool). + +## Keep tests passing when you change a function or class + +Whenever you change a function or class, run its corresponding tests. If +they fail, resolve it — don't leave it — by figuring out whether the test +needs updating (e.g. it exercised behavior that's no longer supported) or +the implementation has a real bug, and fixing whichever one is wrong. + +## Keep docs in sync with transformer changes + +When new functionality is introduced in a transformer, update its +corresponding `docs/user_guide//.rst` with a short +worked example showing the new functionality. When behaviour changes, check +that the outputs shown in the user guide examples are still correct. + +User guides and other user-facing documentation are written for users: +assume readers don't know the source code, and certainly not narwhals. Explain +what a feature does and when to use it, in plain terms, without implementation +details or references to how the code used to behave. + +## Verify before applying + +Benchmark before claiming a speedup, and diff old-vs-new output across +realistic and edge cases (empty/all-NaN, both backends, both dtype +branches) before trusting a rewrite — logic mistakes here are easy to make +and easy to miss without an actual comparison. Compare like with like: time +the same work (for example the whole `fit()`) before and after. + +Benchmark a range of data sizes, but base the decision mainly on the sizes +each backend is typically used with: 10k to 500k rows for pandas, and 500k +rows and more for polars. Smaller and larger sizes are worth measuring, but +they weigh less in the decision. + +## Tests + +Every transformer test file has the same structure, so they are easy to +maintain: + +```python +# init parameters +def test_error_if__not_allowed(...) # one test per error message +def test_init_param_assignment(...) # several valid value combinations + +# fit and transform +... +``` + +- Init error tests are parametrized with wrong values and wrong types. +- `test_init_param_assignment` checks every init parameter except `variables` + and `return_empty`, which are tested elsewhere. +- Fit and transform tests don't assert init parameters. +- Every `pytest.raises` and `pytest.warns` matches the full message with + `match=re.escape(msg)`, including `NotFittedError` and messages that come + from scikit-learn. Never use + `with pytest.raises() as record: ... assert str(record.value) == msg`. + Matching the full message catches tests that pass for the wrong reason. + +Backends and data: + +- Dataframe-agnostic means one test, both backends: request the `make_df` + fixture from `tests/conftest.py`, which runs the test with `pd.DataFrame` + and `pl.DataFrame`, and assert the same input produces the same output + values on both. Never write a separate pandas-only test and a + separate polars-only test for the same behavior — that duplicates + the test and hides the point of being dataframe-agnostic, which is + that the same input gives the same output regardless of backend. + Keep a test single-backend only when the behavior itself is + backend-specific (e.g. integer column names, which polars doesn't + support; pandas category or nullable extension dtypes), and check those + with `pd.testing.assert_frame_equal`. +- Use the helpers in `tests/backend_helpers.py`: `frame_to_dict`, `null_count` + and `make_series`. Don't add per-file helpers that do the same. +- Data used by several test files of a module lives in that module's + `conftest.py`, as fixtures that return plain dicts, with `None` for missing + values. Data used by one file stays in that file. +- Pass the target as a series built with `make_series`, and add one test with + the target as a list and as a numpy array. +- Check outputs with `assert isinstance(Xt, make_df)` and compare + `frame_to_dict(Xt)` with a dict. Compare floats with `pytest.approx`. +- Don't call polars' `to_pandas()` in tests: pyarrow is not installed locally + or in CI. +- Name helpers after what they return (`frame_to_dict`, not `_cols`). + +## API changes + +- New parameters default to preserve current behavior. +- When adding a parameter to a function called from multiple sites (or a + shared private helper), thread it through every call site, not just the + one you're looking at. + +## Before pushing + +- Run the tests of the changed code, `flake8 feature_engine tests` (lines of 88 + characters at most) and `mypy feature_engine`. Running mypy on single files + ignores the exclusions in `pyproject.toml`. +- If the target branch already has failing tests, compare the failing tests + before and after the change instead of expecting a clean run. + +## Pull requests + +- When a PR is built on another open PR and that one is squash-merged, rebase + with `git rebase --onto origin/ ` so the PR shows + only its own files. Push with `--force-with-lease`. +- Don't end PR descriptions with an AI tool attribution line, such as + "Generated with Claude Code". diff --git a/README.md b/README.md index 8f04b87e0..3622ba9a3 100644 --- a/README.md +++ b/README.md @@ -274,7 +274,7 @@ Feature-engine documentation is built using [Sphinx](https://www.sphinx-doc.org) To build the documentation make sure you have the dependencies installed: from the root directory: ``` -pip install -r docs/requirements.txt +pip install -e ".[docs]" ``` Now you can build the docs using: diff --git a/docs/contribute/contribute_code.rst b/docs/contribute/contribute_code.rst index 4e85515ec..d5413e051 100644 --- a/docs/contribute/contribute_code.rst +++ b/docs/contribute/contribute_code.rst @@ -395,7 +395,7 @@ To do this, first make sure you have all the documentation dependencies installe set up the environment as we described previously, they should be installed. Alternatively, from the windows cmd or mac terminal, run:: - $ pip install -r docs/requirements.txt + $ pip install -e ".[docs]" Make sure you are within the feature_engine module when you run the previous command. diff --git a/docs/contribute/contribute_docs.rst b/docs/contribute/contribute_docs.rst index 856e22d85..ee16ed9f4 100644 --- a/docs/contribute/contribute_docs.rst +++ b/docs/contribute/contribute_docs.rst @@ -79,7 +79,7 @@ dependencies. If you set up the development environment as we described in the Alternatively, first activate your environment. Then navigate to the root folder of feature-engine. And now install the requirements for the documentation:: - $ pip install -r docs/requirements.txt + $ pip install -e ".[docs]" To build the documentation (and test if it is working properly) run:: diff --git a/docs/index.rst b/docs/index.rst index 98e8a3bac..2d7931cc9 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -127,7 +127,7 @@ The following characteristics make feature-engine unique: Installation ------------ -Feature-engine is a Python 3 package and works well with 3.9 or later. +Feature-engine is a Python 3 package and works well with 3.11 or later. The simplest way to install feature-engine is from PyPI with pip: diff --git a/docs/user_guide/creation/CyclicalFeatures.rst b/docs/user_guide/creation/CyclicalFeatures.rst index 59b26567a..5226afe23 100644 --- a/docs/user_guide/creation/CyclicalFeatures.rst +++ b/docs/user_guide/creation/CyclicalFeatures.rst @@ -208,6 +208,60 @@ This returns the name of all the variables in the final output: ['day_sin', 'day_cos', 'months_sin', 'months_cos'] +With polars +----------- + +:class:`CyclicalFeatures()` works in the same way with a polars dataframe. +Let's create an equivalent toy dataframe: + +.. code:: python + + import polars as pl + from feature_engine.creation import CyclicalFeatures + + df = pl.DataFrame({ + "day": [6, 7, 5, 3, 1, 2, 4], + "months": [3, 7, 9, 12, 4, 6, 12], + }) + + cyclical = CyclicalFeatures(variables=None, drop_original=False) + X = cyclical.fit_transform(df) + + cyclical.max_values_ + +The maximum values match those found with pandas: + +.. code:: python + + {'day': 7, 'months': 12} + +And the transformed dataframe contains the same cyclical features: + +.. code:: python + + print(X) + +.. code:: text + + shape: (7, 6) + ┌─────┬────────┬─────────────┬───────────┬─────────────┬─────────────┐ + │ day ┆ months ┆ day_sin ┆ day_cos ┆ months_sin ┆ months_cos │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ i64 ┆ i64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞═════╪════════╪═════════════╪═══════════╪═════════════╪═════════════╡ + │ 6 ┆ 3 ┆ -0.781831 ┆ 0.62349 ┆ 1.0 ┆ 6.1232e-17 │ + │ 7 ┆ 7 ┆ -2.4493e-16 ┆ 1.0 ┆ -0.5 ┆ -0.866025 │ + │ 5 ┆ 9 ┆ -0.974928 ┆ -0.222521 ┆ -1.0 ┆ -1.8370e-16 │ + │ 3 ┆ 12 ┆ 0.433884 ┆ -0.900969 ┆ -2.4493e-16 ┆ 1.0 │ + │ 1 ┆ 4 ┆ 0.781831 ┆ 0.62349 ┆ 0.866025 ┆ -0.5 │ + │ 2 ┆ 6 ┆ 0.974928 ┆ -0.222521 ┆ 1.2246e-16 ┆ -1.0 │ + │ 4 ┆ 12 ┆ -0.433884 ┆ -0.900969 ┆ -2.4493e-16 ┆ 1.0 │ + └─────┴────────┴─────────────┴───────────┴─────────────┴─────────────┘ + +`drop_original=True` and `get_feature_names_out()` work identically to the +pandas example above. + + Understanding cyclical encoding ------------------------------- diff --git a/docs/user_guide/creation/DecisionTreeFeatures.rst b/docs/user_guide/creation/DecisionTreeFeatures.rst index 56c6a4056..ba59800d2 100644 --- a/docs/user_guide/creation/DecisionTreeFeatures.rst +++ b/docs/user_guide/creation/DecisionTreeFeatures.rst @@ -485,6 +485,53 @@ are not there: 2670 1.843904 15709 1.843904 +With polars +----------- + +:class:`DecisionTreeFeatures()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.creation import DecisionTreeFeatures + + X = pl.DataFrame({ + "Age": [20, 44, 19, 33, 51, 40, 41, 37, 30, 54], + "Height": [164, 150, 178, 158, 188, 190, 168, 174, 176, 171], + }) + y = [4.1, 5.8, 3.9, 6.2, 4.3, 4.5, 7.2, 4.4, 4.1, 6.7] + + dtf = DecisionTreeFeatures(features_to_combine=2, drop_original=True) + dtf.fit(X, y) + + print(dtf.transform(X)) + +The resulting values match those found with pandas: + +.. code:: text + + shape: (10, 3) + ┌───────────┬──────────────┬─────────────────────────┐ + │ tree(Age) ┆ tree(Height) ┆ tree(['Age', 'Height']) │ + │ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 │ + ╞═══════════╪══════════════╪═════════════════════════╡ + │ 4.533333 ┆ 5.366667 ┆ 4.1 │ + │ 6.0 ┆ 5.366667 ┆ 6.475 │ + │ 4.533333 ┆ 4.133333 ┆ 4.0 │ + │ 4.533333 ┆ 5.366667 ┆ 6.475 │ + │ 6.0 ┆ 4.4 ┆ 4.4 │ + │ 4.533333 ┆ 4.4 ┆ 4.4 │ + │ 6.0 ┆ 6.95 ┆ 6.475 │ + │ 4.533333 ┆ 4.133333 ┆ 4.4 │ + │ 4.533333 ┆ 4.133333 ┆ 4.0 │ + │ 6.0 ┆ 6.95 ┆ 6.475 │ + └───────────┴──────────────┴─────────────────────────┘ + +`get_feature_names_out()`, classification, and every other parameter shown +above with pandas work identically with polars. + + Creating features for classification ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -498,6 +545,46 @@ identical. We just need to set the parameter `regression` to False. classification, on the other hand, the features will contain the prediction of the class. +Training trees in parallel +~~~~~~~~~~~~~~~~~~~~~~~~~~ + +Each tree is trained on its own feature combination independently of the others, so +when there are many combinations (a large number of variables and/or a high +`features_to_combine`) or a large `param_grid` to search, training can be +parallelized across combinations with the `n_jobs` parameter: + +.. code:: python + + import pandas as pd + from feature_engine.creation import DecisionTreeFeatures + + X = pd.DataFrame({ + "Age": [20, 44, 19, 33, 51, 40, 41, 37, 30, 54], + "Height": [164, 150, 178, 158, 188, 190, 168, 174, 176, 171], + "Marks": [1.0, 0.8, 0.6, 0.1, 0.3, 0.4, 0.8, 0.6, 0.5, 0.2], + }) + y = [4.1, 5.8, 3.9, 6.2, 4.3, 4.5, 7.2, 4.4, 4.1, 6.7] + + dtf = DecisionTreeFeatures(features_to_combine=3, n_jobs=2, random_state=0) + dtf.fit(X, y) + + print(dtf.transform(X).columns.tolist()) + +.. code:: text + + ['Age', 'Height', 'Marks', 'tree(Age)', 'tree(Height)', 'tree(Marks)', + "tree(['Age', 'Height'])", "tree(['Age', 'Marks'])", + "tree(['Height', 'Marks'])", "tree(['Age', 'Height', 'Marks'])"] + +`n_jobs` defaults to `None`, which trains the trees sequentially, matching this +transformer's original behaviour. Setting it trains multiple trees at the same +time using threads, which only pays off once there are enough feature +combinations or a large enough `param_grid` to outweigh the overhead of +dispatching work to threads — with just a handful of combinations, sequential +training is faster. The resulting trees and predictions are identical +regardless of `n_jobs`; only training speed changes. + + Additional resources -------------------- diff --git a/docs/user_guide/creation/GeoDistanceFeatures.rst b/docs/user_guide/creation/GeoDistanceFeatures.rst index 9744d61e9..c142c4009 100644 --- a/docs/user_guide/creation/GeoDistanceFeatures.rst +++ b/docs/user_guide/creation/GeoDistanceFeatures.rst @@ -77,11 +77,11 @@ In the following output we see the trip ID followed by the distance travelled in .. code:: python - trip_id distance_km - 0 1 3935.746254 - 1 2 2808.517344 - 2 3 1144.286561 - 3 4 1634.724892 + trip_id distance_km + 0 1 3935.746255 + 1 2 2803.971507 + 2 3 1144.291274 + 3 4 1632.166882 Using different distance methods ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -108,10 +108,10 @@ for Earth's curvature: .. code:: python trip_id distance_euclidean - 0 1 4940.252715 - 1 2 3493.298968 - 2 3 1519.295694 - 3 4 1720.178310 + 0 1 4965.730734 + 1 2 3507.416606 + 2 3 1517.763567 + 3 4 1898.819227 Alternatively, we can use the Manhattan distance, which is useful for grid-based city layouts: @@ -133,10 +133,10 @@ The Manhattan distance sums the absolute differences in latitude and longitude: .. code:: python trip_id distance_manhattan - 0 1 5628.24000 - 1 2 4684.15800 - 2 3 1637.36700 - 3 4 2279.96460 + 0 1 5649.7113 + 1 2 4266.8178 + 2 3 1641.5901 + 3 4 2263.5342 Using different output units ~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -162,10 +162,10 @@ The distances are now expressed in miles instead of kilometres: .. code:: python trip_id distance_miles - 0 1 2445.258392 - 1 2 1745.046817 - 2 3 711.000629 - 3 4 1015.643614 + 0 1 2445.586607 + 1 2 1742.326542 + 2 3 711.037560 + 3 4 1014.192788 Dropping original coordinate columns ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -193,6 +193,55 @@ After transformation, only the non-coordinate columns and the new distance colum ['trip_id', 'geo_distance'] +With polars +----------- + +:class:`GeoDistanceFeatures()` works in the same way with a polars dataframe. +Let's create an equivalent toy dataset: + +.. code:: python + + import polars as pl + from feature_engine.creation import GeoDistanceFeatures + + X = pl.DataFrame({ + 'origin_lat': [40.7128, 34.0522, 41.8781, 29.7604], + 'origin_lon': [-74.0060, -118.2437, -87.6298, -95.3698], + 'dest_lat': [34.0522, 41.8781, 40.7128, 33.4484], + 'dest_lon': [-118.2437, -87.6298, -74.0060, -112.0740], + 'trip_id': [1, 2, 3, 4] + }) + + gdt = GeoDistanceFeatures( + lat1='origin_lat', lon1='origin_lon', + lat2='dest_lat', lon2='dest_lon', + method='haversine', output_unit='km', output_col='distance_km' + ) + + gdt.fit(X) + X_transformed = gdt.transform(X) + + print(X_transformed.select(['trip_id', 'distance_km'])) + +We see the resulting distances: + +.. code:: text + + shape: (4, 2) + ┌─────────┬─────────────┐ + │ trip_id ┆ distance_km │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════════╪═════════════╡ + │ 1 ┆ 3935.746255 │ + │ 2 ┆ 2803.971507 │ + │ 3 ┆ 1144.291274 │ + │ 4 ┆ 1632.166882 │ + └─────────┴─────────────┘ + +`drop_original=True` and the different distance methods and output units +work identically to the pandas examples above. + Calculating distance within a Pipeline ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -232,7 +281,7 @@ The pipeline successfully trains and returns predictions: .. code:: python - Predictions: [100. 150. 80. 200.] + Predictions: [116.67298659 120.75252844 88.47598336 204.09850161] Additional resources -------------------- diff --git a/docs/user_guide/creation/MathFeatures.rst b/docs/user_guide/creation/MathFeatures.rst index 6b79af525..0c119d348 100644 --- a/docs/user_guide/creation/MathFeatures.rst +++ b/docs/user_guide/creation/MathFeatures.rst @@ -143,11 +143,11 @@ We obtain the following dataframe: 2 krish Liverpool 19 0.7 2020-02-24 00:02:00 19.7 3 jack Bristol 18 0.6 2020-02-24 00:03:00 18.6 - prod_Age_Marks amin_Age_Marks amax_Age_Marks std_Age_Marks - 0 18.0 0.9 20.0 13.505740 - 1 16.8 0.8 21.0 14.283557 - 2 13.3 0.7 19.0 12.940054 - 3 10.8 0.6 18.0 12.303658 + prod_Age_Marks min_Age_Marks max_Age_Marks std_Age_Marks + 0 18.0 0.9 20.0 9.55 + 1 16.8 0.8 21.0 10.10 + 2 13.3 0.7 19.0 9.15 + 3 10.8 0.6 18.0 8.70 We have the option to set the parameter `drop_original` to True to drop the variables after performing the calculations. @@ -169,11 +169,60 @@ Which will return the names of all the variables in the transformed data: 'dob', 'sum_Age_Marks', 'prod_Age_Marks', - 'amin_Age_Marks', - 'amax_Age_Marks', + 'min_Age_Marks', + 'max_Age_Marks', 'std_Age_Marks'] +With polars +----------- + +:class:`MathFeatures()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.creation import MathFeatures + + df = pl.DataFrame({ + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + }) + + transformer = MathFeatures( + variables=["Age", "Marks"], + func=["sum", "prod", "min", "max", "std"], + ) + + print(transformer.fit_transform(df)) + +The resulting values match those found with pandas: + +.. code:: text + + shape: (4, 7) + ┌─────┬───────┬───────────────┬────────────────┬───────────────┬───────────────┬───────────────┐ + │ Age ┆ Marks ┆ sum_Age_Marks ┆ prod_Age_Marks ┆ min_Age_Marks ┆ max_Age_Marks ┆ std_Age_Marks │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ i64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞═════╪═══════╪═══════════════╪════════════════╪═══════════════╪═══════════════╪═══════════════╡ + │ 20 ┆ 0.9 ┆ 20.9 ┆ 18.0 ┆ 0.9 ┆ 20.0 ┆ 13.50574 │ + │ 21 ┆ 0.8 ┆ 21.8 ┆ 16.8 ┆ 0.8 ┆ 21.0 ┆ 14.283557 │ + │ 19 ┆ 0.7 ┆ 19.7 ┆ 13.3 ┆ 0.7 ┆ 19.0 ┆ 12.940054 │ + │ 18 ┆ 0.6 ┆ 18.6 ┆ 10.8 ┆ 0.6 ┆ 18.0 ┆ 12.303658 │ + └─────┴───────┴───────────────┴────────────────┴───────────────┴───────────────┴───────────────┘ + +`new_variables_names`, `drop_original`, and `get_feature_names_out()` work +identically to the pandas examples above. + +If you pass a custom Python callable as `func` (instead of a string or one +of the common aggregations above, which are always NumPy-vectorized), note +that the callable receives a **plain tuple** of values for polars input, +not a pandas `Series` — so `lambda row: max(row) - min(row)` works on both +backends, but `lambda row: row.max() - row.min()` (which relies on `Series` +methods) only works with pandas. + + New variables names ^^^^^^^^^^^^^^^^^^^ diff --git a/docs/user_guide/creation/RelativeFeatures.rst b/docs/user_guide/creation/RelativeFeatures.rst index 5611aa204..870b5c1cd 100644 --- a/docs/user_guide/creation/RelativeFeatures.rst +++ b/docs/user_guide/creation/RelativeFeatures.rst @@ -141,6 +141,51 @@ Which will return the names of all the variables in the transformed data: 'Marks_pow_Age'] +With polars +----------- + +:class:`RelativeFeatures()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.creation import RelativeFeatures + + df = pl.DataFrame({ + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + }) + + transformer = RelativeFeatures( + variables=["Age", "Marks"], + reference=["Age"], + func = ["sub", "div", "mod", "pow"], + ) + + print(transformer.fit_transform(df)) + +The resulting values match those found with pandas (`Age_pow_Age`'s large +values are genuine `int64` overflow from raising `Age` to the power of +itself, not an error - the same happens with pandas): + +.. code:: text + + shape: (4, 10) + ┌─────┬───────┬─────────────┬───────────────┬─────────────┬───────────────┬─────────────┬───────────────┬──────────────────────┬───────────────┐ + │ Age ┆ Marks ┆ Age_sub_Age ┆ Marks_sub_Age ┆ Age_div_Age ┆ Marks_div_Age ┆ Age_mod_Age ┆ Marks_mod_Age ┆ Age_pow_Age ┆ Marks_pow_Age │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ i64 ┆ f64 ┆ i64 ┆ f64 ┆ f64 ┆ f64 ┆ i64 ┆ f64 ┆ i64 ┆ f64 │ + ╞═════╪═══════╪═════════════╪═══════════════╪═════════════╪═══════════════╪═════════════╪═══════════════╪══════════════════════╪═══════════════╡ + │ 20 ┆ 0.9 ┆ 0 ┆ -19.1 ┆ 1.0 ┆ 0.045 ┆ 0 ┆ 0.9 ┆ -2101438300051996672 ┆ 0.121577 │ + │ 21 ┆ 0.8 ┆ 0 ┆ -20.2 ┆ 1.0 ┆ 0.038095 ┆ 0 ┆ 0.8 ┆ -1595931050845505211 ┆ 0.009223 │ + │ 19 ┆ 0.7 ┆ 0 ┆ -18.3 ┆ 1.0 ┆ 0.036842 ┆ 0 ┆ 0.7 ┆ 6353754964178307979 ┆ 0.00114 │ + │ 18 ┆ 0.6 ┆ 0 ┆ -17.4 ┆ 1.0 ┆ 0.033333 ┆ 0 ┆ 0.6 ┆ -497033925936021504 ┆ 0.000102 │ + └─────┴───────┴─────────────┴───────────────┴─────────────┴───────────────┴─────────────┴───────────────┴──────────────────────┴───────────────┘ + +`fill_value`, `drop_original`, and `get_feature_names_out()` work +identically to the pandas examples above. + + Additional resources -------------------- diff --git a/docs/user_guide/datetime/DatetimeFeatures.rst b/docs/user_guide/datetime/DatetimeFeatures.rst index 54667f8d3..8a1769a35 100644 --- a/docs/user_guide/datetime/DatetimeFeatures.rst +++ b/docs/user_guide/datetime/DatetimeFeatures.rst @@ -743,6 +743,59 @@ In the following output we see the resulting dataframe: As you can see, we do not have the constant features in the transformed dataset. +With polars +----------- + +:class:`DatetimeFeatures()` also works with polars dataframes. + +.. code:: python + + import polars as pl + from feature_engine.datetime import DatetimeFeatures + + toy_df = pl.DataFrame({ + "id": [1, 2, 3, 4], + "var_date": ["2012-06-21", "1998-02-10", "2010-08-03", "2020-10-31"], + }) + + dfts = DatetimeFeatures( + features_to_extract=["month", "year", "day_of_week", "days_in_month"], + ) + + df_transf = dfts.fit_transform(toy_df) + + df_transf + +We see the new features in the following output: + +.. code:: text + + shape: (4, 5) + ┌─────┬────────────────┬───────────────┬──────────────────────┬────────────────────────┐ + │ id ┆ var_date_month ┆ var_date_year ┆ var_date_day_of_week ┆ var_date_days_in_month │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ i64 ┆ i8 ┆ i32 ┆ i8 ┆ i8 │ + ╞═════╪════════════════╪═══════════════╪══════════════════════╪════════════════════════╡ + │ 1 ┆ 6 ┆ 2012 ┆ 3 ┆ 30 │ + │ 2 ┆ 2 ┆ 1998 ┆ 1 ┆ 28 │ + │ 3 ┆ 8 ┆ 2010 ┆ 1 ┆ 31 │ + │ 4 ┆ 10 ┆ 2020 ┆ 5 ┆ 31 │ + └─────┴────────────────┴───────────────┴──────────────────────┴────────────────────────┘ + +.. note:: + + When parsing string columns, pandas relies on `dateutil` and can infer loose or + ambiguous formats. Polars parses dates natively and needs the format to be + unambiguous and consistent across the column: ISO 8601 (e.g. *2012-06-21*) parses + reliably, but looser formats, such as day-first dates like *21/06/2012*, require an + explicit `format`. The `dayfirst`, `yearfirst` and `utc` parameters are + `pandas.to_datetime`-only options and have no effect on polars input. + +.. note:: + + `variables="index"` is only supported when `X` is a pandas dataframe, since only pandas + dataframes have an index. + Working with different timezones -------------------------------- diff --git a/docs/user_guide/datetime/DatetimeOrdinal.rst b/docs/user_guide/datetime/DatetimeOrdinal.rst index b2c4bc041..512899140 100644 --- a/docs/user_guide/datetime/DatetimeOrdinal.rst +++ b/docs/user_guide/datetime/DatetimeOrdinal.rst @@ -55,7 +55,8 @@ Datetime ordinal with feature-engine ordinal numbers. It works with variables whose dtype is datetime, as well as with object-type variables, provided that they can be parsed into datetime format. -:class:`DatetimeOrdinal()` uses pandas `toordinal()` under the hood. The main +:class:`DatetimeOrdinal()` computes the same proleptic Gregorian ordinal that +Python's `toordinal()` returns, vectorized under the hood for speed. The main functionalities are: - It can convert multiple datetime variables at once. @@ -111,6 +112,51 @@ We see the new ordinal feature in the output: By default, :class:`DatetimeOrdinal()` drops the original datetime variable. To keep it, you can set `drop_original=False`. +With polars +~~~~~~~~~~~ + +:class:`DatetimeOrdinal()` works the same way with polars dataframes: + +.. code:: python + + import polars as pl + from feature_engine.datetime import DatetimeOrdinal + + toy_df = pl.DataFrame({ + "var_date1": ["1989-05-15", "2020-12-01", "1999-01-20", "2002-02-14"], + "var_date2": ["2012-06-21", "1998-02-10", "2010-08-03", "2020-10-31"], + "other_var": [1, 2, 3, 4] + }) + + dtfs = DatetimeOrdinal(variables="var_date2") + + df_transf = dtfs.fit_transform(toy_df) + + df_transf + +.. code:: text + + shape: (4, 3) + ┌────────────┬───────────┬───────────────────┐ + │ var_date1 ┆ other_var ┆ var_date2_ordinal │ + │ --- ┆ --- ┆ --- │ + │ str ┆ i64 ┆ i64 │ + ╞════════════╪═══════════╪═══════════════════╡ + │ 1989-05-15 ┆ 1 ┆ 734675 │ + │ 2020-12-01 ┆ 2 ┆ 729430 │ + │ 1999-01-20 ┆ 3 ┆ 733987 │ + │ 2002-02-14 ┆ 4 ┆ 737729 │ + └────────────┴───────────┴───────────────────┘ + +.. note:: + + For string variables, pandas leans on `dateutil` and can guess its way through + loosely-formatted or ambiguous dates (e.g. ``"May-1989"``, ``"06/21/2012"``). + Polars parses dates natively and needs the format to be unambiguous and + consistent across the column - ISO 8601 (e.g. ``"1989-05-15"``) parses + reliably, but looser formats may raise an error. If your dates arrive in a + looser format, convert them to a native `Date`/`Datetime` column upstream. + Calculate days from a start date ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -135,9 +181,9 @@ The new feature now represents the number of days between `var_date2` and Januar var_date1 other_var var_date2_ordinal 0 May-1989 1 903 - 1 Dec-2020 2 -4343 + 1 Dec-2020 2 -4342 2 Jan-1999 3 215 - 3 Feb-2002 4 3956 + 3 Feb-2002 4 3957 Missing timestamps @@ -150,7 +196,8 @@ If `missing_values="raise"`, the transformer will raise an error if NaT values a found in the datetime variables during `fit()` or `transform()`. If `missing_values="ignore"`, the transformer will ignore NaT values, and the resulting -ordinal feature will contain `NaN` (or `pd.NA`) in their place. +ordinal feature will contain a missing value in their place - `NaN` (`float64`) for +pandas, and `null` (`Int64`) for polars, following each library's own convention. Additional resources diff --git a/docs/user_guide/datetime/DatetimeSubtraction.rst b/docs/user_guide/datetime/DatetimeSubtraction.rst index 662657692..638502472 100644 --- a/docs/user_guide/datetime/DatetimeSubtraction.rst +++ b/docs/user_guide/datetime/DatetimeSubtraction.rst @@ -156,6 +156,42 @@ original variables and also the new variables with the time difference: 4 2019-03-09 2018-04-08 0.917199 +With polars +~~~~~~~~~~~ + +:class:`DatetimeSubtraction()` also works with polars dataframes: + +.. code:: python + + import polars as pl + from feature_engine.datetime import DatetimeSubtraction + + data = pl.DataFrame({ + "date1" : ["2022-09-01", "2022-10-01", "2022-12-01"], + "date2" : ["2022-09-15", "2022-10-15", "2022-12-15"], + "date3" : ["2022-08-01", "2022-09-01", "2022-11-01"], + "date4" : ["2022-08-15", "2022-09-15", "2022-11-15"], + }) + + dtf = DatetimeSubtraction(variables=["date1", "date2"], reference=["date3", "date4"]) + + data = dtf.fit_transform(data) + + print(data) + +.. code:: text + + shape: (3, 8) + ┌────────────┬────────────┬────────────┬────────────┬─────────────────┬─────────────────┬─────────────────┬─────────────────┐ + │ date1 ┆ date2 ┆ date3 ┆ date4 ┆ date1_sub_date3 ┆ date2_sub_date3 ┆ date1_sub_date4 ┆ date2_sub_date4 │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ str ┆ str ┆ str ┆ str ┆ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞════════════╪════════════╪════════════╪════════════╪═════════════════╪═════════════════╪═════════════════╪═════════════════╡ + │ 2022-09-01 ┆ 2022-09-15 ┆ 2022-08-01 ┆ 2022-08-15 ┆ 31.0 ┆ 45.0 ┆ 17.0 ┆ 31.0 │ + │ 2022-10-01 ┆ 2022-10-15 ┆ 2022-09-01 ┆ 2022-09-15 ┆ 30.0 ┆ 44.0 ┆ 16.0 ┆ 30.0 │ + │ 2022-12-01 ┆ 2022-12-15 ┆ 2022-11-01 ┆ 2022-11-15 ┆ 30.0 ┆ 44.0 ┆ 16.0 ┆ 30.0 │ + └────────────┴────────────┴────────────┴────────────┴─────────────────┴─────────────────┴─────────────────┴─────────────────┘ + Drop original variables after computation ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ diff --git a/docs/user_guide/discretisation/ArbitraryDiscretiser.rst b/docs/user_guide/discretisation/ArbitraryDiscretiser.rst index b4d81e604..9c42a2cd1 100644 --- a/docs/user_guide/discretisation/ArbitraryDiscretiser.rst +++ b/docs/user_guide/discretisation/ArbitraryDiscretiser.rst @@ -110,6 +110,42 @@ obtain monotonic relationships between the variable and the target, you can do s seamlessly by setting `return_object` to True. You can find an example of discretisation followed by encoding to obtain monotonic releationships `here `_. +With polars +----------- + +:class:`ArbitraryDiscretiser()` also works with polars dataframes. + +.. code:: python + + import polars as pl + import numpy as np + from feature_engine.discretisation import ArbitraryDiscretiser + + X = pl.DataFrame({ + "MedInc": [1.5, 3.0, 5.0, 8.0, 0.8], + }) + + user_dict = {"MedInc": [0, 2, 4, 6, np.inf]} + + transformer = ArbitraryDiscretiser(binning_dict=user_dict, return_boundaries=False) + X_t = transformer.fit_transform(X) + print(X_t) + +.. code:: text + + shape: (5, 1) + ┌────────┐ + │ MedInc │ + │ --- │ + │ i64 │ + ╞════════╡ + │ 0 │ + │ 1 │ + │ 2 │ + │ 3 │ + │ 0 │ + └────────┘ + Additional resources -------------------- diff --git a/docs/user_guide/discretisation/DecisionTreeDiscretiser.rst b/docs/user_guide/discretisation/DecisionTreeDiscretiser.rst index 80e994b8d..c15d4e418 100644 --- a/docs/user_guide/discretisation/DecisionTreeDiscretiser.rst +++ b/docs/user_guide/discretisation/DecisionTreeDiscretiser.rst @@ -443,6 +443,88 @@ were sorted: 799 0 9 380 0 9 +With polars +----------- + +:class:`DecisionTreeDiscretiser()` also accepts polars dataframes as input, and returns a polars +dataframe from `transform()`: + +.. code:: python + + import polars as pl + + X_train_pl = pl.DataFrame(X_train[["LotArea", "GrLivArea"]]) + + disc = DecisionTreeDiscretiser( + bin_output="prediction", + cv=3, + scoring="neg_mean_squared_error", + regression=True, + ) + disc.fit(X_train_pl, y_train) + + train_t = disc.transform(X_train_pl) + print(train_t.head()) + +.. code:: text + + shape: (5, 2) + ┌───────────────┬───────────────┐ + │ LotArea ┆ GrLivArea │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞═══════════════╪═══════════════╡ + │ 144174.283688 ┆ 152471.713568 │ + │ 144174.283688 ┆ 191760.966667 │ + │ 176117.741848 ┆ 97156.25 │ + │ 144174.283688 ┆ 202178.409091 │ + │ 144174.283688 ┆ 202178.409091 │ + └───────────────┴───────────────┘ + +The predictions match those obtained with the pandas dataframe above. + +Training trees in parallel +--------------------------- + +:class:`DecisionTreeDiscretiser()` fits one decision tree per variable, independently of the +others. When there are many variables to discretise, or a large `param_grid` to search, training +can be parallelized across variables with the `n_jobs` parameter: + +.. code:: python + + import pandas as pd + from feature_engine.discretisation import DecisionTreeDiscretiser + + X = pd.DataFrame({ + "Age": [20, 44, 19, 33, 51, 40, 41, 37, 30, 54], + "Height": [164, 150, 178, 158, 188, 190, 168, 174, 176, 171], + "Marks": [1.0, 0.8, 0.6, 0.1, 0.3, 0.4, 0.8, 0.6, 0.5, 0.2], + }) + y = [4.1, 5.8, 3.9, 6.2, 4.3, 4.5, 7.2, 4.4, 4.1, 6.7] + + dtd = DecisionTreeDiscretiser(n_jobs=2, random_state=0) + dtd.fit(X, y) + + print(dtd.transform(X)) + +.. code:: text + + Age Height Marks + 0 4.533333 5.366667 4.100000 + 1 6.000000 5.366667 6.500000 + 2 4.533333 4.133333 4.133333 + 3 4.533333 5.366667 6.200000 + 4 6.000000 4.400000 4.400000 + 5 4.533333 4.400000 4.400000 + 6 6.000000 6.950000 6.500000 + 7 4.533333 4.133333 4.133333 + 8 4.533333 4.133333 4.133333 + 9 6.000000 6.950000 6.700000 + +`n_jobs` is the number of jobs to run in parallel. `fit` is parallelized over the variables, +training one decision tree per variable. `None` means 1 unless in a `joblib.parallel_backend` +context. `-1` means using all processors. + Additional considerations ------------------------- diff --git a/docs/user_guide/discretisation/EqualFrequencyDiscretiser.rst b/docs/user_guide/discretisation/EqualFrequencyDiscretiser.rst index 9e2c408d7..f29064f0a 100644 --- a/docs/user_guide/discretisation/EqualFrequencyDiscretiser.rst +++ b/docs/user_guide/discretisation/EqualFrequencyDiscretiser.rst @@ -46,8 +46,9 @@ would potentially impact the model's performance in this scenario. EqualFrequencyDiscretiser ------------------------- -Feature-engine's :class:`EqualFrequencyDiscretiser` applies equal frequency discretisation to numerical variables. It uses -the `pandas.qcut()` function under the hood to determine the interval limits. +Feature-engine's :class:`EqualFrequencyDiscretiser` applies equal frequency discretisation to numerical variables. It +determines the interval limits from the variable's quantiles, matching the limits that `pandas.qcut()` would return, and +works with both pandas and polars dataframes. You can specify the variables to be discretised by passing their names in a list when setting up the transformer. Alternatively, :class:`EqualFrequencyDiscretiser` will automatically infer the data types and compute the interval limits for all numeric variables. @@ -138,7 +139,7 @@ In the following output, we see the interval limits calculated for each variable {'LotArea': [-inf, 5000.0, 7105.6, - 8099.200000000003, + 8099.200000000004, 8874.0, 9600.0, 10318.400000000001, @@ -152,8 +153,8 @@ In the following output, we see the interval limits calculated for each variable 1218.0, 1348.4, 1476.5, - 1601.6000000000001, - 1717.6999999999998, + 1601.6000000000004, + 1717.7000000000003, 1893.0000000000005, 2166.3999999999996, inf]} @@ -393,6 +394,49 @@ the value range. .. image:: ../../images/equalfrequencydiscretisation_skewed.png +With polars +----------- + +:class:`EqualFrequencyDiscretiser` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.discretisation import EqualFrequencyDiscretiser + + df = pl.DataFrame({ + "Age": [20, 21, 19, 18, 25, 30, 45, 60, 15, 22], + "Marks": [0.9, 0.8, 0.7, 0.6, 0.5, 0.4, 0.3, 0.2, 0.1, 0.95], + }) + + disc = EqualFrequencyDiscretiser(q=5, variables=["Age", "Marks"]) + + print(disc.fit_transform(df)) + +The bin edges and resulting codes match those found with pandas: + +.. code:: text + + shape: (10, 2) + ┌─────┬───────┐ + │ Age ┆ Marks │ + │ --- ┆ --- │ + │ i64 ┆ i64 │ + ╞═════╪═══════╡ + │ 1 ┆ 4 │ + │ 2 ┆ 3 │ + │ 1 ┆ 3 │ + │ 0 ┆ 2 │ + │ 3 ┆ 2 │ + │ 3 ┆ 1 │ + │ 4 ┆ 1 │ + │ 4 ┆ 0 │ + │ 0 ┆ 0 │ + │ 2 ┆ 4 │ + └─────┴───────┘ + +`return_object`, `return_boundaries`, and `get_feature_names_out()` work identically to the pandas examples above. + See Also -------- diff --git a/docs/user_guide/discretisation/EqualWidthDiscretiser.rst b/docs/user_guide/discretisation/EqualWidthDiscretiser.rst index bd149df24..16bfda029 100644 --- a/docs/user_guide/discretisation/EqualWidthDiscretiser.rst +++ b/docs/user_guide/discretisation/EqualWidthDiscretiser.rst @@ -48,9 +48,10 @@ potentially impact the model's performance in this scenario. EqualWidthDiscretiser --------------------- -Feture-engine's :class:`EqualWidthDiscretiser()` applies equal width discretisation to numerical variables. It uses -the `pandas.cut()` function under the hood to find the interval limits and then sort the continuous variables into -the bins. +Feture-engine's :class:`EqualWidthDiscretiser()` applies equal width discretisation to numerical variables. It finds +the interval limits from each variable's minimum and maximum value, then sorts the continuous variables into the +bins. It works with pandas, polars, and any other dataframe library supported by +`narwhals `_. You can specify the variables to be discretised by passing their names in a list when you set up the transformer. Alternatively, :class:`EqualWidthDiscretiser()` will automatically infer the data types and compute the interval limits for all numeric @@ -271,7 +272,7 @@ If we want to output the intervals limits instead of integers, we can set `retur .. code:: python # Set up the discretisation transformer - disc = EqualFrequencyDiscretiser( + disc = EqualWidthDiscretiser( bins=10, variables=['LotArea','GrLivArea'], return_boundaries=True) @@ -301,6 +302,48 @@ While we can't use these variables to train machine learning models, as opposed to the variables discretised into integers, they are very useful in this format for data analysis, and we can use any feature-engine encoder for further processing. +With polars +~~~~~~~~~~~ + +:class:`EqualWidthDiscretiser()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.discretisation import EqualWidthDiscretiser + + df = pl.DataFrame({ + "x": [10400, 3675, 8640, 11670, 10667, 6120, 9500, 14000, 7200, 5300], + }) + + disc = EqualWidthDiscretiser(bins=5) + + print(disc.fit_transform(df)) + +The resulting values match those found with pandas: + +.. code:: text + + shape: (10, 1) + ┌─────┐ + │ x │ + │ --- │ + │ i64 │ + ╞═════╡ + │ 3 │ + │ 0 │ + │ 2 │ + │ 3 │ + │ 3 │ + │ 1 │ + │ 2 │ + │ 4 │ + │ 1 │ + │ 0 │ + └─────┘ + +`return_object`, `return_boundaries`, and `binner_dict_` work identically to the pandas examples above. + See Also -------- diff --git a/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst b/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst index 74e150763..940746d9d 100644 --- a/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst +++ b/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst @@ -144,6 +144,66 @@ In the following output, we see the interval limits determined for each variable 2212.974, inf]} +With polars +----------- + +:class:`GeometricWidthDiscretiser()` works in the same way with a polars dataframe: + +.. code:: python + + import numpy as np + import polars as pl + from feature_engine.discretisation import GeometricWidthDiscretiser + + np.random.seed(42) + df = pl.DataFrame({"x": np.random.randint(1, 100, 100).astype(float)}) + + disc = GeometricWidthDiscretiser(bins=10) + Xt = disc.fit_transform(df) + + print(Xt["x"].value_counts().sort("x")) + +The resulting bin counts: + +.. code:: text + + shape: (9, 2) + ┌─────┬───────┐ + │ x ┆ count │ + │ --- ┆ --- │ + │ i64 ┆ u32 │ + ╞═════╪═══════╡ + │ 0 ┆ 6 │ + │ 1 ┆ 3 │ + │ 3 ┆ 3 │ + │ 4 ┆ 1 │ + │ 5 ┆ 5 │ + │ 6 ┆ 9 │ + │ 7 ┆ 8 │ + │ 8 ┆ 25 │ + │ 9 ┆ 40 │ + └─────┴───────┘ + +And the fitted bin edges, matching what we'd get fitting on the same values with pandas: + +.. code:: python + + disc.binner_dict_ + +.. code:: python + + {'x': [-inf, + 3.573433146226546, + 4.475691865644366, + 5.895335641248283, + 8.129050213617685, + 11.643650760992958, + 17.173639757979174, + 25.874707744105372, + 39.565256521047, + 61.106419756718246, + inf]} + Interval width ~~~~~~~~~~~~~~ diff --git a/docs/user_guide/encoding/CountEncoder.rst b/docs/user_guide/encoding/CountEncoder.rst index 929507984..8616433f1 100644 --- a/docs/user_guide/encoding/CountEncoder.rst +++ b/docs/user_guide/encoding/CountEncoder.rst @@ -265,6 +265,55 @@ With the method `inverse_transform`, we can transform the encoded dataframes bac original representation, that is, we can replace the encoding with the original categorical values. +With polars +----------- + +:class:`CountEncoder()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.encoding import CountEncoder + + df = pl.DataFrame({ + "cabin": ["M", "C", "M", "B", "M"], + "sex": ["male", "female", "male", "female", "male"], + "embarked": ["S", "C", "S", "S", "Q"], + }) + + encoder = CountEncoder( + encoding_method="count", + variables=["cabin", "sex", "embarked"], + ) + encoder.fit(df) + + print(encoder.encoder_dict_) + +.. code:: python + + {'cabin': {'M': 3, 'C': 1, 'B': 1}, 'sex': {'male': 3, 'female': 2}, 'embarked': {'S': 3, 'C': 1, 'Q': 1}} + +.. code:: python + + Xt = encoder.transform(df) + + print(Xt) + +.. code:: text + + shape: (5, 3) + ┌───────┬─────┬──────────┐ + │ cabin ┆ sex ┆ embarked │ + │ --- ┆ --- ┆ --- │ + │ i64 ┆ i64 ┆ i64 │ + ╞═══════╪═════╪══════════╡ + │ 3 ┆ 3 ┆ 3 │ + │ 1 ┆ 2 ┆ 1 │ + │ 3 ┆ 3 ┆ 3 │ + │ 1 ┆ 2 ┆ 3 │ + │ 3 ┆ 3 ┆ 1 │ + └───────┴─────┴──────────┘ + Additional resources -------------------- diff --git a/docs/user_guide/encoding/DecisionTreeEncoder.rst b/docs/user_guide/encoding/DecisionTreeEncoder.rst index 9c5ab968d..5e220597c 100644 --- a/docs/user_guide/encoding/DecisionTreeEncoder.rst +++ b/docs/user_guide/encoding/DecisionTreeEncoder.rst @@ -438,6 +438,63 @@ In the following image we also see a monotonic relationship after the encoding: be some sort of relationship between the target and the categories that can be captured by the decision tree. Use with caution. +With polars +----------- + +:class:`DecisionTreeEncoder()` works the same way with a polars dataframe. Let's create a toy +dataset: + +.. code:: python + + import polars as pl + from feature_engine.encoding import DecisionTreeEncoder + + X = pl.DataFrame({ + "city": ["London", "Manchester", "Liverpool", "London", "Manchester", "Liverpool"], + "price": [500, 300, 250, 520, 310, 260], + }) + y = pl.Series("target", [1, 0, 0, 1, 0, 1]) + +Let's set up :class:`DecisionTreeEncoder()` to encode `city` with a classification tree, and fit +it to the data: + +.. code:: python + + encoder = DecisionTreeEncoder(variables=["city"], regression=False, cv=2) + encoder.fit(X, y) + + encoder.encoder_dict_ + +We see the resulting mappings from category to the tree's predictions: + +.. code:: python + + {'city': {'London': 1.0, 'Manchester': 0.25, 'Liverpool': 0.25}} + +Now let's transform the data: + +.. code:: python + + encoder.transform(X) + +We obtain a polars dataframe with the categories in `city` replaced by the tree's predictions: + +.. code:: text + + shape: (6, 2) + ┌──────┬───────┐ + │ city ┆ price │ + │ --- ┆ --- │ + │ f64 ┆ i64 │ + ╞══════╪═══════╡ + │ 1.0 ┆ 500 │ + │ 0.25 ┆ 300 │ + │ 0.25 ┆ 250 │ + │ 1.0 ┆ 520 │ + │ 0.25 ┆ 310 │ + │ 0.25 ┆ 260 │ + └──────┴───────┘ + Additional resources -------------------- diff --git a/docs/user_guide/encoding/MeanEncoder.rst b/docs/user_guide/encoding/MeanEncoder.rst index 086eef00f..2419e8b74 100644 --- a/docs/user_guide/encoding/MeanEncoder.rst +++ b/docs/user_guide/encoding/MeanEncoder.rst @@ -364,6 +364,62 @@ After encoding the features we can use the data sets to train machine learning a encoded variable. Hence, this encoding method is suitable for predictive modelling that uses models that are sensitive to the size of the feature space. +With polars +~~~~~~~~~~~ + +:class:`MeanEncoder()` works the same way with a polars dataframe. Let's create a toy dataset: + +.. code:: python + + import polars as pl + from feature_engine.encoding import MeanEncoder + + X = pl.DataFrame({ + "city": ["London", "Manchester", "Liverpool", "London", "Manchester", "Liverpool"], + "price": [500, 300, 250, 520, 310, 260], + }) + y = pl.Series("target", [1, 0, 0, 1, 0, 1]) + +Let's set up :class:`MeanEncoder()` to encode `city` with the target mean, and fit it to the data: + +.. code:: python + + encoder = MeanEncoder(variables=["city"]) + encoder.fit(X, y) + + encoder.encoder_dict_ + +We see the resulting mappings from category to target mean: + +.. code:: python + + {'city': {'London': 1.0, 'Liverpool': 0.5, 'Manchester': 0.0}} + +Now let's transform the data: + +.. code:: python + + encoder.transform(X) + +We obtain a polars dataframe with the categories in `city` replaced by the target mean: + +.. code:: text + + shape: (6, 2) + ┌──────┬───────┐ + │ city ┆ price │ + │ --- ┆ --- │ + │ f64 ┆ i64 │ + ╞══════╪═══════╡ + │ 1.0 ┆ 500 │ + │ 0.0 ┆ 300 │ + │ 0.5 ┆ 250 │ + │ 1.0 ┆ 520 │ + │ 0.0 ┆ 310 │ + │ 0.5 ┆ 260 │ + └──────┴───────┘ + + Additional resources -------------------- diff --git a/docs/user_guide/encoding/OneHotEncoder.rst b/docs/user_guide/encoding/OneHotEncoder.rst index 02901671e..94cc1a284 100644 --- a/docs/user_guide/encoding/OneHotEncoder.rst +++ b/docs/user_guide/encoding/OneHotEncoder.rst @@ -521,6 +521,39 @@ We see the names of the columns below: 'embarked_S', 'embarked_C'] +With polars +----------- + +:class:`OneHotEncoder()` works the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.encoding import OneHotEncoder + + X = pl.DataFrame({"x1": ["b", "b", "b", "a", "a"], "x2": [1, 2, 3, 4, 5]}) + + ohe = OneHotEncoder(variables=["x1"]) + ohe.fit(X) + + print(ohe.transform(X)) + +.. code:: text + + shape: (5, 3) + ┌─────┬──────┬──────┐ + │ x2 ┆ x1_b ┆ x1_a │ + │ --- ┆ --- ┆ --- │ + │ i64 ┆ i8 ┆ i8 │ + ╞═════╪══════╪══════╡ + │ 1 ┆ 1 ┆ 0 │ + │ 2 ┆ 1 ┆ 0 │ + │ 3 ┆ 1 ┆ 0 │ + │ 4 ┆ 0 ┆ 1 │ + │ 5 ┆ 0 ┆ 1 │ + └─────┴──────┴──────┘ + + Considerations -------------- diff --git a/docs/user_guide/encoding/OrdinalEncoder.rst b/docs/user_guide/encoding/OrdinalEncoder.rst index cff284c08..08f5c9417 100644 --- a/docs/user_guide/encoding/OrdinalEncoder.rst +++ b/docs/user_guide/encoding/OrdinalEncoder.rst @@ -532,6 +532,62 @@ might otherwise go unnoticed. The power of ordinal ordered encoder resides in its intrinsic capacity of finding monotonic relationships. +With polars +~~~~~~~~~~~ + +:class:`OrdinalEncoder()` works the same way with a polars dataframe. Let's create a toy dataset: + +.. code:: python + + import polars as pl + from feature_engine.encoding import OrdinalEncoder + + X = pl.DataFrame({ + "city": ["London", "Manchester", "Liverpool", "London", "Manchester", "Liverpool"], + "price": [500, 300, 250, 520, 310, 260], + }) + y = pl.Series("target", [1, 0, 0, 1, 0, 1]) + +Let's set up :class:`OrdinalEncoder()` to encode `city` with ordered ordinal encoding, and fit it to the data: + +.. code:: python + + encoder = OrdinalEncoder(encoding_method="ordered", variables=["city"]) + encoder.fit(X, y) + + encoder.encoder_dict_ + +We see the resulting mappings from category to integer: + +.. code:: python + + {'city': {'Manchester': 0, 'Liverpool': 1, 'London': 2}} + +Now let's transform the data: + +.. code:: python + + encoder.transform(X) + +We obtain a polars dataframe with the categories in `city` replaced by their ordinal number: + +.. code:: text + + shape: (6, 2) + ┌──────┬───────┐ + │ city ┆ price │ + │ --- ┆ --- │ + │ i64 ┆ i64 │ + ╞══════╪═══════╡ + │ 2 ┆ 500 │ + │ 0 ┆ 300 │ + │ 1 ┆ 250 │ + │ 2 ┆ 520 │ + │ 0 ┆ 310 │ + │ 1 ┆ 260 │ + └──────┴───────┘ + + Additional resources -------------------- diff --git a/docs/user_guide/encoding/RareLabelEncoder.rst b/docs/user_guide/encoding/RareLabelEncoder.rst index c19719c66..f9173395a 100644 --- a/docs/user_guide/encoding/RareLabelEncoder.rst +++ b/docs/user_guide/encoding/RareLabelEncoder.rst @@ -179,11 +179,12 @@ In the following output, we see the number of observations per category: .. code:: python + var_A A 10 B 10 C 2 D 1 - Name: var_A, dtype: int64 + Name: count, dtype: int64 Now, we group categories only for variables with more than 3 unique categories: @@ -216,10 +217,43 @@ a new category called `Rare`: .. code:: python + var_A A 10 B 10 Rare 3 - Name: var_A, dtype: int64 + Name: count, dtype: int64 + +With polars +----------- + +:class:`RareLabelEncoder()` works the same way with polars dataframes: + +.. code:: python + + import polars as pl + from feature_engine.encoding import RareLabelEncoder + + data = {'var_A': ['A'] * 10 + ['B'] * 10 + ['C'] * 2 + ['D'] * 1} + data = pl.DataFrame(data) + + rare_encoder = RareLabelEncoder(tol=0.05, n_categories=3, max_n_categories=2) + Xt = rare_encoder.fit_transform(data) + Xt['var_A'].value_counts().sort('var_A') + +We see the same grouping as with the pandas dataframe: + +.. code:: text + + shape: (3, 2) + ┌───────┬───────┐ + │ var_A ┆ count │ + │ --- ┆ --- │ + │ str ┆ u32 │ + ╞═══════╪═══════╡ + │ A ┆ 10 │ + │ B ┆ 10 │ + │ Rare ┆ 3 │ + └───────┴───────┘ Considerations -------------- diff --git a/docs/user_guide/encoding/StringSimilarityEncoder.rst b/docs/user_guide/encoding/StringSimilarityEncoder.rst index 3fbfaec27..0397db198 100644 --- a/docs/user_guide/encoding/StringSimilarityEncoder.rst +++ b/docs/user_guide/encoding/StringSimilarityEncoder.rst @@ -299,6 +299,37 @@ Below, we see the resulting dataframe: 393 0.0 0.437500 0.666667 0.666667 +With polars +----------- + +:class:`StringSimilarityEncoder()` works the same way with polars dataframes: + +.. code:: python + + import polars as pl + from feature_engine.encoding import StringSimilarityEncoder + + df = pl.DataFrame({"words": ["dog", "dig", "cat"]}) + + encoder = StringSimilarityEncoder() + dft = encoder.fit_transform(df) + dft + +We see the same similarity values as with the pandas dataframe: + +.. code:: text + + shape: (3, 3) + ┌───────────┬───────────┬───────────┐ + │ words_dog ┆ words_dig ┆ words_cat │ + │ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 │ + ╞═══════════╪═══════════╪═══════════╡ + │ 1.0 ┆ 0.666667 ┆ 0.0 │ + │ 0.666667 ┆ 1.0 ┆ 0.0 │ + │ 0.0 ┆ 0.0 ┆ 1.0 │ + └───────────┴───────────┴───────────┘ + Additional resources -------------------- diff --git a/docs/user_guide/encoding/WoEEncoder.rst b/docs/user_guide/encoding/WoEEncoder.rst index a25d83074..4f43cc576 100644 --- a/docs/user_guide/encoding/WoEEncoder.rst +++ b/docs/user_guide/encoding/WoEEncoder.rst @@ -112,8 +112,10 @@ This occurs when a category shows only 1 of the possible values of the target (e always takes 1 or 0). In practice, this happens mostly when a category has a low frequency in the dataset, that is, when only very few observations show that category. -To overcome this limitation, consider using a variable transformation method to group -those categories together, for example by using feature-engine's :class:`RareLabelEncoder()`. +A common way to obtain a WoE for these categories is to replace the zero count by 0.5, +which is what :class:`WoEEncoder()` does. Still, WoE values calculated from very few +observations are unreliable, so consider grouping infrequent categories first, for +example with feature-engine's :class:`RareLabelEncoder()`. Taking into account the above considerations, conducting a detailed exploratory data analysis (EDA) is essential as part of the data science and model-building process. @@ -162,9 +164,46 @@ with feature-engine's imputers. :class:`WoEEncoder()` will ignore unseen categories by default, in which case, they will be replaced by np.nan after the encoding. You have the option to make the encoder raise -an error instead, by setting `unseen='raise'`. You can also replace unseen categories -by an arbitrary value you need to define in `fill_value`, although we do not recommend -this option because it may lead to unpredictable results. +an error instead, by setting `unseen='raise'`. + +Categories with no positive or no negative cases +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +.. attention:: + + **New in version 2.0:** :class:`WoEEncoder()` used to raise an error when a category + had no positive or no negative cases, unless you set the parameter `fill_value`. + `fill_value` was removed. The encoder now replaces zero counts by 0.5 and lists the + affected variables in the attribute `variables_with_zero_counts_`. + +When a category has no positive or no negative cases in the training set, +:class:`WoEEncoder()` replaces the zero count by 0.5 to calculate the WoE, and stores the +names of the affected variables in `variables_with_zero_counts_`. In the following +example, the category red has only positive cases: + +.. code:: python + + import pandas as pd + from feature_engine.encoding import WoEEncoder + + X = pd.DataFrame( + {"colour": ["blue", "blue", "blue", "red", "red", "green", "green", "green"]} + ) + y = pd.Series([1, 0, 1, 1, 1, 0, 1, 0]) + + woe = WoEEncoder() + woe.fit(X, y) + + print(woe.encoder_dict_) + print(woe.variables_with_zero_counts_) + +There are 5 positive and 3 negative cases. Red has 2 positive cases and no negative +cases, so its WoE is log((2 / 5) / (0.5 / 3)) = 0.88: + +.. code:: python + + {'colour': {'blue': 0.1823215567939548, 'green': -1.203972804325936, 'red': 0.8754687373539001}} + ['colour'] Python example -------------- @@ -280,6 +319,41 @@ variable values: 686 -0.584173 female 22.000000 0 0 7.7250 -0.357528 0.012075 +With polars +~~~~~~~~~~~ + +:class:`WoEEncoder()` also works with polars dataframes: + +.. code:: python + + import polars as pl + from feature_engine.encoding import WoEEncoder + + X = pl.DataFrame(dict(x1 = [1,2,3,4,5], x2 = ["b", "b", "b", "a", "a"])) + y = pl.Series([0,1,1,1,0]) + + woe = WoEEncoder() + woe.fit(X, y) + woe.transform(X) + +We see the resulting dataframe below: + +.. code:: text + + shape: (5, 2) + ┌─────┬───────────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════╪═══════════╡ + │ 1 ┆ 0.287682 │ + │ 2 ┆ 0.287682 │ + │ 3 ┆ 0.287682 │ + │ 4 ┆ -0.405465 │ + │ 5 ┆ -0.405465 │ + └─────┴───────────┘ + + WoE in categorical and numerical variables ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ diff --git a/docs/user_guide/imputation/ArbitraryImputer.rst b/docs/user_guide/imputation/ArbitraryImputer.rst index 19a10f9f1..4a0084af8 100644 --- a/docs/user_guide/imputation/ArbitraryImputer.rst +++ b/docs/user_guide/imputation/ArbitraryImputer.rst @@ -134,6 +134,45 @@ imputation (in red the imputed variable): .. image:: ../../images/arbitraryvalueimputation.png +With polars +----------- + +:class:`ArbitraryImputer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.imputation import ArbitraryImputer + + df = pl.DataFrame({ + "LotFrontage": [65.0, None, 68.0, None, 84.0], + "MasVnrArea": [196.0, None, 162.0, None, 350.0], + }) + + transformer = ArbitraryImputer( + arbitrary_number=-999, + variables=["LotFrontage", "MasVnrArea"], + ) + + print(transformer.fit_transform(df)) + +The resulting values match those found with pandas: + +.. code:: text + + shape: (5, 2) + ┌─────────────┬────────────┐ + │ LotFrontage ┆ MasVnrArea │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞═════════════╪════════════╡ + │ 65.0 ┆ 196.0 │ + │ -999.0 ┆ -999.0 │ + │ 68.0 ┆ 162.0 │ + │ -999.0 ┆ -999.0 │ + │ 84.0 ┆ 350.0 │ + └─────────────┴────────────┘ + Additional resources -------------------- diff --git a/docs/user_guide/imputation/CategoricalImputer.rst b/docs/user_guide/imputation/CategoricalImputer.rst index 2382d47ad..427dee929 100644 --- a/docs/user_guide/imputation/CategoricalImputer.rst +++ b/docs/user_guide/imputation/CategoricalImputer.rst @@ -255,8 +255,9 @@ Categorical features with 2 modes ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ It is possible that one variable has more than one mode. In that case, the -transformer will raise an error. For example, when you set the transformer to -impute the variable ‘PoolQC` with the most frequent value: +transformer imputes with the first one, taking the modes in sorted order. For +example, when you set the transformer to impute the variable `'PoolQC'` with +the most frequent value: .. code:: python @@ -267,32 +268,117 @@ impute the variable ‘PoolQC` with the most frequent value: imputer.fit(X_train) -'PoolQC` has more than 1 mode, so the transformer raises the following error: +`'PoolQC'` has more than 1 mode: .. code:: python - 196 self.imputer_dict_ = {var: mode_vals[0]} - 198 # imputing multiple variables: - 199 else: - 200 # Returns a dataframe with 1 row if there is one mode per - 201 # variable, or more rows if there are more modes: + X_train['PoolQC'].mode() - ValueError: The variable PoolQC contains multiple frequent categories. +We see that this variable has 3 categories with a similar maximum number of +observations: -We can check that the variable has various modes like this: +.. code:: python + + 0 Ex + 1 Fa + 2 Gd + Name: PoolQC, dtype: str + +so the transformer picks the first one, `'Ex'`: .. code:: python - X_train['PoolQC'].mode() + imputer.imputer_dict_ -We see that this variable has 3 categories with similar maximum number of observations: +.. code:: python + + {'PoolQC': 'Ex'} + +The pick is deterministic and is the same for pandas and polars. + +With polars +----------- + +:class:`CategoricalImputer()` works in the same way with a polars dataframe: .. code:: python - 0 Ex - 1 Fa - 2 Gd - Name: PoolQC, dtype: object + import polars as pl + from feature_engine.imputation import CategoricalImputer + + df = pl.DataFrame({ + "City": ["London", "Manchester", None, "Bristol", "London", None], + "Studies": ["Bachelor", None, "Bachelor", "PhD", None, "Masters"], + }) + + imputer = CategoricalImputer(imputation_method="frequent") + print(imputer.fit_transform(df)) + +The most frequent category imputation gives the same result as with pandas: + +.. code:: text + + shape: (6, 2) + ┌────────────┬──────────┐ + │ City ┆ Studies │ + │ --- ┆ --- │ + │ str ┆ str │ + ╞════════════╪══════════╡ + │ London ┆ Bachelor │ + │ Manchester ┆ Bachelor │ + │ London ┆ Bachelor │ + │ Bristol ┆ PhD │ + │ London ┆ Bachelor │ + │ London ┆ Masters │ + └────────────┴──────────┘ + +Imputing with an arbitrary string also works the same way: + +.. code:: python + + imputer = CategoricalImputer(fill_value="Missing") + print(imputer.fit_transform(df)) + +.. code:: text + + shape: (6, 2) + ┌────────────┬──────────┐ + │ City ┆ Studies │ + │ --- ┆ --- │ + │ str ┆ str │ + ╞════════════╪══════════╡ + │ London ┆ Bachelor │ + │ Manchester ┆ Missing │ + │ Missing ┆ Bachelor │ + │ Bristol ┆ PhD │ + │ London ┆ Missing │ + │ Missing ┆ Masters │ + └────────────┴──────────┘ + +.. note:: + + polars' ``Categorical`` dtype accepts a brand-new fill value automatically, + unlike pandas' ``category`` dtype, which needs its categories widened first + (:class:`CategoricalImputer()` handles that difference for you on both + backends). polars' ``Enum`` dtype, however, has a *fixed* set of categories + that cannot be widened. If you impute a fixed-category ``Enum`` column with + a `fill_value` that isn't already one of its categories, the transformer + raises a clear error instead of silently writing null: + + .. code:: python + + enum_dtype = pl.Enum(["London", "Manchester", "Bristol"]) + df_enum = df.with_columns(pl.col("City").cast(enum_dtype)) + + imputer = CategoricalImputer(fill_value="Missing", variables=["City"]) + imputer.fit_transform(df_enum) + + .. code:: text + + ValueError: Cannot fill variable 'City' with 'Missing': it is a polars + Enum with fixed categories ('London', 'Manchester', 'Bristol') that do + not include the fill value. Cast the column to Categorical or String + before imputing. Considerations -------------- diff --git a/docs/user_guide/imputation/DropMissingData.rst b/docs/user_guide/imputation/DropMissingData.rst index dcae91e21..8b805d7c6 100644 --- a/docs/user_guide/imputation/DropMissingData.rst +++ b/docs/user_guide/imputation/DropMissingData.rst @@ -550,6 +550,60 @@ In the following output we see the predictions made by the pipeline: array([2., 2.]) +With polars +^^^^^^^^^^^ + +:class:`DropMissingData()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.imputation import DropMissingData + + X = pl.DataFrame( + { + "x1": [2, 1, 1, 0, None], + "x2": ["a", None, "b", None, "a"], + "x3": [2, 3, 4, 5, 5], + } + ) + + dmd = DropMissingData() + dmd.fit_transform(X) + +We get the same complete-case rows as with pandas: + +.. code:: text + + shape: (2, 3) + ┌─────┬─────┬─────┐ + │ x1 ┆ x2 ┆ x3 │ + │ --- ┆ --- ┆ --- │ + │ i64 ┆ str ┆ i64 │ + ╞═════╪═════╪═════╡ + │ 2 ┆ a ┆ 2 │ + │ 1 ┆ b ┆ 4 │ + └─────┴─────┴─────┘ + +``return_na_data()`` and ``threshold`` behave identically on polars too: + +.. code:: python + + dmd.return_na_data(X) + +.. code:: text + + shape: (3, 3) + ┌──────┬──────┬─────┐ + │ x1 ┆ x2 ┆ x3 │ + │ --- ┆ --- ┆ --- │ + │ i64 ┆ str ┆ i64 │ + ╞══════╪══════╪═════╡ + │ 1 ┆ null ┆ 3 │ + │ 0 ┆ null ┆ 5 │ + │ null ┆ a ┆ 5 │ + └──────┴──────┴─────┘ + Dropna or fillna? ^^^^^^^^^^^^^^^^^ diff --git a/docs/user_guide/imputation/EndTailImputer.rst b/docs/user_guide/imputation/EndTailImputer.rst index e908cf518..6725fbfff 100644 --- a/docs/user_guide/imputation/EndTailImputer.rst +++ b/docs/user_guide/imputation/EndTailImputer.rst @@ -119,6 +119,53 @@ imputation (in red the imputed variable): The second peak corresponds to the missing data, which were replaced with a value at that side of the distribution. +With polars +----------- + +:class:`EndTailImputer()` also works with polars dataframes: + +.. code:: python + + import polars as pl + from feature_engine.imputation import EndTailImputer + + X = pl.DataFrame({ + "LotFrontage": [65.0, 80.0, None, 60.0, 84.0, None, 75.0], + "MasVnrArea": [196.0, None, 162.0, 0.0, 350.0, None, 0.0], + }) + + # set up the imputer + tail_imputer = EndTailImputer( + imputation_method='gaussian', + tail='right', + fold=3, + variables=['LotFrontage', 'MasVnrArea'], + ) + + # fit the imputer + tail_imputer.fit(X) + + # transform the data + X_t = tail_imputer.transform(X) + X_t + +.. code:: text + + shape: (7, 2) + ┌─────────────┬────────────┐ + │ LotFrontage ┆ MasVnrArea │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞═════════════╪════════════╡ + │ 65.0 ┆ 196.0 │ + │ 80.0 ┆ 583.800407 │ + │ 103.053925 ┆ 162.0 │ + │ 60.0 ┆ 0.0 │ + │ 84.0 ┆ 350.0 │ + │ 103.053925 ┆ 583.800407 │ + │ 75.0 ┆ 0.0 │ + └─────────────┴────────────┘ + Additional resources -------------------- diff --git a/docs/user_guide/imputation/MeanImputer.rst b/docs/user_guide/imputation/MeanImputer.rst index b0d0df877..66d8a0873 100644 --- a/docs/user_guide/imputation/MeanImputer.rst +++ b/docs/user_guide/imputation/MeanImputer.rst @@ -310,6 +310,55 @@ center of the distribution: Because of the increase in the number of observations at the center, the variance of the variable decreases, and the kurtosis coefficient increases. +With polars +----------- + +:class:`MeanImputer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.imputation import MeanImputer + + df = pl.DataFrame({ + "Age": [20, 21, 19, None, 23, 40, 41, 37], + "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], + }) + + transformer = MeanImputer(imputation_method="mean") + transformer.fit(df) + + print(transformer.imputer_dict_) + +The learned mean values match those found with pandas: + +.. code:: text + + {'Age': 28.714285714285715, 'Marks': 0.6833333333333332} + +.. code:: python + + print(transformer.transform(df)) + +.. code:: text + + shape: (8, 2) + ┌───────────┬──────────┐ + │ Age ┆ Marks │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞═══════════╪══════════╡ + │ 20.0 ┆ 0.9 │ + │ 21.0 ┆ 0.8 │ + │ 19.0 ┆ 0.7 │ + │ 28.714286 ┆ 0.683333 │ + │ 23.0 ┆ 0.3 │ + │ 40.0 ┆ 0.683333 │ + │ 41.0 ┆ 0.8 │ + │ 37.0 ┆ 0.6 │ + └───────────┴──────────┘ + + Additional resources -------------------- diff --git a/docs/user_guide/imputation/RandomSampleImputer.rst b/docs/user_guide/imputation/RandomSampleImputer.rst index b5686272d..70a59332c 100644 --- a/docs/user_guide/imputation/RandomSampleImputer.rst +++ b/docs/user_guide/imputation/RandomSampleImputer.rst @@ -28,12 +28,12 @@ missing data and `seed` is the number you entered in the `random_state`. If `seed = 'observation'`, then the random_state should be a variable name or a list of variable names. The seed will be calculated observation per -observation, either by adding or multiplying the values of the variables -indicated in the `random_state`. Then, a value will be extracted from the train set -using that seed and used to replace the NAN in that particular observation. This is the -equivalent of `pandas.sample(1, random_state=var1+var2)` if the `seeding_method` is -set to `add` or `pandas.sample(1, random_state=var1*var2)` if the `seeding_method` -is set to `multiply`. +observation from the values of the variables indicated in the `random_state`. +Then, a value will be extracted from the train set using that seed and used to +replace the NAN in that particular observation. + +Observations with the same values in the `random_state` variables receive the same +imputation, regardless of their position in the dataframe. For example, if the observation shows variables colour: np.nan, height: 152, weight:52, and we set the imputer as: @@ -43,20 +43,56 @@ and we set the imputer as: RandomSampleImputer( random_state=['height', 'weight'], seed='observation', - seeding_method='add', ) -the np.nan in the variable colour will be replaced using pandas sample as follows: +the np.nan in the variable colour will be replaced with a value extracted from the train +set, using a seed derived from the values 152 and 52. Any other observation with +height 152 and weight 52 will receive the same value. + +.. note:: + + The variables indicated in the `random_state` must be numerical, otherwise the + imputer will return an error. Missing values in those variables are treated as 0. + +With polars +----------- + +:class:`RandomSampleImputer()` also accepts polars dataframes as input to `fit()` and +`transform()`. .. code:: python - observation.sample(1, random_state=int(152+52)) + import polars as pl + from feature_engine.imputation import RandomSampleImputer -.. note:: + X_train = pl.DataFrame({ + "MSSubClass": [60, 20, 60, 20, 50], + "YrSold": [2008, 2007, 2008, 2007, 2009], + "LotFrontage": [65.0, None, 68.0, 60.0, None], + }) - Note, if the variables indicated in the `random_state` list are not numerical - the imputer will return an error. In addition, the variables indicated as seed - should not contain missing values themselves. + imputer = RandomSampleImputer( + variables=["LotFrontage"], + random_state=["MSSubClass", "YrSold"], + seed="observation", + ) + imputer.fit(X_train) + imputer.transform(X_train) + +.. code:: text + + shape: (5, 3) + ┌────────────┬────────┬─────────────┐ + │ MSSubClass ┆ YrSold ┆ LotFrontage │ + │ --- ┆ --- ┆ --- │ + │ i64 ┆ i64 ┆ f64 │ + ╞════════════╪════════╪═════════════╡ + │ 60 ┆ 2008 ┆ 65.0 │ + │ 20 ┆ 2007 ┆ 68.0 │ + │ 60 ┆ 2008 ┆ 68.0 │ + │ 20 ┆ 2007 ┆ 60.0 │ + │ 50 ┆ 2009 ┆ 65.0 │ + └────────────┴────────┴─────────────┘ Important for GDPR ------------------ @@ -105,8 +141,8 @@ First, let's load the data and separate it into train and test: ) In this example, we sample values at random, observation per observation, using as seed -the value of the variable 'MSSubClass' plus the value of the variable 'YrSold'. Note -that the seed's value is different for each observation. +the values of the variables 'MSSubClass' and 'YrSold'. Observations with the same values +in these variables receive the same imputed values. The :class:`RandomSampleImputer()` will impute all variables in the data, as we left the default value of the parameter `variables` to `None`. @@ -117,7 +153,6 @@ default value of the parameter `variables` to `None`. imputer = RandomSampleImputer( random_state=['MSSubClass', 'YrSold'], seed='observation', - seeding_method='add' ) # fit the imputer @@ -162,4 +197,4 @@ For tutorials about missing data imputation methods check out these resources: Both our book and courses are suitable for beginners and more advanced data scientists alike. By purchasing them you are supporting `Sole `_, -the main developer of feature-engine. \ No newline at end of file +the main developer of feature-engine. diff --git a/docs/user_guide/outliers/ArbitraryOutlierCapper.rst b/docs/user_guide/outliers/ArbitraryOutlierCapper.rst index 70153b25b..ce916e9c4 100644 --- a/docs/user_guide/outliers/ArbitraryOutlierCapper.rst +++ b/docs/user_guide/outliers/ArbitraryOutlierCapper.rst @@ -96,6 +96,44 @@ values: dtype: float64 +With polars +----------- + +:class:`ArbitraryOutlierCapper()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.outliers import ArbitraryOutlierCapper + + df = pl.DataFrame({ + "age": [20.0, 21.0, 19.0, 45.0, 67.0, 18.0, 90.0, 34.0, 55.0, 23.0], + "fare": [7.5, 8.0, 71.3, 13.0, 30.5, 7.9, 512.3, 26.0, 15.5, 8.6], + }) + + capper = ArbitraryOutlierCapper( + max_capping_dict={"age": 50, "fare": 200}, + min_capping_dict=None, + ) + + capper.fit(df) + Xt = capper.transform(df) + + print(Xt.select(["age", "fare"]).max()) + +The resulting maximum values, capped at the values we entered in the dictionary: + +.. code:: text + + shape: (1, 2) + ┌──────┬───────┐ + │ age ┆ fare │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞══════╪═══════╡ + │ 50.0 ┆ 200.0 │ + └──────┴───────┘ + Additional resources -------------------- diff --git a/docs/user_guide/outliers/OutlierTrimmer.rst b/docs/user_guide/outliers/OutlierTrimmer.rst index 929824d45..e7ed0f0ca 100644 --- a/docs/user_guide/outliers/OutlierTrimmer.rst +++ b/docs/user_guide/outliers/OutlierTrimmer.rst @@ -125,6 +125,13 @@ and percentile methods stay closer to where the observations actually lie: they are true outliers or faithful data points. That requires further examination and domain knowledge. +.. note:: + + If all or most of the values of a variable are the same, the method may return a + spread of 0 (for example, an IQR of 0 when over half of the values are 0). The + variable then has no outliers, so :class:`OutlierTrimmer()` sets its limits to infinity + in `right_tail_caps_` and `left_tail_caps_`, and leaves it untouched. + Let’s move on to removing outliers in Python. Removing outliers in Python @@ -293,7 +300,7 @@ In the following output, we see the maximum of the variables after removing the .. code:: python fare 65.0 - age 53.0 + age 74.0 dtype: float64 Finally, we can check the boxplot of the transformed variables to corroborate the effect on their distribution. @@ -521,7 +528,7 @@ We see the adjusted data size compared to the original size here: .. code:: python - ((916, 8), (736, 76)) + ((916, 8), (828, 142)) Feature-engine's pipeline can also adjust the target: @@ -535,7 +542,7 @@ We see the adjusted data size compared to the original size here: .. code:: python - ((916,), (736,)) + ((916,), (828,)) To wrap up, let's add a machine learning algorithm to the pipeline. We'll use logistic regression to predict survival: @@ -565,7 +572,7 @@ We see the following output: .. code:: python - array([1, 1, 1, 0, 1, 0, 1, 1, 0, 1], dtype=int64) + array([1, 1, 0, 1, 0, 1, 0, 0, 1, 0]) We can obtain the probability of survival: @@ -580,16 +587,16 @@ We see the following output: .. code:: python - array([[0.13027536, 0.86972464], - [0.14982143, 0.85017857], - [0.2783799 , 0.7216201 ], - [0.86907159, 0.13092841], - [0.31794531, 0.68205469], - [0.86905145, 0.13094855], - [0.1396715 , 0.8603285 ], - [0.48403632, 0.51596368], - [0.6299007 , 0.3700993 ], - [0.49712853, 0.50287147]]) + array([[0.23320943, 0.76679057], + [0.22089305, 0.77910695], + [0.85469885, 0.14530115], + [0.28510312, 0.71489688], + [0.85468117, 0.14531883], + [0.0494853 , 0.9505147 ], + [0.58079146, 0.41920854], + [0.536129 , 0.463871 ], + [0.36885157, 0.63114843], + [0.81102131, 0.18897869]]) We can obtain the accuracy of the predictions over the test set: @@ -601,7 +608,7 @@ That returns the following accuracy: .. code:: python - 0.7823343848580442 + 0.804093567251462 We can obtain the names of the features after the transformation: @@ -635,7 +642,7 @@ We see the resulting sizes here: .. code:: python - ((393, 8), (317, 76)) + ((393, 8), (342, 142)) Setting up the stringency (param `fold`) @@ -656,6 +663,49 @@ The default values for fold are as follows: You can manually adjust the fold value to make the outlier detection process more or less conservative, thus customising the extent of outlier trimming. +With polars +----------- + +:class:`OutlierTrimmer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.outliers import OutlierTrimmer + + df = pl.DataFrame({ + "Age": [20, 21, 19, 18, 95], + "Marks": [0.9, 0.8, 0.7, 0.6, 0.1], + }) + + transformer = OutlierTrimmer( + capping_method="quantiles", + tail="both", + fold=0.2, + ) + + print(transformer.fit_transform(df)) + +Only the rows where both `Age` and `Marks` fall within the 20th-80th +percentile range survive; the other three rows breach the bound on at +least one of the two variables: + +.. code:: text + + shape: (2, 2) + ┌─────┬───────┐ + │ Age ┆ Marks │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════╪═══════╡ + │ 21 ┆ 0.8 │ + │ 19 ┆ 0.7 │ + └─────┴───────┘ + +`transform_x_y()` and `get_feature_names_out()` work identically to the +pandas examples above. + + Additional resources -------------------- diff --git a/docs/user_guide/outliers/Winsoriser.rst b/docs/user_guide/outliers/Winsoriser.rst index 31babf233..a374f8e07 100644 --- a/docs/user_guide/outliers/Winsoriser.rst +++ b/docs/user_guide/outliers/Winsoriser.rst @@ -67,6 +67,13 @@ Percentiles or quantiles The values used by default by :class:`Winsoriser()` are those suggested as optimal in statistical studies. +.. note:: + + If all or most of the values of a variable are the same, the method may return a + spread of 0 (for example, an IQR of 0 when over half of the values are 0). The + variable then has no outliers, so :class:`Winsoriser()` sets its limits to infinity + in `right_tail_caps_` and `left_tail_caps_`, and leaves it untouched. + The following image shows the four methods applied to a normal distribution. Their capping values are close together because, when the data is roughly symmetric and bell-shaped, the mean, median, standard deviation, IQR, and MAD all describe the same thing. @@ -354,6 +361,62 @@ The default values for fold are as follows: You can manually adjust the `fold` value to make the outlier detection process more or less conservative, thus customising the extent of outlier capping. +With polars +----------- + +:class:`Winsoriser()` works in the same way with a polars dataframe, including the +`add_indicators` option, which flags the rows that were capped on each tail: + +.. code:: python + + import polars as pl + from feature_engine.outliers import Winsoriser + + df = pl.DataFrame({ + "Age": [20, 21, 19, 18, 23, 40, 41, 97], + "Marks": [0.9, 0.8, 0.7, 0.6, 0.3, 0.5, 0.8, 0.05], + }) + + transformer = Winsoriser( + capping_method="iqr", tail="both", fold=1.5, add_indicators=True, + ) + transformer.fit(df) + + print(transformer.right_tail_caps_) + print(transformer.left_tail_caps_) + +The learned capping values match those found with pandas: + +.. code:: text + + {'Age': 71.0, 'Marks': 1.3250000000000002} + {'Age': -11.0, 'Marks': -0.07500000000000001} + +.. code:: python + + print(transformer.transform(df)) + +`Age`'s outlier, 97, was capped to 71 and flagged in `Age_right`; none of the values +in `Marks` were extreme enough to be capped: + +.. code:: text + + shape: (8, 6) + ┌──────┬───────┬──────────┬───────────┬────────────┬─────────────┐ + │ Age ┆ Marks ┆ Age_left ┆ Age_right ┆ Marks_left ┆ Marks_right │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞══════╪═══════╪══════════╪═══════════╪════════════╪═════════════╡ + │ 20.0 ┆ 0.9 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ + │ 21.0 ┆ 0.8 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ + │ 19.0 ┆ 0.7 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ + │ 18.0 ┆ 0.6 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ + │ 23.0 ┆ 0.3 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ + │ 40.0 ┆ 0.5 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ + │ 41.0 ┆ 0.8 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ + │ 71.0 ┆ 0.05 ┆ 0.0 ┆ 1.0 ┆ 0.0 ┆ 0.0 │ + └──────┴───────┴──────────┴───────────┴────────────┴─────────────┘ + Additional resources -------------------- diff --git a/docs/user_guide/preprocessing/MatchCategories.rst b/docs/user_guide/preprocessing/MatchCategories.rst index ed5c46fc2..730934170 100644 --- a/docs/user_guide/preprocessing/MatchCategories.rst +++ b/docs/user_guide/preprocessing/MatchCategories.rst @@ -6,7 +6,8 @@ MatchCategories =============== :class:`MatchCategories()` ensures that categorical variables are encoded as pandas -'categorical' dtype instead of generic python 'object' or other dtypes. +'categorical' dtype, or polars 'Enum' dtype, instead of generic python 'object', +string or other dtypes. Under the hood, 'categorical' dtype is a representation that maps each category to an integer, thus providing a more memory-efficient object @@ -79,10 +80,10 @@ Here are the mappings learnt for each categorical variable: .. code:: python - {'pclass': Int64Index([1, 2, 3], dtype='int64'), - 'sex': Index(['female', 'male'], dtype='object'), - 'cabin': Index(['A', 'B', 'C', 'D', 'E', 'F', 'M', 'T'], dtype='object'), - 'embarked': Index(['C', 'Missing', 'Q', 'S'], dtype='object')} + {'pclass': Index([1, 2, 3], dtype='int64'), + 'sex': Index(['female', 'male'], dtype='str'), + 'cabin': Index(['A', 'B', 'C', 'D', 'E', 'F', 'M', 'T'], dtype='str'), + 'embarked': Index(['C', 'Missing', 'Q', 'S'], dtype='str')} To see why this matters, let's compare the order in which the categories of `embarked` appear in the raw train and test sets. This is the order in the train set: @@ -95,7 +96,9 @@ We obtain the following order: .. code:: python - array(['S', 'C', 'Missing', 'Q'], dtype=object) + + ['S', 'C', 'Missing', 'Q'] + Length: 4, dtype: str And this is the order in the test set: @@ -107,7 +110,9 @@ Which is different from the train set: .. code:: python - array(['Q', 'S', 'C'], dtype=object) + + ['Q', 'S', 'C'] + Length: 3, dtype: str The categories appear in a different order in each set. If we transform the dataframes using the same `match_categories` object, categorical variables will be converted to a @@ -122,7 +127,7 @@ Which is: .. code:: python - Index(['C', 'Missing', 'Q', 'S'], dtype='object') + Index(['C', 'Missing', 'Q', 'S'], dtype='str') And this is the order we now obtain for the test set: @@ -134,11 +139,10 @@ The 2 sets now show exactly the same category order: .. code:: python - Index(['C', 'Missing', 'Q', 'S'], dtype='object') + Index(['C', 'Missing', 'Q', 'S'], dtype='str') If some category was not present in the training data, it will not be mapped -to any integer and will thus not get encoded. This behaviour can be modified through the -parameter `errors`. Let's illustrate this with the `cabin` variable. These are the +to any integer and will become a missing value instead. Let's illustrate this with the `cabin` variable. These are the categories present in the train set: .. code:: python @@ -149,7 +153,9 @@ We obtain the following categories: .. code:: python - array(['B', 'C', 'E', 'D', 'A', 'M', 'T', 'F'], dtype=object) + + ['B', 'C', 'E', 'D', 'A', 'M', 'T', 'F'] + Length: 8, dtype: str And these are the categories present in the test set, which include a category, 'G', that was not seen during training: @@ -162,7 +168,9 @@ We obtain the following categories, including the unseen 'G': .. code:: python - array(['M', 'F', 'E', 'G'], dtype=object) + + ['M', 'F', 'E', 'G'] + Length: 4, dtype: str After transforming the train set, we obtain the same categories as before, now correctly typed as 'category' dtype: @@ -176,7 +184,7 @@ Which are: .. code:: python ['B', 'C', 'E', 'D', 'A', 'M', 'T', 'F'] - Categories (8, object): ['A', 'B', 'C', 'D', 'E', 'F', 'M', 'T'] + Categories (8, str): ['A', 'B', 'C', 'D', 'E', 'F', 'M', 'T'] But when we transform the test set, the unseen category 'G' is not mapped to any integer, and becomes a missing value instead: @@ -190,7 +198,130 @@ We see that 'G' has been replaced by a missing value: .. code:: python ['M', 'F', 'E', NaN] - Categories (8, object): ['A', 'B', 'C', 'D', 'E', 'F', 'M', 'T'] + Categories (8, str): ['A', 'B', 'C', 'D', 'E', 'F', 'M', 'T'] + +Because we set `missing_values="ignore"`, :class:`MatchCategories()` warns us that +missing values were introduced: + +.. code:: python + + UserWarning: During the encoding, NaN values were introduced in the feature(s) cabin. + +With the default `missing_values="raise"`, :class:`MatchCategories()` raises an error +instead, both when the data contains missing values and when unseen categories would +introduce them. + +With polars +^^^^^^^^^^^ + +:class:`MatchCategories()` also works with polars dataframes. In polars, the variables +are cast to the 'Enum' dtype, which, like pandas 'categorical', holds a fixed list of +categories. Let's create a toy train set and test set: + +.. code:: python + + import polars as pl + from feature_engine.preprocessing import MatchCategories + + train = pl.DataFrame({ + "city": ["London", "Paris", "Madrid", "Paris"], + "rooms": [2, 3, 1, 3], + }) + test = pl.DataFrame({ + "city": ["Madrid", "Rome", "London", "Paris"], + "rooms": [1, 2, 4, 3], + }) + +We fit :class:`MatchCategories()` to the train set: + +.. code:: python + + match_categories = MatchCategories(missing_values="ignore") + match_categories.fit(train) + + match_categories.category_dict_ + +With polars, the categories are stored in lists: + +.. code:: python + + {'city': ['London', 'Madrid', 'Paris']} + +Now we transform the test set: + +.. code:: python + + test_t = match_categories.transform(test) + test_t + +The variable `city` is now an 'Enum', and the unseen category 'Rome' became a missing +value: + +.. code:: text + + shape: (4, 2) + ┌────────┬───────┐ + │ city ┆ rooms │ + │ --- ┆ --- │ + │ enum ┆ i64 │ + ╞════════╪═══════╡ + │ Madrid ┆ 1 │ + │ null ┆ 2 │ + │ London ┆ 4 │ + │ Paris ┆ 3 │ + └────────┴───────┘ + +We can check the categories in the schema: + +.. code:: python + + test_t.schema + +The categories are the ones learned from the train set: + +.. code:: python + + Schema({'city': Enum(categories=['London', 'Madrid', 'Paris']), 'rooms': Int64}) + +The polars 'Enum' dtype only takes strings. Hence, if we cast numerical variables by +setting `ignore_format=True`, their values become strings. The categories are sorted +in numerical order: + +.. code:: python + + match_categories = MatchCategories( + variables=["rooms"], ignore_format=True, missing_values="ignore" + ) + match_categories.fit(train) + + match_categories.category_dict_ + +We see the categories of `rooms` as strings: + +.. code:: python + + {'rooms': ['1', '2', '3']} + +And these are the values after the transformation, where the unseen value 4 became a +missing value: + +.. code:: python + + match_categories.transform(test) + +.. code:: text + + shape: (4, 2) + ┌────────┬───────┐ + │ city ┆ rooms │ + │ --- ┆ --- │ + │ str ┆ enum │ + ╞════════╪═══════╡ + │ Madrid ┆ 1 │ + │ Rome ┆ 2 │ + │ London ┆ null │ + │ Paris ┆ 3 │ + └────────┴───────┘ When to use the transformer diff --git a/docs/user_guide/preprocessing/MatchVariables.rst b/docs/user_guide/preprocessing/MatchVariables.rst index 325aaafa0..2203aacc9 100644 --- a/docs/user_guide/preprocessing/MatchVariables.rst +++ b/docs/user_guide/preprocessing/MatchVariables.rst @@ -86,11 +86,11 @@ We see that `sex` and `age` are no longer in the dataframe: .. code:: python pclass survived sibsp parch fare cabin embarked - 1000 3 1 0 0 7.7500 n Q - 1001 3 1 2 0 23.2500 n Q - 1002 3 1 2 0 23.2500 n Q - 1003 3 1 2 0 23.2500 n Q - 1004 3 1 0 0 7.7875 n Q + 1000 3 1 0 0 7.7500 NaN Q + 1001 3 1 2 0 23.2500 NaN Q + 1002 3 1 2 0 23.2500 NaN Q + 1003 3 1 2 0 23.2500 NaN Q + 1004 3 1 0 0 7.7875 NaN Q If we transform the dataframe with the dropped columns using :class:`MatchVariables()`, we see that the new dataframe contains all the variables, and those that were missing @@ -107,13 +107,13 @@ Indeed, `sex` and `age` are back, filled with missing values: .. code:: python - The following variables are added to the DataFrame: ['age', 'sex'] + The following variables are added to the DataFrame: ['sex', 'age'] pclass survived sex age sibsp parch fare cabin embarked - 1000 3 1 NaN NaN 0 0 7.7500 n Q - 1001 3 1 NaN NaN 2 0 23.2500 n Q - 1002 3 1 NaN NaN 2 0 23.2500 n Q - 1003 3 1 NaN NaN 2 0 23.2500 n Q - 1004 3 1 NaN NaN 0 0 7.7875 n Q + 1000 3 1 NaN NaN 0 0 7.7500 NaN Q + 1001 3 1 NaN NaN 2 0 23.2500 NaN Q + 1002 3 1 NaN NaN 2 0 23.2500 NaN Q + 1003 3 1 NaN NaN 2 0 23.2500 NaN Q + 1004 3 1 NaN NaN 0 0 7.7875 NaN Q Note how the missing columns were added back to the transformed test set, with missing values, in the position (i.e., order) in which they were in the train set. @@ -133,11 +133,11 @@ We now have 2 extra columns, `var_a` and `var_b`, that were not present in the t .. code:: python pclass survived sibsp parch fare cabin embarked var_a var_b - 1000 3 1 0 0 7.7500 n Q 0 0 - 1001 3 1 2 0 23.2500 n Q 0 0 - 1002 3 1 2 0 23.2500 n Q 0 0 - 1003 3 1 2 0 23.2500 n Q 0 0 - 1004 3 1 0 0 7.7875 n Q 0 0 + 1000 3 1 0 0 7.7500 NaN Q 0 0 + 1001 3 1 2 0 23.2500 NaN Q 0 0 + 1002 3 1 2 0 23.2500 NaN Q 0 0 + 1003 3 1 2 0 23.2500 NaN Q 0 0 + 1004 3 1 0 0 7.7875 NaN Q 0 0 And now, we transform the data with :class:`MatchVariables()`: @@ -152,14 +152,14 @@ the additional columns from the resulting dataset: .. code:: python - The following variables are added to the DataFrame: ['age', 'sex'] - The following variables are dropped from the DataFrame: ['var_b', 'var_a'] + The following variables are added to the DataFrame: ['sex', 'age'] + The following variables are dropped from the DataFrame: ['var_a', 'var_b'] pclass survived sex age sibsp parch fare cabin embarked - 1000 3 1 NaN NaN 0 0 7.7500 n Q - 1001 3 1 NaN NaN 2 0 23.2500 n Q - 1002 3 1 NaN NaN 2 0 23.2500 n Q - 1003 3 1 NaN NaN 2 0 23.2500 n Q - 1004 3 1 NaN NaN 0 0 7.7875 n Q + 1000 3 1 NaN NaN 0 0 7.7500 NaN Q + 1001 3 1 NaN NaN 2 0 23.2500 NaN Q + 1002 3 1 NaN NaN 2 0 23.2500 NaN Q + 1003 3 1 NaN NaN 2 0 23.2500 NaN Q + 1004 3 1 NaN NaN 0 0 7.7875 NaN Q However, if we look closely, the dtypes for the `sex` variable do not match. This could cause issues if other transformations depend upon having the correct dtypes. This is the @@ -173,7 +173,7 @@ Which is: .. code:: python - dtype('O') + And this is the dtype in the transformed test set: @@ -201,8 +201,8 @@ We see in the messages that the `sex` dtype was changed to match that of the tra .. code:: python The following variables are added to the DataFrame: ['sex', 'age'] - The following variables are dropped from the DataFrame: ['var_b', 'var_a'] - The sex dtype is changing from float64 to object + The following variables are dropped from the DataFrame: ['var_a', 'var_b'] + The sex dtype is changing from float64 to str Now the dtype matches: @@ -214,11 +214,61 @@ Which is: .. code:: python - dtype('O') + By default, :class:`MatchVariables()` will print out messages indicating which variables were added, removed and altered. We can switch off the messages through the parameter `verbose`. +Working with polars +^^^^^^^^^^^^^^^^^^^ + +:class:`MatchVariables()` also works with polars dataframes, and returns a polars +dataframe. With polars, the variables added with the default `fill_value` contain +nulls, which is how polars represents missing data: + +.. code:: python + + import polars as pl + from feature_engine.preprocessing import MatchVariables + + train = pl.DataFrame({ + "pclass": [1, 1, 3, 2], + "sex": ["female", "male", "male", "female"], + "age": [29.0, 2.0, 30.0, 25.0], + "fare": [211.34, 151.55, 7.75, 26.0], + }) + + test = pl.DataFrame({ + "fare": [7.75, 23.25, 7.78], + "pclass": [3, 3, 3], + "var_a": [0, 0, 0], + }) + + match_cols = MatchVariables(match_dtypes=True) + match_cols.fit(train) + + test_t = match_cols.transform(test) + print(test_t) + +The transformer added `sex` and `age`, removed `var_a`, sorted the variables as in the +train set and cast `sex` to the string dtype that it had in the train set: + +.. code:: text + + The following variables are added to the DataFrame: ['sex', 'age'] + The following variables are dropped from the DataFrame: ['var_a'] + The sex dtype is changing from Float64 to String + shape: (3, 4) + ┌────────┬──────┬──────┬───────┐ + │ pclass ┆ sex ┆ age ┆ fare │ + │ --- ┆ --- ┆ --- ┆ --- │ + │ i64 ┆ str ┆ f64 ┆ f64 │ + ╞════════╪══════╪══════╪═══════╡ + │ 3 ┆ null ┆ null ┆ 7.75 │ + │ 3 ┆ null ┆ null ┆ 23.25 │ + │ 3 ┆ null ┆ null ┆ 7.78 │ + └────────┴──────┴──────┴───────┘ + When to use the transformer ^^^^^^^^^^^^^^^^^^^^^^^^^^^ diff --git a/docs/user_guide/scaling/MeanNormalisationScaler.rst b/docs/user_guide/scaling/MeanNormalisationScaler.rst index 30952f06c..2255d8d2d 100644 --- a/docs/user_guide/scaling/MeanNormalisationScaler.rst +++ b/docs/user_guide/scaling/MeanNormalisationScaler.rst @@ -137,11 +137,56 @@ In the following data, we see the scaled variables returned to their original re .. code:: python - Name City Age Height Marks dob - 0 tom London 20 1.80 0.9 2020-02-24 00:00:00 - 1 nick Manchester 21 1.77 0.8 2020-02-24 00:01:00 - 2 krish Liverpool 19 1.90 0.7 2020-02-24 00:02:00 - 3 jack Bristol 18 2.00 0.6 2020-02-24 00:03:00 + Name City Age Height Marks dob + 0 tom London 20.0 1.80 0.9 2020-02-24 00:00:00 + 1 nick Manchester 21.0 1.77 0.8 2020-02-24 00:01:00 + 2 krish Liverpool 19.0 1.90 0.7 2020-02-24 00:02:00 + 3 jack Bristol 18.0 2.00 0.6 2020-02-24 00:03:00 + +Note that **Age** comes back as a float, not the original integer: multiplying and +adding floats (the range and mean) always produces a float in both pandas and +polars, so the inverse transformation cannot restore the original integer dtype. + +With polars +----------- + +:class:`MeanNormalisationScaler()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.scaling import MeanNormalisationScaler + + df = pl.DataFrame( + { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Height": [1.80, 1.77, 1.90, 2.00], + "Marks": [0.9, 0.8, 0.7, 0.6], + } + ) + + scaler = MeanNormalisationScaler(variables=["Age", "Marks", "Height"]) + scaler.fit(df) + + print(scaler.transform(df)) + +The resulting values match those found with pandas: + +.. code:: text + + shape: (4, 5) + ┌───────┬────────────┬───────────┬───────────┬───────────┐ + │ Name ┆ City ┆ Age ┆ Height ┆ Marks │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ str ┆ str ┆ f64 ┆ f64 ┆ f64 │ + ╞═══════╪════════════╪═══════════╪═══════════╪═══════════╡ + │ tom ┆ London ┆ 0.166667 ┆ -0.293478 ┆ 0.5 │ + │ nick ┆ Manchester ┆ 0.5 ┆ -0.423913 ┆ 0.166667 │ + │ krish ┆ Liverpool ┆ -0.166667 ┆ 0.141304 ┆ -0.166667 │ + │ jack ┆ Bristol ┆ -0.5 ┆ 0.576087 ┆ -0.5 │ + └───────┴────────────┴───────────┴───────────┴───────────┘ Additional resources diff --git a/docs/user_guide/text/TextFeatures.rst b/docs/user_guide/text/TextFeatures.rst index cf31dd0b9..1ad66bd4b 100644 --- a/docs/user_guide/text/TextFeatures.rst +++ b/docs/user_guide/text/TextFeatures.rst @@ -38,27 +38,30 @@ Text features :class:`TextFeatures()` can extract the following features from a text piece: -- **char_count**: Number of characters in the text +- **char_count**: Number of characters, excluding whitespace - **word_count**: Number of words (whitespace-separated tokens) - **sentence_count**: Number of sentences (based on .!? punctuation) -- **avg_word_length**: Average length of words +- **avg_word_length**: Average number of characters per word - **digit_count**: Number of digit characters -- **letter_count**: Number of alphabetic characters (a-z, A-Z) -- **uppercase_count**: Number of uppercase letters -- **lowercase_count**: Number of lowercase letters -- **special_char_count**: Number of special characters (non-alphanumeric) +- **letter_count**: Number of letters a-z and A-Z +- **uppercase_count**: Number of uppercase letters A-Z +- **lowercase_count**: Number of lowercase letters a-z +- **special_char_count**: Number of characters that are not a-z, A-Z, 0-9 or whitespace - **whitespace_count**: Number of whitespace characters - **whitespace_ratio**: Ratio of whitespace to total characters -- **digit_ratio**: Ratio of digits to total characters -- **uppercase_ratio**: Ratio of uppercase to total characters +- **digit_ratio**: Ratio of digits to non-whitespace characters +- **uppercase_ratio**: Ratio of uppercase letters to non-whitespace characters - **has_digits**: Binary indicator if text contains digits -- **has_uppercase**: Binary indicator if text contains uppercase +- **has_uppercase**: Binary indicator if text contains uppercase letters A-Z - **is_empty**: Binary indicator if text is empty -- **starts_with_uppercase**: Binary indicator if text starts with uppercase +- **starts_with_uppercase**: Binary indicator if text starts with A-Z - **ends_with_punctuation**: Binary indicator if text ends with .!? - **unique_word_count**: Number of unique words (case-insensitive) - **lexical_diversity**: Ratio of unique words to total words +Letters with accents or from other alphabets, like é or ß, are not counted as letters +or uppercase letters; they are counted as special characters. + The **number of sentences** is inferred by :class:`TextFeatures()` by counting blocks of sentence-ending punctuation (., !, ?) as a proxy for sentence boundaries. This means that multiple consecutive punctuation marks (e.g., "!!!" or "??") are counted as a single @@ -160,7 +163,7 @@ The input dataframe looks like this: Now let's extract 5 specific text features: the number of words, the number of characters, the number of sentences, whether the text has digits, and the ratio of -upper- to lowercase: +uppercase letters to non-whitespace characters: .. code:: python @@ -221,10 +224,10 @@ The output dataframe contains all 20 text features extracted from the `review` c 3 TERRIBLE!!! DO NOT BUY! Awful 20 4 review_sentence_count review_avg_word_length review_digit_count review_letter_count - 0 2 6.285714 0 36 - 1 2 6.200000 0 25 - 2 2 3.888889 2 23 - 3 2 5.750000 0 16 + 0 2 5.428571 0 36 + 1 2 5.400000 0 25 + 2 2 3.000000 2 23 + 3 2 5.000000 0 16 review_uppercase_count review_lowercase_count review_special_char_count review_whitespace_count 0 9 27 2 6 @@ -279,6 +282,57 @@ extracted features remain: 2 Average 9 27 3 Awful 4 20 +With polars +~~~~~~~~~~~ + +:class:`TextFeatures()` works the same way with a polars dataframe, and returns a polars +dataframe. Let's create a toy dataset with a missing value: + +.. code:: python + + import polars as pl + from feature_engine.text import TextFeatures + + X = pl.DataFrame({ + 'review': [ + 'This product is AMAZING! Best purchase ever.', + 'Not great. Would not recommend.', + 'OK for the price. 3 out of 5 stars.', + None, + ], + }) + +Let's extract the number of words, whether the text has digits, and the ratio of +uppercase letters: + +.. code:: python + + tf = TextFeatures( + variables=['review'], + features=['word_count', 'has_digits', 'uppercase_ratio'], + ) + + X_transformed = tf.fit_transform(X) + + print(X_transformed) + +We obtain a polars dataframe with the new features. The missing value was replaced by +an empty string, which has 0 words: + +.. code-block:: none + + shape: (4, 4) + ┌─────────────────────────────────┬───────────────────┬───────────────────┬────────────────────────┐ + │ review ┆ review_word_count ┆ review_has_digits ┆ review_uppercase_ratio │ + │ --- ┆ --- ┆ --- ┆ --- │ + │ str ┆ i64 ┆ i64 ┆ f64 │ + ╞═════════════════════════════════╪═══════════════════╪═══════════════════╪════════════════════════╡ + │ This product is AMAZING! Best … ┆ 7 ┆ 0 ┆ 0.236842 │ + │ Not great. Would not recommend… ┆ 5 ┆ 0 ┆ 0.074074 │ + │ OK for the price. 3 out of 5 s… ┆ 9 ┆ 1 ┆ 0.074074 │ + │ ┆ 0 ┆ 0 ┆ 0.0 │ + └─────────────────────────────────┴───────────────────┴───────────────────┴────────────────────────┘ + Combining with sklearn's bag-of-words ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ diff --git a/docs/user_guide/transformation/ArcSinhTransformer.rst b/docs/user_guide/transformation/ArcSinhTransformer.rst index 36dd44862..39806e94a 100644 --- a/docs/user_guide/transformation/ArcSinhTransformer.rst +++ b/docs/user_guide/transformation/ArcSinhTransformer.rst @@ -553,6 +553,44 @@ The recovered data: 493 -3.258723 9405.785347 122 30.047946 1448.874284 +With polars +----------- + +:class:`ArcSinhTransformer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.transformation import ArcSinhTransformer + + df = pl.DataFrame({ + "profit": [12.14, 6.43, 14.12, 33.89, 5.85, -2.5, 0.0], + "net_worth": [-8516.91, -277.74, 1920.33, -163.47, -10337.21, 500.0, 0.0], + }) + + tf = ArcSinhTransformer(variables=["profit", "net_worth"]) + tf.fit(df) + Xt = tf.transform(df) + + print(Xt) + +.. code:: text + + shape: (7, 2) + ┌───────────┬───────────┐ + │ profit ┆ net_worth │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞═══════════╪═══════════╡ + │ 3.191345 ┆ -9.742956 │ + │ 2.560114 ┆ -6.319836 │ + │ 3.341991 ┆ 8.2534 │ + │ 4.216485 ┆ -5.789786 │ + │ 2.466815 ┆ -9.936652 │ + │ -1.647231 ┆ 6.907756 │ + │ 0.0 ┆ 0.0 │ + └───────────┴───────────┘ + References ---------- diff --git a/docs/user_guide/transformation/ArcsinTransformer.rst b/docs/user_guide/transformation/ArcsinTransformer.rst index d69fa20ff..5d82e1280 100644 --- a/docs/user_guide/transformation/ArcsinTransformer.rst +++ b/docs/user_guide/transformation/ArcsinTransformer.rst @@ -134,6 +134,42 @@ shape after the transformation: .. image:: ../../images/breast_cancer_arcsin.png +With polars +----------- + +:class:`ArcsinTransformer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.transformation import ArcsinTransformer + + df = pl.DataFrame({ + "proportion_1": [0.1, 0.2, 0.3, 0.4, 0.5], + "proportion_2": [0.9, 0.8, 0.7, 0.6, 0.5], + }) + + tf = ArcsinTransformer(variables=None) + tf.fit(df) + Xt = tf.transform(df) + + print(Xt) + +.. code:: text + + shape: (5, 2) + ┌──────────────┬──────────────┐ + │ proportion_1 ┆ proportion_2 │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞══════════════╪══════════════╡ + │ 0.321751 ┆ 1.249046 │ + │ 0.463648 ┆ 1.107149 │ + │ 0.57964 ┆ 0.991157 │ + │ 0.684719 ┆ 0.886077 │ + │ 0.785398 ┆ 0.785398 │ + └──────────────┴──────────────┘ + Additional resources -------------------- diff --git a/docs/user_guide/transformation/BoxCoxTransformer.rst b/docs/user_guide/transformation/BoxCoxTransformer.rst index 5340d88b1..ad3b66043 100644 --- a/docs/user_guide/transformation/BoxCoxTransformer.rst +++ b/docs/user_guide/transformation/BoxCoxTransformer.rst @@ -206,6 +206,43 @@ In the following plots we see that the variables are non-normally distributed, b .. image:: ../../images/nonnormalvars2.png +With polars +----------- + +:class:`BoxCoxTransformer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.transformation import BoxCoxTransformer + + df = pl.DataFrame({ + "var_1": [4.0, 9.0, 16.0, 25.0, 100.0], + "var_2": [1.0, 8.0, 27.0, 64.0, 125.0], + }) + + boxcox = BoxCoxTransformer(variables=None) + boxcox.fit(df) + Xt = boxcox.transform(df) + + print(Xt) + +.. code:: text + + shape: (5, 2) + ┌──────────┬──────────┐ + │ var_1 ┆ var_2 │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞══════════╪══════════╡ + │ 1.225161 ┆ 0.0 │ + │ 1.810873 ┆ 2.666746 │ + │ 2.177001 ┆ 4.93175 │ + │ 2.435721 ┆ 6.969848 │ + │ 3.117497 ┆ 8.854291 │ + └──────────┴──────────┘ + + Additional resources -------------------- diff --git a/docs/user_guide/transformation/LogCpTransformer.rst b/docs/user_guide/transformation/LogCpTransformer.rst index 0f90f85af..b4b224d86 100644 --- a/docs/user_guide/transformation/LogCpTransformer.rst +++ b/docs/user_guide/transformation/LogCpTransformer.rst @@ -93,7 +93,7 @@ before applying the logarithm transformation: .. code:: python - {'MedInc': 0, 'HouseAge': 0} + {'MedInc': 0.0, 'HouseAge': 0.0} .. note:: @@ -298,6 +298,47 @@ And the constant values will be those from the dictionary: You can now apply `transform()` to transform all these variables. +With polars +----------- + +:class:`LogCpTransformer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.transformation import LogCpTransformer + + df = pl.DataFrame({"var_1": [-2.0, -1.0, 0.0, 1.0, 2.0]}) + + tf = LogCpTransformer(variables=None) + tf.fit(df) + + print(tf.C_) + +.. code:: text + + {'var_1': 3.0} + +.. code:: python + + print(tf.transform(df)) + +.. code:: text + + shape: (5, 1) + ┌──────────┐ + │ var_1 │ + │ --- │ + │ f64 │ + ╞══════════╡ + │ 0.0 │ + │ 0.693147 │ + │ 1.098612 │ + │ 1.386294 │ + │ 1.609438 │ + └──────────┘ + + Additional resources -------------------- diff --git a/docs/user_guide/transformation/LogTransformer.rst b/docs/user_guide/transformation/LogTransformer.rst index b73d317d8..8bab8fd38 100644 --- a/docs/user_guide/transformation/LogTransformer.rst +++ b/docs/user_guide/transformation/LogTransformer.rst @@ -217,6 +217,48 @@ mapping each variable to its own constant (``C={"bmi": 2, "s3": 3}``), the same way you would with the deprecated :class:`LogCpTransformer()`. +With polars +----------- + +:class:`LogTransformer()` works in the same way with a polars dataframe, including +the ``C="auto"`` shift for variables that contain zero or negative values: + +.. code:: python + + import polars as pl + from feature_engine.transformation import LogTransformer + + df = pl.DataFrame({"var_1": [-2.0, -1.0, 0.0, 1.0, 2.0]}) + + logt = LogTransformer(variables=None, C="auto") + logt.fit(df) + + print(logt.C_) + +.. code:: text + + {'var_1': 3.0} + +.. code:: python + + print(logt.transform(df)) + +.. code:: text + + shape: (5, 1) + ┌──────────┐ + │ var_1 │ + │ --- │ + │ f64 │ + ╞══════════╡ + │ 0.0 │ + │ 0.693147 │ + │ 1.098612 │ + │ 1.386294 │ + │ 1.609438 │ + └──────────┘ + + Additional resources -------------------- diff --git a/docs/user_guide/transformation/PowerTransformer.rst b/docs/user_guide/transformation/PowerTransformer.rst index 7aef1c6d4..009fcfb68 100644 --- a/docs/user_guide/transformation/PowerTransformer.rst +++ b/docs/user_guide/transformation/PowerTransformer.rst @@ -423,6 +423,43 @@ Result of the inverse transformation: As we can see, the original data and the inverse transformed one are identical. +With polars +----------- + +:class:`PowerTransformer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.transformation import PowerTransformer + + df = pl.DataFrame({ + "var_1": [4.0, 9.0, 16.0, 25.0, 100.0], + "var_2": [1.0, 8.0, 27.0, 64.0, 125.0], + }) + + tf = PowerTransformer(variables=None, exp=0.5) + tf.fit(df) + Xt = tf.transform(df) + + print(Xt) + +.. code:: text + + shape: (5, 2) + ┌───────┬──────────┐ + │ var_1 ┆ var_2 │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞═══════╪══════════╡ + │ 2.0 ┆ 1.0 │ + │ 3.0 ┆ 2.828427 │ + │ 4.0 ┆ 5.196152 │ + │ 5.0 ┆ 8.0 │ + │ 10.0 ┆ 11.18034 │ + └───────┴──────────┘ + + Considerations -------------- diff --git a/docs/user_guide/transformation/ReciprocalTransformer.rst b/docs/user_guide/transformation/ReciprocalTransformer.rst index e4aacf030..9f6336a12 100644 --- a/docs/user_guide/transformation/ReciprocalTransformer.rst +++ b/docs/user_guide/transformation/ReciprocalTransformer.rst @@ -227,6 +227,43 @@ symmetrically distributed across their value ranges: That's it! We've now applied different mathematical functions to stabilise the variance of the variables in the dataset. +With polars +----------- + +:class:`ReciprocalTransformer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.transformation import ReciprocalTransformer + + df = pl.DataFrame({ + "ratio_1": [4.0, 5.0, 10.0, 20.0, 2.0], + "ratio_2": [0.5, 0.25, 0.2, 0.1, 1.0], + }) + + tf = ReciprocalTransformer(variables=None) + tf.fit(df) + Xt = tf.transform(df) + + print(Xt) + +.. code:: text + + shape: (5, 2) + ┌─────────┬─────────┐ + │ ratio_1 ┆ ratio_2 │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞═════════╪═════════╡ + │ 0.25 ┆ 2.0 │ + │ 0.2 ┆ 4.0 │ + │ 0.1 ┆ 5.0 │ + │ 0.05 ┆ 10.0 │ + │ 0.5 ┆ 1.0 │ + └─────────┴─────────┘ + + Alternatives to the reciprocal function --------------------------------------- diff --git a/docs/user_guide/transformation/YeoJohnsonTransformer.rst b/docs/user_guide/transformation/YeoJohnsonTransformer.rst index 160e6f795..7c6257356 100644 --- a/docs/user_guide/transformation/YeoJohnsonTransformer.rst +++ b/docs/user_guide/transformation/YeoJohnsonTransformer.rst @@ -201,6 +201,43 @@ values, using the `inverse_transform` method. test_unt = tf.inverse_transform(test_t) +With polars +----------- + +:class:`YeoJohnsonTransformer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.transformation import YeoJohnsonTransformer + + df = pl.DataFrame({ + "var_1": [-4.0, -1.0, 0.0, 3.0, 10.0], + "var_2": [1.0, 8.0, 27.0, 64.0, 125.0], + }) + + tf = YeoJohnsonTransformer(variables=None) + tf.fit(df) + Xt = tf.transform(df) + + print(Xt) + +.. code:: text + + shape: (5, 2) + ┌───────────┬──────────┐ + │ var_1 ┆ var_2 │ + │ --- ┆ --- │ + │ f64 ┆ f64 │ + ╞═══════════╪══════════╡ + │ -5.520511 ┆ 0.740576 │ + │ -1.129157 ┆ 2.723463 │ + │ 0.0 ┆ 4.64057 │ + │ 2.323426 ┆ 6.353727 │ + │ 6.13494 ┆ 7.905058 │ + └───────────┴──────────┘ + + Additional resources -------------------- diff --git a/docs/user_guide/variable_handling/check_all_variables.rst b/docs/user_guide/variable_handling/check_all_variables.rst index d35f7f53d..cb0e74e95 100644 --- a/docs/user_guide/variable_handling/check_all_variables.rst +++ b/docs/user_guide/variable_handling/check_all_variables.rst @@ -38,8 +38,8 @@ Now, we create the dataset: X["cat_var1"] = ["Hello"] * 1000 X["cat_var2"] = ["Bye"] * 1000 - X["date1"] = pd.date_range("2020-02-24", periods=1000, freq="T") - X["date2"] = pd.date_range("2021-09-29", periods=1000, freq="H") + X["date1"] = pd.date_range("2020-02-24", periods=1000, freq="min") + X["date2"] = pd.date_range("2021-09-29", periods=1000, freq="h") X["date3"] = ["2020-02-24"] * 1000 print(X.head()) @@ -89,4 +89,48 @@ Below we see the error message: .. code:: python - KeyError: 'Some of the variables are not in the dataframe.' \ No newline at end of file + KeyError: 'Some of the variables are not in the dataframe.' + +With polars +----------- + +:class:`check_all_variables()` works in the same way with a polars dataframe. +Let's create an equivalent toy dataset: + +.. code:: python + + import polars as pl + from datetime import datetime, timedelta + from sklearn.datasets import make_classification + from feature_engine.variable_handling import check_all_variables + + X, y = make_classification( + n_samples=1000, + n_features=4, + n_redundant=1, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + + colnames = [f"num_var_{i+1}" for i in range(4)] + X = pl.DataFrame(X, schema=colnames) + + X = X.with_columns( + pl.lit("Hello").alias("cat_var1"), + pl.lit("Bye").alias("cat_var2"), + pl.Series("date1", [datetime(2020, 2, 24) + timedelta(minutes=i) for i in range(1000)]), + pl.Series("date2", [datetime(2021, 9, 29) + timedelta(hours=i) for i in range(1000)]), + pl.lit("2020-02-24").alias("date3"), + ) + + checked_vars = check_all_variables(X, ["num_var_1", "cat_var1", "date1"]) + + checked_vars + +The output is the list of variable names passed to the function: + +.. code:: python + + ['num_var_1', 'cat_var1', 'date1'] \ No newline at end of file diff --git a/docs/user_guide/variable_handling/check_categorical_variables.rst b/docs/user_guide/variable_handling/check_categorical_variables.rst index d311decd6..1aa610ec1 100644 --- a/docs/user_guide/variable_handling/check_categorical_variables.rst +++ b/docs/user_guide/variable_handling/check_categorical_variables.rst @@ -38,8 +38,8 @@ Now, we create the dataset: X["cat_var1"] = ["Hello"] * 1000 X["cat_var2"] = ["Bye"] * 1000 - X["date1"] = pd.date_range("2020-02-24", periods=1000, freq="T") - X["date2"] = pd.date_range("2021-09-29", periods=1000, freq="H") + X["date1"] = pd.date_range("2020-02-24", periods=1000, freq="min") + X["date2"] = pd.date_range("2021-09-29", periods=1000, freq="h") X["date3"] = ["2020-02-24"] * 1000 print(X.head()) @@ -89,4 +89,49 @@ Below we see the error message: .. code:: python TypeError: Some of the variables are not categorical. Please cast them as object - or categorical before using this transformer. \ No newline at end of file + or categorical before using this transformer. + +With polars +----------- + +:class:`check_categorical_variables()` works in the same way with a polars dataframe. +Let's create an equivalent toy dataset: + +.. code:: python + + import polars as pl + from datetime import datetime, timedelta + from sklearn.datasets import make_classification + from feature_engine.variable_handling import check_categorical_variables + + X, y = make_classification( + n_samples=1000, + n_features=4, + n_redundant=1, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + + colnames = [f"num_var_{i+1}" for i in range(4)] + X = pl.DataFrame(X, schema=colnames) + + X = X.with_columns( + pl.lit("Hello").alias("cat_var1"), + pl.lit("Bye").alias("cat_var2"), + pl.Series("date1", [datetime(2020, 2, 24) + timedelta(minutes=i) for i in range(1000)]), + pl.Series("date2", [datetime(2021, 9, 29) + timedelta(hours=i) for i in range(1000)]), + pl.lit("2020-02-24").alias("date3"), + ) + + var_cat = check_categorical_variables(X, ["cat_var1", "date3"]) + + var_cat + +Both variables are of type string and hence, will be in the resulting list: + +.. code:: python + + ['cat_var1', 'date3'] + diff --git a/docs/user_guide/variable_handling/check_datetime_variables.rst b/docs/user_guide/variable_handling/check_datetime_variables.rst index e1fb86168..ab28a9eb4 100644 --- a/docs/user_guide/variable_handling/check_datetime_variables.rst +++ b/docs/user_guide/variable_handling/check_datetime_variables.rst @@ -38,8 +38,8 @@ Now, we create the dataset: X["cat_var1"] = ["Hello"] * 1000 X["cat_var2"] = ["Bye"] * 1000 - X["date1"] = pd.date_range("2020-02-24", periods=1000, freq="T") - X["date2"] = pd.date_range("2021-09-29", periods=1000, freq="H") + X["date1"] = pd.date_range("2020-02-24", periods=1000, freq="min") + X["date2"] = pd.date_range("2021-09-29", periods=1000, freq="h") X["date3"] = ["2020-02-24"] * 1000 print(X.head()) @@ -92,3 +92,49 @@ Below the error message: .. code:: python TypeError: Some of the variables are not or cannot be parsed as datetime. + +With polars +----------- + +:class:`check_datetime_variables()` works in the same way with a polars dataframe. +Let's create an equivalent toy dataset: + +.. code:: python + + import polars as pl + from datetime import datetime, timedelta + from sklearn.datasets import make_classification + from feature_engine.variable_handling import check_datetime_variables + + X, y = make_classification( + n_samples=1000, + n_features=4, + n_redundant=1, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + + colnames = [f"num_var_{i+1}" for i in range(4)] + X = pl.DataFrame(X, schema=colnames) + + X = X.with_columns( + pl.lit("Hello").alias("cat_var1"), + pl.lit("Bye").alias("cat_var2"), + pl.Series("date1", [datetime(2020, 2, 24) + timedelta(minutes=i) for i in range(1000)]), + pl.Series("date2", [datetime(2021, 9, 29) + timedelta(hours=i) for i in range(1000)]), + pl.lit("2020-02-24").alias("date3"), + ) + + var_date = check_datetime_variables(X, ["date2", "date3"]) + + var_date + +In this case, both variables, if they can be parsed as datetime, will be in the +resulting list: + +.. code:: python + + ['date2', 'date3'] + diff --git a/docs/user_guide/variable_handling/check_numerical_variables.rst b/docs/user_guide/variable_handling/check_numerical_variables.rst index 795376850..b95255b5c 100644 --- a/docs/user_guide/variable_handling/check_numerical_variables.rst +++ b/docs/user_guide/variable_handling/check_numerical_variables.rst @@ -25,7 +25,7 @@ Now, we create the dataset: "City": ["London", "Manchester", "Liverpool", "Bristol"], "Age": [20, 21, 19, 18], "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="T"), + "dob": pd.date_range("2020-02-24", periods=4, freq="min"), }) print(df.head()) @@ -67,3 +67,32 @@ Below we see the error message: TypeError: Some of the variables are not numerical. Please cast them as numerical before using this transformer. + +With polars +----------- + +:class:`check_numerical_variables()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from datetime import datetime + from feature_engine.variable_handling import check_numerical_variables + + df = pl.DataFrame({ + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, i) for i in range(4)], + }) + + var_num = check_numerical_variables(df, ['Age', 'Marks']) + + var_num + +If the variables are numerical, the function returns their names in a list: + +.. code:: python + + ['Age', 'Marks'] diff --git a/docs/user_guide/variable_handling/find_all_variables.rst b/docs/user_guide/variable_handling/find_all_variables.rst index 055fac788..8c7835dd0 100644 --- a/docs/user_guide/variable_handling/find_all_variables.rst +++ b/docs/user_guide/variable_handling/find_all_variables.rst @@ -125,4 +125,68 @@ However, this command returns an empty list: X[[ 'date1', 'date2', 'date3']], exclude_datetime=True, return_empty=True, - ) \ No newline at end of file + ) + +With polars +----------- + +:class:`find_all_variables()` works in the same way with a polars dataframe. +Let's create an equivalent toy dataset: + +.. code:: python + + import polars as pl + from datetime import datetime, timedelta + from sklearn.datasets import make_classification + from feature_engine.variable_handling import find_all_variables + + X, y = make_classification( + n_samples=1000, + n_features=4, + n_redundant=1, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + + colnames = [f"num_var_{i+1}" for i in range(4)] + X = pl.DataFrame(X, schema=colnames) + + X = X.with_columns( + pl.lit("Hello").alias("cat_var1"), + pl.lit("Bye").alias("cat_var2"), + pl.Series("date1", [datetime(2020, 2, 24) + timedelta(minutes=i) for i in range(1000)]), + pl.Series("date2", [datetime(2021, 9, 29) + timedelta(hours=i) for i in range(1000)]), + pl.lit("2020-02-24").alias("date3"), + ) + + vars_all = find_all_variables(X) + + vars_all + +We see the variable names in the list below: + +.. code:: python + + ['num_var_1', + 'num_var_2', + 'num_var_3', + 'num_var_4', + 'cat_var1', + 'cat_var2', + 'date1', + 'date2', + 'date3'] + +And, as with pandas, we can exclude the datetime variables: + +.. code:: python + + vars_all = find_all_variables(X, exclude_datetime=True) + + vars_all + +.. code:: python + + ['num_var_1', 'num_var_2', 'num_var_3', 'num_var_4', 'cat_var1', 'cat_var2'] \ No newline at end of file diff --git a/docs/user_guide/variable_handling/find_categorical_and_numerical_variables.rst b/docs/user_guide/variable_handling/find_categorical_and_numerical_variables.rst index 05d5807cc..0b27dc70a 100644 --- a/docs/user_guide/variable_handling/find_categorical_and_numerical_variables.rst +++ b/docs/user_guide/variable_handling/find_categorical_and_numerical_variables.rst @@ -126,4 +126,50 @@ To return empty lists instead, we set `return_empty` to `True`: find_categorical_and_numerical_variables( X[[ 'date1', 'date2', 'date3']], return_empty = True - ) \ No newline at end of file + ) + +With polars +----------- + +:class:`find_categorical_and_numerical_variables()` works in the same way with a +polars dataframe. Let's create an equivalent toy dataset: + +.. code:: python + + import polars as pl + from datetime import datetime, timedelta + from sklearn.datasets import make_classification + from feature_engine.variable_handling import find_categorical_and_numerical_variables + + X, y = make_classification( + n_samples=1000, + n_features=4, + n_redundant=1, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + + colnames = [f"num_var_{i+1}" for i in range(4)] + X = pl.DataFrame(X, schema=colnames) + + X = X.with_columns( + pl.lit("Hello").alias("cat_var1"), + pl.lit("Bye").alias("cat_var2"), + pl.Series("date1", [datetime(2020, 2, 24) + timedelta(minutes=i) for i in range(1000)]), + pl.Series("date2", [datetime(2021, 9, 29) + timedelta(hours=i) for i in range(1000)]), + pl.lit("2020-02-24").alias("date3"), + ) + + var_cat, var_num = find_categorical_and_numerical_variables(X) + + var_cat, var_num + +Below we see the names of the categorical variables, followed by the names of the +numerical variables: + +.. code:: python + + (['cat_var1', 'cat_var2'], + ['num_var_1', 'num_var_2', 'num_var_3', 'num_var_4']) \ No newline at end of file diff --git a/docs/user_guide/variable_handling/find_categorical_variables.rst b/docs/user_guide/variable_handling/find_categorical_variables.rst index a00de72af..1ed6a23de 100644 --- a/docs/user_guide/variable_handling/find_categorical_variables.rst +++ b/docs/user_guide/variable_handling/find_categorical_variables.rst @@ -90,3 +90,52 @@ To return an empty list instead of the error we need to set `return_empty` to `T follows: `find_categorical_variables(X[colnames], return_empty=True)`. The previous command returns an empty list: `[]`. + +With polars +----------- + +:class:`find_categorical_variables()` works in the same way with a polars dataframe. +Let's create an equivalent toy dataset: + +.. code:: python + + import polars as pl + from datetime import datetime, timedelta + from sklearn.datasets import make_classification + from feature_engine.variable_handling import find_categorical_variables + + X, y = make_classification( + n_samples=1000, + n_features=4, + n_redundant=1, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + + colnames = [f"num_var_{i+1}" for i in range(4)] + X = pl.DataFrame(X, schema=colnames) + + X = X.with_columns( + pl.lit("Hello").alias("cat_var1"), + pl.lit("Bye").alias("cat_var2"), + pl.Series("date1", [datetime(2020, 2, 24) + timedelta(minutes=i) for i in range(1000)]), + pl.Series("date2", [datetime(2021, 9, 29) + timedelta(hours=i) for i in range(1000)]), + pl.lit("2020-02-24").alias("date3"), + ) + +Now let's find the categorical variables: + +.. code:: python + + var_cat = find_categorical_variables(X) + + var_cat + +We see the variable names in the list below: + +.. code:: python + + ['cat_var1', 'cat_var2'] + diff --git a/docs/user_guide/variable_handling/find_datetime_variables.rst b/docs/user_guide/variable_handling/find_datetime_variables.rst index 443a43d4e..baa327873 100644 --- a/docs/user_guide/variable_handling/find_datetime_variables.rst +++ b/docs/user_guide/variable_handling/find_datetime_variables.rst @@ -87,3 +87,53 @@ can be parsed as datetime, it will be captured in the list as well. If there are no datetime variables, :class:`find_datetime_variables()` will raise an error. To return an empty list instead, use the argument `return_empty` to `True`. + +With polars +----------- + +:class:`find_datetime_variables()` works in the same way with a polars dataframe. +Let's create an equivalent toy dataset: + +.. code:: python + + import polars as pl + from datetime import datetime, timedelta + from sklearn.datasets import make_classification + from feature_engine.variable_handling import find_datetime_variables + + X, y = make_classification( + n_samples=1000, + n_features=4, + n_redundant=1, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + + colnames = [f"num_var_{i+1}" for i in range(4)] + X = pl.DataFrame(X, schema=colnames) + + X = X.with_columns( + pl.lit("Hello").alias("cat_var1"), + pl.lit("Bye").alias("cat_var2"), + pl.Series("date1", [datetime(2020, 2, 24) + timedelta(minutes=i) for i in range(1000)]), + pl.Series("date2", [datetime(2021, 9, 29) + timedelta(hours=i) for i in range(1000)]), + pl.lit("2020-02-24").alias("date3"), + ) + +The dataframe has 3 datetime variables: two of them are native polars `Datetime` +columns, and one, `date3`, is an ISO-8601 string. Let's capture all 3: + +.. code:: python + + var_date = find_datetime_variables(X) + + var_date + +Below we see the variable names in the list: + +.. code:: python + + ['date1', 'date2', 'date3'] + diff --git a/docs/user_guide/variable_handling/find_numerical_variables.rst b/docs/user_guide/variable_handling/find_numerical_variables.rst index fcb4b40c4..aa315114d 100644 --- a/docs/user_guide/variable_handling/find_numerical_variables.rst +++ b/docs/user_guide/variable_handling/find_numerical_variables.rst @@ -68,3 +68,32 @@ need to set `return_empty` to `True`: find_numerical_variables(df[["Name", "City", "dob"]], return_empty=True) The previous command returns an empty list: `[]`. + +With polars +----------- + +:class:`find_numerical_variables()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from datetime import datetime + from feature_engine.variable_handling import find_numerical_variables + + df = pl.DataFrame({ + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, i) for i in range(4)], + }) + + var_num = find_numerical_variables(df) + + var_num + +We see the names of the numerical variables in the list below: + +.. code:: python + + ['Age', 'Marks'] diff --git a/docs/user_guide/variable_handling/retain_variables_if_in_df.rst b/docs/user_guide/variable_handling/retain_variables_if_in_df.rst index 4f4ab71cd..c5f36ecf7 100644 --- a/docs/user_guide/variable_handling/retain_variables_if_in_df.rst +++ b/docs/user_guide/variable_handling/retain_variables_if_in_df.rst @@ -25,7 +25,7 @@ Now, we create the dataset: "City": ["London", "Manchester", "Liverpool", "Bristol"], "Age": [20, 21, 19, 18], "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="T"), + "dob": pd.date_range("2020-02-24", periods=4, freq="min"), }) print(df.head()) @@ -59,6 +59,35 @@ We see the names of the subset of variables that are in the dataframe below: If none of variables in the list are in the dataset, :class:`retain_variables_if_in_df()` will raise an error. +With polars +----------- + +:class:`retain_variables_if_in_df()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from datetime import datetime + from feature_engine.variable_handling import retain_variables_if_in_df + + df = pl.DataFrame({ + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, i) for i in range(4)], + }) + + vars_in_df = retain_variables_if_in_df(df, variables = ["Name", "City", "Dogs"]) + + vars_in_df + +We see the names of the subset of variables that are in the dataframe below: + +.. code:: python + + ['Name', 'City'] + Uses ---- diff --git a/feature_engine/_base_transformers/base_numerical.py b/feature_engine/_base_transformers/base_numerical.py index fed663213..a88856820 100644 --- a/feature_engine/_base_transformers/base_numerical.py +++ b/feature_engine/_base_transformers/base_numerical.py @@ -3,7 +3,11 @@ shared by most transformers, like checking that input is a df, the size, NA, etc. """ -import pandas as pd +from typing import List, Tuple, Union + +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -28,7 +32,9 @@ class BaseNumericalTransformer( variable transformers, discretisers, math combination. """ - def _fit_setup(self, X: pd.DataFrame): + def _fit_setup( + self, X: IntoDataFrame + ) -> Tuple[nw.DataFrame, List[Union[str, int]]]: """ Checks that input is a dataframe, finds numerical variables, or alternatively checks that variables entered by the user are of type numerical, and checks @@ -38,12 +44,12 @@ def _fit_setup(self, X: pd.DataFrame): Parameters ---------- - X : Pandas DataFrame + X : dataframe Raises ------ TypeError - If the input is not a Pandas DataFrame or a numpy array + If the input is not a recognised dataframe If any of the user provided variables are not numerical ValueError If there are no numerical variables in the df or the df is empty @@ -51,77 +57,70 @@ def _fit_setup(self, X: pd.DataFrame): Returns ------- - X : Pandas DataFrame - The same dataframe entered as parameter + nw_X : narwhals dataframe + The dataframe entered as parameter, as a narwhals dataframe. variables_ : List The variables that were found or checked. """ + nw_X = check_X(X) - # check input dataframe - X = check_X(X) - - # find or check for numerical variables if self.variables is None: variables_ = find_numerical_variables(X, return_empty=self.return_empty) else: variables_ = check_numerical_variables(X, self.variables) - # check if dataset contains na or inf _check_contains_na(X, variables_) _check_contains_inf(X, variables_) - return X, variables_ + return nw_X, variables_ def _get_feature_names_in(self, X): """Get the names and number of features in the train set (the dataframe used during fit).""" - self.feature_names_in_ = X.columns.tolist() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self - def _check_transform_input_and_state(self, X: pd.DataFrame) -> pd.DataFrame: + def _check_transform_input_and_state(self, X: IntoDataFrame) -> nw.DataFrame: """ Checks that the input is a dataframe and of the same size than the one used in the fit() method. Checks absence of NA and Inf. Parameters ---------- - X : Pandas DataFrame + X : dataframe Raises ------ TypeError - If the input is not a Pandas DataFrame + If the input is not a recognised dataframe ValueError - If the variable(s) contain null values - If the df has different number of features than the df used in fit() Returns ------- - X : Pandas DataFrame. - The same dataframe entered by the user. + nw_X : narwhals dataframe + The dataframe entered by the user, as a narwhals dataframe, with the + variables in the same order as in the train set. """ - - # Check method fit has been called check_is_fitted(self) - - # check that input is a dataframe - X = check_X(X) - - # Check if input data contains same number of columns as dataframe used to fit. + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - - # check if dataset contains na or inf _check_contains_na(X, self.variables_) _check_contains_inf(X, self.variables_) - # reorder variables to match train set - X = X[self.feature_names_in_] - - return X + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return nw.from_native(X[self.feature_names_in_], eager_only=True) + else: + return nw_X.select(nw.col(*self.feature_names_in_)) # for the check_estimator tests def _more_tags(self): diff --git a/feature_engine/_base_transformers/mixins.py b/feature_engine/_base_transformers/mixins.py index 9207873be..a1e5ef29b 100644 --- a/feature_engine/_base_transformers/mixins.py +++ b/feature_engine/_base_transformers/mixins.py @@ -1,6 +1,8 @@ from typing import Dict, List, Tuple, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from numpy import ndarray from numpy.typing import ArrayLike from sklearn.utils.validation import check_is_fitted @@ -15,7 +17,7 @@ class TransformXyMixin: - def transform_x_y(self, X: pd.DataFrame, y: pd.Series): + def transform_x_y(self, X: IntoDataFrame, y: IntoSeries): """ Transform, align and adjust both X and y based on the transformations applied to X, ensuring that they correspond to the same set of rows if any were @@ -23,32 +25,64 @@ def transform_x_y(self, X: pd.DataFrame, y: pd.Series): Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The dataframe to transform. - y: pandas Series or Dataframe of length = n_samples + y: Series or Dataframe of length = n_samples The target variable to transform. Can be multi-output. Returns ------- - X_new: pandas dataframe + X_new: dataframe The transformed dataframe of shape [n_samples - n_rows, n_features]. It may contain less rows than the original dataset. - y_new: pandas Series or DataFrame + y_new: Series or DataFrame The transformed target variable of length [n_samples - n_rows]. It contains as many rows as those left in X_new. """ - X, y = check_X_y(X, y) - X = self.transform(X) - y = y.loc[X.index] + # check_X_y validates X against y and returns X as a narwhals dataframe; + # y stays native. + nw_X, y = check_X_y(X, y) + + if nw_X.implementation.is_pandas(): + # pandas rows carry labels, so y can be realigned on the index of + # the rows that transform() kept. + X = self.transform(X) + y = y.loc[X.index] + else: + # positional backends (polars, pyarrow, ...) have no row labels: + # tag each row with its position, transform, then use the tags that + # survived to subset y. + row_index_col = "__feature_engine_row_index__" + nw_X = nw_X.with_row_index(row_index_col) + # the tag column leaves X one column wider than the training set, + # which transform()'s column-count check (when the transformer has + # one) would reject. Widen the reference by one for the duration of + # this internal call only. + has_count_check = hasattr(self, "n_features_in_") + if has_count_check: + self.n_features_in_ += 1 + try: + X = self.transform(nw_X.to_native()) + finally: + if has_count_check: + self.n_features_in_ -= 1 + nw_X = nw.from_native(X, eager_only=True) + row_positions = nw_X.get_column(row_index_col).to_list() + X = nw_X.drop(row_index_col).to_native() + if nwd.is_into_series(y): + y = nw.from_native(y, series_only=True)[row_positions].to_native() + else: + y = nw.from_native(y, eager_only=True)[row_positions].to_native() + return X, y class FitFromDictMixin: def _fit_from_dict( - self, X: pd.DataFrame, user_dict_: Dict - ) -> Tuple[pd.DataFrame, List[Union[str, int]]]: + self, X: IntoDataFrame, user_dict_: Dict + ) -> Tuple[nw.DataFrame, List[Union[str, int]]]: """ Checks that input is a dataframe, checks that variables in the dictionary entered by the user are of type numerical. Does not assign any @@ -57,7 +91,7 @@ def _fit_from_dict( Parameters ---------- - X : Pandas DataFrame + X : dataframe user_dict_ : Dictionary. Default = None Any dictionary allowed by the transformer and entered by user. @@ -65,7 +99,7 @@ def _fit_from_dict( Raises ------ TypeError - If the input is not a Pandas DataFrame or a numpy array + If the input is not a recognised dataframe If any of the variables in the dictionary are not numerical ValueError If there are no numerical variables in the df or the df is empty @@ -73,24 +107,18 @@ def _fit_from_dict( Returns ------- - X : Pandas DataFrame - The same dataframe entered as parameter + nw_X : narwhals dataframe + The dataframe entered as parameter, as a narwhals dataframe. variables_ : List The variables in the dictionary. """ - # check input dataframe - X = check_X(X) - - # find or check for numerical variables - variables = list(user_dict_.keys()) - variables_ = check_numerical_variables(X, variables) - - # check if dataset contains na or inf + nw_X = check_X(X) + variables_ = check_numerical_variables(X, list(user_dict_.keys())) _check_contains_na(X, variables_) _check_contains_inf(X, variables_) - return X, variables_ + return nw_X, variables_ class GetFeatureNamesOutMixin: @@ -118,47 +146,20 @@ def get_feature_names_out( check_is_fitted(self) if input_features is not None: - # If input to fit is an array, then the variable names in - # feature_names_in_ are "x0", "x1","x2" ..."xn". - if self.feature_names_in_ == [f"x{i}" for i in range(self.n_features_in_)]: - - # If the input was an array, we let the user enter the variable names. - if len(input_features) == self.n_features_in_: - if isinstance(input_features, list): - feature_names = input_features - else: - feature_names = list(input_features) - - # For transformers that add features to the data. - feature_names = self._add_new_feature_names(feature_names) - - # For transformers that remove features from data, i..e, selectors. - feature_names = self._remove_feature_names( - feature_names, indices=True - ) - - return feature_names - - else: - raise ValueError( - "The number of input_features does not match the number of " - "features seen in the dataframe used in fit." - ) + msg = "input_features is not equal to feature_names_in_" + if isinstance(input_features, list): + if input_features != self.feature_names_in_: + raise ValueError(msg) + elif isinstance(input_features, ndarray) or ( + nwd.is_pandas_index(input_features) is True + ): + if list(input_features) != self.feature_names_in_: + raise ValueError(msg) else: - msg = "input_features is not equal to feature_names_in_" - if isinstance(input_features, list): - if input_features != self.feature_names_in_: - raise ValueError(msg) - elif isinstance(input_features, ndarray) or isinstance( - input_features, pd.core.indexes.base.Index - ): - if list(input_features) != self.feature_names_in_: - raise ValueError(msg) - else: - raise ValueError( - "input_features must be a list or an array. " - "Got {input_features} instead." - ) + raise ValueError( + "input_features must be a list or an array. " + f"Got {input_features} instead." + ) feature_names = self.feature_names_in_ @@ -166,7 +167,7 @@ def get_feature_names_out( feature_names = self._add_new_feature_names(feature_names) # For transformers that remove features from data, i..e, selectors. - feature_names = self._remove_feature_names(feature_names, indices=False) + feature_names = self._remove_feature_names(feature_names) return feature_names @@ -183,14 +184,10 @@ def _add_new_feature_names(self, feature_names): return feature_names - def _remove_feature_names(self, feature_names, indices=False) -> List: + def _remove_feature_names(self, feature_names) -> List: # For transformers that remove features from data, i..e, selectors. if hasattr(self, "features_to_drop_"): - if indices is True: - mask = self.get_support(indices=True) - feature_names = [feature_names[i] for i in mask] - else: - feature_names = [ - f for f in feature_names if f not in self.features_to_drop_ - ] + feature_names = [ + f for f in feature_names if f not in self.features_to_drop_ + ] return feature_names diff --git a/feature_engine/_check_init_parameters/check_init_input_params.py b/feature_engine/_check_init_parameters/check_init_input_params.py index e1000accc..a89b3b500 100644 --- a/feature_engine/_check_init_parameters/check_init_input_params.py +++ b/feature_engine/_check_init_parameters/check_init_input_params.py @@ -1,5 +1,8 @@ def _check_param_missing_values(missing_values): - if missing_values not in ["raise", "ignore"]: + if not isinstance(missing_values, str) or missing_values not in [ + "raise", + "ignore", + ]: raise ValueError( "missing_values takes only values 'raise' or 'ignore'. " f"Got {missing_values} instead." diff --git a/feature_engine/_prediction/base_predictor.py b/feature_engine/_prediction/base_predictor.py index c7e2618fd..819b6a5f0 100644 --- a/feature_engine/_prediction/base_predictor.py +++ b/feature_engine/_prediction/base_predictor.py @@ -233,16 +233,16 @@ def _make_discretiser(self): def _transform(self, X: pd.DataFrame) -> pd.DataFrame: """ - Replace original values by the average of the target mean value per bin or - category in each one of the variables. + Replace original values by the average of the target mean value per bin or + category in each one of the variables. - Parameters - ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The input samples. + Parameters + ---------- + X : pandas dataframe of shape = [n_samples, n_features] + The input samples. - Return - ------- + Returns + ------- X_new: pandas dataframe of shape = [n_samples, n_features] The transformed data with the discrete variables. """ @@ -279,7 +279,7 @@ def _predict(self, X: pd.DataFrame) -> np.ndarray: X : pandas dataframe of shape = [n_samples, n_features] The input samples. - Return + Returns ------- y_pred: numpy array of shape = (n_samples, ) The mean target value per observation. diff --git a/feature_engine/_prediction/target_mean_classifier.py b/feature_engine/_prediction/target_mean_classifier.py index bc88b0c2f..b0c1a71ab 100644 --- a/feature_engine/_prediction/target_mean_classifier.py +++ b/feature_engine/_prediction/target_mean_classifier.py @@ -139,7 +139,7 @@ def predict_proba(self, X: pd.DataFrame) -> np.ndarray: X : pandas dataframe of shape = [n_samples, n_features] The input samples. - Return + Returns ------- p: array-like of shape (n_samples, n_classes) Returns the probability of the sample for each class in the model, where @@ -159,7 +159,7 @@ def predict_log_proba(self, X: pd.DataFrame) -> np.ndarray: X : pandas dataframe of shape = [n_samples, n_features] The input samples. - Return + Returns ------- p: array-like of shape (n_samples, n_classes) Returns the log-probability of the sample for each class in the model, @@ -178,7 +178,7 @@ def predict(self, X: pd.DataFrame) -> np.ndarray: X : pandas dataframe of shape = [n_samples, n_features] The input samples. - Return + Returns ------- y_pred: ndarray of shape (n_samples,) Vector containing the class labels for each sample. diff --git a/feature_engine/_prediction/target_mean_regressor.py b/feature_engine/_prediction/target_mean_regressor.py index 26fa27875..231d7c268 100644 --- a/feature_engine/_prediction/target_mean_regressor.py +++ b/feature_engine/_prediction/target_mean_regressor.py @@ -115,7 +115,7 @@ def predict(self, X: pd.DataFrame) -> np.ndarray: X : pandas dataframe of shape = [n_samples, ] The input samples. - Return + Returns ------- y_pred: ndarray of shape (n_samples,) Returns predicted values. diff --git a/feature_engine/creation/base_creation.py b/feature_engine/creation/base_creation.py index c294045f4..e2fb424b0 100644 --- a/feature_engine/creation/base_creation.py +++ b/feature_engine/creation/base_creation.py @@ -1,6 +1,8 @@ from typing import Optional -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -37,21 +39,20 @@ def __init__( self.missing_values = missing_values self.drop_original = drop_original - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. - y: pandas Series, or np.array. Defaults to None. + y: Series, or np.array. Defaults to None. It is not needed in this transformer. You can pass y or None. """ - # check input dataframe - X = check_X(X) + check_X(X) # check variables are numerical if self.variables is None: @@ -71,33 +72,35 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): _check_contains_inf(X, self.reference) # save input features - self.feature_names_in_ = X.columns.tolist() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns # save train set shape self.n_features_in_ = X.shape[1] return self - def _check_transform_input_and_state(self, X: pd.DataFrame) -> pd.DataFrame: + def _check_transform_input_and_state(self, X: IntoDataFrame) -> IntoDataFrame: """ Common input and transformer checks. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: Pandas dataframe + X_new: dataframe The dataframe with the original variables plus the new variables. """ # Check method fit has been called check_is_fitted(self) - # check that input is a dataframe - X = check_X(X) + check_X(X) # Check if input data contains same number of columns as dataframe used to fit. _check_X_matches_training_df(X, self.n_features_in_) @@ -111,7 +114,12 @@ def _check_transform_input_and_state(self, X: pd.DataFrame) -> pd.DataFrame: _check_contains_inf(X, self.reference) # reorder variables to match train set - X = X[self.feature_names_in_] + if nwd.is_pandas_dataframe(X) is True: + X = X[self.feature_names_in_] + else: + X = nw.from_native(X, eager_only=True).select( + self.feature_names_in_ + ).to_native() return X diff --git a/feature_engine/creation/cyclical_features.py b/feature_engine/creation/cyclical_features.py index 24018b0cd..3418e0af1 100644 --- a/feature_engine/creation/cyclical_features.py +++ b/feature_engine/creation/cyclical_features.py @@ -1,7 +1,8 @@ from typing import Dict, List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._base_transformers.mixins import ( @@ -122,6 +123,30 @@ class CyclicalFeatures( 5 2 1.224647e-16 -1.000000e+00 6 1 1.000000e+00 6.123234e-17 7 2 1.224647e-16 -1.000000e+00 + + With polars: + + >>> import polars as pl + >>> from feature_engine.creation import CyclicalFeatures + >>> X = pl.DataFrame({"x": [1, 4, 3, 3, 4, 2, 1, 2]}) + >>> cf = CyclicalFeatures() + >>> cf.fit(X) + >>> cf.transform(X) + shape: (8, 3) + ┌─────┬─────────────┬─────────────┐ + │ x ┆ x_sin ┆ x_cos │ + │ --- ┆ --- ┆ --- │ + │ i64 ┆ f64 ┆ f64 │ + ╞═════╪═════════════╪═════════════╡ + │ 1 ┆ 1.0 ┆ 6.1232e-17 │ + │ 4 ┆ -2.4493e-16 ┆ 1.0 │ + │ 3 ┆ -1.0 ┆ -1.8370e-16 │ + │ 3 ┆ -1.0 ┆ -1.8370e-16 │ + │ 4 ┆ -2.4493e-16 ┆ 1.0 │ + │ 2 ┆ 1.2246e-16 ┆ -1.0 │ + │ 1 ┆ 1.0 ┆ 6.1232e-17 │ + │ 2 ┆ 1.2246e-16 ┆ -1.0 │ + └─────┴─────────────┴─────────────┘ """ def __init__( @@ -141,24 +166,33 @@ def __init__( self.max_values = max_values self.drop_original = drop_original - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learns the maximum value of each variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ if self.max_values is None: - X, variables_ = self._fit_setup(X) - max_values_ = X[variables_].max().to_dict() + nw_X, variables_ = self._fit_setup(X) + if len(variables_) == 0: + # return_empty=True can leave variables_ empty; narwhals' + # select([]) collapses row count too, so .to_numpy().max() + # would fail on a genuinely empty selection. + max_values_: dict = {} + else: + max_arr = nw_X.select(nw.col(variables_)).to_numpy().max(axis=0) + # .tolist() converts numpy scalars to plain Python int/float, + # matching the dtype .to_dict() used to return. + max_values_ = dict(zip(variables_, max_arr.tolist())) else: - X, variables_ = super()._fit_from_dict(X, self.max_values) + _, variables_ = super()._fit_from_dict(X, self.max_values) max_values_ = self.max_values self.variables_ = variables_ @@ -167,29 +201,31 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame): + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Creates new features using the cyclical transformations. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe. + X_new: dataframe. The original dataframe plus the additional features. """ - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) + new_cols = [] for variable in self.variables_: - max_value = self.max_values_[variable] - X[f"{variable}_sin"] = np.sin(X[variable] * (2.0 * np.pi / max_value)) - X[f"{variable}_cos"] = np.cos(X[variable] * (2.0 * np.pi / max_value)) - - if self.drop_original: - X.drop(columns=self.variables_, inplace=True) + scaled = nw.col(variable) * (2.0 * np.pi / self.max_values_[variable]) + new_cols.append(scaled.sin().alias(f"{variable}_sin")) + new_cols.append(scaled.cos().alias(f"{variable}_cos")) + nw_X = nw_X.with_columns(*new_cols) + if self.drop_original is True: + nw_X = nw_X.drop(self.variables_) + X = nw_X.to_native() return X diff --git a/feature_engine/creation/decision_tree_features.py b/feature_engine/creation/decision_tree_features.py index aa76d4bbd..ba20a292b 100644 --- a/feature_engine/creation/decision_tree_features.py +++ b/feature_engine/creation/decision_tree_features.py @@ -1,8 +1,11 @@ import itertools from typing import Any, Dict, Iterable, List, Optional, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from joblib import Parallel, delayed +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.base import BaseEstimator, TransformerMixin from sklearn.model_selection import GridSearchCV from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor @@ -131,6 +134,11 @@ class DecisionTreeFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMi DecisionTreeClassifier(). For reproducibility it is recommended to set the random_state to an integer. + n_jobs: int, default=None + The number of jobs to run in parallel. `fit` is parallelized over the feature + combinations, training one decision tree per combination. `None` means 1 + unless in a `joblib.parallel_backend` context. `-1` means using all processors. + {missing_values} {drop_original} @@ -210,6 +218,36 @@ class DecisionTreeFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMi 7 4.24 8 4.24 9 6.00 + + With polars: + + >>> import polars as pl + >>> from feature_engine.creation import DecisionTreeFeatures + >>> X = pl.DataFrame({ + ... "Age": [20, 44, 19, 33, 51, 40, 41, 37, 30, 54], + ... "Height": [164, 150, 178, 158, 188, 190, 168, 174, 176, 171], + ... }) + >>> y = [4.1, 5.8, 3.9, 6.2, 4.3, 4.5, 7.2, 4.4, 4.1, 6.7] + >>> dtf = DecisionTreeFeatures(features_to_combine=1) + >>> dtf.fit(X, y) + >>> dtf.transform(X) + shape: (10, 4) + ┌─────┬────────┬───────────┬──────────────┐ + │ Age ┆ Height ┆ tree(Age) ┆ tree(Height) │ + │ --- ┆ --- ┆ --- ┆ --- │ + │ i64 ┆ i64 ┆ f64 ┆ f64 │ + ╞═════╪════════╪═══════════╪══════════════╡ + │ 20 ┆ 164 ┆ 4.533333 ┆ 5.366667 │ + │ 44 ┆ 150 ┆ 6.0 ┆ 5.366667 │ + │ 19 ┆ 178 ┆ 4.533333 ┆ 4.133333 │ + │ 33 ┆ 158 ┆ 4.533333 ┆ 5.366667 │ + │ 51 ┆ 188 ┆ 6.0 ┆ 4.4 │ + │ 40 ┆ 190 ┆ 4.533333 ┆ 4.4 │ + │ 41 ┆ 168 ┆ 6.0 ┆ 6.95 │ + │ 37 ┆ 174 ┆ 4.533333 ┆ 4.133333 │ + │ 30 ┆ 176 ┆ 4.533333 ┆ 4.133333 │ + │ 54 ┆ 171 ┆ 6.0 ┆ 6.95 │ + └─────┴────────┴───────────┴──────────────┘ """ def __init__( @@ -223,6 +261,7 @@ def __init__( param_grid: Optional[Dict[str, Union[str, int, float, List[int]]]] = None, regression: bool = True, random_state: int = 0, + n_jobs: Optional[int] = None, missing_values: str = "raise", drop_original: bool = False, ) -> None: @@ -251,21 +290,22 @@ def __init__( self.param_grid = param_grid self.regression = regression self.random_state = random_state + self.n_jobs = n_jobs self.missing_values = missing_values self.drop_original = drop_original - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Fits decision trees based on the input variable combinations with cross-validation and grid-search for hyperparameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series or np.array = [n_samples,] + y: Series or np.array = [n_samples,] The target variable that is used to train the decision tree. """ # confirm model type and target variables are compatible. @@ -280,7 +320,7 @@ def fit(self, X: pd.DataFrame, y: pd.Series): check_classification_targets(y) self._is_binary = type_of_target(y) - X, y = check_X_y(X, y) + _, y = check_X_y(X, y) # find or check for numerical variables if self.variables is None: @@ -302,47 +342,54 @@ def fit(self, X: pd.DataFrame, y: pd.Series): how_to_combine=self.features_to_combine, variables=variables_ ) - estimators_ = [] - for features in input_features: - estimator = self._make_decision_tree(param_grid=param_grid) + nw_X = nw.from_native(X, eager_only=True) + X_subs = [] + for features in input_features: # single feature models - if isinstance(features, str): - estimator.fit(X[features].to_frame(), y) + if isinstance(features, (str, int)): + X_sub = nw_X.get_column(features).to_frame().to_native() # multi feature models + elif nwd.is_pandas_dataframe(X) is True: + X_sub = X[features] else: - estimator.fit(X[features], y) + X_sub = nw_X.select(features).to_native() + X_subs.append(X_sub) - estimators_.append(estimator) + estimators_ = Parallel(n_jobs=self.n_jobs, prefer="threads")( + delayed(self._fit_one_tree)(X_sub, y, param_grid) for X_sub in X_subs + ) self.variables_ = variables_ self.input_features_ = input_features self.estimators_ = estimators_ - self.feature_names_in_ = X.columns.tolist() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw_X.columns self.n_features_in_ = X.shape[1] return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Create and add new variables. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe. + X_new: dataframe. Either the original dataframe plus the new features or a dataframe of only the new features. """ # Check method fit has been called check_is_fitted(self) - # check that input is a dataframe - X = check_X(X) + check_X(X) # Check if input data contains same number of columns as dataframe used to fit. _check_X_matches_training_df(X, self.n_features_in_) @@ -352,49 +399,61 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: _check_contains_inf(X, self.variables_) # reorder variables to match train set - X = X[self.feature_names_in_] - - # create new features and add them to the original dataframe - # if regression or multiclass, we return the output of predict() - if self.regression is True: - for features, estimator in zip(self.input_features_, self.estimators_): - if isinstance(features, str): - preds = estimator.predict(X[features].to_frame()) - if self.precision is not None: - preds = np.round(preds, self.precision) - X.loc[:, f"tree({features})"] = preds - else: - preds = estimator.predict(X[features]) - if self.precision is not None: - preds = np.round(preds, self.precision) - X.loc[:, f"tree({features})"] = preds - + if nwd.is_pandas_dataframe(X) is True: + X = X[self.feature_names_in_] + else: + X = ( + nw.from_native(X, eager_only=True) + .select(self.feature_names_in_) + .to_native() + ) + nw_X = nw.from_native(X, eager_only=True) + + def get_x_sub(features): + if isinstance(features, (str, int)): + return nw_X.get_column(features).to_frame().to_native() + if nwd.is_pandas_dataframe(X) is True: + return X[features] + return nw_X.select(features).to_native() + + new_series = [] + new_columns = {} + # if regression or multiclass, we return the output of predict(); # if binary classification, we return the probability - elif self._is_binary == "binary": - for features, estimator in zip(self.input_features_, self.estimators_): - if isinstance(features, str): - preds = estimator.predict_proba(X[features].to_frame()) - if self.precision is not None: - preds = np.round(preds, self.precision) - X.loc[:, f"tree({features})"] = preds[:, 1] - else: - preds = estimator.predict_proba(X[features]) - if self.precision is not None: - preds = np.round(preds, self.precision) - X.loc[:, f"tree({features})"] = preds[:, 1] + for features, estimator in zip(self.input_features_, self.estimators_): + X_sub = get_x_sub(features) + col_name = f"tree({features})" + + if self.regression is True: + preds = estimator.predict(X_sub) + if self.precision is not None: + preds = np.round(preds, self.precision) + elif self._is_binary == "binary": + preds = estimator.predict_proba(X_sub)[:, 1] + if self.precision is not None: + preds = np.round(preds, self.precision) + else: + preds = estimator.predict(X_sub) - # if multiclass, we return the output of predict() - else: - for features, estimator in zip(self.input_features_, self.estimators_): - if isinstance(features, str): - preds = estimator.predict(X[features].to_frame()) - X.loc[:, f"tree({features})"] = preds - else: - preds = estimator.predict(X[features]) - X.loc[:, f"tree({features})"] = preds + if nwd.is_pandas_dataframe(X) is True: + new_columns[col_name] = preds + else: + new_series.append( + nw.new_series(col_name, preds, backend=nw_X.implementation) + ) - if self.drop_original: - X.drop(columns=self.variables_, inplace=True) + if nwd.is_pandas_dataframe(X) is True: + # assign() still inserts columns one at a time internally, so it + # doesn't avoid fragmentation with many feature combinations; + # building one DataFrame and joining it does (single insertion). + X = X.join(type(X)(new_columns, index=X.index)) + if self.drop_original is True: + X = X.drop(columns=self.variables_) + else: + nw_X = nw.from_native(X, eager_only=True).with_columns(*new_series) + if self.drop_original is True: + nw_X = nw_X.drop(self.variables_) + X = nw_X.to_native() return X @@ -414,6 +473,12 @@ def _make_decision_tree(self, param_grid: Dict): return tree_model + def _fit_one_tree(self, X_sub: IntoDataFrame, y: IntoSeries, param_grid: Dict): + """Instantiate and fit one decision tree on one feature combination.""" + estimator = self._make_decision_tree(param_grid=param_grid) + estimator.fit(X_sub, y) + return estimator + def _create_variable_combinations( self, variables: List, diff --git a/feature_engine/creation/geo_features.py b/feature_engine/creation/geo_features.py index bb2698d07..274a26bfa 100644 --- a/feature_engine/creation/geo_features.py +++ b/feature_engine/creation/geo_features.py @@ -3,8 +3,10 @@ from typing import List, Literal, Optional, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -142,10 +144,39 @@ class GeoDistanceFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMix >>> gdt.fit(X) >>> X = gdt.transform(X) >>> X - origin_lat origin_lon dest_lat dest_lon geo_distance - 0 40.7128 -74.0060 34.0522 -118.2437 3935.746254 - 1 34.0522 -118.2437 41.8781 -87.6298 2808.517344 - 2 41.8781 -87.6298 40.7128 -74.0060 1144.286561 + origin_lat origin_lon dest_lat dest_lon geo_distance + 0 40.7128 -74.0060 34.0522 -118.2437 3935.746255 + 1 34.0522 -118.2437 41.8781 -87.6298 2803.971507 + 2 41.8781 -87.6298 40.7128 -74.0060 1144.291274 + + With polars: + + >>> import polars as pl + >>> from feature_engine.creation import GeoDistanceFeatures + >>> X = pl.DataFrame({ + ... "origin_lat": [40.7128, 34.0522, 41.8781], + ... "origin_lon": [-74.0060, -118.2437, -87.6298], + ... "dest_lat": [34.0522, 41.8781, 40.7128], + ... "dest_lon": [-118.2437, -87.6298, -74.0060], + ... }) + >>> gdt = GeoDistanceFeatures( + ... lat1="origin_lat", lon1="origin_lon", + ... lat2="dest_lat", lon2="dest_lon", + ... method="haversine", output_unit="km" + ... ) + >>> gdt.fit(X) + >>> X = gdt.transform(X) + >>> X + shape: (3, 5) + ┌────────────┬────────────┬──────────┬───────────┬──────────────┐ + │ origin_lat ┆ origin_lon ┆ dest_lat ┆ dest_lon ┆ geo_distance │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞════════════╪════════════╪══════════╪═══════════╪══════════════╡ + │ 40.7128 ┆ -74.006 ┆ 34.0522 ┆ -118.2437 ┆ 3935.746255 │ + │ 34.0522 ┆ -118.2437 ┆ 41.8781 ┆ -87.6298 ┆ 2803.971507 │ + │ 41.8781 ┆ -87.6298 ┆ 40.7128 ┆ -74.006 ┆ 1144.291274 │ + └────────────┴────────────┴──────────┴───────────┴──────────────┘ """ def __init__( @@ -213,16 +244,16 @@ def __init__( self.drop_original = drop_original self.validate_ranges = validate_ranges - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. - y: pandas Series, or np.array. Defaults to None. + y: Series, or np.array. Defaults to None. It is not needed in this transformer. You can pass y or None. Returns @@ -231,8 +262,7 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): The fitted transformer. """ - # check input dataframe - X = check_X(X) + check_X(X) # Coordinate variables variables: List[Union[str, int]] = [ @@ -243,7 +273,11 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): ] # Check all coordinate columns exist - missing = set(variables) - set(X.columns) + if nwd.is_pandas_dataframe(X) is True: + columns = set(X.columns) + else: + columns = set(nw.from_native(X, eager_only=True).columns) + missing = set(variables) - columns if missing: raise ValueError( f"Coordinate columns {missing} are not present in the dataframe." @@ -256,50 +290,68 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): _check_contains_na(X, variables) # Validate coordinate ranges if enabled - if self.validate_ranges: - for lat_col in [self.lat1, self.lat2]: - if (X[lat_col].abs() > 90).any(): - raise ValueError( - f"Latitude values in '{lat_col}' must be between -90 and 90." - ) - - for lon_col in [self.lon1, self.lon2]: - if (X[lon_col].abs() > 180).any(): - raise ValueError( - f"Longitude values in '{lon_col}' must be between -180 and 180." - ) + if self.validate_ranges is True: + self._validate_coordinate_ranges(X) # save coordinate variables self.variables_ = variables # save input features - self.feature_names_in_ = X.columns.tolist() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns # save train set shape self.n_features_in_ = X.shape[1] return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def _validate_coordinate_ranges(self, X: IntoDataFrame) -> None: + """Raise if any latitude/longitude value falls outside its valid range.""" + if nwd.is_pandas_dataframe(X) is True: + for lat_col in [self.lat1, self.lat2]: + if (X[lat_col].abs() > 90).any(): + raise ValueError( + f"Latitude values in '{lat_col}' must be between -90 and 90." + ) + for lon_col in [self.lon1, self.lon2]: + if (X[lon_col].abs() > 180).any(): + raise ValueError( + f"Longitude values in '{lon_col}' must be between -180 and 180." + ) + else: + nw_X = nw.from_native(X, eager_only=True) + for lat_col in [self.lat1, self.lat2]: + if nw_X.select((nw.col(lat_col).abs() > 90).any()).to_numpy().any(): + raise ValueError( + f"Latitude values in '{lat_col}' must be between -90 and 90." + ) + for lon_col in [self.lon1, self.lon2]: + if nw_X.select((nw.col(lon_col).abs() > 180).any()).to_numpy().any(): + raise ValueError( + f"Longitude values in '{lon_col}' must be between -180 and 180." + ) + + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Calculate distances and add them as a new column. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the new distance column added. """ # Check method fit has been called check_is_fitted(self) - # check that input is a dataframe - X = check_X(X) + check_X(X) # Check if input data contains same number of columns as dataframe used to fit. _check_X_matches_training_df(X, self.n_features_in_) @@ -307,36 +359,41 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: # Check for missing values _check_contains_na(X, self.variables_) - # reorder variables to match train set - X = X[self.feature_names_in_] + # reorder variables to match train set, and extract coordinate arrays + if nwd.is_pandas_dataframe(X) is True: + X = X[self.feature_names_in_] + lat1 = X[self.lat1].to_numpy() + lon1 = X[self.lon1].to_numpy() + lat2 = X[self.lat2].to_numpy() + lon2 = X[self.lon2].to_numpy() + else: + nw_X = nw.from_native(X, eager_only=True).select(self.feature_names_in_) + lat1 = nw_X.get_column(self.lat1).to_numpy() + lon1 = nw_X.get_column(self.lon1).to_numpy() + lat2 = nw_X.get_column(self.lat2).to_numpy() + lon2 = nw_X.get_column(self.lon2).to_numpy() # Calculate distance based on method if self.method == "haversine": - distances = self._haversine_distance( - X[self.lat1].values, - X[self.lon1].values, - X[self.lat2].values, - X[self.lon2].values, - ) + distances = self._haversine_distance(lat1, lon1, lat2, lon2) elif self.method == "euclidean": - distances = self._euclidean_distance( - X[self.lat1].values, - X[self.lon1].values, - X[self.lat2].values, - X[self.lon2].values, - ) + distances = self._euclidean_distance(lat1, lon1, lat2, lon2) else: # manhattan - distances = self._manhattan_distance( - X[self.lat1].values, - X[self.lon1].values, - X[self.lat2].values, - X[self.lon2].values, - ) - - X[self.output_col] = distances + distances = self._manhattan_distance(lat1, lon1, lat2, lon2) - if self.drop_original: - X = X.drop(columns=self.variables_) + if nwd.is_pandas_dataframe(X) is True: + X[self.output_col] = distances + if self.drop_original is True: + X = X.drop(columns=self.variables_) + else: + nw_X = nw_X.with_columns( + nw.new_series( + self.output_col, distances, backend=nw_X.implementation + ) + ) + if self.drop_original is True: + nw_X = nw_X.drop(self.variables_) + X = nw_X.to_native() return X diff --git a/feature_engine/creation/math_features.py b/feature_engine/creation/math_features.py index bea520bb1..a4d43563b 100644 --- a/feature_engine/creation/math_features.py +++ b/feature_engine/creation/math_features.py @@ -1,8 +1,10 @@ import warnings from typing import Any, List, Optional, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, @@ -21,7 +23,10 @@ from feature_engine._docstrings.substitute import Substitution from feature_engine.creation.base_creation import BaseCreation -_PANDAS_LT_3 = int(pd.__version__.split(".")[0]) < 3 + +def _pandas_version() -> int: + return int(nwd.get_pandas().__version__.split(".")[0]) + # In pandas < 3, agg() maps these callables to the pandas methods and warns that # this will change; the string alias keeps that behaviour (e.g., np.std -> @@ -83,7 +88,11 @@ class MathFeatures(BaseCreation): """ MathFeatures() applies functions across multiple features returning one or more additional features as a result. Common reductions use vectorized NumPy - operations. Other functions fall back to `pandas.agg()` with `axis=1`. + operations. Other functions fall back to `pandas.agg()` with `axis=1` for + pandas input, or to polars' native `map_rows()` for polars input — in that + case, the callable receives each row as a plain tuple, not a `Series`, so + it must not rely on `Series` methods (e.g. use `max(row)` instead of + `row.max()`) to work on both backends. For supported aggregation functions, see `pandas documentation `_. @@ -174,11 +183,30 @@ class MathFeatures(BaseCreation): >>> mf = MathFeatures(variables = ["x1","x2"], func = "mean") >>> mf.fit(X) - >>> mf.transform(X)) + >>> mf.transform(X) x1 x2 mean_x1_x2 0 1 4 2.5 1 2 5 3.5 2 3 6 4.5 + + With polars: + + >>> import polars as pl + >>> from feature_engine.creation import MathFeatures + >>> X = pl.DataFrame({"x1": [1, 2, 3], "x2": [4, 5, 6]}) + >>> mf = MathFeatures(variables=["x1", "x2"], func="sum") + >>> mf.fit(X) + >>> mf.transform(X) + shape: (3, 3) + ┌─────┬─────┬───────────┐ + │ x1 ┆ x2 ┆ sum_x1_x2 │ + │ --- ┆ --- ┆ --- │ + │ i64 ┆ i64 ┆ i64 │ + ╞═════╪═════╪═══════════╡ + │ 1 ┆ 4 ┆ 5 │ + │ 2 ┆ 5 ┆ 7 │ + │ 3 ┆ 6 ┆ 9 │ + └─────┴─────┴───────────┘ """ def __init__( @@ -237,18 +265,18 @@ def __init__( self.func = func self.new_variables_names = new_variables_names - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Create and add new variables. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe, shape = [n_samples, n_features + n_operations] + X_new: dataframe, shape = [n_samples, n_features + n_operations] The input dataframe plus the new variables. """ X = self._check_transform_input_and_state(X) @@ -256,41 +284,71 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: new_variable_names = self._get_new_features_name() func = self.func - if _PANDAS_LT_3: + if nwd.is_pandas_dataframe(X) is True and _pandas_version() < 3: if isinstance(func, list): func = [_FUNC_TO_STRING_ALIAS.get(fun, fun) for fun in func] else: func = _FUNC_TO_STRING_ALIAS.get(func, func) - variables = X[self.variables] functions = func if isinstance(func, list) else [func] reducers = [_get_numpy_reducer(fun) for fun in functions] - values = variables.to_numpy() + + nw_X = nw.from_native(X, eager_only=True) + if nwd.is_pandas_dataframe(X) is True: + values = X[self.variables].to_numpy() + else: + values = nw_X.select(self.variables).to_numpy() # Nullable extension dtypes produce object arrays. Keep those, custom - # callables, and less common pandas aggregations on the exact legacy path. + # callables, and less common aggregations on the fallback path below. if reducers and values.dtype.kind in "biuf" and all(reducers): - results = [] - for reducer, kwargs in reducers: + new_series = [] + for (reducer, kwargs), name in zip(reducers, new_variable_names): # pandas' named reductions do not warn for empty/all-missing rows. # NumPy returns the same values but emits RuntimeWarning for some # reducers, so silence only those warnings on this equivalent path. with warnings.catch_warnings(): warnings.simplefilter("ignore", RuntimeWarning) result = reducer(values, axis=1, **kwargs) - results.append(pd.Series(result, index=X.index)) - - result = results[0] if len(results) == 1 else pd.concat(results, axis=1) - else: - result = variables.agg(func, axis=1) - - if len(new_variable_names) == 1: - X[new_variable_names[0]] = result + new_series.append( + nw.new_series(name, result, backend=nw_X.implementation) + ) + nw_X = nw_X.with_columns(*new_series) + if self.drop_original is True: + nw_X = nw_X.drop(self.variables) + X = nw_X.to_native() + elif nwd.is_pandas_dataframe(X) is True: + result = X[self.variables].agg(func, axis=1) + if len(new_variable_names) == 1: + X[new_variable_names[0]] = result + else: + X[new_variable_names] = result + if self.drop_original is True: + X = X.drop(columns=self.variables) else: - X[new_variable_names] = result - - if self.drop_original: - X.drop(columns=self.variables, inplace=True) + # polars has no equivalent to pandas' agg(func, axis=1): apply each + # function natively via map_rows, one call per function. map_rows + # passes each row as a plain tuple, not a Series, so callables that + # rely on Series methods (e.g. `row.max()`) need `max(row)` instead. + sub_native = nw_X.select(self.variables).to_native() + new_series = [] + for fun, name in zip(functions, new_variable_names): + if not callable(fun): + raise NotImplementedError( + f"'{fun}' has no NumPy-vectorized implementation, and " + "non-callable aggregation names are not supported for " + "polars input. Pass a Python callable instead." + ) + result_df = sub_native.map_rows(fun) + new_series.append( + nw.new_series( + name, result_df.to_series(0), backend=nw_X.implementation + ) + ) + nw_X = nw_X.with_columns(*new_series) + if self.drop_original is True: + nw_X = nw_X.drop(self.variables) + X = nw_X.to_native() return X diff --git a/feature_engine/creation/relative_features.py b/feature_engine/creation/relative_features.py index 5b2957bff..d664c2448 100644 --- a/feature_engine/creation/relative_features.py +++ b/feature_engine/creation/relative_features.py @@ -1,6 +1,8 @@ from typing import List, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, @@ -31,6 +33,20 @@ "pow", ] +_NUMPY_OPS = { + "add": np.add, + "sub": np.subtract, + "mul": np.multiply, + "div": np.divide, + "truediv": np.true_divide, + "floordiv": np.floor_divide, + "mod": np.mod, + "pow": np.power, +} + +# these can divide by zero; fill_value handling applies only to them. +_DIVISION_LIKE = {"div", "truediv", "floordiv", "mod"} + @Substitution( variables=_variables_numerical_docstring, @@ -54,12 +70,10 @@ class RelativeFeatures(BaseCreation): features to / by a group of reference variables. The features resulting from these functions are added to the dataframe. - This transformer works only with numerical variables. It uses the pandas methods - `pd.DataFrame.add`, `pd.DataFrame.sub`, `pd.DataFrame.mul`, `pd.DataFrame.div`, - `pd.DataFrame.truediv`, `pd.DataFrame.floordiv`, `pd.DataFrame.mod` and - `pd.DataFrame.pow`. - Find out more in `pandas documentation - `_. + This transformer works only with numerical variables. It uses NumPy's `add`, + `subtract`, `multiply`, `divide`, `true_divide`, `floor_divide`, `mod` and + `power` under the hood, matching the semantics of the equivalent pandas + `DataFrame.add`, `DataFrame.sub`, etc. methods. More details in the :ref:`User Guide `. @@ -125,6 +139,27 @@ class RelativeFeatures(BaseCreation): 0 1 4 3 0.333333 1.333333 1 2 5 4 0.500000 1.250000 2 3 6 5 0.600000 1.200000 + + With polars: + + >>> import polars as pl + >>> from feature_engine.creation import RelativeFeatures + >>> X = pl.DataFrame({"x1": [1, 2, 3], "x2": [4, 5, 6], "x3": [3, 4, 5]}) + >>> rf = RelativeFeatures(variables=["x1", "x2"], + >>> reference=["x3"], + >>> func=["div"]) + >>> rf.fit(X) + >>> rf.transform(X) + shape: (3, 5) + ┌─────┬─────┬─────┬───────────┬───────────┐ + │ x1 ┆ x2 ┆ x3 ┆ x1_div_x3 ┆ x2_div_x3 │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ i64 ┆ i64 ┆ i64 ┆ f64 ┆ f64 │ + ╞═════╪═════╪═════╪═══════════╪═══════════╡ + │ 1 ┆ 4 ┆ 3 ┆ 0.333333 ┆ 1.333333 │ + │ 2 ┆ 5 ┆ 4 ┆ 0.5 ┆ 1.25 │ + │ 3 ┆ 6 ┆ 5 ┆ 0.6 ┆ 1.2 │ + └─────┴─────┴─────┴───────────┴───────────┘ """ def __init__( @@ -179,124 +214,70 @@ def __init__( self.func = func self.fill_value = fill_value - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Add new features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe + X_new: dataframe The input dataframe plus the new variables. """ X = self._check_transform_input_and_state(X) - methods_dict = { - "add": self._add, - "mul": self._mul, - "sub": self._sub, - "div": self._div, - "truediv": self._truediv, - "floordiv": self._floordiv, - "mod": self._mod, - "pow": self._pow, - } + nw_X = nw.from_native(X, eager_only=True) + # Extract each column as its own 1D array (not one batched 2D array + # via select().to_numpy()) so mixed int/float variables each keep + # their own dtype promotion, matching pandas' per-column .sub()/ + # .div()/etc. instead of upcasting everything to a common dtype. + var_arrays = {var: nw_X.get_column(var).to_numpy() for var in self.variables} + ref_arrays = {ref: nw_X.get_column(ref).to_numpy() for ref in self.reference} + new_series = [] for func in self.func: - methods_dict[func](X) - - if self.drop_original: - X.drop( - columns=set(self.variables + self.reference), - inplace=True, - ) - - return X - - def _sub(self, X): - for reference in self.reference: - varname = [f"{var}_sub_{reference}" for var in self.variables] - X[varname] = X[self.variables].sub(X[reference], axis=0) - return X - - def _add(self, X): - for reference in self.reference: - varname = [f"{var}_add_{reference}" for var in self.variables] - X[varname] = X[self.variables].add(X[reference], axis=0) - return X - - def _mul(self, X): - for reference in self.reference: - varname = [f"{var}_mul_{reference}" for var in self.variables] - X[varname] = X[self.variables].mul(X[reference], axis=0) - return X - - def _div(self, X): - for reference in self.reference: - zeros_ix, contains_zero = self._find_zeroes_in_reference(X, reference) - - if self.fill_value is None and contains_zero: - self._raise_error_when_zero_in_denominator() - - varname = [f"{var}_div_{reference}" for var in self.variables] - X[varname] = X[self.variables].div(X[reference], axis=0) - - if contains_zero: - X.loc[zeros_ix, varname] = self.fill_value - return X - - def _truediv(self, X): - for reference in self.reference: - zeros_ix, contains_zero = self._find_zeroes_in_reference(X, reference) - - if self.fill_value is None and contains_zero: - self._raise_error_when_zero_in_denominator() - - varname = [f"{var}_truediv_{reference}" for var in self.variables] - X[varname] = X[self.variables].truediv(X[reference], axis=0) - - if contains_zero: - X.loc[zeros_ix, varname] = self.fill_value - return X - - def _floordiv(self, X): - for reference in self.reference: - zeros_ix, contains_zero = self._find_zeroes_in_reference(X, reference) - - if self.fill_value is None and contains_zero: - self._raise_error_when_zero_in_denominator() - - varname = [f"{var}_floordiv_{reference}" for var in self.variables] - X[varname] = X[self.variables].floordiv(X[reference], axis=0) - - if contains_zero: - X.loc[zeros_ix, varname] = self.fill_value - return X - - def _mod(self, X): - for reference in self.reference: - zeros_ix, contains_zero = self._find_zeroes_in_reference(X, reference) - - if self.fill_value is None and contains_zero: - self._raise_error_when_zero_in_denominator() - - varname = [f"{var}_mod_{reference}" for var in self.variables] - X[varname] = X[self.variables].mod(X[reference], axis=0) - - if contains_zero: - X.loc[zeros_ix, varname] = self.fill_value - return X - - def _pow(self, X): - for reference in self.reference: - varname = [f"{var}_pow_{reference}" for var in self.variables] - X[varname] = X[self.variables].pow(X[reference], axis=0) - return X + op = _NUMPY_OPS[func] + for reference in self.reference: + ref_arr = ref_arrays[reference] + + if func in _DIVISION_LIKE: + zero_mask = ref_arr == 0 + contains_zero = zero_mask.any() + if self.fill_value is None and contains_zero: + self._raise_error_when_zero_in_denominator() + + for var in self.variables: + name = f"{var}_{func}_{reference}" + if func in _DIVISION_LIKE: + with np.errstate(divide="ignore", invalid="ignore"): + result = op(var_arrays[var], ref_arr) + if contains_zero: + # floordiv/mod on integer input stay integer-typed; + # widen to match fill_value if it wouldn't fit, + # mirroring pandas' automatic dtype promotion. + fill_arr = np.asarray(self.fill_value) + if not np.can_cast(fill_arr, result.dtype, casting="safe"): + result = result.astype( + np.result_type(result.dtype, fill_arr.dtype) + ) + result[zero_mask] = self.fill_value + else: + result = op(var_arrays[var], ref_arr) + + new_series.append( + nw.new_series(name, result, backend=nw_X.implementation) + ) + + nw_X = nw_X.with_columns(*new_series) + if self.drop_original is True: + nw_X = nw_X.drop(list(set(self.variables + self.reference))) + + return nw_X.to_native() def _raise_error_when_zero_in_denominator(self): raise ValueError( @@ -305,11 +286,6 @@ def _raise_error_when_zero_in_denominator(self): "or set `fill_value` to a number." ) - def _find_zeroes_in_reference(self, X, var): - zero_ix = X[var] == 0 - zero_bool = (zero_ix).any() - return zero_ix, zero_bool - def _get_new_features_name(self) -> List: """Return names of the created features.""" diff --git a/feature_engine/dataframe_checks.py b/feature_engine/dataframe_checks.py index fe313a627..efd207b8f 100644 --- a/feature_engine/dataframe_checks.py +++ b/feature_engine/dataframe_checks.py @@ -4,118 +4,84 @@ from typing import List, Tuple, Union +import narwhals as nw +import narwhals.dependencies as nwd +import narwhals.selectors as nws import numpy as np -import pandas as pd -from scipy.sparse import issparse +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.utils.validation import _check_y, check_consistent_length, column_or_1d -from feature_engine.variable_handling._variable_type_checks import is_object - -def check_X(X: Union[np.generic, np.ndarray, pd.DataFrame]) -> pd.DataFrame: +def check_X(X: IntoDataFrame) -> IntoDataFrame: """ - Checks if the input is a DataFrame and then creates a copy. This is an important - step not to accidentally transform the original dataset entered by the user. - - If the input is a numpy array, it converts it to a pandas Dataframe. The column - names are strings representing the column index starting at 0. - - Feature-engine was originally designed to work with pandas dataframes. However, - allowing numpy arrays as input allows 2 things: - - We can use the Scikit-learn tests for transformers provided by the - `check_estimator` function to test the compatibility of our transformers with - sklearn functionality. - - Feature-engine transformers can be used within a Scikit-learn Pipeline together - with Scikit-learn transformers like the `SimpleImputer`, which return by default - Numpy arrays. + Checks that X is a dataframe from any library supported by narwhals (for example + pandas, polars, modin, cuDF, or PyArrow). Parameters ---------- - X : pandas Dataframe or numpy array. - The input to check and copy or transform. + X : dataframe (pandas, polars, PyArrow, modin, or cuDF). Feature-engine does + not support libraries that build a deferred query plan (for example Dask, + DuckDB, PySpark, Ibis, or a polars LazyFrame). Convert those to an eager + dataframe (e.g. `LazyFrame.collect()`) before passing them in. + The input to check and transform. Raises ------ TypeError - If the input is not a Pandas DataFrame or a numpy array. + If the input is not a recognised dataframe. ValueError - If the input is an empty dataframe. + If the input has duplicated column names, or 0 columns or rows. Returns ------- - X : pandas Dataframe. - A copy of original DataFrame or a converted Numpy array. + X : narwhals dataframe. + The validated dataframe in narwhals format. """ - if isinstance(X, pd.DataFrame): - if not X.columns.is_unique: - raise ValueError("Input data contains duplicated variable names.") - X = X.copy() - - elif isinstance(X, (np.generic, np.ndarray)): - # If input is scalar raise error - if X.ndim == 0: - raise ValueError( - "Expected 2D array, got scalar array instead:\narray={}.\n" - "Reshape your data either using array.reshape(-1, 1) if " - "your data has a single feature or array.reshape(1, -1) " - "if it contains a single sample.".format(X) - ) - # If input is 1D raise error - if X.ndim == 1: + if nwd.is_into_dataframe(X): + # from_native() raises narwhals.exceptions.DuplicateError, a ValueError + # subclass, when the dataframe has duplicated column names. + nw_X = nw.from_native(X, eager_only=True) + if nw_X.is_empty() or nw_X.shape[1] == 0: raise ValueError( - "Expected 2D array, got 1D array instead:\narray={}.\n" - "Reshape your data either using array.reshape(-1, 1) if " - "your data has a single feature or array.reshape(1, -1) " - "if it contains a single sample.".format(X) + f"Found array with 0 feature(s) (shape={nw_X.shape}) while a " + "minimum of 1 is required." ) - if np.any(np.iscomplex(X)): - raise TypeError("Complex data not supported by this transformer.") - - X = pd.DataFrame(X) - X.columns = [f"x{i}" for i in range(X.shape[1])] - - elif issparse(X): - raise TypeError("This transformer does not support sparse matrices.") - else: raise TypeError( - f"X must be a numpy array or pandas dataframe. Got {type(X)} instead." - ) - - if X.empty: - raise ValueError( - "0 feature(s) (shape=%s) while a minimum of %d is required." % (X.shape, 1) + "X must be a dataframe from a library supported by narwhals " + f"(e.g. pandas, polars, PyArrow). Got {type(X)} instead." ) - return X + return nw_X def check_y( - y: Union[np.generic, np.ndarray, pd.Series, pd.DataFrame, List], + y: Union[IntoSeries, IntoDataFrame, np.generic, np.ndarray, List], y_numeric: bool = False, -) -> pd.Series: +): """ - Checks that y is a series or a dataframe, or alternatively, if it can be converted - to a series or dataframe. + Checks that y is a Series or DataFrame from a library supported by narwhals (for + example pandas or polars), or alternatively, if it can be converted to a numpy + array. Parameters ---------- - y : pd.Series, pd.DataFrame, np.array, list - The input to check and copy or transform. + y : Series or DataFrame (pandas, polars, PyArrow, modin, or cuDF), np.array, + list. Feature-engine does not support libraries that build a deferred + query plan (for example Dask, DuckDB, PySpark, Ibis, or a polars + LazyFrame). Convert those to an eager dataframe (e.g. `LazyFrame.collect()`) + before passing them in. + The input to check. y_numeric : bool, default=False - Whether to ensure that y has a numeric type. If dtype of y is object, - it is converted to float64. Should only be used for regression - algorithms. + Whether to ensure that y has a numeric type. If dtype of y is not numeric, + it is cast to float64. Should only be used for regression algorithms. Returns ------- - y: pd.Series or pd.DataFrame + y: Series, DataFrame, or numpy array """ - if y is None: raise ValueError( "requires y to be passed, but the target y is None", @@ -123,110 +89,97 @@ def check_y( "y should be a 1d array", ) - elif isinstance(y, pd.Series): - if y.isnull().any(): + if nwd.is_into_series(y): + nw_y = nw.from_native(y, series_only=True) + if nw_y.is_null().any() or ( + nw_y.dtype.is_numeric() and nw_y.is_nan().any() + ): raise ValueError("y contains NaN values.") - if not is_object(y) and not np.isfinite(y).all(): - raise ValueError("y contains infinity values.") - if y_numeric and is_object(y): - y = y.astype("float64") - y = y.copy() - - elif isinstance(y, pd.DataFrame): - if y.isnull().any().any(): + if nw_y.dtype.is_numeric(): + if not np.isfinite(nw_y.to_numpy()).all(): + raise ValueError("y contains infinity values.") + elif y_numeric: + nw_y = nw_y.cast(nw.Float64()) + return nw_y.to_native() + + if nwd.is_into_dataframe(y): + nw_y = nw.from_native(y, eager_only=True) + if ( + nw_y.select(nw.all().is_null().any()).to_numpy().any() + or nw_y.select(nws.numeric().is_nan().any()).to_numpy().any() + ): raise ValueError("y contains NaN values.") - if not np.isfinite(y).all().all(): + if not np.isfinite(nw_y.to_numpy()).all(): raise ValueError("y contains infinity values.") - y = y.copy() + return nw_y.to_native() - else: - try: - y = column_or_1d(y) - y = _check_y(y, multi_output=False, y_numeric=y_numeric) - y = pd.Series(y).copy() - except ValueError: - y = _check_y(y, multi_output=True, y_numeric=y_numeric) - y = pd.DataFrame(y).copy() - return y + try: + y = column_or_1d(y) + return _check_y(y, multi_output=False, y_numeric=y_numeric) + except ValueError: + return _check_y(y, multi_output=True, y_numeric=y_numeric) def check_X_y( - X: Union[np.generic, np.ndarray, pd.DataFrame], - y: Union[np.generic, np.ndarray, pd.Series, List], + X: IntoDataFrame, + y: Union[IntoSeries, IntoDataFrame, np.generic, np.ndarray, List], y_numeric: bool = False, -) -> Tuple[pd.DataFrame, pd.Series]: +) -> Tuple[IntoDataFrame, Union[IntoSeries, IntoDataFrame, np.ndarray]]: """ - Ensures X and y are compatible pandas DataFrame and Series. If both are pandas - objects, checks that their indexes match. If any is a numpy array, converts to - pandas object with compatible index. - - This transformer ensures that we can concatenate X and y using `pandas.concat`, - functionality needed in the encoders. + Ensures X and y are compatible dataframe/array-like objects with a consistent + number of rows. If both are pandas objects, checks that their indexes match. Parameters ---------- - X: Pandas DataFrame or numpy ndarray - The input to check and copy or transform. + X: dataframe (pandas, polars, PyArrow, modin, or cuDF). Feature-engine does + not support libraries that build a deferred query plan (for example Dask, + DuckDB, PySpark, Ibis, or a polars LazyFrame). Convert those to an eager + dataframe (e.g. `LazyFrame.collect()`) before passing them in. + The input to check. - y: pd.Series, np.array, list - The input to check and copy or transform. + y: Series, DataFrame (pandas, polars, or any other library supported by + narwhals), np.array, list + The input to check. y_numeric : bool, default=False - Whether to ensure that y has a numeric type. If dtype of y is object, - it is converted to float64. Should only be used for regression - algorithms. + Whether to ensure that y has a numeric type. If dtype of y is not numeric, + it is cast to float64. Should only be used for regression algorithms. Raises ------ - ValueError: if X and y are pandas objects with inconsistent indexes. - TypeError: if X is sparse matrix, empty dataframe or not a dataframe. - TypeError: if y can't be parsed as pandas Series. + TypeError + If X is not a recognised dataframe. + ValueError + If X has duplicated column names, 0 columns, or 0 rows; if y is None, or + contains NaN or infinity values; if X and y have a different number of + rows; or if X and y are pandas objects with mismatched indexes. Returns ------- - X: Pandas DataFrame - y: Pandas Series + X: narwhals dataframe + y: Series, DataFrame, or numpy array """ + X = check_X(X) + y = check_y(y, y_numeric=y_numeric) + check_consistent_length(X, y) - def _check_X_y(X, y): - X = check_X(X) - y = check_y(y, y_numeric=y_numeric) - check_consistent_length(X, y) - return X, y - - # case 1: both are pandas objects - if isinstance(X, pd.DataFrame) and isinstance(y, (pd.Series, pd.DataFrame)): - X, y = _check_X_y(X, y) - # Check that their indexes match. - if X.index.equals(y.index) is False: - raise ValueError("The indexes of X and y do not match.") - - # case 2: X is dataframe and y is something else - if isinstance(X, pd.DataFrame) and not isinstance(y, (pd.Series, pd.DataFrame)): - X, y = _check_X_y(X, y) - y.index = X.index - - # case 3: X is not a dataframe and y is a series - elif not isinstance(X, pd.DataFrame) and isinstance(y, (pd.Series, pd.DataFrame)): - X, y = _check_X_y(X, y) - X.index = y.index - - # all other cases - else: - X, y = _check_X_y(X, y) + if X.implementation.is_pandas(): + if nwd.is_pandas_series(y) or nwd.is_pandas_dataframe(y): + if not X.to_native().index.equals(y.index): + raise ValueError("The indexes of X and y do not match.") return X, y -def _check_X_matches_training_df(X: pd.DataFrame, reference: int) -> None: +def _check_X_matches_training_df(X: IntoDataFrame, reference: int) -> None: """ - Checks that DataFrame to transform has the same number of columns that the - DataFrame used with the fit() method. + Checks that the dataframe to transform has the same number of columns as the + dataframe used with the fit() method. Parameters ---------- - X : Pandas DataFrame - The df to be checked + X : dataframe (pandas, polars, or any other library supported by narwhals) + The df to be checked. reference : int The number of columns in the dataframe that was used with the fit() method. @@ -234,92 +187,93 @@ def _check_X_matches_training_df(X: pd.DataFrame, reference: int) -> None: ------ ValueError If the number of columns does not match. - - Returns - ------- - None """ - if X.shape[1] != reference: raise ValueError( "The number of columns in this dataset is different from the one used to " "fit this transformer (when using the fit() method)." ) - return None - def _check_contains_na( - X: pd.DataFrame, + X: IntoDataFrame, variables: List[Union[str, int]], + error_msg: str = "simple", ) -> None: """ - Checks if DataFrame contains null values in the selected columns. + Checks if the dataframe contains null values in the selected columns. Parameters ---------- - X : Pandas DataFrame + X : dataframe variables : List The selected group of variables in which null values will be examined. - Raises - ------ - ValueError - If the variable(s) contain null values. - """ - - if X[variables].isnull().any().any(): - raise ValueError( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer." - ) - - -def _check_optional_contains_na( - X: pd.DataFrame, variables: List[Union[str, int]] -) -> None: - """ - Checks if DataFrame contains null values in the selected columns. - - Parameters - ---------- - X : Pandas DataFrame - - variables : List - The selected group of variables in which null values will be examined. + error_msg : str, default="simple" + The message in the error. Some transformers can ignore null values. Raises ------ ValueError If the variable(s) contain null values. """ - - if X[variables].isnull().any().any(): - raise ValueError( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer or set the parameter " - "`missing_values='ignore'` when initialising this transformer." - ) - - -def _check_contains_inf(X: pd.DataFrame, variables: List[Union[str, int]]) -> None: + error_msg_simple = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." + ) + error_msg_ignore = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." + ) + if len(variables) == 0: + return + + if _contains_na(X, variables) is True: + if error_msg == "simple": + raise ValueError(error_msg_simple) + else: + raise ValueError(error_msg_ignore) + + +def _contains_na(X: IntoDataFrame, variables: List[Union[str, int]]) -> bool: + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return bool(X[variables].isna().to_numpy().any()) + else: + nw_X = nw.from_native(X, eager_only=True) + # polars stores the null count, so check it before scanning floats for NaN. + if sum(nw_X.select(nw.col(variables).null_count()).row(0)) > 0: + return True + schema = nw_X.schema + floats = [var for var in variables if schema[var].is_float() is True] + if len(floats) == 0: + return False + return True in nw_X.select(nw.col(floats).is_nan().any()).row(0) + + +def _check_contains_inf(X: IntoDataFrame, variables: List[Union[str, int]]) -> None: """ - Checks if DataFrame contains inf values in the selected columns. + Checks if the dataframe contains inf values in the selected columns. Parameters ---------- - X : Pandas DataFrame + X : dataframe variables : List - The selected group of variables in which null values will be examined. + The selected group of variables in which infinite values will be examined. Raises ------ ValueError If the variable(s) contain np.inf values """ + # polars can't select an empty list of columns + if len(variables) == 0: + return None - if np.isinf(X[variables]).any().any(): + values = nw.from_native(X, eager_only=True).select(nw.col(variables)).to_numpy() + if np.isinf(values.astype(float)).any(): raise ValueError( "Some of the variables to transform contain inf values. Check and " "remove those before using this transformer." diff --git a/feature_engine/datetime/_datetime_constants.py b/feature_engine/datetime/_datetime_constants.py index 198b6d5a5..5b2cb6755 100644 --- a/feature_engine/datetime/_datetime_constants.py +++ b/feature_engine/datetime/_datetime_constants.py @@ -1,3 +1,4 @@ +import narwhals as nw import numpy as np FEATURES_SUPPORTED = [ @@ -78,3 +79,105 @@ "minute": lambda x: x.dt.minute, "second": lambda x: x.dt.second, } + + +def _nw_quarter(x: nw.Series) -> nw.Series: + return ((x.dt.month() - 1) // 3) + 1 + + +def _nw_semester(x: nw.Series) -> nw.Series: + return (x.dt.month() > 6).cast(nw.Int64()) + 1 + + +def _nw_week(x: nw.Series) -> nw.Series: + # narwhals has no isocalendar(); the "%V" strftime code (ISO week) round-trips + # correctly on every backend tested (pandas, polars) via to_string(). + return x.dt.to_string("%V").cast(nw.Int64()) + + +def _nw_day_of_week(x: nw.Series) -> nw.Series: + # narwhals weekday() is 1=Monday..7=Sunday; pandas dayofweek is 0=Monday..6=Sunday. + return x.dt.weekday() - 1 + + +def _nw_weekend(x: nw.Series) -> nw.Series: + return (_nw_day_of_week(x) >= 5).cast(nw.Int64()) + + +def _nw_is_month_start(x: nw.Series) -> nw.Series: + return x.dt.day() == 1 + + +def _nw_is_month_end(x: nw.Series) -> nw.Series: + # no days_in_month()/is_month_end() in narwhals: a day belongs to the last + # day of its month iff the next day rolls over into a different month. + return x.dt.offset_by("1d").dt.month() != x.dt.month() + + +def _nw_month_start(x: nw.Series) -> nw.Series: + return _nw_is_month_start(x).cast(nw.Int64()) + + +def _nw_month_end(x: nw.Series) -> nw.Series: + return _nw_is_month_end(x).cast(nw.Int64()) + + +def _nw_quarter_start(x: nw.Series) -> nw.Series: + # quarters start in Jan/Apr/Jul/Oct, the only months where month % 3 == 1. + return (_nw_is_month_start(x) & (x.dt.month() % 3 == 1)).cast(nw.Int64()) + + +def _nw_quarter_end(x: nw.Series) -> nw.Series: + # quarters end in Mar/Jun/Sep/Dec, the only months where month % 3 == 0. + return (_nw_is_month_end(x) & (x.dt.month() % 3 == 0)).cast(nw.Int64()) + + +def _nw_year_start(x: nw.Series) -> nw.Series: + return (_nw_is_month_start(x) & (x.dt.month() == 1)).cast(nw.Int64()) + + +def _nw_year_end(x: nw.Series) -> nw.Series: + return (_nw_is_month_end(x) & (x.dt.month() == 12)).cast(nw.Int64()) + + +def _nw_leap_year(x: nw.Series) -> nw.Series: + year = x.dt.year() + return (((year % 4 == 0) & (year % 100 != 0)) | (year % 400 == 0)).cast( + nw.Int64() + ) + + +def _nw_days_in_month(x: nw.Series) -> nw.Series: + # start of month, plus a month, minus a day = last day of the original month; + # its day number is the month's length. Handles leap years automatically. + return x.dt.truncate("1mo").dt.offset_by("1mo").dt.offset_by("-1d").dt.day() + + +# narwhals-native equivalents of FEATURES_FUNCTIONS above, used for dataframe +# backends other than pandas. Kept separate from FEATURES_FUNCTIONS because +# roughly a third of these features (week, month_end, quarter_end, quarter_start, +# year_start, year_end, leap_year, days_in_month) benchmarked 2x-53x slower than +# pandas-native when run through narwhals on a pandas backend, so pandas keeps its +# fast, unchanged native path. +FEATURES_FUNCTIONS_NARWHALS = { + "month": lambda x: x.dt.month(), + "quarter": _nw_quarter, + "semester": _nw_semester, + "year": lambda x: x.dt.year(), + "week": _nw_week, + "day_of_week": _nw_day_of_week, + "day_of_month": lambda x: x.dt.day(), + "day_of_year": lambda x: x.dt.ordinal_day(), + "weekend": _nw_weekend, + "month_start": _nw_month_start, + "month_end": _nw_month_end, + "quarter_start": _nw_quarter_start, + "quarter_end": _nw_quarter_end, + "year_start": _nw_year_start, + "year_end": _nw_year_end, + "leap_year": _nw_leap_year, + "days_in_month": _nw_days_in_month, + "hour": lambda x: x.dt.hour(), + "minute": lambda x: x.dt.minute(), + "second": lambda x: x.dt.second(), +} diff --git a/feature_engine/datetime/datetime.py b/feature_engine/datetime/datetime.py index 106d50277..a78cd751c 100644 --- a/feature_engine/datetime/datetime.py +++ b/feature_engine/datetime/datetime.py @@ -2,9 +2,9 @@ from typing import List, Optional, Union -import pandas as pd -from pandas.api.types import is_datetime64_any_dtype as is_datetime -from pandas.api.types import is_numeric_dtype as is_numeric +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -35,6 +35,7 @@ from feature_engine.datetime._datetime_constants import ( FEATURES_DEFAULT, FEATURES_FUNCTIONS, + FEATURES_FUNCTIONS_NARWHALS, FEATURES_SUFFIXES, FEATURES_SUPPORTED, ) @@ -58,8 +59,9 @@ class DatetimeFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin) new columns to the dataset. DatetimeFeatures can extract datetime information from existing datetime or object-like variables or from the dataframe index. - DatetimeFeatures uses `pandas.to_datetime` to convert object variables to datetime - and pandas.dt to extract the features from datetime. + DatetimeFeatures works with pandas and polars dataframes. `dayfirst`, + `yearfirst` and `utc` are pandas-only parsing options and have no effect on + polars input, so use `format` there instead. The transformer supports the extraction of the following features: @@ -93,7 +95,8 @@ class DatetimeFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin) If None, the transformer will find and select all datetime variables, including variables of type object that can be converted to datetime. If "index", the transformer will extract datetime features from the - index of the dataframe. + index of the dataframe. "index" is only supported when `X` is a pandas + dataframe, since only pandas dataframes have an index. {return_empty} @@ -119,11 +122,14 @@ class DatetimeFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin) dayfirst: bool, default="False" Specify a date parse order if arg is str or is list-like. If True, parses dates with the day first, e.g. 10/11/12 is parsed as 2012-11-10. Same as in - `pandas.to_datetime`. + `pandas.to_datetime`. Only applied when `X` is a pandas dataframe; ignored + for other backends, which have no equivalent parsing option. yearfirst: bool, default="False" Specify a date parse order if arg is str or is list-like. - Same as in `pandas.to_datetime`. + Same as in `pandas.to_datetime`. Only applied when `X` is a pandas + dataframe; ignored for other backends, which have no equivalent parsing + option. - If True parses dates with the year first, e.g. 10/11/12 is parsed as 2010-11-12. @@ -131,7 +137,9 @@ class DatetimeFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin) utc: bool, default=None Return UTC DatetimeIndex if True (converting any tz-aware datetime.datetime - objects as well). Same as in `pandas.to_datetime`. + objects as well). Same as in `pandas.to_datetime`. Only applied when `X` is + a pandas dataframe; ignored for other backends, which have no equivalent + parsing option. format: str, default None The strftime to parse time, e.g. "%d/%m/%Y". Check pandas `to_datetime()` for @@ -182,6 +190,25 @@ class DatetimeFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin) 0 2022 9 18 1 2022 10 27 2 2022 12 24 + + With polars: + + >>> import polars as pl + >>> from feature_engine.datetime import DatetimeFeatures + >>> X = pl.DataFrame(dict(date = ["2022-09-18", "2022-10-27", "2022-12-24"])) + >>> dtf = DatetimeFeatures(features_to_extract = ["year", "month", "day_of_month"]) + >>> dtf.fit(X) + >>> dtf.transform(X) + shape: (3, 3) + ┌───────────┬────────────┬───────────────────┐ + │ date_year ┆ date_month ┆ date_day_of_month │ + │ --- ┆ --- ┆ --- │ + │ i32 ┆ i8 ┆ i8 │ + ╞═══════════╪════════════╪═══════════════════╡ + │ 2022 ┆ 9 ┆ 18 │ + │ 2022 ┆ 10 ┆ 27 │ + │ 2022 ┆ 12 ┆ 24 │ + └───────────┴────────────┴───────────────────┘ """ def __init__( @@ -240,7 +267,7 @@ def __init__( self.features_to_extract = features_to_extract self.format = format - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn any parameter. @@ -249,23 +276,35 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) # special case index if self.variables == "index": + # polars and other narwhals backends have no index concept. + if not nwd.is_pandas_dataframe(X): + raise TypeError( + "variables='index' requires a pandas dataframe, since only " + f"pandas dataframes have an index. Got {type(X)} instead." + ) + pd_ = nw_X.__native_namespace__() + index_is_dt = pd_.api.types.is_datetime64_any_dtype(X.index) + index_is_numeric = pd_.api.types.is_numeric_dtype(X.index) if not ( - is_datetime(X.index) + index_is_dt or ( - not is_numeric(X.index) and _is_categorical_and_is_datetime(X.index) + index_is_numeric is False + and _is_categorical_and_is_datetime( + nw.from_native(pd_.Series(X.index), series_only=True) + ) ) ): raise TypeError("The dataframe index is not datetime.") @@ -294,26 +333,24 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): else: self.features_to_extract_ = self.features_to_extract - # save input features - self.feature_names_in_ = X.columns.tolist() - - # save train set shape - self.n_features_in_ = X.shape[1] + # save input features and train set shape + self.feature_names_in_ = nw_X.columns + self.n_features_in_ = nw_X.shape[1] return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Extract the date and time features and add them to the dataframe. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe, shape = [n_samples, n_features x n_df_features] + X_new: dataframe, shape = [n_samples, n_features x n_df_features] The dataframe with the original variables plus the new variables. """ @@ -321,23 +358,22 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: check_is_fitted(self) # check that input is a dataframe - X = check_X(X) + nw_X = check_X(X) # Check if input data contains same number of columns as dataframe used to fit. _check_X_matches_training_df(X, self.n_features_in_) - # reorder variables to match train set - X = X[self.feature_names_in_] - - # special case index + # special case index: only reachable for pandas, fit() already raised + # TypeError for any other backend, since only pandas has an index. if self.variables == "index": # check if dataset contains na if self.missing_values == "raise": self._check_index_contains_na(X.index) + pd_ = nw_X.__native_namespace__() # convert index to a datetime series - idx_datetime = pd.Series( - pd.to_datetime( + idx_datetime = pd_.Series( + pd_.to_datetime( X.index, dayfirst=self.dayfirst, yearfirst=self.yearfirst, @@ -347,9 +383,14 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: index=X.index, ) - # create new features - for feat in self.features_to_extract_: - X[FEATURES_SUFFIXES[feat][1:]] = FEATURES_FUNCTIONS[feat](idx_datetime) + # add the new features without mutating the input dataframe + new_columns = { + FEATURES_SUFFIXES[feat][1:]: FEATURES_FUNCTIONS[feat](idx_datetime) + for feat in self.features_to_extract_ + } + X = pd_.concat( + [X, pd_.DataFrame(new_columns, index=X.index)], axis=1 + ) else: # check if dataset contains na @@ -359,32 +400,64 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: if len(self.variables_) == 0: return X - # convert datetime variables - datetime_df = pd.concat( - [ - pd.to_datetime( - X[variable], - dayfirst=self.dayfirst, - yearfirst=self.yearfirst, - utc=self.utc, - format=self.format, - ) - for variable in self.variables_ - ], - axis=1, - ) + if nwd.is_pandas_dataframe(X): + pd_ = nw_X.__native_namespace__() + # convert datetime variables + datetime_df = pd_.concat( + [ + pd_.to_datetime( + X[variable], + dayfirst=self.dayfirst, + yearfirst=self.yearfirst, + utc=self.utc, + format=self.format, + ) + for variable in self.variables_ + ], + axis=1, + ) - # create new features - for var in self.variables_: - for feat in self.features_to_extract_: - X[str(var) + FEATURES_SUFFIXES[feat]] = FEATURES_FUNCTIONS[feat]( + # build all the new features, then add them in a single insertion + # (avoids fragmenting the frame when many features are extracted) + # and without mutating the input dataframe. + new_columns = { + str(var) + FEATURES_SUFFIXES[feat]: FEATURES_FUNCTIONS[feat]( datetime_df[var] ) - if self.drop_original: - X.drop(self.variables_, axis=1, inplace=True) + for var in self.variables_ + for feat in self.features_to_extract_ + } + X = pd_.concat( + [X, pd_.DataFrame(new_columns, index=X.index)], axis=1 + ) + if self.drop_original: + X = X.drop(columns=self.variables_) + else: + # dayfirst/yearfirst/utc are pandas.to_datetime-only knobs with no + # equivalent on other backends, so only `format` is honoured here. + new_series = [ + FEATURES_FUNCTIONS_NARWHALS[feat]( + self._to_nw_datetime(nw_X.get_column(var)) + ).alias(str(var) + FEATURES_SUFFIXES[feat]) + for var in self.variables_ + for feat in self.features_to_extract_ + ] + nw_X = nw_X.with_columns(*new_series) + if self.drop_original: + nw_X = nw_X.drop(self.variables_) + X = nw_X.to_native() return X + def _to_nw_datetime(self, col: nw.Series) -> nw.Series: + """Ensure a narwhals Series has Datetime dtype, parsing strings/categoricals + and casting bare Dates (whose `.dt` methods reject hour/minute/second).""" + if isinstance(col.dtype, nw.Datetime): + return col + if isinstance(col.dtype, nw.Date): + return col.cast(nw.Datetime()) + return col.cast(nw.String()).str.to_datetime(format=self.format) + def _get_new_features_name(self) -> List: """create the names for the datetime features.""" @@ -401,7 +474,7 @@ def _get_new_features_name(self) -> List: return feature_names - def _check_index_contains_na(self, index: pd.Index): + def _check_index_contains_na(self, index) -> None: if index.isnull().any(): raise ValueError( "The dataframe index contains missing data. " diff --git a/feature_engine/datetime/datetime_ordinal.py b/feature_engine/datetime/datetime_ordinal.py index 9e3904642..3c53e7cd6 100644 --- a/feature_engine/datetime/datetime_ordinal.py +++ b/feature_engine/datetime/datetime_ordinal.py @@ -1,7 +1,11 @@ -from typing import List, Optional, Union import datetime +from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +import numpy as np +from dateutil.parser import parse as _parse_datetime +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -32,6 +36,12 @@ from feature_engine.variable_handling.check_variables import check_datetime_variables from feature_engine.variable_handling.find_variables import find_datetime_variables +# datetime.date(1970, 1, 1).toordinal() - the proleptic Gregorian ordinal of the +# Unix epoch, used to convert epoch-based timestamps into the same "days since +# January 1, 0001" ordinal that datetime.date.toordinal() returns. +_UNIX_EPOCH_ORDINAL = 719_163 +_MICROSECONDS_PER_DAY = 86_400_000_000 + @Substitution( return_empty=_return_empty_docstring, @@ -66,7 +76,7 @@ class DatetimeOrdinal(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): contain missing values. If 'ignore', missing data will be ignored when performing the transformation. - start_date: str, datetime.datetime, default=None + start_date: str, datetime.date, datetime.datetime, default=None A reference date from which the ordinal values will be calculated. If provided, the ordinal value of `start_date` will be 1, the day after will be 2, and so on. Days before `start_date` will take negative values. @@ -116,6 +126,25 @@ class DatetimeOrdinal(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): 0 1 1 2 2 3 + + With polars: + + >>> import polars as pl + >>> from feature_engine.datetime import DatetimeOrdinal + >>> X = pl.DataFrame(dict(date = ["2023-01-01", "2023-01-02", "2023-01-03"])) + >>> dtf = DatetimeOrdinal(start_date="2023-01-01") + >>> dtf.fit(X) + >>> dtf.transform(X) + shape: (3, 1) + ┌──────────────┐ + │ date_ordinal │ + │ --- │ + │ i64 │ + ╞══════════════╡ + │ 1 │ + │ 2 │ + │ 3 │ + └──────────────┘ """ def __init__( @@ -133,17 +162,6 @@ def __init__( f"Got {missing_values} instead." ) - if start_date is not None: - try: - self.start_date_ = pd.to_datetime(start_date) - except Exception as e: - raise ValueError( - f"start_date could not be converted to datetime. " - f"Got {start_date} instead. Error: {e}" - ) - else: - self.start_date_ = None - if not isinstance(drop_original, bool): raise ValueError( "drop_original takes only booleans True or False. " @@ -155,26 +173,46 @@ def __init__( self.variables = _check_variables_input_value(variables) self.return_empty = return_empty self.missing_values = missing_values + self.start_date = start_date self.drop_original = drop_original - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn any parameter. Finds datetime variables or checks that the variables selected by the user - can be converted to datetime. + can be converted to datetime. Also parses `start_date`, if provided, into + its ordinal representation. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series=None + y: Series=None It is not needed in this transformer. You can pass y or None. + """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) + + # parse the user-provided start_date into its ordinal representation. + # datetime.datetime is a subclass of datetime.date, so both are handled + # by the isinstance branch; strings are parsed with dateutil. + self.start_date_ordinal_: Optional[int] + if self.start_date is None: + self.start_date_ordinal_ = None + elif isinstance(self.start_date, datetime.date): + self.start_date_ordinal_ = self.start_date.toordinal() + else: + try: + self.start_date_ordinal_ = _parse_datetime(self.start_date).toordinal() + except Exception as e: + raise ValueError( + f"start_date could not be converted to datetime. " + f"Got {self.start_date} instead. Error: {e}" + ) if self.variables is None: self.variables_ = find_datetime_variables( @@ -184,80 +222,114 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): self.variables_ = check_datetime_variables(X, self.variables) # check if datetime variables contains na - if self.missing_values == "raise": + # nw.col([]) errors on the polars backend, so skip when there's + # nothing to check (happens when return_empty=True found no variables). + if self.missing_values == "raise" and len(self.variables_) > 0: _check_contains_na(X, self.variables_) - if self.start_date_ is not None: - self.start_date_ordinal_ = self.start_date_.toordinal() - else: - self.start_date_ordinal_ = None - # save input features - self.feature_names_in_ = X.columns.tolist() + self.feature_names_in_ = nw_X.columns # save train set shape - self.n_features_in_ = X.shape[1] + self.n_features_in_ = nw_X.shape[1] return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Calculate ordinal representation of datetime features and add them to the dataframe. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe, shape = [n_samples, n_features x n_df_features] + X_new: dataframe, shape = [n_samples, n_features x n_df_features] The dataframe with the original variables plus the new features. """ - # Check method fit has been called check_is_fitted(self) # check that input is a dataframe - X = check_X(X) + nw_X = check_X(X) # Check if input data contains same number of columns as dataframe used to fit. _check_X_matches_training_df(X, self.n_features_in_) - # reorder variables to match train set - X = X[self.feature_names_in_] - if len(self.variables_) == 0: return X - # create a copy(to protect original data) - X_new = X.copy() - # check if dataset contains na if self.missing_values == "raise": - _check_contains_na(X_new, self.variables_) + _check_contains_na(X, self.variables_) - for var in self.variables_: - # Convert to datetime, then to ordinal - datetime_series = pd.to_datetime(X_new[var]) - # Handle NaT values: toordinal() raises ValueError for NaT - ordinal_series = datetime_series.apply( - lambda x: x.toordinal() if pd.notna(x) else pd.NA + # variables can be native Date/Datetime columns, or string/categorical + # columns holding parseable date values - the latter need parsing into + # a real datetime dtype before the ordinal can be computed. + schema = nw_X.schema + to_parse = [ + var + for var in self.variables_ + if not isinstance(schema[var], (nw.Date, nw.Datetime)) + ] + if len(to_parse) > 0: + nw_X = nw_X.with_columns( + nw.col(var).cast(nw.String).str.to_datetime() for var in to_parse ) - if self.start_date_ordinal_ is not None: - # Only apply offset if not NaT - ordinal_series = ordinal_series.apply( - lambda x: x - self.start_date_ordinal_ + 1 if pd.notna(x) else pd.NA - ) + if nwd.is_pandas_dataframe(X): + return self._transform_pandas(nw_X.to_native()) + return self._transform_narwhals(nw_X) - X_new[str(var) + "_ordinal"] = ordinal_series + def _transform_pandas(self, X): + """Vectorized ordinal computation via numpy datetime64[D] arithmetic. - if self.drop_original: - X_new.drop(self.variables_, axis=1, inplace=True) + Benchmarked ~3.5-12x faster than the narwhals-generic dt.timestamp path + at 10k-100k rows x 1-10 columns (the gap widens with more columns), so + pandas keeps its own numpy fast path here. + """ + new_columns = {} + for var in self.variables_: + days = X[var].to_numpy().astype("datetime64[D]") + na_mask = np.isnat(days) + ordinal = days.astype("int64") + _UNIX_EPOCH_ORDINAL + if self.start_date_ordinal_ is not None: + ordinal = ordinal - self.start_date_ordinal_ + 1 + if na_mask.any(): + # int64 arithmetic on the NaT sentinel can wrap around, but that's + # harmless - the masked slots are overwritten with NaN right after. + ordinal = ordinal.astype("float64") + ordinal[na_mask] = np.nan + new_columns[str(var) + "_ordinal"] = ordinal + + # assign() still inserts columns one at a time internally, so it doesn't + # avoid fragmentation with many variables; building one DataFrame and + # joining it does (single insertion), same pattern as DecisionTreeFeatures. + X = X.join(type(X)(new_columns, index=X.index)) + if self.drop_original is True: + X = X.drop(columns=self.variables_) + return X + + def _transform_narwhals(self, nw_X): + """Ordinal computation via narwhals' dt.timestamp, already vectorized and + fast enough on polars that a numpy round-trip wouldn't pay for itself.""" + exprs = [] + for var in self.variables_: + ordinal_expr = ( + nw.col(var).dt.timestamp("us") // _MICROSECONDS_PER_DAY + + _UNIX_EPOCH_ORDINAL + ) + if self.start_date_ordinal_ is not None: + ordinal_expr = ordinal_expr - self.start_date_ordinal_ + 1 + exprs.append(ordinal_expr.alias(str(var) + "_ordinal")) - return X_new + nw_X = nw_X.with_columns(*exprs) + if self.drop_original is True: + nw_X = nw_X.drop(self.variables_) + return nw_X.to_native() def _get_new_features_name(self) -> List: """create the names for the new features.""" diff --git a/feature_engine/datetime/datetime_subtraction.py b/feature_engine/datetime/datetime_subtraction.py index 3688253fc..bc809f953 100644 --- a/feature_engine/datetime/datetime_subtraction.py +++ b/feature_engine/datetime/datetime_subtraction.py @@ -1,8 +1,10 @@ -from typing import List, Optional, Union +from datetime import timezone +from typing import Dict, List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd -from pandas.api.types import is_datetime64_any_dtype as is_datetime +from dateutil.parser import parse as _dateutil_parse +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.utils.validation import check_is_fitted from feature_engine._check_init_parameters.check_init_input_params import ( @@ -47,6 +49,27 @@ 0 2022-09-18 2022-08-18 31.0 1 2022-10-27 2022-08-27 61.0 2 2022-12-24 2022-06-24 183.0 + + With polars: + + >>> import polars as pl + >>> from feature_engine.datetime import DatetimeSubtraction + >>> X = pl.DataFrame({ + >>> "date1": ["2022-09-18", "2022-10-27", "2022-12-24"], + >>> "date2": ["2022-08-18", "2022-08-27", "2022-06-24"]}) + >>> dtf = DatetimeSubtraction(variables=["date1"], reference=["date2"]) + >>> dtf.fit(X) + >>> dtf.transform(X) + shape: (3, 3) + ┌────────────┬────────────┬─────────────────┐ + │ date1 ┆ date2 ┆ date1_sub_date2 │ + │ --- ┆ --- ┆ --- │ + │ str ┆ str ┆ f64 │ + ╞════════════╪════════════╪═════════════════╡ + │ 2022-09-18 ┆ 2022-08-18 ┆ 31.0 │ + │ 2022-10-27 ┆ 2022-08-27 ┆ 61.0 │ + │ 2022-12-24 ┆ 2022-06-24 ┆ 183.0 │ + └────────────┴────────────┴─────────────────┘ """.rstrip() @@ -219,21 +242,21 @@ def __init__( self.utc = utc self.format = format - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn any parameter. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, or np.array. Default=None. + y: Series, or np.array. Default=None. It is not needed in this transformer. You can pass y or None. """ # Common checks and attributes - X = check_X(X) + nw_X = check_X(X) # check variables are datetime if self.variables is None: @@ -267,25 +290,25 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): _check_contains_na(X, vars) # save input features - self.feature_names_in_ = X.columns.tolist() + self.feature_names_in_ = nw_X.columns # save train set shape - self.n_features_in_ = X.shape[1] + self.n_features_in_ = nw_X.shape[1] return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Add new features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe + X_new: dataframe The input dataframe plus the new variables. """ @@ -293,7 +316,7 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: check_is_fitted(self) # check that input is a dataframe - X = check_X(X) + nw_X = check_X(X) # Check if input data contains same number of columns as dataframe used to fit. _check_X_matches_training_df(X, self.n_features_in_) @@ -302,40 +325,50 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: vars = list(set(self.variables_ + self.reference_)) _check_contains_na(X, vars) - # reorder variables to match train set - X = X[self.feature_names_in_] + dt_arrays = self._to_datetime(nw_X) - X_dt = self._to_datetime(X) + new_series = self._sub(dt_arrays, nw_X.implementation) - new_features = self._sub(X_dt) + nw_X = nw_X.with_columns(*new_series) - X = pd.concat([X, new_features], axis=1) + if self.drop_original is True: + nw_X = nw_X.drop(list(set(self.variables_ + self.reference_))) - if self.drop_original: - X = X.drop( - columns=set(self.variables_ + self.reference_), - ) + return nw_X.to_native() + + def _to_datetime( + self, nw_X: nw.DataFrame + ) -> Dict[Union[str, int], np.ndarray]: + """Convert the variables and reference columns to numpy datetime64 arrays.""" + needed = sorted(set(self.variables_ + self.reference_)) + is_pandas = nw_X.implementation.is_pandas() - return X + # pandas.to_datetime honours dayfirst/yearfirst/utc precisely; grab the + # native namespace once (no `import pandas`) rather than per-column. + if is_pandas is True: + native_ns = nw.get_native_namespace(nw_X) - def _to_datetime(self, X: pd.DataFrame): - """convert variables to datetime.""" - # convert datetime variables - datetime_df = pd.concat( - [ - pd.to_datetime( - X[variable], + arrays = {} + non_dt_columns = [] + for variable in needed: + col = nw_X.get_column(variable) + if is_pandas is True: + parsed_native = native_ns.to_datetime( + col.to_native(), dayfirst=self.dayfirst, yearfirst=self.yearfirst, utc=self.utc, format=self.format, ) - for variable in set(self.variables_ + self.reference_) - ], - axis=1, - ) + parsed = nw.from_native(parsed_native, series_only=True) + else: + parsed = self._parse_non_pandas_column(col) - non_dt_columns = datetime_df.columns[~datetime_df.apply(is_datetime)].tolist() + if not isinstance(parsed.dtype, nw.Datetime): + non_dt_columns.append(variable) + continue + + arrays[variable] = parsed.to_numpy() if non_dt_columns: raise ValueError( @@ -343,23 +376,69 @@ def _to_datetime(self, X: pd.DataFrame): + (len(non_dt_columns) * "{} ").format(*non_dt_columns) + "could not be converted to datetime. Try setting utc=True" ) - return datetime_df - def _sub(self, dt_df: pd.DataFrame): - """make datetime subtraction""" - new_df = pd.DataFrame() - for reference in self.reference_: - new_varnames = [f"{var}_sub_{reference}" for var in self.variables_] - new_df[new_varnames] = ( - dt_df[self.variables_] - .sub(dt_df[reference], axis=0) - .div(np.timedelta64(1, self.output_unit).astype("timedelta64[ns]")) + return arrays + + def _parse_non_pandas_column(self, col: "nw.Series") -> "nw.Series": + """Parse a single non-pandas column to a narwhals Datetime series.""" + if isinstance(col.dtype, nw.Datetime): + return col + if isinstance(col.dtype, nw.Date): + return col.cast(nw.Datetime) + if isinstance(col.dtype, (nw.Categorical, nw.Enum)): + col = col.cast(nw.String) + + try: + return col.str.to_datetime(format=self.format) + except Exception: + if self.format is not None: + raise + # narwhals' vectorized parser needs a single unambiguous format; + # fall back to dateutil per value, same flexible guessing that + # check_datetime_variables already promises across backends. + return self._flexible_parse(col) + + def _flexible_parse(self, col: "nw.Series") -> "nw.Series": + values = [ + None + if value is None + else _dateutil_parse( + value, dayfirst=self.dayfirst, yearfirst=self.yearfirst ) + for value in col.to_list() + ] + if self.utc is True: + values = [ + None + if v is None + else ( + v.astimezone(timezone.utc) + if v.tzinfo is not None + else v.replace(tzinfo=timezone.utc) + ) + for v in values + ] + return nw.new_series(col.name, values, backend=col.implementation) - if self.new_variables_names is not None: - new_df.columns = self.new_variables_names - - return new_df + def _sub(self, dt_arrays: Dict[Union[str, int], np.ndarray], backend) -> List: + """make datetime subtraction""" + names = self._get_new_features_name() + # "Y"/"M" are non-linear units: numpy can only divide timedeltas by + # them once both sides are cast to a common linear unit (ns), which + # is also what pandas does internally for Timedelta / Timedelta. + unit_td = np.timedelta64(1, self.output_unit).astype("timedelta64[ns]") + + new_series = [] + idx = 0 + for reference in self.reference_: + ref_arr = dt_arrays[reference] + for var in self.variables_: + diff = (dt_arrays[var] - ref_arr).astype("timedelta64[ns]") + result = diff / unit_td + new_series.append(nw.new_series(names[idx], result, backend=backend)) + idx += 1 + + return new_series def _get_new_features_name(self) -> List: """Return names of the created features.""" diff --git a/feature_engine/discretisation/arbitrary.py b/feature_engine/discretisation/arbitrary.py index 5776cd71d..fd3056940 100644 --- a/feature_engine/discretisation/arbitrary.py +++ b/feature_engine/discretisation/arbitrary.py @@ -4,7 +4,9 @@ import warnings from typing import Dict, List, Optional, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.mixins import FitFromDictMixin from feature_engine._docstrings.fit_attributes import ( @@ -110,6 +112,27 @@ class ArbitraryDiscretiser(BaseDiscretiser, FitFromDictMixin): 3 25 1 17 Name: x, dtype: int64 + + With polars: + + >>> import polars as pl + >>> from feature_engine.discretisation import ArbitraryDiscretiser + >>> X = pl.DataFrame({"x": [10, 30, 60, 90]}) + >>> bins = dict(x=[0, 25, 50, 75, 100]) + >>> ad = ArbitraryDiscretiser(binning_dict=bins) + >>> ad.fit(X) + >>> ad.transform(X) + shape: (4, 1) + ┌─────┐ + │ x │ + │ --- │ + │ i64 │ + ╞═════╡ + │ 0 │ + │ 1 │ + │ 2 │ + │ 3 │ + └─────┘ """ def __init__( @@ -127,7 +150,7 @@ def __init__( f"variable. Got {binning_dict} instead." ) - if errors not in ["ignore", "raise"]: + if not isinstance(errors, str) or errors not in ["ignore", "raise"]: raise ValueError( "errors only takes values 'ignore' and 'raise'. " f"Got {errors} instead." @@ -138,21 +161,20 @@ def __init__( self.binning_dict = binning_dict self.errors = errors - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn any parameter. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. y: None y is not needed in this transformer. You can pass y or None. """ - # check input dataframe - X, variables_ = super()._fit_from_dict(X, self.binning_dict) + _, variables_ = super()._fit_from_dict(X, self.binning_dict) self.variables_ = variables_ # for consistency with the rest of the discretisers, we add this attribute @@ -161,30 +183,44 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Sort the variable values into the intervals. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The transformed data with the discrete variables. """ X = super().transform(X) - # check if NaN values were introduced by the discretisation procedure. - if X[self.variables_].isnull().sum().sum() > 0: - - # obtain the name(s) of the columns with null values - nan_columns = ( - X[self.variables_].columns[X[self.variables_].isnull().any()].tolist() - ) + # check if NaN values were introduced by the discretisation procedure. + nw_X = nw.from_native(X, eager_only=True) + if self.return_boundaries is True: + # missing labels are set to None by _bin_labels() + nan_columns = [ + var + for var in self.variables_ + if nw_X.get_column(var).is_null().any() + ] + else: + # codes are numeric; when return_object=True they're boxed as + # python floats in an Object column, where polars' is_null()/ + # is_nan() can't see NaN - a numpy float cast is reliable on + # both backends. + nan_columns = [ + var + for var in self.variables_ + if np.isnan(nw_X.get_column(var).to_numpy().astype(float)).any() + ] + + if len(nan_columns) > 0: if len(nan_columns) > 1: nan_columns_str = ", ".join(nan_columns) else: diff --git a/feature_engine/discretisation/base_discretiser.py b/feature_engine/discretisation/base_discretiser.py index 6c61d05d3..09ec7a741 100644 --- a/feature_engine/discretisation/base_discretiser.py +++ b/feature_engine/discretisation/base_discretiser.py @@ -1,7 +1,11 @@ # Authors: Morgan Sell # License: BSD 3 clause -import pandas as pd +from typing import List + +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer @@ -41,45 +45,125 @@ def __init__( self.return_boundaries = return_boundaries self.precision = precision - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """Sort the variable values into the intervals. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The transformed data with the discrete variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) + + # bin edges are already fixed by fit(), so sorting values into them is a + # plain numpy searchsorted - vectorizable identically for every backend, + # no pandas/polars-specific path needed. + native_namespace = nw_X.__native_namespace__() - # transform variables if self.return_boundaries is True: - for feature in self.variables_: - X[feature] = pd.cut( - X[feature], - self.binner_dict_[feature], - precision=self.precision, - include_lowest=True, + new_columns = [ + nw.new_series( + feature, + self._bin_labels( + nw_X.get_column(feature).to_numpy(), + self.binner_dict_[feature], + self.precision, + ), + backend=native_namespace, ) - X[self.variables_] = X[self.variables_].astype(str) - + for feature in self.variables_ + ] else: - for feature in self.variables_: - X[feature] = pd.cut( - X[feature], - self.binner_dict_[feature], - labels=False, - include_lowest=True, + # nw.Object mirrors the pandas "O" dtype astype() used to produce, + # and is what feature-engine's categorical encoders detect on + # every narwhals-supported backend (see variable_handling). + dtype = nw.Object if self.return_object is True else None + new_columns = [ + nw.new_series( + feature, + self._bin_codes( + nw_X.get_column(feature).to_numpy(), + self.binner_dict_[feature], + self.return_object, + ), + dtype=dtype, + backend=native_namespace, ) + for feature in self.variables_ + ] - # return object - if self.return_object: - X[self.variables_] = X[self.variables_].astype("O") + X = nw_X.with_columns(*new_columns).to_native() return X + + def _digitize(self, values: np.ndarray, bins_arr: np.ndarray): + """0-based bin index per value, right-closed intervals with the lowest edge + included - mirrors pandas.cut(bins=bins, include_lowest=True), which is + itself built on this same bins.searchsorted() call. Values outside the + bin range, and NaNs, are flagged via na_mask rather than given a code. + """ + ids = np.asarray(np.searchsorted(bins_arr, values, side="left")) + ids[values == bins_arr[0]] = 1 + na_mask: np.ndarray = np.isnan(values) | (ids == len(bins_arr)) | (ids == 0) + return ids - 1, na_mask + + def _bin_codes(self, values: np.ndarray, bins: List[float], return_object: bool): + bins_arr: np.ndarray = np.asarray(bins, dtype=float) + codes, na_mask = self._digitize(values, bins_arr) + + # match pandas.cut(labels=False): int codes, upcast to float only when a + # NaN placeholder is actually needed. + if na_mask.any(): + codes = codes.astype(np.float64) + codes[na_mask] = np.nan + if return_object is True: + codes = codes.astype(object) + + return codes + + def _bin_labels(self, values: np.ndarray, bins: List[float], precision: int): + bins_arr: np.ndarray = np.asarray(bins, dtype=float) + codes, na_mask = self._digitize(values, bins_arr) + + labels = np.asarray(self._format_bin_labels(bins_arr, precision), dtype=object) + out: np.ndarray = np.empty(len(values), dtype=object) + out[~na_mask] = labels[codes[~na_mask]] + out[na_mask] = None + + return out + + def _format_bin_labels(self, bins_arr: np.ndarray, precision: int) -> List[str]: + """"(lower, upper]" text per bin, replicating pandas.cut's own label + formatting: widen precision until break values are unique, then shrink + the lowest edge so include_lowest values still read as inside the first + interval. + """ + precision = self._infer_precision(precision, bins_arr) + breaks = [self._round_frac(b, precision) for b in bins_arr] + breaks[0] = breaks[0] - 10 ** (-precision) + return [f"({breaks[i]}, {breaks[i + 1]}]" for i in range(len(breaks) - 1)] + + def _round_frac(self, x: float, precision: int) -> float: + if not np.isfinite(x) or x == 0: + return float(x) + frac, whole = np.modf(x) + if whole == 0: + digits = -int(np.floor(np.log10(abs(frac)))) - 1 + precision + else: + digits = precision + return float(np.around(x, digits)) + + def _infer_precision(self, base_precision: int, bins_arr: np.ndarray) -> int: + # widen precision until every rounded break is unique - otherwise two + # adjacent bins could render with identical label text. + for precision in range(base_precision, 20): + levels = [self._round_frac(b, precision) for b in bins_arr] + if len(set(levels)) == len(bins_arr): + return precision + return base_precision diff --git a/feature_engine/discretisation/decision_tree.py b/feature_engine/discretisation/decision_tree.py index 8af4b9b60..6869cc629 100644 --- a/feature_engine/discretisation/decision_tree.py +++ b/feature_engine/discretisation/decision_tree.py @@ -3,8 +3,10 @@ from typing import Dict, List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from joblib import Parallel, delayed +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.model_selection import GridSearchCV from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor from sklearn.utils.multiclass import check_classification_targets, type_of_target @@ -115,6 +117,11 @@ class DecisionTreeDiscretiser(BaseNumericalTransformer): DecisionTreeClassifier(). For reproducibility it is recommended to set the random_state to an integer. + n_jobs: int, default=None + The number of jobs to run in parallel. `fit` is parallelized over the variables, + training one decision tree per variable. `None` means 1 unless in a + `joblib.parallel_backend` context. `-1` means using all processors. + Attributes ---------- binner_dict_: @@ -163,9 +170,10 @@ class DecisionTreeDiscretiser(BaseNumericalTransformer): >>> dtd = DecisionTreeDiscretiser(random_state=42) >>> dtd.fit(X, y_reg) >>> dtd.transform(X)["x"].value_counts() + x -0.090091 90 - 0.479454 10 - Name: x, dtype: int64 + 0.479454 10 + Name: count, dtype: int64 You can also apply this for classification problems adjusting the scoring metric. @@ -173,9 +181,27 @@ class DecisionTreeDiscretiser(BaseNumericalTransformer): >>> dtd = DecisionTreeDiscretiser(regression=False, scoring="f1", random_state=42) >>> dtd.fit(X, y_clf) >>> dtd.transform(X)["x"].value_counts() + x 0.480769 52 0.687500 48 - Name: x, dtype: int64 + Name: count, dtype: int64 + + With polars: + + >>> import polars as pl + >>> X = pl.DataFrame({"x": X["x"].to_list()}) + >>> dtd = DecisionTreeDiscretiser(random_state=42) + >>> dtd.fit(X, y_reg) + >>> dtd.transform(X)["x"].value_counts() + shape: (2, 2) + ┌───────────┬───────┐ + │ x ┆ count │ + │ --- ┆ --- │ + │ f64 ┆ u32 │ + ╞═══════════╪═══════╡ + │ -0.090091 ┆ 90 │ + │ 0.479454 ┆ 10 │ + └───────────┴───────┘ """ def __init__( @@ -189,9 +215,14 @@ def __init__( param_grid: Optional[Dict[str, Union[str, int, float, List[int]]]] = None, regression: bool = True, random_state: Optional[int] = None, + n_jobs: Optional[int] = None, ) -> None: - if bin_output not in ["prediction", "bin_number", "boundaries"]: + if not isinstance(bin_output, str) or bin_output not in [ + "prediction", + "bin_number", + "boundaries", + ]: raise ValueError( "bin_output takes values 'prediction', 'bin_number' or 'boundaries'. " f"Got {bin_output} instead." @@ -223,9 +254,10 @@ def __init__( self.variables = _check_variables_input_value(variables) self.param_grid = param_grid self.random_state = random_state + self.n_jobs = n_jobs self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Fit one decision tree per variable to discretise with cross-validation and grid-search for hyperparameters. @@ -233,11 +265,11 @@ def fit(self, X: pd.DataFrame, y: pd.Series): Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. - y: pandas series. + y: Series. Target variable. Required to train the decision tree. """ # confirm model type and target variables are compatible. @@ -251,33 +283,24 @@ def fit(self, X: pd.DataFrame, y: pd.Series): else: check_classification_targets(y) - # check input dataframe - X, variables_ = self._fit_setup(X) + nw_X, variables_ = self._fit_setup(X) if self.param_grid: param_grid = self.param_grid else: param_grid = {"max_depth": [1, 2, 3, 4]} - binner_dict_ = {} - scores_dict_ = {} + X_subs = [nw_X.get_column(var).to_frame().to_native() for var in variables_] - for var in variables_: + fitted = Parallel(n_jobs=self.n_jobs, prefer="threads")( + delayed(self._fit_one_tree)(X_sub, y, param_grid) for X_sub in X_subs + ) - if self.regression: - model = DecisionTreeRegressor(random_state=self.random_state) - else: - model = DecisionTreeClassifier(random_state=self.random_state) - - tree_model = GridSearchCV( - model, cv=self.cv, scoring=self.scoring, param_grid=param_grid - ) - - # fit the model to the variable - tree_model.fit(X[var].to_frame(), y) - - binner_dict_[var] = tree_model - scores_dict_[var] = tree_model.score(X[var].to_frame(), y) + binner_dict_ = dict(zip(variables_, fitted)) + scores_dict_ = { + var: tree_model.score(X_sub, y) + for var, X_sub, tree_model in zip(variables_, X_subs, fitted) + } if self.bin_output != "prediction": for var in variables_: @@ -296,64 +319,76 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Replaces original variable values with the predictions of the tree. The decision tree predictions are finite, aka, discrete. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input samples. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The dataframe with transformed variables. """ + nw_X = self._check_transform_input_and_state(X) - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + # add all new columns in one step, instead of one per variable, and leave + # the user's dataframe unchanged + new_columns: Dict[Union[str, int], np.ndarray] = {} if self.bin_output == "prediction": for feature in self.variables_: - if self.regression: - preds = self.binner_dict_[feature].predict(X[feature].to_frame()) - if self.precision is None: - X[feature] = preds - else: - X[feature] = np.round(preds, self.precision) + X_sub = nw_X.get_column(feature).to_frame().to_native() + if self.regression is True: + preds = self.binner_dict_[feature].predict(X_sub) else: - tmp = self.binner_dict_[feature].predict_proba( - X[feature].to_frame() - ) - preds = tmp[:, 1] - if self.precision is None: - X[feature] = preds - else: - X[feature] = np.round(preds, self.precision) + preds = self.binner_dict_[feature].predict_proba(X_sub)[:, 1] + if self.precision is not None: + preds = np.round(preds, self.precision) + new_columns[feature] = preds elif self.bin_output == "boundaries": + # __init__ already guarantees precision is set when bin_output is + # "boundaries"; assert narrows the type for mypy. + assert self.precision is not None for feature in self.variables_: - X[feature] = pd.cut( - X[feature], - self.binner_dict_[feature], - precision=self.precision, - include_lowest=True, - ) - X[self.variables_] = X[self.variables_].astype(str) + thresholds = self.binner_dict_[feature] + labels = self._bin_labels(thresholds, self.precision) + values = nw_X.get_column(feature).to_numpy() + bin_idx = self._bin_index(values, thresholds) + new_columns[feature] = np.array(labels)[bin_idx] else: for feature in self.variables_: - X[feature] = pd.cut( - X[feature], - self.binner_dict_[feature], - labels=False, - include_lowest=True, - ) + thresholds = self.binner_dict_[feature] + values = nw_X.get_column(feature).to_numpy() + new_columns[feature] = self._bin_index(values, thresholds) + + new_series = [ + nw.new_series(name, values, backend=nw_X.implementation) + for name, values in new_columns.items() + ] + X = nw_X.with_columns(*new_series).to_native() return X + def _fit_one_tree(self, X_sub: IntoDataFrame, y: IntoSeries, param_grid: Dict): + """Instantiate and fit one decision tree on one variable.""" + if self.regression is True: + model = DecisionTreeRegressor(random_state=self.random_state) + else: + model = DecisionTreeClassifier(random_state=self.random_state) + + tree_model = GridSearchCV( + model, cv=self.cv, scoring=self.scoring, param_grid=param_grid + ) + tree_model.fit(X_sub, y) + return tree_model + def _more_tags(self): tags_dict = _return_tags() tags_dict["variables"] = "numerical" @@ -363,3 +398,50 @@ def _more_tags(self): def __sklearn_tags__(self): tags = super().__sklearn_tags__() return tags + + def _round_bin_edge(self, x: float, precision: int) -> float: + """Round a bin edge the way pandas.cut historically formatted Interval labels: + -inf/inf/0 pass through unrounded, and numbers with magnitude < 1 get extra + decimals so that `precision` significant digits survive past the leading + zeros (e.g. -0.0942 at precision=3 keeps 4 decimals, not 3, since + round(-0.0942, 3) == -0.094 would only keep 2 significant digits). + """ + if not np.isfinite(x) or x == 0: + return x + frac, whole = np.modf(x) + if whole == 0: + digits = -int(np.floor(np.log10(abs(frac)))) - 1 + precision + else: + digits = precision + return round(x, digits) + + def _infer_bin_precision(self, thresholds: List[float], precision: int) -> int: + """Find the smallest precision >= `precision` at which every rounded + threshold is still distinct, mirroring pandas.cut's behaviour of bumping + precision (for every edge, not just the colliding pair) when the requested + precision would make two adjacent bin edges collide.""" + for prec in range(precision, 20): + rounded = [self._round_bin_edge(t, prec) for t in thresholds] + if len(set(rounded)) == len(thresholds): + return prec + return precision + + def _format_bin_edge(self, x: float, precision: int) -> str: + if x == -np.inf: + return "-inf" + if x == np.inf: + return "inf" + return str(self._round_bin_edge(x, precision)) + + def _bin_labels(self, thresholds: List[float], precision: int) -> List[str]: + """Build the `(left, right]` interval label for every bin delimited by + `thresholds`, which starts with -inf and ends with inf.""" + precision = self._infer_bin_precision(thresholds, precision) + edges = [self._format_bin_edge(t, precision) for t in thresholds] + return [f"({edges[i]}, {edges[i + 1]}]" for i in range(len(edges) - 1)] + + def _bin_index(self, values: np.ndarray, thresholds: List[float]) -> np.ndarray: + """Map each value to the 0-indexed bin delimited by `thresholds` (which + starts with -inf and ends with inf), bins being closed on the right. + """ + return np.digitize(values, thresholds[1:-1], right=True) diff --git a/feature_engine/discretisation/equal_frequency.py b/feature_engine/discretisation/equal_frequency.py index a2137870f..028c95da7 100644 --- a/feature_engine/discretisation/equal_frequency.py +++ b/feature_engine/discretisation/equal_frequency.py @@ -3,7 +3,8 @@ from typing import List, Optional, Union -import pandas as pd +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -156,26 +157,41 @@ def __init__( self.return_empty = return_empty self.q = q - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the limits of the equal frequency intervals. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. y: None y is not needed in this encoder. You can pass y or None. """ - # check input dataframe - X, variables_ = self._fit_setup(X) + nw_X, variables_ = self._fit_setup(X) + quantiles = np.linspace(0, 1, self.q + 1) + # pandas.qcut nudges each quantile that isn't exactly representable in + # base 2 up via nextafter, to round up rather than to nearest (verified + # against pandas.core.reshape.tile.qcut source); skipping this shifts + # bin edges by ~1e-13 versus the pre-migration pd.qcut output. + np.putmask( + quantiles, + self.q * quantiles != np.arange(self.q + 1), + np.nextafter(quantiles, 1), + ) binner_dict_ = {} for var in variables_: - tmp, bins = pd.qcut(x=X[var], q=self.q, retbins=True, duplicates="drop") + # _fit_setup() already rejects NaN in variables_, so no NaN-masking + # is needed here. np.quantile replicates pandas.qcut's own quantile + # computation (verified bit-exact against real pd.qcut(retbins=True) + # output); np.unique both sorts and drops duplicate edges, matching + # qcut(duplicates="drop"). + values = nw_X.get_column(var).to_numpy() + bins = np.unique(np.quantile(values, quantiles, method="linear")) # Prepend/Append infinities to accommodate outliers bins = list(bins) diff --git a/feature_engine/discretisation/equal_width.py b/feature_engine/discretisation/equal_width.py index bab5c5396..9dbc1802a 100644 --- a/feature_engine/discretisation/equal_width.py +++ b/feature_engine/discretisation/equal_width.py @@ -3,7 +3,10 @@ from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -164,43 +167,59 @@ def __init__( self.return_empty = return_empty self.bins = bins - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the boundaries of the equal width intervals / bins for each variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. y: None y is not needed in this encoder. You can pass y or None. """ - # check input dataframe - X, variables_ = self._fit_setup(X) + nw_X, variables_ = self._fit_setup(X) # fit binner_dict_ = {} - for var in variables_: - tmp, bins = pd.cut( - x=X[var], - bins=self.bins, - retbins=True, - duplicates="drop", - include_lowest=True, - ) - - # Prepend/Append infinities - bins = list(bins) - bins[0] = float("-inf") - bins[len(bins) - 1] = float("inf") - binner_dict_[var] = bins + if len(variables_) > 0: + # one call for all variables; pandas is faster than narwhals here + if nwd.is_pandas_dataframe(X) is True: + arr = X[variables_].to_numpy() + else: + arr = nw_X.select(nw.col(variables_)).to_numpy() + mins = arr.min(axis=0) + maxs = arr.max(axis=0) + for var, mn, mx in zip(variables_, mins, maxs): + binner_dict_[var] = self._equal_width_edges(mn, mx, self.bins) self.binner_dict_ = binner_dict_ self.variables_ = variables_ self._get_feature_names_in(X) return self + + def _equal_width_edges(self, mn: float, mx: float, bins: int) -> List[float]: + """Bin-edge computation matching pandas.cut(bins=int, duplicates="drop"): + widen a constant [mn, mx] by 0.1% so linspace still produces positive- + width bins, then collapse duplicate edges the same way. The outer edges + are then clipped to +-inf, same as the pre-migration code did to the + retbins output, so transform() never needs an out-of-range branch. + """ + if mn == mx: + mn = mn - 0.001 * abs(mn) if mn != 0 else -0.001 + mx = mx + 0.001 * abs(mx) if mx != 0 else 0.001 + + edges = np.linspace(mn, mx, bins + 1) + unique_edges = np.unique(edges) + if len(unique_edges) < len(edges) and len(edges) != 2: + edges = unique_edges + + edges_: List[float] = edges.tolist() + edges_[0] = float("-inf") + edges_[-1] = float("inf") + return edges_ diff --git a/feature_engine/discretisation/geometric_width.py b/feature_engine/discretisation/geometric_width.py index 709381c71..6d9b5a462 100644 --- a/feature_engine/discretisation/geometric_width.py +++ b/feature_engine/discretisation/geometric_width.py @@ -1,7 +1,7 @@ from typing import List, Optional, Union import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -159,28 +159,28 @@ def __init__( self.return_empty = return_empty self.bins = bins - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the boundaries of the geometric width intervals / bins for each variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. y: None y is not needed in this encoder. You can pass y or None. """ - # check input dataframe - X, variables_ = self._fit_setup(X) + nw_X, variables_ = self._fit_setup(X) # fit binner_dict_ = {} for var in variables_: - min_, max_ = X[var].min(), X[var].max() + col = nw_X.get_column(var) + min_, max_ = col.min(), col.max() increment = np.power(max_ - min_, 1.0 / self.bins) bins = np.r_[ -np.inf, min_ + np.power(increment, np.arange(1, self.bins)), np.inf diff --git a/feature_engine/encoding/_helper_functions.py b/feature_engine/encoding/_helper_functions.py index 521f6e31e..d9d047311 100644 --- a/feature_engine/encoding/_helper_functions.py +++ b/feature_engine/encoding/_helper_functions.py @@ -1,3 +1,9 @@ +import narwhals as nw +import narwhals.dependencies as nwd + +TARGET_NAME = "__feature_engine_target__" + + def check_parameter_unseen(unseen, accepted_values): if not isinstance(accepted_values, list) or not all( isinstance(item, str) for item in accepted_values @@ -6,8 +12,20 @@ def check_parameter_unseen(unseen, accepted_values): "accepted_values should be a list of strings. " f" Got {accepted_values} instead." ) - if unseen not in accepted_values: + if not isinstance(unseen, str) or unseen not in accepted_values: raise ValueError( f"Parameter `unseen` takes only values {', '.join(accepted_values)}." f" Got {unseen} instead." ) + + +def add_target_to_X(nw_X, y): + """Add y to X as the column TARGET_NAME, pairing rows by position. + + y can be a series, list or array. With pandas, the column takes the index of X. + """ + if nwd.is_into_series(y): + y_nw = nw.from_native(y, series_only=True) + else: + y_nw = nw.new_series(name=TARGET_NAME, values=y, backend=nw_X.implementation) + return nw_X.with_columns(y_nw.alias(TARGET_NAME)) diff --git a/feature_engine/encoding/base_encoder.py b/feature_engine/encoding/base_encoder.py index 35427f260..77ebbb5d5 100644 --- a/feature_engine/encoding/base_encoder.py +++ b/feature_engine/encoding/base_encoder.py @@ -1,7 +1,7 @@ import warnings from typing import List, Union -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -20,7 +20,7 @@ from feature_engine._docstrings.init_parameters.encoders import _ignore_format_docstring from feature_engine._docstrings.substitute import Substitution from feature_engine.dataframe_checks import ( - _check_optional_contains_na, + _check_contains_na, _check_X_matches_training_df, check_X, ) @@ -96,7 +96,10 @@ def __init__( ignore_format: bool = False, ) -> None: - if missing_values not in ["raise", "ignore"]: + if not isinstance(missing_values, str) or missing_values not in [ + "raise", + "ignore", + ]: raise ValueError( "missing_values takes only values 'raise' or 'ignore'. " f"Got {missing_values} instead." @@ -121,11 +124,11 @@ class CategoricalMethodsMixin(TransformerMixin, BaseEstimator, GetFeatureNamesOu - GetFeatureNamesOutMixin brings method get_feature_names_out(). """ - def _check_na(self, X: pd.DataFrame, variables): + def _check_na(self, X: IntoDataFrame, variables): if self.missing_values == "raise": - _check_optional_contains_na(X, variables) + _check_contains_na(X, variables, error_msg="optional") - def _check_or_select_variables(self, X: pd.DataFrame): + def _check_or_select_variables(self, X: IntoDataFrame): """ Finds categorical variables, or alternatively checks that the variables entered by the user are of type object (categorical). @@ -133,7 +136,7 @@ def _check_or_select_variables(self, X: pd.DataFrame): Parameters ---------- - X: Pandas DataFrame + X: dataframe Raises ------ @@ -159,115 +162,105 @@ def _check_or_select_variables(self, X: pd.DataFrame): return variables_ - def _get_feature_names_in(self, X: pd.DataFrame): + def _get_feature_names_in(self, X: IntoDataFrame): """ - Returns attributes `featrure_names_in_` and `n_feature_names_in_`, which are + Sets attributes `feature_names_in_` and `n_features_in_`, which are standard for all transformers in the library. + + Parameters + ---------- + X: narwhals dataframe + The dataframe returned by `check_X` / `check_X_y` at the start of `fit`. """ - # save input features - self.feature_names_in_ = X.columns.tolist() + # save input features. list() normalises both a narwhals `.columns` + # (already a list) and a pandas `Index` to a plain list. + self.feature_names_in_ = list(X.columns) # save train set shape self.n_features_in_ = X.shape[1] - def _check_transform_input_and_state(self, X: pd.DataFrame) -> pd.DataFrame: + def _check_transform_input_and_state(self, X: IntoDataFrame) -> IntoDataFrame: """ Checks that the input is a dataframe and of the same size than the one used - in the fit method. Checks absence of NA. + in the fit method. Parameters ---------- - X: Pandas DataFrame + X: dataframe + The dataframe entered by the user, in any library supported by narwhals. Raises ------ TypeError - If the input is not a Pandas DataFrame + If the input is not a dataframe ValueError - - If the variable(s) contain null values. - - If the df has different number of features than the df used in fit() + If the df has a different number of features than the df used in fit() Returns ------- - X: Pandas DataFrame - The same dataframe entered by the user. + nw_X: narwhals dataframe + The narwhalified version of the dataframe entered by the user. """ - # Check method fit has been called check_is_fitted(self) - # check that input is a dataframe - X = check_X(X) + nw_X = check_X(X) - # Check input data contains same number of columns as df used to fit _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + return nw_X - return X - - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """Replace categories with the learned parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The dataset to transform. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features]. + X_new: dataframe of shape = [n_samples, n_features]. The dataframe containing the categories replaced by numbers. """ - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) # check if dataset contains na if self.missing_values == "raise": - _check_optional_contains_na(X, self.variables_) + _check_contains_na(X, self.variables_, error_msg="optional") - X = self._encode(X) + X = self._encode(nw_X) return X - def _encode(self, X: pd.DataFrame) -> pd.DataFrame: - # replace categories by the learned parameters - for feature in self.encoder_dict_.keys(): - X[feature] = X[feature].map(self.encoder_dict_[feature]) - - # if original variables are cast as categorical, they will remain - # categorical after the encoding, and this is probably not desired - if X[feature].dtype.name == "category": - if all(isinstance(x, int) for x in X[feature]): - X[feature] = X[feature].astype("int") - else: - X[feature] = X[feature].astype("float") - - if self.unseen == "encode": - X[self.variables_] = X[self.variables_].fillna(self._unseen) - else: + def _encode(self, X: IntoDataFrame) -> IntoDataFrame: + default = self._unseen if self.unseen == "encode" else None + new_series = [ + X.get_column(feature).replace_strict(mapping, default=default) + for feature, mapping in self.encoder_dict_.items() + ] + X = X.with_columns(*new_series) + + if self.unseen != "encode": # check if nan values were introduced by the transformation self._check_nan_values_after_transformation(X) - return X + return X.to_native() - def _check_nan_values_after_transformation(self, X): + def _check_nan_values_after_transformation(self, X: IntoDataFrame): + nan_columns = [ + feature + for feature in self.encoder_dict_.keys() + if X.get_column(feature).null_count() > 0 + ] - # check if NaN values were introduced by the encoding - if X[self.variables_].isnull().sum().sum() > 0: - - # obtain the name(s) of the columns have null values - nan_columns = ( - X[self.encoder_dict_.keys()] - .columns[X[self.encoder_dict_.keys()].isnull().any()] - .tolist() - ) + if len(nan_columns) > 0: if len(nan_columns) > 1: - nan_columns_str = ", ".join(nan_columns) + nan_columns_str = ", ".join(str(col) for col in nan_columns) else: - nan_columns_str = nan_columns[0] + nan_columns_str = str(nan_columns[0]) if self.unseen == "ignore": warnings.warn( @@ -280,27 +273,32 @@ def _check_nan_values_after_transformation(self, X): f"{nan_columns_str}." ) - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """Convert the encoded variable back to the original values. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The transformed dataframe. Returns ------- - X_tr: pandas dataframe of shape = [n_samples, n_features]. + X_tr: dataframe of shape = [n_samples, n_features]. The un-transformed dataframe, with the categorical variables containing the original values. """ - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) - # replace encoded categories by the original values - for feature in self.encoder_dict_.keys(): - inv_map = {v: k for k, v in self.encoder_dict_[feature].items()} - X[feature] = X[feature].map(inv_map) + # replace encoded categories by the original values. get_column() + # rather than nw.col() again, to support pandas integer column names. + new_series = [ + nw_X.get_column(feature).replace_strict( + {v: k for k, v in mapping.items()}, default=None + ) + for feature, mapping in self.encoder_dict_.items() + ] + X = nw_X.with_columns(*new_series).to_native() return X diff --git a/feature_engine/encoding/count_frequency.py b/feature_engine/encoding/count_frequency.py index 682c90680..317b1ca1b 100644 --- a/feature_engine/encoding/count_frequency.py +++ b/feature_engine/encoding/count_frequency.py @@ -4,7 +4,7 @@ import warnings from typing import List, Optional, Union -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -157,6 +157,26 @@ class CountEncoder(CategoricalMethodsMixin, CategoricalInitMixinNA): 1 2 0.25 2 3 0.25 3 4 0.50 + + With polars + + >>> import polars as pl + >>> from feature_engine.encoding import CountEncoder + >>> X = pl.DataFrame(dict(x1 = [1,2,3,4], x2 = ["c", "a", "b", "c"])) + >>> cf = CountEncoder(encoding_method='count') + >>> cf.fit(X) + >>> cf.transform(X) + shape: (4, 2) + ┌─────┬─────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ i64 │ + ╞═════╪═════╡ + │ 1 ┆ 2 │ + │ 2 ┆ 1 │ + │ 3 ┆ 1 │ + │ 4 ┆ 2 │ + └─────┴─────┘ """ def __init__( @@ -169,7 +189,10 @@ def __init__( unseen: str = "ignore", ) -> None: - if encoding_method not in ["count", "frequency"]: + if not isinstance(encoding_method, str) or encoding_method not in [ + "count", + "frequency", + ]: raise ValueError( "encoding_method takes only values 'count' and 'frequency'. " f"Got {encoding_method} instead." @@ -183,37 +206,35 @@ def __init__( self.unseen = unseen self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the counts or frequencies which will be used to replace the categories. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. - y: pandas Series, default = None + y: Series, default = None y is not needed in this encoder. You can pass y or None. """ - X = check_X(X) + nw_X = check_X(X) variables_ = self._check_or_select_variables(X) self._check_na(X, variables_) + normalize = self.encoding_method == "frequency" + self.encoder_dict_ = {} - # learn encoding maps + # learn encoding maps. for var in variables_: - if self.encoding_method == "count": - self.encoder_dict_[var] = X[var].value_counts().to_dict() - - elif self.encoding_method == "frequency": - self.encoder_dict_[var] = X[var].value_counts(normalize=True).to_dict() - else: - raise ValueError( - "Unrecognized value for encoding_method. It should be 'count' or " - f"'frequency'. Got {self.encoding_method} instead." - ) + counts = nw_X.get_column(var).drop_nulls().value_counts( + sort=True, normalize=normalize + ) + keys = counts.get_column(counts.columns[0]).to_list() + values = counts.get_column(counts.columns[1]).to_list() + self.encoder_dict_[var] = dict(zip(keys, values)) # unseen categories are replaced by 0 if self.unseen == "encode": diff --git a/feature_engine/encoding/decision_tree.py b/feature_engine/encoding/decision_tree.py index 9b66fe32d..ee6dfbbd8 100644 --- a/feature_engine/encoding/decision_tree.py +++ b/feature_engine/encoding/decision_tree.py @@ -3,8 +3,9 @@ from typing import List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.pipeline import Pipeline from sklearn.utils.multiclass import check_classification_targets, type_of_target @@ -139,6 +140,11 @@ class DecisionTreeEncoder(CategoricalMethodsMixin, CategoricalInitMixin): fill_value: float, default=None The value used to encode unseen categories. Only used when `unseen='encode'`. + n_jobs: int, default=None + The number of jobs to run in parallel. `fit` is parallelized over the variables, + training one decision tree per variable. `None` means 1 unless in a + `joblib.parallel_backend` context. `-1` means using all processors. + Attributes ---------- encoder_dict_: @@ -212,6 +218,27 @@ class DecisionTreeEncoder(CategoricalMethodsMixin, CategoricalInitMixin): 2 3 0.666667 3 4 0.500000 4 5 0.500000 + + With polars: + + >>> import polars as pl + >>> X = pl.DataFrame(dict(x1 = [1,2,3,4,5], x2 = ["b", "b", "b", "a", "a"])) + >>> y = [0, 1, 1, 1, 0] + >>> dte = DecisionTreeEncoder(regression=False, cv=2) + >>> dte.fit(X, y) + >>> dte.transform(X) + shape: (5, 2) + ┌─────┬──────────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════╪══════════╡ + │ 1 ┆ 0.666667 │ + │ 2 ┆ 0.666667 │ + │ 3 ┆ 0.666667 │ + │ 4 ┆ 0.5 │ + │ 5 ┆ 0.5 │ + └─────┴──────────┘ """ def __init__( @@ -228,9 +255,13 @@ def __init__( precision: Optional[int] = None, unseen: str = "ignore", fill_value: Optional[float] = None, + n_jobs: Optional[int] = None, ) -> None: - if encoding_method not in ["ordered", "arbitrary"]: + if not isinstance(encoding_method, str) or encoding_method not in [ + "ordered", + "arbitrary", + ]: raise ValueError( "`encoding_method` takes only values 'ordered' and 'arbitrary'." f" Got {encoding_method} instead." @@ -261,22 +292,23 @@ def __init__( self.precision = precision self.unseen = unseen self.fill_value = fill_value + self.n_jobs = n_jobs - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Fit a decision tree per variable. Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the categorical variables. - y : pandas series. + y: Series. The target variable. Required to train the decision tree and for ordered ordinal encoding. """ - X, y = check_X_y(X, y) + nw_X, y = check_X_y(X, y) # confirm model type and target variables are compatible. if self.regression is True: @@ -309,59 +341,58 @@ def fit(self, X: pd.DataFrame, y: pd.Series): missing_values="raise", ignore_format=self.ignore_format, ) - tree = DecisionTreeDiscretiser( + variables=variables_, cv=self.cv, scoring=self.scoring, - variables=variables_, param_grid=param_grid, regression=self.regression, random_state=self.random_state, + n_jobs=self.n_jobs, ) - - # pipeline for the encoder - pipe = Pipeline( - [ - ("encoder", encoder), - ("tree", tree), - ] - ) - - Xt = pipe.fit_transform(X, y) - - encoder_ = {} - if self.precision is None: - for var in variables_: - encoder_[var] = dict(zip(X[var], Xt[var])) - else: - for var in variables_: - encoder_[var] = dict(zip(X[var], np.round(Xt[var], self.precision))) + Xt = Pipeline([("encoder", encoder), ("tree", tree)]).fit_transform(X, y) + nw_Xt = nw.from_native(Xt, eager_only=True) + + # map each category to the prediction of its tree + encoder_dict_ = {} + for var in variables_: + pairs = ( + nw_X.get_column(var) + .alias("__category__") + .to_frame() + .with_columns(nw_Xt.get_column(var).alias("__prediction__")) + .unique() + ) + preds = pairs["__prediction__"].to_numpy() + if self.precision is not None: + preds = np.round(preds, self.precision) + encoder_dict_[var] = dict(zip(pairs["__category__"].to_list(), preds)) if self.unseen == "encode": self._unseen = self.fill_value - self.encoder_dict_ = encoder_ + self.encoder_dict_ = encoder_dict_ self.variables_ = variables_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Replace categorical variables by the predictions of the decision tree. Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input samples. Returns ------- - X_new : pandas dataframe of shape = [n_samples, n_features]. + X_new: dataframe of shape = [n_samples, n_features]. Dataframe with variables encoded with decision tree predictions. """ - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) _check_contains_na(X, self.variables_) - X = self._encode(X) + X = self._encode(nw_X) return X diff --git a/feature_engine/encoding/mean_encoding.py b/feature_engine/encoding/mean_encoding.py index 145aa6394..10b223e5f 100644 --- a/feature_engine/encoding/mean_encoding.py +++ b/feature_engine/encoding/mean_encoding.py @@ -2,7 +2,9 @@ # License: BSD 3 clause from typing import List, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -28,7 +30,11 @@ ) from feature_engine._docstrings.substitute import Substitution from feature_engine.dataframe_checks import check_X_y -from feature_engine.encoding._helper_functions import check_parameter_unseen +from feature_engine.encoding._helper_functions import ( + TARGET_NAME, + add_target_to_X, + check_parameter_unseen, +) from feature_engine.encoding.base_encoder import ( CategoricalInitMixinNA, CategoricalMethodsMixin, @@ -203,64 +209,88 @@ def __init__( check_parameter_unseen(unseen, ["ignore", "raise", "encode"]) self.unseen = unseen - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Learn the mean value of the target for each category of the variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to be encoded. - y: pandas series + y: Series The target. """ - X, y = check_X_y(X, y) + nw_X, y = check_X_y(X, y) variables_ = self._check_or_select_variables(X) self._check_na(X, variables_) self.encoder_dict_ = {} - y_prior = y.mean() + nw_Xy = add_target_to_X(nw_X, y) + y_nw = nw_Xy[TARGET_NAME] + y_prior = y_nw.mean() if self.unseen == "encode": self._unseen = y_prior if self.smoothing == "auto": - y_var = y.var(ddof=0) - for var in variables_: - if self.smoothing == "auto": - damping = y.groupby(X[var]).var(ddof=0) / y_var - else: - damping = self.smoothing - counts = X[var].value_counts() - counts.index = counts.index.infer_objects() - _lambda = counts / (counts + damping) - self.encoder_dict_[var] = ( - _lambda * y.groupby(X[var], observed=False).mean() - + (1.0 - _lambda) * y_prior - ).to_dict() + y_var = y_nw.var(ddof=0) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X): + # pandas series with the index of X + y = y_nw.to_native() + for var in variables_: + if self.smoothing == "auto": + damping = y.groupby(X[var]).var(ddof=0) / y_var + else: + damping = self.smoothing + counts = X[var].value_counts() + counts.index = counts.index.infer_objects() + _lambda = counts / (counts + damping) + self.encoder_dict_[var] = ( + _lambda * y.groupby(X[var], observed=False).mean() + + (1.0 - _lambda) * y_prior + ).to_dict() + else: + for var in variables_: + stats = nw_Xy.group_by(var, drop_null_keys=True).agg( + nw.col(TARGET_NAME).mean().alias("__mean__"), + nw.col(TARGET_NAME).len().alias("__count__"), + nw.col(TARGET_NAME).var(ddof=0).alias("__var__"), + ) + if self.smoothing == "auto": + damping = nw.col("__var__") / y_var + else: + damping = self.smoothing + _lambda = nw.col("__count__") / (nw.col("__count__") + damping) + encoding = _lambda * nw.col("__mean__") + (1.0 - _lambda) * y_prior + stats = stats.select(var, encoding.alias("__encoding__")) + self.encoder_dict_[var] = dict( + zip(stats[var].to_list(), stats["__encoding__"].to_list()) + ) # assign underscore parameters at the end in case code above fails self.variables_ = variables_ self._get_feature_names_in(X) return self - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """Convert the encoded variable back to the original values. Note that if unseen was set to 'encode', then this method is not implemented. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The transformed dataframe. Returns ------- - X_tr: pandas dataframe of shape = [n_samples, n_features]. + X_tr: dataframe of shape = [n_samples, n_features]. The un-transformed dataframe, with the categorical variables containing the original values. """ diff --git a/feature_engine/encoding/one_hot.py b/feature_engine/encoding/one_hot.py index 9d028475b..2fae8cfae 100644 --- a/feature_engine/encoding/one_hot.py +++ b/feature_engine/encoding/one_hot.py @@ -3,8 +3,8 @@ from typing import List, Optional, Union -import numpy as np -import pandas as pd +import narwhals as nw +from narwhals.typing import IntoDataFrame from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, @@ -196,7 +196,7 @@ def __init__( self.drop_last = drop_last self.drop_last_binary = drop_last_binary - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoDataFrame] = None): """ Learns the unique categories per variable. If top_categories is indicated, it will learn the most popular categories. Alternatively, it learns all @@ -205,7 +205,7 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: pandas or polars dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just selected variables. @@ -214,84 +214,91 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): None. """ - X = check_X(X) + nw_X = check_X(X) variables_ = self._check_or_select_variables(X) _check_contains_na(X, variables_) - self.encoder_dict_ = {} + encoder_dict_ = {} for var in variables_: + col = nw_X.get_column(var) - # make dummies only for the most popular categories - if self.top_categories: - self.encoder_dict_[var] = [ - x - for x in X[var] - .value_counts() - .sort_values(ascending=False) - .head(self.top_categories) - .index - ] + if self.top_categories is not None: + top = col.value_counts(sort=True, name="count").head( + self.top_categories + ) + encoder_dict_[var] = top.get_column(var).to_list() else: - category_ls = list(X[var].unique()) - - # return k-1 dummies - if self.drop_last: - self.encoder_dict_[var] = category_ls[:-1] - - # return k dummies - else: - self.encoder_dict_[var] = category_ls + category_ls = col.unique(maintain_order=True).to_list() + # return k-1 vs k dummies + encoder_dict_[var] = ( + category_ls[:-1] if self.drop_last is True else category_ls + ) - self.variables_binary_ = [var for var in variables_ if X[var].nunique() == 2] + self.variables_binary_ = [ + var for var in variables_ if nw_X.get_column(var).n_unique() == 2 + ] # automatically encode binary variables as 1 dummy - if self.drop_last_binary: + if self.drop_last_binary is True: for var in self.variables_binary_: - category = X[var].unique()[0] - self.encoder_dict_[var] = [category] + category = nw_X.get_column(var).unique(maintain_order=True)[0] + encoder_dict_[var] = [category] self.variables_ = variables_ + self.encoder_dict_ = encoder_dict_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Replaces the categorical variables by the binary variables. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: pandas or polars dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe. + X_new: pandas or polars dataframe. The transformed dataframe. The shape of the dataframe will be different from the original as it includes the dummy variables in place of the original categorical ones. """ - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) # check if dataset contains na _check_contains_na(X, self.variables_) + dummy_frames = [] + # to_dummies() skips the prefix for falsy names (e.g. a column called 0), so + # use a placeholder name and swap in the real one by slicing its length + tmp_name = "__ohe_tmp__" for feature in self.variables_: - for category in self.encoder_dict_[feature]: - dummy_df = pd.DataFrame( - {f"{feature}_{category}": np.where(X[feature] == category, 1, 0)}, - index=X.index, + desired = [ + f"{feature}_{category}" for category in self.encoder_dict_[feature] + ] + dummies = nw_X.get_column(feature).alias(tmp_name).to_dummies(separator="_") + dummies = dummies.rename( + {c: f"{feature}{c[len(tmp_name):]}" for c in dummies.columns} + ) + # add all-0 columns for learned categories missing in X, and drop the + # columns of unseen categories, so these are encoded as 0 + missing = [c for c in desired if c not in dummies.columns] + if len(missing) > 0: + dummies = dummies.with_columns( + **{c: nw.lit(0, dtype=nw.Int8) for c in missing} ) - X = pd.concat([X, dummy_df], axis=1) + dummy_frames.append(dummies.select(desired)) - # drop the original non-encoded variables. - X.drop(labels=self.variables_, axis=1, inplace=True) + nw_X = nw.concat([nw_X.drop(*self.variables_), *dummy_frames], how="horizontal") - return X + return nw_X.to_native() - def inverse_transform(self, X: pd.DataFrame): + def inverse_transform(self, X: IntoDataFrame): """inverse_transform is not implemented for this transformer.""" raise NotImplementedError( "inverse_transform is not implemented for this transformer." diff --git a/feature_engine/encoding/ordinal.py b/feature_engine/encoding/ordinal.py index 10417f1d0..6e7a0f7a5 100644 --- a/feature_engine/encoding/ordinal.py +++ b/feature_engine/encoding/ordinal.py @@ -3,7 +3,9 @@ from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -29,7 +31,11 @@ ) from feature_engine._docstrings.substitute import Substitution from feature_engine.dataframe_checks import check_X, check_X_y -from feature_engine.encoding._helper_functions import check_parameter_unseen +from feature_engine.encoding._helper_functions import ( + TARGET_NAME, + add_target_to_X, + check_parameter_unseen, +) from feature_engine.encoding.base_encoder import ( CategoricalInitMixinNA, CategoricalMethodsMixin, @@ -177,9 +183,13 @@ def __init__( unseen: str = "ignore", ) -> None: - if encoding_method not in ["ordered", "arbitrary"]: + if not isinstance(encoding_method, str) or encoding_method not in [ + "ordered", + "arbitrary", + ]: raise ValueError( - "encoding_method takes only values 'ordered' and 'arbitrary'" + "encoding_method takes only values 'ordered' and 'arbitrary'. " + f"Got {encoding_method} instead." ) check_parameter_unseen(unseen, ["ignore", "raise", "encode"]) @@ -190,48 +200,63 @@ def __init__( self.unseen = unseen self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """Learn the numbers to be used to replace the categories in each variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to be encoded. - y: pandas series, default=None + y: Series, default=None The Target. Can be None if `encoding_method='arbitrary'`. Otherwise, y needs to be passed when fitting the transformer. """ if self.encoding_method == "ordered": - X, y = check_X_y(X, y) + nw_X, y = check_X_y(X, y) + nw_Xy = add_target_to_X(nw_X, y) else: - X = check_X(X) + nw_X = check_X(X) variables_ = self._check_or_select_variables(X) self._check_na(X, variables_) self.encoder_dict_ = {} - for var in variables_: + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X): if self.encoding_method == "ordered": - t = y.groupby(X[var], observed=False).mean() # type: ignore - t = t.sort_values(ascending=True).index - - elif self.encoding_method == "arbitrary": - if self.missing_values == "ignore": + # pandas series with the index of X + y_pd = nw_Xy[TARGET_NAME].to_native() + for var in variables_: + if self.encoding_method == "ordered": + t = y_pd.groupby(X[var], observed=False).mean().sort_values().index + elif self.missing_values == "ignore": t = X[var].dropna().unique() else: t = X[var].unique() - else: - raise ValueError( - "Unrecognized value for encoding_method. It should be 'arbitrary' " - f"or 'frequency'. Got {self.encoding_method} instead." - ) - - self.encoder_dict_[var] = {k: i for i, k in enumerate(t, 0)} + self.encoder_dict_[var] = {k: i for i, k in enumerate(t)} + else: + for var in variables_: + if self.encoding_method == "ordered": + # sort by mean, then category, so ties get the same order + # in every backend + t = ( + nw_Xy.group_by(var, drop_null_keys=True) + .agg(nw.col(TARGET_NAME).mean()) + .sort([TARGET_NAME, var]) + .get_column(var) + .to_list() + ) + else: + col = nw_X.get_column(var) + if self.missing_values == "ignore": + col = col.drop_nulls() + t = col.unique(maintain_order=True).to_list() + self.encoder_dict_[var] = {k: i for i, k in enumerate(t)} if self.unseen == "encode": self._unseen = -1 diff --git a/feature_engine/encoding/rare_label.py b/feature_engine/encoding/rare_label.py index 84c1f6910..720bf3dc9 100644 --- a/feature_engine/encoding/rare_label.py +++ b/feature_engine/encoding/rare_label.py @@ -4,8 +4,8 @@ import warnings from typing import List, Optional, Union -import numpy as np -import pandas as pd +import narwhals as nw +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -23,7 +23,7 @@ from feature_engine._docstrings.init_parameters.encoders import _ignore_format_docstring from feature_engine._docstrings.methods import _fit_transform_docstring from feature_engine._docstrings.substitute import Substitution -from feature_engine.dataframe_checks import _check_optional_contains_na, check_X +from feature_engine.dataframe_checks import _check_contains_na, check_X from feature_engine.encoding.base_encoder import ( CategoricalInitMixinNA, CategoricalMethodsMixin, @@ -137,6 +137,28 @@ class RareLabelEncoder(CategoricalMethodsMixin, CategoricalInitMixinNA): 3 4 b 4 5 b 5 6 Rare + + With polars + + >>> import polars as pl + >>> from feature_engine.encoding import RareLabelEncoder + >>> X = pl.DataFrame(dict(x1 = [1,2,3,4,5,6], x2 = ["b", "b", "b", "b", "b", "a"])) + >>> rle = RareLabelEncoder(n_categories = 1, tol=0.2) + >>> rle.fit(X) + >>> rle.transform(X) + shape: (6, 2) + ┌─────┬──────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ str │ + ╞═════╪══════╡ + │ 1 ┆ b │ + │ 2 ┆ b │ + │ 3 ┆ b │ + │ 4 ┆ b │ + │ 5 ┆ b │ + │ 6 ┆ Rare │ + └─────┴──────┘ """ def __init__( @@ -173,7 +195,7 @@ def __init__( if not isinstance(replace_with, (str, int, float)): raise ValueError( - "replace_with can should be a string, integer or float. " + "replace_with should be a string, integer or float. " f"Got {replace_with} instead." ) @@ -186,13 +208,13 @@ def __init__( self.replace_with = replace_with self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the frequent categories for each variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just selected variables @@ -200,26 +222,34 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): y is not required. You can pass y or None. """ - X = check_X(X) + nw_X = check_X(X) variables_ = self._check_or_select_variables(X) self._check_na(X, variables_) self.encoder_dict_ = {} + # n_unique() counts a null as its own category, matching pandas' + # plain unique() (used for the cardinality check below), unlike + # pandas' nunique() which drops nulls by default. for var in variables_: - if len(X[var].unique()) > self.n_categories: + col = nw_X.get_column(var) - # if the variable has more than the indicated number of categories - # the encoder will learn the most frequent categories - t = X[var].value_counts(normalize=True) + if col.n_unique() > self.n_categories: + + # learn the most frequent categories, dropping nulls and sorting + # by frequency like pandas' value_counts() + counts = col.drop_nulls().value_counts(sort=True, normalize=True) + cat_col, freq_col = counts.columns # non-rare labels: - freq_idx = t[t >= self.tol].index + freq_idx = counts.filter( + counts.get_column(freq_col) >= self.tol + ).get_column(cat_col).to_list() if self.max_n_categories: - self.encoder_dict_[var] = list(freq_idx[: self.max_n_categories]) + self.encoder_dict_[var] = freq_idx[: self.max_n_categories] else: - self.encoder_dict_[var] = list(freq_idx) + self.encoder_dict_[var] = freq_idx else: # if the total number of categories is smaller than the indicated @@ -229,56 +259,64 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): "indicated in n_categories. Thus, all categories will be " "considered frequent".format(var) ) - self.encoder_dict_[var] = list(X[var].unique()) + self.encoder_dict_[var] = col.unique(maintain_order=True).to_list() self.variables_ = variables_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Group infrequent categories. Replace infrequent categories by the string 'Rare' or any other name provided by the user. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input samples. Returns ------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The dataframe where rare categories have been grouped. """ - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) # check if dataset contains na if self.missing_values == "raise": - _check_optional_contains_na(X, self.variables_) - with_nan = [] - else: - with_nan = [np.nan] - + _check_contains_na(X, self.variables_, error_msg="optional") + + # pandas categorical columns need replace_with added to their categories; + # work on a copy so the user's dataframe is not changed + if nw_X.implementation.is_pandas(): + native_X = nw_X.to_native().copy() + for feature in self.variables_: + if native_X[feature].dtype == "category": + native_X[feature] = native_X[feature].cat.add_categories( + self.replace_with + ) + nw_X = nw.from_native(native_X, eager_only=True) + + # narwhals resolves the dtype when mixing each column with replace_with; + # series from get_column() also work with pandas integer column names + new_columns = [] for feature in self.variables_: - # Setting an item of incompatible dtype is deprecated - # and will raise an error in a future version of pandas - if self.ignore_format is True and isinstance(self.replace_with, str): - num_vars = list( - X[self.variables_].select_dtypes(include="number").columns + col = nw_X.get_column(feature) + keep = col.is_in(self.encoder_dict_[feature]) + if self.missing_values == "ignore": + keep = keep | col.is_null() + new_columns.append( + nw.when(keep).then(col).otherwise(nw.lit(self.replace_with)).alias( + feature ) - X[num_vars] = X[num_vars].astype("O") - - if X[feature].dtype == "category": - X[feature] = X[feature].cat.add_categories(self.replace_with) - - X.loc[~X[feature].isin(self.encoder_dict_[feature] + with_nan), feature] = ( - self.replace_with ) + X = nw_X.with_columns(*new_columns).to_native() + return X - def inverse_transform(self, X: pd.DataFrame): + def inverse_transform(self, X: IntoDataFrame): """inverse_transform is not implemented for this transformer.""" raise NotImplementedError( "inverse_transform is not implemented for this transformer." diff --git a/feature_engine/encoding/similarity_encoder.py b/feature_engine/encoding/similarity_encoder.py index c438487d6..f56219bd6 100644 --- a/feature_engine/encoding/similarity_encoder.py +++ b/feature_engine/encoding/similarity_encoder.py @@ -1,8 +1,9 @@ from difflib import SequenceMatcher from typing import List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.utils.validation import check_is_fitted from feature_engine._docstrings.fit_attributes import ( @@ -17,7 +18,7 @@ from feature_engine._docstrings.init_parameters.encoders import _ignore_format_docstring from feature_engine._docstrings.methods import _fit_transform_docstring from feature_engine._docstrings.substitute import Substitution -from feature_engine.dataframe_checks import _check_optional_contains_na, check_X +from feature_engine.dataframe_checks import _check_contains_na, check_X from feature_engine.encoding.base_encoder import ( CategoricalInitMixin, CategoricalMethodsMixin, @@ -183,6 +184,26 @@ class StringSimilarityEncoder(CategoricalMethodsMixin, CategoricalInitMixin): 1 2 0.666667 1.000000 0.444444 0.4 2 3 0.444444 0.444444 1.000000 0.0 3 4 0.000000 0.400000 0.000000 1.0 + + With polars + + >>> import polars as pl + >>> from feature_engine.encoding import StringSimilarityEncoder + >>> X = pl.DataFrame(dict(x1 = [1,2,3,4], x2 = ["dog", "dig", "dagger", "hi"])) + >>> sse = StringSimilarityEncoder() + >>> sse.fit(X) + >>> sse.transform(X) + shape: (4, 5) + ┌─────┬──────────┬──────────┬───────────┬───────┐ + │ x1 ┆ x2_dog ┆ x2_dig ┆ x2_dagger ┆ x2_hi │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ i64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞═════╪══════════╪══════════╪═══════════╪═══════╡ + │ 1 ┆ 1.0 ┆ 0.666667 ┆ 0.444444 ┆ 0.0 │ + │ 2 ┆ 0.666667 ┆ 1.0 ┆ 0.444444 ┆ 0.4 │ + │ 3 ┆ 0.444444 ┆ 0.444444 ┆ 1.0 ┆ 0.0 │ + │ 4 ┆ 0.0 ┆ 0.4 ┆ 0.0 ┆ 1.0 │ + └─────┴──────────┴──────────┴───────────┴───────┘ """ def __init__( @@ -198,7 +219,11 @@ def __init__( raise ValueError( f"top_categories takes only integers. Got {top_categories!r} instead." ) - if missing_values not in ("raise", "impute", "ignore"): + if not isinstance(missing_values, str) or missing_values not in ( + "raise", + "impute", + "ignore", + ): raise ValueError( "missing_values should be one of 'raise', 'impute' or 'ignore'." f" Got {missing_values!r} instead." @@ -217,7 +242,7 @@ def __init__( self.missing_values = missing_values self.keywords = keywords - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learns the unique categories per variable. If top_categories is indicated, it will learn the most popular categories. Alternatively, it learns all @@ -226,15 +251,15 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to encode. - y: pandas series, default=None + y: Series, default=None Target. It is not needed in this encoder. You can pass y or None. """ - X = check_X(X) + nw_X = check_X(X) variables_ = self._check_or_select_variables(X) if self.keywords and not all( @@ -247,117 +272,104 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): # if data contains nan, fail before running any logic if self.missing_values == "raise": - _check_optional_contains_na(X, variables_) + _check_contains_na(X, variables_, error_msg="optional") - self.encoder_dict_ = {} + encoder_dict_ = {} if self.keywords: - self.encoder_dict_.update(self.keywords) + encoder_dict_.update(self.keywords) cols_to_iterate = [x for x in variables_ if x not in self.keywords] else: cols_to_iterate = variables_ - if self.missing_values == "raise": - for var in cols_to_iterate: - self.encoder_dict_[var] = ( - X[var] - .astype(str) - .value_counts() - .head(self.top_categories) - .index.tolist() - ) - elif self.missing_values == "impute": - for var in cols_to_iterate: - series = X[var] - self.encoder_dict_[var] = ( - series.astype(str) - .mask(series.isna(), "") - .value_counts() - .head(self.top_categories) - .index.tolist() - ) - elif self.missing_values == "ignore": - for var in cols_to_iterate: - self.encoder_dict_[var] = ( - X[var] - .dropna() - .astype(str) - .value_counts(dropna=True) - .head(self.top_categories) - .index.tolist() - ) - else: - raise ValueError( - "Unrecognized value for missing_values. It should be 'raise', 'ignore' " - f"or 'impute'. Got {self.missing_values} instead." - ) + # cast(nw.String) keeps nulls as nulls, unlike pandas' astype(str) + for var in cols_to_iterate: + col = nw_X.get_column(var) + if self.missing_values == "impute": + col = col.cast(nw.String).fill_null("") + elif self.missing_values == "ignore": + col = col.drop_nulls().cast(nw.String) + else: + col = col.cast(nw.String) + + # sort=True mirrors pandas' own value_counts() default order + # (descending by count, ties broken by first appearance), so + # encoder_dict_ keeps the same category order as before. + counts = col.value_counts(sort=True) + categories = counts.get_column(counts.columns[0]).to_list() + encoder_dict_[var] = categories[: self.top_categories] # assign underscore parameters at the end in case code above fails self.variables_ = variables_ + self.encoder_dict_ = encoder_dict_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Replaces the categorical variables with the similarity variables. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe. + X_new: dataframe. The transformed dataframe. The shape of the dataframe will be different from the original as it includes the similarity variables in place of the original categorical ones. """ check_is_fitted(self) - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) if self.missing_values == "raise": - _check_optional_contains_na(X, self.variables_) + _check_contains_na(X, self.variables_, error_msg="optional") if len(self.variables_) == 0: - return X + return nw_X.to_native() - new_values = [] + # difflib has no vectorised equivalent, so similarities are computed in numpy + # once per unique value, then mapped back to the rows + new_series = [] for var in self.variables_: + col = nw_X.get_column(var) + categories = self.encoder_dict_[var] + + null_mask = None if self.missing_values == "impute": - series = X[var] - series = series.astype(str).mask(series.isna(), "") + str_col = col.cast(nw.String).fill_null("") else: - series = X[var].astype(str) - - categories = series.unique() - column_encoder_dict = { - x: _gpm_fast_vec(x, self.encoder_dict_[var]) for x in categories - } - # Ensure map result is always an array of the correct size. - # Missing values in categories or unknown categories will map to NaN. - default_nan = np.full(len(self.encoder_dict_[var]), np.nan) - if "nan" not in column_encoder_dict: - column_encoder_dict["nan"] = default_nan - if "" not in column_encoder_dict: - column_encoder_dict[""] = default_nan - - encoded_series = series.map(column_encoder_dict) - - # Robust stacking: replace any float NaNs (from unknown values) with arrays - encoded_list = [ - v if isinstance(v, (list, np.ndarray)) else default_nan - for v in encoded_series - ] - encoded = np.vstack(encoded_list) - if self.missing_values == "ignore": - encoded[X[var].isna(), :] = np.nan - new_values.append(encoded) - - new_features = self._get_new_features_name() - X.loc[:, new_features] = np.hstack(new_values) - - return X.drop(self.variables_, axis=1) + str_col = col.cast(nw.String) + if self.missing_values == "ignore": + null_mask = np.array(col.is_null().to_list()) + + values = np.asarray(str_col.to_list(), dtype=object) + if null_mask is not None: + # placeholder value for null rows: overwritten with NaN + # below, the string itself is never used. + values = np.where(null_mask, "", values) + + unique_vals, inverse = np.unique(values, return_inverse=True) + cats_arr = np.asarray(categories, dtype=object) + sim_matrix = _gpm_fast_vec( + unique_vals.reshape(-1, 1), cats_arr.reshape(1, -1) + ) + encoded = sim_matrix[inverse] + + if null_mask is not None: + encoded[null_mask, :] = np.nan + + for j, category in enumerate(categories): + name = f"{var}_nan" if category == "" else f"{var}_{category}" + new_series.append( + nw.new_series(name, encoded[:, j], backend=nw_X.implementation) + ) + + nw_X = nw_X.with_columns(*new_series).drop(self.variables_) + + return nw_X.to_native() def _get_new_features_name(self) -> List[str]: """Return names of the created features.""" @@ -378,7 +390,7 @@ def _add_new_feature_names(self, feature_names: List[str]) -> List[str]: return feature_names - def inverse_transform(self, X: pd.DataFrame): + def inverse_transform(self, X: IntoDataFrame): """inverse_transform is not implemented for this transformer.""" raise NotImplementedError( "inverse_transform is not implemented for this transformer." diff --git a/feature_engine/encoding/woe.py b/feature_engine/encoding/woe.py index bd1538a2c..8f435fa11 100644 --- a/feature_engine/encoding/woe.py +++ b/feature_engine/encoding/woe.py @@ -3,8 +3,8 @@ from typing import List, Union -import numpy as np -import pandas as pd +import narwhals as nw +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, @@ -26,7 +26,11 @@ ) from feature_engine._docstrings.substitute import Substitution from feature_engine.dataframe_checks import _check_contains_na, check_X_y -from feature_engine.encoding._helper_functions import check_parameter_unseen +from feature_engine.encoding._helper_functions import ( + TARGET_NAME, + add_target_to_X, + check_parameter_unseen, +) from feature_engine.encoding.base_encoder import ( CategoricalInitMixin, CategoricalMethodsMixin, @@ -35,14 +39,16 @@ class WoE: - def _check_fit_input(self, X: pd.DataFrame, y: pd.Series): + def _check_fit_input(self, X: IntoDataFrame, y: IntoSeries): """ Check that X is dataframe, and y a binary series with values 0 and 1. """ - X, y = check_X_y(X, y) + nw_X, y = check_X_y(X, y) + # with pandas, y takes the index of X + y_nw = add_target_to_X(nw_X, y)[TARGET_NAME] # check that y is binary - if y.nunique() != 2: + if y_nw.n_unique() != 2: raise ValueError( "This encoder is designed for binary classification. The target " "used has more than 2 unique values." @@ -50,38 +56,50 @@ def _check_fit_input(self, X: pd.DataFrame, y: pd.Series): # if target does not have values 0 and 1, we need to remap, to be able to # compute the averages. - if y.min() != 0 or y.max() != 1: - y = pd.Series(np.where(y == y.min(), 0, 1)) - return X, y + y_min, y_max = y_nw.min(), y_nw.max() + if y_min != 0 or y_max != 1: + y_nw = (y_nw != y_min).cast(nw.Int64()).alias("target") + + return X, y_nw.to_native() def _calculate_woe( self, - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y: IntoSeries, variable: Union[str, int], - fill_value: Union[float, None] = None, ): - total_pos = y.sum() - inverse_y = y.ne(1).copy() - total_neg = inverse_y.sum() - - pos = y.groupby(X[variable], observed=False).sum() / total_pos - neg = inverse_y.groupby(X[variable], observed=False).sum() / total_neg - - if not (pos[:] == 0).sum() == 0 or not (neg[:] == 0).sum() == 0: - if fill_value is None: - raise ValueError( - "The proportion of one of the classes for a category in " - "variable {} is zero, and log of zero is not defined".format( - variable - ) - ) - else: - pos[pos[:] == 0] = fill_value - neg[neg[:] == 0] = fill_value - - woe = np.log(pos / neg) - return pos, neg, woe + """ + Return a narwhals dataframe with one row per category of the variable and the + columns __category__, __pos__ and __neg__, the fraction of positive and + negative cases, and __woe__, the weight of evidence. Also return whether any + category has no positive or no negative cases. + """ + # narwhals expressions need string column names, pandas allows integers + col = nw.from_native(X, eager_only=True).get_column(variable) + nw_Xy = add_target_to_X(col.alias("__category__").to_frame(), y) + total_pos = nw_Xy[TARGET_NAME].sum() + total_neg = len(nw_Xy) - total_pos + + counts = ( + nw_Xy.group_by("__category__", drop_null_keys=True) + .agg(nw.col(TARGET_NAME).sum().alias("__pos__"), nw.len().alias("__n__")) + .sort("__category__") + .with_columns((nw.col("__n__") - nw.col("__pos__")).alias("__neg__")) + ) + pos, neg = nw.col("__pos__"), nw.col("__neg__") + has_zero_counts = bool(counts.select(((pos == 0) | (neg == 0)).any()).item()) + + # the WoE is not defined for zero counts, so they are replaced by 0.5 + pos = nw.when(pos == 0).then(0.5).otherwise(pos) / total_pos + neg = nw.when(neg == 0).then(0.5).otherwise(neg) / total_neg + + woe = counts.select( + "__category__", + pos.alias("__pos__"), + neg.alias("__neg__"), + (pos / neg).log().alias("__woe__"), + ) + return woe, has_zero_counts @Substitution( @@ -119,10 +137,10 @@ class WoEEncoder(CategoricalMethodsMixin, CategoricalInitMixin, WoE): **Note** - The log(0) is not defined and the division by 0 is not defined. Thus, if any of the - terms in the WoE equation are 0 for a given category, the encoder will return an - error. If this happens, try grouping less frequent categories. Alternatively, - you can now add a fill_value (see parameter below). + The WoE is not defined for categories with no positive or no negative cases. For + those categories, the encoder replaces the zero count by 0.5, and lists the + variables in `variables_with_zero_counts_`. Grouping infrequent categories before + the encoding reduces how often this happens. More details in the :ref:`User Guide `. @@ -136,17 +154,15 @@ class WoEEncoder(CategoricalMethodsMixin, CategoricalInitMixin, WoE): {unseen} - fill_value: int, float, default=None - When the numerator or denominator of the WoE calculation are zero, the WoE - calculation is not possible. If `fill_value` is None (recommended), an error - will be raised in those cases. Alternatively, fill_value will be used in place - of denominators or numerators that equal zero. - Attributes ---------- encoder_dict_: Dictionary with the WoE per variable. + variables_with_zero_counts_: + List of variables with categories that have no positive or no negative cases. + For those categories, 0.5 replaces the zero count to calculate the WoE. + {variables_} {feature_names_in_} @@ -198,6 +214,28 @@ class WoEEncoder(CategoricalMethodsMixin, CategoricalInitMixin, WoE): 2 3 0.287682 3 4 -0.405465 4 5 -0.405465 + + With polars + + >>> import polars as pl + >>> from feature_engine.encoding import WoEEncoder + >>> X = pl.DataFrame(dict(x1 = [1,2,3,4,5], x2 = ["b", "b", "b", "a", "a"])) + >>> y = pl.Series([0,1,1,1,0]) + >>> woe = WoEEncoder() + >>> woe.fit(X, y) + >>> woe.transform(X) + shape: (5, 2) + ┌─────┬───────────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════╪═══════════╡ + │ 1 ┆ 0.287682 │ + │ 2 ┆ 0.287682 │ + │ 3 ┆ 0.287682 │ + │ 4 ┆ -0.405465 │ + │ 5 ┆ -0.405465 │ + └─────┴───────────┘ """ def __init__( @@ -206,29 +244,23 @@ def __init__( return_empty: bool = False, ignore_format: bool = False, unseen: str = "ignore", - fill_value: Union[int, float, None] = None, ) -> None: super().__init__(variables, return_empty, ignore_format) check_parameter_unseen(unseen, ["ignore", "raise"]) - if fill_value is not None and not isinstance(fill_value, (int, float)): - raise ValueError( - f"fill_value takes None, integer or float. Got {fill_value} instead." - ) self.unseen = unseen - self.fill_value = fill_value - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Learn the WoE. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the categorical variables. - y: pandas series. + y: Series. Target, must be binary. """ X, y = self._check_fit_input(X, y) @@ -236,63 +268,45 @@ def fit(self, X: pd.DataFrame, y: pd.Series): _check_contains_na(X, variables_) encoder_dict_ = {} - vars_that_fail = [] + variables_with_zero_counts_ = [] for var in variables_: - try: - _, _, woe = self._calculate_woe(X, y, var, self.fill_value) - encoder_dict_[var] = woe.to_dict() - except ValueError: - vars_that_fail.append(var) - - if len(vars_that_fail) > 0: - vars_that_fail_str = ( - ", ".join(vars_that_fail) - if len(vars_that_fail) > 1 - else vars_that_fail[0] - ) - - raise ValueError( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - f"and hence the WoE can't be calculated: {vars_that_fail_str}." + woe, has_zero_counts = self._calculate_woe(X, y, var) + encoder_dict_[var] = dict( + zip(woe["__category__"].to_list(), woe["__woe__"].to_list()) ) + if has_zero_counts is True: + variables_with_zero_counts_.append(var) self.encoder_dict_ = encoder_dict_ + self.variables_with_zero_counts_ = variables_with_zero_counts_ self.variables_ = variables_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """Replace categories with the learned parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The dataset to transform. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features]. + X_new: dataframe of shape = [n_samples, n_features]. The dataframe containing the categories replaced by numbers. """ - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) _check_contains_na(X, self.variables_) - X = self._encode(X) + X = self._encode(nw_X) return X def _more_tags(self): tags_dict = _return_tags() tags_dict["variables"] = "categorical" tags_dict["requires_y"] = True - # in the current format, the tests are performed using continuous np.arrays - # this means that when we encode some of the values, the denominator is 0 - # and this the transformer raises an error, and the test fails. - # For this reason, most sklearn tests will fail. And it has nothing to - # do with the class not being compatible, it is just that the inputs passed - # are not suitable - tags_dict["_skip_test"] = True return tags_dict def __sklearn_tags__(self): diff --git a/feature_engine/imputation/arbitrary_imputer.py b/feature_engine/imputation/arbitrary_imputer.py index 333a81918..be130fba9 100644 --- a/feature_engine/imputation/arbitrary_imputer.py +++ b/feature_engine/imputation/arbitrary_imputer.py @@ -1,11 +1,10 @@ # Authors: Soledad Galli # License: BSD 3 clause +import warnings from typing import List, Optional, Union -import pandas as pd - -import warnings +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_input_dictionary import ( _check_numerical_dict, @@ -120,6 +119,28 @@ class ArbitraryImputer(BaseImputer): 2 1.0 b 3 0.0 NaN 4 -999.0 a + + With polars: + + >>> import polars as pl + >>> from feature_engine.imputation import ArbitraryImputer + >>> X = pl.DataFrame({"x1": [None, 1, 1, 0, None], + >>> "x2": ["a", None, "b", None, "a"]}) + >>> ai = ArbitraryImputer(arbitrary_number=-999) + >>> ai.fit(X) + >>> ai.transform(X) + shape: (5, 2) + ┌──────┬──────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ str │ + ╞══════╪══════╡ + │ -999 ┆ a │ + │ 1 ┆ null │ + │ 1 ┆ b │ + │ 0 ┆ null │ + │ -999 ┆ a │ + └──────┴──────┘ """ def __init__( @@ -133,7 +154,10 @@ def __init__( if isinstance(arbitrary_number, int) or isinstance(arbitrary_number, float): self.arbitrary_number = arbitrary_number else: - raise ValueError("arbitrary_number must be numeric of type int or float") + raise ValueError( + "arbitrary_number must be numeric of type int or float. " + f"Got {arbitrary_number} instead." + ) _check_numerical_dict(imputer_dict) @@ -144,13 +168,13 @@ def __init__( self.imputer_dict = imputer_dict - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This method does not learn any parameter. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. y: None @@ -158,11 +182,11 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): """ # check input dataframe - X = check_X(X) + check_X(X) # find or check for numerical variables # create the imputer dictionary - if self.imputer_dict: + if self.imputer_dict is not None: variables_ = check_numerical_variables( X, list(self.imputer_dict.keys()) ) diff --git a/feature_engine/imputation/base_imputer.py b/feature_engine/imputation/base_imputer.py index f9c3a2fea..c65ce4a3f 100644 --- a/feature_engine/imputation/base_imputer.py +++ b/feature_engine/imputation/base_imputer.py @@ -1,4 +1,6 @@ -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -6,13 +8,11 @@ from feature_engine.dataframe_checks import _check_X_matches_training_df, check_X from feature_engine.tags import _return_tags -_PANDAS_LT_3 = int(pd.__version__.split(".")[0]) < 3 - class BaseImputer(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): """shared set-up checks and methods across imputers""" - def _transform(self, X: pd.DataFrame) -> pd.DataFrame: + def _transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Common checks before transforming data: @@ -23,59 +23,56 @@ def _transform(self, X: pd.DataFrame) -> pd.DataFrame: Parameters ---------- - X: Pandas DataFrame + X: dataframe of shape = [n_samples, n_features] Returns ------- - X: Pandas DataFrame - The same dataframe entered by the user. + X: narwhals dataframe. + The narwhalified version of the dataframe entered by the user. """ - # Check method fit has been called check_is_fitted(self) - - # check that input is a dataframe - X = check_X(X) - - # Check that input df contains same number of columns as df used to fit + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + return nw_X - return X - - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Replace missing data with the learned parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The dataframe without missing values in the selected variables. """ + nw_X = self._transform(X) - X = self._transform(X) - - # Replace missing data with learned parameters. In pandas < 3, fillna - # downcasts object columns and warns; the option applies the pandas 3 - # behavior: no downcasting, and infer_objects restores numeric dtypes. - if _PANDAS_LT_3: - with pd.option_context("future.no_silent_downcasting", True): - X = X.fillna(value=self.imputer_dict_) - else: + # pandas-native fillna is ~1.3-1.6x faster than narwhals-generic + # fill_null equivalent at the 10k-100k + if nwd.is_pandas_dataframe(X): X = X.fillna(value=self.imputer_dict_) - return X.infer_objects() + X = X.infer_objects() + else: + nw_X = nw_X.with_columns( + nw.col(var).fill_null(value) + for var, value in self.imputer_dict_.items() + ) + X = nw_X.to_native() + + return X def _get_feature_names_in(self, X): """Get the names and number of features in the train set (the dataframe used during fit).""" - - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X): + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/feature_engine/imputation/categorical.py b/feature_engine/imputation/categorical.py index 2cd4a00a3..94b9e0d79 100644 --- a/feature_engine/imputation/categorical.py +++ b/feature_engine/imputation/categorical.py @@ -3,7 +3,9 @@ from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_variables import ( _check_variables_input_value, @@ -146,13 +148,26 @@ def __init__( return_object: bool = False, ignore_format: bool = False, ) -> None: - if imputation_method not in ["missing", "frequent"]: + if not isinstance(imputation_method, str) or imputation_method not in [ + "missing", + "frequent", + ]: raise ValueError( - "imputation_method takes only values 'missing' or 'frequent'" + "imputation_method takes only values 'missing' or 'frequent'. " + f"Got {imputation_method} instead." ) if not isinstance(ignore_format, bool): - raise ValueError("ignore_format takes only booleans True and False") + raise ValueError( + "ignore_format takes only booleans True and False. " + f"Got {ignore_format} instead." + ) + + if not isinstance(return_object, bool): + raise ValueError( + "return_object takes only booleans True and False. " + f"Got {return_object} instead." + ) self.imputation_method = imputation_method self.fill_value = fill_value @@ -162,21 +177,22 @@ def __init__( _check_return_empty_is_bool(return_empty) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the most frequent category if the imputation method is set to frequent. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The training dataset. + X: dataframe of shape = [n_samples, n_features] + The training dataset. Can be a pandas, polars, or any other dataframe + supported by narwhals. - y: pandas Series, default=None + y: Series, default=None y is not needed in this imputation. You can pass None or y. """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) # select variables to encode if self.ignore_format is True: @@ -194,38 +210,14 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): imputer_dict_ = {var: self.fill_value for var in variables_} elif self.imputation_method == "frequent": - # if imputing only 1 variable: - if len(variables_) == 1: - var = variables_[0] - mode_vals = X[var].mode() - - # Some variables may contain more than 1 mode: - if len(mode_vals) > 1: - raise ValueError( - f"The variable {var} contains multiple frequent categories." - ) - - imputer_dict_ = {var: mode_vals[0]} - - # imputing multiple variables: - else: - # Returns a dataframe with 1 row if there is one mode per - # variable, or more rows if there are more modes: - mode_vals = X[variables_].mode() - - # Careful: some variables contain multiple modes - if len(mode_vals) > 1: - varnames = mode_vals.dropna(axis=1).columns.to_list() - if len(varnames) > 1: - varnames_str = ", ".join(varnames) - else: - varnames_str = varnames[0] - raise ValueError( - f"The variable(s) {varnames_str} contain(s) multiple frequent " - f"categories." - ) - - imputer_dict_ = mode_vals.iloc[0].to_dict() + imputer_dict_ = {} + for var in variables_: + # polars' mode() keeps nulls (unlike pandas' default), so drop + # them first. When a variable has several equally-frequent + # categories, sort and take the smallest so fit() is + # reproducible and pandas and polars agree. + modes = sorted(nw_X[var].drop_nulls().mode(keep="all").to_list()) + imputer_dict_[var] = modes[0] self.variables_ = variables_ self.imputer_dict_ = imputer_dict_ @@ -233,28 +225,65 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: # Frequent category imputation if self.imputation_method == "frequent": X = super().transform(X) # Imputation with string else: - X = self._transform(X) - - # if variable is of type category, we need to add the new - # category, before filling in the nan - for variable in self.variables_: - if X[variable].dtype.name == "category": - X[variable] = X[variable].cat.add_categories( - self.imputer_dict_[variable] - ) - - X = X.fillna(self.imputer_dict_) + nw_X = self._transform(X) + + if nwd.is_pandas_dataframe(X): + # if variable is of type category, we need to add the new + # category, before filling in the nan. Copy first so the + # in-place column reassignment doesn't mutate the caller's + # dataframe (BaseImputer._transform no longer returns a copy). + cat_vars = [ + var + for var in self.variables_ + if X[var].dtype.name == "category" + ] + if cat_vars: + X = X.copy() + for variable in cat_vars: + X[variable] = X[variable].cat.add_categories( + self.imputer_dict_[variable] + ) + + X = X.fillna(self.imputer_dict_) + else: + schema = nw_X.schema + for variable in self.variables_: + dtype = schema[variable] + fill_value = self.imputer_dict_[variable] + # polars' Categorical widens itself on fill_null, but its + # Enum has a fixed category set and silently fills with + # null (no error) if fill_value isn't already a member. + if isinstance(dtype, nw.Enum) and ( + fill_value not in dtype.categories + ): + raise ValueError( + f"Cannot fill variable '{variable}' with " + f"'{fill_value}': it is a polars Enum with fixed " + f"categories {dtype.categories} that do not include " + "the fill value. Cast the column to Categorical or " + "String before imputing." + ) + + nw_X = nw_X.with_columns( + nw.col(var).fill_null(value) + for var, value in self.imputer_dict_.items() + ) + X = nw_X.to_native() # add additional step to return variables cast as object - if self.return_object: - X[self.variables_] = X[self.variables_].astype("O") + if self.return_object is True: + if nwd.is_pandas_dataframe(X): + X[self.variables_] = X[self.variables_].astype("O") + # polars/narwhals backends never silently upcast a string-typed + # column back to numeric (unlike pandas' fillna+infer_objects), + # so there is nothing to recast there. return X diff --git a/feature_engine/imputation/drop_missing_data.py b/feature_engine/imputation/drop_missing_data.py index 5be7a19cb..af91e99fc 100644 --- a/feature_engine/imputation/drop_missing_data.py +++ b/feature_engine/imputation/drop_missing_data.py @@ -3,7 +3,9 @@ from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.mixins import TransformXyMixin from feature_engine._check_init_parameters.check_variables import ( @@ -114,6 +116,26 @@ class DropMissingData(BaseImputer, TransformXyMixin): >>> dmd.transform(X) x1 x2 2 1.0 b + + With polars: + + >>> import polars as pl + >>> from feature_engine.imputation import DropMissingData + >>> X = pl.DataFrame(dict( + ... x1 = [None, 1, 1, 0, None], + ... x2 = ["a", None, "b", None, "a"], + ... )) + >>> dmd = DropMissingData() + >>> dmd.fit(X) + >>> dmd.transform(X) + shape: (1, 2) + ┌─────┬─────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ str │ + ╞═════╪═════╡ + │ 1 ┆ b │ + └─────┴─────┘ """ def __init__( @@ -144,69 +166,69 @@ def __init__( _check_return_empty_is_bool(return_empty) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Find the variables for which missing data should be evaluated to decide if a row should be dropped. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training data set. - y: pandas Series or dataframe, default=None + y: Series or dataframe, default=None y is not needed in this imputation. You can pass None or y. """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) # find variables for which indicator should be added if self.variables is None: - variables_ = find_all_variables(X, self.return_empty) + variables_ = find_all_variables(X, return_empty=self.return_empty) else: variables_ = check_all_variables(X, self.variables) # If user passes a threshold, then missing_only is ignored: if self.threshold is None and self.missing_only is True: - variables_ = [var for var in variables_ if X[var].isnull().sum() > 0] + # Benchmarked: a per-column isnull().sum() loop beats a narwhals- + # generic call on pandas input, matching MissingIndicator's split. + if nwd.is_pandas_dataframe(X): + variables_ = [ + var for var in variables_ if X[var].isnull().sum() > 0 + ] + else: + null_counts = nw_X.select(variables_).null_count().row(0) + variables_ = [ + var for var, count in zip(variables_, null_counts) if count > 0 + ] self.variables_ = variables_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Remove rows with missing data. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The dataframe to be transformed. Returns ------- - X_new: pandas dataframe + X_new: dataframe The complete case dataframe for the selected variables, of shape [n_samples - n_samples_with_na, n_features] """ - X = self._transform(X) - - if self.threshold: - X.dropna( - thresh=len(self.variables_) * self.threshold, - subset=self.variables_, - axis=0, - inplace=True, - ) - else: - X.dropna(axis=0, how="any", subset=self.variables_, inplace=True) - - return X + nw_X = self._transform(X) + # TODO + return self._select_rows(nw_X, keep=True) - def return_na_data(self, X: pd.DataFrame) -> pd.DataFrame: + def return_na_data(self, X: IntoDataFrame) -> IntoDataFrame: """ Returns the subset of the dataframe with the rows with missing values. That is, the subset of the dataframe that would be removed with the `transform()` method. @@ -215,20 +237,62 @@ def return_na_data(self, X: pd.DataFrame) -> pd.DataFrame: Parameters ---------- - X_na: pandas dataframe of shape = [n_samples_with_na, features] + X: dataframe of shape = [n_samples, n_features] + The dataframe to be transformed. + + Returns + ------- + X_na: dataframe of shape = [n_samples_with_na, features] The subset of the dataframe with the rows with missing data. """ X = self._transform(X) + return self._select_rows(X, keep=False) - if self.threshold: - idx = pd.isnull(X[self.variables_]).mean(axis=1) >= self.threshold - idx = idx[idx] + def _select_rows(self, X: IntoDataFrame, keep: bool) -> IntoDataFrame: + """ + Shared row-selection logic for transform() (keep=True, rows without + missing data) and return_na_data() (keep=False, rows with missing + data). Deriving both from the same "keep" condition, negated for the + drop side, guarantees the two outputs are always an exact partition + of X - they can never overlap or leave a row out. + """ + if len(self.variables_) == 0: + # dropna(subset=[]) keeps every row: there are no variables to + # evaluate missingness on, so nothing can ever be "missing". + if keep is True: + return X.to_native() + return X.head(0).to_native() + + # Benchmarked: a numpy-backed mask beats both pandas' own axis=1 + # isnull()/notna().sum() (a known-slow reduction) and the narwhals + # path below, so pandas keeps this dedicated fast path. X is the + # narwhals frame returned by check_X, so branch on its implementation. + if X.implementation.is_pandas(): + X = X.to_native() + if self.threshold is not None: + non_null_count = X[self.variables_].notna().to_numpy().sum(axis=1) + mask = non_null_count >= len(self.variables_) * self.threshold + else: + mask = ~X[self.variables_].isnull().to_numpy().any(axis=1) + if keep is False: + mask = ~mask + return X[mask] else: - idx = pd.isnull(X[self.variables_]).any(axis=1) - idx = idx[idx] - - return X.loc[idx.index, :] + if self.threshold is not None: + non_null_count = nw.sum_horizontal( + (~nw.col(var).is_null()).cast(nw.Int64) + for var in self.variables_ + ) + expr = non_null_count >= len(self.variables_) * self.threshold + else: + expr = ~nw.any_horizontal( + (nw.col(var).is_null() for var in self.variables_), + ignore_nulls=True, + ) + if keep is False: + expr = ~expr + return X.filter(expr).to_native() def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/imputation/end_tail.py b/feature_engine/imputation/end_tail.py index e52500056..677a43b68 100644 --- a/feature_engine/imputation/end_tail.py +++ b/feature_engine/imputation/end_tail.py @@ -3,7 +3,8 @@ from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_variables import ( _check_variables_input_value, @@ -140,6 +141,27 @@ class EndTailImputer(BaseImputer): 2 0.500000 3 0.000000 4 1.199359 + + With polars: + + >>> import polars as pl + >>> from feature_engine.imputation import EndTailImputer + >>> X = pl.DataFrame({"x1": [None, 0.5, 0.5, 0.0, None]}) + >>> eti = EndTailImputer(imputation_method='gaussian', tail='right', fold=3) + >>> eti.fit(X) + >>> eti.transform(X) + shape: (5, 1) + ┌──────────┐ + │ x1 │ + │ --- │ + │ f64 │ + ╞══════════╡ + │ 1.199359 │ + │ 0.5 │ + │ 0.5 │ + │ 0.0 │ + │ 1.199359 │ + └──────────┘ """ def __init__( @@ -151,16 +173,23 @@ def __init__( return_empty: bool = False, ) -> None: - if imputation_method not in ["gaussian", "iqr", "max"]: + if not isinstance(imputation_method, str) or imputation_method not in [ + "gaussian", + "iqr", + "max", + ]: raise ValueError( - "imputation_method takes only values 'gaussian', 'iqr' or 'max'" + "imputation_method takes only values 'gaussian', 'iqr' or 'max'. " + f"Got {imputation_method} instead." ) - if tail not in ["right", "left"]: - raise ValueError("tail takes only values 'right' or 'left'") + if not isinstance(tail, str) or tail not in ["right", "left"]: + raise ValueError( + f"tail takes only values 'right' or 'left'. Got {tail} instead." + ) - if fold <= 0: - raise ValueError("fold takes only positive numbers") + if not isinstance(fold, (int, float)) or isinstance(fold, bool) or fold <= 0: + raise ValueError(f"fold takes only positive numbers. Got {fold} instead.") self.imputation_method = imputation_method self.tail = tail @@ -170,20 +199,20 @@ def __init__( _check_return_empty_is_bool(return_empty) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the values at the end of the variable distribution. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. y: pandas Series, default=None y is not needed in this imputation. You can pass None or y. """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) # find or check for numerical variables if self.variables is None: @@ -191,33 +220,33 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): else: variables_ = check_numerical_variables(X, self.variables) - # estimate imputation values - if self.imputation_method == "max": - imputer_dict_ = (X[variables_].max() * self.fold).to_dict() - - elif self.imputation_method == "gaussian": - if self.tail == "right": - imputer_dict_ = ( - X[variables_].mean() + self.fold * X[variables_].std() - ).to_dict() - elif self.tail == "left": - imputer_dict_ = ( - X[variables_].mean() - self.fold * X[variables_].std() - ).to_dict() - - elif self.imputation_method == "iqr": - IQR = X[variables_].quantile(0.75) - X[variables_].quantile(0.25) - if self.tail == "right": - imputer_dict_ = ( - X[variables_].quantile(0.75) + (IQR * self.fold) - ).to_dict() - elif self.tail == "left": - imputer_dict_ = ( - X[variables_].quantile(0.25) - (IQR * self.fold) - ).to_dict() + # Narwhals aggregation matches/beats pandas-native on pandas and is + # 3-10x faster on polars (benchmarked), so one path serves both backends. + exprs = [self._end_value_expr(v) for v in variables_] + agg = nw_X.select(*exprs) + imputer_dict_ = {k: v[0] for k, v in agg.to_dict(as_series=False).items()} self.variables_ = variables_ self.imputer_dict_ = imputer_dict_ self._get_feature_names_in(X) return self + + def _end_value_expr(self, variable: Union[str, int]) -> nw.Expr: + """Build the narwhals expression that computes the end-of-distribution + replacement value for one variable, per `imputation_method` and `tail`.""" + col = nw.col(variable) + + if self.imputation_method == "max": + return (col.max() * self.fold).alias(variable) + + if self.imputation_method == "gaussian": + if self.tail == "right": + return (col.mean() + self.fold * col.std()).alias(variable) + return (col.mean() - self.fold * col.std()).alias(variable) + + # imputation_method == "iqr" + iqr = col.quantile(0.75, "linear") - col.quantile(0.25, "linear") + if self.tail == "right": + return (col.quantile(0.75, "linear") + self.fold * iqr).alias(variable) + return (col.quantile(0.25, "linear") - self.fold * iqr).alias(variable) diff --git a/feature_engine/imputation/mean_median.py b/feature_engine/imputation/mean_median.py index 049b768e3..a809ca665 100644 --- a/feature_engine/imputation/mean_median.py +++ b/feature_engine/imputation/mean_median.py @@ -4,7 +4,8 @@ import warnings from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_variables import ( _check_variables_input_value, @@ -102,6 +103,30 @@ class MeanImputer(BaseImputer): 2 1.0 b 3 0.0 NaN 4 1.0 a + + With polars: + + >>> import polars as pl + >>> from feature_engine.imputation import MeanImputer + >>> X = pl.DataFrame(dict( + >>> x1 = [None, 1, 1, 0, None], + >>> x2 = ["a", None, "b", None, "a"], + >>> )) + >>> mmi = MeanImputer(imputation_method='median') + >>> mmi.fit(X) + >>> mmi.transform(X) + shape: (5, 2) + ┌─────┬──────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ f64 ┆ str │ + ╞═════╪══════╡ + │ 1.0 ┆ a │ + │ 1.0 ┆ null │ + │ 1.0 ┆ b │ + │ 0.0 ┆ null │ + │ 1.0 ┆ a │ + └─────┴──────┘ """ def __init__( @@ -111,8 +136,14 @@ def __init__( return_empty: bool = False, ) -> None: - if imputation_method not in ["median", "mean"]: - raise ValueError("imputation_method takes only values 'median' or 'mean'") + if not isinstance(imputation_method, str) or imputation_method not in [ + "median", + "mean", + ]: + raise ValueError( + "imputation_method takes only values 'median' or 'mean'. " + f"Got {imputation_method} instead." + ) self.imputation_method = imputation_method self.variables = _check_variables_input_value(variables) @@ -120,21 +151,22 @@ def __init__( _check_return_empty_is_bool(return_empty) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the mean or median values. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The training dataset. + X: dataframe of shape = [n_samples, n_features] + The training dataset. Can be a pandas, polars, or any other dataframe + supported by narwhals. - y: pandas series or None, default=None + y: Series or None, default=None y is not needed in this imputation. You can pass None or y. """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) # find or check for numerical variables if self.variables is None: @@ -143,11 +175,16 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): variables_ = check_numerical_variables(X, self.variables) # find imputation parameters: mean or median - if self.imputation_method == "mean": - imputer_dict_ = X[variables_].mean().to_dict() - - elif self.imputation_method == "median": - imputer_dict_ = X[variables_].median().to_dict() + if len(variables_) == 0: + imputer_dict_ = {} + else: + stats = nw_X.select( + *[ + getattr(nw.col(var), self.imputation_method)() + for var in variables_ + ] + ) + imputer_dict_ = stats.rows(named=True)[0] self.variables_ = variables_ self.imputer_dict_ = imputer_dict_ diff --git a/feature_engine/imputation/missing_indicator.py b/feature_engine/imputation/missing_indicator.py index 012cf7b23..45aa6f719 100644 --- a/feature_engine/imputation/missing_indicator.py +++ b/feature_engine/imputation/missing_indicator.py @@ -3,7 +3,10 @@ from typing import List, Optional, Union import warnings -import pandas as pd + +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_variables import ( _check_variables_input_value, @@ -105,6 +108,30 @@ class MissingIndicator(BaseImputer): 2 1.0 b 0 0 3 0.0 NaN 0 1 4 NaN a 1 0 + + With polars: + + >>> import polars as pl + >>> from feature_engine.imputation import MissingIndicator + >>> X = pl.DataFrame(dict( + ... x1 = [None, 1, 1, 0, None], + ... x2 = ["a", None, "b", None, "a"], + ... )) + >>> ami = MissingIndicator() + >>> ami.fit(X) + >>> ami.transform(X) + shape: (5, 4) + ┌──────┬──────┬───────┬───────┐ + │ x1 ┆ x2 ┆ x1_na ┆ x2_na │ + │ --- ┆ --- ┆ --- ┆ --- │ + │ i64 ┆ str ┆ i8 ┆ i8 │ + ╞══════╪══════╪═══════╪═══════╡ + │ null ┆ a ┆ 1 ┆ 0 │ + │ 1 ┆ null ┆ 0 ┆ 1 │ + │ 1 ┆ b ┆ 0 ┆ 0 │ + │ 0 ┆ null ┆ 0 ┆ 1 │ + │ null ┆ a ┆ 1 ┆ 0 │ + └──────┴──────┴───────┴───────┘ """ def __init__( @@ -115,7 +142,10 @@ def __init__( ) -> None: if not isinstance(missing_only, bool): - raise ValueError("missing_only takes values True or False") + raise ValueError( + "missing_only takes values True or False. " + f"Got {missing_only} instead." + ) self.variables = _check_variables_input_value(variables) self.missing_only = missing_only @@ -123,21 +153,21 @@ def __init__( _check_return_empty_is_bool(return_empty) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the variables for which the missing indicators will be created. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. - y: pandas Series, default=None + y: Series, default=None y is not needed in this imputation. You can pass None or y. """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) # find variables for which indicator should be added if self.variables is None: @@ -146,38 +176,64 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): variables_ = check_all_variables(X, self.variables) if self.missing_only is True: - variables_ = [var for var in variables_ if X[var].isnull().sum() > 0] + # Benchmarked: a per-column isnull().sum() loop is ~2-5x faster + # than narwhals' single null_count() call on pandas input (the + # loop calls straight into pandas' C implementation with no + # narwhals overhead), so pandas keeps its own fast path here. + if nwd.is_pandas_dataframe(X): + variables_ = [ + var for var in variables_ if X[var].isnull().sum() > 0 + ] + else: + null_counts = nw_X.select(variables_).null_count().row(0) + variables_ = [ + var + for var, count in zip(variables_, null_counts) + if count > 0 + ] self.variables_ = variables_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Add the binary missing indicators. Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The dataframe to be transformed. Returns ------- - X_new : pandas dataframe of shape = [n_samples, n_features] + X_new : dataframe of shape = [n_samples, n_features] The dataframe containing the additional binary variables. """ - X = self._transform(X) - X_indicators = ( - X[self.variables_] - .isna() - .astype("int8") - .add_suffix("_na") - ) - X = pd.concat([X, X_indicators], axis=1) + nw_X = self._transform(X) + + # Benchmarked: building a separate indicator frame and concatenating + # it (pandas-native) is ~2-5x faster than narwhals' with_columns + # equivalent on pandas input, so pandas keeps its own fast path here. + if nwd.is_pandas_dataframe(X): + pd = nw.from_native(X, eager_only=True).__native_namespace__() + X_indicators = ( + X[self.variables_] + .isna() + .astype("int8") + .add_suffix("_na") + ) + X = pd.concat([X, X_indicators], axis=1) + else: + nw_X = nw_X.with_columns( + nw.col(var).is_null().cast(nw.Int8).alias(f"{var}_na") + for var in self.variables_ + ) + X = nw_X.to_native() return X diff --git a/feature_engine/imputation/random_sample.py b/feature_engine/imputation/random_sample.py index bc11e0dac..bc9804217 100644 --- a/feature_engine/imputation/random_sample.py +++ b/feature_engine/imputation/random_sample.py @@ -1,10 +1,12 @@ # Authors: Soledad Galli # License: BSD 3 clause +import hashlib from typing import List, Optional, Union +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_variables import ( _check_variables_input_value, @@ -31,20 +33,27 @@ from feature_engine.variable_handling import check_all_variables, find_all_variables -# for RandomSampleImputer -def _define_seed( - X: pd.DataFrame, - index: int, - seed_variables: Union[str, int, List[Union[str, int]]], - how: str = "add", -) -> int: - # determine seed by adding or multiplying the value of 1 or - # more variables - if how == "add": - internal_seed = int(np.round(X.loc[index, seed_variables].sum(), 0)) - elif how == "multiply": - internal_seed = int(np.round(X.loc[index, seed_variables].product(), 0)) - return internal_seed +def _hash_seeds(values) -> np.ndarray: + """Return one seed per row, in [0, 2**32), derived from the row's values. + + Rows with the same values get the same seed, regardless of their position. + Values are compared as floats (25 and 25.0 are equal) and missing values + count as 0. hashlib, unlike hash(), gives the same seed in every session. + """ + values = np.asarray(values, dtype="float64") + values = values.reshape(len(values), -1) + # + 0.0 turns -0.0 into 0.0, so both give the same bytes + values = np.where(np.isnan(values), 0.0, values) + 0.0 + values = np.ascontiguousarray(values, dtype=">> x1 = [np.nan,1,1,0,np.nan], >>> x2 = ["a", np.nan, "b", np.nan, "a"], >>> )) - >>> rsi = RandomSampleImputer() + >>> rsi = RandomSampleImputer(random_state=42) >>> rsi.fit(X) >>> rsi.transform(X) x1 x2 - 0 1.0 a - 1 1.0 b + 0 0.0 a + 1 1.0 a 2 1.0 b 3 0.0 a 4 1.0 a + + With polars: + + >>> import polars as pl + >>> X = pl.DataFrame(dict( + ... x1 = [None, 1, 1, 0, None], + ... x2 = ["a", None, "b", None, "a"], + ... )) + >>> rsi = RandomSampleImputer(random_state=42) + >>> rsi.fit(X) + >>> rsi.transform(X) + shape: (5, 2) + ┌─────┬─────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ str │ + ╞═════╪═════╡ + │ 0 ┆ a │ + │ 1 ┆ a │ + │ 1 ┆ b │ + │ 0 ┆ a │ + │ 1 ┆ a │ + └─────┴─────┘ """ def __init__( @@ -147,25 +176,26 @@ def __init__( return_empty: bool = False, random_state: Union[None, int, str, List[Union[str, int]]] = None, seed: str = "general", - seeding_method: str = "add", ) -> None: - if seed not in ["general", "observation"]: - raise ValueError("seed takes only values 'general' or 'observation'") - - if seeding_method not in ["add", "multiply"]: - raise ValueError("seeding_method takes only values 'add' or 'multiply'") + if not isinstance(seed, str) or seed not in ["general", "observation"]: + raise ValueError( + "seed takes only values 'general' or 'observation'. " + f"Got {seed} instead." + ) if seed == "general" and random_state: if not isinstance(random_state, int): raise ValueError( - "if seed == 'general' then random_state must take an integer" + "if seed == 'general' then random_state must take an integer. " + f"Got {random_state} instead." ) if seed == "observation" and not random_state: raise ValueError( "if seed == 'observation' the random state must take the name of one " - "or more variables which will be used to seed the imputer" + "or more variables which will be used to seed the imputer. " + f"Got {random_state} instead." ) self.variables = _check_variables_input_value(variables) @@ -175,9 +205,8 @@ def __init__( self.random_state = random_state self.seed = seed - self.seeding_method = seeding_method - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Makes a copy of the train set. Only stores a copy of the variables to impute. This copy is then used to randomly extract the values to fill the missing data @@ -186,15 +215,16 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The training dataset. + X: dataframe of shape = [n_samples, n_features] + The training dataset. Can be a pandas, polars, or any other dataframe + supported by narwhals. y: None y is not needed in this imputation. You can pass None or y. """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) # find variables to impute if self.variables is None: @@ -203,7 +233,10 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): variables_ = check_all_variables(X, self.variables) # take a copy of the selected variables - X_ = X[variables_].copy() + if nwd.is_pandas_dataframe(X): + X_ = X[variables_].copy() + else: + X_ = nw_X.select(variables_) # check the variables assigned to the random state if self.seed == "observation": @@ -215,7 +248,7 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): ): raise ValueError( "There are variables assigned as random state which are not part " - "of the training dataframe." + f"of the training dataframe. Got {self.random_state} instead." ) self.random_state = random_state @@ -225,23 +258,37 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Replace missing data with random values taken from the train set. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The dataframe to be transformed. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The dataframe without missing values in the transformed variables. """ - X = self._transform(X) + nw_X = self._transform(X) + + if nwd.is_pandas_dataframe(X): + X = self._transform_pandas(X) + else: + X = self._transform_narwhals(nw_X) + + return X + + def _transform_pandas(self, X): + # copy first: the .loc assignments below fill NaNs in place, and + # BaseImputer._transform no longer returns a copy (#1002), so without + # this the caller's dataframe (and self.X_ when it is the same object) + # would be mutated. + X = X.copy() # random sampling with a general seed if self.seed == "general": @@ -265,27 +312,64 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: # random sampling observation per observation elif self.seed == "observation" and self.random_state: + # seeds come from the values before any variable is imputed; rows are + # addressed by position, so duplicated index labels don't matter + seeds = _hash_seeds(X[self.random_state].to_numpy()) for feature in self.variables_: - if X[feature].isnull().sum() > 0: + is_null = X[feature].isnull().to_numpy() + if is_null.any(): + pool = self.X_[feature].dropna() + positions = np.flatnonzero(is_null) + random_values = [ + pool.sample( + 1, replace=True, random_state=int(seeds[pos]) + ).iloc[0] + for pos in positions + ] + X.iloc[positions, X.columns.get_loc(feature)] = random_values + return X - # loop over each observation with missing data - for i in X[X[feature].isnull()].index: - # find the seed using additional variables - internal_seed = _define_seed( - X, i, self.random_state, how=self.seeding_method - ) + def _transform_narwhals(self, X): - # extract 1 value at random - random_sample = ( - self.X_[feature] - .dropna() - .sample(1, replace=True, random_state=internal_seed) + if self.seed == "general": + for feature in self.variables_: + col = X[feature] + null_mask = col.is_null() + n_samples = int(null_mask.sum()) + if n_samples > 0: + positions = null_mask.arg_true() + random_sample = ( + self.X_[feature] + .drop_nulls() + .sample( + n_samples, with_replacement=True, seed=self.random_state ) - random_sample = random_sample.values[0] + ) + # reassign X so each variable's imputation is carried over + # to the next iteration + X = X.with_columns(col.scatter(positions, random_sample)) - # replace the missing data point - X.loc[i, feature] = random_sample - return X + elif self.seed == "observation" and self.random_state: + # seeds come from the values before any variable is imputed + internal_seeds = _hash_seeds(X.select(self.random_state).to_numpy()) + + for feature in self.variables_: + col = X[feature] + null_mask = col.is_null() + if int(null_mask.sum()) > 0: + positions = null_mask.arg_true().to_list() + pool = self.X_[feature].drop_nulls() + random_values = [ + pool.sample( + 1, with_replacement=True, seed=int(internal_seeds[pos]) + ).item() + for pos in positions + ] + # reassign X so each variable's imputation is carried over + # to the next iteration + X = X.with_columns(col.scatter(positions, random_values)) + + return X.to_native() def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/outliers/artbitrary.py b/feature_engine/outliers/artbitrary.py index 6088520da..98e983c8a 100644 --- a/feature_engine/outliers/artbitrary.py +++ b/feature_engine/outliers/artbitrary.py @@ -4,7 +4,7 @@ from typing import Optional -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_input_dictionary import ( _check_numerical_dict, @@ -119,80 +119,78 @@ def __init__( missing_values: str = "raise", ) -> None: - if not max_capping_dict and not min_capping_dict: + _check_numerical_dict(max_capping_dict) + _check_numerical_dict(min_capping_dict) + + if (max_capping_dict is None or len(max_capping_dict) == 0) and ( + min_capping_dict is None or len(min_capping_dict) == 0 + ): raise ValueError( "Please provide at least 1 dictionary with the capping values." ) - if missing_values not in ["raise", "ignore"]: - raise ValueError("missing_values takes only values 'raise' or 'ignore'") - - _check_numerical_dict(max_capping_dict) - _check_numerical_dict(min_capping_dict) + if not isinstance(missing_values, str) or missing_values not in [ + "raise", + "ignore", + ]: + raise ValueError( + "missing_values must be 'raise' or 'ignore'. " + f"Got {missing_values} instead." + ) self.max_capping_dict = max_capping_dict self.min_capping_dict = min_capping_dict self.missing_values = missing_values - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn any parameter. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. - y: pandas Series, default=None + y: Series, default=None y is not needed in this transformer. You can pass y or None. """ - X = check_X(X) - - # find variables to be capped - if self.min_capping_dict is None and self.max_capping_dict: - self.variables_ = [x for x in self.max_capping_dict.keys()] - elif self.max_capping_dict is None and self.min_capping_dict: - self.variables_ = [x for x in self.min_capping_dict.keys()] - elif self.min_capping_dict and self.max_capping_dict: - tmp = self.min_capping_dict.copy() - tmp.update(self.max_capping_dict) - self.variables_ = [x for x in tmp.keys()] - - if self.missing_values == "raise": - # check if dataset contains na - _check_contains_na(X, self.variables_) - _check_contains_inf(X, self.variables_) - - # find or check for numerical variables - self.variables_ = check_numerical_variables(X, self.variables_) + nw_X = check_X(X) - if self.max_capping_dict is not None: - self.right_tail_caps_ = self.max_capping_dict - else: + if self.max_capping_dict is None: self.right_tail_caps_ = {} - - if self.min_capping_dict is not None: - self.left_tail_caps_ = self.min_capping_dict else: + self.right_tail_caps_ = self.max_capping_dict + + if self.min_capping_dict is None: self.left_tail_caps_ = {} + else: + self.left_tail_caps_ = self.min_capping_dict + + variables = list({**self.left_tail_caps_, **self.right_tail_caps_}) + + if self.missing_values == "raise": + _check_contains_na(X, variables) + _check_contains_inf(X, variables) + + self.variables_ = check_numerical_variables(X, variables) - self.feature_names_in_ = X.columns.to_list() - self.n_features_in_ = X.shape[1] + self.feature_names_in_ = nw_X.columns + self.n_features_in_ = nw_X.shape[1] return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Cap the variable values. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The dataframe with the capped variables. """ return super()._transform(X) diff --git a/feature_engine/outliers/base_outlier.py b/feature_engine/outliers/base_outlier.py index 2f914df86..5daa4efff 100644 --- a/feature_engine/outliers/base_outlier.py +++ b/feature_engine/outliers/base_outlier.py @@ -1,6 +1,9 @@ from typing import List, Literal, Optional, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -27,34 +30,35 @@ class BaseOutlier(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): """shared set-up checks and methods across outlier transformers""" - def _check_transform_input_and_state(self, X: pd.DataFrame) -> pd.DataFrame: + def _check_transform_input_and_state(self, X: IntoDataFrame) -> IntoDataFrame: """Checks that the input is a dataframe and of the same size as the one used in the fit method. Checks absence of NA. Parameters ---------- - X: pandas DataFrame + X: dataframe Raises ------ TypeError - If the input is not a pandas DataFrame + If the input is not a recognised dataframe ValueError If the dataframe is not of same size as that used in fit() Returns ------- - X: pandas DataFrame - The same dataframe entered by the user. + nw_X: narwhals dataframe + The narwhalified version of the dataframe entered by the user, with + the variables in the same order as in the train set. """ # check if class was fitted check_is_fitted(self) # check that input is a dataframe - X = check_X(X) + nw_X = check_X(X) # Check that the dataframe contains the same number of columns - # than the dataframe used to fit the imputer. + # than the dataframe used to fit the transformer. _check_X_matches_training_df(X, self.n_features_in_) if self.missing_values == "raise": @@ -62,37 +66,76 @@ def _check_transform_input_and_state(self, X: pd.DataFrame) -> pd.DataFrame: _check_contains_na(X, self.variables_) _check_contains_inf(X, self.variables_) - # reorder to match training set - X = X[self.feature_names_in_] + # reorder to match training set. pandas selects by label, which also + # supports integer column names. + if nwd.is_pandas_dataframe(X): + return nw.from_native(X[self.feature_names_in_], eager_only=True) + return nw_X.select(nw.col(*self.feature_names_in_)) - return X - - def _transform(self, X: pd.DataFrame) -> pd.DataFrame: + def _transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Cap the variable values. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The dataframe with the capped variables. """ # check if class was fitted - X = self._check_transform_input_and_state(X) - - # replace outliers - for feature in self.right_tail_caps_.keys(): - X[feature] = X[feature].clip(upper=self.right_tail_caps_[feature]) - - for feature in self.left_tail_caps_.keys(): - X[feature] = X[feature].clip(lower=self.left_tail_caps_[feature]) - - return X + nw_X = self._check_transform_input_and_state(X) + + # infinite limits don't cap, and clipping to them turns integers into floats + right = {v: c for v, c in self.right_tail_caps_.items() if np.isfinite(c)} + left = {v: c for v, c in self.left_tail_caps_.items() if np.isfinite(c)} + + both = [var for var in self.variables_ if var in right and var in left] + right_only = [ + var for var in self.variables_ if var in right and var not in left + ] + left_only = [var for var in self.variables_ if var in left and var not in right] + + # Grouping columns by which bound(s) apply turns the per-column .clip() + # loop into up to 3 vectorized numpy calls (benchmarked 2-6x faster than + # pandas-native at 10k-100k rows). Using np.clip/minimum/maximum only with + # the bounds that actually apply (never an inf sentinel for a missing + # side) keeps int-dtype columns int, matching pandas .clip() exactly. + new_series = [] + if len(both) > 0: + values = nw_X.select(nw.col(*both)).to_numpy() + lower = np.array([left[var] for var in both]) + upper = np.array([right[var] for var in both]) + clipped = np.clip(values, lower, upper) + new_series += [ + nw.new_series(var, clipped[:, i], backend=nw_X.implementation) + for i, var in enumerate(both) + ] + if len(right_only) > 0: + values = nw_X.select(nw.col(*right_only)).to_numpy() + upper = np.array([right[var] for var in right_only]) + clipped = np.minimum(values, upper) + new_series += [ + nw.new_series(var, clipped[:, i], backend=nw_X.implementation) + for i, var in enumerate(right_only) + ] + if len(left_only) > 0: + values = nw_X.select(nw.col(*left_only)).to_numpy() + lower = np.array([left[var] for var in left_only]) + clipped = np.maximum(values, lower) + new_series += [ + nw.new_series(var, clipped[:, i], backend=nw_X.implementation) + for i, var in enumerate(left_only) + ] + + if len(new_series) > 0: + nw_X = nw_X.with_columns(*new_series) + + return nw_X.to_native() def _more_tags(self): tags_dict = _return_tags() @@ -205,21 +248,21 @@ def __init__( self.return_empty = return_empty self.missing_values = missing_values - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the values that should be used to replace outliers. Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training input samples. - y : pandas Series, default=None + y : Series, default=None y is not needed in this transformer. You can pass y or None. """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) # find or check for numerical variables if self.variables is None: @@ -242,49 +285,75 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): else: self.fold_ = self.fold + values = nw_X.select(nw.col(*self.variables_)).to_numpy() + + # nan-aware reductions: with missing_values="ignore", values may contain + # NaN, and pandas' mean/std/quantile/median skip NaN by default. if self.capping_method == "gaussian": - bias = X[self.variables_].mean() - scale = X[self.variables_].std(ddof=0) + bias = np.nanmean(values, axis=0) + scale = np.nanstd(values, axis=0, ddof=0) elif self.capping_method == "iqr": - bias = X[self.variables_].quantile((0.75, 0.25)) - scale = bias.loc[0.75] - bias.loc[0.25] + q75 = np.nanquantile(values, 0.75, axis=0) + q25 = np.nanquantile(values, 0.25, axis=0) + scale = q75 - q25 elif self.capping_method == "quantiles": - bias = X[self.variables_].quantile((1 - self.fold_, self.fold_)) - scale = bias.loc[1 - self.fold_] - bias.loc[self.fold_] + q_hi = np.nanquantile(values, 1 - self.fold_, axis=0) + q_lo = np.nanquantile(values, self.fold_, axis=0) + scale = q_hi - q_lo elif self.capping_method == "mad": - bias = X[self.variables_].median() + bias = np.nanmedian(values, axis=0) # scaling factor for normal distribution - scale = (X[self.variables_] - bias).abs().median() / 0.67449 - if (scale == 0).any(): - raise ValueError( - f"Input columns {scale[scale == 0].index.tolist()!r}" - f" have low variation for method {self.capping_method!r}." - f" Try other capping methods or drop these columns." - ) + scale = np.nanmedian(np.abs(values - bias), axis=0) / 0.67449 # estimate the end values if self.tail in ("right", "both"): if self.capping_method in ("gaussian", "mad"): - self.right_tail_caps_ = (bias + self.fold_ * scale).to_dict() + self.right_tail_caps_ = { + var: float(b + self.fold_ * s) + for var, b, s in zip(self.variables_, bias, scale) + } elif self.capping_method == "iqr": - self.right_tail_caps_ = (bias.loc[0.75] + self.fold_ * scale).to_dict() + self.right_tail_caps_ = { + var: float(q + self.fold_ * s) + for var, q, s in zip(self.variables_, q75, scale) + } elif self.capping_method == "quantiles": - self.right_tail_caps_ = bias.loc[1 - self.fold_].to_dict() + self.right_tail_caps_ = { + var: float(q) for var, q in zip(self.variables_, q_hi) + } if self.tail in ("left", "both"): if self.capping_method in ("gaussian", "mad"): - self.left_tail_caps_ = (bias - self.fold_ * scale).to_dict() + self.left_tail_caps_ = { + var: float(b - self.fold_ * s) + for var, b, s in zip(self.variables_, bias, scale) + } elif self.capping_method == "iqr": - self.left_tail_caps_ = (bias.loc[0.25] - self.fold_ * scale).to_dict() + self.left_tail_caps_ = { + var: float(q - self.fold_ * s) + for var, q, s in zip(self.variables_, q25, scale) + } elif self.capping_method == "quantiles": - self.left_tail_caps_ = bias.loc[self.fold_].to_dict() - - self.feature_names_in_ = X.columns.to_list() - self.n_features_in_ = X.shape[1] + self.left_tail_caps_ = { + var: float(q) for var, q in zip(self.variables_, q_lo) + } + + # variables without variation have no outliers, so they get infinite limits + for var, s in zip(self.variables_, scale): + if s == 0: + if var in self.right_tail_caps_: + self.right_tail_caps_[var] = float("inf") + if var in self.left_tail_caps_: + self.left_tail_caps_[var] = float("-inf") + + # list() normalises both a narwhals `.columns` (already a list) and a + # pandas Index to a plain list. + self.feature_names_in_ = list(nw_X.columns) + self.n_features_in_ = nw_X.shape[1] return self diff --git a/feature_engine/outliers/trimmer.py b/feature_engine/outliers/trimmer.py index 41cb48145..e86281be6 100644 --- a/feature_engine/outliers/trimmer.py +++ b/feature_engine/outliers/trimmer.py @@ -1,9 +1,10 @@ # Authors: Soledad Galli # License: BSD 3 clause -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries -from feature_engine._base_transformers.mixins import TransformXyMixin from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, _left_tail_caps_docstring, @@ -23,6 +24,7 @@ ) from feature_engine._docstrings.methods import _fit_transform_docstring from feature_engine._docstrings.substitute import Substitution +from feature_engine.dataframe_checks import check_X_y from feature_engine.outliers.base_outlier import WinsorizerBase @@ -41,7 +43,7 @@ n_features_in_=_n_features_in_docstring, fit_transform=_fit_transform_docstring, ) -class OutlierTrimmer(WinsorizerBase, TransformXyMixin): +class OutlierTrimmer(WinsorizerBase): """The OutlierTrimmer() removes observations with outliers from the dataset. The OutlierTrimmer() first calculates the maximum and/or minimum values @@ -174,29 +176,63 @@ class OutlierTrimmer(WinsorizerBase, TransformXyMixin): 9 0.54256 """ - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Remove observations with outliers from the dataframe. Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The dataframe without outlier observations. """ + nw_X = self._check_transform_input_and_state(X) + return self._remove_outliers(nw_X).to_native() - X = self._check_transform_input_and_state(X) + def transform_x_y(self, X: IntoDataFrame, y: IntoSeries): + """ + Remove observations with outliers from the dataframe and the target. + + Parameters + ---------- + X: dataframe of shape = [n_samples, n_features] + The dataframe to transform. + + y: Series or Dataframe of length = n_samples + The target variable to transform. Can be multi-output. + + Returns + ------- + X_new: dataframe + The dataframe without outlier observations. It may contain less rows + than the original dataset. + + y_new: Series or DataFrame + The target variable, with as many rows as those left in X_new. + """ + _, y = check_X_y(X, y) + + row_index = "__row_index__" + nw_X = self._check_transform_input_and_state(X).with_row_index(row_index) + nw_X = self._remove_outliers(nw_X) + rows = nw_X.get_column(row_index).to_list() + + if nwd.is_into_series(y): + y = nw.from_native(y, series_only=True)[rows].to_native() + else: + y = nw.from_native(y, eager_only=True)[rows].to_native() + + return nw_X.drop(row_index).to_native(), y - for feature in self.right_tail_caps_.keys(): - inliers = X[feature].le(self.right_tail_caps_[feature]) - X = X.loc[inliers] + def _remove_outliers(self, nw_X: nw.DataFrame) -> nw.DataFrame: + conditions = [nw.col(f) <= c for f, c in self.right_tail_caps_.items()] + conditions += [nw.col(f) >= c for f, c in self.left_tail_caps_.items()] - for feature in self.left_tail_caps_.keys(): - inliers = X[feature].ge(self.left_tail_caps_[feature]) - X = X.loc[inliers] + if len(conditions) > 0: + nw_X = nw_X.filter(nw.all_horizontal(*conditions, ignore_nulls=False)) - return X + return nw_X diff --git a/feature_engine/outliers/winsorizer.py b/feature_engine/outliers/winsorizer.py index ad2320ab9..63454f404 100644 --- a/feature_engine/outliers/winsorizer.py +++ b/feature_engine/outliers/winsorizer.py @@ -4,8 +4,9 @@ import warnings from typing import List, Literal, Union -import numpy as np -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, @@ -26,7 +27,6 @@ ) from feature_engine._docstrings.methods import _fit_transform_docstring from feature_engine._docstrings.substitute import Substitution -from feature_engine.dataframe_checks import check_X from feature_engine.outliers.base_outlier import WinsorizerBase @@ -145,25 +145,33 @@ class Winsoriser(WinsorizerBase): 8 -0.469474 9 0.542560 + With polars: + >>> import numpy as np - >>> import pandas as pd + >>> import polars as pl >>> from feature_engine.outliers import Winsoriser >>> np.random.seed(42) - >>> X = pd.DataFrame(dict(x = np.random.normal(size = 10))) + >>> X = pl.DataFrame(dict(x = np.random.normal(size = 10))) >>> wz = Winsoriser(capping_method='mad', tail='both', fold=3) >>> wz.fit(X) >>> wz.transform(X) - x - 0 0.496714 - 1 -0.138264 - 2 0.647689 - 3 1.523030 - 4 -0.234153 - 5 -0.234137 - 6 1.579213 - 7 0.767435 - 8 -0.469474 - 9 0.542560 + shape: (10, 1) + ┌───────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞═══════════╡ + │ 0.496714 │ + │ -0.138264 │ + │ 0.647689 │ + │ 1.52303 │ + │ -0.234153 │ + │ -0.234137 │ + │ 1.579213 │ + │ 0.767435 │ + │ -0.469474 │ + │ 0.54256 │ + └───────────┘ """ def __init__( @@ -178,7 +186,7 @@ def __init__( ) -> None: if not isinstance(add_indicators, bool): raise ValueError( - "add_indicators takes only booleans True and False" + "add_indicators takes only booleans True and False. " f"Got {add_indicators} instead." ) super().__init__( @@ -186,53 +194,74 @@ def __init__( ) self.add_indicators = add_indicators - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Cap the variable values. Optionally, add outlier indicators. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features + n_ind] + X_new: dataframe of shape = [n_samples, n_features + n_ind] The dataframe with the capped variables and indicators. The number of output variables depends on the values for 'tail' and 'add_indicators': if passing 'add_indicators=False', will be equal to 'n_features', otherwise, will have an additional indicator column per processed feature for each tail. """ - if not self.add_indicators: - X_out = super()._transform(X) + X_out = super()._transform(X) - else: - X_orig = check_X(X) - X_out = super()._transform(X_orig) - X_orig = X_orig[self.variables_] - X_out_filtered = X_out[self.variables_] - - if self.tail in ["left", "both"]: - X_left = X_out_filtered > X_orig - X_left.columns = [str(cl) + "_left" for cl in self.variables_] - if self.tail in ["right", "both"]: - X_right = X_out_filtered < X_orig - X_right.columns = [str(cl) + "_right" for cl in self.variables_] - if self.tail == "left": - X_out = pd.concat([X_out, X_left.astype(np.float64)], axis=1) - elif self.tail == "right": - X_out = pd.concat([X_out, X_right.astype(np.float64)], axis=1) - else: - X_both = pd.concat([X_left, X_right], axis=1).astype(np.float64) - X_both = X_both[ - [ - cl1 - for cl2 in zip(X_left.columns.values, X_right.columns.values) - for cl1 in cl2 + if self.add_indicators is True: + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X_out) is True: + pd = nw.from_native(X_out, eager_only=True).__native_namespace__() + X_orig_filtered = X[self.variables_] + X_out_filtered = X_out[self.variables_] + + if self.tail in ["left", "both"]: + X_left = X_out_filtered > X_orig_filtered + X_left.columns = [str(cl) + "_left" for cl in self.variables_] + if self.tail in ["right", "both"]: + X_right = X_out_filtered < X_orig_filtered + X_right.columns = [str(cl) + "_right" for cl in self.variables_] + if self.tail == "left": + X_out = pd.concat([X_out, X_left.astype("float64")], axis=1) + elif self.tail == "right": + X_out = pd.concat([X_out, X_right.astype("float64")], axis=1) + else: + X_both = pd.concat([X_left, X_right], axis=1).astype("float64") + X_both = X_both[ + [ + cl1 + for cl2 in zip( + X_left.columns.values, X_right.columns.values + ) + for cl1 in cl2 + ] ] - ] - X_out = pd.concat([X_out, X_both], axis=1) + X_out = pd.concat([X_out, X_both], axis=1) + else: + nw_orig = nw.from_native(X, eager_only=True) + nw_out = nw.from_native(X_out, eager_only=True) + + new_cols = [] + for var in self.variables_: + if self.tail in ["left", "both"]: + new_cols.append( + (nw_out[var] > nw_orig[var]) + .cast(nw.Float64) + .alias(f"{var}_left") + ) + if self.tail in ["right", "both"]: + new_cols.append( + (nw_out[var] < nw_orig[var]) + .cast(nw.Float64) + .alias(f"{var}_right") + ) + X_out = nw_out.with_columns(*new_cols).to_native() return X_out diff --git a/feature_engine/preprocessing/match_categories.py b/feature_engine/preprocessing/match_categories.py index e0e863c1f..981f1cc0b 100644 --- a/feature_engine/preprocessing/match_categories.py +++ b/feature_engine/preprocessing/match_categories.py @@ -1,7 +1,9 @@ import warnings from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.mixins import GetFeatureNamesOutMixin from feature_engine._check_init_parameters.check_init_input_params import ( @@ -19,7 +21,7 @@ ) from feature_engine._docstrings.init_parameters.encoders import _ignore_format_docstring from feature_engine._docstrings.substitute import Substitution -from feature_engine.dataframe_checks import _check_optional_contains_na, check_X +from feature_engine.dataframe_checks import check_X from feature_engine.encoding.base_encoder import ( CategoricalInitMixinNA, CategoricalMethodsMixin, @@ -40,7 +42,8 @@ class MatchCategories( ): """ MatchCategories() ensures that categorical variables are encoded as pandas - `'categorical'` dtype, instead of generic python `'object'` or other dtypes. + `'categorical'` dtype, or polars `'Enum'` dtype, instead of generic python + `'object'`, string or other dtypes. Under the hood, `'categorical'` dtype is a representation that maps each category to an integer, thus providing a more memory-efficient object @@ -51,7 +54,11 @@ class MatchCategories( category, and can thus be used to ensure that the correct encoding gets applied when passing categorical data to modelling packages that support this dtype, or to prevent unseen categories from reaching a further transformer - or estimator in a pipeline, for example. + or estimator in a pipeline, for example. Categories not seen during fit become + missing values. + + The polars `'Enum'` dtype only takes strings, so with polars, numerical + variables cast with `ignore_format=True` become strings. More details in the :ref:`User Guide `. @@ -68,7 +75,8 @@ class MatchCategories( Attributes ---------- category_dict_: - Dictionary with the category encodings assigned to each variable. + Dictionary with the categories learned for each variable. With polars, the + categories are stored as lists of strings. {variables_} @@ -94,7 +102,7 @@ class MatchCategories( Set the parameters of this estimator. transform: - Enforce the type of categorical variables as dtype `categorical`. + Cast the categorical variables to a categorical dtype. Examples -------- @@ -131,84 +139,103 @@ def __init__( super().__init__(variables, missing_values, ignore_format) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ - Learn the encodings or levels to use for representing categorical variables. + Learn the categories of each categorical variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. - y: pandas Series, default = None - y is not needed in this encoder. You can pass y or None. + y: Series, default = None + y is not needed in this transformer. You can pass y or None. """ - X = check_X(X) + nw_X = check_X(X) variables_ = self._check_or_select_variables(X) - - if self.missing_values == "raise": - _check_optional_contains_na(X, variables_) - - self.category_dict_ = dict() - for var in variables_: - self.category_dict_[var] = pd.Categorical(X[var]).categories + self._check_na(X, variables_) + + if nwd.is_pandas_dataframe(X) is True: + # pandas is faster than narwhals. + self.category_dict_ = { + var: X[var].astype("category").cat.categories for var in variables_ + } + else: + self.category_dict_ = { + var: self._find_categories(nw_X.get_column(var)) for var in variables_ + } self.variables_ = variables_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ - Encode categorical variables as pandas categorical dtype. + Cast the categorical variables to a categorical dtype with the categories + learned during fit. Categories not seen during fit become missing values. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. - The dataset to encode. + X: dataframe of shape = [n_samples, n_features]. + The dataset to transform. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features]. - The dataframe with the variables encoded as pandas categorical dtype. + X_new: dataframe of shape = [n_samples, n_features]. + The dataframe with the variables cast to pandas `category` or polars + `Enum` dtype. """ - X = self._check_transform_input_and_state(X) - - if self.missing_values == "raise": - _check_optional_contains_na(X, self.variables_) - - for feature, levels in self.category_dict_.items(): - X[feature] = pd.Categorical( - X[feature].where(X[feature].isin(levels)), - categories=levels + nw_X = self._check_transform_input_and_state(X) + self._check_na(X, self.variables_) + + if nwd.is_pandas_dataframe(X) is True: + # pandas is faster than narwhals. + X = X.copy() + categorical = nw.get_native_namespace(nw_X).Categorical + for feature, levels in self.category_dict_.items(): + # get_indexer returns -1 for unseen categories, which from_codes + # turns into NaN. + X[feature] = categorical.from_codes( + levels.get_indexer(X[feature]), categories=levels + ) + nw_X = nw.from_native(X, eager_only=True) + else: + nw_X = nw_X.with_columns( + *[ + nw.when(nw.col(feature).cast(nw.String).is_in(levels)) + .then(nw.col(feature).cast(nw.String)) + .cast(nw.Enum(levels)) + for feature, levels in self.category_dict_.items() + ] ) - self._check_nas_in_result(X) - return X + self._check_nas_in_result(nw_X) + return nw_X.to_native() - def _check_nas_in_result(self, X: pd.DataFrame): - # check if NaN values were introduced by the encoding - if X[self.category_dict_.keys()].isnull().sum().sum() > 0: + def _check_nas_in_result(self, nw_X: nw.DataFrame): + nan_columns = [ + str(feature) + for feature in self.category_dict_ + if nw_X.get_column(feature).null_count() > 0 + ] - # obtain the name(s) of the columns that have null values - nan_columns = ( - X[self.category_dict_.keys()] - .columns[X[self.category_dict_.keys()].isnull().any()] - .tolist() + if len(nan_columns) > 0: + msg = ( + "During the encoding, NaN values were introduced in the feature(s) " + f"{', '.join(nan_columns)}." ) - - if len(nan_columns) > 1: - nan_columns_str = ", ".join(nan_columns) - else: - nan_columns_str = nan_columns[0] - if self.missing_values == "ignore": - warnings.warn( - "During the encoding, NaN values were introduced in the feature(s) " - f"{nan_columns_str}." - ) + warnings.warn(msg) elif self.missing_values == "raise": - raise ValueError( - "During the encoding, NaN values were introduced in the feature(s) " - f"{nan_columns_str}." - ) + raise ValueError(msg) + + def _find_categories(self, series: nw.Series) -> List[str]: + if series.dtype == nw.Enum: + return list(series.dtype.categories) + if series.dtype.is_float() is True: + # NaN is a missing value in pandas, but a regular value in polars. + series = series.filter(~series.is_nan()) + # polars Enum only takes strings, so categories are sorted in the original + # dtype, to keep numbers in numeric order, and then cast to string. + return series.drop_nulls().unique().sort().cast(nw.String).to_list() diff --git a/feature_engine/preprocessing/match_columns.py b/feature_engine/preprocessing/match_columns.py index 4a1598d49..8b92e0923 100644 --- a/feature_engine/preprocessing/match_columns.py +++ b/feature_engine/preprocessing/match_columns.py @@ -1,11 +1,16 @@ -from typing import Dict, List, Union +from typing import Dict, List, Optional, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted from feature_engine._base_transformers.mixins import GetFeatureNamesOutMixin +from feature_engine._check_init_parameters.check_init_input_params import ( + _check_param_missing_values, +) from feature_engine.dataframe_checks import _check_contains_na, check_X from feature_engine.tags import _return_tags @@ -47,11 +52,10 @@ class MatchVariables(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): df_transformed - Name City Age Marks - 0 tom np.nan 20 0.9 - 1 sam np.nan 22 0.7 - 2 nick np.nan 23 0.6 - + Name City Age Marks + 0 tom NaN 20 0.9 + 1 sam NaN 22 0.7 + 2 nick NaN 23 0.6 The order of the variables in the transformed dataset is also adjusted to match that observed in the train set. @@ -62,6 +66,7 @@ class MatchVariables(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): ---------- fill_value: integer, float or string. Default=np.nan The values for the variables that will be added to the transformed dataset. + With polars dataframes, np.nan adds the variables as nulls. missing_values: string, default='raise' Indicates if missing values should be ignored or raised. If 'raise' the @@ -85,10 +90,6 @@ class MatchVariables(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): n_features_in_: The number of features in the train set used in fit. - dtype_dict_: - If `match_dtypes` is set to `True`, then this attribute will exist, and it will - contain a dictionary of variables and their corresponding dtypes. - Methods ------- fit: @@ -116,12 +117,12 @@ class MatchVariables(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): >>> from feature_engine.preprocessing import MatchVariables >>> X_train = pd.DataFrame(dict(x1 = ["a","b","c"], x2 = [4,5,6])) >>> X_test = pd.DataFrame(dict(x1 = ["c","b","a","d"], - >>> x2 = [5,6,4,7], - >>> x3 = [1,1,1,1])) + ... x2 = [5,6,4,7], + ... x3 = [1,1,1,1])) >>> mv = MatchVariables(missing_values="ignore") >>> mv.fit(X_train) >>> mv.transform(X_train) - x1 x2 + x1 x2 0 a 4 1 b 5 2 c 6 @@ -136,7 +137,7 @@ class MatchVariables(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): >>> import pandas as pd >>> from feature_engine.preprocessing import MatchVariables >>> X_train = pd.DataFrame(dict(x1 = ["a","b","c"], - >>> x2 = [4,5,6], x3 = [1,1,1])) + ... x2 = [4,5,6], x3 = [1,1,1])) >>> X_test = pd.DataFrame(dict(x1 = ["c","b","a","d"], x2 = [5,6,4,7])) >>> mv = MatchVariables(missing_values="ignore") >>> mv.fit(X_train) @@ -152,6 +153,32 @@ class MatchVariables(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): 1 b 6 NaN 2 a 4 NaN 3 d 7 NaN + + With polars: + + >>> import polars as pl + >>> from feature_engine.preprocessing import MatchVariables + >>> X_train = pl.DataFrame(dict(x1 = ["a","b","c"], + ... x2 = [4,5,6], x3 = [1,1,1])) + >>> X_test = pl.DataFrame(dict(x2 = [5,6,4,7], + ... x1 = ["c","b","a","d"], + ... x4 = [0,0,0,0])) + >>> mv = MatchVariables(missing_values="ignore") + >>> mv.fit(X_train) + >>> mv.transform(X_test) + The following variables are added to the DataFrame: ['x3'] + The following variables are dropped from the DataFrame: ['x4'] + shape: (4, 3) + ┌─────┬─────┬──────┐ + │ x1 ┆ x2 ┆ x3 │ + │ --- ┆ --- ┆ --- │ + │ str ┆ i64 ┆ f64 │ + ╞═════╪═════╪══════╡ + │ c ┆ 5 ┆ null │ + │ b ┆ 6 ┆ null │ + │ a ┆ 4 ┆ null │ + │ d ┆ 7 ┆ null │ + └─────┴─────┴──────┘ """ def __init__( @@ -161,28 +188,24 @@ def __init__( match_dtypes: bool = False, verbose: bool = True, ): - if missing_values not in ["raise", "ignore"]: - raise ValueError( - "missing_values takes only values 'raise' or 'ignore'." - f"Got '{missing_values} instead." - ) + _check_param_missing_values(missing_values) if not isinstance(match_dtypes, bool): raise ValueError( "match_dtypes takes only booleans True and False. " - f"Got '{match_dtypes} instead." + f"Got {match_dtypes} instead." ) if not isinstance(verbose, bool): raise ValueError( - f"verbose takes only booleans True and False. Got '{verbose} instead." + f"verbose takes only booleans True and False. Got {verbose} instead." ) # note: np.nan is an instance of float!!! if not isinstance(fill_value, (str, int, float)): raise ValueError( - "fill_value takes integers, floats or strings." - f"Got '{fill_value} instead." + "fill_value takes integers, floats or strings. " + f"Got {fill_value} instead." ) self.fill_value = fill_value @@ -190,35 +213,36 @@ def __init__( self.match_dtypes = match_dtypes self.verbose = verbose - def fit(self, X: pd.DataFrame, y: pd.Series = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """Learns and stores the names of the variables in the training dataset. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe. y: None y is not needed for this transformer. You can pass y or None. """ - X = check_X(X) + nw_X = check_X(X) if self.missing_values == "raise": - # check if dataset contains na - _check_contains_na(X, X.columns) + _check_contains_na(X, nw_X.columns) - # save input features - self.feature_names_in_: List[Union[str, int]] = X.columns.tolist() + self.feature_names_in_: List[Union[str, int]] = nw_X.columns + self.n_features_in_ = nw_X.shape[1] - self.n_features_in_ = X.shape[1] - - if self.match_dtypes: - self.dtype_dict_: Dict = X.dtypes.to_dict() + if self.match_dtypes is True: + # narwhals dtypes don't carry the categories of pandas categoricals. + if nwd.is_pandas_dataframe(X) is True: + self._dtype_dict: Dict = X.dtypes.to_dict() + else: + self._dtype_dict = dict(nw_X.schema) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Drops variables that were not seen in the train set and adds variables that were in the train set but not in the data to transform. In other words, it @@ -226,29 +250,30 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: Pandas dataframe, shape = [n_samples, n_features] - The dataframe with variables that match those observed in the train set. + X_new: dataframe of shape = [n_samples, n_features] + The dataframe with variables that match those observed in the train set. """ check_is_fitted(self) - X = check_X(X) + nw_X = check_X(X) + columns = set(nw_X.columns) if self.missing_values == "raise": - # Some variables from the train set may not be present in the test set - # and vice versa. We'll check for nan only in the variables seen during - # training. - vars = [var for var in self.feature_names_in_ if var in X.columns] - _check_contains_na(X, vars) + # Variables from the train set may be missing from X. + _check_contains_na( + X, [var for var in self.feature_names_in_ if var in columns] + ) - _columns_to_drop = list(set(X.columns) - set(self.feature_names_in_)) - _columns_to_add = list(set(self.feature_names_in_) - set(X.columns)) + train_columns = set(self.feature_names_in_) + _columns_to_add = [var for var in self.feature_names_in_ if var not in columns] + _columns_to_drop = [var for var in nw_X.columns if var not in train_columns] - if self.verbose: + if self.verbose is True: if len(_columns_to_add) > 0: print( "The following variables are added to the DataFrame: " @@ -260,36 +285,90 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: f"{_columns_to_drop}" ) - X = X.drop(_columns_to_drop, axis=1) - - # Add missing columns first and then reorder to avoid - # Pandas 3 StringDtype reindex issue (before we used reindex) - X[_columns_to_add] = self.fill_value - X = X[self.feature_names_in_] - - if self.match_dtypes: - _current_dtypes = X.dtypes.to_dict() - _columns_to_update = { - column: new_dtype - for column, new_dtype in self.dtype_dict_.items() - if new_dtype != _current_dtypes[column] - } - - for column, new_dtype in _columns_to_update.items(): - if self.verbose: - print( - f"The {column} dtype is changing from ", - f"{_current_dtypes[column]} to {new_dtype}", - ) - - # Handle pandas 4 future warning - if isinstance(new_dtype, pd.CategoricalDtype): - cats = new_dtype.categories - X[column] = X[column].where(X[column].isin(cats)) - - X = X.astype(_columns_to_update) - - return X + if nwd.is_pandas_dataframe(X) is True: + # pandas is faster than narwhals. + X = X.reindex(columns=self.feature_names_in_, fill_value=self.fill_value) + if self.match_dtypes is True: + X = self._match_dtypes_pandas(X) + return X + + if len(_columns_to_add) > 0: + fill_value = self._fill_value_expression() + nw_X = nw_X.with_columns(fill_value.alias(var) for var in _columns_to_add) + nw_X = nw_X.select(self.feature_names_in_) + + if self.match_dtypes is True: + nw_X = self._match_dtypes_narwhals(nw_X) + + return nw_X.to_native() + + def _fill_value_expression(self) -> nw.Expr: + if isinstance(self.fill_value, float) and np.isnan(self.fill_value): + # polars treats NaN as a value, not as missing data. + return nw.lit(None, dtype=nw.Float64()) + if isinstance(self.fill_value, int) and not isinstance(self.fill_value, bool): + # polars would store integers as Int32, pandas uses int64. + return nw.lit(self.fill_value, dtype=nw.Int64()) + return nw.lit(self.fill_value) + + def _dtypes_to_update(self, current_dtypes: Dict) -> Dict: + dtypes_to_update = { + column: new_dtype + for column, new_dtype in self._dtype_dict.items() + if new_dtype != current_dtypes[column] + } + if self.verbose is True: + for column, new_dtype in dtypes_to_update.items(): + print( + f"The {column} dtype is changing from ", + f"{current_dtypes[column]} to {new_dtype}", + ) + return dtypes_to_update + + def _match_dtypes_pandas(self, X): + dtypes_to_update = self._dtypes_to_update(X.dtypes.to_dict()) + + for column, new_dtype in dtypes_to_update.items(): + # Handle pandas 4 future warning + if new_dtype.name == "category": + X[column] = X[column].where(X[column].isin(new_dtype.categories)) + elif new_dtype.kind in "iub" and X[column].hasnans is True: + # numpy integers can't hold NaN and booleans turn it into True. + dtypes_to_update[column] = self._nullable_dtype(new_dtype) + + return X.astype(dtypes_to_update) + + def _nullable_dtype(self, dtype) -> str: + if dtype.kind == "b": + return "boolean" + prefix = "UInt" if dtype.kind == "u" else "Int" + return f"{prefix}{dtype.itemsize * 8}" + + def _match_dtypes_narwhals(self, nw_X: nw.DataFrame) -> nw.DataFrame: + current_dtypes = nw_X.schema + dtypes_to_update = self._dtypes_to_update(current_dtypes) + if len(dtypes_to_update) == 0: + return nw_X + + expressions = [] + for column, new_dtype in dtypes_to_update.items(): + expression = nw.col(column) + if isinstance(new_dtype, nw.Enum): + # polars raises on values outside the categories, pandas sets NaN. + # Strings, because is_in on an Enum rejects values it doesn't have. + expression = expression.cast(nw.String()) + expression = nw.when(expression.is_in(new_dtype.categories)).then( + expression + ) + elif current_dtypes[column] == nw.String: + # polars does not parse strings when casting them to dates. + if new_dtype == nw.Datetime: + expression = expression.str.to_datetime() + elif new_dtype == nw.Date: + expression = expression.str.to_date() + expressions.append(expression.cast(new_dtype)) + + return nw_X.with_columns(expressions) # for the check_estimator tests def _more_tags(self): diff --git a/feature_engine/scaling/mean_normalization.py b/feature_engine/scaling/mean_normalization.py index 620865735..eec86217a 100644 --- a/feature_engine/scaling/mean_normalization.py +++ b/feature_engine/scaling/mean_normalization.py @@ -4,7 +4,8 @@ import warnings from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._check_init_parameters.check_init_input_params import ( @@ -96,12 +97,36 @@ class MeanNormalisationScaler(BaseNumericalTransformer): >>> mns.fit(X) >>> X = mns.transform(X) >>> X.head() - x - 0 0.496714 - 1 -0.138264 - 2 0.647689 - 3 1.523030 - 4 -0.234153 + x + 0 0.051125 + 1 -0.071456 + 2 0.093623 + 3 0.518122 + 4 -0.084093 + + With polars: + + >>> import numpy as np + >>> import polars as pl + >>> from feature_engine.scaling import MeanNormalisationScaler + >>> np.random.seed(42) + >>> X = pl.DataFrame(dict(x = np.random.lognormal(size = 100))) + >>> mns = MeanNormalisationScaler() + >>> mns.fit(X) + >>> X = mns.transform(X) + >>> X.head() + shape: (5, 1) + ┌───────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞═══════════╡ + │ 0.051125 │ + │ -0.071456 │ + │ 0.093623 │ + │ 0.518122 │ + │ -0.084093 │ + └───────────┘ """ def __init__( @@ -115,25 +140,35 @@ def __init__( self.variables = _check_variables_input_value(variables) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Finds the mean and value range of each variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ - # check input dataframe - X, variables_ = self._fit_setup(X) - - mean_ = X[variables_].mean().to_dict() - range_ = (X[variables_].max() - X[variables_].min()).to_dict() + nw_X, variables_ = self._fit_setup(X) + + if len(variables_) == 0: + # return_empty=True can leave variables_ empty; narwhals' select([]) + # collapses row count too, so .to_numpy() would reduce over 0 rows. + mean_: dict = {} + range_: dict = {} + else: + values = nw_X.select(nw.col(variables_)).to_numpy() + mean_arr = values.mean(axis=0) + range_arr = values.max(axis=0) - values.min(axis=0) + # .tolist() converts numpy scalars to plain Python int/float, + # matching the dtype the old pandas .to_dict() used to return. + mean_ = dict(zip(variables_, mean_arr.tolist())) + range_ = dict(zip(variables_, range_arr.tolist())) # check for constant columns constant_columns = [col for col, value in range_.items() if value == 0] @@ -150,51 +185,65 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Transform the variables using mean normalisation. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) # transformation - X[self.variables_] = (X[self.variables_] - self.mean_) / self.range_ + new_series = [ + nw.new_series( + var, + (nw_X.get_column(var).to_numpy() - self.mean_[var]) / self.range_[var], + backend=nw_X.implementation, + ) + for var in self.variables_ + ] + nw_X = nw_X.with_columns(*new_series) - return X + return nw_X.to_native() - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Convert the data back to the original representation. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_tr: pandas dataframe + X_tr: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) # inverse transform - X[self.variables_] = X[self.variables_] * self.range_ + self.mean_ + new_series = [ + nw.new_series( + var, + nw_X.get_column(var).to_numpy() * self.range_[var] + self.mean_[var], + backend=nw_X.implementation, + ) + for var in self.variables_ + ] + nw_X = nw_X.with_columns(*new_series) - return X + return nw_X.to_native() # TODO: remove in version 2.1.0 diff --git a/feature_engine/selection/information_value.py b/feature_engine/selection/information_value.py index b58dbeb9d..fd0dd53fe 100644 --- a/feature_engine/selection/information_value.py +++ b/feature_engine/selection/information_value.py @@ -230,8 +230,12 @@ def fit(self, X: pd.DataFrame, y: pd.Series): self.information_values_ = {} for var in self.variables_: - total_pos, total_neg, woe = self._calculate_woe(X, y, var) - iv = self._calculate_iv(total_pos, total_neg, woe) + woe, _ = self._calculate_woe(X, y, var) + iv = self._calculate_iv( + woe["__pos__"].to_numpy(), + woe["__neg__"].to_numpy(), + woe["__woe__"].to_numpy(), + ) self.information_values_[var] = iv self.features_to_drop_ = [ diff --git a/feature_engine/tags.py b/feature_engine/tags.py index ad36b030a..0c15bb1a0 100644 --- a/feature_engine/tags.py +++ b/feature_engine/tags.py @@ -1,9 +1,3 @@ -import sklearn -from sklearn.utils.fixes import parse_version - -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) - - def _return_tags(): tags = { "preserves_dtype": [], @@ -32,14 +26,13 @@ def _return_tags(): }, } - if sklearn_version > parse_version("1.6"): - msg1 = "against Feature-engines design." - msg2 = "Our transformers do not preserve dtype." - all_fail = { - "check_do_not_raise_errors_in_init_or_set_params": msg1, - "check_transformer_preserve_dtypes": msg2, - # TODO: investigate this test further. - "check_n_features_in_after_fitting": "not sure why it fails, we do check.", - } - tags["_xfail_checks"].update(all_fail) # type: ignore + msg1 = "against Feature-engines design." + msg2 = "Our transformers do not preserve dtype." + all_fail = { + "check_do_not_raise_errors_in_init_or_set_params": msg1, + "check_transformer_preserve_dtypes": msg2, + # TODO: investigate this test further. + "check_n_features_in_after_fitting": "not sure why it fails, we do check.", + } + tags["_xfail_checks"].update(all_fail) # type: ignore return tags diff --git a/feature_engine/text/text_features.py b/feature_engine/text/text_features.py index 96dd283c0..3d492edf6 100644 --- a/feature_engine/text/text_features.py +++ b/feature_engine/text/text_features.py @@ -1,8 +1,12 @@ # Authors: Ankit Hemant Lade (contributor) # License: BSD 3 clause +import string +from functools import cached_property from typing import List, Optional, Union, cast -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -12,40 +16,175 @@ _check_param_missing_values, ) from feature_engine.dataframe_checks import ( - _check_optional_contains_na, + _check_contains_na, _check_X_matches_training_df, check_X, ) -# Available text features and their computation functions +# The characters Python treats as whitespace. Listed explicitly because the \s of +# polars' regex engine misses \x1c-\x1f, and we want the same counts everywhere. +_WHITESPACE = ( + "\t\n\x0b\x0c\r\x1c\x1d\x1e\x1f \x85\xa0\u1680\u2000\u2001\u2002\u2003\u2004" + "\u2005\u2006\u2007\u2008\u2009\u200a\u2028\u2029\u202f\u205f\u3000" +) +_WORD = f"[^{_WHITESPACE}]+" + +# Each feature is computed from the text statistics of one of the classes below, +# so all backends share the same definitions. TEXT_FEATURES = { - "char_count": lambda x: x.str.replace(r"\s+", "", regex=True).str.len(), - "word_count": lambda x: x.str.strip().str.split().str.len(), - "sentence_count": lambda x: x.str.count(r"[.!?]+"), - "avg_word_length": lambda x: x.str.strip().str.len() - / x.str.strip().str.split().str.len(), - "digit_count": lambda x: x.str.count(r"\d"), - "letter_count": lambda x: x.str.count(r"[a-zA-Z]"), - "uppercase_count": lambda x: x.str.count(r"[A-Z]"), - "lowercase_count": lambda x: x.str.count(r"[a-z]"), - "special_char_count": lambda x: x.str.count(r"[^a-zA-Z0-9\s]"), - "whitespace_count": lambda x: x.str.count(r"\s"), - "whitespace_ratio": lambda x: x.str.count(r"\s") / x.str.len().replace(0, 1), - "digit_ratio": lambda x: x.str.count(r"\d") - / x.str.replace(r"\s+", "", regex=True).str.len().replace(0, 1), - "uppercase_ratio": lambda x: x.str.count(r"[A-Z]") - / x.str.replace(r"\s+", "", regex=True).str.len().replace(0, 1), - "has_digits": lambda x: x.str.contains(r"\d", regex=True).astype(int), - "has_uppercase": lambda x: x.str.contains(r"[A-Z]", regex=True).astype(int), - "is_empty": lambda x: (x.str.len() == 0).astype(int), - "starts_with_uppercase": lambda x: x.str.match(r"^[A-Z]").astype(int), - "ends_with_punctuation": lambda x: x.str.match(r".*[.!?]$").astype(int), - "unique_word_count": lambda x: (x.str.lower().str.split().apply(set).str.len()), - "lexical_diversity": lambda x: x.str.lower().str.split().apply(set).str.len() - / x.str.strip().str.split().str.len(), + "char_count": lambda t: t.length - t.count_in(_WHITESPACE), + "word_count": lambda t: t.word_count, + "sentence_count": lambda t: t.count(r"[.!?]+"), + "avg_word_length": lambda t: (t.length - t.count_in(_WHITESPACE)) + / t.word_count.clip(1), + "digit_count": lambda t: t.count(r"\d"), + "letter_count": lambda t: t.count_in(string.ascii_letters), + "uppercase_count": lambda t: t.count(r"[A-Z]"), + "lowercase_count": lambda t: t.count_in(string.ascii_lowercase), + "special_char_count": lambda t: t.count_not_in( + string.ascii_letters + string.digits + _WHITESPACE + ), + "whitespace_count": lambda t: t.count_in(_WHITESPACE), + "whitespace_ratio": lambda t: t.count_in(_WHITESPACE) / t.length.clip(1), + "digit_ratio": lambda t: t.count(r"\d") + / (t.length - t.count_in(_WHITESPACE)).clip(1), + "uppercase_ratio": lambda t: t.count(r"[A-Z]") + / (t.length - t.count_in(_WHITESPACE)).clip(1), + "has_digits": lambda t: t.contains(r"\d"), + "has_uppercase": lambda t: t.contains(r"[A-Z]"), + "is_empty": lambda t: t.is_empty(), + "starts_with_uppercase": lambda t: t.contains(r"^[A-Z]"), + "ends_with_punctuation": lambda t: t.ends_with_punctuation(), + "unique_word_count": lambda t: t.unique_word_count, + "lexical_diversity": lambda t: t.unique_word_count / t.word_count.clip(1), } +class _PandasText: + """Text statistics of a pandas Series of strings.""" + + def __init__(self, text, native_namespace): + self.text = text + self._pd = native_namespace + # several features share the same counts, and pandas computes them eagerly + self._counts: dict = {} + + @cached_property + def length(self): + return self.text.str.len() + + @cached_property + def word_count(self): + # a Python loop is 2x faster than pandas' str.split().str.len() + words = [len(s.split()) for s in self.text.tolist()] + return self._pd.Series(words, index=self.text.index) + + @cached_property + def unique_word_count(self): + words = [len(set(s.lower().split())) for s in self.text.tolist()] + return self._pd.Series(words, index=self.text.index) + + def count(self, pattern): + if ("count", pattern) not in self._counts: + self._counts[("count", pattern)] = self.text.str.count(pattern) + return self._counts[("count", pattern)] + + def count_in(self, characters): + # deleting the characters with translate is faster than a regex count + if ("count_in", characters) not in self._counts: + self._counts[("count_in", characters)] = ( + self.length - self.count_not_in(characters) + ) + return self._counts[("count_in", characters)] + + def count_not_in(self, characters): + table = str.maketrans("", "", characters) + return self.text.str.translate(table).str.len() + + def contains(self, pattern): + return self.text.str.contains(pattern, regex=True).astype(int) + + def is_empty(self): + return self.text.eq("").astype(int) + + def ends_with_punctuation(self): + return self.text.str.match(r".*[.!?]$").astype(int) + + +class _NarwhalsText: + """Text statistics of a string column, as narwhals expressions.""" + + def __init__(self, text, namespace): + self.text = text + self._ns = namespace + + @property + def length(self): + return self.text.str.len_chars().cast(self._ns.Int64) + + @property + def word_count(self): + return self.count(_WORD) + + @property + def unique_word_count(self): + words = ( + self.text.str.to_lowercase() + .str.replace_all(f"[{_WHITESPACE}]+", " ") + .str.strip_chars(" ") + .str.split(" ") + ) + # splitting a text without words returns one empty word + return ( + self._ns.when(self.word_count == 0) + .then(0) + .otherwise(words.list.unique().list.len()) + .cast(self._ns.Int64) + ) + + def count(self, pattern): + # narwhals can't count matches, but replacing each match with 2 + # characters instead of 1 makes the text 1 character longer per match + return ( + self.text.str.replace_all(pattern, "ab").str.len_chars() + - self.text.str.replace_all(pattern, "a").str.len_chars() + ).cast(self._ns.Int64) + + def count_in(self, characters): + return self.length - self.count_not_in(characters) + + def count_not_in(self, characters): + kept = self.text.str.replace_all(f"[{characters}]", "") + return kept.str.len_chars().cast(self._ns.Int64) + + def contains(self, pattern): + return self.text.str.contains(pattern).cast(self._ns.Int64) + + def is_empty(self): + return (self.text.str.len_chars() == 0).cast(self._ns.Int64) + + def ends_with_punctuation(self): + # Python's regex $ also matches before a final \n, which would let + # "x.\n\n" match on backends that use it + ends = self.text.str.contains(r"^[^\n]*[.!?]\n?$") + return (ends & ~self.text.str.ends_with("\n\n")).cast(self._ns.Int64) + + +class _PolarsText(_NarwhalsText): + """Text statistics of a string column, as polars expressions.""" + + @property + def unique_word_count(self): + words = self.text.str.to_lowercase().str.extract_all(_WORD) + return words.list.n_unique().cast(self._ns.Int64) + + def count(self, pattern): + return self.text.str.count_matches(pattern).cast(self._ns.Int64) + + def count_not_in(self, characters): + return self.count(f"[^{characters}]") + + class TextFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): """ TextFeatures() extracts numerical features from text/string variables. This @@ -64,23 +203,25 @@ class TextFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): features: list, default=None List of text features to extract. Available features are: - - 'char_count': Number of characters in the text + - 'char_count': Number of characters, excluding whitespace - 'word_count': Number of words (whitespace-separated tokens) - 'sentence_count': Number of sentences (based on .!? punctuation) - - 'avg_word_length': Average length of words + - 'avg_word_length': Average number of characters per word - 'digit_count': Number of digit characters - - 'letter_count': Number of alphabetic characters (a-z, A-Z) - - 'uppercase_count': Number of uppercase letters - - 'lowercase_count': Number of lowercase letters - - 'special_char_count': Number of special characters (non-alphanumeric) + - 'letter_count': Number of letters a-z and A-Z + - 'uppercase_count': Number of uppercase letters A-Z + - 'lowercase_count': Number of lowercase letters a-z + - 'special_char_count': Number of characters that are not a-z, A-Z, 0-9 + or whitespace - 'whitespace_count': Number of whitespace characters - 'whitespace_ratio': Ratio of whitespace to total characters - - 'digit_ratio': Ratio of digits to total characters - - 'uppercase_ratio': Ratio of uppercase to total characters + - 'digit_ratio': Ratio of digits to non-whitespace characters + - 'uppercase_ratio': Ratio of uppercase letters to non-whitespace + characters - 'has_digits': Binary indicator if text contains digits - - 'has_uppercase': Binary indicator if text contains uppercase + - 'has_uppercase': Binary indicator if text contains uppercase letters A-Z - 'is_empty': Binary indicator if text is empty - - 'starts_with_uppercase': Binary indicator if text starts with uppercase + - 'starts_with_uppercase': Binary indicator if text starts with A-Z - 'ends_with_punctuation': Binary indicator if text ends with .!? - 'unique_word_count': Number of unique words (case-insensitive) - 'lexical_diversity': Ratio of unique words to total words @@ -88,9 +229,9 @@ class TextFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): If None, extracts all available features. missing_values: string, default='ignore' - If 'ignore', NaNs will be filled with an empty string before feature - extraction. If 'raise', the transformer will raise an error if missing data - is found. + If 'ignore', missing values will be filled with an empty string before + feature extraction. If 'raise', the transformer will raise an error if + missing data is found. drop_original: bool, default=False Whether to drop the original text columns after transformation. @@ -141,8 +282,6 @@ class TextFeatures(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): ... features=['char_count', 'word_count', 'has_digits'] ... ) >>> tf.fit(X) - TextFeatures(features=['char_count', 'word_count', 'has_digits'], - variables=['text']) >>> X = tf.transform(X) >>> pd.options.display.max_columns = 10 >>> print(X) @@ -160,7 +299,6 @@ def __init__( drop_original: bool = False, ) -> None: - # Validate variables if isinstance(variables, str): variables = [variables] if not isinstance(variables, list) or not all( @@ -168,24 +306,17 @@ def __init__( ): raise ValueError( "variables must be a string or a list of strings. " - f"Got {type(variables).__name__} instead." + f"Got {variables} instead." ) - # Validate features - if features is not None: - if not isinstance(features, list) or not all( - isinstance(f, str) for f in features - ): - raise ValueError( - "features must be None or a list of strings. " - f"Got {type(features).__name__} instead." - ) - invalid_features = set(features) - set(TEXT_FEATURES.keys()) - if invalid_features: - raise ValueError( - f"Invalid features: {invalid_features}. " - f"Available features are: {list(TEXT_FEATURES.keys())}" - ) + if features is not None and ( + not isinstance(features, list) + or not all(isinstance(f, str) and f in TEXT_FEATURES for f in features) + ): + raise ValueError( + "features must be None or a list with any of " + f"{list(TEXT_FEATURES.keys())}. Got {features} instead." + ) _check_param_drop_original(drop_original) _check_param_missing_values(missing_values) @@ -195,38 +326,38 @@ def __init__( self.missing_values = missing_values self.drop_original = drop_original - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y=None): """ This transformer does not learn any parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, or np.array. Defaults to None. + y: Series, or np.array. Defaults to None. The target. It is not needed in this transformer. You can pass y or None. """ + nw_X = check_X(X) - # check input dataframe - X = check_X(X) - - # Validate user-specified variables exist - missing = set(self.variables) - set(X.columns) - if missing: + missing = set(self.variables) - set(nw_X.columns) + if len(missing) > 0: raise ValueError(f"Variables {missing} are not present in the dataframe.") - # Validate that the variables are object or string - non_text = [ - col - for col in self.variables - if not ( - pd.api.types.is_string_dtype(X[col]) - or pd.api.types.is_object_dtype(X[col]) - ) - ] - if non_text: + non_text = [] + for var in self.variables: + dtype = nw_X.get_column(var).dtype + # pandas categories can be numbers, polars categories are always strings + if isinstance(dtype, nw.Categorical) and nwd.is_pandas_dataframe(X) is True: + is_text = X[var].cat.categories.inferred_type == "string" + else: + is_text = isinstance( + dtype, (nw.String, nw.Object, nw.Categorical, nw.Enum) + ) + if is_text is False: + non_text.append(var) + if len(non_text) > 0: raise ValueError( f"Variables {non_text} are not object or string. " "Please provide text variables only." @@ -234,99 +365,99 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): self.variables_ = self.variables - # check if dataset contains na if self.missing_values == "raise": - _check_optional_contains_na(X, cast(list[Union[str, int]], self.variables_)) + _check_contains_na( + X, cast(list[Union[str, int]], self.variables_), error_msg="optional" + ) - # Set features to extract if self.features is None: self.features_ = list(TEXT_FEATURES.keys()) else: self.features_ = self.features - # save input features - self.feature_names_in_ = X.columns.tolist() - - # save train set shape - self.n_features_in_ = X.shape[1] + self.feature_names_in_ = nw_X.columns + self.n_features_in_ = nw_X.shape[1] return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Extract text features and add them to the dataframe. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the original columns plus the new text features. """ - - # Check method fit has been called check_is_fitted(self) + nw_X = check_X(X) + _check_X_matches_training_df(nw_X, self.n_features_in_) - # check that input is a dataframe - X = check_X(X) - - # Check if input data contains same number of columns as dataframe used to fit. - _check_X_matches_training_df(X, self.n_features_in_) - - # check if dataset contains na if self.missing_values == "raise": - _check_optional_contains_na(X, cast(list[Union[str, int]], self.variables_)) - else: - X[self.variables_] = X[self.variables_].fillna("") - - # reorder variables to match train set - X = X[self.feature_names_in_] - - # Extract features for each text variable - for var in self.variables_: - for feature_name in self.features_: - new_col_name = f"{var}_{feature_name}" - feature_func = TEXT_FEATURES[feature_name] - X[new_col_name] = feature_func(X[var]) + _check_contains_na( + X, cast(list[Union[str, int]], self.variables_), error_msg="optional" + ) - # Fill any NaN values resulting from computation with 0 - X[new_col_name] = X[new_col_name].fillna(0) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + X_new = self._transform_pandas(X, nw.get_native_namespace(nw_X)) + elif nwd.is_polars_dataframe(X) is True: + # polars counts regex matches natively, narwhals needs two replacements. + X_new = self._transform_expressions( + X, nw.get_native_namespace(nw_X), _PolarsText + ) + else: + X_new = self._transform_expressions(nw_X, nw, _NarwhalsText).to_native() - if self.drop_original: - X = X.drop(columns=self.variables_) + return X_new - return X + def _transform_pandas(self, X, native_namespace): + X_new = X[self.feature_names_in_] + if self.missing_values == "ignore": + X_new = X_new.fillna({var: "" for var in self.variables_}) - def get_feature_names_out(self, input_features=None) -> List[str]: - """ - Get output feature names for transformation. + new_features = [] + for var in self.variables_: + statistics = _PandasText(X_new[var], native_namespace) + new_features += [ + TEXT_FEATURES[feature](statistics).rename(f"{var}_{feature}") + for feature in self.features_ + ] - Parameters - ---------- - input_features : array-like of str or None, default=None - Input features. If ``None``, uses ``feature_names_in_``. + X_new = native_namespace.concat([X_new, *new_features], axis=1) + if self.drop_original is True: + X_new = X_new.drop(columns=self.variables_) - Returns - ------- - feature_names_out : list of str - Output feature names. - """ - check_is_fitted(self) + return X_new - # Start with original features - if self.drop_original: - feature_names = [ - f for f in self.feature_names_in_ if f not in self.variables_ + def _transform_expressions(self, X, namespace, text_class): + # polars and narwhals expressions share the API used here + filled_text, new_features = [], [] + for var in self.variables_: + if self.missing_values == "ignore": + filled_text.append(namespace.col(var).fill_null("")) + text = namespace.col(var).cast(namespace.String).fill_null("") + statistics = text_class(text, namespace) + new_features += [ + TEXT_FEATURES[feature](statistics).alias(f"{var}_{feature}") + for feature in self.features_ ] - else: - feature_names = list(self.feature_names_in_) - # Add new text feature names - for var in self.variables_: - for feature_name in self.features_: - feature_names.append(f"{var}_{feature_name}") + X_new = X.select(self.feature_names_in_).with_columns( + *filled_text, *new_features + ) + if self.drop_original is True: + X_new = X_new.drop(self.variables_) + + return X_new - return feature_names + def _get_new_features_name(self) -> List[str]: + """Return the names of the created features.""" + return [ + f"{var}_{feature}" for var in self.variables_ for feature in self.features_ + ] diff --git a/feature_engine/transformation/arcsin.py b/feature_engine/transformation/arcsin.py index da4045fa4..ab079780f 100644 --- a/feature_engine/transformation/arcsin.py +++ b/feature_engine/transformation/arcsin.py @@ -3,8 +3,9 @@ from typing import List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._check_init_parameters.check_init_input_params import ( @@ -105,6 +106,30 @@ class ArcsinTransformer(BaseNumericalTransformer): 2 0.144664 3 0.783236 4 0.650777 + + With polars: + + >>> import numpy as np + >>> import polars as pl + >>> from feature_engine.transformation import ArcsinTransformer + >>> np.random.seed(42) + >>> X = pl.DataFrame({"x": list(np.random.beta(1, 1, size=6))}) + >>> ast = ArcsinTransformer() + >>> ast.fit(X) + >>> ast.transform(X) + shape: (6, 1) + ┌──────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞══════════╡ + │ 0.785437 │ + │ 0.253389 │ + │ 0.144664 │ + │ 0.783236 │ + │ 0.650777 │ + │ 0.883313 │ + └──────────┘ """ def __init__( @@ -118,25 +143,25 @@ def __init__( self.variables = _check_variables_input_value(variables) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn parameters. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ - # check input dataframe - X, variables_ = self._fit_setup(X) + nw_X, variables_ = self._fit_setup(X) # check if the variables are in the correct range - if ((X[variables_] < 0) | (X[variables_] > 1)).any().any(): + values = nw_X.select(nw.col(variables_)).to_numpy() + if np.any((values < 0) | (values > 1)): raise ValueError( "Some variables contain values outside the possible range 0-1. " "Can't apply the arcsin transformation. " @@ -147,52 +172,65 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Apply the arcsin transformation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) + values = nw_X.select(nw.col(self.variables_)).to_numpy() # check if the variables are in the correct range - if ((X[self.variables_] < 0) | (X[self.variables_] > 1)).any().any(): + if np.any((values < 0) | (values > 1)): raise ValueError( "Some variables contain values outside the possible range 0-1. " "Can't apply the arcsin transformation." ) # transform - X.loc[:, self.variables_] = np.arcsin(np.sqrt(X.loc[:, self.variables_])) + result = np.arcsin(np.sqrt(values)) + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Convert the data back to the original representation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_tr: pandas dataframe + X_tr: dataframe The dataframe with the transformed variables. """ + nw_X = nw.from_native(X, eager_only=True) + values = nw_X.select(nw.col(self.variables_)).to_numpy() + # inverse_transform - X.loc[:, self.variables_] = (np.sin(X.loc[:, self.variables_])) ** 2 + result = np.sin(values) ** 2 + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X diff --git a/feature_engine/transformation/arcsinh.py b/feature_engine/transformation/arcsinh.py index 92ebf2c0a..5f132ae77 100644 --- a/feature_engine/transformation/arcsinh.py +++ b/feature_engine/transformation/arcsinh.py @@ -3,8 +3,9 @@ from typing import List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._check_init_parameters.check_init_input_params import ( @@ -119,11 +120,35 @@ class ArcSinhTransformer(BaseNumericalTransformer): >>> X = ast.transform(X) >>> X.head() x - 0 7.516076 - 1 -6.330816 - 2 7.780254 - 3 8.825252 - 4 -6.995893 + 0 6.901163 + 1 -5.622327 + 2 7.166558 + 3 8.021604 + 4 -6.149128 + + With polars: + + >>> import numpy as np + >>> import polars as pl + >>> from feature_engine.transformation import ArcSinhTransformer + >>> np.random.seed(42) + >>> X = pl.DataFrame({"x": list(np.random.randn(6) * 1000)}) + >>> ast = ArcSinhTransformer() + >>> ast.fit(X) + >>> ast.transform(X) + shape: (6, 1) + ┌───────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞═══════════╡ + │ 6.901163 │ + │ -5.622327 │ + │ 7.166558 │ + │ 8.021604 │ + │ -6.149128 │ + │ -6.149058 │ + └───────────┘ """ def __init__( @@ -152,17 +177,17 @@ def __init__( self.loc = float(loc) self.scale = float(scale) - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Selects the numerical variables and stores feature names. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. Returns @@ -171,64 +196,66 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): The fitted transformer. """ - # check input dataframe and find/check numerical variables - X, variables_ = self._fit_setup(X) + _, variables_ = self._fit_setup(X) self.variables_ = variables_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Transform the variables using the arcsinh function. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) - - # Ensure float dtype for the transformation - X[self.variables_] = X[self.variables_].astype(float) + nw_X = self._check_transform_input_and_state(X) # Apply arcsinh transformation: arcsinh((x - loc) / scale) - X.loc[:, self.variables_] = np.arcsinh( - (X.loc[:, self.variables_] - self.loc) / self.scale - ) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) + result = np.arcsinh((values - self.loc) / self.scale) + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Convert the data back to the original representation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be inverse transformed. Returns ------- - X_tr: pandas dataframe + X_tr: dataframe The dataframe with the inverse transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) # Inverse transform: x = sinh(y) * scale + loc - X.loc[:, self.variables_] = ( - np.sinh(X.loc[:, self.variables_]) * self.scale + self.loc - ) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) + result = np.sinh(values) * self.scale + self.loc + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X diff --git a/feature_engine/transformation/boxcox.py b/feature_engine/transformation/boxcox.py index 52d5feffb..2085b6a54 100644 --- a/feature_engine/transformation/boxcox.py +++ b/feature_engine/transformation/boxcox.py @@ -3,9 +3,11 @@ from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import numpy as np +import scipy.special as spsp import scipy.stats as stats -from scipy.special import inv_boxcox +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._check_init_parameters.check_init_input_params import ( @@ -119,6 +121,30 @@ class BoxCoxTransformer(BaseNumericalTransformer): 2 0.662654 3 1.607518 4 -0.232237 + + With polars: + + >>> import numpy as np + >>> import polars as pl + >>> from feature_engine.transformation import BoxCoxTransformer + >>> np.random.seed(42) + >>> X = pl.DataFrame({"x": list(np.random.lognormal(size=6))}) + >>> bct = BoxCoxTransformer() + >>> bct.fit(X) + >>> bct.transform(X) + shape: (6, 1) + ┌───────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞═══════════╡ + │ 0.403681 │ + │ -0.146883 │ + │ 0.495725 │ + │ 0.845914 │ + │ -0.259585 │ + │ -0.259565 │ + └───────────┘ """ def __init__( @@ -132,27 +158,28 @@ def __init__( self.variables = _check_variables_input_value(variables) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the optimal lambda for the BoxCox transformation. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ - # check input dataframe - X, variables_ = self._fit_setup(X) + nw_X, variables_ = self._fit_setup(X) + values = nw_X.select(nw.col(variables_)).to_numpy().astype(float) lambda_dict_ = {} - - for var in variables_: - _, lambda_dict_[var] = stats.boxcox(X[var]) + # lambda search is per-column and not vectorizable across columns, + # unlike transform()'s elementwise application once lambdas are known + for i, var in enumerate(variables_): + _, lambda_dict_[var] = stats.boxcox(values[:, i]) self.variables_ = variables_ self.lambda_dict_ = lambda_dict_ @@ -160,55 +187,65 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Apply the BoxCox transformation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) # check contains zero or negative values - if (X[self.variables_] <= 0).any().any(): + if (values <= 0).any(): raise ValueError("Data must be positive.") # transform - for feature in self.variables_: - X[feature] = stats.boxcox(X[feature], lmbda=self.lambda_dict_[feature]) + lmbdas = np.array([self.lambda_dict_[var] for var in self.variables_]) + result = spsp.boxcox(values, lmbdas) + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Convert the data back to the original representation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be inverse transformed. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the original variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) # inverse transform - for feature in self.variables_: - X[feature] = inv_boxcox(X[feature], self.lambda_dict_[feature]) + lmbdas = np.array([self.lambda_dict_[var] for var in self.variables_]) + result = spsp.inv_boxcox(values, lmbdas) + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X diff --git a/feature_engine/transformation/log.py b/feature_engine/transformation/log.py index 9cd12c307..1c289ee5a 100644 --- a/feature_engine/transformation/log.py +++ b/feature_engine/transformation/log.py @@ -4,8 +4,9 @@ import warnings from typing import Dict, List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._base_transformers.mixins import FitFromDictMixin @@ -126,6 +127,30 @@ class LogTransformer(BaseNumericalTransformer, FitFromDictMixin): 2 0.647689 3 1.523030 4 -0.234153 + + With polars: + + >>> import numpy as np + >>> import polars as pl + >>> from feature_engine.transformation import LogTransformer + >>> np.random.seed(42) + >>> X = pl.DataFrame({"x": list(np.random.lognormal(size=6))}) + >>> lt = LogTransformer() + >>> lt.fit(X) + >>> lt.transform(X) + shape: (6, 1) + ┌───────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞═══════════╡ + │ 0.496714 │ + │ -0.138264 │ + │ 0.647689 │ + │ 1.52303 │ + │ -0.234153 │ + │ -0.234137 │ + └───────────┘ """ def __init__( @@ -154,7 +179,7 @@ def __init__( self.base = base self.C = C - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the constant C to add to the variable before the logarithm transformation, if C="auto". Otherwise, this transformer does not learn @@ -162,35 +187,34 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ - # check input dataframe if isinstance(self.C, dict): - X, variables_ = super()._fit_from_dict(X, self.C) + nw_X, variables_ = super()._fit_from_dict(X, self.C) else: - X, variables_ = self._fit_setup(X) + nw_X, variables_ = self._fit_setup(X) + + values = nw_X.select(nw.col(variables_)).to_numpy() + values = values.astype(float) C_ = self.C - # calculate C to add to each variable + # 0 for strictly positive variables, abs(min) + 1 (shift to positive) + # otherwise. if self.C == "auto": - # we add 0 to positive variables - c_dict = {var: 0 for var in variables_ if X[var].min() > 0} - - # we add the minimum plus 1 to non-positive variables - non_positive_vars = [var for var in variables_ if var not in c_dict.keys()] - c_dict.update(dict(X[non_positive_vars].min(axis=0).abs() + 1)) - C_ = c_dict # type:ignore + mins = values.min(axis=0) + c_values = np.where(mins > 0, 0, np.abs(mins) + 1) + C_ = dict(zip(variables_, c_values.tolist())) # C=0 is the original LogTransformer contract: no constant is added, # so fail fast at fit time exactly as before this class supported C. - if C_ == 0 and (X[variables_] <= 0).any().any(): + if C_ == 0 and np.any(values <= 0): raise ValueError( "Some variables contain zero or negative values, can't apply log" ) @@ -201,23 +225,29 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def _c_as_array(self) -> Union[int, float, np.ndarray]: + """Broadcastable form of C_: a plain scalar, or a numpy array ordered + to line up column-wise with self.variables_ when C_ is a dict.""" + if isinstance(self.C_, dict): + return np.array([self.C_[var] for var in self.variables_], dtype=float) + return self.C_ + + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Transform the variables with the logarithm of x plus the constant C. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) if self.C_ == 0: error_msg = ( @@ -229,42 +259,56 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: + " constant C, can't apply log." ) - if (X[self.variables_] + self.C_ <= 0).any().any(): - raise ValueError(error_msg) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) + shifted = values + self._c_as_array() - X[self.variables_] = X[self.variables_].astype(float) + if np.any(shifted <= 0): + raise ValueError(error_msg) # transform if self.base == "e": - X.loc[:, self.variables_] = np.log(X.loc[:, self.variables_] + self.C_) - elif self.base == "10": - X.loc[:, self.variables_] = np.log10(X.loc[:, self.variables_] + self.C_) + result = np.log(shifted) + else: + result = np.log10(shifted) + + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Convert the data back to the original representation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_tr: pandas dataframe + X_tr: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) + c_arr = self._c_as_array() # inverse_transform if self.base == "e": - X.loc[:, self.variables_] = np.exp(X.loc[:, self.variables_]) - self.C_ - elif self.base == "10": - X.loc[:, self.variables_] = 10 ** X.loc[:, self.variables_] - self.C_ + result = np.exp(values) - c_arr + else: + result = 10**values - c_arr + + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X diff --git a/feature_engine/transformation/power.py b/feature_engine/transformation/power.py index 89aea9bf2..5949e20c2 100644 --- a/feature_engine/transformation/power.py +++ b/feature_engine/transformation/power.py @@ -3,8 +3,9 @@ from typing import List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._check_init_parameters.check_init_input_params import ( @@ -99,6 +100,30 @@ class PowerTransformer(BaseNumericalTransformer): 2 1.382432 3 2.141518 4 0.889517 + + With polars: + + >>> import numpy as np + >>> import polars as pl + >>> from feature_engine.transformation import PowerTransformer + >>> np.random.seed(42) + >>> X = pl.DataFrame({"x": list(np.random.lognormal(size=6))}) + >>> pt = PowerTransformer() + >>> pt.fit(X) + >>> pt.transform(X) + shape: (6, 1) + ┌──────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞══════════╡ + │ 1.281918 │ + │ 0.933203 │ + │ 1.382432 │ + │ 2.141518 │ + │ 0.889517 │ + │ 0.889524 │ + └──────────┘ """ def __init__( @@ -117,71 +142,79 @@ def __init__( self.return_empty = return_empty self.exp = exp - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The training input samples. - Can be the entire dataframe, not just the variables to transform. + X: dataframe of shape = [n_samples, n_features]. + The training input samples. Can be the entire dataframe, not just the + variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ - # check input dataframe - X, variables_ = self._fit_setup(X) + _, variables_ = self._fit_setup(X) self.variables_ = variables_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Apply the power transformation to the variables. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas Dataframe + X_new: dataframe The dataframe with the power transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) # transform - X[self.variables_] = X[self.variables_].astype(float) - X.loc[:, self.variables_] = np.power(X.loc[:, self.variables_], self.exp) + result = np.power(values, self.exp) + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Convert the data back to the original representation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_tr: pandas Dataframe + X_tr: dataframe The dataframe with the power transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) # inverse_transform - X.loc[:, self.variables_] = np.power(X.loc[:, self.variables_], 1 / self.exp) + result = np.power(values, 1 / self.exp) + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X diff --git a/feature_engine/transformation/reciprocal.py b/feature_engine/transformation/reciprocal.py index 22678544c..3f8788aa3 100644 --- a/feature_engine/transformation/reciprocal.py +++ b/feature_engine/transformation/reciprocal.py @@ -3,7 +3,9 @@ from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._check_init_parameters.check_init_input_params import ( @@ -97,6 +99,30 @@ class ReciprocalTransformer(BaseNumericalTransformer): 2 0.115164 3 0.110047 4 0.101726 + + With polars: + + >>> import numpy as np + >>> import polars as pl + >>> from feature_engine.transformation import ReciprocalTransformer + >>> np.random.seed(42) + >>> X = pl.DataFrame({"x": list(10 - np.random.exponential(size=6))}) + >>> rt = ReciprocalTransformer() + >>> rt.fit(X) + >>> rt.transform(X) + shape: (6, 1) + ┌──────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞══════════╡ + │ 0.104924 │ + │ 0.143064 │ + │ 0.115164 │ + │ 0.110047 │ + │ 0.101726 │ + │ 0.101725 │ + └──────────┘ """ def __init__( @@ -109,25 +135,25 @@ def __init__( self.variables = _check_variables_input_value(variables) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ This transformer does not learn parameters. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ - # check input dataframe - X, variables_ = self._fit_setup(X) + nw_X, variables_ = self._fit_setup(X) # check if the variables contain the value 0 - if (X[variables_] == 0).any().any(): + values = nw_X.select(nw.col(variables_)).to_numpy() + if np.any(values == 0): raise ValueError( "Some variables contain the value zero, can't apply reciprocal " "transformation." @@ -138,49 +164,53 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Apply the reciprocal 1 / x transformation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) + values = nw_X.select(nw.col(self.variables_)).to_numpy() # check if the variables contain the value 0 - if (X[self.variables_] == 0).any().any(): + if np.any(values == 0): raise ValueError( "Some variables contain the value zero, can't apply reciprocal " "transformation." ) # transform - X[self.variables_] = X[self.variables_].astype(float) - X.loc[:, self.variables_] = 1 / X.loc[:, self.variables_] + result = 1 / values + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Convert the data back to the original representation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_tr: pandas dataframe + X_tr: dataframe The dataframe with the transformed variables. """ # inverse_transform diff --git a/feature_engine/transformation/yeojohnson.py b/feature_engine/transformation/yeojohnson.py index 82fa53dac..2b45a8d7d 100644 --- a/feature_engine/transformation/yeojohnson.py +++ b/feature_engine/transformation/yeojohnson.py @@ -3,9 +3,10 @@ from typing import List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd import scipy.stats as stats +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._check_init_parameters.check_init_input_params import ( @@ -108,11 +109,35 @@ class YeoJohnsonTransformer(BaseNumericalTransformer): >>> X = yjt.transform(X) >>> X.head() x - 0 -267042.906453 - 1 -444357.138990 - 2 -221626.115742 - 3 -23647.632651 - 4 -467264.993249 + 0 -267042.661354 + 1 -444356.715596 + 2 -221625.915167 + 3 -23647.614887 + 4 -467264.546413 + + With polars: + + >>> import numpy as np + >>> import polars as pl + >>> from feature_engine.transformation import YeoJohnsonTransformer + >>> np.random.seed(42) + >>> X = pl.DataFrame({"x": list(np.random.lognormal(size=6) - 10)}) + >>> yjt = YeoJohnsonTransformer() + >>> yjt.fit(X) + >>> yjt.transform(X) + shape: (6, 1) + ┌────────────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞════════════════╡ + │ -467714.164249 │ + │ -795057.401919 │ + │ -385148.281012 │ + │ -37417.353351 │ + │ -837807.71099 │ + │ -837800.580457 │ + └────────────────┘ """ def __init__( @@ -125,27 +150,30 @@ def __init__( self.variables = _check_variables_input_value(variables) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the optimal lambda for the Yeo-Johnson transformation. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ - # check input dataframe - X, variables_ = self._fit_setup(X) + nw_X, variables_ = self._fit_setup(X) - lambda_dict_ = {} + values = nw_X.select(nw.col(variables_)).to_numpy() + values = values.astype(float) - for var in variables_: - _, lambda_dict_[var] = stats.yeojohnson(X[var]) + # scipy searches the optimal lambda one column at a time, there is no + # vectorized multi-column form of the search. + lambda_dict_ = {} + for i, var in enumerate(variables_): + _, lambda_dict_[var] = stats.yeojohnson(values[:, i]) self.variables_ = variables_ self.lambda_dict_ = lambda_dict_ @@ -153,55 +181,71 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Apply the Yeo-Johnson transformation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X: pandas dataframe + X_new: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted + nw_X = self._check_transform_input_and_state(X) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) - X = self._check_transform_input_and_state(X) - for feature in self.variables_: - X[feature] = stats.yeojohnson(X[feature], lmbda=self.lambda_dict_[feature]) + # transform + result = np.empty_like(values) + for i, var in enumerate(self.variables_): + result[:, i] = stats.yeojohnson(values[:, i], lmbda=self.lambda_dict_[var]) + + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() return X - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Convert the data back to the original representation. Parameters ---------- - X: pandas DataFrame of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_tr: pandas dataframe + X_tr: dataframe The dataframe with the transformed variables. """ - # check input dataframe and if class was fitted - X = self._check_transform_input_and_state(X) - - for feature in self.variables_: - X[feature] = self._inverse_transform_series( - X[feature], lmbda=self.lambda_dict_[feature] + nw_X = self._check_transform_input_and_state(X) + values = nw_X.select(nw.col(self.variables_)).to_numpy().astype(float) + + # inverse_transform + result = np.empty_like(values) + for i, var in enumerate(self.variables_): + result[:, i] = self._inverse_transform_array( + values[:, i], lmbda=self.lambda_dict_[var] ) + new_series = [ + nw.new_series(var, result[:, i], backend=nw_X.implementation) + for i, var in enumerate(self.variables_) + ] + X = nw_X.with_columns(*new_series).to_native() + return X - def _inverse_transform_series(self, X: pd.Series, lmbda: float) -> pd.Series: - x_inv = pd.Series(np.zeros_like(X), index=X.index) + def _inverse_transform_array(self, X: np.ndarray, lmbda: float) -> np.ndarray: + x_inv = np.zeros_like(X) pos = X >= 0 # when x >= 0 diff --git a/feature_engine/variable_handling/_variable_type_checks.py b/feature_engine/variable_handling/_variable_type_checks.py index 17eb4e41d..a787178df 100644 --- a/feature_engine/variable_handling/_variable_type_checks.py +++ b/feature_engine/variable_handling/_variable_type_checks.py @@ -1,62 +1,113 @@ -import pandas as pd -from pandas.api.types import is_object_dtype, is_string_dtype -from pandas.core.dtypes.common import is_datetime64_any_dtype as is_datetime -from pandas.core.dtypes.common import is_numeric_dtype as is_numeric +import warnings +from datetime import date, datetime +import narwhals as nw +from dateutil.parser import parser -def is_object(s) -> bool: - return is_object_dtype(s) or is_string_dtype(s) +def _is_date_or_datetime(dtype) -> bool: + # nw.selectors.datetime() only matches Datetime, not Date, so this needs + # its own explicit check. + return isinstance(dtype, (nw.Date, nw.Datetime)) -def _is_categorical_and_is_not_datetime(column: pd.Series) -> bool: - # check for datetime only if the type of the categories is not numeric - # because pd.to_datetime throws an error when it is an integer - if isinstance(column.dtype, pd.CategoricalDtype): - is_cat = _is_categories_num(column) or not _is_convertible_to_dt(column) - # check for datetime only if object cannot be cast as numeric because - # if it could pd.to_datetime would convert it to datetime regardless - elif is_object(column): - is_cat = _is_convertible_to_num(column) or not _is_convertible_to_dt(column) - - else: - is_cat = False - - return is_cat +def _looks_like_date_string(value) -> bool: + # taken from pandas + # https://github.com/pandas-dev/pandas/blob/cbae8aea4a31a4052736ab0d23f284ff1e78aa06/pandas/_libs/tslibs/parsing.pyx#L666 + try: + result, _ = parser()._parse(value) + except TypeError: + return False + if result is None: + return False -def _is_categories_num(column: pd.Series) -> bool: - return is_numeric(column.dtype.categories) + fields = ("year", "month", "day", "hour", "minute", "second") + found_fields = sum(1 for field in fields if getattr(result, field) is not None) + return found_fields >= 2 -def _is_convertible_to_dt(column: pd.Series) -> bool: - try: - var = pd.to_datetime(column, utc=True) - return is_datetime(var) - except Exception: +def _is_convertible_to_num(s: "nw.Series") -> bool: + values = s.drop_nulls().to_list() + if len(values) == 0: return False - - -def _is_convertible_to_num(column: pd.Series) -> bool: try: - ser = pd.to_numeric(column) + for value in values[:100]: + float(value) except (ValueError, TypeError): - ser = column - return is_numeric(ser) + return False + return True -def _is_categorical_and_is_datetime(column: pd.Series) -> bool: - # check for datetime only if the type of the categories is not numeric - # because pd.to_datetime throws an error when it is an integer - if isinstance(column.dtype, pd.CategoricalDtype): - is_dt = not _is_categories_num(column) and _is_convertible_to_dt(column) +def _is_convertible_to_dt(s: "nw.Series") -> bool: + values = s.drop_nulls() + values_list = values.to_list() + if len(values_list) == 0: + return False + + first_value = values_list[0] + if not isinstance(first_value, (date, datetime)): + if _looks_like_date_string(first_value) is False: + return False + + # Try the backend's own vectorized parser first (faster). + # Fall back to the per-value check (below) when it fails. + with warnings.catch_warnings(): + warnings.simplefilter("ignore", UserWarning) + try: + values.str.to_datetime() + return True + except Exception: + pass + + for value in values_list[:100]: + if isinstance(value, (date, datetime)): + continue + if _looks_like_date_string(value) is False: + return False + return True + + +def _is_categories_num(s: "nw.Series") -> bool: + return s.cat.get_categories().dtype.is_numeric() + + +def _is_categorical_and_is_not_datetime(s: "nw.Series") -> bool: + if isinstance(s.dtype, nw.Enum): + # an explicit, user-defined category set is an unambiguous categorical + # signal, unlike a generic string column, so skip the datetime check + return True + + if isinstance(s.dtype, nw.Categorical): + # check for datetime only if the categories are not numeric, because + # a numeric-backed categorical (pandas-only - polars categories are + # always string-backed) can never hold dates + categories_are_numeric = _is_categories_num(s) + is_convertible_to_dt = _is_convertible_to_dt(s) + return categories_are_numeric is True or is_convertible_to_dt is False + + if isinstance(s.dtype, (nw.String, nw.Object)): + # check for datetime only if the column cannot be cast as numeric, + # because if it could, it would be a numeric column, not a date + is_convertible_to_num = _is_convertible_to_num(s) + is_convertible_to_dt = _is_convertible_to_dt(s) + return is_convertible_to_num is True or is_convertible_to_dt is False + + return False + + +def _is_categorical_and_is_datetime(s: "nw.Series") -> bool: + if isinstance(s.dtype, nw.Enum): + return False - # check for datetime only if object cannot be cast as numeric because - # if it could pd.to_datetime would convert it to datetime regardless - elif is_object(column): - is_dt = not _is_convertible_to_num(column) and _is_convertible_to_dt(column) + if isinstance(s.dtype, nw.Categorical): + categories_are_numeric = _is_categories_num(s) + is_convertible_to_dt = _is_convertible_to_dt(s) + return categories_are_numeric is False and is_convertible_to_dt is True - else: - is_dt = False + if isinstance(s.dtype, (nw.String, nw.Object)): + is_convertible_to_num = _is_convertible_to_num(s) + is_convertible_to_dt = _is_convertible_to_dt(s) + return is_convertible_to_num is False and is_convertible_to_dt is True - return is_dt + return False diff --git a/feature_engine/variable_handling/check_variables.py b/feature_engine/variable_handling/check_variables.py index 76c4ea7c3..dbff2b434 100644 --- a/feature_engine/variable_handling/check_variables.py +++ b/feature_engine/variable_handling/check_variables.py @@ -2,19 +2,19 @@ from typing import List, Union -import pandas as pd -from pandas.api.types import is_numeric_dtype as is_numeric +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame from feature_engine.variable_handling._variable_type_checks import ( _is_categorical_and_is_datetime, ) -from feature_engine.variable_handling.dtypes import DATETIME_TYPES Variables = Union[int, str, List[Union[str, int]]] def check_numerical_variables( - X: pd.DataFrame, variables: Variables + X: IntoDataFrame, variables: Variables ) -> List[Union[str, int]]: """ Checks that the variables in the list are of type numerical. @@ -23,8 +23,9 @@ def check_numerical_variables( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The dataset. + X : dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. variables : List The list with the names of the variables to check. @@ -41,7 +42,7 @@ def check_numerical_variables( >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> var_ = check_numerical_variables(X, variables=["var_num"]) >>> var_ @@ -51,7 +52,15 @@ def check_numerical_variables( if isinstance(variables, (str, int)): variables = [variables] - if len(X[variables].select_dtypes(exclude="number").columns) > 0: + if nwd.is_pandas_dataframe(X) is True: + not_numerical = len(X[variables].select_dtypes(exclude="number").columns) > 0 + else: + sub_X = nw.from_native(X, eager_only=True).select(variables) + not_numerical = len(sub_X.select(nw.selectors.numeric()).columns) != len( + sub_X.columns + ) + + if not_numerical is True: raise TypeError( "Some of the variables are not numerical. Please cast them as " "numerical before using this transformer." @@ -61,7 +70,7 @@ def check_numerical_variables( def check_categorical_variables( - X: pd.DataFrame, variables: Variables + X: IntoDataFrame, variables: Variables ) -> List[Union[str, int]]: """ Checks that the variables in the list are of type object or categorical. @@ -70,8 +79,9 @@ def check_categorical_variables( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The dataset + X : dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. variables : list The list with the names of the variables to check. @@ -81,6 +91,13 @@ def check_categorical_variables( variables: List The names of the categorical variables. + Notes + ----- + For polars (and other non-pandas dataframes), plain string columns are + accepted as categorical. Polars has no separate "object" dtype the way + pandas does, so its `String` dtype is the only way to represent free-form + text and is treated as categorical here. + Examples -------- >>> import pandas as pd @@ -88,7 +105,7 @@ def check_categorical_variables( >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> var_ = check_categorical_variables(X, "var_cat") >>> var_ @@ -98,7 +115,24 @@ def check_categorical_variables( if isinstance(variables, (str, int)): variables = [variables] - if len(X[variables].select_dtypes(exclude=["O", "category"]).columns) > 0: + if nwd.is_pandas_dataframe(X) is True: + not_categorical = ( + len( + X[variables] + .select_dtypes(exclude=["O", "category", "string"]) + .columns + ) + > 0 + ) + else: + sub_X = nw.from_native(X, eager_only=True).select(variables) + not_categorical = len( + sub_X.select( + nw.selectors.categorical() | nw.selectors.enum() | nw.selectors.string() + ).columns + ) != len(sub_X.columns) + + if not_categorical is True: raise TypeError( "Some of the variables are not categorical. Please cast them as " "object or categorical before using this transformer." @@ -108,7 +142,7 @@ def check_categorical_variables( def check_datetime_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, ) -> List[Union[str, int]]: """ @@ -119,8 +153,9 @@ def check_datetime_variables( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The dataset + X : dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. variables : list The list with the names of the variables to check. @@ -130,6 +165,12 @@ def check_datetime_variables( variables: List The names of the datetime variables. + Notes + ----- + String columns are parsed with flexible, dateutil-backed date guessing, in + addition to ISO-8601 strings and native `Date`/`Datetime` columns, + regardless of the dataframe library backing `X`. + Examples -------- >>> import pandas as pd @@ -137,7 +178,7 @@ def check_datetime_variables( >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> var_date = check_datetime_variables(X, "var_date") >>> var_date @@ -147,13 +188,27 @@ def check_datetime_variables( if isinstance(variables, (str, int)): variables = [variables] - # find non datetime variables, if any: - non_datetime_vars = [] - for column in X[variables].select_dtypes(exclude=DATETIME_TYPES): - if is_numeric(X[column]) or not _is_categorical_and_is_datetime(X[column]): - non_datetime_vars.append(column) + if nwd.is_pandas_dataframe(X) is True: + sub_X = X[variables] + candidates = sub_X.select_dtypes(exclude=["datetime", "datetimetz"]).columns + numeric_cols = set(sub_X.select_dtypes(include="number").columns) + nw_X = nw.from_native(sub_X, eager_only=True) + non_datetime = any( + column in numeric_cols + or not _is_categorical_and_is_datetime(nw_X.get_column(column)) + for column in candidates + ) + else: + sub_X = nw.from_native(X, eager_only=True).select(variables) + candidates = sub_X.select(~nw.selectors.by_dtype(nw.Date, nw.Datetime)).columns + numeric_cols = set(sub_X.select(nw.selectors.numeric()).columns) + non_datetime = any( + column in numeric_cols + or not _is_categorical_and_is_datetime(sub_X.get_column(column)) + for column in candidates + ) - if len(non_datetime_vars) > 0: + if non_datetime is True: raise TypeError( "Some of the variables are not or cannot be parsed as datetime." ) @@ -162,7 +217,7 @@ def check_datetime_variables( def check_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, ) -> List[Union[str, int]]: """ @@ -172,8 +227,9 @@ def check_all_variables( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The dataset + X : dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. variables : list The list with the names of the variables to check. @@ -190,19 +246,24 @@ def check_all_variables( >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> vars_all = check_all_variables(X, ['var_num', 'var_cat', 'var_date']) >>> vars_all ['var_num', 'var_cat', 'var_date'] """ + if nwd.is_pandas_dataframe(X) is True: + columns = set(X.columns) + else: + columns = set(nw.from_native(X, eager_only=True).columns) + if isinstance(variables, (str, int)): - if variables not in X.columns.to_list(): + if variables not in columns: raise KeyError(f"The variable {variables} is not in the dataframe.") variables_ = [variables] else: - if not set(variables).issubset(set(X.columns)): + if set(variables).issubset(columns) is False: raise KeyError("Some of the variables are not in the dataframe.") variables_ = variables diff --git a/feature_engine/variable_handling/dtypes.py b/feature_engine/variable_handling/dtypes.py deleted file mode 100644 index c7d93950c..000000000 --- a/feature_engine/variable_handling/dtypes.py +++ /dev/null @@ -1 +0,0 @@ -DATETIME_TYPES = ("datetimetz", "datetime") diff --git a/feature_engine/variable_handling/find_variables.py b/feature_engine/variable_handling/find_variables.py index 5d072eb56..1a19dfaf5 100644 --- a/feature_engine/variable_handling/find_variables.py +++ b/feature_engine/variable_handling/find_variables.py @@ -1,21 +1,55 @@ -"""Functions to select certain types of variables.""" +"""Functions to select different types of variables.""" import warnings -from typing import List, Tuple, Union +from typing import List, Optional, Tuple, Union -import pandas as pd -from pandas.api.types import is_datetime64_any_dtype as is_datetime -from pandas.core.dtypes.common import is_numeric_dtype as is_numeric +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame from feature_engine.variable_handling._variable_type_checks import ( _is_categorical_and_is_datetime, _is_categorical_and_is_not_datetime, ) -from feature_engine.variable_handling.dtypes import DATETIME_TYPES + + +def _find_nw_categoricals( + X: IntoDataFrame, + variables: Optional[List[Union[str, int]]] = None, + exclude_datetime: bool = True, +) -> List[Union[str, int]]: + if nwd.is_pandas_dataframe(X) is True: + sub_X = X if variables is None else X[variables] + candidates = list( + sub_X.select_dtypes(include=["object", "category", "string"]).columns + ) + nw_X = nw.from_native(sub_X, eager_only=True) + else: + nw_X = nw.from_native(X, eager_only=True) + if variables is not None: + nw_X = nw_X.select(variables) + _NW_SELECTOR = ( + nw.selectors.categorical() + | nw.selectors.enum() + | nw.selectors.string() + | nw.selectors.by_dtype(nw.Object) + ) + # `|`-combined selectors don't preserve column order, + # so re-filter over nw_X.columns to restore it. + matched = set(nw_X.select(_NW_SELECTOR).columns) + candidates = [column for column in nw_X.columns if column in matched] + + if exclude_datetime is True: + candidates = [ + column + for column in candidates + if _is_categorical_and_is_not_datetime(nw_X.get_column(column)) + ] + return candidates def find_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, return_empty: bool = False, ) -> List[Union[str, int]]: """ @@ -25,8 +59,9 @@ def find_numerical_variables( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The dataset. + X : dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. return_empty : bool, default=False Whether to return an empty list when no numerical variables are found. @@ -50,13 +85,18 @@ def find_numerical_variables( >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> var_ = find_numerical_variables(X) >>> var_ ['var_num'] """ - variables = list(X.select_dtypes(include="number").columns) + if nwd.is_pandas_dataframe(X) is True: + variables = list(X.select_dtypes(include="number").columns) + else: + nw_X = nw.from_native(X, eager_only=True) + variables = nw_X.select(nw.selectors.numeric()).columns + if len(variables) == 0: if return_empty is False: raise TypeError( @@ -73,8 +113,9 @@ def find_numerical_variables( def find_categorical_variables( - X: pd.DataFrame, + X: IntoDataFrame, return_empty: bool = False, + exclude_datetime: bool = True, ) -> List[Union[str, int]]: """ Returns a list with the names of all the categorical variables in a dataframe. @@ -85,8 +126,9 @@ def find_categorical_variables( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The dataset. + X : dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. return_empty : bool, default=False Whether to return an empty list when no categorical variables are found. @@ -98,6 +140,9 @@ def find_categorical_variables( warning, explicitly set `return_empty=False` instead of relying on the default. + exclude_datetime: bool, default=True + Whether to exclude variables that can be parsed as datetime. + Returns ------- variables: List @@ -110,17 +155,14 @@ def find_categorical_variables( >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> var_ = find_categorical_variables(X) >>> var_ ['var_cat'] """ - variables = [ - column - for column in X.select_dtypes(include=["O", "category", "string"]).columns - if _is_categorical_and_is_not_datetime(X[column]) - ] + variables = _find_nw_categoricals(X, exclude_datetime=exclude_datetime) + if len(variables) == 0: if return_empty is False: raise TypeError( @@ -138,7 +180,7 @@ def find_categorical_variables( def find_datetime_variables( - X: pd.DataFrame, + X: IntoDataFrame, return_empty: bool = False, ) -> List[Union[str, int]]: """ @@ -152,8 +194,9 @@ def find_datetime_variables( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The dataset. + X : dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. return_empty : bool, default=False Whether to return an empty list when no datetime variables are found. @@ -170,6 +213,13 @@ def find_datetime_variables( variables: List The names of the datetime variables. + Notes + ----- + String columns are parsed with flexible, dateutil-backed date guessing, so + formats like "01-Jan-2010" or "10/11/12" are recognised, in addition to + ISO-8601 strings and native `Date`/`Datetime` columns, regardless of the + dataframe library backing `X`. + Examples -------- >>> import pandas as pd @@ -177,18 +227,30 @@ def find_datetime_variables( >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> var_date = find_datetime_variables(X) >>> var_date ['var_date'] """ + if nwd.is_pandas_dataframe(X) is True: + non_numeric = X.select_dtypes(exclude="number").columns + datetime_cols = set(X.select_dtypes(include=["datetime", "datetimetz"]).columns) + nw_X = nw.from_native(X, eager_only=True) + else: + nw_X = nw.from_native(X, eager_only=True) + non_numeric = nw_X.select(~nw.selectors.numeric()).columns + datetime_cols = set( + nw_X.select(nw.selectors.by_dtype(nw.Date, nw.Datetime)).columns + ) variables = [ column - for column in X.select_dtypes(exclude="number").columns - if is_datetime(X[column]) or _is_categorical_and_is_datetime(X[column]) + for column in non_numeric + if column in datetime_cols + or _is_categorical_and_is_datetime(nw_X.get_column(column)) ] + if len(variables) == 0: if return_empty is False: raise TypeError( @@ -205,7 +267,7 @@ def find_datetime_variables( def find_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, exclude_datetime: bool = False, return_empty: bool = False, ) -> List[Union[str, int]]: @@ -217,8 +279,9 @@ def find_all_variables( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The dataset. + X : dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. exclude_datetime: bool, default=False Whether to exclude datetime variables. @@ -245,21 +308,40 @@ def find_all_variables( >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> vars_all = find_all_variables(X) >>> vars_all ['var_num', 'var_cat', 'var_date'] """ - if exclude_datetime is True: - variables = X.select_dtypes(exclude=DATETIME_TYPES).columns.to_list() - variables = [ - var - for var in variables - if is_numeric(X[var]) or not _is_categorical_and_is_datetime(X[var]) - ] + if nwd.is_pandas_dataframe(X) is True: + if exclude_datetime is True: + variables = X.select_dtypes(exclude=["datetime", "datetimetz"]).columns + numeric_cols = set(X.select_dtypes(include="number").columns) + nw_X = nw.from_native(X, eager_only=True) + variables = [ + var + for var in variables + if var in numeric_cols + or not _is_categorical_and_is_datetime(nw_X.get_column(var)) + ] + else: + variables = list(X.columns) else: - variables = X.columns.to_list() + nw_X = nw.from_native(X, eager_only=True) + if exclude_datetime is True: + variables = nw_X.select( + ~nw.selectors.by_dtype(nw.Date, nw.Datetime) + ).columns + numeric_cols = set(nw_X.select(nw.selectors.numeric()).columns) + variables = [ + var + for var in variables + if var in numeric_cols + or not _is_categorical_and_is_datetime(nw_X.get_column(var)) + ] + else: + variables = nw_X.columns if len(variables) == 0: if return_empty is False: @@ -276,9 +358,10 @@ def find_all_variables( def find_categorical_and_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Union[None, int, str, List[Union[str, int]]] = None, return_empty: bool = False, + exclude_datetime: bool = True, ) -> Tuple[List[Union[str, int]], List[Union[str, int]]]: """ Find numerical and categorical variables in a dataframe or from a list. @@ -290,8 +373,9 @@ def find_categorical_and_numerical_variables( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] - The dataset. + X : dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. variables : list, default=None If `None`, the function finds all categorical and numerical variables in X. @@ -308,6 +392,9 @@ def find_categorical_and_numerical_variables( warning, explicitly set `return_empty=False` instead of relying on the default. + exclude_datetime: bool, default=True + Whether to exclude variables that can be parsed as datetime. + Returns ------- variables: tuple @@ -323,21 +410,28 @@ def find_categorical_and_numerical_variables( >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> var_cat, var_num = find_categorical_and_numerical_variables(X) >>> var_cat, var_num (['var_cat'], ['var_num']) """ + nw_X = nw.from_native(X, eager_only=True) # If the user passes just 1 variable outside a list. if isinstance(variables, (str, int)): - if X[variables].dtype.name == "category" or _is_categorical_and_is_not_datetime( - X[variables] - ): + s = nw_X.get_column(variables) + is_cat = bool( + _find_nw_categoricals( + X, variables=[variables], exclude_datetime=exclude_datetime + ) + ) + is_num = s.dtype.is_numeric() + + if is_cat: variables_cat = [variables] variables_num = [] - elif is_numeric(X[variables]): + elif is_num: variables_num = [variables] variables_cat = [] else: @@ -358,12 +452,11 @@ def find_categorical_and_numerical_variables( # If user leaves default None parameter. elif variables is None: - variables_cat = [ - column - for column in X.select_dtypes(include=["O", "category", "string"]).columns - if _is_categorical_and_is_not_datetime(X[column]) - ] - variables_num = list(X.select_dtypes(include="number").columns) + variables_cat = _find_nw_categoricals(X, exclude_datetime=exclude_datetime) + if nwd.is_pandas_dataframe(X) is True: + variables_num = list(X.select_dtypes(include="number").columns) + else: + variables_num = nw_X.select(nw.selectors.numeric()).columns if len(variables_num) == 0 and len(variables_cat) == 0: if return_empty is False: @@ -399,15 +492,15 @@ def find_categorical_and_numerical_variables( variables_num = [] else: - # find categorical variables - variables_cat = [ - column - for column in X[variables] - .select_dtypes(include=["O", "category", "string"]) - .columns - if _is_categorical_and_is_not_datetime(X[column]) - ] - # find numerical variables - variables_num = list(X[variables].select_dtypes(include="number").columns) + variables_cat = _find_nw_categoricals( + X, variables=variables, exclude_datetime=exclude_datetime + ) + if nwd.is_pandas_dataframe(X) is True: + variables_num = list( + X[variables].select_dtypes(include="number").columns + ) + else: + sub_X = nw_X.select(variables) + variables_num = sub_X.select(nw.selectors.numeric()).columns return variables_cat, variables_num diff --git a/feature_engine/variable_handling/retain_variables.py b/feature_engine/variable_handling/retain_variables.py index 2a161d066..fa3b947ff 100644 --- a/feature_engine/variable_handling/retain_variables.py +++ b/feature_engine/variable_handling/retain_variables.py @@ -2,18 +2,23 @@ from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame + Variables = Union[int, str, List[Union[str, int]]] -def retain_variables_if_in_df(X, variables): +def retain_variables_if_in_df(X: IntoDataFrame, variables): """Returns the subset of variables in the list that are present in the dataframe. More details in the :ref:`User Guide `. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The dataset. + X: dataframe of shape = [n_samples, n_features] + The dataset. Can be a pandas, polars, or any other dataframe supported by + narwhals. variables: string, int or list of strings or int. The names of the variables to check. @@ -30,7 +35,7 @@ def retain_variables_if_in_df(X, variables): >>> X = pd.DataFrame({ >>> "var_num": [1, 2, 3], >>> "var_cat": ["A", "B", "C"], - >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="T") + >>> "var_date": pd.date_range("2020-02-24", periods=3, freq="min") >>> }) >>> vars_in_df = retain_variables_if_in_df(X, ['var_num', 'var_cat', 'var_other']) >>> vars_in_df @@ -39,7 +44,11 @@ def retain_variables_if_in_df(X, variables): if isinstance(variables, (str, int)): variables = [variables] - variables_in_df = [var for var in variables if var in X.columns] + if nwd.is_pandas_dataframe(X) is True: + columns = set(X.columns) + else: + columns = set(nw.from_native(X, eager_only=True).columns) + variables_in_df = [var for var in variables if var in columns] # Raise an error if no column is left to work with. if len(variables_in_df) == 0: diff --git a/pyproject.toml b/pyproject.toml index 9e5fb0199..64dbb8156 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,18 +10,17 @@ license = {text = "BSD 3 clause"} authors = [ { name = "Soledad Galli", email = "solegalli@protonmail.com" } ] -requires-python = ">=3.9.0" +requires-python = ">=3.11.0" dependencies = [ "numpy>=1.18.2", - "pandas>=2.2.0", - "scikit-learn>=1.4.0", + "scikit-learn>=1.7.0", "scipy>=1.4.1", + "narwhals>=2.0.0", + "python-dateutil>=2.8.2", ] classifiers = [ "License :: OSI Approved :: BSD License", - "Programming Language :: Python :: 3.9", - "Programming Language :: Python :: 3.10", "Programming Language :: Python :: 3.11", "Programming Language :: Python :: 3.12", "Programming Language :: Python :: 3.13", @@ -39,10 +38,13 @@ docs = [ "pydata_sphinx_theme>=0.7.2", "sphinx_autodoc_typehints>=1.11.1,<=1.21.3", "numpydoc>=0.9.2", + "pandas>=2.2.0", ] tests = [ "pytest>=5.4.1", + "pandas>=2.2.0", + "polars>=1.0.0", # repo maintenance tooling "black>=21.5b1", diff --git a/requirements.txt b/requirements.txt deleted file mode 100644 index 6df436a29..000000000 --- a/requirements.txt +++ /dev/null @@ -1,4 +0,0 @@ -numpy>=1.18.2 -pandas>=2.2.0 -scikit-learn>=1.4.0 -scipy>=1.4.1 diff --git a/tests/backend_helpers.py b/tests/backend_helpers.py new file mode 100644 index 000000000..44032496f --- /dev/null +++ b/tests/backend_helpers.py @@ -0,0 +1,38 @@ +"""Helpers for tests that run on several dataframe backends. +Use them together with the `make_df` fixture in `tests/conftest.py`. +""" + +import narwhals as nw +import pandas as pd +import polars as pl + + +def frame_to_dict(X): + """Return the dataframe contents as ``{column: list of values}``. + + pandas represents missing values as NaN and polars as None, so NaN (and + pd.NA) are normalised to None and the same expected values work for both + backends. + """ + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + return { + col: [none_if_missing(v) for v in values] for col, values in result.items() + } + + +def null_count(X, col): + """Return the number of missing values in column ``col``.""" + return nw.from_native(X, eager_only=True).get_column(col).null_count() + + +def make_series(make_df, values, name=None): + """Build a Series on the same backend as ``make_df``.""" + if make_df is pd.DataFrame: + return pd.Series(values, name=name) + return pl.Series(name=name or "", values=values) + + +def none_if_missing(value): + if value is pd.NA or (isinstance(value, float) and value != value): + return None + return value diff --git a/tests/check_estimators_with_parametrize_tests.py b/tests/check_estimators_with_parametrize_tests.py deleted file mode 100644 index 039bd50c2..000000000 --- a/tests/check_estimators_with_parametrize_tests.py +++ /dev/null @@ -1,200 +0,0 @@ -""" -This file is only intended to help understand check_estimator tests on Feature-engine -transformers. It is not run as part of the battery of acceptance tests. Works up to -sklearn < 1.6. -""" - -from sklearn.impute import SimpleImputer -from sklearn.linear_model import LogisticRegression -from sklearn.utils.estimator_checks import parametrize_with_checks - -from feature_engine.creation import ( - CyclicalFeatures, - DecisionTreeFeatures, - MathFeatures, - RelativeFeatures, -) -from feature_engine.encoding import ( - CountEncoder, - DecisionTreeEncoder, - MeanEncoder, - OneHotEncoder, - OrdinalEncoder, - RareLabelEncoder, - StringSimilarityEncoder, - WoEEncoder, -) -from feature_engine.imputation import ( - AddMissingIndicator, - ArbitraryImputer, - CategoricalImputer, - DropMissingData, - EndTailImputer, - MeanImputer, - RandomSampleImputer, -) -from feature_engine.outliers import ArbitraryOutlierCapper, OutlierTrimmer, Winsoriser -from feature_engine.selection import ( - MRMR, - DropConstantFeatures, - DropCorrelatedFeatures, - DropDuplicateFeatures, - DropFeatures, - DropHighPSIFeatures, - ProbeFeatureSelection, - RecursiveFeatureAddition, - RecursiveFeatureElimination, - SelectByInformationValue, - SelectByShuffling, - SelectBySingleFeaturePerformance, - SelectByTargetEncoding, - SmartCorrelatedSelection, -) -from feature_engine.timeseries.forecasting import ( - ExpandingWindowFeatures, - LagFeatures, - WindowFeatures, -) -from feature_engine.transformation import ( - ArcsinTransformer, - BoxCoxTransformer, - LogTransformer, - PowerTransformer, - ReciprocalTransformer, - YeoJohnsonTransformer, -) -from feature_engine.wrappers import SklearnWrapper - - -# creation -@parametrize_with_checks( - [ - DecisionTreeFeatures(regression=False), - CyclicalFeatures(), - MathFeatures(variables=["x0", "x1"], func="mean", missing_values="ignore"), - RelativeFeatures( - variables=["x0", "x1"], - reference=["x0"], - func=["add"], - missing_values="ignore", - ), - ] -) -def test_sklearn_compatible_creator(estimator, check): - check(estimator) - - -# imputation -@parametrize_with_checks( - [ - MeanImputer(), - ArbitraryImputer(), - CategoricalImputer(fill_value=0, ignore_format=True), - EndTailImputer(), - AddMissingIndicator(), - RandomSampleImputer(), - DropMissingData(), - ] -) -def test_sklearn_compatible_imputer(estimator, check): - check(estimator) - - -# encoding -@parametrize_with_checks( - [ - CountEncoder(ignore_format=True), - DecisionTreeEncoder(regression=False, ignore_format=True), - MeanEncoder(ignore_format=True), - OneHotEncoder(ignore_format=True), - OrdinalEncoder(ignore_format=True), - RareLabelEncoder( - tol=0.00000000001, - n_categories=100000000000, - replace_with=10, - ignore_format=True, - ), - WoEEncoder(ignore_format=True), - StringSimilarityEncoder(ignore_format=True), - ] -) -def test_sklearn_compatible_encoder(estimator, check): - check(estimator) - - -# outliers -@parametrize_with_checks( - [ - ArbitraryOutlierCapper(max_capping_dict={"x0": 10}), - OutlierTrimmer(), - Winsoriser(), - ] -) -def test_sklearn_compatible_outliers(estimator, check): - check(estimator) - - -# transformers -@parametrize_with_checks( - [ - ArcsinTransformer(), - BoxCoxTransformer(), - LogTransformer(), - PowerTransformer(), - ReciprocalTransformer(), - YeoJohnsonTransformer(), - ] -) -def test_sklearn_compatible_transformer(estimator, check): - check(estimator) - - -# selectors -@parametrize_with_checks( - [ - DropFeatures(features_to_drop=["x0"]), - DropConstantFeatures(missing_values="ignore"), - DropDuplicateFeatures(), - DropCorrelatedFeatures(), - SmartCorrelatedSelection(), - DropHighPSIFeatures(bins=5), - SelectByShuffling( - LogisticRegression(max_iter=2, random_state=1), scoring="accuracy" - ), - SelectBySingleFeaturePerformance( - LogisticRegression(max_iter=2, random_state=1), scoring="accuracy" - ), - RecursiveFeatureAddition( - LogisticRegression(max_iter=2, random_state=1), scoring="accuracy" - ), - RecursiveFeatureElimination( - LogisticRegression(max_iter=2, random_state=1), - scoring="accuracy", - threshold=-100, - ), - SelectByTargetEncoding(scoring="roc_auc", bins=3, regression=False), - SelectByInformationValue(), - MRMR(), - ProbeFeatureSelection(estimator=LogisticRegression()), - ] -) -def test_sklearn_compatible_selectors(estimator, check): - check(estimator) - - -# wrappers -@parametrize_with_checks([SklearnWrapper(SimpleImputer())]) -def test_sklearn_compatible_wrapper(estimator, check): - check(estimator) - - -# test_forecasting -@parametrize_with_checks( - [ - LagFeatures(missing_values="ignore"), - WindowFeatures(missing_values="ignore"), - ExpandingWindowFeatures(missing_values="ignore"), - ] -) -def test_sklearn_compatible_forecasters(estimator, check): - check(estimator) diff --git a/tests/conftest.py b/tests/conftest.py index 721b8b5f3..88fe8f3d5 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,8 +1,19 @@ import numpy as np import pandas as pd +import polars as pl import pytest +@pytest.fixture(params=[pd.DataFrame, pl.DataFrame], ids=["pandas", "polars"]) +def make_df(request): + """Dataframe constructor of the backend under test: pandas or polars. + + A test that requests this fixture runs once per backend. Build the input + with ``make_df(data)`` and check the output with ``isinstance(X, make_df)``. + """ + return request.param + + @pytest.fixture(scope="module") def df_vartypes(): data = { diff --git a/tests/test_base_transformers/test_base_numerical_transformer.py b/tests/test_base_transformers/test_base_numerical_transformer.py index 4934e4b57..3d60cf41d 100644 --- a/tests/test_base_transformers/test_base_numerical_transformer.py +++ b/tests/test_base_transformers/test_base_numerical_transformer.py @@ -1,76 +1,139 @@ +import re + +import narwhals as nw +import pandas as pd import pytest -from numpy import inf -from pandas.testing import assert_frame_equal from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer +from tests.backend_helpers import frame_to_dict from tests.estimator_checks.non_fitted_error_checks import check_raises_non_fitted_error +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) +MSG_INF = ( + "Some of the variables to transform contain inf values. Check and " + "remove those before using this transformer." +) + class MockClass(BaseNumericalTransformer): - def __init__(self): - self.variables = None - self.return_empty = False + def __init__(self, variables=None, return_empty=False): + self.variables = variables + self.return_empty = return_empty def fit(self, X): - X, variables_ = self._fit_setup(X) + _, variables_ = self._fit_setup(X) self.variables_ = variables_ self._get_feature_names_in(X) - return X + return self def transform(self, X): - return self._check_transform_input_and_state(X) + return self._check_transform_input_and_state(X).to_native() -def test_empty_find_numerical_variables(df_vartypes): - transformer = MockClass() - with pytest.raises(TypeError): - transformer.fit(df_vartypes.drop(columns=["Age", "Marks"])) - transformer = MockClass() - transformer.return_empty = True - transformer.fit(df_vartypes.drop(columns=["Age", "Marks"])) - assert transformer.variables_ == [] +# fit and transform +def test_fit_setup_returns_narwhals_frame_and_numerical_variables(make_df): + nw_X, variables_ = MockClass()._fit_setup(make_df(DATA)) + assert isinstance(nw_X, nw.DataFrame) + assert isinstance(nw_X.to_native(), make_df) + assert frame_to_dict(nw_X.to_native()) == DATA + assert variables_ == ["Age", "Marks"] -def test_fit_method(df_vartypes, df_na): - transformer = MockClass() - res = transformer.fit(df_vartypes) - assert transformer.feature_names_in_ == list(df_vartypes.columns) - assert transformer.n_features_in_ == len(df_vartypes.columns) - assert_frame_equal(res, df_vartypes) - with pytest.raises(ValueError): - transformer.fit(df_na) +def test_fit_setup_checks_user_variables(make_df): + _, variables_ = MockClass(variables=["Marks"])._fit_setup(make_df(DATA)) + assert variables_ == ["Marks"] - df_na = df_na.fillna(inf) - with pytest.raises(ValueError): - assert transformer.fit(df_na) + msg = ( + "Some of the variables are not numerical. Please cast them as numerical " + "before using this transformer." + ) + with pytest.raises(TypeError, match=re.escape(msg)): + MockClass(variables=["Name"])._fit_setup(make_df(DATA)) -def test_transform_method(df_vartypes, df_na): - transformer = MockClass() - transformer.fit(df_vartypes) - assert_frame_equal( - transformer._check_transform_input_and_state(df_vartypes), df_vartypes - ) - assert_frame_equal( - transformer._check_transform_input_and_state( - df_vartypes[["City", "Age", "Name", "Marks", "dob"]] - ), - df_vartypes, +def test_fit_setup_when_there_are_no_numerical_variables(make_df): + X = make_df({"Name": DATA["Name"], "City": DATA["City"]}) + msg = ( + "No numerical variables found in this dataframe. Check variable dtypes or " + "set return_empty to True to return an empty list instead." ) + with pytest.raises(TypeError, match=re.escape(msg)): + MockClass()._fit_setup(X) + + msg = "No numerical variables found in this dataframe. Returning an empty list." + with pytest.warns(UserWarning, match=re.escape(msg)): + _, variables_ = MockClass(return_empty=True)._fit_setup(X) + assert variables_ == [] + + +@pytest.mark.parametrize("value, msg", [(None, MSG_NA), (float("inf"), MSG_INF)]) +def test_fit_setup_raises_error_if_na_or_inf(make_df, value, msg): + X = make_df({**DATA, "Marks": [0.9, value, 0.7, 0.6]}) + with pytest.raises(ValueError, match=re.escape(msg)): + MockClass()._fit_setup(X) + + +def test_get_feature_names_in(make_df): + transformer = MockClass().fit(make_df(DATA)) + assert transformer.feature_names_in_ == list(DATA) + assert transformer.n_features_in_ == 4 - with pytest.raises(ValueError): - transformer.fit(df_na) - df_na = df_na.fillna(inf) - with pytest.raises(ValueError): - assert transformer.fit(df_na) +def test_check_transform_input_and_state_returns_narwhals_frame(make_df): + transformer = MockClass().fit(make_df(DATA)) + reordered = make_df({k: DATA[k] for k in ["Marks", "City", "Age", "Name"]}) - with pytest.raises(ValueError): - assert transformer._check_transform_input_and_state( - df_vartypes[["Age", "Marks"]] + nw_X = transformer._check_transform_input_and_state(reordered) + + assert isinstance(nw_X, nw.DataFrame) + assert isinstance(nw_X.to_native(), make_df) + # the columns are returned in the order seen in fit + assert nw_X.columns == list(DATA) + assert frame_to_dict(nw_X.to_native()) == DATA + + +def test_check_transform_input_and_state_with_integer_column_names(): + # integer column names are pandas-only + X = pd.DataFrame({0: [1.0, 2.0], 1: ["a", "b"], 2: [3, 4]}) + transformer = MockClass().fit(X) + + nw_X = transformer._check_transform_input_and_state(X[[2, 0, 1]]) + + pd.testing.assert_frame_equal(nw_X.to_native(), X) + + +def test_check_transform_input_and_state_raises_error_if_columns_differ(make_df): + transformer = MockClass().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer._check_transform_input_and_state( + make_df({k: DATA[k] for k in ["Age", "Marks"]}) ) +@pytest.mark.parametrize("value, msg", [(None, MSG_NA), (float("inf"), MSG_INF)]) +def test_check_transform_input_and_state_raises_error_if_na_or_inf( + make_df, value, msg +): + transformer = MockClass().fit(make_df(DATA)) + X = make_df({**DATA, "Marks": [0.9, value, 0.7, 0.6]}) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer._check_transform_input_and_state(X) + + def test_raises_non_fitted_error(): check_raises_non_fitted_error(MockClass()) diff --git a/tests/test_base_transformers/test_get_feature_names_out_mixin.py b/tests/test_base_transformers/test_get_feature_names_out_mixin.py index e4b67ed33..2cfb83c91 100644 --- a/tests/test_base_transformers/test_get_feature_names_out_mixin.py +++ b/tests/test_base_transformers/test_get_feature_names_out_mixin.py @@ -1,4 +1,8 @@ +import re + import numpy as np +import pandas as pd +import polars as pl import pytest from sklearn.base import BaseEstimator from sklearn.exceptions import NotFittedError @@ -9,9 +13,14 @@ from feature_engine._base_transformers.mixins import GetFeatureNamesOutMixin from feature_engine.dataframe_checks import check_X -variables_str = ["Name", "City", "Age", "Marks", "dob"] -variables_arr = ["x0", "x1", "x2", "x3", "x4"] -variables_user = ["Dog", "Cat", "Bird", "Frog", "Duck"] +VARTYPES_DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "dob": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} +variables_str = list(VARTYPES_DATA.keys()) class MockTransformer(BaseEstimator, GetFeatureNamesOutMixin): @@ -25,40 +34,45 @@ def transform(self, X): return X.copy() -def test_non_fitted_error(df_vartypes): +def test_non_fitted_error(): transformer = MockTransformer() with pytest.raises(NotFittedError): - transformer.get_feature_names_out(df_vartypes) + transformer.get_feature_names_out() # ======== Tests for transformers that do not add new features to the data ======== -def test_when_input_is_pandas_columns(df_vartypes): - input_features = df_vartypes.columns +def test_when_input_is_pandas_columns(): + df = pd.DataFrame(VARTYPES_DATA) transformer = MockTransformer() - - transformer.fit(df_vartypes) + transformer.fit(df) assert ( - transformer.get_feature_names_out(input_features=input_features) - == variables_str + transformer.get_feature_names_out(input_features=df.columns) == variables_str ) - transformer.fit(df_vartypes.to_numpy()) + +def test_when_input_is_polars_columns(): + # polars' .columns is already a plain list, so this exercises the + # `isinstance(input_features, list)` branch, not `nwd.is_pandas_index`. + df = pl.DataFrame(VARTYPES_DATA) + transformer = MockTransformer() + transformer.fit(df) assert ( - transformer.get_feature_names_out(input_features=input_features) - == variables_str + transformer.get_feature_names_out(input_features=df.columns) == variables_str ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "input_features", [None, variables_str, np.array(variables_str)] ) -def test_with_df(df_vartypes, input_features): +def test_with_df(make_df, input_features): # When the data used to train the class is a dataframe, the variable names are # stored in feature_names_in_. Those should be returned by get_feature_names_out() + df = make_df(VARTYPES_DATA) transformer = MockTransformer() - transformer.fit(df_vartypes) + transformer.fit(df) assert ( transformer.get_feature_names_out(input_features=input_features) == transformer.feature_names_in_ @@ -67,48 +81,16 @@ def test_with_df(df_vartypes, input_features): transformer.get_feature_names_out(input_features=input_features) == variables_str ) - assert ( - transformer.get_feature_names_out(input_features=df_vartypes.columns) - == variables_str - ) - - -@pytest.mark.parametrize( - "input_features", - [ - None, - variables_arr, - np.array(variables_arr), - variables_str, - np.array(variables_str), - variables_user, - ], -) -def test_with_array(df_vartypes, input_features): - # When the data used to train the class is a numpy array, the names stored in - # feature_names_in_ are x0, x1, etc. Those should be returned by - # get_feature_names_out() when input_features is None. Alternatively, it returns - # a list of the variables entered by the user. - transformer = MockTransformer() - transformer.fit(df_vartypes.to_numpy()) - - if input_features is None: - assert ( - transformer.get_feature_names_out(input_features=input_features) - == variables_arr - ) - else: - assert transformer.get_feature_names_out(input_features=input_features) == list( - input_features - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "input_features", [None, variables_str, np.array(variables_str)] ) -def test_with_pipeline_and_df(df_vartypes, input_features): +def test_with_pipeline_and_df(make_df, input_features): + df = make_df(VARTYPES_DATA) pipe = Pipeline([("transformer", MockTransformer())]) - pipe.fit(df_vartypes) + pipe.fit(df) assert ( pipe.get_feature_names_out(input_features=input_features) == pipe.named_steps["transformer"].feature_names_in_ @@ -116,113 +98,35 @@ def test_with_pipeline_and_df(df_vartypes, input_features): assert pipe.get_feature_names_out(input_features=input_features) == variables_str -@pytest.mark.parametrize( - "input_features", - [ - None, - variables_arr, - np.array(variables_arr), - variables_str, - np.array(variables_str), - variables_user, - ], -) -def test_with_pipeline_and_array(df_vartypes, input_features): - pipe = Pipeline([("transformer", MockTransformer())]) - pipe.fit(df_vartypes.to_numpy()) - - if input_features is None: - assert ( - pipe.get_feature_names_out(input_features=input_features) == variables_arr - ) - else: - assert pipe.get_feature_names_out(input_features=input_features) == list( - input_features - ) - - @pytest.mark.parametrize( "input_features", [None, variables_str, np.array(variables_str)] ) -def test_with_pipe_and_skl_transformer_input_df(df_vartypes, input_features): +def test_with_pipe_and_skl_transformer_input_df(input_features): + # SimpleImputer outputs a numpy array by default, which check_X now + # rejects, so it must be configured to output a dataframe. + df = pd.DataFrame(VARTYPES_DATA) pipe = Pipeline( [ - ("imputer", SimpleImputer(strategy="constant")), + ( + "imputer", + SimpleImputer(strategy="constant").set_output(transform="pandas"), + ), ("transformer", MockTransformer()), ] ) - pipe.fit(df_vartypes) + pipe.fit(df) assert pipe.get_feature_names_out(input_features=input_features) == variables_str -@pytest.mark.parametrize( - "input_features", - [ - None, - variables_arr, - np.array(variables_arr), - variables_str, - np.array(variables_str), - variables_user, - ], -) -def test_with_pipe_and_skl_transformer_input_array(df_vartypes, input_features): +def test_pipe_with_skl_transformer_that_adds_features(): + df = pd.DataFrame({"Age": VARTYPES_DATA["Age"], "Marks": VARTYPES_DATA["Marks"]}) pipe = Pipeline( [ - ("imputer", SimpleImputer(strategy="constant")), + ("poly", PolynomialFeatures().set_output(transform="pandas")), ("transformer", MockTransformer()), ] ) - pipe.fit(df_vartypes.to_numpy()) - - if input_features is None: - assert ( - pipe.get_feature_names_out(input_features=input_features) == variables_arr - ) - else: - assert pipe.get_feature_names_out(input_features=input_features) == list( - input_features - ) - - -def test_pipe_with_skl_transformer_that_adds_features(df_vartypes): - pipe = Pipeline( - [ - ("poly", PolynomialFeatures()), - ("transformer", MockTransformer()), - ] - ) - - # when input is array - pipe.fit(df_vartypes[["Age", "Marks"]].to_numpy()) - assert pipe.get_feature_names_out(input_features=None) == [ - "1", - "x0", - "x1", - "x0^2", - "x0 x1", - "x1^2", - ] - - assert pipe.get_feature_names_out(input_features=["Age", "Marks"]) == [ - "1", - "Age", - "Marks", - "Age^2", - "Age Marks", - "Marks^2", - ] - assert pipe.get_feature_names_out(input_features=["Dog", "Cat"]) == [ - "1", - "Dog", - "Cat", - "Dog^2", - "Dog Cat", - "Cat^2", - ] - - # when input is df - pipe.fit(df_vartypes[["Age", "Marks"]]) + pipe.fit(df) assert pipe.get_feature_names_out(input_features=None) == [ "1", "Age", @@ -242,32 +146,24 @@ def test_pipe_with_skl_transformer_that_adds_features(df_vartypes): ] -def test_raise_error_when_input_feature_non_permitted(df_vartypes): +def test_raise_error_when_input_feature_non_permitted(): + df = pd.DataFrame(VARTYPES_DATA) transformer = MockTransformer() + transformer.fit(df) - # when input is dataframe - transformer.fit(df_vartypes) - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match="feature_names_in_"): transformer.get_feature_names_out(input_features=["Name"]) - assert "feature_names_in_" in str(record) - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match="feature_names_in_"): transformer.get_feature_names_out(input_features=np.array(["Name", "Age"])) - assert "feature_names_in_" in str(record) - with pytest.raises(ValueError) as record: + msg = "input_features must be a list or an array. Got var1 instead." + with pytest.raises(ValueError, match=re.escape(msg)): transformer.get_feature_names_out(input_features="var1") - assert "list or an array" in str(record) - with pytest.raises(ValueError) as record: + msg = "input_features must be a list or an array. Got True instead." + with pytest.raises(ValueError, match=re.escape(msg)): transformer.get_feature_names_out(input_features=True) - assert "list or an array" in str(record) - - # when input is array - transformer.fit(df_vartypes.to_numpy()) - with pytest.raises(ValueError) as record: - transformer.get_feature_names_out(input_features=["Name", "Age"]) - assert "number of input_features does not match" in str(record) # ================ Tests for transformers that add features to the data ======= @@ -292,21 +188,23 @@ def _get_new_features_name(self): return [f"{i}_plus" for i in self.variables_] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("features_in", [["Age", "Marks"], ["Name", "dob"]]) @pytest.mark.parametrize( "input_features", [None, variables_str, np.array(variables_str)] ) -def test_new_feature_names_with_df(df_vartypes, features_in, input_features): +def test_new_feature_names_with_df(make_df, features_in, input_features): + df = make_df(VARTYPES_DATA) transformer = MockCreator(variables=features_in, drop_original=False) - transformer.fit(df_vartypes) - features_out = list(df_vartypes.columns) + [f"{i}_plus" for i in features_in] + transformer.fit(df) + features_out = variables_str + [f"{i}_plus" for i in features_in] assert ( transformer.get_feature_names_out(input_features=input_features) == features_out ) transformer = MockCreator(variables=features_in, drop_original=True) - transformer.fit(df_vartypes) - features_out = [f for f in df_vartypes.columns if f not in features_in] + [ + transformer.fit(df) + features_out = [f for f in variables_str if f not in features_in] + [ f"{i}_plus" for i in features_in ] assert ( @@ -314,18 +212,20 @@ def test_new_feature_names_with_df(df_vartypes, features_in, input_features): ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("features_in", [["Age", "Marks"], ["Name", "dob"]]) @pytest.mark.parametrize( "input_features", [None, variables_str, np.array(variables_str)] ) -def test_new_feature_names_within_pipeline(df_vartypes, features_in, input_features): +def test_new_feature_names_within_pipeline(make_df, features_in, input_features): + df = make_df(VARTYPES_DATA) transformer = Pipeline( [ ("transformer", MockCreator(variables=features_in, drop_original=False)), ] ) - transformer.fit(df_vartypes) - features_out = list(df_vartypes.columns) + [f"{i}_plus" for i in features_in] + transformer.fit(df) + features_out = variables_str + [f"{i}_plus" for i in features_in] assert ( transformer.get_feature_names_out(input_features=input_features) == features_out ) @@ -335,8 +235,8 @@ def test_new_feature_names_within_pipeline(df_vartypes, features_in, input_featu ("transformer", MockCreator(variables=features_in, drop_original=True)), ] ) - transformer.fit(df_vartypes) - features_out = [f for f in df_vartypes.columns if f not in features_in] + [ + transformer.fit(df) + features_out = [f for f in variables_str if f not in features_in] + [ f"{i}_plus" for i in features_in ] assert ( @@ -349,25 +249,33 @@ def test_new_feature_names_within_pipeline(df_vartypes, features_in, input_featu "input_features", [None, variables_str, np.array(variables_str)] ) def test_new_feature_names_pipe_with_skl_transformer_and_df( - df_vartypes, features_in, input_features + features_in, input_features ): + df = pd.DataFrame(VARTYPES_DATA) pipe = Pipeline( [ - ("imputer", SimpleImputer(strategy="constant")), + ( + "imputer", + SimpleImputer(strategy="constant").set_output(transform="pandas"), + ), ("transformer", MockCreator(variables=features_in, drop_original=False)), ] ) - pipe.fit(df_vartypes) - features_out = list(df_vartypes.columns) + [f"{i}_plus" for i in features_in] + pipe.fit(df) + features_out = variables_str + [f"{i}_plus" for i in features_in] assert pipe.get_feature_names_out(input_features=input_features) == features_out + pipe = Pipeline( [ - ("imputer", SimpleImputer(strategy="constant")), + ( + "imputer", + SimpleImputer(strategy="constant").set_output(transform="pandas"), + ), ("transformer", MockCreator(variables=features_in, drop_original=True)), ] ) - pipe.fit(df_vartypes) - features_out = [f for f in df_vartypes.columns if f not in features_in] + [ + pipe.fit(df) + features_out = [f for f in variables_str if f not in features_in] + [ f"{i}_plus" for i in features_in ] assert pipe.get_feature_names_out(input_features=input_features) == features_out @@ -376,15 +284,13 @@ def test_new_feature_names_pipe_with_skl_transformer_and_df( @pytest.mark.parametrize( "input_features", [None, ["Age", "Marks"], np.array(["Age", "Marks"])] ) -def test_new_feature_names_pipe_and_skl_transformer_that_adds_features( - df_vartypes, input_features -): +def test_new_feature_names_pipe_and_skl_transformer_that_adds_features(input_features): features_in = ["Age", "Marks"] - df = df_vartypes[features_in].copy() + df = pd.DataFrame({"Age": VARTYPES_DATA["Age"], "Marks": VARTYPES_DATA["Marks"]}) pipe = Pipeline( [ - ("poly", PolynomialFeatures()), + ("poly", PolynomialFeatures().set_output(transform="pandas")), ("transformer", MockCreator(variables=features_in, drop_original=False)), ] ) @@ -419,41 +325,29 @@ def get_support(self, indices=False): return mask if not indices else np.where(mask)[0] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "input_features", [None, variables_str, np.array(variables_str)] ) -def test_remove_features_in_df(df_vartypes, input_features): - transformer = MockSelector() - transformer.fit(df_vartypes) - features_out = list(df_vartypes.columns)[2:] - assert ( - transformer.get_feature_names_out(input_features=input_features) == features_out - ) - - -@pytest.mark.parametrize( - "input_features", - [None, variables_arr, np.array(variables_arr), variables_str, variables_user], -) -def test_remove_features_in_array(df_vartypes, input_features): +def test_remove_features_in_df(make_df, input_features): + df = make_df(VARTYPES_DATA) transformer = MockSelector() - transformer.fit(df_vartypes.to_numpy()) - if input_features is None: - features_out = ["x2", "x3", "x4"] - else: - features_out = list(input_features)[2:] + transformer.fit(df) + features_out = variables_str[2:] assert ( transformer.get_feature_names_out(input_features=input_features) == features_out ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "input_features", [None, variables_str, np.array(variables_str)] ) -def test_remove_feature_names_within_pipeline_when_df(df_vartypes, input_features): +def test_remove_feature_names_within_pipeline_when_df(make_df, input_features): + df = make_df(VARTYPES_DATA) transformer = Pipeline([("transformer", MockSelector())]) - transformer.fit(df_vartypes) - features_out = list(df_vartypes.columns)[2:] + transformer.fit(df) + features_out = variables_str[2:] assert ( transformer.get_feature_names_out(input_features=input_features) == features_out ) @@ -462,73 +356,60 @@ def test_remove_feature_names_within_pipeline_when_df(df_vartypes, input_feature @pytest.mark.parametrize( "input_features", [None, variables_str, np.array(variables_str)] ) -def test_remove_feature_names_pipe_with_skl_transformer_and_df( - df_vartypes, input_features -): - df_vartypes = df_vartypes.drop(["dob"], axis=1) - if input_features is not None: - input_features = input_features[0:-1] +def test_remove_feature_names_pipe_with_skl_transformer_and_df(input_features): + df = pd.DataFrame( + {k: v for k, v in VARTYPES_DATA.items() if k != "dob"} + ) + variables_no_dob = [v for v in variables_str if v != "dob"] + trimmed_input_features = ( + input_features[0:-1] if input_features is not None else None + ) pipe = Pipeline( [ ("transformer", MockSelector()), - ("imputer", SimpleImputer(strategy="constant")), + ( + "imputer", + SimpleImputer(strategy="constant").set_output(transform="pandas"), + ), ] ) - pipe.fit(df_vartypes) - features_out = list(df_vartypes.columns)[2:] + pipe.fit(df) + features_out = variables_no_dob[2:] + # sklearn's Pipeline.get_feature_names_out() returns a numpy array here + # when the feature-removing transformer isn't the last step. assert all( - pipe.get_feature_names_out(input_features=input_features) == features_out + pipe.get_feature_names_out(input_features=trimmed_input_features) + == features_out ) pipe = Pipeline( [ - ("imputer", SimpleImputer(strategy="constant")), + ( + "imputer", + SimpleImputer(strategy="constant").set_output(transform="pandas"), + ), ("transformer", MockSelector()), ] ) - pipe.fit(df_vartypes) - features_out = list(df_vartypes.columns)[2:] - assert pipe.get_feature_names_out(input_features=input_features) == features_out - - -@pytest.mark.parametrize( - "input_features", [None, variables_str, variables_arr, variables_user] -) -def test_new_feature_names_pipe_with_skl_transformer_and_array( - df_vartypes, input_features -): - df_vartypes = df_vartypes.drop(["dob"], axis=1) - - pipe = Pipeline( - [ - ("imputer", SimpleImputer(strategy="constant")), - ("transformer", MockSelector()), - ] + pipe.fit(df) + assert ( + pipe.get_feature_names_out(input_features=trimmed_input_features) + == features_out ) - pipe.fit(df_vartypes.to_numpy()) - - if input_features is not None: - input_features = input_features[0:-1] - features_out = input_features[2:] - assert pipe.get_feature_names_out(input_features=input_features) == features_out - else: - features_out = ["x2", "x3"] - assert pipe.get_feature_names_out(input_features=input_features) == features_out @pytest.mark.parametrize( "input_features", [None, ["Age", "Marks"], np.array(["Age", "Marks"])] ) def test_remove_feature_names_pipe_and_skl_transformer_that_adds_features( - df_vartypes, input_features + input_features, ): - features_in = ["Age", "Marks"] - df = df_vartypes[features_in].copy() + df = pd.DataFrame({"Age": VARTYPES_DATA["Age"], "Marks": VARTYPES_DATA["Marks"]}) pipe = Pipeline( [ - ("poly", PolynomialFeatures()), + ("poly", PolynomialFeatures().set_output(transform="pandas")), ("transformer", MockSelector()), ] ) diff --git a/tests/test_base_transformers/test_transform_xy_mixin.py b/tests/test_base_transformers/test_transform_xy_mixin.py index 03a34f3d5..0bd6b4d5c 100644 --- a/tests/test_base_transformers/test_transform_xy_mixin.py +++ b/tests/test_base_transformers/test_transform_xy_mixin.py @@ -1,36 +1,60 @@ -import numpy as np +import narwhals as nw import pandas as pd +import polars as pl +import pytest from feature_engine._base_transformers.mixins import TransformXyMixin +BACKENDS = [(pd.DataFrame, pd.Series), (pl.DataFrame, pl.Series)] + class MockTransformer(TransformXyMixin): def transform(self, X): - return X.iloc[1:-1].copy() + # drops rows at positions 2 and 4, backend-agnostic + nw_X = nw.from_native(X, eager_only=True) + keep = [i for i in range(len(nw_X)) if i not in (2, 4)] + return nw_X[keep].to_native() -def test_transform_x_y_method(df_vartypes): - # single target - y = pd.Series(0, index=np.arange(len(df_vartypes))) +@pytest.mark.parametrize("make_df, make_series", BACKENDS) +def test_transform_x_y_single_target(make_df, make_series): + X = make_df({"a": [0, 1, 2, 3, 4, 5], "b": [10, 11, 12, 13, 14, 15]}) + y = make_series([0, 1, 2, 3, 4, 5]) transformer = MockTransformer() - Xt, yt = transformer.transform_x_y(df_vartypes, y) - assert len(Xt) == len(yt) - assert len(Xt) != len(df_vartypes) - assert len(yt) != len(y) - assert (Xt.index == yt.index).all() - assert (Xt.index == [1, 2]).all() + Xt, yt = transformer.transform_x_y(X, y) + + assert len(Xt) == 4 + assert len(yt) == 4 + assert nw.from_native(yt, series_only=True).to_list() == [0, 1, 3, 5] + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_x_y_multioutput_target(make_df): + X = make_df({"a": [0, 1, 2, 3, 4, 5], "b": [10, 11, 12, 13, 14, 15]}) + y = make_df({"t1": [0, 1, 2, 3, 4, 5], "t2": [0, 10, 20, 30, 40, 50]}) + transformer = MockTransformer() + + Xt, yt = transformer.transform_x_y(X, y) + + assert len(Xt) == 4 + assert len(yt) == 4 + nw_yt = nw.from_native(yt, eager_only=True) + assert nw_yt["t1"].to_list() == [0, 1, 3, 5] + assert nw_yt["t2"].to_list() == [0, 10, 30, 50] + + +def test_transform_x_y_pandas_index_alignment(df_vartypes): + # pandas branch keeps the original (non-default) index aligned between X and y + class DropFirstAndLast(TransformXyMixin): + def transform(self, X): + return X.iloc[1:-1].copy() - # multioutput target - y = ( - pd.DataFrame(columns=["vara", "varb"], index=df_vartypes.index) - .astype(float) - .fillna(0) - ) + y = pd.Series(range(len(df_vartypes)), index=df_vartypes.index) + transformer = DropFirstAndLast() Xt, yt = transformer.transform_x_y(df_vartypes, y) assert len(Xt) == len(yt) assert len(Xt) != len(df_vartypes) - assert len(yt) != len(y) assert (Xt.index == yt.index).all() assert (Xt.index == [1, 2]).all() diff --git a/tests/test_check_init_parameters/test_check_init_input_params.py b/tests/test_check_init_parameters/test_check_init_input_params.py index 4f4b7f631..d2c08abff 100644 --- a/tests/test_check_init_parameters/test_check_init_input_params.py +++ b/tests/test_check_init_parameters/test_check_init_input_params.py @@ -1,3 +1,5 @@ +import re + import pytest from feature_engine._check_init_parameters.check_init_input_params import ( @@ -6,13 +8,23 @@ ) -@pytest.mark.parametrize("missing_vals", [None, ["Hola"], True, "Hola"]) +@pytest.mark.parametrize( + "missing_vals", [None, ["Hola"], ["raise"], ("ignore",), True, 1, "Hola", "Raise"] +) def test_check_param_missing_values(missing_vals): - with pytest.raises(ValueError): + msg = ( + "missing_values takes only values 'raise' or 'ignore'. " + f"Got {missing_vals} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): _check_param_missing_values(missing_vals) @pytest.mark.parametrize("drop_orig", [None, ["Hola"], 10, "Hola"]) def test_check_param_drop_original(drop_orig): - with pytest.raises(ValueError): + msg = ( + "drop_original takes only boolean values True and False. " + f"Got {drop_orig} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): _check_param_drop_original(drop_orig) diff --git a/tests/test_creation/test_base_creation.py b/tests/test_creation/test_base_creation.py new file mode 100644 index 000000000..d32976007 --- /dev/null +++ b/tests/test_creation/test_base_creation.py @@ -0,0 +1,122 @@ +import narwhals as nw +import pandas as pd +import polars as pl +import pytest + +from feature_engine.creation.base_creation import BaseCreation + +BASIC_DATA = { + "var_a": [1, 2, 3, 4], + "var_b": [10, 20, 30, 40], + "var_c": [100, 200, 300, 400], +} + + +class StubCreation(BaseCreation): + def __init__(self, variables=None, missing_values="raise", drop_original=False): + self.variables = variables + super().__init__(missing_values=missing_values, drop_original=drop_original) + + def transform(self, X): + return self._check_transform_input_and_state(X) + + +class StubWithReference(StubCreation): + def __init__( + self, reference, variables=None, missing_values="raise", drop_original=False + ): + self.reference = reference + super().__init__( + variables=variables, + missing_values=missing_values, + drop_original=drop_original, + ) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_transform_round_trip(make_df): + X = make_df(BASIC_DATA) + transformer = StubCreation() + transformer.fit(X) + + assert transformer.variables_ == ["var_a", "var_b", "var_c"] + assert transformer.feature_names_in_ == ["var_a", "var_b", "var_c"] + assert transformer.n_features_in_ == 3 + + Xt = transformer.transform(X) + assert list(nw.from_native(Xt, eager_only=True).columns) == [ + "var_a", + "var_b", + "var_c", + ] + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_reorders_columns_to_match_fit(make_df): + X = make_df(BASIC_DATA) + transformer = StubCreation() + transformer.fit(X) + + reordered = make_df( + { + "var_c": BASIC_DATA["var_c"], + "var_a": BASIC_DATA["var_a"], + "var_b": BASIC_DATA["var_b"], + } + ) + Xt = transformer.transform(reordered) + assert list(nw.from_native(Xt, eager_only=True).columns) == [ + "var_a", + "var_b", + "var_c", + ] + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_raises_when_column_count_differs(make_df): + X = make_df(BASIC_DATA) + transformer = StubCreation() + transformer.fit(X) + + X_fewer_cols = make_df( + {"var_a": BASIC_DATA["var_a"], "var_b": BASIC_DATA["var_b"]} + ) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer" + ) + with pytest.raises(ValueError, match=msg): + transformer.transform(X_fewer_cols) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_missing_values_raise_vs_ignore(make_df): + data_with_na = {**BASIC_DATA, "var_a": [1, None, 3, 4]} + X = make_df(data_with_na) + + transformer_raise = StubCreation(missing_values="raise") + msg = "Some of the variables in the dataset contain NaN" + with pytest.raises(ValueError, match=msg): + transformer_raise.fit(X) + + transformer_ignore = StubCreation(missing_values="ignore") + transformer_ignore.fit(X) + Xt = transformer_ignore.transform(X) + assert len(nw.from_native(Xt, eager_only=True)) == 4 + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_reference_attribute_is_checked_in_fit(make_df): + X = make_df(BASIC_DATA) + transformer = StubWithReference(reference=["var_a"]) + transformer.fit(X) + assert transformer.variables_ == ["var_a", "var_b", "var_c"] + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_reference_must_be_numerical(make_df): + X = make_df({**BASIC_DATA, "var_d": ["a", "b", "c", "d"]}) + transformer = StubWithReference(reference=["var_d"]) + msg = "Some of the variables are not numerical" + with pytest.raises(TypeError, match=msg): + transformer.fit(X) diff --git a/tests/test_creation/test_check_estimator_creation.py b/tests/test_creation/test_check_estimator_creation.py index e50f0cbef..f472e3ec5 100644 --- a/tests/test_creation/test_check_estimator_creation.py +++ b/tests/test_creation/test_check_estimator_creation.py @@ -1,9 +1,7 @@ import pandas as pd import pytest -import sklearn from sklearn.pipeline import Pipeline from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine.creation import ( CyclicalFeatures, @@ -18,8 +16,6 @@ ) from tests.estimator_checks.sklearn_check_wrapper import wrap_for_check_estimator -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) - # Estimators for sklearn's check_estimator # Note: GeoDistanceFeatures is not included here because it requires 4 specific # named coordinate columns, but sklearn's check_estimator generates test data @@ -33,20 +29,13 @@ DecisionTreeFeatures(regression=False), ] -if sklearn_version > parse_version("1.6"): - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator( - estimator=wrap_for_check_estimator(estimator), - expected_failed_checks=estimator._more_tags()["_xfail_checks"], - ) - -else: - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator(estimator) +@pytest.mark.parametrize("estimator", _estimators) +def test_check_estimator_from_sklearn(estimator): + return check_estimator( + estimator=wrap_for_check_estimator(estimator), + expected_failed_checks=estimator._more_tags()["_xfail_checks"], + ) _estimators = [ @@ -84,12 +73,14 @@ def test_transformers_in_pipeline_with_set_output_pandas(transformer): # Test GeoDistanceFeatures in pipeline with proper column names def test_geo_distance_transformer_in_pipeline(): """Test GeoDistanceFeatures works in a sklearn pipeline.""" - X = pd.DataFrame({ - "lat1": [40.7128, 34.0522], - "lon1": [-74.0060, -118.2437], - "lat2": [34.0522, 41.8781], - "lon2": [-118.2437, -87.6298], - }) + X = pd.DataFrame( + { + "lat1": [40.7128, 34.0522], + "lon1": [-74.0060, -118.2437], + "lat2": [34.0522, 41.8781], + "lon2": [-118.2437, -87.6298], + } + ) y = pd.Series([0, 1]) transformer = GeoDistanceFeatures( diff --git a/tests/test_creation/test_cyclical_features.py b/tests/test_creation/test_cyclical_features.py index 5bc1df88f..b0187bbab 100644 --- a/tests/test_creation/test_cyclical_features.py +++ b/tests/test_creation/test_cyclical_features.py @@ -1,29 +1,33 @@ +import narwhals as nw import pandas as pd +import polars as pl import pytest from numpy import array from feature_engine.creation import CyclicalFeatures +CYCLICAL_DATA = { + "day": [6, 7, 5, 3, 1, 2, 4], + "months": [3, 7, 9, 12, 4, 6, 12], +} -@pytest.fixture -def df_cyclical(): - df = { - "day": [6, 7, 5, 3, 1, 2, 4], - "months": [3, 7, 9, 12, 4, 6, 12], - } - df = pd.DataFrame(df) - return df + +def assert_df_equal(X, expected: dict) -> None: + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + assert list(result.keys()) == list(expected.keys()) + for col, values in expected.items(): + assert result[col] == pytest.approx(values, abs=1e-5) -def test_general_transformation_without_dropping_variables(df_cyclical): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_general_transformation_without_dropping_variables(make_df): # test case 1: just one variable. + df = make_df(CYCLICAL_DATA) cyclical = CyclicalFeatures(variables=["day"]) - X = cyclical.fit_transform(df_cyclical) - - transf_df = df_cyclical.copy() + X = cyclical.fit_transform(df) - # expected output - transf_df["day_sin"] = [ + expected = dict(CYCLICAL_DATA) + expected["day_sin"] = [ -0.78183, 0.0, -0.97493, @@ -32,7 +36,7 @@ def test_general_transformation_without_dropping_variables(df_cyclical): 0.97493, -0.43388, ] - transf_df["day_cos"] = [ + expected["day_cos"] = [ 0.623490, 1.0, -0.222521, @@ -46,18 +50,18 @@ def test_general_transformation_without_dropping_variables(df_cyclical): assert cyclical.max_values_ == {"day": 7} # test transform output - pd.testing.assert_frame_equal(X, transf_df) + assert_df_equal(X, expected) -def test_general_transformation_dropping_original_variables(df_cyclical): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_general_transformation_dropping_original_variables(make_df): # test case 1: just one variable, but dropping the variable after transformation + df = make_df(CYCLICAL_DATA) cyclical = CyclicalFeatures(variables=["day"], drop_original=True) - X = cyclical.fit_transform(df_cyclical) + X = cyclical.fit_transform(df) - transf_df = df_cyclical.copy() - - # expected output - transf_df["day_sin"] = [ + expected = dict(CYCLICAL_DATA) + expected["day_sin"] = [ -0.78183, 0.0, -0.97493, @@ -66,7 +70,7 @@ def test_general_transformation_dropping_original_variables(df_cyclical): 0.97493, -0.43388, ] - transf_df["day_cos"] = [ + expected["day_cos"] = [ 0.623490, 1.0, -0.222521, @@ -75,60 +79,61 @@ def test_general_transformation_dropping_original_variables(df_cyclical): -0.222521, -0.900969, ] - transf_df = transf_df.drop(columns="day") + del expected["day"] # test fit attr assert cyclical.n_features_in_ == 2 assert cyclical.max_values_ == {"day": 7} # test transform output - pd.testing.assert_frame_equal(X, transf_df) + assert_df_equal(X, expected) -def test_automatically_find_variables(df_cyclical): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_automatically_find_variables(make_df): # test case 2: automatically select variables + df = make_df(CYCLICAL_DATA) cyclical = CyclicalFeatures(variables=None, drop_original=True) - X = cyclical.fit_transform(df_cyclical) - transf_df = df_cyclical.copy() - - # expected output - transf_df["day_sin"] = [ - -0.78183, - 0.0, - -0.97493, - 0.43388, - 0.78183, - 0.97493, - -0.43388, - ] - transf_df["day_cos"] = [ - 0.62349, - 1.0, - -0.222521, - -0.900969, - 0.62349, - -0.222521, - -0.900969, - ] - transf_df["months_sin"] = [ - 1.0, - -0.5, - -1.0, - 0.0, - 0.86603, - 0.0, - 0.0, - ] - transf_df["months_cos"] = [ - 0.0, - -0.86603, - -0.0, - 1.0, - -0.5, - -1.0, - 1.0, - ] - transf_df = transf_df.drop(columns=["day", "months"]) + X = cyclical.fit_transform(df) + + expected = { + "day_sin": [ + -0.78183, + 0.0, + -0.97493, + 0.43388, + 0.78183, + 0.97493, + -0.43388, + ], + "day_cos": [ + 0.62349, + 1.0, + -0.222521, + -0.900969, + 0.62349, + -0.222521, + -0.900969, + ], + "months_sin": [ + 1.0, + -0.5, + -1.0, + 0.0, + 0.86603, + 0.0, + 0.0, + ], + "months_cos": [ + 0.0, + -0.86603, + -0.0, + 1.0, + -0.5, + -1.0, + 1.0, + ], + } # test fit attr assert cyclical.max_values_ == { @@ -137,44 +142,53 @@ def test_automatically_find_variables(df_cyclical): } # test transform output - pd.testing.assert_frame_equal(X, transf_df) + assert_df_equal(X, expected) -def test_fit_raises_error_if_na_in_df(df_na): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_raises_error_if_na_in_df(make_df): # test case 3: when dataset contains na, fit method - with pytest.raises(ValueError): - transformer = CyclicalFeatures() - transformer.fit(df_na) + df = make_df({"day": [1, 2, None, 4], "months": [1, 2, 3, 4]}) + msg = "Some of the variables in the dataset contain NaN" + with pytest.raises(ValueError, match=msg): + CyclicalFeatures().fit(df) -def test_fit_raises_error_if_user_dictionary_key_not_in_df(df_cyclical): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_raises_error_if_user_dictionary_key_not_in_df(make_df): + df = make_df(CYCLICAL_DATA) + # message differs by backend (pandas KeyError vs narwhals + # ColumnNotFoundError, a KeyError subclass), so no match= here. with pytest.raises(KeyError): - transformer = CyclicalFeatures(max_values={"dayi": 31}) - transformer.fit(df_cyclical) + CyclicalFeatures(max_values={"dayi": 31}).fit(df) -def test_raises_error_when_init_parameters_not_permitted(df_cyclical): - - with pytest.raises(TypeError): +def test_raises_error_when_init_parameters_not_permitted(): + msg = "The parameter can only take a dictionary or None" + with pytest.raises(TypeError, match=msg): # when max_values is not a dictionary CyclicalFeatures(max_values=("dayi", 31)) - with pytest.raises(ValueError): + msg = "All values in the dictionary must be integer or float" + with pytest.raises(ValueError, match=msg): # when max_values values are not integers or string CyclicalFeatures(max_values={"day": "31"}) - with pytest.raises(ValueError): + msg = "drop_original takes only boolean values True and False" + with pytest.raises(ValueError, match=msg): # when drop original is not a boolean CyclicalFeatures(drop_original="True") -def test_max_values_mapping(df_cyclical): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_max_values_mapping(make_df): + df = make_df(CYCLICAL_DATA) cyclical = CyclicalFeatures(variables="day", max_values={"day": 31}) - X = cyclical.fit_transform(df_cyclical) + X = cyclical.fit_transform(df) - transf_df = df_cyclical.copy() - transf_df["day_sin"] = [ + expected = dict(CYCLICAL_DATA) + expected["day_sin"] = [ 0.937752, 0.988468, 0.848644, @@ -183,7 +197,7 @@ def test_max_values_mapping(df_cyclical): 0.394355, 0.724792, ] - transf_df["day_cos"] = [ + expected["day_cos"] = [ 0.347305, 0.151428, 0.528964, @@ -192,34 +206,53 @@ def test_max_values_mapping(df_cyclical): 0.918958, 0.688967, ] - pd.testing.assert_frame_equal(X, transf_df) + assert_df_equal(X, expected) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "input_features", [None, ["day", "months"], array(["day", "months"])] ) -def test_get_feature_names_out(df_cyclical, input_features): +def test_get_feature_names_out(make_df, input_features): # default features from all variables + df = make_df(CYCLICAL_DATA) transformer = CyclicalFeatures() - X = transformer.fit_transform(df_cyclical) - feat_out = list(df_cyclical.columns) + [ + X = transformer.fit_transform(df) + feat_out = list(CYCLICAL_DATA.keys()) + [ "day_sin", "day_cos", "months_sin", "months_cos", ] - assert list(X.columns) == transformer.get_feature_names_out() + assert ( + list(nw.from_native(X, eager_only=True).columns) + == transformer.get_feature_names_out() + ) assert transformer.get_feature_names_out(input_features=input_features) == feat_out - with pytest.raises(ValueError): + msg = "input_features is not equal to feature_names_in_" + with pytest.raises(ValueError, match=msg): transformer.get_feature_names_out(input_features=["day"]) - with pytest.raises(ValueError): + with pytest.raises(ValueError, match=msg): transformer.get_feature_names_out(input_features=["sandia", "banana"]) transformer = CyclicalFeatures(drop_original=True) - X = transformer.fit_transform(df_cyclical) + X = transformer.fit_transform(df) feat_out = ["day_sin", "day_cos", "months_sin", "months_cos"] - assert list(X.columns) == transformer.get_feature_names_out() + assert ( + list(nw.from_native(X, eager_only=True).columns) + == transformer.get_feature_names_out() + ) assert transformer.get_feature_names_out(input_features=input_features) == feat_out + + +def test_integer_column_names(): + # integer column names are pandas-only + X = pd.DataFrame({0: [1.0, 2.0, 3.0], 1: [10.0, 20.0, 40.0]}) + Xt = CyclicalFeatures().fit_transform(X) + expected = CyclicalFeatures().fit_transform(X.rename(columns=str)) + + assert list(Xt.columns) == [0, 1, "0_sin", "0_cos", "1_sin", "1_cos"] + assert Xt.to_numpy().tolist() == expected.to_numpy().tolist() diff --git a/tests/test_creation/test_decision_tree_features.py b/tests/test_creation/test_decision_tree_features.py index 4e8a93e8c..a97f68475 100644 --- a/tests/test_creation/test_decision_tree_features.py +++ b/tests/test_creation/test_decision_tree_features.py @@ -1,5 +1,9 @@ +import warnings + +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.model_selection import GridSearchCV from sklearn.pipeline import Pipeline @@ -8,44 +12,87 @@ from feature_engine.creation import DecisionTreeFeatures from tests.estimator_checks.fit_functionality_checks import check_return_empty - -@pytest.fixture(scope="module") -def df_creation(): - data = { - "Name": [ - "tom", - "nick", - "krish", - "megan", - "peter", - "jordan", - "fred", - "sam", - "alexa", - "brittany", - ], - "Age": [20, 44, 19, 33, 51, 40, 41, 37, 30, 54], - "Height": [164, 150, 178, 158, 188, 190, 168, 174, 176, 171], - "Marks": [1.0, 0.8, 0.6, 0.1, 0.3, 0.4, 0.8, 0.6, 0.5, 0.2], - } - - df = pd.DataFrame(data) - return df - - -@pytest.fixture(scope="module") -def regression_target(): - return pd.Series([4.1, 5.8, 3.9, 6.2, 4.3, 4.5, 7.2, 4.4, 4.1, 6.7]) - - -@pytest.fixture(scope="module") -def classification_target(): - return pd.Series([1, 1, 1, 0, 0, 1, 0, 1, 0, 0]) - - -@pytest.fixture(scope="module") -def multiclass_target(): - return pd.Series([1, 1, 2, 2, 0, 1, 0, 1, 0, 0]) +DATA = { + "Name": [ + "tom", + "nick", + "krish", + "megan", + "peter", + "jordan", + "fred", + "sam", + "alexa", + "brittany", + ], + "Age": [20, 44, 19, 33, 51, 40, 41, 37, 30, 54], + "Height": [164, 150, 178, 158, 188, 190, 168, 174, 176, 171], + "Marks": [1.0, 0.8, 0.6, 0.1, 0.3, 0.4, 0.8, 0.6, 0.5, 0.2], +} +REGRESSION_Y = [4.1, 5.8, 3.9, 6.2, 4.3, 4.5, 7.2, 4.4, 4.1, 6.7] +BINARY_Y = [1, 1, 1, 0, 0, 1, 0, 1, 0, 0] +MULTICLASS_Y = [1, 1, 2, 2, 0, 1, 0, 1, 0, 0] + +COMBOS = [ + "Age", + "Height", + "Marks", + ["Age", "Height"], + ["Age", "Marks"], + ["Height", "Marks"], + ["Age", "Height", "Marks"], +] + + +def _select(X, combo): + cols = combo if isinstance(combo, list) else [combo] + return nw.from_native(X, eager_only=True).select(cols).to_native() + + +def _expected_tree_predictions( + X, + y, + scoring, + random_state, + regression=True, + binary=False, + precision=None, + param_grid=None, +): + # Fits a fresh GridSearchCV per combo on the same backend as X, so this + # works as the reference for both pandas and polars input alike. + if param_grid is None: + param_grid = {"max_depth": [1, 2, 3, 4]} + if regression is True: + est = DecisionTreeRegressor(random_state=random_state) + else: + est = DecisionTreeClassifier(random_state=random_state) + tree = GridSearchCV(est, cv=3, scoring=scoring, param_grid=param_grid) + + expected = {} + for combo in COMBOS: + X_sub = _select(X, combo) + tree.fit(X_sub, y) + if regression is True: + preds = tree.predict(X_sub) + elif binary is True: + preds = tree.predict_proba(X_sub)[:, 1] + else: + preds = tree.predict(X_sub) + if precision is not None: + preds = np.round(preds, precision) + expected[f"tree({combo})"] = list(preds) + return expected + + +def assert_df_equal(X, expected: dict) -> None: + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + assert list(result.keys()) == list(expected.keys()) + for col, values in expected.items(): + if all(isinstance(v, (int, float, np.integer, np.floating)) for v in values): + assert result[col] == pytest.approx(values, abs=1e-6) + else: + assert result[col] == values @pytest.mark.parametrize("precision", ["string", 0.1, -1, np.nan]) @@ -204,390 +251,226 @@ def test_create_variable_combinations_when_tuple(input_features, expected): assert combos == expected -def test_feature_creation_regression(df_creation, regression_target): - X = df_creation.copy() - y = regression_target.copy() - +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_feature_creation_regression(make_df): + X = make_df(DATA) scoring = "neg_mean_squared_error" rs = 0 tr = DecisionTreeFeatures(scoring=scoring, random_state=rs) - Xt = tr.fit_transform(X, y) - - # get expected - est = DecisionTreeRegressor(random_state=rs) - tree = GridSearchCV( - est, - cv=3, - scoring=scoring, - param_grid={"max_depth": [1, 2, 3, 4]}, - ) - - combos = [ - "Age", - "Height", - "Marks", - ["Age", "Height"], - ["Age", "Marks"], - ["Height", "Marks"], - ["Age", "Height", "Marks"], - ] - var_names = [f"tree({item})" for item in combos] - - X_exp = df_creation.copy() - for i in range(len(combos)): - varn = var_names[i] - combon = combos[i] - if isinstance(combon, str): - tree.fit(X[combon].to_frame(), y) - X_exp[varn] = tree.predict(X[combon].to_frame()) - else: - tree.fit(X[combon], y) - X_exp[varn] = tree.predict(X[combon]) - - pd.testing.assert_frame_equal(Xt, X_exp) + Xt = tr.fit_transform(X, REGRESSION_Y) + expected = dict(DATA) + expected.update(_expected_tree_predictions(X, REGRESSION_Y, scoring, rs)) + assert_df_equal(Xt, expected) -def test_feature_creation_regression_and_precision(df_creation, regression_target): - X = df_creation.copy() - y = regression_target.copy() +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_feature_creation_regression_and_precision(make_df): + X = make_df(DATA) scoring = "neg_mean_squared_error" rs = 0 tr = DecisionTreeFeatures(scoring=scoring, random_state=rs, precision=1) - Xt = tr.fit_transform(X, y) - - # get expected - est = DecisionTreeRegressor(random_state=rs) - tree = GridSearchCV( - est, - cv=3, - scoring=scoring, - param_grid={"max_depth": [1, 2, 3, 4]}, - ) - - combos = [ - "Age", - "Height", - "Marks", - ["Age", "Height"], - ["Age", "Marks"], - ["Height", "Marks"], - ["Age", "Height", "Marks"], - ] - var_names = [f"tree({item})" for item in combos] - - X_exp = df_creation.copy() - for i in range(len(combos)): - varn = var_names[i] - combon = combos[i] - if isinstance(combon, str): - tree.fit(X[combon].to_frame(), y) - preds = tree.predict(X[combon].to_frame()) - X_exp[varn] = np.round(preds, 1) - else: - tree.fit(X[combon], y) - preds = tree.predict(X[combon]) - X_exp[varn] = np.round(preds, 1) - - pd.testing.assert_frame_equal(Xt, X_exp) + Xt = tr.fit_transform(X, REGRESSION_Y) + expected = dict(DATA) + expected.update( + _expected_tree_predictions(X, REGRESSION_Y, scoring, rs, precision=1) + ) + assert_df_equal(Xt, expected) -def test_feature_creation_regression_drop_original(df_creation, regression_target): - X = df_creation.copy() - y = regression_target.copy() +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_feature_creation_regression_drop_original(make_df): + X = make_df(DATA) scoring = "neg_mean_squared_error" rs = 0 tr = DecisionTreeFeatures(scoring=scoring, random_state=rs, drop_original=True) - Xt = tr.fit_transform(X, y) - - # get expected - est = DecisionTreeRegressor(random_state=rs) - tree = GridSearchCV( - est, - cv=3, - scoring=scoring, - param_grid={"max_depth": [1, 2, 3, 4]}, - ) - - combos = [ - "Age", - "Height", - "Marks", - ["Age", "Height"], - ["Age", "Marks"], - ["Height", "Marks"], - ["Age", "Height", "Marks"], - ] - var_names = [f"tree({item})" for item in combos] - - X_exp = df_creation.copy() - for i in range(len(combos)): - varn = var_names[i] - combon = combos[i] - if isinstance(combon, str): - tree.fit(X[combon].to_frame(), y) - X_exp[varn] = tree.predict(X[combon].to_frame()) - else: - tree.fit(X[combon], y) - X_exp[varn] = tree.predict(X[combon]) - X_exp.drop(["Age", "Height", "Marks"], axis=1, inplace=True) - - pd.testing.assert_frame_equal(Xt, X_exp) + Xt = tr.fit_transform(X, REGRESSION_Y) + expected = {"Name": DATA["Name"]} + expected.update(_expected_tree_predictions(X, REGRESSION_Y, scoring, rs)) + assert_df_equal(Xt, expected) -def test_feature_creation_binary_classif(df_creation, classification_target): - X = df_creation.copy() - y = classification_target.copy() +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_feature_creation_binary_classif(make_df): + X = make_df(DATA) scoring = "roc_auc" rs = 0 tr = DecisionTreeFeatures(scoring=scoring, random_state=rs, regression=False) - Xt = tr.fit_transform(X, y) - - # get expected - est = DecisionTreeClassifier(random_state=rs) - tree = GridSearchCV( - est, - cv=3, - scoring=scoring, - param_grid={"max_depth": [1, 2, 3, 4]}, - ) - - combos = [ - "Age", - "Height", - "Marks", - ["Age", "Height"], - ["Age", "Marks"], - ["Height", "Marks"], - ["Age", "Height", "Marks"], - ] - var_names = [f"tree({item})" for item in combos] - - X_exp = df_creation.copy() - for i in range(len(combos)): - varn = var_names[i] - combon = combos[i] - if isinstance(combon, str): - tree.fit(X[combon].to_frame(), y) - preds = tree.predict_proba(X[combon].to_frame()) - X_exp[varn] = preds[:, 1] - else: - tree.fit(X[combon], y) - preds = tree.predict_proba(X[combon]) - X_exp[varn] = preds[:, 1] - - pd.testing.assert_frame_equal(Xt, X_exp) + Xt = tr.fit_transform(X, BINARY_Y) + expected = dict(DATA) + expected.update( + _expected_tree_predictions( + X, BINARY_Y, scoring, rs, regression=False, binary=True + ) + ) + assert_df_equal(Xt, expected) -def test_feature_creation_binary_classif_w_precision( - df_creation, classification_target -): - X = df_creation.copy() - y = classification_target.copy() +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_feature_creation_binary_classif_w_precision(make_df): + X = make_df(DATA) scoring = "roc_auc" rs = 0 tr = DecisionTreeFeatures( scoring=scoring, random_state=rs, regression=False, precision=2 ) - Xt = tr.fit_transform(X, y) - - # get expected - est = DecisionTreeClassifier(random_state=rs) - tree = GridSearchCV( - est, - cv=3, - scoring=scoring, - param_grid={"max_depth": [1, 2, 3, 4]}, - ) - - combos = [ - "Age", - "Height", - "Marks", - ["Age", "Height"], - ["Age", "Marks"], - ["Height", "Marks"], - ["Age", "Height", "Marks"], - ] - var_names = [f"tree({item})" for item in combos] - - X_exp = df_creation.copy() - for i in range(len(combos)): - varn = var_names[i] - combon = combos[i] - if isinstance(combon, str): - tree.fit(X[combon].to_frame(), y) - preds = tree.predict_proba(X[combon].to_frame()) - X_exp[varn] = np.round(preds[:, 1], 2) - else: - tree.fit(X[combon], y) - preds = tree.predict_proba(X[combon]) - X_exp[varn] = np.round(preds[:, 1], 2) - - pd.testing.assert_frame_equal(Xt, X_exp) + Xt = tr.fit_transform(X, BINARY_Y) + expected = dict(DATA) + expected.update( + _expected_tree_predictions( + X, BINARY_Y, scoring, rs, regression=False, binary=True, precision=2 + ) + ) + assert_df_equal(Xt, expected) -def test_feature_creation_binary_multiclass(df_creation, multiclass_target): - X = df_creation.copy() - y = multiclass_target.copy() +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_feature_creation_binary_multiclass(make_df): + X = make_df(DATA) scoring = "roc_auc" rs = 0 tr = DecisionTreeFeatures(scoring=scoring, random_state=rs, regression=False) - Xt = tr.fit_transform(X, y) - - # get expected - est = DecisionTreeClassifier(random_state=rs) - tree = GridSearchCV( - est, - cv=3, - scoring=scoring, - param_grid={"max_depth": [1, 2, 3, 4]}, - ) - - combos = [ - "Age", - "Height", - "Marks", - ["Age", "Height"], - ["Age", "Marks"], - ["Height", "Marks"], - ["Age", "Height", "Marks"], - ] - var_names = [f"tree({item})" for item in combos] - - X_exp = df_creation.copy() - for i in range(len(combos)): - varn = var_names[i] - combon = combos[i] - if isinstance(combon, str): - tree.fit(X[combon].to_frame(), y) - preds = tree.predict(X[combon].to_frame()) - X_exp[varn] = preds - else: - tree.fit(X[combon], y) - preds = tree.predict(X[combon]) - X_exp[varn] = preds - - pd.testing.assert_frame_equal(Xt, X_exp) - - -def test_get_feature_names_out(df_creation, regression_target): - X = df_creation.copy() - y = regression_target.copy() + Xt = tr.fit_transform(X, MULTICLASS_Y) - tr = DecisionTreeFeatures( - variables=["Age", "Marks"], + expected = dict(DATA) + expected.update( + _expected_tree_predictions( + X, MULTICLASS_Y, scoring, rs, regression=False, binary=False + ) ) + assert_df_equal(Xt, expected) - Xt = tr.fit_transform(X, y) - feat_out = Xt.columns.to_list() - assert tr.get_feature_names_out() == feat_out - assert tr.get_feature_names_out(X.columns.to_list()) == feat_out +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_get_feature_names_out(make_df): + X = make_df(DATA) + tr = DecisionTreeFeatures(variables=["Age", "Marks"]) + Xt = tr.fit_transform(X, REGRESSION_Y) + feat_out = list(nw.from_native(Xt, eager_only=True).columns) + assert tr.get_feature_names_out() == feat_out + assert tr.get_feature_names_out(list(DATA.keys())) == feat_out -def test_get_feature_names_out_from_pipeline(df_creation, regression_target): - X = df_creation.copy() - y = regression_target.copy() - - # set up transformer - tr = DecisionTreeFeatures( - variables=["Age", "Marks"], - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_get_feature_names_out_from_pipeline(make_df): + X = make_df(DATA) + tr = DecisionTreeFeatures(variables=["Age", "Marks"]) pipe = Pipeline([("transformer", tr)]) - - Xt = pipe.fit_transform(X, y) - feat_out = Xt.columns.to_list() + Xt = pipe.fit_transform(X, REGRESSION_Y) + feat_out = list(nw.from_native(Xt, eager_only=True).columns) assert pipe.get_feature_names_out(input_features=None) == feat_out - assert pipe.get_feature_names_out(input_features=X.columns.to_list()) == feat_out + assert pipe.get_feature_names_out(input_features=list(DATA.keys())) == feat_out +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("_input_features", ["hola", ["Age", "Marks"]]) -def test_get_feature_names_out_raises_error_when_wrong_param( - _input_features, df_creation, regression_target -): - X = df_creation.copy() - y = regression_target.copy() - - tr = DecisionTreeFeatures( - variables=["Age", "Marks"], - ) - tr.fit(X, y) - +def test_get_feature_names_out_raises_error_when_wrong_param(make_df, _input_features): + X = make_df(DATA) + tr = DecisionTreeFeatures(variables=["Age", "Marks"]) + tr.fit(X, REGRESSION_Y) with pytest.raises(ValueError): tr.get_feature_names_out(input_features=_input_features) -def test_error_when_regression_true_and_target_binary( - df_creation, classification_target -): - X = df_creation.copy() - y = classification_target.copy() +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_when_regression_true_and_target_binary(make_df): + X = make_df(DATA) tr = DecisionTreeFeatures(regression=True) msg = ( "Trying to fit a regression to a binary target is not " - + "allowed by this transformer. Check the target values " - + "or set regression to False." + "allowed by this transformer. Check the target values " + "or set regression to False." ) with pytest.raises(ValueError, match=msg): - tr.fit(X, y) + tr.fit(X, BINARY_Y) -def test_user_enter_param_grid(df_creation, classification_target): - X = df_creation.copy() - y = classification_target.copy() +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_user_enter_param_grid(make_df): + X = make_df(DATA) scoring = "roc_auc" rs = 0 grid = {"max_depth": [1, 2, 3, 4]} tr = DecisionTreeFeatures( scoring=scoring, random_state=rs, regression=False, param_grid=grid ) - Xt = tr.fit_transform(X, y) - - # get expected - est = DecisionTreeClassifier(random_state=rs) - tree = GridSearchCV( - est, - cv=3, - scoring=scoring, - param_grid={"max_depth": [1, 2, 3, 4]}, - ) - - combos = [ - "Age", - "Height", - "Marks", - ["Age", "Height"], - ["Age", "Marks"], - ["Height", "Marks"], - ["Age", "Height", "Marks"], - ] - var_names = [f"tree({item})" for item in combos] - - X_exp = df_creation.copy() - for i in range(len(combos)): - varn = var_names[i] - combon = combos[i] - if isinstance(combon, str): - tree.fit(X[combon].to_frame(), y) - preds = tree.predict_proba(X[combon].to_frame()) - X_exp[varn] = preds[:, 1] - else: - tree.fit(X[combon], y) - preds = tree.predict_proba(X[combon]) - X_exp[varn] = preds[:, 1] + Xt = tr.fit_transform(X, BINARY_Y) - pd.testing.assert_frame_equal(Xt, X_exp) + expected = dict(DATA) + expected.update( + _expected_tree_predictions( + X, BINARY_Y, scoring, rs, regression=False, binary=True, param_grid=grid + ) + ) + assert_df_equal(Xt, expected) def test_check_return_empty(): # DecisionTreeFeatures is not part of the check_feature_engine_estimator # pipeline (test_check_estimator_creation.py only feeds MathFeatures, # RelativeFeatures and CyclicalFeatures into it), so return_empty is - # tested directly here instead. + # tested directly here instead. check_return_empty is a shared, + # pandas-only estimator-check helper used across the library. check_return_empty(DecisionTreeFeatures(regression=False)) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_n_jobs_parallel_matches_sequential(make_df): + # core correctness check for n_jobs: parallelizing tree training across + # feature combinations must produce identical trees, and therefore + # identical predictions, to sequential training (n_jobs=None). + X = make_df(DATA) + tr_seq = DecisionTreeFeatures(n_jobs=None, random_state=0) + tr_seq.fit(X, REGRESSION_Y) + tr_par = DecisionTreeFeatures(n_jobs=2, random_state=0) + tr_par.fit(X, REGRESSION_Y) + + Xt_seq = tr_seq.transform(X) + Xt_par = tr_par.transform(X) + + expected = nw.from_native(Xt_seq, eager_only=True).to_dict(as_series=False) + assert_df_equal(Xt_par, expected) + + +def test_transform_does_not_fragment_pandas_output(): + # regression test: transform() used to assign one new tree column at a + # time (X[col_name] = preds), which triggers pandas' "DataFrame is + # highly fragmented" PerformanceWarning once there are enough feature + # combinations - fixed by building all new columns in one DataFrame + # and joining once. Needs enough variables to cross pandas' internal + # fragmentation threshold (a handful of combos won't trigger it). + rng = np.random.RandomState(0) + n_vars = 9 + X = pd.DataFrame( + rng.rand(200, n_vars), columns=[f"v{i}" for i in range(n_vars)] + ) + y = rng.rand(200) + + tr = DecisionTreeFeatures( + features_to_combine=3, param_grid={"max_depth": [1, 2]}, random_state=0 + ) + tr.fit(X, y) + + with warnings.catch_warnings(): + warnings.simplefilter("error", pd.errors.PerformanceWarning) + tr.transform(X) + + +def test_single_int_named_feature_combo(): + # regression test: a single-variable combo with an integer column name + # used to crash (isinstance(features, str) missed the int case), since + # X[features] for a bare int returns a 1D Series, not the 2D input + # sklearn requires - fixed to check isinstance(features, (str, int)). + # Integer column names are pandas-only - polars requires string columns. + df = pd.DataFrame({0: [1.0, 2, 3, 4, 5, 6, 7, 8], 1: [2.0, 3, 4, 5, 6, 7, 8, 9]}) + y = [1.0, 2, 3, 4, 5, 6, 7, 8] + transformer = DecisionTreeFeatures(features_to_combine=1, random_state=0) + transformer.fit(df, y) + Xt = transformer.transform(df) + assert "tree(0)" in Xt.columns + assert "tree(1)" in Xt.columns diff --git a/tests/test_creation/test_geo_features.py b/tests/test_creation/test_geo_features.py index bbd800044..f137e4ef1 100644 --- a/tests/test_creation/test_geo_features.py +++ b/tests/test_creation/test_geo_features.py @@ -1,81 +1,84 @@ +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from feature_engine.creation import GeoDistanceFeatures +COORDS_DATA = { + "lat1": [40.7128], + "lon1": [-74.0060], + "lat2": [34.0522], + "lon2": [-118.2437], +} -@pytest.fixture -def df_coords(): - """Fixture providing sample coordinate data for a single route.""" - return pd.DataFrame({ - "lat1": [40.7128], - "lon1": [-74.0060], - "lat2": [34.0522], - "lon2": [-118.2437], - }) - - -@pytest.fixture -def df_multi_coords(): - """Fixture providing sample coordinate data with multiple rows.""" - return pd.DataFrame({ - "origin_lat": [40.7128, 34.0522, 41.8781], - "origin_lon": [-74.0060, -118.2437, -87.6298], - "dest_lat": [34.0522, 41.8781, 40.7128], - "dest_lon": [-118.2437, -87.6298, -74.0060], - }) - - -@pytest.fixture -def df_with_extra(): - """Fixture for DataFrame with coordinates and extra columns.""" - return pd.DataFrame({ - "lat1": [40.0], - "lon1": [-74.0], - "lat2": [34.0], - "lon2": [-118.0], - "other": [1], - }) - - -def test_haversine_distance_default(df_coords): +MULTI_COORDS_DATA = { + "origin_lat": [40.7128, 34.0522, 41.8781], + "origin_lon": [-74.0060, -118.2437, -87.6298], + "dest_lat": [34.0522, 41.8781, 40.7128], + "dest_lon": [-118.2437, -87.6298, -74.0060], +} + +COORDS_WITH_EXTRA_DATA = { + "lat1": [40.0], + "lon1": [-74.0], + "lat2": [34.0], + "lon2": [-118.0], + "other": [1], +} + + +def get_value(X, col: str, idx: int = 0): + """Extract a single scalar from a pandas or polars dataframe column.""" + return nw.from_native(X, eager_only=True).get_column(col).to_list()[idx] + + +def assert_df_equal(X, expected: dict, abs_tol: float = 1e-5) -> None: + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + assert list(result.keys()) == list(expected.keys()) + for col, values in expected.items(): + assert result[col] == pytest.approx(values, abs=abs_tol) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_haversine_distance_default(make_df): """Test Haversine distance calculation with default parameters.""" + df = make_df(COORDS_DATA) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2" ) - X_tr = transformer.fit_transform(df_coords) + X_tr = transformer.fit_transform(df) assert "geo_distance" in X_tr.columns - assert 3900 < X_tr["geo_distance"].iloc[0] < 4000 + assert 3900 < get_value(X_tr, "geo_distance") < 4000 -def test_haversine_distance_miles(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_haversine_distance_miles(make_df): """Test Haversine distance in miles.""" - X = pd.DataFrame({ - "lat1": [40.7128], - "lon1": [-74.0060], - "lat2": [34.0522], - "lon2": [-118.2437], - }) + X = make_df(COORDS_DATA) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2", output_unit="miles" ) X_tr = transformer.fit_transform(X) - assert 2400 < X_tr["geo_distance"].iloc[0] < 2500 + assert 2400 < get_value(X_tr, "geo_distance") < 2500 +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("method", ["haversine", "euclidean", "manhattan"]) @pytest.mark.parametrize("output_unit", ["km", "miles", "meters", "feet"]) -def test_same_location_zero_distance(method, output_unit): +def test_same_location_zero_distance(make_df, method, output_unit): """Test that same location returns zero distance for all methods and units.""" - X = pd.DataFrame({ - "lat1": [40.7128, 34.0522], - "lon1": [-74.0060, -118.2437], - "lat2": [40.7128, 34.0522], - "lon2": [-74.0060, -118.2437], - }) + X = make_df( + { + "lat1": [40.7128, 34.0522], + "lon1": [-74.0060, -118.2437], + "lat2": [40.7128, 34.0522], + "lon2": [-74.0060, -118.2437], + } + ) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", @@ -86,14 +89,14 @@ def test_same_location_zero_distance(method, output_unit): ) X_tr = transformer.fit_transform(X) - np.testing.assert_array_almost_equal( - X_tr["geo_distance"].values, [0.0, 0.0], decimal=10 - ) + values = nw.from_native(X_tr, eager_only=True).get_column("geo_distance") + np.testing.assert_array_almost_equal(values.to_list(), [0.0, 0.0], decimal=10) -def test_euclidean_method(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_euclidean_method(make_df): """Test Euclidean distance method returns expected values.""" - X = pd.DataFrame({"lat1": [0.0], "lon1": [0.0], "lat2": [1.0], "lon2": [1.0]}) + X = make_df({"lat1": [0.0], "lon1": [0.0], "lat2": [1.0], "lon2": [1.0]}) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2", method="euclidean" ) @@ -101,13 +104,14 @@ def test_euclidean_method(): expected_distance = np.sqrt(2) * 111.0 np.testing.assert_almost_equal( - X_tr["geo_distance"].iloc[0], expected_distance, decimal=1 + get_value(X_tr, "geo_distance"), expected_distance, decimal=1 ) -def test_manhattan_method(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_manhattan_method(make_df): """Test Manhattan distance method returns expected values.""" - X = pd.DataFrame({"lat1": [0.0], "lon1": [0.0], "lat2": [1.0], "lon2": [1.0]}) + X = make_df({"lat1": [0.0], "lon1": [0.0], "lat2": [1.0], "lon2": [1.0]}) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2", method="manhattan" ) @@ -115,30 +119,35 @@ def test_manhattan_method(): expected_distance = 2 * 111.0 np.testing.assert_almost_equal( - X_tr["geo_distance"].iloc[0], expected_distance, decimal=1 + get_value(X_tr, "geo_distance"), expected_distance, decimal=1 ) -def test_custom_output_column_name(df_coords): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_custom_output_column_name(make_df): """Test custom output column name.""" + df = make_df(COORDS_DATA) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2", output_col="distance_km" ) - X_tr = transformer.fit_transform(df_coords) + X_tr = transformer.fit_transform(df) assert "distance_km" in X_tr.columns assert "geo_distance" not in X_tr.columns -def test_drop_original_columns(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_drop_original_columns(make_df): """Test drop_original parameter removes coordinate columns.""" - X = pd.DataFrame({ - "lat1": [40.7128], - "lon1": [-74.0060], - "lat2": [34.0522], - "lon2": [-118.2437], - "other": [1], - }) + X = make_df( + { + "lat1": [40.7128], + "lon1": [-74.0060], + "lat2": [34.0522], + "lon2": [-118.2437], + "other": [1], + } + ) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2", drop_original=True ) @@ -153,26 +162,23 @@ def test_drop_original_columns(): assert list(X_tr.columns) == ["other", "geo_distance"] -def test_multiple_rows(df_multi_coords): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_multiple_rows(make_df): """Test transformation with multiple rows returns expected distances.""" + df = make_df(MULTI_COORDS_DATA) transformer = GeoDistanceFeatures( lat1="origin_lat", lon1="origin_lon", lat2="dest_lat", lon2="dest_lon" ) - X_tr = transformer.fit_transform(df_multi_coords) + X_tr = transformer.fit_transform(df) - expected = df_multi_coords.copy() + expected = dict(MULTI_COORDS_DATA) expected["geo_distance"] = [ 3935.746254609723, 2803.971506975193, 1144.2912739463475, ] - pd.testing.assert_frame_equal( - X_tr, - expected, - check_exact=False, - atol=0.001, - ) + assert_df_equal(X_tr, expected, abs_tol=0.001) @pytest.mark.parametrize("invalid_method", ["invalid", True, 123]) @@ -197,9 +203,10 @@ def test_invalid_output_unit_raises_error(invalid_unit): ) -def test_missing_columns_raises_error(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_missing_columns_raises_error(make_df): """Test that missing columns raise ValueError on fit.""" - X = pd.DataFrame({"lat1": [1], "lon1": [1]}) + X = make_df({"lat1": [1], "lon1": [1]}) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2" ) @@ -207,15 +214,18 @@ def test_missing_columns_raises_error(): transformer.fit(X) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("invalid_lat", [100, -100]) -def test_invalid_latitude_range_raises_error(invalid_lat): +def test_invalid_latitude_range_raises_error(make_df, invalid_lat): """Test that latitude outside [-90, 90] raises ValueError.""" - X = pd.DataFrame({ - "lat1": [invalid_lat], - "lon1": [0], - "lat2": [0], - "lon2": [0], - }) + X = make_df( + { + "lat1": [invalid_lat], + "lon1": [0], + "lat2": [0], + "lon2": [0], + } + ) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2" ) @@ -223,15 +233,18 @@ def test_invalid_latitude_range_raises_error(invalid_lat): transformer.fit(X) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("invalid_lon", [200, -200]) -def test_invalid_longitude_range_raises_error(invalid_lon): +def test_invalid_longitude_range_raises_error(make_df, invalid_lon): """Test that longitude outside [-180, 180] raises ValueError.""" - X = pd.DataFrame({ - "lat1": [0], - "lon1": [invalid_lon], - "lat2": [0], - "lon2": [0], - }) + X = make_df( + { + "lat1": [0], + "lon1": [invalid_lon], + "lat2": [0], + "lon2": [0], + } + ) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2" ) @@ -239,14 +252,17 @@ def test_invalid_longitude_range_raises_error(invalid_lon): transformer.fit(X) -def test_validate_ranges_disabled(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_validate_ranges_disabled(make_df): """Test that invalid coordinates don't raise error when validate_ranges=False.""" - X = pd.DataFrame({ - "lat1": [100], - "lon1": [200], - "lat2": [0], - "lon2": [0], - }) + X = make_df( + { + "lat1": [100], + "lon1": [200], + "lat2": [0], + "lon2": [0], + } + ) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2", validate_ranges=False ) @@ -268,11 +284,10 @@ def test_validate_ranges_parameter_validation(invalid_value): ) -def test_fit_stores_attributes(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_stores_attributes(make_df): """Test that fit stores expected attributes with correct values.""" - X = pd.DataFrame( - {"lat1": [40.0], "lon1": [-74.0], "lat2": [34.0], "lon2": [-118.0]} - ) + X = make_df({"lat1": [40.0], "lon1": [-74.0], "lat2": [34.0], "lon2": [-118.0]}) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2" ) @@ -286,38 +301,38 @@ def test_fit_stores_attributes(): assert transformer.n_features_in_ == 4 -def test_get_feature_names_out(df_with_extra): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_get_feature_names_out(make_df): """Test get_feature_names_out returns correct feature names.""" + df = make_df(COORDS_WITH_EXTRA_DATA) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2" ) - transformer.fit(df_with_extra) + transformer.fit(df) feature_names = transformer.get_feature_names_out() expected_names = ["lat1", "lon1", "lat2", "lon2", "other", "geo_distance"] assert feature_names == expected_names -def test_get_feature_names_out_with_drop_original(df_with_extra): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_get_feature_names_out_with_drop_original(make_df): """Test get_feature_names_out when drop_original=True.""" + df = make_df(COORDS_WITH_EXTRA_DATA) transformer = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2", drop_original=True ) - transformer.fit(df_with_extra) + transformer.fit(df) feature_names = transformer.get_feature_names_out() expected_names = ["other", "geo_distance"] assert feature_names == expected_names -def test_output_units_conversion(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_output_units_conversion(make_df): """Test different output units give consistent results with correct conversion.""" - X = pd.DataFrame({ - "lat1": [40.7128], - "lon1": [-74.0060], - "lat2": [34.0522], - "lon2": [-118.2437], - }) + data = COORDS_DATA transformer_km = GeoDistanceFeatures( lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2", output_unit="km" @@ -326,8 +341,10 @@ def test_output_units_conversion(): lat1="lat1", lon1="lon1", lat2="lat2", lon2="lon2", output_unit="miles" ) - dist_km = transformer_km.fit_transform(X.copy())["geo_distance"].iloc[0] - dist_miles = transformer_miles.fit_transform(X.copy())["geo_distance"].iloc[0] + dist_km = get_value(transformer_km.fit_transform(make_df(data)), "geo_distance") + dist_miles = get_value( + transformer_miles.fit_transform(make_df(data)), "geo_distance" + ) expected_miles = dist_km * 0.621371 np.testing.assert_almost_equal(dist_miles, expected_miles, decimal=0) @@ -356,7 +373,5 @@ def test_more_tags_and_sklearn_tags(): == "transformer has mandatory parameters" ) - # basic check for sklearn tags if available (new sklearn versions) - if hasattr(transformer, "__sklearn_tags__"): - tags = transformer.__sklearn_tags__() - assert tags is not None + tags = transformer.__sklearn_tags__() + assert tags is not None diff --git a/tests/test_creation/test_math_features.py b/tests/test_creation/test_math_features.py index 9c4c6b10c..1c695ca88 100644 --- a/tests/test_creation/test_math_features.py +++ b/tests/test_creation/test_math_features.py @@ -1,13 +1,35 @@ import warnings +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.pipeline import Pipeline from feature_engine.creation import MathFeatures -dob_datrange = pd.date_range("2020-02-24", periods=4, freq="min") +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} + + +def _none_to_nan(values): + # Missing values print as None for polars, NaN for pandas float columns + # - both mean "missing" here, so normalize both sides before comparing. + return [np.nan if v is None else v for v in values] + + +def assert_df_equal(X, expected: dict, abs_tol: float = 1e-5) -> None: + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + assert list(result.keys()) == list(expected.keys()) + for col, values in expected.items(): + assert _none_to_nan(result[col]) == pytest.approx( + _none_to_nan(values), abs=abs_tol, nan_ok=True + ) # test param variables_to_combine @@ -83,138 +105,95 @@ def test_error_new_variable_names_not_permitted(): ) -def test_aggregations_with_strings(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_aggregations_with_strings(make_df): + df = make_df(DATA) transformer = MathFeatures( variables=["Age", "Marks"], func=["sum", "prod", "mean", "std", "max", "min"] ) - X = transformer.fit_transform(df_vartypes) + Xt = transformer.fit_transform(df) - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": dob_datrange, - "sum_Age_Marks": [20.9, 21.8, 19.7, 18.6], - "prod_Age_Marks": [18.0, 16.8, 13.299999999999999, 10.799999999999999], - "mean_Age_Marks": [10.45, 10.9, 9.85, 9.3], - "std_Age_Marks": [ - 13.505739520663058, - 14.28355697996826, - 12.94005409571382, - 12.303657992645928, - ], - "max_Age_Marks": [20.0, 21.0, 19.0, 18.0], - "min_Age_Marks": [0.9, 0.8, 0.7, 0.6], - } - ) + expected = dict(DATA) + expected["sum_Age_Marks"] = [20.9, 21.8, 19.7, 18.6] + expected["prod_Age_Marks"] = [18.0, 16.8, 13.3, 10.8] + expected["mean_Age_Marks"] = [10.45, 10.9, 9.85, 9.3] + expected["std_Age_Marks"] = [13.505740, 14.283557, 12.940054, 12.303658] + expected["max_Age_Marks"] = [20.0, 21.0, 19.0, 18.0] + expected["min_Age_Marks"] = [0.9, 0.8, 0.7, 0.6] - # transform params - pd.testing.assert_frame_equal(X, ref) + assert_df_equal(Xt, expected) -def test_aggregations_with_functions(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_aggregations_with_functions(make_df): + df = make_df(DATA) transformer = MathFeatures( variables=["Age", "Marks"], func=[np.sum, np.mean, np.std] ) - X = transformer.fit_transform(df_vartypes) + Xt = transformer.fit_transform(df) - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": dob_datrange, - "sum_Age_Marks": [20.9, 21.8, 19.7, 18.6], - "mean_Age_Marks": [10.45, 10.9, 9.85, 9.3], - "std_Age_Marks": [ - 13.505739520663058, - 14.28355697996826, - 12.94005409571382, - 12.303657992645928, - ], - } - ) + expected = dict(DATA) + expected["sum_Age_Marks"] = [20.9, 21.8, 19.7, 18.6] + expected["mean_Age_Marks"] = [10.45, 10.9, 9.85, 9.3] - # TODO: Remove pandas < 3 support when dropping older pandas versions - # In pandas >=3, when the user passes np.std, agg will use numpy. - # In pandas <3, when the user passes np.std, agg will use pd.std. - # Hence the difference in results - if pd.__version__ >= "3": - ref["std_Age_Marks"] = np.std(df_vartypes[["Age", "Marks"]], axis=1) + # np.std uses ddof=0 (population std) everywhere now, except pandas < 3, + # where agg() still routes np.std through pandas' own ddof=1 Series.std(). + # TODO: remove the pandas < 3 branch when dropping older pandas support. + if make_df is pd.DataFrame and int(pd.__version__.split(".")[0]) < 3: + expected["std_Age_Marks"] = [13.505740, 14.283557, 12.940054, 12.303658] + else: + arr = np.array([DATA["Age"], DATA["Marks"]], dtype=float) + expected["std_Age_Marks"] = np.std(arr, axis=0).tolist() - # transform params - pd.testing.assert_frame_equal(X, ref) + assert_df_equal(Xt, expected) -def test_user_enters_two_operations(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_user_enters_two_operations(make_df): + df = make_df(DATA) transformer = MathFeatures(variables=["Age", "Marks"], func=["sum", np.mean]) + Xt = transformer.fit_transform(df) - X = transformer.fit_transform(df_vartypes) + expected = dict(DATA) + expected["sum_Age_Marks"] = [20.9, 21.8, 19.7, 18.6] + expected["mean_Age_Marks"] = [10.45, 10.9, 9.85, 9.3] - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": dob_datrange, - "sum_Age_Marks": [20.9, 21.8, 19.7, 18.6], - "mean_Age_Marks": [10.45, 10.9, 9.85, 9.3], - } - ) - - pd.testing.assert_frame_equal(X, ref) + assert_df_equal(Xt, expected) -def test_new_variable_names(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_new_variable_names(make_df): + df = make_df(DATA) transformer = MathFeatures( variables=["Age", "Marks"], func=["sum", "mean"], new_variables_names=["sum_of_two_vars", "mean_of_two_vars"], ) + Xt = transformer.fit_transform(df) - X = transformer.fit_transform(df_vartypes) + expected = dict(DATA) + expected["sum_of_two_vars"] = [20.9, 21.8, 19.7, 18.6] + expected["mean_of_two_vars"] = [10.45, 10.9, 9.85, 9.3] - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": dob_datrange, - "sum_of_two_vars": [20.9, 21.8, 19.7, 18.6], - "mean_of_two_vars": [10.45, 10.9, 9.85, 9.3], - } - ) + assert_df_equal(Xt, expected) - pd.testing.assert_frame_equal(X, ref) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_one_mathematical_operation(make_df): + df = make_df(DATA) + expected = dict(DATA) + expected["sum_Age_Marks"] = [20.9, 21.8, 19.7, 18.6] -def test_one_mathematical_operation(df_vartypes): transformer = MathFeatures(variables=["Age", "Marks"], func="sum") - X = transformer.fit_transform(df_vartypes) - - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": dob_datrange, - "sum_Age_Marks": [20.9, 21.8, 19.7, 18.6], - } - ) - pd.testing.assert_frame_equal(X, ref) + assert_df_equal(transformer.fit_transform(df), expected) transformer = MathFeatures(variables=["Age", "Marks"], func=["sum"]) - X = transformer.fit_transform(df_vartypes) - pd.testing.assert_frame_equal(X, ref) + assert_df_equal(transformer.fit_transform(df), expected) def test_variable_names_when_df_cols_are_integers(df_numeric_columns): + # polars requires string column names, so int-named columns are + # pandas-only - no polars equivalent to parametrize against here. transformer = MathFeatures( variables=[2, 3], func=["sum", "prod", "mean", "std", "max", "min"] ) @@ -227,7 +206,7 @@ def test_variable_names_when_df_cols_are_integers(df_numeric_columns): 1: ["London", "Manchester", "Liverpool", "Bristol"], 2: [20, 21, 19, 18], 3: [0.9, 0.8, 0.7, 0.6], - 4: dob_datrange, + 4: pd.date_range("2020-02-24", periods=4, freq="min"), "sum_2_3": [20.9, 21.8, 19.7, 18.6], "prod_2_3": [18.0, 16.8, 13.299999999999999, 10.799999999999999], "mean_2_3": [10.45, 10.9, 9.85, 9.3], @@ -245,9 +224,11 @@ def test_variable_names_when_df_cols_are_integers(df_numeric_columns): pd.testing.assert_frame_equal(X, ref) -def test_error_when_null_values_in_variable(df_vartypes): - df_na = df_vartypes.copy() - df_na.loc[1, "Age"] = np.nan +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_when_null_values_in_variable(make_df): + data_na = dict(DATA) + data_na["Age"] = [20, None, 19, 18] + df_na = make_df(data_na) math_combinator = MathFeatures( variables=["Age", "Marks"], @@ -258,65 +239,65 @@ def test_error_when_null_values_in_variable(df_vartypes): with pytest.raises(ValueError): math_combinator.fit(df_na) - math_combinator.fit(df_vartypes) + math_combinator.fit(make_df(DATA)) with pytest.raises(ValueError): math_combinator.transform(df_na) -def test_no_error_when_null_values_in_variable(df_vartypes): - df_na = df_vartypes.copy() - df_na.loc[1, "Age"] = np.nan +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_no_error_when_null_values_in_variable(make_df): + data_na = dict(DATA) + data_na["Age"] = [20, None, 19, 18] + df_na = make_df(data_na) transformer = MathFeatures( variables=["Age", "Marks"], func=["sum", "mean"], missing_values="ignore", ) + Xt = transformer.fit_transform(df_na) - X = transformer.fit_transform(df_na) + expected = dict(data_na) + expected["sum_Age_Marks"] = [20.9, 0.8, 19.7, 18.6] + expected["mean_Age_Marks"] = [10.45, 0.8, 9.85, 9.3] - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, np.nan, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": dob_datrange, - "sum_Age_Marks": [20.9, 0.8, 19.7, 18.6], - "mean_Age_Marks": [10.45, 0.8, 9.85, 9.3], - } - ) - # transform params - pd.testing.assert_frame_equal(X, ref) + assert_df_equal(Xt, expected) -def test_standard_aggregations_match_pandas_with_missing_values(): - X = pd.DataFrame( - { - "a": [1.0, np.nan, np.nan, 4.0], - "b": [3.0, 4.0, np.nan, 6.0], - "c": [5.0, 8.0, np.nan, np.nan], - } - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_standard_aggregations_match_pandas_with_missing_values(make_df): + data = { + "a": [1.0, np.nan, np.nan, 4.0], + "b": [3.0, 4.0, np.nan, 6.0], + "c": [5.0, 8.0, np.nan, np.nan], + } functions = ["sum", "mean", "std", "var", "min", "max", "prod", "median"] names = [f"result_{function}" for function in functions] + + # pandas' own agg() is the ground truth both backends are checked against. + X_pd = pd.DataFrame(data) with warnings.catch_warnings(): warnings.simplefilter("ignore", RuntimeWarning) - expected = X.agg(functions, axis=1) - expected.columns = names + expected_df = X_pd.agg(functions, axis=1) + expected = {name: expected_df[fn].tolist() for name, fn in zip(names, functions)} + df = make_df(data) transformer = MathFeatures( - variables=list(X.columns), + variables=list(data.keys()), func=functions, new_variables_names=names, missing_values="ignore", ) - result = transformer.fit_transform(X) + result = transformer.fit_transform(df) - pd.testing.assert_frame_equal(result[names], expected) + result_dict = nw.from_native(result, eager_only=True).to_dict(as_series=False) + for name in names: + assert result_dict[name] == pytest.approx(expected[name], nan_ok=True) def test_nullable_dtypes_use_backwards_compatible_aggregation(): + # pandas' nullable "Int64" dtype is pandas-specific - no polars + # equivalent to parametrize against here. X = pd.DataFrame( { "a": pd.Series([1, pd.NA, 3], dtype="Int64"), @@ -339,65 +320,111 @@ def test_nullable_dtypes_use_backwards_compatible_aggregation(): pd.testing.assert_frame_equal(result[names], expected) -def test_custom_function_uses_pandas_aggregation_fallback(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_custom_function_fallback(make_df): + # max()/min()/sum() are built-ins, so they work identically whether + # func receives a pandas Series (pandas' agg(axis=1) fallback) or a + # plain tuple (polars' map_rows fallback) - one callable, one test. def peak_to_peak(row): - return row.max() - row.min() + return max(row) - min(row) - expected = df_vartypes[["Age", "Marks"]].agg(peak_to_peak, axis=1) + df = make_df(DATA) transformer = MathFeatures( variables=["Age", "Marks"], func=peak_to_peak, new_variables_names=["age_marks_range"], ) + Xt = transformer.fit_transform(df) - result = transformer.fit_transform(df_vartypes) + expected = dict(DATA) + expected["age_marks_range"] = [ + a - m for a, m in zip(DATA["Age"], DATA["Marks"]) + ] + assert_df_equal(Xt, expected) - pd.testing.assert_series_equal( - result["age_marks_range"], expected, check_names=False - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_multiple_custom_functions_fallback(make_df): + def total(row): + return sum(row) -def test_drop_original_variables(df_vartypes): + def spread(row): + return max(row) - min(row) + + df = make_df(DATA) + transformer = MathFeatures( + variables=["Age", "Marks"], + func=[total, spread], + new_variables_names=["total", "spread"], + ) + Xt = transformer.fit_transform(df) + + expected = dict(DATA) + expected["total"] = [a + m for a, m in zip(DATA["Age"], DATA["Marks"])] + expected["spread"] = [a - m for a, m in zip(DATA["Age"], DATA["Marks"])] + assert_df_equal(Xt, expected) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_uncommon_aggregation_string_only_supported_for_pandas(make_df): + # a genuine, documented backend asymmetry, not an oversight: pandas' + # agg() accepts any of its own aggregation strings (even ones outside + # our NumPy-vectorized table), but polars has no way to resolve an + # arbitrary pandas-specific string without pandas itself, so it raises + # instead of silently doing the wrong thing. + df = make_df(DATA) + transformer = MathFeatures(variables=["Age", "Marks"], func="sem") + + if make_df is pd.DataFrame: + Xt = transformer.fit_transform(df) + expected = dict(DATA) + expected["sem_Age_Marks"] = [9.55, 10.10, 9.15, 8.70] + assert_df_equal(Xt, expected) + else: + with pytest.raises(NotImplementedError, match="has no NumPy-vectorized"): + transformer.fit_transform(df) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_drop_original_variables(make_df): + df = make_df(DATA) transformer = MathFeatures( variables=["Age", "Marks"], func=["sum", "mean"], drop_original=True, ) + Xt = transformer.fit_transform(df) - X = transformer.fit_transform(df_vartypes) - - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "dob": dob_datrange, - "sum_Age_Marks": [20.9, 21.8, 19.7, 18.6], - "mean_Age_Marks": [10.45, 10.9, 9.85, 9.3], - } - ) - - pd.testing.assert_frame_equal(X, ref) + expected = { + "Name": DATA["Name"], + "City": DATA["City"], + "sum_Age_Marks": [20.9, 21.8, 19.7, 18.6], + "mean_Age_Marks": [10.45, 10.9, 9.85, 9.3], + } + assert_df_equal(Xt, expected) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("_varnames", [None, ["var1", "var2"]]) @pytest.mark.parametrize("_drop", [True, False]) -def test_get_feature_names_out(_varnames, _drop, df_vartypes): +def test_get_feature_names_out(make_df, _varnames, _drop): + df = make_df(DATA) tr = MathFeatures( variables=["Age", "Marks"], func=["sum", "mean"], new_variables_names=_varnames, drop_original=_drop, ) - X = tr.fit_transform(df_vartypes) - feat_out = list(X.columns) + Xt = tr.fit_transform(df) + feat_out = list(nw.from_native(Xt, eager_only=True).columns) assert tr.get_feature_names_out(input_features=None) == feat_out - assert tr.get_feature_names_out(input_features=df_vartypes.columns) == feat_out +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("_varnames", [None, ["var1", "var2"]]) @pytest.mark.parametrize("_drop", [True, False]) -def test_get_feature_names_out_from_pipeline(_varnames, _drop, df_vartypes): - # set up transformer +def test_get_feature_names_out_from_pipeline(make_df, _varnames, _drop): + df = make_df(DATA) transformer = MathFeatures( variables=["Age", "Marks"], func=["sum", "mean"], @@ -406,24 +433,21 @@ def test_get_feature_names_out_from_pipeline(_varnames, _drop, df_vartypes): ) pipe = Pipeline([("transformer", transformer)]) + Xt = pipe.fit_transform(df) - # fit transformer - X = pipe.fit_transform(df_vartypes) - - feat_out = list(X.columns) + feat_out = list(nw.from_native(Xt, eager_only=True).columns) assert pipe.get_feature_names_out(input_features=None) == feat_out - assert pipe.get_feature_names_out(input_features=df_vartypes.columns) == feat_out +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("_input_features", ["hola", ["Age", "Marks"]]) -def test_get_feature_names_out_raises_error_when_wrong_param( - _input_features, df_vartypes -): +def test_get_feature_names_out_raises_error_when_wrong_param(make_df, _input_features): + df = make_df(DATA) transformer = MathFeatures( variables=["Age", "Marks"], func=["sum", "mean"], ) - transformer.fit(df_vartypes) + transformer.fit(df) with pytest.raises(ValueError): transformer.get_feature_names_out(input_features=_input_features) diff --git a/tests/test_creation/test_relative_features.py b/tests/test_creation/test_relative_features.py index dbfa4972c..e8dc5971c 100644 --- a/tests/test_creation/test_relative_features.py +++ b/tests/test_creation/test_relative_features.py @@ -1,10 +1,34 @@ +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.pipeline import Pipeline from feature_engine.creation import RelativeFeatures +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} + + +def _none_to_nan(values): + # Missing values print as None for polars, NaN for pandas float columns + # - both mean "missing" here, so normalize both sides before comparing. + return [np.nan if v is None else v for v in values] + + +def assert_df_equal(X, expected: dict, abs_tol: float = 1e-5) -> None: + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + assert list(result.keys()) == list(expected.keys()) + for col, values in expected.items(): + assert _none_to_nan(result[col]) == pytest.approx( + _none_to_nan(values), abs=abs_tol, nan_ok=True + ) + def test_mandatory_init_parameters(): with pytest.raises(TypeError): @@ -75,14 +99,16 @@ def test_error_when_drop_original_not_bool(): ) -def test_error_when_variables_not_numeric(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_when_variables_not_numeric(make_df): + df = make_df(DATA) transformer = RelativeFeatures( variables=["Name", "Age", "Marks"], reference=["Age", "Name"], func=["sub"], ) with pytest.raises(TypeError): - transformer.fit_transform(df_vartypes) + transformer.fit_transform(df) transformer = RelativeFeatures( reference=["Name", "Age", "Marks"], @@ -90,17 +116,19 @@ def test_error_when_variables_not_numeric(df_vartypes): func=["sub"], ) with pytest.raises(TypeError): - transformer.fit_transform(df_vartypes) + transformer.fit_transform(df) -def test_error_when_entered_variables_not_in_df(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_when_entered_variables_not_in_df(make_df): + df = make_df(DATA) transformer = RelativeFeatures( variables=["FeatOutsideDataset", "Age"], reference=["Age", "Name"], func=["sub"], ) with pytest.raises(KeyError): - transformer.fit_transform(df_vartypes) + transformer.fit_transform(df) transformer = RelativeFeatures( reference=["FeatOutsideDataset", "Age"], @@ -108,146 +136,126 @@ def test_error_when_entered_variables_not_in_df(df_vartypes): func=["sub"], ) with pytest.raises(TypeError): - transformer.fit_transform(df_vartypes) - + transformer.fit_transform(df) -def test_classic_binary_operation(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_classic_binary_operation(make_df): + df = make_df(DATA) transformer = RelativeFeatures( variables=["Age"], reference=["Marks"], func=["sub", "div", "add", "mul"], ) + Xt = transformer.fit_transform(df) - X = transformer.fit_transform(df_vartypes) + expected = dict(DATA) + expected["Age_sub_Marks"] = [19.1, 20.2, 18.3, 17.4] + expected["Age_div_Marks"] = [22.22222222222222, 26.25, 27.142857142857146, 30.0] + expected["Age_add_Marks"] = [20.9, 21.8, 19.7, 18.6] + expected["Age_mul_Marks"] = [18.0, 16.8, 13.3, 10.8] - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - "Age_sub_Marks": [19.1, 20.2, 18.3, 17.4], - "Age_div_Marks": [22.22222222222222, 26.25, 27.142857142857146, 30.0], - "Age_add_Marks": [20.9, 21.8, 19.7, 18.6], - "Age_mul_Marks": [18.0, 16.8, 13.299999999999999, 10.799999999999999], - } - ) - - pd.testing.assert_frame_equal(X, ref) - - -def test_alternative_operation(df_vartypes): + assert_df_equal(Xt, expected) - # input df - df = df_vartypes.copy() - - # Expected result - dft = df.copy() - dft["Age_truediv_Marks"] = dft["Age"].truediv(dft["Marks"]) - dft["Age_floordiv_Marks"] = dft["Age"].floordiv(dft["Marks"]) - dft["Age_mod_Marks"] = dft["Age"].mod(dft["Marks"]) - dft["Age_pow_Marks"] = dft["Age"].pow(dft["Marks"]) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_alternative_operation(make_df): + df = make_df(DATA) transformer = RelativeFeatures( variables=["Age"], reference=["Marks"], func=["truediv", "floordiv", "mod", "pow"], ) - X = transformer.fit_transform(df) + Xt = transformer.fit_transform(df) + + expected = dict(DATA) + expected["Age_truediv_Marks"] = [22.22222222222222, 26.25, 27.142857142857146, 30.0] + expected["Age_floordiv_Marks"] = [22.0, 26.0, 27.0, 30.0] + expected["Age_mod_Marks"] = [ + 0.1999999999999995, + 0.19999999999999885, + 0.1000000000000012, + 6.661338147750939e-16, + ] + expected["Age_pow_Marks"] = [ + 14.822688982138954, + 11.42287530066645, + 7.85466234994081, + 5.664525067769412, + ] - pd.testing.assert_frame_equal(X, dft) + assert_df_equal(Xt, expected) -def test_operations_with_multiple_variables(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_operations_with_multiple_variables(make_df): + df = make_df(DATA) transformer = RelativeFeatures( variables=["Age", "Marks"], reference=["Age", "Marks"], func=["sub"], ) + Xt = transformer.fit_transform(df) - X = transformer.fit_transform(df_vartypes) + expected = dict(DATA) + expected["Age_sub_Age"] = [0, 0, 0, 0] + expected["Marks_sub_Age"] = [-19.1, -20.2, -18.3, -17.4] + expected["Age_sub_Marks"] = [19.1, 20.2, 18.3, 17.4] + expected["Marks_sub_Marks"] = [0.0, 0.0, 0.0, 0.0] - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - "Age_sub_Age": [0, 0, 0, 0], - "Marks_sub_Age": [-19.1, -20.2, -18.3, -17.4], - "Age_sub_Marks": [19.1, 20.2, 18.3, 17.4], - "Marks_sub_Marks": [0.0, 0.0, 0.0, 0.0], - } - ) + assert_df_equal(Xt, expected) - pd.testing.assert_frame_equal(X, ref) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_multiple_operations_with_multiple_variables(make_df): + df = make_df(DATA) -def test_multiple_operations_with_multiple_variables(df_vartypes): + # column order follows func order: sub's 4 columns, then add's 4 transformer = RelativeFeatures( variables=["Age", "Marks"], reference=["Age", "Marks"], func=["sub", "add"], ) + Xt = transformer.fit_transform(df) - X = transformer.fit_transform(df_vartypes) - - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - "Age_sub_Age": [0, 0, 0, 0], - "Marks_sub_Age": [-19.1, -20.2, -18.3, -17.4], - "Age_sub_Marks": [19.1, 20.2, 18.3, 17.4], - "Marks_sub_Marks": [0.0, 0.0, 0.0, 0.0], - "Age_add_Age": [40, 42, 38, 36], - "Marks_add_Age": [20.9, 21.8, 19.7, 18.6], - "Age_add_Marks": [20.9, 21.8, 19.7, 18.6], - "Marks_add_Marks": [1.8, 1.6, 1.4, 1.2], - } - ) + expected = dict(DATA) + expected["Age_sub_Age"] = [0, 0, 0, 0] + expected["Marks_sub_Age"] = [-19.1, -20.2, -18.3, -17.4] + expected["Age_sub_Marks"] = [19.1, 20.2, 18.3, 17.4] + expected["Marks_sub_Marks"] = [0.0, 0.0, 0.0, 0.0] + expected["Age_add_Age"] = [40, 42, 38, 36] + expected["Marks_add_Age"] = [20.9, 21.8, 19.7, 18.6] + expected["Age_add_Marks"] = [20.9, 21.8, 19.7, 18.6] + expected["Marks_add_Marks"] = [1.8, 1.6, 1.4, 1.2] - pd.testing.assert_frame_equal(X, ref) + assert_df_equal(Xt, expected) + # reversing func order reverses the corresponding column block order transformer = RelativeFeatures( variables=["Age", "Marks"], reference=["Age", "Marks"], func=["add", "sub"], ) + Xt = transformer.fit_transform(df) - X = transformer.fit_transform(df_vartypes) - - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - "Age_add_Age": [40, 42, 38, 36], - "Marks_add_Age": [20.9, 21.8, 19.7, 18.6], - "Age_add_Marks": [20.9, 21.8, 19.7, 18.6], - "Marks_add_Marks": [1.8, 1.6, 1.4, 1.2], - "Age_sub_Age": [0, 0, 0, 0], - "Marks_sub_Age": [-19.1, -20.2, -18.3, -17.4], - "Age_sub_Marks": [19.1, 20.2, 18.3, 17.4], - "Marks_sub_Marks": [0.0, 0.0, 0.0, 0.0], - } - ) - - pd.testing.assert_frame_equal(X, ref) + expected = dict(DATA) + expected["Age_add_Age"] = [40, 42, 38, 36] + expected["Marks_add_Age"] = [20.9, 21.8, 19.7, 18.6] + expected["Age_add_Marks"] = [20.9, 21.8, 19.7, 18.6] + expected["Marks_add_Marks"] = [1.8, 1.6, 1.4, 1.2] + expected["Age_sub_Age"] = [0, 0, 0, 0] + expected["Marks_sub_Age"] = [-19.1, -20.2, -18.3, -17.4] + expected["Age_sub_Marks"] = [19.1, 20.2, 18.3, 17.4] + expected["Marks_sub_Marks"] = [0.0, 0.0, 0.0, 0.0] + assert_df_equal(Xt, expected) -def test_when_missing_values_is_ignore(df_vartypes): - df_na = df_vartypes.copy() - df_na.loc[1, "Age"] = np.nan +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_when_missing_values_is_ignore(make_df): + data_na = dict(DATA) + data_na["Age"] = [20, None, 19, 18] + df_na = make_df(data_na) transformer = RelativeFeatures( variables=["Age", "Marks"], @@ -255,30 +263,22 @@ def test_when_missing_values_is_ignore(df_vartypes): func=["sub"], missing_values="ignore", ) + Xt = transformer.fit_transform(df_na) - X = transformer.fit_transform(df_na) - - ref = pd.DataFrame.from_dict( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, np.nan, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - "Age_sub_Age": [0, np.nan, 0, 0], - "Marks_sub_Age": [-19.1, np.nan, -18.3, -17.4], - "Age_sub_Marks": [19.1, np.nan, 18.3, 17.4], - "Marks_sub_Marks": [0.0, 0.0, 0.0, 0.0], - } - ) - - pd.testing.assert_frame_equal(X, ref) + expected = dict(data_na) + expected["Age_sub_Age"] = [0, np.nan, 0, 0] + expected["Marks_sub_Age"] = [-19.1, np.nan, -18.3, -17.4] + expected["Age_sub_Marks"] = [19.1, np.nan, 18.3, 17.4] + expected["Marks_sub_Marks"] = [0.0, 0.0, 0.0, 0.0] + assert_df_equal(Xt, expected) -def test_error_when_null_values_in_variable(df_vartypes): - df_na = df_vartypes.copy() - df_na.loc[1, "Age"] = np.nan +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_when_null_values_in_variable(make_df): + data_na = dict(DATA) + data_na["Age"] = [20, None, 19, 18] + df_na = make_df(data_na) transformer = RelativeFeatures( variables=["Age", "Marks"], @@ -290,14 +290,16 @@ def test_error_when_null_values_in_variable(df_vartypes): with pytest.raises(ValueError): transformer.fit(df_na) - transformer.fit(df_vartypes) + transformer.fit(make_df(DATA)) with pytest.raises(ValueError): transformer.transform(df_na) -def test_when_df_cols_are_integers(df_vartypes): - df = df_vartypes.copy() - df.columns = [0, 1, 2, 3, 4] +def test_when_df_cols_are_integers(): + # polars requires string column names, so int-named columns are + # pandas-only - no polars equivalent to parametrize against here. + df = pd.DataFrame(DATA) + df.columns = [0, 1, 2, 3] transformer = RelativeFeatures( variables=[2, 3], @@ -313,7 +315,6 @@ def test_when_df_cols_are_integers(df_vartypes): 1: ["London", "Manchester", "Liverpool", "Bristol"], 2: [20, 21, 19, 18], 3: [0.9, 0.8, 0.7, 0.6], - 4: pd.date_range("2020-02-24", periods=4, freq="min"), "2_sub_2": [0, 0, 0, 0], "3_sub_2": [-19.1, -20.2, -18.3, -17.4], "2_sub_3": [19.1, 20.2, 18.3, 17.4], @@ -328,31 +329,30 @@ def test_when_df_cols_are_integers(df_vartypes): pd.testing.assert_frame_equal(X, ref) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("_func", [["div"], ["truediv"], ["floordiv"], ["mod"]]) -def test_error_when_division_by_zero_and_fill_value_is_none(_func, df_vartypes): - - df_zero = df_vartypes.copy() - df_zero.loc[1, "Marks"] = 0 +def test_error_when_division_by_zero_and_fill_value_is_none(make_df, _func): + data_zero = dict(DATA) + data_zero["Marks"] = [0.9, 0, 0.7, 0.6] + df_zero = make_df(data_zero) transformer = RelativeFeatures( variables=["Age"], reference=["Marks"], func=_func, ) - transformer.fit(df_vartypes) - - with pytest.raises(ValueError) as record: - transformer.transform(df_zero) + transformer.fit(make_df(DATA)) msg = ( "Some of the reference variables contain zeroes. Division by zero " "does not exist. Replace zeros before using this transformer for division " "or set `fill_value` to a number." ) - # check that the error message matches - assert str(record.value) == msg + with pytest.raises(ValueError, match=msg): + transformer.transform(df_zero) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "_fill_value, _func", [ @@ -366,11 +366,13 @@ def test_error_when_division_by_zero_and_fill_value_is_none(_func, df_vartypes): (999, ["mod"]), ], ) -def test_fill_values_when_division_by_zero(_fill_value, _func, df_vartypes): - df_zero = df_vartypes.copy() - df_zero.loc[2, "Marks"] = 0 - df_zero.loc[1, "Age"] = np.nan - df_zero.loc[3, "Age"] = np.inf +def test_fill_values_when_division_by_zero(make_df, _fill_value, _func): + data_zero = dict(DATA) + data_zero["Marks"] = [0.9, 0.8, 0, 0.6] + # Age must be float from the start: polars can't build an Int64 column + # from a mix of ints and NaN/inf the way pandas silently upcasts to. + data_zero["Age"] = [20.0, np.nan, 19.0, np.inf] + df_zero = make_df(data_zero) transformer = RelativeFeatures( variables=["Age"], @@ -379,18 +381,20 @@ def test_fill_values_when_division_by_zero(_fill_value, _func, df_vartypes): func=_func, missing_values="ignore", ) - - X = transformer.fit_transform(df_zero) + Xt = transformer.fit_transform(df_zero) new_var = f"Age_{_func[0]}_Marks" + result = nw.from_native(Xt, eager_only=True).to_dict(as_series=False) - assert X.loc[2, new_var] == _fill_value - np.testing.assert_equal(X.loc[1, "Age"], np.nan) - np.testing.assert_equal(X.loc[3, "Age"], np.inf) + assert result[new_var][2] == pytest.approx(_fill_value) + np.testing.assert_equal(result["Age"][1], np.nan) + np.testing.assert_equal(result["Age"][3], np.inf) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("_drop", [True, False]) -def test_get_feature_names_out(_drop, df_vartypes): +def test_get_feature_names_out(make_df, _drop): + df = make_df(DATA) transformer = RelativeFeatures( variables=["Age", "Marks"], reference=["Age", "Marks"], @@ -407,55 +411,88 @@ def test_get_feature_names_out(_drop, df_vartypes): "Age_sub_Marks", "Marks_sub_Marks", ] - X = transformer.fit_transform(df_vartypes) - feat_out = list(X.columns) + Xt = transformer.fit_transform(df) + feat_out = list(nw.from_native(Xt, eager_only=True).columns) assert feat_out == transformer.get_feature_names_out(input_features=None) - assert feat_out == transformer.get_feature_names_out( - input_features=df_vartypes.columns - ) assert all([f for f in varnames if f in feat_out]) + if _drop is True: + # drop_original only drops columns that are in variables/reference + # (here Age, Marks) - Name and City are neither, so they remain. + assert feat_out == ["Name", "City"] + varnames + else: + assert feat_out == list(DATA.keys()) + varnames +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("_drop", [True, False]) -def test_get_feature_names_out_from_pipeline(_drop, df_vartypes): +def test_get_feature_names_out_from_pipeline(make_df, _drop): + df = make_df(DATA) transformer = RelativeFeatures( variables=["Age", "Marks"], reference=["Age", "Marks"], func=["add", "sub"], drop_original=_drop, ) - pipe = Pipeline([("transformer", transformer)]) - varnames = [ - "Age_add_Age", - "Marks_add_Age", - "Age_add_Marks", - "Marks_add_Marks", - "Age_sub_Age", - "Marks_sub_Age", - "Age_sub_Marks", - "Marks_sub_Marks", - ] - - X = pipe.fit_transform(df_vartypes) - assert list(X.columns) == pipe.get_feature_names_out(input_features=None) - assert list(X.columns) == pipe.get_feature_names_out( - input_features=df_vartypes.columns - ) - assert all([f for f in varnames if f in X.columns]) + Xt = pipe.fit_transform(df) + feat_out = list(nw.from_native(Xt, eager_only=True).columns) + assert feat_out == pipe.get_feature_names_out(input_features=None) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("_input_features", ["hola", ["Age", "Marks"]]) -def test_get_feature_names_out_raises_error_when_wrong_param( - _input_features, df_vartypes -): +def test_get_feature_names_out_raises_error_when_wrong_param(make_df, _input_features): + df = make_df(DATA) transformer = RelativeFeatures( variables=["Age", "Marks"], reference=["Age", "Marks"], func=["add", "sub"], ) - transformer.fit(df_vartypes) + transformer.fit(df) with pytest.raises(ValueError): transformer.get_feature_names_out(input_features=_input_features) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_mixed_int_float_variables_preserve_own_dtype(make_df): + # a regression check: extracting variables as one batched 2D array + # upcasts everything to a common dtype, losing e.g. an int column's + # own int result for subtraction. Each variable must keep its own + # dtype promotion, independent of the other variables in the list. + df = make_df(DATA) + transformer = RelativeFeatures( + variables=["Age", "Marks"], reference=["Age"], func=["sub"] + ) + Xt = transformer.fit_transform(df) + nw_Xt = nw.from_native(Xt, eager_only=True) + assert nw_Xt.get_column("Age_sub_Age").dtype.is_integer() + assert not nw_Xt.get_column("Marks_sub_Age").dtype.is_integer() + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_floordiv_zero_with_float_fill_value_widens_dtype(make_df): + # floordiv on integer input stays integer-typed; a float fill_value + # must widen the result column rather than truncating or erroring, + # matching pandas' own automatic dtype promotion here. + df = make_df({"v": [7, 8], "ref": [0, 2]}) + transformer = RelativeFeatures( + variables=["v"], reference=["ref"], func=["floordiv"], fill_value=-1.5 + ) + Xt = transformer.fit_transform(df) + result = nw.from_native(Xt, eager_only=True).get_column("v_floordiv_ref").to_list() + assert result == pytest.approx([-1.5, 4.0]) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_drop_original_both_backends(make_df): + df = make_df({"x1": [1, 2, 3], "x2": [4, 5, 6], "x3": [3, 4, 5]}) + transformer = RelativeFeatures( + variables=["x1", "x2"], reference=["x3"], func=["div"], drop_original=True + ) + Xt = transformer.fit_transform(df) + assert list(nw.from_native(Xt, eager_only=True).columns) == [ + "x1_div_x3", + "x2_div_x3", + ] diff --git a/tests/test_dataframe_checks.py b/tests/test_dataframe_checks.py index 711b1aea7..bd630d6bc 100644 --- a/tests/test_dataframe_checks.py +++ b/tests/test_dataframe_checks.py @@ -1,304 +1,465 @@ +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from pandas.testing import assert_frame_equal, assert_series_equal +from polars.testing import assert_frame_equal as pl_assert_frame_equal +from polars.testing import assert_series_equal as pl_assert_series_equal from scipy.sparse import csr_matrix from feature_engine.dataframe_checks import ( _check_contains_inf, _check_contains_na, - _check_optional_contains_na, _check_X_matches_training_df, check_X, check_X_y, check_y, ) +# ------------------------ +# test check_X +# ------------------------ -def test_check_X_returns_df(df_vartypes): - assert_frame_equal(check_X(df_vartypes), df_vartypes) +@pytest.mark.parametrize( + "make_df, assert_equal_fn", + [(pd.DataFrame, assert_frame_equal), (pl.DataFrame, pl_assert_frame_equal)], +) +def test_check_X_returns_df_unchanged(make_df, assert_equal_fn): + df = make_df({"a": [1, 2, 3], "b": [4.0, 5.0, 6.0]}) + X = check_X(df) + assert isinstance(X, nw.DataFrame) + assert_equal_fn(X.to_native(), df) -def test_check_X_converts_numpy_to_pandas(): - a1D = np.array([1, 2, 3, 4]) - a2D = np.array([[1, 2], [3, 4]]) - a3D = np.array([[[1, 2], [3, 4]], [[5, 6], [7, 8]]]) - - df_2D = pd.DataFrame(a2D, columns=["x0", "x1"]) - assert_frame_equal(df_2D, check_X(a2D)) +@pytest.mark.parametrize( + "make_df, assert_equal_fn", + [(pd.DataFrame, assert_frame_equal), (pl.DataFrame, pl_assert_frame_equal)], +) +def test_check_X_returns_df_with_mixed_dtypes(make_df, assert_equal_fn): + data = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "dob": pd.date_range("2020-02-24", periods=4, freq="min"), + } + df = make_df(data) + X = check_X(df) + assert isinstance(X, nw.DataFrame) + assert_equal_fn(X.to_native(), df) + + +@pytest.mark.parametrize( + "df", + [ + pd.DataFrame([]), + pd.DataFrame({"a": []}), + pl.DataFrame({"a": []}), + ], +) +def test_raises_error_if_empty_df(df): with pytest.raises(ValueError): - check_X(a3D) + check_X(df) + + +def test_check_X_raises_error_if_0_columns(): + # A dataframe with rows but no columns is not caught by `is_empty()`, which + # only looks at the row count, so it needs its own explicit check. Polars has + # no representation for "rows with 0 columns", so this case is pandas-only. + df = pd.DataFrame(index=range(3)) + assert df.shape == (3, 0) with pytest.raises(ValueError): - check_X(a1D) + check_X(df) -def test_check_X_raises_error_sparse_matrix(): - sparse_mx = csr_matrix([[5]]) - with pytest.raises(TypeError): - assert check_X(sparse_mx) +def test_check_X_raises_error_on_duplicated_column_names(): + # only relevant for pandas + df = pd.DataFrame( + { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + } + ) + df.columns = ["var_A", "var_A", "var_B", "var_C"] + msg = "Expected unique column names" + with pytest.raises(ValueError, match=msg): + check_X(df) -def test_check_X_raises_error_with_complex_data(): - msg = "Complex data not supported" - rng = np.random.RandomState(0) - X = rng.uniform(size=10) + 1j * rng.uniform(size=10) - X = X.reshape(-1, 1) - with pytest.raises(TypeError, match=msg): - assert check_X(X) +@pytest.mark.parametrize( + "X", + [ + np.array([[1, 2], [3, 4]]), + np.array([1, 2, 3]), + np.array(1), + [1, 2, 3], + {"a": [1, 2, 3]}, + "not a dataframe", + None, + csr_matrix([[1, 2], [3, 4]]), + ], +) +def test_check_X_raises_error_on_non_dataframe_input(X): + with pytest.raises(TypeError) as record: + check_X(X) + assert record.match("X must be a dataframe from a library supported by narwhals") -def test_raises_error_if_empty_df(): - df = pd.DataFrame([]) - with pytest.raises(ValueError): - check_X(df) +# ------------------------ +# test check_y +# ------------------------ + +# --- series input --- + + +@pytest.mark.parametrize( + "make_series, assert_equal_fn", + [(pd.Series, assert_series_equal), (pl.Series, pl_assert_series_equal)], +) +def test_check_y_series_returns_values_unchanged(make_series, assert_equal_fn): + s = make_series([0, 1, 2, 3, 4]) + assert_equal_fn(check_y(s), s) -def test_check_y_returns_series(): - s = pd.Series([0, 1, 2, 3, 4]) - assert_series_equal(check_y(s), s) +@pytest.mark.parametrize( + "make_series", + [pd.Series, pl.Series], +) +def test_check_y_series_raises_nan_error(make_series): + s = make_series([0.0, None, 2.0]) + with pytest.raises(ValueError, match="y contains NaN values."): + check_y(s) -def test_check_y_returns_dataframe(): - d = pd.DataFrame({"t1": [0, 1, 2, 3, 4], "t2": [5, 6, 7, 8, 9]}) - assert_frame_equal(check_y(d), d) +@pytest.mark.parametrize( + "make_series", + [pd.Series, pl.Series], +) +def test_check_y_series_raises_nan_error_for_explicit_nan(make_series): + # in polars, an explicit float("nan") is not a null value, so it is only + # caught if is_nan() is checked in addition to is_null() + s = make_series([0.0, float("nan"), 2.0]) + with pytest.raises(ValueError, match="y contains NaN values."): + check_y(s) -def test_check_y_converts_np_array(): - a1D = np.array([1, 2, 3, 4]) - s = pd.Series(a1D) - assert_series_equal(check_y(a1D), s) +@pytest.mark.parametrize( + "make_series", + [pd.Series, pl.Series], +) +def test_check_y_series_raises_inf_error(make_series): + s = make_series([0.0, float("inf"), 2.0]) + with pytest.raises(ValueError, match="y contains infinity values."): + check_y(s) -def test_check_y_converts_np_array_2D(): - a2D = np.array([1, 2, 3, 4, 5, 6, 7, 8]).reshape(2, 4) - d = pd.DataFrame(a2D) - assert_frame_equal(check_y(a2D), d) +@pytest.mark.parametrize( + "make_series, assert_equal_fn", + [(pd.Series, assert_series_equal), (pl.Series, pl_assert_series_equal)], +) +def test_check_y_series_converts_string_to_number_when_y_numeric( + make_series, assert_equal_fn +): + s = make_series(["0", "1", "2"]) + y = check_y(s, y_numeric=True) + expected = make_series([0.0, 1.0, 2.0]) + assert_equal_fn(y, expected) + + +@pytest.mark.parametrize( + "make_series, assert_equal_fn", + [(pd.Series, assert_series_equal), (pl.Series, pl_assert_series_equal)], +) +def test_check_y_series_leaves_non_numeric_unchanged_by_default( + make_series, assert_equal_fn +): + # y_numeric defaults to False: a non-numeric series (e.g. classification + # labels) should be returned as-is, without being cast to float. + s = make_series(["a", "b", "c"]) + assert_equal_fn(check_y(s), s) -def test_check_y_raises_none_error(): - with pytest.raises(ValueError): - check_y(None) +# --- dataframe (multioutput) input --- -def test_check_y_raises_nan_error(): - msg = "y contains NaN values." +@pytest.mark.parametrize( + "make_df, assert_equal_fn", + [(pd.DataFrame, assert_frame_equal), (pl.DataFrame, pl_assert_frame_equal)], +) +def test_check_y_dataframe_returns_values_unchanged(make_df, assert_equal_fn): + d = make_df({"t1": [0, 1, 2, 3, 4], "t2": [5, 6, 7, 8, 9]}) + assert_equal_fn(check_y(d), d) - # y is series - s = pd.Series([0, np.nan, 2, 3, 4]) - with pytest.raises(ValueError) as record: - check_y(s) - assert str(record.value) == msg - # y is multioutput - d = pd.DataFrame(np.array([1, np.nan, 3, 4, 5, 6, np.nan, 8]).reshape(2, 4)) - with pytest.raises(ValueError) as record: +@pytest.mark.parametrize( + "make_df", + [pd.DataFrame, pl.DataFrame], +) +def test_check_y_dataframe_raises_nan_error(make_df): + d = make_df({"t1": [0.0, None, 2.0], "t2": [5.0, 6.0, 7.0]}) + with pytest.raises(ValueError, match="y contains NaN values."): check_y(d) - assert str(record.value) == msg -def test_check_y_raises_inf_error(): - msg = "y contains infinity values." +@pytest.mark.parametrize( + "make_df", + [pd.DataFrame, pl.DataFrame], +) +def test_check_y_dataframe_raises_nan_error_for_explicit_nan(make_df): + # in polars, an explicit float("nan") is not a null value, so it is only + # caught if is_nan() is checked in addition to is_null() + d = make_df({"t1": [0.0, float("nan"), 2.0], "t2": [5.0, 6.0, 7.0]}) + with pytest.raises(ValueError, match="y contains NaN values."): + check_y(d) - # y is series - s = pd.Series([0, np.inf, 2, 3, 4]) - with pytest.raises(ValueError) as record: - check_y(s) - assert str(record.value) == msg - # y is multioutput - d = pd.DataFrame(np.array([1, np.inf, 3, 4, 5, 6, np.inf, 8]).reshape(2, 4)) - with pytest.raises(ValueError) as record: +@pytest.mark.parametrize( + "make_df", + [pd.DataFrame, pl.DataFrame], +) +def test_check_y_dataframe_raises_inf_error(make_df): + d = make_df({"t1": [0.0, 0.4, 2.0], "t2": [5.0, float("inf"), 7.0]}) + with pytest.raises(ValueError, match="y contains infinity values."): check_y(d) - assert str(record.value) == msg -def test_check_y_converts_string_to_number(): - s = pd.Series(["0", "1", "2", "3", "4"]) - assert_series_equal(check_y(s, y_numeric=True), s.astype("float")) +# --- array-like input --- -def test_check_x_y_returns_pandas_from_pandas(df_vartypes): - # when s is series - s = pd.Series([0, 1, 2, 3]) - x, y = check_X_y(df_vartypes, s) - assert_frame_equal(df_vartypes, x) - assert_series_equal(s, y) +@pytest.mark.parametrize( + "a", + [ + np.array([1, 2, 3, 4]), + np.array([1, 2, 3, 4, 5, 6, 7, 8]).reshape(2, 4), + [1, 2, 3, 4], + ], +) +def test_check_y_array_returns_unchanged(a): + y = check_y(a) + assert isinstance(y, np.ndarray) + np.testing.assert_array_equal(a, y) + + +def test_check_y_raises_none_error(): + msg = "requires y to be passed, but the target y" + with pytest.raises(ValueError, match=msg): + check_y(None) - # when y is multioutput - d = pd.DataFrame(np.array([1, 2, 3, 4, 5, 6, 7, 8]).reshape(4, 2)) - x, y = check_X_y(df_vartypes, d) - assert_frame_equal(df_vartypes, x) - assert_frame_equal(d, y) +# ------------------------ +# test check_X_y +# ------------------------ -def test_check_X_y_returns_pandas_from_pandas_with_non_typical_index(): + +@pytest.mark.parametrize( + "make_df, assert_frame_fn, make_series, assert_series_fn", + [ + (pd.DataFrame, assert_frame_equal, pd.Series, assert_series_equal), + (pl.DataFrame, pl_assert_frame_equal, pl.Series, pl_assert_series_equal), + ], +) +def test_check_X_y_returns_df_and_series_unchanged( + make_df, assert_frame_fn, make_series, assert_series_fn +): + df = make_df({"a": [1, 2, 3], "b": [4, 5, 6]}) + s = make_series([0, 1, 2]) + X, y = check_X_y(df, s) + assert isinstance(X, nw.DataFrame) and isinstance(y, type(s)) + assert_frame_fn(X.to_native(), df) + assert_series_fn(y, s) + + +@pytest.mark.parametrize( + "make_df, assert_frame_fn", + [(pd.DataFrame, assert_frame_equal), (pl.DataFrame, pl_assert_frame_equal)], +) +def test_check_X_y_returns_df_and_multioutput_y_unchanged(make_df, assert_frame_fn): + df = make_df({"a": [1, 2, 3, 4], "b": [5, 6, 7, 8]}) + d = make_df({"t1": [1, 2, 3, 4], "t2": [5, 6, 7, 8]}) + X, y = check_X_y(df, d) + assert isinstance(X, nw.DataFrame) + assert_frame_fn(X.to_native(), df) + assert_frame_fn(y, d) + + +@pytest.mark.parametrize( + "make_df, assert_frame_fn", + [(pd.DataFrame, assert_frame_equal), (pl.DataFrame, pl_assert_frame_equal)], +) +@pytest.mark.parametrize( + "y", + [ + np.array([0, 1, 2]), + [0, 1, 2], + np.array([[0, 1], [2, 3], [4, 5]]), + ], +) +def test_check_X_y_with_array_like_y_returns_check_y_output( + make_df, assert_frame_fn, y +): + df = make_df({"a": [1, 2, 3], "b": [4, 5, 6]}) + X, y_out = check_X_y(df, y) + assert isinstance(X, nw.DataFrame) + assert_frame_fn(X.to_native(), df) + np.testing.assert_array_equal(y_out, check_y(y)) + + +def test_check_X_y_returns_pandas_with_non_typical_index(): + # only relevant for pandas: polars has no index to reconcile df = pd.DataFrame({"0": [1, 2, 3, 4], "1": [5, 6, 7, 8]}, index=[22, 99, 101, 212]) s = pd.Series([1, 2, 3, 4], index=[22, 99, 101, 212]) x, y = check_X_y(df, s) - assert_frame_equal(df, x) + assert isinstance(x, nw.DataFrame) + assert_frame_equal(df, x.to_native()) assert_series_equal(s, y) def test_check_X_y_raises_error_when_pandas_index_dont_match(): + # only relevant for pandas: polars has no index to reconcile msg = "The indexes of X and y do not match." df = pd.DataFrame({"0": [1, 2, 3, 4], "1": [5, 6, 7, 8]}, index=[22, 99, 101, 212]) s = pd.Series([1, 2, 3, 4], index=[22, 99, 101, 999]) - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=msg): check_X_y(df, s) - assert str(record.value) == msg # when y is multioutput d = pd.DataFrame( np.array([1, 2, 3, 4, 5, 6, 7, 8]).reshape(4, 2), index=[22, 99, 101, 999] ) - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=msg): check_X_y(df, d) - assert str(record.value) == msg -def test_check_x_y_reassings_index_when_only_one_input_is_pandas(): - # X is dataframe, y is 1D array - df = pd.DataFrame({"0": [1, 2, 3, 4], "1": [5, 6, 7, 8]}, index=[22, 99, 101, 212]) - s = np.array([1, 2, 3, 4]) - s_exp = pd.Series([1, 2, 3, 4], index=[22, 99, 101, 212]) - x, y = check_X_y(df, s) - assert_frame_equal(df, x) - assert_series_equal(s_exp.astype(int), y.astype(int)) +@pytest.mark.parametrize( + "make_df, make_series", + [(pd.DataFrame, pd.Series), (pl.DataFrame, pl.Series)], +) +def test_check_x_y_raises_error_when_inconsistent_length(make_df, make_series): + df = make_df({"a": [1, 2, 3]}) + s = make_series([0, 1]) + with pytest.raises(ValueError): + check_X_y(df, s) - # X is dataframe, y is 2d array - s = np.array([1, 2, 3, 4, 5, 6, 7, 8]).reshape(4, 2) - s_exp = pd.DataFrame(s, index=[22, 99, 101, 212]) - x, y = check_X_y(df, s) - assert_frame_equal(df, x) - assert_frame_equal(s_exp.astype(int), y.astype(int)) - # X is not a df, y is a series - df = np.array([[1, 2, 3, 4], [5, 6, 7, 8]]).T - s = pd.Series([1, 2, 3, 4], index=[22, 99, 101, 212]) - df_exp = pd.DataFrame(df, columns=["x0", "x1"]) - df_exp.index = s.index - x, y = check_X_y(df, s) - assert_frame_equal(df_exp, x) - assert_series_equal(s, y) +# ----------------------------------- +# test _check_X_matches_training_df +# ----------------------------------- - # X is not a df, y is a dataframe - s = np.array([1, 2, 3, 4, 5, 6, 7, 8]).reshape(4, 2) - s = pd.DataFrame(s, index=[22, 99, 101, 212]) - df = np.array([[1, 2, 3, 4], [5, 6, 7, 8]]).T - df_exp = pd.DataFrame(df, columns=["x0", "x1"]) - df_exp.index = s.index - x, y = check_X_y(df, s) - assert_frame_equal(df_exp, x) - assert_frame_equal(s, y) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_check_X_matches_training_df_passes_when_columns_match(make_df): + df = make_df({"a": [1, 2], "b": [3, 4]}) + assert _check_X_matches_training_df(df, 2) is None -def test_check_x_y_converts_numpy_to_pandas(): - a2D = np.array([[1, 2], [3, 4], [3, 4], [3, 4]]) - df2D = pd.DataFrame(a2D, columns=["x0", "x1"]) - a1D = np.array([1, 2, 3, 4]) - s1D = pd.Series(a1D) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_check_X_matches_training_df_raises_error_when_columns_dont_match(make_df): + msg = "The number of columns in this dataset is different from" + df = make_df({"a": [1, 2], "b": [3, 4]}) + with pytest.raises(ValueError, match=msg): + _check_X_matches_training_df(df, 3) - # X is df and y is array - x, y = check_X_y(df2D, a1D) - assert_frame_equal(df2D, x) - assert_series_equal(s1D, y) - # X is array and y is series - x, y = check_X_y(a2D, s1D) - assert_frame_equal(df2D, x) - assert_series_equal(s1D, y) +# ------------------------- +# test _check_contains_na +# ------------------------- - # X is df and y is 2d array - y2D = pd.DataFrame(a2D, columns=[0, 1]) - x, y = check_X_y(df2D, a2D) - assert_frame_equal(df2D, x) - assert_frame_equal(y2D, y) - # X is array and y multioutput df - x, y = check_X_y(a2D, df2D) - assert_frame_equal(df2D, x) - assert_frame_equal(df2D, y) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_contains_na_raises_when_nan(make_df): + msg1 = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." + ) + msg2 = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." + ) + df = make_df({"Name": ["tom", None], "City": ["London", "Manchester"]}) + with pytest.raises(ValueError, match=msg1): + _check_contains_na(df, ["Name", "City"]) -def test_check_x_y_raises_error_when_inconsistent_length(df_vartypes): - s = pd.Series([0, 1, 2, 3, 5]) - with pytest.raises(ValueError): - check_X_y(df_vartypes, s) + with pytest.raises(ValueError, match=msg2): + _check_contains_na(df, ["Name", "City"], error_msg="other") -def test_check_X_matches_training_df(df_vartypes): - with pytest.raises(ValueError): - assert _check_X_matches_training_df(df_vartypes, 4) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_contains_na_passes_when_no_nan(make_df): + df = make_df({"Name": ["tom", "nick"], "City": ["London", "Manchester"]}) + assert _check_contains_na(df, ["Name", "City"]) is None -def test_contains_na(df_na): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_contains_na_ignores_columns_not_in_variables(make_df): + df = make_df({"Name": ["tom", None], "City": ["London", "Manchester"]}) + assert _check_contains_na(df, ["City"]) is None + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_contains_na_raises_for_explicit_nan_in_numeric_column(make_df): + # in polars, an explicit float("nan") is not a null value, so it is only + # caught if is_nan() is checked in addition to is_null() msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer." ) - - with pytest.raises(ValueError) as record: - assert _check_contains_na(df_na, ["Name", "City"]) - assert str(record.value) == msg + df = make_df({"Age": [20.0, float("nan"), 19.0], "City": ["a", "b", "c"]}) + with pytest.raises(ValueError, match=msg): + _check_contains_na(df, ["Age", "City"]) -def test_optional_contains_na(df_na): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_contains_na_raises_for_mix_of_null_and_nan_across_dtypes(make_df): + # a numeric column with a NaN and a string column with a null should both + # still be caught, and the numeric-only is_nan() scoping must not error out + # on the string column msg = ( "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer or set the parameter " - "`missing_values='ignore'` when initialising this transformer." + "remove those before using this transformer." ) + df = make_df({"Age": [20.0, float("nan"), 19.0], "City": ["a", None, "c"]}) + with pytest.raises(ValueError, match=msg): + _check_contains_na(df, ["Age", "City"]) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_contains_na_no_ops_when_variables_is_empty(make_df): + # on the narwhals/polars backend, nw.col([]) raises a TypeError, so an + # empty `variables` list must short-circuit before any column selection + df = make_df({"Name": ["tom", None], "City": ["London", "Manchester"]}) + assert _check_contains_na(df, []) is None + - with pytest.raises(ValueError) as record: - assert _check_optional_contains_na(df_na, ["Name", "City"]) - assert str(record.value) == msg +# -------------------------- +# test _check_contains_inf +# -------------------------- -def test_contains_inf_raises_on_inf(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_contains_inf_raises_on_inf(make_df): msg = ( "Some of the variables to transform contain inf values. Check and " "remove those before using this transformer." ) - df = pd.DataFrame({"A": [1.1, np.inf, 3.3]}) + df = make_df({"A": [1.1, np.inf, 3.3]}) with pytest.raises(ValueError, match=msg): _check_contains_inf(df, ["A"]) -def test_contains_inf_passes_without_inf(): - df = pd.DataFrame({"A": [1.1, 2.2, 3.3]}) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_contains_inf_passes_without_inf(make_df): + df = make_df({"A": [1.1, 2.2, 3.3]}) assert _check_contains_inf(df, ["A"]) is None -def test_check_X_raises_error_on_duplicated_column_names(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - } - ) - df.columns = ["var_A", "var_A", "var_B", "var_C"] - with pytest.raises(ValueError) as err_txt: - check_X(df) - assert err_txt.match("Input data contains duplicated variable names.") - - -def test_check_X_errors(): - # Test scalar array error (line 58) - with pytest.raises(ValueError) as record: - check_X(np.array(1)) - assert record.match("Expected 2D array, got scalar array instead") - - # Test 1D array error (line 65) - with pytest.raises(ValueError) as record: - check_X(np.array([1, 2, 3])) - assert record.match("Expected 2D array, got 1D array instead") - - # Test incorrect type error (line 80) - with pytest.raises(TypeError) as record: - check_X("not a dataframe") - assert record.match("X must be a numpy array or pandas dataframe") +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_contains_inf_ignores_columns_not_in_variables(make_df): + df = make_df({"A": [1.1, float("inf"), 3.3], "B": [1.0, 2.0, 3.0]}) + assert _check_contains_inf(df, ["B"]) is None diff --git a/tests/test_datetime/test_datetime_features.py b/tests/test_datetime/test_datetime_features.py index c22239492..4934741e3 100644 --- a/tests/test_datetime/test_datetime_features.py +++ b/tests/test_datetime/test_datetime_features.py @@ -1,5 +1,7 @@ +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from sklearn.pipeline import Pipeline @@ -7,6 +9,7 @@ from feature_engine.datetime import DatetimeFeatures from feature_engine.datetime._datetime_constants import ( FEATURES_DEFAULT, + FEATURES_FUNCTIONS, FEATURES_SUFFIXES, FEATURES_SUPPORTED, ) @@ -23,6 +26,39 @@ index=pd.date_range("2003-02-27", periods=4, freq="D"), ) +# ISO-8601 strings parse identically on pandas and polars/narwhals (unlike the +# dateutil-style formats in df_datetime above, which are pandas-only), so these +# back the cross-backend tests. Covers a leap day and a year/quarter/month +# boundary, so "all" features exercise every derived (non-1:1) narwhals feature. +CROSS_BACKEND_DATES = [ + "2020-01-01 00:00:00", + "2020-02-29 12:30:45", + "2020-12-31 23:59:59", + "2021-07-15 06:07:08", +] +CROSS_BACKEND_DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "Age": [20, 21, 19, 18], + "date": CROSS_BACKEND_DATES, +} +feat_names_default_cb = [f"date{FEATURES_SUFFIXES[feat]}" for feat in FEATURES_DEFAULT] + + +def _expected_cross_backend_features(feats): + """Reference feature values computed with pandas' native FEATURES_FUNCTIONS, + the ground truth both the pandas and the narwhals extraction paths must match.""" + dt = pd.Series(pd.to_datetime(CROSS_BACKEND_DATES)) + return { + f"date{FEATURES_SUFFIXES[feat]}": list(FEATURES_FUNCTIONS[feat](dt)) + for feat in feats + } + + +def _to_py_values(column): + # normalise pandas/numpy and polars scalar containers to plain Python ints + # so the two backends' outputs compare equal regardless of dtype width. + return [int(v) for v in column] + _false_input_params = [ (["not_supported"], 3.519, "wrong_option"), @@ -173,72 +209,48 @@ def test_raises_non_fitted_error(df_datetime): DatetimeFeatures().transform(df_datetime) -def test_extract_datetime_features_with_default_options( - df_datetime, df_datetime_transformed -): - transformer = DatetimeFeatures() - X = transformer.fit_transform(df_datetime) - pd.testing.assert_frame_equal( - X, - df_datetime_transformed[ - vars_non_dt + [var + feat for var in vars_dt for feat in feat_names_default] - ], - check_dtype=False, - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_extract_datetime_features_with_default_options(make_df): + X = make_df(CROSS_BACKEND_DATA) + Xt = DatetimeFeatures().fit_transform(X) + result = nw.from_native(Xt, eager_only=True) + assert result.columns == vars_non_dt + feat_names_default_cb + for col, expected in _expected_cross_backend_features(FEATURES_DEFAULT).items(): + assert _to_py_values(result.get_column(col)) == expected -def test_extract_datetime_features_from_specified_variables( - df_datetime, df_datetime_transformed -): - # single datetime variable - X = DatetimeFeatures(variables="date_obj1").fit_transform(df_datetime) - pd.testing.assert_frame_equal( - X, - df_datetime_transformed[ - vars_non_dt - + ["datetime_range", "date_obj2", "time_obj"] - + ["date_obj1" + feat for feat in feat_names_default] - ], - check_dtype=False, - ) - # multiple datetime variables - X = DatetimeFeatures(variables=["datetime_range", "date_obj2"]).fit_transform( - df_datetime - ) - pd.testing.assert_frame_equal( - X, - df_datetime_transformed[ - vars_non_dt - + ["date_obj1", "time_obj"] - + [ - var + feat - for var in ["datetime_range", "date_obj2"] - for feat in feat_names_default - ] - ], - check_dtype=False, - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_extract_datetime_features_from_specified_variables(make_df): + data = dict(CROSS_BACKEND_DATA) + data["date2"] = CROSS_BACKEND_DATES + X = make_df(data) - # multiple datetime variables in different order than they appear in the df - X = DatetimeFeatures(variables=["date_obj2", "date_obj1"]).fit_transform( - df_datetime - ) - pd.testing.assert_frame_equal( - X, - df_datetime_transformed[ - vars_non_dt - + ["datetime_range", "time_obj"] - + [ - var + feat - for var in ["date_obj2", "date_obj1"] - for feat in feat_names_default - ] - ], - check_dtype=False, + # single datetime variable + Xt = DatetimeFeatures(variables="date").fit_transform(X) + result = nw.from_native(Xt, eager_only=True) + assert result.columns == vars_non_dt + ["date2"] + feat_names_default_cb + for col, expected in _expected_cross_backend_features(FEATURES_DEFAULT).items(): + assert _to_py_values(result.get_column(col)) == expected + + # multiple datetime variables, in different order than they appear in X + Xt = DatetimeFeatures(variables=["date2", "date"]).fit_transform(X) + result = nw.from_native(Xt, eager_only=True) + expected_cols = ( + vars_non_dt + + [f"date2{FEATURES_SUFFIXES[feat]}" for feat in FEATURES_DEFAULT] + + feat_names_default_cb ) + assert result.columns == expected_cols + for col, expected in _expected_cross_backend_features(FEATURES_DEFAULT).items(): + assert _to_py_values(result.get_column(col)) == expected + assert _to_py_values(result.get_column(col.replace("date", "date2"))) == ( + expected + ) + - # datetime variable is index +def test_extract_datetime_features_from_index(): + # "index" is pandas-only: polars and other narwhals backends have no index. X = DatetimeFeatures( variables="index", features_to_extract=["month", "day_of_month"] ).fit_transform(dates_idx_dt) @@ -259,38 +271,41 @@ def test_extract_datetime_features_from_specified_variables( ) -def test_extract_all_datetime_features(df_datetime, df_datetime_transformed): - X = DatetimeFeatures(features_to_extract="all").fit_transform(df_datetime) - pd.testing.assert_frame_equal( - X, df_datetime_transformed.drop(vars_dt, axis=1), check_dtype=False - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_variables_index_raises_on_non_pandas(make_df): + X = make_df(CROSS_BACKEND_DATA) + transformer = DatetimeFeatures(variables="index") + if make_df is pd.DataFrame: + with pytest.raises(TypeError, match="The dataframe index is not datetime."): + transformer.fit(X) + else: + with pytest.raises(TypeError, match="variables='index' requires a pandas"): + transformer.fit(X) -def test_extract_specified_datetime_features(df_datetime, df_datetime_transformed): - X = DatetimeFeatures(features_to_extract=["semester", "week"]).fit_transform( - df_datetime - ) - pd.testing.assert_frame_equal( - X, - df_datetime_transformed[ - vars_non_dt - + [var + "_" + feat for var in vars_dt for feat in ["semester", "week"]] - ], - check_dtype=False, - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_extract_all_datetime_features(make_df): + X = make_df(CROSS_BACKEND_DATA) + Xt = DatetimeFeatures(features_to_extract="all").fit_transform(X) + + result = nw.from_native(Xt, eager_only=True) + expected = _expected_cross_backend_features(FEATURES_SUPPORTED) + assert result.columns == vars_non_dt + list(expected.keys()) + for col, values in expected.items(): + assert _to_py_values(result.get_column(col)) == values - # different order than they appear in the glossary - X = DatetimeFeatures(features_to_extract=["hour", "day_of_week"]).fit_transform( - df_datetime - ) - pd.testing.assert_frame_equal( - X, - df_datetime_transformed[ - vars_non_dt - + [var + "_" + feat for var in vars_dt for feat in ["hour", "day_of_week"]] - ], - check_dtype=False, - ) + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +@pytest.mark.parametrize("features", [["semester", "week"], ["hour", "day_of_week"]]) +def test_extract_specified_datetime_features(make_df, features): + X = make_df(CROSS_BACKEND_DATA) + Xt = DatetimeFeatures(features_to_extract=features).fit_transform(X) + + result = nw.from_native(Xt, eager_only=True) + expected = _expected_cross_backend_features(features) + assert result.columns == vars_non_dt + list(expected.keys()) + for col, values in expected.items(): + assert _to_py_values(result.get_column(col)) == values def test_extract_features_from_categorical_variable( @@ -418,43 +433,53 @@ def test_extract_features_from_localized_tz_variables(): pd.testing.assert_frame_equal(X, df_expected, check_dtype=False) -def test_extract_features_without_dropping_original_variables( - df_datetime, df_datetime_transformed -): - X = DatetimeFeatures( - variables=["datetime_range", "date_obj2"], +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_extract_features_without_dropping_original_variables(make_df): + data = dict(CROSS_BACKEND_DATA) + data["date2"] = CROSS_BACKEND_DATES + X = make_df(data) + + Xt = DatetimeFeatures( + variables=["date", "date2"], features_to_extract=["week", "quarter"], drop_original=False, - ).fit_transform(df_datetime) - - pd.testing.assert_frame_equal( - X, - pd.concat( - [df_datetime_transformed[column] for column in vars_non_dt] - + [df_datetime[var] for var in vars_dt] - + [ - df_datetime_transformed[feat] - for feat in [ - var + "_" + feat - for var in ["datetime_range", "date_obj2"] - for feat in ["week", "quarter"] - ] - ], - axis=1, - ), - check_dtype=False, + ).fit_transform(X) + + result = nw.from_native(Xt, eager_only=True) + expected_cols = ( + vars_non_dt + + ["date", "date2"] + + [ + f"{var}{FEATURES_SUFFIXES[feat]}" + for var in ["date", "date2"] + for feat in ["week", "quarter"] + ] ) + assert result.columns == expected_cols + for col, values in _expected_cross_backend_features(["week", "quarter"]).items(): + assert _to_py_values(result.get_column(col)) == values + assert _to_py_values(result.get_column(col.replace("date", "date2"))) == ( + values + ) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_extract_features_from_variables_containing_nans(make_df): + X = make_df({"dates_na": ["2010-02-01", None, "1922-06-01", None]}) + Xt = DatetimeFeatures( + features_to_extract=["year"], missing_values="ignore" + ).fit_transform(X) + result = nw.from_native(Xt, eager_only=True).get_column("dates_na_year") + values = result.to_list() + assert values[0] == 2010 or values[0] == 2010.0 + assert values[1] is None or (isinstance(values[1], float) and np.isnan(values[1])) + assert values[2] == 1922 or values[2] == 1922.0 + assert values[3] is None or (isinstance(values[3], float) and np.isnan(values[3])) -def test_extract_features_from_variables_containing_nans(): - X = DatetimeFeatures( - features_to_extract=["year"], missing_values="ignore" - ).fit_transform(dates_nan) - pd.testing.assert_frame_equal( - X, - pd.DataFrame({"dates_na_year": [2010, np.nan, 1922, np.nan]}), - ) - # dt variable is index + +def test_extract_features_from_index_containing_nans(): + # "index" is pandas-only: polars and other narwhals backends have no index. X = DatetimeFeatures( variables="index", features_to_extract=["month"], missing_values="ignore" ).fit_transform(dates_idx_nan) @@ -472,6 +497,26 @@ def test_extract_features_from_variables_containing_nans(): ) +def test_polars_string_parsing_needs_explicit_format_for_ambiguous_dates(): + # dayfirst/yearfirst are pandas.to_datetime-only: narwhals' generic + # str.to_datetime() has no day/year-first heuristic, so an ambiguous, + # non-ISO format needs an explicit `format` on non-pandas input. + X = pl.DataFrame({"date_obj1": ["01-Jan-2010", "24-Feb-1945"]}) + transformer = DatetimeFeatures(variables="date_obj1", features_to_extract=["year"]) + transformer.fit(X) + with pytest.raises(Exception, match="could not find an appropriate format"): + transformer.transform(X) + + transformer = DatetimeFeatures( + variables="date_obj1", features_to_extract=["year"], format="%d-%b-%Y" + ) + transformer.fit(X) + Xt = transformer.transform(X) + assert nw.from_native(Xt, eager_only=True).get_column( + "date_obj1_year" + ).to_list() == [2010, 1945] + + def test_ignore_nan_for_week_and_days_in_month(): # week and days_in_month must propagate NaN like the other features when # missing_values="ignore", instead of raising on the int cast of a NaT. @@ -509,6 +554,20 @@ def test_extract_features_with_different_datetime_parsing_options(df_datetime): ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_get_feature_names_out_cross_backend(make_df): + X = make_df(CROSS_BACKEND_DATA) + transformer = DatetimeFeatures() + Xt = transformer.fit_transform(X) + result = nw.from_native(Xt, eager_only=True) + assert result.columns == transformer.get_feature_names_out() + + transformer = DatetimeFeatures(drop_original=False) + Xt = transformer.fit_transform(X) + result = nw.from_native(Xt, eager_only=True) + assert result.columns == transformer.get_feature_names_out() + + def test_get_feature_names_out(df_datetime, df_datetime_transformed): # default features from all variables transformer = DatetimeFeatures() diff --git a/tests/test_datetime/test_datetime_ordinal.py b/tests/test_datetime/test_datetime_ordinal.py index aabeee395..6fe3c46ac 100644 --- a/tests/test_datetime/test_datetime_ordinal.py +++ b/tests/test_datetime/test_datetime_ordinal.py @@ -1,180 +1,200 @@ import datetime +import math + import pandas as pd +import polars as pl import pytest from feature_engine.datetime import DatetimeOrdinal +DATE_COLS = ["date_col_1", "date_col_2"] -@pytest.fixture(scope="module") -def df_datetime_ordinal(): - df = pd.DataFrame( - { - "date_col_1": pd.to_datetime( - ["2023-01-01", "2023-01-02", "2023-01-03", "2023-01-04", "2023-01-05"] - ), - "date_col_2": pd.to_datetime( - ["2024-02-10", "2024-02-11", "2024-02-12", "2024-02-13", "2024-02-14"] - ), - "non_date_col": [1, 2, 3, 4, 5], - } - ) - return df - - -@pytest.fixture(scope="module") -def df_datetime_ordinal_na(): - df = pd.DataFrame( - { - "date_col_1": pd.to_datetime( - ["2023-01-01", "2023-01-02", None, "2023-01-04", "2023-01-05"] - ), - "date_col_2": pd.to_datetime( - ["2024-02-10", "2024-02-11", "2024-02-12", None, "2024-02-14"] - ), - } - ) - return df - - +DATE_DATA = { + "date_col_1": [ + "2023-01-01", + "2023-01-02", + "2023-01-03", + "2023-01-04", + "2023-01-05", + ], + "date_col_2": [ + "2024-02-10", + "2024-02-11", + "2024-02-12", + "2024-02-13", + "2024-02-14", + ], + "non_date_col": [1, 2, 3, 4, 5], +} + +DATE_DATA_NA = { + "date_col_1": ["2023-01-01", "2023-01-02", None, "2023-01-04", "2023-01-05"], + "date_col_2": ["2024-02-10", "2024-02-11", "2024-02-12", None, "2024-02-14"], +} + + +def _make_datetime_df(make_df, data: dict, date_cols=DATE_COLS): + """Build a dataframe where `date_cols` hold a native Date/Datetime dtype + (not strings), the same way real datetime columns arrive in practice - + constructed differently per backend since pandas and polars have no + shared literal syntax for it.""" + if make_df is pd.DataFrame: + return pd.DataFrame( + { + col: pd.to_datetime(values) if col in date_cols else values + for col, values in data.items() + } + ) + df = pl.DataFrame(data) + return df.with_columns([pl.col(c).str.to_datetime() for c in date_cols]) + + +def _expected_ordinal(date_strings, start_date_ordinal=None): + result = [] + for s in date_strings: + if s is None: + result.append(None) + continue + ordinal = datetime.date.fromisoformat(s).toordinal() + if start_date_ordinal is not None: + ordinal = ordinal - start_date_ordinal + 1 + result.append(ordinal) + return result + + +def _as_comparable_ints(values): + """Normalize a result column to plain ints/None regardless of whether the + backend represented missing ordinals as NaN (pandas float64) or null + (polars Int64) - same values, different native missing-data convention.""" + out = [] + for v in values: + if v is None or (isinstance(v, float) and math.isnan(v)): + out.append(None) + else: + out.append(int(v)) + return out + + +def _get_col(X, col): + if isinstance(X, pd.DataFrame): + return X[col].tolist() + return X[col].to_list() + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "variables_param", - [ - ["date_col_1", "date_col_2"], # Case 1: 'variables' are specified - None, # Case 2: 'variables' not specified - ], - ids=[ - "variables_specified", - "variables_auto_find", - ], # Optional but recommended for test readability + [["date_col_1", "date_col_2"], None], + ids=["variables_specified", "variables_auto_find"], ) -def test_datetime_ordinal_feature_creation(df_datetime_ordinal, variables_param): - """ - Tests that the ordinal features are created correctly, - both when variables are specified and when they are auto-detected. - """ +def test_datetime_ordinal_feature_creation(make_df, variables_param): + X = _make_datetime_df(make_df, DATE_DATA) transformer = DatetimeOrdinal(variables=variables_param) - X_transformed = transformer.fit_transform(df_datetime_ordinal) - - # --- Common validation logic for both tests --- - expected_ordinal_1 = pd.Series( - [d.toordinal() for d in df_datetime_ordinal["date_col_1"]], - name="date_col_1_ordinal", - ) - expected_ordinal_2 = pd.Series( - [d.toordinal() for d in df_datetime_ordinal["date_col_2"]], - name="date_col_2_ordinal", - ) + Xt = transformer.fit_transform(X) - pd.testing.assert_series_equal( - X_transformed["date_col_1_ordinal"], expected_ordinal_1 + assert _as_comparable_ints(_get_col(Xt, "date_col_1_ordinal")) == _expected_ordinal( + DATE_DATA["date_col_1"] ) - pd.testing.assert_series_equal( - X_transformed["date_col_2_ordinal"], expected_ordinal_2 + assert _as_comparable_ints(_get_col(Xt, "date_col_2_ordinal")) == _expected_ordinal( + DATE_DATA["date_col_2"] ) - # Check if original columns are dropped and non-date column remains - assert "non_date_col" in X_transformed.columns - assert "date_col_1" not in X_transformed.columns - assert "date_col_2" not in X_transformed.columns + columns = Xt.columns + assert "non_date_col" in columns + assert "date_col_1" not in columns + assert "date_col_2" not in columns -def test_datetime_ordinal_with_start_date(df_datetime_ordinal): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_with_start_date(make_df): start_date_str = "2023-01-01" + X = _make_datetime_df(make_df, DATE_DATA) transformer = DatetimeOrdinal(variables=["date_col_1"], start_date=start_date_str) - X_transformed = transformer.fit_transform(df_datetime_ordinal) + Xt = transformer.fit_transform(X) - start_ordinal = pd.to_datetime(start_date_str).toordinal() - expected_ordinal = pd.Series( - [d.toordinal() - start_ordinal + 1 for d in df_datetime_ordinal["date_col_1"]], - name="date_col_1_ordinal", + start_ordinal = datetime.date.fromisoformat(start_date_str).toordinal() + expected = _expected_ordinal( + DATE_DATA["date_col_1"], start_date_ordinal=start_ordinal ) - pd.testing.assert_series_equal( - X_transformed["date_col_1_ordinal"], expected_ordinal - ) - assert "date_col_2" in X_transformed.columns - assert "date_col_1" not in X_transformed.columns + assert _as_comparable_ints(_get_col(Xt, "date_col_1_ordinal")) == expected + assert "date_col_2" in Xt.columns + assert "date_col_1" not in Xt.columns -def test_datetime_ordinal_with_start_date_datetime_object(df_datetime_ordinal): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_with_start_date_datetime_object(make_df): start_date_obj = datetime.date(2023, 1, 1) + X = _make_datetime_df(make_df, DATE_DATA) transformer = DatetimeOrdinal(variables=["date_col_1"], start_date=start_date_obj) - X_transformed = transformer.fit_transform(df_datetime_ordinal) - - start_ordinal = pd.to_datetime(start_date_obj).toordinal() - expected_ordinal = pd.Series( - [d.toordinal() - start_ordinal + 1 for d in df_datetime_ordinal["date_col_1"]], - name="date_col_1_ordinal", - ) + Xt = transformer.fit_transform(X) - pd.testing.assert_series_equal( - X_transformed["date_col_1_ordinal"], expected_ordinal + expected = _expected_ordinal( + DATE_DATA["date_col_1"], start_date_ordinal=start_date_obj.toordinal() ) + assert _as_comparable_ints(_get_col(Xt, "date_col_1_ordinal")) == expected -def test_datetime_ordinal_missing_values_raise(df_datetime_ordinal_na): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_missing_values_raise(make_df): + X = _make_datetime_df(make_df, DATE_DATA_NA) transformer = DatetimeOrdinal(missing_values="raise") with pytest.raises( ValueError, match="Some of the variables in the dataset contain NaN" ): - transformer.fit(df_datetime_ordinal_na) + transformer.fit(X) -def test_datetime_ordinal_missing_values_ignore(df_datetime_ordinal_na): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_missing_values_ignore(make_df): + X = _make_datetime_df(make_df, DATE_DATA_NA) transformer = DatetimeOrdinal(missing_values="ignore") - X_transformed = transformer.fit_transform(df_datetime_ordinal_na) - - # Expected values for date_col_1_ordinal, handling None - expected_ordinal_1 = pd.Series( - [ - d.toordinal() if pd.notna(d) else pd.NA - for d in df_datetime_ordinal_na["date_col_1"] - ], - name="date_col_1_ordinal", - dtype=object, - ) - expected_ordinal_2 = pd.Series( - [ - d.toordinal() if pd.notna(d) else pd.NA - for d in df_datetime_ordinal_na["date_col_2"] - ], - name="date_col_2_ordinal", - dtype=object, - ) + Xt = transformer.fit_transform(X) - pd.testing.assert_series_equal( - X_transformed["date_col_1_ordinal"], expected_ordinal_1 - ) - pd.testing.assert_series_equal( - X_transformed["date_col_2_ordinal"], expected_ordinal_2 - ) + assert _as_comparable_ints( + _get_col(Xt, "date_col_1_ordinal") + ) == _expected_ordinal(DATE_DATA_NA["date_col_1"]) + assert _as_comparable_ints( + _get_col(Xt, "date_col_2_ordinal") + ) == _expected_ordinal(DATE_DATA_NA["date_col_2"]) def test_datetime_ordinal_invalid_start_date(): + # start_date is parsed in fit(), not __init__, so __init__ only stores it. + transformer = DatetimeOrdinal(start_date="not-a-date") + assert transformer.start_date == "not-a-date" + + X = pd.DataFrame(DATE_DATA) with pytest.raises( ValueError, match="start_date could not be converted to datetime" ): - DatetimeOrdinal(start_date="not-a-date") + transformer.fit(X) -def test_datetime_ordinal_non_datetime_variable_error(df_datetime_ordinal): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_non_datetime_variable_error(make_df): + X = make_df(DATE_DATA) transformer = DatetimeOrdinal(variables=["non_date_col"]) with pytest.raises(TypeError): - transformer.fit(df_datetime_ordinal) + transformer.fit(X) -def test_datetime_ordinal_drop_original_false(df_datetime_ordinal): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_drop_original_false(make_df): + X = _make_datetime_df(make_df, DATE_DATA) transformer = DatetimeOrdinal(variables=["date_col_1"], drop_original=False) - X_transformed = transformer.fit_transform(df_datetime_ordinal) + Xt = transformer.fit_transform(X) - assert "date_col_1" in X_transformed.columns - assert "date_col_1_ordinal" in X_transformed.columns - assert "date_col_2" in X_transformed.columns + assert "date_col_1" in Xt.columns + assert "date_col_1_ordinal" in Xt.columns + assert "date_col_2" in Xt.columns -def test_datetime_ordinal_get_feature_names_out(df_datetime_ordinal): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_get_feature_names_out(make_df): + X = _make_datetime_df(make_df, DATE_DATA) transformer = DatetimeOrdinal(variables=["date_col_1", "date_col_2"]) - transformer.fit(df_datetime_ordinal) + transformer.fit(X) feature_names_out = transformer.get_feature_names_out() expected_feature_names = [ @@ -185,13 +205,13 @@ def test_datetime_ordinal_get_feature_names_out(df_datetime_ordinal): assert sorted(feature_names_out) == sorted(expected_feature_names) -def test_datetime_ordinal_get_feature_names_out_with_input_features( - df_datetime_ordinal, -): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_get_feature_names_out_with_input_features(make_df): + X = _make_datetime_df(make_df, DATE_DATA) transformer = DatetimeOrdinal(variables=["date_col_1"], drop_original=False) - transformer.fit(df_datetime_ordinal) + transformer.fit(X) feature_names_out = transformer.get_feature_names_out( - input_features=df_datetime_ordinal.columns.tolist() + input_features=list(X.columns) ) expected_feature_names = [ @@ -203,52 +223,49 @@ def test_datetime_ordinal_get_feature_names_out_with_input_features( assert sorted(feature_names_out) == sorted(expected_feature_names) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_datetime_ordinal_get_feature_names_out_with_input_features_drop_original( - df_datetime_ordinal, + make_df, ): + X = _make_datetime_df(make_df, DATE_DATA) transformer = DatetimeOrdinal(variables=["date_col_1"], drop_original=True) - transformer.fit(df_datetime_ordinal) + transformer.fit(X) feature_names_out = transformer.get_feature_names_out( - input_features=df_datetime_ordinal.columns.tolist() + input_features=list(X.columns) ) expected_feature_names = ["date_col_1_ordinal", "date_col_2", "non_date_col"] assert sorted(feature_names_out) == sorted(expected_feature_names) -def test_datetime_ordinal_non_datetime_variable_in_transform(df_datetime_ordinal): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_non_datetime_variable_in_transform(make_df): + X = _make_datetime_df(make_df, DATE_DATA) transformer = DatetimeOrdinal(variables=["date_col_1"]) - transformer.fit(df_datetime_ordinal) - # Create a new dataframe where 'date_col_1' is no longer datetime - X_test = df_datetime_ordinal.copy() - X_test["date_col_1"] = ["a", "b", "c", "d", "e"] + transformer.fit(X) - with pytest.raises(ValueError): + junk_data = {**DATE_DATA, "date_col_1": ["a", "b", "c", "d", "e"]} + X_test = make_df(junk_data) + + # pandas raises ValueError, polars raises its own ComputeError - different + # exception classes, but both signal the same "not a real date" failure. + with pytest.raises(Exception): transformer.transform(X_test) -def test_datetime_ordinal_missing_values_raise_in_transform( - df_datetime_ordinal, df_datetime_ordinal_na -): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_ordinal_missing_values_raise_in_transform(make_df): + X = _make_datetime_df(make_df, DATE_DATA) transformer = DatetimeOrdinal(missing_values="raise") + transformer.fit(X) - # 1. Fit using the 3-column dataframe (df_datetime_ordinal) - transformer.fit(df_datetime_ordinal) + na_data = {**DATE_DATA_NA, "non_date_col": [1, 2, 3, 4, 5]} + X_test = _make_datetime_df(make_df, na_data) - # 2. Copy the NA dataframe (which initially has 2 columns) - X_test = df_datetime_ordinal_na.copy() - - # 3. Add 'non_date_col' to match the column count (3) from the fit data. - # (The content doesn't matter, matching the column count is what's important - # to avoid the column mismatch error). - X_test["non_date_col"] = [1, 2, 3, 4, 5] - - # 4. Now, test that it raises the NaN error (not the column mismatch error). - # The match string is aligned with the error found in the fit test (Failure 1). with pytest.raises( ValueError, match="Some of the variables in the dataset contain NaN" ): - transformer.transform(X_test) # 3 columns + NA data + transformer.transform(X_test) def test_raises_error_for_invalid_missing_values(): @@ -271,14 +288,12 @@ def test_more_tags_returns_expected_tags(): assert transformer._more_tags() == expected_tags -def test_return_empty(): - # DatetimeOrdinal.__init__ does not store `self.start_date = start_date` - # (only the derived `self.start_date_`), which breaks sklearn's - # get_params()/clone() for this transformer. Because of that, it cannot go - # through the shared, clone-based check_return_empty check, nor through - # check_feature_engine_estimator at all. This test instantiates the - # transformer directly instead. - X = pd.DataFrame({"var_num": [1.0, 2.0, 3.0]}) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_return_empty(make_df): + # Instantiated directly rather than via the shared, clone-based + # check_return_empty helper, which parametrizes over a fixed transformer list + # this one is not part of. + X = make_df({"var_num": [1.0, 2.0, 3.0]}) transformer = DatetimeOrdinal(variables=None, return_empty=False) with pytest.raises( @@ -294,8 +309,7 @@ def test_return_empty(): transformer.fit(X) assert transformer.variables_ == [] - # if return_empty=True, transformer should return same df - # after transformation - dft = transformer.transform(X) - pd.testing.assert_frame_equal(dft, X) + # if return_empty=True, transformer should return same df after transformation + Xt = transformer.transform(X) + assert _get_col(Xt, "var_num") == _get_col(X, "var_num") assert transformer.get_feature_names_out() == list(X.columns) diff --git a/tests/test_datetime/test_datetime_subtraction.py b/tests/test_datetime/test_datetime_subtraction.py index 4e854d04e..a8c64e614 100644 --- a/tests/test_datetime/test_datetime_subtraction.py +++ b/tests/test_datetime/test_datetime_subtraction.py @@ -1,5 +1,8 @@ -import numpy as np +from datetime import datetime as _datetime + +import narwhals as nw import pandas as pd +import polars as pl import pytest from feature_engine.datetime import DatetimeSubtraction @@ -15,6 +18,38 @@ ) from tests.estimator_checks.non_fitted_error_checks import check_raises_non_fitted_error +DATA_DATETIME = { + "Name": ["tom", "nick", "krish", "jack"], + "Age": [20, 21, 19, 18], + "datetime_range": [ + _datetime(2020, 2, 24), + _datetime(2020, 2, 25), + _datetime(2020, 2, 26), + _datetime(2020, 2, 27), + ], + "date_obj1": ["01-Jan-2010", "24-Feb-1945", "14-Jun-2100", "17-May-1999"], + "date_obj2": ["10/11/12", "12/31/09", "06/30/95", "03/17/04"], + "time_obj": ["21:45:23", "09:15:33", "12:34:59", "03:27:02"], +} + +DATA_NAN = { + "dates_na": ["Feb-2010", None, "Jun-1922", None], + "dates_full": ["Feb-2010", "Mar-2010", "Jun-1922", "Feb-2011"], +} + +DATA_NAN_FILLED = { + "dates_na": ["Feb-2010", "Mar-2010", "Jun-1922", "Mar-2010"], + "dates_full": ["Feb-2010", "Mar-2010", "Jun-1922", "Feb-2011"], +} + + +def assert_df_equal(X, expected: dict, abs_tol: float = 1e-5) -> None: + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + assert list(result.keys()) == list(expected.keys()) + for col, values in expected.items(): + assert result[col] == pytest.approx(values, abs=abs_tol) + + # ========= init functionality tests @@ -138,24 +173,30 @@ def test_missing_values_raises_error_when_not_valid(param): # ==== fit functionality +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("input_vars", [["Age", "date_obj2"], "Age"]) -def test_raises_error_when_variables_not_datetime(df_datetime, input_vars): +def test_raises_error_when_variables_not_datetime(make_df, input_vars): + df = make_df(DATA_DATETIME) tr = DatetimeSubtraction(variables=input_vars, reference="date_obj1") with pytest.raises(TypeError): - tr.fit(df_datetime) + tr.fit(df) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("input_vars", [["Age", "date_obj2"], "Age"]) -def test_raises_error_when_reference_not_datetime(df_datetime, input_vars): +def test_raises_error_when_reference_not_datetime(make_df, input_vars): + df = make_df(DATA_DATETIME) tr = DatetimeSubtraction(variables=["date_obj1"], reference=input_vars) with pytest.raises(TypeError): - tr.fit(df_datetime) + tr.fit(df) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("input_vars", [["time_obj", "date_obj2"], "date_obj2", None]) -def test_sets_variables_if_datetime(df_datetime, input_vars): +def test_sets_variables_if_datetime(make_df, input_vars): + df = make_df(DATA_DATETIME) tr = DatetimeSubtraction(variables=input_vars, reference=input_vars) - tr.fit(df_datetime) + tr.fit(df) if input_vars is None: dt_vars = ["datetime_range", "date_obj1", "date_obj2", "time_obj"] assert tr.variables_ == dt_vars @@ -168,29 +209,24 @@ def test_sets_variables_if_datetime(df_datetime, input_vars): assert tr.reference_ == ["time_obj", "date_obj2"] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("new", [["new1", "new2"], ["new1", "new2", "new3"]]) -def test_new_variables_raise_error_if_not_adequate_number(df_datetime, new): +def test_new_variables_raise_error_if_not_adequate_number(make_df, new): + df = make_df(DATA_DATETIME) tr = DatetimeSubtraction( variables="date_obj1", reference="date_obj1", new_variables_names=new ) with pytest.raises(ValueError): - tr.fit(df_datetime) - - -@pytest.fixture -def df_nan(): - df = pd.DataFrame( - { - "dates_na": ["Feb-2010", np.nan, "Jun-1922", np.nan], - "dates_full": ["Feb-2010", "Mar-2010", "Jun-1922", "Feb-2011"], - } - ) - return df + tr.fit(df) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("input_vars_1", ["dates_full", None]) @pytest.mark.parametrize("input_vars_2", ["dates_na", ["dates_full", "dates_na"], None]) -def test_raises_error_when_nan_in_variables_in_fit(df_nan, input_vars_1, input_vars_2): +def test_raises_error_when_nan_in_variables_in_fit( + make_df, input_vars_1, input_vars_2 +): + df_nan = make_df(DATA_NAN) tr = DatetimeSubtraction( variables=input_vars_2, reference=input_vars_1, missing_values="raise" ) @@ -198,9 +234,13 @@ def test_raises_error_when_nan_in_variables_in_fit(df_nan, input_vars_1, input_v tr.fit(df_nan) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("input_vars_1", ["dates_full", None]) @pytest.mark.parametrize("input_vars_2", ["dates_na", ["dates_full", "dates_na"], None]) -def test_raises_error_when_nan_in_reference_in_fit(df_nan, input_vars_1, input_vars_2): +def test_raises_error_when_nan_in_reference_in_fit( + make_df, input_vars_1, input_vars_2 +): + df_nan = make_df(DATA_NAN) tr = DatetimeSubtraction( variables=input_vars_1, reference=input_vars_2, missing_values="raise" ) @@ -209,32 +249,35 @@ def test_raises_error_when_nan_in_reference_in_fit(df_nan, input_vars_1, input_v # transform tests +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("input_vars_1", ["dates_full", None]) @pytest.mark.parametrize("input_vars_2", ["dates_na", ["dates_full", "dates_na"], None]) def test_raises_error_when_nan_in_variables_in_transform( - df_nan, input_vars_1, input_vars_2 + make_df, input_vars_1, input_vars_2 ): tr = DatetimeSubtraction( variables=input_vars_2, reference=input_vars_1, missing_values="raise" ) - tr.fit(df_nan.fillna("Mar-2010")) + tr.fit(make_df(DATA_NAN_FILLED)) with pytest.raises(ValueError): - tr.transform(df_nan) + tr.transform(make_df(DATA_NAN)) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("input_vars_1", ["dates_full", None]) @pytest.mark.parametrize("input_vars_2", ["dates_na", ["dates_full", "dates_na"], None]) def test_raises_error_when_nan_in_reference_in_transform( - df_nan, input_vars_1, input_vars_2 + make_df, input_vars_1, input_vars_2 ): tr = DatetimeSubtraction( variables=input_vars_1, reference=input_vars_2, missing_values="raise" ) - tr.fit(df_nan.fillna("Mar-2010")) + tr.fit(make_df(DATA_NAN_FILLED)) with pytest.raises(ValueError): - tr.transform(df_nan) + tr.transform(make_df(DATA_NAN)) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "unit, expected", [ @@ -243,91 +286,77 @@ def test_raises_error_when_nan_in_reference_in_transform( ("h", [744.0, 1464.0, 4392.0]), ], ) -def test_subtraction_units(unit, expected): - df_input = pd.DataFrame( - { - "date1": ["2022-09-18", "2022-10-27", "2022-12-24"], - "date2": ["2022-08-18", "2022-08-27", "2022-06-24"], - } - ) - df_expected = pd.DataFrame( - { - "date1": ["2022-09-18", "2022-10-27", "2022-12-24"], - "date2": ["2022-08-18", "2022-08-27", "2022-06-24"], - "date1_sub_date2": expected, - } - ) +def test_subtraction_units(make_df, unit, expected): + data = { + "date1": ["2022-09-18", "2022-10-27", "2022-12-24"], + "date2": ["2022-08-18", "2022-08-27", "2022-06-24"], + } + df_input = make_df(data) dtf = DatetimeSubtraction( variables=["date1"], reference=["date2"], output_unit=unit ) df_output = dtf.fit_transform(df_input) - pd.testing.assert_frame_equal(df_output, df_expected, check_dtype=False) + expected_dict = dict(data) + expected_dict["date1_sub_date2"] = expected + assert_df_equal(df_output, expected_dict) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_multiple_subtractions(make_df): + data = { + "date1": ["2022-09-01", "2022-10-01", "2022-12-01"], + "date2": ["2022-09-15", "2022-10-15", "2022-12-15"], + "date3": ["2022-08-01", "2022-09-01", "2022-11-01"], + "date4": ["2022-08-15", "2022-09-15", "2022-11-15"], + } + df_input = make_df(data) + + expected = dict(data) + expected["date1_sub_date3"] = [31, 30, 30] + expected["date2_sub_date3"] = [45, 44, 44] + expected["date1_sub_date4"] = [17, 16, 16] + expected["date2_sub_date4"] = [31, 30, 30] -def test_multiple_subtractions(): - df_input = pd.DataFrame( - { - "date1": ["2022-09-01", "2022-10-01", "2022-12-01"], - "date2": ["2022-09-15", "2022-10-15", "2022-12-15"], - "date3": ["2022-08-01", "2022-09-01", "2022-11-01"], - "date4": ["2022-08-15", "2022-09-15", "2022-11-15"], - } - ) - df_expected = pd.DataFrame( - { - "date1": ["2022-09-01", "2022-10-01", "2022-12-01"], - "date2": ["2022-09-15", "2022-10-15", "2022-12-15"], - "date3": ["2022-08-01", "2022-09-01", "2022-11-01"], - "date4": ["2022-08-15", "2022-09-15", "2022-11-15"], - "date1_sub_date3": [31, 30, 30], - "date2_sub_date3": [45, 44, 44], - "date1_sub_date4": [17, 16, 16], - "date2_sub_date4": [31, 30, 30], - } - ) dtf = DatetimeSubtraction( variables=["date1", "date2"], reference=["date3", "date4"] ) df_output = dtf.fit_transform(df_input) - pd.testing.assert_frame_equal(df_output, df_expected, check_dtype=False) + assert_df_equal(df_output, expected) -def test_assigns_new_variable_names(): - df_input = pd.DataFrame( - { - "date1": ["2022-09-01", "2022-10-01", "2022-12-01"], - "date2": ["2022-09-15", "2022-10-15", "2022-12-15"], - "date3": ["2022-08-01", "2022-09-01", "2022-11-01"], - "date4": ["2022-08-15", "2022-09-15", "2022-11-15"], - } - ) - df_expected = pd.DataFrame( - { - "date1": ["2022-09-01", "2022-10-01", "2022-12-01"], - "date2": ["2022-09-15", "2022-10-15", "2022-12-15"], - "date3": ["2022-08-01", "2022-09-01", "2022-11-01"], - "date4": ["2022-08-15", "2022-09-15", "2022-11-15"], - "new1": [31, 30, 30], - "new2": [45, 44, 44], - "new3": [17, 16, 16], - "new4": [31, 30, 30], - } - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_assigns_new_variable_names(make_df): + data = { + "date1": ["2022-09-01", "2022-10-01", "2022-12-01"], + "date2": ["2022-09-15", "2022-10-15", "2022-12-15"], + "date3": ["2022-08-01", "2022-09-01", "2022-11-01"], + "date4": ["2022-08-15", "2022-09-15", "2022-11-15"], + } + df_input = make_df(data) + + expected = dict(data) + expected["new1"] = [31, 30, 30] + expected["new2"] = [45, 44, 44] + expected["new3"] = [17, 16, 16] + expected["new4"] = [31, 30, 30] + dtf = DatetimeSubtraction( variables=["date1", "date2"], reference=["date3", "date4"], new_variables_names=["new1", "new2", "new3", "new4"], ) df_output = dtf.fit_transform(df_input) - pd.testing.assert_frame_equal(df_output, df_expected, check_dtype=False) + assert_df_equal(df_output, expected) # additional methods -def test_get_feature_names_out(): - df = pd.DataFrame( +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_get_feature_names_out(make_df): + df = make_df( { "d1": ["Feb-2010", "Mar-2010", "Jun-1922", "Feb-2011"], "d2": ["Feb-2010", "Mar-2010", "Jun-1922", "Feb-2011"], @@ -335,7 +364,7 @@ def test_get_feature_names_out(): "d4": ["Feb-2010", "Mar-2010", "Jun-1922", "Feb-2011"], } ) - input_vars = df.columns.to_list() + input_vars = list(nw.from_native(df, eager_only=True).columns) tr = DatetimeSubtraction(variables="d1", reference="d2") tr.fit(df) diff --git a/tests/test_discretisation/conftest.py b/tests/test_discretisation/conftest.py new file mode 100644 index 000000000..ace9121c4 --- /dev/null +++ b/tests/test_discretisation/conftest.py @@ -0,0 +1,73 @@ +"""Data shared by the discretiser tests. + +Each fixture returns a fresh dict, so tests can build the dataframe on the +backend under test with ``make_df(data)``. Missing values are written as None, +which both pandas and polars read as missing. +""" + +import datetime +from functools import lru_cache + +import numpy as np +import pytest +from sklearn.datasets import fetch_california_housing + + +@lru_cache(maxsize=1) +def _california_housing(): + dataset = fetch_california_housing() + return dataset.feature_names, dataset.data + + +@pytest.fixture +def data_california(): + feature_names, values = _california_housing() + return {name: values[:, i].tolist() for i, name in enumerate(feature_names)} + + +@pytest.fixture +def data_normal_dist(): + # same seed and parameters as the pandas df_normal_dist fixture in + # tests/conftest.py + return {"var": np.random.RandomState(0).normal(0, 0.1, 100).tolist()} + + +@pytest.fixture +def data_vartypes(): + return { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "dob": [datetime.datetime(2020, 2, 24, 0, i) for i in range(4)], + } + + +@pytest.fixture +def data_na(): + return { + "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], + "City": [ + "London", + "Manchester", + None, + None, + "London", + "London", + "Bristol", + "Manchester", + ], + "Studies": [ + "Bachelor", + "Bachelor", + None, + None, + "Bachelor", + "PhD", + "None", + "Masters", + ], + "Age": [20, 21, 19, None, 23, 40, 41, 37], + "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], + "dob": [datetime.datetime(2020, 2, 24, 0, i) for i in range(8)], + } diff --git a/tests/test_discretisation/test_arbitrary_discretiser.py b/tests/test_discretisation/test_arbitrary_discretiser.py index f1b2db712..79e2bf1fc 100644 --- a/tests/test_discretisation/test_arbitrary_discretiser.py +++ b/tests/test_discretisation/test_arbitrary_discretiser.py @@ -1,94 +1,118 @@ +import re + import numpy as np import pandas as pd import pytest -from numpy.random import default_rng -from scipy.stats import skewnorm -from sklearn.datasets import fetch_california_housing from feature_engine.discretisation import ArbitraryDiscretiser +from tests.backend_helpers import frame_to_dict + +BINS = [0, 20, 40, 60, np.inf] + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) -def test_arbitrary_discretiser(): - california_dataset = fetch_california_housing() - data = pd.DataFrame( - california_dataset.data, columns=california_dataset.feature_names +# init parameters +@pytest.mark.parametrize("binning_dict", ["HOLA", 1, False, None, [0, 10, 20]]) +def test_error_if_binning_dict_not_dict_type(binning_dict): + msg = ( + "binning_dict must be a dictionary with the interval limits per " + f"variable. Got {binning_dict} instead." ) - user_dict = {"HouseAge": [0, 20, 40, 60, np.inf]} + with pytest.raises(ValueError, match=re.escape(msg)): + ArbitraryDiscretiser(binning_dict=binning_dict) - data_t1 = data.copy() - data_t2 = data.copy() - # HouseAge is the median house age in the block group. - data_t1["HouseAge"] = pd.cut( - data["HouseAge"], bins=[0, 20, 40, 60, np.inf], include_lowest=True +@pytest.mark.parametrize( + "errors", ["medialuna", "Ignore", "", 1, None, ["ignore"], ("raise",)] +) +def test_error_if_errors_not_permitted_value(errors): + msg = f"errors only takes values 'ignore' and 'raise'. Got {errors} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + ArbitraryDiscretiser(binning_dict={"Age": BINS}, errors=errors) + + +@pytest.mark.parametrize( + "binning_dict, return_object, return_boundaries, precision, errors", + [ + ({"HouseAge": BINS}, False, False, 3, "ignore"), + ({"HouseAge": BINS, "MedInc": [0, 5, np.inf]}, True, False, 1, "raise"), + ({"Age": [0, 10, np.inf]}, False, True, 10, "raise"), + ], +) +def test_init_param_assignment( + binning_dict, return_object, return_boundaries, precision, errors +): + transformer = ArbitraryDiscretiser( + binning_dict=binning_dict, + return_object=return_object, + return_boundaries=return_boundaries, + precision=precision, + errors=errors, ) - data_t1["HouseAge"] = data_t1["HouseAge"].astype(str) - data_t2["HouseAge"] = pd.cut( - data["HouseAge"], - bins=[0, 20, 40, 60, np.inf], - labels=False, - include_lowest=True, + assert transformer.binning_dict == binning_dict + assert transformer.return_object is return_object + assert transformer.return_boundaries is return_boundaries + assert transformer.precision == precision + assert transformer.errors == errors + + +# fit and transform +def test_arbitrary_discretiser(make_df, data_california): + user_dict = {"HouseAge": BINS} + + # ground truth via pandas.cut - bins are user-supplied and fixed, so both + # backends must reproduce this exact output. + house_age = pd.Series(data_california["HouseAge"]) + expected_codes = pd.cut( + house_age, bins=BINS, labels=False, include_lowest=True + ).tolist() + expected_labels = ( + pd.cut(house_age, bins=BINS, include_lowest=True).astype(str).tolist() ) + data = make_df(data_california) + transformer = ArbitraryDiscretiser( binning_dict=user_dict, return_object=False, return_boundaries=False ) X = transformer.fit_transform(data) - # init params - assert transformer.return_object is False - assert transformer.return_boundaries is False # fit params assert transformer.variables_ == ["HouseAge"] assert transformer.binner_dict_ == user_dict # transform params - pd.testing.assert_frame_equal(X, data_t2) + assert isinstance(X, make_df) + assert frame_to_dict(X)["HouseAge"] == expected_codes transformer = ArbitraryDiscretiser( binning_dict=user_dict, return_object=False, return_boundaries=True ) X = transformer.fit_transform(data) - pd.testing.assert_frame_equal(X, data_t1) + assert isinstance(X, make_df) + assert frame_to_dict(X)["HouseAge"] == expected_labels -def test_error_if_input_df_contains_na_in_transform(df_vartypes, df_na): - # test case 1: when dataset contains na, transform method +def test_error_if_input_df_contains_na_in_transform(make_df): + # test case 1: when dataset contains na, transform method raises age_dict = {"Age": [0, 10, 20, 30, np.inf]} + data = make_df({"Age": [20.0, 21.0, 19.0, 18.0]}) + data_na = make_df({"Age": [20.0, 21.0, None, 18.0]}) - with pytest.raises(ValueError): - transformer = ArbitraryDiscretiser(binning_dict=age_dict) - transformer.fit(df_vartypes) - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) - - -def test_error_when_nan_introduced_during_transform(): - # test error when NA are introduced during the discretisation. - rng = default_rng() + transformer = ArbitraryDiscretiser(binning_dict=age_dict) + transformer.fit(data) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(data_na) - # create dataframe with 2 variables, 1 normal and 1 skewed - random = skewnorm.rvs(a=-50, loc=4, size=100) - random = random - min(random) # Shift so the minimum value is equal to zero. - - train = pd.concat( - [ - pd.Series(rng.standard_normal(100)), - pd.Series(random), - ], - axis=1, - ) - train.columns = ["var_a", "var_b"] - - # create a dataframe with 2 variables normally distributed - test = pd.concat( - [ - pd.Series(rng.standard_normal(100)), - pd.Series(rng.standard_normal(100)), - ], - axis=1, - ) - - test.columns = ["var_a", "var_b"] +@pytest.mark.parametrize("return_object", [False, True]) +def test_error_when_nan_introduced_during_transform(make_df, return_object): + # values outside the bin edges learned in fit become NaN, which warns or raises + train = make_df({"var_a": [-4.0, -1.0, 1.0, 4.0], "var_b": [1.0, 2.0, 3.0, 4.0]}) + test = make_df({"var_a": [-4.0, -1.0, 1.0, 4.0], "var_b": [10.0, 20.0, 30.0, 40.0]}) msg = ( "During the discretisation, NaN values were introduced " @@ -98,40 +122,17 @@ def test_error_when_nan_introduced_during_transform(): limits_dict = {"var_a": [-5, -2, 0, 2, 5], "var_b": [0, 2, 5]} # check for warning when errors equals 'ignore' - with pytest.warns(UserWarning) as record: - transformer = ArbitraryDiscretiser(binning_dict=limits_dict, errors="ignore") - transformer.fit(train) + transformer = ArbitraryDiscretiser( + binning_dict=limits_dict, return_object=return_object, errors="ignore" + ) + transformer.fit(train) + with pytest.warns(UserWarning, match=re.escape(msg)): transformer.transform(test) - # check that only one warning was returned - assert len(record) == 1 - # check that message matches - assert record[0].message.args[0] == msg - # check for error when errors equals 'raise' - with pytest.raises(ValueError) as record: - transformer = ArbitraryDiscretiser(binning_dict=limits_dict, errors="raise") - transformer.fit(train) - transformer.transform(test) - - # check that error message matches - assert str(record.value) == msg - - -def test_error_if_not_permitted_value_is_errors(): - age_dict = {"Age": [0, 10, 20, 30, np.inf]} - with pytest.raises(ValueError): - ArbitraryDiscretiser(binning_dict=age_dict, errors="medialuna") - - -@pytest.mark.parametrize("binning_dict", ["HOLA", 1, False]) -def test_error_if_binning_dict_not_dict_type(binning_dict): - msg = ( - "binning_dict must be a dictionary with the interval limits per " - f"variable. Got {binning_dict} instead." + transformer = ArbitraryDiscretiser( + binning_dict=limits_dict, return_object=return_object, errors="raise" ) - with pytest.raises(ValueError) as record: - ArbitraryDiscretiser(binning_dict=binning_dict) - - # check that error message matches - assert str(record.value) == msg + transformer.fit(train) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.transform(test) diff --git a/tests/test_discretisation/test_base_discretizer.py b/tests/test_discretisation/test_base_discretizer.py index fc8110ff1..6cde4bd55 100644 --- a/tests/test_discretisation/test_base_discretizer.py +++ b/tests/test_discretisation/test_base_discretizer.py @@ -1,79 +1,85 @@ +import re + import numpy as np import pandas as pd import pytest -from sklearn.datasets import fetch_california_housing from feature_engine.discretisation.base_discretiser import BaseDiscretiser +from tests.backend_helpers import frame_to_dict + +BINS = [0, 20, 40, 60, np.inf] -# test init params -@pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2]) +# init parameters +@pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2, None]) def test_raises_error_when_return_object_not_bool(param): - with pytest.raises(ValueError): + msg = f"return_object must be True or False. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): BaseDiscretiser(return_object=param) -@pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2]) +@pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2, None]) def test_raises_error_when_return_boundaries_not_bool(param): - with pytest.raises(ValueError): + msg = f"return_boundaries must be True or False. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): BaseDiscretiser(return_boundaries=param) -@pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 0, -1]) +@pytest.mark.parametrize( + "param", [0.1, "hola", (True, False), {"a": True}, 0, -1, None] +) def test_raises_error_when_precision_not_int(param): - with pytest.raises(ValueError): + msg = f"precision must be a positive integer. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): BaseDiscretiser(precision=param) -@pytest.mark.parametrize("params", [(False, 1), (True, 10)]) -def test_correct_param_assignment_at_init(params): - param1, param2 = params - t = BaseDiscretiser( - return_object=param1, return_boundaries=param1, precision=param2 +@pytest.mark.parametrize( + "return_object, return_boundaries, precision", + [(False, False, 1), (True, False, 10), (False, True, 3)], +) +def test_init_param_assignment(return_object, return_boundaries, precision): + transformer = BaseDiscretiser( + return_object=return_object, + return_boundaries=return_boundaries, + precision=precision, ) - assert t.return_object is param1 - assert t.return_boundaries is param1 - assert t.precision == param2 + assert transformer.return_object is return_object + assert transformer.return_boundaries is return_boundaries + assert transformer.precision == precision +# fit and transform class MockClassFit(BaseDiscretiser): def fit(self, X): - california_dataset = fetch_california_housing() - data = pd.DataFrame( - california_dataset.data, columns=california_dataset.feature_names - ) + # bins are hard-coded rather than learnt, so this mock works unchanged + # on both pandas and polars input. self.variables_ = ["HouseAge"] - self.binner_dict_ = {"HouseAge": [0, 20, 40, 60, np.inf]} - self.n_features_in_ = data.shape[1] - self.feature_names_in_ = california_dataset.feature_names + self.binner_dict_ = {"HouseAge": BINS} + self.n_features_in_ = X.shape[1] + self.feature_names_in_ = list(X.columns) return self -def test_transform(): - california_dataset = fetch_california_housing() - data = pd.DataFrame( - california_dataset.data, columns=california_dataset.feature_names +def test_transform(make_df, data_california): + # ground truth via pandas.cut: bins are fixed by MockClassFit, so both + # backends must reproduce this exact output. + house_age = pd.Series(data_california["HouseAge"]) + expected_codes = pd.cut( + house_age, bins=BINS, labels=False, include_lowest=True + ).tolist() + expected_labels = ( + pd.cut(house_age, bins=BINS, include_lowest=True).astype(str).tolist() ) - data_t1 = data.copy() - data_t2 = data.copy() - - # HouseAge is the median house age in the block group. - data_t1["HouseAge"] = pd.cut( - data["HouseAge"], bins=[0, 20, 40, 60, np.inf], include_lowest=True - ) - data_t1["HouseAge"] = data_t1["HouseAge"].astype(str) - data_t2["HouseAge"] = pd.cut( - data["HouseAge"], - bins=[0, 20, 40, 60, np.inf], - labels=False, - include_lowest=True, - ) + data = make_df(data_california) transformer = MockClassFit(return_boundaries=False) X = transformer.fit_transform(data) - pd.testing.assert_frame_equal(X, data_t2) + assert isinstance(X, make_df) + assert frame_to_dict(X)["HouseAge"] == expected_codes transformer = MockClassFit(return_object=False, return_boundaries=True) X = transformer.fit_transform(data) - pd.testing.assert_frame_equal(X, data_t1) + assert isinstance(X, make_df) + assert frame_to_dict(X)["HouseAge"] == expected_labels diff --git a/tests/test_discretisation/test_check_estimator_discretisers.py b/tests/test_discretisation/test_check_estimator_discretisers.py index 7867035d0..9d47e6593 100644 --- a/tests/test_discretisation/test_check_estimator_discretisers.py +++ b/tests/test_discretisation/test_check_estimator_discretisers.py @@ -1,10 +1,8 @@ import numpy as np import pandas as pd import pytest -import sklearn from sklearn.pipeline import Pipeline from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine.discretisation import ( ArbitraryDiscretiser, @@ -19,9 +17,6 @@ ) from tests.estimator_checks.sklearn_check_wrapper import wrap_for_check_estimator -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) - - _estimators = [ DecisionTreeDiscretiser(regression=False), EqualFrequencyDiscretiser(), @@ -30,20 +25,13 @@ GeometricWidthDiscretiser(), ] -if sklearn_version < parse_version("1.6"): - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator(estimator) -else: - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator( - estimator=wrap_for_check_estimator(estimator), - expected_failed_checks=estimator._more_tags()["_xfail_checks"], - ) +@pytest.mark.parametrize("estimator", _estimators) +def test_check_estimator_from_sklearn(estimator): + return check_estimator( + estimator=wrap_for_check_estimator(estimator), + expected_failed_checks=estimator._more_tags()["_xfail_checks"], + ) @pytest.mark.parametrize("estimator", _estimators) diff --git a/tests/test_discretisation/test_decision_tree_discretiser.py b/tests/test_discretisation/test_decision_tree_discretiser.py index a90d64ab8..c380384ca 100644 --- a/tests/test_discretisation/test_decision_tree_discretiser.py +++ b/tests/test_discretisation/test_decision_tree_discretiser.py @@ -1,44 +1,49 @@ +import re + import numpy as np import pandas as pd import pytest from sklearn.exceptions import NotFittedError -from feature_engine.discretisation import DecisionTreeDiscretiser, EqualWidthDiscretiser +from feature_engine.discretisation import DecisionTreeDiscretiser +from tests.backend_helpers import make_series, frame_to_dict + +_rng = np.random.RandomState(42) +DATA_TWO_VARS = { + "var_A": _rng.normal(0, 3, 20).tolist(), + "var_B": _rng.normal(3, 5, 20).tolist(), +} +TARGET_TWO_VARS = [0, 1, 1, 0, 1, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1] + + +def _binary_target(): + np.random.seed(0) + return np.random.binomial(1, 0.7, 100).tolist() + + +def _continuous_target(): + np.random.seed(0) + return np.random.normal(0, 0.1, 100).tolist() # init parameters @pytest.mark.parametrize( - "params", - [("prediction", 3, True), ("bin_number", 10, False), ("boundaries", 1, False)], + "bin_output_", ["arbitrary", "Prediction", "", False, 1, None, ["prediction"]] ) -def test_init_param_assignment(params): - dsc = DecisionTreeDiscretiser( - bin_output=params[0], - precision=params[1], - regression=params[2], - ) - assert dsc.bin_output == params[0] - assert dsc.precision == params[1] - assert dsc.regression == params[2] - - -@pytest.mark.parametrize("bin_output_", ["arbitrary", False, 1]) def test_error_if_binoutput_not_permitted_value(bin_output_): msg = ( "bin_output takes values 'prediction', 'bin_number' or 'boundaries'. " f"Got {bin_output_} instead." ) - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(msg)): DecisionTreeDiscretiser(bin_output=bin_output_) - assert str(record.value) == msg @pytest.mark.parametrize("precision_", ["arbitrary", -1, 0.3]) def test_error_if_precision_not_permitted_value(precision_): msg = "precision must be None or a positive integer. " f"Got {precision_} instead." - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(msg)): DecisionTreeDiscretiser(precision=precision_) - assert str(record.value) == msg def test_precision_errors_if_none_when_bin_output_is_boundaries(): @@ -46,42 +51,95 @@ def test_precision_errors_if_none_when_bin_output_is_boundaries(): "When `bin_output == 'boundaries', `precision` cannot be None. " "Change precision's value to a positive integer." ) - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(msg)): DecisionTreeDiscretiser(precision=None, bin_output="boundaries") - assert str(record.value) == msg - - dsc = DecisionTreeDiscretiser(precision=None, bin_output="bin_number") - assert dsc.precision is None -@pytest.mark.parametrize("regression_", ["arbitrary", -1, 0.3]) +@pytest.mark.parametrize("regression_", ["arbitrary", -1, 0.3, 1, None]) def test_error_if_regression_is_not_bool(regression_): msg = "regression can only take True or False. " f"Got {regression_} instead." - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(msg)): DecisionTreeDiscretiser(regression=regression_) - assert str(record.value) == msg -# fit -def test_error_if_y_not_passed(df_normal_dist): +@pytest.mark.parametrize( + "params", + [ + { + "bin_output": "prediction", + "precision": 3, + "cv": 3, + "scoring": "neg_mean_squared_error", + "param_grid": None, + "regression": True, + "random_state": None, + "n_jobs": None, + }, + { + "bin_output": "bin_number", + "precision": None, + "cv": 5, + "scoring": "roc_auc", + "param_grid": {"max_depth": [1, 2]}, + "regression": False, + "random_state": 0, + "n_jobs": -1, + }, + { + "bin_output": "boundaries", + "precision": 1, + "cv": 2, + "scoring": "accuracy", + "param_grid": {"max_depth": [3]}, + "regression": False, + "random_state": 42, + "n_jobs": 2, + }, + ], +) +def test_init_param_assignment(params): + transformer = DecisionTreeDiscretiser(**params) + for param, value in params.items(): + assert getattr(transformer, param) == value + + +# fit and transform +def test_error_if_y_not_passed(make_df, data_normal_dist): encoder = DecisionTreeDiscretiser() - with pytest.raises(TypeError): - encoder.fit(df_normal_dist) + msg = "DecisionTreeDiscretiser.fit() missing 1 required positional argument: 'y'" + with pytest.raises(TypeError, match=re.escape(msg)): + encoder.fit(make_df(data_normal_dist)) -def test_error_when_regression_is_true_and_target_is_binary(df_discretise): +def test_error_when_regression_is_true_and_target_is_binary(make_df): + X = make_df(DATA_TWO_VARS) + y = make_series(make_df, TARGET_TWO_VARS) msg = ( "Trying to fit a regression to a binary target is not " "allowed by this transformer. Check the target values " "or set regression to False." ) transformer = DecisionTreeDiscretiser(regression=True) - with pytest.raises(ValueError) as record: - transformer.fit(df_discretise[["var_A", "var_B"]], df_discretise["target"]) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.fit(X, y) -def test_classification_predictions(df_normal_dist): +def test_error_when_regression_is_false_and_target_is_continuous(make_df): + X = make_df(DATA_TWO_VARS) + np.random.seed(42) + y = make_series(make_df, np.random.normal(0, 3, 20).tolist()) + transformer = DecisionTreeDiscretiser(regression=False) + msg = ( + "Unknown label type: continuous. Maybe you are trying to fit a classifier, " + "which expects discrete classes on a regression target with continuous values." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.fit(X, y) + + +def test_classification_predictions(make_df, data_normal_dist): + X = make_df(data_normal_dist) + y = make_series(make_df, _binary_target()) transformer = DecisionTreeDiscretiser( cv=3, @@ -91,26 +149,44 @@ def test_classification_predictions(df_normal_dist): regression=False, random_state=0, ) - np.random.seed(0) - y = pd.Series(np.random.binomial(1, 0.7, 100)) - X = transformer.fit_transform(df_normal_dist, y) + Xt = transformer.fit_transform(X, y) X_t = [1.0, 0.71, 0.93, 0.0] - # init params - assert transformer.cv == 3 - assert transformer.variables is None - assert transformer.scoring == "roc_auc" - assert transformer.regression is False # fit params assert transformer.variables_ == ["var"] assert transformer.n_features_in_ == 1 # transform params - assert all(x for x in np.round(X["var"].unique(), 2) if x not in X_t) + assert isinstance(Xt, make_df) + unique_vals = sorted(set(frame_to_dict(Xt)["var"])) + assert all(x for x in np.round(unique_vals, 2) if x not in X_t) assert np.round(transformer.scores_dict_["var"], 3) == np.round( 0.717391304347826, 3 ) +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_or_array(make_df, data_normal_dist, to_target): + # a list or numpy array target must give the same result as a Series + X = make_df(data_normal_dist) + params = dict( + bin_output="bin_number", + scoring="roc_auc", + param_grid={"max_depth": [1, 2, 3, 4]}, + regression=False, + random_state=0, + ) + + from_series = DecisionTreeDiscretiser(**params) + from_series.fit(X, make_series(make_df, _binary_target())) + transformer = DecisionTreeDiscretiser(**params) + transformer.fit(X, to_target(_binary_target())) + Xt = transformer.transform(X) + + assert transformer.binner_dict_ == from_series.binner_dict_ + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == frame_to_dict(from_series.transform(X)) + + @pytest.mark.parametrize( "params", [ @@ -119,7 +195,9 @@ def test_classification_predictions(df_normal_dist): (3, [1.0, 0.712, 0.933, 0.0]), ], ) -def test_classification_rounds_predictions(df_normal_dist, params): +def test_classification_rounds_predictions(make_df, data_normal_dist, params): + X = make_df(data_normal_dist) + y = make_series(make_df, _binary_target()) transformer = DecisionTreeDiscretiser( precision=params[0], @@ -130,15 +208,15 @@ def test_classification_rounds_predictions(df_normal_dist, params): regression=False, random_state=0, ) - np.random.seed(0) - y = pd.Series(np.random.binomial(1, 0.7, 100)) - X = transformer.fit_transform(df_normal_dist, y) - bins = params[1] + Xt = transformer.fit_transform(X, y) - assert list(X["var"].unique()) == bins + assert isinstance(Xt, make_df) + assert sorted(set(frame_to_dict(Xt)["var"])) == sorted(params[1]) -def test_classification_bin_number(df_normal_dist): +def test_classification_bin_number(make_df, data_normal_dist): + X = make_df(data_normal_dist) + y = make_series(make_df, _binary_target()) transformer = DecisionTreeDiscretiser( bin_output="bin_number", scoring="roc_auc", @@ -146,10 +224,8 @@ def test_classification_bin_number(df_normal_dist): regression=False, random_state=0, ) - np.random.seed(0) - y = pd.Series(np.random.binomial(1, 0.7, 100)) - X = transformer.fit_transform(df_normal_dist, y) - bins = [4, 2, 1, 0, 3] + Xt = transformer.fit_transform(X, y) + bins = [0, 1, 2, 3, 4] limits = [ -np.inf, -0.22668930888175964, @@ -163,10 +239,13 @@ def test_classification_bin_number(df_normal_dist): assert np.round(transformer.scores_dict_["var"], 3) == np.round( 0.717391304347826, 3 ) - assert list(X["var"].unique()) == bins + assert isinstance(Xt, make_df) + assert sorted(set(frame_to_dict(Xt)["var"])) == bins -def test_classification_boundaries(df_normal_dist): +def test_classification_boundaries(make_df, data_normal_dist): + X = make_df(data_normal_dist) + y = make_series(make_df, _binary_target()) transformer = DecisionTreeDiscretiser( bin_output="boundaries", precision=3, @@ -175,16 +254,16 @@ def test_classification_boundaries(df_normal_dist): regression=False, random_state=0, ) - np.random.seed(0) - y = pd.Series(np.random.binomial(1, 0.7, 100)) - X = transformer.fit_transform(df_normal_dist, y) - bins = [ - "(0.116, inf]", - "(-0.0942, 0.102]", - "(-0.227, -0.0942]", - "(-inf, -0.227]", - "(0.102, 0.116]", - ] + Xt = transformer.fit_transform(X, y) + bins = sorted( + [ + "(0.116, inf]", + "(-0.0942, 0.102]", + "(-0.227, -0.0942]", + "(-inf, -0.227]", + "(0.102, 0.116]", + ] + ) limits = [ -np.inf, -0.22668930888175964, @@ -198,10 +277,13 @@ def test_classification_boundaries(df_normal_dist): assert np.round(transformer.scores_dict_["var"], 3) == np.round( 0.717391304347826, 3 ) - assert list(X["var"].unique()) == bins + assert isinstance(Xt, make_df) + assert sorted(set(frame_to_dict(Xt)["var"])) == bins -def test_regression(df_normal_dist): +def test_regression(make_df, data_normal_dist): + X = make_df(data_normal_dist) + y = make_series(make_df, _continuous_target()) transformer = DecisionTreeDiscretiser( cv=3, @@ -211,9 +293,7 @@ def test_regression(df_normal_dist): regression=True, random_state=0, ) - np.random.seed(0) - y = pd.Series(pd.Series(np.random.normal(0, 0.1, 100))) - X = transformer.fit_transform(df_normal_dist, y) + Xt = transformer.fit_transform(X, y) X_t = [ 0.19, 0.04, @@ -233,11 +313,6 @@ def test_regression(df_normal_dist): -0.12, ] - # init params - assert transformer.cv == 3 - assert transformer.variables is None - assert transformer.scoring == "neg_mean_squared_error" - assert transformer.regression is True # fit params assert transformer.variables_ == ["var"] assert transformer.n_features_in_ == 1 @@ -245,7 +320,9 @@ def test_regression(df_normal_dist): -4.4373314584616444e-05, 3 ) # transform params - assert all(x for x in np.round(X["var"].unique(), 2) if x not in X_t) + assert isinstance(Xt, make_df) + unique_vals = sorted(set(frame_to_dict(Xt)["var"])) + assert all(x for x in np.round(unique_vals, 2) if x not in X_t) @pytest.mark.parametrize( @@ -275,7 +352,9 @@ def test_regression(df_normal_dist): ), ], ) -def test_regression_rounds_predictions(df_normal_dist, params): +def test_regression_rounds_predictions(make_df, data_normal_dist, params): + X = make_df(data_normal_dist) + y = make_series(make_df, _continuous_target()) transformer = DecisionTreeDiscretiser( precision=params[0], @@ -286,43 +365,53 @@ def test_regression_rounds_predictions(df_normal_dist, params): regression=True, random_state=0, ) - np.random.seed(0) - y = pd.Series(pd.Series(np.random.normal(0, 0.1, 100))) - X = transformer.fit_transform(df_normal_dist, y) - bins = params[1] + Xt = transformer.fit_transform(X, y) - assert list(X["var"].unique()) == bins + assert isinstance(Xt, make_df) + assert sorted(set(frame_to_dict(Xt)["var"])) == sorted(params[1]) -# transform -def test_non_fitted_error(df_vartypes): - with pytest.raises(NotFittedError): - transformer = EqualWidthDiscretiser() - transformer.transform(df_vartypes) +@pytest.mark.parametrize("bin_output", ["prediction", "bin_number", "boundaries"]) +def test_integer_column_names(bin_output): + # integer column names are pandas-only + X = pd.DataFrame({0: DATA_TWO_VARS["var_A"], 1: DATA_TWO_VARS["var_B"]}) + y = pd.Series(TARGET_TWO_VARS) + params = dict(bin_output=bin_output, precision=3, regression=False, random_state=0) + Xt = DecisionTreeDiscretiser(**params).fit_transform(X, y) + expected = DecisionTreeDiscretiser(**params).fit_transform(X.rename(columns=str), y) -@pytest.fixture(scope="module") -def df_discretise(): - np.random.seed(42) - mu1, sigma1 = 0, 3 - s1 = np.random.normal(mu1, sigma1, 20) - mu2, sigma2 = 3, 5 - s2 = np.random.normal(mu2, sigma2, 20) - data = { - "var_A": s1, - "var_B": s2, - "target": [0, 1, 1, 0, 1, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1], - } + assert list(Xt.columns) == [0, 1] + assert Xt.to_numpy().tolist() == expected.to_numpy().tolist() - df = pd.DataFrame(data) - return df +def test_non_fitted_error(make_df, data_normal_dist): + transformer = DecisionTreeDiscretiser() + msg = ( + "This DecisionTreeDiscretiser instance is not fitted yet. Call 'fit' " + "with appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + transformer.transform(make_df(data_normal_dist)) -def test_error_when_regression_is_false_and_target_is_continuous(df_discretise): - np.random.seed(42) - mu, sigma = 0, 3 - y = np.random.normal(mu, sigma, len(df_discretise)) - transformer = DecisionTreeDiscretiser(regression=False) - with pytest.raises(ValueError): - transformer.fit(df_discretise[["var_A", "var_B"]], y) +def test_n_jobs_parallel_matches_sequential(make_df): + # parallel tree training must give the same trees and predictions as sequential + X = make_df(DATA_TWO_VARS) + np.random.seed(0) + y = make_series(make_df, np.random.normal(0, 1, 20).tolist()) + + tr_seq = DecisionTreeDiscretiser( + n_jobs=None, random_state=0, param_grid={"max_depth": [1, 2, 3]} + ) + tr_seq.fit(X, y) + tr_par = DecisionTreeDiscretiser( + n_jobs=2, random_state=0, param_grid={"max_depth": [1, 2, 3]} + ) + tr_par.fit(X, y) + + Xt_seq = tr_seq.transform(X) + Xt_par = tr_par.transform(X) + + assert isinstance(Xt_par, make_df) + assert frame_to_dict(Xt_par) == frame_to_dict(Xt_seq) diff --git a/tests/test_discretisation/test_equal_frequency_discretiser.py b/tests/test_discretisation/test_equal_frequency_discretiser.py index 112262dd1..1291903c6 100644 --- a/tests/test_discretisation/test_equal_frequency_discretiser.py +++ b/tests/test_discretisation/test_equal_frequency_discretiser.py @@ -1,70 +1,113 @@ +import re +from collections import Counter + +import narwhals as nw import pandas as pd import pytest from sklearn.exceptions import NotFittedError from feature_engine.discretisation import EqualFrequencyDiscretiser - - -def test_automatically_find_variables_and_return_as_numeric(df_normal_dist): +from tests.backend_helpers import frame_to_dict + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) + + +# init parameters +@pytest.mark.parametrize("q", ["other", 1.5, None, [10]]) +def test_error_when_q_not_number(q): + msg = f"q must be an integer. Got {q} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EqualFrequencyDiscretiser(q=q) + + +@pytest.mark.parametrize("return_object", ["other", 1, None]) +def test_error_if_return_object_not_bool(return_object): + msg = f"return_object must be True or False. Got {return_object} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EqualFrequencyDiscretiser(return_object=return_object) + + +@pytest.mark.parametrize( + "q, return_object, return_boundaries, precision", + [(10, False, False, 3), (5, True, False, 1), (2, False, True, 7)], +) +def test_init_param_assignment(q, return_object, return_boundaries, precision): + transformer = EqualFrequencyDiscretiser( + q=q, + return_object=return_object, + return_boundaries=return_boundaries, + precision=precision, + ) + assert transformer.q == q + assert transformer.return_object is return_object + assert transformer.return_boundaries is return_boundaries + assert transformer.precision == precision + + +# fit and transform + +def test_automatically_find_variables_and_return_as_numeric( + make_df, data_normal_dist +): # test case 1: automatically select variables, return_object=False transformer = EqualFrequencyDiscretiser(q=10, variables=None, return_object=False) - X = transformer.fit_transform(df_normal_dist) - - # output expected for fit attr - _, bins = pd.qcut(x=df_normal_dist["var"], q=10, retbins=True, duplicates="drop") + X = transformer.fit_transform(make_df(data_normal_dist)) + + # output expected for fit attr, computed via pandas.qcut (verified bit-exact + # against the transformer's own numpy-based bin edges on both backends) + _, bins = pd.qcut( + x=pd.Series(data_normal_dist["var"]), q=10, retbins=True, duplicates="drop" + ) + bins = list(bins) bins[0] = float("-inf") bins[len(bins) - 1] = float("inf") - # expected transform output - X_t = [x for x in range(0, 10)] - - # test init params - assert transformer.q == 10 - assert transformer.variables is None - assert transformer.return_object is False # test fit attr assert transformer.variables_ == ["var"] assert transformer.n_features_in_ == 1 + assert transformer.binner_dict_["var"] == bins # test transform output - assert (transformer.binner_dict_["var"] == bins).all() - assert all(x for x in X["var"].unique() if x not in X_t) + assert isinstance(X, make_df) + values = frame_to_dict(X)["var"] + assert set(values) == set(range(10)) # in equal frequency discretisation, all intervals get same proportion of values - assert len((X["var"].value_counts()).unique()) == 1 + assert len(set(Counter(values).values())) == 1 -def test_automatically_find_variables_and_return_as_object(df_normal_dist): +def test_automatically_find_variables_and_return_as_object(make_df, data_normal_dist): # test case 2: return variables cast as object transformer = EqualFrequencyDiscretiser(q=10, variables=None, return_object=True) - X = transformer.fit_transform(df_normal_dist) - assert X["var"].dtypes == "O" - + X = transformer.fit_transform(make_df(data_normal_dist)) + assert isinstance(X, make_df) + assert nw.from_native(X, eager_only=True).schema["var"] == nw.Object -def test_error_when_q_not_number(): - with pytest.raises(ValueError): - EqualFrequencyDiscretiser(q="other") - -def test_error_if_return_object_not_bool(): - with pytest.raises(ValueError): - EqualFrequencyDiscretiser(return_object="other") - - -def test_error_if_input_df_contains_na_in_fit(df_na): +def test_error_if_input_df_contains_na_in_fit(make_df, data_na): # test case 3: when dataset contains na, fit method - with pytest.raises(ValueError): - transformer = EqualFrequencyDiscretiser() - transformer.fit(df_na) + transformer = EqualFrequencyDiscretiser() + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.fit(make_df(data_na)) -def test_error_if_input_df_contains_na_in_transform(df_vartypes, df_na): +def test_error_if_input_df_contains_na_in_transform(make_df, data_vartypes, data_na): # test case 4: when dataset contains na, transform method - with pytest.raises(ValueError): - transformer = EqualFrequencyDiscretiser() - transformer.fit(df_vartypes) - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) - - -def test_non_fitted_error(df_vartypes): - with pytest.raises(NotFittedError): - transformer = EqualFrequencyDiscretiser() - transformer.transform(df_vartypes) + transform_data = make_df( + {k: data_na[k] for k in ["Name", "City", "Age", "Marks", "dob"]} + ) + transformer = EqualFrequencyDiscretiser() + transformer.fit(make_df(data_vartypes)) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(transform_data) + + +def test_non_fitted_error(make_df, data_vartypes): + transformer = EqualFrequencyDiscretiser() + msg = ( + "This EqualFrequencyDiscretiser instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + transformer.transform(make_df(data_vartypes)) diff --git a/tests/test_discretisation/test_equal_width_discretiser.py b/tests/test_discretisation/test_equal_width_discretiser.py index 88782af91..7a85f1b94 100644 --- a/tests/test_discretisation/test_equal_width_discretiser.py +++ b/tests/test_discretisation/test_equal_width_discretiser.py @@ -1,70 +1,139 @@ +import re + +import narwhals as nw +import numpy as np import pandas as pd import pytest from sklearn.exceptions import NotFittedError from feature_engine.discretisation import EqualWidthDiscretiser +from tests.backend_helpers import frame_to_dict + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) + + +# init parameters +@pytest.mark.parametrize("bins", ["other", 1.5, None, [10]]) +def test_error_when_bins_not_number(bins): + msg = f"bins must be an integer. Got {bins} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EqualWidthDiscretiser(bins=bins) + + +@pytest.mark.parametrize("return_object", ["other", 1, None]) +def test_error_if_return_object_not_bool(return_object): + msg = f"return_object must be True or False. Got {return_object} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EqualWidthDiscretiser(return_object=return_object) + + +@pytest.mark.parametrize( + "bins, return_object, return_boundaries, precision", + [(10, False, False, 3), (5, True, False, 1), (2, False, True, 7)], +) +def test_init_param_assignment(bins, return_object, return_boundaries, precision): + transformer = EqualWidthDiscretiser( + bins=bins, + return_object=return_object, + return_boundaries=return_boundaries, + precision=precision, + ) + assert transformer.bins == bins + assert transformer.return_object is return_object + assert transformer.return_boundaries is return_boundaries + assert transformer.precision == precision + + +# fit and transform + +def _expected_bins_and_codes(values, n_bins): + # ground truth bin edges via pandas.cut, same widening/duplicates-drop + # rules the fit() replicates in plain numpy. + series = pd.Series(values) + _, bins = pd.cut(x=series, bins=n_bins, retbins=True, duplicates="drop") + bins[0] = float("-inf") + bins[len(bins) - 1] = float("inf") + codes = pd.cut(series, bins=list(bins), labels=False, include_lowest=True) + return bins, codes.tolist() -def test_automatically_find_variables_and_return_as_numeric(df_normal_dist): - # test case 1: automatically select variables, return_object=False +def test_automatically_find_variables_and_return_as_numeric( + make_df, data_normal_dist +): transformer = EqualWidthDiscretiser(bins=10, variables=None, return_object=False) - X = transformer.fit_transform(df_normal_dist) - - # fit parameters - _, bins = pd.cut(x=df_normal_dist["var"], bins=10, retbins=True, duplicates="drop") - bins[0] = float("-inf") - bins[len(bins) - 1] = float("inf") + X = transformer.fit_transform(make_df(data_normal_dist)) - # transform output - X_t = [x for x in range(0, 10)] - val_counts = [18, 17, 16, 13, 11, 7, 7, 5, 5, 1] + bins, expected_codes = _expected_bins_and_codes(data_normal_dist["var"], 10) - # init params - assert transformer.bins == 10 - assert transformer.variables is None - assert transformer.return_object is False # fit params assert transformer.variables_ == ["var"] assert transformer.n_features_in_ == 1 - # transform params - assert (transformer.binner_dict_["var"] == bins).all() - assert all(x for x in X["var"].unique() if x not in X_t) - # in equal width discretisation, intervals get different number of values - assert all(x for x in X["var"].value_counts() if x not in val_counts) + assert np.allclose(transformer.binner_dict_["var"], bins) + # transform params: same bin codes on both backends + assert isinstance(X, make_df) + assert frame_to_dict(X)["var"] == expected_codes -def test_automatically_find_variables_and_return_as_object(df_normal_dist): +def test_automatically_find_variables_and_return_as_object(make_df, data_normal_dist): transformer = EqualWidthDiscretiser(bins=10, variables=None, return_object=True) - X = transformer.fit_transform(df_normal_dist) - assert X["var"].dtypes == "O" + X = transformer.fit_transform(make_df(data_normal_dist)) + assert isinstance(X, make_df) + assert nw.from_native(X, eager_only=True).schema["var"] == nw.Object + + +def test_constant_variable_produces_single_bin(make_df): + # a constant variable still fits, with every value in the same bin, as with + # pandas.cut(bins=10) + data = {"var": [5.0] * 10} + transformer = EqualWidthDiscretiser(bins=10) + X = transformer.fit_transform(make_df(data)) + + _, expected_codes = _expected_bins_and_codes(data["var"], 10) + + assert transformer.binner_dict_["var"][0] == float("-inf") + assert transformer.binner_dict_["var"][-1] == float("inf") + assert isinstance(X, make_df) + assert frame_to_dict(X)["var"] == expected_codes -def test_error_when_bins_not_number(): - with pytest.raises(ValueError): - EqualWidthDiscretiser(bins="other") +def test_error_if_input_df_contains_na_in_fit(make_df, data_na): + transformer = EqualWidthDiscretiser() + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.fit(make_df(data_na)) -def test_error_if_return_object_not_bool(): - with pytest.raises(ValueError): - EqualWidthDiscretiser(return_object="other") +def test_error_if_input_df_contains_na_in_transform(make_df, data_vartypes, data_na): + transform_data = make_df( + {k: data_na[k] for k in ["Name", "City", "Age", "Marks", "dob"]} + ) + transformer = EqualWidthDiscretiser() + transformer.fit(make_df(data_vartypes)) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(transform_data) -def test_error_if_input_df_contains_na_in_fit(df_na): - # test case 3: when dataset contains na, fit method - with pytest.raises(ValueError): - transformer = EqualWidthDiscretiser() - transformer.fit(df_na) +def test_integer_column_names(data_normal_dist): + # integer column names are pandas-only + values = data_normal_dist["var"] + X = pd.DataFrame({0: values, 1: [2 * v for v in values]}) + transformer = EqualWidthDiscretiser(bins=10) + Xt = transformer.fit_transform(X) + expected = EqualWidthDiscretiser(bins=10).fit_transform(X.rename(columns=str)) -def test_error_if_input_df_contains_na_in_transform(df_vartypes, df_na): - # test case 4: when dataset contains na, transform method - with pytest.raises(ValueError): - transformer = EqualWidthDiscretiser() - transformer.fit(df_vartypes) - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + assert list(transformer.binner_dict_) == [0, 1] + assert list(Xt.columns) == [0, 1] + assert Xt.to_numpy().tolist() == expected.to_numpy().tolist() -def test_non_fitted_error(df_vartypes): - with pytest.raises(NotFittedError): - transformer = EqualWidthDiscretiser() - transformer.transform(df_vartypes) +def test_non_fitted_error(make_df, data_vartypes): + transformer = EqualWidthDiscretiser() + msg = ( + "This EqualWidthDiscretiser instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + transformer.transform(make_df(data_vartypes)) diff --git a/tests/test_discretisation/test_geometric_width_discretiser.py b/tests/test_discretisation/test_geometric_width_discretiser.py index 6a4b56c2d..82a70add2 100644 --- a/tests/test_discretisation/test_geometric_width_discretiser.py +++ b/tests/test_discretisation/test_geometric_width_discretiser.py @@ -1,38 +1,51 @@ +import re + +import narwhals as nw import numpy as np import pandas as pd import pytest from sklearn.exceptions import NotFittedError from feature_engine.discretisation import GeometricWidthDiscretiser +from tests.backend_helpers import frame_to_dict + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) -# test init params +# init parameters @pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2]) def test_raises_error_when_return_object_not_bool(param): - with pytest.raises(ValueError): + msg = f"return_object must be True or False. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): GeometricWidthDiscretiser(return_object=param) @pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2]) def test_raises_error_when_return_boundaries_not_bool(param): - with pytest.raises(ValueError): + msg = f"return_boundaries must be True or False. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): GeometricWidthDiscretiser(return_boundaries=param) @pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 0, -1]) def test_raises_error_when_precision_not_int(param): - with pytest.raises(ValueError): + msg = f"precision must be a positive integer. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): GeometricWidthDiscretiser(precision=param) -@pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}]) +@pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, None]) def test_raises_error_when_bins_not_int(param): - with pytest.raises(ValueError): + msg = f"bins must be an integer. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): GeometricWidthDiscretiser(bins=param) @pytest.mark.parametrize("params", [(False, 1), (True, 10)]) -def test_correct_param_assignment_at_init(params): +def test_init_param_assignment(params): param1, param2 = params t = GeometricWidthDiscretiser( return_object=param1, return_boundaries=param1, precision=param2, bins=param2 @@ -43,14 +56,16 @@ def test_correct_param_assignment_at_init(params): assert t.bins == param2 -def test_fit_and_transform_methods(df_normal_dist): +# fit and transform +def test_fit_and_transform_methods(make_df, data_normal_dist): transformer = GeometricWidthDiscretiser( bins=10, variables=None, return_object=False ) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(make_df(data_normal_dist)) # manual calculation - min_, max_ = df_normal_dist["var"].min(), df_normal_dist["var"].max() + arr = np.array(data_normal_dist["var"]) + min_, max_ = arr.min(), arr.max() increment = np.power(max_ - min_, 1.0 / 10) bins = np.r_[-np.inf, min_ + np.power(increment, np.arange(1, 10)), np.inf] bins = np.sort(bins) @@ -58,34 +73,43 @@ def test_fit_and_transform_methods(df_normal_dist): # fit params assert (transformer.binner_dict_["var"] == bins).all() - # transform params - assert ( - X["var"] == pd.cut(df_normal_dist["var"], bins=bins, precision=7).cat.codes - ).all() + # transform params - ground truth from pandas.cut on the same bins; values + # must match regardless of which backend the input dataframe uses. + expected = pd.cut(pd.Series(arr), bins=bins, precision=7).cat.codes.tolist() + assert isinstance(X, make_df) + assert frame_to_dict(X)["var"] == expected -def test_automatically_find_variables_and_return_as_object(df_normal_dist): +def test_automatically_find_variables_and_return_as_object(make_df, data_normal_dist): transformer = GeometricWidthDiscretiser(bins=10, variables=None, return_object=True) - X = transformer.fit_transform(df_normal_dist) - assert X["var"].dtypes == "O" + X = transformer.fit_transform(make_df(data_normal_dist)) + assert isinstance(X, make_df) + assert nw.from_native(X, eager_only=True).schema["var"] == nw.Object -def test_error_if_input_df_contains_na_in_fit(df_na): - # test case 3: when dataset contains na, fit method +def test_error_if_input_df_contains_na_in_fit(make_df): + df_na = make_df({"Age": [20.0, 21.0, None, 23.0]}) transformer = GeometricWidthDiscretiser() - with pytest.raises(ValueError): + with pytest.raises(ValueError, match=re.escape(MSG_NA)): transformer.fit(df_na) -def test_error_if_input_df_contains_na_in_transform(df_vartypes, df_na): - # test case 4: when dataset contains na, transform method +def test_error_if_input_df_contains_na_in_transform(make_df): + df = make_df({"Age": [20.0, 21.0, 19.0, 23.0]}) + df_na = make_df({"Age": [20.0, 21.0, None, 23.0]}) + transformer = GeometricWidthDiscretiser() - transformer.fit(df_vartypes) - with pytest.raises(ValueError): - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.fit(df) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(df_na) -def test_non_fitted_error(df_vartypes): +def test_non_fitted_error(make_df): + df = make_df({"Age": [20.0, 21.0, 19.0, 23.0]}) transformer = GeometricWidthDiscretiser() - with pytest.raises(NotFittedError): - transformer.transform(df_vartypes) + msg = ( + "This GeometricWidthDiscretiser instance is not fitted yet. Call 'fit' " + "with appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + transformer.transform(df) diff --git a/tests/test_encoding/conftest.py b/tests/test_encoding/conftest.py new file mode 100644 index 000000000..24387fa0c --- /dev/null +++ b/tests/test_encoding/conftest.py @@ -0,0 +1,111 @@ +"""Data shared by the encoder tests. + +Each fixture returns a fresh dict, so tests can build the dataframe on the +backend under test with `make_df(data)`. Missing values are written as None, +which both pandas and polars read as missing. +""" + +import pytest + +TARGET = [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0] + + +@pytest.fixture +def data_enc(): + return { + "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "target": list(TARGET), + } + + +@pytest.fixture +def data_enc_rare(): + return { + "var_A": ["B"] * 9 + ["A"] * 6 + ["C"] * 4 + ["D"] * 1, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "target": list(TARGET), + } + + +@pytest.fixture +def data_enc_na(): + return { + "var_A": [None] + ["B"] * 8 + ["A"] * 6 + ["C"] * 4 + ["D"] * 1, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "target": list(TARGET), + } + + +@pytest.fixture +def data_enc_numeric(): + return { + "var_A": [1] * 6 + [2] * 10 + [3] * 4, + "var_B": [1] * 10 + [2] * 6 + [3] * 4, + "target": list(TARGET), + } + + +def _data_enc_big(): + return { + "var_A": ["A"] * 6 + + ["B"] * 10 + + ["C"] * 4 + + ["D"] * 10 + + ["E"] * 2 + + ["F"] * 2 + + ["G"] * 6, + "var_B": ["A"] * 10 + + ["B"] * 6 + + ["C"] * 4 + + ["D"] * 10 + + ["E"] * 2 + + ["F"] * 2 + + ["G"] * 6, + "var_C": ["A"] * 4 + + ["B"] * 6 + + ["C"] * 10 + + ["D"] * 10 + + ["E"] * 2 + + ["F"] * 2 + + ["G"] * 6, + } + + +@pytest.fixture +def data_enc_big(): + return _data_enc_big() + + +@pytest.fixture +def data_enc_big_na(): + data = _data_enc_big() + data["var_A"][0] = None + return data + + +@pytest.fixture +def data_enc_top(): + return { + "var_A": ["A"] * 5 + + ["B"] * 11 + + ["C"] * 4 + + ["D"] * 9 + + ["E"] * 2 + + ["F"] * 2 + + ["G"] * 7, + "var_B": ["A"] * 11 + + ["B"] * 7 + + ["C"] * 4 + + ["D"] * 9 + + ["E"] * 2 + + ["F"] * 2 + + ["G"] * 5, + "var_C": ["A"] * 4 + + ["B"] * 5 + + ["C"] * 11 + + ["D"] * 9 + + ["E"] * 2 + + ["F"] * 2 + + ["G"] * 7, + } diff --git a/tests/test_encoding/test_base_encoders/test_categorical_init_mixin.py b/tests/test_encoding/test_base_encoders/test_categorical_init_mixin.py index 2c6e65fb4..daada0742 100644 --- a/tests/test_encoding/test_base_encoders/test_categorical_init_mixin.py +++ b/tests/test_encoding/test_base_encoders/test_categorical_init_mixin.py @@ -1,3 +1,5 @@ +import re + import pytest from feature_engine.encoding.base_encoder import CategoricalInitMixin @@ -5,10 +7,9 @@ @pytest.mark.parametrize("param", [1, "hola", [1, 2, 0], (True, False)]) def test_raises_error_when_ignore_format_not_permitted(param): - with pytest.raises(ValueError) as record: - CategoricalInitMixin(ignore_format=param) msg = f"ignore_format takes only booleans True and False. Got {param} instead." - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + CategoricalInitMixin(ignore_format=param) @pytest.mark.parametrize("param", [True, False]) diff --git a/tests/test_encoding/test_base_encoders/test_categorical_init_mixin_na.py b/tests/test_encoding/test_base_encoders/test_categorical_init_mixin_na.py index a7c112816..4eaebb376 100644 --- a/tests/test_encoding/test_base_encoders/test_categorical_init_mixin_na.py +++ b/tests/test_encoding/test_base_encoders/test_categorical_init_mixin_na.py @@ -1,3 +1,5 @@ +import re + import pytest from feature_engine.encoding.base_encoder import CategoricalInitMixinNA @@ -5,18 +7,18 @@ @pytest.mark.parametrize("param", [1, "hola", [1, 2, 0], (True, False)]) def test_raises_error_when_ignore_format_not_permitted(param): - with pytest.raises(ValueError) as record: - CategoricalInitMixinNA(ignore_format=param) msg = f"ignore_format takes only booleans True and False. Got {param} instead." - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + CategoricalInitMixinNA(ignore_format=param) -@pytest.mark.parametrize("param", [1, "hola", [1, 2, 0], (True, False)]) +@pytest.mark.parametrize( + "param", [1, "hola", "Raise", None, [1, 2, 0], ["raise"], (True, False)] +) def test_raises_error_when_missing_values_not_permitted(param): - with pytest.raises(ValueError) as record: - CategoricalInitMixinNA(missing_values=param) msg = f"missing_values takes only values 'raise' or 'ignore'. Got {param} instead." - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + CategoricalInitMixinNA(missing_values=param) @pytest.mark.parametrize("param", [(True, "ignore"), (False, "raise")]) diff --git a/tests/test_encoding/test_base_encoders/test_categorical_method_mixin.py b/tests/test_encoding/test_base_encoders/test_categorical_method_mixin.py index 80dc30813..609a4f9c4 100644 --- a/tests/test_encoding/test_base_encoders/test_categorical_method_mixin.py +++ b/tests/test_encoding/test_base_encoders/test_categorical_method_mixin.py @@ -1,3 +1,6 @@ +import re + +import narwhals as nw import numpy as np import pandas as pd import pytest @@ -23,14 +26,13 @@ def test_underscore_check_na_method(): variables = ["words", "animals"] enc = MockClassFit(missing_values="raise") - with pytest.raises(ValueError) as record: - enc._check_na(input_df, variables) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer or set the parameter " "`missing_values='ignore'` when initialising this transformer." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + enc._check_na(input_df, variables) def test_check_or_select_variables(): @@ -102,13 +104,11 @@ def test_raises_error_when_nan_introduced(): enc = MockClass(unseen="raise") msg = "During the encoding, NaN values were introduced in the feature(s) words." - with pytest.raises(ValueError) as record: - enc._check_nan_values_after_transformation(output_df) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + enc._check_nan_values_after_transformation(nw.from_native(output_df)) - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(msg)): enc.transform(input_df) - assert str(record.value) == msg def test_raises_warning_when_nan_introduced(): @@ -117,26 +117,23 @@ def test_raises_warning_when_nan_introduced(): enc = MockClass(unseen="ignore") msg = "During the encoding, NaN values were introduced in the feature(s) words." - with pytest.warns(UserWarning) as record: + with pytest.warns(UserWarning, match=re.escape(msg)): enc.transform(input_df) - assert record[0].message.args[0] == msg - with pytest.warns(UserWarning) as record: - enc._check_nan_values_after_transformation(output_df) - assert record[0].message.args[0] == msg + with pytest.warns(UserWarning, match=re.escape(msg)): + enc._check_nan_values_after_transformation(nw.from_native(output_df)) def test_transform_raises_error_when_df_has_nan(): input_df = pd.DataFrame({"words": ["dog", "dig", "cat", np.nan]}) enc = MockClass() - with pytest.raises(ValueError) as record: - enc.transform(input_df) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer or set the parameter " "`missing_values='ignore'` when initialising this transformer." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + enc.transform(input_df) def test_transform_ignores_nan_in_df_to_transform(): diff --git a/tests/test_encoding/test_check_estimator_encoders.py b/tests/test_encoding/test_check_estimator_encoders.py index fbe9ce00f..db6e3b7be 100644 --- a/tests/test_encoding/test_check_estimator_encoders.py +++ b/tests/test_encoding/test_check_estimator_encoders.py @@ -1,12 +1,10 @@ import pandas as pd import pytest -import sklearn from numpy import nan from sklearn import clone from sklearn.exceptions import NotFittedError from sklearn.pipeline import Pipeline from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine.encoding import ( CountEncoder, @@ -26,8 +24,6 @@ ) from tests.estimator_checks.sklearn_check_wrapper import wrap_for_check_estimator -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) - _estimators = [ CountEncoder(ignore_format=True), CountFrequencyEncoder(ignore_format=True), @@ -47,23 +43,17 @@ ] -if sklearn_version < parse_version("1.6"): - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator(estimator) +expected_fails = _return_tags()["_xfail_checks"] +expected_fails.update({"check_estimators_nan_inf": "transformer allows NA"}) -else: - expected_fails = _return_tags()["_xfail_checks"] - expected_fails.update({"check_estimators_nan_inf": "transformer allows NA"}) - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - if estimator.__class__.__name__ != "WoEEncoder": - return check_estimator( - estimator=wrap_for_check_estimator(estimator), - expected_failed_checks=expected_fails, - ) +@pytest.mark.parametrize("estimator", _estimators) +def test_check_estimator_from_sklearn(estimator): + if estimator.__class__.__name__ != "WoEEncoder": + return check_estimator( + estimator=wrap_for_check_estimator(estimator), + expected_failed_checks=expected_fails, + ) _estimators = [ diff --git a/tests/test_encoding/test_count_frequency_encoder.py b/tests/test_encoding/test_count_frequency_encoder.py index 88998755b..0b166f5bd 100644 --- a/tests/test_encoding/test_count_frequency_encoder.py +++ b/tests/test_encoding/test_count_frequency_encoder.py @@ -1,17 +1,33 @@ +import re import warnings import pandas as pd import pytest -from numpy import nan from sklearn.exceptions import NotFittedError from feature_engine.encoding import CountEncoder, CountFrequencyEncoder +from tests.backend_helpers import null_count, frame_to_dict + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "dob": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} # init parameters -@pytest.mark.parametrize("enc_method", ["arbitrary", False, 1]) +@pytest.mark.parametrize( + "enc_method", + ["arbitrary", "Count", "", False, 1, None, ["count"], ("frequency",)], +) def test_error_if_encoding_method_not_permitted_value(enc_method): - with pytest.raises(ValueError): + msg = ( + "encoding_method takes only values 'count' and 'frequency'. " + f"Got {enc_method} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): CountEncoder(encoding_method=enc_method) @@ -19,7 +35,11 @@ def test_error_if_encoding_method_not_permitted_value(enc_method): "errors", ["empanada", False, 1, ("raise", "ignore"), ["ignore"]] ) def test_error_if_unseen_gets_not_permitted_value(errors): - with pytest.raises(ValueError): + msg = ( + "Parameter `unseen` takes only values ignore, raise, encode. " + f"Got {errors} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): CountEncoder(unseen=errors) @@ -40,102 +60,29 @@ def test_init_param_assignment(params): # fit and transform -def test_encode_1_variable_with_counts(df_enc): +def test_encode_1_variable_with_counts(make_df, data_enc): # test case 1: 1 variable, counts encoder = CountEncoder(encoding_method="count", variables=["var_A"]) - X = encoder.fit_transform(df_enc) + X = encoder.fit_transform(make_df(data_enc)) - # expected result - transf_df = df_enc.copy() - transf_df["var_A"] = [ - 6, - 6, - 6, - 6, - 6, - 6, - 10, - 10, - 10, - 10, - 10, - 10, - 10, - 10, - 10, - 10, - 4, - 4, - 4, - 4, - ] - - # init params - assert encoder.encoding_method == "count" - assert encoder.variables == ["var_A"] # fit params assert encoder.variables_ == ["var_A"] assert encoder.encoder_dict_ == {"var_A": {"A": 6, "B": 10, "C": 4}} assert encoder.n_features_in_ == 3 # transform params - pd.testing.assert_frame_equal(X, transf_df) + assert isinstance(X, make_df) + assert frame_to_dict(X) == { + "var_A": [6] * 6 + [10] * 10 + [4] * 4, + "var_B": data_enc["var_B"], + "target": data_enc["target"], + } -def test_automatically_select_variables_encode_with_frequency(df_enc): +def test_automatically_select_variables_encode_with_frequency(make_df, data_enc): # test case 2: automatically select variables, frequency encoder = CountEncoder(encoding_method="frequency", variables=None) - X = encoder.fit_transform(df_enc) + X = encoder.fit_transform(make_df(data_enc)) - # expected output - transf_df = df_enc.copy() - transf_df["var_A"] = [ - 0.3, - 0.3, - 0.3, - 0.3, - 0.3, - 0.3, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.2, - 0.2, - 0.2, - 0.2, - ] - transf_df["var_B"] = [ - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.5, - 0.3, - 0.3, - 0.3, - 0.3, - 0.3, - 0.3, - 0.2, - 0.2, - 0.2, - 0.2, - ] - - # init params - assert encoder.encoding_method == "frequency" - assert encoder.variables is None # fit params assert encoder.variables_ == ["var_A", "var_B"] assert encoder.encoder_dict_ == { @@ -144,64 +91,53 @@ def test_automatically_select_variables_encode_with_frequency(df_enc): } assert encoder.n_features_in_ == 3 # transform params - pd.testing.assert_frame_equal(X, transf_df) - + assert isinstance(X, make_df) + assert frame_to_dict(X) == { + "var_A": [0.3] * 6 + [0.5] * 10 + [0.2] * 4, + "var_B": [0.5] * 10 + [0.3] * 6 + [0.2] * 4, + "target": data_enc["target"], + } -def test_encoding_when_nan_in_fit_df(df_enc): - df = df_enc.copy() - df.loc[len(df)] = [nan, nan, nan] +def test_encoding_when_nan_in_fit_df(make_df, data_enc): encoder = CountEncoder( encoding_method="frequency", missing_values="ignore", ) - encoder.fit(df_enc) + encoder.fit(make_df(data_enc)) X = encoder.transform( - pd.DataFrame({"var_A": ["A", nan], "var_B": ["A", nan], "target": [1, 0]}) + make_df({"var_A": ["A", None], "var_B": ["A", None], "target": [1, 0]}) ) # transform params - pd.testing.assert_frame_equal( - X, - pd.DataFrame({"var_A": [0.3, nan], "var_B": [0.5, nan], "target": [1, 0]}), - ) - - -@pytest.mark.parametrize("enc_method", ["arbitrary", False, 1]) -def test_error_if_encoding_method_not_recognized_in_fit(enc_method, df_enc): - enc = CountEncoder() - enc.encoding_method = enc_method - with pytest.raises(ValueError) as record: - enc.fit(df_enc) - msg = ( - "Unrecognized value for encoding_method. It should be 'count' or " - f"'frequency'. Got {enc_method} instead." - ) - assert str(record.value) == msg + assert isinstance(X, make_df) + assert frame_to_dict(X) == { + "var_A": [0.3, None], + "var_B": [0.5, None], + "target": [1, 0], + } -def test_warning_when_df_contains_unseen_categories(df_enc, df_enc_rare): +def test_warning_when_df_contains_unseen_categories( + make_df, data_enc, data_enc_rare +): # dataset to be transformed contains categories not present in # training dataset (unseen categories), unseen set to ignore. - msg = "During the encoding, NaN values were introduced in the feature(s) var_A." # check for warning when unseen equals 'ignore' encoder = CountEncoder(unseen="ignore") - encoder.fit(df_enc) - with pytest.warns(UserWarning) as record: - encoder.transform(df_enc_rare) - - # check that only one warning was raised - assert len(record) == 1 - # check that the message matches - assert record[0].message.args[0] == msg + encoder.fit(make_df(data_enc)) + with pytest.warns(UserWarning, match=re.escape(msg)): + encoder.transform(make_df(data_enc_rare)) -def test_error_when_df_contains_unseen_categories(df_enc, df_enc_rare): +def test_error_when_df_contains_unseen_categories(make_df, data_enc, data_enc_rare): # dataset to be transformed contains categories not present in # training dataset (unseen categories), unseen set to raise. + df_enc = make_df(data_enc) + df_enc_rare = make_df(data_enc_rare) msg = "During the encoding, NaN values were introduced in the feature(s) var_A." @@ -209,146 +145,115 @@ def test_error_when_df_contains_unseen_categories(df_enc, df_enc_rare): encoder.fit(df_enc) # check for exception when unseen equals 'raise' - with pytest.raises(ValueError) as record: - encoder.transform(df_enc_rare) - - # check that the error message matches - assert str(record.value) == msg - - # check for no error and no warning when unseen equals 'encode' - with warnings.catch_warnings(): - warnings.simplefilter("error") - encoder = CountEncoder(unseen="encode") - encoder.fit(df_enc) - encoder.transform(df_enc_rare) - - -def test_no_error_triggered_when_df_contains_unseen_categories_and_unseen_is_encode( - df_enc, df_enc_rare -): - # dataset to be transformed contains categories not present in - # training dataset (unseen categories). - - # check for no error and no warning when unseen equals 'encode' - warnings.simplefilter("error") - encoder = CountEncoder(unseen="encode") - encoder.fit(df_enc) - with warnings.catch_warnings(): + with pytest.raises(ValueError, match=re.escape(msg)): encoder.transform(df_enc_rare) @pytest.mark.parametrize("errors", ["raise", "ignore", "encode"]) -def test_fit_raises_error_if_df_contains_na(errors, df_enc_na): +def test_fit_raises_error_if_df_contains_na(errors, make_df, data_enc_na): # test case 4: when dataset contains na, fit method encoder = CountEncoder(unseen=errors) - with pytest.raises(ValueError) as record: - encoder.fit(df_enc_na) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer or set the parameter " "`missing_values='ignore'` when initialising this transformer." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(make_df(data_enc_na)) @pytest.mark.parametrize("errors", ["raise", "ignore", "encode"]) -def test_transform_raises_error_if_df_contains_na(errors, df_enc, df_enc_na): +def test_transform_raises_error_if_df_contains_na( + errors, make_df, data_enc, data_enc_na +): # test case 4: when dataset contains na, transform method encoder = CountEncoder(unseen=errors) - encoder.fit(df_enc) - with pytest.raises(ValueError) as record: - encoder.transform(df_enc_na) + encoder.fit(make_df(data_enc)) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer or set the parameter " "`missing_values='ignore'` when initialising this transformer." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.transform(make_df(data_enc_na)) -def test_zero_encoding_for_new_categories(): - - df_fit = pd.DataFrame( +def test_zero_encoding_for_unseen_categories_if_unseen_is_encode(make_df): + df_fit = make_df( {"col1": ["a", "a", "b", "a", "c"], "col2": ["1", "2", "3", "1", "2"]} ) - df_transf = pd.DataFrame( - {"col1": ["a", "d", "b", "a", "c"], "col2": ["1", "2", "3", "1", "4"]} - ) - encoder = CountEncoder(unseen="encode").fit(df_fit) - - result = encoder.transform(df_transf) - - # check that no NaNs are added - assert pd.isnull(result).sum().sum() == 0 - - # check that the counts are correct for both new and old - expected_result = pd.DataFrame({"col1": [3, 0, 1, 3, 1], "col2": [2, 2, 1, 2, 0]}) - pd.testing.assert_frame_equal(result, expected_result, check_dtype=False) - - -def test_zero_encoding_for_unseen_categories_if_unseen_is_encode(): - df_fit = pd.DataFrame( - {"col1": ["a", "a", "b", "a", "c"], "col2": ["1", "2", "3", "1", "2"]} - ) - df_transform = pd.DataFrame( + df_transform = make_df( {"col1": ["a", "d", "b", "a", "c"], "col2": ["1", "2", "3", "1", "4"]} ) # count encoding encoder = CountEncoder(unseen="encode").fit(df_fit) - result = encoder.transform(df_transform) + # unseen categories are encoded without raising or warning + with warnings.catch_warnings(): + warnings.simplefilter("error") + result = encoder.transform(df_transform) + assert isinstance(result, make_df) # check that no NaNs are added - assert pd.isnull(result).sum().sum() == 0 + assert null_count(result, "col1") == 0 + assert null_count(result, "col2") == 0 # check that the counts are correct - expected_result = pd.DataFrame({"col1": [3, 0, 1, 3, 1], "col2": [2, 2, 1, 2, 0]}) - pd.testing.assert_frame_equal(result, expected_result, check_dtype=False) + assert frame_to_dict(result) == {"col1": [3, 0, 1, 3, 1], "col2": [2, 2, 1, 2, 0]} # with frequency - encoder = CountEncoder(encoding_method="frequency", unseen="encode").fit( - df_fit - ) - result = encoder.transform(df_transform) + encoder = CountEncoder(encoding_method="frequency", unseen="encode").fit(df_fit) + # unseen categories are encoded without raising or warning + with warnings.catch_warnings(): + warnings.simplefilter("error") + result = encoder.transform(df_transform) + assert isinstance(result, make_df) # check that no NaNs are added - assert pd.isnull(result).sum().sum() == 0 + assert null_count(result, "col1") == 0 + assert null_count(result, "col2") == 0 # check that the frequencies are correct - expected_result = pd.DataFrame( - {"col1": [0.6, 0, 0.2, 0.6, 0.2], "col2": [0.4, 0.4, 0.2, 0.4, 0]} - ) - pd.testing.assert_frame_equal(result, expected_result) + assert frame_to_dict(result) == { + "col1": [0.6, 0, 0.2, 0.6, 0.2], + "col2": [0.4, 0.4, 0.2, 0.4, 0], + } -def test_nan_encoding_for_new_categories_if_unseen_is_ignore(): - df_fit = pd.DataFrame( +def test_nan_encoding_for_new_categories_if_unseen_is_ignore(make_df): + df_fit = make_df( {"col1": ["a", "a", "b", "a", "c"], "col2": ["1", "2", "3", "1", "2"]} ) - df_transf = pd.DataFrame( + df_transf = make_df( {"col1": ["a", "d", "b", "a", "c"], "col2": ["1", "2", "3", "1", "4"]} ) encoder = CountEncoder(unseen="ignore").fit(df_fit) result = encoder.transform(df_transf) + assert isinstance(result, make_df) - # check that no NaNs are added - assert pd.isnull(result).sum().sum() == 2 + # check that 1 NaN is added per variable + assert null_count(result, "col1") == 1 + assert null_count(result, "col2") == 1 # check that the counts are correct for both new and old - expected_result = pd.DataFrame( - {"col1": [3, nan, 1, 3, 1], "col2": [2, 2, 1, 2, nan]} - ) - pd.testing.assert_frame_equal(result, expected_result) + assert frame_to_dict(result) == { + "col1": [3, None, 1, 3, 1], + "col2": [2, 2, 1, 2, None], + } -def test_ignore_variable_format_with_frequency(df_vartypes): +def test_ignore_variable_format_with_frequency(make_df): encoder = CountEncoder( encoding_method="frequency", variables=None, ignore_format=True ) - X = encoder.fit_transform(df_vartypes) + X = encoder.fit_transform(make_df(DATA_VARTYPES)) - # expected output - transf_df = { + # fit params + assert encoder.variables_ == ["Name", "City", "Age", "Marks", "dob"] + assert encoder.n_features_in_ == 5 + # transform params + assert isinstance(X, make_df) + assert frame_to_dict(X) == { "Name": [0.25, 0.25, 0.25, 0.25], "City": [0.25, 0.25, 0.25, 0.25], "Age": [0.25, 0.25, 0.25, 0.25], @@ -356,19 +261,9 @@ def test_ignore_variable_format_with_frequency(df_vartypes): "dob": [0.25, 0.25, 0.25, 0.25], } - transf_df = pd.DataFrame(transf_df) - - # init params - assert encoder.encoding_method == "frequency" - assert encoder.variables is None - # fit params - assert encoder.variables_ == ["Name", "City", "Age", "Marks", "dob"] - assert encoder.n_features_in_ == 5 - # transform params - pd.testing.assert_frame_equal(X, transf_df) - def test_column_names_are_numbers(df_numeric_columns): + # integer column names are not supported by polars - pandas only. encoder = CountEncoder( encoding_method="frequency", variables=[0, 1, 2, 3], ignore_format=True ) @@ -385,9 +280,6 @@ def test_column_names_are_numbers(df_numeric_columns): transf_df = pd.DataFrame(transf_df) - # init params - assert encoder.encoding_method == "frequency" - assert encoder.variables == [0, 1, 2, 3] # fit params assert encoder.variables_ == [0, 1, 2, 3] assert encoder.n_features_in_ == 5 @@ -396,33 +288,13 @@ def test_column_names_are_numbers(df_numeric_columns): def test_variables_cast_as_category(df_enc_category_dtypes): + # pandas category dtype is not a polars concept - pandas only. encoder = CountEncoder(encoding_method="count", variables=["var_A"]) X = encoder.fit_transform(df_enc_category_dtypes) # expected result transf_df = df_enc_category_dtypes.copy() - transf_df["var_A"] = [ - 6, - 6, - 6, - 6, - 6, - 6, - 10, - 10, - 10, - 10, - 10, - 10, - 10, - 10, - 10, - 10, - 4, - 4, - 4, - 4, - ] + transf_df["var_A"] = [6] * 6 + [10] * 10 + [4] * 4 # transform params pd.testing.assert_frame_equal(X, transf_df, check_dtype=False) assert X["var_A"].dtypes == int @@ -432,55 +304,69 @@ def test_variables_cast_as_category(df_enc_category_dtypes): assert X["var_A"].dtypes == float -def test_inverse_transform_when_no_unseen(): - df = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) +def test_inverse_transform_when_no_unseen(make_df): + words = ["dog", "dog", "cat", "cat", "cat", "bird"] + df = make_df({"words": words}) enc = CountEncoder() enc.fit(df) dft = enc.transform(df) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df) + X = enc.inverse_transform(dft) + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"words": words} -def test_inverse_transform_when_ignore_unseen(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) - df2 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) - df3 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", nan]}) +def test_inverse_transform_when_ignore_unseen(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + df2 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) enc = CountEncoder(unseen="ignore") enc.fit(df1) dft = enc.transform(df2) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df3) + X = enc.inverse_transform(dft) + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"words": ["dog", "dog", "cat", "cat", "cat", None]} -def test_inverse_transform_when_encode_unseen(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) - df2 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) - df3 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", nan]}) +def test_inverse_transform_when_encode_unseen(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + df2 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) enc = CountEncoder(unseen="encode") enc.fit(df1) dft = enc.transform(df2) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df3) + X = enc.inverse_transform(dft) + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"words": ["dog", "dog", "cat", "cat", "cat", None]} -def test_inverse_transform_raises_non_fitted_error(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) +def test_inverse_transform_raises_non_fitted_error(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) enc = CountEncoder() + msg = ( + "This CountEncoder instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + msg_na = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." + ) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): + with pytest.raises(NotFittedError, match=re.escape(msg)): enc.inverse_transform(df1) - df1.loc[len(df1) - 1] = nan + df1_na = make_df({"words": ["dog", "dog", "cat", "cat", "cat", None]}) - with pytest.raises(ValueError): - enc.fit(df1) + with pytest.raises(ValueError, match=re.escape(msg_na)): + enc.fit(df1_na) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): - enc.inverse_transform(df1) + with pytest.raises(NotFittedError, match=re.escape(msg)): + enc.inverse_transform(df1_na) -def test_count_frequency_encoder_is_deprecated(): +def test_count_frequency_encoder_is_deprecated(make_df): """CountFrequencyEncoder should emit a FutureWarning and still work.""" - X = pd.DataFrame({"var_A": ["A"] * 6 + ["B"] * 2 + ["C"] * 2}) + X = make_df({"var_A": ["A"] * 6 + ["B"] * 2 + ["C"] * 2}) with pytest.warns(FutureWarning, match="CountFrequencyEncoder was deprecated"): enc = CountFrequencyEncoder(encoding_method="count") @@ -488,6 +374,12 @@ def test_count_frequency_encoder_is_deprecated(): enc_new = CountEncoder(encoding_method="count") - pd.testing.assert_frame_equal( - enc.fit_transform(X), enc_new.fit_transform(X) + X_old = enc.fit_transform(X) + X_new = enc_new.fit_transform(X) + assert isinstance(X_old, make_df) + assert isinstance(X_new, make_df) + assert ( + frame_to_dict(X_old) + == frame_to_dict(X_new) + == {"var_A": [6] * 6 + [2] * 2 + [2] * 2} ) diff --git a/tests/test_encoding/test_decision_tree_encoder.py b/tests/test_encoding/test_decision_tree_encoder.py index fd4cef789..0cd7d48cc 100644 --- a/tests/test_encoding/test_decision_tree_encoder.py +++ b/tests/test_encoding/test_decision_tree_encoder.py @@ -3,20 +3,41 @@ import numpy as np import pandas as pd import pytest - from sklearn.exceptions import NotFittedError from feature_engine.encoding import DecisionTreeEncoder +from tests.backend_helpers import make_series, frame_to_dict + +# Tree: var_A <= 1.5 -> 0.25 else 0.5 +# Tree: var_B <= 0.5 -> 0.2 else 0.4 +ENCODED = { + "var_A": [0.25] * 16 + [0.5] * 4, + "var_B": [0.2] * 10 + [0.4] * 10, +} +ENCODED_REGRESSION = { + "var_A": [0.034348] * 6 + [-0.024679] * 10 + [-0.075473] * 4, + "var_B": [0.044806] * 10 + [-0.079066] * 10, +} + + +def _rounded(X, decimals=6): + return { + col: [round(v, decimals) for v in values] + for col, values in frame_to_dict(X).items() + } # init parameters -@pytest.mark.parametrize("enc_method", ["count", False, 1]) +@pytest.mark.parametrize( + "enc_method", + ["count", "Ordered", "", False, 1, None, ["ordered"], ("arbitrary",)], +) def test_error_if_encoding_method_not_permitted_value(enc_method): msg = ( "`encoding_method` takes only values 'ordered' and 'arbitrary'." f" Got {enc_method} instead." ) - with pytest.raises(ValueError, match=msg): + with pytest.raises(ValueError, match=re.escape(msg)): DecisionTreeEncoder(encoding_method=enc_method) @@ -24,11 +45,11 @@ def test_error_if_encoding_method_not_permitted_value(enc_method): "unseen", ["string", False, ("raise", "ignore"), ["ignore"], np.nan] ) def test_error_if_unseen_gets_not_permitted_value(unseen): - msg = re.escape( + msg = ( "Parameter `unseen` takes only values ignore, raise, encode. " - rf"Got {unseen} instead." + f"Got {unseen} instead." ) - with pytest.raises(ValueError, match=msg): + with pytest.raises(ValueError, match=re.escape(msg)): DecisionTreeEncoder(unseen=unseen) @@ -37,41 +58,74 @@ def test_error_if_unseen_is_encode_and_fill_value_is_none(): "When `unseen='encode'` you need to pass a number to `fill_value`. " f"Got {None} instead." ) - with pytest.raises(ValueError, match=msg): + with pytest.raises(ValueError, match=re.escape(msg)): DecisionTreeEncoder(unseen="encode", fill_value=None) @pytest.mark.parametrize("precision", ["string", 0.1, -1, np.nan]) def test_error_if_precision_gets_not_permitted_value(precision): msg = "Parameter `precision` takes integers or None. " f"Got {precision} instead." - with pytest.raises(ValueError, match=msg): + with pytest.raises(ValueError, match=re.escape(msg)): DecisionTreeEncoder(precision=precision) @pytest.mark.parametrize( - "encoding_method,ignore_format,precision,unseen,fill_value", + "params", [ - ("arbitrary", True, 1, "raise", None), - ("ordered", False, 2, "ignore", 1), - ("ordered", False, None, "encode", 0.1), + { + "encoding_method": "arbitrary", + "cv": 3, + "scoring": "neg_mean_squared_error", + "regression": True, + "param_grid": None, + "random_state": None, + "ignore_format": True, + "precision": 1, + "unseen": "raise", + "fill_value": None, + "n_jobs": None, + }, + { + "encoding_method": "ordered", + "cv": 5, + "scoring": "roc_auc", + "regression": False, + "param_grid": {"max_depth": [1, 2]}, + "random_state": 0, + "ignore_format": False, + "precision": None, + "unseen": "encode", + "fill_value": 0.1, + "n_jobs": -1, + }, + { + "encoding_method": "ordered", + "cv": 2, + "scoring": "accuracy", + "regression": False, + "param_grid": {"max_depth": [3]}, + "random_state": 42, + "ignore_format": False, + "precision": 2, + "unseen": "ignore", + "fill_value": 1, + "n_jobs": 2, + }, ], ) -def test_init_param_assignment( - encoding_method, ignore_format, precision, unseen, fill_value -): - DecisionTreeEncoder( - encoding_method=encoding_method, - ignore_format=ignore_format, - precision=precision, - unseen=unseen, - fill_value=fill_value, - ) +def test_init_param_assignment(params): + encoder = DecisionTreeEncoder(**params) + for param, value in params.items(): + assert getattr(encoder, param) == value # fit attributes -def test_encoding_dictionary(df_enc): +def test_encoding_dictionary(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = DecisionTreeEncoder(regression=False) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) + encoder.fit(X, y) # Tree: var_A <= 1.5 -> 0.25 else 0.5 # Tree: var_B <= 0.5 -> 0.2 else 0.4 @@ -82,9 +136,28 @@ def test_encoding_dictionary(df_enc): assert encoder.encoder_dict_ == expected_encodings -def test_precision(df_enc): +def test_ordered_encoding_dictionary(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + + encoder = DecisionTreeEncoder(regression=False, encoding_method="ordered") + encoder.fit(X, y) + + # codes by target mean: var_A B=0, A=1, C=2 and var_B A=0, B=1, C=2 + # both trees split code 0 from the rest + expected_encodings = { + "var_A": {"B": 0.2, "A": 0.4, "C": 0.4}, + "var_B": {"A": 0.2, "B": 0.4, "C": 0.4}, + } + assert encoder.encoder_dict_ == expected_encodings + + +def test_precision(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = DecisionTreeEncoder(regression=False, precision=1) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) + encoder.fit(X, y) # Tree: var_A <= 1.5 -> 0.25 else 0.5 # Tree: var_B <= 0.5 -> 0.2 else 0.4 @@ -95,92 +168,108 @@ def test_precision(df_enc): assert encoder.encoder_dict_ == expected_encodings -def test_classification(df_enc): +def test_classification(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = DecisionTreeEncoder(regression=False) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = encoder.transform(df_enc[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) - transf_df = df_enc.copy() - transf_df["var_A"] = [0.25] * 16 + [0.5] * 4 # Tree: var_A <= 1.5 -> 0.25 else 0.5 - transf_df["var_B"] = [0.2] * 10 + [0.4] * 10 # Tree: var_B <= 0.5 -> 0.2 else 0.4 - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == ENCODED -def test_regression(df_enc): +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_or_array(make_df, data_enc, to_target): + # a list or numpy array target takes a different code path than a Series + X = make_df(data_enc)[["var_A", "var_B"]] + y = to_target(data_enc["target"]) + + encoder = DecisionTreeEncoder(regression=False) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == ENCODED + + +def test_regression(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] random = np.random.RandomState(42) - y = random.normal(0, 0.1, len(df_enc)) + y = make_series(make_df, random.normal(0, 0.1, len(data_enc["target"]))) encoder = DecisionTreeEncoder( regression=True, random_state=random, ) - encoder.fit(df_enc[["var_A", "var_B"]], y) - X = encoder.transform(df_enc[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert isinstance(Xt, make_df) + assert _rounded(Xt) == ENCODED_REGRESSION - transf_df = df_enc.copy() - transf_df["var_A"] = ( - [0.034348] * 6 + [-0.024679] * 10 + [-0.075473] * 4 - ) # Tree: var_A <= 1.5 -> 0.25 else 0.5 - transf_df["var_B"] = [0.044806] * 10 + [-0.079066] * 10 - pd.testing.assert_frame_equal(X.round(6), transf_df[["var_A", "var_B"]]) +def test_fit_raises_error_if_df_contains_na(make_df, data_enc_na): + X = make_df(data_enc_na)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_na["target"]) -def test_fit_raises_error_if_df_contains_na(df_enc_na): - # test case 4: when dataset contains na, fit method encoder = DecisionTreeEncoder(regression=False) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer." ) - with pytest.raises(ValueError, match=msg): - encoder.fit(df_enc_na[["var_A", "var_B"]], df_enc_na["target"]) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(X, y) -def test_transform_raises_error_if_df_contains_na(df_enc, df_enc_na): - # test case 4: when dataset contains na, transform method +def test_transform_raises_error_if_df_contains_na(make_df, data_enc, data_enc_na): + X = make_df(data_enc)[["var_A", "var_B"]] + X_na = make_df(data_enc_na)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = DecisionTreeEncoder(regression=False) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) + encoder.fit(X, y) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer." ) - with pytest.raises(ValueError, match=msg): - encoder.transform(df_enc_na[["var_A", "var_B"]]) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.transform(X_na) + +def test_classification_ignore_format(make_df, data_enc_numeric): + X = make_df(data_enc_numeric)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_numeric["target"]) -def test_classification_ignore_format(df_enc_numeric): encoder = DecisionTreeEncoder( regression=False, ignore_format=True, ) - encoder.fit(df_enc_numeric[["var_A", "var_B"]], df_enc_numeric["target"]) - X = encoder.transform(df_enc_numeric[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) - transf_df = df_enc_numeric.copy() - transf_df["var_A"] = [0.25] * 16 + [0.5] * 4 # Tree: var_A <= 1.5 -> 0.25 else 0.5 - transf_df["var_B"] = [0.2] * 10 + [0.4] * 10 # Tree: var_B <= 0.5 -> 0.2 else 0.4 - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == ENCODED -def test_regression_ignore_format(df_enc_numeric): +def test_regression_ignore_format(make_df, data_enc_numeric): + X = make_df(data_enc_numeric)[["var_A", "var_B"]] random = np.random.RandomState(42) - y = random.normal(0, 0.1, len(df_enc_numeric)) + y = make_series(make_df, random.normal(0, 0.1, len(data_enc_numeric["target"]))) encoder = DecisionTreeEncoder( regression=True, random_state=random, ignore_format=True, ) - encoder.fit(df_enc_numeric[["var_A", "var_B"]], y) - X = encoder.transform(df_enc_numeric[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) - transf_df = df_enc_numeric.copy() - transf_df["var_A"] = ( - [0.034348] * 6 + [-0.024679] * 10 + [-0.075473] * 4 - ) # Tree: var_A <= 1.5 -> 0.25 else 0.5 - transf_df["var_B"] = [0.044806] * 10 + [-0.079066] * 10 - pd.testing.assert_frame_equal(X.round(6), transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert _rounded(Xt) == ENCODED_REGRESSION def test_variables_cast_as_category(df_enc_category_dtypes): + # pandas Categorical dtype has no direct polars equivalent - pandas-only. df = df_enc_category_dtypes.copy() encoder = DecisionTreeEncoder(regression=False) encoder.fit(df[["var_A", "var_B"]], df["target"]) @@ -193,24 +282,45 @@ def test_variables_cast_as_category(df_enc_category_dtypes): assert X["var_A"].dtypes == float -def test_error_when_regression_is_true_and_target_is_binary(df_enc): +def test_integer_column_names(data_enc): + # integer column names are pandas-only + X = pd.DataFrame({0: data_enc["var_A"], 1: data_enc["var_B"]}) + y = pd.Series(data_enc["target"]) + + encoder = DecisionTreeEncoder(regression=False).fit(X, y) + expected = DecisionTreeEncoder(regression=False).fit(X.rename(columns=str), y) + + assert encoder.encoder_dict_ == { + 0: expected.encoder_dict_["0"], + 1: expected.encoder_dict_["1"], + } + + +def test_error_when_regression_is_true_and_target_is_binary(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = DecisionTreeEncoder(regression=True) msg = ( "Trying to fit a regression to a binary target is not " "allowed by this transformer. Check the target values " "or set regression to False." ) - with pytest.raises(ValueError, match=msg): - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(X, y) -def test_error_when_regression_is_false_and_target_is_continuous(df_enc): +def test_error_when_regression_is_false_and_target_is_continuous(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] random = np.random.RandomState(42) - y = random.normal(0, 10, len(df_enc)) + y = make_series(make_df, random.normal(0, 10, len(data_enc["target"]))) encoder = DecisionTreeEncoder(regression=False) - # the error message comes from sklearn api - won't test - with pytest.raises(ValueError): - encoder.fit(df_enc[["var_A", "var_B"]], y) + msg = ( + "Unknown label type: continuous. Maybe you are trying to fit a classifier, " + "which expects discrete classes on a regression target with continuous values." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(X, y) @pytest.mark.parametrize( @@ -225,116 +335,132 @@ def test_assigns_param_grid(grid): assert encoder._assign_param_grid() == grid -def test_unseen_is_encode(df_enc): +def test_unseen_is_encode(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = DecisionTreeEncoder(unseen="encode", regression=False, fill_value=-1) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) + encoder.fit(X, y) - X_unseen_input = pd.DataFrame( + X_unseen_input = make_df( { "var_A": ["A", "ZZZ", "YYY"], "var_B": ["C", "YYY", "ZZZ"], } ) + Xt = encoder.transform(X_unseen_input) - X_unseen_output = pd.DataFrame( - { - "var_A": [0.25, -1, -1], - "var_B": [0.4, -1, -1], - } - ) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"var_A": [0.25, -1, -1], "var_B": [0.4, -1, -1]} - Xt = encoder.transform(X_unseen_input) - pd.testing.assert_frame_equal(Xt, X_unseen_output) +def test_unseen_is_ignore(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) -def test_unseen_is_ignore(df_enc): encoder = DecisionTreeEncoder(unseen="ignore", regression=False) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) + encoder.fit(X, y) - X_unseen_input = pd.DataFrame( + X_unseen_input = make_df( { "var_A": ["A", "ZZZ", "YYY"], "var_B": ["C", "YYY", "ZZZ"], } ) + Xt = encoder.transform(X_unseen_input) - X_unseen_output = pd.DataFrame( - { - "var_A": [0.25, np.nan, np.nan], - "var_B": [0.4, np.nan, np.nan], - } - ) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [0.25, None, None], + "var_B": [0.4, None, None], + } - Xt = encoder.transform(X_unseen_input) - pd.testing.assert_frame_equal(Xt, X_unseen_output) +def test_fit_errors_if_new_cat_values_and_unseen_is_raise_param(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) -def test_fit_errors_if_new_cat_values_and_unseen_is_raise_param(df_enc): encoder = DecisionTreeEncoder(unseen="raise", regression=False) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = pd.DataFrame( + encoder.fit(X, y) + X_unseen = make_df( { "var_A": ["A", "ZZZ", "YYY"], "var_B": ["C", "YYY", "ZZZ"], } ) - var_ls = "var_A, var_B" msg = ( "During the encoding, NaN values were introduced in the " - rf"feature\(s\) {var_ls}." + "feature(s) var_A, var_B." ) # new categories will raise an error - with pytest.raises(ValueError, match=msg): - encoder.transform(X) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.transform(X_unseen) -def test_inverse_transform_when_no_unseen(): - X = pd.DataFrame({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "bird"]}) - y = pd.Series([0, 0, 1, 1, 1, 1, 0]) +def test_inverse_transform_when_no_unseen(make_df): + words = ["dog", "dog", "dog", "cat", "cat", "cat", "bird"] + X = make_df({"words": words}) + y = make_series(make_df, [0, 0, 1, 1, 1, 1, 0]) enc = DecisionTreeEncoder(regression=False) enc.fit(X, y) dft = enc.transform(X) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), X) + Xi = enc.inverse_transform(dft) + assert isinstance(Xi, make_df) + assert frame_to_dict(Xi) == {"words": words} -def test_inverse_transform_when_ignore_unseen(): - X = pd.DataFrame({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "bird"]}) - y = pd.Series([0, 0, 1, 1, 1, 1, 0]) +def test_inverse_transform_when_ignore_unseen(make_df): + X = make_df({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "bird"]}) + y = make_series(make_df, [0, 0, 1, 1, 1, 1, 0]) enc = DecisionTreeEncoder(regression=False, unseen="ignore") enc.fit(X, y) - df1 = pd.DataFrame({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "frog"]}) - df2 = pd.DataFrame({"words": ["dog", "dog", "dog", "cat", "cat", "cat", np.nan]}) + df1 = make_df({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "frog"]}) dft = enc.transform(df1) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df2) + Xi = enc.inverse_transform(dft) + assert isinstance(Xi, make_df) + assert frame_to_dict(Xi) == { + "words": ["dog", "dog", "dog", "cat", "cat", "cat", None] + } -def test_inverse_transform_when_encode_unseen(): - X = pd.DataFrame({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "bird"]}) - y = pd.Series([0, 0, 1, 1, 1, 1, 0]) +def test_inverse_transform_when_encode_unseen(make_df): + X = make_df({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "bird"]}) + y = make_series(make_df, [0, 0, 1, 1, 1, 1, 0]) enc = DecisionTreeEncoder(regression=False, unseen="encode", fill_value=1000) enc.fit(X, y) - df1 = pd.DataFrame({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "frog"]}) - df2 = pd.DataFrame({"words": ["dog", "dog", "dog", "cat", "cat", "cat", np.nan]}) + df1 = make_df({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "frog"]}) dft = enc.transform(df1) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df2) + Xi = enc.inverse_transform(dft) + assert isinstance(Xi, make_df) + assert frame_to_dict(Xi) == { + "words": ["dog", "dog", "dog", "cat", "cat", "cat", None] + } -def test_inverse_transform_raises_non_fitted_error(): - X = pd.DataFrame({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "bird"]}) - y = pd.Series([0, 0, 1, 1, 1, 1, 0]) - enc = DecisionTreeEncoder() +def test_inverse_transform_raises_non_fitted_error(make_df): + X = make_df({"words": ["dog", "dog", "dog", "cat", "cat", "cat", "bird"]}) + y = make_series(make_df, [0, 0, 1, 1, 1, 1, 0]) + enc = DecisionTreeEncoder(regression=False) + msg = ( + "This DecisionTreeEncoder instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + msg_na = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." + ) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): + with pytest.raises(NotFittedError, match=re.escape(msg)): enc.inverse_transform(X) - X.loc[len(X) - 1] = np.nan + X_na = make_df({"words": ["dog", "dog", "dog", "cat", "cat", "cat", None]}) - with pytest.raises(ValueError): - enc.fit(X, y) + with pytest.raises(ValueError, match=re.escape(msg_na)): + enc.fit(X_na, y) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): - enc.inverse_transform(X) + with pytest.raises(NotFittedError, match=re.escape(msg)): + enc.inverse_transform(X_na) diff --git a/tests/test_encoding/test_helper_functions.py b/tests/test_encoding/test_helper_functions.py index 022c051c3..6616cf2ef 100644 --- a/tests/test_encoding/test_helper_functions.py +++ b/tests/test_encoding/test_helper_functions.py @@ -1,22 +1,49 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd import pytest -from feature_engine.encoding._helper_functions import check_parameter_unseen +from feature_engine.encoding._helper_functions import ( + TARGET_NAME, + add_target_to_X, + check_parameter_unseen, +) +from tests.backend_helpers import frame_to_dict, make_series @pytest.mark.parametrize("accepted", ["one", False, [1, 2], ("one", "two"), 1]) def test_raises_error_when_accepted_values_not_permitted(accepted): - with pytest.raises(ValueError) as record: - check_parameter_unseen("zero", accepted) msg = "accepted_values should be a list of strings. " f" Got {accepted} instead." - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + check_parameter_unseen("zero", accepted) -@pytest.mark.parametrize("accepted", [["one", "two"], ["three", "four"]]) -def test_raises_error_when_error_not_in_accepted_values(accepted): - with pytest.raises(ValueError) as record: - check_parameter_unseen("zero", accepted) - msg = ( - f"Parameter `unseen` takes only values {', '.join(accepted)}." - " Got zero instead." - ) - assert str(record.value) == msg +@pytest.mark.parametrize("unseen", ["zero", "One", "", 1, None, ["one"], ("one",)]) +def test_raises_error_when_unseen_not_in_accepted_values(unseen): + msg = f"Parameter `unseen` takes only values one, two. Got {unseen} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + check_parameter_unseen(unseen, ["one", "two"]) + + +@pytest.mark.parametrize("to_target", [list, np.array, "series"]) +def test_add_target_to_X_pairs_rows_by_position(make_df, to_target): + X = make_df({"var_A": ["a", "b", "c"]}) + values = [1, 0, 1] + if to_target == "series": + y = make_series(make_df, values) + else: + y = to_target(values) + + Xy = add_target_to_X(nw.from_native(X), y).to_native() + + assert isinstance(Xy, make_df) + assert frame_to_dict(Xy) == {"var_A": ["a", "b", "c"], TARGET_NAME: values} + + +def test_add_target_to_X_keeps_the_pandas_index(): + X = pd.DataFrame({"var_A": ["a", "b", "c"]}, index=[12, 10, 11]) + Xy = add_target_to_X(nw.from_native(X), np.array([1, 0, 1])).to_native() + assert Xy.index.tolist() == [12, 10, 11] + assert Xy[TARGET_NAME].tolist() == [1, 0, 1] diff --git a/tests/test_encoding/test_mean_encoder.py b/tests/test_encoding/test_mean_encoder.py index a13d0e5bf..164e27e23 100644 --- a/tests/test_encoding/test_mean_encoder.py +++ b/tests/test_encoding/test_mean_encoder.py @@ -1,339 +1,245 @@ +import re + +import numpy as np import pandas as pd import pytest -from numpy import nan from sklearn.exceptions import NotFittedError from feature_engine.encoding import MeanEncoder +from tests.backend_helpers import make_series, frame_to_dict - -# test init params -@pytest.mark.parametrize("params", [("raise", True, "auto"), ("ignore", False, 1)]) -def test_init_param_assignment(params): - MeanEncoder( - missing_values=params[0], - ignore_format=params[1], - unseen=params[0], - smoothing=params[2], - ) +ENC_DICT_VAR_A = {"A": 0.3333333333333333, "B": 0.2, "C": 0.5} +ENC_DICT_VAR_B = {"A": 0.2, "B": 0.3333333333333333, "C": 0.5} +# init parameters @pytest.mark.parametrize( - "errors", ["empanada", False, 1, ("raise", "ignore"), ["ignore"]] + "unseen", ["empanada", False, 1, None, ("raise", "ignore"), ["ignore"]] ) -def test_error_if_unseen_gets_not_permitted_value(errors): - with pytest.raises(ValueError): - MeanEncoder(unseen=errors) +def test_error_if_unseen_gets_not_permitted_value(unseen): + msg = ( + "Parameter `unseen` takes only values ignore, raise, encode. " + f"Got {unseen} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + MeanEncoder(unseen=unseen) @pytest.mark.parametrize("smoothing", ["hello", ["auto"], -1]) def test_raises_error_when_not_allowed_smoothing_param_in_init(smoothing): - with pytest.raises(ValueError): + msg = f"smoothing must be greater than 0 or 'auto'. Got {smoothing} instead." + with pytest.raises(ValueError, match=re.escape(msg)): MeanEncoder(smoothing=smoothing) +@pytest.mark.parametrize( + "missing_values, ignore_format, unseen, smoothing", + [ + ("raise", True, "ignore", "auto"), + ("ignore", False, "encode", 1), + ("raise", False, "raise", 0.5), + ], +) +def test_init_param_assignment(missing_values, ignore_format, unseen, smoothing): + encoder = MeanEncoder( + missing_values=missing_values, + ignore_format=ignore_format, + unseen=unseen, + smoothing=smoothing, + ) + assert encoder.missing_values == missing_values + assert encoder.ignore_format is ignore_format + assert encoder.unseen == unseen + assert encoder.smoothing == smoothing + + # fit and transform -def test_user_enters_1_variable(df_enc): +def test_user_enters_1_variable(make_df, data_enc): # test case 1: 1 variable + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = MeanEncoder(variables=["var_A"]) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = encoder.transform(df_enc[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) - # expected output - transf_df = df_enc.copy() - transf_df["var_A"] = [ - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.5, - 0.5, - 0.5, - 0.5, - ] - - # test init params - assert encoder.variables == ["var_A"] # test fit attr assert encoder.variables_ == ["var_A"] - assert encoder.encoder_dict_ == { - "var_A": {"A": 0.3333333333333333, "B": 0.2, "C": 0.5} - } + assert encoder.encoder_dict_ == {"var_A": ENC_DICT_VAR_A} assert encoder.n_features_in_ == 2 # test transform output - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [ENC_DICT_VAR_A[v] for v in data_enc["var_A"]], + "var_B": data_enc["var_B"], + } -def test_automatically_find_variables(df_enc): +def test_automatically_find_variables(make_df, data_enc): # test case 2: automatically select variables + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = MeanEncoder(variables=None) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = encoder.transform(df_enc[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) - # expected output - transf_df = df_enc.copy() - transf_df["var_A"] = [ - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.5, - 0.5, - 0.5, - 0.5, - ] - transf_df["var_B"] = [ - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.5, - 0.5, - 0.5, - 0.5, - ] - - # test init params - assert encoder.variables is None # test fit attr assert encoder.variables_ == ["var_A", "var_B"] - assert encoder.encoder_dict_ == { - "var_A": {"A": 0.3333333333333333, "B": 0.2, "C": 0.5}, - "var_B": {"A": 0.2, "B": 0.3333333333333333, "C": 0.5}, - } + assert encoder.encoder_dict_ == {"var_A": ENC_DICT_VAR_A, "var_B": ENC_DICT_VAR_B} assert encoder.n_features_in_ == 2 # test transform output - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [ENC_DICT_VAR_A[v] for v in data_enc["var_A"]], + "var_B": [ENC_DICT_VAR_B[v] for v in data_enc["var_B"]], + } + + +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_or_array(make_df, data_enc, to_target): + # a list or numpy array target takes a different code path than a Series + X = make_df(data_enc)[["var_A", "var_B"]] + y = to_target(data_enc["target"]) + + encoder = MeanEncoder() + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": ENC_DICT_VAR_A, "var_B": ENC_DICT_VAR_B} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [ENC_DICT_VAR_A[v] for v in data_enc["var_A"]], + "var_B": [ENC_DICT_VAR_B[v] for v in data_enc["var_B"]], + } -def test_encoding_when_nan_in_fit_df(df_enc): - df = df_enc.copy() - df.loc[len(df)] = [nan, nan, 0] +def test_encoding_when_nan_in_fit_df(make_df, data_enc): + data = { + "var_A": data_enc["var_A"] + [None], + "var_B": data_enc["var_B"] + [None], + "target": data_enc["target"] + [0], + } + X = make_df(data)[["var_A", "var_B"]] + y = make_series(make_df, data["target"]) encoder = MeanEncoder(missing_values="ignore") - encoder.fit(df[["var_A", "var_B"]], df["target"]) + encoder.fit(X, y) - X = encoder.transform( - pd.DataFrame( - { - "var_A": ["A", nan], - "var_B": ["A", nan], - } - ) - ) + Xt = encoder.transform(make_df({"var_A": ["A", None], "var_B": ["A", None]})) - # transform params - pd.testing.assert_frame_equal( - X, - pd.DataFrame( - { - "var_A": [0.3333333333333333, nan], - "var_B": [0.2, nan], - } - ), - ) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [0.3333333333333333, None], + "var_B": [0.2, None], + } -def test_warning_if_transform_df_contains_categories_not_present_in_fit_df( - df_enc, df_enc_rare +def test_raises_if_transform_df_contains_categories_not_present_in_fit_df( + make_df, data_enc, data_enc_rare ): # test case 4: when dataset to be transformed contains categories not present # in training dataset + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + X_rare = make_df(data_enc_rare)[["var_A", "var_B"]] msg = "During the encoding, NaN values were introduced in the feature(s) var_A." - # check for warning when rare_labels equals 'ignore' - with pytest.warns(UserWarning) as record: - encoder = MeanEncoder(unseen="ignore") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - encoder.transform(df_enc_rare[["var_A", "var_B"]]) - - # check that at least one warning was raised (Pandas 3 may emit additional - # deprecation warnings) - assert len(record) >= 1 - # check that the message matches - assert any(r.message.args[0] == msg for r in record) + # check for warning when unseen equals 'ignore' + encoder = MeanEncoder(unseen="ignore") + encoder.fit(X, y) + with pytest.warns(UserWarning, match=re.escape(msg)): + encoder.transform(X_rare) - # check for error when rare_labels equals 'raise' - with pytest.raises(ValueError) as record: - encoder = MeanEncoder(unseen="raise") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - encoder.transform(df_enc_rare[["var_A", "var_B"]]) + # check for error when unseen equals 'raise' + encoder = MeanEncoder(unseen="raise") + encoder.fit(X, y) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.transform(X_rare) - # check that the error message matches - assert str(record.value) == msg - -def test_fit_raises_error_if_df_contains_na(df_enc_na): +def test_fit_raises_error_if_df_contains_na(make_df, data_enc_na): # test case 4: when dataset contains na, fit method + X = make_df(data_enc_na)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_na["target"]) + encoder = MeanEncoder() - with pytest.raises(ValueError) as record: - encoder.fit(df_enc_na[["var_A", "var_B"]], df_enc_na["target"]) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer or set the parameter " "`missing_values='ignore'` when initialising this transformer." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(X, y) -def test_transform_raises_error_if_df_contains_na(df_enc, df_enc_na): +def test_transform_raises_error_if_df_contains_na(make_df, data_enc, data_enc_na): # test case 4: when dataset contains na, transform method + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + X_na = make_df(data_enc_na)[["var_A", "var_B"]] + encoder = MeanEncoder() - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - with pytest.raises(ValueError) as record: - encoder.transform(df_enc_na[["var_A", "var_B"]]) + encoder.fit(X, y) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer or set the parameter " "`missing_values='ignore'` when initialising this transformer." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.transform(X_na) -def test_user_enters_1_variable_ignore_format(df_enc_numeric): +def test_user_enters_1_variable_ignore_format(make_df, data_enc_numeric): # test case 1: 1 variable + X = make_df(data_enc_numeric)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_numeric["target"]) + encoder = MeanEncoder(variables=["var_A"], ignore_format=True) - encoder.fit(df_enc_numeric[["var_A", "var_B"]], df_enc_numeric["target"]) - X = encoder.transform(df_enc_numeric[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + enc_dict_var_a = {1: 0.3333333333333333, 2: 0.2, 3: 0.5} - # expected output - transf_df = df_enc_numeric.copy() - transf_df["var_A"] = [ - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.5, - 0.5, - 0.5, - 0.5, - ] - - # test init params - assert encoder.variables == ["var_A"] # test fit attr assert encoder.variables_ == ["var_A"] - assert encoder.encoder_dict_ == {"var_A": {1: 0.3333333333333333, 2: 0.2, 3: 0.5}} + assert encoder.encoder_dict_ == {"var_A": enc_dict_var_a} assert encoder.n_features_in_ == 2 # test transform output - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [enc_dict_var_a[v] for v in data_enc_numeric["var_A"]], + "var_B": data_enc_numeric["var_B"], + } -def test_automatically_find_variables_ignore_format(df_enc_numeric): +def test_automatically_find_variables_ignore_format(make_df, data_enc_numeric): # test case 2: automatically select variables + X = make_df(data_enc_numeric)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_numeric["target"]) + encoder = MeanEncoder(variables=None, ignore_format=True) - encoder.fit(df_enc_numeric[["var_A", "var_B"]], df_enc_numeric["target"]) - X = encoder.transform(df_enc_numeric[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + enc_dict_var_a = {1: 0.3333333333333333, 2: 0.2, 3: 0.5} + enc_dict_var_b = {1: 0.2, 2: 0.3333333333333333, 3: 0.5} - # expected output - transf_df = df_enc_numeric.copy() - transf_df["var_A"] = [ - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.5, - 0.5, - 0.5, - 0.5, - ] - transf_df["var_B"] = [ - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.5, - 0.5, - 0.5, - 0.5, - ] - - # test init params - assert encoder.variables is None # test fit attr assert encoder.variables_ == ["var_A", "var_B"] - assert encoder.encoder_dict_ == { - "var_A": {1: 0.3333333333333333, 2: 0.2, 3: 0.5}, - "var_B": {1: 0.2, 2: 0.3333333333333333, 3: 0.5}, - } + assert encoder.encoder_dict_ == {"var_A": enc_dict_var_a, "var_B": enc_dict_var_b} assert encoder.n_features_in_ == 2 # test transform output - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [enc_dict_var_a[v] for v in data_enc_numeric["var_A"]], + "var_B": [enc_dict_var_b[v] for v in data_enc_numeric["var_B"]], + } def test_variables_cast_as_category(df_enc_category_dtypes): + # pandas-only. df = df_enc_category_dtypes.copy() encoder = MeanEncoder(variables=["var_A"]) encoder.fit(df[["var_A", "var_B"]], df["target"]) @@ -341,40 +247,21 @@ def test_variables_cast_as_category(df_enc_category_dtypes): # expected output transf_df = df.copy() - transf_df["var_A"] = [ - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.3333333333333333, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.2, - 0.5, - 0.5, - 0.5, - 0.5, - ] + transf_df["var_A"] = [0.3333333333333333] * 6 + [0.2] * 10 + [0.5] * 4 pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]], check_dtype=False) assert X["var_A"].dtypes.name == "float64" -def test_auto_smoothing(df_enc): +def test_auto_smoothing(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = MeanEncoder(smoothing="auto") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = encoder.transform(df_enc[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) # expected output - transf_df = df_enc.copy() var_A_dict = { "A": 0.328335832083958, "B": 0.20707964601769913, @@ -385,29 +272,28 @@ def test_auto_smoothing(df_enc): "B": 0.328335832083958, "C": 0.4541284403669725, } - transf_df["var_A"] = transf_df["var_A"].map(var_A_dict) - transf_df["var_B"] = transf_df["var_B"].map(var_B_dict) - # test init params - assert encoder.variables is None # test fit attr assert encoder.variables_ == ["var_A", "var_B"] - assert encoder.encoder_dict_ == { - "var_A": var_A_dict, - "var_B": var_B_dict, - } + assert encoder.encoder_dict_ == {"var_A": var_A_dict, "var_B": var_B_dict} assert encoder.n_features_in_ == 2 # test transform output - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [var_A_dict[v] for v in data_enc["var_A"]], + "var_B": [var_B_dict[v] for v in data_enc["var_B"]], + } -def test_value_smoothing(df_enc): +def test_value_smoothing(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + encoder = MeanEncoder(smoothing=100) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = encoder.transform(df_enc[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) # expected output - transf_df = df_enc.copy() var_A_dict = { "A": 0.3018867924528302, "B": 0.2909090909090909, @@ -418,80 +304,95 @@ def test_value_smoothing(df_enc): "B": 0.3018867924528302, "C": 0.30769230769230765, } - transf_df["var_A"] = transf_df["var_A"].map(var_A_dict) - transf_df["var_B"] = transf_df["var_B"].map(var_B_dict) - # test init params - assert encoder.variables is None # test fit attr assert encoder.variables_ == ["var_A", "var_B"] - assert encoder.encoder_dict_ == { - "var_A": var_A_dict, - "var_B": var_B_dict, - } + assert encoder.encoder_dict_ == {"var_A": var_A_dict, "var_B": var_B_dict} assert encoder.n_features_in_ == 2 # test transform output - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [var_A_dict[v] for v in data_enc["var_A"]], + "var_B": [var_B_dict[v] for v in data_enc["var_B"]], + } + +def test_encoding_new_categories(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + df_unseen = make_df({"var_A": ["D"], "var_B": ["D"]}) -def test_encoding_new_categories(df_enc): - df_unseen = pd.DataFrame({"var_A": ["D"], "var_B": ["D"]}) encoder = MeanEncoder(unseen="encode") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - df_transformed = encoder.transform(df_unseen) - assert (df_transformed == df_enc["target"].mean()).all(axis=None) + encoder.fit(X, y) + Xt = encoder.transform(df_unseen) + target_mean = sum(data_enc["target"]) / len(data_enc["target"]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"var_A": [target_mean], "var_B": [target_mean]} -def test_inverse_transform_when_no_unseen(): - df = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) - y = [1, 0, 1, 0, 1, 0] + +def test_inverse_transform_when_no_unseen(make_df): + words = ["dog", "dog", "cat", "cat", "cat", "bird"] + df = make_df({"words": words}) + y = make_series(make_df, [1, 0, 1, 0, 1, 0]) enc = MeanEncoder() enc.fit(df, y) dft = enc.transform(df) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df) + Xi = enc.inverse_transform(dft) + assert isinstance(Xi, make_df) + assert frame_to_dict(Xi) == {"words": words} -def test_inverse_transform_when_ignore_unseen(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) - df2 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) - df3 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", nan]}) - y = [1, 0, 1, 0, 1, 0] +def test_inverse_transform_when_ignore_unseen(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + df2 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) + y = make_series(make_df, [1, 0, 1, 0, 1, 0]) enc = MeanEncoder(unseen="ignore") enc.fit(df1, y) dft = enc.transform(df2) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df3) + Xi = enc.inverse_transform(dft) + assert isinstance(Xi, make_df) + assert frame_to_dict(Xi) == {"words": ["dog", "dog", "cat", "cat", "cat", None]} -def test_inverse_transform_when_encode_unseen(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) - df2 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) - y = [1, 0, 1, 0, 1, 0] +def test_inverse_transform_when_encode_unseen(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + df2 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) + y = make_series(make_df, [1, 0, 1, 0, 1, 0]) enc = MeanEncoder(unseen="encode") enc.fit(df1, y) dft = enc.transform(df2) - with pytest.raises(NotImplementedError) as record: - enc.inverse_transform(dft) msg = ( "inverse_transform is not implemented for this transformer when " "`unseen='encode'`." ) - assert str(record.value) == msg + with pytest.raises(NotImplementedError, match=re.escape(msg)): + enc.inverse_transform(dft) -def test_inverse_transform_raises_non_fitted_error(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) - y = [1, 0, 1, 0, 1, 0] +def test_inverse_transform_raises_non_fitted_error(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + y = make_series(make_df, [1, 0, 1, 0, 1, 0]) enc = MeanEncoder() + msg = ( + "This MeanEncoder instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + msg_na = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." + ) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): + with pytest.raises(NotFittedError, match=re.escape(msg)): enc.inverse_transform(df1) - df1.loc[len(df1) - 1] = nan + df1_na = make_df({"words": ["dog", "dog", "cat", "cat", "cat", None]}) - with pytest.raises(ValueError): - enc.fit(df1, y) + with pytest.raises(ValueError, match=re.escape(msg_na)): + enc.fit(df1_na, y) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): - enc.inverse_transform(df1) + with pytest.raises(NotFittedError, match=re.escape(msg)): + enc.inverse_transform(df1_na) diff --git a/tests/test_encoding/test_onehot_encoder.py b/tests/test_encoding/test_onehot_encoder.py index aca3448be..5ff6f1a84 100644 --- a/tests/test_encoding/test_onehot_encoder.py +++ b/tests/test_encoding/test_onehot_encoder.py @@ -1,60 +1,105 @@ +import re + import pandas as pd import pytest from sklearn.pipeline import Pipeline from feature_engine.encoding import OneHotEncoder +from tests.backend_helpers import frame_to_dict + +DATA_ENC_BINARY = { + "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "var_C": ["AHA"] * 12 + ["UHU"] * 8, + "var_D": ["OHO"] * 5 + ["EHE"] * 15, + "var_num": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], +} + + +# init parameters +@pytest.mark.parametrize("top_cat", ["empanada", [1], 0.5, -1]) +def test_error_if_top_categories_not_integer(top_cat): + msg = f"top_categories takes only positive integers. Got {top_cat} instead" + with pytest.raises(ValueError, match=re.escape(msg)): + OneHotEncoder(top_categories=top_cat) + + +@pytest.mark.parametrize("drop_last", ["empanada", [1], 0.5, -1, 1, None]) +def test_error_if_drop_last_not_bool(drop_last): + msg = f"drop_last takes only True or False. Got {drop_last} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + OneHotEncoder(drop_last=drop_last) + + +@pytest.mark.parametrize("drop_binary", ["hello", ["auto"], -1, 100, 0.5, None]) +def test_error_if_drop_last_binary_not_bool(drop_binary): + msg = f"drop_last_binary takes only True or False. Got {drop_binary} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + OneHotEncoder(drop_last_binary=drop_binary) + + +@pytest.mark.parametrize( + "top_categories, drop_last, drop_last_binary, ignore_format", + [ + (None, False, False, False), + (1, True, False, True), + (10, False, True, False), + (0, True, True, True), + ], +) +def test_init_param_assignment( + top_categories, drop_last, drop_last_binary, ignore_format +): + encoder = OneHotEncoder( + top_categories=top_categories, + drop_last=drop_last, + drop_last_binary=drop_last_binary, + ignore_format=ignore_format, + ) + assert encoder.top_categories == top_categories + assert encoder.drop_last is drop_last + assert encoder.drop_last_binary is drop_last_binary + assert encoder.ignore_format is ignore_format +# fit and transform @pytest.mark.parametrize("index_", [[1, 2, 3], [3, 2, 1], [4, 9, 2]]) -def test_concat_with_non_ordered_index(index_): - df = pd.DataFrame({"varA": ["a", "b", "c"], "varB": ["d", "d", "a"]}, index=index_) +def test_concat_with_non_ordered_index(make_df, index_): + data = {"varA": ["a", "b", "c"], "varB": ["d", "d", "a"]} + # only pandas has a row index to scramble + if make_df is pd.DataFrame: + df = make_df(data, index=index_) + else: + df = make_df(data) encoder = OneHotEncoder() dft = encoder.fit_transform(df) - df_expected = pd.DataFrame( - { - "varA_a": [1, 0, 0], - "varA_b": [0, 1, 0], - "varA_c": [0, 0, 1], - "varB_d": [1, 1, 0], - "varB_a": [0, 0, 1], - }, - index=index_, - ) - pd.testing.assert_frame_equal(dft, df_expected, check_dtype=False) + expected = { + "varA_a": [1, 0, 0], + "varA_b": [0, 1, 0], + "varA_c": [0, 0, 1], + "varB_d": [1, 1, 0], + "varB_a": [0, 0, 1], + } + assert isinstance(dft, make_df) + assert list(dft.columns) == list(expected) + assert frame_to_dict(dft) == expected -def test_encode_categories_in_k_binary_plus_select_vars_automatically(df_enc_big): + +def test_encode_categories_in_k_binary_plus_select_vars_automatically( + make_df, data_enc_big +): # test case 1: encode all categories into k binary variables, select variables # automatically encoder = OneHotEncoder(top_categories=None, variables=None, drop_last=False) - X = encoder.fit_transform(df_enc_big) + X = encoder.fit_transform(make_df(data_enc_big)) - # test init params - assert encoder.top_categories is None - assert encoder.variables is None - assert encoder.drop_last is False # test fit attr transf = { - "var_A_A": 6, - "var_A_B": 10, - "var_A_C": 4, - "var_A_D": 10, - "var_A_E": 2, - "var_A_F": 2, - "var_A_G": 6, - "var_B_A": 10, - "var_B_B": 6, - "var_B_C": 4, - "var_B_D": 10, - "var_B_E": 2, - "var_B_F": 2, - "var_B_G": 6, - "var_C_A": 4, - "var_C_B": 6, - "var_C_C": 10, - "var_C_D": 10, - "var_C_E": 2, - "var_C_F": 2, + "var_A_A": 6, "var_A_B": 10, "var_A_C": 4, "var_A_D": 10, "var_A_E": 2, + "var_A_F": 2, "var_A_G": 6, "var_B_A": 10, "var_B_B": 6, "var_B_C": 4, + "var_B_D": 10, "var_B_E": 2, "var_B_F": 2, "var_B_G": 6, "var_C_A": 4, + "var_C_B": 6, "var_C_C": 10, "var_C_D": 10, "var_C_E": 2, "var_C_F": 2, "var_C_G": 6, } @@ -67,36 +112,27 @@ def test_encode_categories_in_k_binary_plus_select_vars_automatically(df_enc_big "var_C": ["A", "B", "C", "D", "E", "F", "G"], } # test transform output - assert X.sum().to_dict() == transf - assert "var_A" not in X.columns + assert isinstance(X, make_df) + result = frame_to_dict(X) + assert {col: sum(result[col]) for col in transf} == transf + assert "var_A" not in result -def test_encode_categories_in_k_minus_1_binary_plus_list_of_variables(df_enc_big): +def test_encode_categories_in_k_minus_1_binary_plus_list_of_variables( + make_df, data_enc_big +): # test case 2: encode all categories into k-1 binary variables, # pass list of variables encoder = OneHotEncoder( top_categories=None, variables=["var_A", "var_B"], drop_last=True ) - X = encoder.fit_transform(df_enc_big) + X = encoder.fit_transform(make_df(data_enc_big)) - # test init params - assert encoder.top_categories is None - assert encoder.variables == ["var_A", "var_B"] - assert encoder.drop_last is True # test fit attr transf = { - "var_A_A": 6, - "var_A_B": 10, - "var_A_C": 4, - "var_A_D": 10, - "var_A_E": 2, - "var_A_F": 2, - "var_B_A": 10, - "var_B_B": 6, - "var_B_C": 4, - "var_B_D": 10, - "var_B_E": 2, - "var_B_F": 2, + "var_A_A": 6, "var_A_B": 10, "var_A_C": 4, "var_A_D": 10, "var_A_E": 2, + "var_A_F": 2, "var_B_A": 10, "var_B_B": 6, "var_B_C": 4, "var_B_D": 10, + "var_B_E": 2, "var_B_F": 2, } assert encoder.variables_ == ["var_A", "var_B"] @@ -107,61 +143,23 @@ def test_encode_categories_in_k_minus_1_binary_plus_list_of_variables(df_enc_big "var_B": ["A", "B", "C", "D", "E", "F"], } # test transform output - for col in transf.keys(): - assert X[col].sum() == transf[col] - assert "var_B" not in X.columns - assert "var_B_G" not in X.columns - assert "var_C" in X.columns + assert isinstance(X, make_df) + result = frame_to_dict(X) + assert {col: sum(result[col]) for col in transf} == transf + assert "var_B" not in result + assert "var_B_G" not in result + assert result["var_C"] == data_enc_big["var_C"] -def test_encode_top_categories(): +def test_encode_top_categories(make_df, data_enc_top): # test case 3: encode only the most popular categories - - df = pd.DataFrame( - { - "var_A": ["A"] * 5 - + ["B"] * 11 - + ["C"] * 4 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 7, - "var_B": ["A"] * 11 - + ["B"] * 7 - + ["C"] * 4 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 5, - "var_C": ["A"] * 4 - + ["B"] * 5 - + ["C"] * 11 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 7, - } - ) - encoder = OneHotEncoder(top_categories=4, variables=None, drop_last=False) - X = encoder.fit_transform(df) + X = encoder.fit_transform(make_df(data_enc_top)) - # test init params - assert encoder.top_categories == 4 - # test fit attr transf = { - "var_A_D": 9, - "var_A_B": 11, - "var_A_A": 5, - "var_A_G": 7, - "var_B_A": 11, - "var_B_D": 9, - "var_B_G": 5, - "var_B_B": 7, - "var_C_D": 9, - "var_C_C": 11, - "var_C_G": 7, - "var_C_B": 5, + "var_A_D": 9, "var_A_B": 11, "var_A_A": 5, "var_A_G": 7, + "var_B_A": 11, "var_B_D": 9, "var_B_G": 5, "var_B_B": 7, + "var_C_D": 9, "var_C_C": 11, "var_C_G": 7, "var_C_B": 5, } # test fit attr @@ -174,53 +172,32 @@ def test_encode_top_categories(): "var_C": ["C", "D", "G", "B"], } # test transform output - for col in transf.keys(): - assert X[col].sum() == transf[col] - assert "var_B" not in X.columns - assert "var_B_F" not in X.columns - - -# init params -@pytest.mark.parametrize("top_cat", ["empanada", [1], 0.5, -1]) -def test_error_if_top_categories_not_integer(top_cat): - with pytest.raises(ValueError): - OneHotEncoder(top_categories=top_cat) - - -@pytest.mark.parametrize("drop_last", ["empanada", [1], 0.5, -1, 1]) -def test_error_if_drop_last_not_bool(drop_last): - with pytest.raises(ValueError): - OneHotEncoder(drop_last=drop_last) - + assert isinstance(X, make_df) + result = frame_to_dict(X) + assert {col: sum(result[col]) for col in transf} == transf + assert "var_B" not in result + assert "var_B_F" not in result -@pytest.mark.parametrize("drop_binary", ["hello", ["auto"], -1, 100, 0.5]) -def test_raises_error_when_not_allowed_smoothing_param_in_init(drop_binary): - with pytest.raises(ValueError): - OneHotEncoder(drop_last_binary=drop_binary) - -def test_raises_error_if_df_contains_na(df_enc_big, df_enc_big_na): - # test case 4: when dataset contains na, fit method +def test_raises_error_if_df_contains_na(make_df, data_enc_big, data_enc_big_na): msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer." ) + # test case 4: when dataset contains na, fit method encoder = OneHotEncoder() - with pytest.raises(ValueError) as record: - encoder.fit(df_enc_big_na) - - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(make_df(data_enc_big_na)) # test case 4: when dataset contains na, transform method encoder = OneHotEncoder() - encoder.fit(df_enc_big) - with pytest.raises(ValueError): - encoder.transform(df_enc_big_na) - assert str(record.value) == msg + encoder.fit(make_df(data_enc_big)) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.transform(make_df(data_enc_big_na)) -def test_encode_numerical_variables(df_enc_numeric): +def test_encode_numerical_variables(make_df, data_enc_numeric): encoder = OneHotEncoder( top_categories=None, variables=None, @@ -228,50 +205,48 @@ def test_encode_numerical_variables(df_enc_numeric): ignore_format=True, ) - X = encoder.fit_transform(df_enc_numeric[["var_A", "var_B"]]) + X = encoder.fit_transform(make_df(data_enc_numeric)[["var_A", "var_B"]]) # test fit attr transf = { - "var_A_1": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_A_2": [0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_A_3": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1], - "var_B_1": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_B_2": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_B_3": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1], + "var_A_1": [1] * 6 + [0] * 14, + "var_A_2": [0] * 6 + [1] * 10 + [0] * 4, + "var_A_3": [0] * 16 + [1] * 4, + "var_B_1": [1] * 10 + [0] * 10, + "var_B_2": [0] * 10 + [1] * 6 + [0] * 4, + "var_B_3": [0] * 16 + [1] * 4, } - transf = pd.DataFrame(transf).astype("int32") - X = pd.DataFrame(X).astype("int32") - assert encoder.variables_ == ["var_A", "var_B"] assert encoder.variables_binary_ == [] assert encoder.n_features_in_ == 2 assert encoder.encoder_dict_ == {"var_A": [1, 2, 3], "var_B": [1, 2, 3]} # test transform output - pd.testing.assert_frame_equal(X, transf) + assert isinstance(X, make_df) + assert frame_to_dict(X) == transf def test_variables_cast_as_category(df_enc_numeric): + # pandas-specific: category dtype has no polars equivalent behavior + # under test here (encoding categorical-dtype columns). + df = df_enc_numeric[["var_A", "var_B"]].copy() + df[["var_A", "var_B"]] = df[["var_A", "var_B"]].astype("category") + encoder = OneHotEncoder( top_categories=None, variables=None, drop_last=False, ignore_format=True, ) + X = encoder.fit_transform(df) - df = df_enc_numeric.copy() - df[["var_A", "var_B"]] = df[["var_A", "var_B"]].astype("category") - - X = encoder.fit_transform(df[["var_A", "var_B"]]) - - # test fit attr transf = { - "var_A_1": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_A_2": [0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_A_3": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1], - "var_B_1": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_B_2": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_B_3": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1], + "var_A_1": [1] * 6 + [0] * 14, + "var_A_2": [0] * 6 + [1] * 10 + [0] * 4, + "var_A_3": [0] * 16 + [1] * 4, + "var_B_1": [1] * 10 + [0] * 10, + "var_B_2": [0] * 10 + [1] * 6 + [0] * 4, + "var_B_3": [0] * 16 + [1] * 4, } transf = pd.DataFrame(transf).astype("int32") @@ -284,40 +259,24 @@ def test_variables_cast_as_category(df_enc_numeric): pd.testing.assert_frame_equal(X, transf) -@pytest.fixture(scope="module") -def df_enc_binary(): - df = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "var_C": ["AHA"] * 12 + ["UHU"] * 8, - "var_D": ["OHO"] * 5 + ["EHE"] * 15, - "var_num": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = pd.DataFrame(df) - - return df - - -def test_encode_into_k_dummy_plus_drop_binary(df_enc_binary): +def test_encode_into_k_dummy_plus_drop_binary(make_df): encoder = OneHotEncoder( top_categories=None, variables=None, drop_last=False, drop_last_binary=True ) - X = encoder.fit_transform(df_enc_binary) - X = X.astype("int32") + X = encoder.fit_transform(make_df(DATA_ENC_BINARY)) # test fit attr transf = { - "var_num": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - "var_A_A": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_A_B": [0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_A_C": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1], - "var_B_A": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_B_B": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_B_C": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1], - "var_C_AHA": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0], - "var_D_OHO": [1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], + "var_num": DATA_ENC_BINARY["var_num"], + "var_A_A": [1] * 6 + [0] * 14, + "var_A_B": [0] * 6 + [1] * 10 + [0] * 4, + "var_A_C": [0] * 16 + [1] * 4, + "var_B_A": [1] * 10 + [0] * 10, + "var_B_B": [0] * 10 + [1] * 6 + [0] * 4, + "var_B_C": [0] * 16 + [1] * 4, + "var_C_AHA": [1] * 12 + [0] * 8, + "var_D_OHO": [1] * 5 + [0] * 15, } - transf = pd.DataFrame(transf).astype("int32") assert encoder.variables_ == ["var_A", "var_B", "var_C", "var_D"] assert encoder.variables_binary_ == ["var_C", "var_D"] @@ -329,28 +288,27 @@ def test_encode_into_k_dummy_plus_drop_binary(df_enc_binary): "var_D": ["OHO"], } # test transform output - pd.testing.assert_frame_equal(X, transf) - assert "var_C_B" not in X.columns + assert isinstance(X, make_df) + assert list(X.columns) == list(transf) + assert frame_to_dict(X) == transf -def test_encode_into_kminus1_dummyy_plus_drop_binary(df_enc_binary): +def test_encode_into_kminus1_dummyy_plus_drop_binary(make_df): encoder = OneHotEncoder( top_categories=None, variables=None, drop_last=True, drop_last_binary=True ) - X = encoder.fit_transform(df_enc_binary) - X = X.astype("int32") + X = encoder.fit_transform(make_df(DATA_ENC_BINARY)) # test fit attr transf = { - "var_num": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - "var_A_A": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_A_B": [0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_B_A": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_B_B": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_C_AHA": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0], - "var_D_OHO": [1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], + "var_num": DATA_ENC_BINARY["var_num"], + "var_A_A": [1] * 6 + [0] * 14, + "var_A_B": [0] * 6 + [1] * 10 + [0] * 4, + "var_B_A": [1] * 10 + [0] * 10, + "var_B_B": [0] * 10 + [1] * 6 + [0] * 4, + "var_C_AHA": [1] * 12 + [0] * 8, + "var_D_OHO": [1] * 5 + [0] * 15, } - transf = pd.DataFrame(transf).astype("int32") assert encoder.variables_ == ["var_A", "var_B", "var_C", "var_D"] assert encoder.variables_binary_ == ["var_C", "var_D"] @@ -362,27 +320,27 @@ def test_encode_into_kminus1_dummyy_plus_drop_binary(df_enc_binary): "var_D": ["OHO"], } # test transform output - pd.testing.assert_frame_equal(X, transf) - assert "var_C_B" not in X.columns + assert isinstance(X, make_df) + assert list(X.columns) == list(transf) + assert frame_to_dict(X) == transf -def test_encode_into_top_categories_plus_drop_binary(df_enc_binary): +def test_encode_into_top_categories_plus_drop_binary(make_df): + df = make_df(DATA_ENC_BINARY) # top_categories = 1 encoder = OneHotEncoder( top_categories=1, variables=None, drop_last=False, drop_last_binary=True ) - X = encoder.fit_transform(df_enc_binary) - X = X.astype("int32") + X = encoder.fit_transform(df) # test fit attr transf = { - "var_num": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - "var_A_B": [0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_B_A": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_C_AHA": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0], - "var_D_OHO": [1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], + "var_num": DATA_ENC_BINARY["var_num"], + "var_A_B": [0] * 6 + [1] * 10 + [0] * 4, + "var_B_A": [1] * 10 + [0] * 10, + "var_C_AHA": [1] * 12 + [0] * 8, + "var_D_OHO": [1] * 5 + [0] * 15, } - transf = pd.DataFrame(transf).astype("int32") assert encoder.variables_ == ["var_A", "var_B", "var_C", "var_D"] assert encoder.variables_binary_ == ["var_C", "var_D"] @@ -394,27 +352,26 @@ def test_encode_into_top_categories_plus_drop_binary(df_enc_binary): "var_D": ["OHO"], } # test transform output - pd.testing.assert_frame_equal(X, transf) - assert "var_C_B" not in X.columns + assert isinstance(X, make_df) + assert list(X.columns) == list(transf) + assert frame_to_dict(X) == transf # top_categories = 2 encoder = OneHotEncoder( top_categories=2, variables=None, drop_last=False, drop_last_binary=True ) - X = encoder.fit_transform(df_enc_binary) - X = X.astype("int32") + X = encoder.fit_transform(df) # test fit attr transf = { - "var_num": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - "var_A_B": [0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_A_A": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_B_A": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - "var_B_B": [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0], - "var_C_AHA": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0], - "var_D_OHO": [1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], + "var_num": DATA_ENC_BINARY["var_num"], + "var_A_B": [0] * 6 + [1] * 10 + [0] * 4, + "var_A_A": [1] * 6 + [0] * 14, + "var_B_A": [1] * 10 + [0] * 10, + "var_B_B": [0] * 10 + [1] * 6 + [0] * 4, + "var_C_AHA": [1] * 12 + [0] * 8, + "var_D_OHO": [1] * 5 + [0] * 15, } - transf = pd.DataFrame(transf).astype("int32") assert encoder.variables_ == ["var_A", "var_B", "var_C", "var_D"] assert encoder.variables_binary_ == ["var_C", "var_D"] @@ -426,28 +383,22 @@ def test_encode_into_top_categories_plus_drop_binary(df_enc_binary): "var_D": ["OHO"], } # test transform output - pd.testing.assert_frame_equal(X, transf) - assert "var_C_B" not in X.columns + assert isinstance(X, make_df) + assert list(X.columns) == list(transf) + assert frame_to_dict(X) == transf -def test_get_feature_names_out(df_enc_binary): +def test_get_feature_names_out(make_df): + df = make_df(DATA_ENC_BINARY) original_features = ["var_num"] - input_features = df_enc_binary.columns + input_features = list(DATA_ENC_BINARY) tr = OneHotEncoder() - tr.fit(df_enc_binary) + tr.fit(df) out = [ - "var_A_A", - "var_A_B", - "var_A_C", - "var_B_A", - "var_B_B", - "var_B_C", - "var_C_AHA", - "var_C_UHU", - "var_D_OHO", - "var_D_EHE", + "var_A_A", "var_A_B", "var_A_C", "var_B_A", "var_B_B", "var_B_C", + "var_C_AHA", "var_C_UHU", "var_D_OHO", "var_D_EHE", ] feat_out = original_features + out @@ -456,33 +407,20 @@ def test_get_feature_names_out(df_enc_binary): assert tr.get_feature_names_out(input_features=input_features) == feat_out tr = OneHotEncoder(drop_last=True) - tr.fit(df_enc_binary) + tr.fit(df) - out = [ - "var_A_A", - "var_A_B", - "var_B_A", - "var_B_B", - "var_C_AHA", - "var_D_OHO", - ] + out = ["var_A_A", "var_A_B", "var_B_A", "var_B_B", "var_C_AHA", "var_D_OHO"] feat_out = original_features + out assert tr.get_feature_names_out(input_features=None) == feat_out assert tr.get_feature_names_out(input_features=input_features) == feat_out tr = OneHotEncoder(drop_last_binary=True) - tr.fit(df_enc_binary) + tr.fit(df) out = [ - "var_A_A", - "var_A_B", - "var_A_C", - "var_B_A", - "var_B_B", - "var_B_C", - "var_C_AHA", - "var_D_OHO", + "var_A_A", "var_A_B", "var_A_C", "var_B_A", "var_B_B", "var_B_C", + "var_C_AHA", "var_D_OHO", ] feat_out = original_features + out @@ -490,7 +428,7 @@ def test_get_feature_names_out(df_enc_binary): assert tr.get_feature_names_out(input_features=input_features) == feat_out tr = OneHotEncoder(top_categories=1) - tr.fit(df_enc_binary) + tr.fit(df) out = ["var_A_B", "var_B_A", "var_C_AHA", "var_D_EHE"] feat_out = original_features + out @@ -498,31 +436,26 @@ def test_get_feature_names_out(df_enc_binary): assert tr.get_feature_names_out(input_features=None) == feat_out assert tr.get_feature_names_out(input_features=input_features) == feat_out - with pytest.raises(ValueError): + msg = "input_features must be a list or an array. Got var_A instead." + with pytest.raises(ValueError, match=re.escape(msg)): tr.get_feature_names_out("var_A") - with pytest.raises(ValueError): + msg = "input_features is not equal to feature_names_in_" + with pytest.raises(ValueError, match=re.escape(msg)): tr.get_feature_names_out(["var_A", "hola"]) -def test_get_feature_names_out_from_pipeline(df_enc_binary): +def test_get_feature_names_out_from_pipeline(make_df): + df = make_df(DATA_ENC_BINARY) original_features = ["var_num"] - input_features = df_enc_binary.columns + input_features = list(DATA_ENC_BINARY) tr = Pipeline([("transformer", OneHotEncoder())]) - tr.fit(df_enc_binary) + tr.fit(df) out = [ - "var_A_A", - "var_A_B", - "var_A_C", - "var_B_A", - "var_B_B", - "var_B_C", - "var_C_AHA", - "var_C_UHU", - "var_D_OHO", - "var_D_EHE", + "var_A_A", "var_A_B", "var_A_C", "var_B_A", "var_B_B", "var_B_C", + "var_C_AHA", "var_C_UHU", "var_D_OHO", "var_D_EHE", ] feat_out = original_features + out @@ -530,7 +463,9 @@ def test_get_feature_names_out_from_pipeline(df_enc_binary): assert tr.get_feature_names_out(input_features=input_features) == feat_out -def test_inverse_transform_raises_not_implemented_error(df_enc_binary): - enc = OneHotEncoder().fit(df_enc_binary) - with pytest.raises(NotImplementedError): - enc.inverse_transform(df_enc_binary) +def test_inverse_transform_raises_not_implemented_error(make_df): + df = make_df(DATA_ENC_BINARY) + enc = OneHotEncoder().fit(df) + msg = "inverse_transform is not implemented for this transformer." + with pytest.raises(NotImplementedError, match=re.escape(msg)): + enc.inverse_transform(df) diff --git a/tests/test_encoding/test_ordinal_encoder.py b/tests/test_encoding/test_ordinal_encoder.py index e447c4176..b1d0f48fe 100644 --- a/tests/test_encoding/test_ordinal_encoder.py +++ b/tests/test_encoding/test_ordinal_encoder.py @@ -1,45 +1,112 @@ +import re + +import numpy as np import pandas as pd import pytest -from numpy import nan from sklearn.exceptions import NotFittedError from feature_engine.encoding import OrdinalEncoder +from tests.backend_helpers import make_series, frame_to_dict + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." +) + + +# init parameters +@pytest.mark.parametrize( + "enc_method", + ["other", "Ordered", "", False, 1, 0.5, None, ["ordered"], ("arbitrary",)], +) +def test_error_if_encoding_method_not_allowed(enc_method): + msg = ( + "encoding_method takes only values 'ordered' and 'arbitrary'. " + f"Got {enc_method} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + OrdinalEncoder(encoding_method=enc_method) + + +@pytest.mark.parametrize( + "unseen", ["empanada", False, 1, None, ("raise", "ignore"), ["ignore"]] +) +def test_error_if_unseen_not_permitted_value(unseen): + msg = ( + "Parameter `unseen` takes only values ignore, raise, encode. " + f"Got {unseen} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + OrdinalEncoder(unseen=unseen) -def test_ordered_encoding_1_variable(df_enc): +@pytest.mark.parametrize( + "encoding_method, missing_values, ignore_format, unseen", + [ + ("ordered", "raise", False, "ignore"), + ("arbitrary", "ignore", True, "raise"), + ("ordered", "ignore", True, "encode"), + ], +) +def test_init_param_assignment(encoding_method, missing_values, ignore_format, unseen): + encoder = OrdinalEncoder( + encoding_method=encoding_method, + missing_values=missing_values, + ignore_format=ignore_format, + unseen=unseen, + ) + assert encoder.encoding_method == encoding_method + assert encoder.missing_values == missing_values + assert encoder.ignore_format is ignore_format + assert encoder.unseen == unseen + + +# fit and transform +def test_ordered_encoding_1_variable(make_df, data_enc): # test case 1: 1 variable, ordered encoding - encoder = OrdinalEncoder(encoding_method="ordered", variables=["var_A"]) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = encoder.transform(df_enc[["var_A", "var_B"]]) + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) - # expected output - transf_df = df_enc.copy() - transf_df["var_A"] = [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2, 2, 2, 2] + encoder = OrdinalEncoder(encoding_method="ordered", variables=["var_A"]) + encoder.fit(X, y) + Xt = encoder.transform(X) - # test init params - assert encoder.encoding_method == "ordered" - assert encoder.variables == ["var_A"] # test fit attr assert encoder.variables_ == ["var_A"] assert encoder.encoder_dict_ == {"var_A": {"A": 1, "B": 0, "C": 2}} assert encoder.n_features_in_ == 2 # test transform output - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [1] * 6 + [0] * 10 + [2] * 4, + "var_B": data_enc["var_B"], + } + +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_ordered_encoding_with_target_as_list_or_array(make_df, data_enc, to_target): + # a list or numpy array target takes a different code path than a Series + X = make_df(data_enc)[["var_A", "var_B"]] + y = to_target(data_enc["target"]) -def test_arbitrary_encoding_automatically_find_variables(df_enc): + encoder = OrdinalEncoder(encoding_method="ordered", variables=["var_A"]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": {"A": 1, "B": 0, "C": 2}} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [1] * 6 + [0] * 10 + [2] * 4, + "var_B": data_enc["var_B"], + } + + +def test_arbitrary_encoding_automatically_find_variables(make_df, data_enc): # test case 2: automatically select variables, unordered encoding encoder = OrdinalEncoder(encoding_method="arbitrary", variables=None) - X = encoder.fit_transform(df_enc) - - # expected output - transf_df = df_enc.copy() - transf_df["var_A"] = [0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2] - transf_df["var_B"] = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2] + Xt = encoder.fit_transform(make_df(data_enc)) - # test init params - assert encoder.encoding_method == "arbitrary" - assert encoder.variables is None # test fit attr assert encoder.variables_ == ["var_A", "var_B"] assert encoder.encoder_dict_ == { @@ -48,179 +115,115 @@ def test_arbitrary_encoding_automatically_find_variables(df_enc): } assert encoder.n_features_in_ == 3 # test transform output - pd.testing.assert_frame_equal(X, transf_df) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [0] * 6 + [1] * 10 + [2] * 4, + "var_B": [0] * 10 + [1] * 6 + [2] * 4, + "target": data_enc["target"], + } -def test_encoding_when_nan_in_fit_df(df_enc): - df = df_enc.copy() - df.loc[len(df)] = [nan, nan, 0] +def test_encoding_when_nan_in_fit_df(make_df, data_enc): + data = { + "var_A": data_enc["var_A"] + [None], + "var_B": data_enc["var_B"] + [None], + "target": data_enc["target"] + [0], + } + X = make_df(data)[["var_A", "var_B"]] + y = make_series(make_df, data["target"]) + X_new = make_df({"var_A": ["A", None], "var_B": ["A", None]}) encoder = OrdinalEncoder(encoding_method="arbitrary", missing_values="ignore") - encoder.fit(df[["var_A", "var_B"]]) - - X = encoder.transform( - pd.DataFrame( - { - "var_A": ["A", nan], - "var_B": ["A", nan], - } - ) - ) - - # transform params - pd.testing.assert_frame_equal( - X, - pd.DataFrame( - { - "var_A": [0, nan], - "var_B": [0, nan], - } - ), - check_dtype=False, - ) + encoder.fit(X) + Xt = encoder.transform(X_new) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"var_A": [0, None], "var_B": [0, None]} encoder = OrdinalEncoder(encoding_method="ordered", missing_values="ignore") - encoder.fit(df[["var_A", "var_B"]], df["target"]) - - X = encoder.transform( - pd.DataFrame( - { - "var_A": ["A", nan], - "var_B": ["A", nan], - } - ) - ) - - # transform params - pd.testing.assert_frame_equal( - X, - pd.DataFrame( - { - "var_A": [1, nan], - "var_B": [0, nan], - } - ), - check_dtype=False, - ) - - -@pytest.mark.parametrize("enc_method", ["other", False, 1]) -def test_error_if_encoding_method_not_allowed(enc_method): - with pytest.raises(ValueError): - OrdinalEncoder(encoding_method=enc_method) - - -@pytest.mark.parametrize("enc_method", ["other", False, 1]) -def test_error_if_encoding_method_not_recognized_in_fit(enc_method, df_enc): - enc = OrdinalEncoder() - enc.encoding_method = enc_method - with pytest.raises(ValueError): - enc.fit(df_enc) + encoder.fit(X, y) + Xt = encoder.transform(X_new) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"var_A": [1, None], "var_B": [0, None]} -def test_error_if_ordinal_encoding_and_no_y_passed(df_enc): +def test_error_if_ordinal_encoding_and_no_y_passed(make_df, data_enc): # test case 3: raises error if target is not passed - with pytest.raises(ValueError): - encoder = OrdinalEncoder(encoding_method="ordered") - encoder.fit(df_enc) + encoder = OrdinalEncoder(encoding_method="ordered") + msg = "requires y to be passed, but the target y is None" + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(make_df(data_enc)) def test_error_if_input_df_contains_categories_not_present_in_training_df( - df_enc, df_enc_rare + make_df, data_enc, data_enc_rare ): # test case 4: when dataset to be transformed contains categories not present # in training dataset + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + X_rare = make_df(data_enc_rare)[["var_A", "var_B"]] msg = "During the encoding, NaN values were introduced in the feature(s) var_A." - # check for warning when rare_labels equals 'ignore' - with pytest.warns(UserWarning) as record: - encoder = OrdinalEncoder(unseen="ignore") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - encoder.transform(df_enc_rare[["var_A", "var_B"]]) + # check for warning when unseen equals 'ignore' + encoder = OrdinalEncoder(unseen="ignore") + encoder.fit(X, y) + with pytest.warns(UserWarning, match=re.escape(msg)): + encoder.transform(X_rare) - # check that at least one warning was raised (Pandas 3 may emit additional - # deprecation warnings) - assert len(record) >= 1 - # check that the message matches - assert any(r.message.args[0] == msg for r in record) + # check for error when unseen equals 'raise' + encoder = OrdinalEncoder(unseen="raise") + encoder.fit(X, y) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.transform(X_rare) - # check for error when rare_labels equals 'raise' - with pytest.raises(ValueError) as record: - encoder = OrdinalEncoder(unseen="raise") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - encoder.transform(df_enc_rare[["var_A", "var_B"]]) - # check that the error message matches - assert str(record.value) == msg - - -def test_fit_raises_error_if_df_contains_na(df_enc_na): +def test_fit_raises_error_if_df_contains_na(make_df, data_enc_na): # test case 4: when dataset contains na, fit method encoder = OrdinalEncoder(encoding_method="arbitrary") - with pytest.raises(ValueError) as record: - encoder.fit(df_enc_na) - - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer or set the parameter " - "`missing_values='ignore'` when initialising this transformer." - ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.fit(make_df(data_enc_na)) -def test_transform_raises_error_if_df_contains_na(df_enc, df_enc_na): +def test_transform_raises_error_if_df_contains_na(make_df, data_enc, data_enc_na): # test case 4: when dataset contains na, transform method encoder = OrdinalEncoder(encoding_method="arbitrary") - encoder.fit(df_enc) - with pytest.raises(ValueError) as record: - encoder.transform(df_enc_na) - - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer or set the parameter " - "`missing_values='ignore'` when initialising this transformer." - ) - assert str(record.value) == msg + encoder.fit(make_df(data_enc)) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.transform(make_df(data_enc_na)) -def test_ordered_encoding_1_variable_ignore_format(df_enc_numeric): +def test_ordered_encoding_1_variable_ignore_format(make_df, data_enc_numeric): + X = make_df(data_enc_numeric)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_numeric["target"]) encoder = OrdinalEncoder( encoding_method="ordered", variables=["var_A"], ignore_format=True ) - encoder.fit(df_enc_numeric[["var_A", "var_B"]], df_enc_numeric["target"]) - X = encoder.transform(df_enc_numeric[["var_A", "var_B"]]) - - # expected output - transf_df = df_enc_numeric.copy() - transf_df["var_A"] = [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2, 2, 2, 2] + encoder.fit(X, y) + Xt = encoder.transform(X) - # test init params - assert encoder.encoding_method == "ordered" - assert encoder.variables == ["var_A"] # test fit attr assert encoder.variables_ == ["var_A"] assert encoder.encoder_dict_ == {"var_A": {1: 1, 2: 0, 3: 2}} assert encoder.n_features_in_ == 2 # test transform output - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [1] * 6 + [0] * 10 + [2] * 4, + "var_B": data_enc_numeric["var_B"], + } -def test_arbitrary_encoding_automatically_find_variables_ignore_format(df_enc_numeric): +def test_arbitrary_encoding_automatically_find_variables_ignore_format( + make_df, data_enc_numeric +): + X = make_df(data_enc_numeric)[["var_A", "var_B"]] encoder = OrdinalEncoder( encoding_method="arbitrary", variables=None, ignore_format=True ) - X = encoder.fit_transform(df_enc_numeric[["var_A", "var_B"]]) + Xt = encoder.fit_transform(X) - # expected output - transf_df = df_enc_numeric[["var_A", "var_B"]].copy() - transf_df["var_A"] = [0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2] - transf_df["var_B"] = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2] - - # test init params - assert encoder.encoding_method == "arbitrary" - assert encoder.variables is None # test fit attr assert encoder.variables_ == ["var_A", "var_B"] assert encoder.encoder_dict_ == { @@ -229,10 +232,15 @@ def test_arbitrary_encoding_automatically_find_variables_ignore_format(df_enc_nu } assert encoder.n_features_in_ == 2 # test transform output - pd.testing.assert_frame_equal(X, transf_df) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": [0] * 6 + [1] * 10 + [2] * 4, + "var_B": [0] * 10 + [1] * 6 + [2] * 4, + } def test_variables_cast_as_category(df_enc_category_dtypes): + # pandas-only. df = df_enc_category_dtypes.copy() encoder = OrdinalEncoder(encoding_method="ordered", variables=["var_A"]) encoder.fit(df[["var_A", "var_B"]], df["target"]) @@ -240,70 +248,73 @@ def test_variables_cast_as_category(df_enc_category_dtypes): # expected output transf_df = df.copy() - transf_df["var_A"] = [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2, 2, 2, 2] + transf_df["var_A"] = [1] * 6 + [0] * 10 + [2] * 4 # test transform output pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]], check_dtype=False) assert X["var_A"].dtypes.name == "int64" -@pytest.mark.parametrize( - "unseen", ["empanada", False, 1, ("raise", "ignore"), ["ignore"]] -) -def test_error_if_unseen_not_permitted_value(unseen): - with pytest.raises(ValueError): - OrdinalEncoder(unseen=unseen) - - -def test_inverse_transform_when_no_unseen(): - df = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) +def test_inverse_transform_when_no_unseen(make_df): + words = ["dog", "dog", "cat", "cat", "cat", "bird"] + df = make_df({"words": words}) enc = OrdinalEncoder(encoding_method="arbitrary") enc.fit(df) dft = enc.transform(df) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df) + Xi = enc.inverse_transform(dft) + assert isinstance(Xi, make_df) + assert frame_to_dict(Xi) == {"words": words} -def test_inverse_transform_when_ignore_unseen(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) - df2 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) - df3 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", nan]}) +def test_inverse_transform_when_ignore_unseen(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + df2 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) enc = OrdinalEncoder(encoding_method="arbitrary", unseen="ignore") enc.fit(df1) dft = enc.transform(df2) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df3) + Xi = enc.inverse_transform(dft) + assert isinstance(Xi, make_df) + assert frame_to_dict(Xi) == {"words": ["dog", "dog", "cat", "cat", "cat", None]} -def test_inverse_transform_when_encode_unseen(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) - df2 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) - df3 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", nan]}) +def test_inverse_transform_when_encode_unseen(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + df2 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "frog"]}) enc = OrdinalEncoder(encoding_method="arbitrary", unseen="encode") enc.fit(df1) dft = enc.transform(df2) - pd.testing.assert_frame_equal(enc.inverse_transform(dft), df3) + Xi = enc.inverse_transform(dft) + assert isinstance(Xi, make_df) + assert frame_to_dict(Xi) == {"words": ["dog", "dog", "cat", "cat", "cat", None]} -def test_inverse_transform_raises_non_fitted_error(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) +def test_inverse_transform_raises_non_fitted_error(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) enc = OrdinalEncoder(encoding_method="arbitrary") + msg = ( + "This OrdinalEncoder instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): + with pytest.raises(NotFittedError, match=re.escape(msg)): enc.inverse_transform(df1) - df1.loc[len(df1) - 1] = nan + df1_na = make_df({"words": ["dog", "dog", "cat", "cat", "cat", None]}) - with pytest.raises(ValueError): - enc.fit(df1) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + enc.fit(df1_na) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): - enc.inverse_transform(df1) + with pytest.raises(NotFittedError, match=re.escape(msg)): + enc.inverse_transform(df1_na) -def test_encoding_new_categories(df_enc): - df_unseen = pd.DataFrame({"var_A": ["D"], "var_B": ["D"]}) +def test_encoding_new_categories(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + df_unseen = make_df({"var_A": ["D"], "var_B": ["D"]}) encoder = OrdinalEncoder(encoding_method="arbitrary", unseen="encode") - encoder.fit(df_enc[["var_A", "var_B"]]) - df_transformed = encoder.transform(df_unseen) - assert (df_transformed == -1).all(axis=None) + encoder.fit(X) + Xt = encoder.transform(df_unseen) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"var_A": [-1], "var_B": [-1]} diff --git a/tests/test_encoding/test_rare_label_encoder.py b/tests/test_encoding/test_rare_label_encoder.py index 9594e1cc3..7b2858d15 100644 --- a/tests/test_encoding/test_rare_label_encoder.py +++ b/tests/test_encoding/test_rare_label_encoder.py @@ -1,63 +1,115 @@ +import re from collections import Counter -import numpy as np import pandas as pd +import polars as pl import pytest from feature_engine.encoding import RareLabelEncoder +from tests.backend_helpers import frame_to_dict + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." +) + +FREQUENT_CATEGORIES = { + "var_A": ["B", "D", "A", "G", "C"], + "var_B": ["A", "D", "B", "G", "C"], + "var_C": ["C", "D", "B", "G", "A"], +} + +# data_enc_big after grouping categories E and F as "Rare" +ENC_BIG_RARE = { + "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4 + ["D"] * 10 + ["Rare"] * 4 + ["G"] * 6, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4 + ["D"] * 10 + ["Rare"] * 4 + ["G"] * 6, + "var_C": ["A"] * 4 + ["B"] * 6 + ["C"] * 10 + ["D"] * 10 + ["Rare"] * 4 + ["G"] * 6, +} + + +# init parameters +@pytest.mark.parametrize("tol", ["hello", [0.5], -1, 1.5, None]) +def test_error_if_tol_not_between_0_and_1(tol): + msg = f"tol takes values between 0 and 1. Got {tol} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + RareLabelEncoder(tol=tol) + + +@pytest.mark.parametrize("n_cat", ["hello", [0.5], -0.1, 1.5, -1, None]) +def test_error_if_n_categories_not_int(n_cat): + msg = f"n_categories takes only positive integer numbers. Got {n_cat} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + RareLabelEncoder(n_categories=n_cat) -def test_defo_params_plus_automatically_find_variables(df_enc_big): - # test case 1: defo params, automatically select variables +@pytest.mark.parametrize("max_n_categories", ["hello", ["auto"], -1, 0.5]) +def test_raises_error_when_max_n_categories_not_allowed(max_n_categories): + msg = ( + "max_n_categories takes only positive integer numbers. " + f"Got {max_n_categories} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + RareLabelEncoder(max_n_categories=max_n_categories) + + +@pytest.mark.parametrize("replace_with", [set("hello"), ["auto"], None]) +def test_error_if_replace_with_not_string(replace_with): + msg = ( + "replace_with should be a string, integer or float. " + f"Got {replace_with} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + RareLabelEncoder(replace_with=replace_with) + + +@pytest.mark.parametrize( + "tol, n_categories, max_n_categories, replace_with, missing_values, ignore_format", + [ + (0.05, 10, None, "Rare", "raise", False), + (0, 0, 3, "Other", "ignore", True), + (1, 5, 0, 1, "raise", True), + (0.5, 2, 10, 0.5, "ignore", False), + ], +) +def test_init_param_assignment( + tol, n_categories, max_n_categories, replace_with, missing_values, ignore_format +): encoder = RareLabelEncoder( - tol=0.06, n_categories=5, variables=None, replace_with="Rare" + tol=tol, + n_categories=n_categories, + max_n_categories=max_n_categories, + replace_with=replace_with, + missing_values=missing_values, + ignore_format=ignore_format, ) - X = encoder.fit_transform(df_enc_big) + assert encoder.tol == tol + assert encoder.n_categories == n_categories + assert encoder.max_n_categories == max_n_categories + assert encoder.replace_with == replace_with + assert encoder.missing_values == missing_values + assert encoder.ignore_format is ignore_format - # expected output - df = { - "var_A": ["A"] * 6 - + ["B"] * 10 - + ["C"] * 4 - + ["D"] * 10 - + ["Rare"] * 4 - + ["G"] * 6, - "var_B": ["A"] * 10 - + ["B"] * 6 - + ["C"] * 4 - + ["D"] * 10 - + ["Rare"] * 4 - + ["G"] * 6, - "var_C": ["A"] * 4 - + ["B"] * 6 - + ["C"] * 10 - + ["D"] * 10 - + ["Rare"] * 4 - + ["G"] * 6, - } - df = pd.DataFrame(df) - frequenc_cat = { - "var_A": ["B", "D", "A", "G", "C"], - "var_B": ["A", "D", "B", "G", "C"], - "var_C": ["C", "D", "B", "G", "A"], - } +# fit and transform +def test_defo_params_plus_automatically_find_variables(make_df, data_enc_big): + encoder = RareLabelEncoder( + tol=0.06, n_categories=5, variables=None, replace_with="Rare" + ) + X = encoder.fit_transform(make_df(data_enc_big)) - # test init params - assert encoder.tol == 0.06 - assert encoder.n_categories == 5 - assert encoder.replace_with == "Rare" - assert encoder.variables is None # test fit attr assert encoder.variables_ == ["var_A", "var_B", "var_C"] assert encoder.n_features_in_ == 3 - assert encoder.encoder_dict_ == frequenc_cat + assert encoder.encoder_dict_ == FREQUENT_CATEGORIES # test transform output - pd.testing.assert_frame_equal(X, df) + assert isinstance(X, make_df) + assert frame_to_dict(X) == ENC_BIG_RARE -def test_when_varnames_are_numbers(df_enc_big): - input_df = df_enc_big.copy() +def test_when_varnames_are_numbers(data_enc_big): + # integer column names are pandas-only + input_df = pd.DataFrame(data_enc_big) input_df.columns = [1, 2, 3] encoder = RareLabelEncoder( @@ -66,73 +118,60 @@ def test_when_varnames_are_numbers(df_enc_big): X = encoder.fit_transform(input_df) # expected output - df = { - 1: ["A"] * 6 + ["B"] * 10 + ["C"] * 4 + ["D"] * 10 + ["Rare"] * 4 + ["G"] * 6, - 2: ["A"] * 10 + ["B"] * 6 + ["C"] * 4 + ["D"] * 10 + ["Rare"] * 4 + ["G"] * 6, - 3: ["A"] * 4 + ["B"] * 6 + ["C"] * 10 + ["D"] * 10 + ["Rare"] * 4 + ["G"] * 6, - } - df = pd.DataFrame(df) - - frequenc_cat = { - 1: ["B", "D", "A", "G", "C"], - 2: ["A", "D", "B", "G", "C"], - 3: ["C", "D", "B", "G", "A"], - } + df = pd.DataFrame( + { + 1: ENC_BIG_RARE["var_A"], + 2: ENC_BIG_RARE["var_B"], + 3: ENC_BIG_RARE["var_C"], + } + ) assert encoder.variables_ == [1, 2, 3] - assert encoder.encoder_dict_ == frequenc_cat + assert encoder.encoder_dict_ == { + 1: FREQUENT_CATEGORIES["var_A"], + 2: FREQUENT_CATEGORIES["var_B"], + 3: FREQUENT_CATEGORIES["var_C"], + } pd.testing.assert_frame_equal(X, df) -def test_correctly_ignores_nan_in_transform(df_enc_big): +def test_correctly_ignores_nan_in_transform(make_df, data_enc_big): encoder = RareLabelEncoder( tol=0.06, n_categories=5, missing_values="ignore", ) - X = encoder.fit_transform(df_enc_big) - - # expected: - frequenc_cat = { - "var_A": ["B", "D", "A", "G", "C"], - "var_B": ["A", "D", "B", "G", "C"], - "var_C": ["C", "D", "B", "G", "A"], - } - assert encoder.encoder_dict_ == frequenc_cat - - # input - t = pd.DataFrame( - { - "var_A": ["A", np.nan, "J"], - "var_B": ["A", np.nan, "J"], - "var_C": ["C", np.nan, "J"], - } - ) - - # expected - tt = pd.DataFrame( - { - "var_A": ["A", np.nan, "Rare"], - "var_B": ["A", np.nan, "Rare"], - "var_C": ["C", np.nan, "Rare"], - } + encoder.fit(make_df(data_enc_big)) + assert encoder.encoder_dict_ == FREQUENT_CATEGORIES + + X = encoder.transform( + make_df( + { + "var_A": ["A", None, "J"], + "var_B": ["A", None, "J"], + "var_C": ["C", None, "J"], + } + ) ) - X = encoder.transform(t) - pd.testing.assert_frame_equal(X, tt) - + assert isinstance(X, make_df) + assert frame_to_dict(X) == { + "var_A": ["A", None, "Rare"], + "var_B": ["A", None, "Rare"], + "var_C": ["C", None, "Rare"], + } -def test_correctly_ignores_nan_in_fit(df_enc_big): - df = df_enc_big.copy() - df.loc[df["var_C"] == "G", "var_C"] = np.nan +def test_correctly_ignores_nan_in_fit(make_df, data_enc_big): + data = data_enc_big + data["var_C"] = [None if v == "G" else v for v in data["var_C"]] encoder = RareLabelEncoder( tol=0.06, n_categories=3, missing_values="ignore", ) - encoder.fit(df) + encoder.fit(make_df(data)) # expected: frequent_cat = { @@ -143,72 +182,31 @@ def test_correctly_ignores_nan_in_fit(df_enc_big): for key in frequent_cat.keys(): assert Counter(encoder.encoder_dict_[key]) == Counter(frequent_cat[key]) - # input - t = pd.DataFrame( - { - "var_A": ["A", np.nan, "J", "G"], - "var_B": ["A", np.nan, "J", "G"], - "var_C": ["C", np.nan, "J", "G"], - } - ) - - # expected - tt = pd.DataFrame( - { - "var_A": ["A", np.nan, "Rare", "G"], - "var_B": ["A", np.nan, "Rare", "G"], - "var_C": ["C", np.nan, "Rare", "Rare"], - } + X = encoder.transform( + make_df( + { + "var_A": ["A", None, "J", "G"], + "var_B": ["A", None, "J", "G"], + "var_C": ["C", None, "J", "G"], + } + ) ) - X = encoder.transform(t) - pd.testing.assert_frame_equal(X, tt) - + assert isinstance(X, make_df) + assert frame_to_dict(X) == { + "var_A": ["A", None, "Rare", "G"], + "var_B": ["A", None, "Rare", "G"], + "var_C": ["C", None, "Rare", "Rare"], + } -def test_correctly_ignores_nan_in_fit_when_var_is_numerical(df_enc_big): - df = df_enc_big.copy() +def test_correctly_ignores_nan_in_fit_when_var_is_numerical(data_enc_big): + # pandas only + df = pd.DataFrame(data_enc_big) df["var_C"] = [ - 1, - 1, - 1, - 1, - 2, - 2, - 2, - 2, - 2, - 2, - 3, - 3, - 3, - 3, - 3, - 3, - 3, - 3, - 3, - 3, - 4, - 4, - 4, - 4, - 4, - 4, - 4, - 4, - 4, - 4, - 5, - 5, - 6, - 6, - np.nan, - np.nan, - np.nan, - np.nan, - np.nan, - np.nan, + 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, + 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 5, 5, 6, 6, + None, None, None, None, None, None, ] encoder = RareLabelEncoder( @@ -231,18 +229,19 @@ def test_correctly_ignores_nan_in_fit_when_var_is_numerical(df_enc_big): # input t = pd.DataFrame( { - "var_A": ["A", np.nan, "J", "G"], - "var_B": ["A", np.nan, "J", "G"], - "var_C": [3, np.nan, 9, 10], + "var_A": ["A", None, "J", "G"], + "var_B": ["A", None, "J", "G"], + "var_C": [3, None, 9, 10], } ) - # expected + # var_C mixes floats and strings after transform, so its + # missing value must be an actual nan tt = pd.DataFrame( { - "var_A": ["A", np.nan, "Rare", "G"], - "var_B": ["A", np.nan, "Rare", "G"], - "var_C": [3.0, np.nan, "Rare", "Rare"], + "var_A": ["A", None, "Rare", "G"], + "var_B": ["A", None, "Rare", "G"], + "var_C": [3.0, float("nan"), "Rare", "Rare"], } ) @@ -250,15 +249,18 @@ def test_correctly_ignores_nan_in_fit_when_var_is_numerical(df_enc_big): pd.testing.assert_frame_equal(X, tt, check_dtype=False) -def test_user_provides_grouping_label_name_and_variable_list(df_enc_big): - # test case 2: user provides alternative grouping value and variable list +def test_user_provides_grouping_label_name_and_variable_list(make_df, data_enc_big): encoder = RareLabelEncoder( tol=0.15, n_categories=5, variables=["var_A", "var_B"], replace_with="Other" ) - X = encoder.fit_transform(df_enc_big) + X = encoder.fit_transform(make_df(data_enc_big)) - # expected output - df = { + # test fit attr + assert encoder.variables_ == ["var_A", "var_B"] + assert encoder.n_features_in_ == 3 + # test transform output + assert isinstance(X, make_df) + assert frame_to_dict(X) == { "var_A": ["A"] * 6 + ["B"] * 10 + ["Other"] * 4 @@ -271,92 +273,44 @@ def test_user_provides_grouping_label_name_and_variable_list(df_enc_big): + ["D"] * 10 + ["Other"] * 4 + ["G"] * 6, - "var_C": ["A"] * 4 - + ["B"] * 6 - + ["C"] * 10 - + ["D"] * 10 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 6, + "var_C": data_enc_big["var_C"], } - df = pd.DataFrame(df) - - # test init params - assert encoder.tol == 0.15 - assert encoder.n_categories == 5 - assert encoder.replace_with == "Other" - assert encoder.variables == ["var_A", "var_B"] - # test fit attr - assert encoder.variables_ == ["var_A", "var_B"] - assert encoder.n_features_in_ == 3 - # test transform output - pd.testing.assert_frame_equal(X, df) - -# init params -@pytest.mark.parametrize("tol", ["hello", [0.5], -1, 1.5]) -def test_error_if_tol_not_between_0_and_1(tol): - with pytest.raises(ValueError): - RareLabelEncoder(tol=tol) - - -@pytest.mark.parametrize("n_cat", ["hello", [0.5], -0.1, 1.5]) -def test_error_if_n_categories_not_int(n_cat): - with pytest.raises(ValueError): - RareLabelEncoder(n_categories=n_cat) - - -@pytest.mark.parametrize("max_n_categories", ["hello", ["auto"], -1, 0.5]) -def test_raises_error_when_max_n_categories_not_allowed(max_n_categories): - with pytest.raises(ValueError): - RareLabelEncoder(max_n_categories=max_n_categories) - -@pytest.mark.parametrize("replace_with", [set("hello"), ["auto"]]) -def test_error_if_replace_with_not_string(replace_with): - with pytest.raises(ValueError): - RareLabelEncoder(replace_with=replace_with) - - -def test_warning_if_variable_cardinality_less_than_n_categories(df_enc_big): - # test case 3: when the variable has low cardinality - with pytest.warns(UserWarning): - encoder = RareLabelEncoder(n_categories=10) - encoder.fit(df_enc_big) +def test_warning_if_variable_cardinality_less_than_n_categories( + make_df, data_enc_big +): + msg = ( + "The number of unique categories for variable var_A is less than that " + "indicated in n_categories. Thus, all categories will be " + "considered frequent" + ) + encoder = RareLabelEncoder(n_categories=10) + with pytest.warns(UserWarning, match=re.escape(msg)): + encoder.fit(make_df(data_enc_big)) -def test_fit_raises_error_if_df_contains_na(df_enc_big_na): - # test case 4: when dataset contains na, fit method +def test_fit_raises_error_if_df_contains_na(make_df, data_enc_big_na): encoder = RareLabelEncoder(n_categories=4) - with pytest.raises(ValueError) as record: - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer or set the parameter " - "`missing_values='ignore'` when initialising this transformer." - ) - encoder.fit(df_enc_big_na) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.fit(make_df(data_enc_big_na)) -def test_transform_raises_error_if_df_contains_na(df_enc_big, df_enc_big_na): - # test case 5: when dataset contains na, transform method +def test_transform_raises_error_if_df_contains_na( + make_df, data_enc_big, data_enc_big_na +): encoder = RareLabelEncoder(n_categories=4) - encoder.fit(df_enc_big) - with pytest.raises(ValueError) as record: - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer or set the parameter " - "`missing_values='ignore'` when initialising this transformer." - ) - encoder.transform(df_enc_big_na) - assert str(record.value) == msg + encoder.fit(make_df(data_enc_big)) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.transform(make_df(data_enc_big_na)) -def test_max_n_categories(df_enc_big): - # test case 6: user provides the maximum number of categories they want +def test_max_n_categories(make_df, data_enc_big): rare_encoder = RareLabelEncoder(tol=0.10, max_n_categories=4, n_categories=5) - X = rare_encoder.fit_transform(df_enc_big) - df = { + X = rare_encoder.fit_transform(make_df(data_enc_big)) + + assert isinstance(X, make_df) + assert frame_to_dict(X) == { "var_A": ["A"] * 6 + ["B"] * 10 + ["Rare"] * 4 @@ -376,12 +330,11 @@ def test_max_n_categories(df_enc_big): + ["Rare"] * 4 + ["G"] * 6, } - df = pd.DataFrame(df) - pd.testing.assert_frame_equal(X, df) -def test_max_n_categories_with_numeric_var(df_enc_numeric): - # ignore_format=True +def test_max_n_categories_with_numeric_var(data_enc_numeric): + # pandas only + df_enc_numeric = pd.DataFrame(data_enc_numeric) rare_encoder = RareLabelEncoder( tol=0.10, max_n_categories=2, n_categories=1, ignore_format=True ) @@ -399,39 +352,44 @@ def test_max_n_categories_with_numeric_var(df_enc_numeric): assert str(list(X["var_B"])[i]) == str(list(df["var_B"])[i]) -def test_variables_cast_as_category(df_enc_big): - # test case 1: defo params, automatically select variables - encoder = RareLabelEncoder( - tol=0.06, n_categories=5, variables=None, replace_with="Rare" +def test_max_n_categories_with_numeric_var_polars(data_enc_numeric): + # polars can't mix int and str in one column, so a numeric variable with a + # string replace_with is cast to string + df_enc_numeric = pl.DataFrame(data_enc_numeric) + rare_encoder = RareLabelEncoder( + tol=0.10, max_n_categories=2, n_categories=1, ignore_format=True ) - df_enc_big = df_enc_big.copy() + X = rare_encoder.fit_transform(df_enc_numeric.select(["var_A", "var_B"])) + + assert isinstance(X, pl.DataFrame) + assert frame_to_dict(X) == { + "var_A": ["1"] * 6 + ["2"] * 10 + ["Rare"] * 4, + "var_B": ["1"] * 10 + ["2"] * 6 + ["Rare"] * 4, + } + + +def test_inverse_transform_raises_not_implemented_error(make_df, data_enc_big): + df_enc_big = make_df(data_enc_big) + enc = RareLabelEncoder().fit(df_enc_big) + msg = "inverse_transform is not implemented for this transformer." + with pytest.raises(NotImplementedError, match=re.escape(msg)): + enc.inverse_transform(df_enc_big) + + +def test_variables_cast_as_category(data_enc_big): + # pandas category dtype is backend-specific: polars has no equivalent + # concept in the same sense. + df_enc_big = pd.DataFrame(data_enc_big) df_enc_big["var_B"] = df_enc_big["var_B"].astype("category") + encoder = RareLabelEncoder( + tol=0.06, n_categories=5, variables=None, replace_with="Rare" + ) X = encoder.fit_transform(df_enc_big) # expected output - df = { - "var_A": ["A"] * 6 - + ["B"] * 10 - + ["C"] * 4 - + ["D"] * 10 - + ["Rare"] * 4 - + ["G"] * 6, - "var_B": ["A"] * 10 - + ["B"] * 6 - + ["C"] * 4 - + ["D"] * 10 - + ["Rare"] * 4 - + ["G"] * 6, - "var_C": ["A"] * 4 - + ["B"] * 6 - + ["C"] * 10 - + ["D"] * 10 - + ["Rare"] * 4 - + ["G"] * 6, - } - df = pd.DataFrame(df) + df = pd.DataFrame(ENC_BIG_RARE) df["var_B"] = pd.Categorical(df["var_B"]) # test fit attr @@ -441,7 +399,11 @@ def test_variables_cast_as_category(df_enc_big): pd.testing.assert_frame_equal(X, df, check_categorical=False) -def test_variables_cast_as_category_with_na_in_transform(df_enc_big): +def test_variables_cast_as_category_with_na_in_transform(data_enc_big): + # pandas category dtype is backend-specific. + df_enc_big = pd.DataFrame(data_enc_big) + df_enc_big["var_B"] = df_enc_big["var_B"].astype("category") + encoder = RareLabelEncoder( tol=0.06, n_categories=5, @@ -449,17 +411,14 @@ def test_variables_cast_as_category_with_na_in_transform(df_enc_big): replace_with="Rare", missing_values="ignore", ) - - df_enc_big = df_enc_big.copy() - df_enc_big["var_B"] = df_enc_big["var_B"].astype("category") encoder.fit(df_enc_big) # input t = pd.DataFrame( { - "var_A": ["A", np.nan, "J", "G"], - "var_B": ["A", np.nan, "J", "G"], - "var_C": ["A", np.nan, "J", "G"], + "var_A": ["A", None, "J", "G"], + "var_B": ["A", None, "J", "G"], + "var_C": ["A", None, "J", "G"], } ) t["var_B"] = pd.Categorical(t["var_B"]) @@ -467,19 +426,19 @@ def test_variables_cast_as_category_with_na_in_transform(df_enc_big): # expected tt = pd.DataFrame( { - "var_A": ["A", np.nan, "Rare", "G"], - "var_B": ["A", np.nan, "Rare", "G"], - "var_C": ["A", np.nan, "Rare", "G"], + "var_A": ["A", None, "Rare", "G"], + "var_B": ["A", None, "Rare", "G"], + "var_C": ["A", None, "Rare", "G"], } ) tt["var_B"] = pd.Categorical(tt["var_B"]) pd.testing.assert_frame_equal(encoder.transform(t), tt, check_categorical=False) -def test_variables_cast_as_category_with_na_in_fit(df_enc_big): - - df = df_enc_big.copy() - df.loc[df["var_C"] == "G", "var_C"] = np.nan +def test_variables_cast_as_category_with_na_in_fit(data_enc_big): + # pandas category dtype is backend-specific. + df = pd.DataFrame(data_enc_big) + df.loc[df["var_C"] == "G", "var_C"] = None df["var_C"] = df["var_C"].astype("category") encoder = RareLabelEncoder( @@ -492,9 +451,9 @@ def test_variables_cast_as_category_with_na_in_fit(df_enc_big): # input t = pd.DataFrame( { - "var_A": ["A", np.nan, "J", "G"], - "var_B": ["A", np.nan, "J", "G"], - "var_C": ["C", np.nan, "J", "G"], + "var_A": ["A", None, "J", "G"], + "var_B": ["A", None, "J", "G"], + "var_C": ["C", None, "J", "G"], } ) t["var_C"] = pd.Categorical(t["var_C"]) @@ -502,17 +461,11 @@ def test_variables_cast_as_category_with_na_in_fit(df_enc_big): # expected tt = pd.DataFrame( { - "var_A": ["A", np.nan, "Rare", "G"], - "var_B": ["A", np.nan, "Rare", "G"], - "var_C": ["C", np.nan, "Rare", "Rare"], + "var_A": ["A", None, "Rare", "G"], + "var_B": ["A", None, "Rare", "G"], + "var_C": ["C", None, "Rare", "Rare"], } ) tt["var_C"] = pd.Categorical(tt["var_C"]) pd.testing.assert_frame_equal(encoder.transform(t), tt, check_categorical=False) - - -def test_inverse_transform_raises_not_implemented_error(df_enc_big): - enc = RareLabelEncoder().fit(df_enc_big) - with pytest.raises(NotImplementedError): - enc.inverse_transform(df_enc_big) diff --git a/tests/test_encoding/test_similarity_encoder.py b/tests/test_encoding/test_similarity_encoder.py index 09c17443b..c11bc1fc2 100644 --- a/tests/test_encoding/test_similarity_encoder.py +++ b/tests/test_encoding/test_similarity_encoder.py @@ -1,3 +1,4 @@ +import re from difflib import SequenceMatcher import numpy as np @@ -6,8 +7,73 @@ from feature_engine.encoding import StringSimilarityEncoder from feature_engine.encoding.similarity_encoder import _gpm_fast +from tests.backend_helpers import frame_to_dict + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." +) + + +# init parameters +@pytest.mark.parametrize("top_cat", ["hello", 0.5, [1]]) +def test_error_if_top_categories_not_integer(top_cat): + msg = f"top_categories takes only integers. Got {top_cat!r} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + StringSimilarityEncoder(top_categories=top_cat) + + +@pytest.mark.parametrize( + "missing_values", + ["error", "propagate", "Raise", ["raise"], ("impute",), 1, 0.1, False, None], +) +def test_error_if_missing_values_not_allowed(missing_values): + msg = ( + "missing_values should be one of 'raise', 'impute' or 'ignore'. " + f"Got {missing_values!r} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + StringSimilarityEncoder(missing_values=missing_values) + + +@pytest.mark.parametrize("keywords", ["hello", 0.5, [1]]) +def test_keywords_bad_type(keywords): + msg = f"keywords should be a dictionary or None. Got {keywords!r} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + StringSimilarityEncoder(keywords=keywords) + + +@pytest.mark.parametrize("item", ["hello", 0.5, 1]) +def test_keywords_bad_items(item): + keywords = {"var_A": item} + msg = f"The items in keywords should be lists. Got {keywords.values()!r} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + StringSimilarityEncoder(keywords=keywords) +@pytest.mark.parametrize( + "top_categories, keywords, missing_values, ignore_format", + [ + (None, None, "impute", False), + (2, {"var_A": ["XYZ"]}, "raise", True), + (10, {"var_A": ["X"], "var_B": ["Y", "Z"]}, "ignore", False), + ], +) +def test_init_param_assignment(top_categories, keywords, missing_values, ignore_format): + encoder = StringSimilarityEncoder( + top_categories=top_categories, + keywords=keywords, + missing_values=missing_values, + ignore_format=ignore_format, + ) + assert encoder.top_categories == top_categories + assert encoder.keywords == keywords + assert encoder.missing_values == missing_values + assert encoder.ignore_format is ignore_format + + +# fit and transform @pytest.mark.parametrize( "strings", [("hola", "chau"), ("hi there", "hi here"), (100, 1000)] ) @@ -18,39 +84,10 @@ def test_gpm_fast(strings): ) -def test_encode_top_categories(): - df = pd.DataFrame( - { - "var_A": ["A"] * 5 - + ["B"] * 11 - + ["C"] * 4 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 7, - "var_B": ["A"] * 11 - + ["B"] * 7 - + ["C"] * 4 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 5, - "var_C": ["A"] * 4 - + ["B"] * 5 - + ["C"] * 11 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 7, - } - ) - +def test_encode_top_categories(make_df, data_enc_top): encoder = StringSimilarityEncoder(top_categories=4) - X = encoder.fit_transform(df) + X = encoder.fit_transform(make_df(data_enc_top)) - # test init params - assert encoder.top_categories == 4 - # test fit attr transf = { "var_A_D": 9, "var_A_B": 11, @@ -75,69 +112,38 @@ def test_encode_top_categories(): "var_C": ["C", "D", "G", "B"], } # test transform output - for col in transf.keys(): - assert X[col].sum() == transf[col] - assert "var_B" not in X.columns - assert "var_B_F" not in X.columns - + assert isinstance(X, make_df) + result = frame_to_dict(X) + assert {col: sum(result[col]) for col in transf} == transf + assert "var_B" not in result + assert "var_B_F" not in result -@pytest.mark.parametrize("top_cat", ["hello", 0.5, [1]]) -def test_error_if_top_categories_not_integer(top_cat): - with pytest.raises(ValueError): - StringSimilarityEncoder(top_categories=top_cat) - - -@pytest.mark.parametrize( - "handle_missing", ["error", "propagate", ["raise"], 1, 0.1, False] -) -def test_error_if_handle_missing_invalid(handle_missing): - with pytest.raises(ValueError): - StringSimilarityEncoder(missing_values=handle_missing) - - -@pytest.mark.parametrize("missing_vals", ["other", False, 1]) -def test_error_if_missing_values_not_recognized_in_fit(missing_vals, df_enc): - enc = StringSimilarityEncoder() - enc.missing_values = missing_vals - with pytest.raises(ValueError): - enc.fit(df_enc) - -def test_nan_behaviour_error_fit(df_enc_big_na): +def test_nan_behaviour_error_fit(make_df, data_enc_big_na): encoder = StringSimilarityEncoder(missing_values="raise") - with pytest.raises(ValueError) as record: - encoder.fit(df_enc_big_na) - - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer or set the parameter " - "`missing_values='ignore'` when initialising this transformer." - ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.fit(make_df(data_enc_big_na)) +# pandas offers several NA sentinels (np.nan, pd.NA, None); polars only has +# a single null representation, so this stays pandas-only. @pytest.mark.parametrize("nan_value", [np.nan, pd.NA, None]) -def test_nan_behaviour_error_transform(df_enc_big, nan_value): +def test_nan_behaviour_error_transform(nan_value, data_enc_big): + df_enc_big = pd.DataFrame(data_enc_big) encoder = StringSimilarityEncoder(missing_values="raise") encoder.fit(df_enc_big) df_enc_big_na = df_enc_big.copy() df_enc_big_na.loc[0, "var_A"] = nan_value - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(MSG_NA)): encoder.transform(df_enc_big_na) - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer or set the parameter " - "`missing_values='ignore'` when initialising this transformer." - ) - assert str(record.value) == msg +# pandas-only: several NA sentinels, see above. @pytest.mark.parametrize("nan_value", [np.nan, pd.NA, None]) -def test_nan_behaviour_impute(df_enc_big, nan_value): - - df_enc_big_na = df_enc_big.copy() +def test_nan_behaviour_impute(nan_value, data_enc_big): + df_enc_big_na = pd.DataFrame(data_enc_big) df_enc_big_na.loc[0, "var_A"] = nan_value encoder = StringSimilarityEncoder(missing_values="impute") @@ -151,9 +157,10 @@ def test_nan_behaviour_impute(df_enc_big, nan_value): } +# pandas-only: several NA sentinels, see above. @pytest.mark.parametrize("nan_value", [np.nan, pd.NA, None]) -def test_nan_behaviour_ignore(df_enc_big, nan_value): - df_enc_big_na = df_enc_big.copy() +def test_nan_behaviour_ignore(nan_value, data_enc_big): + df_enc_big_na = pd.DataFrame(data_enc_big) df_enc_big_na.loc[0, "var_A"] = nan_value encoder = StringSimilarityEncoder(missing_values="ignore") @@ -167,18 +174,17 @@ def test_nan_behaviour_ignore(df_enc_big, nan_value): def test_string_dtype_with_pd_na(): - # Test StringDtype with pd.NA to hit "" branch in transform + # pandas nullable "string" dtype is pandas-specific. df = pd.DataFrame({"var_A": ["A", "B", pd.NA]}, dtype="string") encoder = StringSimilarityEncoder(missing_values="impute") X = encoder.fit_transform(df) assert (X.isna().sum() == 0).all(axis=None) - # The categories will include "" or the string version of it assert "" in encoder.encoder_dict_["var_A"] def test_string_dtype_with_literal_nan_strings(): - # Test with literal "nan" and "" strings to hit skips in - # transform (line 339, 341 False) + # literal "nan"/"" strings (not real nulls) must be treated as + # ordinary categories; pandas nullable "string" dtype is pandas-specific. df = pd.DataFrame({"var_A": ["nan", "", "A", "B"]}, dtype="string") encoder = StringSimilarityEncoder(missing_values="impute") X = encoder.fit_transform(df) @@ -187,15 +193,17 @@ def test_string_dtype_with_literal_nan_strings(): assert "" in encoder.encoder_dict_["var_A"] -def test_inverse_transform_error(df_enc_big): +def test_inverse_transform_error(make_df, data_enc_big): encoder = StringSimilarityEncoder() - X = encoder.fit_transform(df_enc_big) - with pytest.raises(NotImplementedError): + X = encoder.fit_transform(make_df(data_enc_big)) + msg = "inverse_transform is not implemented for this transformer." + with pytest.raises(NotImplementedError, match=re.escape(msg)): encoder.inverse_transform(X) -def test_get_feature_names_out(df_enc_big): - input_features = df_enc_big.columns.tolist() +def test_get_feature_names_out(make_df, data_enc_big): + df_enc_big = make_df(data_enc_big) + input_features = list(data_enc_big) tr = StringSimilarityEncoder() tr.fit(df_enc_big) @@ -236,18 +244,20 @@ def test_get_feature_names_out(df_enc_big): assert tr.get_feature_names_out(input_features=None) == out assert tr.get_feature_names_out(input_features=input_features) == out - with pytest.raises(ValueError): + msg = "input_features must be a list or an array. Got var_A instead." + with pytest.raises(ValueError, match=re.escape(msg)): tr.get_feature_names_out("var_A") - with pytest.raises(ValueError): + msg = "input_features is not equal to feature_names_in_" + with pytest.raises(ValueError, match=re.escape(msg)): tr.get_feature_names_out(["var_A", "hola"]) -def test_get_feature_names_out_na(df_enc_big_na): - input_features = df_enc_big_na.columns.tolist() +def test_get_feature_names_out_na(make_df, data_enc_big_na): + input_features = list(data_enc_big_na) tr = StringSimilarityEncoder() - tr.fit(df_enc_big_na) + tr.fit(make_df(data_enc_big_na)) out = [ "var_A_B", @@ -284,58 +294,18 @@ def test_get_feature_names_out_na(df_enc_big_na): assert tr.get_feature_names_out(input_features=input_features) == out -@pytest.mark.parametrize("keywords", ["hello", 0.5, [1]]) -def test_keywords_bad_type(keywords): - with pytest.raises(ValueError): - StringSimilarityEncoder(keywords=keywords) - - -@pytest.mark.parametrize("item", ["hello", 0.5, 1]) -def test_keywords_bad_items(item): - with pytest.raises(ValueError): - StringSimilarityEncoder(keywords={"var_A": item}) - - @pytest.mark.parametrize("key", ["hello", 0.5, 1]) -def test_keywords_bad_keys(df_enc_big, key): +def test_keywords_bad_keys(key, make_df, data_enc_big): encoder = StringSimilarityEncoder(keywords={key: ["A"]}) - with pytest.raises(ValueError): - encoder.fit(df_enc_big) - - -def test_encode_partial_keywords(): - df = pd.DataFrame( - { - "var_A": ["A"] * 5 - + ["B"] * 11 - + ["C"] * 4 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 7, - "var_B": ["A"] * 11 - + ["B"] * 7 - + ["C"] * 4 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 5, - "var_C": ["A"] * 4 - + ["B"] * 5 - + ["C"] * 11 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 7, - } - ) + msg = "There are variables in keywords that are not present in the dataset." + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(make_df(data_enc_big)) + +def test_encode_partial_keywords(make_df, data_enc_top): encoder = StringSimilarityEncoder(top_categories=2, keywords={"var_A": ["XYZ"]}) - X = encoder.fit_transform(df) + X = encoder.fit_transform(make_df(data_enc_top)) - # test init params - assert encoder.top_categories == 2 - # test fit attr transf = { "var_A_XYZ": 0, "var_B_A": 11, @@ -353,43 +323,18 @@ def test_encode_partial_keywords(): "var_C": ["C", "D"], } # test transform output - for col in transf.keys(): - assert X[col].sum() == transf[col] - assert "var_B" not in X.columns - assert "var_B_F" not in X.columns - - -def test_encode_complete_keywords(): - df = pd.DataFrame( - { - "var_A": ["A"] * 5 - + ["B"] * 11 - + ["C"] * 4 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 7, - "var_B": ["A"] * 11 - + ["B"] * 7 - + ["C"] * 4 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 5, - "var_C": ["A"] * 4 - + ["B"] * 5 - + ["C"] * 11 - + ["D"] * 9 - + ["E"] * 2 - + ["F"] * 2 - + ["G"] * 7, - } - ) + assert isinstance(X, make_df) + result = frame_to_dict(X) + assert {col: sum(result[col]) for col in transf} == transf + assert "var_B" not in result + assert "var_B_F" not in result + +def test_encode_complete_keywords(make_df, data_enc_top): encoder = StringSimilarityEncoder( keywords={"var_A": ["X"], "var_B": ["Y"], "var_C": ["Z"]} ) - X = encoder.fit_transform(df) + X = encoder.fit_transform(make_df(data_enc_top)) # test fit attr transf = { @@ -407,17 +352,18 @@ def test_encode_complete_keywords(): "var_C": ["Z"], } # test transform output - for col in transf.keys(): - assert X[col].sum() == transf[col] - assert "var_B" not in X.columns - assert "var_B_F" not in X.columns + assert isinstance(X, make_df) + result = frame_to_dict(X) + assert {col: sum(result[col]) for col in transf} == transf + assert "var_B" not in result + assert "var_B_F" not in result -def test_get_feature_names_out_w_keywords(df_enc_big_na): - input_features = df_enc_big_na.columns.tolist() +def test_get_feature_names_out_w_keywords(make_df, data_enc_big_na): + input_features = list(data_enc_big_na) tr = StringSimilarityEncoder(keywords={"var_A": ["XYZ"]}) - tr.fit(df_enc_big_na) + tr.fit(make_df(data_enc_big_na)) out = [ "var_A_XYZ", diff --git a/tests/test_encoding/test_woe/test_woe_class.py b/tests/test_encoding/test_woe/test_woe_class.py index f26253786..3e6f16111 100644 --- a/tests/test_encoding/test_woe/test_woe_class.py +++ b/tests/test_encoding/test_woe/test_woe_class.py @@ -1,66 +1,50 @@ -import numpy as np -import pandas as pd +import math + import pytest from feature_engine.encoding.woe import WoE - - -def test_woe_calculation(df_enc): - pos_exp = pd.Series({"A": 0.333333, "B": 0.333333, "C": 0.333333}) - neg_exp = pd.Series({"A": 0.285714, "B": 0.571429, "C": 0.142857}) - - woe_class = WoE() - pos, neg, woe = woe_class._calculate_woe(df_enc, df_enc["target"], "var_A") - - pd.testing.assert_series_equal(pos, pos_exp, check_names=False) - pd.testing.assert_series_equal(neg, neg_exp, check_names=False) - pd.testing.assert_series_equal(np.log(pos_exp / neg_exp), woe, check_names=False) - - -def test_woe_error(): - df = { - "var_A": ["B"] * 9 + ["A"] * 6 + ["C"] * 3 + ["D"] * 2, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], +from tests.backend_helpers import frame_to_dict, make_series + + +def test_woe_calculation(make_df, data_enc): + X = make_df(data_enc) + y = make_series(make_df, data_enc["target"]) + + woe, has_zero_counts = WoE()._calculate_woe(X, y, "var_A") + woe = woe.to_native() + + # 6 positive and 14 negative cases + pos = [2 / 6, 2 / 6, 2 / 6] + neg = [4 / 14, 8 / 14, 2 / 14] + assert has_zero_counts is False + assert isinstance(woe, make_df) + assert frame_to_dict(woe) == { + "__category__": ["A", "B", "C"], + "__pos__": pytest.approx(pos), + "__neg__": pytest.approx(neg), + "__woe__": pytest.approx([math.log(p / n) for p, n in zip(pos, neg)]), } - df = pd.DataFrame(df) - woe_class = WoE() - - with pytest.raises(ValueError): - woe_class._calculate_woe(df, df["target"], "var_A") -@pytest.mark.parametrize("fill_value", [1, 10, 0.1]) -def test_fill_value(fill_value): - df = { +def test_zero_counts_are_replaced_by_half(make_df): + data = { "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], } - df = pd.DataFrame(df) - - pos_exp = pd.Series( - { - "A": 0.2857142857142857, - "B": 0.2857142857142857, - "C": 0.42857142857142855, - "D": fill_value, - } - ) - neg_exp = pd.Series( - { - "A": 0.5384615384615384, - "B": 0.3076923076923077, - "C": fill_value, - "D": 0.15384615384615385, - } - ) - - woe_class = WoE() - pos, neg, woe = woe_class._calculate_woe( - df, df["target"], "var_A", fill_value=fill_value - ) - - pd.testing.assert_series_equal(pos, pos_exp, check_names=False) - pd.testing.assert_series_equal(neg, neg_exp, check_names=False) - pd.testing.assert_series_equal(np.log(pos_exp / neg_exp), woe, check_names=False) + X = make_df(data) + y = make_series(make_df, data["target"]) + + woe, has_zero_counts = WoE()._calculate_woe(X, y, "var_A") + woe = woe.to_native() + + # 7 positive and 13 negative cases; C has no negatives and D no positives + pos = [2 / 7, 2 / 7, 3 / 7, 0.5 / 7] + neg = [7 / 13, 4 / 13, 0.5 / 13, 2 / 13] + assert has_zero_counts is True + assert isinstance(woe, make_df) + assert frame_to_dict(woe) == { + "__category__": ["A", "B", "C", "D"], + "__pos__": pytest.approx(pos), + "__neg__": pytest.approx(neg), + "__woe__": pytest.approx([math.log(p / n) for p, n in zip(pos, neg)]), + } diff --git a/tests/test_encoding/test_woe/test_woe_encoder.py b/tests/test_encoding/test_woe/test_woe_encoder.py index a38caa6fa..95e148c38 100644 --- a/tests/test_encoding/test_woe/test_woe_encoder.py +++ b/tests/test_encoding/test_woe/test_woe_encoder.py @@ -1,4 +1,5 @@ import math +import re import numpy as np import pandas as pd @@ -6,102 +7,98 @@ from sklearn.exceptions import NotFittedError from feature_engine.encoding import WoEEncoder +from tests.backend_helpers import make_series, frame_to_dict + +WOE_A = { + "A": 0.15415067982725836, + "B": -0.5389965007326869, + "C": 0.8472978603872037, +} +WOE_B = { + "A": -0.5389965007326869, + "B": 0.15415067982725836, + "C": 0.8472978603872037, +} +VAR_A = [WOE_A["A"]] * 6 + [WOE_A["B"]] * 10 + [WOE_A["C"]] * 4 +VAR_B = [WOE_B["A"]] * 10 + [WOE_B["B"]] * 6 + [WOE_B["C"]] * 4 + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) -VAR_A = [ - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, -] -VAR_B = [ - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, -] +# init parameters +@pytest.mark.parametrize( + "unseen", ["empanada", "encode", False, 1, None, ("raise", "ignore"), ["ignore"]] +) +def test_error_if_unseen_not_permitted_value(unseen): + msg = f"Parameter `unseen` takes only values ignore, raise. Got {unseen} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + WoEEncoder(unseen=unseen) -def test_automatically_select_variables(df_enc): - encoder = WoEEncoder(variables=None) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = encoder.transform(df_enc[["var_A", "var_B"]]) +@pytest.mark.parametrize( + "ignore_format, unseen", + [(False, "ignore"), (True, "raise"), (False, "raise"), (True, "ignore")], +) +def test_init_param_assignment(ignore_format, unseen): + encoder = WoEEncoder(ignore_format=ignore_format, unseen=unseen) + assert encoder.ignore_format is ignore_format + assert encoder.unseen == unseen - # transformed dataframe - transf_df = df_enc.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B - assert encoder.encoder_dict_ == { - "var_A": { - "A": 0.15415067982725836, - "B": -0.5389965007326869, - "C": 0.8472978603872037, - }, - "var_B": { - "A": -0.5389965007326869, - "B": 0.15415067982725836, - "C": 0.8472978603872037, - }, +# fit and transform +def test_automatically_select_variables(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + + encoder = WoEEncoder(variables=None) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert encoder.variables_with_zero_counts_ == [] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), } - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) -def test_user_passes_variables(df_enc): - encoder = WoEEncoder(variables=["var_A", "var_B"]) - encoder.fit(df_enc, df_enc["target"]) - X = encoder.transform(df_enc) +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_or_array(make_df, data_enc, to_target): + # a list or numpy array target takes a different code path than a Series + X = make_df(data_enc)[["var_A", "var_B"]] + y = to_target(data_enc["target"]) - # transformed dataframe - transf_df = df_enc.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B + encoder = WoEEncoder(variables=None) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + } - assert encoder.encoder_dict_ == { - "var_A": { - "A": 0.15415067982725836, - "B": -0.5389965007326869, - "C": 0.8472978603872037, - }, - "var_B": { - "A": -0.5389965007326869, - "B": 0.15415067982725836, - "C": 0.8472978603872037, - }, + +def test_user_passes_variables(make_df, data_enc): + X = make_df(data_enc) + y = make_series(make_df, data_enc["target"]) + + encoder = WoEEncoder(variables=["var_A", "var_B"]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + "target": data_enc["target"], } - pd.testing.assert_frame_equal(X, transf_df) _targets = [ @@ -112,279 +109,183 @@ def test_user_passes_variables(df_enc): @pytest.mark.parametrize("target", _targets) -def test_when_target_class_not_0_1(df_enc, target): - encoder = WoEEncoder(variables=["var_A", "var_B"]) - df_enc["target"] = target - encoder.fit(df_enc, df_enc["target"]) - X = encoder.transform(df_enc) - - # transformed dataframe - transf_df = df_enc.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B +def test_when_target_class_not_0_1(make_df, data_enc, target): + data = dict(data_enc) + data["target"] = target + X = make_df(data) + y = make_series(make_df, target) - assert encoder.encoder_dict_ == { - "var_A": { - "A": 0.15415067982725836, - "B": -0.5389965007326869, - "C": 0.8472978603872037, - }, - "var_B": { - "A": -0.5389965007326869, - "B": 0.15415067982725836, - "C": 0.8472978603872037, - }, + encoder = WoEEncoder(variables=["var_A", "var_B"]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + "target": target, } - pd.testing.assert_frame_equal(X, transf_df) -def test_warn_if_transform_df_contains_categories_not_seen_in_fit(df_enc, df_enc_rare): +def test_warn_if_transform_df_contains_categories_not_seen_in_fit( + make_df, data_enc, data_enc_rare +): # test case 3: when dataset to be transformed contains categories not present # in training dataset + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + X_rare = make_df(data_enc_rare)[["var_A", "var_B"]] msg = "During the encoding, NaN values were introduced in the feature(s) var_A." - # check for error when rare_labels equals 'raise' - with pytest.warns(UserWarning) as record: - encoder = WoEEncoder(unseen="ignore") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - encoder.transform(df_enc_rare[["var_A", "var_B"]]) - - # check that at least one warning was raised (Pandas 3 may emit additional - # deprecation warnings) - assert len(record) >= 1 - # check that the message matches - assert any(r.message.args[0] == msg for r in record) - - # check for error when rare_labels equals 'raise' - with pytest.raises(ValueError) as record: - encoder = WoEEncoder(unseen="raise") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - encoder.transform(df_enc_rare[["var_A", "var_B"]]) + # check for warning when unseen equals 'ignore' + encoder = WoEEncoder(unseen="ignore") + encoder.fit(X, y) + with pytest.warns(UserWarning, match=re.escape(msg)): + encoder.transform(X_rare) - # check that the error message matches - assert str(record.value) == msg + # check for error when unseen equals 'raise' + encoder = WoEEncoder(unseen="raise") + encoder.fit(X, y) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.transform(X_rare) -def test_error_if_target_not_binary(): +def test_error_if_target_not_binary(make_df): # test case 4: the target is not binary - encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError): - df = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 2, 2, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = pd.DataFrame(df) - encoder.fit(df[["var_A", "var_B"]], df["target"]) - - -def test_error_if_denominator_probability_is_zero_1_var(): - df = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = pd.DataFrame(df) - encoder = WoEEncoder(variables=None) - - with pytest.raises(ValueError) as record: - encoder.fit(df[["var_A", "var_B"]], df["target"]) - - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A." - ) - assert str(record.value) == msg - - df = { - "var_A": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "var_B": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = pd.DataFrame(df) - encoder = WoEEncoder(variables=None) - - with pytest.raises(ValueError) as record: - encoder.fit(df[["var_A", "var_B"]], df["target"]) - - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_B." - ) - assert str(record.value) == msg - - -def test_error_if_denominator_probability_is_zero_2_vars(): - df = { + data = { "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], + "target": [1, 1, 2, 2, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = pd.DataFrame(df) - encoder = WoEEncoder(variables=None) - - with pytest.raises(ValueError) as record: - encoder.fit(df, df["target"]) + X = make_df(data)[["var_A", "var_B"]] + y = make_series(make_df, data["target"]) - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A, var_C." - ) - assert str(record.value) == msg - - -def test_error_if_numerator_probability_is_zero(): - df = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "target": [0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = pd.DataFrame(df) encoder = WoEEncoder(variables=None) - - with pytest.raises(ValueError) as record: - encoder.fit(df, df["target"]) - - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A, var_C." - ) - assert str(record.value) == msg - - with pytest.raises(ValueError) as record: - encoder.fit(df[["var_A", "var_B"]], df["target"]) - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A." + "This encoder is designed for binary classification. The target " + "used has more than 2 unique values." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(X, y) -def test_fill_value(): - df = { +def test_zero_counts_are_replaced_by_half(make_df): + # in var_A, C has no negative cases and D no positive cases + data = { "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], } - df = pd.DataFrame(df) - encoder = WoEEncoder(variables=None, fill_value=1) - encoder.fit(df, df["target"]) - woe_exp_a = { - "A": -0.6337237600891445, - "B": -0.07410797215372196, - "C": -0.8472978603872037, - "D": 1.8718021769015913, + X = make_df(data)[["var_A", "var_B"]] + y = make_series(make_df, data["target"]) + + encoder = WoEEncoder().fit(X, y) + Xt = encoder.transform(X) + + # 7 positive and 13 negative cases + woe_a = { + "A": math.log((2 / 7) / (7 / 13)), + "B": math.log((2 / 7) / (4 / 13)), + "C": math.log((3 / 7) / (0.5 / 13)), + "D": math.log((0.5 / 7) / (2 / 13)), + } + woe_b = { + "A": math.log((2 / 7) / (8 / 13)), + "B": math.log((3 / 7) / (3 / 13)), + "C": math.log((2 / 7) / (2 / 13)), } - woe_exp_b = { - "A": -0.7672551527136673, - "B": 0.6190392084062234, - "C": 0.6190392084062234, + assert encoder.encoder_dict_ == { + "var_A": pytest.approx(woe_a), + "var_B": pytest.approx(woe_b), } - woe_exp = {"var_A": woe_exp_a, "var_B": woe_exp_b} - - for var in ["var_A", "var_B"]: - for k, i in woe_exp[var].items(): - assert math.isclose(encoder.encoder_dict_[var][k], woe_exp[var][k]) - - encoder = WoEEncoder(variables=None, fill_value=10) - encoder.fit(df, df["target"]) - woe_exp_a = { - "A": -0.6337237600891445, - "B": -0.07410797215372196, - "C": -3.1498829533812494, - "D": 4.174387269895637, + assert encoder.variables_with_zero_counts_ == ["var_A"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx([woe_a[v] for v in data["var_A"]]), + "var_B": pytest.approx([woe_b[v] for v in data["var_B"]]), } - woe_exp = {"var_A": woe_exp_a, "var_B": woe_exp_b} - for var in ["var_A", "var_B"]: - for k, i in woe_exp[var].items(): - assert math.isclose(encoder.encoder_dict_[var][k], woe_exp[var][k]) -@pytest.mark.parametrize("fill_value", ["hola", [10]]) -def test_error_if_fill_value_not_allowed(fill_value): - with pytest.raises(ValueError): - WoEEncoder(fill_value=fill_value) +def test_variables_with_zero_counts(make_df): + # category A of var_A and var_C has no negative cases + data = { + "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, + "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], + } + X = make_df(data)[["var_A", "var_B", "var_C"]] + y = make_series(make_df, data["target"]) + encoder = WoEEncoder().fit(X, y) -@pytest.mark.parametrize("fill_value", [0, 1, 10, 0.5, 0.002, None]) -def test_assigns_fill_value_at_init(fill_value): - encoder = WoEEncoder(fill_value=fill_value) - assert encoder.fill_value == fill_value + assert encoder.variables_with_zero_counts_ == ["var_A", "var_C"] -def test_error_if_contains_na_in_fit(df_enc_na): +def test_error_if_contains_na_in_fit(make_df, data_enc_na): # test case 9: when dataset contains na, fit method + X = make_df(data_enc_na)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_na["target"]) + encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError) as record: - encoder.fit(df_enc_na[["var_A", "var_B"]], df_enc_na["target"]) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.fit(X, y) - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer." - ) - assert str(record.value) == msg +def test_error_if_df_contains_na_in_transform(make_df, data_enc, data_enc_na): + # test case 10: when dataset contains na, transform method + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + X_na = make_df(data_enc_na)[["var_A", "var_B"]] -def test_error_if_df_contains_na_in_transform(df_enc, df_enc_na): - # test case 10: when dataset contains na, transform method} encoder = WoEEncoder(variables=None) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - with pytest.raises(ValueError) as record: - encoder.transform(df_enc_na[["var_A", "var_B"]]) - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer." - ) - assert str(record.value) == msg + encoder.fit(X, y) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.transform(X_na) -def test_on_numerical_variables(df_enc_numeric): +def test_on_numerical_variables(make_df, data_enc_numeric): # ignore_format=True - encoder = WoEEncoder(variables=None, ignore_format=True) - encoder.fit(df_enc_numeric[["var_A", "var_B"]], df_enc_numeric["target"]) - X = encoder.transform(df_enc_numeric[["var_A", "var_B"]]) + X = make_df(data_enc_numeric)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_numeric["target"]) - # transformed dataframe - transf_df = df_enc_numeric.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B + encoder = WoEEncoder(variables=None, ignore_format=True) + encoder.fit(X, y) + Xt = encoder.transform(X) - # init params - assert encoder.variables is None # fit params assert encoder.variables_ == ["var_A", "var_B"] assert encoder.encoder_dict_ == { - "var_A": { - 1: 0.15415067982725836, - 2: -0.5389965007326869, - 3: 0.8472978603872037, - }, - "var_B": { - 1: -0.5389965007326869, - 2: 0.15415067982725836, - 3: 0.8472978603872037, - }, + "var_A": {1: WOE_A["A"], 2: WOE_A["B"], 3: WOE_A["C"]}, + "var_B": {1: WOE_B["A"], 2: WOE_B["B"], 3: WOE_B["C"]}, } assert encoder.n_features_in_ == 2 # transform params - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + } + + +def test_integer_column_names(data_enc): + # integer column names are pandas-only + X = pd.DataFrame({0: data_enc["var_A"], 1: data_enc["var_B"]}) + y = pd.Series(data_enc["target"]) + + encoder = WoEEncoder().fit(X, y) + + assert encoder.encoder_dict_ == {0: WOE_A, 1: WOE_B} def test_variables_cast_as_category(df_enc_category_dtypes): + # pandas Categorical dtype has no direct polars equivalent. df = df_enc_category_dtypes.copy() encoder = WoEEncoder(variables=None) encoder.fit(df[["var_A", "var_B"]], df["target"]) X = encoder.transform(df[["var_A", "var_B"]]) - # transformed dataframe transf_df = df.copy() transf_df["var_A"] = VAR_A transf_df["var_B"] = VAR_B @@ -393,27 +294,24 @@ def test_variables_cast_as_category(df_enc_category_dtypes): assert X["var_A"].dtypes.name == "float64" -@pytest.mark.parametrize( - "errors", ["empanada", False, 1, ("raise", "ignore"), ["ignore"]] -) -def test_error_if_rare_labels_not_permitted_value(errors): - with pytest.raises(ValueError): - WoEEncoder(unseen=errors) - - -def test_inverse_transform_raises_non_fitted_error(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) +def test_inverse_transform_raises_non_fitted_error(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + y = make_series(make_df, [0, 1, 0, 1, 1, 0]) enc = WoEEncoder() + msg = ( + "This WoEEncoder instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): + with pytest.raises(NotFittedError, match=re.escape(msg)): enc.inverse_transform(df1) - df1.loc[len(df1) - 1] = np.nan + df1_na = make_df({"words": ["dog", "dog", "cat", "cat", "cat", None]}) - with pytest.raises(ValueError): - enc.fit(df1, pd.Series([0, 1, 0, 1, 1, 0])) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + enc.fit(df1_na, y) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): - enc.inverse_transform(df1) + with pytest.raises(NotFittedError, match=re.escape(msg)): + enc.inverse_transform(df1_na) diff --git a/tests/test_imputation/conftest.py b/tests/test_imputation/conftest.py new file mode 100644 index 000000000..906215f99 --- /dev/null +++ b/tests/test_imputation/conftest.py @@ -0,0 +1,49 @@ +"""Data shared by the imputer tests. + +Each fixture returns a fresh dict, so tests can build the dataframe on the +backend under test with ``make_df(data)``. Missing values are written as None, +not np.nan: polars treats np.nan as a real float value (not a null), so +mean/std/quantile would not skip it, unlike pandas. None becomes a null on +both backends. +""" + +import datetime +import pytest + + +@pytest.fixture +def data_na(): + return { + "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], + "City": [ + "London", + "Manchester", + None, + None, + "London", + "London", + "Bristol", + "Manchester", + ], + "Studies": [ + "Bachelor", + "Bachelor", + None, + None, + "Bachelor", + "PhD", + "None", + "Masters", + ], + "Age": [20, 21, 19, None, 23, 40, 41, 37], + "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], + } + + +@pytest.fixture +def data_na_dob(data_na): + # dob is never null: exercises a datetime variable that missing_only=True + # should exclude from variables_. + dob = [datetime.datetime(2020, 2, 24, 0, i) for i in range(8)] + # returns a new dict with every key of data_na plus dob + return data_na | {"dob": dob} diff --git a/tests/test_imputation/test_arbitrary_imputer.py b/tests/test_imputation/test_arbitrary_imputer.py index a8f2ce1c4..8d26ec3e0 100644 --- a/tests/test_imputation/test_arbitrary_imputer.py +++ b/tests/test_imputation/test_arbitrary_imputer.py @@ -1,87 +1,114 @@ +import re + import pytest -import pandas as pd from feature_engine.imputation import ArbitraryImputer, ArbitraryNumberImputer - - -def test_impute_with_99_and_automatically_select_variables(df_na): - # set up the transformer +from tests.backend_helpers import frame_to_dict, null_count + + +# init parameters +@pytest.mark.parametrize("arbitrary_number", ["arbitrary", [1], None]) +def test_error_when_arbitrary_number_not_numeric(arbitrary_number): + msg = ( + "arbitrary_number must be numeric of type int or float. " + f"Got {arbitrary_number} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + ArbitraryImputer(arbitrary_number=arbitrary_number) + + +@pytest.mark.parametrize( + "imputer_dict", [{"Age": "arbitrary_number"}, {"Age": 1, "Marks": [2]}] +) +def test_error_when_imputer_dict_values_not_numeric(imputer_dict): + msg = ( + "All values in the dictionary must be integer or float. " + f"Got {imputer_dict} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + ArbitraryImputer(imputer_dict=imputer_dict) + + +@pytest.mark.parametrize("imputer_dict", ["Age", ["Age", 1], 1]) +def test_error_when_imputer_dict_not_dict(imputer_dict): + msg = ( + "The parameter can only take a dictionary or None. " + f"Got {imputer_dict} instead." + ) + with pytest.raises(TypeError, match=re.escape(msg)): + ArbitraryImputer(imputer_dict=imputer_dict) + + +@pytest.mark.parametrize( + "arbitrary_number, imputer_dict", + [ + (999, None), + (-1, None), + (0.5, {"Age": -42, "Marks": -999}), + (99, {"Age": 1.5}), + ], +) +def test_init_param_assignment(arbitrary_number, imputer_dict): + imputer = ArbitraryImputer( + arbitrary_number=arbitrary_number, imputer_dict=imputer_dict + ) + assert imputer.arbitrary_number == arbitrary_number + assert imputer.imputer_dict == imputer_dict + + +# fit and transform +def test_impute_with_99_and_automatically_select_variables(make_df, data_na): imputer = ArbitraryImputer(arbitrary_number=99, variables=None) - X_transformed = imputer.fit_transform(df_na) - - # set up output reference - X_reference = df_na.copy() - X_reference["Age"] = X_reference["Age"].fillna(99) - X_reference["Marks"] = X_reference["Marks"].fillna(99) - - # test init params - assert imputer.arbitrary_number == 99 - assert imputer.variables is None + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Age", "Marks"] - assert imputer.n_features_in_ == 6 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"Age": 99, "Marks": 99} - # test transform output - # selected variables should not contain NA - # non selected variables should still contain NA - assert X_transformed[["Age", "Marks"]].isnull().sum().sum() == 0 - assert X_transformed[["Name", "City"]].isnull().sum().sum() > 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) - + # selected variables should not contain NA, non-selected should still + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 + assert null_count(X_transformed, "Name") > 0 + assert null_count(X_transformed, "City") > 0 -def test_impute_with_1_and_single_variable_entered_by_user(df_na): - # set up transformer - imputer = ArbitraryImputer(arbitrary_number=-1, variables=["Age"]) - X_transformed = imputer.fit_transform(df_na) + result = frame_to_dict(X_transformed) + assert result["Age"] == [20, 21, 19, 99, 23, 40, 41, 37] + assert result["Marks"] == [0.9, 0.8, 0.7, 99, 0.3, 99, 0.8, 0.6] - # set up output reference - X_reference = df_na.copy() - X_reference["Age"] = X_reference["Age"].fillna(-1) - # test init params - assert imputer.arbitrary_number == -1 - assert imputer.variables == ["Age"] +def test_impute_with_1_and_single_variable_entered_by_user(make_df, data_na): + imputer = ArbitraryImputer(arbitrary_number=-1, variables=["Age"]) + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Age"] - assert imputer.n_features_in_ == 6 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"Age": -1} - # test transform output - assert X_transformed["Age"].isnull().sum() == 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) - + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert frame_to_dict(X_transformed)["Age"] == [20, 21, 19, -1, 23, 40, 41, 37] -def test_error_when_arbitrary_number_is_string(): - with pytest.raises(ValueError): - ArbitraryImputer(arbitrary_number="arbitrary") - -def test_dictionary_of_imputation_values(df_na): - # set up transformer +def test_dictionary_of_imputation_values(make_df, data_na): imputer = ArbitraryImputer(imputer_dict={"Age": -42, "Marks": -999}) - X_transformed = imputer.fit_transform(df_na) - - # set up expected output - X_reference = df_na.copy() - X_reference["Age"] = X_reference["Age"].fillna(-42) - X_reference["Marks"] = X_reference["Marks"].fillna(-999) + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit params - assert imputer.n_features_in_ == 6 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"Age": -42, "Marks": -999} - # test transform params - assert X_transformed[["Age", "Marks"]].isnull().sum().sum() == 0 - assert X_transformed[["Name", "City"]].isnull().sum().sum() > 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) - + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 + assert null_count(X_transformed, "Name") > 0 + assert null_count(X_transformed, "City") > 0 -def test_imputer_error_when_dictionary_value_is_string(): - with pytest.raises(ValueError): - ArbitraryImputer(imputer_dict={"Age": "arbitrary_number"}) + result = frame_to_dict(X_transformed) + assert result["Age"] == [20, 21, 19, -42, 23, 40, 41, 37] + assert result["Marks"] == [0.9, 0.8, 0.7, -999, 0.3, -999, 0.8, 0.6] def test_arbitrary_number_imputer_is_deprecated(): @@ -89,4 +116,3 @@ def test_arbitrary_number_imputer_is_deprecated(): with pytest.warns(FutureWarning, match="ArbitraryNumberImputer was deprecated"): imputer = ArbitraryNumberImputer(arbitrary_number=99) assert isinstance(imputer, ArbitraryImputer) - assert imputer.arbitrary_number == 99 diff --git a/tests/test_imputation/test_categorical_imputer.py b/tests/test_imputation/test_categorical_imputer.py index 182e8826b..94004bf24 100644 --- a/tests/test_imputation/test_categorical_imputer.py +++ b/tests/test_imputation/test_categorical_imputer.py @@ -1,27 +1,87 @@ +import re + import pandas as pd +import polars as pl import pytest from feature_engine.imputation import CategoricalImputer +from tests.backend_helpers import frame_to_dict, null_count -def test_impute_with_string_missing_and_automatically_find_variables(df_na): - # set up transformer - imputer = CategoricalImputer(imputation_method="missing", variables=None) - X_transformed = imputer.fit_transform(df_na) +# init parameters +@pytest.mark.parametrize( + "imputation_method", + ["arbitrary", "mean", 1, None, ("missing",), ["frequent"]], +) +def test_error_when_imputation_method_not_frequent_or_missing(imputation_method): + msg = ( + "imputation_method takes only values 'missing' or 'frequent'. " + f"Got {imputation_method} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + CategoricalImputer(imputation_method=imputation_method) - # set up expected output - X_reference = df_na.copy() - X_reference["Name"] = X_reference["Name"].fillna("Missing") - X_reference["City"] = X_reference["City"].fillna("Missing") - X_reference["Studies"] = X_reference["Studies"].fillna("Missing") - # test init params - assert imputer.imputation_method == "missing" - assert imputer.variables is None +@pytest.mark.parametrize( + "ignore_format", + [22.3, 1, "HOLA", {"key1": "value1", "key2": "value2", "key3": "value3"}], +) +def test_error_when_ignore_format_is_not_boolean(ignore_format): + msg = ( + "ignore_format takes only booleans True and False. " + f"Got {ignore_format} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + CategoricalImputer(imputation_method="missing", ignore_format=ignore_format) + + +@pytest.mark.parametrize( + "return_object", + [22.3, 1, "HOLA", {"key1": "value1", "key2": "value2", "key3": "value3"}], +) +def test_error_when_return_object_is_not_boolean(return_object): + msg = ( + "return_object takes only booleans True and False. " + f"Got {return_object} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + CategoricalImputer(imputation_method="missing", return_object=return_object) + + +@pytest.mark.parametrize( + "imputation_method, fill_value, return_object, ignore_format", + [ + ("missing", "Missing", False, False), + ("missing", 0, True, True), + ("frequent", "Unknown", False, True), + ("frequent", 1.5, True, False), + ], +) +def test_init_param_assignment( + imputation_method, fill_value, return_object, ignore_format +): + imputer = CategoricalImputer( + imputation_method=imputation_method, + fill_value=fill_value, + return_object=return_object, + ignore_format=ignore_format, + ) + assert imputer.imputation_method == imputation_method + assert imputer.fill_value == fill_value + assert imputer.return_object is return_object + assert imputer.ignore_format is ignore_format + + +# fit and transform +def test_impute_with_string_missing_and_automatically_find_variables( + make_df, data_na +): + imputer = CategoricalImputer(imputation_method="missing", variables=None) + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Name", "City", "Studies"] - assert imputer.n_features_in_ == 6 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == { "Name": "Missing", "City": "Missing", @@ -31,84 +91,98 @@ def test_impute_with_string_missing_and_automatically_find_variables(df_na): # test transform output # selected columns should have no NA # non selected columns should still have NA - assert X_transformed[["Name", "City", "Studies"]].isnull().sum().sum() == 0 - assert X_transformed[["Age", "Marks"]].isnull().sum().sum() > 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) - - -def test_user_defined_string_and_automatically_find_variables(df_na): - # set up imputer + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Name") == 0 + assert null_count(X_transformed, "City") == 0 + assert null_count(X_transformed, "Studies") == 0 + assert null_count(X_transformed, "Age") > 0 + assert null_count(X_transformed, "Marks") > 0 + result = frame_to_dict(X_transformed) + assert result["Name"] == [ + "tom", "nick", "krish", "Missing", "peter", "Missing", "fred", "sam", + ] + assert result["City"] == [ + "London", "Manchester", "Missing", "Missing", "London", "London", + "Bristol", "Manchester", + ] + assert result["Studies"] == [ + "Bachelor", "Bachelor", "Missing", "Missing", "Bachelor", "PhD", + "None", "Masters", + ] + + +def test_user_defined_string_and_automatically_find_variables(make_df, data_na): imputer = CategoricalImputer( imputation_method="missing", fill_value="Unknown", variables=None ) - X_transformed = imputer.fit_transform(df_na) - - # set up expected output - X_reference = df_na.copy() - X_reference["Name"] = X_reference["Name"].fillna("Unknown") - X_reference["City"] = X_reference["City"].fillna("Unknown") - X_reference["Studies"] = X_reference["Studies"].fillna("Unknown") - - # test init params - assert imputer.imputation_method == "missing" - assert imputer.fill_value == "Unknown" - assert imputer.variables is None + X_transformed = imputer.fit_transform(make_df(data_na)) - # tes fit attributes + # test fit attributes assert imputer.variables_ == ["Name", "City", "Studies"] - assert imputer.n_features_in_ == 6 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == { "Name": "Unknown", "City": "Unknown", "Studies": "Unknown", } - # test transform output: - assert X_transformed[["Name", "City", "Studies"]].isnull().sum().sum() == 0 - assert X_transformed[["Age", "Marks"]].isnull().sum().sum() > 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) - - -def test_mode_imputation_and_single_variable(df_na): - # set up imputer + # test transform output + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Name") == 0 + assert null_count(X_transformed, "City") == 0 + assert null_count(X_transformed, "Studies") == 0 + assert null_count(X_transformed, "Age") > 0 + assert null_count(X_transformed, "Marks") > 0 + assert frame_to_dict(X_transformed)["City"] == [ + "London", "Manchester", "Unknown", "Unknown", "London", "London", + "Bristol", "Manchester", + ] + + +def test_mode_imputation_and_single_variable(make_df, data_na): imputer = CategoricalImputer(imputation_method="frequent", variables="City") - X_transformed = imputer.fit_transform(df_na) - - # set up expected result - X_reference = df_na.copy() - X_reference["City"] = X_reference["City"].fillna("London") + X_transformed = imputer.fit_transform(make_df(data_na)) - # test init, fit and transform params, attr and output - assert imputer.imputation_method == "frequent" - assert imputer.variables == "City" + # test fit attr and transform output assert imputer.variables_ == ["City"] - assert imputer.n_features_in_ == 6 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"City": "London"} - assert X_transformed["City"].isnull().sum() == 0 - assert X_transformed[["Age", "Marks"]].isnull().sum().sum() > 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "City") == 0 + assert null_count(X_transformed, "Age") > 0 + assert null_count(X_transformed, "Marks") > 0 + assert frame_to_dict(X_transformed)["City"] == [ + "London", "Manchester", "London", "London", "London", "London", + "Bristol", "Manchester", + ] -def test_mode_imputation_with_multiple_variables(df_na): - # set up imputer +def test_mode_imputation_with_multiple_variables(make_df, data_na): imputer = CategoricalImputer( imputation_method="frequent", variables=["Studies", "City"] ) - X_transformed = imputer.fit_transform(df_na) - - # set up expected output - X_reference = df_na.copy() - X_reference["City"] = X_reference["City"].fillna("London") - X_reference["Studies"] = X_reference["Studies"].fillna("Bachelor") + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attr and transform output assert imputer.imputer_dict_ == {"Studies": "Bachelor", "City": "London"} - pd.testing.assert_frame_equal(X_transformed, X_reference) - - -def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_numerical(df_na): - # test case: imputing of numerical variables cast as object + return numeric - df_na = df_na.copy() + assert isinstance(X_transformed, make_df) + result = frame_to_dict(X_transformed) + assert result["Studies"] == [ + "Bachelor", "Bachelor", "Bachelor", "Bachelor", "Bachelor", "PhD", + "None", "Masters", + ] + assert result["City"] == [ + "London", "Manchester", "London", "London", "London", "London", + "Bristol", "Manchester", + ] + + +def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_numerical( + data_na, +): + # casting a numeric column to pandas' "object" dtype while keeping + # numeric values is a pandas quirk with no polars equivalent. + df_na = pd.DataFrame(data_na) df_na["Marks"] = df_na["Marks"].astype("O") imputer = CategoricalImputer( imputation_method="frequent", variables=["City", "Studies", "Marks"] @@ -119,7 +193,6 @@ def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_numerical(d X_reference["Marks"] = X_reference["Marks"].astype(float).fillna(0.8) X_reference["City"] = X_reference["City"].fillna("London") X_reference["Studies"] = X_reference["Studies"].fillna("Bachelor") - assert imputer.variables == ["City", "Studies", "Marks"] assert imputer.variables_ == ["City", "Studies", "Marks"] assert imputer.imputer_dict_ == { "Studies": "Bachelor", @@ -130,10 +203,11 @@ def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_numerical(d pd.testing.assert_frame_equal(X_transformed, X_reference) -def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_object(df_na): - # test case 6: imputing of numerical variables cast as object + return as object - # after imputation - df_na = df_na.copy() +def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_object( + data_na, +): + # pandas only: see comment on the test above. + df_na = pd.DataFrame(data_na) df_na["Marks"] = df_na["Marks"].astype("O") imputer = CategoricalImputer( imputation_method="frequent", @@ -144,83 +218,81 @@ def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_object(df_n assert X_transformed["Marks"].dtype == "O" -def test_error_when_imputation_method_not_frequent_or_missing(): - with pytest.raises(ValueError): - CategoricalImputer(imputation_method="arbitrary") +def test_polars_return_object_is_a_no_op(): + # polars never casts String back to numeric, so return_object has no effect + df_na = pl.DataFrame( + {"Marks": ["0.9", "0.8", "0.7", None, "0.3", None, "0.8", "0.6"]} + ) + imputer = CategoricalImputer( + imputation_method="frequent", + variables=["Marks"], + ignore_format=True, + return_object=True, + ) + X_transformed = imputer.fit_transform(df_na) + assert X_transformed.schema["Marks"] == pl.String -def test_error_when_variable_contains_multiple_modes(df_na): - msg = "The variable Name contains multiple frequent categories." - imputer = CategoricalImputer(imputation_method="frequent", variables="Name") - with pytest.raises(ValueError) as record: - imputer.fit(df_na) - # check that error message matches - assert str(record.value) == msg +def test_uses_smallest_mode_when_variable_has_multiple_modes(make_df, data_na): + # every non-null value of "Name" is unique, so all are modes. The imputer + # picks the sorted-smallest one ("fred") deterministically. + df_na = make_df(data_na) - msg = "The variable(s) Name contain(s) multiple frequent categories." - imputer = CategoricalImputer(imputation_method="frequent") - with pytest.raises(ValueError) as record: - imputer.fit(df_na) - # check that error message matches - assert str(record.value) == msg - - df_ = df_na.copy() - df_["Name_dup"] = df_["Name"] - msg = "The variable(s) Name, Name_dup contain(s) multiple frequent categories." + # explicit variable + imputer = CategoricalImputer(imputation_method="frequent", variables="Name") + imputer.fit(df_na) + assert imputer.imputer_dict_ == {"Name": "fred"} + X_transformed = imputer.transform(df_na) + assert isinstance(X_transformed, make_df) + assert frame_to_dict(X_transformed)["Name"] == [ + "tom", + "nick", + "krish", + "fred", + "peter", + "fred", + "fred", + "sam", + ] + + # auto-selected: only "Name" is multi-mode; "City" has + # a single mode and is unaffected. imputer = CategoricalImputer(imputation_method="frequent") - with pytest.raises(ValueError) as record: - imputer.fit(df_) - # check that error message matches - assert str(record.value) == msg + imputer.fit(df_na) + assert imputer.imputer_dict_["Name"] == "fred" + assert imputer.imputer_dict_["City"] == "London" -def test_impute_numerical_variables(df_na): - # set up transformer +def test_impute_numerical_variables(make_df, data_na): imputer = CategoricalImputer( imputation_method="missing", fill_value=0, variables=["Name", "City", "Studies", "Age", "Marks"], ignore_format=True, ) - X_transformed = imputer.fit_transform(df_na) - - # set up expected output - X_reference = df_na.copy() - X_reference = X_reference.fillna(0) - - # test init params - assert imputer.imputation_method == "missing" - assert imputer.variables == ["Name", "City", "Studies", "Age", "Marks"] + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] - assert imputer.n_features_in_ == 6 + assert imputer.n_features_in_ == 5 - # test transform params - pd.testing.assert_frame_equal(X_transformed, X_reference) + # test transform params: no nulls left anywhere + assert isinstance(X_transformed, make_df) + for col in ["Name", "City", "Studies", "Age", "Marks"]: + assert null_count(X_transformed, col) == 0 -def test_impute_numerical_variables_with_mode(df_na): - # set up transformer +def test_impute_numerical_variables_with_mode(make_df, data_na): imputer = CategoricalImputer( imputation_method="frequent", variables=["City", "Studies", "Marks"], ignore_format=True, ) - X_transformed = imputer.fit_transform(df_na) - - # set up expected output - X_reference = df_na.copy() - X_reference["City"] = X_reference["City"].fillna("London") - X_reference["Studies"] = X_reference["Studies"].fillna("Bachelor") - X_reference["Marks"] = X_reference["Marks"].fillna(0.8) - - # test init params - assert imputer.variables == ["City", "Studies", "Marks"] + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["City", "Studies", "Marks"] - assert imputer.n_features_in_ == 6 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == { "City": "London", "Studies": "Bachelor", @@ -228,80 +300,91 @@ def test_impute_numerical_variables_with_mode(df_na): } # test transform output - pd.testing.assert_frame_equal(X_transformed, X_reference) + assert isinstance(X_transformed, make_df) + for col in ["City", "Studies", "Marks"]: + assert null_count(X_transformed, col) == 0 -def test_variables_cast_as_category_missing(df_na): - # string missing - df_na = df_na.copy() +def test_variables_cast_as_category_missing(data_na): + # pandas only + df_na = pd.DataFrame(data_na) df_na["City"] = df_na["City"].astype("category") imputer = CategoricalImputer(imputation_method="missing", variables=None) X_transformed = imputer.fit_transform(df_na) - # set up expected output X_reference = df_na.copy() X_reference["Name"] = X_reference["Name"].fillna("Missing") X_reference["Studies"] = X_reference["Studies"].fillna("Missing") - X_reference["City"] = ( X_reference["City"].cat.add_categories("Missing").fillna("Missing") ) - # test fit attributes assert imputer.variables_ == ["Name", "City", "Studies"] assert imputer.imputer_dict_ == { "Name": "Missing", "City": "Missing", "Studies": "Missing", } - - # test transform output - # selected columns should have no NA - # non selected columns should still have NA assert X_transformed[["Name", "City", "Studies"]].isnull().sum().sum() == 0 assert X_transformed[["Age", "Marks"]].isnull().sum().sum() > 0 pd.testing.assert_frame_equal(X_transformed, X_reference) -def test_variables_cast_as_category_frequent(df_na): - df_na = df_na.copy() +def test_variables_cast_as_category_frequent(data_na): + # pandas only + df_na = pd.DataFrame(data_na) df_na["City"] = df_na["City"].astype("category") - - # this variable does not have a mode, so drop - df_na.drop(labels=["Name"], axis=1, inplace=True) + df_na = df_na.drop(columns=["Name"]) # this variable has no mode imputer = CategoricalImputer(imputation_method="frequent", variables=None) X_transformed = imputer.fit_transform(df_na) - # set up expected output X_reference = df_na.copy() X_reference["Studies"] = X_reference["Studies"].fillna("Bachelor") X_reference["City"] = X_reference["City"].fillna("London") - # test fit attributes assert imputer.variables_ == ["City", "Studies"] assert imputer.imputer_dict_ == { "City": "London", "Studies": "Bachelor", } - - # test transform output - # selected columns should have no NA - # non selected columns should still have NA assert X_transformed[["City", "Studies"]].isnull().sum().sum() == 0 assert X_transformed[["Age", "Marks"]].isnull().sum().sum() > 0 pd.testing.assert_frame_equal(X_transformed, X_reference) -@pytest.mark.parametrize( - "ignore_format", - [22.3, 1, "HOLA", {"key1": "value1", "key2": "value2", "key3": "value3"}], -) -def test_error_when_ignore_format_is_not_boolean(ignore_format): - msg = "ignore_format takes only booleans True and False" - with pytest.raises(ValueError) as record: - CategoricalImputer(imputation_method="missing", ignore_format=ignore_format) +def test_polars_categorical_dtype_widens_on_missing_fill(data_na): + # polars only. + df_na = pl.DataFrame(data_na).with_columns(pl.col("City").cast(pl.Categorical)) + + imputer = CategoricalImputer( + imputation_method="missing", fill_value="Missing", variables=["City"] + ) + X_transformed = imputer.fit_transform(df_na) - # check that error message matches - assert str(record.value) == msg + assert X_transformed.schema["City"] == pl.Categorical + assert null_count(X_transformed, "City") == 0 + assert frame_to_dict(X_transformed)["City"] == [ + "London", "Manchester", "Missing", "Missing", "London", "London", + "Bristol", "Manchester", + ] + + +def test_polars_enum_fixed_categories_raises_on_missing_fill(data_na): + # polars only. + enum_dtype = pl.Enum(["London", "Manchester", "Bristol"]) + df_na = pl.DataFrame(data_na).with_columns(pl.col("City").cast(enum_dtype)) + + imputer = CategoricalImputer( + imputation_method="missing", fill_value="Missing", variables=["City"] + ) + with pytest.raises(ValueError, match="polars Enum with fixed categories"): + imputer.fit_transform(df_na) + + # a fill value that is already a member of the fixed category set works + imputer_ok = CategoricalImputer( + imputation_method="missing", fill_value="London", variables=["City"] + ) + X_transformed = imputer_ok.fit_transform(df_na) + assert null_count(X_transformed, "City") == 0 diff --git a/tests/test_imputation/test_check_estimator_imputers.py b/tests/test_imputation/test_check_estimator_imputers.py index eb865bb41..7b437ee65 100644 --- a/tests/test_imputation/test_check_estimator_imputers.py +++ b/tests/test_imputation/test_check_estimator_imputers.py @@ -1,9 +1,7 @@ import pandas as pd import pytest -import sklearn from sklearn.pipeline import Pipeline from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine.imputation import ( MissingIndicator, @@ -30,22 +28,13 @@ DropMissingData(), ] -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) -if sklearn_version < parse_version("1.6"): - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator(estimator) - -else: - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator( - estimator=wrap_for_check_estimator(estimator), - expected_failed_checks=estimator._more_tags()["_xfail_checks"], - ) +@pytest.mark.parametrize("estimator", _estimators) +def test_check_estimator_from_sklearn(estimator): + return check_estimator( + estimator=wrap_for_check_estimator(estimator), + expected_failed_checks=estimator._more_tags()["_xfail_checks"], + ) @pytest.mark.parametrize("estimator", _estimators) @@ -87,18 +76,14 @@ def test_raises_non_fitted_error_when_error_during_fit(estimator): X = pd.DataFrame({"cat1": ["a", "b", "c", "a", "b"]}) elif estimator.__class__.__name__ == "ArbitraryImputer": X = pd.DataFrame({"cat1": ["a", "b", "c", "a", "b"]}) - elif estimator.__class__.__name__ == "CategoricalImputer": - # equally frequent categories: fails after variables_ would have been - # selected, inside the "frequent" imputation logic itself. - estimator = estimator.__class__(imputation_method="frequent") - X = pd.DataFrame({"cat1": ["a", "a", "b", "b"]}) elif estimator.__class__.__name__ == "RandomSampleImputer": # invalid random_state: fails after variables_/X_ would have been set. estimator = RandomSampleImputer(seed="observation", random_state="not_a_col") X = pd.DataFrame({"num1": [1.0, 2.0, 3.0, 4.0, 5.0]}) else: - # AddMissingIndicator, DropMissingData: no reachable failure point - # once variables are selected, so fail at input validation instead. + # CategoricalImputer, AddMissingIndicator, DropMissingData: no + # reachable failure point once variables are selected, so fail at + # input validation instead. X = pd.DataFrame() check_raises_non_fitted_error_when_fit_fails(estimator, X) diff --git a/tests/test_imputation/test_drop_missing_data.py b/tests/test_imputation/test_drop_missing_data.py index ee49fee82..715fa898f 100644 --- a/tests/test_imputation/test_drop_missing_data.py +++ b/tests/test_imputation/test_drop_missing_data.py @@ -1,137 +1,199 @@ -import numpy as np -import pandas as pd +import re + import pytest from feature_engine.imputation import DropMissingData +from tests.backend_helpers import frame_to_dict, make_series, null_count + + +# init parameters +@pytest.mark.parametrize("missing_only", ["missing_only", 1, None]) +def test_error_when_missing_only_not_bool(missing_only): + msg = f"missing_only takes values True or False. Got {missing_only} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + DropMissingData(missing_only=missing_only) + + +@pytest.mark.parametrize("threshold", [1.01, -0.01, 0, "0.5"]) +def test_error_when_threshold_not_between_0_and_1(threshold): + msg = f"threshold must be a value between 0 < x <= 1. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + DropMissingData(threshold=threshold) + +@pytest.mark.parametrize( + "missing_only, threshold", + [(True, None), (False, None), (True, 0.5), (False, 1)], +) +def test_init_param_assignment(missing_only, threshold): + imputer = DropMissingData(missing_only=missing_only, threshold=threshold) + assert imputer.missing_only is missing_only + assert imputer.threshold == threshold -def test_detect_variables_with_na(df_na): + +# fit and transform +def test_detect_variables_with_na(make_df, data_na_dob): # test case 1: automatically detect variables with missing data imputer = DropMissingData(missing_only=True, variables=None) - X_transformed = imputer.fit_transform(df_na) - # init params - assert imputer.missing_only is True - assert imputer.threshold is None - assert imputer.variables is None + X_transformed = imputer.fit_transform(make_df(data_na_dob)) # fit params assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] assert imputer.n_features_in_ == 6 - # transform outputs + # transform outputs: only rows complete in variables_ survive + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (5, 6) - assert X_transformed["Name"].shape[0] == 5 - assert X_transformed.isna().sum().sum() == 0 + assert frame_to_dict(X_transformed)["Age"] == [20, 21, 23, 41, 37] + for var in imputer.variables_: + assert null_count(X_transformed, var) == 0 -def test_transform_x_y(df_na): - y = pd.Series(np.zeros(len(df_na))) +def test_transform_x_y(make_df, data_na_dob): + df_na = make_df(data_na_dob) + y = make_series(make_df, list(range(8))) imputer = DropMissingData(missing_only=True, variables=None) X_transformed = imputer.fit_transform(df_na) - # transform outputs assert X_transformed.shape == (5, 6) - assert X_transformed.isna().sum().sum() == 0 assert len(X_transformed) != len(y) Xt, yt = imputer.transform_x_y(df_na, y) + assert isinstance(Xt, make_df) + assert isinstance(yt, type(y)) + # rows 0, 1, 4, 6, 7 are the ones complete in Name/City/Studies/Age/Marks + assert list(yt) == [0, 1, 4, 6, 7] + assert frame_to_dict(Xt)["Age"] == [20, 21, 23, 41, 37] assert len(Xt) == len(yt) - assert (Xt.index == yt.index).all() assert len(df_na) != len(Xt) -def test_selelct_all_variables_when_variables_is_none(df_na): +def test_selelct_all_variables_when_variables_is_none(make_df, data_na_dob): imputer = DropMissingData(missing_only=False, variables=None) - X_transformed = imputer.fit_transform(df_na) + X_transformed = imputer.fit_transform(make_df(data_na_dob)) assert imputer.n_features_in_ == 6 - assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks", "dob"] + assert imputer.variables_ == [ + "Name", "City", "Studies", "Age", "Marks", "dob" + ] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (5, 6) - assert X_transformed[imputer.variables_].isna().sum().sum() == 0 + for var in imputer.variables_: + assert null_count(X_transformed, var) == 0 -def test_detect_variables_with_na_in_variables_entered_by_user(df_na): +def test_detect_variables_with_na_in_variables_entered_by_user(make_df, data_na_dob): imputer = DropMissingData( missing_only=True, variables=["City", "Studies", "Age", "dob"] ) - X_transformed = imputer.fit_transform(df_na) - assert imputer.variables == ["City", "Studies", "Age", "dob"] + X_transformed = imputer.fit_transform(make_df(data_na_dob)) + # dob never has NA in the train set, so it's dropped from variables_ assert imputer.variables_ == ["City", "Studies", "Age"] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (6, 6) + assert frame_to_dict(X_transformed)["Age"] == [20, 21, 23, 40, 41, 37] -def test_return_na_data_method(df_na): +def test_return_na_data_method(make_df, data_na_dob): + df_na = make_df(data_na_dob) - # test with vars + # test with vars and threshold: return_na_data must return the exact + # complement of transform() - row 2 has 2 of 4 variables present, which + # meets thresh=2 and is therefore *kept* by transform(), so it must NOT + # also show up here. imputer = DropMissingData( threshold=0.5, variables=["City", "Studies", "Age", "Marks"] ) imputer.fit_transform(df_na) X_nona = imputer.return_na_data(df_na) - assert list(X_nona.index) == [2, 3] + assert isinstance(X_nona, make_df) + assert X_nona.shape[0] == 1 + assert frame_to_dict(X_nona)["Age"] == [None] # test without vars & threshold imputer = DropMissingData() imputer.fit_transform(df_na) X_nona = imputer.return_na_data(df_na) - assert list(X_nona.index) == [2, 3, 5] - - -def test_error_when_missing_only_not_bool(): - with pytest.raises(ValueError): - DropMissingData(missing_only="missing_only") - - -def test_threshold(df_na): + assert isinstance(X_nona, make_df) + assert X_nona.shape[0] == 3 + assert frame_to_dict(X_nona)["Age"] == [19, None, 40] + + +def test_transform_and_return_na_data_partition_input(make_df, data_na_dob): + # transform() (rows kept) and return_na_data() (rows dropped) must + # partition the input exactly: no row in both, no row in neither. + df_na = make_df(data_na_dob) + for threshold in [None, 1, 0.75, 0.5, 0.25, 0.01]: + imputer = DropMissingData( + threshold=threshold, variables=["City", "Studies", "Age", "Marks"] + ) + imputer.fit(df_na) + kept = imputer.transform(df_na) + dropped = imputer.return_na_data(df_na) + assert kept.shape[0] + dropped.shape[0] == df_na.shape[0] + kept_age = set(frame_to_dict(kept)["Age"]) + dropped_age = set(frame_to_dict(dropped)["Age"]) + assert kept_age.isdisjoint(dropped_age) + + +def test_threshold(make_df, data_na_dob): + df_na = make_df(data_na_dob) # Each row must have 100% data available imputer = DropMissingData(threshold=1) X = imputer.fit_transform(df_na) - assert list(X.index) == [0, 1, 4, 6, 7] + assert isinstance(X, make_df) + assert frame_to_dict(X)["Age"] == [20, 21, 23, 41, 37] # Each row must have at least 1% data available imputer = DropMissingData(threshold=0.01) X = imputer.fit_transform(df_na) - assert list(X.index) == [0, 1, 2, 3, 4, 5, 6, 7] + assert frame_to_dict(X)["Age"] == [20, 21, 19, None, 23, 40, 41, 37] # Each row must have at least 50% data available imputer = DropMissingData(threshold=0.50) X = imputer.fit_transform(df_na) - assert list(X.index) == [0, 1, 2, 4, 5, 6, 7] + assert frame_to_dict(X)["Age"] == [20, 21, 19, 23, 40, 41, 37] - # Each row must have 100% data available + # threshold overrides missing_only, so the same 3 checks hold verbatim + # with missing_only=False: imputer = DropMissingData(threshold=1, missing_only=False) X = imputer.fit_transform(df_na) - assert list(X.index) == [0, 1, 4, 6, 7] + assert frame_to_dict(X)["Age"] == [20, 21, 23, 41, 37] - # Each row must have at least 1% data available imputer = DropMissingData(threshold=0.01, missing_only=False) X = imputer.fit_transform(df_na) - assert list(X.index) == [0, 1, 2, 3, 4, 5, 6, 7] + assert frame_to_dict(X)["Age"] == [20, 21, 19, None, 23, 40, 41, 37] - # Each row must have at least 50% data available imputer = DropMissingData(threshold=0.50, missing_only=False) X = imputer.fit_transform(df_na) - assert list(X.index) == [0, 1, 2, 4, 5, 6, 7] - + assert frame_to_dict(X)["Age"] == [20, 21, 19, 23, 40, 41, 37] -def test_threshold_value_error(df_na): - with pytest.raises(ValueError): - DropMissingData(threshold=1.01) - with pytest.raises(ValueError): - DropMissingData(threshold=-0.01) +def test_threshold_with_variables(make_df, data_na_dob): + df_na = make_df(data_na_dob) - with pytest.raises(ValueError): - DropMissingData(threshold=0) - - -def test_threshold_with_variables(df_na): - - # Each row must have 100% data avaiable for columns ['Marks'] + # Each row must have 100% data available for column ['Marks'] imputer = DropMissingData(threshold=1, variables=["Marks"]) X = imputer.fit_transform(df_na) - assert list(X.index) == [0, 1, 2, 4, 6, 7] + assert isinstance(X, make_df) + assert frame_to_dict(X)["Age"] == [20, 21, 19, 23, 41, 37] - # Each row must have 25% data avaiable for ['City', 'Studies', 'Age', 'Marks'] + # Each row must have 75% data available for ['City', 'Studies', 'Age', 'Marks'] imputer = DropMissingData( threshold=0.75, variables=["City", "Studies", "Age", "Marks"] ) X = imputer.fit_transform(df_na) - assert list(X.index) == [0, 1, 4, 5, 6, 7] + assert frame_to_dict(X)["Age"] == [20, 21, 23, 40, 41, 37] + + +def test_missing_only_finds_no_variables_leaves_data_unchanged(make_df): + # A clean training set has nothing for missing_only=True to select: + # variables_ ends up empty, and transform()/return_na_data() must not + # error on the narwhals horizontal-expression path with 0 columns. + clean_data = {"x1": [1, 2, 3], "x2": [4, 5, 6]} + X = make_df(clean_data) + imputer = DropMissingData() + Xt = imputer.fit_transform(X) + assert imputer.variables_ == [] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == clean_data + X_nona = imputer.return_na_data(X) + assert isinstance(X_nona, make_df) + assert X_nona.shape == (0, 2) diff --git a/tests/test_imputation/test_end_tail_imputer.py b/tests/test_imputation/test_end_tail_imputer.py index 88998d658..4a6d0e532 100644 --- a/tests/test_imputation/test_end_tail_imputer.py +++ b/tests/test_imputation/test_end_tail_imputer.py @@ -1,103 +1,120 @@ +import re + import numpy as np -import pandas as pd import pytest from feature_engine.imputation import EndTailImputer +from tests.backend_helpers import frame_to_dict, null_count + + +# init parameters +@pytest.mark.parametrize( + "imputation_method", ["arbitrary", "mean", 1, ("iqr",), ["iqr"]] +) +def test_error_when_imputation_method_is_not_permitted(imputation_method): + msg = ( + "imputation_method takes only values 'gaussian', 'iqr' or 'max'. " + f"Got {imputation_method} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + EndTailImputer(imputation_method=imputation_method) + + +@pytest.mark.parametrize("tail", ["arbitrary", "both", 1, ("right",), ["right"]]) +def test_error_when_tail_is_not_permitted(tail): + msg = f"tail takes only values 'right' or 'left'. Got {tail} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EndTailImputer(tail=tail) -def test_automatically_find_variables_and_gaussian_imputation_on_right_tail(df_na): - # set up transformer +@pytest.mark.parametrize("fold", [-1, 0, -0.5, "3", None, [3], True]) +def test_error_when_fold_is_not_positive_number(fold): + msg = f"fold takes only positive numbers. Got {fold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EndTailImputer(fold=fold) + + +@pytest.mark.parametrize( + "imputation_method, tail, fold", + [("gaussian", "right", 3), ("iqr", "left", 1.5), ("max", "right", 2)], +) +def test_init_param_assignment(imputation_method, tail, fold): + imputer = EndTailImputer(imputation_method=imputation_method, tail=tail, fold=fold) + assert imputer.imputation_method == imputation_method + assert imputer.tail == tail + assert imputer.fold == fold + + +# fit and transform +def test_automatically_find_variables_and_gaussian_imputation_on_right_tail( + make_df, data_na +): imputer = EndTailImputer( imputation_method="gaussian", tail="right", fold=3, variables=None ) - X_transformed = imputer.fit_transform(df_na) - - # set up expected output - X_reference = df_na.copy() - X_reference["Age"] = X_reference["Age"].fillna(58.94908118478389) - X_reference["Marks"] = X_reference["Marks"].fillna(1.3244261503263175) - - # test init params - assert imputer.imputation_method == "gaussian" - assert imputer.tail == "right" - assert imputer.fold == 3 - assert imputer.variables is None + X_transformed = imputer.fit_transform(make_df(data_na)) + # test fit attr assert imputer.variables_ == ["Age", "Marks"] - assert imputer.n_features_in_ == 6 - imputer.imputer_dict_ = { - key: round(value, 3) for (key, value) in imputer.imputer_dict_.items() - } - assert imputer.imputer_dict_ == { - "Age": 58.949, - "Marks": 1.324, - } + assert imputer.n_features_in_ == 5 + rounded = {k: round(v, 3) for k, v in imputer.imputer_dict_.items()} + assert rounded == {"Age": 58.949, "Marks": 1.324} + # transform output: indicated vars ==> no NA, not indicated vars with NA - assert X_transformed[["Age", "Marks"]].isnull().sum().sum() == 0 - assert X_transformed[["City", "Name"]].isnull().sum().sum() > 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 + assert null_count(X_transformed, "City") > 0 + assert null_count(X_transformed, "Name") > 0 + + expected = dict(data_na) + expected["Age"] = pytest.approx([20, 21, 19, 58.94908118478389, 23, 40, 41, 37]) + expected["Marks"] = pytest.approx( + [0.9, 0.8, 0.7, 1.3244261503263175, 0.3, 1.3244261503263175, 0.8, 0.6] + ) + assert frame_to_dict(X_transformed) == expected -def test_user_enters_variables_and_iqr_imputation_on_right_tail(df_na): - # set up transformer +def test_user_enters_variables_and_iqr_imputation_on_right_tail(make_df, data_na): imputer = EndTailImputer( imputation_method="iqr", tail="right", fold=1.5, variables=["Age", "Marks"] ) - X_transformed = imputer.fit_transform(df_na) - - # set up expected result - X_reference = df_na.copy() - X_reference["Age"] = X_reference["Age"].fillna(65.5) - X_reference["Marks"] = X_reference["Marks"].fillna(1.0625) + X_transformed = imputer.fit_transform(make_df(data_na)) - # test fit and transform attr and output assert imputer.imputer_dict_ == {"Age": 65.5, "Marks": 1.0625} - assert X_transformed[["Age", "Marks"]].isnull().sum().sum() == 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 + + expected = dict(data_na) + expected["Age"] = pytest.approx([20, 21, 19, 65.5, 23, 40, 41, 37]) + expected["Marks"] = pytest.approx([0.9, 0.8, 0.7, 1.0625, 0.3, 1.0625, 0.8, 0.6]) + assert frame_to_dict(X_transformed) == expected -def test_user_enters_variables_and_max_value_imputation(df_na): +def test_user_enters_variables_and_max_value_imputation(make_df, data_na): imputer = EndTailImputer( imputation_method="max", tail="right", fold=2, variables=["Age", "Marks"] ) - imputer.fit(df_na) + imputer.fit(make_df(data_na)) assert imputer.imputer_dict_ == {"Age": 82.0, "Marks": 1.8} -def test_automatically_select_variables_and_gaussian_imputation_on_left_tail(df_na): +def test_automatically_select_variables_and_gaussian_imputation_on_left_tail( + make_df, data_na +): imputer = EndTailImputer(imputation_method="gaussian", tail="left", fold=3) - imputer.fit(df_na) - imputer.imputer_dict_ = { - key: round(value, 3) for (key, value) in imputer.imputer_dict_.items() - } - assert imputer.imputer_dict_ == { - "Age": -1.521, - "Marks": 0.042, - } - - -def test_user_enters_variables_and_iqr_imputation_on_left_tail(df_na): - # test case 5: IQR + left tail + imputer.fit(make_df(data_na)) + rounded = {k: round(v, 3) for k, v in imputer.imputer_dict_.items()} + assert rounded == {"Age": -1.521, "Marks": 0.042} + + +def test_user_enters_variables_and_iqr_imputation_on_left_tail(make_df, data_na): imputer = EndTailImputer( imputation_method="iqr", tail="left", fold=1.5, variables=["Age", "Marks"] ) - imputer.fit(df_na) + imputer.fit(make_df(data_na)) assert imputer.imputer_dict_["Age"] == -6.5 assert np.round(imputer.imputer_dict_["Marks"], 3) == np.round( 0.36249999999999993, 3 ) - - -def test_error_when_imputation_method_is_not_permitted(): - with pytest.raises(ValueError): - EndTailImputer(imputation_method="arbitrary") - - -def test_error_when_tail_is_string(): - with pytest.raises(ValueError): - EndTailImputer(tail="arbitrary") - - -def test_error_when_fold_is_1(): - with pytest.raises(ValueError): - EndTailImputer(fold=-1) diff --git a/tests/test_imputation/test_mean_median_imputer.py b/tests/test_imputation/test_mean_median_imputer.py index c3603ecf7..a3ca0dc33 100644 --- a/tests/test_imputation/test_mean_median_imputer.py +++ b/tests/test_imputation/test_mean_median_imputer.py @@ -1,9 +1,9 @@ import re -import pandas as pd import pytest from feature_engine.imputation import MeanImputer, MeanMedianImputer +from tests.backend_helpers import frame_to_dict, null_count DEPRECATION_WARNING = ( "MeanMedianImputer was deprecated in favour of MeanImputer in version " @@ -27,66 +27,75 @@ def make_imputer(imputer_class, **kwargs): return imputer_class(**kwargs) -def test_mean_median_imputer_raises_future_warning(): - with pytest.warns(FutureWarning, match=re.escape(DEPRECATION_WARNING)): - MeanMedianImputer() +# init parameters +@pytest.mark.parametrize( + "imputation_method", ["arbitrary", "mode", 1, None, ("mean",), ["median"]] +) +def test_error_with_wrong_imputation_method(imputer_class, imputation_method): + msg = ( + "imputation_method takes only values 'median' or 'mean'. " + f"Got {imputation_method} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + make_imputer(imputer_class, imputation_method=imputation_method) -def test_mean_imputation_and_automatically_select_variables(df_na, imputer_class): - # set up transformer - imputer = make_imputer(imputer_class, imputation_method="mean", variables=None) - X_transformed = imputer.fit_transform(df_na) +@pytest.mark.parametrize("imputation_method", ["mean", "median"]) +def test_init_param_assignment(imputer_class, imputation_method): + imputer = make_imputer(imputer_class, imputation_method=imputation_method) + assert imputer.imputation_method == imputation_method - # set up reference result - X_reference = df_na.copy() - X_reference["Age"] = X_reference["Age"].fillna(28.714285714285715) - X_reference["Marks"] = X_reference["Marks"].fillna(0.6833333333333332) - # test init params - assert imputer.imputation_method == "mean" - assert imputer.variables is None +# fit and transform +def test_mean_imputation_and_automatically_select_variables( + make_df, data_na, imputer_class +): + imputer = make_imputer(imputer_class, imputation_method="mean", variables=None) + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Age", "Marks"] - imputer.imputer_dict_ = { + rounded_dict = { key: round(value, 3) for (key, value) in imputer.imputer_dict_.items() } - assert imputer.imputer_dict_ == { - "Age": 28.714, - "Marks": 0.683, - } - assert imputer.n_features_in_ == 6 + assert rounded_dict == {"Age": 28.714, "Marks": 0.683} + assert imputer.n_features_in_ == 5 # test transform output: # selected variables should have no NA # not selected variables should still have NA - assert X_transformed[["Age", "Marks"]].isnull().sum().sum() == 0 - assert X_transformed[["Name", "City"]].isnull().sum().sum() > 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) - - -def test_median_imputation_when_user_enters_single_variables(df_na, imputer_class): - # set up trasnformer - imputer = make_imputer(imputer_class, imputation_method="median", variables=["Age"]) - X_transformed = imputer.fit_transform(df_na) - - # set up reference output - X_reference = df_na.copy() - X_reference["Age"] = X_reference["Age"].fillna(23.0) - - # test init params - assert imputer.imputation_method == "median" - assert imputer.variables == ["Age"] + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 + assert null_count(X_transformed, "Name") > 0 + assert null_count(X_transformed, "City") > 0 + result = frame_to_dict(X_transformed) + assert result["Age"] == pytest.approx( + [20, 21, 19, 28.714285714285715, 23, 40, 41, 37] + ) + assert result["Marks"] == pytest.approx( + [0.9, 0.8, 0.7, 0.6833333333333332, 0.3, 0.6833333333333332, 0.8, 0.6] + ) + + +def test_median_imputation_when_user_enters_single_variables( + make_df, data_na, imputer_class +): + imputer = make_imputer( + imputer_class, imputation_method="median", variables=["Age"] + ) + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes - assert imputer.n_features_in_ == 6 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"Age": 23.0} # test transform output - assert X_transformed["Age"].isnull().sum() == 0 - pd.testing.assert_frame_equal(X_transformed, X_reference) + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert frame_to_dict(X_transformed)["Age"] == [20, 21, 19, 23.0, 23, 40, 41, 37] -def test_error_with_wrong_imputation_method(imputer_class): - with pytest.raises(ValueError): - make_imputer(imputer_class, imputation_method="arbitrary") +def test_mean_median_imputer_raises_future_warning(): + with pytest.warns(FutureWarning, match=re.escape(DEPRECATION_WARNING)): + MeanMedianImputer() diff --git a/tests/test_imputation/test_missing_indicator.py b/tests/test_imputation/test_missing_indicator.py index 386d3b61e..70cf224c8 100644 --- a/tests/test_imputation/test_missing_indicator.py +++ b/tests/test_imputation/test_missing_indicator.py @@ -1,49 +1,60 @@ +import re import warnings import numpy as np import pandas as pd import pytest - from sklearn.pipeline import Pipeline -from feature_engine.imputation import MissingIndicator, AddMissingIndicator +from feature_engine.imputation import AddMissingIndicator, MissingIndicator +from tests.backend_helpers import frame_to_dict + +INDICATORS = [MissingIndicator, AddMissingIndicator] + + +# init parameters +@pytest.mark.parametrize("indicator_cls", INDICATORS) +@pytest.mark.parametrize("missing_only", ["missing_only", 1, None]) +def test_error_when_missing_only_not_bool(indicator_cls, missing_only): + msg = f"missing_only takes values True or False. Got {missing_only} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + indicator_cls(missing_only=missing_only) -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) +@pytest.mark.parametrize("indicator_cls", INDICATORS) +@pytest.mark.parametrize("missing_only", [True, False]) +def test_init_param_assignment(indicator_cls, missing_only): + imputer = indicator_cls(missing_only=missing_only) + assert imputer.missing_only is missing_only + + +# fit and transform +@pytest.mark.parametrize("indicator_cls", INDICATORS) def test_detect_variables_with_missing_data_when_variables_is_none( - df_na, indicator_cls + make_df, data_na_dob, indicator_cls ): # test case 1: automatically detect variables with missing data imputer = indicator_cls(missing_only=True, variables=None) - X_transformed = imputer.fit_transform(df_na) - - # init params - assert imputer.missing_only is True - assert imputer.variables is None + X_transformed = imputer.fit_transform(make_df(data_na_dob)) # fit params assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] assert imputer.n_features_in_ == 6 # transform outputs + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (8, 11) - assert "Name_na" in X_transformed.columns - assert X_transformed["Name_na"].sum() == 2 + result = frame_to_dict(X_transformed) + assert "Name_na" in result + assert sum(result["Name_na"]) == 2 -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) +@pytest.mark.parametrize("indicator_cls", INDICATORS) def test_add_indicators_to_all_variables_when_variables_is_none( - df_na, indicator_cls + make_df, data_na_dob, indicator_cls ): imputer = indicator_cls(missing_only=False, variables=None) - - X_transformed = imputer.fit_transform(df_na) + X_transformed = imputer.fit_transform(make_df(data_na_dob)) assert imputer.variables_ == [ "Name", @@ -53,66 +64,52 @@ def test_add_indicators_to_all_variables_when_variables_is_none( "Marks", "dob", ] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (8, 12) - assert "dob_na" in X_transformed.columns - assert X_transformed["dob_na"].sum() == 0 + result = frame_to_dict(X_transformed) + assert "dob_na" in result + assert sum(result["dob_na"]) == 0 -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) -def test_add_indicators_to_one_variable(df_na, indicator_cls): +@pytest.mark.parametrize("indicator_cls", INDICATORS) +def test_add_indicators_to_one_variable(make_df, data_na_dob, indicator_cls): imputer = indicator_cls(variables="Name") - - X_transformed = imputer.fit_transform(df_na) + X_transformed = imputer.fit_transform(make_df(data_na_dob)) assert imputer.variables_ == ["Name"] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (8, 7) - assert "Name_na" in X_transformed.columns - assert X_transformed["Name_na"].sum() == 2 + result = frame_to_dict(X_transformed) + assert "Name_na" in result + assert sum(result["Name_na"]) == 2 -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) +@pytest.mark.parametrize("indicator_cls", INDICATORS) def test_detect_variables_with_missing_data_in_variables_entered_by_user( - df_na, indicator_cls + make_df, data_na_dob, indicator_cls ): imputer = indicator_cls( missing_only=True, variables=["City", "Studies", "Age", "dob"], ) + X_transformed = imputer.fit_transform(make_df(data_na_dob)) - X_transformed = imputer.fit_transform(df_na) - - assert imputer.variables == ["City", "Studies", "Age", "dob"] assert imputer.variables_ == ["City", "Studies", "Age"] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (8, 9) - assert "City_na" in X_transformed.columns - assert "dob_na" not in X_transformed.columns - assert X_transformed["City_na"].sum() == 2 + result = frame_to_dict(X_transformed) + assert "City_na" in result + assert "dob_na" not in result + assert sum(result["City_na"]) == 2 -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) -def test_error_when_missing_only_not_bool(indicator_cls): - with pytest.raises(ValueError): - indicator_cls(missing_only="missing_only") - - -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) -def test_get_feature_names_out(df_na, indicator_cls): - original_features = df_na.columns.to_list() +@pytest.mark.parametrize("indicator_cls", INDICATORS) +def test_get_feature_names_out(make_df, data_na_dob, indicator_cls): + X = make_df(data_na_dob) + original_features = list(data_na_dob) tr = indicator_cls(missing_only=False) - tr.fit(df_na) + tr.fit(X) out = [f + "_na" for f in original_features] feat_out = original_features + out @@ -121,7 +118,7 @@ def test_get_feature_names_out(df_na, indicator_cls): assert tr.get_feature_names_out(input_features=original_features) == feat_out tr = indicator_cls(missing_only=True) - tr.fit(df_na) + tr.fit(X) out = [f + "_na" for f in original_features[0:-1]] feat_out = original_features + out @@ -136,18 +133,13 @@ def test_get_feature_names_out(df_na, indicator_cls): tr.get_feature_names_out(["Name", "hola"]) -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) -def test_get_feature_names_out_from_pipeline(df_na, indicator_cls): - original_features = df_na.columns.to_list() - - tr = Pipeline( - [("transformer", indicator_cls(missing_only=False))] - ) +@pytest.mark.parametrize("indicator_cls", INDICATORS) +def test_get_feature_names_out_from_pipeline(make_df, data_na_dob, indicator_cls): + X = make_df(data_na_dob) + original_features = list(data_na_dob) - tr.fit(df_na) + tr = Pipeline([("transformer", indicator_cls(missing_only=False))]) + tr.fit(X) out = [f + "_na" for f in original_features] feat_out = original_features + out @@ -156,11 +148,9 @@ def test_get_feature_names_out_from_pipeline(df_na, indicator_cls): assert tr.get_feature_names_out(input_features=original_features) == feat_out -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) +@pytest.mark.parametrize("indicator_cls", INDICATORS) def test_no_performance_warning_with_many_variables(indicator_cls): + # pandas-only. n_cols = 101 df = pd.DataFrame( diff --git a/tests/test_imputation/test_random_sample_imputer.py b/tests/test_imputation/test_random_sample_imputer.py index cd296b7c8..feca6ca57 100644 --- a/tests/test_imputation/test_random_sample_imputer.py +++ b/tests/test_imputation/test_random_sample_imputer.py @@ -1,30 +1,119 @@ # Authors: Soledad Galli # License: BSD 3 clause +import re + import numpy as np import pandas as pd +import polars as pl import pytest from feature_engine.imputation import RandomSampleImputer -from feature_engine.imputation.random_sample import _define_seed +from feature_engine.imputation.random_sample import _hash_seeds +from tests.backend_helpers import frame_to_dict, null_count + + +# init parameters +@pytest.mark.parametrize( + "seed", ["arbitrary", "both", 1, None, ("general",), ["observation"]] +) +def test_error_if_seed_not_permitted_value(seed): + msg = f"seed takes only values 'general' or 'observation'. Got {seed} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + RandomSampleImputer(seed=seed) + + +@pytest.mark.parametrize("random_state", ["arbitrary", 0.5, ["Age"]]) +def test_error_if_random_state_not_integer_when_seed_is_general(random_state): + msg = ( + "if seed == 'general' then random_state must take an integer. " + f"Got {random_state} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + RandomSampleImputer(seed="general", random_state=random_state) + + +@pytest.mark.parametrize("random_state", [None, [], ""]) +def test_error_if_random_state_is_empty_when_seed_is_observation(random_state): + msg = ( + "if seed == 'observation' the random state must take the name of one " + "or more variables which will be used to seed the imputer. " + f"Got {random_state} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + RandomSampleImputer(seed="observation", random_state=random_state) + + +@pytest.mark.parametrize( + "random_state, seed", + [ + (None, "general"), + (5, "general"), + ("Age", "observation"), + (["Age", "Marks"], "observation"), + ], +) +def test_init_param_assignment(random_state, seed): + imputer = RandomSampleImputer(random_state=random_state, seed=seed) + assert imputer.random_state == random_state + assert imputer.seed == seed + + +# fit and transform +def test_hash_seeds(): + values = np.array( + [ + [25, 0.7], + [25.0, 0.7], + [0.0, 0.7], + [np.nan, 0.7], + [-0.0, 0.7], + [-30.0, 1e20], + ] + ) + seeds = _hash_seeds(values) + # same values, same seed: ints and floats are equal, nan and -0.0 count as 0 + assert seeds[0] == seeds[1] + assert seeds[2] == seeds[3] == seeds[4] + assert seeds[0] != seeds[2] + # negative and large values give valid numpy seeds + assert all(0 <= seed < 2**32 for seed in seeds) + # the seed must not change between sessions or releases + assert _hash_seeds(np.array([[25.0, 0.7]]))[0] == 2067629302 -def test_define_seed(df_vartypes): - assert _define_seed(df_vartypes, 0, ["Age", "Marks"], how="add") == 21 - assert _define_seed(df_vartypes, 0, ["Age", "Marks"], how="multiply") == 18 - assert _define_seed(df_vartypes, 2, ["Age", "Marks"], how="add") == 20 - assert _define_seed(df_vartypes, 2, ["Age", "Marks"], how="multiply") == 13 - assert _define_seed(df_vartypes, 1, ["Age"], how="add") == 21 - assert _define_seed(df_vartypes, 3, ["Marks"], how="multiply") == 1 +def test_general_seed_plus_automatically_select_variables(make_df, data_na): + df_na = make_df(data_na) + imputer = RandomSampleImputer(variables=None, random_state=5, seed="general") + X_transformed = imputer.fit_transform(df_na) -def test_general_seed_plus_automatically_select_variables(df_na): - # set up transformer + # test fit attrs + assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] + assert imputer.n_features_in_ == 5 + assert frame_to_dict(imputer.X_) == frame_to_dict(df_na) + + # no missing data left in any imputed variable, and every value used to + # fill NA came from the training data itself + assert isinstance(X_transformed, make_df) + result = frame_to_dict(X_transformed) + for col in imputer.variables_: + assert null_count(X_transformed, col) == 0 + assert set(result[col]) <= {v for v in data_na[col] if v is not None} + + # pandas and polars draw different values for the same seed, so we only check + # that the same seed on the same backend gives the same result. + imputer2 = RandomSampleImputer(variables=None, random_state=5, seed="general") + X_transformed2 = imputer2.fit_transform(df_na) + assert frame_to_dict(X_transformed) == frame_to_dict(X_transformed2) + + +def test_pandas_general_seed_reproduces_historic_values(df_na): + # pandas only: with a fixed seed, pandas must return the same values as before + # the narwhals migration. polars uses a different random number generator. imputer = RandomSampleImputer(variables=None, random_state=5, seed="general") X_transformed = imputer.fit_transform(df_na) - # expected output: - # fillna based on seed used (found experimenting on Jupyter notebook) ref = { "Name": ["tom", "nick", "krish", "peter", "peter", "sam", "fred", "sam"], "City": [ @@ -53,257 +142,172 @@ def test_general_seed_plus_automatically_select_variables(df_na): } ref = pd.DataFrame(ref) - # test init params - assert imputer.variables is None - assert imputer.random_state == 5 - assert imputer.seed == "general" + pd.testing.assert_frame_equal(X_transformed, ref, check_dtype=False) - # test fit attr - assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks", "dob"] - assert imputer.n_features_in_ == 6 - pd.testing.assert_frame_equal(imputer.X_, df_na) - # test transform output - pd.testing.assert_frame_equal(X_transformed, ref, check_dtype=False) +def _data_without_na_in(data, columns): + # the variables used as seed should not have missing data + data = dict(data) + for col in columns: + data[col] = [v if v is not None else 1 for v in data[col]] + return data -def test_seed_per_observation_and_multiple_variables_in_random_state(df_na): - # test case 2: imputer seed per observation using multiple variables to determine - # the random_state - # Note the variables used as seed should not have missing data, this I fill - df_na = df_na.copy() - df_na[["Marks", "Age"]] = df_na[["Marks", "Age"]].fillna(1) +@pytest.mark.parametrize("random_state", [["Marks", "Age"], "Age"]) +def test_seed_per_observation(make_df, data_na, random_state): + seed_vars = [random_state] if isinstance(random_state, str) else random_state + data = _data_without_na_in(data_na, seed_vars) + df_na = make_df(data) imputer = RandomSampleImputer( - variables=["City", "Studies"], random_state=["Marks", "Age"], seed="observation" + variables=["City", "Studies"], + random_state=random_state, + seed="observation", ) - X_transformed = imputer.fit_transform(df_na) - # expected output - ref = { - "Name": ["tom", "nick", "krish", np.nan, "peter", np.nan, "fred", "sam"], - "City": [ - "London", - "Manchester", - "London", - "London", - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - "PhD", - "Bachelor", - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, np.nan, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, np.nan, 0.3, np.nan, 0.8, 0.6], - "dob": pd.date_range("2020-02-24", periods=8, freq="min"), - } - ref = pd.DataFrame(ref) - - assert imputer.variables == ["City", "Studies"] - assert imputer.random_state == ["Marks", "Age"] - assert imputer.seed == "observation" - pd.testing.assert_frame_equal( - imputer.X_[["City", "Studies"]], df_na[["City", "Studies"]] - ) - - pd.testing.assert_frame_equal( - X_transformed[["City", "Studies"]], ref[["City", "Studies"]] - ) - - -def test_seed_per_observation_plus_product_of_seeding_variables(df_na): - # test case 3: observation seed, 2 variables as seed, product of seed variables - # need to fill variables used as seed - df_na = df_na.copy() - df_na[["Marks", "Age"]] = df_na[["Marks", "Age"]].fillna(1) - - imputer = RandomSampleImputer( + # fit() turns a single seeding variable name into a list + assert imputer.random_state == seed_vars + assert isinstance(X_transformed, make_df) + result = frame_to_dict(X_transformed) + for col in ["City", "Studies"]: + assert frame_to_dict(imputer.X_)[col] == data[col] + assert null_count(X_transformed, col) == 0 + assert set(result[col]) <= {v for v in data[col] if v is not None} + # variables not selected for imputation are untouched + assert result["Age"] == data["Age"] + + # same seed, same backend -> same result + imputer2 = RandomSampleImputer( variables=["City", "Studies"], - random_state=["Marks", "Age"], + random_state=random_state, seed="observation", - seeding_method="multiply", ) + X_transformed2 = imputer2.fit_transform(df_na) + assert frame_to_dict(X_transformed) == frame_to_dict(X_transformed2) - X_transformed = imputer.fit_transform(df_na) - - # expected output - ref = { - "Name": ["tom", "nick", "krish", np.nan, "peter", np.nan, "fred", "sam"], - "City": [ - "London", - "Manchester", - "London", - "Manchester", - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - "Bachelor", - "Masters", - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, np.nan, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, np.nan, 0.3, np.nan, 0.8, 0.6], - "dob": pd.date_range("2020-02-24", periods=8, freq="min"), - } - ref = pd.DataFrame(ref) - assert imputer.variables == ["City", "Studies"] - assert imputer.random_state == ["Marks", "Age"] - assert imputer.seed == "observation" +DATA_SEED_TRAIN = { + "City": ["London", "Manchester", "Bristol", "Leeds", "York", "Bath", "Hull"], + "Age": [20.0, 21.0, 19.0, 23.0, 40.0, 41.0, 37.0], + "Marks": [0.9, 0.8, 0.7, 0.3, 0.6, 0.8, 0.5], +} +# rows 0 and 3 have identical seeding values and City missing +DATA_SEED_TEST = { + "City": [None, "Leeds", None, None, None], + "Age": [25.0, 30.0, 40.0, 25.0, 33.0], + "Marks": [0.7, 0.4, 0.6, 0.7, 0.2], +} - pd.testing.assert_frame_equal( - imputer.X_[["City", "Studies"]], df_na[["City", "Studies"]] - ) - pd.testing.assert_frame_equal( - X_transformed[["City", "Studies"]], - ref[["City", "Studies"]], - check_dtype=False, +def test_seed_per_observation_imputes_identical_rows_equally(make_df): + imputer = RandomSampleImputer( + variables=["City"], random_state=["Age", "Marks"], seed="observation" ) + imputer.fit(make_df(DATA_SEED_TRAIN)) + X_transformed = imputer.transform(make_df(DATA_SEED_TEST)) + city = frame_to_dict(X_transformed)["City"] + assert city[0] == city[3] -def test_seed_per_observation_with_only_1_variable_as_seed(df_na): - # test case 4: observation seed, only variable indicated as seed, method: addition - # Note the variable used as seed should not have missing data - df_na = df_na.copy() - df_na["Age"] = df_na["Age"].fillna(1) +def test_seed_per_observation_does_not_depend_on_row_position(make_df): imputer = RandomSampleImputer( - variables=["City", "Studies"], random_state="Age", seed="observation" + variables=["City"], random_state=["Age", "Marks"], seed="observation" ) + imputer.fit(make_df(DATA_SEED_TRAIN)) + X = make_df(DATA_SEED_TEST) + expected = frame_to_dict(imputer.transform(X))["City"] - X_transformed = imputer.fit_transform(df_na) + # same rows in reverse order + X_reversed = make_df({k: v[::-1] for k, v in DATA_SEED_TEST.items()}) + reversed_city = frame_to_dict(imputer.transform(X_reversed))["City"] + assert reversed_city == expected[::-1] - # expected output - ref = { - "Name": ["tom", "nick", "krish", np.nan, "peter", np.nan, "fred", "sam"], - "City": [ - "London", - "Manchester", - "Manchester", - "Manchester", - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - "Masters", - "Masters", - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, np.nan, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, np.nan, 0.3, np.nan, 0.8, 0.6], - "dob": pd.date_range("2020-02-24", periods=8, freq="min"), - } - ref = pd.DataFrame(ref) + # each row imputed on its own + for i in range(len(expected)): + row_city = frame_to_dict(imputer.transform(X[i:i + 1]))["City"] + assert row_city == [expected[i]] - assert imputer.random_state == ["Age"] - pd.testing.assert_frame_equal( - imputer.X_[["City", "Studies"]], df_na[["City", "Studies"]] +def test_seed_per_observation_with_negative_and_large_seeding_values(make_df): + imputer = RandomSampleImputer( + variables=["City"], random_state=["Age", "Marks"], seed="observation" ) - - pd.testing.assert_frame_equal( - X_transformed[["City", "Studies"]], - ref[["City", "Studies"]], - check_dtype=False, + imputer.fit(make_df(DATA_SEED_TRAIN)) + X = make_df( + { + "City": [None, None, "Leeds"], + "Age": [-30.0, 1e20, 20.0], + "Marks": [0.1, 1e20, 0.9], + } ) + X_transformed = imputer.transform(X) - -def test_error_if_seed_not_permitted_value(): - with pytest.raises(ValueError): - RandomSampleImputer(seed="arbitrary") + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "City") == 0 + assert set(frame_to_dict(X_transformed)["City"]) <= set(DATA_SEED_TRAIN["City"]) -def test_error_if_seeding_method_not_permitted_value(): - with pytest.raises(ValueError): - RandomSampleImputer(seeding_method="arbitrary") - +def test_seed_per_observation_uses_values_before_imputation_with_missing_as_zero( + make_df, +): + # Age is imputed and also seeds City: row 0 (Age missing) must seed like + # row 1 (Age 0), not with its imputed Age. + imputer = RandomSampleImputer( + variables=["Age", "City"], random_state=["Age", "Marks"], seed="observation" + ) + imputer.fit(make_df(DATA_SEED_TRAIN)) + X = make_df( + { + "City": [None, None, "Leeds"], + "Age": [None, 0.0, 30.0], + "Marks": [0.7, 0.7, 0.4], + } + ) + X_transformed = imputer.transform(X) -def test_error_if_random_state_takes_not_permitted_value(): - with pytest.raises(ValueError): - RandomSampleImputer(seed="general", random_state="arbitrary") + result = frame_to_dict(X_transformed) + assert null_count(X_transformed, "Age") == 0 + assert result["City"][0] == result["City"][1] -def test_error_if_random_state_is_none_when_seed_is_observation(): - with pytest.raises(ValueError): - RandomSampleImputer(seed="observation", random_state=None) +def test_seed_per_observation_with_duplicated_index(): + # pandas only: polars has no index + imputer = RandomSampleImputer( + variables=["City"], random_state=["Age", "Marks"], seed="observation" + ) + imputer.fit(pd.DataFrame(DATA_SEED_TRAIN)) + X = pd.DataFrame(DATA_SEED_TEST) + expected = frame_to_dict(imputer.transform(X))["City"] + X.index = [0, 0, 1, 1, 2] + assert frame_to_dict(imputer.transform(X))["City"] == expected -def test_error_if_random_state_is_string(df_na): - with pytest.raises(ValueError): - imputer = RandomSampleImputer(seed="observation", random_state="arbitrary") - imputer.fit(df_na) +def test_error_if_random_state_variables_not_in_dataframe(make_df, data_na): + imputer = RandomSampleImputer(seed="observation", random_state="arbitrary") + msg = ( + "There are variables assigned as random state which are not part " + "of the training dataframe. Got arbitrary instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + imputer.fit(make_df(data_na)) -def test_variables_cast_as_category(df_na): - df_na = df_na.copy() - df_na["City"] = df_na["City"].astype("category") +def test_variables_cast_as_category(make_df, data_na): + df_na = make_df(data_na) + if make_df is pd.DataFrame: + df_na["City"] = df_na["City"].astype("category") + else: + df_na = df_na.with_columns(pl.col("City").cast(pl.Categorical)) - # set up transformer imputer = RandomSampleImputer(variables=None, random_state=5, seed="general") X_transformed = imputer.fit_transform(df_na) - # expected output: - # fillna based on seed used (found experimenting on Jupyter notebook) - ref = { - "Name": ["tom", "nick", "krish", "peter", "peter", "sam", "fred", "sam"], - "City": [ - "London", - "Manchester", - "London", - "Manchester", - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - "PhD", - "Masters", - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, 23, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, 0.3, 0.3, 0.6, 0.8, 0.6], - "dob": pd.date_range("2020-02-24", periods=8, freq="min"), - } - ref = pd.DataFrame(ref) - ref["City"] = ref["City"].astype("category") - - # test fit attr - assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks", "dob"] - assert imputer.n_features_in_ == 6 - pd.testing.assert_frame_equal(imputer.X_, df_na) - - # test transform output - pd.testing.assert_frame_equal(X_transformed, ref, check_dtype=False) + assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] + assert imputer.n_features_in_ == 5 + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "City") == 0 + city_pool = {v for v in data_na["City"] if v is not None} + assert set(frame_to_dict(X_transformed)["City"]) <= city_pool diff --git a/tests/test_outliers/conftest.py b/tests/test_outliers/conftest.py new file mode 100644 index 000000000..a00ac097f --- /dev/null +++ b/tests/test_outliers/conftest.py @@ -0,0 +1,45 @@ +"""Data shared by the outlier transformer tests. + +Each fixture returns a fresh dict, so tests can build the dataframe on the +backend under test with ``make_df(data)``. Missing values are written as None, +which both pandas and polars read as missing. +""" + +import numpy as np +import pytest + + +@pytest.fixture +def data_normal_dist(): + # same seed and parameters as the pandas df_normal_dist fixture in + # tests/conftest.py + return {"var": np.random.RandomState(0).normal(0, 0.1, 100).tolist()} + + +@pytest.fixture +def data_na(): + return { + "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], + "City": [ + "London", + "Manchester", + None, + None, + "London", + "London", + "Bristol", + "Manchester", + ], + "Studies": [ + "Bachelor", + "Bachelor", + None, + None, + "Bachelor", + "PhD", + "None", + "Masters", + ], + "Age": [20, 21, 19, None, 23, 40, 41, 37], + "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], + } diff --git a/tests/test_outliers/test_arbitrary_capper.py b/tests/test_outliers/test_arbitrary_capper.py index 5cba357b8..641daa5ce 100644 --- a/tests/test_outliers/test_arbitrary_capper.py +++ b/tests/test_outliers/test_arbitrary_capper.py @@ -1,176 +1,165 @@ +import re + import numpy as np -import pandas as pd import pytest from feature_engine.outliers import ArbitraryOutlierCapper +from tests.backend_helpers import frame_to_dict +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) -def test_right_end_capping(df_normal_dist): - # test case 1: right end capping - transformer = ArbitraryOutlierCapper( - max_capping_dict={"var": 0.10727677848029868}, min_capping_dict=None - ) - X = transformer.fit_transform(df_normal_dist) - - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = np.where( - df_transf["var"] > 0.10727677848029868, 0.10727677848029868, df_transf["var"] - ) - # test init params - assert np.round(transformer.max_capping_dict["var"], 3) == np.round( - 0.10727677848029868, 3 - ) - assert transformer.min_capping_dict is None - assert transformer.variables_ == ["var"] - # test fit attrs - assert np.round(transformer.right_tail_caps_["var"], 3) == np.round( - 0.10727677848029868, 3 - ) - assert transformer.left_tail_caps_ == {} - assert transformer.n_features_in_ == 1 - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert np.round(X["var"].max(), 3) <= np.round(0.10727677848029868, 3) - assert np.round(df_normal_dist["var"].max(), 3) > np.round(0.10727677848029868, 3) +# init parameters +@pytest.mark.parametrize("param", ["max_capping_dict", "min_capping_dict"]) +@pytest.mark.parametrize("value", ["other", 1, ["var"], ("var", 1)]) +def test_error_if_capping_dict_not_dict(param, value): + msg = f"The parameter can only take a dictionary or None. Got {value} instead." + with pytest.raises(TypeError, match=re.escape(msg)): + ArbitraryOutlierCapper(**{param: value}) -def test_both_ends_capping(df_normal_dist): - # test case 2: both tails - transformer = ArbitraryOutlierCapper( - max_capping_dict={"var": 0.20857275540714884}, - min_capping_dict={"var": -0.19661115230025186}, +@pytest.mark.parametrize("param", ["max_capping_dict", "min_capping_dict"]) +@pytest.mark.parametrize("value", [{"var": "a"}, {"var": None}, {"a": 1, "b": [2]}]) +def test_error_if_capping_dict_values_not_numerical(param, value): + msg = ( + "All values in the dictionary must be integer or float. " + f"Got {value} instead." ) - X = transformer.fit_transform(df_normal_dist) + with pytest.raises(ValueError, match=re.escape(msg)): + ArbitraryOutlierCapper(**{param: value}) - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = np.where( - df_transf["var"] > 0.20857275540714884, 0.20857275540714884, df_transf["var"] - ) - df_transf["var"] = np.where( - df_transf["var"] < -0.19661115230025186, -0.19661115230025186, df_transf["var"] - ) - # test fit params - assert np.round(transformer.right_tail_caps_["var"], 3) == np.round( - 0.20857275540714884, 3 - ) - assert np.round(transformer.left_tail_caps_["var"], 3) == np.round( - -0.19661115230025186, 3 - ) - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert np.round(X["var"].max(), 3) <= np.round(0.20857275540714884, 3) - assert np.round(X["var"].min(), 3) >= np.round(-0.19661115230025186, 3) - assert np.round(df_normal_dist["var"].max(), 3) > np.round(0.20857275540714884, 3) - assert np.round(df_normal_dist["var"].min(), 3) < np.round(-0.19661115230025186, 3) +@pytest.mark.parametrize( + "max_capping_dict, min_capping_dict", + [(None, None), ({}, None), (None, {}), ({}, {})], +) +def test_error_if_no_capping_values(max_capping_dict, min_capping_dict): + msg = "Please provide at least 1 dictionary with the capping values." + with pytest.raises(ValueError, match=re.escape(msg)): + ArbitraryOutlierCapper( + max_capping_dict=max_capping_dict, min_capping_dict=min_capping_dict + ) -def test_left_tail_capping(df_normal_dist): - # test case 3: left tail - transformer = ArbitraryOutlierCapper( - max_capping_dict=None, min_capping_dict={"var": -0.17486039103044} +@pytest.mark.parametrize( + "missing_values", ["HOLA", "Raise", 1, True, None, ["raise"], {"key": "raise"}] +) +def test_error_if_missing_values_not_permitted(missing_values): + msg = ( + "missing_values must be 'raise' or 'ignore'. " + f"Got {missing_values} instead." ) - X = transformer.fit_transform(df_normal_dist) + with pytest.raises(ValueError, match=re.escape(msg)): + ArbitraryOutlierCapper( + min_capping_dict={"var": -0.15}, missing_values=missing_values + ) - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = np.where( - df_transf["var"] < -0.17486039103044, -0.17486039103044, df_transf["var"] - ) - # test init param - assert transformer.max_capping_dict is None - assert np.round(transformer.min_capping_dict["var"], 3) == np.round( - -0.17486039103044, 3 - ) - # test fit attr - assert transformer.right_tail_caps_ == {} - assert np.round(transformer.left_tail_caps_["var"], 3) == np.round( - -0.17486039103044, 3 +@pytest.mark.parametrize( + "max_capping_dict, min_capping_dict, missing_values", + [ + ({"var": 0.1}, None, "raise"), + (None, {"var": -0.15}, "ignore"), + ({"var": 0.1}, {"var": -0.15, "other": 2}, "raise"), + ({"var": 1}, {}, "ignore"), + ], +) +def test_init_param_assignment(max_capping_dict, min_capping_dict, missing_values): + transformer = ArbitraryOutlierCapper( + max_capping_dict=max_capping_dict, + min_capping_dict=min_capping_dict, + missing_values=missing_values, ) - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert np.round(X["var"].min(), 3) >= np.round(-0.17486039103044, 3) - assert np.round(df_normal_dist["var"].min(), 3) < np.round(-0.17486039103044, 3) + assert transformer.max_capping_dict == max_capping_dict + assert transformer.min_capping_dict == min_capping_dict + assert transformer.missing_values == missing_values -def test_ignores_na_in_input_df(df_na): - # test case 4: dataset contains na and transformer is asked to ignore them +# fit and transform +@pytest.mark.parametrize( + "max_capping_dict, min_capping_dict, right_tail_caps, left_tail_caps", + [ + ({"var": 0.1}, None, {"var": 0.1}, {}), + (None, {"var": -0.15}, {}, {"var": -0.15}), + ({"var": 0.1}, {"var": -0.15}, {"var": 0.1}, {"var": -0.15}), + ], +) +def test_capping( + make_df, + data_normal_dist, + max_capping_dict, + min_capping_dict, + right_tail_caps, + left_tail_caps, +): transformer = ArbitraryOutlierCapper( - max_capping_dict=None, min_capping_dict={"Age": 20}, missing_values="ignore" + max_capping_dict=max_capping_dict, min_capping_dict=min_capping_dict ) - X = transformer.fit_transform(df_na) + Xt = transformer.fit_transform(make_df(data_normal_dist)) - # expected output - df_transf = df_na.copy() - df_transf["Age"] = np.where(df_transf["Age"] < 20, 20, df_transf["Age"]) + # a tail without a limit is not capped + upper = right_tail_caps.get("var", np.inf) + lower = left_tail_caps.get("var", -np.inf) + expected = np.clip(data_normal_dist["var"], lower, upper).tolist() - # test fit params - assert transformer.max_capping_dict is None - assert transformer.min_capping_dict == {"Age": 20} - assert transformer.n_features_in_ == 6 - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert X["Age"].min() >= 20 - assert df_na["Age"].min() < 20 + assert transformer.right_tail_caps_ == right_tail_caps + assert transformer.left_tail_caps_ == left_tail_caps + assert transformer.variables_ == ["var"] + assert transformer.feature_names_in_ == ["var"] + assert transformer.n_features_in_ == 1 + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"var": pytest.approx(expected)} -def test_error_if_max_capping_dict_wrong_input(): - with pytest.raises(TypeError): - ArbitraryOutlierCapper(max_capping_dict="other") - with pytest.raises(ValueError): - ArbitraryOutlierCapper(max_capping_dict={"a": "a"}) +def test_variables_are_taken_from_both_dicts(make_df): + X = make_df({"a": [0, 5, 10], "b": [0, 5, 10], "c": [0, 5, 10]}) + transformer = ArbitraryOutlierCapper( + max_capping_dict={"a": 8}, min_capping_dict={"b": 2, "a": 1} + ) + Xt = transformer.fit_transform(X) + assert transformer.variables_ == ["b", "a"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"a": [1, 5, 8], "b": [2, 5, 10], "c": [0, 5, 10]} -def test_error_if_min_capping_dict_wrong_input(): - with pytest.raises(TypeError): - ArbitraryOutlierCapper(min_capping_dict="other") - with pytest.raises(ValueError): - ArbitraryOutlierCapper(min_capping_dict={"a": "a"}) +def test_empty_dict_is_ignored(make_df): + X = make_df({"a": [0, 5, 10], "b": [0, 5, 10]}) + transformer = ArbitraryOutlierCapper(max_capping_dict={"a": 8}, min_capping_dict={}) + Xt = transformer.fit_transform(X) -def test_error_if_both_capping_dicts_are_none(): - with pytest.raises(ValueError): - ArbitraryOutlierCapper(min_capping_dict=None, max_capping_dict=None) + assert transformer.variables_ == ["a"] + assert transformer.left_tail_caps_ == {} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"a": [0, 5, 8], "b": [0, 5, 10]} -def test_error_if_missing_values_not_bool(): - with pytest.raises(ValueError): - ArbitraryOutlierCapper(missing_values="other") +def test_ignores_na_in_input_df(make_df, data_na): + transformer = ArbitraryOutlierCapper( + min_capping_dict={"Age": 21}, missing_values="ignore" + ) + Xt = transformer.fit_transform(make_df(data_na)) + expected = [None if v is None else max(v, 21) for v in data_na["Age"]] -def test_fit_and_transform_raise_error_if_df_contains_na(df_normal_dist): - df_na = df_normal_dist.copy() - df_na.loc[1, "var"] = np.nan + assert transformer.n_features_in_ == 5 + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt)["Age"] == expected - # test case 5: when dataset contains na, fit method - with pytest.raises(ValueError): - transformer = ArbitraryOutlierCapper( - min_capping_dict={"var": -0.17486039103044} - ) - transformer.fit(df_na) - # test case 6: when dataset contains na, transform method - with pytest.raises(ValueError): - transformer = ArbitraryOutlierCapper( - min_capping_dict={"var": -0.17486039103044} - ) - transformer.fit(df_normal_dist) - transformer.transform(df_na) +def test_fit_raises_error_if_df_contains_na(make_df, data_na): + transformer = ArbitraryOutlierCapper(min_capping_dict={"Age": 21}) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.fit(make_df(data_na)) -@pytest.mark.parametrize( - "missing_values", - ["HOLA", 1, True, {"key1": "value1", "key2": "value2", "key3": "value3"}], -) -def test_error_if_missing_values_wrong_type(missing_values): - msg = "missing_values takes only values 'raise' or 'ignore'" - with pytest.raises(ValueError) as record: - ArbitraryOutlierCapper( - min_capping_dict={"var": -0.17486039103044}, missing_values="missing_values" - ) - # check that error message matches - assert str(record.value) == msg +def test_transform_raises_error_if_df_contains_na(make_df, data_normal_dist): + data_na = {"var": list(data_normal_dist["var"])} + data_na["var"][1] = None + transformer = ArbitraryOutlierCapper(min_capping_dict={"var": -0.15}) + transformer.fit(make_df(data_normal_dist)) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(make_df(data_na)) diff --git a/tests/test_outliers/test_base_outlier.py b/tests/test_outliers/test_base_outlier.py new file mode 100644 index 000000000..58923aadb --- /dev/null +++ b/tests/test_outliers/test_base_outlier.py @@ -0,0 +1,285 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.exceptions import NotFittedError + +from feature_engine.outliers.base_outlier import BaseOutlier, WinsorizerBase +from tests.backend_helpers import frame_to_dict + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) + + +# init parameters +@pytest.mark.parametrize( + "capping_method", ["arbitrary", "Gaussian", "", 1, None, ["iqr"]] +) +def test_error_if_capping_method_not_permitted(capping_method): + msg = ( + "capping_method must be 'gaussian', 'iqr', 'mad', 'quantiles'. " + f"Got {capping_method} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + WinsorizerBase(capping_method=capping_method) + + +@pytest.mark.parametrize("tail", ["other", "Right", "", 1, None, ["right"]]) +def test_error_if_tail_not_permitted(tail): + msg = f"tail must be 'right', 'left' or 'both'. Got {tail} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + WinsorizerBase(tail=tail) + + +@pytest.mark.parametrize("fold", ["other", "Auto", 0, -1, -0.5]) +def test_error_if_fold_not_permitted(fold): + msg = f"fold must be a positive number or 'auto'. Got {fold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + WinsorizerBase(fold=fold) + + +@pytest.mark.parametrize("fold", [0.3, 1, 5]) +def test_error_if_fold_above_0_2_with_quantiles(fold): + msg = ( + "with capping_method ='quantiles', fold takes values between 0 and " + "0.20 only." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + WinsorizerBase(capping_method="quantiles", fold=fold) + + +@pytest.mark.parametrize("missing_values", ["other", "Raise", 1, True, None]) +def test_error_if_missing_values_not_permitted(missing_values): + msg = ( + "missing_values must be 'raise' or 'ignore'. " + f"Got {missing_values} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + WinsorizerBase(missing_values=missing_values) + + +@pytest.mark.parametrize( + "capping_method, tail, fold, missing_values", + [ + ("gaussian", "right", "auto", "raise"), + ("iqr", "left", 2, "ignore"), + ("mad", "both", 1.5, "raise"), + ("quantiles", "both", 0.1, "ignore"), + ], +) +def test_init_param_assignment(capping_method, tail, fold, missing_values): + transformer = WinsorizerBase( + capping_method=capping_method, + tail=tail, + fold=fold, + missing_values=missing_values, + ) + assert transformer.capping_method == capping_method + assert transformer.tail == tail + assert transformer.fold == fold + assert transformer.missing_values == missing_values + + +# fit and transform +def _expected_caps(values, capping_method, fold): + # reference limits computed with pandas + s = pd.Series(values) + if capping_method == "gaussian": + return s.mean() + fold * s.std(ddof=0), s.mean() - fold * s.std(ddof=0) + if capping_method == "iqr": + iqr = s.quantile(0.75) - s.quantile(0.25) + return s.quantile(0.75) + fold * iqr, s.quantile(0.25) - fold * iqr + if capping_method == "mad": + mad = (s - s.median()).abs().median() / 0.67449 + return s.median() + fold * mad, s.median() - fold * mad + return s.quantile(1 - fold), s.quantile(fold) + + +@pytest.mark.parametrize( + "capping_method, fold", + [("gaussian", 3), ("gaussian", 1), ("iqr", 1.5), ("mad", 2), ("quantiles", 0.1)], +) +def test_fit_learns_caps(make_df, data_normal_dist, capping_method, fold): + transformer = WinsorizerBase(capping_method=capping_method, tail="both", fold=fold) + transformer.fit(make_df(data_normal_dist)) + + right, left = _expected_caps(data_normal_dist["var"], capping_method, fold) + assert transformer.right_tail_caps_ == {"var": pytest.approx(right)} + assert transformer.left_tail_caps_ == {"var": pytest.approx(left)} + assert transformer.variables_ == ["var"] + assert transformer.feature_names_in_ == ["var"] + assert transformer.n_features_in_ == 1 + + +@pytest.mark.parametrize("tail", ["right", "left"]) +def test_fit_learns_caps_for_one_tail(make_df, data_normal_dist, tail): + transformer = WinsorizerBase(tail=tail, fold=3).fit(make_df(data_normal_dist)) + + right, left = _expected_caps(data_normal_dist["var"], "gaussian", 3) + if tail == "right": + assert transformer.right_tail_caps_ == {"var": pytest.approx(right)} + assert transformer.left_tail_caps_ == {} + else: + assert transformer.left_tail_caps_ == {"var": pytest.approx(left)} + assert transformer.right_tail_caps_ == {} + + +@pytest.mark.parametrize( + "capping_method, expected", + [("gaussian", 3.0), ("iqr", 1.5), ("mad", 3.29), ("quantiles", 0.05)], +) +def test_auto_fold(make_df, data_normal_dist, capping_method, expected): + transformer = WinsorizerBase(capping_method=capping_method, fold="auto") + transformer.fit(make_df(data_normal_dist)) + assert transformer.fold_ == expected + + +def test_fold_is_kept_when_given(make_df, data_normal_dist): + transformer = WinsorizerBase(fold=2.5).fit(make_df(data_normal_dist)) + assert transformer.fold_ == 2.5 + + +def test_fit_selects_numerical_variables_and_ignores_na(make_df, data_na): + transformer = WinsorizerBase(tail="both", fold=1, missing_values="ignore") + transformer.fit(make_df(data_na)) + + assert transformer.variables_ == ["Age", "Marks"] + assert transformer.feature_names_in_ == list(data_na) + assert transformer.n_features_in_ == 5 + # missing values are skipped when learning the caps + for var in ["Age", "Marks"]: + values = [v for v in data_na[var] if v is not None] + right, left = _expected_caps(values, "gaussian", 1) + assert transformer.right_tail_caps_[var] == pytest.approx(right) + assert transformer.left_tail_caps_[var] == pytest.approx(left) + + +def test_fit_raises_error_if_na(make_df, data_na): + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + WinsorizerBase().fit(make_df(data_na)) + + +@pytest.mark.parametrize("capping_method", ["gaussian", "iqr", "mad", "quantiles"]) +@pytest.mark.parametrize("tail", ["right", "left", "both"]) +def test_variables_without_variation_get_infinite_caps(make_df, capping_method, tail): + X = make_df({"var": [1.0] * 10, "other": [float(v) for v in range(10)]}) + transformer = WinsorizerBase(capping_method=capping_method, tail=tail) + transformer.fit(X) + + if tail in ("right", "both"): + assert transformer.right_tail_caps_["var"] == np.inf + assert np.isfinite(transformer.right_tail_caps_["other"]) + if tail in ("left", "both"): + assert transformer.left_tail_caps_["var"] == -np.inf + assert np.isfinite(transformer.left_tail_caps_["other"]) + + +def test_fit_with_integer_column_names(data_normal_dist): + # integer column names are pandas-only + X = pd.DataFrame({0: data_normal_dist["var"]}) + transformer = WinsorizerBase(tail="both", fold=3).fit(X) + + right, left = _expected_caps(data_normal_dist["var"], "gaussian", 3) + assert transformer.right_tail_caps_ == {0: pytest.approx(right)} + assert transformer.left_tail_caps_ == {0: pytest.approx(left)} + + +class MockCapper(BaseOutlier): + # caps are set by hand to test the shared transform logic + def __init__(self, missing_values="raise"): + self.missing_values = missing_values + + def fit(self, X, y=None): + self.variables_ = ["a", "b", "c"] + self.right_tail_caps_ = {"a": 2, "b": 2.5} + self.left_tail_caps_ = {"a": 0, "c": 1} + self.feature_names_in_ = list(X.columns) + self.n_features_in_ = X.shape[1] + return self + + def transform(self, X): + return self._transform(X) + + +DATA_CAP = { + "a": [-1.0, 1.0, 3.0], + "b": [1.0, 2.0, 3.0], + "c": [0, 1, 2], + "d": ["x", "y", "z"], +} + + +def test_transform_caps_values(make_df): + X = make_df(DATA_CAP) + Xt = MockCapper().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "a": [0.0, 1.0, 2.0], + "b": [1.0, 2.0, 2.5], + "c": [1, 1, 2], + "d": ["x", "y", "z"], + } + # capping with a left bound only keeps integer columns as integers + assert nw.from_native(Xt, eager_only=True)["c"].dtype.is_integer() + + +def test_transform_reorders_columns_to_match_fit(make_df): + transformer = MockCapper().fit(make_df(DATA_CAP)) + reordered = make_df({k: DATA_CAP[k] for k in ["d", "c", "b", "a"]}) + + Xt = transformer.transform(reordered) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["a", "b", "c", "d"] + + +def test_transform_raises_error_if_different_number_of_columns(make_df): + transformer = MockCapper().fit(make_df(DATA_CAP)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.transform(make_df({k: DATA_CAP[k] for k in ["a", "b", "c"]})) + + +def test_transform_raises_error_if_na(make_df): + transformer = MockCapper().fit(make_df(DATA_CAP)) + X_na = make_df({**DATA_CAP, "a": [-1.0, None, 3.0]}) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(X_na) + + +def test_transform_keeps_na_when_ignored(make_df): + X_na = make_df({**DATA_CAP, "a": [-1.0, None, 3.0]}) + Xt = MockCapper(missing_values="ignore").fit(X_na).transform(X_na) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt)["a"] == [0.0, None, 2.0] + + +def test_transform_raises_non_fitted_error(make_df): + msg = ( + "This MockCapper instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockCapper().transform(make_df(DATA_CAP)) + + +def test_transform_leaves_variables_with_infinite_caps_untouched(make_df): + transformer = MockCapper().fit(make_df(DATA_CAP)) + transformer.right_tail_caps_ = {"a": np.inf, "b": np.inf} + transformer.left_tail_caps_ = {"a": -np.inf, "c": -np.inf} + + Xt = transformer.transform(make_df(DATA_CAP)) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == DATA_CAP + # the integer column is not cast to float + assert nw.from_native(Xt, eager_only=True)["c"].dtype.is_integer() diff --git a/tests/test_outliers/test_check_estimator_outliers.py b/tests/test_outliers/test_check_estimator_outliers.py index bb9773c12..7b85d3f99 100644 --- a/tests/test_outliers/test_check_estimator_outliers.py +++ b/tests/test_outliers/test_check_estimator_outliers.py @@ -1,9 +1,7 @@ import pandas as pd import pytest -import sklearn from sklearn.pipeline import Pipeline from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine.outliers import ArbitraryOutlierCapper, OutlierTrimmer, Winsoriser from feature_engine.tags import _return_tags @@ -16,45 +14,35 @@ Winsoriser(), ] -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) - -if sklearn_version < parse_version("1.6"): - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator(estimator) - -else: - FAILED_CHECKS = _return_tags()["_xfail_checks"] - FAILED_CHECKS_AOC = _return_tags()["_xfail_checks"] - - msg1 = ( - "transformers raise errors when data variation is low, " "thus this check fails" - ) - - msg2 = "transformer has 1 mandatory parameter" - - FAILED_CHECKS.update({"check_fit2d_1sample": msg1}) - FAILED_CHECKS_AOC.update( - { - "check_fit2d_1sample": msg1, - "check_parameters_default_constructible": msg2, - } - ) - - @pytest.mark.parametrize( - "estimator, failed_tests", - [ - (_estimators[0], FAILED_CHECKS_AOC), - (_estimators[1], FAILED_CHECKS), - (_estimators[2], FAILED_CHECKS), - ], +FAILED_CHECKS = _return_tags()["_xfail_checks"] +FAILED_CHECKS_AOC = _return_tags()["_xfail_checks"] + +msg1 = "transformers raise errors when data variation is low, " "thus this check fails" + +msg2 = "transformer has 1 mandatory parameter" + +FAILED_CHECKS.update({"check_fit2d_1sample": msg1}) +FAILED_CHECKS_AOC.update( + { + "check_fit2d_1sample": msg1, + "check_parameters_default_constructible": msg2, + } +) + + +@pytest.mark.parametrize( + "estimator, failed_tests", + [ + (_estimators[0], FAILED_CHECKS_AOC), + (_estimators[1], FAILED_CHECKS), + (_estimators[2], FAILED_CHECKS), + ], +) +def test_check_estimator_from_sklearn(estimator, failed_tests): + return check_estimator( + estimator=wrap_for_check_estimator(estimator), + expected_failed_checks=failed_tests, ) - def test_check_estimator_from_sklearn(estimator, failed_tests): - return check_estimator( - estimator=wrap_for_check_estimator(estimator), - expected_failed_checks=failed_tests, - ) @pytest.mark.parametrize("estimator", _estimators) diff --git a/tests/test_outliers/test_outlier_trimmer.py b/tests/test_outliers/test_outlier_trimmer.py index b4f6f8534..bc6525a12 100644 --- a/tests/test_outliers/test_outlier_trimmer.py +++ b/tests/test_outliers/test_outlier_trimmer.py @@ -2,70 +2,90 @@ # License: BSD 3 clause import numpy as np -import pandas as pd import pytest from feature_engine.outliers import OutlierTrimmer +from tests.backend_helpers import make_series, frame_to_dict +# row 0 is an outlier in both variables, row 1 only in var_b, row 4 only in var_a +DATA_TWO_VARS = {"var_a": [1, 2, 3, 4, 100], "var_b": [1000, 6, 7, 8, 9]} -def test_gaussian_right_tail_capping_when_fold_is_1(df_normal_dist): - # test case 1: mean and std, right tail + +# init parameters +# the errors come from WinsorizerBase and are tested in test_base_outlier.py +@pytest.mark.parametrize( + "capping_method, tail, fold, missing_values", + [ + ("gaussian", "right", "auto", "raise"), + ("iqr", "left", 2, "ignore"), + ("mad", "both", 1.5, "raise"), + ("quantiles", "both", 0.1, "ignore"), + ], +) +def test_init_param_assignment(capping_method, tail, fold, missing_values): + transformer = OutlierTrimmer( + capping_method=capping_method, + tail=tail, + fold=fold, + missing_values=missing_values, + ) + assert transformer.capping_method == capping_method + assert transformer.tail == tail + assert transformer.fold == fold + assert transformer.missing_values == missing_values + + +# fit and transform +def test_gaussian_right_tail_capping_when_fold_is_1(make_df, data_normal_dist): transformer = OutlierTrimmer(capping_method="gaussian", tail="right", fold=1) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(make_df(data_normal_dist)) - # expected output - df_transf = df_normal_dist.copy() - inliers = df_transf["var"].le(0.10727677848029868) - df_transf = df_transf.loc[inliers] + cap = transformer.right_tail_caps_["var"] + expected = [v for v in data_normal_dist["var"] if v <= cap] - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 83 + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var": pytest.approx(expected)} + assert X.shape[0] == 83 -def test_gaussian_both_tails_capping_with_fold_2(df_normal_dist): - # test case 2: mean and std, both tails, different fold value +def test_gaussian_both_tails_capping_with_fold_2(make_df, data_normal_dist): transformer = OutlierTrimmer(capping_method="gaussian", tail="both", fold=2) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(make_df(data_normal_dist)) - # expected output - df_transf = df_normal_dist.copy() - inliers = df_transf["var"].between(-0.1955956473898675, 0.2075572504967645) - df_transf = df_transf.loc[inliers] + lower = transformer.left_tail_caps_["var"] + upper = transformer.right_tail_caps_["var"] + expected = [v for v in data_normal_dist["var"] if lower <= v <= upper] - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 96 + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var": pytest.approx(expected)} + assert X.shape[0] == 96 -def test_iqr_left_tail_capping_with_fold_2(df_normal_dist): - # test case 3: IQR, left tail, fold 2 +def test_iqr_left_tail_capping_with_fold_0_8(make_df, data_normal_dist): transformer = OutlierTrimmer(capping_method="iqr", tail="left", fold=0.8) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(make_df(data_normal_dist)) - df_transf = df_normal_dist.copy() - inliers = df_transf["var"].ge(-0.17486039103044) - df_transf = df_transf.loc[inliers] + lower = transformer.left_tail_caps_["var"] + expected = [v for v in data_normal_dist["var"] if v >= lower] - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 98 + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var": pytest.approx(expected)} + assert X.shape[0] == 98 -def test_mad_right_tail_capping_with_fold_1(df_normal_dist): - # test case 4: MAD, right tail, fold 1 +def test_mad_right_tail_capping_with_fold_1(make_df, data_normal_dist): transformer = OutlierTrimmer(capping_method="mad", tail="right", fold=1) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(make_df(data_normal_dist)) - df_transf = df_normal_dist.copy() - inliers = df_transf["var"].le(0.10995521088494983) - df_transf = df_transf.loc[inliers] + cap = transformer.right_tail_caps_["var"] + expected = [v for v in data_normal_dist["var"] if v <= cap] - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 83 + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var": pytest.approx(expected)} + assert X.shape[0] == 83 -def test_transformer_ignores_na_in_df(df_na): - # test case 5: dataset contains na, and transformer is asked to ignore +def test_transformer_ignores_na_in_df(make_df, data_na): transformer = OutlierTrimmer( capping_method="gaussian", tail="right", @@ -73,39 +93,53 @@ def test_transformer_ignores_na_in_df(df_na): variables=["Age"], missing_values="ignore", ) - X = transformer.fit_transform(df_na) + X = transformer.fit_transform(make_df(data_na)) + + assert transformer.right_tail_caps_["Age"] == pytest.approx(38.04494616731882) + assert isinstance(X, make_df) + # rows with missing values are removed too + assert frame_to_dict(X)["Age"] == [20, 21, 19, 23, 37] - df_transf = df_na.copy() - inliers = df_transf["Age"].le(38.04494616731882) - df_transf = df_transf.loc[inliers] - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 5 +def test_rows_are_removed_if_any_variable_is_an_outlier(make_df): + transformer = OutlierTrimmer(capping_method="quantiles", tail="both", fold=0.2) + X = transformer.fit_transform(make_df(DATA_TWO_VARS)) + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var_a": [3, 4], "var_b": [7, 8]} -def test_transform_x_t(df_normal_dist): - y = pd.Series(np.zeros(len(df_normal_dist))) + +def test_transform_x_y(make_df, data_normal_dist): + df = make_df(data_normal_dist) + y = make_series(make_df, [0.0] * len(data_normal_dist["var"])) transformer = OutlierTrimmer(capping_method="mad", tail="right", fold=1) - X = transformer.fit_transform(df_normal_dist) - assert len(X) != len(y) + X = transformer.fit_transform(df) + assert X.shape[0] != len(y) - Xt, yt = transformer.transform_x_y(df_normal_dist, y) - assert len(Xt) == len(yt) - assert len(Xt) != len(df_normal_dist) - assert (Xt.index == yt.index).all() + Xt, yt = transformer.transform_x_y(df, y) + assert isinstance(Xt, make_df) + assert isinstance(yt, type(y)) + assert frame_to_dict(Xt) == frame_to_dict(X) + assert Xt.shape[0] == len(yt) + assert Xt.shape[0] != len(data_normal_dist["var"]) @pytest.mark.parametrize( - "strings,expected", + "capping_method, expected", [("gaussian", 3), ("iqr", 1.5), ("mad", 3.29), ("quantiles", 0.05)], ) -def test_auto_fold_default_value(strings, expected, df_normal_dist): - transformer = OutlierTrimmer(capping_method=strings, fold="auto") - transformer.fit(df_normal_dist) +def test_auto_fold_default_value(capping_method, expected, make_df, data_normal_dist): + transformer = OutlierTrimmer(capping_method=capping_method, fold="auto") + transformer.fit(make_df(data_normal_dist)) assert transformer.fold_ == expected -def test_low_variation(df_normal_dist): - transformer = OutlierTrimmer(capping_method="mad") - with pytest.raises(ValueError): - transformer.fit(df_normal_dist // 10) +def test_variables_without_variation_are_left_untouched(make_df, data_normal_dist): + data = {"var": [v // 10 for v in data_normal_dist["var"]]} + transformer = OutlierTrimmer(capping_method="mad", tail="both") + Xt = transformer.fit_transform(make_df(data)) + + assert transformer.right_tail_caps_ == {"var": np.inf} + assert transformer.left_tail_caps_ == {"var": -np.inf} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == data diff --git a/tests/test_outliers/test_winsorizer.py b/tests/test_outliers/test_winsorizer.py index 1264bece3..ad1a2308d 100644 --- a/tests/test_outliers/test_winsorizer.py +++ b/tests/test_outliers/test_winsorizer.py @@ -1,17 +1,28 @@ -import math import re import numpy as np -import pandas as pd import pytest from feature_engine.outliers import Winsoriser, Winsorizer +from tests.backend_helpers import frame_to_dict DEPRECATION_WARNING = ( "Winsorizer was deprecated in favour of Winsoriser in version 2.0.0 and will " "be removed in version 2.1.0. To silence this warning, use Winsoriser instead." ) +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) + +VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} + @pytest.fixture( params=[Winsoriser, Winsorizer], @@ -28,263 +39,153 @@ def make_transformer(transformer_class, **kwargs): return transformer_class(**kwargs) +# init parameters +# the errors of the parameters from WinsorizerBase are tested in test_base_outlier.py def test_winsorizer_raises_future_warning(): with pytest.warns(FutureWarning, match=re.escape(DEPRECATION_WARNING)): Winsorizer() -def test_gaussian_capping_right_tail_with_fold_1(df_normal_dist, transformer_class): - # test case 1: mean and std, right tail - transformer = make_transformer( - transformer_class, capping_method="gaussian", tail="right", fold=1 - ) - X = transformer.fit_transform(df_normal_dist) - - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = df_transf["var"].clip(upper=0.1067690260251065) - - # test init params - assert transformer.capping_method == "gaussian" - assert transformer.tail == "right" - assert transformer.fold == 1 - # test fit attr - assert math.isclose(transformer.right_tail_caps_["var"], 0.1067690260251065) - assert transformer.left_tail_caps_ == {} - assert transformer.n_features_in_ == 1 - # test transform outputs - pd.testing.assert_frame_equal(X, df_transf) - assert math.isclose(X["var"].max(), 0.10676902602510658) - assert math.isclose(df_transf["var"].max(), 0.1067690260251065) - - -def test_gaussian_capping_both_tails_with_fold_2(df_normal_dist, transformer_class): - # test case 2: mean and std, both tails, different fold value - transformer = make_transformer( - transformer_class, capping_method="gaussian", tail="both", fold=2 - ) - X = transformer.fit_transform(df_normal_dist) - - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = df_transf["var"].clip(-0.1955956473898675, 0.2075572504967645) - - # test fit params - assert math.isclose(transformer.right_tail_caps_["var"], 0.2075572504967645) - assert math.isclose(transformer.left_tail_caps_["var"], -0.1955956473898675) - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert math.isclose(X["var"].max(), 0.2075572504967645) - assert math.isclose(X["var"].min(), -0.1955956473898675) - assert math.isclose(df_transf["var"].max(), 0.2075572504967645) - assert math.isclose(df_transf["var"].min(), -0.1955956473898675) - - -def test_iqr_capping_both_tails_with_fold_1(df_normal_dist, transformer_class): - # test case 3: IQR, both tails, fold 1 - transformer = make_transformer( - transformer_class, capping_method="iqr", tail="both", fold=1 - ) - X = transformer.fit_transform(df_normal_dist) - - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = df_transf["var"].clip(-0.20247907173293223, 0.21180113880445128) - - # test fit params - assert math.isclose(transformer.right_tail_caps_["var"], 0.21180113880445128) - assert math.isclose(transformer.left_tail_caps_["var"], -0.20247907173293223) - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert math.isclose(X["var"].max(), 0.21180113880445128) - assert math.isclose(X["var"].min(), -0.20247907173293223) - assert math.isclose(df_transf["var"].max(), 0.21180113880445128) - assert math.isclose(df_transf["var"].min(), -0.20247907173293223) - - -def test_iqr_capping_left_tail_with_fold_2(df_normal_dist, transformer_class): - # test case 4: IQR, left tail, fold 2 - transformer = make_transformer( - transformer_class, capping_method="iqr", tail="left", fold=0.8 +@pytest.mark.parametrize("add_indicators", [-1, 1, "True", None, (), [True]]) +def test_error_if_add_indicators_not_permitted(add_indicators, transformer_class): + msg = ( + "add_indicators takes only booleans True and False. " + f"Got {add_indicators} instead." ) - X = transformer.fit_transform(df_normal_dist) - - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = df_transf["var"].clip(lower=-0.17486039103044) + with pytest.raises(ValueError, match=re.escape(msg)): + make_transformer(transformer_class, add_indicators=add_indicators) - # test fit params - assert transformer.right_tail_caps_ == {} - assert math.isclose(transformer.left_tail_caps_["var"], -0.17486039103044) - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert math.isclose(X["var"].min(), -0.17486039103044) - assert math.isclose(df_transf["var"].min(), -0.17486039103044) - -def test_quantile_capping_both_tails_with_fold_10_percent( - df_normal_dist, transformer_class +@pytest.mark.parametrize( + "capping_method, tail, fold, add_indicators, missing_values", + [ + ("gaussian", "right", "auto", False, "raise"), + ("iqr", "left", 2, True, "ignore"), + ("mad", "both", 1.5, False, "ignore"), + ("quantiles", "both", 0.1, True, "raise"), + ], +) +def test_init_param_assignment( + capping_method, tail, fold, add_indicators, missing_values, transformer_class ): - # test case 5: quantiles, both tails, fold 10% transformer = make_transformer( - transformer_class, capping_method="quantiles", tail="both", fold=0.1 + transformer_class, + capping_method=capping_method, + tail=tail, + fold=fold, + add_indicators=add_indicators, + missing_values=missing_values, ) - X = transformer.fit_transform(df_normal_dist) - - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = df_transf["var"].clip(-0.12366227743232801, 0.14712481122898166) + assert transformer.capping_method == capping_method + assert transformer.tail == tail + assert transformer.fold == fold + assert transformer.add_indicators == add_indicators + assert transformer.missing_values == missing_values - # test fit params - assert math.isclose(transformer.right_tail_caps_["var"], 0.14712481122898166) - assert math.isclose(transformer.left_tail_caps_["var"], -0.12366227743232801) - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert math.isclose(X["var"].max(), 0.14712481122898166) - assert math.isclose(X["var"].min(), -0.12366227743232801) - assert math.isclose(df_transf["var"].max(), 0.14712481122898166) - assert math.isclose(df_transf["var"].min(), -0.12366227743232801) - -def test_quantile_capping_right_tail_with_fold_15_percent( - df_normal_dist, transformer_class +# fit and transform +@pytest.mark.parametrize( + "capping_method, tail, fold, right, left", + [ + ("gaussian", "right", 1, 0.1067690260251065, None), + ("gaussian", "both", 2, 0.2075572504967645, -0.1955956473898675), + ("iqr", "both", 1, 0.21180113880445128, -0.20247907173293223), + ("iqr", "left", 0.8, None, -0.17486039103044), + ("quantiles", "both", 0.1, 0.14712481122898166, -0.12366227743232801), + ("quantiles", "right", 0.15, 0.11823196128033647, None), + ("mad", "right", 1, 0.10995521088494983, None), + ("mad", "both", 2, 0.21050080982609987, -0.1916815859385002), + ], +) +def test_capping( + make_df, + data_normal_dist, + transformer_class, + capping_method, + tail, + fold, + right, + left, ): - # test case 6: quantiles, right tail, fold 15% transformer = make_transformer( - transformer_class, capping_method="quantiles", tail="right", fold=0.15 + transformer_class, capping_method=capping_method, tail=tail, fold=fold ) - X = transformer.fit_transform(df_normal_dist) - - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = df_transf["var"].clip(upper=0.11823196128033647) - - # test fit params - assert math.isclose(transformer.right_tail_caps_["var"], 0.11823196128033647) - assert transformer.left_tail_caps_ == {} - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert math.isclose(X["var"].max(), 0.11823196128033647) - assert math.isclose(df_transf["var"].max(), 0.11823196128033647) + X_out = transformer.fit_transform(make_df(data_normal_dist)) + + upper = np.inf if right is None else right + lower = -np.inf if left is None else left + expected = [min(max(v, lower), upper) for v in data_normal_dist["var"]] + + if right is None: + assert transformer.right_tail_caps_ == {} + else: + assert transformer.right_tail_caps_ == {"var": pytest.approx(right)} + if left is None: + assert transformer.left_tail_caps_ == {} + else: + assert transformer.left_tail_caps_ == {"var": pytest.approx(left)} + assert transformer.n_features_in_ == 1 + assert isinstance(X_out, make_df) + assert frame_to_dict(X_out) == {"var": pytest.approx(expected)} @pytest.mark.parametrize( - "strings,expected", + "capping_method, expected", [("gaussian", 3), ("iqr", 1.5), ("mad", 3.29), ("quantiles", 0.05)], ) -def test_auto_fold_default_value(strings, expected, df_normal_dist, transformer_class): +def test_auto_fold_default_value( + make_df, data_normal_dist, capping_method, expected, transformer_class +): transformer = make_transformer( - transformer_class, capping_method=strings, fold="auto" + transformer_class, capping_method=capping_method, fold="auto" ) - transformer.fit(df_normal_dist) + transformer.fit(make_df(data_normal_dist)) assert transformer.fold_ == expected -def test_mad_capping_right_tail_with_fold_1(df_normal_dist, transformer_class): - # test case 1: median and mad, right tail - transformer = make_transformer( - transformer_class, capping_method="mad", tail="right", fold=1 - ) - X = transformer.fit_transform(df_normal_dist) - - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = df_transf["var"].clip(upper=0.10995521088494983) - - # test init params - assert transformer.capping_method == "mad" - assert transformer.tail == "right" - assert transformer.fold == 1 - # test fit attr - assert math.isclose(transformer.right_tail_caps_["var"], 0.10995521088494983) - assert transformer.left_tail_caps_ == {} - assert transformer.n_features_in_ == 1 - # test transform outputs - pd.testing.assert_frame_equal(X, df_transf) - assert math.isclose(X["var"].max(), 0.10995521088494983) - assert math.isclose(df_transf["var"].max(), 0.10995521088494983) - - -def test_mad_capping_both_tails_with_fold_2(df_normal_dist, transformer_class): - # test case 2: mean and std, both tails, different fold value - transformer = make_transformer( - transformer_class, capping_method="mad", tail="both", fold=2 - ) - X = transformer.fit_transform(df_normal_dist) - - # expected output - df_transf = df_normal_dist.copy() - df_transf["var"] = df_transf["var"].clip(-0.1916815859385002, 0.21050080982609987) - - # test fit params - assert math.isclose(transformer.right_tail_caps_["var"], 0.21050080982609987) - assert math.isclose(transformer.left_tail_caps_["var"], -0.1916815859385002) - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert math.isclose(X["var"].max(), 0.21050080982609987) - assert math.isclose(X["var"].min(), -0.1916815859385002) - assert math.isclose(df_transf["var"].max(), 0.21050080982609987) - assert math.isclose(df_transf["var"].min(), -0.1916815859385002) - - -def test_indicators_are_added(df_normal_dist, transformer_class): - transformer = make_transformer( - transformer_class, - tail="both", - capping_method="quantiles", - fold=0.1, - add_indicators=True, - ) - X = transformer.fit_transform(df_normal_dist) - # test that the number of output variables is correct - assert X.shape[1] == 3 * df_normal_dist.shape[1] - assert np.all(X.iloc[:, df_normal_dist.shape[1]:].sum(axis=0) > 0) - +@pytest.mark.parametrize("tail, n_indicators", [("both", 2), ("left", 1), ("right", 1)]) +def test_indicators_are_added( + make_df, data_normal_dist, transformer_class, tail, n_indicators +): + X = make_df(data_normal_dist) transformer = make_transformer( transformer_class, - tail="left", + tail=tail, capping_method="quantiles", fold=0.1, add_indicators=True, ) - X = transformer.fit_transform(df_normal_dist) - assert X.shape[1] == 2 * df_normal_dist.shape[1] - assert np.all(X.iloc[:, df_normal_dist.shape[1]:].sum(axis=0) > 0) + X_out = transformer.fit_transform(X) - transformer = make_transformer( - transformer_class, - tail="right", - capping_method="quantiles", - fold=0.1, - add_indicators=True, - ) - X = transformer.fit_transform(df_normal_dist) - assert X.shape[1] == 2 * df_normal_dist.shape[1] - assert np.all(X.iloc[:, df_normal_dist.shape[1]:].sum(axis=0) > 0) + assert isinstance(X_out, make_df) + assert X_out.shape[1] == 1 + n_indicators + result = frame_to_dict(X_out) + for col in list(X_out.columns)[1:]: + assert sum(result[col]) > 0 -def test_indicators_filter_variables(df_vartypes, transformer_class): +@pytest.mark.parametrize("tail, n_indicators", [("both", 4), ("left", 2), ("right", 2)]) +def test_indicators_filter_variables(make_df, transformer_class, tail, n_indicators): + X = make_df(VARTYPES) transformer = make_transformer( transformer_class, variables=["Age", "Marks"], - tail="both", + tail=tail, capping_method="quantiles", fold=0.1, add_indicators=True, ) - X = transformer.fit_transform(df_vartypes) - assert X.shape[1] == df_vartypes.shape[1] + 4 + X_out = transformer.fit_transform(X) - transformer.set_params(tail="left") - X = transformer.fit_transform(df_vartypes) - assert X.shape[1] == df_vartypes.shape[1] + 2 + assert isinstance(X_out, make_df) + assert X_out.shape[1] == len(VARTYPES) + n_indicators - transformer.set_params(tail="right") - X = transformer.fit_transform(df_vartypes) - assert X.shape[1] == df_vartypes.shape[1] + 2 +def test_indicators_are_correct(make_df, transformer_class): + X = make_df({"col": [float(i) for i in range(100)]}) + expected_left = [1.0] * 10 + [0.0] * 90 + expected_right = [0.0] * 90 + [1.0] * 10 -def test_indicators_are_correct(transformer_class): transformer = make_transformer( transformer_class, tail="left", @@ -292,39 +193,23 @@ def test_indicators_are_correct(transformer_class): fold=0.1, add_indicators=True, ) - df = pd.DataFrame({"col": np.arange(100).astype(np.float64)}) - df_out = transformer.fit_transform(df) - expected_ind = np.r_[np.repeat(True, 10), np.repeat(False, 90)].astype(np.float64) - pd.testing.assert_frame_equal( - df_out.drop("col", axis=1), df.assign(col_left=expected_ind).drop("col", axis=1) - ) + X_out = transformer.fit_transform(X) + assert isinstance(X_out, make_df) + assert frame_to_dict(X_out)["col_left"] == expected_left transformer.set_params(tail="right") - df_out = transformer.fit_transform(df) - expected_ind = np.r_[np.repeat(False, 90), np.repeat(True, 10)].astype(np.float64) - pd.testing.assert_frame_equal( - df_out.drop("col", axis=1), - df.assign(col_right=expected_ind).drop("col", axis=1), - ) + X_out = transformer.fit_transform(X) + assert frame_to_dict(X_out)["col_right"] == expected_right transformer.set_params(tail="both") - df_out = transformer.fit_transform(df) - expected_ind_left = np.r_[np.repeat(True, 10), np.repeat(False, 90)].astype( - np.float64 - ) - expected_ind_right = np.r_[np.repeat(False, 90), np.repeat(True, 10)].astype( - np.float64 - ) - pd.testing.assert_frame_equal( - df_out.drop("col", axis=1), - df.assign(col_left=expected_ind_left, col_right=expected_ind_right).drop( - "col", axis=1 - ), - ) + X_out = transformer.fit_transform(X) + result = frame_to_dict(X_out) + assert result["col_left"] == expected_left + assert result["col_right"] == expected_right + assert list(X_out.columns) == ["col", "col_left", "col_right"] -def test_transformer_ignores_na_in_df(df_na, transformer_class): - # test case 7: dataset contains na and transformer is asked to ignore them +def test_transformer_ignores_na_in_df(make_df, data_na, transformer_class): transformer = make_transformer( transformer_class, capping_method="gaussian", @@ -333,124 +218,74 @@ def test_transformer_ignores_na_in_df(df_na, transformer_class): variables=["Age", "Marks"], missing_values="ignore", ) - X = transformer.fit_transform(df_na) + X_out = transformer.fit_transform(make_df(data_na)) - # expected output - df_transf = df_na.copy() - df_transf["Age"] = df_transf["Age"].clip(upper=38.04494616731882) - df_transf["Marks"] = df_transf["Marks"].clip(upper=0.8784116651786605) - - # test fit params - assert math.isclose(transformer.right_tail_caps_["Age"], 38.04494616731882) - assert math.isclose(transformer.right_tail_caps_["Marks"], 0.8784116651786605) + assert transformer.right_tail_caps_ == { + "Age": pytest.approx(38.04494616731882), + "Marks": pytest.approx(0.8784116651786605), + } assert transformer.left_tail_caps_ == {} - assert transformer.n_features_in_ == 6 - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert math.isclose(X["Age"].max(), 38.04494616731882) - assert math.isclose(X["Age"].max(), 38.04494616731882) - assert math.isclose(X["Marks"].max(), 0.8784116651786605) - assert math.isclose(df_transf["Marks"].max(), 0.8784116651786605) - - -def test_error_if_capping_method_not_permitted(transformer_class): - # test error raises - with pytest.raises(ValueError): - make_transformer(transformer_class, capping_method="other") - - -def test_error_if_tail_value_not_permitted(transformer_class): - with pytest.raises(ValueError): - make_transformer(transformer_class, tail="other") - - -def test_error_if_missing_values_not_permited(transformer_class): - with pytest.raises(ValueError): - make_transformer(transformer_class, missing_values="other") - - -def test_error_if_fold_value_not_permitted(transformer_class): - with pytest.raises(ValueError): - make_transformer(transformer_class, fold=-1) - - -def test_error_if_capping_method_quantiles_and_fold_value_not_permitted( - transformer_class, -): - with pytest.raises(ValueError): - make_transformer(transformer_class, capping_method="quantiles", fold=0.3) - - -def test_error_if_add_incators_not_permitted(transformer_class): - with pytest.raises(ValueError): - make_transformer(transformer_class, add_indicators=-1) - with pytest.raises(ValueError): - make_transformer(transformer_class, add_indicators=()) - with pytest.raises(ValueError): - make_transformer(transformer_class, add_indicators=[True]) + assert transformer.n_features_in_ == 5 + assert isinstance(X_out, make_df) + result = frame_to_dict(X_out) + for var, cap in [("Age", 38.04494616731882), ("Marks", 0.8784116651786605)]: + expected = [None if v is None else min(v, cap) for v in data_na[var]] + assert result[var] == pytest.approx(expected) -def test_fit_raises_error_if_na_in_inut_df(df_na, transformer_class): - # test case 8: when dataset contains na, fit method - with pytest.raises(ValueError): - transformer = make_transformer(transformer_class) - transformer.fit(df_na) +def test_fit_raises_error_if_na_in_input_df(make_df, data_na, transformer_class): + transformer = make_transformer(transformer_class) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.fit(make_df(data_na)) def test_transform_raises_error_if_na_in_input_df( - df_vartypes, df_na, transformer_class + make_df, data_na, transformer_class ): - # test case 9: when dataset contains na, transform method - with pytest.raises(ValueError): - transformer = make_transformer(transformer_class) - transformer.fit(df_vartypes) - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + X_na = make_df({k: data_na[k] for k in ["Name", "City", "Age", "Marks"]}) + transformer = make_transformer(transformer_class) + transformer.fit(make_df(VARTYPES)) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(X_na) -def test_get_feature_names_out(df_na, transformer_class): - original_features = df_na.columns.to_list() - input_features = ["Age", "Marks"] - - # when indicators is false, we've got the generic check. - # We need to test only when true +# without indicators, the feature names are covered by the generic checks +@pytest.mark.parametrize( + "tail, indicators", + [ + ("left", ["Age_left", "Marks_left"]), + ("right", ["Age_right", "Marks_right"]), + ("both", ["Age_left", "Age_right", "Marks_left", "Marks_right"]), + ], +) +def test_get_feature_names_out(make_df, data_na, transformer_class, tail, indicators): + original_features = list(data_na) tr = make_transformer( - transformer_class, - tail="left", - add_indicators=True, - missing_values="ignore", + transformer_class, tail=tail, add_indicators=True, missing_values="ignore" ) - tr.fit(df_na) + tr.fit(make_df(data_na)) - out = [f + "_left" for f in input_features] - assert tr.get_feature_names_out() == original_features + out - assert tr.get_feature_names_out(original_features) == original_features + out + expected = original_features + indicators + assert tr.get_feature_names_out() == expected + assert tr.get_feature_names_out(original_features) == expected - tr = make_transformer( - transformer_class, - tail="right", - add_indicators=True, - missing_values="ignore", - ) - tr.fit(df_na) - - out = [f + "_right" for f in input_features] - assert tr.get_feature_names_out() == original_features + out - assert tr.get_feature_names_out(original_features) == original_features + out - tr = make_transformer( - transformer_class, - tail="both", - add_indicators=True, - missing_values="ignore", +def test_variables_without_variation_are_left_untouched( + make_df, data_normal_dist, transformer_class +): + data = { + "var": [v // 10 for v in data_normal_dist["var"]], + "other": data_normal_dist["var"], + } + transformer = make_transformer( + transformer_class, capping_method="mad", tail="both", add_indicators=True ) - tr.fit(df_na) - - out = ["Age_left", "Age_right", "Marks_left", "Marks_right"] - assert tr.get_feature_names_out() == original_features + out - assert tr.get_feature_names_out(original_features) == original_features + out - - -def test_low_variation(df_normal_dist, transformer_class): - transformer = make_transformer(transformer_class, capping_method="mad") - with pytest.raises(ValueError): - transformer.fit(df_normal_dist // 10) + Xt = transformer.fit_transform(make_df(data)) + + assert transformer.right_tail_caps_["var"] == np.inf + assert transformer.left_tail_caps_["var"] == -np.inf + assert isinstance(Xt, make_df) + result = frame_to_dict(Xt) + assert result["var"] == data["var"] + assert result["var_left"] == [0.0] * len(data["var"]) + assert result["var_right"] == [0.0] * len(data["var"]) diff --git a/tests/test_prediction/test_check_estimator_prediction.py b/tests/test_prediction/test_check_estimator_prediction.py index 3618933b3..afe45db71 100644 --- a/tests/test_prediction/test_check_estimator_prediction.py +++ b/tests/test_prediction/test_check_estimator_prediction.py @@ -1,11 +1,8 @@ import numpy as np import pandas as pd import pytest -import sklearn from sklearn.base import clone from sklearn.exceptions import NotFittedError -from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine._prediction.base_predictor import BaseTargetMeanEstimator from feature_engine._prediction.target_mean_classifier import TargetMeanClassifier @@ -18,18 +15,14 @@ from tests.estimator_checks.dataframe_for_checks import test_df from tests.estimator_checks.fit_functionality_checks import check_error_if_y_not_passed -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) - _estimators = [BaseTargetMeanEstimator(), TargetMeanClassifier(), TargetMeanRegressor()] _predictors = [TargetMeanRegressor(), TargetMeanClassifier()] -if sklearn_version < parse_version("1.6"): - # In sklearn version 1.6, changes into the developer api were introduced - # that break the tests. Need to dig further into it. - # TODO: add tests for sklearn version > 1.6 - @pytest.mark.parametrize("estimator", [BaseTargetMeanEstimator()]) - def test_check_estimator_from_sklearn(estimator): - return check_estimator(estimator) + +# TODO: no test_check_estimator_from_sklearn exists for this module — the previous +# sklearn<1.6 version of this test was removed when dropping sklearn<=1.6 support, +# and a sklearn>=1.6-compatible replacement (using expected_failed_checks=...) was +# never written. See the module's git history for the removed sklearn<1.6 branch. @pytest.mark.parametrize("estimator", _estimators) diff --git a/tests/test_preprocessing/test_check_estimator_preprocessing.py b/tests/test_preprocessing/test_check_estimator_preprocessing.py index d76134053..d5293807e 100644 --- a/tests/test_preprocessing/test_check_estimator_preprocessing.py +++ b/tests/test_preprocessing/test_check_estimator_preprocessing.py @@ -1,12 +1,10 @@ import pandas as pd import pytest -import sklearn from numpy import nan from sklearn import clone from sklearn.exceptions import NotFittedError from sklearn.pipeline import Pipeline from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine.preprocessing import MatchCategories, MatchVariables from feature_engine.tags import _return_tags @@ -16,46 +14,38 @@ ) from tests.estimator_checks.sklearn_check_wrapper import wrap_for_check_estimator -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) - _estimators = [MatchCategories(ignore_format=True), MatchVariables()] -if sklearn_version < parse_version("1.6"): - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator(estimator) +FAILED_CHECKS = _return_tags()["_xfail_checks"] +FAILED_CHECKS_MATCHCOLS = _return_tags()["_xfail_checks"] -else: - FAILED_CHECKS = _return_tags()["_xfail_checks"] - FAILED_CHECKS_MATCHCOLS = _return_tags()["_xfail_checks"] +msg1 = "input shape of dataframes in fit and transform can differ" +msg2 = ( + "transformer takes categorical variables, and inf cannot be determined" + "on these variables. Thus, check is not implemented" +) - msg1 = "input shape of dataframes in fit and transform can differ" - msg2 = ( - "transformer takes categorical variables, and inf cannot be determined" - "on these variables. Thus, check is not implemented" - ) +FAILED_CHECKS.update({"check_estimators_nan_inf": msg2}) +FAILED_CHECKS_MATCHCOLS.update( + { + "check_transformer_general": msg1, + "check_estimators_nan_inf": msg2, + } +) - FAILED_CHECKS.update({"check_estimators_nan_inf": msg2}) - FAILED_CHECKS_MATCHCOLS.update( - { - "check_transformer_general": msg1, - "check_estimators_nan_inf": msg2, - } - ) - @pytest.mark.parametrize( - "estimator, failed_tests", - [ - (_estimators[0], FAILED_CHECKS), - (_estimators[1], FAILED_CHECKS_MATCHCOLS), - ], +@pytest.mark.parametrize( + "estimator, failed_tests", + [ + (_estimators[0], FAILED_CHECKS), + (_estimators[1], FAILED_CHECKS_MATCHCOLS), + ], +) +def test_check_estimator_from_sklearn(estimator, failed_tests): + return check_estimator( + estimator=wrap_for_check_estimator(estimator), + expected_failed_checks=failed_tests, ) - def test_check_estimator_from_sklearn(estimator, failed_tests): - return check_estimator( - estimator=wrap_for_check_estimator(estimator), - expected_failed_checks=failed_tests, - ) @pytest.mark.parametrize("estimator", [MatchCategories(), MatchVariables()]) diff --git a/tests/test_preprocessing/test_match_categories.py b/tests/test_preprocessing/test_match_categories.py index dff612c3c..bb6a2d3e0 100644 --- a/tests/test_preprocessing/test_match_categories.py +++ b/tests/test_preprocessing/test_match_categories.py @@ -1,67 +1,310 @@ -import warnings +import re -import numpy as np +import narwhals as nw import pandas as pd +import polars as pl import pytest +from sklearn.exceptions import NotFittedError from feature_engine.preprocessing import MatchCategories +from tests.backend_helpers import frame_to_dict +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." +) -def test_category_encoder_outputs_correct_dtype(): - df_str = pd.DataFrame({"col1": ["a", "b", "c"]}) - res_str = MatchCategories().fit(df_str).transform(df_str) - assert res_str.dtypes["col1"] == "category" +MSG_NA_INTRODUCED = ( + "During the encoding, NaN values were introduced in the feature(s) {}." +) - df_float = pd.DataFrame({"col1": [1.0, 2.0, 3.0]}) - tr = MatchCategories(variables=["col1"], ignore_format=True) - res_float = tr.fit(df_float).transform(df_float) - assert res_float.dtypes["col1"] == "category" +TRAIN = {"x1": ["b", "a", "c", "a"], "x2": [4, 5, 6, 7], "x3": ["z", "y", "z", "y"]} +TEST = {"x1": ["c", "d", "a", "b"], "x2": [5, 6, 4, 7], "x3": ["y", "w", "z", "y"]} - df_obj = pd.DataFrame({"col1": ["a", None, -1.0]}) - with warnings.catch_warnings(): - warnings.simplefilter("ignore") - res_obj = MatchCategories(missing_values="ignore").fit(df_obj).transform(df_obj) - assert res_obj.dtypes["col1"] == "category" - df_categ = pd.DataFrame({"col1": pd.Categorical(pd.Series(["a", "b", "c"]))}) - res_categ = MatchCategories().fit(df_categ).transform(df_categ) - assert res_categ.dtypes["col1"] == "category" +def dtype_categories(X, variable): + if isinstance(X, pd.DataFrame): + return list(X[variable].cat.categories) + return list(X.schema[variable].categories) -def test_category_encoder_handles_missing(): - df_no_nas = pd.DataFrame({"col1": ["a", "b", "c"]}) - df_nas = pd.DataFrame({"col1": ["a", "b", None]}) - df_new = pd.DataFrame({"col1": ["a", "b", "d"]}) +# init parameters +@pytest.mark.parametrize( + "missing_values", ["other", "Raise", "", 1, 0.5, True, None, ["raise"]] +) +def test_error_if_missing_values_not_allowed(missing_values): + msg = ( + "missing_values takes only values 'raise' or 'ignore'. " + f"Got {missing_values} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + MatchCategories(missing_values=missing_values) + + +@pytest.mark.parametrize("ignore_format", ["True", 1, 0, None, [True]]) +def test_error_if_ignore_format_not_bool(ignore_format): + msg = ( + "ignore_format takes only booleans True and False. " + f"Got {ignore_format} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + MatchCategories(ignore_format=ignore_format) + + +@pytest.mark.parametrize("return_empty", ["True", 1, 0, None, [True]]) +def test_error_if_return_empty_not_bool(return_empty): + msg = ( + "return_empty takes only boolean values True and False. " + f"Got {return_empty} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + MatchCategories(return_empty=return_empty) + + +@pytest.mark.parametrize( + "missing_values, ignore_format", [("raise", False), ("ignore", True)] +) +def test_init_param_assignment(missing_values, ignore_format): + transformer = MatchCategories( + missing_values=missing_values, ignore_format=ignore_format + ) + assert transformer.missing_values == missing_values + assert transformer.ignore_format is ignore_format + + +# fit and transform +def test_learns_categories_and_casts_to_categorical(make_df): + transformer = MatchCategories() + transformer.fit(make_df(TRAIN)) + Xt = transformer.transform(make_df(TRAIN)) + + assert transformer.variables_ == ["x1", "x3"] + assert {k: list(v) for k, v in transformer.category_dict_.items()} == { + "x1": ["a", "b", "c"], + "x3": ["y", "z"], + } + assert transformer.n_features_in_ == 3 + assert transformer.feature_names_in_ == ["x1", "x2", "x3"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == TRAIN + assert dtype_categories(Xt, "x1") == ["a", "b", "c"] + assert dtype_categories(Xt, "x3") == ["y", "z"] + + +def test_categories_are_the_same_in_train_and_test(make_df): + train = make_df({"x1": ["b", "a", "c"]}) + test = make_df({"x1": ["c", "b", "c"]}) + transformer = MatchCategories().fit(train) + + assert dtype_categories(transformer.transform(train), "x1") == ["a", "b", "c"] + assert dtype_categories(transformer.transform(test), "x1") == ["a", "b", "c"] - # check that it fails for missing values when using 'raise' - tr = MatchCategories(missing_values="raise") - with pytest.raises(ValueError): - tr.fit(df_nas) - tr.fit(df_no_nas) - with pytest.raises(ValueError): - tr.transform(df_nas) +def test_unseen_categories_become_nan_and_warn(make_df): + transformer = MatchCategories(missing_values="ignore").fit(make_df(TRAIN)) - # check that it doens't fail for missing values when using 'ignore' - tr = MatchCategories(missing_values="ignore").fit(df_nas) - with pytest.warns(UserWarning): - tr.transform(df_nas) + with pytest.warns(UserWarning, match=re.escape(MSG_NA_INTRODUCED.format("x1, x3"))): + Xt = transformer.transform(make_df(TEST)) - # check that it doesn't fail at transforming new values when using 'ignore' - tr = MatchCategories(missing_values="ignore").fit(df_no_nas) - with pytest.warns(UserWarning): - tr.transform(df_new) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "x1": ["c", None, "a", "b"], + "x2": [5, 6, 4, 7], + "x3": ["y", None, "z", "y"], + } + assert dtype_categories(Xt, "x1") == ["a", "b", "c"] -def test_category_outputs_correct_results(): - df = pd.DataFrame({"col1": ["a", "b", "c"], "col2": [1.0, 2.0, 3.0]}) - res = MatchCategories(variables=["col1", "col2"], ignore_format=True).fit_transform( - df +def test_error_if_unseen_categories_when_missing_values_raise(make_df): + transformer = MatchCategories().fit(make_df(TRAIN)) + with pytest.raises(ValueError, match=re.escape(MSG_NA_INTRODUCED.format("x1, x3"))): + transformer.transform(make_df(TEST)) + + +def test_error_if_nan_in_fit_when_missing_values_raise(make_df): + X = make_df({"x1": ["a", None, "b"]}) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + MatchCategories().fit(X) + + +def test_error_if_nan_in_transform_when_missing_values_raise(make_df): + transformer = MatchCategories().fit(make_df({"x1": ["a", "b", "b"]})) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(make_df({"x1": ["a", None, "b"]})) + + +def test_nan_is_not_a_category_when_missing_values_ignore(make_df): + X = make_df({"x1": ["b", None, "a", "b"], "x2": [1.0, None, 2.0, 3.0]}) + transformer = MatchCategories(missing_values="ignore").fit(X) + + with pytest.warns(UserWarning, match=re.escape(MSG_NA_INTRODUCED.format("x1"))): + Xt = transformer.transform(X) + + assert {k: list(v) for k, v in transformer.category_dict_.items()} == { + "x1": ["a", "b"] + } + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "x1": ["b", None, "a", "b"], + "x2": [1.0, None, 2.0, 3.0], + } + + +@pytest.mark.parametrize("variables", ["x3", ["x3"]]) +def test_transforms_only_selected_variables(make_df, variables): + transformer = MatchCategories(variables=variables, missing_values="ignore") + transformer.fit(make_df(TRAIN)) + + with pytest.warns(UserWarning, match=re.escape(MSG_NA_INTRODUCED.format("x3"))): + Xt = transformer.transform(make_df(TEST)) + + assert transformer.variables_ == ["x3"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "x1": ["c", "d", "a", "b"], + "x2": [5, 6, 4, 7], + "x3": ["y", None, "z", "y"], + } + assert nw.from_native(Xt).schema["x1"] == nw.String + + +def test_keeps_categories_of_categorical_input(make_df): + # the categories come from the dtype, so unused ones and their order are kept. + X = ( + nw.from_native(make_df({"x1": ["b", "a", "b"]})) + .with_columns(nw.col("x1").cast(nw.Enum(["z", "b", "a"]))) + .to_native() ) - pd.testing.assert_frame_equal(df, res, check_dtype=False, check_categorical=False) + transformer = MatchCategories().fit(X) + Xt = transformer.transform(make_df({"x1": ["a", "b", "a"]})) + + assert list(transformer.category_dict_["x1"]) == ["z", "b", "a"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"x1": ["a", "b", "a"]} + assert dtype_categories(Xt, "x1") == ["z", "b", "a"] + + +def test_ignore_format_casts_numerical_variables(make_df): + # polars categorical dtypes only take strings, so numbers become strings. + X = make_df({"x1": [3, 1, 2, 10], "x2": [1.5, float("nan"), 2.5, 10.0]}) + X_test = make_df({"x1": [1, 5, 10, 3], "x2": [2.5, 1.5, 7.0, 10.0]}) + transformer = MatchCategories(ignore_format=True, missing_values="ignore") + transformer.fit(X) + + with pytest.warns(UserWarning, match=re.escape(MSG_NA_INTRODUCED.format("x1, x2"))): + Xt = transformer.transform(X_test) + + expected_categories = { + pd.DataFrame: {"x1": [1, 2, 3, 10], "x2": [1.5, 2.5, 10.0]}, + pl.DataFrame: {"x1": ["1", "2", "3", "10"], "x2": ["1.5", "2.5", "10.0"]}, + } + expected_values = { + pd.DataFrame: {"x1": [1.0, None, 10.0, 3.0], "x2": [2.5, 1.5, None, 10.0]}, + pl.DataFrame: { + "x1": ["1", None, "10", "3"], + "x2": ["2.5", "1.5", None, "10.0"], + }, + } + assert { + k: list(v) for k, v in transformer.category_dict_.items() + } == expected_categories[make_df] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == expected_values[make_df] + assert dtype_categories(Xt, "x1") == expected_categories[make_df]["x1"] + + +def test_return_empty_when_no_categorical_variables(make_df): + X = make_df({"x1": [1, 2, 3], "x2": [1.0, 2.0, 3.0]}) + transformer = MatchCategories(return_empty=True) + + with pytest.warns( + UserWarning, + match=re.escape( + "No categorical variables found in this dataframe. " + "Returning an empty list." + ), + ): + transformer.fit(X) + Xt = transformer.transform(X) + + assert transformer.variables_ == [] + assert transformer.category_dict_ == {} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"x1": [1, 2, 3], "x2": [1.0, 2.0, 3.0]} + + +def test_error_if_no_categorical_variables(make_df): + msg = ( + "No categorical variables found in this dataframe. Check variable " + "dtypes or set return_empty to True to return an empty list instead." + ) + with pytest.raises(TypeError, match=re.escape(msg)): + MatchCategories().fit(make_df({"x1": [1, 2, 3]})) + + +def test_does_not_modify_input(make_df): + X = make_df(TEST) + transformer = MatchCategories(missing_values="ignore").fit(make_df(TRAIN)) + + with pytest.warns(UserWarning, match=re.escape(MSG_NA_INTRODUCED.format("x1, x3"))): + transformer.transform(X) + + assert frame_to_dict(X) == TEST + assert nw.from_native(X).schema["x1"] == nw.String + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MatchCategories instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MatchCategories().transform(make_df(TRAIN)) + + +def test_output_dtype_is_pandas_category(): + Xt = MatchCategories(missing_values="ignore").fit(pd.DataFrame(TRAIN)) + + with pytest.warns(UserWarning, match=re.escape(MSG_NA_INTRODUCED.format("x1, x3"))): + Xt = Xt.transform(pd.DataFrame(TEST)) + + expected = pd.DataFrame( + { + "x1": pd.Categorical(["c", None, "a", "b"], categories=["a", "b", "c"]), + "x2": [5, 6, 4, 7], + "x3": pd.Categorical(["y", None, "z", "y"], categories=["y", "z"]), + } + ) + pd.testing.assert_frame_equal(Xt, expected) + + +def test_output_dtype_is_polars_enum(): + Xt = MatchCategories().fit_transform(pl.DataFrame(TRAIN)) + assert Xt.schema == pl.Schema( + {"x1": pl.Enum(["a", "b", "c"]), "x2": pl.Int64, "x3": pl.Enum(["y", "z"])} + ) + + +def test_integer_column_names(): + X = pd.DataFrame({0: ["a", "b", "c"], 1: ["x", "y", "x"], "n": [1, 2, 3]}) + X_test = pd.DataFrame({0: ["a", "q", "c"], 1: ["x", "y", "w"], "n": [1, 2, 3]}) + transformer = MatchCategories(missing_values="ignore").fit(X) + + with pytest.warns(UserWarning, match=re.escape(MSG_NA_INTRODUCED.format("0, 1"))): + Xt = transformer.transform(X_test) + + expected = pd.DataFrame( + { + 0: pd.Categorical(["a", None, "c"], categories=["a", "b", "c"]), + 1: pd.Categorical(["x", "y", None], categories=["x", "y"]), + "n": [1, 2, 3], + } + ) + pd.testing.assert_frame_equal(Xt, expected) + - df = pd.DataFrame({"col1": ["a", "b", "d"], "col2": [1.0, 2.0, np.nan]}) - res = MatchCategories( - variables=["col1", "col2"], ignore_format=True, missing_values="ignore" - ).fit_transform(df) - pd.testing.assert_frame_equal(df, res, check_dtype=False, check_categorical=False) +def test_keeps_pandas_index(): + X = pd.DataFrame(TRAIN, index=[10, 20, 30, 40]) + Xt = MatchCategories().fit_transform(X) + pd.testing.assert_index_equal(Xt.index, X.index) diff --git a/tests/test_preprocessing/test_match_columns.py b/tests/test_preprocessing/test_match_columns.py index 6726b33f9..dbd1437d2 100644 --- a/tests/test_preprocessing/test_match_columns.py +++ b/tests/test_preprocessing/test_match_columns.py @@ -1,370 +1,469 @@ +import datetime +import re + +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.preprocessing import MatchVariables +from tests.backend_helpers import frame_to_dict, null_count + +DOB = [datetime.datetime(2020, 2, 24, 0, minute) for minute in range(4)] + +DATA_TRAIN = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "dob": DOB, +} + +# lacks City and Age, has two extra variables and a different column order +DATA_TEST = { + "extra_1": ["a", "b", "c", "d"], + "Marks": [0.5, 0.4, 0.3, 0.2], + "Name": ["sam", "fred", "peter", "bob"], + "dob": DOB, + "extra_2": [1, 2, 3, 4], +} + +DATA_TRAIN_NA = { + "Name": ["tom", None, "krish", "jack"], + "City": ["London", "Manchester", None, "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], +} + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) -_params_fill_value = [ - (1, [1, 1, 1, 1], [1, 1, 1, 1]), - (0.1, [0.1, 0.1, 0.1, 0.1], [0.1, 0.1, 0.1, 0.1]), - ("none", ["none", "none", "none", "none"], ["none", "none", "none", "none"]), - (np.nan, [np.nan, np.nan, np.nan, np.nan], [np.nan, np.nan, np.nan, np.nan]), -] -_params_allowed = [ - ([0, 1], "ignore", True, True), - ("nan", "hola", True, True), - ("nan", "ignore", True, "hallo"), - ("nan", "ignore", "hallo", True), -] +# init parameters +@pytest.mark.parametrize("fill_value", [[0, 1], None, {"a": 1}, (1,)]) +def test_error_if_fill_value_not_allowed(fill_value): + msg = f"fill_value takes integers, floats or strings. Got {fill_value} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + MatchVariables(fill_value=fill_value) -@pytest.mark.parametrize( - "fill_value, expected_studies, expected_age", _params_fill_value -) -def test_drop_and_add_columns( - fill_value, expected_studies, expected_age, df_vartypes, df_na -): - train = df_na.copy() - test = df_vartypes.copy() - test = test.drop("Age", axis=1) # to add more than one column - - # adding columns to test if they are removed - for new_col in ["test1", "test2"]: - test.loc[:, new_col] = new_col - - match_columns = MatchVariables( - fill_value=fill_value, - missing_values="ignore", +@pytest.mark.parametrize("missing_values", ["hola", 1, None, ["raise"]]) +def test_error_if_missing_values_not_allowed(missing_values): + msg = ( + "missing_values takes only values 'raise' or 'ignore'. " + f"Got {missing_values} instead." ) - match_columns.fit(train) + with pytest.raises(ValueError, match=re.escape(msg)): + MatchVariables(missing_values=missing_values) - transformed_df = match_columns.transform(test) - expected_result = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Studies": expected_studies, - "Age": expected_age, - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - } +@pytest.mark.parametrize("match_dtypes", ["hallo", 1, None, [True]]) +def test_error_if_match_dtypes_not_bool(match_dtypes): + msg = ( + "match_dtypes takes only booleans True and False. " + f"Got {match_dtypes} instead." ) + with pytest.raises(ValueError, match=re.escape(msg)): + MatchVariables(match_dtypes=match_dtypes) - # test init params - if fill_value is np.nan: - assert match_columns.fill_value is np.nan - else: - assert match_columns.fill_value == fill_value - assert match_columns.verbose is True - assert match_columns.missing_values == "ignore" - assert match_columns.match_dtypes is False - # test fit attrs - assert list(match_columns.feature_names_in_) == list(train.columns) - assert match_columns.n_features_in_ == 6 - # test transform output - pd.testing.assert_frame_equal(expected_result, transformed_df) + +@pytest.mark.parametrize("verbose", ["hallo", 1, None, [True]]) +def test_error_if_verbose_not_bool(verbose): + msg = f"verbose takes only booleans True and False. Got {verbose} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + MatchVariables(verbose=verbose) @pytest.mark.parametrize( - "fill_value, expected_studies, expected_age", _params_fill_value + "fill_value, missing_values, match_dtypes, verbose", + [ + (np.nan, "raise", False, True), + (1, "ignore", True, False), + (0.1, "raise", True, True), + ("none", "ignore", False, False), + ], ) -def test_columns_addition_when_more_columns_in_train_than_test( - fill_value, expected_studies, expected_age, df_vartypes, df_na -): - train = df_na.copy() - test = df_vartypes.copy() - test = test.drop("Age", axis=1) # to add more than one column - - match_columns = MatchVariables( +def test_init_param_assignment(fill_value, missing_values, match_dtypes, verbose): + transformer = MatchVariables( fill_value=fill_value, - missing_values="ignore", + missing_values=missing_values, + match_dtypes=match_dtypes, + verbose=verbose, ) - match_columns.fit(train) + assert transformer.fill_value is fill_value + assert transformer.missing_values == missing_values + assert transformer.match_dtypes is match_dtypes + assert transformer.verbose is verbose - transformed_df = match_columns.transform(test) - expected_result = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Studies": expected_studies, - "Age": expected_age, - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - } - ) +# fit and transform +def test_fit_attributes(make_df): + transformer = MatchVariables().fit(make_df(DATA_TRAIN)) + assert transformer.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert transformer.n_features_in_ == 5 + assert not hasattr(transformer, "_dtype_dict") - # test init params - if fill_value is np.nan: - assert match_columns.fill_value is np.nan - else: - assert match_columns.fill_value == fill_value - assert match_columns.verbose is True - assert match_columns.missing_values == "ignore" - assert match_columns.match_dtypes is False - # test fit attrs - assert list(match_columns.feature_names_in_) == list(train.columns) - assert match_columns.n_features_in_ == 6 - # test transform output - pd.testing.assert_frame_equal(expected_result, transformed_df) - - -def test_drop_columns_when_more_columns_in_test_than_train(df_vartypes, df_na): - train = df_vartypes.copy() - train = train.drop("City", axis=1) # to remove more than one column - test = df_na.copy() - - match_columns = MatchVariables(missing_values="ignore") - match_columns.fit(train) - - transformed_df = match_columns.transform(test) - - expected_result = test.drop(columns=["Studies", "City"]) - - # test init params - assert match_columns.fill_value is np.nan - assert match_columns.verbose is True - assert match_columns.missing_values == "ignore" - assert match_columns.match_dtypes is False - # test fit attrs - assert list(match_columns.feature_names_in_) == list(train.columns) - assert match_columns.n_features_in_ == 4 - # test transform output - pd.testing.assert_frame_equal(expected_result, transformed_df) - - -def test_match_dtypes_string_to_numbers(df_vartypes): - train = df_vartypes.copy().select_dtypes("number") - test = train.copy().astype("string") - - match_columns = MatchVariables(match_dtypes=True) - match_columns.fit(train) - - transformed_df = match_columns.transform(test) - - # test init params - assert match_columns.match_dtypes is True - # test fit attrs - assert match_columns.dtype_dict_ == { - "Age": np.dtype("int64"), - "Marks": np.dtype("float64"), + +@pytest.mark.parametrize( + "fill_value, expected", + [(np.nan, None), (1, 1), (0.1, 0.1), ("none", "none")], +) +def test_add_drop_and_reorder_variables(make_df, fill_value, expected): + transformer = MatchVariables(fill_value=fill_value, verbose=False) + transformer.fit(make_df(DATA_TRAIN)) + Xt = transformer.transform(make_df(DATA_TEST)) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "Name": ["sam", "fred", "peter", "bob"], + "City": [expected] * 4, + "Age": [expected] * 4, + "Marks": [0.5, 0.4, 0.3, 0.2], + "dob": DOB, } + assert transformer.get_feature_names_out() == [ + "Name", + "City", + "Age", + "Marks", + "dob", + ] - # test transform output - pd.testing.assert_series_equal(train.dtypes, transformed_df.dtypes) - pd.testing.assert_frame_equal(transformed_df, train) +@pytest.mark.parametrize( + "fill_value, expected_dtype", + [(np.nan, nw.Float64), (1, nw.Int64), (0.1, nw.Float64), ("none", nw.String)], +) +def test_dtype_of_added_variables(make_df, fill_value, expected_dtype): + transformer = MatchVariables(fill_value=fill_value, verbose=False) + transformer.fit(make_df(DATA_TRAIN)) + Xt = transformer.transform(make_df(DATA_TEST)) + + schema = nw.from_native(Xt).schema + assert schema["City"] == expected_dtype + assert schema["Age"] == expected_dtype + + +def test_nan_fill_value_adds_missing_data(make_df): + # polars treats NaN as a value, so the added variables must hold nulls. + transformer = MatchVariables(verbose=False).fit(make_df(DATA_TRAIN)) + Xt = transformer.transform(make_df(DATA_TEST)) + assert null_count(Xt, "City") == 4 + assert null_count(Xt, "Age") == 4 + + +def test_only_reorder_variables(make_df): + X = make_df({"Age": [1, 2], "Name": ["a", "b"], "Marks": [0.1, 0.2]}) + train = make_df({"Name": ["c"], "Marks": [0.3], "Age": [3]}) + transformer = MatchVariables().fit(train) + Xt = transformer.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["Name", "Marks", "Age"] + assert frame_to_dict(Xt) == { + "Name": ["a", "b"], + "Marks": [0.1, 0.2], + "Age": [1, 2], + } -def test_match_dtypes_numbers_to_string(df_vartypes): - train = df_vartypes.copy().select_dtypes("number").astype("string") - test = df_vartypes.copy().select_dtypes("number") - match_columns = MatchVariables(match_dtypes=True) - match_columns.fit(train) +def test_no_variable_in_common(make_df): + transformer = MatchVariables(verbose=False).fit(make_df({"a": [1], "b": ["x"]})) + Xt = transformer.transform(make_df({"c": [1, 2]})) - transformed_df = match_columns.transform(test) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"a": [None, None], "b": [None, None]} - # test init params - assert match_columns.match_dtypes is True - # test fit attrs - assert isinstance(match_columns.dtype_dict_, dict) - # test transform output - pd.testing.assert_series_equal(train.dtypes, transformed_df.dtypes) - pd.testing.assert_frame_equal(transformed_df, train) +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA_TEST) + transformer = MatchVariables(fill_value=0, verbose=False) + transformer.fit(make_df(DATA_TRAIN)) + Xt = transformer.transform(X) -def test_match_dtypes_string_to_datetime(df_vartypes): - train = df_vartypes.copy().loc[:, ["dob"]] - test = train.copy().astype("string") + assert list(X.columns) == ["extra_1", "Marks", "Name", "dob", "extra_2"] + assert frame_to_dict(X) == DATA_TEST + assert frame_to_dict(Xt)["City"] == [0, 0, 0, 0] - match_columns = MatchVariables(match_dtypes=True, verbose=False) - match_columns.fit(train) - transformed_df = match_columns.transform(test) +def test_verbose_print_out(capsys, make_df): + transformer = MatchVariables(verbose=True).fit(make_df(DATA_TRAIN)) + transformer.transform(make_df(DATA_TEST)) - # test init params - assert match_columns.match_dtypes is True - assert match_columns.verbose is False - # test fit attrs - # TODO: Remove pandas < 3 support when dropping older pandas versions - if pd.__version__ >= "3": - assert match_columns.dtype_dict_ == {"dob": np.dtype(" 2? No, 100%!", + "Hello. World", + "Hello. World.", + "Hello... World!?!", + "This is a proper sentence containing " + "supercalifragilisticexpialidocious and exceptionally long words.", +] + +# non-ASCII letters and digits, non-breaking space (\xa0), file separator (\x1c), +# new lines around the final punctuation and a Greek final sigma +TEXT_EDGE_CASES = [ + "", + None, + " ", + "Hello World!", + "HELLO", + "\N{LATIN CAPITAL LETTER E WITH ACUTE}COLE " + "na\N{LATIN SMALL LETTER I WITH DIAERESIS}ve 123", + "\N{ARABIC-INDIC DIGIT THREE} digits", + "a\xa0b\x1cc", + "x.\n", + "x.\n\n", + "a\nb.", + "Dog dog DOG", + "\N{GREEK CAPITAL LETTER OMICRON}\N{GREEK CAPITAL LETTER DELTA}" + "\N{GREEK CAPITAL LETTER OMICRON}\N{GREEK CAPITAL LETTER SIGMA} " + "\N{GREEK SMALL LETTER OMICRON}\N{GREEK SMALL LETTER DELTA}" + "\N{GREEK SMALL LETTER OMICRON}\N{GREEK SMALL LETTER FINAL SIGMA}", + "Is 1 > 2? No, 100%!", +] + +EXPECTED = { + "char_count": [11, 5, 5, 8, 0, 8, 6, 0, 0, 6, 5, 5, 12, 3, 16, 14, 11, 12, 16, 91], + "word_count": [2, 1, 1, 2, 0, 1, 1, 0, 0, 3, 1, 2, 2, 1, 4, 6, 2, 2, 2, 11], + "sentence_count": [1, 0, 0, 4, 0, 0, 1, 0, 0, 3, 0, 1, 1, 1, 2, 2, 1, 2, 2, 1], + "avg_word_length": [ + 11 / 2, 5, 5, 4, 0, 8, 6, 0, 0, 2, 5, 5 / 2, 6, 3, 4, 14 / 6, 11 / 2, 6, 8, + 91 / 11, + ], + "digit_count": [0, 0, 5, 0, 0, 0, 0, 0, 0, 0, 0, 0, 4, 0, 0, 5, 0, 0, 0, 0], + "letter_count": [10, 5, 0, 4, 0, 8, 3, 0, 0, 3, 5, 2, 4, 0, 13, 4, 10, 10, 10, 90], + "uppercase_count": [2, 5, 0, 0, 0, 0, 0, 0, 0, 3, 3, 1, 2, 0, 0, 2, 2, 2, 2, 1], + "lowercase_count": [8, 0, 0, 4, 0, 8, 3, 0, 0, 0, 2, 1, 2, 0, 13, 2, 8, 8, 8, 89], + "special_char_count": [1, 0, 0, 4, 0, 0, 3, 0, 0, 3, 0, 3, 4, 3, 3, 5, 1, 2, 6, 1], + "whitespace_count": [1, 0, 0, 1, 3, 2, 0, 0, 0, 2, 0, 1, 1, 0, 3, 5, 1, 1, 1, 10], + "whitespace_ratio": [ + 1 / 12, 0, 0, 1 / 9, 1, 2 / 10, 0, 0, 0, 2 / 8, 0, 1 / 6, 1 / 13, 0, 3 / 19, + 5 / 19, 1 / 12, 1 / 13, 1 / 17, 10 / 101, + ], + "digit_ratio": [ + 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 4 / 12, 0, 0, 5 / 14, 0, 0, 0, 0, + ], + "uppercase_ratio": [ + 2 / 11, 1, 0, 0, 0, 0, 0, 0, 0, 3 / 6, 3 / 5, 1 / 5, 2 / 12, 0, 0, 2 / 14, + 2 / 11, 2 / 12, 2 / 16, 1 / 91, + ], + "has_digits": [0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 1, 0, 0, 0, 0], + "has_uppercase": [1, 1, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 0, 0, 1, 1, 1, 1, 1], + "is_empty": [0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], + "starts_with_uppercase": [ + 1, 1, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 0, 0, 1, 1, 1, 1, 1, + ], + "ends_with_punctuation": [ + 1, 0, 0, 1, 0, 0, 1, 0, 0, 1, 0, 0, 0, 1, 0, 1, 0, 1, 1, 1, + ], + "unique_word_count": [2, 1, 1, 2, 0, 1, 1, 0, 0, 3, 1, 2, 2, 1, 4, 6, 2, 2, 2, 11], + "lexical_diversity": [1, 1, 1, 1, 0, 1, 1, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1], +} + +EXPECTED_EDGE_CASES = { + "char_count": [0, 0, 0, 11, 5, 13, 7, 3, 2, 2, 3, 9, 8, 14], + "word_count": [0, 0, 0, 2, 1, 3, 2, 3, 1, 1, 2, 3, 2, 6], + "sentence_count": [0, 0, 0, 1, 0, 0, 0, 0, 1, 1, 1, 0, 0, 2], + "avg_word_length": [ + 0, 0, 0, 11 / 2, 5, 13 / 3, 7 / 2, 1, 2, 2, 3 / 2, 3, 4, 14 / 6, + ], + "digit_count": [0, 0, 0, 0, 0, 3, 1, 0, 0, 0, 0, 0, 0, 5], + "letter_count": [0, 0, 0, 10, 5, 8, 6, 3, 1, 1, 2, 9, 0, 4], + "uppercase_count": [0, 0, 0, 2, 5, 4, 0, 0, 0, 0, 0, 4, 0, 2], + "lowercase_count": [0, 0, 0, 8, 0, 4, 6, 3, 1, 1, 2, 5, 0, 2], + "special_char_count": [0, 0, 0, 1, 0, 2, 1, 0, 1, 1, 1, 0, 8, 5], + "whitespace_count": [0, 0, 3, 1, 0, 2, 1, 2, 1, 2, 1, 2, 1, 5], + "whitespace_ratio": [ + 0, 0, 1, 1 / 12, 0, 2 / 15, 1 / 8, 2 / 5, 1 / 3, 2 / 4, 1 / 4, 2 / 11, 1 / 9, + 5 / 19, + ], + "digit_ratio": [0, 0, 0, 0, 0, 3 / 13, 1 / 7, 0, 0, 0, 0, 0, 0, 5 / 14], + "uppercase_ratio": [0, 0, 0, 2 / 11, 1, 4 / 13, 0, 0, 0, 0, 0, 4 / 9, 0, 2 / 14], + "has_digits": [0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 0, 0, 1], + "has_uppercase": [0, 0, 0, 1, 1, 1, 0, 0, 0, 0, 0, 1, 0, 1], + "is_empty": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], + "starts_with_uppercase": [0, 0, 0, 1, 1, 0, 0, 0, 0, 0, 0, 1, 0, 1], + "ends_with_punctuation": [0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1], + "unique_word_count": [0, 0, 0, 2, 1, 3, 2, 3, 1, 1, 2, 1, 1, 6], + "lexical_diversity": [0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1 / 3, 1 / 2, 1], +} + + +# init parameters +@pytest.mark.parametrize( + "variables", [123, True, None, [1, 2], ["text", 123], ("text",), {"text": 1}] +) +def test_error_if_variables_not_string_or_list_of_strings(variables): + msg = f"variables must be a string or a list of strings. Got {variables} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + TextFeatures(variables=variables) @pytest.mark.parametrize( - "invalid_variables", + "features", [ + "char_count", 123, True, + ("char_count",), + {"char_count": 1}, [1, 2], - ["text", 123], - {"text": 1}, + ["char_count", True], + ["invalid_feature"], + ["char_count", "invalid_feature"], ], ) -def test_invalid_variables_raises_error(invalid_variables): - with pytest.raises(ValueError, match="variables must be a string or a list of"): - TextFeatures(variables=invalid_variables) +def test_error_if_features_not_permitted(features): + msg = ( + f"features must be None or a list with any of {list(TEXT_FEATURES.keys())}. " + f"Got {features} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + TextFeatures(variables=["text"], features=features) + + +@pytest.mark.parametrize("missing_values", ["empanada", True, 1, None, ["raise"]]) +def test_error_if_missing_values_not_permitted(missing_values): + msg = ( + "missing_values takes only values 'raise' or 'ignore'. " + f"Got {missing_values} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + TextFeatures(variables=["text"], missing_values=missing_values) + + +@pytest.mark.parametrize("drop_original", ["True", 1, None, [True]]) +def test_error_if_drop_original_not_bool(drop_original): + msg = ( + "drop_original takes only boolean values True and False. " + f"Got {drop_original} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + TextFeatures(variables=["text"], drop_original=drop_original) @pytest.mark.parametrize( - "invalid_features, err_msg", + "features, missing_values, drop_original", [ - ("some_string", "features must be"), - ([1, 2], "features must be"), - (123, "features must be"), - (True, "features must be"), - (["some_string", True], "features must be"), - ({"some_string": 1}, "features must be"), - (["invalid_feature"], "Invalid features"), - (["char_count", "invalid_feature"], "Invalid features"), + (None, "ignore", False), + (["char_count"], "raise", True), + (["word_count", "lexical_diversity"], "ignore", True), ], ) -def test_invalid_features_raises_error(invalid_features, err_msg): - with pytest.raises(ValueError, match=err_msg): - TextFeatures(variables=["text"], features=invalid_features) - - -# ============================================================================== -# FIT TESTS -# ============================================================================== +def test_init_param_assignment(features, missing_values, drop_original): + transformer = TextFeatures( + variables=["text"], + features=features, + missing_values=missing_values, + drop_original=drop_original, + ) + assert transformer.features == features + assert transformer.missing_values == missing_values + assert transformer.drop_original is drop_original +# fit and transform @pytest.mark.parametrize( - "variables, features", + "variables, features, variables_, features_", [ - ("text", None), - (["string"], ["char_count"]), - (["text", "string"], ["sentence_count", "avg_word_length"]), + ("text", None, ["text"], list(TEXT_FEATURES.keys())), + (["string"], ["char_count"], ["string"], ["char_count"]), + (["text", "string"], ["word_count"], ["text", "string"], ["word_count"]), ], ) -def test_fit_stores_attributes(variables, features): - X = pd.DataFrame({"text": ["Hello"], "string": ["Bye"]}) - transformer = TextFeatures(variables=variables, features=features) - transformer.fit(X) - - assert ( - transformer.variables_ == variables - if isinstance(variables, list) - else transformer.variables_ == [variables] - ) - assert ( - transformer.features_ == list(TEXT_FEATURES.keys()) - if features is None - else transformer.features_ == features - ) - assert transformer.feature_names_in_ == ["text", "string"] - assert transformer.n_features_in_ == 2 +def test_fit_attributes(make_df, variables, features, variables_, features_): + X = make_df({"text": ["Hello"], "string": ["Bye"], "number": [1]}) + transformer = TextFeatures(variables=variables, features=features).fit(X) + + assert transformer.variables_ == variables_ + assert transformer.features_ == features_ + assert transformer.feature_names_in_ == ["text", "string", "number"] + assert transformer.n_features_in_ == 3 + + +@pytest.mark.parametrize("target", ["series", "list", "array"]) +def test_fit_ignores_the_target(make_df, target): + X = make_df({"text": ["Hello", "World"]}) + y = { + "series": make_series(make_df, [0, 1]), + "list": [0, 1], + "array": np.array([0, 1]), + }[target] + transformer = TextFeatures(variables=["text"], features=["char_count"]) + Xt = transformer.fit(X, y).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"text": ["Hello", "World"], "text_char_count": [5, 5]} -def test_missing_variable_raises_error(): - X = pd.DataFrame({"text": ["Hello"]}) +def test_error_if_variable_not_in_dataframe(make_df): + X = make_df({"text": ["Hello"]}) transformer = TextFeatures(variables=["nonexistent"]) - with pytest.raises(ValueError, match="not present in the dataframe"): + msg = "Variables {'nonexistent'} are not present in the dataframe." + with pytest.raises(ValueError, match=re.escape(msg)): transformer.fit(X) -@pytest.mark.parametrize("variables", ["Age", "Marks", "dob"]) -def test_no_text_columns_raises_error(df_vartypes, variables): - transformer = TextFeatures(variables=variables) - with pytest.raises(ValueError, match="not object or string"): - transformer.fit(df_vartypes) - - -def test_nan_handling_raise_error_fit(df_na): - transformer = TextFeatures( - variables=["City"], features=["char_count"], missing_values="raise" +@pytest.mark.parametrize( + "variable, values", + [ + ("Age", [20, 21]), + ("Marks", [0.9, 0.8]), + ("dob", [datetime(2020, 2, 24), datetime(2020, 2, 25)]), + ], +) +def test_error_if_variable_not_text(make_df, variable, values): + X = make_df({"Name": ["tom", "nick"], variable: values}) + transformer = TextFeatures(variables=["Name", variable]) + msg = ( + f"Variables ['{variable}'] are not object or string. " + "Please provide text variables only." ) - msg = "`missing_values='ignore'` when initialising this transformer" - with pytest.raises(ValueError, match=msg): - transformer.fit(df_na) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.fit(X) -# ============================================================================== -# TRANSFORM TESTS - GENERAL -# ============================================================================== +def test_categorical_variables_with_string_categories(make_df): + X = make_df({"text": ["Hello World", "Hi", "Hello World"]}) + X = nw.from_native(X).with_columns(nw.col("text").cast(nw.Categorical)) + transformer = TextFeatures(variables=["text"], features=["word_count"]) + Xt = transformer.fit_transform(X.to_native()) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "text": ["Hello World", "Hi", "Hello World"], + "text_word_count": [2, 1, 2], + } -def test_transform_on_new_data(): - X_train = pd.DataFrame({"text": ["Hello World", "Foo Bar"]}) - X_test = pd.DataFrame({"text": ["New Data", "Test 123"]}) - transformer = TextFeatures( - variables=["text"], features=["char_count", "has_digits"] +def test_error_if_categories_are_not_strings(): + # polars categories are always strings + X = pd.DataFrame({"text": pd.Series([1, 2], dtype="category")}) + transformer = TextFeatures(variables=["text"]) + msg = ( + "Variables ['text'] are not object or string. " + "Please provide text variables only." ) - transformer.fit(X_train) - X_tr = transformer.transform(X_test) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.fit(X) - assert X_tr["text_char_count"].tolist() == [7, 7] - assert X_tr["text_has_digits"].tolist() == [0, 1] +def test_error_if_missing_values_in_fit(make_df): + X = make_df({"text": ["Hello", None, "World"]}) + transformer = TextFeatures(variables=["text"], missing_values="raise") + msg = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.fit(X) -def test_nan_handling_raise_error_transform(): - X_train = pd.DataFrame({"text": ["Hello", "World"]}) - X_test = pd.DataFrame({"text": ["Hello", None, "World"]}) - transformer = TextFeatures( - variables=["text"], features=["char_count"], missing_values="raise" + +def test_error_if_missing_values_in_transform(make_df): + transformer = TextFeatures(variables=["text"], missing_values="raise") + transformer.fit(make_df({"text": ["Hello", "World"]})) + msg = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer or set the parameter " + "`missing_values='ignore'` when initialising this transformer." ) - transformer.fit(X_train) - msg = "`missing_values='ignore'` when initialising this transformer" - with pytest.raises(ValueError, match=msg): - transformer.transform(X_test) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.transform(make_df({"text": ["Hello", None, "World"]})) -def test_nan_handling(): - X = pd.DataFrame({"text": ["Hello", None, "World"]}) +def test_missing_values_are_treated_as_empty_strings(make_df): + X = make_df({"text": ["Hello", None, "World"]}) transformer = TextFeatures(variables=["text"], features=["char_count"]) - X_tr = transformer.fit_transform(X) + Xt = transformer.fit_transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "text": ["Hello", "", "World"], + "text_char_count": [5, 0, 5], + } + assert frame_to_dict(X) == {"text": ["Hello", None, "World"]} + - # NaN should be filled with empty string, resulting in char_count of 0 - assert X_tr["text_char_count"].tolist() == [5, 0, 5] +def test_missing_values_raise_returns_same_values(make_df): + X = make_df({"text": ["Hello World", "Hi"]}) + transformer = TextFeatures( + variables=["text"], features=["word_count"], missing_values="raise" + ) + Xt = transformer.fit_transform(X) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "text": ["Hello World", "Hi"], + "text_word_count": [2, 1], + } -def test_default_all_features(): - """Test extracting all features with default parameters.""" - X = pd.DataFrame({"text": ["Hello World!", "Python 123", "AI"]}) + +def test_error_if_not_fitted(make_df): transformer = TextFeatures(variables=["text"]) - X_tr = transformer.fit_transform(X) + msg = ( + "This TextFeatures instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + transformer.transform(make_df({"text": ["Hello"]})) - # Spot check a few features to ensure they were added and computed - assert X_tr["text_char_count"].tolist() == [11, 9, 2] - assert X_tr["text_word_count"].tolist() == [2, 2, 1] - assert X_tr["text_digit_count"].tolist() == [0, 3, 0] + +def test_error_if_transform_gets_different_number_of_columns(make_df): + transformer = TextFeatures(variables=["text"]).fit(make_df({"text": ["Hello"]})) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.transform(make_df({"text": ["Hello"], "other": [1]})) -def test_specific_features(): - """Test extracting specific features only.""" - X = pd.DataFrame({"text": ["Hello", "World"]}) +def test_transform_on_new_data(make_df): transformer = TextFeatures( - variables=["text"], features=["char_count", "word_count"] + variables=["text"], features=["char_count", "has_digits"] ) - X_tr = transformer.fit_transform(X) + transformer.fit(make_df({"text": ["Hello World", "Foo Bar"]})) + Xt = transformer.transform(make_df({"text": ["New Data", "Test 123"]})) - # Check only specified features are extracted - assert X_tr.columns.tolist() == ["text", "text_char_count", "text_word_count"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "text": ["New Data", "Test 123"], + "text_char_count": [7, 7], + "text_has_digits": [0, 1], + } -def test_specific_variables(): - """Test extracting features from specific variables only.""" - X = pd.DataFrame( - {"text1": ["Hello", "World"], "text2": ["Foo", "Bar"], "numeric": [1, 2]} - ) - transformer = TextFeatures(variables=["text1"], features=["char_count"]) - X_tr = transformer.fit_transform(X) +def test_transform_reorders_columns_as_in_fit(make_df): + transformer = TextFeatures(variables=["text"], features=["char_count"]) + transformer.fit(make_df({"text": ["Hello"], "other": [1]})) + Xt = transformer.transform(make_df({"other": [2], "text": ["Hi"]})) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"text": ["Hi"], "other": [2], "text_char_count": [2]} + + +def test_default_extracts_all_features(make_df): + X = make_df({"text": ["Hello World!", "Python 123", "AI"]}) + Xt = TextFeatures(variables=["text"]).fit_transform(X) - # Only text1 should have features extracted - assert X_tr.columns.tolist() == ["text1", "text2", "numeric", "text1_char_count"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["text"] + [f"text_{f}" for f in TEXT_FEATURES] -def test_drop_original(): - """Test drop_original parameter.""" - X = pd.DataFrame({"text": ["Hello", "World"], "other": [1, 2]}) +def test_only_selected_variables_and_features_are_added(make_df): + X = make_df({"a": ["Hello", "World"], "b": ["Foo", "Bar"], "numeric": [1, 2]}) + transformer = TextFeatures( + variables=["b", "a"], features=["word_count", "is_empty"] + ) + Xt = transformer.fit_transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "a": ["Hello", "World"], + "b": ["Foo", "Bar"], + "numeric": [1, 2], + "b_word_count": [1, 1], + "b_is_empty": [0, 0], + "a_word_count": [1, 1], + "a_is_empty": [0, 0], + } + + +def test_drop_original(make_df): + X = make_df({"text": ["Hello", "World"], "other": [1, 2]}) transformer = TextFeatures( variables=["text"], features=["char_count"], drop_original=True ) - X_tr = transformer.fit_transform(X) + Xt = transformer.fit_transform(X) - assert X_tr.columns.tolist() == ["other", "text_char_count"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"other": [1, 2], "text_char_count": [5, 5]} -def test_string_variable_input(): - """Test that passing a single string variable works (auto-converted to list).""" - X = pd.DataFrame({"text": ["Hello", "World"], "other": ["A", "B"]}) - transformer = TextFeatures(variables="text", features=["char_count"]) - X_tr = transformer.fit_transform(X) +@pytest.mark.parametrize("feature", list(TEXT_FEATURES.keys())) +def test_feature_values(make_df, feature): + X = make_df({"text": TEXT}) + Xt = TextFeatures(variables=["text"], features=[feature]).fit_transform(X) - assert transformer.variables_ == ["text"] - assert X_tr.columns.tolist() == ["text", "other", "text_char_count"] - assert X_tr["text_char_count"].tolist() == [5, 5] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt)[f"text_{feature}"] == pytest.approx(EXPECTED[feature]) -def test_multiple_text_columns(): - """Test extracting features from multiple text columns.""" - X = pd.DataFrame({"a": ["Hello", "World"], "b": ["Foo", "Bar"]}) - transformer = TextFeatures( - variables=["a", "b"], features=["char_count", "word_count"] +@pytest.mark.parametrize("feature", list(TEXT_FEATURES.keys())) +def test_feature_values_on_edge_cases(make_df, feature): + X = make_df({"text": TEXT_EDGE_CASES}) + Xt = TextFeatures(variables=["text"], features=[feature]).fit_transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt)[f"text_{feature}"] == pytest.approx( + EXPECTED_EDGE_CASES[feature] ) - X_tr = transformer.fit_transform(X) - - assert X_tr.columns.tolist() == [ - "a", - "b", - "a_char_count", - "a_word_count", - "b_char_count", - "b_word_count", - ] -# ============================================================================== -# TRANSFORM - TEST TEXT FEATURES -# ============================================================================== +@pytest.mark.parametrize("feature", list(TEXT_FEATURES.keys())) +def test_feature_values_on_other_backends(monkeypatch, feature): + # makes polars take the path used by backends other than pandas and polars + backend_checks = SimpleNamespace( + is_pandas_dataframe=lambda X: False, is_polars_dataframe=lambda X: False + ) + monkeypatch.setattr(text_features, "nwd", backend_checks) + X = pl.DataFrame({"text": TEXT_EDGE_CASES}) + Xt = TextFeatures(variables=["text"], features=[feature]).fit_transform(X) + + assert isinstance(Xt, pl.DataFrame) + assert frame_to_dict(Xt)[f"text_{feature}"] == pytest.approx( + EXPECTED_EDGE_CASES[feature] + ) -@pytest.fixture(scope="module") -def df_text(): - df = pd.DataFrame( +def test_lexical_diversity_is_unique_words_over_total_words(make_df): + X = make_df( { "text": [ - "Hello World!", - "HELLO", - "12345", - "e.g. i.e.", - " ", - " trailing ", - "abc...", - "", - None, - "A? B! C.", - "HeLLo", - "Hi! @#", - "A1b2 C3d4!@#$", - "???", - "i.e., this is wrong", - "Is 1 > 2? No, 100%!", - "Hello. World", - "Hello. World.", - "Hello... World!?!", - "This is a proper sentence containing " - "supercalifragilisticexpialidocious and exceptionally long words.", + "the cat sat on the mat", # 6 words, 5 unique + "good good good good", # 4 words, 1 unique + "all words here are distinct", # 5 words, 5 unique ] } ) - return df - - -def test_whitespace_features(df_text): - text_features = ["whitespace_count", "whitespace_ratio"] - transformer = TextFeatures(variables=["text"], features=text_features) - X_tr = transformer.fit_transform(df_text) - assert X_tr["text_whitespace_count"].tolist() == [ - 1, - 0, - 0, - 1, - 3, - 2, - 0, - 0, - 0, - 2, - 0, - 1, - 1, - 0, - 3, - 5, - 1, - 1, - 1, - 10, - ] - assert X_tr["text_whitespace_ratio"].tolist() == [ - 0.08333333333333333, - 0.0, - 0.0, - 0.1111111111111111, - 1.0, - 0.2, - 0.0, - 0.0, - 0.0, - 0.25, - 0.0, - 0.16666666666666666, - 0.07692307692307693, - 0.0, - 0.15789473684210525, - 0.2631578947368421, - 0.08333333333333333, - 0.07692307692307693, - 0.058823529411764705, - 0.09900990099009901, - ] - + transformer = TextFeatures(variables=["text"], features=["lexical_diversity"]) + Xt = transformer.fit_transform(X) -def test_digit_features(df_text): - transformer = TextFeatures( - variables=["text"], features=["digit_count", "digit_ratio", "has_digits"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt)["text_lexical_diversity"] == pytest.approx( + [5 / 6, 1 / 4, 1] ) - X_tr = transformer.fit_transform(df_text) - assert X_tr["text_digit_count"].tolist() == [ - 0, - 0, - 5, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 4, - 0, - 0, - 5, - 0, - 0, - 0, - 0, - ] - assert X_tr["text_digit_ratio"].tolist() == [ - 0.0, - 0.0, - 1.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.3333333333333333, - 0.0, - 0.0, - 0.35714285714285715, - 0.0, - 0.0, - 0.0, - 0.0, - ] - assert X_tr["text_has_digits"].tolist() == [ - 0, - 0, - 1, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 1, - 0, - 0, - 1, - 0, - 0, - 0, - 0, - ] -def test_uppercase_features(df_text): - transformer = TextFeatures( - variables=["text"], - features=[ - "uppercase_count", - "uppercase_ratio", - "has_uppercase", - "starts_with_uppercase", - ], - ) - X_tr = transformer.fit_transform(df_text) - assert X_tr["text_uppercase_count"].tolist() == [ - 2, - 5, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 3, - 3, - 1, - 2, - 0, - 0, - 2, - 2, - 2, - 2, - 1, - ] - assert X_tr["text_uppercase_ratio"].tolist() == [ - 0.18181818181818182, - 1.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.5, - 0.6, - 0.2, - 0.16666666666666666, - 0.0, - 0.0, - 0.14285714285714285, - 0.18181818181818182, - 0.16666666666666666, - 0.125, - 0.01098901098901099, - ] - assert X_tr["text_has_uppercase"].tolist() == [ - 1, - 1, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 1, - 1, - 1, - 1, - 0, - 0, - 1, - 1, - 1, - 1, - 1, - ] - assert X_tr["text_starts_with_uppercase"].tolist() == [ - 1, - 1, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 1, - 1, - 1, - 1, - 0, - 0, - 1, - 1, - 1, - 1, - 1, - ] - +def test_output_dtypes(make_df): + X = make_df({"text": ["Hello World", "Hi"]}) + features = ["char_count", "whitespace_ratio", "has_digits", "unique_word_count"] + Xt = TextFeatures(variables=["text"], features=features).fit_transform(X) -def test_punctuation_features(df_text): - transformer = TextFeatures( - variables=["text"], features=["special_char_count", "ends_with_punctuation"] - ) - X_tr = transformer.fit_transform(df_text) - assert X_tr["text_special_char_count"].tolist() == [ - 1, - 0, - 0, - 4, - 0, - 0, - 3, - 0, - 0, - 3, - 0, - 3, - 4, - 3, - 3, - 5, - 1, - 2, - 6, - 1, - ] - assert X_tr["text_ends_with_punctuation"].tolist() == [ - 1, - 0, - 0, - 1, - 0, - 0, - 1, - 0, - 0, - 1, - 0, - 0, - 0, - 1, - 0, - 1, - 0, - 1, - 1, - 1, + schema = nw.from_native(Xt).schema + assert [schema[f"text_{f}"] for f in features] == [ + nw.Int64, + nw.Float64, + nw.Int64, + nw.Int64, ] -def test_word_features(df_text): +@pytest.mark.parametrize( + "drop_original, expected", + [ + (False, ["text", "other", "text_char_count", "text_word_count"]), + (True, ["other", "text_char_count", "text_word_count"]), + ], +) +def test_get_feature_names_out(make_df, drop_original, expected): + X = make_df({"text": ["Hello"], "other": [1]}) transformer = TextFeatures( variables=["text"], - features=[ - "word_count", - "unique_word_count", - "lexical_diversity", - "avg_word_length", - ], + features=["char_count", "word_count"], + drop_original=drop_original, ) - X_tr = transformer.fit_transform(df_text) - assert X_tr["text_word_count"].tolist() == [ - 2, - 1, - 1, - 2, - 0, - 1, - 1, - 0, - 0, - 3, - 1, - 2, - 2, - 1, - 4, - 6, - 2, - 2, - 2, - 11, - ] - assert X_tr["text_unique_word_count"].tolist() == [ - 2, - 1, - 1, - 2, - 0, - 1, - 1, - 0, - 0, - 3, - 1, - 2, - 2, - 1, - 4, - 6, - 2, - 2, - 2, - 11, - ] - assert X_tr["text_lexical_diversity"].tolist() == [ - 1.0, - 1.0, - 1.0, - 1.0, - 0.0, - 1.0, - 1.0, - 0.0, - 0.0, - 1.0, - 1.0, - 1.0, - 1.0, - 1.0, - 1.0, - 1.0, - 1.0, - 1.0, - 1.0, - 1.0, - ] - assert X_tr["text_avg_word_length"].tolist() == [ - 6.0, - 5.0, - 5.0, - 4.5, - 0.0, - 8.0, - 6.0, - 0.0, - 0.0, - 2.6666666666666665, - 5.0, - 3.0, - 6.5, - 3.0, - 4.75, - 3.1666666666666665, - 6.0, - 6.5, - 8.5, - 9.181818181818182, - ] + Xt = transformer.fit_transform(X) + assert transformer.get_feature_names_out() == expected + assert list(Xt.columns) == expected -def test_basic_features(df_text): - transformer = TextFeatures( - variables=["text"], - features=[ - "char_count", - "sentence_count", - "letter_count", - "lowercase_count", - "is_empty", - ], - ) - X_tr = transformer.fit_transform(df_text) - assert X_tr["text_char_count"].tolist() == [ - 11, - 5, - 5, - 8, - 0, - 8, - 6, - 0, - 0, - 6, - 5, - 5, - 12, - 3, - 16, - 14, - 11, - 12, - 16, - 91, - ] - assert X_tr["text_sentence_count"].tolist() == [ - 1, - 0, - 0, - 4, - 0, - 0, - 1, - 0, - 0, - 3, - 0, - 1, - 1, - 1, - 2, - 2, - 1, - 2, - 2, - 1, - ] - assert X_tr["text_letter_count"].tolist() == [ - 10, - 5, - 0, - 4, - 0, - 8, - 3, - 0, - 0, - 3, - 5, - 2, - 4, - 0, - 13, - 4, - 10, - 10, - 10, - 90, - ] - assert X_tr["text_lowercase_count"].tolist() == [ - 8, - 0, - 0, - 4, - 0, - 8, - 3, - 0, - 0, - 0, - 2, - 1, - 2, - 0, - 13, - 2, - 8, - 8, - 8, - 89, - ] - assert X_tr["text_is_empty"].tolist() == [ - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 1, - 1, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 0, - 0, + +@pytest.mark.parametrize( + "input_features", [["text", "other"], np.array(["text", "other"])] +) +def test_get_feature_names_out_with_input_features(make_df, input_features): + X = make_df({"text": ["Hello"], "other": [1]}) + transformer = TextFeatures(variables=["text"], features=["char_count"]).fit(X) + assert transformer.get_feature_names_out(input_features) == [ + "text", + "other", + "text_char_count", ] -# ============================================================================== -# OTHER METHOD TESTS -# ============================================================================== +@pytest.mark.parametrize("input_features", [["other", "text"], ["text"]]) +def test_error_if_input_features_not_feature_names_in(make_df, input_features): + X = make_df({"text": ["Hello"], "other": [1]}) + transformer = TextFeatures(variables=["text"], features=["char_count"]).fit(X) + msg = "input_features is not equal to feature_names_in_" + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.get_feature_names_out(input_features) -def test_get_feature_names_out(): - X = pd.DataFrame({"text": ["Hello"], "other": [1]}) - transformer = TextFeatures( - variables=["text"], features=["char_count", "word_count"] - ) - transformer.fit(X) +@pytest.mark.parametrize("input_features", ["text", 1, {"text": 1}]) +def test_error_if_input_features_not_list_or_array(make_df, input_features): + X = make_df({"text": ["Hello"], "other": [1]}) + transformer = TextFeatures(variables=["text"], features=["char_count"]).fit(X) + msg = f"input_features must be a list or an array. Got {input_features} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + transformer.get_feature_names_out(input_features) - feature_names = transformer.get_feature_names_out() - expected_features = ["text", "other", "text_char_count", "text_word_count"] - assert feature_names == expected_features +def test_integer_column_names(): + X = pd.DataFrame({0: [1, 2], "text": ["Hello World", None], 1: ["a", "b"]}) + transformer = TextFeatures(variables=["text"], features=["word_count"]) + Xt = transformer.fit_transform(X) -def test_get_feature_names_out_with_drop(): - """Test get_feature_names_out with drop_original=True.""" - X = pd.DataFrame({"text": ["Hello"], "other": [1]}) - transformer = TextFeatures( - variables=["text"], features=["char_count"], drop_original=True + expected = pd.DataFrame( + { + 0: [1, 2], + "text": ["Hello World", ""], + 1: ["a", "b"], + "text_word_count": [2, 0], + } ) - transformer.fit(X) + pd.testing.assert_frame_equal(Xt, expected) + assert transformer.get_feature_names_out() == [0, "text", 1, "text_word_count"] - feature_names = transformer.get_feature_names_out() - expected_features = ["other", "text_char_count"] - assert feature_names == expected_features +def test_pandas_index_is_kept(): + X = pd.DataFrame({"text": ["Hello World", "Hi", "Hey"]}, index=[10, 10, 3]) + transformer = TextFeatures( + variables=["text"], features=["char_count", "unique_word_count"] + ) + Xt = transformer.fit_transform(X) -def test_lexical_diversity_is_unique_words_over_total_words(): - X = pd.DataFrame( + expected = pd.DataFrame( { - "text": [ - "the cat sat on the mat", # 6 words, 5 unique - "good good good good", # 4 words, 1 unique - "all words here are distinct", # 5 words, 5 unique - ] - } + "text": ["Hello World", "Hi", "Hey"], + "text_char_count": [10, 2, 3], + "text_unique_word_count": [2, 1, 1], + }, + index=[10, 10, 3], ) - transformer = TextFeatures(variables=["text"], features=["lexical_diversity"]) - X_tr = transformer.fit_transform(X) - - assert X_tr["text_lexical_diversity"].tolist() == [5 / 6, 1 / 4, 1.0] - # a ratio of unique words to total words never exceeds 1 - assert (X_tr["text_lexical_diversity"] <= 1.0).all() + pd.testing.assert_frame_equal(Xt, expected) diff --git a/tests/test_time_series/test_forecasting/test_check_estimator_forecasting.py b/tests/test_time_series/test_forecasting/test_check_estimator_forecasting.py index 6f4bcacae..bc887f918 100644 --- a/tests/test_time_series/test_forecasting/test_check_estimator_forecasting.py +++ b/tests/test_time_series/test_forecasting/test_check_estimator_forecasting.py @@ -1,11 +1,9 @@ import numpy as np import pandas as pd import pytest -import sklearn from sklearn.base import clone from sklearn.pipeline import Pipeline from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine.timeseries.forecasting import ( ExpandingWindowFeatures, @@ -22,28 +20,19 @@ ] -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) - -if sklearn_version < parse_version("1.6"): - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator(estimator) - -else: - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - extra_failing_checks = { - "check_estimators_nan_inf": "Time Series transformers do not handle NaNs " - "or infinity." - } - return check_estimator( - estimator=wrap_for_check_estimator(estimator), - expected_failed_checks={ - **extra_failing_checks, - **estimator._more_tags()["_xfail_checks"], - }, - ) +@pytest.mark.parametrize("estimator", _estimators) +def test_check_estimator_from_sklearn(estimator): + extra_failing_checks = { + "check_estimators_nan_inf": "Time Series transformers do not handle NaNs " + "or infinity." + } + return check_estimator( + estimator=wrap_for_check_estimator(estimator), + expected_failed_checks={ + **extra_failing_checks, + **estimator._more_tags()["_xfail_checks"], + }, + ) @pytest.mark.parametrize("estimator", _estimators) diff --git a/tests/test_transformation/test_arcsin_transformer.py b/tests/test_transformation/test_arcsin_transformer.py index e476c27d5..b8a161132 100644 --- a/tests/test_transformation/test_arcsin_transformer.py +++ b/tests/test_transformation/test_arcsin_transformer.py @@ -1,68 +1,83 @@ +import narwhals as nw +import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.transformation import ArcsinTransformer - -def test_transform_and_inverse_transform(df_vartypes): +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} +DATA_NA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20.0, 21.0, 19.0, np.nan], + "Marks": [0.9, 0.8, 0.7, np.nan], +} + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_and_inverse_transform(make_df): + X = make_df(DATA) transformer = ArcsinTransformer(variables=["Marks"]) - X = transformer.fit_transform(df_vartypes) - - # expected output - transf_df = df_vartypes.copy() - transf_df["Marks"] = [1.24905, 1.10715, 0.99116, 0.88607] - - # test transform output - pd.testing.assert_frame_equal(X, transf_df) - - # test inverse_transform - Xit = transformer.inverse_transform(X) + Xt = transformer.fit_transform(X) - # convert numbers to original format. - Xit["Marks"] = Xit["Marks"].round(1) + result = nw.from_native(Xt, eager_only=True).to_dict(as_series=False) + assert result["Marks"] == pytest.approx( + [1.24905, 1.10715, 0.99116, 0.88607], abs=1e-5 + ) - # test - pd.testing.assert_frame_equal(Xit, df_vartypes) + Xit = transformer.inverse_transform(Xt) + result_it = nw.from_native(Xit, eager_only=True).to_dict(as_series=False) + assert [round(v, 1) for v in result_it["Marks"]] == DATA["Marks"] -def test_fit_raises_error_if_na_in_df(df_na): - # test case 2: when dataset contains na, fit method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_raises_error_if_na_in_df(make_df): + X = make_df(DATA_NA) transformer = ArcsinTransformer(variables=["Marks"]) with pytest.raises(ValueError): - transformer.fit(df_na) + transformer.fit(X) -def test_transform_raises_error_if_na_in_df(df_vartypes, df_na): - # test case 3: when dataset contains na, transform method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_raises_error_if_na_in_df(make_df): + X = make_df(DATA) + X_na = make_df(DATA_NA) transformer = ArcsinTransformer(variables=["Marks"]) - transformer.fit(df_vartypes) + transformer.fit(X) with pytest.raises(ValueError): - transformer.transform(df_na[df_vartypes.columns]) + transformer.transform(X_na) -def test_error_if_df_contains_outside_range_values(df_vartypes): - # test error when data contains value outside range [0, +1] - df_out_range = df_vartypes.copy() - df_out_range.loc[1, "Marks"] = 2 +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_df_contains_outside_range_values(make_df): + data_out_range = dict(DATA) + data_out_range["Marks"] = [0.9, 2, 0.7, 0.6] + X = make_df(DATA) + X_out_range = make_df(data_out_range) transformer = ArcsinTransformer(variables=["Marks"]) - # test case 4: when variable contains value outside range, fit with pytest.raises(ValueError): - transformer.fit(df_out_range) + transformer.fit(X_out_range) - # test case 5: when variable contains value outside range, transform - transformer.fit(df_vartypes) + transformer.fit(X) with pytest.raises(ValueError): - transformer.transform(df_out_range) + transformer.transform(X_out_range) - # when selecting variables automatically and some are outside range transformer = ArcsinTransformer() with pytest.raises(ValueError): - transformer.fit(df_vartypes) + transformer.fit(X_out_range) -def test_non_fitted_error(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df): + X = make_df(DATA) transformer = ArcsinTransformer(variables="Marks") with pytest.raises(NotFittedError): - transformer.transform(df_vartypes) + transformer.transform(X) diff --git a/tests/test_transformation/test_arcsinh.py b/tests/test_transformation/test_arcsinh.py index a3d8b8d4d..aa90b10af 100644 --- a/tests/test_transformation/test_arcsinh.py +++ b/tests/test_transformation/test_arcsinh.py @@ -1,128 +1,142 @@ +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from feature_engine.transformation import ArcSinhTransformer +DATA_NUMERICAL = { + "a": [-100.0, -10.0, 0.0, 10.0, 100.0], + "b": [1.0, 2.0, 3.0, 4.0, 5.0], +} +DATA_MULTI_COLUMN = { + "a": [1.0, 2.0, 3.0], + "b": [4.0, 5.0, 6.0], + "c": [7.0, 8.0, 9.0], +} -@pytest.fixture -def df_numerical(): - """Fixture providing sample numerical data with positive and negative values.""" - return pd.DataFrame({ - "a": [-100, -10, 0, 10, 100], - "b": [1, 2, 3, 4, 5], - }) +def _col(X, name): + return nw.from_native(X, eager_only=True).get_column(name).to_numpy() -@pytest.fixture -def df_multi_column(): - """Fixture providing DataFrame with multiple columns.""" - return pd.DataFrame({ - "a": [1, 2, 3], - "b": [4, 5, 6], - "c": [7, 8, 9], - }) - -def test_default_parameters(df_numerical): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_default_parameters(make_df): """Test transformer with default parameters applies arcsinh to all columns.""" + X = make_df(DATA_NUMERICAL) transformer = ArcSinhTransformer() - X_tr = transformer.fit_transform(df_numerical.copy()) + X_tr = transformer.fit_transform(X) - expected_a = np.arcsinh(df_numerical["a"]) - expected_b = np.arcsinh(df_numerical["b"]) - np.testing.assert_array_almost_equal(X_tr["a"], expected_a) - np.testing.assert_array_almost_equal(X_tr["b"], expected_b) + expected_a = np.arcsinh(np.array(DATA_NUMERICAL["a"])) + expected_b = np.arcsinh(np.array(DATA_NUMERICAL["b"])) + np.testing.assert_array_almost_equal(_col(X_tr, "a"), expected_a) + np.testing.assert_array_almost_equal(_col(X_tr, "b"), expected_b) -def test_specific_variables(df_multi_column): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_specific_variables(make_df): """Test transformer with specific variables selected.""" + X = make_df(DATA_MULTI_COLUMN) transformer = ArcSinhTransformer(variables=["a", "b"]) - X_tr = transformer.fit_transform(df_multi_column.copy()) + X_tr = transformer.fit_transform(X) np.testing.assert_array_almost_equal( - X_tr["a"], np.arcsinh(df_multi_column["a"]) + _col(X_tr, "a"), np.arcsinh(np.array(DATA_MULTI_COLUMN["a"])) ) np.testing.assert_array_almost_equal( - X_tr["b"], np.arcsinh(df_multi_column["b"]) + _col(X_tr, "b"), np.arcsinh(np.array(DATA_MULTI_COLUMN["b"])) ) - np.testing.assert_array_equal(X_tr["c"], df_multi_column["c"]) + np.testing.assert_array_equal(_col(X_tr, "c"), np.array(DATA_MULTI_COLUMN["c"])) -def test_with_loc_and_scale(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_with_loc_and_scale(make_df): """Test transformer with loc and scale parameters.""" - X = pd.DataFrame({"a": [10, 20, 30, 40, 50]}) + data = {"a": [10.0, 20.0, 30.0, 40.0, 50.0]} + X = make_df(data) loc = 30.0 scale = 10.0 transformer = ArcSinhTransformer(loc=loc, scale=scale) - X_tr = transformer.fit_transform(X.copy()) + X_tr = transformer.fit_transform(X) - expected = np.arcsinh((X["a"] - loc) / scale) - np.testing.assert_array_almost_equal(X_tr["a"], expected) - np.testing.assert_almost_equal(X_tr["a"].iloc[2], 0.0, decimal=10) + expected = np.arcsinh((np.array(data["a"]) - loc) / scale) + np.testing.assert_array_almost_equal(_col(X_tr, "a"), expected) + np.testing.assert_almost_equal(_col(X_tr, "a")[2], 0.0, decimal=10) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("loc", [0.0, 10.0, -10.0, 100.5]) -def test_various_loc_values(loc): +def test_various_loc_values(make_df, loc): """Test that various loc values work correctly.""" - X = pd.DataFrame({"a": [1, 2, 3, 4, 5]}) + data = {"a": [1.0, 2.0, 3.0, 4.0, 5.0]} + X = make_df(data) transformer = ArcSinhTransformer(loc=loc) - X_tr = transformer.fit_transform(X.copy()) + X_tr = transformer.fit_transform(X) - expected = np.arcsinh((X["a"] - loc) / 1.0) - np.testing.assert_array_almost_equal(X_tr["a"], expected) + expected = np.arcsinh((np.array(data["a"]) - loc) / 1.0) + np.testing.assert_array_almost_equal(_col(X_tr, "a"), expected) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("scale", [0.5, 1.0, 2.0, 10.0, 100.0]) -def test_various_scale_values(scale): +def test_various_scale_values(make_df, scale): """Test that various scale values work correctly.""" - X = pd.DataFrame({"a": [1, 2, 3, 4, 5]}) + data = {"a": [1.0, 2.0, 3.0, 4.0, 5.0]} + X = make_df(data) transformer = ArcSinhTransformer(scale=scale) - X_tr = transformer.fit_transform(X.copy()) + X_tr = transformer.fit_transform(X) - expected = np.arcsinh((X["a"] - 0.0) / scale) - np.testing.assert_array_almost_equal(X_tr["a"], expected) + expected = np.arcsinh((np.array(data["a"]) - 0.0) / scale) + np.testing.assert_array_almost_equal(_col(X_tr, "a"), expected) -def test_inverse_transform(df_numerical): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_inverse_transform(make_df): """Test inverse_transform returns original values.""" - X_original = df_numerical.copy() + X = make_df(DATA_NUMERICAL) transformer = ArcSinhTransformer() - X_tr = transformer.fit_transform(df_numerical.copy()) + X_tr = transformer.fit_transform(X) X_inv = transformer.inverse_transform(X_tr) - np.testing.assert_array_almost_equal(X_inv["a"], X_original["a"], decimal=10) - np.testing.assert_array_almost_equal(X_inv["b"], X_original["b"], decimal=10) + np.testing.assert_array_almost_equal( + _col(X_inv, "a"), np.array(DATA_NUMERICAL["a"]), decimal=10 + ) + np.testing.assert_array_almost_equal( + _col(X_inv, "b"), np.array(DATA_NUMERICAL["b"]), decimal=10 + ) -def test_inverse_transform_with_loc_scale(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_inverse_transform_with_loc_scale(make_df): """Test inverse_transform with loc and scale parameters.""" - X = pd.DataFrame({"a": [10, 20, 30, 40, 50]}) - X_original = X.copy() + data = {"a": [10.0, 20.0, 30.0, 40.0, 50.0]} + X = make_df(data) transformer = ArcSinhTransformer(loc=25.0, scale=5.0) - X_tr = transformer.fit_transform(X.copy()) + X_tr = transformer.fit_transform(X) X_inv = transformer.inverse_transform(X_tr) - np.testing.assert_array_almost_equal(X_inv["a"], X_original["a"], decimal=10) + np.testing.assert_array_almost_equal( + _col(X_inv, "a"), np.array(data["a"]), decimal=10 + ) -def test_negative_values(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_negative_values(make_df): """Test that transformer handles negative values correctly.""" - X = pd.DataFrame({"a": [-1000, -500, 0, 500, 1000]}) + data = {"a": [-1000.0, -500.0, 0.0, 500.0, 1000.0]} + X = make_df(data) transformer = ArcSinhTransformer() - X_tr = transformer.fit_transform(X.copy()) + X_tr = transformer.fit_transform(X) # Expected values: arcsinh([ -1000, -500, 0, 500, 1000 ]) expected = [-7.600902, -6.907755, 0.0, 6.907755, 7.600902] - np.testing.assert_array_almost_equal(X_tr["a"], expected, decimal=5) + result = _col(X_tr, "a") + np.testing.assert_array_almost_equal(result, expected, decimal=5) # Verify symmetry property: arcsinh(-x) = -arcsinh(x) - np.testing.assert_almost_equal( - X_tr["a"].iloc[0], -X_tr["a"].iloc[4], decimal=10 - ) - np.testing.assert_almost_equal( - X_tr["a"].iloc[1], -X_tr["a"].iloc[3], decimal=10 - ) + np.testing.assert_almost_equal(result[0], -result[4], decimal=10) + np.testing.assert_almost_equal(result[1], -result[3], decimal=10) @pytest.mark.parametrize("invalid_scale", [0, -1, -0.5, -100, "string", False]) @@ -139,9 +153,10 @@ def test_invalid_loc_raises_error(invalid_loc): ArcSinhTransformer(loc=invalid_loc) -def test_fit_stores_attributes(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_stores_attributes(make_df): """Test that fit stores expected attributes with correct values.""" - X = pd.DataFrame({"a": [1, 2, 3], "b": [4, 5, 6]}) + X = make_df({"a": [1.0, 2.0, 3.0], "b": [4.0, 5.0, 6.0]}) transformer = ArcSinhTransformer() transformer.fit(X) @@ -153,9 +168,10 @@ def test_fit_stores_attributes(): assert transformer.feature_names_in_ == ["a", "b"] -def test_get_feature_names_out(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_get_feature_names_out(make_df): """Test get_feature_names_out returns correct feature names.""" - X = pd.DataFrame({"a": [1, 2, 3], "b": [4, 5, 6]}) + X = make_df({"a": [1.0, 2.0, 3.0], "b": [4.0, 5.0, 6.0]}) transformer = ArcSinhTransformer() transformer.fit(X) @@ -163,9 +179,10 @@ def test_get_feature_names_out(): assert feature_names == ["a", "b"] -def test_get_feature_names_out_with_subset(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_get_feature_names_out_with_subset(make_df): """Test get_feature_names_out with subset of variables.""" - X = pd.DataFrame({"a": [1, 2, 3], "b": [4, 5, 6], "c": [7, 8, 9]}) + X = make_df({"a": [1.0, 2.0, 3.0], "b": [4.0, 5.0, 6.0], "c": [7.0, 8.0, 9.0]}) transformer = ArcSinhTransformer(variables=["a"]) transformer.fit(X) @@ -173,29 +190,36 @@ def test_get_feature_names_out_with_subset(): assert feature_names == ["a", "b", "c"] -def test_behavior_like_log_for_large_values(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_behavior_like_log_for_large_values(make_df): """Test that arcsinh behaves like log for large positive values.""" - X = pd.DataFrame({"a": [1000, 10000, 100000]}) + data = {"a": [1000.0, 10000.0, 100000.0]} + X = make_df(data) transformer = ArcSinhTransformer() - X_tr = transformer.fit_transform(X.copy()) + X_tr = transformer.fit_transform(X) - log_approx = np.log(2 * X["a"]) - np.testing.assert_array_almost_equal(X_tr["a"], log_approx, decimal=1) + log_approx = np.log(2 * np.array(data["a"])) + np.testing.assert_array_almost_equal(_col(X_tr, "a"), log_approx, decimal=1) -def test_behavior_like_identity_for_small_values(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_behavior_like_identity_for_small_values(make_df): """Test that arcsinh behaves like identity for small values.""" - X = pd.DataFrame({"a": [0.001, 0.01, 0.1]}) + data = {"a": [0.001, 0.01, 0.1]} + X = make_df(data) transformer = ArcSinhTransformer() - X_tr = transformer.fit_transform(X.copy()) + X_tr = transformer.fit_transform(X) - np.testing.assert_array_almost_equal(X_tr["a"], X["a"], decimal=2) + np.testing.assert_array_almost_equal( + _col(X_tr, "a"), np.array(data["a"]), decimal=2 + ) -def test_zero_input_returns_zero(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_zero_input_returns_zero(make_df): """Test that arcsinh(0) = 0.""" - X = pd.DataFrame({"a": [0.0]}) + X = make_df({"a": [0.0]}) transformer = ArcSinhTransformer() - X_tr = transformer.fit_transform(X.copy()) + X_tr = transformer.fit_transform(X) - assert X_tr["a"].iloc[0] == 0.0 + assert _col(X_tr, "a")[0] == 0.0 diff --git a/tests/test_transformation/test_boxcox_transformer.py b/tests/test_transformation/test_boxcox_transformer.py index 25fd20c40..172ea03f6 100644 --- a/tests/test_transformation/test_boxcox_transformer.py +++ b/tests/test_transformation/test_boxcox_transformer.py @@ -1,72 +1,98 @@ +import narwhals as nw import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.transformation import BoxCoxTransformer +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} -def test_automatically_finds_variables(df_vartypes): - # test case 1: automatically select variables - transformer = BoxCoxTransformer(variables=None) - X = transformer.fit_transform(df_vartypes) +DATA_NA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, None, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} + + +def assert_df_equal(X, expected: dict, abs_tol: float = 1e-5) -> None: + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + assert list(result.keys()) == list(expected.keys()) + for col, values in expected.items(): + assert result[col] == pytest.approx(values, abs=abs_tol, nan_ok=True) - # expected output - transf_df = df_vartypes.copy() - transf_df["Age"] = [9.78731, 10.1666, 9.40189, 9.0099] - transf_df["Marks"] = [-0.101687, -0.207092, -0.316843, -0.431788] + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_automatically_finds_variables_and_inverse_transform(make_df): + df = make_df(DATA) + + transformer = BoxCoxTransformer(variables=None) + X = transformer.fit_transform(df) # test init params assert transformer.variables is None # test fit attr assert transformer.variables_ == ["Age", "Marks"] - assert transformer.n_features_in_ == 5 - # test transform output - pd.testing.assert_frame_equal(X, transf_df) + assert transformer.n_features_in_ == 4 + + expected = dict(DATA) + expected["Age"] = [9.78731, 10.1666, 9.40189, 9.0099] + expected["Marks"] = [-0.101687, -0.207092, -0.316843, -0.431788] + assert_df_equal(X, expected) # test inverse_transform Xit = transformer.inverse_transform(X) - - # convert numbers to original format. - Xit["Age"] = Xit["Age"].round().astype("int64") - Xit["Marks"] = Xit["Marks"].round(1) - - # test - pd.testing.assert_frame_equal(Xit, df_vartypes) + result = nw.from_native(Xit, eager_only=True).to_dict(as_series=False) + assert [round(v) for v in result["Age"]] == DATA["Age"] + assert [round(v, 1) for v in result["Marks"]] == DATA["Marks"] -def test_fit_raises_error_if_df_contains_na(df_na): - # test case 2: when dataset contains na, fit method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_raises_error_if_df_contains_na(make_df): + df_na = make_df(DATA_NA) transformer = BoxCoxTransformer() with pytest.raises(ValueError): transformer.fit(df_na) -def test_transform_raises_error_if_df_contains_na(df_vartypes, df_na): - # test case 3: when dataset contains na, transform method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_raises_error_if_df_contains_na(make_df): + df = make_df(DATA) + df_na = make_df(DATA_NA) transformer = BoxCoxTransformer() - transformer.fit(df_vartypes) + transformer.fit(df) with pytest.raises(ValueError): - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.transform(df_na) -def test_error_if_df_contains_negative_values(df_vartypes): - # test error when data contains negative values - df_neg = df_vartypes.copy() - df_neg.loc[1, "Age"] = -1 +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_df_contains_negative_values(make_df): + data_neg = {k: list(v) for k, v in DATA.items()} + data_neg["Age"][1] = -1 + df_neg = make_df(data_neg) + df = make_df(DATA) - # test case 4: when variable contains negative value, fit + # when variable contains negative value, fit transformer = BoxCoxTransformer() with pytest.raises(ValueError): transformer.fit(df_neg) - # test case 5: when variable contains negative value, transform + # when variable contains negative value, transform transformer = BoxCoxTransformer() - transformer.fit(df_vartypes) + transformer.fit(df) with pytest.raises(ValueError): transformer.transform(df_neg) -def test_non_fitted_error(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df): + df = make_df(DATA) transformer = BoxCoxTransformer() with pytest.raises(NotFittedError): - transformer.transform(df_vartypes) + transformer.transform(df) diff --git a/tests/test_transformation/test_check_estimator_transformers.py b/tests/test_transformation/test_check_estimator_transformers.py index e2b82ff69..738e160fa 100644 --- a/tests/test_transformation/test_check_estimator_transformers.py +++ b/tests/test_transformation/test_check_estimator_transformers.py @@ -1,9 +1,8 @@ import pandas as pd import pytest -import sklearn +from sklearn.base import clone from sklearn.pipeline import Pipeline from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine.transformation import ( ArcsinTransformer, @@ -32,53 +31,45 @@ YeoJohnsonTransformer(), ] -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) - -if sklearn_version < parse_version("1.6"): - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - return check_estimator(estimator) - -else: - checks_with_negative_values = [ - "check_readonly_memmap_input", - "check_fit_score_takes_y", - "check_dont_overwrite_parameters", - "check_estimators_nan_inf", - "check_f_contiguous_array_estimator", - "check_fit2d_1feature", - "check_fit2d_1sample", - "check_dict_unchanged", - "check_fit_check_is_fitted", - "check_n_features_in", - "check_positive_only_tag_during_fit", - "check_methods_subset_invariance", - ] - estimators_not_supporting_negative_values = [ - "BoxCoxTransformer", - "LogTransformer", - "ArcsinTransformer", - ] - extra_failing_checks = { - estimator_name: dict.fromkeys( - checks_with_negative_values, - "this checks passes a negative value which is not supported by " - "the transformer", - ) - for estimator_name in estimators_not_supporting_negative_values - } - - @pytest.mark.parametrize("estimator", _estimators) - def test_check_estimator_from_sklearn(estimator): - expected_failed_checks = estimator._more_tags()["_xfail_checks"] - expected_failed_checks.update( - extra_failing_checks.get(estimator.__class__.__name__, {}) - ) - return check_estimator( - estimator=wrap_for_check_estimator(estimator), - expected_failed_checks=expected_failed_checks, - ) +checks_with_negative_values = [ + "check_readonly_memmap_input", + "check_fit_score_takes_y", + "check_dont_overwrite_parameters", + "check_estimators_nan_inf", + "check_f_contiguous_array_estimator", + "check_fit2d_1feature", + "check_fit2d_1sample", + "check_dict_unchanged", + "check_fit_check_is_fitted", + "check_n_features_in", + "check_positive_only_tag_during_fit", + "check_methods_subset_invariance", +] +estimators_not_supporting_negative_values = [ + "BoxCoxTransformer", + "LogTransformer", + "ArcsinTransformer", +] +extra_failing_checks = { + estimator_name: dict.fromkeys( + checks_with_negative_values, + "this checks passes a negative value which is not supported by " + "the transformer", + ) + for estimator_name in estimators_not_supporting_negative_values +} + + +@pytest.mark.parametrize("estimator", _estimators) +def test_check_estimator_from_sklearn(estimator): + expected_failed_checks = estimator._more_tags()["_xfail_checks"] + expected_failed_checks.update( + extra_failing_checks.get(estimator.__class__.__name__, {}) + ) + return check_estimator( + estimator=wrap_for_check_estimator(estimator), + expected_failed_checks=expected_failed_checks, + ) @pytest.mark.parametrize("estimator", _estimators[4:]) @@ -126,3 +117,15 @@ def test_raises_non_fitted_error_when_error_during_fit(estimator): X = pd.DataFrame({"cat1": ["a", "b", "c", "a", "b"]}) check_raises_non_fitted_error_when_fit_fails(estimator, X) + + +@pytest.mark.parametrize("estimator", _estimators) +def test_integer_column_names(estimator): + # integer column names are pandas-only + X = pd.DataFrame({0: [0.1, 0.3, 0.5, 0.9], 1: [0.2, 0.4, 0.6, 0.8]}) + transformer = clone(estimator) + Xt = transformer.fit_transform(X) + expected = clone(estimator).fit_transform(X.rename(columns=str)) + + pd.testing.assert_frame_equal(Xt, expected.set_axis([0, 1], axis=1)) + pd.testing.assert_frame_equal(transformer.inverse_transform(Xt), X) diff --git a/tests/test_transformation/test_log_transformer.py b/tests/test_transformation/test_log_transformer.py index 23a74104c..757becf5d 100644 --- a/tests/test_transformation/test_log_transformer.py +++ b/tests/test_transformation/test_log_transformer.py @@ -1,175 +1,195 @@ +import re + +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.transformation import LogTransformer - -def test_transforming_int_vars(): - df = pd.DataFrame( - { - "var1": [1, 2, 3], - "var2": [4, 5, 3], - } - ) - dft = np.log(df) +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} +DATA_NA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20.0, 21.0, 19.0, np.nan], + "Marks": [0.9, 0.8, 0.7, np.nan], +} +DATA_C = { + "vara": [0, 1, 2, 3], + "varb": [5, 5, 6, 7], + "varc": [-2, -1, 0, 4], + "vard": [-3, -2, -1, -5], + "vare": ["a", "b", "c", "d"], +} +DATA_C_VARS = ["vara", "varb", "varc", "vard"] +DATA_C_AUTO = {"vara": 1, "varb": 0, "varc": 3, "vard": 6} + + +def _to_dict(X): + return nw.from_native(X, eager_only=True).to_dict(as_series=False) + + +def _expected_log(c, base): + fn = np.log if base == "e" else np.log10 + out = {} + for var in DATA_C_VARS: + c_var = c[var] if isinstance(c, dict) else c + out[var] = [fn(x + c_var) for x in DATA_C[var]] + return out + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transforming_int_vars(make_df): + X = make_df({"var1": [1, 2, 3], "var2": [4, 5, 3]}) transformer = LogTransformer(base="e", variables=None) - X = transformer.fit_transform(df) - pd.testing.assert_frame_equal(X, dft) + Xt = transformer.fit_transform(X) + result = _to_dict(Xt) + assert result["var1"] == pytest.approx(list(np.log([1, 2, 3]))) + assert result["var2"] == pytest.approx(list(np.log([4, 5, 3]))) -def test_log_base_e_plus_automatically_find_variables(df_vartypes): - # test case 1: log base e, automatically select variables +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_log_base_e_plus_automatically_find_variables(make_df): + X = make_df(DATA) transformer = LogTransformer(base="e", variables=None) - X = transformer.fit_transform(df_vartypes) - - # expected output - transf_df = df_vartypes.copy() - transf_df["Age"] = [2.99573, 3.04452, 2.94444, 2.89037] - transf_df["Marks"] = [-0.105361, -0.223144, -0.356675, -0.510826] + Xt = transformer.fit_transform(X) # test init params assert transformer.base == "e" assert transformer.variables is None # test fit attr assert transformer.variables_ == ["Age", "Marks"] - assert transformer.n_features_in_ == 5 + assert transformer.n_features_in_ == 4 + # test transform output - pd.testing.assert_frame_equal(X, transf_df) + result = _to_dict(Xt) + assert result["Age"] == pytest.approx( + [2.99573, 3.04452, 2.94444, 2.89037], abs=1e-5 + ) + assert result["Marks"] == pytest.approx( + [-0.105361, -0.223144, -0.356675, -0.510826], abs=1e-5 + ) # test inverse_transform - Xit = transformer.inverse_transform(X) - - # convert numbers to original format. - Xit["Age"] = Xit["Age"].round().astype("int64") - Xit["Marks"] = Xit["Marks"].round(1) + Xit = transformer.inverse_transform(Xt) + result_it = _to_dict(Xit) + assert [round(v) for v in result_it["Age"]] == DATA["Age"] + assert [round(v, 1) for v in result_it["Marks"]] == DATA["Marks"] - # test - pd.testing.assert_frame_equal(Xit, df_vartypes) - -def test_log_base_10_plus_user_passes_var_list(df_vartypes): - # test case 2: log base 10, user passes variables +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_log_base_10_plus_user_passes_var_list(make_df): + X = make_df(DATA) transformer = LogTransformer(base="10", variables="Age") - X = transformer.fit_transform(df_vartypes) - - # expected output - transf_df = df_vartypes.copy() - transf_df["Age"] = [1.30103, 1.32222, 1.27875, 1.25527] + Xt = transformer.fit_transform(X) # test init params assert transformer.base == "10" assert transformer.variables == "Age" # test fit attr assert transformer.variables_ == ["Age"] - assert transformer.n_features_in_ == 5 + assert transformer.n_features_in_ == 4 + # test transform output - pd.testing.assert_frame_equal(X, transf_df) + result = _to_dict(Xt) + assert result["Age"] == pytest.approx( + [1.30103, 1.32222, 1.27875, 1.25527], abs=1e-5 + ) # test inverse_transform - Xit = transformer.inverse_transform(X) - - # convert numbers to original format. - Xit["Age"] = Xit["Age"].round().astype("int64") - - # test - pd.testing.assert_frame_equal(Xit, df_vartypes) + Xit = transformer.inverse_transform(Xt) + result_it = _to_dict(Xit) + assert [round(v) for v in result_it["Age"]] == DATA["Age"] def test_error_if_base_value_not_allowed(): - with pytest.raises(ValueError) as record: + msg = "base can take only '10' or 'e' as values. Got other instead." + with pytest.raises(ValueError, match=re.escape(msg)): LogTransformer(base="other") - assert str(record.value) == ( - "base can take only '10' or 'e' as values. Got other instead." - ) -def test_fit_raises_error_if_na_in_df(df_na): - # test case 3: when dataset contains na, fit method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_raises_error_if_na_in_df(make_df): + X = make_df(DATA_NA) with pytest.raises(ValueError): transformer = LogTransformer() - transformer.fit(df_na) + transformer.fit(X) -def test_transform_raises_error_if_na_in_df(df_vartypes, df_na): - # test case 4: when dataset contains na, transform method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_raises_error_if_na_in_df(make_df): + X = make_df(DATA) + X_na = make_df(DATA_NA) + transformer = LogTransformer() + transformer.fit(X) with pytest.raises(ValueError): - transformer = LogTransformer() - transformer.fit(df_vartypes) - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.transform(X_na) -def test_error_if_df_contains_negative_values(df_vartypes): - # test error when data contains negative values - df_neg = df_vartypes.copy() - df_neg.loc[1, "Age"] = -1 +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_df_contains_negative_values(make_df): + data_neg = dict(DATA) + data_neg["Age"] = [20, -1, 19, 18] + X = make_df(DATA) + X_neg = make_df(data_neg) - # test case 5: when variable contains negative value, fit + # when variable contains negative value, fit with pytest.raises(ValueError): transformer = LogTransformer() - transformer.fit(df_neg) + transformer.fit(X_neg) - # test case 6: when variable contains negative value, transform + # when variable contains negative value, transform with pytest.raises(ValueError): transformer = LogTransformer() - transformer.fit(df_vartypes) - transformer.transform(df_neg) + transformer.fit(X) + transformer.transform(X_neg) -def test_non_fitted_error(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df): + X = make_df(DATA) with pytest.raises(NotFittedError): transformer = LogTransformer() - transformer.transform(df_vartypes) + transformer.transform(X) -def test_inverse_e_plus_user_passes_var_list(df_vartypes): - # test case 7: inverse log, user passes variables +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_inverse_e_plus_user_passes_var_list(make_df): + X = make_df(DATA) transformer = LogTransformer(variables="Age") - Xt = transformer.fit_transform(df_vartypes) - X = transformer.inverse_transform(Xt) - - # convert floats to int - X["Age"] = X["Age"].round().astype("int64") + Xt = transformer.fit_transform(X) + Xit = transformer.inverse_transform(Xt) # test init params assert transformer.base == "e" assert transformer.variables == "Age" # test fit attr assert transformer.variables_ == ["Age"] - assert transformer.n_features_in_ == 5 + assert transformer.n_features_in_ == 4 # test transform output - pd.testing.assert_frame_equal(X, df_vartypes) + result_it = _to_dict(Xit) + assert [round(v) for v in result_it["Age"]] == DATA["Age"] -def test_default_C_preserves_original_fail_fast_behavior(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_default_C_preserves_original_fail_fast_behavior(make_df): """LogTransformer()'s default C=0 must raise at fit() time, with the original exact message, matching pre-merge behavior. See #957.""" - df = pd.DataFrame({"x": [1, 2, 0, 4]}) + X = make_df({"x": [1, 2, 0, 4]}) tr = LogTransformer() assert tr.C == 0 - with pytest.raises(ValueError) as record: - tr.fit(df) - - assert str(record.value) == ( - "Some variables contain zero or negative values, can't apply log" - ) - - -@pytest.fixture(scope="module") -def df_c(): - df = pd.DataFrame( - { - "vara": [0, 1, 2, 3], - "varb": [5, 5, 6, 7], - "varc": [-2, -1, 0, 4], - "vard": [-3, -2, -1, -5], - "vare": ["a", "b", "c", "d"], - } - ) - return df + msg = "Some variables contain zero or negative values, can't apply log" + with pytest.raises(ValueError, match=re.escape(msg)): + tr.fit(X) @pytest.mark.parametrize("c", [1, 0.1, {"var1": 1, "var2": 2}, "auto"]) @@ -181,86 +201,86 @@ def test_c_parameter(c): @pytest.mark.parametrize("c", ["string", [1, 2]]) def test_c_raises_error(c): msg = f"C can take only 'auto', integers, floats or dictionaries. Got {c} instead." - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(msg)): LogTransformer(C=c) - assert str(record.value) == msg -def test_C_when_auto(df_c): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_C_when_auto(make_df): + X = make_df(DATA_C) tr = LogTransformer(C="auto") - tr.fit(df_c) - c = {"vara": 1, "varb": 0, "varc": 3, "vard": 6} - assert tr.C_ == c + tr.fit(X) + assert tr.C_ == DATA_C_AUTO -def test_C_when_dict(df_c): - c = {"vara": 1, "varb": 0, "varc": 3, "vard": 6} - tr = LogTransformer(C=c) - tr.fit(df_c) - assert tr.C_ == c +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_C_when_dict(make_df): + X = make_df(DATA_C) + tr = LogTransformer(C=DATA_C_AUTO) + tr.fit(X) + assert tr.C_ == DATA_C_AUTO -def test_C_when_int(df_c): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_C_when_int(make_df): + X = make_df(DATA_C) tr = LogTransformer(C=10) - tr.fit(df_c) + tr.fit(X) assert tr.C_ == 10 -def test_raises_error_when_transformed_data_has_negative_values_with_C(df_c): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_raises_error_when_transformed_data_has_negative_values_with_C(make_df): + X = make_df(DATA_C) tr = LogTransformer(C="auto") - tr.fit(df_c) - dft = df_c.copy() - dft["vara"] = dft["vara"] - 2 + tr.fit(X) + + data_shifted = dict(DATA_C) + data_shifted["vara"] = [v - 2 for v in DATA_C["vara"]] + Xt = make_df(data_shifted) + msg = ( "Some variables contain zero or negative values after adding constant C, " "can't apply log." ) - with pytest.raises(ValueError) as record: - tr.transform(dft) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + tr.transform(Xt) -def test_log_base_e_with_C(df_c): - dft = LogTransformer(C="auto").fit_transform(df_c) - exp = np.log( - df_c[["vara", "varb", "varc", "vard"]] - + {"vara": 1, "varb": 0, "varc": 3, "vard": 6} - ) - exp["vare"] = df_c["vare"] - pd.testing.assert_frame_equal(dft, exp) +@pytest.mark.parametrize("base", ["e", "10"]) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_log_with_C(make_df, base): + X = make_df(DATA_C) - dft = LogTransformer(C=10).fit_transform(df_c) - exp = np.log(df_c[["vara", "varb", "varc", "vard"]] + 10) - exp["vare"] = df_c["vare"] - pd.testing.assert_frame_equal(dft, exp) + dft = LogTransformer(C="auto", base=base).fit_transform(X) + result = _to_dict(dft) + expected = _expected_log(DATA_C_AUTO, base) + for var in DATA_C_VARS: + assert result[var] == pytest.approx(expected[var], abs=1e-6) + assert result["vare"] == DATA_C["vare"] + dft = LogTransformer(C=10, base=base).fit_transform(X) + result = _to_dict(dft) + expected = _expected_log(10, base) + for var in DATA_C_VARS: + assert result[var] == pytest.approx(expected[var], abs=1e-6) + assert result["vare"] == DATA_C["vare"] -def test_log_base_10_with_C(df_c): - dft = LogTransformer(C="auto", base="10").fit_transform(df_c) - exp = np.log10( - df_c[["vara", "varb", "varc", "vard"]] - + {"vara": 1, "varb": 0, "varc": 3, "vard": 6} - ) - exp["vare"] = df_c["vare"] - pd.testing.assert_frame_equal(dft, exp) - dft = LogTransformer(C=10, base="10").fit_transform(df_c) - exp = np.log10(df_c[["vara", "varb", "varc", "vard"]] + 10) - exp["vare"] = df_c["vare"] - pd.testing.assert_frame_equal(dft, exp) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_inverse_transform_with_C(make_df): + X = make_df(DATA_C) - -def test_inverse_transform_with_C(df_c): tr = LogTransformer(C="auto", base="10") - dft = tr.fit_transform(df_c) + dft = tr.fit_transform(X) orig = tr.inverse_transform(dft) - pd.testing.assert_frame_equal( - orig, df_c, check_dtype=False, check_exact=False, rtol=0.1 - ) + result = _to_dict(orig) + for var in DATA_C_VARS: + assert result[var] == pytest.approx(DATA_C[var], abs=0.1) tr = LogTransformer(C=10, base="e") - dft = tr.fit_transform(df_c) + dft = tr.fit_transform(X) orig = tr.inverse_transform(dft) - pd.testing.assert_frame_equal( - orig, df_c, check_dtype=False, check_exact=False, rtol=0.1 - ) + result = _to_dict(orig) + for var in DATA_C_VARS: + assert result[var] == pytest.approx(DATA_C[var], abs=0.1) diff --git a/tests/test_transformation/test_logcp_transformer.py b/tests/test_transformation/test_logcp_transformer.py index 753cc43bf..98e154735 100644 --- a/tests/test_transformation/test_logcp_transformer.py +++ b/tests/test_transformation/test_logcp_transformer.py @@ -1,10 +1,50 @@ +import re + +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.transformation import LogCpTransformer +DATA = { + "vara": [0, 1, 2, 3], + "varb": [5, 5, 6, 7], + "varc": [-2, -1, 0, 4], + "vard": [-3, -2, -1, -5], + "vare": ["a", "b", "c", "d"], +} +DATA_VARS = ["vara", "varb", "varc", "vard"] +DATA_AUTO_C = {"vara": 1, "varb": 0, "varc": 3, "vard": 6} + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} +DATA_NA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20.0, 21.0, 19.0, np.nan], + "Marks": [0.9, 0.8, 0.7, np.nan], +} + + +def _to_dict(X): + return nw.from_native(X, eager_only=True).to_dict(as_series=False) + + +def _expected_log(c, base): + fn = np.log if base == "e" else np.log10 + out = {} + for var in DATA_VARS: + c_var = c[var] if isinstance(c, dict) else c + out[var] = [fn(x + c_var) for x in DATA[var]] + return out + @pytest.mark.parametrize("base", ["e", "10"]) def test_base_parameter(base): @@ -15,9 +55,8 @@ def test_base_parameter(base): @pytest.mark.parametrize("base", [False, 1, 10]) def test_base_raises_error(base): msg = f"base can take only '10' or 'e' as values. Got {base} instead." - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(msg)): LogCpTransformer(base=base) - assert str(record.value) == msg @pytest.mark.parametrize("c", [1, 0.1, {"var1": 1, "var2": 2}, "auto"]) @@ -29,9 +68,8 @@ def test_c_parameter(c): @pytest.mark.parametrize("c", ["string", [1, 2]]) def test_c_raises_error(c): msg = f"C can take only 'auto', integers, floats or dictionaries. Got {c} instead." - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(msg)): LogCpTransformer(C=c) - assert str(record.value) == msg def test_instantiation_raises_future_warning(): @@ -40,119 +78,111 @@ def test_instantiation_raises_future_warning(): "LogTransformer and will be removed in version 2.1.0. " 'Use LogTransformer(C="auto") instead.' ) - with pytest.warns(FutureWarning) as record: + with pytest.warns(FutureWarning, match=re.escape(msg)): LogCpTransformer() - assert str(record[0].message) == msg - - -@pytest.fixture(scope="module") -def df(): - df = pd.DataFrame( - { - "vara": [0, 1, 2, 3], - "varb": [5, 5, 6, 7], - "varc": [-2, -1, 0, 4], - "vard": [-3, -2, -1, -5], - "vare": ["a", "b", "c", "d"], - } - ) - return df -def test_C_when_auto(df): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_C_when_auto(make_df): + X = make_df(DATA) tr = LogCpTransformer(C="auto") - tr.fit(df) - c = {"vara": 1, "varb": 0, "varc": 3, "vard": 6} - assert tr.C_ == c + tr.fit(X) + assert tr.C_ == DATA_AUTO_C -def test_C_when_dict(df): - c = {"vara": 1, "varb": 0, "varc": 3, "vard": 6} - tr = LogCpTransformer(C=c) - tr.fit(df) - assert tr.C_ == c +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_C_when_dict(make_df): + X = make_df(DATA) + tr = LogCpTransformer(C=DATA_AUTO_C) + tr.fit(X) + assert tr.C_ == DATA_AUTO_C -def test_C_when_int(df): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_C_when_int(make_df): + X = make_df(DATA) tr = LogCpTransformer(C=10) - tr.fit(df) + tr.fit(X) assert tr.C_ == 10 -def test_raises_error_when_transformed_data_has_negative_values(df): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_raises_error_when_transformed_data_has_negative_values(make_df): + X = make_df(DATA) tr = LogCpTransformer(C="auto") - tr.fit(df) - dft = df.copy() - dft["vara"] = dft["vara"] - 2 + tr.fit(X) + + data_shifted = dict(DATA) + data_shifted["vara"] = [v - 2 for v in DATA["vara"]] + Xt = make_df(data_shifted) + msg = ( "Some variables contain zero or negative values after adding constant C, " "can't apply log." ) - with pytest.raises(ValueError) as record: - tr.transform(dft) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + tr.transform(Xt) -def test_log_base_e(df): - dft = LogCpTransformer(C="auto").fit_transform(df) - exp = np.log( - df[["vara", "varb", "varc", "vard"]] - + {"vara": 1, "varb": 0, "varc": 3, "vard": 6} - ) - exp["vare"] = df["vare"] - pd.testing.assert_frame_equal(dft, exp) - - dft = LogCpTransformer(C=10).fit_transform(df) - exp = np.log(df[["vara", "varb", "varc", "vard"]] + 10) - exp["vare"] = df["vare"] - pd.testing.assert_frame_equal(dft, exp) +@pytest.mark.parametrize("base", ["e", "10"]) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_log_with_C(make_df, base): + X = make_df(DATA) + dft = LogCpTransformer(C="auto", base=base).fit_transform(X) + result = _to_dict(dft) + expected = _expected_log(DATA_AUTO_C, base) + for var in DATA_VARS: + assert result[var] == pytest.approx(expected[var], abs=1e-6) + assert result["vare"] == DATA["vare"] -def test_log_base_10(df): - dft = LogCpTransformer(C="auto", base="10").fit_transform(df) - exp = np.log10( - df[["vara", "varb", "varc", "vard"]] - + {"vara": 1, "varb": 0, "varc": 3, "vard": 6} - ) - exp["vare"] = df["vare"] - pd.testing.assert_frame_equal(dft, exp) + dft = LogCpTransformer(C=10, base=base).fit_transform(X) + result = _to_dict(dft) + expected = _expected_log(10, base) + for var in DATA_VARS: + assert result[var] == pytest.approx(expected[var], abs=1e-6) + assert result["vare"] == DATA["vare"] - dft = LogCpTransformer(C=10, base="10").fit_transform(df) - exp = np.log10(df[["vara", "varb", "varc", "vard"]] + 10) - exp["vare"] = df["vare"] - pd.testing.assert_frame_equal(dft, exp) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_inverse_transform(make_df): + X = make_df(DATA) -def test_inverse_transform(df): tr = LogCpTransformer(C="auto", base="10") - dft = tr.fit_transform(df) + dft = tr.fit_transform(X) orig = tr.inverse_transform(dft) - pd.testing.assert_frame_equal( - orig, df, check_dtype=False, check_exact=False, rtol=0.1 - ) + result = _to_dict(orig) + for var in DATA_VARS: + assert result[var] == pytest.approx(DATA[var], abs=0.1) tr = LogCpTransformer(C=10, base="e") - dft = tr.fit_transform(df) + dft = tr.fit_transform(X) orig = tr.inverse_transform(dft) - pd.testing.assert_frame_equal( - orig, df, check_dtype=False, check_exact=False, rtol=0.1 - ) + result = _to_dict(orig) + for var in DATA_VARS: + assert result[var] == pytest.approx(DATA[var], abs=0.1) + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_raises_error_if_na_in_df(make_df): + X_na = make_df(DATA_NA) + X = make_df(DATA_VARTYPES) -def test_raises_error_if_na_in_df(df_na, df_vartypes): # when dataset contains na, fit method transformer = LogCpTransformer() with pytest.raises(ValueError): - transformer.fit(df_na) + transformer.fit(X_na) # when dataset contains na, transform method transformer = LogCpTransformer() - transformer.fit(df_vartypes) + transformer.fit(X) with pytest.raises(ValueError): - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.transform(X_na) -def test_non_fitted_error(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df): + X = make_df(DATA_VARTYPES) transformer = LogCpTransformer() with pytest.raises(NotFittedError): - transformer.transform(df_vartypes) + transformer.transform(X) diff --git a/tests/test_transformation/test_power_transformer.py b/tests/test_transformation/test_power_transformer.py index 4f39d8eb0..09fb3eb44 100644 --- a/tests/test_transformation/test_power_transformer.py +++ b/tests/test_transformation/test_power_transformer.py @@ -1,38 +1,57 @@ +import narwhals as nw +import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.transformation import PowerTransformer +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} +DATA_NA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20.0, 21.0, 19.0, np.nan], + "Marks": [0.9, 0.8, 0.7, np.nan], +} + +_exp_ls = [0.001, 0.1, 2, 3, 4, 10] -def test_defo_params_plus_automatically_find_variables(df_vartypes): - # test case 1: automatically select variables - transformer = PowerTransformer(variables=None) - X = transformer.fit_transform(df_vartypes) - # expected output - transf_df = df_vartypes.copy() - transf_df["Age"] = [4.47214, 4.58258, 4.3589, 4.24264] - transf_df["Marks"] = [0.948683, 0.894427, 0.83666, 0.774597] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_defo_params_plus_automatically_find_variables(make_df): + X = make_df(DATA) + transformer = PowerTransformer(variables=None) + Xt = transformer.fit_transform(X) # test init params assert transformer.exp == 0.5 assert transformer.variables is None # test fit attr assert transformer.variables_ == ["Age", "Marks"] - assert transformer.n_features_in_ == 5 + assert transformer.n_features_in_ == 4 + # test transform output - pd.testing.assert_frame_equal(X, transf_df) + result = nw.from_native(Xt, eager_only=True).to_dict(as_series=False) + assert result["Age"] == pytest.approx( + [4.47214, 4.58258, 4.3589, 4.24264], abs=1e-5 + ) + assert result["Marks"] == pytest.approx( + [0.948683, 0.894427, 0.83666, 0.774597], abs=1e-5 + ) # inverse transform - Xit = transformer.inverse_transform(X) + Xit = transformer.inverse_transform(Xt) + result_it = nw.from_native(Xit, eager_only=True).to_dict(as_series=False) # convert numbers to original format. - Xit["Age"] = Xit["Age"].round().astype("int64") - Xit["Marks"] = Xit["Marks"].round(1) - - # test - pd.testing.assert_frame_equal(Xit, df_vartypes) + assert [round(v) for v in result_it["Age"]] == DATA["Age"] + assert [round(v, 1) for v in result_it["Marks"]] == DATA["Marks"] def test_error_if_exp_value_not_allowed(): @@ -40,45 +59,48 @@ def test_error_if_exp_value_not_allowed(): PowerTransformer(exp="other") -def test_fit_raises_error_if_na_in_df(df_na): - # test case 2: when dataset contains na, fit method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_raises_error_if_na_in_df(make_df): + X = make_df(DATA_NA) with pytest.raises(ValueError): transformer = PowerTransformer() - transformer.fit(df_na) + transformer.fit(X) -def test_transform_raises_error_if_na_in_df(df_vartypes, df_na): - # test case 3: when dataset contains na, transform method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_raises_error_if_na_in_df(make_df): + X = make_df(DATA) + X_na = make_df(DATA_NA) with pytest.raises(ValueError): transformer = PowerTransformer() - transformer.fit(df_vartypes) - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.fit(X) + transformer.transform(X_na) -def test_non_fitted_error(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df): + X = make_df(DATA) with pytest.raises(NotFittedError): transformer = PowerTransformer() - transformer.transform(df_vartypes) - - -_exp_ls = [0.001, 0.1, 2, 3, 4, 10] + transformer.transform(X) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("exp_base", _exp_ls) -def test_inverse_transform_exp_no_default(exp_base, df_vartypes): +def test_inverse_transform_exp_no_default(make_df, exp_base): + X = make_df(DATA) transformer = PowerTransformer(exp=exp_base) - Xt = transformer.fit_transform(df_vartypes) - X = transformer.inverse_transform(Xt) + Xt = transformer.fit_transform(X) + Xit = transformer.inverse_transform(Xt) + + result_it = nw.from_native(Xit, eager_only=True).to_dict(as_series=False) # convert numbers to original format. - X["Age"] = X["Age"].round().astype("int64") - X["Marks"] = X["Marks"].round(1) + assert [round(v) for v in result_it["Age"]] == DATA["Age"] + assert [round(v, 1) for v in result_it["Marks"]] == DATA["Marks"] # test init params - # assert transformer.exp == 100 assert transformer.variables is None # test fit attr assert transformer.variables_ == ["Age", "Marks"] - assert transformer.n_features_in_ == 5 - # test transform output - pd.testing.assert_frame_equal(X, df_vartypes) + assert transformer.n_features_in_ == 4 diff --git a/tests/test_transformation/test_reciprocal_transformer.py b/tests/test_transformation/test_reciprocal_transformer.py index a8ac99aff..c149e18e8 100644 --- a/tests/test_transformation/test_reciprocal_transformer.py +++ b/tests/test_transformation/test_reciprocal_transformer.py @@ -1,72 +1,94 @@ +import narwhals as nw +import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.transformation import ReciprocalTransformer - -def test_automatically_find_variables(df_vartypes): - # test case 1: automatically select variables +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} +DATA_NA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20.0, 21.0, 19.0, np.nan], + "Marks": [0.9, 0.8, 0.7, np.nan], +} + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_automatically_find_variables_and_inverse_transform(make_df): + X = make_df(DATA) transformer = ReciprocalTransformer(variables=None) - X = transformer.fit_transform(df_vartypes) - - # expected output - transf_df = df_vartypes.copy() - transf_df["Age"] = [0.05, 0.047619, 0.0526316, 0.0555556] - transf_df["Marks"] = [1.11111, 1.25, 1.42857, 1.66667] + Xt = transformer.fit_transform(X) # test init params assert transformer.variables is None # test fit attr assert transformer.variables_ == ["Age", "Marks"] - assert transformer.n_features_in_ == 5 + assert transformer.n_features_in_ == 4 + # test transform output - pd.testing.assert_frame_equal(X, transf_df) + result = nw.from_native(Xt, eager_only=True).to_dict(as_series=False) + assert result["Age"] == pytest.approx( + [0.05, 0.047619, 0.052632, 0.055556], abs=1e-5 + ) + assert result["Marks"] == pytest.approx( + [1.111111, 1.25, 1.428571, 1.666667], abs=1e-5 + ) # test inverse_transform - Xit = transformer.inverse_transform(X) - - # convert numbers to original format. - Xit["Age"] = Xit["Age"].round().astype("int64") - Xit["Marks"] = Xit["Marks"].round(1) + Xit = transformer.inverse_transform(Xt) + result_it = nw.from_native(Xit, eager_only=True).to_dict(as_series=False) + assert [round(v) for v in result_it["Age"]] == DATA["Age"] + assert [round(v, 1) for v in result_it["Marks"]] == DATA["Marks"] - # test - pd.testing.assert_frame_equal(Xit, df_vartypes) - -def test_fit_raises_error_if_na_in_df(df_na): - # test case 2: when dataset contains na, fit method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_raises_error_if_na_in_df(make_df): + X = make_df(DATA_NA) with pytest.raises(ValueError): transformer = ReciprocalTransformer() - transformer.fit(df_na) + transformer.fit(X) -def test_transform_raises_error_if_na_in_df(df_vartypes, df_na): - # test case 3: when dataset contains na, transform method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_raises_error_if_na_in_df(make_df): + X = make_df(DATA) + X_na = make_df(DATA_NA) + transformer = ReciprocalTransformer() + transformer.fit(X) with pytest.raises(ValueError): - transformer = ReciprocalTransformer() - transformer.fit(df_vartypes) - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.transform(X_na) -def test_error_if_df_contains_0_as_value(df_vartypes): - # test error when data contains value zero - df_neg = df_vartypes.copy() - df_neg.loc[1, "Age"] = 0 +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_df_contains_0_as_value(make_df): + data_zero = dict(DATA) + data_zero["Age"] = [20, 0, 19, 18] + X = make_df(DATA) + X_zero = make_df(data_zero) - # test case 4: when variable contains zero, fit + # when variable contains zero, fit with pytest.raises(ValueError): transformer = ReciprocalTransformer() - transformer.fit(df_neg) + transformer.fit(X_zero) - # test case 5: when variable contains zero, transform + # when variable contains zero, transform + transformer = ReciprocalTransformer() + transformer.fit(X) with pytest.raises(ValueError): - transformer = ReciprocalTransformer() - transformer.fit(df_vartypes) - transformer.transform(df_neg) + transformer.transform(X_zero) -def test_non_fitted_error(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df): + X = make_df(DATA) with pytest.raises(NotFittedError): transformer = ReciprocalTransformer() - transformer.transform(df_vartypes) + transformer.transform(X) diff --git a/tests/test_transformation/test_yeojohnson_transformer.py b/tests/test_transformation/test_yeojohnson_transformer.py index f4eb32f93..b411bddcf 100644 --- a/tests/test_transformation/test_yeojohnson_transformer.py +++ b/tests/test_transformation/test_yeojohnson_transformer.py @@ -1,170 +1,194 @@ +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.transformation import YeoJohnsonTransformer - -def test_automatically_select_variables(df_vartypes): - # test case 1: automatically select variables +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} +DATA_NA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20.0, 21.0, 19.0, np.nan], + "Marks": [0.9, 0.8, 0.7, np.nan], +} + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_automatically_select_variables_and_inverse_transform(make_df): + X = make_df(DATA) transformer = YeoJohnsonTransformer(variables=None) - X = transformer.fit_transform(df_vartypes) - - # expected result - transf_df = df_vartypes.copy() - transf_df["Age"] = [10.167, 10.5406, 9.78774, 9.40229] - transf_df["Marks"] = [0.804449, 0.722367, 0.638807, 0.553652] + Xt = transformer.fit_transform(X) # test init params assert transformer.variables is None - # test fit attr + # test fit attrs assert transformer.variables_ == ["Age", "Marks"] - assert transformer.n_features_in_ == 5 + assert transformer.n_features_in_ == 4 + # test transform output - pd.testing.assert_frame_equal(X, transf_df) + result = nw.from_native(Xt, eager_only=True).to_dict(as_series=False) + assert result["Age"] == pytest.approx( + [10.167048, 10.540602, 9.787738, 9.402289], abs=1e-5 + ) + assert result["Marks"] == pytest.approx( + [0.804449, 0.722367, 0.638807, 0.553652], abs=1e-5 + ) + + # test inverse_transform, including non-transformed columns + Xit = transformer.inverse_transform(Xt) + result_it = nw.from_native(Xit, eager_only=True).to_dict(as_series=False) + assert [round(v) for v in result_it["Age"]] == DATA["Age"] + assert [round(v, 1) for v in result_it["Marks"]] == DATA["Marks"] + assert result_it["Name"] == DATA["Name"] + assert result_it["City"] == DATA["City"] -def test_transformer_on_integer_variables(): - df = pd.DataFrame( +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transformer_on_integer_variables(make_df): + X = make_df( { "var1": [0, 1, 0, 2, 3, 4, 5, 6, 8, 10], "var2": [12, 11, 10, 15, 13, 12, 11, 10, 10, 20], } ) - dft = pd.DataFrame( - { - "var1": { - 0: 0.0, - 1: 0.7871467037957388, - 2: 0.0, - 3: 1.34716625120788, - 4: 1.797027857352365, - 5: 2.1794549065159363, - 6: 2.5155129679774246, - 7: 2.817344570368886, - 8: 3.346739213848269, - 9: 3.8051709334268566, - }, - "var2": { - 0: 0.2891005444159968, - 1: 0.2890875957028113, - 2: 0.2890687942494933, - 3: 0.2891213447054929, - 4: 0.2891097235906253, - 5: 0.2891005444159968, - 6: 0.2890875957028113, - 7: 0.2890687942494933, - 8: 0.2890687942494933, - 9: 0.28913341330818815, - }, - } + Xt = YeoJohnsonTransformer().fit_transform(X) + result = nw.from_native(Xt, eager_only=True).to_dict(as_series=False) + + assert result["var1"] == pytest.approx( + [ + 0.0, + 0.787147, + 0.0, + 1.347166, + 1.797028, + 2.179455, + 2.515513, + 2.817345, + 3.346739, + 3.805171, + ], + abs=1e-5, + ) + assert result["var2"] == pytest.approx( + [ + 0.289101, + 0.289088, + 0.289069, + 0.289121, + 0.289110, + 0.289101, + 0.289088, + 0.289069, + 0.289069, + 0.289133, + ], + abs=1e-5, ) - - X_tr = YeoJohnsonTransformer().fit_transform(df) - pd.testing.assert_frame_equal(X_tr, dft) -def test_fit_raises_error_if_na_in_df(df_na): - # test case 2: when dataset contains na, fit method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_raises_error_if_na_in_df(make_df): + X = make_df(DATA_NA) with pytest.raises(ValueError): transformer = YeoJohnsonTransformer() - transformer.fit(df_na) + transformer.fit(X) -def test_transform_raises_error_if_na_in_df(df_vartypes, df_na): - # test case 3: when dataset contains na, transform method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_raises_error_if_na_in_df(make_df): + X = make_df(DATA) + X_na = make_df(DATA_NA) + transformer = YeoJohnsonTransformer() + transformer.fit(X) with pytest.raises(ValueError): - transformer = YeoJohnsonTransformer() - transformer.fit(df_vartypes) - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.transform(X_na) -def test_non_fitted_error(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df): + X = make_df(DATA) with pytest.raises(NotFittedError): transformer = YeoJohnsonTransformer() - transformer.transform(df_vartypes) - - -def test_inverse_transform_automatically_select_only_transformed_columns(df_vartypes): - X = df_vartypes.copy(deep=True) - transformer = YeoJohnsonTransformer(variables=None) - X_trans = transformer.fit_transform(X) + transformer.transform(X) - X_inverse = transformer.inverse_transform(X_trans) - X_inverse["Age"] = X_inverse["Age"].round(0).astype(int) - pd.testing.assert_frame_equal(X, X_inverse, check_dtype=False) - - -def test_inverse_with_X_negative_and_positive(): - X = pd.DataFrame( +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_inverse_with_x_negative_and_positive(make_df): + X = make_df( { - "var1": np.arange(-20, 0), - "var2": np.arange(0, 20), - "var3": np.arange(-10, 10), + "var1": list(np.arange(-20, 0)), + "var2": list(np.arange(0, 20)), + "var3": list(np.arange(-10, 10)), } ) transformer = YeoJohnsonTransformer(variables=None) - X_trans = transformer.fit_transform(X) - - X_inverse = transformer.inverse_transform(X_trans) - X_inverse = X_inverse.round(0).astype(int) + Xt = transformer.fit_transform(X) + Xi = transformer.inverse_transform(Xt) + result = nw.from_native(Xi, eager_only=True).to_dict(as_series=False) - pd.testing.assert_frame_equal(X, X_inverse, check_dtype=False) + assert [round(v) for v in result["var1"]] == list(np.arange(-20, 0)) + assert [round(v) for v in result["var2"]] == list(np.arange(0, 20)) + assert [round(v) for v in result["var3"]] == list(np.arange(-10, 10)) -def test_inverse_with_with_non_linear_index(): +def test_inverse_with_non_linear_index(): + # pandas-specific: exercises index-preserving behaviour, which has no + # polars equivalent (polars has no row index). X = pd.DataFrame( { "var1": np.arange(-20, 0), "var2": np.arange(0, 20), "var3": np.arange(-10, 10), }, - index=[13, 15, 12, 11, 17, 9, 4, 0, 1, 14, 18, 2, 3, 6, 5, 7, 8, 2, 16, 10] + index=[13, 15, 12, 11, 17, 9, 4, 0, 1, 14, 18, 2, 3, 6, 5, 7, 8, 2, 16, 10], ) transformer = YeoJohnsonTransformer(variables=None) - X_trans = transformer.fit_transform(X) + Xt = transformer.fit_transform(X) - X_inverse = transformer.inverse_transform(X_trans) - X_inverse = X_inverse.round(0).astype(int) + Xi = transformer.inverse_transform(Xt) + Xi = Xi.round(0).astype(int) - pd.testing.assert_frame_equal(X, X_inverse, check_dtype=False) + pd.testing.assert_frame_equal(X, Xi, check_dtype=False) -def test_lambda_equals_lambda_equal_0(): - X = pd.DataFrame( - { - "var1": np.arange(0, 20), - "var2": np.arange(20, 40), - } - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_lambda_equal_0(make_df): + X = make_df({"var1": list(np.arange(0, 20)), "var2": list(np.arange(20, 40))}) transformer = YeoJohnsonTransformer(variables=None) transformer = transformer.fit(X) - transformer.lambda_dict_ = {"var1": 0, "var2": 0} - X_trans = transformer.transform(X) - X_inverse = transformer.inverse_transform(X_trans) - X_inverse = X_inverse.round(0).astype(int) + Xt = transformer.transform(X) + Xi = transformer.inverse_transform(Xt) + result = nw.from_native(Xi, eager_only=True).to_dict(as_series=False) - pd.testing.assert_frame_equal(X, X_inverse, check_dtype=False) + assert [round(v) for v in result["var1"]] == list(np.arange(0, 20)) + assert [round(v) for v in result["var2"]] == list(np.arange(20, 40)) -def test_lambda_equals_lambda_equal_2(): - X = pd.DataFrame({"var1": np.arange(-21, -1), "var2": np.arange(-41, -21)}) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_lambda_equal_2(make_df): + X = make_df({"var1": list(np.arange(-21, -1)), "var2": list(np.arange(-41, -21))}) transformer = YeoJohnsonTransformer(variables=None) transformer = transformer.fit(X) - transformer.lambda_dict_ = {"var1": 2, "var2": 2} - X_trans = transformer.transform(X) - X_inverse = transformer.inverse_transform(X_trans) - X_inverse = X_inverse.round(0).astype(int) + Xt = transformer.transform(X) + Xi = transformer.inverse_transform(Xt) + result = nw.from_native(Xi, eager_only=True).to_dict(as_series=False) - pd.testing.assert_frame_equal(X, X_inverse, check_dtype=False) + assert [round(v) for v in result["var1"]] == list(np.arange(-21, -1)) + assert [round(v) for v in result["var2"]] == list(np.arange(-41, -21)) diff --git a/tests/test_variable_handling/conftest.py b/tests/test_variable_handling/conftest.py index 841656da2..536776c93 100644 --- a/tests/test_variable_handling/conftest.py +++ b/tests/test_variable_handling/conftest.py @@ -1,7 +1,40 @@ +from datetime import datetime, timezone + import pandas as pd +import polars as pl import pytest +def cast_categorical(df, columns): + """Cast `columns` to the backend's categorical dtype, whichever backend `df` + (pandas or polars) happens to be. Used to build matched pandas/polars data + for tests parametrized over both libraries. + """ + if isinstance(df, pd.DataFrame): + df = df.copy() + df[columns] = df[columns].astype("category") + return df + return df.with_columns([pl.col(c).cast(pl.Categorical) for c in columns]) + + +# Data shared between the pandas and polars variants of a test. +BASIC_DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} + +DATETIME_DATA = { + **BASIC_DATA, + "date_range": [datetime(2020, 2, 24, 0, i) for i in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], + "date_range_tz": [ + datetime(2020, 2, 24, 0, i, tzinfo=timezone.utc) for i in range(4) + ], +} + + @pytest.fixture def df(): df = pd.DataFrame( diff --git a/tests/test_variable_handling/test_check_variables.py b/tests/test_variable_handling/test_check_variables.py index 8eba88cb0..eb1bb018a 100644 --- a/tests/test_variable_handling/test_check_variables.py +++ b/tests/test_variable_handling/test_check_variables.py @@ -1,4 +1,5 @@ import pandas as pd +import polars as pl import pytest from feature_engine.variable_handling import ( @@ -7,103 +8,139 @@ check_datetime_variables, check_numerical_variables, ) +from tests.test_variable_handling.conftest import ( + BASIC_DATA, + DATETIME_DATA, + cast_categorical, +) -def test_check_numerical_variables_returns_numerical_variables(df, df_int): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_check_numerical_variables_returns_numerical_variables(make_df): + df = make_df(BASIC_DATA) assert check_numerical_variables(df, ["Age", "Marks"]) == ["Age", "Marks"] assert check_numerical_variables(df, ["Age"]) == ["Age"] assert check_numerical_variables(df, "Age") == ["Age"] + + +def test_check_numerical_variables_returns_numerical_variables_int_names(df_int): + # polars requires string column names, so int-named columns are pandas-only assert check_numerical_variables(df_int, [3, 4]) == [3, 4] assert check_numerical_variables(df_int, [3]) == [3] assert check_numerical_variables(df_int, 4) == [4] -def test_check_numerical_variables_raises_errors_when_not_numerical(df, df_int): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_check_numerical_variables_raises_errors_when_not_numerical(make_df): + df = make_df(BASIC_DATA) msg = ( "Some of the variables are not numerical. Please cast them as " "numerical before using this transformer." ) - with pytest.raises(TypeError) as record: - assert check_numerical_variables(df, "Name") - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_numerical_variables(df, "Name") + + with pytest.raises(TypeError, match=msg): + check_numerical_variables(df, ["Name"]) - with pytest.raises(TypeError) as record: - assert check_numerical_variables(df, ["Name"]) - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_numerical_variables(df, ["Name", "Marks"]) - with pytest.raises(TypeError) as record: - assert check_numerical_variables(df_int, 1) - assert str(record.value) == msg - with pytest.raises(TypeError) as record: - assert check_numerical_variables(df_int, [1]) - assert str(record.value) == msg +def test_check_numerical_variables_raises_errors_int_names(df_int): + msg = ( + "Some of the variables are not numerical. Please cast them as " + "numerical before using this transformer." + ) + with pytest.raises(TypeError, match=msg): + check_numerical_variables(df_int, 1) - with pytest.raises(TypeError) as record: - assert check_numerical_variables(df, ["Name", "Marks"]) - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_numerical_variables(df_int, [1]) - with pytest.raises(TypeError) as record: - assert check_numerical_variables(df_int, [2, 3]) - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_numerical_variables(df_int, [2, 3]) -def test_check_categorical_variables_returns_categorical_variables(df, df_int): - assert check_categorical_variables(df, ["Name", "date_obj0"]) == [ - "Name", - "date_obj0", - ] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_check_categorical_variables_returns_categorical_variables(make_df): + df = make_df(BASIC_DATA) + assert check_categorical_variables(df, ["Name", "City"]) == ["Name", "City"] assert check_categorical_variables(df, ["Name"]) == ["Name"] - assert check_categorical_variables(df, "date_obj0") == ["date_obj0"] + assert check_categorical_variables(df, "Name") == ["Name"] + + +def test_check_categorical_variables_numeric_categories_pandas_only(): + # polars categoricals are always string-backed, so casting a + # numeric column to Categorical isn't a realistic polars scenario. + df = pd.DataFrame(BASIC_DATA) + df = cast_categorical(df, ["Age", "Marks"]) + assert check_categorical_variables(df, ["Age", "Marks"]) == ["Age", "Marks"] + + +def test_check_categorical_variables_returns_categorical_variables_int_names(df_int): assert check_categorical_variables(df_int, [1, 2]) == [1, 2] assert check_categorical_variables(df_int, [2]) == [2] assert check_categorical_variables(df_int, 2) == [2] - df[["Age", "Marks"]] = df[["Age", "Marks"]].astype(pd.CategoricalDtype) - assert check_categorical_variables(df, ["Age", "Marks"]) == ["Age", "Marks"] - -def test_check_categorical_variables_raises_errors_when_not_categorical(df, df_int): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_check_categorical_variables_raises_errors_when_not_categorical(make_df): + df = make_df(BASIC_DATA) msg = ( "Some of the variables are not categorical. Please cast them as " "object or categorical before using this transformer." ) - with pytest.raises(TypeError) as record: - assert check_categorical_variables(df, "Age") - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_categorical_variables(df, "Age") + + with pytest.raises(TypeError, match=msg): + check_categorical_variables(df, ["Age"]) - with pytest.raises(TypeError) as record: - assert check_categorical_variables(df, ["Age"]) - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_categorical_variables(df, ["Name", "Marks"]) - with pytest.raises(TypeError) as record: - assert check_categorical_variables(df_int, 3) - assert str(record.value) == msg - with pytest.raises(TypeError) as record: - assert check_categorical_variables(df_int, [3]) - assert str(record.value) == msg +def test_check_categorical_variables_raises_errors_int_names(df_int): + msg = ( + "Some of the variables are not categorical. Please cast them as " + "object or categorical before using this transformer." + ) + with pytest.raises(TypeError, match=msg): + check_categorical_variables(df_int, 3) - with pytest.raises(TypeError) as record: - assert check_categorical_variables(df, ["Name", "Marks"]) - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_categorical_variables(df_int, [3]) - with pytest.raises(TypeError) as record: - assert check_categorical_variables(df_int, [2, 3]) - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_categorical_variables(df_int, [2, 3]) -def test_check_datetime_variables_returns_datetime_variables(df_datetime): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_check_datetime_variables_returns_datetime_variables(make_df): + df = make_df(DATETIME_DATA) var_dt = ["date_range"] var_dt_str = "date_range" + vars_dt = ["date_range", "date_obj0", "date_range_tz"] + tz_time = "date_range_tz" + + assert check_datetime_variables(df, var_dt_str) == [var_dt_str] + assert check_datetime_variables(df, var_dt) == var_dt + assert check_datetime_variables(df, vars_dt) == vars_dt + assert check_datetime_variables(df, tz_time) == [tz_time] + + # only the string column can be cast to categorical. Native Datetime + # columns can't be cast to Categorical in polars + df = cast_categorical(df, ["date_obj0"]) + assert check_datetime_variables(df, "date_obj0") == ["date_obj0"] + + +def test_check_datetime_variables_returns_pandas_only_string_formats(df_datetime): + # "01-Jan-2010"-style and "10/11/12"-style strings are recognised via + # flexible, dateutil-backed guessing. vars_convertible_to_dt = ["date_range", "date_obj1", "date_obj2", "time_obj"] var_convertible_to_dt = "date_obj1" - tz_time = "time_objTZ" - tz_time_obj = "date_range_tz" - # when variables are specified - assert check_datetime_variables(df_datetime, var_dt_str) == [var_dt_str] - assert check_datetime_variables(df_datetime, var_dt) == var_dt assert check_datetime_variables(df_datetime, var_convertible_to_dt) == [ var_convertible_to_dt ] @@ -111,8 +148,6 @@ def test_check_datetime_variables_returns_datetime_variables(df_datetime): check_datetime_variables(df_datetime, vars_convertible_to_dt) == vars_convertible_to_dt ) - assert check_datetime_variables(df_datetime, tz_time) == [tz_time] - assert check_datetime_variables(df_datetime, tz_time_obj) == [tz_time_obj] df_datetime[vars_convertible_to_dt] = df_datetime[vars_convertible_to_dt].astype( pd.CategoricalDtype @@ -123,55 +158,48 @@ def test_check_datetime_variables_returns_datetime_variables(df_datetime): ) -def test_check_datetime_variables_raises_errors_when_not_datetime(df_datetime): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_check_datetime_variables_raises_errors_when_not_datetime(make_df): + df = make_df(DATETIME_DATA) msg = "Some of the variables are not or cannot be parsed as datetime." - with pytest.raises(TypeError) as record: - assert check_datetime_variables(df_datetime, variables="Age") - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_datetime_variables(df, variables="Age") - with pytest.raises(TypeError) as record: - assert check_datetime_variables(df_datetime, variables=["Age", "Name"]) - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_datetime_variables(df, variables=["Age", "Name"]) - with pytest.raises(TypeError): - assert check_datetime_variables(df_datetime, variables=["date_range", "Age"]) - assert str(record.value) == msg + with pytest.raises(TypeError, match=msg): + check_datetime_variables(df, variables=["date_range", "Age"]) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "input_vars", [ - ["Name", "City", "Age", "Marks", "dob"], - [ - "Name", - "City", - "Age", - "Marks", - ], + ["Name", "City", "Age", "Marks"], + ["Name", "City", "Age"], "Name", ["Age"], ], ) -def test_check_all_variables_returns_all_variables(df_vartypes, input_vars): +def test_check_all_variables_returns_all_variables(make_df, input_vars): + df = make_df(BASIC_DATA) if isinstance(input_vars, list): - assert check_all_variables(df_vartypes, input_vars) == input_vars + assert check_all_variables(df, input_vars) == input_vars else: - assert check_all_variables(df_vartypes, input_vars) == [input_vars] + assert check_all_variables(df, input_vars) == [input_vars] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize( "input_vars", [["Name", "City", "Absent"], "Absent", ["Absent"]] ) -def test_check_all_variables_raises_errors_when_not_in_dataframe( - df_vartypes, input_vars -): +def test_check_all_variables_raises_errors_when_not_in_dataframe(make_df, input_vars): + df = make_df(BASIC_DATA) msg_ls = "'Some of the variables are not in the dataframe.'" msg_single = "'The variable Absent is not in the dataframe.'" + msg = msg_ls if isinstance(input_vars, list) else msg_single - with pytest.raises(KeyError) as record: - assert check_all_variables(df_vartypes, input_vars) - if isinstance(input_vars, list): - assert str(record.value) == msg_ls - else: - assert str(record.value) == msg_single + with pytest.raises(KeyError, match=msg): + check_all_variables(df, input_vars) diff --git a/tests/test_variable_handling/test_fe_type_checks.py b/tests/test_variable_handling/test_fe_type_checks.py deleted file mode 100644 index de4bc2d38..000000000 --- a/tests/test_variable_handling/test_fe_type_checks.py +++ /dev/null @@ -1,93 +0,0 @@ -import pandas as pd - -from feature_engine.variable_handling._variable_type_checks import ( - _is_categorical_and_is_datetime, - _is_categorical_and_is_not_datetime, - _is_categories_num, - _is_convertible_to_dt, - _is_convertible_to_num, -) - - -def test_is_categories_num(df): - assert _is_categories_num(df["Name"]) is False - - df["Age"] = df["Age"].astype("category") - assert _is_categories_num(df["Age"]) is True - - -def test_is_convertible_to_num(df): - assert _is_convertible_to_num(df["Name"]) is False - assert _is_convertible_to_num(df["date_obj0"]) is False - - df["age_str"] = ["20", "21", "19", "18"] - assert _is_convertible_to_num(df["age_str"]) is True - - -def test_is_convertible_to_dt(df): - assert _is_convertible_to_dt(df["date_obj0"]) is True - assert _is_convertible_to_dt(df["date_range"]) is True - assert _is_convertible_to_dt(df["Name"]) is False - - df["age_str"] = ["20", "21", "19", "18"] - assert _is_convertible_to_dt(df["age_str"]) is False - - -def test_is_categorical_and_is_datetime(df, df_datetime): - assert _is_categorical_and_is_datetime(df["date_obj0"]) is True - assert _is_categorical_and_is_datetime(df["Name"]) is False - assert _is_categorical_and_is_datetime(df_datetime["date_obj1"]) is True - - df["age_str"] = ["20", "21", "19", "18"] - assert _is_categorical_and_is_datetime(df["age_str"]) is False - - df = df.copy() - # from pandas 3 onwards, object types that contain strings are not recognised as - # objects any more - df["Age"] = df["Age"].astype("O") - assert _is_categorical_and_is_datetime(df["Age"]) is False - - # Object Datetime - s_obj_dt = pd.Series([pd.Timestamp("2020-01-01")], dtype="object") - assert _is_categorical_and_is_datetime(s_obj_dt) is True - - # StringDtype Datetime (if convertible) - s_str_dt = pd.Series(["2020-01-01", "2020-01-02"], dtype="string") - assert _is_categorical_and_is_datetime(s_str_dt) is True - - # Numeric (should be False for both if and elif branches) - s_num = pd.Series([1, 2, 3]) - assert _is_categorical_and_is_datetime(s_num) is False - - # Categorical (should hit the 'if' branch) - s_cat = pd.Series(["a", "b"], dtype="category") - assert _is_categorical_and_is_datetime(s_cat) is False - - -def test_is_categorical_and_is_not_datetime(df): - assert _is_categorical_and_is_not_datetime(df["date_obj0"]) is False - assert _is_categorical_and_is_not_datetime(df["date_obj0"]) is False - assert _is_categorical_and_is_not_datetime(df["Name"]) is True - - df["age_str"] = ["20", "21", "19", "18"] - assert _is_categorical_and_is_not_datetime(df["age_str"]) is True - - # Object Integer - s_obj_int = pd.Series([1, 2], dtype="object") - assert _is_categorical_and_is_not_datetime(s_obj_int) is True - - # Object Datetime should be False - s_obj_dt = pd.Series([pd.Timestamp("2020-01-01")], dtype="object") - assert _is_categorical_and_is_not_datetime(s_obj_dt) is False - - # StringDtype (not convertible to numeric/datetime) should be True - s_str = pd.Series(["a", "b"], dtype="string") - assert _is_categorical_and_is_not_datetime(s_str) is True - - # Numeric should be False - s_num = pd.Series([1, 2, 3]) - assert _is_categorical_and_is_not_datetime(s_num) is False - - # Categorical should be True (it hits the 'if' branch) - s_cat = pd.Series(["a", "b"], dtype="category") - assert _is_categorical_and_is_not_datetime(s_cat) is True diff --git a/tests/test_variable_handling/test_find_variables.py b/tests/test_variable_handling/test_find_variables.py index 6ae29384d..249eb0b37 100644 --- a/tests/test_variable_handling/test_find_variables.py +++ b/tests/test_variable_handling/test_find_variables.py @@ -1,4 +1,5 @@ import pandas as pd +import polars as pl import pytest from feature_engine.variable_handling import ( @@ -8,89 +9,100 @@ find_datetime_variables, find_numerical_variables, ) +from tests.test_variable_handling.conftest import ( + BASIC_DATA, + DATETIME_DATA, + cast_categorical, +) # --- find_numerical_variables --- # -def test_numerical_variables_finds_variables(df, df_int): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numerical_variables_finds_variables(make_df): + df = make_df(BASIC_DATA) assert find_numerical_variables(df) == ["Age", "Marks"] + + +def test_numerical_variables_finds_variables_with_int_column_names(df_int): + # polars requires string column names. int-named columns are pandas-only assert find_numerical_variables(df_int) == [3, 4] -def test_numerical_variables_raises_error(df, df_int): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numerical_variables_raises_error(make_df): + df = make_df(BASIC_DATA) msg = "No numerical variables found in this dataframe." with pytest.raises(TypeError, match=msg): - find_numerical_variables(df.drop(["Age", "Marks"], axis=1)) - - with pytest.raises(TypeError, match=msg): - find_numerical_variables(df_int.drop([3, 4], axis=1)) + find_numerical_variables(df[["Name", "City"]]) -def test_numerical_variables_raises_warning(df, df_int): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numerical_variables_raises_warning(make_df): + df = make_df(BASIC_DATA) msg = "No numerical variables found in this dataframe." - - # Test with a regular DataFrame with pytest.warns(UserWarning, match=msg): - find_numerical_variables(df.drop(["Age", "Marks"], axis=1), return_empty=True) - - # Test with integer-only DataFrame - with pytest.warns(UserWarning, match=msg): - find_numerical_variables(df_int.drop([3, 4], axis=1), return_empty=True) + find_numerical_variables(df[["Name", "City"]], return_empty=True) -def test_numerical_variables_returns_empty_list(df, df_int): - assert ( - find_numerical_variables(df.drop(["Age", "Marks"], axis=1), return_empty=True) - == [] - ) - assert ( - find_numerical_variables(df_int.drop([3, 4], axis=1), return_empty=True) == [] - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numerical_variables_returns_empty_list(make_df): + df = make_df(BASIC_DATA) + assert find_numerical_variables(df[["Name", "City"]], return_empty=True) == [] # --- find_categorical_variables --- # -def test_categorical_variables_finds_variables(df, df_int): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_categorical_variables_finds_variables(make_df): + df = make_df(BASIC_DATA) assert find_categorical_variables(df) == ["Name", "City"] + + +def test_categorical_variables_finds_variables_with_int_column_names(df_int): assert find_categorical_variables(df_int) == [1, 2] -def test_categorical_variables_raises_error(df, df_int): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_categorical_variables_raises_error(make_df): + df = make_df(BASIC_DATA) msg = "No categorical variables found in this dataframe." with pytest.raises(TypeError, match=msg): - find_categorical_variables(df.drop(["Name", "City"], axis=1)) - - with pytest.raises(TypeError, match=msg): - find_categorical_variables(df_int.drop([1, 2], axis=1)) + find_categorical_variables(df[["Age", "Marks"]]) -def test_categorical_variables_raises_warning(df, df_int): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_categorical_variables_raises_warning(make_df): + df = make_df(BASIC_DATA) msg = "No categorical variables found in this dataframe." - - # Test with a regular DataFrame with pytest.warns(UserWarning, match=msg): - find_categorical_variables(df.drop(["Name", "City"], axis=1), return_empty=True) - - # Test with integer-only DataFrame - with pytest.warns(UserWarning, match=msg): - find_categorical_variables(df_int.drop([1, 2], axis=1), return_empty=True) + find_categorical_variables(df[["Age", "Marks"]], return_empty=True) -def test_categorical_variables_returns_empty_list(df, df_int): - assert ( - find_categorical_variables(df.drop(["Name", "City"], axis=1), return_empty=True) - == [] - ) - assert ( - find_categorical_variables(df_int.drop([1, 2], axis=1), return_empty=True) == [] - ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_categorical_variables_returns_empty_list(make_df): + df = make_df(BASIC_DATA) + assert find_categorical_variables(df[["Age", "Marks"]], return_empty=True) == [] # --- find_datetime_variables --- # -def test_datetime_variables_finds_variables(df_datetime): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_variables_finds_variables(make_df): + df = make_df(DATETIME_DATA) + vars_dt = ["date_range", "date_obj0", "date_range_tz"] + assert find_datetime_variables(df) == vars_dt + + assert find_datetime_variables( + df[["date_obj0", "date_range", "date_range_tz"]], + ) == ["date_obj0", "date_range", "date_range_tz"] + + +def test_datetime_variables_finds_pandas_only_string_formats(df_datetime): + # "01-Jan-2010"-style, "10/11/12"-style and bare-time strings are + # recognised through flexible, dateutil-backed guessing. vars_dt = [ "date_range", "date_obj0", @@ -100,206 +112,206 @@ def test_datetime_variables_finds_variables(df_datetime): "time_obj", "time_objTZ", ] - assert find_datetime_variables(df_datetime) == vars_dt - assert find_datetime_variables( - df_datetime[vars_dt].reindex(columns=["date_obj1", "date_range", "date_obj2"]), - ) == ["date_obj1", "date_range", "date_obj2"] +def test_datetime_variables_finds_flexible_string_formats_in_polars_too(): + # flexible, dateutil-backed date guessing is backend-agnostic, so polars + # now also recognises non-ISO formats it previously could not. + df = pl.DataFrame( + { + "var_num": [1, 2, 3], + "date_obj1": ["01-Jan-2010", "24-Feb-1945", "14-Jun-2100"], + "date_obj2": ["10/11/12", "12/31/09", "06/30/95"], + } + ) + assert find_datetime_variables(df) == ["date_obj1", "date_obj2"] -def test_datetime_variables_raises_error(df_datetime): - msg = "No datetime variables found in this dataframe." +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_variables_raises_error(make_df): + df = make_df(DATETIME_DATA) + msg = "No datetime variables found in this dataframe." vars_nondt = ["Marks", "Age", "Name"] - with pytest.raises(TypeError, match=msg): - find_datetime_variables(df_datetime.loc[:, vars_nondt]) + find_datetime_variables(df[vars_nondt]) -def test_datetime_variables_raises_warning(df_datetime): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_variables_raises_warning(make_df): + df = make_df(DATETIME_DATA) msg = "No datetime variables found in this dataframe." vars_nondt = ["Marks", "Age", "Name"] with pytest.warns(UserWarning, match=msg): - find_datetime_variables(df_datetime.loc[:, vars_nondt], return_empty=True) + find_datetime_variables(df[vars_nondt], return_empty=True) -def test_datetime_variables_returns_empty_list(df_datetime): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_datetime_variables_returns_empty_list(make_df): + df = make_df(DATETIME_DATA) vars_nondt = ["Marks", "Age", "Name"] - assert ( - find_datetime_variables(df_datetime.loc[:, vars_nondt], return_empty=True) == [] - ) + assert find_datetime_variables(df[vars_nondt], return_empty=True) == [] # --- find_all_variables --- # -def test_find_all_variables(df): - all_vars = [ - "Name", - "City", - "Age", - "Marks", - "date_range", - "date_obj0", - "date_range_tz", - ] - assert find_all_variables(df, exclude_datetime=False) == all_vars +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_find_all_variables(make_df): + df = make_df(BASIC_DATA) + assert find_all_variables(df, exclude_datetime=False) == list(BASIC_DATA.keys()) -def test_find_all_variables_excludes_dt(df): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_find_all_variables_excludes_dt(make_df): + df = make_df(DATETIME_DATA) all_vars_no_dt = ["Name", "City", "Age", "Marks"] assert find_all_variables(df, exclude_datetime=True) == all_vars_no_dt -def test_find_all_variables_raises_error(df): - dt_vars = [ - "date_range", - "date_obj0", - "date_range_tz", - ] - df = df[dt_vars] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_find_all_variables_raises_error(make_df): + dt_vars = ["date_range", "date_obj0", "date_range_tz"] + df = make_df(DATETIME_DATA)[dt_vars] msg = "No variables found in this dataframe" with pytest.raises(TypeError, match=msg): find_all_variables(df, exclude_datetime=True) -def test_find_all_variables_raises_warning(df): - dt_vars = [ - "date_range", - "date_obj0", - "date_range_tz", - ] - df = df[dt_vars] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_find_all_variables_raises_warning(make_df): + dt_vars = ["date_range", "date_obj0", "date_range_tz"] + df = make_df(DATETIME_DATA)[dt_vars] msg = "No variables found in this dataframe" with pytest.warns(UserWarning, match=msg): find_all_variables(df, exclude_datetime=True, return_empty=True) -def test_find_all_variables_returns_empty(df): - dt_vars = [ - "date_range", - "date_obj0", - "date_range_tz", - ] - df = df[dt_vars] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_find_all_variables_returns_empty(make_df): + dt_vars = ["date_range", "date_obj0", "date_range_tz"] + df = make_df(DATETIME_DATA)[dt_vars] assert find_all_variables(df, exclude_datetime=True, return_empty=True) == [] # --- find_categorical_and_numerical_variables --- # -def test_numcat_user_passes_varlist(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numcat_user_passes_varlist(make_df): + df = make_df(BASIC_DATA) + # Case 1: user passes 1 variable that is categorical - assert find_categorical_and_numerical_variables(df_vartypes, ["Name"]) == ( - ["Name"], - [], - ) - assert find_categorical_and_numerical_variables(df_vartypes, "Name") == ( - ["Name"], - [], - ) + assert find_categorical_and_numerical_variables(df, ["Name"]) == (["Name"], []) + assert find_categorical_and_numerical_variables(df, "Name") == (["Name"], []) # Case 2: user passes 1 variable that is numerical - assert find_categorical_and_numerical_variables(df_vartypes, ["Age"]) == ( - [], - ["Age"], - ) - assert find_categorical_and_numerical_variables(df_vartypes, "Age") == ( - [], - ["Age"], - ) + assert find_categorical_and_numerical_variables(df, ["Age"]) == ([], ["Age"]) + assert find_categorical_and_numerical_variables(df, "Age") == ([], ["Age"]) # Case 3: user passes 1 categorical and 1 numerical variable - assert find_categorical_and_numerical_variables(df_vartypes, ["Age", "Name"]) == ( + assert find_categorical_and_numerical_variables(df, ["Age", "Name"]) == ( ["Name"], ["Age"], ) -def test_numcat_when_var_is_none(df_vartypes): - # Case 4: automatically identify variables - assert find_categorical_and_numerical_variables(df_vartypes, None) == ( +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numcat_when_var_is_none(make_df): + df = make_df(BASIC_DATA) + + assert find_categorical_and_numerical_variables(df, None) == ( ["Name", "City"], ["Age", "Marks"], ) - assert find_categorical_and_numerical_variables( - df_vartypes[["Name", "City"]], None - ) == (["Name", "City"], []) - assert find_categorical_and_numerical_variables( - df_vartypes[["Age", "Marks"]], None - ) == ([], ["Age", "Marks"]) - - -@pytest.fixture(scope="module") -def dfdt(): - X = pd.DataFrame() - X["date1"] = pd.date_range("2020-02-24", periods=1000, freq="min") - X["date2"] = pd.date_range("2021-09-29", periods=1000, freq="h") - X["date3"] = ["2020-02-24"] * 1000 - return X + assert find_categorical_and_numerical_variables(df[["Name", "City"]], None) == ( + ["Name", "City"], + [], + ) + assert find_categorical_and_numerical_variables(df[["Age", "Marks"]], None) == ( + [], + ["Age", "Marks"], + ) -def test_numcat_raises_no_var_error(dfdt): +@pytest.mark.parametrize( + "make_df, assert_error", [(pd.DataFrame, TypeError), (pl.DataFrame, TypeError)] +) +def test_numcat_raises_no_var_error(make_df, assert_error): # Case 5: error when no variable is numerical or categorical + df = make_df( + { + "date1": DATETIME_DATA["date_range"], + "date2": DATETIME_DATA["date_range_tz"], + } + ) msg = "There are no numerical or categorical variables" - with pytest.raises(TypeError, match=msg): - find_categorical_and_numerical_variables(dfdt, None) + with pytest.raises(assert_error, match=msg): + find_categorical_and_numerical_variables(df, None) msg = "The variable entered is neither numerical nor categorical." - with pytest.raises(TypeError, match=msg): - find_categorical_and_numerical_variables(dfdt, "date1") + with pytest.raises(assert_error, match=msg): + find_categorical_and_numerical_variables(df, "date1") -def test_numcat_raises_no_var_warn(dfdt): - # Case 6: warning when no variable is numerical or categorical +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numcat_raises_no_var_warn(make_df): + df = make_df( + { + "date1": DATETIME_DATA["date_range"], + "date2": DATETIME_DATA["date_range_tz"], + } + ) msg = "There are no numerical or categorical variables" with pytest.warns(UserWarning, match=msg): - find_categorical_and_numerical_variables( - dfdt, - None, - return_empty=True, - ) + find_categorical_and_numerical_variables(df, None, return_empty=True) msg = "The variable entered is neither numerical nor" with pytest.warns(UserWarning, match=msg): find_categorical_and_numerical_variables( - dfdt, variables="date1", return_empty=True + df, variables="date1", return_empty=True ) -def test_numcat_returns_empty_lists(dfdt): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numcat_returns_empty_lists(make_df): + df = make_df( + { + "date1": DATETIME_DATA["date_range"], + "date2": DATETIME_DATA["date_range_tz"], + } + ) assert find_categorical_and_numerical_variables( - dfdt, - None, - return_empty=True, + df, None, return_empty=True ) == ([], []) assert find_categorical_and_numerical_variables( - dfdt, - "date1", - return_empty=True, + df, "date1", return_empty=True ) == ([], []) -def test_numcat_on_user_empty_list(df_vartypes): - # Case 7: user passes empty list +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numcat_on_user_empty_list(make_df): + df = make_df(BASIC_DATA) + msg = "The list of variables provided is empty. If this was" with pytest.raises(ValueError, match=msg): - find_categorical_and_numerical_variables(df_vartypes, []) + find_categorical_and_numerical_variables(df, []) msg = "The list of variables provided is empty. Returning " with pytest.warns(UserWarning, match=msg): - find_categorical_and_numerical_variables(df_vartypes, [], return_empty=True) + find_categorical_and_numerical_variables(df, [], return_empty=True) - assert find_categorical_and_numerical_variables( - df_vartypes, [], return_empty=True - ) == ([], []) + assert find_categorical_and_numerical_variables(df, [], return_empty=True) == ( + [], + [], + ) def test_numcat_when_dt_as_object(df_vartypes): - # Case 8: datetime cast as object + # Case 8: datetime cast as object - pandas-only, `df_vartypes["dob"]` is a + # pandas datetime64 column relying on pandas' `.astype("O")`, which has no + # polars equivalent (polars has no generic object dtype to cast into). df = df_vartypes.copy() df["dob"] = df["dob"].astype("O") - # datetime variable is skipped when automatically finding variables, assert find_categorical_and_numerical_variables(df, None) == ( ["Name", "City"], ["Age", "Marks"], @@ -310,12 +322,43 @@ def test_numcat_when_dt_as_object(df_vartypes): ) -def test_numcat_vars_as_category(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numcat_vars_as_category(make_df): # Case 9: variables cast as category - df = df_vartypes.copy() - df["City"] = df["City"].astype("category") + df = make_df(BASIC_DATA) + df = cast_categorical(df, ["City"]) assert find_categorical_and_numerical_variables(df, None) == ( ["Name", "City"], ["Age", "Marks"], ) assert find_categorical_and_numerical_variables(df, "City") == (["City"], []) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numcat_agrees_with_find_categorical_on_date_like_category(make_df): + # Regression test: the single-variable path used to disagree with + # find_categorical_variables on a date-like category column. + df = make_df({"date_cat": DATETIME_DATA["date_obj0"], "num": BASIC_DATA["Age"]}) + df = cast_categorical(df, ["date_cat"]) + + assert find_categorical_variables(df, return_empty=True) == [] + assert find_categorical_and_numerical_variables(df, None) == ([], ["num"]) + assert find_categorical_and_numerical_variables( + df, "date_cat", return_empty=True + ) == ([], []) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_numcat_exclude_datetime_false_keeps_date_like_category(make_df): + # exclude_datetime=False must be honoured consistently across all three + # entry points, including the single-variable branch. + df = make_df({"date_cat": DATETIME_DATA["date_obj0"], "num": BASIC_DATA["Age"]}) + df = cast_categorical(df, ["date_cat"]) + + assert find_categorical_variables(df, exclude_datetime=False) == ["date_cat"] + assert find_categorical_and_numerical_variables( + df, None, exclude_datetime=False + ) == (["date_cat"], ["num"]) + assert find_categorical_and_numerical_variables( + df, "date_cat", exclude_datetime=False + ) == (["date_cat"], []) diff --git a/tests/test_variable_handling/test_remove_variables.py b/tests/test_variable_handling/test_remove_variables.py deleted file mode 100644 index 3984d2c45..000000000 --- a/tests/test_variable_handling/test_remove_variables.py +++ /dev/null @@ -1,28 +0,0 @@ -import pandas as pd -import pytest - -from feature_engine.variable_handling.retain_variables import retain_variables_if_in_df - -test_dict = [ - ( - pd.DataFrame(columns=["A", "B", "C", "D", "E"]), - ["A", "C", "B", "G", "H"], - ["A", "C", "B"], - ["X", "Y"], - ), - (pd.DataFrame(columns=[1, 2, 3, 4, 5]), [1, 2, 4, 6], [1, 2, 4], [6, 7]), - (pd.DataFrame(columns=[1, 2, 3, 4, 5]), 1, [1], 7), - (pd.DataFrame(columns=["A", "B", "C", "D", "E"]), "C", ["C"], "G"), -] - - -@pytest.mark.parametrize("df, variables, overlap, col_not_in_df", test_dict) -def test_retain_variables_if_in_df(df, variables, overlap, col_not_in_df): - - msg = "None of the variables in the list are present in the dataframe." - - assert retain_variables_if_in_df(df, variables) == overlap - - with pytest.raises(ValueError) as record: - retain_variables_if_in_df(df, col_not_in_df) - assert str(record.value) == msg diff --git a/tests/test_variable_handling/test_retain_variables.py b/tests/test_variable_handling/test_retain_variables.py new file mode 100644 index 000000000..3b839d73f --- /dev/null +++ b/tests/test_variable_handling/test_retain_variables.py @@ -0,0 +1,39 @@ +import pandas as pd +import polars as pl +import pytest + +from feature_engine.variable_handling.retain_variables import retain_variables_if_in_df + +test_dict = [ + (["A", "C", "B", "G", "H"], ["A", "C", "B"], ["X", "Y"]), + ("C", ["C"], "G"), +] + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +@pytest.mark.parametrize("variables, overlap, col_not_in_df", test_dict) +def test_retain_variables_if_in_df(make_df, variables, overlap, col_not_in_df): + df = make_df({"A": [1], "B": [1], "C": [1], "D": [1], "E": [1]}) + + msg = "None of the variables in the list are present in the dataframe." + + assert retain_variables_if_in_df(df, variables) == overlap + + with pytest.raises(ValueError, match=msg): + retain_variables_if_in_df(df, col_not_in_df) + + +def test_retain_variables_if_in_df_int_column_names(): + # polars requires string column names. int-named columns are pandas-only + df = pd.DataFrame({1: [1], 2: [1], 3: [1], 4: [1], 5: [1]}) + + msg = "None of the variables in the list are present in the dataframe." + + assert retain_variables_if_in_df(df, [1, 2, 4, 6]) == [1, 2, 4] + assert retain_variables_if_in_df(df, 1) == [1] + + with pytest.raises(ValueError, match=msg): + retain_variables_if_in_df(df, [6, 7]) + + with pytest.raises(ValueError, match=msg): + retain_variables_if_in_df(df, 7) diff --git a/tests/test_variable_handling/test_variable_type_checks.py b/tests/test_variable_handling/test_variable_type_checks.py new file mode 100644 index 000000000..e09da4438 --- /dev/null +++ b/tests/test_variable_handling/test_variable_type_checks.py @@ -0,0 +1,246 @@ +from datetime import date + +import narwhals as nw +import pandas as pd +import polars as pl + +from feature_engine.variable_handling._variable_type_checks import ( + _is_categorical_and_is_datetime, + _is_categorical_and_is_not_datetime, + _is_categories_num, + _is_convertible_to_dt, + _is_convertible_to_num, + _is_date_or_datetime, + _looks_like_date_string, +) + + +def nw_series(values, dtype=None): + s = pl.Series("x", values) + if dtype is not None: + s = s.cast(dtype) + return nw.from_native(s, series_only=True) + + +def nw_pandas_series(values, dtype=None): + s = pd.Series(values, dtype=dtype) + return nw.from_native(s, series_only=True) + + +def test_is_date_or_datetime(): + """A dtype is a date or datetime if it is narwhals' Date or Datetime type.""" + assert _is_date_or_datetime(nw_series([date(2020, 1, 1)]).dtype) is True + assert ( + _is_date_or_datetime(nw_series(["2020-01-01"]).str.to_datetime().dtype) + is True + ) + assert _is_date_or_datetime(nw_series(["a", "b"]).dtype) is False + assert _is_date_or_datetime(nw_series([1, 2, 3]).dtype) is False + + +def test_looks_like_date_string(): + """A string looks like a date if dateutil finds at least 2 date/time fields + in it - this rejects bare numbers that dateutil would otherwise happily + "parse" as a single field (e.g. a day), while still accepting real dates + in non-ISO formats and bare times. + """ + # real dates, including non-ISO formats + assert _looks_like_date_string("2020-01-01") is True + assert _looks_like_date_string("01-Jan-2010") is True + assert _looks_like_date_string("10/11/12") is True + + # bare times + assert _looks_like_date_string("21:45:23") is True + assert _looks_like_date_string("08:00") is True + + # partial dates + assert _looks_like_date_string("Jan 2020") is True + + # bare numbers dateutil could misparse as a single date/time field + assert _looks_like_date_string("20") is False + assert _looks_like_date_string("1999") is False + assert _looks_like_date_string("12") is False + + # non-date garbage + assert _looks_like_date_string("hello") is False + assert _looks_like_date_string("") is False + + # non-string values (e.g. from a mixed-type pandas Object column) must not + # raise, they simply aren't date strings + assert _looks_like_date_string(20) is False + assert _looks_like_date_string(1.5) is False + assert _looks_like_date_string(None) is False + + +def test_is_convertible_to_num(): + """A series is convertible to numeric if every non-null value can be cast + to float. + """ + assert _is_convertible_to_num(nw_series(["20", "21", "19"])) is True + assert _is_convertible_to_num(nw_series(["a", "b"])) is False + assert ( + _is_convertible_to_num(nw_series(["20", "21"], dtype=pl.Categorical)) + is True + ) + + # object dtype columns (pandas-only concept - narwhals classifies a plain + # object dtype column of ints as `nw.Object`, not `nw.String`) + assert _is_convertible_to_num(nw_pandas_series([1, 2], dtype="object")) is True + assert ( + _is_convertible_to_num( + nw_pandas_series([pd.Timestamp("2020-01-01")], dtype="object") + ) + is False + ) + + +def test_is_convertible_to_dt(): + """A series is convertible to datetime if every non-null value is either a + real date/datetime object, or a string that looks like a date. + """ + assert _is_convertible_to_dt(nw_series(["2020-01-01", "2020-01-02"])) is True + assert _is_convertible_to_dt(nw_series(["a", "b"])) is False + assert _is_convertible_to_dt(nw_series(["20", "21"])) is False + + # flexible, dateutil-backed date guessing works for every backend now, not + # just pandas - so non-ISO formats are recognised here too + assert _is_convertible_to_dt(nw_series(["01-Jan-2010"])) is True + assert _is_convertible_to_dt(nw_series(["10/11/12"])) is True + + # an object dtype column holding actual datetime objects (e.g. pandas + # Timestamps) is trivially convertible, without needing to parse anything + assert ( + _is_convertible_to_dt( + nw_pandas_series([pd.Timestamp("2020-01-01")], dtype="object") + ) + is True + ) + + +def test_is_categories_num(): + """A categorical series' categories are numeric if their dtype is numeric - + only possible for pandas, since polars categories are always string-backed. + """ + non_numeric_cat = nw_series(["a", "b", "c"], dtype=pl.Categorical) + assert _is_categories_num(non_numeric_cat) is False + + numeric_cat = nw_pandas_series([20, 21, 19, 18], dtype="category") + assert _is_categories_num(numeric_cat) is True + + +def test_is_categorical_and_is_datetime(): + """A series is categorical-and-datetime if it is a Categorical/String/Object + column whose values are dates, but not an Enum (an explicit category set is + never treated as a datetime) or a numeric-backed categorical. + """ + assert ( + _is_categorical_and_is_datetime( + nw_series(["2020-01-01", "2020-01-02"], dtype=pl.Categorical) + ) + is True + ) + assert ( + _is_categorical_and_is_datetime(nw_series(["a", "b"], dtype=pl.Categorical)) + is False + ) + assert _is_categorical_and_is_datetime(nw_series(["2020-01-01"])) is True + assert _is_categorical_and_is_datetime(nw_series(["20", "21"])) is False + assert _is_categorical_and_is_datetime(nw_series(["a", "b"])) is False + + # an explicit Enum is always treated as categorical, never as datetime + enum_dtype = pl.Enum(["2020-01-01", "2020-01-02"]) + assert ( + _is_categorical_and_is_datetime( + nw_series(["2020-01-01", "2020-01-02"], dtype=enum_dtype) + ) + is False + ) + + # numeric should be False + assert _is_categorical_and_is_datetime(nw_series([1, 2, 3])) is False + + # a numeric-backed categorical (pandas-only - polars categories are always + # string-backed) can never be a datetime, regardless of the categories + numeric_cat = nw_pandas_series([20, 21, 19, 18], dtype="category") + assert _is_categorical_and_is_datetime(numeric_cat) is False + + # a string-dtype pandas column with datetime-like values + assert ( + _is_categorical_and_is_datetime( + nw_pandas_series(["2020-01-01", "2020-01-02"], dtype="string") + ) + is True + ) + + # object dtype column holding actual Timestamp objects + assert ( + _is_categorical_and_is_datetime( + nw_pandas_series([pd.Timestamp("2020-01-01")], dtype="object") + ) + is True + ) + + # object dtype column holding plain ints - not a datetime + assert ( + _is_categorical_and_is_datetime(nw_pandas_series([1, 2], dtype="object")) + is False + ) + + +def test_is_categorical_and_is_not_datetime(): + """A series is categorical-and-not-datetime if it is a Categorical/String/ + Object/Enum column whose values are not dates. + """ + assert ( + _is_categorical_and_is_not_datetime( + nw_series(["2020-01-01", "2020-01-02"], dtype=pl.Categorical) + ) + is False + ) + assert ( + _is_categorical_and_is_not_datetime( + nw_series(["a", "b"], dtype=pl.Categorical) + ) + is True + ) + assert _is_categorical_and_is_not_datetime(nw_series(["2020-01-01"])) is False + assert _is_categorical_and_is_not_datetime(nw_series(["20", "21"])) is True + assert _is_categorical_and_is_not_datetime(nw_series(["a", "b"])) is True + + # an explicit Enum is always treated as categorical + assert ( + _is_categorical_and_is_not_datetime( + nw_series(["a", "b"], dtype=pl.Enum(["a", "b"])) + ) + is True + ) + + # numeric should be False + assert _is_categorical_and_is_not_datetime(nw_series([1, 2, 3])) is False + + # a numeric-backed categorical is categorical-and-not-datetime + numeric_cat = nw_pandas_series([20, 21, 19, 18], dtype="category") + assert _is_categorical_and_is_not_datetime(numeric_cat) is True + + # object dtype column of plain ints + assert ( + _is_categorical_and_is_not_datetime(nw_pandas_series([1, 2], dtype="object")) + is True + ) + + # object dtype column holding actual Timestamp objects - is a datetime, so + # not "categorical and not datetime" + assert ( + _is_categorical_and_is_not_datetime( + nw_pandas_series([pd.Timestamp("2020-01-01")], dtype="object") + ) + is False + ) + + # string-dtype pandas column not convertible to numeric or datetime + assert ( + _is_categorical_and_is_not_datetime( + nw_pandas_series(["a", "b"], dtype="string") + ) + is True + ) diff --git a/tests/test_wrappers/test_check_estimator_wrappers.py b/tests/test_wrappers/test_check_estimator_wrappers.py index 1125e3a08..75ab4b9c0 100644 --- a/tests/test_wrappers/test_check_estimator_wrappers.py +++ b/tests/test_wrappers/test_check_estimator_wrappers.py @@ -1,10 +1,8 @@ import pandas as pd import pytest -import sklearn from sklearn.impute import SimpleImputer from sklearn.preprocessing import OrdinalEncoder, StandardScaler from sklearn.utils.estimator_checks import check_estimator -from sklearn.utils.fixes import parse_version from feature_engine.wrappers import SklearnWrapper from tests.estimator_checks.estimator_checks import ( @@ -18,29 +16,17 @@ check_numerical_variables_assignment, ) -sklearn_version = parse_version(parse_version(sklearn.__version__).base_version) -if sklearn_version < parse_version("1.6"): - - def test_sklearn_transformer_wrapper(): - check_estimator(SklearnWrapper(transformer=SimpleImputer())) - -else: - - def test_sklearn_transformer_wrapper(): - check_estimator( - estimator=wrap_for_check_estimator( - SklearnWrapper(transformer=SimpleImputer()) - ), - expected_failed_checks=SklearnWrapper( - transformer=SimpleImputer() - )._more_tags()["_xfail_checks"], - ) +def test_sklearn_transformer_wrapper(): + check_estimator( + estimator=wrap_for_check_estimator(SklearnWrapper(transformer=SimpleImputer())), + expected_failed_checks=SklearnWrapper(transformer=SimpleImputer())._more_tags()[ + "_xfail_checks" + ], + ) -@pytest.mark.parametrize( - "estimator", [SklearnWrapper(transformer=OrdinalEncoder())] -) +@pytest.mark.parametrize("estimator", [SklearnWrapper(transformer=OrdinalEncoder())]) def test_check_estimator_from_feature_engine(estimator): check_raises_non_fitted_error(estimator) check_raises_error_when_input_not_a_df(estimator) @@ -48,12 +34,8 @@ def test_check_estimator_from_feature_engine(estimator): def test_check_variables_assignment(): - check_numerical_variables_assignment( - SklearnWrapper(transformer=StandardScaler()) - ) - check_all_types_variables_assignment( - SklearnWrapper(transformer=OrdinalEncoder()) - ) + check_numerical_variables_assignment(SklearnWrapper(transformer=StandardScaler())) + check_all_types_variables_assignment(SklearnWrapper(transformer=OrdinalEncoder())) def test_raises_error_when_no_transformer_passed(): diff --git a/tests/test_wrappers/test_sklearn_wrapper.py b/tests/test_wrappers/test_sklearn_wrapper.py index f15063cd9..76e816c7e 100644 --- a/tests/test_wrappers/test_sklearn_wrapper.py +++ b/tests/test_wrappers/test_sklearn_wrapper.py @@ -60,13 +60,7 @@ def _OneHotEncoder(sparse, drop=None, dtype=np.float64) -> OneHotEncoder: - """OneHotEncoder sparse argument has been renamed as sparse_output - in scikitlearn >=1.2""" - - if skl_version.split(".")[0] == "1" and int(skl_version.split(".")[1]) >= 2: - return OneHotEncoder(sparse_output=sparse, drop=drop, dtype=dtype) - else: - return OneHotEncoder(sparse=sparse, drop=drop, dtype=dtype) + return OneHotEncoder(sparse_output=sparse, drop=drop, dtype=dtype) @pytest.mark.parametrize( diff --git a/tox.ini b/tox.ini index c31b3edd1..4a21bd8ee 100644 --- a/tox.ini +++ b/tox.ini @@ -1,9 +1,5 @@ [tox] envlist = - py39 - py310 - py311-sklearn150 - py311-sklearn160 py311-sklearn170 py312-pandas230 py312-pandas300 @@ -32,14 +28,6 @@ commands = # Python versions # ------------------------- -[testenv:py39] -deps = - .[tests] - -[testenv:py310] -deps = - .[tests] - [testenv:py313] deps = .[tests] @@ -53,16 +41,6 @@ deps = # scikit-learn matrix # ------------------------- -[testenv:py311-sklearn150] -deps = - .[tests] - scikit-learn==1.5.1 - -[testenv:py311-sklearn160] -deps = - .[tests] - scikit-learn==1.6.1 - [testenv:py311-sklearn170] deps = .[tests]