From c28163d7dccc998c378b87a0b65062fda07b4657 Mon Sep 17 00:00:00 2001 From: adrinjalali Date: Sun, 27 Sep 2026 10:34:03 +0100 Subject: [PATCH 1/8] FEAT add pandas support --- docs/changes.rst | 14 + docs/persistence.rst | 9 +- docs/requirements.txt | 2 +- pixi.lock | 1109 ++++++++++++++++++------ pyproject.toml | 15 +- skops/io/_pandas.py | 443 ++++++++++ skops/io/_persist.py | 14 +- skops/io/_trusted_types.py | 48 + skops/io/tests/data/pandas-2.0.3.skops | Bin 0 -> 66609 bytes skops/io/tests/data/pandas-3.0.3.skops | Bin 0 -> 78724 bytes skops/io/tests/test_pandas.py | 359 ++++++++ 11 files changed, 1732 insertions(+), 281 deletions(-) create mode 100644 skops/io/_pandas.py create mode 100644 skops/io/tests/data/pandas-2.0.3.skops create mode 100644 skops/io/tests/data/pandas-3.0.3.skops create mode 100644 skops/io/tests/test_pandas.py diff --git a/docs/changes.rst b/docs/changes.rst index 5ff4a358..8d893b8b 100644 --- a/docs/changes.rst +++ b/docs/changes.rst @@ -9,6 +9,20 @@ skops Changelog :depth: 1 :local: +v0.17 +----- +- Add support for pandas objects: :class:`~pandas.DataFrame`, + :class:`~pandas.Series`, every kind of :class:`~pandas.Index`, extension + arrays and extension dtypes can now be saved and loaded. They are stored as + the numpy arrays and scalars they are made of and rebuilt through the public + pandas constructors, so no pandas internals end up in the file, and they are + trusted by default. Estimators from other libraries that keep pandas objects + in their fitted attributes, such as ``category_encoders``, can now be + persisted. A file written with one pandas version loads with any other from + 2.0 on, keeping the dtypes of the version that wrote it. The ``freq`` of + datetime-like indexes and the ``attrs`` of a Series or DataFrame are not + preserved. :issue:`450` and :pr:`XXX` by `Adrin Jalali`_. + v0.16 ----- - Fix loading of time-zone-aware ``datetime.datetime`` and ``datetime.time`` diff --git a/docs/persistence.rst b/docs/persistence.rst index 7aba33d9..8ed87e07 100644 --- a/docs/persistence.rst +++ b/docs/persistence.rst @@ -246,7 +246,14 @@ Supported libraries Skops intends to support all of **scikit-learn**, that is, not only its estimators, but also other classes like cross validation splitters. Furthermore, most types from **numpy** and **scipy** should be supported, such as (sparse) -arrays, dtypes, random generators, and ufuncs. +arrays, dtypes, random generators, and ufuncs. **pandas** objects, that is +``DataFrame``, ``Series``, every kind of ``Index``, extension arrays and +extension dtypes, are supported as well with pandas 2.0 or later: they are +stored as the arrays they are made of and rebuilt through the public pandas +constructors, so that no pandas internals end up in the file, and a file +written with one pandas version loads with any other. The ``freq`` of +datetime-like indexes and the ``attrs`` of a ``Series`` or ``DataFrame`` are not +preserved. Apart from this core, we plan to support machine learning libraries commonly used be the community. So far, we have tested the following libraries: diff --git a/docs/requirements.txt b/docs/requirements.txt index 3c5b8c06..6cac6248 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -1,6 +1,6 @@ # to be synced with the versions in pyproject.toml matplotlib>=3.3 -pandas>=1 +pandas>=2 fairlearn>=0.7.0 sphinx>=3.2.0 sphinx-gallery>=0.7.0 diff --git a/pixi.lock b/pixi.lock index 15aa943e..fbe426c5 100644 --- a/pixi.lock +++ b/pixi.lock @@ -117,7 +117,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/openjpeg-2.5.4-heb1ab33_2.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/openldap-2.6.13-hbde042b_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/openssl-3.6.4-h781a0a9_0.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-3.0.5-py312h8ecdadd_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pandoc-3.11-ha770c72_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pcre2-10.47-h8b3dc9c_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pillow-12.3.0-py312h38079b3_4.conda @@ -210,6 +209,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/99/c7/bd05c5c430feb347aa040fcc8870135d70b256718deee9bc7d2ca74a77ff/xgboost-3.4.1-py3-none-manylinux_2_28_x86_64.whl - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl osx-64: @@ -307,7 +307,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/numpy-2.5.3-py314he0ba898_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/openjpeg-2.5.4-he4eae51_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/openssl-3.6.4-h332eb6d_0.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/pandas-3.0.5-py314h99bb933_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pandoc-3.11-h694c41f_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pcre2-10.47-h31793e3_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pillow-12.3.0-py314h909093e_4.conda @@ -335,6 +334,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/cd/05/7213965863cba1ed0150ad045bceed6276a1afaaaedbaeff4699ec4f0ccb/lightgbm-4.7.0-py3-none-macosx_10_15_x86_64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_10_15_x86_64.whl osx-arm64: @@ -432,7 +432,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/numpy-2.5.3-py314he06036b_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openjpeg-2.5.4-h4d1e80c_2.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openssl-3.6.4-h55eecbc_0.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-3.0.5-py314he609de1_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandoc-3.11-hce30654_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pcre2-10.47-he63d830_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pillow-12.3.0-py314h709c99d_4.conda @@ -460,6 +459,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/f7/94/e5c37a8972ad780edc1d8459d1931356344ca133f7f99ba9cfda516b5bba/xgboost-3.4.1-py3-none-macosx_12_0_arm64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_11_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_12_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_12_0_arm64.whl win-64: @@ -494,7 +494,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/pytest-cov-7.1.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-discovery-1.6.0-pyhcf101f3_0.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.3-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.14-9_cp314.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rich-15.0.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/setuptools-84.0.0-pyh332efcf_0.conda @@ -560,7 +559,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/numpy-2.5.3-py314h02f10f6_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/openjpeg-2.5.4-h90fa87c_2.conda - conda: https://conda.anaconda.org/conda-forge/win-64/openssl-3.6.4-hf411b9b_0.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/pandas-3.0.5-py314hf700ef7_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pandoc-3.11-h57928b3_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pcre2-10.47-h8466c1e_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pillow-12.3.0-py314h6acc0e1_4.conda @@ -591,7 +589,9 @@ environments: - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/d5/0b/c5c17d862b12ce292f24cd85d40f2f8f8981668fbdbd43fdc2625eccbc79/lightgbm-4.7.0-py3-none-win_amd64.whl + - pypi: https://files.pythonhosted.org/packages/f9/bc/8737e8d54cf51106118039b83f485a4783112fab49ea9d044b234978a46e/tzdata-2026.4-py2.py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-win_amd64.whl ci-sklearn12: @@ -5959,7 +5959,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/openjpeg-2.5.4-h55fea9a_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/openldap-2.6.13-hbde042b_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/openssl-3.6.3-h35e630c_0.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-3.0.3-py314hb4ffadd_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pandoc-3.11-ha770c72_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pcre2-10.47-haa7fec5_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pillow-12.2.0-py314h8ec4b1a_0.conda @@ -6101,6 +6100,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/7b/91/984aca2ec129e2757d1e4e3c81c3fcda9d0f85b74670a094cc443d9ee949/joblib-1.5.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl osx-64: @@ -6248,7 +6248,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/numpy-2.5.0-py314h7b24d9b_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/openjpeg-2.5.4-h52bb76a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/openssl-3.6.3-hc881268_0.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/pandas-3.0.3-py314h99bb933_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pandoc-3.11-h694c41f_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pcre2-10.47-h13923f0_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pillow-12.2.0-py314hc904d5e_0.conda @@ -6276,6 +6275,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/f2/75/cffc9962cca296bc5536896b7e65b4a7cdeb8db208e71b9c0133c08f8f7e/lightgbm-4.6.0-py3-none-macosx_10_15_x86_64.whl - pypi: https://files.pythonhosted.org/packages/fc/72/3b68983c0215ef65d48e9eeb1f168c3c6e3d62a61ece605de3209c79cae1/xgboost-3.3.0-py3-none-macosx_10_15_x86_64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_10_15_x86_64.whl osx-arm64: @@ -6423,7 +6423,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/numpy-2.5.0-py314hb79c6fa_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openjpeg-2.5.4-hd9e9057_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openssl-3.6.3-hd24854e_0.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-3.0.3-py314he609de1_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandoc-3.11-hce30654_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pcre2-10.47-h30297fc_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pillow-12.2.0-py314hab283cf_0.conda @@ -6451,6 +6450,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/c9/62/b49e756822b29909d0c95ed334662dc6c7c81a99ec6bc10dc18e69f3d6e7/xgboost-3.3.0-py3-none-macosx_12_0_arm64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_11_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_12_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_12_0_arm64.whl win-64: @@ -6511,7 +6511,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/pytest-cov-7.1.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-discovery-1.4.2-pyhcf101f3_0.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.14-8_cp314.conda - conda: https://conda.anaconda.org/conda-forge/noarch/requests-2.34.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rich-15.0.0-pyhcf101f3_0.conda @@ -6605,7 +6604,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/onemkl-license-2026.0.0-h57928b3_908.conda - conda: https://conda.anaconda.org/conda-forge/win-64/openjpeg-2.5.4-h0e57b4f_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/openssl-3.6.3-hf411b9b_0.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/pandas-3.0.3-py314hf700ef7_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pandoc-3.11-h57928b3_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pcre2-10.47-hd2b5f0e_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pillow-12.2.0-py314h61b30b5_0.conda @@ -6639,7 +6637,9 @@ environments: - pypi: https://files.pythonhosted.org/packages/5e/23/f8b28ca248bb629b9e08f877dd2965d1994e1674a03d67cd10c5246da248/lightgbm-4.6.0-py3-none-win_amd64.whl - pypi: https://files.pythonhosted.org/packages/7b/91/984aca2ec129e2757d1e4e3c81c3fcda9d0f85b74670a094cc443d9ee949/joblib-1.5.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl + - pypi: https://files.pythonhosted.org/packages/f9/bc/8737e8d54cf51106118039b83f485a4783112fab49ea9d044b234978a46e/tzdata-2026.4-py2.py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-win_amd64.whl docs: @@ -6734,7 +6734,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/openjpeg-2.5.4-h55fea9a_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/openldap-2.6.13-hbde042b_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/openssl-3.6.3-h35e630c_0.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-3.0.3-py314hb4ffadd_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pcre2-10.47-haa7fec5_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pillow-12.2.0-py314h8ec4b1a_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pixman-0.46.4-h54a6638_1.conda @@ -6835,6 +6834,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/7f/c4/bc41eb19b0fd0db868f4132920879019318d80cc522ad8f2bca4611af808/scipy-1.18.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/bd/6e/95b0e537de1f4d4301f76f944642c6da50d1511cc7b3d64dc418a66c7509/wcwidth-0.8.1-py3-none-any.whl + - pypi: https://files.pythonhosted.org/packages/ca/ba/ffdcb19be4ff6bfe7d969e7cef2c567c633df5a3a1cc1053394ad053bca8/pandas-3.0.6-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl - pypi: https://files.pythonhosted.org/packages/f0/af/4d72d9e475ac83719160c662619e4bf7b95c19507cd582e7d0167a3c3dae/scikit_learn-1.9.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl osx-64: @@ -6946,7 +6946,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/numpy-2.5.0-py314h7b24d9b_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/openjpeg-2.5.4-h52bb76a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/openssl-3.6.3-hc881268_0.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/pandas-3.0.3-py314h99bb933_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pcre2-10.47-h13923f0_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pillow-12.2.0-py314hc904d5e_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pixman-0.46.4-ha059160_1.conda @@ -6964,6 +6963,7 @@ environments: - pypi: ./ - pypi: https://files.pythonhosted.org/packages/32/d5/f9a850d79b0851d1d4ef6456097579a9005b31fea68726a4ae5f2d82ddd9/threadpoolctl-3.6.0-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/48/ca/36339329c4604adbcc99c899b7eb1ce1a555c499b6a6860757dc9bfed36d/narwhals-2.22.1-py3-none-any.whl + - pypi: https://files.pythonhosted.org/packages/75/55/1a8875395b05ccd572cbca0b9255dcd2db6e6508e632a558c1a6884b39ad/pandas-3.0.6-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://files.pythonhosted.org/packages/78/b5/915a19b3de2f7430062b509653563db1633ddbb6f021b06731521115d4e2/scipy-1.18.0-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://files.pythonhosted.org/packages/7b/91/984aca2ec129e2757d1e4e3c81c3fcda9d0f85b74670a094cc443d9ee949/joblib-1.5.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl @@ -7079,7 +7079,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/numpy-2.5.0-py314hb79c6fa_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openjpeg-2.5.4-hd9e9057_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openssl-3.6.3-hd24854e_0.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-3.0.3-py314he609de1_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pcre2-10.47-h30297fc_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pillow-12.2.0-py314hab283cf_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pixman-0.46.4-h81086ad_1.conda @@ -7096,6 +7095,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/zstd-1.5.7-hbf9d68e_6.conda - pypi: ./ - pypi: https://files.pythonhosted.org/packages/32/d5/f9a850d79b0851d1d4ef6456097579a9005b31fea68726a4ae5f2d82ddd9/threadpoolctl-3.6.0-py3-none-any.whl + - pypi: https://files.pythonhosted.org/packages/35/61/47ae13476995cc8a40cd609e93e7cf11f273d8692925c2903cb6d38aa0d1/pandas-3.0.6-cp314-cp314-macosx_11_0_arm64.whl - pypi: https://files.pythonhosted.org/packages/3c/a7/552a7821597c632b907f7bfe8f36f9f572777af8ef8a48353041cf8e091a/scikit_learn-1.9.0-cp314-cp314-macosx_12_0_arm64.whl - pypi: https://files.pythonhosted.org/packages/48/ca/36339329c4604adbcc99c899b7eb1ce1a555c499b6a6860757dc9bfed36d/narwhals-2.22.1-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/7b/91/984aca2ec129e2757d1e4e3c81c3fcda9d0f85b74670a094cc443d9ee949/joblib-1.5.3-py3-none-any.whl @@ -7135,7 +7135,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/pyparsing-3.3.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/pysocks-1.7.1-pyh09c184e_7.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda - - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.14-8_cp314.conda - conda: https://conda.anaconda.org/conda-forge/noarch/requests-2.34.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rich-15.0.0-pyhcf101f3_0.conda @@ -7221,7 +7220,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/onemkl-license-2026.0.0-h57928b3_908.conda - conda: https://conda.anaconda.org/conda-forge/win-64/openjpeg-2.5.4-h0e57b4f_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/openssl-3.6.3-hf411b9b_0.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/pandas-3.0.3-py314hf700ef7_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pcre2-10.47-hd2b5f0e_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pillow-12.2.0-py314h61b30b5_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pixman-0.46.4-h5112557_1.conda @@ -7250,7 +7248,9 @@ environments: - pypi: https://files.pythonhosted.org/packages/93/3e/902d836831474b0ab5a37d16404f7bc5fafd9efba632890e271ba952635f/scipy-1.18.0-cp314-cp314-win_amd64.whl - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/a2/d5/6a58eea2cb9abbb9b3f2bb8b2cfb3243d1152d69f442d256c7af71304769/scikit_learn-1.9.0-cp314-cp314-win_amd64.whl + - pypi: https://files.pythonhosted.org/packages/b7/e9/f43410fada510b43fec09993c08f552086c3d247d3ee801a678f3cb10ea5/pandas-3.0.6-cp314-cp314-win_amd64.whl - pypi: https://files.pythonhosted.org/packages/bd/6e/95b0e537de1f4d4301f76f944642c6da50d1511cc7b3d64dc418a66c7509/wcwidth-0.8.1-py3-none-any.whl + - pypi: https://files.pythonhosted.org/packages/f9/bc/8737e8d54cf51106118039b83f485a4783112fab49ea9d044b234978a46e/tzdata-2026.4-py2.py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl lint: channels: @@ -12411,64 +12411,6 @@ packages: run_exports: {} size: 15303815 timestamp: 1778602611222 -- conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-3.0.5-py312h8ecdadd_1.conda - sha256: 393e529c0574c020a9b790275af10ff35e21cbb4e12740888bd3f214f2f07034 - md5: 85eb29ade84d1bd97be5ec55607efb00 - depends: - - python - - numpy >=1.26.0 - - python-dateutil >=2.8.2 - - __glibc >=2.17,<3.0.a0 - - libgcc >=14 - - libstdcxx >=14 - - numpy >=1.23,<3 - - python_abi 3.12.* *_cp312 - constrains: - - adbc-driver-postgresql >=1.2.0 - - adbc-driver-sqlite >=1.2.0 - - beautifulsoup4 >=4.12.3 - - blosc >=1.21.3 - - bottleneck >=1.4.2 - - fastparquet >=2024.11.0 - - fsspec >=2024.10.0 - - gcsfs >=2024.10.0 - - html5lib >=1.1 - - hypothesis >=6.116.0 - - jinja2 >=3.1.5 - - lxml >=5.3.0 - - matplotlib >=3.9.3 - - numba >=0.60.0 - - numexpr >=2.10.2 - - odfpy >=1.4.1 - - openpyxl >=3.1.5 - - psycopg2 >=2.9.10 - - pyarrow >=13.0.0 - - pyiceberg >=0.8.1 - - pymysql >=1.1.1 - - pyqt5 >=5.15.9 - - pyreadstat >=1.2.8 - - pytables >=3.10.1 - - pytest >=8.3.4 - - pytest-xdist >=3.6.1 - - python-calamine >=0.3.0 - - pytz >=2024.2 - - pyxlsb >=1.0.10 - - qtpy >=2.4.2 - - scipy >=1.14.1 - - s3fs >=2024.10.0 - - sqlalchemy >=2.0.36 - - tabulate >=0.9.0 - - xarray >=2024.10.0 - - xlrd >=2.0.1 - - xlsxwriter >=3.2.0 - - zstandard >=0.23.0 - license: BSD-3-Clause - license_family: BSD - purls: - - pkg:pypi/pandas?source=hash-mapping - run_exports: {} - size: 14936796 - timestamp: 1785276227779 - conda: https://conda.anaconda.org/conda-forge/linux-64/pandoc-3.11-ha770c72_0.conda sha256: cb14839e33fbbb2163f259476091dd99f9e83fbb1690b789cd1e6e6a05d17db7 md5: 2c71f77db75fa5e9dd64abb706c66788 @@ -16491,18 +16433,6 @@ packages: run_exports: {} size: 146639 timestamp: 1777068997932 -- conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.3-pyhd8ed1ab_0.conda - sha256: 3f05db78cf8be33cf6dbc469664b8e3a01f3980d61d6d6bef48669b171896d8a - md5: eefc8d916bd2e708d76d40398ef9a1ee - depends: - - python >=3.10 - license: Apache-2.0 - license_family: APACHE - purls: - - pkg:pypi/tzdata?source=hash-mapping - run_exports: {} - size: 146862 - timestamp: 1783704822814 - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.10-8_cp310.conda build_number: 8 sha256: 7ad76fa396e4bde336872350124c0819032a9e8a0a40590744ff9527b54351c1 @@ -20495,63 +20425,6 @@ packages: run_exports: {} size: 14597208 timestamp: 1778602856255 -- conda: https://conda.anaconda.org/conda-forge/osx-64/pandas-3.0.5-py314h99bb933_1.conda - sha256: 7910c507692b67567627edbed6ff0ce80bb800b704bd0b238729e4c64d9fbc4b - md5: 7c7abc0faf169efe0e103c62f2824857 - depends: - - python - - numpy >=1.26.0 - - python-dateutil >=2.8.2 - - libcxx >=19 - - __osx >=11.0 - - numpy >=1.23,<3 - - python_abi 3.14.* *_cp314 - constrains: - - adbc-driver-postgresql >=1.2.0 - - adbc-driver-sqlite >=1.2.0 - - beautifulsoup4 >=4.12.3 - - blosc >=1.21.3 - - bottleneck >=1.4.2 - - fastparquet >=2024.11.0 - - fsspec >=2024.10.0 - - gcsfs >=2024.10.0 - - html5lib >=1.1 - - hypothesis >=6.116.0 - - jinja2 >=3.1.5 - - lxml >=5.3.0 - - matplotlib >=3.9.3 - - numba >=0.60.0 - - numexpr >=2.10.2 - - odfpy >=1.4.1 - - openpyxl >=3.1.5 - - psycopg2 >=2.9.10 - - pyarrow >=13.0.0 - - pyiceberg >=0.8.1 - - pymysql >=1.1.1 - - pyqt5 >=5.15.9 - - pyreadstat >=1.2.8 - - pytables >=3.10.1 - - pytest >=8.3.4 - - pytest-xdist >=3.6.1 - - python-calamine >=0.3.0 - - pytz >=2024.2 - - pyxlsb >=1.0.10 - - qtpy >=2.4.2 - - scipy >=1.14.1 - - s3fs >=2024.10.0 - - sqlalchemy >=2.0.36 - - tabulate >=0.9.0 - - xarray >=2024.10.0 - - xlrd >=2.0.1 - - xlsxwriter >=3.2.0 - - zstandard >=0.23.0 - license: BSD-3-Clause - license_family: BSD - purls: - - pkg:pypi/pandas?source=hash-mapping - run_exports: {} - size: 14639133 - timestamp: 1785276464478 - conda: https://conda.anaconda.org/conda-forge/osx-64/pandoc-3.11-h694c41f_0.conda sha256: 0ef2d60b6d9ae28d4e70fae3fbb716e751db77743b6f5d18dc298fe13e280827 md5: 932c47502f555da85dcf6f562b3e36a5 @@ -25343,64 +25216,6 @@ packages: run_exports: {} size: 14368928 timestamp: 1778602917992 -- conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-3.0.5-py314he609de1_0.conda - sha256: 9db72f83b107fab7393e02c4005003c97c7a7aaffc33417e78aadb5a82056cd5 - md5: b1dde16791eca59c8316d0ad1487bff0 - depends: - - python - - numpy >=1.26.0 - - python-dateutil >=2.8.2 - - python 3.14.* *_cp314 - - libcxx >=19 - - __osx >=11.0 - - numpy >=1.23,<3 - - python_abi 3.14.* *_cp314 - constrains: - - adbc-driver-postgresql >=1.2.0 - - adbc-driver-sqlite >=1.2.0 - - beautifulsoup4 >=4.12.3 - - blosc >=1.21.3 - - bottleneck >=1.4.2 - - fastparquet >=2024.11.0 - - fsspec >=2024.10.0 - - gcsfs >=2024.10.0 - - html5lib >=1.1 - - hypothesis >=6.116.0 - - jinja2 >=3.1.5 - - lxml >=5.3.0 - - matplotlib >=3.9.3 - - numba >=0.60.0 - - numexpr >=2.10.2 - - odfpy >=1.4.1 - - openpyxl >=3.1.5 - - psycopg2 >=2.9.10 - - pyarrow >=13.0.0 - - pyiceberg >=0.8.1 - - pymysql >=1.1.1 - - pyqt5 >=5.15.9 - - pyreadstat >=1.2.8 - - pytables >=3.10.1 - - pytest >=8.3.4 - - pytest-xdist >=3.6.1 - - python-calamine >=0.3.0 - - pytz >=2024.2 - - pyxlsb >=1.0.10 - - qtpy >=2.4.2 - - scipy >=1.14.1 - - s3fs >=2024.10.0 - - sqlalchemy >=2.0.36 - - tabulate >=0.9.0 - - xarray >=2024.10.0 - - xlrd >=2.0.1 - - xlsxwriter >=3.2.0 - - zstandard >=0.23.0 - license: BSD-3-Clause - license_family: BSD - purls: - - pkg:pypi/pandas?source=hash-mapping - run_exports: {} - size: 14402418 - timestamp: 1784821943837 - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandoc-3.11-hce30654_0.conda sha256: 1520d26e93867d62aa105d6f50e279c446d4bcd79f8db8d5b6a8442b148e8389 md5: a2f9df29541bc9c9ec08950fcedfea84 @@ -30708,65 +30523,6 @@ packages: run_exports: {} size: 14062915 timestamp: 1778602665890 -- conda: https://conda.anaconda.org/conda-forge/win-64/pandas-3.0.5-py314hf700ef7_1.conda - sha256: 399c1d7f7918986d9c1a646f4e789ac24749149598d16b529febc3361e9aad23 - md5: a519b50147401cf6bad2fd188028ad38 - depends: - - python - - numpy >=1.26.0 - - python-dateutil >=2.8.2 - - python-tzdata - - vc >=14.3,<15 - - vc14_runtime >=14.44.35208 - - ucrt >=10.0.20348.0 - - python_abi 3.14.* *_cp314 - - numpy >=1.23,<3 - constrains: - - adbc-driver-postgresql >=1.2.0 - - adbc-driver-sqlite >=1.2.0 - - beautifulsoup4 >=4.12.3 - - blosc >=1.21.3 - - bottleneck >=1.4.2 - - fastparquet >=2024.11.0 - - fsspec >=2024.10.0 - - gcsfs >=2024.10.0 - - html5lib >=1.1 - - hypothesis >=6.116.0 - - jinja2 >=3.1.5 - - lxml >=5.3.0 - - matplotlib >=3.9.3 - - numba >=0.60.0 - - numexpr >=2.10.2 - - odfpy >=1.4.1 - - openpyxl >=3.1.5 - - psycopg2 >=2.9.10 - - pyarrow >=13.0.0 - - pyiceberg >=0.8.1 - - pymysql >=1.1.1 - - pyqt5 >=5.15.9 - - pyreadstat >=1.2.8 - - pytables >=3.10.1 - - pytest >=8.3.4 - - pytest-xdist >=3.6.1 - - python-calamine >=0.3.0 - - pytz >=2024.2 - - pyxlsb >=1.0.10 - - qtpy >=2.4.2 - - scipy >=1.14.1 - - s3fs >=2024.10.0 - - sqlalchemy >=2.0.36 - - tabulate >=0.9.0 - - xarray >=2024.10.0 - - xlrd >=2.0.1 - - xlsxwriter >=3.2.0 - - zstandard >=0.23.0 - license: BSD-3-Clause - license_family: BSD - purls: - - pkg:pypi/pandas?source=hash-mapping - run_exports: {} - size: 14103773 - timestamp: 1785276287020 - conda: https://conda.anaconda.org/conda-forge/win-64/pandoc-3.11-h57928b3_0.conda sha256: 03b03e1c841f25e738bafe60e20cae65a98951683f14e72aae6a8ac359e0501c md5: 4587e6544ff96ef0ee50b042f9d3719d @@ -33186,6 +32942,96 @@ packages: version: 3.6.0 sha256: 43a0b8fd5a2928500110039e43a5eed8480b918967083ea48dc3ab9f13c4a7fb requires_python: '>=3.9' +- pypi: https://files.pythonhosted.org/packages/35/61/47ae13476995cc8a40cd609e93e7cf11f273d8692925c2903cb6d38aa0d1/pandas-3.0.6-cp314-cp314-macosx_11_0_arm64.whl + name: pandas + version: 3.0.6 + sha256: ff51a4459ed036e93d1eb1bb5e6e7b28685d3cb6b7c12b91c05b31024e234729 + requires_dist: + - numpy>=1.26.0 ; python_full_version < '3.14' + - numpy>=2.3.3 ; python_full_version >= '3.14' + - python-dateutil>=2.8.2 + - tzdata ; sys_platform == 'win32' + - tzdata ; sys_platform == 'emscripten' + - hypothesis>=6.116.0 ; extra == 'test' + - pytest>=8.3.4,<9.1 ; extra == 'test' + - pytest-xdist>=3.6.1 ; extra == 'test' + - pyarrow>=13.0.0 ; extra == 'pyarrow' + - bottleneck>=1.4.2 ; extra == 'performance' + - numba>=0.60.0 ; extra == 'performance' + - numexpr>=2.10.2 ; extra == 'performance' + - scipy>=1.14.1 ; extra == 'computation' + - xarray>=2024.10.0 ; extra == 'computation' + - fsspec>=2024.10.0 ; extra == 'fss' + - s3fs>=2024.10.0 ; extra == 'aws' + - gcsfs>=2024.10.0 ; extra == 'gcp' + - odfpy>=1.4.1 ; extra == 'excel' + - openpyxl>=3.1.5 ; extra == 'excel' + - python-calamine>=0.3.0 ; extra == 'excel' + - pyxlsb>=1.0.10 ; extra == 'excel' + - xlrd>=2.0.1 ; extra == 'excel' + - xlsxwriter>=3.2.0 ; extra == 'excel' + - pyarrow>=13.0.0 ; extra == 'parquet' + - pyarrow>=13.0.0 ; extra == 'feather' + - pyiceberg>=0.8.1 ; extra == 'iceberg' + - tables>=3.10.1 ; extra == 'hdf5' + - pyreadstat>=1.2.8 ; extra == 'spss' + - sqlalchemy>=2.0.36 ; extra == 'postgresql' + - psycopg2>=2.9.10 ; extra == 'postgresql' + - adbc-driver-postgresql>=1.2.0 ; extra == 'postgresql' + - sqlalchemy>=2.0.36 ; extra == 'mysql' + - pymysql>=1.1.1 ; extra == 'mysql' + - sqlalchemy>=2.0.36 ; extra == 'sql-other' + - adbc-driver-postgresql>=1.2.0 ; extra == 'sql-other' + - adbc-driver-sqlite>=1.2.0 ; extra == 'sql-other' + - beautifulsoup4>=4.12.3 ; extra == 'html' + - html5lib>=1.1 ; extra == 'html' + - lxml>=5.3.0 ; extra == 'html' + - lxml>=5.3.0 ; extra == 'xml' + - matplotlib>=3.9.3 ; extra == 'plot' + - jinja2>=3.1.5 ; extra == 'output-formatting' + - tabulate>=0.9.0 ; extra == 'output-formatting' + - pyqt5>=5.15.9 ; extra == 'clipboard' + - qtpy>=2.4.2 ; extra == 'clipboard' + - zstandard>=0.23.0 ; extra == 'compression' + - pytz>=2020.1 ; extra == 'timezone' + - adbc-driver-postgresql>=1.2.0 ; extra == 'all' + - adbc-driver-sqlite>=1.2.0 ; extra == 'all' + - beautifulsoup4>=4.12.3 ; extra == 'all' + - bottleneck>=1.4.2 ; extra == 'all' + - fastparquet>=2024.11.0 ; extra == 'all' + - fsspec>=2024.10.0 ; extra == 'all' + - gcsfs>=2024.10.0 ; extra == 'all' + - html5lib>=1.1 ; extra == 'all' + - hypothesis>=6.116.0 ; extra == 'all' + - jinja2>=3.1.5 ; extra == 'all' + - lxml>=5.3.0 ; extra == 'all' + - matplotlib>=3.9.3 ; extra == 'all' + - numba>=0.60.0 ; extra == 'all' + - numexpr>=2.10.2 ; extra == 'all' + - odfpy>=1.4.1 ; extra == 'all' + - openpyxl>=3.1.5 ; extra == 'all' + - psycopg2>=2.9.10 ; extra == 'all' + - pyarrow>=13.0.0 ; extra == 'all' + - pyiceberg>=0.8.1 ; extra == 'all' + - pymysql>=1.1.1 ; extra == 'all' + - pyqt5>=5.15.9 ; extra == 'all' + - pyreadstat>=1.2.8 ; extra == 'all' + - pytest>=8.3.4 ; extra == 'all' + - pytest-xdist>=3.6.1 ; extra == 'all' + - python-calamine>=0.3.0 ; extra == 'all' + - pytz>=2020.1 ; extra == 'all' + - pyxlsb>=1.0.10 ; extra == 'all' + - qtpy>=2.4.2 ; extra == 'all' + - scipy>=1.14.1 ; extra == 'all' + - s3fs>=2024.10.0 ; extra == 'all' + - sqlalchemy>=2.0.36 ; extra == 'all' + - tables>=3.10.1 ; extra == 'all' + - tabulate>=0.9.0 ; extra == 'all' + - xarray>=2024.10.0 ; extra == 'all' + - xlrd>=2.0.1 ; extra == 'all' + - xlsxwriter>=3.2.0 ; extra == 'all' + - zstandard>=0.23.0 ; extra == 'all' + requires_python: '>=3.11' - pypi: https://files.pythonhosted.org/packages/3c/a7/552a7821597c632b907f7bfe8f36f9f572777af8ef8a48353041cf8e091a/scikit_learn-1.9.0-cp314-cp314-macosx_12_0_arm64.whl name: scikit-learn version: 1.9.0 @@ -33389,6 +33235,96 @@ packages: - pandas>=0.24.0 ; extra == 'pandas' - scikit-learn>=0.24.2 ; extra == 'scikit-learn' requires_python: '>=3.7' +- pypi: https://files.pythonhosted.org/packages/75/55/1a8875395b05ccd572cbca0b9255dcd2db6e6508e632a558c1a6884b39ad/pandas-3.0.6-cp314-cp314-macosx_10_15_x86_64.whl + name: pandas + version: 3.0.6 + sha256: ee913a91669056c1de1a6b733fbfeab711de9e54e3bee2dfa5fe79d9457247d1 + requires_dist: + - numpy>=1.26.0 ; python_full_version < '3.14' + - numpy>=2.3.3 ; python_full_version >= '3.14' + - python-dateutil>=2.8.2 + - tzdata ; sys_platform == 'win32' + - tzdata ; sys_platform == 'emscripten' + - hypothesis>=6.116.0 ; extra == 'test' + - pytest>=8.3.4,<9.1 ; extra == 'test' + - pytest-xdist>=3.6.1 ; extra == 'test' + - pyarrow>=13.0.0 ; extra == 'pyarrow' + - bottleneck>=1.4.2 ; extra == 'performance' + - numba>=0.60.0 ; extra == 'performance' + - numexpr>=2.10.2 ; extra == 'performance' + - scipy>=1.14.1 ; extra == 'computation' + - xarray>=2024.10.0 ; extra == 'computation' + - fsspec>=2024.10.0 ; extra == 'fss' + - s3fs>=2024.10.0 ; extra == 'aws' + - gcsfs>=2024.10.0 ; extra == 'gcp' + - odfpy>=1.4.1 ; extra == 'excel' + - openpyxl>=3.1.5 ; extra == 'excel' + - python-calamine>=0.3.0 ; extra == 'excel' + - pyxlsb>=1.0.10 ; extra == 'excel' + - xlrd>=2.0.1 ; extra == 'excel' + - xlsxwriter>=3.2.0 ; extra == 'excel' + - pyarrow>=13.0.0 ; extra == 'parquet' + - pyarrow>=13.0.0 ; extra == 'feather' + - pyiceberg>=0.8.1 ; extra == 'iceberg' + - tables>=3.10.1 ; extra == 'hdf5' + - pyreadstat>=1.2.8 ; extra == 'spss' + - sqlalchemy>=2.0.36 ; extra == 'postgresql' + - psycopg2>=2.9.10 ; extra == 'postgresql' + - adbc-driver-postgresql>=1.2.0 ; extra == 'postgresql' + - sqlalchemy>=2.0.36 ; extra == 'mysql' + - pymysql>=1.1.1 ; extra == 'mysql' + - sqlalchemy>=2.0.36 ; extra == 'sql-other' + - adbc-driver-postgresql>=1.2.0 ; extra == 'sql-other' + - adbc-driver-sqlite>=1.2.0 ; extra == 'sql-other' + - beautifulsoup4>=4.12.3 ; extra == 'html' + - html5lib>=1.1 ; extra == 'html' + - lxml>=5.3.0 ; extra == 'html' + - lxml>=5.3.0 ; extra == 'xml' + - matplotlib>=3.9.3 ; extra == 'plot' + - jinja2>=3.1.5 ; extra == 'output-formatting' + - tabulate>=0.9.0 ; extra == 'output-formatting' + - pyqt5>=5.15.9 ; extra == 'clipboard' + - qtpy>=2.4.2 ; extra == 'clipboard' + - zstandard>=0.23.0 ; extra == 'compression' + - pytz>=2020.1 ; extra == 'timezone' + - adbc-driver-postgresql>=1.2.0 ; extra == 'all' + - adbc-driver-sqlite>=1.2.0 ; extra == 'all' + - beautifulsoup4>=4.12.3 ; extra == 'all' + - bottleneck>=1.4.2 ; extra == 'all' + - fastparquet>=2024.11.0 ; extra == 'all' + - fsspec>=2024.10.0 ; extra == 'all' + - gcsfs>=2024.10.0 ; extra == 'all' + - html5lib>=1.1 ; extra == 'all' + - hypothesis>=6.116.0 ; extra == 'all' + - jinja2>=3.1.5 ; extra == 'all' + - lxml>=5.3.0 ; extra == 'all' + - matplotlib>=3.9.3 ; extra == 'all' + - numba>=0.60.0 ; extra == 'all' + - numexpr>=2.10.2 ; extra == 'all' + - odfpy>=1.4.1 ; extra == 'all' + - openpyxl>=3.1.5 ; extra == 'all' + - psycopg2>=2.9.10 ; extra == 'all' + - pyarrow>=13.0.0 ; extra == 'all' + - pyiceberg>=0.8.1 ; extra == 'all' + - pymysql>=1.1.1 ; extra == 'all' + - pyqt5>=5.15.9 ; extra == 'all' + - pyreadstat>=1.2.8 ; extra == 'all' + - pytest>=8.3.4 ; extra == 'all' + - pytest-xdist>=3.6.1 ; extra == 'all' + - python-calamine>=0.3.0 ; extra == 'all' + - pytz>=2020.1 ; extra == 'all' + - pyxlsb>=1.0.10 ; extra == 'all' + - qtpy>=2.4.2 ; extra == 'all' + - scipy>=1.14.1 ; extra == 'all' + - s3fs>=2024.10.0 ; extra == 'all' + - sqlalchemy>=2.0.36 ; extra == 'all' + - tables>=3.10.1 ; extra == 'all' + - tabulate>=0.9.0 ; extra == 'all' + - xarray>=2024.10.0 ; extra == 'all' + - xlrd>=2.0.1 ; extra == 'all' + - xlsxwriter>=3.2.0 ; extra == 'all' + - zstandard>=0.23.0 ; extra == 'all' + requires_python: '>=3.11' - pypi: https://files.pythonhosted.org/packages/78/b5/915a19b3de2f7430062b509653563db1633ddbb6f021b06731521115d4e2/scipy-1.18.0-cp314-cp314-macosx_10_15_x86_64.whl name: scipy version: 1.18.0 @@ -33720,6 +33656,96 @@ packages: - scikit-learn ; extra == 'pyspark' - scikit-learn ; extra == 'scikit-learn' requires_python: '>=3.8' +- pypi: https://files.pythonhosted.org/packages/b7/e9/f43410fada510b43fec09993c08f552086c3d247d3ee801a678f3cb10ea5/pandas-3.0.6-cp314-cp314-win_amd64.whl + name: pandas + version: 3.0.6 + sha256: 77ccbe5057aece6fc172b9b77f19c04335af6882bc2e10c8f3ee4e6bfb3da553 + requires_dist: + - numpy>=1.26.0 ; python_full_version < '3.14' + - numpy>=2.3.3 ; python_full_version >= '3.14' + - python-dateutil>=2.8.2 + - tzdata ; sys_platform == 'win32' + - tzdata ; sys_platform == 'emscripten' + - hypothesis>=6.116.0 ; extra == 'test' + - pytest>=8.3.4,<9.1 ; extra == 'test' + - pytest-xdist>=3.6.1 ; extra == 'test' + - pyarrow>=13.0.0 ; extra == 'pyarrow' + - bottleneck>=1.4.2 ; extra == 'performance' + - numba>=0.60.0 ; extra == 'performance' + - numexpr>=2.10.2 ; extra == 'performance' + - scipy>=1.14.1 ; extra == 'computation' + - xarray>=2024.10.0 ; extra == 'computation' + - fsspec>=2024.10.0 ; extra == 'fss' + - s3fs>=2024.10.0 ; extra == 'aws' + - gcsfs>=2024.10.0 ; extra == 'gcp' + - odfpy>=1.4.1 ; extra == 'excel' + - openpyxl>=3.1.5 ; extra == 'excel' + - python-calamine>=0.3.0 ; extra == 'excel' + - pyxlsb>=1.0.10 ; extra == 'excel' + - xlrd>=2.0.1 ; extra == 'excel' + - xlsxwriter>=3.2.0 ; extra == 'excel' + - pyarrow>=13.0.0 ; extra == 'parquet' + - pyarrow>=13.0.0 ; extra == 'feather' + - pyiceberg>=0.8.1 ; extra == 'iceberg' + - tables>=3.10.1 ; extra == 'hdf5' + - pyreadstat>=1.2.8 ; extra == 'spss' + - sqlalchemy>=2.0.36 ; extra == 'postgresql' + - psycopg2>=2.9.10 ; extra == 'postgresql' + - adbc-driver-postgresql>=1.2.0 ; extra == 'postgresql' + - sqlalchemy>=2.0.36 ; extra == 'mysql' + - pymysql>=1.1.1 ; extra == 'mysql' + - sqlalchemy>=2.0.36 ; extra == 'sql-other' + - adbc-driver-postgresql>=1.2.0 ; extra == 'sql-other' + - adbc-driver-sqlite>=1.2.0 ; extra == 'sql-other' + - beautifulsoup4>=4.12.3 ; extra == 'html' + - html5lib>=1.1 ; extra == 'html' + - lxml>=5.3.0 ; extra == 'html' + - lxml>=5.3.0 ; extra == 'xml' + - matplotlib>=3.9.3 ; extra == 'plot' + - jinja2>=3.1.5 ; extra == 'output-formatting' + - tabulate>=0.9.0 ; extra == 'output-formatting' + - pyqt5>=5.15.9 ; extra == 'clipboard' + - qtpy>=2.4.2 ; extra == 'clipboard' + - zstandard>=0.23.0 ; extra == 'compression' + - pytz>=2020.1 ; extra == 'timezone' + - adbc-driver-postgresql>=1.2.0 ; extra == 'all' + - adbc-driver-sqlite>=1.2.0 ; extra == 'all' + - beautifulsoup4>=4.12.3 ; extra == 'all' + - bottleneck>=1.4.2 ; extra == 'all' + - fastparquet>=2024.11.0 ; extra == 'all' + - fsspec>=2024.10.0 ; extra == 'all' + - gcsfs>=2024.10.0 ; extra == 'all' + - html5lib>=1.1 ; extra == 'all' + - hypothesis>=6.116.0 ; extra == 'all' + - jinja2>=3.1.5 ; extra == 'all' + - lxml>=5.3.0 ; extra == 'all' + - matplotlib>=3.9.3 ; extra == 'all' + - numba>=0.60.0 ; extra == 'all' + - numexpr>=2.10.2 ; extra == 'all' + - odfpy>=1.4.1 ; extra == 'all' + - openpyxl>=3.1.5 ; extra == 'all' + - psycopg2>=2.9.10 ; extra == 'all' + - pyarrow>=13.0.0 ; extra == 'all' + - pyiceberg>=0.8.1 ; extra == 'all' + - pymysql>=1.1.1 ; extra == 'all' + - pyqt5>=5.15.9 ; extra == 'all' + - pyreadstat>=1.2.8 ; extra == 'all' + - pytest>=8.3.4 ; extra == 'all' + - pytest-xdist>=3.6.1 ; extra == 'all' + - python-calamine>=0.3.0 ; extra == 'all' + - pytz>=2020.1 ; extra == 'all' + - pyxlsb>=1.0.10 ; extra == 'all' + - qtpy>=2.4.2 ; extra == 'all' + - scipy>=1.14.1 ; extra == 'all' + - s3fs>=2024.10.0 ; extra == 'all' + - sqlalchemy>=2.0.36 ; extra == 'all' + - tables>=3.10.1 ; extra == 'all' + - tabulate>=0.9.0 ; extra == 'all' + - xarray>=2024.10.0 ; extra == 'all' + - xlrd>=2.0.1 ; extra == 'all' + - xlsxwriter>=3.2.0 ; extra == 'all' + - zstandard>=0.23.0 ; extra == 'all' + requires_python: '>=3.11' - pypi: https://files.pythonhosted.org/packages/bd/6e/95b0e537de1f4d4301f76f944642c6da50d1511cc7b3d64dc418a66c7509/wcwidth-0.8.1-py3-none-any.whl name: wcwidth version: 0.8.1 @@ -33749,6 +33775,96 @@ packages: - scikit-learn ; extra == 'pyspark' - scikit-learn ; extra == 'scikit-learn' requires_python: '>=3.12' +- pypi: https://files.pythonhosted.org/packages/ca/ba/ffdcb19be4ff6bfe7d969e7cef2c567c633df5a3a1cc1053394ad053bca8/pandas-3.0.6-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl + name: pandas + version: 3.0.6 + sha256: 62f51d7f651c8054c5e82a69265c98082e795d1442df7ca6edc3a545d61214b1 + requires_dist: + - numpy>=1.26.0 ; python_full_version < '3.14' + - numpy>=2.3.3 ; python_full_version >= '3.14' + - python-dateutil>=2.8.2 + - tzdata ; sys_platform == 'win32' + - tzdata ; sys_platform == 'emscripten' + - hypothesis>=6.116.0 ; extra == 'test' + - pytest>=8.3.4,<9.1 ; extra == 'test' + - pytest-xdist>=3.6.1 ; extra == 'test' + - pyarrow>=13.0.0 ; extra == 'pyarrow' + - bottleneck>=1.4.2 ; extra == 'performance' + - numba>=0.60.0 ; extra == 'performance' + - numexpr>=2.10.2 ; extra == 'performance' + - scipy>=1.14.1 ; extra == 'computation' + - xarray>=2024.10.0 ; extra == 'computation' + - fsspec>=2024.10.0 ; extra == 'fss' + - s3fs>=2024.10.0 ; extra == 'aws' + - gcsfs>=2024.10.0 ; extra == 'gcp' + - odfpy>=1.4.1 ; extra == 'excel' + - openpyxl>=3.1.5 ; extra == 'excel' + - python-calamine>=0.3.0 ; extra == 'excel' + - pyxlsb>=1.0.10 ; extra == 'excel' + - xlrd>=2.0.1 ; extra == 'excel' + - xlsxwriter>=3.2.0 ; extra == 'excel' + - pyarrow>=13.0.0 ; extra == 'parquet' + - pyarrow>=13.0.0 ; extra == 'feather' + - pyiceberg>=0.8.1 ; extra == 'iceberg' + - tables>=3.10.1 ; extra == 'hdf5' + - pyreadstat>=1.2.8 ; extra == 'spss' + - sqlalchemy>=2.0.36 ; extra == 'postgresql' + - psycopg2>=2.9.10 ; extra == 'postgresql' + - adbc-driver-postgresql>=1.2.0 ; extra == 'postgresql' + - sqlalchemy>=2.0.36 ; extra == 'mysql' + - pymysql>=1.1.1 ; extra == 'mysql' + - sqlalchemy>=2.0.36 ; extra == 'sql-other' + - adbc-driver-postgresql>=1.2.0 ; extra == 'sql-other' + - adbc-driver-sqlite>=1.2.0 ; extra == 'sql-other' + - beautifulsoup4>=4.12.3 ; extra == 'html' + - html5lib>=1.1 ; extra == 'html' + - lxml>=5.3.0 ; extra == 'html' + - lxml>=5.3.0 ; extra == 'xml' + - matplotlib>=3.9.3 ; extra == 'plot' + - jinja2>=3.1.5 ; extra == 'output-formatting' + - tabulate>=0.9.0 ; extra == 'output-formatting' + - pyqt5>=5.15.9 ; extra == 'clipboard' + - qtpy>=2.4.2 ; extra == 'clipboard' + - zstandard>=0.23.0 ; extra == 'compression' + - pytz>=2020.1 ; extra == 'timezone' + - adbc-driver-postgresql>=1.2.0 ; extra == 'all' + - adbc-driver-sqlite>=1.2.0 ; extra == 'all' + - beautifulsoup4>=4.12.3 ; extra == 'all' + - bottleneck>=1.4.2 ; extra == 'all' + - fastparquet>=2024.11.0 ; extra == 'all' + - fsspec>=2024.10.0 ; extra == 'all' + - gcsfs>=2024.10.0 ; extra == 'all' + - html5lib>=1.1 ; extra == 'all' + - hypothesis>=6.116.0 ; extra == 'all' + - jinja2>=3.1.5 ; extra == 'all' + - lxml>=5.3.0 ; extra == 'all' + - matplotlib>=3.9.3 ; extra == 'all' + - numba>=0.60.0 ; extra == 'all' + - numexpr>=2.10.2 ; extra == 'all' + - odfpy>=1.4.1 ; extra == 'all' + - openpyxl>=3.1.5 ; extra == 'all' + - psycopg2>=2.9.10 ; extra == 'all' + - pyarrow>=13.0.0 ; extra == 'all' + - pyiceberg>=0.8.1 ; extra == 'all' + - pymysql>=1.1.1 ; extra == 'all' + - pyqt5>=5.15.9 ; extra == 'all' + - pyreadstat>=1.2.8 ; extra == 'all' + - pytest>=8.3.4 ; extra == 'all' + - pytest-xdist>=3.6.1 ; extra == 'all' + - python-calamine>=0.3.0 ; extra == 'all' + - pytz>=2020.1 ; extra == 'all' + - pyxlsb>=1.0.10 ; extra == 'all' + - qtpy>=2.4.2 ; extra == 'all' + - scipy>=1.14.1 ; extra == 'all' + - s3fs>=2024.10.0 ; extra == 'all' + - sqlalchemy>=2.0.36 ; extra == 'all' + - tables>=3.10.1 ; extra == 'all' + - tabulate>=0.9.0 ; extra == 'all' + - xarray>=2024.10.0 ; extra == 'all' + - xlrd>=2.0.1 ; extra == 'all' + - xlsxwriter>=3.2.0 ; extra == 'all' + - zstandard>=0.23.0 ; extra == 'all' + requires_python: '>=3.11' - pypi: https://files.pythonhosted.org/packages/cd/05/7213965863cba1ed0150ad045bceed6276a1afaaaedbaeff4699ec4f0ccb/lightgbm-4.7.0-py3-none-macosx_10_15_x86_64.whl name: lightgbm version: 4.7.0 @@ -34048,6 +34164,11 @@ packages: - scikit-learn ; extra == 'pyspark' - cloudpickle ; extra == 'pyspark' requires_python: '>=3.12' +- pypi: https://files.pythonhosted.org/packages/f9/bc/8737e8d54cf51106118039b83f485a4783112fab49ea9d044b234978a46e/tzdata-2026.4-py2.py3-none-any.whl + name: tzdata + version: '2026.4' + sha256: c2169a8b0a7a5e9674da5a135ccdfb2b3e671b333ed9fed17b41f73c34476e81 + requires_python: '>=2' - pypi: https://files.pythonhosted.org/packages/fc/72/3b68983c0215ef65d48e9eeb1f168c3c6e3d62a61ece605de3209c79cae1/xgboost-3.3.0-py3-none-macosx_10_15_x86_64.whl name: xgboost version: 3.3.0 @@ -34077,61 +34198,501 @@ packages: - pytest-lazy-fixtures ; extra == 'tests' - pytest>=9 ; extra == 'tests' requires_python: '>=3.10' +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl + name: pandas + version: 3.1.0.dev0+2043.g7aac401536 + index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple + requires_dist: + - numpy>=2.0.2 ; python_full_version < '3.14' + - numpy>=2.3.3 ; python_full_version >= '3.14' + - python-dateutil>=2.9.0 + - tzdata ; sys_platform == 'win32' + - tzdata ; sys_platform == 'emscripten' + - pytest>=8.3.4 ; extra == 'test' + - pytest-xdist>=3.6.1 ; extra == 'test' + - pyarrow>=16.0.0 ; extra == 'pyarrow' + - bottleneck>=1.5.0 ; extra == 'performance' + - numba>=0.61.2 ; extra == 'performance' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'performance' + - scipy>=1.16.1 ; extra == 'computation' + - xarray>=2025.7.1 ; extra == 'computation' + - fsspec>=2025.7.0 ; extra == 'fss' + - s3fs>=2025.7.0 ; extra == 'aws' + - gcsfs>=2025.7.0 ; extra == 'gcp' + - odfpy>=1.4.1 ; extra == 'excel' + - openpyxl>=3.1.5 ; extra == 'excel' + - python-calamine>=0.4.0 ; extra == 'excel' + - pyxlsb>=1.0.10 ; extra == 'excel' + - xlrd>=2.0.2 ; extra == 'excel' + - xlsxwriter>=3.2.5 ; extra == 'excel' + - pyarrow>=13.0.0 ; extra == 'parquet' + - pyarrow>=13.0.0 ; extra == 'feather' + - pyiceberg>=0.9.1 ; extra == 'iceberg' + - tables>=3.10.2 ; extra == 'hdf5' + - pyreadstat>=1.3.0 ; extra == 'spss' + - sqlalchemy>=2.0.42 ; extra == 'postgresql' + - psycopg2>=2.9.10 ; extra == 'postgresql' + - adbc-driver-postgresql>=1.7.0 ; extra == 'postgresql' + - sqlalchemy>=2.0.42 ; extra == 'mysql' + - pymysql>=1.1.1 ; extra == 'mysql' + - sqlalchemy>=2.0.42 ; extra == 'sql-other' + - adbc-driver-postgresql>=1.7.0 ; extra == 'sql-other' + - adbc-driver-sqlite>=1.7.0 ; extra == 'sql-other' + - beautifulsoup4>=4.13.4 ; extra == 'html' + - html5lib>=1.1 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'xml' + - matplotlib>=3.10.5 ; extra == 'plot' + - jinja2>=3.1.6 ; extra == 'output-formatting' + - tabulate>=0.9.0 ; extra == 'output-formatting' + - pyqt5>=5.15.11 ; extra == 'clipboard' + - qtpy>=2.4.3 ; extra == 'clipboard' + - zstandard>=0.23.0 ; extra == 'compression' + - pytz>=2020.1 ; extra == 'timezone' + - adbc-driver-postgresql>=1.7.0 ; extra == 'all' + - adbc-driver-sqlite>=1.7.0 ; extra == 'all' + - beautifulsoup4>=4.13.4 ; extra == 'all' + - bottleneck>=1.5.0 ; extra == 'all' + - fastparquet>=2024.11.0 ; extra == 'all' + - fsspec>=2025.7.0 ; extra == 'all' + - gcsfs>=2025.7.0 ; extra == 'all' + - html5lib>=1.1 ; extra == 'all' + - jinja2>=3.1.6 ; extra == 'all' + - lxml>=6.0.0 ; extra == 'all' + - matplotlib>=3.10.5 ; extra == 'all' + - numba>=0.61.2 ; extra == 'all' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'all' + - odfpy>=1.4.1 ; extra == 'all' + - openpyxl>=3.1.5 ; extra == 'all' + - psycopg2>=2.9.10 ; extra == 'all' + - pyarrow>=16.0.0 ; extra == 'all' + - pyiceberg>=0.9.1 ; extra == 'all' + - pymysql>=1.1.1 ; extra == 'all' + - pyqt5>=5.15.11 ; extra == 'all' + - pyreadstat>=1.3.0 ; extra == 'all' + - pytest>=8.3.4 ; extra == 'all' + - pytest-xdist>=3.6.1 ; extra == 'all' + - python-calamine>=0.4.0 ; extra == 'all' + - pytz>=2020.1 ; extra == 'all' + - pyxlsb>=1.0.10 ; extra == 'all' + - qtpy>=2.4.3 ; extra == 'all' + - scipy>=1.16.1 ; extra == 'all' + - s3fs>=2025.7.0 ; extra == 'all' + - sqlalchemy>=2.0.42 ; extra == 'all' + - tables>=3.10.2 ; extra == 'all' + - tabulate>=0.9.0 ; extra == 'all' + - xarray>=2025.7.1 ; extra == 'all' + - xlrd>=2.0.2 ; extra == 'all' + - xlsxwriter>=3.2.5 ; extra == 'all' + - zstandard>=0.23.0 ; extra == 'all' + requires_python: '>=3.11' +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_10_15_x86_64.whl + name: pandas + version: 3.1.0.dev0+2043.g7aac401536 + index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple + requires_dist: + - numpy>=2.0.2 ; python_full_version < '3.14' + - numpy>=2.3.3 ; python_full_version >= '3.14' + - python-dateutil>=2.9.0 + - tzdata ; sys_platform == 'win32' + - tzdata ; sys_platform == 'emscripten' + - pytest>=8.3.4 ; extra == 'test' + - pytest-xdist>=3.6.1 ; extra == 'test' + - pyarrow>=16.0.0 ; extra == 'pyarrow' + - bottleneck>=1.5.0 ; extra == 'performance' + - numba>=0.61.2 ; extra == 'performance' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'performance' + - scipy>=1.16.1 ; extra == 'computation' + - xarray>=2025.7.1 ; extra == 'computation' + - fsspec>=2025.7.0 ; extra == 'fss' + - s3fs>=2025.7.0 ; extra == 'aws' + - gcsfs>=2025.7.0 ; extra == 'gcp' + - odfpy>=1.4.1 ; extra == 'excel' + - openpyxl>=3.1.5 ; extra == 'excel' + - python-calamine>=0.4.0 ; extra == 'excel' + - pyxlsb>=1.0.10 ; extra == 'excel' + - xlrd>=2.0.2 ; extra == 'excel' + - xlsxwriter>=3.2.5 ; extra == 'excel' + - pyarrow>=13.0.0 ; extra == 'parquet' + - pyarrow>=13.0.0 ; extra == 'feather' + - pyiceberg>=0.9.1 ; extra == 'iceberg' + - tables>=3.10.2 ; extra == 'hdf5' + - pyreadstat>=1.3.0 ; extra == 'spss' + - sqlalchemy>=2.0.42 ; extra == 'postgresql' + - psycopg2>=2.9.10 ; extra == 'postgresql' + - adbc-driver-postgresql>=1.7.0 ; extra == 'postgresql' + - sqlalchemy>=2.0.42 ; extra == 'mysql' + - pymysql>=1.1.1 ; extra == 'mysql' + - sqlalchemy>=2.0.42 ; extra == 'sql-other' + - adbc-driver-postgresql>=1.7.0 ; extra == 'sql-other' + - adbc-driver-sqlite>=1.7.0 ; extra == 'sql-other' + - beautifulsoup4>=4.13.4 ; extra == 'html' + - html5lib>=1.1 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'xml' + - matplotlib>=3.10.5 ; extra == 'plot' + - jinja2>=3.1.6 ; extra == 'output-formatting' + - tabulate>=0.9.0 ; extra == 'output-formatting' + - pyqt5>=5.15.11 ; extra == 'clipboard' + - qtpy>=2.4.3 ; extra == 'clipboard' + - zstandard>=0.23.0 ; extra == 'compression' + - pytz>=2020.1 ; extra == 'timezone' + - adbc-driver-postgresql>=1.7.0 ; extra == 'all' + - adbc-driver-sqlite>=1.7.0 ; extra == 'all' + - beautifulsoup4>=4.13.4 ; extra == 'all' + - bottleneck>=1.5.0 ; extra == 'all' + - fastparquet>=2024.11.0 ; extra == 'all' + - fsspec>=2025.7.0 ; extra == 'all' + - gcsfs>=2025.7.0 ; extra == 'all' + - html5lib>=1.1 ; extra == 'all' + - jinja2>=3.1.6 ; extra == 'all' + - lxml>=6.0.0 ; extra == 'all' + - matplotlib>=3.10.5 ; extra == 'all' + - numba>=0.61.2 ; extra == 'all' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'all' + - odfpy>=1.4.1 ; extra == 'all' + - openpyxl>=3.1.5 ; extra == 'all' + - psycopg2>=2.9.10 ; extra == 'all' + - pyarrow>=16.0.0 ; extra == 'all' + - pyiceberg>=0.9.1 ; extra == 'all' + - pymysql>=1.1.1 ; extra == 'all' + - pyqt5>=5.15.11 ; extra == 'all' + - pyreadstat>=1.3.0 ; extra == 'all' + - pytest>=8.3.4 ; extra == 'all' + - pytest-xdist>=3.6.1 ; extra == 'all' + - python-calamine>=0.4.0 ; extra == 'all' + - pytz>=2020.1 ; extra == 'all' + - pyxlsb>=1.0.10 ; extra == 'all' + - qtpy>=2.4.3 ; extra == 'all' + - scipy>=1.16.1 ; extra == 'all' + - s3fs>=2025.7.0 ; extra == 'all' + - sqlalchemy>=2.0.42 ; extra == 'all' + - tables>=3.10.2 ; extra == 'all' + - tabulate>=0.9.0 ; extra == 'all' + - xarray>=2025.7.1 ; extra == 'all' + - xlrd>=2.0.2 ; extra == 'all' + - xlsxwriter>=3.2.5 ; extra == 'all' + - zstandard>=0.23.0 ; extra == 'all' + requires_python: '>=3.11' +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_11_0_arm64.whl + name: pandas + version: 3.1.0.dev0+2043.g7aac401536 + index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple + requires_dist: + - numpy>=2.0.2 ; python_full_version < '3.14' + - numpy>=2.3.3 ; python_full_version >= '3.14' + - python-dateutil>=2.9.0 + - tzdata ; sys_platform == 'win32' + - tzdata ; sys_platform == 'emscripten' + - pytest>=8.3.4 ; extra == 'test' + - pytest-xdist>=3.6.1 ; extra == 'test' + - pyarrow>=16.0.0 ; extra == 'pyarrow' + - bottleneck>=1.5.0 ; extra == 'performance' + - numba>=0.61.2 ; extra == 'performance' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'performance' + - scipy>=1.16.1 ; extra == 'computation' + - xarray>=2025.7.1 ; extra == 'computation' + - fsspec>=2025.7.0 ; extra == 'fss' + - s3fs>=2025.7.0 ; extra == 'aws' + - gcsfs>=2025.7.0 ; extra == 'gcp' + - odfpy>=1.4.1 ; extra == 'excel' + - openpyxl>=3.1.5 ; extra == 'excel' + - python-calamine>=0.4.0 ; extra == 'excel' + - pyxlsb>=1.0.10 ; extra == 'excel' + - xlrd>=2.0.2 ; extra == 'excel' + - xlsxwriter>=3.2.5 ; extra == 'excel' + - pyarrow>=13.0.0 ; extra == 'parquet' + - pyarrow>=13.0.0 ; extra == 'feather' + - pyiceberg>=0.9.1 ; extra == 'iceberg' + - tables>=3.10.2 ; extra == 'hdf5' + - pyreadstat>=1.3.0 ; extra == 'spss' + - sqlalchemy>=2.0.42 ; extra == 'postgresql' + - psycopg2>=2.9.10 ; extra == 'postgresql' + - adbc-driver-postgresql>=1.7.0 ; extra == 'postgresql' + - sqlalchemy>=2.0.42 ; extra == 'mysql' + - pymysql>=1.1.1 ; extra == 'mysql' + - sqlalchemy>=2.0.42 ; extra == 'sql-other' + - adbc-driver-postgresql>=1.7.0 ; extra == 'sql-other' + - adbc-driver-sqlite>=1.7.0 ; extra == 'sql-other' + - beautifulsoup4>=4.13.4 ; extra == 'html' + - html5lib>=1.1 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'xml' + - matplotlib>=3.10.5 ; extra == 'plot' + - jinja2>=3.1.6 ; extra == 'output-formatting' + - tabulate>=0.9.0 ; extra == 'output-formatting' + - pyqt5>=5.15.11 ; extra == 'clipboard' + - qtpy>=2.4.3 ; extra == 'clipboard' + - zstandard>=0.23.0 ; extra == 'compression' + - pytz>=2020.1 ; extra == 'timezone' + - adbc-driver-postgresql>=1.7.0 ; extra == 'all' + - adbc-driver-sqlite>=1.7.0 ; extra == 'all' + - beautifulsoup4>=4.13.4 ; extra == 'all' + - bottleneck>=1.5.0 ; extra == 'all' + - fastparquet>=2024.11.0 ; extra == 'all' + - fsspec>=2025.7.0 ; extra == 'all' + - gcsfs>=2025.7.0 ; extra == 'all' + - html5lib>=1.1 ; extra == 'all' + - jinja2>=3.1.6 ; extra == 'all' + - lxml>=6.0.0 ; extra == 'all' + - matplotlib>=3.10.5 ; extra == 'all' + - numba>=0.61.2 ; extra == 'all' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'all' + - odfpy>=1.4.1 ; extra == 'all' + - openpyxl>=3.1.5 ; extra == 'all' + - psycopg2>=2.9.10 ; extra == 'all' + - pyarrow>=16.0.0 ; extra == 'all' + - pyiceberg>=0.9.1 ; extra == 'all' + - pymysql>=1.1.1 ; extra == 'all' + - pyqt5>=5.15.11 ; extra == 'all' + - pyreadstat>=1.3.0 ; extra == 'all' + - pytest>=8.3.4 ; extra == 'all' + - pytest-xdist>=3.6.1 ; extra == 'all' + - python-calamine>=0.4.0 ; extra == 'all' + - pytz>=2020.1 ; extra == 'all' + - pyxlsb>=1.0.10 ; extra == 'all' + - qtpy>=2.4.3 ; extra == 'all' + - scipy>=1.16.1 ; extra == 'all' + - s3fs>=2025.7.0 ; extra == 'all' + - sqlalchemy>=2.0.42 ; extra == 'all' + - tables>=3.10.2 ; extra == 'all' + - tabulate>=0.9.0 ; extra == 'all' + - xarray>=2025.7.1 ; extra == 'all' + - xlrd>=2.0.2 ; extra == 'all' + - xlsxwriter>=3.2.5 ; extra == 'all' + - zstandard>=0.23.0 ; extra == 'all' + requires_python: '>=3.11' +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl + name: pandas + version: 3.1.0.dev0+2043.g7aac401536 + index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple + requires_dist: + - numpy>=2.0.2 ; python_full_version < '3.14' + - numpy>=2.3.3 ; python_full_version >= '3.14' + - python-dateutil>=2.9.0 + - tzdata ; sys_platform == 'win32' + - tzdata ; sys_platform == 'emscripten' + - pytest>=8.3.4 ; extra == 'test' + - pytest-xdist>=3.6.1 ; extra == 'test' + - pyarrow>=16.0.0 ; extra == 'pyarrow' + - bottleneck>=1.5.0 ; extra == 'performance' + - numba>=0.61.2 ; extra == 'performance' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'performance' + - scipy>=1.16.1 ; extra == 'computation' + - xarray>=2025.7.1 ; extra == 'computation' + - fsspec>=2025.7.0 ; extra == 'fss' + - s3fs>=2025.7.0 ; extra == 'aws' + - gcsfs>=2025.7.0 ; extra == 'gcp' + - odfpy>=1.4.1 ; extra == 'excel' + - openpyxl>=3.1.5 ; extra == 'excel' + - python-calamine>=0.4.0 ; extra == 'excel' + - pyxlsb>=1.0.10 ; extra == 'excel' + - xlrd>=2.0.2 ; extra == 'excel' + - xlsxwriter>=3.2.5 ; extra == 'excel' + - pyarrow>=13.0.0 ; extra == 'parquet' + - pyarrow>=13.0.0 ; extra == 'feather' + - pyiceberg>=0.9.1 ; extra == 'iceberg' + - tables>=3.10.2 ; extra == 'hdf5' + - pyreadstat>=1.3.0 ; extra == 'spss' + - sqlalchemy>=2.0.42 ; extra == 'postgresql' + - psycopg2>=2.9.10 ; extra == 'postgresql' + - adbc-driver-postgresql>=1.7.0 ; extra == 'postgresql' + - sqlalchemy>=2.0.42 ; extra == 'mysql' + - pymysql>=1.1.1 ; extra == 'mysql' + - sqlalchemy>=2.0.42 ; extra == 'sql-other' + - adbc-driver-postgresql>=1.7.0 ; extra == 'sql-other' + - adbc-driver-sqlite>=1.7.0 ; extra == 'sql-other' + - beautifulsoup4>=4.13.4 ; extra == 'html' + - html5lib>=1.1 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'xml' + - matplotlib>=3.10.5 ; extra == 'plot' + - jinja2>=3.1.6 ; extra == 'output-formatting' + - tabulate>=0.9.0 ; extra == 'output-formatting' + - pyqt5>=5.15.11 ; extra == 'clipboard' + - qtpy>=2.4.3 ; extra == 'clipboard' + - zstandard>=0.23.0 ; extra == 'compression' + - pytz>=2020.1 ; extra == 'timezone' + - adbc-driver-postgresql>=1.7.0 ; extra == 'all' + - adbc-driver-sqlite>=1.7.0 ; extra == 'all' + - beautifulsoup4>=4.13.4 ; extra == 'all' + - bottleneck>=1.5.0 ; extra == 'all' + - fastparquet>=2024.11.0 ; extra == 'all' + - fsspec>=2025.7.0 ; extra == 'all' + - gcsfs>=2025.7.0 ; extra == 'all' + - html5lib>=1.1 ; extra == 'all' + - jinja2>=3.1.6 ; extra == 'all' + - lxml>=6.0.0 ; extra == 'all' + - matplotlib>=3.10.5 ; extra == 'all' + - numba>=0.61.2 ; extra == 'all' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'all' + - odfpy>=1.4.1 ; extra == 'all' + - openpyxl>=3.1.5 ; extra == 'all' + - psycopg2>=2.9.10 ; extra == 'all' + - pyarrow>=16.0.0 ; extra == 'all' + - pyiceberg>=0.9.1 ; extra == 'all' + - pymysql>=1.1.1 ; extra == 'all' + - pyqt5>=5.15.11 ; extra == 'all' + - pyreadstat>=1.3.0 ; extra == 'all' + - pytest>=8.3.4 ; extra == 'all' + - pytest-xdist>=3.6.1 ; extra == 'all' + - python-calamine>=0.4.0 ; extra == 'all' + - pytz>=2020.1 ; extra == 'all' + - pyxlsb>=1.0.10 ; extra == 'all' + - qtpy>=2.4.3 ; extra == 'all' + - scipy>=1.16.1 ; extra == 'all' + - s3fs>=2025.7.0 ; extra == 'all' + - sqlalchemy>=2.0.42 ; extra == 'all' + - tables>=3.10.2 ; extra == 'all' + - tabulate>=0.9.0 ; extra == 'all' + - xarray>=2025.7.1 ; extra == 'all' + - xlrd>=2.0.2 ; extra == 'all' + - xlsxwriter>=3.2.5 ; extra == 'all' + - zstandard>=0.23.0 ; extra == 'all' + requires_python: '>=3.11' +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-win_amd64.whl + name: pandas + version: 3.1.0.dev0+2043.g7aac401536 + index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple + requires_dist: + - numpy>=2.0.2 ; python_full_version < '3.14' + - numpy>=2.3.3 ; python_full_version >= '3.14' + - python-dateutil>=2.9.0 + - tzdata ; sys_platform == 'win32' + - tzdata ; sys_platform == 'emscripten' + - pytest>=8.3.4 ; extra == 'test' + - pytest-xdist>=3.6.1 ; extra == 'test' + - pyarrow>=16.0.0 ; extra == 'pyarrow' + - bottleneck>=1.5.0 ; extra == 'performance' + - numba>=0.61.2 ; extra == 'performance' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'performance' + - scipy>=1.16.1 ; extra == 'computation' + - xarray>=2025.7.1 ; extra == 'computation' + - fsspec>=2025.7.0 ; extra == 'fss' + - s3fs>=2025.7.0 ; extra == 'aws' + - gcsfs>=2025.7.0 ; extra == 'gcp' + - odfpy>=1.4.1 ; extra == 'excel' + - openpyxl>=3.1.5 ; extra == 'excel' + - python-calamine>=0.4.0 ; extra == 'excel' + - pyxlsb>=1.0.10 ; extra == 'excel' + - xlrd>=2.0.2 ; extra == 'excel' + - xlsxwriter>=3.2.5 ; extra == 'excel' + - pyarrow>=13.0.0 ; extra == 'parquet' + - pyarrow>=13.0.0 ; extra == 'feather' + - pyiceberg>=0.9.1 ; extra == 'iceberg' + - tables>=3.10.2 ; extra == 'hdf5' + - pyreadstat>=1.3.0 ; extra == 'spss' + - sqlalchemy>=2.0.42 ; extra == 'postgresql' + - psycopg2>=2.9.10 ; extra == 'postgresql' + - adbc-driver-postgresql>=1.7.0 ; extra == 'postgresql' + - sqlalchemy>=2.0.42 ; extra == 'mysql' + - pymysql>=1.1.1 ; extra == 'mysql' + - sqlalchemy>=2.0.42 ; extra == 'sql-other' + - adbc-driver-postgresql>=1.7.0 ; extra == 'sql-other' + - adbc-driver-sqlite>=1.7.0 ; extra == 'sql-other' + - beautifulsoup4>=4.13.4 ; extra == 'html' + - html5lib>=1.1 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'html' + - lxml>=6.0.0 ; extra == 'xml' + - matplotlib>=3.10.5 ; extra == 'plot' + - jinja2>=3.1.6 ; extra == 'output-formatting' + - tabulate>=0.9.0 ; extra == 'output-formatting' + - pyqt5>=5.15.11 ; extra == 'clipboard' + - qtpy>=2.4.3 ; extra == 'clipboard' + - zstandard>=0.23.0 ; extra == 'compression' + - pytz>=2020.1 ; extra == 'timezone' + - adbc-driver-postgresql>=1.7.0 ; extra == 'all' + - adbc-driver-sqlite>=1.7.0 ; extra == 'all' + - beautifulsoup4>=4.13.4 ; extra == 'all' + - bottleneck>=1.5.0 ; extra == 'all' + - fastparquet>=2024.11.0 ; extra == 'all' + - fsspec>=2025.7.0 ; extra == 'all' + - gcsfs>=2025.7.0 ; extra == 'all' + - html5lib>=1.1 ; extra == 'all' + - jinja2>=3.1.6 ; extra == 'all' + - lxml>=6.0.0 ; extra == 'all' + - matplotlib>=3.10.5 ; extra == 'all' + - numba>=0.61.2 ; extra == 'all' + - numexpr>=2.11.0,!=2.14.1 ; extra == 'all' + - odfpy>=1.4.1 ; extra == 'all' + - openpyxl>=3.1.5 ; extra == 'all' + - psycopg2>=2.9.10 ; extra == 'all' + - pyarrow>=16.0.0 ; extra == 'all' + - pyiceberg>=0.9.1 ; extra == 'all' + - pymysql>=1.1.1 ; extra == 'all' + - pyqt5>=5.15.11 ; extra == 'all' + - pyreadstat>=1.3.0 ; extra == 'all' + - pytest>=8.3.4 ; extra == 'all' + - pytest-xdist>=3.6.1 ; extra == 'all' + - python-calamine>=0.4.0 ; extra == 'all' + - pytz>=2020.1 ; extra == 'all' + - pyxlsb>=1.0.10 ; extra == 'all' + - qtpy>=2.4.3 ; extra == 'all' + - scipy>=1.16.1 ; extra == 'all' + - s3fs>=2025.7.0 ; extra == 'all' + - sqlalchemy>=2.0.42 ; extra == 'all' + - tables>=3.10.2 ; extra == 'all' + - tabulate>=0.9.0 ; extra == 'all' + - xarray>=2025.7.1 ; extra == 'all' + - xlrd>=2.0.2 ; extra == 'all' + - xlsxwriter>=3.2.5 ; extra == 'all' + - zstandard>=0.23.0 ; extra == 'all' + requires_python: '>=3.11' - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl name: scikit-learn version: 1.10.dev0 index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - - numpy>=1.24.1 - - scipy>=1.10.0 + - numpy>=1.26.0 + - scipy>=1.11.4 - joblib>=1.4.0 - narwhals>=2.0.1 - threadpoolctl>=3.5.0 - requires_python: '>=3.11' + requires_python: '>=3.12' - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_10_15_x86_64.whl name: scikit-learn version: 1.10.dev0 index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - - numpy>=1.24.1 - - scipy>=1.10.0 + - numpy>=1.26.0 + - scipy>=1.11.4 - joblib>=1.4.0 - narwhals>=2.0.1 - threadpoolctl>=3.5.0 - requires_python: '>=3.11' + requires_python: '>=3.12' - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_12_0_arm64.whl name: scikit-learn version: 1.10.dev0 index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - - numpy>=1.24.1 - - scipy>=1.10.0 + - numpy>=1.26.0 + - scipy>=1.11.4 - joblib>=1.4.0 - narwhals>=2.0.1 - threadpoolctl>=3.5.0 - requires_python: '>=3.11' + requires_python: '>=3.12' - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl name: scikit-learn version: 1.10.dev0 index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - - numpy>=1.24.1 - - scipy>=1.10.0 + - numpy>=1.26.0 + - scipy>=1.11.4 - joblib>=1.4.0 - narwhals>=2.0.1 - threadpoolctl>=3.5.0 - requires_python: '>=3.11' + requires_python: '>=3.12' - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-win_amd64.whl name: scikit-learn version: 1.10.dev0 index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - - numpy>=1.24.1 - - scipy>=1.10.0 + - numpy>=1.26.0 + - scipy>=1.11.4 - joblib>=1.4.0 - narwhals>=2.0.1 - threadpoolctl>=3.5.0 - requires_python: '>=3.11' + requires_python: '>=3.12' - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl name: scipy version: 2.0.0.dev0 diff --git a/pyproject.toml b/pyproject.toml index dc70ca3e..979ca807 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -128,7 +128,6 @@ skops = { path = ".", editable = true } [tool.pixi.feature.docs.dependencies] # To be synced with the versions in docs/requirements.txt matplotlib = ">=3.3" -pandas = ">=1" sphinx = ">=3.2.0" sphinx-gallery = ">=0.7.0" sphinx-rtd-theme = ">=1" @@ -138,8 +137,10 @@ sphinx-issues = ">=1.2.0" [tool.pixi.feature.docs.pypi-dependencies] # everything that depends on scikit-learn needs to be a pypi dependency so that this -# spec is compatible with the nightly build environment. +# spec is compatible with the nightly build environment. The same holds for pandas, +# whose dev version the nightly build environment installs from pypi. fairlearn = ">=0.7.0" +pandas = ">=2" [tool.pixi.feature.tests.dependencies] pytest = ">=7" @@ -148,7 +149,6 @@ flaky = ">=3.7.0" pandoc = ">=3.6.4" rich = ">=12" matplotlib = ">=3.3" -pandas = ">=1" [tool.pixi.feature.tests.pypi-dependencies] # these are packages that require scikit-learn. They need to be as a pypi dependency @@ -156,6 +156,11 @@ pandas = ">=1" # when installing pre-release nightly release. lightgbm = ">=3" xgboost = ">=1.6" +# skops.io supports pandas 2.0 and later; each CI environment pins one minor +# version so that the whole range is tested, see the sklearn* features below. +# A pypi dependency for the same reason as above: the nightly environment +# installs the dev version of pandas from pypi. +pandas = ">=2" [tool.pixi.feature.lint.dependencies] pre-commit = "*" @@ -257,7 +262,9 @@ extra-index-urls = ["https://pypi.anaconda.org/scientific-python-nightly-wheels/ # The version value here needs to be exact, hence == instead of ~= scikit-learn = "==1.10.dev0" fairlearn = "*" -pandas = "*" +# The dev version of pandas from the nightly index; "*" would pick the latest +# release, since pre-releases are only considered when named explicitly. +pandas = "==3.1.0.dev0" numpy = "*" scipy = "*" diff --git a/skops/io/_pandas.py b/skops/io/_pandas.py new file mode 100644 index 00000000..6322a42b --- /dev/null +++ b/skops/io/_pandas.py @@ -0,0 +1,443 @@ +"""Persistence of pandas objects. + +pandas objects are not persisted through ``__reduce__`` or ``__getstate__``: +those expose internals such as block managers, index engines and reference +trackers, which change between pandas versions and cannot be rebuilt from data +alone. Instead, every object is taken apart into the public pieces its +constructor accepts, and rebuilt by calling that constructor with an explicit +dtype, so that no type inference happens on load: + +- an :class:`~pandas.Index` is stored as its values and name, +- a :class:`~pandas.Series` as its values, index and name, +- a :class:`~pandas.DataFrame` as its columns, index and one array per column, +- extension arrays as the numpy arrays or scalars they are made of, plus their + dtype, and extension dtypes as their string representation. + +Values with a numpy dtype are stored as numpy arrays, everything else as lists +of scalars. + +pandas is optional and slow to import, so ``skops.io`` does not import it. The +``get_state`` handlers are registered on the first dump after the user has +imported pandas, see :func:`register_if_imported`, and the nodes only import +pandas when they construct an object. + +Not preserved: the ``freq`` of datetime-like indexes and arrays, and the +``attrs`` and ``flags`` of a Series or DataFrame. +""" + +from __future__ import annotations + +import sys +import warnings +from typing import Any + +import numpy as np + +from ._audit import Node, get_tree +from ._protocol import PROTOCOL +from ._trusted_types import PANDAS_TYPE_NAMES +from ._utils import ( + LoadContext, + SaveContext, + TrustedTypes, + _get_state, + get_module, + get_state, + gettype, +) +from .exceptions import UnsupportedTypeException + + +def _public_module(cls: type) -> str: + # The public module of a pandas class: "pandas.arrays" for the array + # classes and "pandas" for everything else. pandas 3 reports these as + # ``__module__`` already, but older versions report the defining module, + # e.g. ``pandas.core.series``, and a file must not depend on that: the + # public name is what the default trusted list knows, and what stays + # importable across versions. Deprecated aliases such as + # ``pandas.arrays.PandasArray`` warn when accessed, hence the suppression. + import pandas as pd + + with warnings.catch_warnings(): + warnings.simplefilter("ignore") + for module in (pd.arrays, pd): + if getattr(module, cls.__name__, None) is cls: + return module.__name__ + return get_module(cls) + + +# Classes that later pandas versions renamed, mapped to their current name, so +# that a file never names a class the loading version may not have. +_RENAMED_CLASSES = { + # renamed in pandas 2.1 + "PandasArray": "NumpyExtensionArray", +} + + +def _pandas_state( + obj: Any, loader: str, content: dict[str, Any], save_context: SaveContext +) -> dict[str, Any]: + cls = type(obj) + # The nodes below rebuild objects with the pandas constructors, so an + # instance of a subclass defined by another library would silently be + # loaded as its pandas base class. Refuse those instead. + if cls.__module__.partition(".")[0] != "pandas": + raise UnsupportedTypeException( + f"{get_module(cls)}.{cls.__name__} is a subclass of a pandas type" + " defined outside pandas, which is not supported: it would be loaded" + " as its pandas base class." + ) + + return { + "__class__": _RENAMED_CLASSES.get(cls.__name__, cls.__name__), + "__module__": _public_module(cls), + "__loader__": loader, + "content": { + key: get_state(value, save_context) for key, value in content.items() + }, + } + + +def _values(obj: Any) -> Any: + # The array behind an Index or Series: a numpy array for numpy dtypes, the + # extension array otherwise. + if isinstance(obj.dtype, np.dtype): + return obj.to_numpy() + return obj.array + + +def index_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + content = {"values": _values(obj), "name": obj.name} + return _pandas_state(obj, "PandasIndexNode", content, save_context) + + +def range_index_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + content = {"start": obj.start, "stop": obj.stop, "step": obj.step, "name": obj.name} + return _pandas_state(obj, "PandasRangeIndexNode", content, save_context) + + +def multi_index_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + content = { + "levels": list(obj.levels), + "codes": list(obj.codes), + "names": list(obj.names), + } + return _pandas_state(obj, "PandasMultiIndexNode", content, save_context) + + +def series_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + content = {"values": _values(obj), "index": obj.index, "name": obj.name} + return _pandas_state(obj, "PandasSeriesNode", content, save_context) + + +def dataframe_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + content = { + "columns": obj.columns, + "index": obj.index, + "data": [_values(obj.iloc[:, i]) for i in range(obj.shape[1])], + } + return _pandas_state(obj, "PandasDataFrameNode", content, save_context) + + +def numpy_backed_array_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + # NumpyExtensionArray, DatetimeArray and TimedeltaArray wrap a numpy array. + # Time zone aware datetimes are stored as naive UTC values plus the zone, + # since ``to_numpy`` would otherwise give an array of Timestamp objects. + tz = getattr(obj, "tz", None) + values = obj if tz is None else obj.tz_convert("UTC").tz_localize(None) + content = {"values": values.to_numpy(), "tz": None if tz is None else str(tz)} + return _pandas_state(obj, "PandasNumpyBackedArrayNode", content, save_context) + + +def masked_array_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + # IntegerArray, FloatingArray and BooleanArray: a numpy array of values and + # a boolean mask of the missing entries, which is what their constructor + # takes. Masked positions hold arbitrary values, so they are zeroed. + numpy_dtype = obj.dtype.numpy_dtype + content = { + "values": obj.to_numpy(dtype=numpy_dtype, na_value=numpy_dtype.type(0)), + "mask": obj.isna(), + } + return _pandas_state(obj, "PandasMaskedArrayNode", content, save_context) + + +def categorical_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + content = {"codes": obj.codes, "dtype": obj.dtype} + return _pandas_state(obj, "PandasCategoricalNode", content, save_context) + + +def period_array_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + content = {"ordinals": obj.asi8, "dtype": obj.dtype} + return _pandas_state(obj, "PandasPeriodArrayNode", content, save_context) + + +def interval_array_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + content = {"left": obj.left, "right": obj.right, "closed": obj.closed} + return _pandas_state(obj, "PandasIntervalArrayNode", content, save_context) + + +def extension_array_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + # Any other extension array, e.g. string, sparse or pyarrow backed arrays: + # stored as its scalars, with ``None`` for missing values, plus its dtype. + content = {"values": obj.to_numpy(dtype=object, na_value=None), "dtype": obj.dtype} + return _pandas_state(obj, "PandasExtensionArrayNode", content, save_context) + + +def extension_dtype_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + # Extension dtypes are rebuilt from their string form, e.g. "Int64", + # "datetime64[ns, UTC]" or "period[M]". + content = {"name": str(obj)} + return _pandas_state(obj, "PandasExtensionDtypeNode", content, save_context) + + +def categorical_dtype_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + # The string form of a categorical dtype does not include its categories. + content = {"categories": obj.categories, "ordered": obj.ordered} + return _pandas_state(obj, "PandasCategoricalDtypeNode", content, save_context) + + +def sparse_dtype_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + # The string form of a sparse dtype can only be parsed back when the fill + # value is the default one for its subtype. + content = {"subtype": str(obj.subtype), "fill_value": obj.fill_value} + return _pandas_state(obj, "PandasSparseDtypeNode", content, save_context) + + +class _PandasNode(Node): + """Base class of the pandas nodes. + + The children are the entries of ``state["content"]``, and ``_construct`` + of each subclass builds the object from the constructed children. + """ + + def __init__( + self, + state: dict[str, Any], + load_context: LoadContext, + trusted: TrustedTypes | None = None, + ) -> None: + super().__init__(state, load_context, trusted) + self.trusted = self._get_trusted(trusted, PANDAS_TYPE_NAMES) + self.content = { + key: get_tree(value, load_context, trusted=trusted) + for key, value in state["content"].items() + } + self.children = dict(self.content) + + def _construct_content(self) -> dict[str, Any]: + return {key: node.construct() for key, node in self.content.items()} + + +class PandasIndexNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + values = content["values"] + # The dtype is passed to prevent inference, and ``tupleize_cols`` keeps + # an object Index of tuples from becoming a MultiIndex. + return pd.Index( + values, dtype=values.dtype, name=content["name"], tupleize_cols=False + ) + + +class PandasRangeIndexNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + return pd.RangeIndex( + content["start"], content["stop"], content["step"], name=content["name"] + ) + + +class PandasMultiIndexNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + return pd.MultiIndex( + levels=content["levels"], codes=content["codes"], names=content["names"] + ) + + +class PandasSeriesNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + values = content["values"] + return pd.Series( + values, index=content["index"], dtype=values.dtype, name=content["name"] + ) + + +class PandasDataFrameNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + columns = [pd.Series(values, dtype=values.dtype) for values in content["data"]] + if columns: + frame = pd.concat(columns, axis=1, ignore_index=True) + frame.index = content["index"] + else: + frame = pd.DataFrame(index=content["index"]) + frame.columns = content["columns"] + return frame + + +class PandasNumpyBackedArrayNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + values = content["values"] + # A numpy dtype makes ``pd.array`` return a NumpyExtensionArray, which + # is what is wanted for all but datetime64 and timedelta64 arrays: for + # those pandas 2.0 also returns one when the dtype is given with a unit + # other than nanoseconds, while without a dtype every version infers + # a DatetimeArray or TimedeltaArray. + dtype = None if values.dtype.kind in "Mm" else values.dtype + array = pd.array(values, dtype=dtype) + if content["tz"] is not None: + array = array.tz_localize("UTC").tz_convert(content["tz"]) + return array + + +class PandasMaskedArrayNode(_PandasNode): + def _construct(self): + content = self._construct_content() + cls = gettype(self.module_name, self.class_name) + return cls(content["values"], content["mask"]) + + +class PandasCategoricalNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + return pd.Categorical.from_codes(content["codes"], dtype=content["dtype"]) + + +class PandasPeriodArrayNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + return pd.arrays.PeriodArray(content["ordinals"], dtype=content["dtype"]) + + +class PandasIntervalArrayNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + return pd.arrays.IntervalArray.from_arrays( + content["left"], content["right"], closed=content["closed"] + ) + + +class PandasExtensionArrayNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + return pd.array(content["values"], dtype=content["dtype"]) + + +class PandasExtensionDtypeNode(_PandasNode): + def _construct(self): + import pandas as pd + + name = self._construct_content()["name"] + dtype = pd.api.types.pandas_dtype(name) + if name == "str" and not isinstance(dtype, pd.api.extensions.ExtensionDtype): + # "str" is the default string dtype of pandas 3. Older versions + # parse it as a numpy unicode dtype, and keep strings in object + # arrays instead, which is what the values are stored as. + return np.dtype(object) + return dtype + + +class PandasCategoricalDtypeNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + return pd.CategoricalDtype(content["categories"], ordered=content["ordered"]) + + +class PandasSparseDtypeNode(_PandasNode): + def _construct(self): + import pandas as pd + + content = self._construct_content() + return pd.SparseDtype(content["subtype"], content["fill_value"]) + + +_registered = False + + +def register_if_imported() -> None: + """Register the ``get_state`` handlers of pandas types, if pandas is imported. + + This is called before every dump. An object can only contain pandas + objects if pandas has been imported, so checking ``sys.modules`` is + enough, and skops itself never imports pandas. + """ + global _registered + if _registered or "pandas" not in sys.modules: + return + + import pandas as pd + + dispatch_functions = [ + (pd.Index, index_get_state), + (pd.RangeIndex, range_index_get_state), + (pd.MultiIndex, multi_index_get_state), + (pd.Series, series_get_state), + (pd.DataFrame, dataframe_get_state), + (pd.api.extensions.ExtensionArray, extension_array_get_state), + (pd.arrays.DatetimeArray, numpy_backed_array_get_state), + (pd.arrays.TimedeltaArray, numpy_backed_array_get_state), + (pd.arrays.IntegerArray, masked_array_get_state), + (pd.arrays.FloatingArray, masked_array_get_state), + (pd.arrays.BooleanArray, masked_array_get_state), + (pd.Categorical, categorical_get_state), + (pd.arrays.PeriodArray, period_array_get_state), + (pd.arrays.IntervalArray, interval_array_get_state), + (pd.api.extensions.ExtensionDtype, extension_dtype_get_state), + (pd.CategoricalDtype, categorical_dtype_get_state), + (pd.SparseDtype, sparse_dtype_get_state), + ] + # pandas.arrays.PandasArray was renamed to NumpyExtensionArray in pandas 2.1 + numpy_backed = getattr(pd.arrays, "NumpyExtensionArray", None) + if numpy_backed is None: + numpy_backed = pd.arrays.PandasArray + dispatch_functions.append((numpy_backed, numpy_backed_array_get_state)) + # StringArray subclasses NumpyExtensionArray, but its numpy form is an + # object array which loses the string dtype, so it takes the generic path. + dispatch_functions.append((pd.arrays.StringArray, extension_array_get_state)) + + for cls, func in dispatch_functions: + _get_state.register(cls)(func) + _registered = True + + +NODE_TYPE_MAPPING = { + ("PandasIndexNode", PROTOCOL): PandasIndexNode, + ("PandasRangeIndexNode", PROTOCOL): PandasRangeIndexNode, + ("PandasMultiIndexNode", PROTOCOL): PandasMultiIndexNode, + ("PandasSeriesNode", PROTOCOL): PandasSeriesNode, + ("PandasDataFrameNode", PROTOCOL): PandasDataFrameNode, + ("PandasNumpyBackedArrayNode", PROTOCOL): PandasNumpyBackedArrayNode, + ("PandasMaskedArrayNode", PROTOCOL): PandasMaskedArrayNode, + ("PandasCategoricalNode", PROTOCOL): PandasCategoricalNode, + ("PandasPeriodArrayNode", PROTOCOL): PandasPeriodArrayNode, + ("PandasIntervalArrayNode", PROTOCOL): PandasIntervalArrayNode, + ("PandasExtensionArrayNode", PROTOCOL): PandasExtensionArrayNode, + ("PandasExtensionDtypeNode", PROTOCOL): PandasExtensionDtypeNode, + ("PandasCategoricalDtypeNode", PROTOCOL): PandasCategoricalDtypeNode, + ("PandasSparseDtypeNode", PROTOCOL): PandasSparseDtypeNode, +} diff --git a/skops/io/_persist.py b/skops/io/_persist.py index 134a63be..7311ee94 100644 --- a/skops/io/_persist.py +++ b/skops/io/_persist.py @@ -9,13 +9,21 @@ import skops +from . import _pandas from ._audit import NODE_TYPE_MAPPING, audit_tree, get_tree from ._utils import SaveContext, TrustedTypes, _get_state, get_state, read_schema # We load the dispatch functions from the corresponding modules and register # them. Old protocols are found in the 'old/' directory, with the protocol # version appended to the corresponding module name. -modules = ["._general", "._numpy", "._scipy", "._sklearn", "._quantile_forest"] +modules = [ + "._general", + "._numpy", + "._scipy", + "._sklearn", + "._quantile_forest", + "._pandas", +] modules.extend([".old._general_v0", ".old._numpy_v0", ".old._numpy_v1"]) for module_name in modules: # register exposed functions for get_state and get_tree @@ -27,6 +35,10 @@ def _save(obj: Any, compression: int, compresslevel: int | None) -> io.BytesIO: + # pandas is optional and only imported by the user, so its get_state + # functions are registered here rather than when skops.io is imported. + _pandas.register_if_imported() + buffer = io.BytesIO() with ZipFile( diff --git a/skops/io/_trusted_types.py b/skops/io/_trusted_types.py index 5fcb40d0..a223e25a 100644 --- a/skops/io/_trusted_types.py +++ b/skops/io/_trusted_types.py @@ -135,3 +135,51 @@ if (type_name := get_type_name(dtype)).startswith("numpy") } ) + +# pandas types which ``skops.io._pandas`` rebuilds from their data through the +# public pandas constructors, by the public names it writes to the file. They +# are listed as strings so that pandas, which is optional, is not imported +# here. +PANDAS_TYPE_NAMES = [ + "pandas.DataFrame", + "pandas.Series", + "pandas.Index", + "pandas.RangeIndex", + "pandas.MultiIndex", + "pandas.CategoricalIndex", + "pandas.DatetimeIndex", + "pandas.TimedeltaIndex", + "pandas.PeriodIndex", + "pandas.IntervalIndex", + "pandas.arrays.ArrowExtensionArray", + "pandas.arrays.ArrowStringArray", + "pandas.arrays.BooleanArray", + "pandas.arrays.Categorical", + "pandas.arrays.DatetimeArray", + "pandas.arrays.FloatingArray", + "pandas.arrays.IntegerArray", + "pandas.arrays.IntervalArray", + "pandas.arrays.NumpyExtensionArray", + "pandas.arrays.PeriodArray", + "pandas.arrays.SparseArray", + "pandas.arrays.StringArray", + "pandas.arrays.TimedeltaArray", + "pandas.ArrowDtype", + "pandas.BooleanDtype", + "pandas.CategoricalDtype", + "pandas.DatetimeTZDtype", + "pandas.Float32Dtype", + "pandas.Float64Dtype", + "pandas.Int8Dtype", + "pandas.Int16Dtype", + "pandas.Int32Dtype", + "pandas.Int64Dtype", + "pandas.IntervalDtype", + "pandas.PeriodDtype", + "pandas.SparseDtype", + "pandas.StringDtype", + "pandas.UInt8Dtype", + "pandas.UInt16Dtype", + "pandas.UInt32Dtype", + "pandas.UInt64Dtype", +] diff --git a/skops/io/tests/data/pandas-2.0.3.skops b/skops/io/tests/data/pandas-2.0.3.skops new file mode 100644 index 0000000000000000000000000000000000000000..e3bce055283e0094cf5b8bc67827318acb139be6 GIT binary patch literal 66609 zcmeG_Ta08^akDlgVQ~^-Rsdt^%J$LW*Z#cDrSMQE@uio&BxBc{4_#J=V3O}8GyWJa( z%62&%6}|46?cFE8c=o~j&VA&>$`dOezineOTi?6wU8irmV_M#J`{~;*?(FUFjkni! z_BJN?{XOH&*#y3yJv!cb_?7YG{5Nmn9k_F}e)-G5Z~Wmr?+u3CGRgBh z=IB80sx<%P%F5)#O>cn?to;3R-+l-`um8RG{cZD(lR#;*KT>m6TP5f1dm zTYvZwwFAKUqRiOB;YEOFoa_IL`{WY%`TK5sK{w`NIOt_0(69qnXUG`&z?kio4`tu~ z!h@erbfg#-S1*YLbVPRGde<>7k%ymq=XX90cko9T^PoLUClyUY+Cn;Vg43KygoeXT zn&u;u2o3Ij;Z{A57!G@xnAVi(3&`YF1kZCM7n$7c8j}36S4Q7Xc+0`Cp9#1Poy`Kc z>}}MUrf~P8o4?gl)nw2fl>Ky|{lp`OmTpQldGVI-ttLn=IyucgJd#f|=W@~Qrq$#c za(U&W-}q5Nvx`B_0bZ*@N+vQH%`S>|I(%M3p1sf+g} zJX-$vYhQgnMR>PNdy%Hmash<9Tu~V2XTS8hn^Y|?hvlfq5Po=Ce&H|v`G4Mi;osl; zyMO+lfAkN(^B?c!oX_PcodsmO3k*QI1a5u#(lalGBh`>osU*IPRq2b&Z7{UZm{&Hd^2%-`GG8N&dSy>}P1 zIlHqlvG>+@w)ZF7`+y9w1OD0m-r96~V{*CrNM<}N`1Q^K3aS)UwCZ2%DkWMN40}y&dmcbWDYh7rKFSOz(PG@VV zGr+z5y@QFDmVX%VXu4ta0Y^0Z?QW;v9hOi1?t$Wnm1O5|-R=Z+o+Xo0KC>;;F^J41@3^2&1(OsLcZEWuZC7()0GI zssjwL9;?vuN#8_Ig62q=@YvI+H zyPt_54uUc{Rh1dgkhX+CVs5Cx#9=4lO{>rx6s0n*(i{*aG*zh?&_Jk60%M00lc_l! zh0LVmQK-%6E(kG@o2cLn#(dCm59Mj0TFD_K(JTBDFD_F19apbw%a=!#fP_?LQaRfR zVRVNyx!R}%=p&NwJCF0RiCSl+jQtbLiom}0Y<;)Mp4*aO+Gt&j; zGvKyi1h>NrZRAj85%`C_jyhj1E|f22yDtYIkPXL>FLDUeiSvr`U(lCM*?Vo|zdH-n z7nloyNw8&IAk#u+j_8=$(f)$9e(Q@LbIfd=9T-A|MRh=WRiaU!%1d$@}Bpc|@G0^VTU^^eEfmnI8K$^$l5E*P$a z!k?#cFu6G13oCaJ)?Uz6Qq3H;XxovoWy(Axm`ldYq{75@yUeTgY=3806ig1;HSBwL zF@5i_>4V4T<5;bJdhafq0JikA?(Eb)5R$)U4oTQA!bzjeq(i4!n1+S zh3DhShHCe4BQ_Z5W59+GYPqtaRY*zUcif1yNb8_#Z3*G<5;m0_d^WIw+~pji;hz1k zHqkW9tIz#G7S6|J5>CWLlTQ45by@kD<>@JdXck_OllGd&C;&Gxi#Q>380)2Ph$J!A zC#Etvvk0XMC`up^`o$t~DM>p6;`>4m7 zIMc^&)Kqx}q;jZkiq1|wi;N#3r8zAF3%&_qaq6%lBg%iG&kXB zB2fX7$2F+D%rkf|^qR>5(#kNmZqI<{t1tr+S7O!qKT!$dNQgqFX~D`6gcL$HjW--c zF)K|JC;TFExyv+?xt}A3RsdrC07CeyPWI**ssbQ}tP)rNr=&@3Fi>m&F&^YZ7=Ho5 z4ZDUd)b5!FB>QBVg(O0gh^0l}-eB*A|A7J`y;K$y;0?{RhiR?%jYKEww7+T=I(a@) zQV)cmX%<*0-ou`6QILyqNmwvHMC-S3YDw2?SMk#?+vmG+5`wA){pxe zwQUKs1^%vh0t)h4*3?DRI|*P3>uVD?VcJ|lZ?Oxa$BkVDv<#hYrRsUcq$agJ zYapG4Elb2C3~gb5&#A3i|5Fs)Lt#nZZ4Y{Vk$}<`HU23^2$RZyk!Q3NVaDC1rdeq7 z`Sb^%=hhr(L(|AcIzzQZU6sqa&|Fmldus**J^qw-B)J}*;AXVeUcArZoAeObuA6 zT>9;>asWxhOQMX>X{f+NOO8r3lW7MaA@YND#*_W&*2J7)5it;8sRV7|F<0Gr7LU*P zz^Kpmq!w-_cZ!*&tT(xQT=XTKhXWv0!@B&tN+L2Cwd&Iywr8Ziv+uHxrg!yvOr%YEcDx1v`vbKfyv(HbbC?j>+F~~ z9?hX}N|>=Q6XVz<(??)jaz6<{vr3aoU`5ot)9JKB0QL4=Jy~f2E*nk;HP3KhEtq{b zCR%f@j<;l9Zwq0Hp#vd4J5&_c(?oK?K6SeUjKsTW_9!Tdtws=)vcHVowPJg#P9A3V z29J_5gQN$}D6WKD$qMx2h#XIW`XE`ncX!edSHuO2JOg4Ycce zxyHN>U?6I66jjsH6bC+_y;;x$KpH5l_0$GD)tHRaPX;>9P0grW(4ch3hUI`L%_-wD zp!s1{#mK5tv6+#%$3})`pesW|s7e9x4=^~UQuaIJrip54K}-lR8Nh`6lEI5e3xp;9lp(|bi3sEKS5sp#-^)xNLly=0Pv8;K$R5lLcQ@bbONOc-O)rd}Fe? zKdwphAMkKT$xUe~2{_k$IsyPqQjz)c{OYw3%(%61VIV6dCMx{8jh~%?J4VRs98eQGGN4xw zEZtj5m_512%AC}`)1kA?PlJZ21Xy8Ne0w~H9Tbtc|84fF+ zO?S4_%OH}%G)R>Z0P{5aK$1&pWr@MSiBFa$4wiNyS)LeJGRbwJW$pI-eXC0FH_-Lt z;lMvB-BG6l`>L395t+};@prt0#EmE@Dm-dr2rqLe!oZkmMi4U;3MFu~COvDIfG>C6 z85v0RPwEa)_ttXfo#mb5VBSW!R;Ctr6U{`d%bj=ph#JEaR)cGP6Ie|he$rtR1{t);=+St2Ke>;$sP=l)>KCy za$rZPIPx?tIwxg#=g3MoCl~1;7-}nCpqM0q;S97828NpAJh_)a7*S~+9C8<)KbxD$ zL2+}J(VWVn9VDe(9~^gp)QjW>$PIWtqc$M%Y*h8?yeO-7B+!J(yf4_064`y2Ol!pu^4uFIdY~YUUcJ z#NCKlEe1K zDG=!l+@+4p4k5*93@Il<`qE@^3?YkoF_kK9(Va>l!5U>-DyhaJtiv0BlPCUgr{!91dbNd}!CLt&t7s2Rg&KXKH)Z^go=|I&UQ zcfM*b5jS#^A*IFMZ-{N%?Ev0IZ)+41YENdK#-9K+^v-Zz|``e@;%b~48KG7 zJfE**eJZ#nF{!b20PiuYDPHeGn5!;Z0~S_C5Qc=rskJ$IViGM{;SxG(2R?EJp-;d| z=#8lFEsTMpnv<&YEAS9D1JA0P*;JuVxJ*kPiWV^;mPdQA4 z6{OeJBcca$1uy;O)s-;jjb7;cO#P9`lZhkq&%w{*#s*WLnD^Ja#37YyOYXj&XL5_7n0eeD+PAnD+`w*d&Uc|KsRS9oF z1(}5e(xBn~Li$|J++|8&&-|-R3B13}BpLAvhT(A7>7Lo%eNr9u(Zp$5RP}&Yvq%%b zH7r*!RNN8#inygQdekYz_IGN2tJH>+=7Vkl?**%#OP&bcBc3GUHh0v@f>eqPR8t45 zYd!!`&ziHNDrZSx((I?#EmiwbWCH7@0lJdO+^+B_vjJW~HwPQ!3+v|6Y~Uf;a7#5C zJ&v6IJldMARbWD^la-fH>}1N&G+n4k`XYr^(*z<8)bD(3KkVK|gGESU<8CJgjS$D2 z`a2?Cz&EWJ7x5W#sG{o+jd5VDJ$&&{&6rboolq6sg>H$V6roXe2qj=?Fy^Gaj0mu1 zl4U`Ws_lX_Ke9Plz5tnd zG^281fP{xGeI=M_!ra@T^OZd#Ie2l>?RR;Fw=iZrhi25OHzu$&u~xlu1slkP zRVT|AAnSZdjpk;YM1@F3vgGO;;`DvvqpmN9^I~XWMys^KMov}L;bYn=napT5H@(zI zHb6BrJ%FT!mm9$*SyAjJbVp&FKqg`HN;99cFwm6hth)ec;xKqtK%QSewrL)#rFbM$ z2nc;V4c|mcAQ76Q3Onrr-l;A0C$cjT8==H55FZY^qM>M(8b#we^_ncC7$-F56+Jpa zM!=Wb$E^knN0Br+YHYl`Ks+_khfBuSP|Fv$1}8!^)9`co;?^ibUAi;eD%~Mbi6{*o z9ecc9+=_deFHQF1t9mI9VA%%?5bLg>cp$GD1acTQh~ zEn;?wZyFjGJrf&ML^R7A*FB(3`_t`9OY1sB>l)0r24&DLpoLJ5NX~}-*J9`@;8>kG zoe{K%*(JUaW|9b9{k9yhAZS+UkQuKF&2R^vKT9_gxTe*Ul_n)FW8ud^&vwUqv&r!~ z&hmakbL>>0$8^vyoeKjXjdL~+t`0&ph?kQYX!7I|%zafRGumh30TU#lX3)fd@bN%A zP_ZcBcmyTOq;_RHD5ID|NRn}FOwd75l{*$(>9q%8X4f)6_4h}cE1fd~HoxsVGQX}s z83CId2J)3rh)|AUGoOqA%V`Yc0=edK6DBHBN1Pl}wGW^pt4Deo12t+m8g|8Nk`yZg zcEw`}#YpVdtzx?4L$i}OY?xg5Ashs46eN*8BBj`av}Om7xG{$KrqPEZ;&d;xE@JUg z^u-brD*>Rie!5JP?XcZ*=ZSkn_$lOzU*Vf%HrLF~xkjT!*D^qz9iLr{EAF>DUDz%G z(;Z6KCaJN?Glkud?{>w`3+mAs+tub>`r3N}zMK<75g2=IOe2-$oY-`bEKsW+#p&H+_2YV!{&J3%;jUgQn>IxSA-wM?ec+xIi(fHJ@fxzm&ObgHmT5q0Br2}x(xh6+qh-Ym@0o#z2FN{O? z6M-9XPzf+jgw`R5m!m;qZ|-z^V{%zk&wzmh@vt4MiqobKT%|=McMz0aHkt=)b3{&U zlb3q}AykV<(3M7p2XK_uMxu?!;cLu{OU4%}$8zCfhv|ZmMN^%2hviu3B4$;YB7^BJ zi?k@?;Kq5_I)7B$8J&nAN^|@cE*@I8pI@~Wzmyw}y6%+vq8jeUOnA1?EH5NtLH{5# zXRgObnGLY}UtRs6vjLzv8^~X)G*Q-?54v44@90B(#Yv?}3CG>IQ#%}3YqE_8ZtSgX zjdyopVG#-LoaYw$fg!?o=-UV%004F(5kTllgR%iA)65D zqbLSEB|CF7DL?llk{G6(i-V+;>!T_TkP@F8Sux;wc^VMybeB#~Ki0&NKt?WXSp>%^ zk<1>U@Txq9{S*IBl}F|tOZc^%!a-41Ws6whE|U4Q!5iePmrDjHpI$op9D6cl69Rg_ ziofF(aSN!A7+sF~VmY>GmSQj+2>AREO}0#DPrY@%W)30;75MN;Qy7|8mJ#egy6G1F0*x7bvU+|zGA zj~m}|Z>y>CRlegOXySBo(|Fi|`Yw-QH8>^z6=<5z+aJ@;+(5X@xf|v-3zey}Hyq>~ zstWz3ikdrv!ypfOdY+^zl6nK&t4&u!I+4*{zcZZuCJZ_qjrl0uydz3Ra0v{hvoON!QDieSkjJ7hm+fSc`-|^?I z_@%o(bpn40kAoK&^~{THB!Zh)KluY4!9Otsu%XYs(n2Em$z3mf@M!?WA5Ov3KtrIQ zY{NTFaY2AY&^`6br*#Sj*Kq`}#8yLa=V<-%m*MI7!zp-)A?OWYkYYXr|9Cy8U})DI z$u7LScIw$@;OY3oDR|}$1i_#UleA`0ghcS&TfY7?od@rKBS8QMA`FHq4{p5mhab@q zT=y#k0Z;(@Wi%fA{CzjRpfmAbC;}*#V6Co(;QL>A@bfx?&;2T)0J<giqhZ-Kxa_$Cp&c+2-zbp(HX1Eav70VWZo zDfp9L=Lp=XSQ5d?N5AnSod+j>gClU~BuNBm3jR$Xa3=>z1n>FEsfToP{^V~m3j7*G ziQvay`|9&Lg5g^@0=IfkBKYi=K6jJOga4uk!YV6?AZ-hO_gjPlum!iIM`9J*Y{{i)9*#-ar literal 0 HcmV?d00001 diff --git a/skops/io/tests/data/pandas-3.0.3.skops b/skops/io/tests/data/pandas-3.0.3.skops new file mode 100644 index 0000000000000000000000000000000000000000..79549f1da7667cf90d5b2460fa2798cbd858eae4 GIT binary patch literal 78724 zcmeG_Ta08!b+a~*u-J*SSYQxICXHkh#aZ|J0Ty7eEr{9GCTk0s-C^9G>9yO;i|HB5 z4j2$9k|PKK@&Ur)11S<8NRh%F|y_m+!c;y}P$N-df$> zU7x7y_m4Mb6SzKme7rM(zu(rr^AGP_erEX)pVOy+EcLm;d@Wrd1GsJF6VImx&}%o- z1GsW*12A;I{4V?bNjiX^-2037eHw15A1;ATdzc)+$EU61lK>_8lXt!R))dd%opzGs zyUz_T?ax+@f$Y;uAOEY-f8P9>rKKOOfAB^4rUQ8Ck;~721^BIg7|(~zW=3gh44`u; z%|E@g{0F!EKn>tiFaPn+)b|Vj@xcGp8_1}CB$7CmftL;nV>7gN!jB zUj%s8y8gek-*f=KeD|3bQ^wpIWNe^t0Eg3m75PpB)Ac7G%D(f(hd!Sg$gqF-SbPKn z;RCqQ4vb3VkuTl#t?@R+C|~)k?ej zk3V{B=_Xf`pTFtbD=CuO&2(;Xj3l3`!R2lyFFQt(IhU6{^z|R6G`lxSE5_Z|Y9hrG znVe=1Gnv>m`(vU_x}4`l=JC4W!M^>ggoWk zcW-*>o;RObQa`VQ7&Tjae6l${`^0Q}>tm;vmm90AYa8R)Y<0Eq_T|R^6_zxFQf`+~l!5qJ_b7$-wtd~&?8KbavqEJnnDi+j7% zt;g=!-5o!z$D<(B2iO!-b#|=a0&x~FCh)$efG4x*_LfFWDAjEwYV{3~V%rU(8dwxj zY@smKTl>F9Tl*UuN5_@^sNL*e2VA+_SUXx}88y4jSXtPW6|Ovuy{C63pq34-p3z-h^k?48 zXe~9nz5SgHvi;3@OKBADB zQZgV5hs=yoXp9GpMbP_EI2fxCtxmt+>5uwEuF3+qelb}`x}_E(2c3Sm!&C$zCMX0A z8*~QE=Fm4l;F4^pQgNA-$O2$0Wq?!$@&M=sQoI@Cf5x=gD7)=er`a2Hx*^axKKs$K zCW^1UQKvI9e6?UYHJ(j&rz9tZrbtaMn%mCA;3n42lH}N6Af|-)5e%gv=9-L}MRt6> zX2(x4drSJg>;@;`0KIv*3vxa{b&fj)*Or;WRdUo%A2nMfFYT<` z4Q<}pay*j;A43dVRyday*>TVp{aYQ2sF%C|^a;t1A<~X|!yXRKhD(v}0dCne9cRZQ z6xCHj)@+d-ll2+k219I`VP*sZ=I(FYiv%Rnyjy0*`=#T&=&FH-tpe}ehuNmJlx0JVEzp5 zVm=rj=8+T>BXE-^Mz@3(hYhh z2L8j^+w>>{m~#Q6?g$>-6Oth@pHYOxnqbJK5|^Td9~L$Js~q{STqU-(P%X-hOnrOr zi7%c!w(&JyJeo|9Az)E725lRn1#S#{ov>~DDrpPkB!5+N73j|(?yRz2*)bS~aYD1@Fx;YQwVu0#int);2} zUrSN({8sNcJ4z?#C$foMe`L{ofUZhm$)Q6*t&`H`lP&-wL$7Ri=8ieW% zkfFJ^l!}+iuvB(fW^+=w>9z(Th#|%mEEfOScs>LvJ6=(AmNR(n~$W+1_OShTb}w zp|hNl4?VH_{Mt%=PJCHsfNC^|agBATt1?re4BME_%tMLBeUwFSK!GFn2LMOTbzLGc ziaIiGwy8=WbcvG+d1nZtE4M4ePe2lwu1w*v9(X#)Czs(2u~^%^6JyY1+(;~n*C&?^ z_9U`GHURX&*>G{O>&e-r`mUkh8a!kV)C3()u7mD6^iz=^B1mkOmD5r-g33uL&KnoI zp4?O6wb&+xC@;UVJs2Q1mZq_#L5EL$^SO+=wz4P!DK-e7a$75l&9sZU_6B_fr_qRW z+JLh)7(`;8x;XM9M?T{31XV=N|xzXD!(v zioIYu?Qf zTYsr?rMCT+D;HC(+;(52S+OC+(?Z1u9p{P?@^XqT)?$MwOAHF6kAt=|iDW{+J))ok z7@v>fuo47rk*QCaJ00+qbk>X;B~_*Oj`t>yZSPLk#_`E7OgLsoTdl`*5ca6vq6He9 zEGJXT4^BL1#URy6eds7sLBT>|k1Ll?EFjjdTroz2yRG~gj8aj?9xEPMza@dEhfv0m z_A@ad>&B_WIxJ>nkVr09ZG$%NLsS4wT22WJfK}3@HpC;07dsS;`-qw?N<_5)Xsj4! z@22h`IVRn#YFrv9`8b`r>poQ5b44cCLVb&)OxZ7M7!SDMkvoE;Wb~TKA`WOsQsv3n zJRyW;Txn}%jW7qQd;w`11}?=*44R6b6ov|4TKuH&<}W7v%0d^$0d4W@r>Z03E@Nq+(`W{xK0id2O|!_BhbkbOX_?NKFXr>h&GbN zBAow0o2BX~0cwS(%i)&!EuUOR(D~==FgOee#m-dnja#VSp%up?O%l$;#kzSQsWWPw9e>OAfrPbT=e zZX9ue7Yv3R5O)TxzPz{Ibx`44vlU2(Lw!@5GZ{}hRb?%)-LdCzFeLt+*__Goe9EJH z&L(>f7t*`xAd=ed4Ckj z%#&F&)ZAtD8rN@<>!AnlQsH?_II>+tlt9zauX~%UCki8q2iVpOAEKN2x(#v=tF zLGpvAL??UG&51thCSoAKVhJ9gNOxWx1pc1!+D9g3#~E^JLEo}dcsiQgw4IG5Jyi!l ztcF$7b3Zf4V3bOA_H}v@&y^+s2|}q3T0~06gsmIz{dvNCDYH_h%NUh)wTxHUJQ|zp z<;H#cyW2aHyWTU|-I#7A0(gOcjX5=flC`*`#CR8RPC(V|bvVUXQ4@V*l~{&SunM0QJZMRyA7Nr)8Zs_5nHg=Ik<~eJym=}k_+}J+a+K`-bJ!U zMp0}vjHs0TW$dmY*qeL%qaS?QPD0?Y=;Ln?8 zECo8iz64$=0B3rS`pB1*=q1D%cxkL=i>#Nc#OqML%%f^Ln&QC6q?QhPOmxP2Dgz!8 zW-unLjLASpxhWZyRc37Jv0+&tZ1E}M(xCY@DA~x&8?cCx*>15C>c`6lLtaqzj!`v% zC`eSwTr0^xnE*wz)A*_P6HQOUp>ogV#zPOjR|r=b1GHV?@f#y-@tsx95LFAp3m6e; zt`6pJ4{;p^fzOAZiZ)tiI24BG6L0!@562ik%Fx0k)x%*2UT~f8P*?P!E8ADD%qI9T zA;5xymGcPm6hIwK3oiw)KN4$XnKVEfo|FpM*A5qYDZV??E1|+a$6o4>nyt4tn}_yQ z`kdoYuh*ozw46yQ77&gw(^N;m!7h>(YL38PxxUJ6@v_Wls7d(xWMgk!k>)?>;O5Cq zNht~FK|UP;fRa=sW8h*6VqeWojGO7CO+Nr~e)Ig}j-OzoO?c%h$?$jJl_1;et0(Hk zAr~~wM;$K_)+h?46M877PPfJz(l#^DNR)a=_M>A-5fUoq1&i8{;_GdC+(a!U$RQ=fS1R^KB6<&^*h zp+1cGcxqv}CVm2z+r|f#-RV?jUt+aSfhNr!|OFf9ehTXSam+V_On<;x!x$p@mO( z6qNcPS79yy!%jqkD*9bDgo>c6L40270Ql3 z3(9(WCrZiTNSdDm(4Ax_JPwjdE@hVvTH~c*K(&o3qyrHd5;Q-L)0Q)S&G2D?{cA2eJ5V6=o&Lu<}SPM z>y7Q><6^sTM2+LA`gU0izUF>k;S*1HLw);T_k?t23xOXF4GaP0%rH_#! zFPL`_)h`u@5H4!q9X)<15q`N2U_5PA9I8S^S+wSUMq3o^k3xOh4O`|&d>o>2Z|H?kE;&%-EXNT8|Gp>o}+t1|$P^Ug{qi_9YbxMrbG4eWDC zSE9VExhDumU^eI+dW@qMKUfmA$_3AhXF7@&%67hsEJfcc+%1hatmfkU3}@}rmB z_9A`R4$vYA&D)XxltIcTs860z z21Y|+%`ps%iBH((%! z&4zSL$s}3n8Tv_QYgJw9&}_}rqi~FMvQghFDR!3?Lox^1x?ZTlR~13(D?C&8lHx~^ z5myyG=(ws3If@!ZrW{3)B4dVD$egQ+95u8~lf%rpvbd4byo~PQethBx1&tv~9G!|K z=T0J`;h@t=SPDuf18zW)T7+^yR2NJH&KED3p0i}j;dS3k$4xHbLvVG`0_j8AiAo8y zBfx#rfmSI!8)T8nv!NC#JQ^NEFQBTTI8j68fgD^VC@4y#^gW?lDHY{sJv#|V=tb0w zaTpP0w6{ydrAiYs&aDW#ssi9RnE$s%6en62fI=E0Sq2*^&(HirEaGQ z1saJz%Aj(`L)_ax3gav3^|4l6}EI|o3l5mlMFPgXJ* zrL79Rj=WVt2}%&>a4>=vOb;TF>c)$`pNb9Y!yOhg0FlvPH!T*TF`5;ev`TVc2r?Q(&N1?*q|$J{gh-fP|(D7kX)3aDE^Jqt$RsOg#U*%hv-XI#W*wFB3`M#_RUcGXL+D+=7<)zyAW zC(|u9=VBUV=2!tsL4jL(@ikyY7RY7JC=#`u(Isj+)^a-E&j=V0=jZZXDxfFAi>+nbl zJjdGh#{TA3K6=M|RXuyk1egG)Au;J< zuwn364r1UT@L>p0g_|;SCy>>383hiCzyj2$JCa_JrGULBP=AEaNn=}XjVR+W!+>SB zCHE>{Cv3ZNm9(vu*be3rSIqYn7cB5aKdXKswWqjZ35`10-0~T7vK98tylmC78FH?& ze~RbtfemyFJP8hP>bH8Muq|r&Dd+=vDw=mOTFU&CWOA@9hI2BCSJIy3au_fKuRy3e zJ}iaFn+Qc~77z>vT{!kWiZepfEQi7z0!S=TmI36KyO3;U-=ACB zu0{`vw)m^DQS{jxs_gYAYOgoj=J;nCFzqk||}< zX)={eCPPU5qaK^=%KLPgBu%PRt6p+$ zJX^fw2Na^=l?OJDHCys>rPUUmW}6m8S@I^Zo2j2dsy-k~WVhF(wsfGAv4-m06mfOe zj!Iw`KWRll9w6g`vrS?9(;@cFyk~oRV=~@a7?%|lo=sI&A`hr7@K{exZE+`DDcZ{^ZQ7e|J+`ne zE34zMuHLAwFNPLOIif6E(|=uB7nn;;X9O)`cCjnWOhiKk>l(D#+y&6}kl3Z=7U{AMLN*3I zH%&PS4TpVj)=g+=Om)^#0CaG{jnraXf%tW|Rsm3nRbGVS8XzvdF@-Vjy5vI>UP>vH1?41KTPO)R7ps2OyPy$mP?JQ6aJPK{ym)m(&U zP&QWeeyXBk(Y55&`SJNkb@UkON9|U(-x@T#t)rU0wmWR?8k;j=dn_&3P7H<^VgdqR ztcNN2mb-6kS#gF}VM{?W$XTDr^{_Z*9c1V{P9eS2GFiN?hKCl=po*{p5I7ztLkhs* zh?oo~SmEboJq&M9LpmcZBIX%Z!mJWTWto(*48YB@H+I;UbMk(ap@m=L>35nFsiw!JkjzM>u}NuIzce4PC7#mA7%!y>pORHP{#NjGD|RX1Uf^YlPA}} zI|FyYdkEA_f1w~;GxP6p{QxxzTM<7_VW&r{0EyZ&xH#FJ!n@oU zTW}^OOb-|-rov?~5t4Lwp{ha^aPIac9l$<&J$Wx9Fi5aZfJUlF+@|K#bZdR`6q8mg zvH0{13sWI93x##a2u?mlLxZPer*9@t)37}uMNc^w3yCS0y`u%h#OL;yY4E%}HHhSM zM>S7Ugnf=|5m%x19z*?MKB>8px=W2xnjcvx%&N#mn10Ov%tn**B#k%7S$@=0169{x zg4(;KP`2{n2y}kc|Bd~R?kqN--U#d`xMl^#n4Ds8HFE)?h_R(iox8~=EGM}jr5R`a zkY`}yBa^3BmBG-Hdno7FgJ;!DW9<6CbVfrK=zY+9br>eMOhHIN7QH7 zX?5W3)I6d0n#X(}Zw`C8)aD>`)PUDNIuD?t}sw85jwrf9xR7#1v2=g!#D zR8{9}9RT=WZOj)<8ZOqUr z3^2O@u0~Ts>ohqwp>XD$T?kkBQ<|62J=_mk6z}uxuL`~SB)tYmiDW=eHkGr7*{Dh8{{!X zFQ7`ul*uR#62D3o)Kw=beNX6?sOzAj(t9*g=tb0`<_}tZ{&Wt@`3;|PxI#P%eBs_?V`fz_|byCRlhlE9khw1NP71ZzfuW2`OP zQMc9P=dz)x5E53xxApP4CQdC@T+u65hY>XB+S3OVxb8tHEmD0z#Vn=*EDka8vqBaU zw96HPz-f=vRP&uGH_JK}N{0Z!4GT71`LdyB^@5^RaHK9J?Js!5%3%vVTG$$J0LMY* z1Q-6ua$vzD6PZwEwTNqBNq7-yPg11SX{pUDLkx7&3|Ju@nM~GXls?(AQ~hRV(1A)( zQ)Q-&o!#xdZFnFN@{G2=usZw5_ReheNqD>JbbAXft8;wLu1}t9oPOrQ1E+4d^~PJ5 z)Te#-J63M`^;>@W1^8Y4yk1?jpHH1qzc_lJGBtpM=hV4e9KmfXpLkwI@XrhZY-W%U z{N&zWyzkQhs(vU1p9UH*1@Qh53Bf1tdikv~1-%<60yx)SLh#Zfm!JO%^sas=1;3yO z;22VI2x$cWcq5@;*fd86aweX+<%Juch2GTx;6hVI=GVz!1KJ%i?#D8T7dghgjoCn|e;zOU85q$19Cf>1oq{#96|5C7jKtU`QIo4_x8gM%9KqXBRKcnn_jx- w&8L>s4``16eCpKhnWd%K+T)YW@maNkcj19ogM(OFdNcg{68!d~J5}iUe^awWAOHXW literal 0 HcmV?d00001 diff --git a/skops/io/tests/test_pandas.py b/skops/io/tests/test_pandas.py new file mode 100644 index 00000000..a2c148b9 --- /dev/null +++ b/skops/io/tests/test_pandas.py @@ -0,0 +1,359 @@ +"""Tests for persisting pandas objects.""" + +from __future__ import annotations + +import datetime as dt +import io +import json +from pathlib import Path +from zipfile import ZipFile + +import numpy as np +import pytest +from sklearn.base import BaseEstimator + +from skops.io import dump, dumps, get_untrusted_types, load, loads, visualize +from skops.io._pandas import _public_module +from skops.io._trusted_types import PANDAS_TYPE_NAMES +from skops.io._utils import get_type_name, gettype +from skops.io.exceptions import UnsupportedTypeException + +pd = pytest.importorskip("pandas") +tm = pytest.importorskip("pandas.testing") + + +def assert_equal(expected, actual): + assert type(actual) is type(expected) + if isinstance(expected, pd.DataFrame): + tm.assert_frame_equal(expected, actual, check_freq=False) + elif isinstance(expected, pd.Series): + tm.assert_series_equal(expected, actual, check_freq=False) + elif isinstance(expected, (pd.DatetimeIndex, pd.TimedeltaIndex)): + # the freq is not preserved, and assert_index_equal starts checking it + # by default in pandas 3.1 + expected = type(expected)(expected, freq=None) + tm.assert_index_equal(expected, actual, exact=True) + elif isinstance(expected, pd.Index): + tm.assert_index_equal(expected, actual, exact=True) + elif isinstance(expected, pd.api.extensions.ExtensionArray): + tm.assert_extension_array_equal(expected, actual) + else: + assert expected == actual + + +INDEXES = [ + pd.Index([1, 2, 3], name="ints"), + pd.Index([1.5, np.nan, 3.0]), + pd.Index([True, False]), + pd.Index(["a", None, "c"], name="strings"), + pd.Index([1, "a", None], dtype=object), + pd.Index([(1, 2), (3, 4)], dtype=object, tupleize_cols=False), + pd.Index([], dtype=object), + pd.Index([1, None, 3], dtype="Int64"), + pd.RangeIndex(5), + pd.RangeIndex(2, 20, 3, name="range"), + pd.date_range("2024-01-01", periods=3, name="dates"), + pd.date_range("2024-01-01", periods=3, tz="Europe/Berlin"), + pd.date_range("2024-01-01", periods=2, tz="UTC"), + # a fixed offset parsed from the strings + pd.to_datetime(["2024-01-01T00:00:00+01:00", "2024-01-02T00:00:00+01:00"]), + pd.DatetimeIndex(["2024-01-01", None]), + pd.timedelta_range("1D", periods=2), + pd.period_range("2024-01", periods=2, freq="M"), + pd.interval_range(0, 3), + pd.CategoricalIndex(["a", "b", "a"], categories=["b", "a"], ordered=True), + pd.MultiIndex.from_tuples([("a", 1), ("b", 2)], names=["letters", None]), +] + +SERIES = [ + pd.Series([1, 2, 3]), + pd.Series([1.5, np.nan], index=["a", "b"], name="floats"), + pd.Series(["x", "y", None]), + pd.Series(["x", 1, None], dtype=object), + pd.Series(np.array(["x", "y"], dtype=object), dtype=object), + pd.Series([1, None, 3], dtype="Int64"), + pd.Series([True, None], dtype="boolean"), + pd.Series([1.5, None], dtype="Float64"), + pd.Series(["a", "b", "a"], dtype="category"), + pd.Series(pd.Categorical(["a", "b"], categories=["b", "a", "c"], ordered=True)), + pd.Series(pd.to_datetime(["2024-01-01", None])), + pd.Series(pd.date_range("2024-01-01", periods=2, tz="UTC")), + pd.Series(pd.to_timedelta([1, 2], unit="D")), + pd.Series(pd.period_range("2024-01", periods=2, freq="M")), + pd.Series(pd.interval_range(0, 2)), + pd.Series(pd.arrays.SparseArray([0, 0, 1.5])), + pd.Series([1, 2], index=pd.MultiIndex.from_tuples([("a", 1), ("b", 2)])), + pd.Series([1, 2, 3], index=[1, 1, 2]), + pd.Series([1], name=("a", "b")), + pd.Series([], dtype=float), + # the shape of category_encoders' TargetEncoder.mapping values + pd.Series([0.49, 0.66, 0.6], index=pd.Index([1, 2, -1]), name="category"), +] + +FRAMES = [ + pd.DataFrame( + {"i": [1, 2], "f": [1.5, np.nan], "s": ["a", None], "b": [True, False]} + ), + pd.DataFrame( + { + "o": pd.Series(["a", "b"], dtype=object), + "n": pd.array([1, None], dtype="Int64"), + "c": pd.Categorical(["x", "y"]), + "t": pd.date_range("2024-01-01", periods=2, tz="Europe/Berlin"), + } + ), + pd.DataFrame([[1, 2], [3, 4]], columns=["a", "a"]), + pd.DataFrame({"a": [1, 2]}, index=pd.Index(["x", "x"], name="dups")), + pd.DataFrame(index=pd.RangeIndex(3)), + pd.DataFrame(columns=["a", "b"]), + pd.DataFrame(), + pd.DataFrame( + np.arange(6).reshape(2, 3), + columns=pd.MultiIndex.from_tuples([("x", 1), ("x", 2), ("y", 1)]), + ), + pd.DataFrame( + {"a": [1]}, index=pd.MultiIndex.from_tuples([("k", 0)], names=["l", "n"]) + ), +] + +ARRAYS = [ + pd.array([1, None], dtype="Int64"), + pd.array([1.5, None], dtype="Float64"), + pd.array([True, None], dtype="boolean"), + pd.array(["a", None], dtype="string"), + pd.Categorical(["a", "b"], categories=["b", "a"], ordered=True), + pd.array(pd.to_datetime(["2024-01-01", None])), + pd.array( + pd.to_datetime(["2024-01-01"]).tz_localize(dt.timezone(dt.timedelta(hours=1))) + ), + pd.array(pd.to_timedelta([1], unit="s")), + pd.array(pd.period_range("2024-01", periods=1, freq="M")), + pd.array(pd.interval_range(0, 2)), + pd.arrays.SparseArray([0, 1]), + pd.Series([1, 2]).array, +] + +DTYPES = [ + pd.Int64Dtype(), + pd.BooleanDtype(), + pd.StringDtype(), + pd.CategoricalDtype(["b", "a"], ordered=True), + pd.CategoricalDtype(), + pd.DatetimeTZDtype("ns", "UTC"), + pd.PeriodDtype("M"), + pd.IntervalDtype("int64", closed="left"), + pd.SparseDtype(float, 0.0), +] + + +def _id(obj): + return f"{type(obj).__name__}-{getattr(obj, 'dtype', '')}" + + +@pytest.mark.parametrize("obj", INDEXES + SERIES + FRAMES + ARRAYS + DTYPES, ids=_id) +def test_roundtrip(obj): + # pandas types are trusted by default, so no trusted list is needed + loaded = loads(dumps(obj)) + assert_equal(obj, loaded) + + +def test_pandas_types_are_trusted_by_default(): + assert get_untrusted_types(data=dumps(FRAMES[1])) == [] + + +class Encoder(BaseEstimator): + """Mirrors the fitted attributes of category_encoders' TargetEncoder.""" + + def fit(self, X, y=None): + self.mapping_ = {"col": pd.Series([0.49, 0.66], index=pd.Index([1, 2]))} + self.categories_ = pd.Index(["A", "B"], name="col") + self.dtypes_ = [pd.StringDtype(), pd.Int64Dtype()] + return self + + +def test_estimator_with_pandas_attributes(): + estimator = Encoder().fit(None) + dumped = dumps(estimator) + # only the estimator itself needs to be trusted + assert get_untrusted_types(data=dumped) == [get_type_name(Encoder)] + + loaded = loads(dumped, trusted=[Encoder]) + tm.assert_series_equal(loaded.mapping_["col"], estimator.mapping_["col"]) + tm.assert_index_equal(loaded.categories_, estimator.categories_, exact=True) + assert loaded.dtypes_ == estimator.dtypes_ + + +def test_subclass_from_other_library_is_unsupported(): + class MySeries(pd.Series): + pass + + with pytest.raises(UnsupportedTypeException, match="subclass of a pandas type"): + dumps(MySeries([1, 2])) + + +def test_visualize(capsys): + visualize(dumps(FRAMES[0])) + assert "pandas.DataFrame" in capsys.readouterr().out + + +def test_file_uses_public_type_names(): + # older pandas versions report the defining module of a class, e.g. + # pandas.core.series.Series, and pandas 3 reports pandas.Series; the file + # always holds the public name, which is the one trusted by default + dumped = dumps(pd.Series([1], index=pd.Index([1]))) + with ZipFile(io.BytesIO(dumped)) as zip_file: + schema = json.loads(zip_file.read("schema.json")) + assert (schema["__module__"], schema["__class__"]) == ("pandas", "Series") + index = schema["content"]["index"] + assert (index["__module__"], index["__class__"]) == ("pandas", "Index") + + +def test_trusted_type_names_are_valid(): + # every name is the public path of a pandas class, as written by dumps, so + # that a rename in pandas does not silently leave a type untrusted + for name in PANDAS_TYPE_NAMES: + module, _, class_name = name.rpartition(".") + try: + cls = gettype(module, class_name) + except AttributeError: + # types that only exist in some pandas versions, e.g. PandasArray + continue + assert f"{_public_module(cls)}.{cls.__name__}" == name + + +FIXTURE_DIR = Path(__file__).parent / "data" + + +def cross_version_objects(): + """Objects whose files, written by one pandas version, load on every other. + + :func:`write_pandas_fixture_file` dumps them with the running pandas version + into ``data/``, and :func:`test_load_file_of_other_pandas_version` loads + every file there. Add a file for a new pandas minor version when it changes + how any of these is represented, and never regenerate an existing one. + """ + return { + "str_index": pd.Index(["a", None, "c"], name="strings"), + "str_series": pd.Series(["x", "y", None], index=["a", "b", "c"]), + "mixed_frame": pd.DataFrame( + { + "i": [1, 2], + "f": [1.5, np.nan], + "s": ["a", None], + "o": pd.Series(["a", 1], dtype=object), + "c": pd.Categorical(["x", "y"], categories=["y", "x"], ordered=True), + "t": pd.date_range("2024-01-01", periods=2, tz="Europe/Berlin"), + } + ), + "datetime_index": pd.date_range("2024-01-01", periods=3, name="dates"), + "datetime_index_tz": pd.date_range("2024-01-01", periods=3, tz="UTC"), + "datetime_fixed_offset": pd.to_datetime(["2024-01-01T00:00:00+01:00"]), + "timedelta_index": pd.timedelta_range("1D", periods=2), + "period_series": pd.Series(pd.period_range("2024-01", periods=2, freq="M")), + "interval_index": pd.interval_range(0, 3), + "categorical_index": pd.CategoricalIndex( + ["a", "b", "a"], categories=["b", "a"], ordered=True + ), + "multi_index": pd.MultiIndex.from_tuples( + [("a", 1), ("b", 2)], names=["letters", None] + ), + "range_index": pd.RangeIndex(2, 20, 3, name="range"), + "nullable_frame": pd.DataFrame( + { + "i": pd.array([1, None], dtype="Int64"), + "b": pd.array([True, None], dtype="boolean"), + "f": pd.array([1.5, None], dtype="Float64"), + } + ), + "sparse_series": pd.Series(pd.arrays.SparseArray([0, 0, 1.5])), + "duplicate_columns": pd.DataFrame([[1, 2]], columns=["a", "a"]), + # the shape of category_encoders' TargetEncoder.mapping + "encoder_mapping": {"col": pd.Series([0.49, 0.66], index=pd.Index([1, 2]))}, + "dtypes": [ + pd.Int64Dtype(), + pd.CategoricalDtype(["b", "a"], ordered=True), + pd.StringDtype(), + ], + } + + +def write_pandas_fixture_file(): + """Dump the cross-version objects with the running pandas version.""" + path = FIXTURE_DIR / f"pandas-{pd.__version__}.skops" + dump(cross_version_objects(), path) + return path + + +def _as_objects(values): + # the values as Python objects, with ``None`` for every missing value + return values.to_numpy(dtype=object, na_value=None) + + +def assert_same_data(expected, actual): + """Equality up to the dtype differences between pandas versions. + + Strings are ``str`` in pandas 3 and ``object`` before, datetimes have a + microsecond resolution in pandas 3 and nanoseconds before, and a file keeps + the dtypes of the version that wrote it. The values are therefore compared + as Python objects, with a single marker for missing values. + """ + assert type(actual) is type(expected) + if isinstance(expected, pd.DataFrame): + assert expected.shape == actual.shape + assert_same_data(expected.columns, actual.columns) + assert_same_data(expected.index, actual.index) + for i in range(expected.shape[1]): + np.testing.assert_array_equal( + _as_objects(expected.iloc[:, i]), _as_objects(actual.iloc[:, i]) + ) + elif isinstance(expected, pd.Series): + assert expected.name == actual.name + assert_same_data(expected.index, actual.index) + np.testing.assert_array_equal(_as_objects(expected), _as_objects(actual)) + elif isinstance(expected, pd.MultiIndex): + assert list(expected.names) == list(actual.names) + assert_same_data(expected.to_frame(index=False), actual.to_frame(index=False)) + elif isinstance(expected, pd.Index): + assert expected.name == actual.name + np.testing.assert_array_equal(_as_objects(expected), _as_objects(actual)) + elif isinstance(expected, dict): + assert expected.keys() == actual.keys() + for key in expected: + assert_same_data(expected[key], actual[key]) + elif isinstance(expected, list): + assert len(expected) == len(actual) + for expected_item, actual_item in zip(expected, actual): + assert_same_data(expected_item, actual_item) + else: + assert expected == actual + + +@pytest.mark.parametrize( + "path", sorted(FIXTURE_DIR.glob("pandas-*.skops")), ids=lambda path: path.stem +) +def test_load_file_of_other_pandas_version(path): + # pandas types are trusted by default, so no trusted list is needed + loaded = load(path) + assert_same_data(cross_version_objects(), loaded) + + +# category_encoders uses deprecated pandas options, which the test setup turns +# into errors +@pytest.mark.filterwarnings("ignore") +def test_category_encoders_target_encoder(): + # the report in https://github.com/skops-dev/skops/issues/450 + ce = pytest.importorskip("category_encoders") + + X = pd.DataFrame({"category": list("ABACBACCBA")}) + y = np.array([0, 1, 0, 1, 1, 0, 1, 1, 1, 0]) + encoder = ce.TargetEncoder().fit(X, y) + + dumped = dumps(encoder) + assert get_untrusted_types(data=dumped) == [ + get_type_name(ce.OrdinalEncoder), + get_type_name(ce.TargetEncoder), + ] + loaded = loads(dumped, trusted=[ce.OrdinalEncoder, ce.TargetEncoder]) + + X_new = pd.DataFrame({"category": ["A", "C", "unseen", None]}) + tm.assert_frame_equal(loaded.transform(X_new), encoder.transform(X_new)) From 0918242d618d4cd446c01be79e0d7aa7c4f793f9 Mon Sep 17 00:00:00 2001 From: adrinjalali Date: Sun, 27 Sep 2026 10:54:43 +0100 Subject: [PATCH 2/8] review --- docs/changes.rst | 7 +- docs/persistence.rst | 7 +- pixi.lock | 251 +++++++++++++++++++++++++ pyproject.toml | 2 + skops/io/_pandas.py | 136 ++++++++++++-- skops/io/tests/data/pandas-2.0.3.skops | Bin 66609 -> 66836 bytes skops/io/tests/data/pandas-3.0.3.skops | Bin 78724 -> 78951 bytes skops/io/tests/test_pandas.py | 63 +++++++ 8 files changed, 446 insertions(+), 20 deletions(-) diff --git a/docs/changes.rst b/docs/changes.rst index 899f1055..e20beae7 100644 --- a/docs/changes.rst +++ b/docs/changes.rst @@ -25,9 +25,10 @@ v0.17 trusted by default. Estimators from other libraries that keep pandas objects in their fitted attributes, such as ``category_encoders``, can now be persisted. A file written with one pandas version loads with any other from - 2.0 on, keeping the dtypes of the version that wrote it. The ``freq`` of - datetime-like indexes and the ``attrs`` of a Series or DataFrame are not - preserved. :issue:`450` and :pr:`XXX` by `Adrin Jalali`_. + 2.0 on, keeping the dtypes of the version that wrote it. Not preserved are + the ``freq`` of datetime-like indexes and arrays, the ``attrs`` and ``flags`` + of a Series or DataFrame, and the storage, python or pyarrow, of a string + dtype. :issue:`450` and :pr:`552` by `Adrin Jalali`_. - Fix a regression since v0.12.0 where saving an object whose ``__reduce__`` raises failed at dump time. ``__reduce__`` is called on every object to detect a plain constructor call, but Cython extension types with a diff --git a/docs/persistence.rst b/docs/persistence.rst index 19a6518a..2a5aecd1 100644 --- a/docs/persistence.rst +++ b/docs/persistence.rst @@ -255,9 +255,10 @@ arrays, dtypes, random generators, and ufuncs. **pandas** objects, that is extension dtypes, are supported as well with pandas 2.0 or later: they are stored as the arrays they are made of and rebuilt through the public pandas constructors, so that no pandas internals end up in the file, and a file -written with one pandas version loads with any other. The ``freq`` of -datetime-like indexes and the ``attrs`` of a ``Series`` or ``DataFrame`` are not -preserved. +written with one pandas version loads with any other. Not preserved are the +``freq`` of datetime-like indexes and arrays, the ``attrs`` and ``flags`` of a +``Series`` or ``DataFrame``, and the storage, python or pyarrow, of a string +dtype, which is an environment choice over the same values. Apart from this core, we plan to support machine learning libraries commonly used be the community. So far, we have tested the following libraries: diff --git a/pixi.lock b/pixi.lock index fbe426c5..19e2d39b 100644 --- a/pixi.lock +++ b/pixi.lock @@ -5333,11 +5333,13 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/readline-8.3-h853b02a_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/scikit-learn-1.9.0-np2py314hf09ca88_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/scipy-1.18.0-py314hf07bd8e_0.conda + - conda: https://conda.anaconda.org/conda-forge/linux-64/statsmodels-0.15.0-np2py314h8874201_2.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/tk-8.6.13-noxft_h366c992_103.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/tornado-6.5.7-py314h5bd0f2a_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/ukkonen-1.1.0-py314h9891dd4_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/unicodedata2-17.0.1-py314h5bd0f2a_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/wayland-1.25.0-hd6090a7_0.conda + - conda: https://conda.anaconda.org/conda-forge/linux-64/wrapt-2.4.1-py314hfe1a184_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/xcb-util-0.4.1-h4f16b4b_2.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/xcb-util-cursor-0.1.6-hb03c661_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/xcb-util-image-0.4.0-hb711507_2.conda @@ -5367,6 +5369,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/zstd-1.5.7-hb78ec9c_6.conda - conda: https://conda.anaconda.org/conda-forge/noarch/adwaita-icon-theme-49.0-unix_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/ca-certificates-2026.6.17-hbd8a1cb_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/category_encoders-2.11.1-pyh5ded981_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/cfgv-3.5.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/colorama-0.4.6-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/cuda-cudart_linux-64-12.9.79-h3f2d84a_0.conda @@ -5384,9 +5387,11 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-ecosystem-1-0.tar.bz2 - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-forge-1-hc364b38_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/fonttools-4.63.0-pyh7db6752_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/formulaic-1.2.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/identify-2.6.19-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/importlib-metadata-9.0.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/iniconfig-2.3.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/interface_meta-1.3.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/joblib-1.5.3-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/markdown-it-py-4.2.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/mdurl-0.1.2-pyhd8ed1ab_1.conda @@ -5394,6 +5399,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/narwhals-2.22.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/nodeenv-1.10.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/patsy-1.0.3-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/platformdirs-4.10.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/plotly-6.8.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/pluggy-1.6.0-pyhf9edf01_1.conda @@ -5412,6 +5418,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/threadpoolctl-3.6.0-pyhecae5ae_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/tomli-2.4.1-pyhcf101f3_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/typing-extensions-4.15.0-h396c80c_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/typing_extensions-4.15.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/tzdata-2025c-hc9c84f9_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/virtualenv-21.5.1-pyhcf101f3_0.conda @@ -5425,6 +5432,7 @@ environments: osx-64: - conda: https://conda.anaconda.org/conda-forge/noarch/adwaita-icon-theme-49.0-unix_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/ca-certificates-2026.6.17-hbd8a1cb_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/category_encoders-2.11.1-pyh5ded981_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/cfgv-3.5.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/colorama-0.4.6-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/cycler-0.12.1-pyhcf101f3_2.conda @@ -5440,9 +5448,11 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-ecosystem-1-0.tar.bz2 - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-forge-1-hc364b38_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/fonttools-4.63.0-pyh7db6752_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/formulaic-1.2.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/identify-2.6.19-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/importlib-metadata-9.0.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/iniconfig-2.3.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/interface_meta-1.3.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/joblib-1.5.3-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/markdown-it-py-4.2.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/mdurl-0.1.2-pyhd8ed1ab_1.conda @@ -5450,6 +5460,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/narwhals-2.22.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/nodeenv-1.10.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/patsy-1.0.3-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/platformdirs-4.10.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/plotly-6.8.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/pluggy-1.6.0-pyhf9edf01_1.conda @@ -5468,6 +5479,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/threadpoolctl-3.6.0-pyhecae5ae_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/tomli-2.4.1-pyhcf101f3_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/typing-extensions-4.15.0-h396c80c_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/typing_extensions-4.15.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/tzdata-2025c-hc9c84f9_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/virtualenv-21.5.1-pyhcf101f3_0.conda @@ -5551,10 +5563,12 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/readline-8.3-h68b038d_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/scikit-learn-1.9.0-np2py314h67cc4f9_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/scipy-1.18.0-py314h5727af0_0.conda + - conda: https://conda.anaconda.org/conda-forge/osx-64/statsmodels-0.15.0-np2py314hc91a65a_2.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/tk-8.6.13-h7142dee_3.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/tornado-6.5.7-py314h217eccc_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/ukkonen-1.1.0-py314h473ef84_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/unicodedata2-17.0.1-py314h4f144dc_0.conda + - conda: https://conda.anaconda.org/conda-forge/osx-64/wrapt-2.4.1-py314h17af80e_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/xorg-libxau-1.0.12-h8616949_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/xorg-libxdmcp-1.1.5-h8616949_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/yaml-0.2.5-h4132b18_3.conda @@ -5568,6 +5582,7 @@ environments: osx-arm64: - conda: https://conda.anaconda.org/conda-forge/noarch/adwaita-icon-theme-49.0-unix_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/ca-certificates-2026.6.17-hbd8a1cb_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/category_encoders-2.11.1-pyh5ded981_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/cfgv-3.5.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/colorama-0.4.6-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/cycler-0.12.1-pyhcf101f3_2.conda @@ -5583,9 +5598,11 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-ecosystem-1-0.tar.bz2 - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-forge-1-hc364b38_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/fonttools-4.63.0-pyh7db6752_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/formulaic-1.2.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/identify-2.6.19-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/importlib-metadata-9.0.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/iniconfig-2.3.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/interface_meta-1.3.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/joblib-1.5.3-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/markdown-it-py-4.2.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/mdurl-0.1.2-pyhd8ed1ab_1.conda @@ -5593,6 +5610,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/narwhals-2.22.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/nodeenv-1.10.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/patsy-1.0.3-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/platformdirs-4.10.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/plotly-6.8.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/pluggy-1.6.0-pyhf9edf01_1.conda @@ -5611,6 +5629,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/threadpoolctl-3.6.0-pyhecae5ae_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/tomli-2.4.1-pyhcf101f3_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/typing-extensions-4.15.0-h396c80c_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/typing_extensions-4.15.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/tzdata-2025c-hc9c84f9_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/virtualenv-21.5.1-pyhcf101f3_0.conda @@ -5694,10 +5713,12 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/readline-8.3-h46df422_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/scikit-learn-1.9.0-np2py314h15f0f0f_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/scipy-1.18.0-py314h18e1515_0.conda + - conda: https://conda.anaconda.org/conda-forge/osx-arm64/statsmodels-0.15.0-np2py314h40f2a13_2.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/tk-8.6.13-h010d191_3.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/tornado-6.5.7-py314h6c2aa35_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/ukkonen-1.1.0-py314h6cfcd04_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/unicodedata2-17.0.1-py314h6c2aa35_0.conda + - conda: https://conda.anaconda.org/conda-forge/osx-arm64/wrapt-2.4.1-py314hd98292b_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/xorg-libxau-1.0.12-hc919400_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/xorg-libxdmcp-1.1.5-hc919400_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/yaml-0.2.5-h925e9cb_3.conda @@ -5710,6 +5731,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl win-64: - conda: https://conda.anaconda.org/conda-forge/noarch/ca-certificates-2026.6.17-h4c7d964_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/category_encoders-2.11.1-pyh5ded981_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/cfgv-3.5.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/colorama-0.4.6-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/cycler-0.12.1-pyhcf101f3_2.conda @@ -5725,9 +5747,11 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-ecosystem-1-0.tar.bz2 - conda: https://conda.anaconda.org/conda-forge/noarch/fonts-conda-forge-1-hc364b38_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/fonttools-4.63.0-pyh7db6752_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/formulaic-1.2.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/identify-2.6.19-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/importlib-metadata-9.0.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/iniconfig-2.3.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/interface_meta-1.3.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/joblib-1.5.3-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/markdown-it-py-4.2.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/mdurl-0.1.2-pyhd8ed1ab_1.conda @@ -5735,6 +5759,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/narwhals-2.22.1-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/nodeenv-1.10.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/packaging-26.2-pyhc364b38_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/patsy-1.0.3-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/platformdirs-4.10.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/plotly-6.8.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/pluggy-1.6.0-pyhf9edf01_1.conda @@ -5754,6 +5779,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/threadpoolctl-3.6.0-pyhecae5ae_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/tomli-2.4.1-pyhcf101f3_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/typing-extensions-4.15.0-h396c80c_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/typing_extensions-4.15.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/tzdata-2025c-hc9c84f9_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/virtualenv-21.5.1-pyhcf101f3_0.conda @@ -5838,6 +5864,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/quantile-forest-1.4.2-py314h13fbf68_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/scikit-learn-1.9.0-np2py314h1b5b07a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/scipy-1.18.0-py314h221f224_0.conda + - conda: https://conda.anaconda.org/conda-forge/win-64/statsmodels-0.15.0-np2py314hea88fa1_2.conda - conda: https://conda.anaconda.org/conda-forge/win-64/tbb-2023.0.0-hd3d4ead_2.conda - conda: https://conda.anaconda.org/conda-forge/win-64/tk-8.6.13-h6ed50ae_3.conda - conda: https://conda.anaconda.org/conda-forge/win-64/tornado-6.5.7-py314h5a2d7ad_0.conda @@ -5847,6 +5874,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/vc-14.5-h1b7c187_39.conda - conda: https://conda.anaconda.org/conda-forge/win-64/vc14_runtime-14.51.36231-h1b9f54f_39.conda - conda: https://conda.anaconda.org/conda-forge/win-64/vcomp14-14.51.36231-h1b9f54f_39.conda + - conda: https://conda.anaconda.org/conda-forge/win-64/wrapt-2.4.1-py314hc5dbbe4_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/xorg-libice-1.1.2-h0e40799_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/xorg-libsm-1.2.6-h0e40799_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/xorg-libx11-1.8.13-hfa52320_0.conda @@ -13943,6 +13971,28 @@ packages: run_exports: {} size: 17260022 timestamp: 1781912924009 +- conda: https://conda.anaconda.org/conda-forge/linux-64/statsmodels-0.15.0-np2py314h8874201_2.conda + sha256: b31c70966a738e052eeda0bb5e6bb711c6a04bc6c75875e13b89bcaadd8c2ed5 + md5: f7ce33f5fb4b72ccfdadcd68571b7a81 + depends: + - python + - numpy >=1.23.5,<3 + - scipy >=1.8,!=1.9.2 + - pandas >=1.4,!=2.1.0 + - patsy >=0.5.6 + - packaging >=21.3 + - formulaic >=1.1.0 + - libgcc >=15 + - __glibc >=2.17,<3.0.a0 + - numpy >=1.25,<3 + - python_abi 3.14.* *_cp314 + license: BSD-3-Clause + license_family: BSD + purls: + - pkg:pypi/statsmodels?source=compressed-mapping + run_exports: {} + size: 14186732 + timestamp: 1790033647794 - conda: https://conda.anaconda.org/conda-forge/linux-64/tk-8.6.13-noxft_h1df4ec4_4.conda build_number: 104 sha256: a1a241d172c1ccab067ba245206dd048bc3c2d1b84504b53c9468e99adfc16a1 @@ -14325,6 +14375,21 @@ packages: - wayland >=1.26.0,<2.0a0 size: 340058 timestamp: 1787793849093 +- conda: https://conda.anaconda.org/conda-forge/linux-64/wrapt-2.4.1-py314hfe1a184_0.conda + sha256: 1d706a003b2bdee04784a8fac90a14e6f00c3dca91bf4137a5f6db50c358bd42 + md5: 4f79d07d0f466843e04c7a54bd63d6d7 + depends: + - python + - libgcc >=15 + - __glibc >=2.17,<3.0.a0 + - python_abi 3.14.* *_cp314 + license: BSD-2-Clause + license_family: BSD + purls: + - pkg:pypi/wrapt?source=compressed-mapping + run_exports: {} + size: 149268 + timestamp: 1789986014273 - conda: https://conda.anaconda.org/conda-forge/linux-64/xcb-util-0.4.1-h4f16b4b_2.conda sha256: ad8cab7e07e2af268449c2ce855cbb51f43f4664936eff679b1f3862e6e4b01d md5: fdc27cb255a7a2cc73b7919a968b48f0 @@ -15080,6 +15145,24 @@ packages: run_exports: {} size: 131780 timestamp: 1784754889428 +- conda: https://conda.anaconda.org/conda-forge/noarch/category_encoders-2.11.1-pyh5ded981_0.conda + sha256: 93dd5e34d287e264fbe548a375a7bb512dcc39d75fe93db4c947127094a347a9 + md5: e0b2a8381b5cb1f6c53e5756180ea258 + depends: + - numpy >=1.14.0 + - pandas >=1.0.5 + - python >=3.11 + - scikit-learn >=1.6.0 + - scipy >=1.0.0 + - statsmodels >=0.9.0 + - python + license: BSD-3-Clause + license_family: BSD + purls: + - pkg:pypi/category-encoders?source=hash-mapping + run_exports: {} + size: 123562 + timestamp: 1788904404770 - conda: https://conda.anaconda.org/conda-forge/noarch/certifi-2025.8.3-pyhd8ed1ab_0.conda sha256: a1ad5b0a2a242f439608f22a538d2175cac4444b7b3f4e2b8c090ac337aaea40 md5: 11f59985f49df4620890f3e746ed7102 @@ -15486,6 +15569,29 @@ packages: run_exports: {} size: 846038 timestamp: 1778770337113 +- conda: https://conda.anaconda.org/conda-forge/noarch/formulaic-1.2.2-pyhd8ed1ab_0.conda + sha256: a6e065128ef1f2b6b2636268409155c1013460babb619e6e8415686305e776ea + md5: f18f43fb3e20ab309d46faebfb3348b0 + depends: + - interface_meta >=1.2 + - narwhals >=1.17 + - numpy >=1.20 + - pandas >=1.3 + - python >=3.10 + - scipy >=1.6 + - typing-extensions >=4.2 + - wrapt >=1.17 + constrains: + - polars >=1 + - pyarrow >=1 + - sympy >=1.3,!=1.10 + license: MIT + license_family: MIT + purls: + - pkg:pypi/formulaic?source=hash-mapping + run_exports: {} + size: 90142 + timestamp: 1780453843138 - conda: https://conda.anaconda.org/conda-forge/noarch/h2-4.2.0-pyhd8ed1ab_0.conda sha256: 0aa1cdc67a9fe75ea95b5644b734a756200d6ec9d0dff66530aec3d1c1e9df75 md5: b4754fb1bdcb70c8fd54f918301582c6 @@ -15668,6 +15774,17 @@ packages: run_exports: {} size: 13387 timestamp: 1760831448842 +- conda: https://conda.anaconda.org/conda-forge/noarch/interface_meta-1.3.0-pyhd8ed1ab_1.conda + sha256: cdd4377e9565f383e3e69f26b43fc1abc892fda591fa724af508e498c2dc0859 + md5: 97f07ae607e01877f0454fce4c833e40 + depends: + - python >=3.9 + license: MIT + license_family: MIT + purls: + - pkg:pypi/interface-meta?source=hash-mapping + size: 18422 + timestamp: 1734284523576 - conda: https://conda.anaconda.org/conda-forge/noarch/ipython-9.14.1-pyh53cf698_0.conda sha256: a3f76e06c31bcf1bda0f633d5c9f1c834286b4f6decc6626067a6cffee283318 md5: fbd58549b374103c1a80577f09a328ef @@ -15951,6 +16068,20 @@ packages: run_exports: {} size: 82472 timestamp: 1777722955579 +- conda: https://conda.anaconda.org/conda-forge/noarch/patsy-1.0.3-pyhcf101f3_0.conda + sha256: d204290be1e2d895ebd945ffd310ff00ea6a7670fe20a92a04ade7d197cdadf5 + md5: 13e1355debaf21a0b8c2726697ebb01b + depends: + - numpy >=1.4.0 + - packaging + - python >=3.10 + - python + license: BSD-2-Clause AND PSF-2.0 + purls: + - pkg:pypi/patsy?source=hash-mapping + run_exports: {} + size: 193838 + timestamp: 1788106018957 - conda: https://conda.anaconda.org/conda-forge/noarch/pexpect-4.9.0-pyhd8ed1ab_1.conda sha256: 202af1de83b585d36445dc1fda94266697341994d1a3328fabde4989e1b3d07a md5: d0d408b1f18883a944376da5cf8101ea @@ -16967,6 +17098,17 @@ packages: run_exports: {} size: 115158 timestamp: 1780507822178 +- conda: https://conda.anaconda.org/conda-forge/noarch/typing-extensions-4.15.0-h396c80c_0.conda + sha256: 7c2df5721c742c2a47b2c8f960e718c930031663ac1174da67c1ed5999f7938c + md5: edd329d7d3a4ab45dcf905899a7a6115 + depends: + - typing_extensions ==4.15.0 pyhcf101f3_0 + license: PSF-2.0 + license_family: PSF + purls: [] + run_exports: {} + size: 91383 + timestamp: 1756220668932 - conda: https://conda.anaconda.org/conda-forge/noarch/typing_extensions-4.14.1-pyhe01879c_0.conda sha256: 4f52390e331ea8b9019b87effaebc4f80c6466d09f68453f52d5cdc2a3e1194f md5: e523f4f1e980ed7a4240d7e27e9ec81f @@ -21449,6 +21591,27 @@ packages: run_exports: {} size: 15667374 timestamp: 1781913667133 +- conda: https://conda.anaconda.org/conda-forge/osx-64/statsmodels-0.15.0-np2py314hc91a65a_2.conda + sha256: 8c115627f8aa864727e23f85a7858d9cef57ee88df4391967441318c09648d5d + md5: f2bce35ac57139912e622ceef1f20d20 + depends: + - python + - numpy >=1.23.5,<3 + - scipy >=1.8,!=1.9.2 + - pandas >=1.4,!=2.1.0 + - patsy >=0.5.6 + - packaging >=21.3 + - formulaic >=1.1.0 + - __osx >=11.0 + - python_abi 3.14.* *_cp314 + - numpy >=1.25,<3 + license: BSD-3-Clause + license_family: BSD + purls: + - pkg:pypi/statsmodels?source=compressed-mapping + run_exports: {} + size: 13777604 + timestamp: 1790033664701 - conda: https://conda.anaconda.org/conda-forge/osx-64/tk-8.6.13-h7142dee_3.conda sha256: 7f0d9c320288532873e2d8486c331ec6d87919c9028208d3f6ac91dc8f99a67b md5: 6e6efb7463f8cef69dbcb4c2205bf60e @@ -21740,6 +21903,20 @@ packages: run_exports: {} size: 406478 timestamp: 1770909238815 +- conda: https://conda.anaconda.org/conda-forge/osx-64/wrapt-2.4.1-py314h17af80e_0.conda + sha256: dda78e5462fb86a2682d00dee62927d3c791ba8e4fbc0851c3460c571fd9db78 + md5: fef84b3d0694187c6c4642a3ebf78882 + depends: + - python + - __osx >=11.0 + - python_abi 3.14.* *_cp314 + license: BSD-2-Clause + license_family: BSD + purls: + - pkg:pypi/wrapt?source=compressed-mapping + run_exports: {} + size: 144915 + timestamp: 1789986067548 - conda: https://conda.anaconda.org/conda-forge/osx-64/xorg-libxau-1.0.12-h8616949_1.conda sha256: 928f28bd278c7da674b57d71b2e7f4ac4e7c7ce56b0bf0f60d6a074366a2e76d md5: 47f1b8b4a76ebd0cd22bd7153e54a4dc @@ -26270,6 +26447,27 @@ packages: run_exports: {} size: 14122215 timestamp: 1781912992503 +- conda: https://conda.anaconda.org/conda-forge/osx-arm64/statsmodels-0.15.0-np2py314h40f2a13_2.conda + sha256: 6762596d9b868d6dfb84e62feb1cd7a29d7b6bcbc38b54afdeac2d6875879413 + md5: ad5201d6bc5426933389149c98c7f08a + depends: + - python + - numpy >=1.23.5,<3 + - scipy >=1.8,!=1.9.2 + - pandas >=1.4,!=2.1.0 + - patsy >=0.5.6 + - packaging >=21.3 + - formulaic >=1.1.0 + - __osx >=11.0 + - python_abi 3.14.* *_cp314 + - numpy >=1.25,<3 + license: BSD-3-Clause + license_family: BSD + purls: + - pkg:pypi/statsmodels?source=compressed-mapping + run_exports: {} + size: 13740298 + timestamp: 1790033597056 - conda: https://conda.anaconda.org/conda-forge/osx-arm64/tk-8.6.13-h010d191_3.conda sha256: 799cab4b6cde62f91f750149995d149bc9db525ec12595e8a1d91b9317f038b3 md5: a9d86bc62f39b94c4661716624eb21b0 @@ -26579,6 +26777,20 @@ packages: run_exports: {} size: 416130 timestamp: 1770909728445 +- conda: https://conda.anaconda.org/conda-forge/osx-arm64/wrapt-2.4.1-py314hd98292b_0.conda + sha256: 149b00f33024a5e25acc59195ecda340672ad3a17701b911a986a225ed6ade06 + md5: d36dc009656ba3dd1dcfed7890d7ac3b + depends: + - python + - __osx >=11.0 + - python_abi 3.14.* *_cp314 + license: BSD-2-Clause + license_family: BSD + purls: + - pkg:pypi/wrapt?source=compressed-mapping + run_exports: {} + size: 144486 + timestamp: 1789986025741 - conda: https://conda.anaconda.org/conda-forge/osx-arm64/xorg-libxau-1.0.12-hb001647_2.conda sha256: cde7cecabb662b435e6537591e5845e870a0648887542d3e538c9f6c20dd6be5 md5: 610f557e75d6a0d3f676914da5822227 @@ -31882,6 +32094,29 @@ packages: run_exports: {} size: 15353018 timestamp: 1781914001107 +- conda: https://conda.anaconda.org/conda-forge/win-64/statsmodels-0.15.0-np2py314hea88fa1_2.conda + sha256: 13b1e307d29164a9f5d18de89cb8e3281026b0e514463fe0b1f05a2987ec8134 + md5: 9fc91bab302b183bcf3d6fee97e0a368 + depends: + - python + - numpy >=1.23.5,<3 + - scipy >=1.8,!=1.9.2 + - pandas >=1.4,!=2.1.0 + - patsy >=0.5.6 + - packaging >=21.3 + - formulaic >=1.1.0 + - vc >=14.3,<15 + - vc14_runtime >=14.44.35208 + - ucrt >=10.0.20348.0 + - python_abi 3.14.* *_cp314 + - numpy >=1.25,<3 + license: BSD-3-Clause + license_family: BSD + purls: + - pkg:pypi/statsmodels?source=compressed-mapping + run_exports: {} + size: 13612207 + timestamp: 1790033589104 - conda: https://conda.anaconda.org/conda-forge/win-64/tbb-2021.13.0-h62715c5_1.conda sha256: 03cc5442046485b03dd1120d0f49d35a7e522930a2ab82f275e938e17b07b302 md5: 9190dd0a23d925f7602f9628b3aed511 @@ -32366,6 +32601,22 @@ packages: run_exports: {} size: 20355 timestamp: 1781320968804 +- conda: https://conda.anaconda.org/conda-forge/win-64/wrapt-2.4.1-py314hc5dbbe4_0.conda + sha256: 87917bc5e8fa8cc18432c07ac8b1d7348df4cad957d741f8c618dd8f667f6ab8 + md5: 2c0ae1b4889d3ff02546a45ac3df3b8e + depends: + - python + - vc >=14.3,<15 + - vc14_runtime >=14.44.35208 + - ucrt >=10.0.20348.0 + - python_abi 3.14.* *_cp314 + license: BSD-2-Clause + license_family: BSD + purls: + - pkg:pypi/wrapt?source=compressed-mapping + run_exports: {} + size: 142909 + timestamp: 1789986015962 - conda: https://conda.anaconda.org/conda-forge/win-64/xorg-kbproto-1.0.7-hcd874cb_1002.tar.bz2 sha256: 5b16e1ca1ecc0d2907f236bc4d8e6ecfd8417db013c862a01afb7f9d78e48c09 md5: 8d11c1dac4756ca57e78c1bfe173bba4 diff --git a/pyproject.toml b/pyproject.toml index da20f209..645b57c9 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -247,6 +247,8 @@ numpy = "~=2.5.0" scipy = "~=1.18.0" catboost = ">=1.0" quantile-forest = "~=1.4.0" +# keeps pandas objects in fitted attributes, see the test for issue #450 +category_encoders = ">=2.6" python = "~=3.14.0" # [tool.pixi.feature.sklearn17] diff --git a/skops/io/_pandas.py b/skops/io/_pandas.py index 6322a42b..4dead483 100644 --- a/skops/io/_pandas.py +++ b/skops/io/_pandas.py @@ -21,8 +21,9 @@ imported pandas, see :func:`register_if_imported`, and the nodes only import pandas when they construct an object. -Not preserved: the ``freq`` of datetime-like indexes and arrays, and the -``attrs`` and ``flags`` of a Series or DataFrame. +Not preserved: the ``freq`` of datetime-like indexes and arrays, the ``attrs`` +and ``flags`` of a Series or DataFrame, and the storage, python or pyarrow, of +a string dtype, which is an environment choice over the same values. """ from __future__ import annotations @@ -34,6 +35,8 @@ import numpy as np from ._audit import Node, get_tree +from ._general import JsonNode, ListNode +from ._numpy import NdArrayNode from ._protocol import PROTOCOL from ._trusted_types import PANDAS_TYPE_NAMES from ._utils import ( @@ -120,6 +123,7 @@ def multi_index_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any] content = { "levels": list(obj.levels), "codes": list(obj.codes), + "sortorder": obj.sortorder, "names": list(obj.names), } return _pandas_state(obj, "PandasMultiIndexNode", content, save_context) @@ -206,8 +210,10 @@ def sparse_dtype_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any class _PandasNode(Node): """Base class of the pandas nodes. - The children are the entries of ``state["content"]``, and ``_construct`` - of each subclass builds the object from the constructed children. + The children are the entries of ``state["content"]``. ``_allowed_types`` + of each subclass names every entry and the node types it may hold, which + is checked while the file is read, and ``_construct`` builds the object + from the constructed children. """ def __init__( @@ -218,17 +224,34 @@ def __init__( ) -> None: super().__init__(state, load_context, trusted) self.trusted = self._get_trusted(trusted, PANDAS_TYPE_NAMES) + allowed_types = self._allowed_types() + if set(state["content"]) != set(allowed_types): + raise ValueError( + f"Expected the entries {sorted(allowed_types)}, got" + f" {sorted(state['content'])}. This is probably due to a corrupted" + " or a malicious file." + ) self.content = { - key: get_tree(value, load_context, trusted=trusted) + key: get_tree( + value, load_context, trusted=trusted, allowed_types=allowed_types[key] + ) for key, value in state["content"].items() } self.children = dict(self.content) + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + # The node types each entry may hold, ``None`` for any: names for + # instance can be any hashable. + raise NotImplementedError + def _construct_content(self) -> dict[str, Any]: return {key: node.construct() for key, node in self.content.items()} class PandasIndexNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"values": _ARRAY_NODES, "name": None} + def _construct(self): import pandas as pd @@ -242,6 +265,14 @@ def _construct(self): class PandasRangeIndexNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return { + "start": (JsonNode,), + "stop": (JsonNode,), + "step": (JsonNode,), + "name": None, + } + def _construct(self): import pandas as pd @@ -252,16 +283,30 @@ def _construct(self): class PandasMultiIndexNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return { + "levels": (ListNode,), + "codes": (ListNode,), + "sortorder": (JsonNode,), + "names": (ListNode,), + } + def _construct(self): import pandas as pd content = self._construct_content() return pd.MultiIndex( - levels=content["levels"], codes=content["codes"], names=content["names"] + levels=content["levels"], + codes=content["codes"], + sortorder=content["sortorder"], + names=content["names"], ) class PandasSeriesNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"values": _ARRAY_NODES, "index": _INDEX_NODES, "name": None} + def _construct(self): import pandas as pd @@ -273,6 +318,9 @@ def _construct(self): class PandasDataFrameNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"columns": _INDEX_NODES, "index": _INDEX_NODES, "data": (ListNode,)} + def _construct(self): import pandas as pd @@ -288,6 +336,9 @@ def _construct(self): class PandasNumpyBackedArrayNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"values": (NdArrayNode,), "tz": (JsonNode,)} + def _construct(self): import pandas as pd @@ -306,6 +357,9 @@ def _construct(self): class PandasMaskedArrayNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"values": (NdArrayNode,), "mask": (NdArrayNode,)} + def _construct(self): content = self._construct_content() cls = gettype(self.module_name, self.class_name) @@ -313,6 +367,9 @@ def _construct(self): class PandasCategoricalNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"codes": (NdArrayNode,), "dtype": (PandasCategoricalDtypeNode,)} + def _construct(self): import pandas as pd @@ -321,6 +378,9 @@ def _construct(self): class PandasPeriodArrayNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"ordinals": (NdArrayNode,), "dtype": (PandasExtensionDtypeNode,)} + def _construct(self): import pandas as pd @@ -329,6 +389,9 @@ def _construct(self): class PandasIntervalArrayNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"left": _INDEX_NODES, "right": _INDEX_NODES, "closed": (JsonNode,)} + def _construct(self): import pandas as pd @@ -339,6 +402,9 @@ def _construct(self): class PandasExtensionArrayNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"values": (NdArrayNode,), "dtype": _DTYPE_NODES} + def _construct(self): import pandas as pd @@ -347,20 +413,38 @@ def _construct(self): class PandasExtensionDtypeNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"name": (JsonNode,)} + def _construct(self): import pandas as pd name = self._construct_content()["name"] - dtype = pd.api.types.pandas_dtype(name) - if name == "str" and not isinstance(dtype, pd.api.extensions.ExtensionDtype): - # "str" is the default string dtype of pandas 3. Older versions - # parse it as a numpy unicode dtype, and keep strings in object - # arrays instead, which is what the values are stored as. - return np.dtype(object) - return dtype + # The declared, trusted dtype class parses the name itself. + # ``pandas.api.types.pandas_dtype`` would look the name up in pandas' + # registry of extension dtypes instead, where any imported library can + # register one, and run that library's code for a name from the file. + cls = gettype(self.module_name, self.class_name) + if not issubclass(cls, pd.api.extensions.ExtensionDtype): + raise ValueError( + f"{self.module_name}.{self.class_name} is not a pandas extension" + " dtype. This is probably due to a corrupted or a malicious file." + ) + try: + return cls.construct_from_string(name) + except TypeError: + if name == "str" and cls is pd.StringDtype: + # "str" is the default string dtype of pandas 3. Older versions + # do not know it, and keep strings in object arrays instead, + # which is what the values are stored as. + return np.dtype(object) + raise class PandasCategoricalDtypeNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"categories": _INDEX_NODES + (JsonNode,), "ordered": (JsonNode,)} + def _construct(self): import pandas as pd @@ -369,11 +453,35 @@ def _construct(self): class PandasSparseDtypeNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"subtype": (JsonNode,), "fill_value": (JsonNode, NdArrayNode)} + def _construct(self): import pandas as pd content = self._construct_content() - return pd.SparseDtype(content["subtype"], content["fill_value"]) + # numpy parses the subtype, so that the name from the file is not + # looked up in pandas' registry of extension dtypes + return pd.SparseDtype(np.dtype(content["subtype"]), content["fill_value"]) + + +# The node types the values, the index and the dtype of a pandas object may be +# stored as. +_ARRAY_NODES = ( + NdArrayNode, + PandasNumpyBackedArrayNode, + PandasMaskedArrayNode, + PandasCategoricalNode, + PandasPeriodArrayNode, + PandasIntervalArrayNode, + PandasExtensionArrayNode, +) +_INDEX_NODES = (PandasIndexNode, PandasRangeIndexNode, PandasMultiIndexNode) +_DTYPE_NODES = ( + PandasExtensionDtypeNode, + PandasCategoricalDtypeNode, + PandasSparseDtypeNode, +) _registered = False diff --git a/skops/io/tests/data/pandas-2.0.3.skops b/skops/io/tests/data/pandas-2.0.3.skops index e3bce055283e0094cf5b8bc67827318acb139be6..21408da0f68585b38fba044607aada5908530fe2 100644 GIT binary patch literal 66836 zcmeG_Ta08^akDlgVQ~^-Ruqq2wbb$S9=F>z=AQ zbcU$Ax>0766oqE@Kp1c3*cbr;MpYtF4z{)j0f8UQ@gnz4__rgzJ^t)NV z(`^_1ywz=<+1h#fFU~&nz`2J{EIqmO$(z=PleO_px1GM}_EGPqTTb6}X?wgo9&D{{ zkJpFl`}+nPlOcRRd2FyVguicY-tzuiPCs+{0H2d5fGqX7%KTb7M>=r*%Cp~&b)eVo z#5-_lW*sne|IZ!v_n*)X{NSE9Kl~g#Qa?-r&Aio$cHmQ^e0CC`BsWj};<*UVn?;e} z`2)QRTa%R;kbQFL(Z6}^wQIh1-O{6P4TcxKb{*}&-Tk$vz6ktQKa}T1JL~nLJimR4 z4ipE{{F6&d!xPuN2Rg9y_b1=l}97Fm~y$8t;F3i8;_~Z~X4V zVh4cptzs@B!1LDi|C#+wCGgWbuYFxM=5C(%5;jmXgdNO~RpbL>HlK7T`_|_l`dp+V zSu0Pt)!B7~ci?KIi3jh`t9~ym;q-# z?Lhn9*Kd^bh#tg(xEa)C`XOX;oaZT$TZ!E52uc3LTm4re-m++=GKZQrdI(&0YGO^3 zx%>Y0UoS*8$?~F|@RpytIJ0zEmDh zUPQB7&4e-DepH1NPh=9B-6|3u{0Mpe!7rV9B$D;E`pvkS)UCb`mhtM;#d#ANY`*{U zmtTtz-p*24Z%sFN2!z{Qks0P^zwp9!qLvr^te0i+%5QdBzWnF^{NFb(|N9-k{ikpI z!9V=gf80SipUP7@t%Iw~AqF5>0zbU@)wS=uv9zRq-VHHovi8_;b8zPI$@bPKPo6&A zSY2J)7)&OstBu=EH`YgMyNz4S*PGkxdmBUZ{l&e}#_niq;+}164`2YwJi7+qaWeFmPYyQrh7&}G#fTUHFB^{sPwT#@r|JVN!W3=oZSDjBn5eVsA5u67K^l)I z>Voa8#&0=g{;dItOn`xe91DR-BI5CQH0hYK-vglsWNCb!w8nB{u-pjWIGU^~odNFc zj`xO6TJB}QqtUw72Q{Jzbzr~MX?2TMi*)oE+RSK(L&4Qz*8Sev+t}EDxBIQ4lXr_= zk!9UZ@A&k3xv_R|Y4@|D-zl0c)^Egx6)gG2?$bL%P^5-dDDp%PWHSrpim|)5vthgq zQJoqF0a}X9;k-g5*h9lx12xpsfvIxZB&F3bHK@~uNGXgOEa3D+Hi|%R;d$C%HBrQ5 z`&p~m>%wp>5}P9F8S*y~q=+Qz7p*3?dt&v}QJATZqR6_<@+}Cwu{GFKih>;lAVB__ z^BbO4a++y@$YadiWfj9lTbC!yP!b%bz0C7YFYmU2y`Hy@t7UI_6Y!wDUNi3-##k^N zO((oRp?9K+!EHsgVA^rWiNwCr16UL_UPU#;M*>m0RE0;(h9;{%=F9Iqm zPn3T64=6Ye`J%Ho(8$HjPe6qLnKzrAP8;h#(SGVEY(Q;wGw&!6c!O7Bs_WEC&}z&+ zPec$4ftehu$~34?TYMnV55!<%v7_)NRcICpQyEri76=m>tJE~8CzL0Fp`#7Sl$@*EA^-PP?4JNsCpeOUoHv(Nl}?dO%{K zJw(FvSkQ>oi5jW5yU+>xWu$H!|E`r4x^GxO1N>~y~s$NQ_ZrMzC}5tS@g;wO$*l896dE$U%fP@ z0~OE~r7ZUhb_e%q2()OWHLA;`iO?#vhj zH5%j2U;#qkrzJ>f8tX8k!HfkEASbagnwW)z&fMTJ9`O+CK7hdi`J;d%<-KURPJJ17 zS~Ht9azogW`KL=GSh@r&pM-tCp}Gs}~h3DhT2D^K>CK~|y7_h;ITB@wdD!8Q7 zchrcaNNb@gZArnY53{N0;IoGH(2&VmhM$Xk->)LS|K@TKf96sNS-PX^@;5gj@yc&XncNZqcQ#G01s1B_`VU zbH4TyMSMjRcUaU}%`C!tX)JtdIUQw`#;OkY7NO`AAQfNCv;eBMVF4`8-)hI2Fw@uc zRhtyNP%ci~PEYCyOgCa{e@-pI%Ukeb9CSnz!ewTS2 zssv+kLBE6cLav!CAXcx`x;+h^uEI2kP>H#dPl6J}l7M`T4VkJ1(+oj3fM4sDqL7tZ zJcwUNF8An0p*Pn}uyL~W1BzVtsgk{Zho}Gw12R;Jas^l06n$lKpPxLYv~Nu z7IBuDbfLbgc=lEd20HvP?TB*SzroKowEE%bP*Qjeebnf1J(BCIw(uU50j2C7Ag7WU zz)vMFEGuJwVjsy=uBIAXq3^fA+Aj7nznj8{n) zi_LYpaqr%EduMp-UBmIlXzS3{*8!_?IGQyJMsW6LU9G$ir?wHMV@Q(=jnLl8lZT1D!K(t9X~4+9$y%!N|%}v5Aq{*9L~Bp-V$kP{`N}0a75N%#eZ@6JRueG5JM<7m^qZ zH$8OzZXsMH3{dVr*kA!?`O*HPzX&gY(tv$Ubp}(}0D;ekp9+en1c!Y0%EX(Vt5wrO z<<_YM9_o_bwzqxh(qxENyuy1IUaIo+Iv?`2CLeb%<&V6tdZ{%^KgJ71-W@b%%w?w}&g zKkwicdH!i31u3tEAja(jj30y1kfmWSbh2pmVY)>##OdS?oGrJ#zIv=qpgX5&KJ2U> zVGU-QQF_QiB@CvHw&2btVK|1yhq+DZ{VgTLo@|8UIjMmzZAw8QyV8wfDm;kbaTMGsd|$Xsr&{;fVt;#w34Ri3x9 z3~CoR^qwjT`ir0tD1rUf1=`<*z!y93j1ZXmM|B6Odte;TVPI3zVW}QSfnuMVFTerheuc&+^gvDAZg>f1w+gig)m8=tMhmV4&6oN zj=oGTa@4+S!K~GNt$rVOzDk$z;r>;mA_O$#t8hslZh=!2{QU|qoZQ254d&zOuD~c% z_2wz6@>LtBm_Mqa^3Vb+e4PzbC|JhLIIVEnIFVI#JwPvO;^*e$l zHS(gG!AolD0zl#jRjJ2gg(jrGaR#XP$J+@|2qYOPy z#YQH;T|*WCRa*`KTU!R0|Iv+0H9WoNP|0_B!xHCtz;BPq^SYPy(D9~RbOeRe9Oh^j zlNzLi&D~xS%e%(r9$d3rQM^8&U=CcidMu=lK#J%Sr^d$c$ziZ)1w@Q2X5J2bqzqEJ zsy;w7wqI7?8yF3REw`#>5qVbPO4cfrhp9S+Q68^R01y*jRi7Z@<)p@|f$Ng_O-WBB zGK};q8ocDos~zPvJg5Q%f=Rj(zXDA`6ThdvFrVYbw^aSMf4(K^HXec$+g8H34dQS! zYSEe^an!xSN>w#gOH8M#0F6zps)ndEwN&ju`;am>-CykGDh*Dzz&)u^FQCpO4^gP8 zI+_GxbX7Hiq&^V^3Xo^bqTPkd+J($9V$Z;SM4g*~758k1Dx9hV>d6(LYD*MgYs(T# zaT&t=F085GI|-EqsZR|_QZ7O1)oizW>Kd_N)Qd|maH15IW_YXJZn5QttQq&YfCQx| zi73obX)uJ**!)P$732hKb77fp*&MwQyhk`mL~U-~%7R#m4OC+XD{DRgVb7Yf zqcUfSV3OP?I!IC=a~&~719)J;1?%NrP&39CO-qk1H zu$>xIyLlSMv`9?6m=cMp77ojSTS>SRuPG5Xg1e}eh4C3)+P!%*Ue0cvl~BVKS#II5 z?TT#IaA*uXl@TSRU|)yw#aJeU#Gx9cZFbhZ4-`-LD?R>OselT{R3-|+$DT5Qo0g38 zFZ9Ed?Iv8yt>cviYmC+1N)^Lu{>G#-fdS+=D6r8Yc#1X3j1qY)HHOus`<4JJN?Ezg z8AYPDGrB}g#}0kRD}4p_`Beubn_5Ve@KJpJLU9QJr)b0F4|5%?i-_)0w*g zgx5+b9?BFvLKjc{H-Qr9eqe-V;5`q|pbhMdQeuO1KS#BIdM|4h8)ye)6*2;dTKWWV zSev49!4jECAw@VLG7nC!^sAQ;_+p2=(ct}vmX;R^Pet^hl3CnM+avX(Vj$8?%}=4! z(%&DcNryL9?MLo#;E*3zm-;T0Vn^TiIdNyrR^+flI1Tl9-A${uZC@Ua!`GkI`%)}uzen9OiGX@4{3UXjU+>`XXdqF!H7JZlg-G6@!vcRC$>X@%_vA3#uF z^C)H!qGX&;v^!+SiYwY}Kq}R?_r=!Q9dKLv%2J8yQ5?b7Cv1T<7 z8!=(>(9an`XoDcBdWxhJdXUCs@1h-J2p?*Fm=mYF+_$h-+Rd1p4R$E7 zEwTzBldt+pom9WEsy7GL8ZA1?0Cje7_K;k$zH6!oMszff_LX5n62OMg5gNjF!4lz6 zF4)psHbCHuIWZ-Y`u-dFgtzsn(P7l8@FP?!%TuG-ggeqqEtSQbSdh{KBP${cK;Y^z zTHpX&t(*nx{G7~*W%!6eq%_jHh&03MFslxul1xh22H>`NQ*0-Lb=zexeG9)5yqCMh z;})#3zPGb6T7&x%SJ$>T_BOZh1_;1_R0ysE9NagCJrZT;gXdU*%B~n0^pF#egy`HmAj?lxLp+2y-ws zJApWOp^V>(Q1zjto9`%@N105;Pbf&3gXx5h$EcjpC!B{N7DT(R`5`K*$-@Kuz?l|ez;E3Zx=Lh zft>*ZP(xi67TvtZC}GW{0Il#>%LpK9NvEz{kY{boNmf(jXhH@tZc(I_s*2-g)a0;e z!*TcuGoxyKtQRiMXz|{nNDDHtg*$5(J+BQ4&>Jyv`tG+ z3QgbuCU-N01#1kq)R`jV)y=`q4lFD}L*ta^>!Y<@TzV&SAZUsM>Cy$~eWv3rTM^o&Z(1=gWxmZX{xjd?30kPJlMpiUlBpICXMo>HSc%8?zhT+$>~SFq3v%bv)1r(r}`yawFz zn9w+Afj7b{RIFKy!PU$K2&{wA(E)(NDp;#X5gA&$PDwBl4+0TR$jrlQ*uV} z)2qs0@K7>WOLme@HPaZoK02DvkOeRB(E$&E`T(Xij|^#t#!*d(#e!vb9y%KYE8SzJ z!p&}{=;4gQF^`z37Wtuq*&S@~1=+Ro#c@xZ{Bo zL>XWS(09W|pH3@WufeFOn>Cc{88ya0Ac!_h5y)vSca<$SGL0BGsA&1+Jcp4Ma-36$Sj})<-q!i)Z57qaA}OkQDyHmj zm47rJYGv+Q9tlDpOdxB;Qr7}h5WA;nOfBJk7vNlx*!dX>y29p4gv5BNF6u#sx5zY}Vi5LS>TmT~!DXlU3Z&-p_Q$0&HxyLj+ztJhgv!|28y0d3Rb`5r zI)g(a_j-DPcyaa6d}2y7jk#&qDSDya%z9F3a?rwe<`^E74bxuTBAC50G7#+<7!<_Q zez^?8G>kTnr6#HYAVz6i%#8w6ZHZ1JODsjPEYTz$eVxa51~M5XF2SkhZ(ZKjavpFn%Oa(G$?63~t1w022AUQ;;=@7< zJpvA3BVX8>FhIQE5pW<8C>B23a$vzDm0u<5poJbOO*wiV=|KydS%OT$RJ4=pn=nmA zDFF0~R@PIQX=7)+y}NCeA~f|EtCJ_TcP6V(!bMFZxMCU3wSrysXV!;LHcmcs?*0>3 zox1welKPzg*audw`T6^P^dkIQ{k&IwX+NJnq5k1#QTjQo-B8o<9KrP~&wg7*@J|c@ zEO3<&{NSE9Kl~hksvkX5HwHy;yIau_Ei)?QHT`W-Cuj^i|}^!Ln(NZ zASl|IITMWYAdcW4ucj1q%wijk;Ag9+UVI+ju6`&5&%Xm9$nqEuzInq}ek}9gL+?Zg z3OG+trr_Eezx%L^;HqCj2!H}OR8Nw^Pw%|;b(x9(f)T(%=iHnW#(D6q&p-4z8Nmy` zj3|IEK$Rv@@aRjoe&aL1K=nhJ_|UH)1keRI14Tj*w}p4T3n4JKQb`Ef_r89ktjhn& z5ZH@+I1l0oo;gV=uvg%41fO`T|B5We?A-`~Ki-Zb=wJW!LhizUF$DI&H;&*ZH@vzc zBlzoUCsdt^%J$LW*Z#cDrSMQE@uio&BxBc{4_#J=V3O}8GyWJa( z%62&%6}|46?cFE8c=o~j&VA&>$`dOezineOTi?6wU8irmV_M#J`{~;*?(FUFjkni! z_BJN?{XOH&*#y3yJv!cb_?7YG{5Nmn9k_F}e)-G5Z~Wmr?+u3CGRgBh z=IB80sx<%P%F5)#O>cn?to;3R-+l-`um8RG{cZD(lR#;*KT>m6TP5f1dm zTYvZwwFAKUqRiOB;YEOFoa_IL`{WY%`TK5sK{w`NIOt_0(69qnXUG`&z?kio4`tu~ z!h@erbfg#-S1*YLbVPRGde<>7k%ymq=XX90cko9T^PoLUClyUY+Cn;Vg43KygoeXT zn&u;u2o3Ij;Z{A57!G@xnAVi(3&`YF1kZCM7n$7c8j}36S4Q7Xc+0`Cp9#1Poy`Kc z>}}MUrf~P8o4?gl)nw2fl>Ky|{lp`OmTpQldGVI-ttLn=IyucgJd#f|=W@~Qrq$#c za(U&W-}q5Nvx`B_0bZ*@N+vQH%`S>|I(%M3p1sf+g} zJX-$vYhQgnMR>PNdy%Hmash<9Tu~V2XTS8hn^Y|?hvlfq5Po=Ce&H|v`G4Mi;osl; zyMO+lfAkN(^B?c!oX_PcodsmO3k*QI1a5u#(lalGBh`>osU*IPRq2b&Z7{UZm{&Hd^2%-`GG8N&dSy>}P1 zIlHqlvG>+@w)ZF7`+y9w1OD0m-r96~V{*CrNM<}N`1Q^K3aS)UwCZ2%DkWMN40}y&dmcbWDYh7rKFSOz(PG@VV zGr+z5y@QFDmVX%VXu4ta0Y^0Z?QW;v9hOi1?t$Wnm1O5|-R=Z+o+Xo0KC>;;F^J41@3^2&1(OsLcZEWuZC7()0GI zssjwL9;?vuN#8_Ig62q=@YvI+H zyPt_54uUc{Rh1dgkhX+CVs5Cx#9=4lO{>rx6s0n*(i{*aG*zh?&_Jk60%M00lc_l! zh0LVmQK-%6E(kG@o2cLn#(dCm59Mj0TFD_K(JTBDFD_F19apbw%a=!#fP_?LQaRfR zVRVNyx!R}%=p&NwJCF0RiCSl+jQtbLiom}0Y<;)Mp4*aO+Gt&j; zGvKyi1h>NrZRAj85%`C_jyhj1E|f22yDtYIkPXL>FLDUeiSvr`U(lCM*?Vo|zdH-n z7nloyNw8&IAk#u+j_8=$(f)$9e(Q@LbIfd=9T-A|MRh=WRiaU!%1d$@}Bpc|@G0^VTU^^eEfmnI8K$^$l5E*P$a z!k?#cFu6G13oCaJ)?Uz6Qq3H;XxovoWy(Axm`ldYq{75@yUeTgY=3806ig1;HSBwL zF@5i_>4V4T<5;bJdhafq0JikA?(Eb)5R$)U4oTQA!bzjeq(i4!n1+S zh3DhShHCe4BQ_Z5W59+GYPqtaRY*zUcif1yNb8_#Z3*G<5;m0_d^WIw+~pji;hz1k zHqkW9tIz#G7S6|J5>CWLlTQ45by@kD<>@JdXck_OllGd&C;&Gxi#Q>380)2Ph$J!A zC#Etvvk0XMC`up^`o$t~DM>p6;`>4m7 zIMc^&)Kqx}q;jZkiq1|wi;N#3r8zAF3%&_qaq6%lBg%iG&kXB zB2fX7$2F+D%rkf|^qR>5(#kNmZqI<{t1tr+S7O!qKT!$dNQgqFX~D`6gcL$HjW--c zF)K|JC;TFExyv+?xt}A3RsdrC07CeyPWI**ssbQ}tP)rNr=&@3Fi>m&F&^YZ7=Ho5 z4ZDUd)b5!FB>QBVg(O0gh^0l}-eB*A|A7J`y;K$y;0?{RhiR?%jYKEww7+T=I(a@) zQV)cmX%<*0-ou`6QILyqNmwvHMC-S3YDw2?SMk#?+vmG+5`wA){pxe zwQUKs1^%vh0t)h4*3?DRI|*P3>uVD?VcJ|lZ?Oxa$BkVDv<#hYrRsUcq$agJ zYapG4Elb2C3~gb5&#A3i|5Fs)Lt#nZZ4Y{Vk$}<`HU23^2$RZyk!Q3NVaDC1rdeq7 z`Sb^%=hhr(L(|AcIzzQZU6sqa&|Fmldus**J^qw-B)J}*;AXVeUcArZoAeObuA6 zT>9;>asWxhOQMX>X{f+NOO8r3lW7MaA@YND#*_W&*2J7)5it;8sRV7|F<0Gr7LU*P zz^Kpmq!w-_cZ!*&tT(xQT=XTKhXWv0!@B&tN+L2Cwd&Iywr8Ziv+uHxrg!yvOr%YEcDx1v`vbKfyv(HbbC?j>+F~~ z9?hX}N|>=Q6XVz<(??)jaz6<{vr3aoU`5ot)9JKB0QL4=Jy~f2E*nk;HP3KhEtq{b zCR%f@j<;l9Zwq0Hp#vd4J5&_c(?oK?K6SeUjKsTW_9!Tdtws=)vcHVowPJg#P9A3V z29J_5gQN$}D6WKD$qMx2h#XIW`XE`ncX!edSHuO2JOg4Ycce zxyHN>U?6I66jjsH6bC+_y;;x$KpH5l_0$GD)tHRaPX;>9P0grW(4ch3hUI`L%_-wD zp!s1{#mK5tv6+#%$3})`pesW|s7e9x4=^~UQuaIJrip54K}-lR8Nh`6lEI5e3xp;9lp(|bi3sEKS5sp#-^)xNLly=0Pv8;K$R5lLcQ@bbONOc-O)rd}Fe? zKdwphAMkKT$xUe~2{_k$IsyPqQjz)c{OYw3%(%61VIV6dCMx{8jh~%?J4VRs98eQGGN4xw zEZtj5m_512%AC}`)1kA?PlJZ21Xy8Ne0w~H9Tbtc|84fF+ zO?S4_%OH}%G)R>Z0P{5aK$1&pWr@MSiBFa$4wiNyS)LeJGRbwJW$pI-eXC0FH_-Lt z;lMvB-BG6l`>L395t+};@prt0#EmE@Dm-dr2rqLe!oZkmMi4U;3MFu~COvDIfG>C6 z85v0RPwEa)_ttXfo#mb5VBSW!R;Ctr6U{`d%bj=ph#JEaR)cGP6Ie|he$rtR1{t);=+St2Ke>;$sP=l)>KCy za$rZPIPx?tIwxg#=g3MoCl~1;7-}nCpqM0q;S97828NpAJh_)a7*S~+9C8<)KbxD$ zL2+}J(VWVn9VDe(9~^gp)QjW>$PIWtqc$M%Y*h8?yeO-7B+!J(yf4_064`y2Ol!pu^4uFIdY~YUUcJ z#NCKlEe1K zDG=!l+@+4p4k5*93@Il<`qE@^3?YkoF_kK9(Va>l!5U>-DyhaJtiv0BlPCUgr{!91dbNd}!CLt&t7s2Rg&KXKH)Z^go=|I&UQ zcfM*b5jS#^A*IFMZ-{N%?Ev0IZ)+41YENdK#-9K+^v-Zz|``e@;%b~48KG7 zJfE**eJZ#nF{!b20PiuYDPHeGn5!;Z0~S_C5Qc=rskJ$IViGM{;SxG(2R?EJp-;d| z=#8lFEsTMpnv<&YEAS9D1JA0P*;JuVxJ*kPiWV^;mPdQA4 z6{OeJBcca$1uy;O)s-;jjb7;cO#P9`lZhkq&%w{*#s*WLnD^Ja#37YyOYXj&XL5_7n0eeD+PAnD+`w*d&Uc|KsRS9oF z1(}5e(xBn~Li$|J++|8&&-|-R3B13}BpLAvhT(A7>7Lo%eNr9u(Zp$5RP}&Yvq%%b zH7r*!RNN8#inygQdekYz_IGN2tJH>+=7Vkl?**%#OP&bcBc3GUHh0v@f>eqPR8t45 zYd!!`&ziHNDrZSx((I?#EmiwbWCH7@0lJdO+^+B_vjJW~HwPQ!3+v|6Y~Uf;a7#5C zJ&v6IJldMARbWD^la-fH>}1N&G+n4k`XYr^(*z<8)bD(3KkVK|gGESU<8CJgjS$D2 z`a2?Cz&EWJ7x5W#sG{o+jd5VDJ$&&{&6rboolq6sg>H$V6roXe2qj=?Fy^Gaj0mu1 zl4U`Ws_lX_Ke9Plz5tnd zG^281fP{xGeI=M_!ra@T^OZd#Ie2l>?RR;Fw=iZrhi25OHzu$&u~xlu1slkP zRVT|AAnSZdjpk;YM1@F3vgGO;;`DvvqpmN9^I~XWMys^KMov}L;bYn=napT5H@(zI zHb6BrJ%FT!mm9$*SyAjJbVp&FKqg`HN;99cFwm6hth)ec;xKqtK%QSewrL)#rFbM$ z2nc;V4c|mcAQ76Q3Onrr-l;A0C$cjT8==H55FZY^qM>M(8b#we^_ncC7$-F56+Jpa zM!=Wb$E^knN0Br+YHYl`Ks+_khfBuSP|Fv$1}8!^)9`co;?^ibUAi;eD%~Mbi6{*o z9ecc9+=_deFHQF1t9mI9VA%%?5bLg>cp$GD1acTQh~ zEn;?wZyFjGJrf&ML^R7A*FB(3`_t`9OY1sB>l)0r24&DLpoLJ5NX~}-*J9`@;8>kG zoe{K%*(JUaW|9b9{k9yhAZS+UkQuKF&2R^vKT9_gxTe*Ul_n)FW8ud^&vwUqv&r!~ z&hmakbL>>0$8^vyoeKjXjdL~+t`0&ph?kQYX!7I|%zafRGumh30TU#lX3)fd@bN%A zP_ZcBcmyTOq;_RHD5ID|NRn}FOwd75l{*$(>9q%8X4f)6_4h}cE1fd~HoxsVGQX}s z83CId2J)3rh)|AUGoOqA%V`Yc0=edK6DBHBN1Pl}wGW^pt4Deo12t+m8g|8Nk`yZg zcEw`}#YpVdtzx?4L$i}OY?xg5Ashs46eN*8BBj`av}Om7xG{$KrqPEZ;&d;xE@JUg z^u-brD*>Rie!5JP?XcZ*=ZSkn_$lOzU*Vf%HrLF~xkjT!*D^qz9iLr{EAF>DUDz%G z(;Z6KCaJN?Glkud?{>w`3+mAs+tub>`r3N}zMK<75g2=IOe2-$oY-`bEKsW+#p&H+_2YV!{&J3%;jUgQn>IxSA-wM?ec+xIi(fHJ@fxzm&ObgHmT5q0Br2}x(xh6+qh-Ym@0o#z2FN{O? z6M-9XPzf+jgw`R5m!m;qZ|-z^V{%zk&wzmh@vt4MiqobKT%|=McMz0aHkt=)b3{&U zlb3q}AykV<(3M7p2XK_uMxu?!;cLu{OU4%}$8zCfhv|ZmMN^%2hviu3B4$;YB7^BJ zi?k@?;Kq5_I)7B$8J&nAN^|@cE*@I8pI@~Wzmyw}y6%+vq8jeUOnA1?EH5NtLH{5# zXRgObnGLY}UtRs6vjLzv8^~X)G*Q-?54v44@90B(#Yv?}3CG>IQ#%}3YqE_8ZtSgX zjdyopVG#-LoaYw$fg!?o=-UV%004F(5kTllgR%iA)65D zqbLSEB|CF7DL?llk{G6(i-V+;>!T_TkP@F8Sux;wc^VMybeB#~Ki0&NKt?WXSp>%^ zk<1>U@Txq9{S*IBl}F|tOZc^%!a-41Ws6whE|U4Q!5iePmrDjHpI$op9D6cl69Rg_ ziofF(aSN!A7+sF~VmY>GmSQj+2>AREO}0#DPrY@%W)30;75MN;Qy7|8mJ#egy6G1F0*x7bvU+|zGA zj~m}|Z>y>CRlegOXySBo(|Fi|`Yw-QH8>^z6=<5z+aJ@;+(5X@xf|v-3zey}Hyq>~ zstWz3ikdrv!ypfOdY+^zl6nK&t4&u!I+4*{zcZZuCJZ_qjrl0uydz3Ra0v{hvoON!QDieSkjJ7hm+fSc`-|^?I z_@%o(bpn40kAoK&^~{THB!Zh)KluY4!9Otsu%XYs(n2Em$z3mf@M!?WA5Ov3KtrIQ zY{NTFaY2AY&^`6br*#Sj*Kq`}#8yLa=V<-%m*MI7!zp-)A?OWYkYYXr|9Cy8U})DI z$u7LScIw$@;OY3oDR|}$1i_#UleA`0ghcS&TfY7?od@rKBS8QMA`FHq4{p5mhab@q zT=y#k0Z;(@Wi%fA{CzjRpfmAbC;}*#V6Co(;QL>A@bfx?&;2T)0J<giqhZ-Kxa_$Cp&c+2-zbp(HX1Eav70VWZo zDfp9L=Lp=XSQ5d?N5AnSod+j>gClU~BuNBm3jR$Xa3=>z1n>FEsfToP{^V~m3j7*G ziQvay`|9&Lg5g^@0=IfkBKYi=K6jJOga4uk!YV6?AZ-hO_gjPlum!iIM`9J*Y{{i)9*#-ar diff --git a/skops/io/tests/data/pandas-3.0.3.skops b/skops/io/tests/data/pandas-3.0.3.skops index 79549f1da7667cf90d5b2460fa2798cbd858eae4..92fb23568172a8416d5f303a15587a7f48470a4c 100644 GIT binary patch delta 9114 zcmeHMdvI3enfHW9F64gs5^^IK2rPu;+%JM$s$^-a&6ioRrBy>BZ3Qs|)V4^q(9w0< zI-2lwdb|}QJDp6oyVInnm8s~KC~LQtVp!`s)|Ls*R@AZf!=1Xjx}*Dh&N=V-&X;@% zjx+6F%S?d0=Q;2DywBzL{GR9Bc`oR}&gQRU zOa7U;wP{g*P92pM-dG}C6xWD0=2S0=G%zeWm|sVyayuv`t)t)P)cczJY1=i=+>c%P zC%!XlV%dQv{LZ_1>u7gw+XdgTZQ0j8_Ktspin8eie}^;pXT~{>%s1$!H%r%%eQ~9K z#6zD-6whHGM>PYr?!GuMW<~qIZJkS2GXr&c3T`BCVO`EuSJQI~L#yD5jw1ZB0yn&j zZpmx)wS8h_%e`}0#W4(MTTjV?<#^@uykmcUb&+pp-qF%oyE=L}|I7YO=86hUc4Soz zZi29AVK*C_-e{l0&6-Lr3+mzJVApvh4MpWUvj)rI>dUBeQIW67p7)#|WVaJM5C&aO z!Oisgq6%Ldr*=%srVVRu``UcB_{)6%@uJA0c;X<=>gjOt&aTXGTyyO_8!Cz)Yj;M< zUi8^Ve=~=&iW|yA@7u#|ge80X~y2ao4ci(+z zYA$I*4KsQL0{1d{t*pW~^0(Z3YUa=z-VTB>S#l#iT3&ad@KDWkc$9E>jB7tBTUsw& zw6xy)W<~dcMV4WznhJO^Y&xHKnvypyrfrwBQPL?2pUJ9i$=>-ykyIE7n>2B@jk<0q z_deLVPa^q52~96uBQ{fP$E3;CRW$uM6+vxUre*1-tlLzstZS+bvG7C8-heoZQZt!TT-_cEjU=BJtrP^ceh9j$!)ZQ?QLB*RZFGQw=EZ71gt=qC8BUo_D0&z>5 zPH%3`eodjOZky_%Ze3i`kY#|=o)gvXjsmI5OPWvS&>wqxgmHaO^`ENp27ghT;gPd@ zDrx+ZU()NB_C$AvG81Pj>G)JFu2KD?&bKA8P3WTpbsphoH8ixmSSSrk^{z_LXh$`# z76u;F4ArKMk5!UPRo>UGdO5>w1chjD%!aIAtaL}E{_88LOzVtp>DZQgwn=KD=1nie zt&FP+8OM~7`NtA8^{pr4A~bc|ionyBr6O2V?|$RNf+FEN!!d2@{jcBB;4folvY`UB zI^6LK+P|hJ5;k?jAq)R#ZCcn=4Y#FHs-bPk6Ljv<-kAnuyEd;++Z7J#zrMyBUSFRR z859A`dt_Kt*u%sN?4QFOzM;i^4OvsY{`aIDEJP_FY?&5b7(BS5HM$v7nwA$Dsqf_` z77W${{z5}nv}OE0zRZ{Nrmws%vbo3w*!AZ?u0~|F7T1bR>D{~jq(psx)0@`cK_nks zRYUTTgb011qiLFqFk?nwW!8mVJ%7}TmkpIc=mT|qbU{@I=md)6f;%YrACFV&aD+a> z#1Z;_b?YV(dT>UVLS-QYJtbvcVb53z!UM-ZpHc7L~UyrQ`A{8hoIIj(_Yj`r|c=+>civu7{pmNyQJgkUUV~ z-dBKy=7$=oS*gZC9&LN5!ksRVz#)4p-Bcx>PCeMpqBMPF15NF!r|I_=QkPYW|Cd7g zQW{-b>I;qx=hJ;ZZ=%t|rQU}Anw$(st{NFGqvl(NuTPz99T_gdPD^RS(v^601x>G0M}|vi zXloIIq`{XO->%~E#gzJ0fjA6siJ&PTXye01hI8Q(S(wpO9fd}oO^gf|vZ}s0MjHx+ zRdvVY$~OR8?P~PizhPijw)6SemYQ}g5ig6q4LLw2Y@QF92JEpidZx%e#IbEscvQNx zjZ&kXG_+HvvYYOonFuTCFtMB`1np59(*ErS8;3xnGoB> zJHU3C`rnK2D7dF?*qRAkm{``}r}O&#`QF;09kEz5Ecbk&6cuR0MmJH`x{UJnz8e%C z-R8WUt`U=Fg3z(t0=NC#V0=0D2F%@L$mm28YJ&S4m{%<4S zs_5PWyMtXgk+~IVxi!R3e}T$Z;Oe(NjbsP>f?ENfp1bX@!YwUZHFU54ON(>TdvI$K zAsFw0ELBoXl)tx{wmp9bZQpl!XeY~53_E~Dp8#1hI45B7K>$@b9R^bg%9l^C!UD$} z!mxFqh?l#+f#U=#IEA~L-suF%dSQv^2!rLIqKeZC@AyQdDhRI3+zGbz&qu7t2Qvz9 z=ApW&d*ffbQi?Y4dnDK^bK!y)Kr*OYp)db>LE3H(_fW7~gsP548v*+A8_T_C@4QbE z%)`=Y^4+Q!ZLl22PLm^DQ)C0ZkswEZ!I~Bvdm-n6oyd;qoxJDQvAZjXs);@A!5!Ic zGVf=7&HA^L1=Z<6LU=wvXH%<`BP$zA?h<;P(KplUo{SXiGb7c>EvZVdd`` z#-r_kC8OFl(f7XnJNOhqVd{#-)Co}%k_!0Fp##Ti<{)5PH5^4lXnAM8QyGg-_zm?s z?OltcO-pFJuH74aU{|gvX*TF6SP&yHpwJzym6{ob88?S^+Q2LwMWF&@w}H{F`NpnT z7ZP+RfFhP=AndxK5ophX)4fahpP?-!#CVT7La8rkjQRfZYk10#-|D4CtA)u`26QSr zvNMQo6%-up22i@55mqc38uuHgH!Sv2M|!frcA!bd3os795FxFIY=a{k%7FkTw>&l% zSz!(x`3oNMCdLw>$^jT>cogb!ZUiqda0GO39_IlQ6kYcQ|KWSNVVuP5x)4{#bYP%C zR1tSPA$^SG4phap7t{FmmF$mM&g8npEyQBK_(Jf^Bdx+2z5{luLH%7@O7c4>WL(Lc zjvlX-q#1;UtuG&`!(R2=cnCw7oC}ky6Yg6L(u#EZ_{gbh^gH#3c7OoRGfPkf*$+n}_4*H<$#g{gvkGJs)g zkAez)Izr809)oNTwh9I`XsnmP>vN~5h&RV4U}2>T7IkDOj`#VreSejYc*8FaXYguG z1=L4X2_D5lMEc?c+#l;eP@HG@3-QWoX^Q4pirdmGRTsA2ynF90yZ3JY{9YQVY^2e* z9}ek|4_FMti@jU_`3pXJ`zC_XzOuwG|Hr>M9tWaB-E~~NgLC-^rbOu0!#e`jRm1dp zCi`YY1a}e7J#{Q%VUB?r6EFVY>BmI?`vg zViOE%I4EcEE_mB?K#%6>$1KLAzgWqP(!(<7{Jt7*_lpftEevLa8QEdUes=ufrC)`? z!Gl4BbA}MIF-AB8gP6!sbk*o>yuO#eA&J`Kj}_0Xi2oY4Z}iqnC!;g`Sz-JPEX1)D zRSp9&(R2Rh%&v5X`=dj=nisM&+V;0K?7+CTAE} zXo{hFL$AEThD3~Gn5Jn8SN^A{`PI!Ku|RIn#7!Jxl|1A!&uxYg0%`PsJh>^;L@Rm0 znVlb-DG4<#i5b9ZtpX1+#`s+WpdSF9zXokT&jIDg&RX*Sy6y3ajylY2Yt{P zbU``7xW)q9J9OfsO(`gxJ?2vC`BkDcAr*agncDn(gd8@{Bo zBLZg|K%DDX78AdZG%6zc>edQ&KpK*5$DNj>4?i=qI1L!s!$3Q)~7h#`e4xR zPV+1=oE$}Y?PGsb;ZT14g^Y>{-fVF8MCAwZsvC}abXlJChjg0zllG{n@=h=;TD_x~ z%HAFm;{kgdNHD~{V7(V!br|WmeVjcK6*n?`_H+8S`lG)uma3#;(KO;?i(E*1uC8+H zN~PLRTG1^TT;b818^!A~&)klFu~aH#?9erI5UOk<5~2iT9CiGH--VG+Bi-9=@yO>W zMn2zPATlaq?x5wfIpPDt?_wSoa;-lHDHQQ);@io_$7*~|DAo*1p+`^g6#OHx!icJy zG>)2leCm>D1+n?+@_8v-+NFxZP5C>4O#;>)k#tmR3$5?LW7Y0dxm0&F*(_Zj4e>3M z@eei`f$LZjzH;AhkpxRZFN{w$uF@)f5E@)o5eDx0rPAI1zrdfd0(*R}Lu%~^i?gD@ zs)fzXw|C&L*1ot-i z_9Z_lwTi>$#TpV{KB))hY% zqme8!w`9m53$n=k1TtyW@WFhxOw66D3JT(O^SiWcFYqUEccUUDRy`Yg0e<5aYyKB* zFHE0+7r3*^_5y!E4VkgQIw4=qHC12Y55y^wjlDi`*d-} z$;^ebZ1-e*J0`tZWUyTnXbvcDHgB$1V=^e`*}QomLk9g@ma`@yBRtN*Z~+`zTBVTW zv|dz%K+({4I==LF%(UtlkusJ+C8E_@W$UzJi!`*_v9?0b`R?-W4etf&O#iDh@g?6q z_k3slo#pY-snm@hrA}{+YGWt)PJZVAJze_)E$KUBEAAbayR~3QhHC@wW5jLbn zVNUwDtBJH|!NUiVkt!k&{aMA(l(Pobb6Jf32OCedZx7H89N+kAOr-kP6` zR@H2E=cL~W)`%#Xa~%mY^&fnGDLzKaWY`VnKH=xxbRUFa)WjK{GN9 zW!%r0w*AZaX<|d6iv{QYd*I{DJmv$Vs9Q=bgU$;A*>xk-XJ(NRVp(2@>Ac93FWcFb zL-CL)M56F?_WkU-%WmEEelqD4!**j+j_vZH*y=1im}nJbpQ%rV7bXrnU4V6IrI?0e z;q<6^Oo$Jke(G>CVqLc5&ZeIZr$V-!9y2!=CJ|gv6j$&&VOMUH*(5nK=z`&eK@Q1& z?QgO^Ym9Q2HAbyR^JWdp)D2lM6hp>OI2(Btn(iuw<}0e9$;fh^sfNg_*4aoFm*o;0 zJPLkZtb~R;a;=@G_HhoWt`iu%QWu8fw~2@lvdBw1ugFN9E7f_Bp=yd~_0$~VGKfTw zR8>;pOR>=*&?HqxHry-;pA)`mlE%aO7Ydjn~K>%uknf|$#8h>SSMQ)Ly`^Yq3wufEmWUNfzUtaxY9G9bo8UzP5YsK;#|-Cx`DyFct(Uh_-M*p z?*;UW^-7)K>hHQE0`^#wr&Lv$WO8$Fh5m|w)*%FILF518pB*% z!WE)T@}X;D0r7AQj*h~BKPO)4@{}|QCF`jvgNm8&I8xb06^_Ub5qe1#R8+VC4nOu% zfCyd5gb^goP*q*Gwki9Ed0HD(h+M7b9GaW%MI#*E8}4iw&djl9Qru34j=15z5u5C@ zlPZu-b+CRy3_hL`w>TOQ!`HTW`Lwwi-W7;Hr}H~o$chC&uiwSNt&J`_w6e_DB{s9n z%goG9EbA&JWaIPsz(d$tcjNDeHgvjE2UG**P6$Its&}lYYm`wAHtv*$hyxCedcD1koI;Ov@0xczjmP;#5Ocx8d%Lu;b5kA^pE%avFHZZ@1zb!_J#?C z!0(P+81_73iyih(@1$Yn$^9?F9&Jxi-fhD~g;efoi#5#C0kvHL14X5TJp(%~z#b29 zADlna?}z@=VGn)I&)?^F97|Vvnu|*12XYyc7j^JKtVTKJ+I`g%nb!8(7jmAQ8l_TC z+ZZR0v42FH^mLDbbvKTORZmnx-&ND$ueWEK6Z5%<*%#)o*cc|)(Vb&l6}qa6plNw7H1E-%>+|8zT9E}0bd6>5 z3>1}jw1<%GX!eNT(*Rvd(qUlv8RrToz;c&Vvn!4_mXI`lXqB70Yu}}A_PIJ zwMBQ$#h};RRmR=oI2!X7YmE(X5%_KQbHa;$c4Rfi!D!>k40RQ4fH@6^}MLzyyPl_k3XsVqkQ#eq&Gy8L}b~@rjd8c|<&x z3^Z5&+FIv!S(kK0wGQ5UP5eeSPE@Xr=OH_w3e&v$ReTRS9PC026*(g#M+8(u7Olni z6{mQD>b!x4X?9h-nco_%gtU7K%=TPvJuF{ymB&mV(6FmGYjQ-LT%M$;**EICgaaEa zzOT@VJ}`p3ZsU)Rx!5|p+N}?oZiBIRZPvqA+t08~QxxEjxD^|7T)=vjQ7a z#nxTxvlq=f0ejKGkbLki>cXuVkEHwTMYE6}!*x37shMsa+_=@fPBR1%xqI=WVYfl} z?6pAHUMkeM{4dHB#;LI3Z8x&1k)?joakwf{Hd%)@znv2I_|MwL$NxCiHV)4J@{_>( zZhMHh?^Q!ZhqLD{x0ZE0LMwYD6Df1 zQo-R{iwKMcx4nb5M+$i|kHkR5UV=_?$)Sj{HjiFUSBV6iGA3?_2x=+0WV zurnBi(C(EEhG~Xu?kwXF)}|D@py?%vD9FUG1QE5w+*-mRv?6j1SPl4@PnF1qvZh}Q zDlVf|;?g20G92D>H}Z7{57ymcEB(xMo;?Hu`uFDJQH9BuaPcZBB4*kBcm$f}J&&_e z8IBZEz?le~PW&i1jD9Bq16%Xq-R=RGh@q=5hMZW`5o0k+(7U$Eh_cu?!iE`vyGnmSG-`a3xlI?D~|! zRjveUO%$DoENinJ8=t`3ehYyuj%iPGRnN66sulb@)djG!>5OMK3*jLAx92 zY$**|8OMWU0|#13_LAsXxUa7i*7wTHVzR{JKe20Kadvyyz9ETWt%F(xlp({VfJe|Xhg3YsX zp=63j<|;H3P5mRZt>Rgc&S8s{23uQ8+_%WJgq^(WG0uP5UTZYt7J(5=ttHdz&|l?Pr+ zG65zXh!c#4&e--hpYYfVyiymK*c`KBXtOj&r`f}<+hS=lG%pQb+T7;3NM@ivI3%%I zpk^e=pzOR%OBovh#*VGB_8z&(KMUZPMQ?wlAu!+Q+d5y5XCHVq9!oxUcwWv~%^%$96-IlT=PQhKz=n_dqle;`;wsOV ziaKIpd_jcKNOLBJvbFQrqdv~G)C)*O2h@26WT$mrfQk-KcIR^Bj%GQ2(p^~K3N6rr zY8tQcxSY7*4=1UnQv0WfB5t&;#y@4)!U%#Sid39)tM)1+uN|bQ%aK_b2C0=}z zrb`=mE10hoYZu=R!~;uUE3IT39@0*gcpSv7g`XY_?jYTL*T9Z)Uu|U#1=ilLFz+M$ zyK64GSJ!0n!6nW6E*9ce3kD+gT<#zDF;bEyaoiT7k$eU7LwIn0p1HMxn-GUgku_r% zx3(i4rm@VHql0!ax}f2L$k(0Q2QL2mb}>47E3k{P54>WAYfiQr^oB6Z032HM-n2CC zlX&*A&2->CT2*7@^v=%U9-_j}?9VpR{?x{9>yoQ1awJq4YQKnqMdL;P+LGSaa=0*; zZSU}Ka+Y~M1KMs1n_rINiX8vw7>2kh*>!K3eSp0Jh6msK^$?fAAW@fa(Tw_ox4|TE zqy2Q-k+4}@$whHtQU+UQX|you#f8a_{Q+(CVqZ1etvf7?sgD)0pDMQcVpW&eZKDxoo2trx9^@!;-Oaj=XVm&mXaHcwP>?!=FO<9KR@tGV+oYx1QT8_jL{ z{{ek_!(!WGl<+Y3+h;SU-cN&Rn1E(ZfChJr3252_G`Q?ZK-2H1 zpfxa^D}%Q!g(ULtb(ewdNh zq>KoB4^In_3tuoJyw7;Lq9u~^<^T;Y4-)X%=BJ@CJ^>9nv$Q5XJs=h~35iH=6Oi;^ zkd%gz1YD6c9wIb|B%s;uqoJ-b5zPq^uRNRSmyY$03D}j3@he%=n1F_{yE!|^PEHsH zT>*9oMTyyYTH<#{iJWg%Nj|Zt#wOsi6lvTbV#V}Ann?@jOI|zX`lNUm>r_TBB&QrL e6?$?5n&Nl632f~Dj@J}ru3^Ln6FK|`68`~>&tBgE diff --git a/skops/io/tests/test_pandas.py b/skops/io/tests/test_pandas.py index a2c148b9..4aed0920 100644 --- a/skops/io/tests/test_pandas.py +++ b/skops/io/tests/test_pandas.py @@ -24,6 +24,8 @@ def assert_equal(expected, actual): assert type(actual) is type(expected) + if isinstance(expected, pd.MultiIndex): + assert actual.sortorder == expected.sortorder if isinstance(expected, pd.DataFrame): tm.assert_frame_equal(expected, actual, check_freq=False) elif isinstance(expected, pd.Series): @@ -63,6 +65,7 @@ def assert_equal(expected, actual): pd.interval_range(0, 3), pd.CategoricalIndex(["a", "b", "a"], categories=["b", "a"], ordered=True), pd.MultiIndex.from_tuples([("a", 1), ("b", 2)], names=["letters", None]), + pd.MultiIndex.from_tuples([("a", 1), ("b", 2)], sortorder=0), ] SERIES = [ @@ -183,6 +186,66 @@ def test_estimator_with_pandas_attributes(): assert loaded.dtypes_ == estimator.dtypes_ +def _with_edited_schema(dumped, edit): + # ``dumped`` with ``edit`` applied to its schema, to mimic a crafted file + with ZipFile(io.BytesIO(dumped)) as zip_file: + schema = json.loads(zip_file.read("schema.json")) + files = { + name: zip_file.read(name) + for name in zip_file.namelist() + if name != "schema.json" + } + edit(schema) + buffer = io.BytesIO() + with ZipFile(buffer, "w") as zip_file: + zip_file.writestr("schema.json", json.dumps(schema)) + for name, data in files.items(): + zip_file.writestr(name, data) + return buffer.getvalue() + + +def test_dtype_node_only_builds_the_declared_class(): + # the name in the file is parsed by the declared, trusted dtype class and + # not looked up in pandas' registry of extension dtypes, where it could + # name the dtype of another library, whose code would then run + dumped = dumps(pd.Int64Dtype()) + + def edit(schema): + schema["content"]["name"]["content"] = json.dumps("period[M]") + + with pytest.raises(TypeError, match="Cannot construct"): + loads(_with_edited_schema(dumped, edit)) + + +def test_child_of_wrong_kind_is_refused(): + # the values of an Index are an array; a file holding something else there + # is refused while it is read, before anything is constructed + dumped = dumps(pd.Index([1, 2])) + + def edit(schema): + schema["content"]["values"] = { + "__class__": "str", + "__module__": "builtins", + "__loader__": "JsonNode", + "content": json.dumps("x"), + "is_json": True, + "__id__": 1, + } + + with pytest.raises(ValueError, match="Expected a node of type"): + loads(_with_edited_schema(dumped, edit)) + + +def test_missing_entry_is_refused(): + dumped = dumps(pd.Index([1, 2])) + + def edit(schema): + del schema["content"]["name"] + + with pytest.raises(ValueError, match="Expected the entries"): + loads(_with_edited_schema(dumped, edit)) + + def test_subclass_from_other_library_is_unsupported(): class MySeries(pd.Series): pass From 9c5e2d83c7a18d2507ccb321a5827cae909a520e Mon Sep 17 00:00:00 2001 From: adrinjalali Date: Sun, 27 Sep 2026 10:55:01 +0100 Subject: [PATCH 3/8] changelog --- docs/changes.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/changes.rst b/docs/changes.rst index e20beae7..9947c8e6 100644 --- a/docs/changes.rst +++ b/docs/changes.rst @@ -28,7 +28,7 @@ v0.17 2.0 on, keeping the dtypes of the version that wrote it. Not preserved are the ``freq`` of datetime-like indexes and arrays, the ``attrs`` and ``flags`` of a Series or DataFrame, and the storage, python or pyarrow, of a string - dtype. :issue:`450` and :pr:`552` by `Adrin Jalali`_. + dtype. :pr:`552` by `Adrin Jalali`_. - Fix a regression since v0.12.0 where saving an object whose ``__reduce__`` raises failed at dump time. ``__reduce__`` is called on every object to detect a plain constructor call, but Cython extension types with a From f84fbc150fb9d9895954f507119847161bff1e86 Mon Sep 17 00:00:00 2001 From: adrinjalali Date: Wed, 30 Sep 2026 20:46:46 +0100 Subject: [PATCH 4/8] reorg tests --- skops/io/tests/_utils.py | 46 +++++++++++++++++++++ skops/io/tests/test_external.py | 39 +++++++++++++++++- skops/io/tests/test_pandas.py | 73 ++------------------------------- skops/io/tests/test_persist.py | 25 ++++++++++- 4 files changed, 112 insertions(+), 71 deletions(-) diff --git a/skops/io/tests/_utils.py b/skops/io/tests/_utils.py index da3ab675..dd205223 100644 --- a/skops/io/tests/_utils.py +++ b/skops/io/tests/_utils.py @@ -14,6 +14,11 @@ from skops.io._protocol import PROTOCOL from skops.io._utils import LoadContext, SaveContext +try: + import pandas as pd +except ImportError: # pandas is optional + pd = None + # TODO: Investigate why that seems to be an issue on MacOS (only observed with # Python 3.8) ATOL = 1e-6 if sys.platform == "darwin" else 1e-7 @@ -76,9 +81,50 @@ def _assert_tuples_equal(val1, val2, path=""): _assert_vals_equal(subval1, subval2, path=f"{path}[]") +def _is_pandas_object(val): + return pd is not None and isinstance( + val, + ( + pd.DataFrame, + pd.Series, + pd.Index, + pd.api.extensions.ExtensionArray, + pd.api.extensions.ExtensionDtype, + ), + ) + + +def _assert_pandas_equal(val1, val2, path=""): + # Strict equality of pandas objects, up to what skops does not preserve: + # the freq of datetime-like indexes and arrays. + assert pd is not None # only called for pandas objects + assert type(val1) is type(val2), f"Path: type({path})" + if isinstance(val1, pd.DataFrame): + pd.testing.assert_frame_equal(val1, val2, check_freq=False, obj=path or "df") + elif isinstance(val1, pd.Series): + pd.testing.assert_series_equal( + val1, val2, check_freq=False, obj=path or "Series" + ) + elif isinstance(val1, pd.MultiIndex): + assert val1.sortorder == val2.sortorder, f"Path: {path}.sortorder" + pd.testing.assert_index_equal(val1, val2, exact=True, obj=path or "Index") + elif isinstance(val1, pd.Index): + if isinstance(val1, (pd.DatetimeIndex, pd.TimedeltaIndex)): + # assert_index_equal starts checking the freq by default in pandas 3.1 + val1 = type(val1)(val1, freq=None) + pd.testing.assert_index_equal(val1, val2, exact=True, obj=path or "Index") + elif isinstance(val1, pd.api.extensions.ExtensionArray): + pd.testing.assert_extension_array_equal(val1, val2) + else: # an extension dtype + assert val1 == val2, f"Path: {path}" + + def _assert_vals_equal(val1, val2, path=""): if isinstance(val1, type): # e.g. could be np.int64 assert val1 is val2, f"Path: {path}" + elif _is_pandas_object(val1): + # before the __getstate__ branch, which would compare pandas internals + _assert_pandas_equal(val1, val2, path=path) elif hasattr(val1, "__getstate__") and (val1.__getstate__() is not None): # This includes BaseEstimator since they implement __getstate__ and # that returns the parameters as well. diff --git a/skops/io/tests/test_external.py b/skops/io/tests/test_external.py index c608fdb2..565ce289 100644 --- a/skops/io/tests/test_external.py +++ b/skops/io/tests/test_external.py @@ -17,7 +17,8 @@ import pytest from sklearn.datasets import make_classification, make_regression -from skops.io import dumps, loads, visualize +from skops.io import dumps, get_untrusted_types, loads, visualize +from skops.io._utils import get_type_name from skops.io.tests._utils import assert_method_outputs_equal, assert_params_equal from skops.utils._fixes import make_xgboost_random_forest @@ -439,3 +440,39 @@ def test_quantile_forest(self, quantile_forest, regr_data, trusted, tree_method) assert_method_outputs_equal(estimator, loaded, X) visualize(dumped, trusted=trusted) + + +class TestCategoryEncoders: + """Tests for category_encoders, whose fitted attributes hold pandas objects""" + + @pytest.fixture(autouse=True) + def ce(self): + return pytest.importorskip("category_encoders") + + @pytest.fixture + def trusted(self, ce): + return [ce.OrdinalEncoder, ce.TargetEncoder] + + # category_encoders uses deprecated pandas options, which the test setup + # turns into errors + @pytest.mark.filterwarnings("ignore") + def test_target_encoder(self, ce, trusted): + # the report in https://github.com/skops-dev/skops/issues/450 + pd = pytest.importorskip("pandas") + X = pd.DataFrame({"category": list("ABACBACCBA")}) + y = [0, 1, 0, 1, 1, 0, 1, 1, 1, 0] + + estimator = ce.TargetEncoder() + loaded = loads(dumps(estimator), trusted=trusted) + assert_params_equal(estimator.get_params(), loaded.get_params()) + + estimator.fit(X, y) + dumped = dumps(estimator) + # the pandas objects in the fitted attributes are trusted by default + assert get_untrusted_types(data=dumped) == [get_type_name(t) for t in trusted] + loaded = loads(dumped, trusted=trusted) + assert_params_equal(estimator.__dict__, loaded.__dict__) + X_new = pd.DataFrame({"category": ["A", "C", "unseen", None]}) + assert_method_outputs_equal(estimator, loaded, X_new) + + visualize(dumped, trusted=trusted) diff --git a/skops/io/tests/test_pandas.py b/skops/io/tests/test_pandas.py index 4aed0920..315593d7 100644 --- a/skops/io/tests/test_pandas.py +++ b/skops/io/tests/test_pandas.py @@ -10,37 +10,15 @@ import numpy as np import pytest -from sklearn.base import BaseEstimator from skops.io import dump, dumps, get_untrusted_types, load, loads, visualize from skops.io._pandas import _public_module from skops.io._trusted_types import PANDAS_TYPE_NAMES -from skops.io._utils import get_type_name, gettype +from skops.io._utils import gettype from skops.io.exceptions import UnsupportedTypeException +from skops.io.tests._utils import _assert_vals_equal pd = pytest.importorskip("pandas") -tm = pytest.importorskip("pandas.testing") - - -def assert_equal(expected, actual): - assert type(actual) is type(expected) - if isinstance(expected, pd.MultiIndex): - assert actual.sortorder == expected.sortorder - if isinstance(expected, pd.DataFrame): - tm.assert_frame_equal(expected, actual, check_freq=False) - elif isinstance(expected, pd.Series): - tm.assert_series_equal(expected, actual, check_freq=False) - elif isinstance(expected, (pd.DatetimeIndex, pd.TimedeltaIndex)): - # the freq is not preserved, and assert_index_equal starts checking it - # by default in pandas 3.1 - expected = type(expected)(expected, freq=None) - tm.assert_index_equal(expected, actual, exact=True) - elif isinstance(expected, pd.Index): - tm.assert_index_equal(expected, actual, exact=True) - elif isinstance(expected, pd.api.extensions.ExtensionArray): - tm.assert_extension_array_equal(expected, actual) - else: - assert expected == actual INDEXES = [ @@ -157,35 +135,14 @@ def _id(obj): def test_roundtrip(obj): # pandas types are trusted by default, so no trusted list is needed loaded = loads(dumps(obj)) - assert_equal(obj, loaded) + # the strict pandas comparison shared with the estimator tests + _assert_vals_equal(obj, loaded) def test_pandas_types_are_trusted_by_default(): assert get_untrusted_types(data=dumps(FRAMES[1])) == [] -class Encoder(BaseEstimator): - """Mirrors the fitted attributes of category_encoders' TargetEncoder.""" - - def fit(self, X, y=None): - self.mapping_ = {"col": pd.Series([0.49, 0.66], index=pd.Index([1, 2]))} - self.categories_ = pd.Index(["A", "B"], name="col") - self.dtypes_ = [pd.StringDtype(), pd.Int64Dtype()] - return self - - -def test_estimator_with_pandas_attributes(): - estimator = Encoder().fit(None) - dumped = dumps(estimator) - # only the estimator itself needs to be trusted - assert get_untrusted_types(data=dumped) == [get_type_name(Encoder)] - - loaded = loads(dumped, trusted=[Encoder]) - tm.assert_series_equal(loaded.mapping_["col"], estimator.mapping_["col"]) - tm.assert_index_equal(loaded.categories_, estimator.categories_, exact=True) - assert loaded.dtypes_ == estimator.dtypes_ - - def _with_edited_schema(dumped, edit): # ``dumped`` with ``edit`` applied to its schema, to mimic a crafted file with ZipFile(io.BytesIO(dumped)) as zip_file: @@ -398,25 +355,3 @@ def test_load_file_of_other_pandas_version(path): # pandas types are trusted by default, so no trusted list is needed loaded = load(path) assert_same_data(cross_version_objects(), loaded) - - -# category_encoders uses deprecated pandas options, which the test setup turns -# into errors -@pytest.mark.filterwarnings("ignore") -def test_category_encoders_target_encoder(): - # the report in https://github.com/skops-dev/skops/issues/450 - ce = pytest.importorskip("category_encoders") - - X = pd.DataFrame({"category": list("ABACBACCBA")}) - y = np.array([0, 1, 0, 1, 1, 0, 1, 1, 1, 0]) - encoder = ce.TargetEncoder().fit(X, y) - - dumped = dumps(encoder) - assert get_untrusted_types(data=dumped) == [ - get_type_name(ce.OrdinalEncoder), - get_type_name(ce.TargetEncoder), - ] - loaded = loads(dumped, trusted=[ce.OrdinalEncoder, ce.TargetEncoder]) - - X_new = pd.DataFrame({"category": ["A", "C", "unseen", None]}) - tm.assert_frame_equal(loaded.transform(X_new), encoder.transform(X_new)) diff --git a/skops/io/tests/test_persist.py b/skops/io/tests/test_persist.py index 89605de3..fcef9190 100644 --- a/skops/io/tests/test_persist.py +++ b/skops/io/tests/test_persist.py @@ -75,7 +75,7 @@ SCIPY_UFUNC_TYPE_NAMES, SKLEARN_ESTIMATOR_TYPE_NAMES, ) -from skops.io._utils import LoadContext, _get_state, get_state, gettype +from skops.io._utils import LoadContext, _get_state, get_state, get_type_name, gettype from skops.io.exceptions import UnsupportedTypeException, UntrustedTypesFoundException from skops.io.tests._utils import ( assert_method_outputs_equal, @@ -1565,6 +1565,29 @@ def test_reduce_raises_falls_back_to_dict(): assert loaded_obj.x == 3 +# This class is here as opposed to inside the test because it needs to be importable. +# It mirrors the fitted attributes of category_encoders' TargetEncoder, see gh-450. +class PandasEncoder(BaseEstimator): + def fit(self, X, y=None): + import pandas as pd + + self.mapping_ = {"col": pd.Series([0.49, 0.66], index=pd.Index([1, 2]))} + self.categories_ = pd.Index(["A", "B"], name="col") + self.dtypes_ = [pd.StringDtype(), pd.Int64Dtype()] + return self + + +def test_estimator_with_pandas_attributes(): + pytest.importorskip("pandas") + estimator = PandasEncoder().fit(None) + dumped = dumps(estimator) + # the pandas objects are trusted by default, only the estimator is not + assert get_untrusted_types(data=dumped) == [get_type_name(PandasEncoder)] + + loaded = loads(dumped, trusted=[PandasEncoder]) + assert_params_equal(estimator.__dict__, loaded.__dict__) + + def test_loss_get_state_unsupported_reduce(): # loss_get_state understands the two shapes of __reduce__ output produced by # scikit-learn's loss classes, and refuses anything else. From 7e8287b3939f232b549455d3f6f7989ef87b9bc9 Mon Sep 17 00:00:00 2001 From: adrinjalali Date: Sun, 4 Oct 2026 18:33:19 +0100 Subject: [PATCH 5/8] more review --- skops/io/_pandas.py | 154 ++++++++++++++++++++++++++++++++-- skops/io/tests/test_pandas.py | 66 ++++++++++++++- 2 files changed, 210 insertions(+), 10 deletions(-) diff --git a/skops/io/_pandas.py b/skops/io/_pandas.py index 4dead483..6b8c8187 100644 --- a/skops/io/_pandas.py +++ b/skops/io/_pandas.py @@ -24,12 +24,20 @@ Not preserved: the ``freq`` of datetime-like indexes and arrays, the ``attrs`` and ``flags`` of a Series or DataFrame, and the storage, python or pyarrow, of a string dtype, which is an environment choice over the same values. + +The pandas types are trusted by default, so loading must stay within pandas' +constructors and the data from the file. In particular, no name from the file +is looked up in pandas' registry of extension dtypes, where any imported +library can register a dtype whose parser would then run, and time zones are +parsed here rather than by pandas, whose parser can open any file on disk. """ from __future__ import annotations +import re import sys import warnings +import zoneinfo from typing import Any import numpy as np @@ -69,11 +77,13 @@ def _public_module(cls: type) -> str: return get_module(cls) -# Classes that later pandas versions renamed, mapped to their current name, so -# that a file never names a class the loading version may not have. +# Classes that later pandas versions renamed, mapped to their current name and +# the pandas version that introduced it, so that a file never names a class +# the loading version may not have. An entry can go once the versions before +# the rename are no longer supported. The test suite checks each entry against +# the running pandas version. _RENAMED_CLASSES = { - # renamed in pandas 2.1 - "PandasArray": "NumpyExtensionArray", + "PandasArray": ("NumpyExtensionArray", "2.1"), } @@ -91,8 +101,12 @@ def _pandas_state( " as its pandas base class." ) + class_name = cls.__name__ + if class_name in _RENAMED_CLASSES: + class_name, _ = _RENAMED_CLASSES[class_name] + return { - "__class__": _RENAMED_CLASSES.get(cls.__name__, cls.__name__), + "__class__": class_name, "__module__": _public_module(cls), "__loader__": loader, "content": { @@ -143,13 +157,58 @@ def dataframe_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: return _pandas_state(obj, "PandasDataFrameNode", content, save_context) +_FIXED_OFFSET = re.compile(r"^UTC([+-])(\d{2}):(\d{2})$") + + +def _timezone_name(tz: Any) -> str: + # The name of a time zone as written to the file: the key of a zoneinfo + # or pytz zone, "UTC", or "UTC+01:00" for a fixed offset. These are the + # only names ``_timezone`` accepts when loading. + name = getattr(tz, "key", None) or getattr(tz, "zone", None) + if name is not None: + return name + offset = tz.utcoffset(None) + if offset is None: + raise UnsupportedTypeException( + f"The time zone {tz!r} has neither a name nor a fixed offset, so it" + " cannot be saved." + ) + seconds = int(offset.total_seconds()) + if seconds == 0: + return "UTC" + sign, seconds = ("-", -seconds) if seconds < 0 else ("+", seconds) + return f"UTC{sign}{seconds // 3600:02d}:{seconds % 3600 // 60:02d}" + + +def _timezone(name: str) -> str: + # The name of a time zone from the file, checked before pandas parses it: + # pandas' own parser would also accept "tzlocal()" and "dateutil/", + # the latter opening any file on disk. Accepted are "UTC", a fixed offset, + # and a key that zoneinfo finds in its own directories; pandas then builds + # the zone of its default implementation from the name, so that it equals + # the zone that was saved. + if name == "UTC" or _FIXED_OFFSET.match(name): + return name + try: + zoneinfo.ZoneInfo(name) + except (ValueError, KeyError, OSError) as err: + raise ValueError( + f"{name!r} is not a known time zone. This is probably due to a" + " corrupted or a malicious file." + ) from err + return name + + def numpy_backed_array_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: # NumpyExtensionArray, DatetimeArray and TimedeltaArray wrap a numpy array. # Time zone aware datetimes are stored as naive UTC values plus the zone, # since ``to_numpy`` would otherwise give an array of Timestamp objects. tz = getattr(obj, "tz", None) values = obj if tz is None else obj.tz_convert("UTC").tz_localize(None) - content = {"values": values.to_numpy(), "tz": None if tz is None else str(tz)} + content = { + "values": values.to_numpy(), + "tz": None if tz is None else _timezone_name(tz), + } return _pandas_state(obj, "PandasNumpyBackedArrayNode", content, save_context) @@ -188,12 +247,28 @@ def extension_array_get_state(obj: Any, save_context: SaveContext) -> dict[str, def extension_dtype_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: - # Extension dtypes are rebuilt from their string form, e.g. "Int64", - # "datetime64[ns, UTC]" or "period[M]". + # Extension dtypes are rebuilt from their string form, e.g. "Int64" or + # "period[M]", except those whose parser consults pandas' registry of + # extension dtypes or the file system, which are stored as parts below. content = {"name": str(obj)} return _pandas_state(obj, "PandasExtensionDtypeNode", content, save_context) +def datetime_tz_dtype_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + content = {"unit": obj.unit, "tz": _timezone_name(obj.tz)} + return _pandas_state(obj, "PandasDatetimeTZDtypeNode", content, save_context) + + +def interval_dtype_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: + # The subtype is a numpy dtype, stored by its name like the subtype of a + # sparse dtype, a tz-aware datetime dtype, or None. + subtype = obj.subtype + if isinstance(subtype, np.dtype): + subtype = str(subtype) + content = {"subtype": subtype, "closed": obj.closed} + return _pandas_state(obj, "PandasIntervalDtypeNode", content, save_context) + + def categorical_dtype_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: # The string form of a categorical dtype does not include its categories. content = {"categories": obj.categories, "ordered": obj.ordered} @@ -352,7 +427,7 @@ def _construct(self): dtype = None if values.dtype.kind in "Mm" else values.dtype array = pd.array(values, dtype=dtype) if content["tz"] is not None: - array = array.tz_localize("UTC").tz_convert(content["tz"]) + array = array.tz_localize("UTC").tz_convert(_timezone(content["tz"])) return array @@ -424,12 +499,23 @@ def _construct(self): # ``pandas.api.types.pandas_dtype`` would look the name up in pandas' # registry of extension dtypes instead, where any imported library can # register one, and run that library's code for a name from the file. + # The parsers of the remaining pandas dtypes, the masked numeric and + # boolean ones, StringDtype, PeriodDtype and ArrowDtype, compare the + # name with their own, or hand it to the offset parser or to pyarrow. cls = gettype(self.module_name, self.class_name) if not issubclass(cls, pd.api.extensions.ExtensionDtype): raise ValueError( f"{self.module_name}.{self.class_name} is not a pandas extension" " dtype. This is probably due to a corrupted or a malicious file." ) + if cls in _dtypes_stored_as_parts(): + # Their parsers consult the registry (the subtype of an interval + # or sparse dtype) or the file system (the time zone of a + # datetime dtype), so they have loaders of their own. + raise ValueError( + f"{self.class_name} is not stored by its name. This is probably" + " due to a corrupted or a malicious file." + ) try: return cls.construct_from_string(name) except TypeError: @@ -465,6 +551,50 @@ def _construct(self): return pd.SparseDtype(np.dtype(content["subtype"]), content["fill_value"]) +class PandasDatetimeTZDtypeNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"unit": (JsonNode,), "tz": (JsonNode,)} + + def _construct(self): + import pandas as pd + + content = self._construct_content() + return pd.DatetimeTZDtype(unit=content["unit"], tz=_timezone(content["tz"])) + + +class PandasIntervalDtypeNode(_PandasNode): + def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: + return {"subtype": (JsonNode, PandasDatetimeTZDtypeNode), "closed": (JsonNode,)} + + def _construct(self): + import pandas as pd + + content = self._construct_content() + subtype = content["subtype"] + # A name is parsed by numpy, so that it is not looked up in pandas' + # registry of extension dtypes; otherwise the subtype is a tz-aware + # datetime dtype, built by its own node, or None. + message = ( + "The subtype of an interval dtype must be a numpy dtype, a datetime" + " dtype or None. This is probably due to a corrupted or a malicious" + " file." + ) + if isinstance(subtype, str): + try: + subtype = np.dtype(subtype) + except TypeError as err: + raise ValueError(message) from err + elif subtype is not None and not isinstance(subtype, pd.DatetimeTZDtype): + raise ValueError(message) + return pd.IntervalDtype(subtype, closed=content["closed"]) + + +def _dtypes_stored_as_parts(): + import pandas as pd + + return (pd.CategoricalDtype, pd.SparseDtype, pd.DatetimeTZDtype, pd.IntervalDtype) + + # The node types the values, the index and the dtype of a pandas object may be # stored as. _ARRAY_NODES = ( @@ -481,6 +611,8 @@ def _construct(self): PandasExtensionDtypeNode, PandasCategoricalDtypeNode, PandasSparseDtypeNode, + PandasDatetimeTZDtypeNode, + PandasIntervalDtypeNode, ) @@ -518,6 +650,8 @@ def register_if_imported() -> None: (pd.api.extensions.ExtensionDtype, extension_dtype_get_state), (pd.CategoricalDtype, categorical_dtype_get_state), (pd.SparseDtype, sparse_dtype_get_state), + (pd.DatetimeTZDtype, datetime_tz_dtype_get_state), + (pd.IntervalDtype, interval_dtype_get_state), ] # pandas.arrays.PandasArray was renamed to NumpyExtensionArray in pandas 2.1 numpy_backed = getattr(pd.arrays, "NumpyExtensionArray", None) @@ -548,4 +682,6 @@ def register_if_imported() -> None: ("PandasExtensionDtypeNode", PROTOCOL): PandasExtensionDtypeNode, ("PandasCategoricalDtypeNode", PROTOCOL): PandasCategoricalDtypeNode, ("PandasSparseDtypeNode", PROTOCOL): PandasSparseDtypeNode, + ("PandasDatetimeTZDtypeNode", PROTOCOL): PandasDatetimeTZDtypeNode, + ("PandasIntervalDtypeNode", PROTOCOL): PandasIntervalDtypeNode, } diff --git a/skops/io/tests/test_pandas.py b/skops/io/tests/test_pandas.py index 315593d7..e80d0a57 100644 --- a/skops/io/tests/test_pandas.py +++ b/skops/io/tests/test_pandas.py @@ -5,14 +5,16 @@ import datetime as dt import io import json +import warnings from pathlib import Path from zipfile import ZipFile import numpy as np import pytest +from packaging.version import Version from skops.io import dump, dumps, get_untrusted_types, load, loads, visualize -from skops.io._pandas import _public_module +from skops.io._pandas import _RENAMED_CLASSES, _public_module from skops.io._trusted_types import PANDAS_TYPE_NAMES from skops.io._utils import gettype from skops.io.exceptions import UnsupportedTypeException @@ -121,8 +123,11 @@ pd.CategoricalDtype(["b", "a"], ordered=True), pd.CategoricalDtype(), pd.DatetimeTZDtype("ns", "UTC"), + pd.DatetimeTZDtype("us", "Europe/Berlin"), pd.PeriodDtype("M"), pd.IntervalDtype("int64", closed="left"), + pd.IntervalDtype("datetime64[ns]"), + pd.IntervalDtype(), pd.SparseDtype(float, 0.0), ] @@ -174,6 +179,49 @@ def edit(schema): loads(_with_edited_schema(dumped, edit)) +def test_dtype_node_refuses_dtypes_stored_as_parts(): + # the parsers of these dtypes consult pandas' registry of extension dtypes + # or the file system, so a file cannot route them through the name based + # loader + dumped = dumps(pd.Int64Dtype()) + + def edit(schema): + schema["__class__"] = "DatetimeTZDtype" + schema["content"]["name"]["content"] = json.dumps( + "datetime64[ns, dateutil//etc/localtime]" + ) + + with pytest.raises(ValueError, match="not stored by its name"): + loads(_with_edited_schema(dumped, edit)) + + +@pytest.mark.parametrize( + "name", ["dateutil//etc/localtime", "tzlocal()", "../../etc/localtime", "/x"] +) +def test_timezone_from_file_is_restricted(name): + # pandas' own parser would open any file on disk for "dateutil/"; + # only "UTC", fixed offsets and valid zoneinfo keys are accepted + dumped = dumps(pd.date_range("2024-01-01", periods=2, tz="UTC")) + + def edit(schema): + schema["content"]["values"]["content"]["tz"]["content"] = json.dumps(name) + + with pytest.raises(ValueError, match="not a known time zone"): + loads(_with_edited_schema(dumped, edit)) + + +def test_interval_dtype_subtype_is_parsed_by_numpy(): + # pandas would look a subtype name up in its registry of extension dtypes, + # where any imported library can register a dtype whose parser then runs + dumped = dumps(pd.IntervalDtype("int64")) + + def edit(schema): + schema["content"]["subtype"]["content"] = json.dumps("probe") + + with pytest.raises(ValueError, match="subtype of an interval dtype"): + loads(_with_edited_schema(dumped, edit)) + + def test_child_of_wrong_kind_is_refused(): # the values of an Index are an array; a file holding something else there # is refused while it is read, before anything is constructed @@ -228,6 +276,22 @@ def test_file_uses_public_type_names(): assert (index["__module__"], index["__class__"]) == ("pandas", "Index") +def test_renamed_classes_match_pandas_version(): + # each rename is a fact about pandas: before the version that introduced + # the new name only the old one exists, from then on the new one does + def exists(name): + with warnings.catch_warnings(): + # the old name may live on as a deprecated alias that warns + warnings.simplefilter("ignore") + return any(hasattr(module, name) for module in (pd, pd.arrays)) + + for old, (new, since) in _RENAMED_CLASSES.items(): + if Version(pd.__version__) < Version(since): + assert exists(old) and not exists(new), old + else: + assert exists(new), new + + def test_trusted_type_names_are_valid(): # every name is the public path of a pandas class, as written by dumps, so # that a rename in pandas does not silently leave a type untrusted From 0b3a13af3675e1db4e03b023da84e62cc1667633 Mon Sep 17 00:00:00 2001 From: adrinjalali Date: Sun, 4 Oct 2026 19:32:00 +0100 Subject: [PATCH 6/8] review --- .github/workflows/build-test.yml | 10 + pixi.lock | 366 ++++++++++++++++++++++--------- pyproject.toml | 21 +- skops/io/_pandas.py | 48 ++-- skops/io/tests/_utils.py | 2 +- skops/io/tests/test_pandas.py | 47 +++- 6 files changed, 366 insertions(+), 128 deletions(-) diff --git a/.github/workflows/build-test.yml b/.github/workflows/build-test.yml index 05c19d22..89df29a3 100644 --- a/.github/workflows/build-test.yml +++ b/.github/workflows/build-test.yml @@ -52,6 +52,16 @@ jobs: # we can freeze the environment and manually bump the dependencies to the # latest version time to time. frozen: true + # the nightly environment is installed below, after refreshing the + # locked pandas nightly wheel, which the index drops after a few days + run-install: ${{ matrix.environment != 'ci-sklearn-nightly' }} + + - name: Install the nightly environment with the latest pandas nightly + if: matrix.environment == 'ci-sklearn-nightly' + run: | + pixi update --environment ci-sklearn-nightly pandas + pixi install --environment ci-sklearn-nightly --frozen + shell: bash - name: Linters run: pixi run -e lint lint diff --git a/pixi.lock b/pixi.lock index 19e2d39b..9b158e2c 100644 --- a/pixi.lock +++ b/pixi.lock @@ -209,7 +209,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/99/c7/bd05c5c430feb347aa040fcc8870135d70b256718deee9bc7d2ca74a77ff/xgboost-3.4.1-py3-none-manylinux_2_28_x86_64.whl - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl osx-64: @@ -334,7 +334,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/cd/05/7213965863cba1ed0150ad045bceed6276a1afaaaedbaeff4699ec4f0ccb/lightgbm-4.7.0-py3-none-macosx_10_15_x86_64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_10_15_x86_64.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_10_15_x86_64.whl osx-arm64: @@ -459,7 +459,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/f7/94/e5c37a8972ad780edc1d8459d1931356344ca133f7f99ba9cfda516b5bba/xgboost-3.4.1-py3-none-macosx_12_0_arm64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_11_0_arm64.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-macosx_11_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_12_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_12_0_arm64.whl win-64: @@ -591,7 +591,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/d5/0b/c5c17d862b12ce292f24cd85d40f2f8f8981668fbdbd43fdc2625eccbc79/lightgbm-4.7.0-py3-none-win_amd64.whl - pypi: https://files.pythonhosted.org/packages/f9/bc/8737e8d54cf51106118039b83f485a4783112fab49ea9d044b234978a46e/tzdata-2026.4-py2.py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-win_amd64.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-win_amd64.whl ci-sklearn12: @@ -5987,6 +5987,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/openjpeg-2.5.4-h55fea9a_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/openldap-2.6.13-hbde042b_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/openssl-3.6.3-h35e630c_0.conda + - conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-3.0.6-np2py314h9e1d7c2_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pandoc-3.11-ha770c72_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pcre2-10.47-haa7fec5_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/pillow-12.2.0-py314h8ec4b1a_0.conda @@ -6128,7 +6129,6 @@ environments: - pypi: https://files.pythonhosted.org/packages/7b/91/984aca2ec129e2757d1e4e3c81c3fcda9d0f85b74670a094cc443d9ee949/joblib-1.5.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl osx-64: @@ -6276,6 +6276,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/numpy-2.5.0-py314h7b24d9b_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/openjpeg-2.5.4-h52bb76a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/openssl-3.6.3-hc881268_0.conda + - conda: https://conda.anaconda.org/conda-forge/osx-64/pandas-3.0.6-np2py314h1f26284_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pandoc-3.11-h694c41f_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pcre2-10.47-h13923f0_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/pillow-12.2.0-py314hc904d5e_0.conda @@ -6303,7 +6304,6 @@ environments: - pypi: https://files.pythonhosted.org/packages/f2/75/cffc9962cca296bc5536896b7e65b4a7cdeb8db208e71b9c0133c08f8f7e/lightgbm-4.6.0-py3-none-macosx_10_15_x86_64.whl - pypi: https://files.pythonhosted.org/packages/fc/72/3b68983c0215ef65d48e9eeb1f168c3c6e3d62a61ece605de3209c79cae1/xgboost-3.3.0-py3-none-macosx_10_15_x86_64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_10_15_x86_64.whl osx-arm64: @@ -6451,6 +6451,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/numpy-2.5.0-py314hb79c6fa_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openjpeg-2.5.4-hd9e9057_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/openssl-3.6.3-hd24854e_0.conda + - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-3.0.6-np2py314hc3b1fbe_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandoc-3.11-hce30654_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pcre2-10.47-h30297fc_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pillow-12.2.0-py314hab283cf_0.conda @@ -6478,7 +6479,6 @@ environments: - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/c9/62/b49e756822b29909d0c95ed334662dc6c7c81a99ec6bc10dc18e69f3d6e7/xgboost-3.3.0-py3-none-macosx_12_0_arm64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_11_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_12_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_12_0_arm64.whl win-64: @@ -6539,6 +6539,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/pytest-cov-7.1.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-discovery-1.4.2-pyhcf101f3_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.5-pyh5ded981_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.14-8_cp314.conda - conda: https://conda.anaconda.org/conda-forge/noarch/requests-2.34.2-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/rich-15.0.0-pyhcf101f3_0.conda @@ -6632,6 +6633,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/onemkl-license-2026.0.0-h57928b3_908.conda - conda: https://conda.anaconda.org/conda-forge/win-64/openjpeg-2.5.4-h0e57b4f_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/openssl-3.6.3-hf411b9b_0.conda + - conda: https://conda.anaconda.org/conda-forge/win-64/pandas-3.0.6-np2py314h1b5b07a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pandoc-3.11-h57928b3_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pcre2-10.47-hd2b5f0e_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/pillow-12.2.0-py314h61b30b5_0.conda @@ -6665,9 +6667,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/5e/23/f8b28ca248bb629b9e08f877dd2965d1994e1674a03d67cd10c5246da248/lightgbm-4.6.0-py3-none-win_amd64.whl - pypi: https://files.pythonhosted.org/packages/7b/91/984aca2ec129e2757d1e4e3c81c3fcda9d0f85b74670a094cc443d9ee949/joblib-1.5.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/99/bd/776182bf82b0a3223773a43b0e7b9510afe6c3d32ed8dd5a27b30a89530b/fairlearn-0.14.0-py3-none-any.whl - - pypi: https://files.pythonhosted.org/packages/f9/bc/8737e8d54cf51106118039b83f485a4783112fab49ea9d044b234978a46e/tzdata-2026.4-py2.py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-win_amd64.whl docs: @@ -12439,6 +12439,64 @@ packages: run_exports: {} size: 15303815 timestamp: 1778602611222 +- conda: https://conda.anaconda.org/conda-forge/linux-64/pandas-3.0.6-np2py314h9e1d7c2_0.conda + sha256: 68a1803d2f72f6911f2cf2a9ea0eae483fc494a09d9feb8da81f03757795c4da + md5: e0f55ab2a87a7f0483a633ac39557098 + depends: + - python + - numpy >=1.26.0 + - python-dateutil >=2.8.2 + - __glibc >=2.17,<3.0.a0 + - libstdcxx >=15 + - libgcc >=15 + - numpy >=1.25,<3 + - python_abi 3.14.* *_cp314 + constrains: + - adbc-driver-postgresql >=1.2.0 + - adbc-driver-sqlite >=1.2.0 + - beautifulsoup4 >=4.12.3 + - blosc >=1.21.3 + - bottleneck >=1.4.2 + - fastparquet >=2024.11.0 + - fsspec >=2024.10.0 + - gcsfs >=2024.10.0 + - html5lib >=1.1 + - hypothesis >=6.116.0 + - jinja2 >=3.1.5 + - lxml >=5.3.0 + - matplotlib >=3.9.3 + - numba >=0.60.0 + - numexpr >=2.10.2 + - odfpy >=1.4.1 + - openpyxl >=3.1.5 + - psycopg2 >=2.9.10 + - pyarrow >=13.0.0 + - pyiceberg >=0.8.1 + - pymysql >=1.1.1 + - pyqt5 >=5.15.9 + - pyreadstat >=1.2.8 + - pytables >=3.10.1 + - pytest >=8.3.4 + - pytest-xdist >=3.6.1 + - python-calamine >=0.3.0 + - pytz >=2024.2 + - pyxlsb >=1.0.10 + - qtpy >=2.4.2 + - scipy >=1.14.1 + - s3fs >=2024.10.0 + - sqlalchemy >=2.0.36 + - tabulate >=0.9.0 + - xarray >=2024.10.0 + - xlrd >=2.0.1 + - xlsxwriter >=3.2.0 + - zstandard >=0.23.0 + license: BSD-3-Clause + license_family: BSD + purls: + - pkg:pypi/pandas?source=compressed-mapping + run_exports: {} + size: 15425802 + timestamp: 1789803361338 - conda: https://conda.anaconda.org/conda-forge/linux-64/pandoc-3.11-ha770c72_0.conda sha256: cb14839e33fbbb2163f259476091dd99f9e83fbb1690b789cd1e6e6a05d17db7 md5: 2c71f77db75fa5e9dd64abb706c66788 @@ -16564,6 +16622,19 @@ packages: run_exports: {} size: 146639 timestamp: 1777068997932 +- conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2026.5-pyh5ded981_0.conda + sha256: b66b783be882069a9f6c00bc44ee561fc3dfba21fe4bab6de31599caf3c63154 + md5: 4a0ae2a06b469e6f8ca95d0655e31c86 + depends: + - python >=3.11 + - python + license: Apache-2.0 + license_family: APACHE + purls: + - pkg:pypi/tzdata?source=compressed-mapping + run_exports: {} + size: 141379 + timestamp: 1791025923003 - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.10-8_cp310.conda build_number: 8 sha256: 7ad76fa396e4bde336872350124c0819032a9e8a0a40590744ff9527b54351c1 @@ -20567,6 +20638,63 @@ packages: run_exports: {} size: 14597208 timestamp: 1778602856255 +- conda: https://conda.anaconda.org/conda-forge/osx-64/pandas-3.0.6-np2py314h1f26284_0.conda + sha256: a6bb85c11ead049fb94e9e6cc8d56f6009ec18e57479ea39e8c73fdaedcac21d + md5: 7fa1ed5a832fcd262d8b812a8c988540 + depends: + - python + - numpy >=1.26.0 + - python-dateutil >=2.8.2 + - libcxx >=21 + - __osx >=11.0 + - python_abi 3.14.* *_cp314 + - numpy >=1.25,<3 + constrains: + - adbc-driver-postgresql >=1.2.0 + - adbc-driver-sqlite >=1.2.0 + - beautifulsoup4 >=4.12.3 + - blosc >=1.21.3 + - bottleneck >=1.4.2 + - fastparquet >=2024.11.0 + - fsspec >=2024.10.0 + - gcsfs >=2024.10.0 + - html5lib >=1.1 + - hypothesis >=6.116.0 + - jinja2 >=3.1.5 + - lxml >=5.3.0 + - matplotlib >=3.9.3 + - numba >=0.60.0 + - numexpr >=2.10.2 + - odfpy >=1.4.1 + - openpyxl >=3.1.5 + - psycopg2 >=2.9.10 + - pyarrow >=13.0.0 + - pyiceberg >=0.8.1 + - pymysql >=1.1.1 + - pyqt5 >=5.15.9 + - pyreadstat >=1.2.8 + - pytables >=3.10.1 + - pytest >=8.3.4 + - pytest-xdist >=3.6.1 + - python-calamine >=0.3.0 + - pytz >=2024.2 + - pyxlsb >=1.0.10 + - qtpy >=2.4.2 + - scipy >=1.14.1 + - s3fs >=2024.10.0 + - sqlalchemy >=2.0.36 + - tabulate >=0.9.0 + - xarray >=2024.10.0 + - xlrd >=2.0.1 + - xlsxwriter >=3.2.0 + - zstandard >=0.23.0 + license: BSD-3-Clause + license_family: BSD + purls: + - pkg:pypi/pandas?source=compressed-mapping + run_exports: {} + size: 14753057 + timestamp: 1789803514551 - conda: https://conda.anaconda.org/conda-forge/osx-64/pandoc-3.11-h694c41f_0.conda sha256: 0ef2d60b6d9ae28d4e70fae3fbb716e751db77743b6f5d18dc298fe13e280827 md5: 932c47502f555da85dcf6f562b3e36a5 @@ -25393,6 +25521,63 @@ packages: run_exports: {} size: 14368928 timestamp: 1778602917992 +- conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandas-3.0.6-np2py314hc3b1fbe_0.conda + sha256: 3e47df60710e7a8f03357d707d18b2e614c5e01ef9e6f79e40e15842a850336f + md5: 4364f79cbf78bb7d745f1ebc9c72bf64 + depends: + - python + - numpy >=1.26.0 + - python-dateutil >=2.8.2 + - __osx >=11.0 + - libcxx >=21 + - python_abi 3.14.* *_cp314 + - numpy >=1.25,<3 + constrains: + - adbc-driver-postgresql >=1.2.0 + - adbc-driver-sqlite >=1.2.0 + - beautifulsoup4 >=4.12.3 + - blosc >=1.21.3 + - bottleneck >=1.4.2 + - fastparquet >=2024.11.0 + - fsspec >=2024.10.0 + - gcsfs >=2024.10.0 + - html5lib >=1.1 + - hypothesis >=6.116.0 + - jinja2 >=3.1.5 + - lxml >=5.3.0 + - matplotlib >=3.9.3 + - numba >=0.60.0 + - numexpr >=2.10.2 + - odfpy >=1.4.1 + - openpyxl >=3.1.5 + - psycopg2 >=2.9.10 + - pyarrow >=13.0.0 + - pyiceberg >=0.8.1 + - pymysql >=1.1.1 + - pyqt5 >=5.15.9 + - pyreadstat >=1.2.8 + - pytables >=3.10.1 + - pytest >=8.3.4 + - pytest-xdist >=3.6.1 + - python-calamine >=0.3.0 + - pytz >=2024.2 + - pyxlsb >=1.0.10 + - qtpy >=2.4.2 + - scipy >=1.14.1 + - s3fs >=2024.10.0 + - sqlalchemy >=2.0.36 + - tabulate >=0.9.0 + - xarray >=2024.10.0 + - xlrd >=2.0.1 + - xlsxwriter >=3.2.0 + - zstandard >=0.23.0 + license: BSD-3-Clause + license_family: BSD + purls: + - pkg:pypi/pandas?source=compressed-mapping + run_exports: {} + size: 14475966 + timestamp: 1789803428830 - conda: https://conda.anaconda.org/conda-forge/osx-arm64/pandoc-3.11-hce30654_0.conda sha256: 1520d26e93867d62aa105d6f50e279c446d4bcd79f8db8d5b6a8442b148e8389 md5: a2f9df29541bc9c9ec08950fcedfea84 @@ -30735,6 +30920,65 @@ packages: run_exports: {} size: 14062915 timestamp: 1778602665890 +- conda: https://conda.anaconda.org/conda-forge/win-64/pandas-3.0.6-np2py314h1b5b07a_0.conda + sha256: 928402fb898126ef23d79baafa389b6e967653bb58a72e7c5d8aacb0c37051bd + md5: 2dabbff823b8623a67918f18b75d0e0a + depends: + - python + - numpy >=1.26.0 + - python-dateutil >=2.8.2 + - python-tzdata + - vc >=14.3,<15 + - vc14_runtime >=14.44.35208 + - ucrt >=10.0.20348.0 + - python_abi 3.14.* *_cp314 + - numpy >=1.25,<3 + constrains: + - adbc-driver-postgresql >=1.2.0 + - adbc-driver-sqlite >=1.2.0 + - beautifulsoup4 >=4.12.3 + - blosc >=1.21.3 + - bottleneck >=1.4.2 + - fastparquet >=2024.11.0 + - fsspec >=2024.10.0 + - gcsfs >=2024.10.0 + - html5lib >=1.1 + - hypothesis >=6.116.0 + - jinja2 >=3.1.5 + - lxml >=5.3.0 + - matplotlib >=3.9.3 + - numba >=0.60.0 + - numexpr >=2.10.2 + - odfpy >=1.4.1 + - openpyxl >=3.1.5 + - psycopg2 >=2.9.10 + - pyarrow >=13.0.0 + - pyiceberg >=0.8.1 + - pymysql >=1.1.1 + - pyqt5 >=5.15.9 + - pyreadstat >=1.2.8 + - pytables >=3.10.1 + - pytest >=8.3.4 + - pytest-xdist >=3.6.1 + - python-calamine >=0.3.0 + - pytz >=2024.2 + - pyxlsb >=1.0.10 + - qtpy >=2.4.2 + - scipy >=1.14.1 + - s3fs >=2024.10.0 + - sqlalchemy >=2.0.36 + - tabulate >=0.9.0 + - xarray >=2024.10.0 + - xlrd >=2.0.1 + - xlsxwriter >=3.2.0 + - zstandard >=0.23.0 + license: BSD-3-Clause + license_family: BSD + purls: + - pkg:pypi/pandas?source=compressed-mapping + run_exports: {} + size: 14059614 + timestamp: 1789803361326 - conda: https://conda.anaconda.org/conda-forge/win-64/pandoc-3.11-h57928b3_0.conda sha256: 03b03e1c841f25e738bafe60e20cae65a98951683f14e72aae6a8ac359e0501c md5: 4587e6544ff96ef0ee50b042f9d3719d @@ -34449,97 +34693,9 @@ packages: - pytest-lazy-fixtures ; extra == 'tests' - pytest>=9 ; extra == 'tests' requires_python: '>=3.10' -- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl - name: pandas - version: 3.1.0.dev0+2043.g7aac401536 - index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple - requires_dist: - - numpy>=2.0.2 ; python_full_version < '3.14' - - numpy>=2.3.3 ; python_full_version >= '3.14' - - python-dateutil>=2.9.0 - - tzdata ; sys_platform == 'win32' - - tzdata ; sys_platform == 'emscripten' - - pytest>=8.3.4 ; extra == 'test' - - pytest-xdist>=3.6.1 ; extra == 'test' - - pyarrow>=16.0.0 ; extra == 'pyarrow' - - bottleneck>=1.5.0 ; extra == 'performance' - - numba>=0.61.2 ; extra == 'performance' - - numexpr>=2.11.0,!=2.14.1 ; extra == 'performance' - - scipy>=1.16.1 ; extra == 'computation' - - xarray>=2025.7.1 ; extra == 'computation' - - fsspec>=2025.7.0 ; extra == 'fss' - - s3fs>=2025.7.0 ; extra == 'aws' - - gcsfs>=2025.7.0 ; extra == 'gcp' - - odfpy>=1.4.1 ; extra == 'excel' - - openpyxl>=3.1.5 ; extra == 'excel' - - python-calamine>=0.4.0 ; extra == 'excel' - - pyxlsb>=1.0.10 ; extra == 'excel' - - xlrd>=2.0.2 ; extra == 'excel' - - xlsxwriter>=3.2.5 ; extra == 'excel' - - pyarrow>=13.0.0 ; extra == 'parquet' - - pyarrow>=13.0.0 ; extra == 'feather' - - pyiceberg>=0.9.1 ; extra == 'iceberg' - - tables>=3.10.2 ; extra == 'hdf5' - - pyreadstat>=1.3.0 ; extra == 'spss' - - sqlalchemy>=2.0.42 ; extra == 'postgresql' - - psycopg2>=2.9.10 ; extra == 'postgresql' - - adbc-driver-postgresql>=1.7.0 ; extra == 'postgresql' - - sqlalchemy>=2.0.42 ; extra == 'mysql' - - pymysql>=1.1.1 ; extra == 'mysql' - - sqlalchemy>=2.0.42 ; extra == 'sql-other' - - adbc-driver-postgresql>=1.7.0 ; extra == 'sql-other' - - adbc-driver-sqlite>=1.7.0 ; extra == 'sql-other' - - beautifulsoup4>=4.13.4 ; extra == 'html' - - html5lib>=1.1 ; extra == 'html' - - lxml>=6.0.0 ; extra == 'html' - - lxml>=6.0.0 ; extra == 'xml' - - matplotlib>=3.10.5 ; extra == 'plot' - - jinja2>=3.1.6 ; extra == 'output-formatting' - - tabulate>=0.9.0 ; extra == 'output-formatting' - - pyqt5>=5.15.11 ; extra == 'clipboard' - - qtpy>=2.4.3 ; extra == 'clipboard' - - zstandard>=0.23.0 ; extra == 'compression' - - pytz>=2020.1 ; extra == 'timezone' - - adbc-driver-postgresql>=1.7.0 ; extra == 'all' - - adbc-driver-sqlite>=1.7.0 ; extra == 'all' - - beautifulsoup4>=4.13.4 ; extra == 'all' - - bottleneck>=1.5.0 ; extra == 'all' - - fastparquet>=2024.11.0 ; extra == 'all' - - fsspec>=2025.7.0 ; extra == 'all' - - gcsfs>=2025.7.0 ; extra == 'all' - - html5lib>=1.1 ; extra == 'all' - - jinja2>=3.1.6 ; extra == 'all' - - lxml>=6.0.0 ; extra == 'all' - - matplotlib>=3.10.5 ; extra == 'all' - - numba>=0.61.2 ; extra == 'all' - - numexpr>=2.11.0,!=2.14.1 ; extra == 'all' - - odfpy>=1.4.1 ; extra == 'all' - - openpyxl>=3.1.5 ; extra == 'all' - - psycopg2>=2.9.10 ; extra == 'all' - - pyarrow>=16.0.0 ; extra == 'all' - - pyiceberg>=0.9.1 ; extra == 'all' - - pymysql>=1.1.1 ; extra == 'all' - - pyqt5>=5.15.11 ; extra == 'all' - - pyreadstat>=1.3.0 ; extra == 'all' - - pytest>=8.3.4 ; extra == 'all' - - pytest-xdist>=3.6.1 ; extra == 'all' - - python-calamine>=0.4.0 ; extra == 'all' - - pytz>=2020.1 ; extra == 'all' - - pyxlsb>=1.0.10 ; extra == 'all' - - qtpy>=2.4.3 ; extra == 'all' - - scipy>=1.16.1 ; extra == 'all' - - s3fs>=2025.7.0 ; extra == 'all' - - sqlalchemy>=2.0.42 ; extra == 'all' - - tables>=3.10.2 ; extra == 'all' - - tabulate>=0.9.0 ; extra == 'all' - - xarray>=2025.7.1 ; extra == 'all' - - xlrd>=2.0.2 ; extra == 'all' - - xlsxwriter>=3.2.5 ; extra == 'all' - - zstandard>=0.23.0 ; extra == 'all' - requires_python: '>=3.11' -- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_10_15_x86_64.whl +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl name: pandas - version: 3.1.0.dev0+2043.g7aac401536 + version: 3.2.0.dev0+74.ga8069d6bdf index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - numpy>=2.0.2 ; python_full_version < '3.14' @@ -34625,9 +34781,9 @@ packages: - xlsxwriter>=3.2.5 ; extra == 'all' - zstandard>=0.23.0 ; extra == 'all' requires_python: '>=3.11' -- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-macosx_11_0_arm64.whl +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-macosx_10_15_x86_64.whl name: pandas - version: 3.1.0.dev0+2043.g7aac401536 + version: 3.2.0.dev0+74.ga8069d6bdf index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - numpy>=2.0.2 ; python_full_version < '3.14' @@ -34713,9 +34869,9 @@ packages: - xlsxwriter>=3.2.5 ; extra == 'all' - zstandard>=0.23.0 ; extra == 'all' requires_python: '>=3.11' -- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-macosx_11_0_arm64.whl name: pandas - version: 3.1.0.dev0+2043.g7aac401536 + version: 3.2.0.dev0+74.ga8069d6bdf index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - numpy>=2.0.2 ; python_full_version < '3.14' @@ -34801,9 +34957,9 @@ packages: - xlsxwriter>=3.2.5 ; extra == 'all' - zstandard>=0.23.0 ; extra == 'all' requires_python: '>=3.11' -- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.1.0.dev0+2043.g7aac401536/pandas-3.1.0.dev0+2043.g7aac401536-cp314-cp314-win_amd64.whl +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-win_amd64.whl name: pandas - version: 3.1.0.dev0+2043.g7aac401536 + version: 3.2.0.dev0+74.ga8069d6bdf index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - numpy>=2.0.2 ; python_full_version < '3.14' diff --git a/pyproject.toml b/pyproject.toml index 645b57c9..785492b0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -170,6 +170,10 @@ pre-commit = "*" [tool.pixi.feature.dev.dependencies] ipython = "*" +# A conda pandas for the default environment, which includes the nightly index +# of the sklearn-nightly feature: that index shadows PyPI for the packages it +# has, so a pypi pandas there would be a dev build. +pandas = ">=2" [tool.pixi.feature.sklearn12.dependencies] scikit-learn = "~=1.2.0" @@ -267,12 +271,21 @@ extra-index-urls = ["https://pypi.anaconda.org/scientific-python-nightly-wheels/ # The version value here needs to be exact, hence == instead of ~= scikit-learn = "==1.10.dev0" fairlearn = "*" -# The dev version of pandas from the nightly index; "*" would pick the latest -# release, since pre-releases are only considered when named explicitly. -pandas = "==3.1.0.dev0" numpy = "*" scipy = "*" +[tool.pixi.feature.pandas-nightly.pypi-dependencies] +# The dev version of pandas from the nightly index; "*" would pick the latest +# release, since pre-releases are only considered when named explicitly. Unlike +# the scikit-learn and scipy nightlies, pandas puts the git revision in the +# version, so every build is a new file and the index drops old ones after a +# few days: the locked wheel goes stale. CI refreshes it before installing the +# environment, see .github/workflows/build-test.yml; do the same locally with +# ``pixi update --environment ci-sklearn-nightly pandas``. This is why the +# feature is not part of the default environment. Like the scikit-learn pin +# above, the version needs a bump after each pandas release. +pandas = "==3.2.0.dev0" + [tool.pixi.feature.lint.tasks] lint = { cmd = "pre-commit install && pre-commit run -v --all-files --show-diff-on-failure" } @@ -291,4 +304,4 @@ ci-sklearn16 = ["rich", "tests", "lint", "sklearn16"] ci-sklearn17 = ["rich", "tests", "lint", "sklearn17"] ci-sklearn18 = ["rich", "tests", "lint", "sklearn18"] ci-sklearn19 = ["rich", "tests", "lint", "sklearn19"] -ci-sklearn-nightly = ["rich", "tests", "lint", "sklearn-nightly"] +ci-sklearn-nightly = ["rich", "tests", "lint", "sklearn-nightly", "pandas-nightly"] diff --git a/skops/io/_pandas.py b/skops/io/_pandas.py index 6b8c8187..d09e2732 100644 --- a/skops/io/_pandas.py +++ b/skops/io/_pandas.py @@ -34,6 +34,7 @@ from __future__ import annotations +import datetime import re import sys import warnings @@ -157,38 +158,51 @@ def dataframe_get_state(obj: Any, save_context: SaveContext) -> dict[str, Any]: return _pandas_state(obj, "PandasDataFrameNode", content, save_context) -_FIXED_OFFSET = re.compile(r"^UTC([+-])(\d{2}):(\d{2})$") +_FIXED_OFFSET = re.compile(r"^UTC([+-])(\d{2}):(\d{2})(?::(\d{2}))?$") def _timezone_name(tz: Any) -> str: # The name of a time zone as written to the file: the key of a zoneinfo - # or pytz zone, "UTC", or "UTC+01:00" for a fixed offset. These are the - # only names ``_timezone`` accepts when loading. + # or pytz zone, "UTC", or a fixed offset as "UTC+01:00", with its seconds + # as "UTC+00:19:32" if it has any. These are the only names ``_timezone`` + # accepts when loading. name = getattr(tz, "key", None) or getattr(tz, "zone", None) if name is not None: return name offset = tz.utcoffset(None) - if offset is None: + if offset is None or offset.microseconds: raise UnsupportedTypeException( - f"The time zone {tz!r} has neither a name nor a fixed offset, so it" - " cannot be saved." + f"The time zone {tz!r} has neither a name nor a fixed offset of whole" + " seconds, so it cannot be saved." ) seconds = int(offset.total_seconds()) if seconds == 0: return "UTC" sign, seconds = ("-", -seconds) if seconds < 0 else ("+", seconds) - return f"UTC{sign}{seconds // 3600:02d}:{seconds % 3600 // 60:02d}" + name = f"UTC{sign}{seconds // 3600:02d}:{seconds % 3600 // 60:02d}" + if seconds % 60: + name += f":{seconds % 60:02d}" + return name -def _timezone(name: str) -> str: - # The name of a time zone from the file, checked before pandas parses it: - # pandas' own parser would also accept "tzlocal()" and "dateutil/", - # the latter opening any file on disk. Accepted are "UTC", a fixed offset, - # and a key that zoneinfo finds in its own directories; pandas then builds - # the zone of its default implementation from the name, so that it equals - # the zone that was saved. - if name == "UTC" or _FIXED_OFFSET.match(name): - return name +def _timezone(name: str) -> str | datetime.tzinfo: + # The time zone for a name from the file, parsed here rather than by + # pandas: pandas' own parser would also accept "tzlocal()" and + # "dateutil/", the latter opening any file on disk, and it drops the + # seconds of a fixed offset. "UTC" and fixed offsets become the + # ``datetime.timezone`` objects pandas builds for them too. A zone key is + # checked with zoneinfo, which only looks in its own directories, and then + # handed to pandas as a name, so that pandas builds the zone of its + # default implementation and the loaded zone equals the saved one. + if name == "UTC": + return datetime.timezone.utc + match = _FIXED_OFFSET.match(name) + if match: + sign, hours, minutes, seconds = match.groups() + offset = datetime.timedelta( + hours=int(hours), minutes=int(minutes), seconds=int(seconds or 0) + ) + return datetime.timezone(-offset if sign == "-" else offset) try: zoneinfo.ZoneInfo(name) except (ValueError, KeyError, OSError) as err: @@ -317,7 +331,7 @@ def __init__( def _allowed_types(self) -> dict[str, tuple[type[Node], ...] | None]: # The node types each entry may hold, ``None`` for any: names for # instance can be any hashable. - raise NotImplementedError + raise NotImplementedError # pragma: no cover def _construct_content(self) -> dict[str, Any]: return {key: node.construct() for key, node in self.content.items()} diff --git a/skops/io/tests/_utils.py b/skops/io/tests/_utils.py index a66ab04c..95da1d60 100644 --- a/skops/io/tests/_utils.py +++ b/skops/io/tests/_utils.py @@ -17,7 +17,7 @@ try: import pandas as pd -except ImportError: # pandas is optional +except ImportError: # pragma: no cover pd = None # TODO: Investigate why that seems to be an issue on MacOS (only observed with diff --git a/skops/io/tests/test_pandas.py b/skops/io/tests/test_pandas.py index e80d0a57..7653b9bd 100644 --- a/skops/io/tests/test_pandas.py +++ b/skops/io/tests/test_pandas.py @@ -16,7 +16,7 @@ from skops.io import dump, dumps, get_untrusted_types, load, loads, visualize from skops.io._pandas import _RENAMED_CLASSES, _public_module from skops.io._trusted_types import PANDAS_TYPE_NAMES -from skops.io._utils import gettype +from skops.io._utils import get_type_name, gettype from skops.io.exceptions import UnsupportedTypeException from skops.io.tests._utils import _assert_vals_equal @@ -109,6 +109,17 @@ pd.array( pd.to_datetime(["2024-01-01"]).tz_localize(dt.timezone(dt.timedelta(hours=1))) ), + # fixed offsets with seconds, as in the local mean time of old data + pd.array( + pd.to_datetime(["1850-01-01"]).tz_localize( + dt.timezone(dt.timedelta(minutes=19, seconds=32)) + ) + ), + pd.array( + pd.to_datetime(["2024-01-01"]).tz_localize( + dt.timezone(-dt.timedelta(hours=3, minutes=30, seconds=45)) + ) + ), pd.array(pd.to_timedelta([1], unit="s")), pd.array(pd.period_range("2024-01", periods=1, freq="M")), pd.array(pd.interval_range(0, 2)), @@ -124,6 +135,7 @@ pd.CategoricalDtype(), pd.DatetimeTZDtype("ns", "UTC"), pd.DatetimeTZDtype("us", "Europe/Berlin"), + pd.DatetimeTZDtype("ns", dt.timezone(dt.timedelta(seconds=30))), pd.PeriodDtype("M"), pd.IntervalDtype("int64", closed="left"), pd.IntervalDtype("datetime64[ns]"), @@ -148,6 +160,13 @@ def test_pandas_types_are_trusted_by_default(): assert get_untrusted_types(data=dumps(FRAMES[1])) == [] +def test_private_pandas_types_are_not_trusted(): + # the dtype of a numpy-backed extension array has no public name, so the + # file holds its defining module, which the default trust does not cover + dtype = pd.Series([1, 2]).array.dtype + assert get_untrusted_types(data=dumps(dtype)) == [get_type_name(type(dtype))] + + def _with_edited_schema(dumped, edit): # ``dumped`` with ``edit`` applied to its schema, to mimic a crafted file with ZipFile(io.BytesIO(dumped)) as zip_file: @@ -179,6 +198,18 @@ def edit(schema): loads(_with_edited_schema(dumped, edit)) +def test_dtype_node_refuses_other_classes(): + # a trusted pandas class that is not an extension dtype cannot be routed + # through the name based dtype loader + dumped = dumps(pd.Int64Dtype()) + + def edit(schema): + schema["__class__"] = "Series" + + with pytest.raises(ValueError, match="not a pandas extension dtype"): + loads(_with_edited_schema(dumped, edit)) + + def test_dtype_node_refuses_dtypes_stored_as_parts(): # the parsers of these dtypes consult pandas' registry of extension dtypes # or the file system, so a file cannot route them through the name based @@ -210,6 +241,14 @@ def edit(schema): loads(_with_edited_schema(dumped, edit)) +def test_sub_second_offset_is_unsupported(): + # the name of a fixed offset holds whole seconds; a finer offset cannot be + # written, and must not be rounded silently + tz = dt.timezone(dt.timedelta(microseconds=1)) + with pytest.raises(UnsupportedTypeException, match="whole seconds"): + dumps(pd.DatetimeTZDtype("ns", tz)) + + def test_interval_dtype_subtype_is_parsed_by_numpy(): # pandas would look a subtype name up in its registry of extension dtypes, # where any imported library can register a dtype whose parser then runs @@ -221,6 +260,12 @@ def edit(schema): with pytest.raises(ValueError, match="subtype of an interval dtype"): loads(_with_edited_schema(dumped, edit)) + def edit_to_number(schema): + schema["content"]["subtype"]["content"] = json.dumps(5) + + with pytest.raises(ValueError, match="subtype of an interval dtype"): + loads(_with_edited_schema(dumped, edit_to_number)) + def test_child_of_wrong_kind_is_refused(): # the values of an Index are an array; a file holding something else there From 0be24e0a4f6feabca43815e0b89bb62d9da2bb45 Mon Sep 17 00:00:00 2001 From: adrinjalali Date: Sun, 4 Oct 2026 21:46:36 +0100 Subject: [PATCH 7/8] reviews --- .github/workflows/build-test.yml | 9 +++++---- docs/changes.rst | 8 +++++--- docs/persistence.rst | 9 ++++++--- skops/io/_pandas.py | 19 ++++++++++++++++--- skops/io/_trusted_types.py | 8 ++++---- skops/io/tests/test_pandas.py | 27 +++++++++++++++++++++++++++ 6 files changed, 63 insertions(+), 17 deletions(-) diff --git a/.github/workflows/build-test.yml b/.github/workflows/build-test.yml index 89df29a3..13ae7a0e 100644 --- a/.github/workflows/build-test.yml +++ b/.github/workflows/build-test.yml @@ -50,10 +50,11 @@ jobs: pixi-version: v0.68.0 environments: ${{ matrix.environment }} # we can freeze the environment and manually bump the dependencies to the - # latest version time to time. - frozen: true - # the nightly environment is installed below, after refreshing the - # locked pandas nightly wheel, which the index drops after a few days + # latest version time to time. The nightly environment is installed + # below, after refreshing the locked pandas nightly wheel, which the + # index drops after a few days; the action refuses ``frozen`` without + # an install, hence the same condition on both. + frozen: ${{ matrix.environment != 'ci-sklearn-nightly' }} run-install: ${{ matrix.environment != 'ci-sklearn-nightly' }} - name: Install the nightly environment with the latest pandas nightly diff --git a/docs/changes.rst b/docs/changes.rst index f6d0264f..1d4ffbbc 100644 --- a/docs/changes.rst +++ b/docs/changes.rst @@ -18,11 +18,13 @@ v0.17 constructed, instead of failing with an unrelated error, or being accepted, during construction. :pr:`547` by `Adrin Jalali`_. - Add support for pandas objects: :class:`~pandas.DataFrame`, - :class:`~pandas.Series`, every kind of :class:`~pandas.Index`, extension - arrays and extension dtypes can now be saved and loaded. They are stored as + :class:`~pandas.Series`, every kind of :class:`~pandas.Index`, and the + extension arrays and extension dtypes of pandas itself, not those of other + libraries, can now be saved and loaded. They are stored as the numpy arrays and scalars they are made of and rebuilt through the public pandas constructors, so no pandas internals end up in the file, and they are - trusted by default. Estimators from other libraries that keep pandas objects + trusted by default, except for the pyarrow backed arrays and dtypes. + Estimators from other libraries that keep pandas objects in their fitted attributes, such as ``category_encoders``, can now be persisted. A file written with one pandas version loads with any other from 2.0 on, keeping the dtypes of the version that wrote it. Not preserved are diff --git a/docs/persistence.rst b/docs/persistence.rst index 2a5aecd1..9d938c93 100644 --- a/docs/persistence.rst +++ b/docs/persistence.rst @@ -251,11 +251,14 @@ Skops intends to support all of **scikit-learn**, that is, not only its estimators, but also other classes like cross validation splitters. Furthermore, most types from **numpy** and **scipy** should be supported, such as (sparse) arrays, dtypes, random generators, and ufuncs. **pandas** objects, that is -``DataFrame``, ``Series``, every kind of ``Index``, extension arrays and -extension dtypes, are supported as well with pandas 2.0 or later: they are +``DataFrame``, ``Series``, every kind of ``Index``, and the extension arrays +and extension dtypes of pandas itself, not those of other libraries, are +supported as well with pandas 2.0 or later: they are stored as the arrays they are made of and rebuilt through the public pandas constructors, so that no pandas internals end up in the file, and a file -written with one pandas version loads with any other. Not preserved are the +written with one pandas version loads with any other. They are trusted by +default, except for the pyarrow backed arrays and dtypes, which need to be +passed as ``trusted`` explicitly. Not preserved are the ``freq`` of datetime-like indexes and arrays, the ``attrs`` and ``flags`` of a ``Series`` or ``DataFrame``, and the storage, python or pyarrow, of a string dtype, which is an environment choice over the same values. diff --git a/skops/io/_pandas.py b/skops/io/_pandas.py index d09e2732..a32ea684 100644 --- a/skops/io/_pandas.py +++ b/skops/io/_pandas.py @@ -264,7 +264,19 @@ def extension_dtype_get_state(obj: Any, save_context: SaveContext) -> dict[str, # Extension dtypes are rebuilt from their string form, e.g. "Int64" or # "period[M]", except those whose parser consults pandas' registry of # extension dtypes or the file system, which are stored as parts below. - content = {"name": str(obj)} + name = str(obj) + try: + rebuilt = type(obj).construct_from_string(name) + except TypeError: + rebuilt = None + if type(rebuilt) is not type(obj): + # e.g. ArrowDtype(pyarrow.string()), whose name "string[pyarrow]" pandas + # reserves for its StringDtype + raise UnsupportedTypeException( + f"The dtype {obj!r} cannot be rebuilt from its name {name!r}, so it" + " cannot be saved." + ) + content = {"name": name} return _pandas_state(obj, "PandasExtensionDtypeNode", content, save_context) @@ -514,8 +526,9 @@ def _construct(self): # registry of extension dtypes instead, where any imported library can # register one, and run that library's code for a name from the file. # The parsers of the remaining pandas dtypes, the masked numeric and - # boolean ones, StringDtype, PeriodDtype and ArrowDtype, compare the - # name with their own, or hand it to the offset parser or to pyarrow. + # boolean ones, StringDtype and PeriodDtype, compare the name with + # their own or hand it to the offset parser. ArrowDtype hands it to + # pyarrow, and is not trusted by default. cls = gettype(self.module_name, self.class_name) if not issubclass(cls, pd.api.extensions.ExtensionDtype): raise ValueError( diff --git a/skops/io/_trusted_types.py b/skops/io/_trusted_types.py index a223e25a..c392daf2 100644 --- a/skops/io/_trusted_types.py +++ b/skops/io/_trusted_types.py @@ -139,7 +139,10 @@ # pandas types which ``skops.io._pandas`` rebuilds from their data through the # public pandas constructors, by the public names it writes to the file. They # are listed as strings so that pandas, which is optional, is not imported -# here. +# here. The pyarrow backed arrays and dtypes, ``pandas.arrays.ArrowExtensionArray``, +# ``pandas.arrays.ArrowStringArray`` and ``pandas.ArrowDtype``, can be saved and +# loaded the same way but are not trusted by default, since loading them runs +# pyarrow's conversion of the loaded values, which has not been reviewed. PANDAS_TYPE_NAMES = [ "pandas.DataFrame", "pandas.Series", @@ -151,8 +154,6 @@ "pandas.TimedeltaIndex", "pandas.PeriodIndex", "pandas.IntervalIndex", - "pandas.arrays.ArrowExtensionArray", - "pandas.arrays.ArrowStringArray", "pandas.arrays.BooleanArray", "pandas.arrays.Categorical", "pandas.arrays.DatetimeArray", @@ -164,7 +165,6 @@ "pandas.arrays.SparseArray", "pandas.arrays.StringArray", "pandas.arrays.TimedeltaArray", - "pandas.ArrowDtype", "pandas.BooleanDtype", "pandas.CategoricalDtype", "pandas.DatetimeTZDtype", diff --git a/skops/io/tests/test_pandas.py b/skops/io/tests/test_pandas.py index 7653b9bd..0fb24e00 100644 --- a/skops/io/tests/test_pandas.py +++ b/skops/io/tests/test_pandas.py @@ -321,6 +321,33 @@ def test_file_uses_public_type_names(): assert (index["__module__"], index["__class__"]) == ("pandas", "Index") +class NamelessDtype(pd.api.extensions.ExtensionDtype): + """A dtype whose name does not rebuild it. + + pandas has such a dtype, ``ArrowDtype(pyarrow.string())``, whose name + "string[pyarrow]" it reserves for its ``StringDtype``; this stand-in needs + no pyarrow. The module is faked since skops only saves pandas' own dtypes. + """ + + __module__ = "pandas.tests" + name = "nameless" + type = object + + @classmethod + def construct_array_type(cls): + return pd.arrays.NumpyExtensionArray # pragma: no cover + + @classmethod + def construct_from_string(cls, string): + raise TypeError(f"Cannot construct a 'NamelessDtype' from '{string}'") + + +def test_dtype_not_rebuildable_from_its_name_is_unsupported(): + # refused when saving rather than failing when loading + with pytest.raises(UnsupportedTypeException, match="cannot be rebuilt"): + dumps(NamelessDtype()) + + def test_renamed_classes_match_pandas_version(): # each rename is a fact about pandas: before the version that introduced # the new name only the old one exists, from then on the new one does From 2707e605bfc8370731c0d9ba073ac09f19e513c8 Mon Sep 17 00:00:00 2001 From: adrinjalali Date: Mon, 5 Oct 2026 19:45:21 +0100 Subject: [PATCH 8/8] CI --- .github/workflows/build-test.yml | 18 ++++++++++++------ docs/changes.rst | 4 +++- docs/persistence.rst | 4 +++- pixi.lock | 24 ++++++++++++------------ 4 files changed, 30 insertions(+), 20 deletions(-) diff --git a/.github/workflows/build-test.yml b/.github/workflows/build-test.yml index 13ae7a0e..80b1e7ef 100644 --- a/.github/workflows/build-test.yml +++ b/.github/workflows/build-test.yml @@ -46,16 +46,22 @@ jobs: shell: bash - uses: prefix-dev/setup-pixi@v0.10.2 + if: matrix.environment != 'ci-sklearn-nightly' with: pixi-version: v0.68.0 environments: ${{ matrix.environment }} # we can freeze the environment and manually bump the dependencies to the - # latest version time to time. The nightly environment is installed - # below, after refreshing the locked pandas nightly wheel, which the - # index drops after a few days; the action refuses ``frozen`` without - # an install, hence the same condition on both. - frozen: ${{ matrix.environment != 'ci-sklearn-nightly' }} - run-install: ${{ matrix.environment != 'ci-sklearn-nightly' }} + # latest version time to time. + frozen: true + + # The nightly environment is installed by hand: the locked pandas nightly + # wheel is dropped from its index after a few days, so it is refreshed + # first. The action accepts no other input together with run-install. + - uses: prefix-dev/setup-pixi@v0.10.2 + if: matrix.environment == 'ci-sklearn-nightly' + with: + pixi-version: v0.68.0 + run-install: false - name: Install the nightly environment with the latest pandas nightly if: matrix.environment == 'ci-sklearn-nightly' diff --git a/docs/changes.rst b/docs/changes.rst index 1d4ffbbc..54ee04b5 100644 --- a/docs/changes.rst +++ b/docs/changes.rst @@ -23,7 +23,9 @@ v0.17 libraries, can now be saved and loaded. They are stored as the numpy arrays and scalars they are made of and rebuilt through the public pandas constructors, so no pandas internals end up in the file, and they are - trusted by default, except for the pyarrow backed arrays and dtypes. + trusted by default, except for the pyarrow backed arrays and dtypes. The + one dtype that cannot be saved is ``ArrowDtype(pyarrow.string())``, whose + name pandas reserves for its ``StringDtype``. Estimators from other libraries that keep pandas objects in their fitted attributes, such as ``category_encoders``, can now be persisted. A file written with one pandas version loads with any other from diff --git a/docs/persistence.rst b/docs/persistence.rst index 9d938c93..60c2888e 100644 --- a/docs/persistence.rst +++ b/docs/persistence.rst @@ -258,7 +258,9 @@ stored as the arrays they are made of and rebuilt through the public pandas constructors, so that no pandas internals end up in the file, and a file written with one pandas version loads with any other. They are trusted by default, except for the pyarrow backed arrays and dtypes, which need to be -passed as ``trusted`` explicitly. Not preserved are the +passed as ``trusted`` explicitly; ``ArrowDtype(pyarrow.string())`` cannot be +saved at all, since its name is the one pandas reserves for ``StringDtype``. +Not preserved are the ``freq`` of datetime-like indexes and arrays, the ``attrs`` and ``flags`` of a ``Series`` or ``DataFrame``, and the storage, python or pyarrow, of a string dtype, which is an environment choice over the same values. diff --git a/pixi.lock b/pixi.lock index 9b158e2c..84f42369 100644 --- a/pixi.lock +++ b/pixi.lock @@ -209,7 +209,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/99/c7/bd05c5c430feb347aa040fcc8870135d70b256718deee9bc7d2ca74a77ff/xgboost-3.4.1-py3-none-manylinux_2_28_x86_64.whl - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+83.g71e11934c8/pandas-3.2.0.dev0+83.g71e11934c8-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl osx-64: @@ -334,7 +334,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/cd/05/7213965863cba1ed0150ad045bceed6276a1afaaaedbaeff4699ec4f0ccb/lightgbm-4.7.0-py3-none-macosx_10_15_x86_64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-macosx_10_15_x86_64.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+83.g71e11934c8/pandas-3.2.0.dev0+83.g71e11934c8-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_10_15_x86_64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_10_15_x86_64.whl osx-arm64: @@ -459,7 +459,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/c4/0e/57f6bb3024a597b2e8ec4aee710ffe62ddc95af2e2bb1ee7a7abdc22c68c/wcwidth-0.8.3-py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/f7/94/e5c37a8972ad780edc1d8459d1931356344ca133f7f99ba9cfda516b5bba/xgboost-3.4.1-py3-none-macosx_12_0_arm64.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-macosx_11_0_arm64.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+83.g71e11934c8/pandas-3.2.0.dev0+83.g71e11934c8-cp314-cp314-macosx_11_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-macosx_12_0_arm64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-macosx_12_0_arm64.whl win-64: @@ -591,7 +591,7 @@ environments: - pypi: https://files.pythonhosted.org/packages/d5/0b/c5c17d862b12ce292f24cd85d40f2f8f8981668fbdbd43fdc2625eccbc79/lightgbm-4.7.0-py3-none-win_amd64.whl - pypi: https://files.pythonhosted.org/packages/f9/bc/8737e8d54cf51106118039b83f485a4783112fab49ea9d044b234978a46e/tzdata-2026.4-py2.py3-none-any.whl - pypi: https://files.pythonhosted.org/packages/fe/be/2e6798ace5cc036f5d05d36b7b2fd85346f1a708c87060890b070d0ec607/prettytable-3.18.0-py3-none-any.whl - - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-win_amd64.whl + - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+83.g71e11934c8/pandas-3.2.0.dev0+83.g71e11934c8-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scikit-learn/1.10.dev0/scikit_learn-1.10.dev0-cp314-cp314-win_amd64.whl - pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/scipy/2.0.0.dev0/scipy-2.0.0.dev0-cp314-cp314-win_amd64.whl ci-sklearn12: @@ -34693,9 +34693,9 @@ packages: - pytest-lazy-fixtures ; extra == 'tests' - pytest>=9 ; extra == 'tests' requires_python: '>=3.10' -- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+83.g71e11934c8/pandas-3.2.0.dev0+83.g71e11934c8-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl name: pandas - version: 3.2.0.dev0+74.ga8069d6bdf + version: 3.2.0.dev0+83.g71e11934c8 index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - numpy>=2.0.2 ; python_full_version < '3.14' @@ -34781,9 +34781,9 @@ packages: - xlsxwriter>=3.2.5 ; extra == 'all' - zstandard>=0.23.0 ; extra == 'all' requires_python: '>=3.11' -- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-macosx_10_15_x86_64.whl +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+83.g71e11934c8/pandas-3.2.0.dev0+83.g71e11934c8-cp314-cp314-macosx_10_15_x86_64.whl name: pandas - version: 3.2.0.dev0+74.ga8069d6bdf + version: 3.2.0.dev0+83.g71e11934c8 index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - numpy>=2.0.2 ; python_full_version < '3.14' @@ -34869,9 +34869,9 @@ packages: - xlsxwriter>=3.2.5 ; extra == 'all' - zstandard>=0.23.0 ; extra == 'all' requires_python: '>=3.11' -- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-macosx_11_0_arm64.whl +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+83.g71e11934c8/pandas-3.2.0.dev0+83.g71e11934c8-cp314-cp314-macosx_11_0_arm64.whl name: pandas - version: 3.2.0.dev0+74.ga8069d6bdf + version: 3.2.0.dev0+83.g71e11934c8 index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - numpy>=2.0.2 ; python_full_version < '3.14' @@ -34957,9 +34957,9 @@ packages: - xlsxwriter>=3.2.5 ; extra == 'all' - zstandard>=0.23.0 ; extra == 'all' requires_python: '>=3.11' -- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+74.ga8069d6bdf/pandas-3.2.0.dev0+74.ga8069d6bdf-cp314-cp314-win_amd64.whl +- pypi: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple/pandas/3.2.0.dev0+83.g71e11934c8/pandas-3.2.0.dev0+83.g71e11934c8-cp314-cp314-win_amd64.whl name: pandas - version: 3.2.0.dev0+74.ga8069d6bdf + version: 3.2.0.dev0+83.g71e11934c8 index: https://pypi.anaconda.org/scientific-python-nightly-wheels/simple requires_dist: - numpy>=2.0.2 ; python_full_version < '3.14'