From ffb84f7abd9da8faf6397439b62d179a5e2fe2c2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Moreira=20Souza?= Date: Thu, 30 Jul 2026 18:02:10 -0300 Subject: [PATCH] docs: bring the README up to date with 0.4.0 through 0.6.0, and release 0.6.1 The package description on PyPI had not been updated since 0.3.0. It listed none of the features added since: not NumPy becoming the only dependency, not save_npz/load_npz, not the training reports, and not the scikit-learn estimator interface that 0.6.0 exists to add. The 0.6.0 description does not contain the word "estimator" or "scikit-learn" anywhere. A PyPI description cannot be edited after upload, so correcting it takes a release. No code changed and results are identical to 0.6.0. The upgrade notes were also short of two things. 0.4.0 changes linear-initialization results for data far from the origin, where the previous PCA path was wrong by up to 5.8%, and it removed pandas and scikit-learn as runtime dependencies. Both now appear alongside the existing 0.3.0 warning, with a note that 0.5.0's string deprecation was withdrawn so anyone who saw the warning knows to stop migrating. Install now shows all three extras, and the quick start shows the estimator interface and save/load, since a reader comparing libraries reads the first code block and stops. Four feature bullets were bolded in a first draft and are not any more. Bolding four of fourteen items to draw the eye is emphasis as decoration, which the writing scan flags as bold-overuse; position does that job without shouting. Release verification gains a check that would have caught this before publishing rather than after: the built wheel's METADATA is grepped for the terms the release claims. twine check confirms the description renders, which says nothing about whether it is accurate. --- CHANGELOG.md | 22 +++++++++++++++-- README.md | 50 +++++++++++++++++++++++++++++++++----- pyproject.toml | 2 +- src/python_som/_version.py | 2 +- uv.lock | 2 +- 5 files changed, 67 insertions(+), 11 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 621df0e..6986d6f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,11 +5,29 @@ All notable changes to this project are documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [0.6.1] - 2026-07-30 + +### Fixed + +- **The package description on PyPI now describes the current package.** It had not been updated + since 0.3.0, so it listed none of the features added in 0.4.0, 0.5.0 or 0.6.0: not the drop to + NumPy as the only dependency, not `save_npz`/`load_npz`, not the training reports, and not the + scikit-learn estimator interface that 0.6.0 exists to add. + + A PyPI description cannot be edited after upload, so correcting it takes a release. That is the + whole of this one: no code changed, and results are identical to 0.6.0. + + The upgrade notes were also short of two things worth knowing. 0.4.0 changes + linear-initialization results for data far from the origin, where the previous PCA path was wrong + by up to 5.8%, and it removed pandas and scikit-learn as runtime dependencies, which matters to + anyone who was importing either transitively. Both are now in the README, alongside a note that 0.5.0's + string deprecation was withdrawn so anyone who saw the warning knows to stop migrating. + ## [0.6.0] - 2026-07-30 A map can now be used as a scikit-learn estimator, and the plain-string deprecation from 0.5.0 is -withdrawn. Nothing breaks and no numerical results change: `train` is untouched, both option spellings -work, and scikit-learn remains optional. +withdrawn. Nothing breaks and no numerical results change: `train` is untouched, both option +spellings work, and scikit-learn remains optional. ### Added diff --git a/README.md b/README.md index 7216b65..17ce664 100644 --- a/README.md +++ b/README.md @@ -14,8 +14,10 @@ arrays, pandas DataFrames, polars, pyarrow, and anything else implementing the ` ## Install ```bash -pip install python-som # requires Python 3.10+ -pip install "python-som[cli]" # adds tqdm progress bars +pip install python-som # requires Python 3.10+; NumPy is the only dependency +pip install "python-som[cli]" # adds tqdm progress bars +pip install "python-som[sklearn]" # adds the scikit-learn estimator adapter +pip install "python-som[examples]" # adds matplotlib and seaborn, for the plots ``` ## Quick start @@ -35,6 +37,17 @@ umatrix = som.distance_matrix() winner = som.winner(data[0]) ``` +The same map through the estimator interface, which `Pipeline` and `GridSearchCV` also understand: + +```python +som.fit(data, n_iteration=len(data), mode="batch") +labels = som.predict(data) # (n_samples,) flat node index +distances = som.transform(data) # (n_samples, x*y) + +som.save_npz("map.npz") # models plus provenance, no pickle +som = python_som.SOM.load_npz("map.npz") +``` + A full worked example with plots is in [examples/iris.py](https://github.com/andremsouza/python-som/blob/master/examples/iris.py) and in the [getting-started guide](https://andremsouza.github.io/python-som/tutorial/first-map/). @@ -42,12 +55,18 @@ A full worked example with plots is in [examples/iris.py](https://github.com/and ## Features +* NumPy is the only runtime dependency; a fresh install is 69 MB across one package * Stepwise and batch training * Random, random-sampling and linear (PCA) weight initialization * Automatic selection of the map size ratio, from PCA * Cyclic arrays, for toroidal maps * Gaussian, bubble and Mexican hat neighborhood functions * Custom decay functions +* Save and load a trained map without `pickle`, so loading one cannot execute code +* Provenance: every run records its seed, iteration count, error and library versions +* Works as a scikit-learn estimator: `fit`, `transform`, `predict`, and an adapter for `Pipeline`, + `GridSearchCV` and `cross_val_score` +* Options accepted as plain strings or enums, with typos caught by a type checker * Visualization support: U-matrix, activation matrix * Supervised labelling, via the label map * Fully type-annotated, with a `py.typed` marker @@ -73,10 +92,27 @@ the derivations, including why the Mexican hat is not an outer product of two 1- ## Upgrading -0.3.0 corrects several methodology defects, so numerical results are not comparable with earlier -versions. In particular **`random_seed` no longer reproduces pre-0.3.0 maps**: the generator is now -per-instance rather than a call to `np.random.seed` on NumPy's global state. To reproduce figures -made with an older version, pin `python-som==0.2.0`. +Two releases change numerical results. If you are reproducing a figure, pin the version that made +it. + +**0.3.0** corrects several methodology defects, so results are not comparable with earlier versions. +In particular **`random_seed` no longer reproduces pre-0.3.0 maps**: the generator is now +per-instance rather than a call to `np.random.seed` on NumPy's global state. To reproduce older +figures, pin `python-som==0.2.0`. + +**0.4.0** corrects linear initialization for data far from the origin. Its PCA previously went +through scikit-learn's `auto` solver, which forms a covariance matrix and loses precision when the +mean is large relative to the spread. On data offset by 1e7 the second explained variance was wrong +by 5.8%. +Near the origin the difference is floating-point noise. Timestamps, coordinates and absolute sensor +readings are the cases that were affected. + +0.4.0 also **removed pandas and scikit-learn as runtime dependencies**. If you imported either +transitively through this package, depend on them directly, or install `python-som[examples]`. + +**0.5.0** briefly made plain-string options emit a `DeprecationWarning`. **0.6.0 withdrew that**: +strings are permanent and 1.0.0 will not remove them. If you saw that warning, you can stop +migrating. Each change and the passage of Kohonen (2013) behind it is in the [changelog](https://github.com/andremsouza/python-som/blob/master/CHANGELOG.md). @@ -89,6 +125,8 @@ uv run pytest --cov # tests and coverage uv run ruff check . # lint uv run ruff format --check . # formatting uv run mypy # type-check +uv run bandit -c pyproject.toml -r src/ # security checks +uv run pip-audit # known vulnerabilities in the resolved set uv run mkdocs serve # docs, locally pre-commit install # optional, run the gates on commit ``` diff --git a/pyproject.toml b/pyproject.toml index a7d72bb..0b45100 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "python-som" -version = "0.6.0" +version = "0.6.1" authors = [{ name = "André Moreira Souza", email = "msouza.andre@hotmail.com" }] description = "Python implementation of the Self-Organizing Map" readme = "README.md" diff --git a/src/python_som/_version.py b/src/python_som/_version.py index 54fa2bf..ba99836 100644 --- a/src/python_som/_version.py +++ b/src/python_som/_version.py @@ -9,4 +9,4 @@ __all__ = ["__version__"] -__version__ = "0.6.0" +__version__ = "0.6.1" diff --git a/uv.lock b/uv.lock index 3167148..5d50574 100644 --- a/uv.lock +++ b/uv.lock @@ -2559,7 +2559,7 @@ wheels = [ [[package]] name = "python-som" -version = "0.6.0" +version = "0.6.1" source = { editable = "." } dependencies = [ { name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" },