From b010e3f9dac510d3d604b918858a7f1202070d20 Mon Sep 17 00:00:00 2001 From: maclandrol Date: Thu, 3 Sep 2026 14:34:38 +0100 Subject: [PATCH] chore!: simplify install extras for 0.2.1 (base=core+viz, [model]=full stack) --- .github/workflows/doc.yml | 4 ++-- .github/workflows/release.yml | 4 ++-- .github/workflows/test.yml | 2 +- CHANGELOG.md | 15 +++++++++++++++ README.md | 34 ++++++++++++++++----------------- docs/index.md | 36 +++++++++++++++++------------------ docs/migration.md | 23 +++++++++++----------- env.yml | 2 +- expts/README.md | 2 +- pyproject.toml | 28 ++++----------------------- safe/__init__.py | 17 ++++++++++------- 11 files changed, 80 insertions(+), 87 deletions(-) diff --git a/.github/workflows/doc.yml b/.github/workflows/doc.yml index 411e20e..ddfbfbf 100644 --- a/.github/workflows/doc.yml +++ b/.github/workflows/doc.yml @@ -31,7 +31,7 @@ jobs: enable-cache: true - name: Install documentation dependencies - run: uv pip install --editable ".[docs,train]" + run: uv pip install --editable ".[docs,model]" - name: Build documentation run: mkdocs build --strict @@ -61,7 +61,7 @@ jobs: enable-cache: true - name: Install documentation dependencies - run: uv pip install --editable ".[docs,train]" + run: uv pip install --editable ".[docs,model]" - name: Configure Git run: | diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index b8f7e00..bb178fa 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -90,7 +90,7 @@ jobs: - name: Build release documentation run: | - uv pip install --editable ".[docs,train]" + uv pip install --editable ".[docs,model]" mkdocs build --strict - name: Upload distributions @@ -174,7 +174,7 @@ jobs: enable-cache: true - name: Install documentation dependencies - run: uv pip install --editable ".[docs,train]" + run: uv pip install --editable ".[docs,model]" - name: Deploy versioned documentation env: diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 6a21559..2557d8a 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -131,7 +131,7 @@ jobs: enable-cache: true - name: Install integration dependencies - run: uv pip install --editable ".[test,docs,all]" + run: uv pip install --editable ".[test,docs,model]" - name: Test training CLI run: safe-train --help diff --git a/CHANGELOG.md b/CHANGELOG.md index 587840b..e705199 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,21 @@ This file records user-visible changes. See the [migration guide](docs/migration for upgrade instructions and [GitHub releases](https://github.com/datamol-io/safe/releases) for earlier release notes. +## 0.2.1 - 2026-09-03 + +### Changed + +- Simplify the install extras. The base `safe-mol` install now includes molecule + visualization, and a single `safe-mol[model]` extra provides the full model + stack (SAFE-GPT inference, the `safe-train` CLI and Weights & Biases logging). + Encoding, decoding and `safe.split` continue to work from the base install + without PyTorch. + +### Removed + +- Remove the `train`, `viz`, `wandb` and `all` extras. Use `safe-mol` for the + notation core (with visualization) and `safe-mol[model]` for everything else. + ## 0.2.0 - 2026-09-03 ### Highlights diff --git a/README.md b/README.md index 705b4f6..e22e2fc 100644 --- a/README.md +++ b/README.md @@ -76,29 +76,27 @@ uv add safe-mol # or: pip install safe-mol mamba install -c conda-forge safe-mol ``` -The core install is lightweight — only encoding, decoding and notation -splitting — and needs no PyTorch, so it also runs on Mac Intel. Optional -features are available as extras: - -| Install | Adds | -| ------------------ | ---------------------------------------------------------- | -| `safe-mol` | Core notation: `safe.encode`, `safe.decode`, `safe.split` | -| `safe-mol[model]` | `SAFETokenizer` and `SAFEDesign` (SAFE-GPT inference) | -| `safe-mol[train]` | The model stack plus the `safe-train` CLI | -| `safe-mol[viz]` | Molecule visualization helpers | -| `safe-mol[wandb]` | Weights & Biases logging | -| `safe-mol[all]` | Every maintained feature above | +The base install covers the notation core (`safe.encode`, `safe.decode`, +`safe.split`) and molecule visualization, and needs no PyTorch — so it runs +anywhere, including Mac Intel. The model stack is a single optional extra: + +| Install | Includes | +| ----------------- | ------------------------------------------------------------------------------ | +| `safe-mol` | Notation core + molecule visualization | +| `safe-mol[model]` | Everything above **plus** SAFE-GPT inference, the `safe-train` CLI and W&B logging | ```bash uv add "safe-mol[model]" # or: pip install "safe-mol[model]" ``` -Model APIs keep their top-level imports but load their dependencies only when -used. The `model` and `train` extras require PyTorch 2.5+; official Mac Intel -wheels stop at 2.2, so use Linux, Windows or Apple Silicon for that stack. The -optional model stack uses Transformers 5, and RDKit 2026.03 is excluded because -of an upstream stereochemistry regression (RDKit 2024.09 through 2025.09 are -covered by CI). See the [migration guide](docs/migration.md) for details. +Installing the extra always includes the base, so `safe-mol[model]` gives you +the core, visualization and the full model stack. Model APIs keep their +top-level imports but load their dependencies only when used. The `model` extra +requires PyTorch 2.5+; official Mac Intel wheels stop at 2.2, so use Linux, +Windows or Apple Silicon for that stack. It uses Transformers 5, and RDKit +2026.03 is excluded because of an upstream stereochemistry regression (RDKit +2024.09 through 2025.09 are covered by CI). See the +[migration guide](docs/migration.md) for details. For GPU workloads, install the PyTorch build matching your CUDA driver before installing SAFE. You can verify the resulting environment with: diff --git a/docs/index.md b/docs/index.md index b5d9b3c..7a558a8 100644 --- a/docs/index.md +++ b/docs/index.md @@ -78,30 +78,28 @@ uv add safe-mol # or: pip install safe-mol mamba install -c conda-forge safe-mol ``` -The core install is lightweight — only encoding, decoding and notation -splitting — and needs no PyTorch, so it also runs on Mac Intel. Optional -features are available as extras: - -| Install | Adds | -| ------------------ | ---------------------------------------------------------- | -| `safe-mol` | Core notation: `safe.encode`, `safe.decode`, `safe.split` | -| `safe-mol[model]` | `SAFETokenizer` and `SAFEDesign` (SAFE-GPT inference) | -| `safe-mol[train]` | The model stack plus the `safe-train` CLI | -| `safe-mol[viz]` | Molecule visualization helpers | -| `safe-mol[wandb]` | Weights & Biases logging | -| `safe-mol[all]` | Every maintained feature above | +The base install covers the notation core (`safe.encode`, `safe.decode`, +`safe.split`) and molecule visualization, and needs no PyTorch — so it runs +anywhere, including Mac Intel. The model stack is a single optional extra: + +| Install | Includes | +| ----------------- | ------------------------------------------------------------------------------ | +| `safe-mol` | Notation core + molecule visualization | +| `safe-mol[model]` | Everything above **plus** SAFE-GPT inference, the `safe-train` CLI and W&B logging | ```bash uv add "safe-mol[model]" # or: pip install "safe-mol[model]" ``` -Model APIs keep their top-level imports but load their dependencies only when -used. The `model` and `train` extras require PyTorch 2.5+; official Mac Intel -wheels stop at 2.2, so use Linux, Windows or Apple Silicon for that stack. The -optional model stack uses Transformers 5, and RDKit 2026.03 is excluded because -of an upstream stereochemistry regression (RDKit 2024.09 through 2025.09 are -covered by CI). Read [Migrating to SAFE 0.2.0](migration.md) before upgrading an -existing environment. +Installing the extra always includes the base, so `safe-mol[model]` gives you +the core, visualization and the full model stack. Model APIs keep their +top-level imports but load their dependencies only when used. The `model` extra +requires PyTorch 2.5+; official Mac Intel wheels stop at 2.2, so use Linux, +Windows or Apple Silicon for that stack. It uses Transformers 5, and RDKit +2026.03 is excluded because of an upstream stereochemistry regression (RDKit +2024.09 through 2025.09 are covered by CI). Read +[Migrating to SAFE 0.2.0](migration.md) before upgrading an existing +environment. For constrained design, `try_hard=True` is the opt-in quality mode. It oversamples, validates the molecular and substructure constraints, and removes diff --git a/docs/migration.md b/docs/migration.md index af8bf67..e8fedac 100644 --- a/docs/migration.md +++ b/docs/migration.md @@ -6,8 +6,8 @@ SAFE 0.2.0 is a maintenance-focused release. It keeps the established encoding, - Python 3.11 through 3.14 is supported. Python 3.9 and 3.10 are no longer tested. - The minimum RDKit release is 2024.09. RDKit 2026.03 is deliberately excluded because that series changes double-bond direction handling during fragmentation and can silently lose stereochemistry in an otherwise valid SAFE round trip. The compatibility matrix uses RDKit 2024.09, 2025.03, and 2025.09. -- Model and training extras require PyTorch 2.5 or newer. Official macOS Intel - wheels stop at PyTorch 2.2, so those extras are not supported natively there. +- The `model` extra requires PyTorch 2.5 or newer. Official macOS Intel + wheels stop at PyTorch 2.2, so it is not supported natively there. The notation core works without PyTorch, including alongside Molfeat on Intel. - Transformers 5 is the supported generation stack. SAFE maintains greedy, multinomial, beam, beam-sampling and the constrained-beam path required by @@ -23,13 +23,13 @@ download or execute this backend. Contrastive and diverse beam search are no longer wrapped by SAFE; advanced Transformers experiments should call the underlying model directly. -The default installation is now the molecular notation core: encoding, -decoding and `safe.split`. PyTorch, Transformers, Tokenizers, tqdm and fsspec -move to the `model` extra. `SAFEDesign` and `SAFETokenizer` keep their public -top-level names and load that stack only when used. The `train` extra includes -the model stack plus Datasets, Evaluate, Accelerate and universal-pathlib. -Matplotlib and Weights & Biases remain isolated in `viz` and `wandb`; `all` -installs every maintained feature. +The base installation is the molecular notation core (encoding, decoding and +`safe.split`) plus molecule visualization, and needs no PyTorch. The single +`model` extra adds the full model stack: PyTorch, Transformers, Tokenizers, +SAFE-GPT inference, the `safe-train` CLI (Datasets, Evaluate, Accelerate, +universal-pathlib) and Weights & Biases logging. `SAFEDesign` and +`SAFETokenizer` keep their public top-level names and load that stack only when +used. Recreate the environment rather than upgrading it in place: @@ -37,9 +37,8 @@ Recreate the environment rather than upgrading it in place: uv sync --all-extras ``` -`env.yml` remains a supported Conda alternative. For a pip training -environment, install `safe-mol[train]` (or `safe-mol[train,wandb]` if -experiment reporting is required). +`env.yml` remains a supported Conda alternative. For a pip environment with the +model, training and experiment-reporting stack, install `safe-mol[model]`. For GPU installations, install the PyTorch build appropriate for the CUDA driver before installing SAFE. diff --git a/env.yml b/env.yml index 7994950..0e8bec5 100644 --- a/env.yml +++ b/env.yml @@ -6,4 +6,4 @@ dependencies: - python >=3.11 - pip >=24 - pip: - - -e .[all,dev,docs,test] + - -e .[model,dev,docs,test] diff --git a/expts/README.md b/expts/README.md index b8017a7..cb3b270 100644 --- a/expts/README.md +++ b/expts/README.md @@ -5,6 +5,6 @@ original SAFE work. The Slurm scripts contain cluster-specific paths and are not portable installation or training examples. They are not part of the installed Python package. -For the maintained training entry point, install `safe-mol[train]` and use +For the maintained training entry point, install `safe-mol[model]` and use `safe-train --help`. See the [migration guide](../docs/migration.md) for the current dependency and tokenizer requirements. diff --git a/pyproject.toml b/pyproject.toml index fbdf4ce..980096e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -37,42 +37,22 @@ keywords = ["safe", "smiles", "de novo", "design", "molecules"] dependencies = [ "datamol>=0.12.5", "loguru>=0.7", + "matplotlib>=3.8", "networkx>=3.2", "numpy>=1.26", + "pillow>=10", "rdkit>=2024.9.1,!=2026.3.*", ] [project.optional-dependencies] +# The model stack: SAFE-GPT inference, the `safe-train` CLI and W&B logging. +# The notation core (and visualization) ship in the base install. model = [ - "fsspec>=2023.9.2", - "huggingface-hub>=1.5,<2", - "tokenizers>=0.23.1,<0.24", - "torch>=2.5", - "tqdm>=4.66", - "transformers>=5,<6", -] -train = [ - "accelerate>=1.1", - "datasets>=4", - "evaluate>=0.4.3", - "fsspec>=2023.9.2", - "huggingface-hub>=1.5,<2", - "tokenizers>=0.23.1,<0.24", - "torch>=2.5", - "tqdm>=4.66", - "transformers>=5,<6", - "universal-pathlib>=0.2", -] -viz = ["matplotlib>=3.8", "pillow>=10"] -wandb = ["fsspec>=2023.9.2", "wandb>=0.18"] -all = [ "accelerate>=1.1", "datasets>=4", "evaluate>=0.4.3", "fsspec>=2023.9.2", "huggingface-hub>=1.5,<2", - "matplotlib>=3.8", - "pillow>=10", "tokenizers>=0.23.1,<0.24", "torch>=2.5", "tqdm>=4.66", diff --git a/safe/__init__.py b/safe/__init__.py index 8d234cb..d3e0aac 100644 --- a/safe/__init__.py +++ b/safe/__init__.py @@ -14,11 +14,14 @@ __version__ = "unknown" +# Third element is the pip install target that provides the feature. The +# notation core and visualization ship in the base install; the model stack +# (inference, training and W&B logging) is the single `model` extra. _LAZY_IMPORTS = { - "SAFEDesign": (".sample", "SAFEDesign", "model"), - "SAFETokenizer": (".tokenizer", "SAFETokenizer", "model"), - "to_image": (".viz", "to_image", "viz"), - "upload_to_wandb": (".io", "upload_to_wandb", "wandb"), + "SAFEDesign": (".sample", "SAFEDesign", "safe-mol[model]"), + "SAFETokenizer": (".tokenizer", "SAFETokenizer", "safe-mol[model]"), + "to_image": (".viz", "to_image", "safe-mol"), + "upload_to_wandb": (".io", "upload_to_wandb", "safe-mol[model]"), } @@ -28,7 +31,7 @@ def __getattr__(name): value = import_module(".trainer", __name__) except ModuleNotFoundError as error: raise ImportError( - 'SAFE training requires: python -m pip install "safe-mol[train]"' + 'SAFE training requires: python -m pip install "safe-mol[model]"' ) from error globals()[name] = value return value @@ -36,12 +39,12 @@ def __getattr__(name): target = _LAZY_IMPORTS.get(name) if target is None: raise AttributeError(f"module {__name__!r} has no attribute {name!r}") - module_name, attribute, extra = target + module_name, attribute, install_spec = target try: value = getattr(import_module(module_name, __name__), attribute) except ModuleNotFoundError as error: raise ImportError( - f'SAFE {name} support requires: python -m pip install "safe-mol[{extra}]"' + f'SAFE {name} support requires: python -m pip install "{install_spec}"' ) from error globals()[name] = value return value