diff --git a/CHANGELOG.md b/CHANGELOG.md index 3fcf353..99542e5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,14 @@ Releases before 0.7.0 are recorded in the git history. +## 0.7.2 + +### Fixed + +- `scoring.neutral_contribution` now neutralizes the rank-Gaussianized meta model + as well as the predictions before orthogonalizing. Predictions identical to + the meta model now have zero neutral contribution, up to floating-point precision. + ## 0.7.1 ### Added diff --git a/README.md b/README.md index dd9ae18..3f4b839 100644 --- a/README.md +++ b/README.md @@ -17,7 +17,9 @@ pip install numerai-tools does not re-rank and re-power the predictions after neutralizing them. - `neutral_contribution` is `correlation_contribution` with that same neutralization step inserted after the rank/gaussianize step. It - neutralizes the submissions only; pass a meta model that is already neutral. + neutralizes both the submissions and the meta model against the same + neutralizers before orthogonalizing. Predictions identical to the meta model + have zero neutral contribution, up to floating-point precision. - The `submissions.py` module provides helper functions to ensure your submissions are valid and formatted correctly. Use this in your automated prediction pipelines to ensure uploads don't fail. diff --git a/RELEASING.md b/RELEASING.md index 9582eb2..b6535c0 100644 --- a/RELEASING.md +++ b/RELEASING.md @@ -201,9 +201,8 @@ attempt that will fail on a duplicate version. and nothing else in the tree hardcodes it, so `[project] version` is the single place to edit. CI reads it with `tomllib`, not a regex. -**There is no CHANGELOG.md.** numerapi's flow includes a changelog edit at each -version bump; this repo has never had one, and the release notes live in the PR -title and `README.md`. Adding one is optional and not assumed anywhere above. +**Update `CHANGELOG.md` when bumping the version.** Record the release's changes +under the matching version heading. **The PyPI secret is `PYPI_API_KEY`**, as in numerai-cli, not `PYPI_API_TOKEN` as in numerapi. Publishing uses `poetry publish --build` — it is the path this diff --git a/numerai_tools/scoring.py b/numerai_tools/scoring.py index fadb397..9e2b85a 100644 --- a/numerai_tools/scoring.py +++ b/numerai_tools/scoring.py @@ -405,20 +405,21 @@ def neutral_contribution( ) -> pd.Series: """Calculate how much the given predictions contribute to the given Meta Model's correlation with the target, after neutralizing the - predictions against the given neutralizers. + predictions and meta model against the given neutralizers. This is correlation_contribution with a neutralization step inserted after the rank/gaussianize step: 1. tie-kept ranking each prediction and the meta model 2. gaussianizing each prediction and the meta model - 3. neutralizing each prediction wrt the neutralizers - 4. orthogonalizing each prediction wrt the meta model + 3. neutralizing each prediction and the meta model wrt the neutralizers + 4. orthogonalizing each neutralized prediction wrt the neutralized meta model 5. dot product the orthogonalized predictions and the targets then normalize by the length of the target (equivalent to covariance) - The meta model is **not** neutralized: this score is defined against an - already-neutral meta model (the Signals v3NUSWMM), so neutralizing it again - would be a no-op. Pass an already-neutral meta model. + Both predictions and the meta model are neutralized after rank/gaussianize, + which can reintroduce neutralizer exposure even for an already-neutral input. + Identical predictions and meta model therefore have zero contribution, + up to floating-point precision. No 1.5 power is applied to the predictions or the targets, and no variance normalization is applied to either: dividing each prediction by its own @@ -427,7 +428,7 @@ def neutral_contribution( Arguments: predictions: pd.DataFrame - the predictions to evaluate - meta_model: pd.Series - the already-neutral meta model to evaluate against + meta_model: pd.Series - the meta model to evaluate against neutralizers: pd.DataFrame - the neutralizer data with features as columns live_targets: pd.Series - the live targets to evaluate against top_bottom: Optional[int] - the number of top and bottom predictions to use @@ -444,9 +445,11 @@ def neutral_contribution( ) # rank and normalize meta model and predictions so mean=0 and std=1, - # then neutralize the predictions wrt the neutralizers + # then neutralize both wrt the same neutralizers p = neutralize(gaussian(tie_kept_rank(predictions)), neutralizers).values - m = gaussian(tie_kept_rank(meta_model.to_frame())).iloc[:, 0].values + m = neutralize( + gaussian(tie_kept_rank(meta_model.to_frame())), neutralizers + ).iloc[:, 0].values # orthogonalize predictions wrt meta model neutral_preds = orthogonalize(p, cast(np.ndarray, m)) diff --git a/pyproject.toml b/pyproject.toml index ed67573..9d7ed08 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,7 @@ [project] name = "numerai-tools" -version = "0.7.1" +version = "0.7.2" + description = "A collection of open-source tools to help interact with Numerai, model data, and automate submissions." authors = [ diff --git a/tests/test_scoring.py b/tests/test_scoring.py index b350b5e..d1a3d28 100644 --- a/tests/test_scoring.py +++ b/tests/test_scoring.py @@ -435,43 +435,53 @@ def test_neutral_contribution(self): predictions, neutralizers, meta_model, targets = neutral_fixture() np.testing.assert_allclose( neutral_contribution(predictions, meta_model, neutralizers, targets), - [0.008166920268846335, 0.056287525943196595, -0.1465734703333085], + [0.0076186790250345705, 0.05538415279085515, -0.1464960350754568], ) np.testing.assert_allclose( neutral_contribution( predictions, meta_model, neutralizers, targets, top_bottom=20 ), - [0.014212728439240246, 0.07892827246673215, -0.13904192918920918], + [0.013225166729302146, 0.09427407739360479, -0.1390320473771724], ) - def test_neutral_contribution_does_not_neutralize_meta_model(self): - # RESOLVED (T-803): the meta model passed in is the v3NUSWMM, which is - # already neutral, so only the submissions are neutralized. This test - # fails if the meta model is neutralized inside the function. + def test_neutral_contribution_neutralizes_meta_model(self): predictions, neutralizers, meta_model, targets = neutral_fixture() - scores = neutral_contribution(predictions, meta_model, neutralizers, targets) neutral_preds = neutralize( gaussian(tie_kept_rank(predictions)), neutralizers ).values - raw_mm = gaussian(tie_kept_rank(meta_model.to_frame()))[meta_model.name] - neutralized_mm = neutralize(raw_mm.to_frame(), neutralizers)[meta_model.name] - np.testing.assert_allclose( - scores, - contribution_scores( - orthogonalize(neutral_preds, raw_mm.values), - targets.copy(), - predictions, - ), - ) - assert not np.allclose( - scores, - contribution_scores( - orthogonalize(neutral_preds, neutralized_mm.values), - targets.copy(), - predictions, - ), - atol=1e-6, - ) + neutral_mm = neutralize( + gaussian(tie_kept_rank(meta_model.to_frame())), neutralizers + ).iloc[:, 0].values + for top_bottom in (None, 20): + with self.subTest(top_bottom=top_bottom): + np.testing.assert_allclose( + neutral_contribution( + predictions, meta_model, neutralizers, targets, top_bottom + ), + contribution_scores( + orthogonalize(neutral_preds, neutral_mm), + targets, + predictions, + top_bottom, + ), + ) + + def test_neutral_contribution_identical_predictions(self): + predictions, neutralizers, _, targets = neutral_fixture() + # Cover different neutralizer exposures and ties, with multiple columns + # scored together so matching the meta model is a per-column property. + predictions["tied"] = predictions["mixed"].round(1) + for column in predictions: + for top_bottom in (None, 20): + with self.subTest(column=column, top_bottom=top_bottom): + scores = neutral_contribution( + predictions, + predictions[column], + neutralizers, + targets, + top_bottom, + ) + np.testing.assert_allclose(scores[column], 0.0, atol=1e-12) def test_neutral_contribution_no_variance_normalize(self): # variance normalizing the neutralized predictions divides each score by @@ -479,11 +489,13 @@ def test_neutral_contribution_no_variance_normalize(self): # contribution to predictions that were mostly neutralizer exposure. predictions, neutralizers, meta_model, targets = neutral_fixture() neutral_preds = neutralize(gaussian(tie_kept_rank(predictions)), neutralizers) - raw_mm = gaussian(tie_kept_rank(meta_model.to_frame()))[meta_model.name] + neutral_mm = neutralize( + gaussian(tie_kept_rank(meta_model.to_frame())), neutralizers + ).iloc[:, 0] assert not np.allclose( neutral_contribution(predictions, meta_model, neutralizers, targets), contribution_scores( - orthogonalize(variance_normalize(neutral_preds).values, raw_mm.values), + orthogonalize(variance_normalize(neutral_preds).values, neutral_mm.values), targets.copy(), predictions, ),