diff --git a/multiverse/experiments/tables.py b/multiverse/experiments/tables.py index 6376ef0..6815471 100644 --- a/multiverse/experiments/tables.py +++ b/multiverse/experiments/tables.py @@ -412,6 +412,12 @@ def _missing_by_estimator(frames, estimators, common, datasets): "the series are length 8 and MRHydra requires at least 9, so the dataset " "cannot complete while MRHydra is a column" ), + "InsectWingbeat": ( + "25,000 training cases of 200 channels. Only the Dummy baseline has a " + "result on it: none of the classifiers in the Multiverse archive paper " + "finished it within the resource limits there, so it cannot enter a " + "table that needs every estimator" + ), "BenzeneConcentration_disc": ( "removed from Multiverse-core. The results here are on version 1, whose " "PT08.S2 channel is a deterministic function of the target; version 2 " diff --git a/results/multiverse/Dummy/Dummy_accuracy.csv b/results/multiverse/Dummy/Dummy_accuracy.csv index b80fb5a..f0f5990 100644 --- a/results/multiverse/Dummy/Dummy_accuracy.csv +++ b/results/multiverse/Dummy/Dummy_accuracy.csv @@ -23,6 +23,8 @@ CharacterTrajectories,0.06476323119777158 CounterMovementJump,0.33519553072625696 Cricket,0.08333333333333333 CrowdSourced,0.5001764913519238 +DREAMERA,0.3934409619021886 +DREAMERV,0.5811267225074305 DuckDuckGeese,0.2 ERing,0.16666666666666666 EigenWorms,0.4198473282442748 @@ -53,6 +55,7 @@ IRDS-STR,0.9166666666666666 ImaginedFeetHands,0.5002583979328166 ImaginedOpenCloseFist,0.49947089947089945 InnerSpeech,0.235 +InsectWingbeat,0.1 JapaneseVowels,0.08378378378378379 KERAAL-CTK,0.5 KERAAL-CTK-MC,0.5 @@ -70,6 +73,7 @@ KINECAL-GGFV,0.9230769230769231 KINECAL-QSEC,0.9444444444444444 KINECAL-QSEO,0.9411764705882353 LSST,0.3150851581508516 +LenDB,0.5092767890950397 Libras,0.06666666666666667 LiveFuelMoistureContent_disc,0.8410596026490066 Locust2022,0.9111852076386784 @@ -92,6 +96,10 @@ PhonemeSpectra,0.025648672830301224 PhotoStimulation,0.4166666666666667 PronouncedSpeech,0.23 RacketSports,0.28289473684210525 +S2Agri-10pc-17,0.6091353993993388 +S2Agri-10pc-34,0.6091353993993388 +S2Agri-17,0.5962028198994797 +S2Agri-34,0.5962028198994797 SPHERE-WUS,0.5 STEW,0.5 SelfRegulationSCP1,0.5017064846416383 diff --git a/results/multiverse/Dummy/Dummy_auroc.csv b/results/multiverse/Dummy/Dummy_auroc.csv index e5e415b..f22abfa 100644 --- a/results/multiverse/Dummy/Dummy_auroc.csv +++ b/results/multiverse/Dummy/Dummy_auroc.csv @@ -23,6 +23,8 @@ CharacterTrajectories,0.5 CounterMovementJump,0.5 Cricket,0.5 CrowdSourced,0.5 +DREAMERA,0.5 +DREAMERV,0.5 DuckDuckGeese,0.5 ERing,0.5 EigenWorms,0.5 @@ -53,6 +55,7 @@ IRDS-STR,0.5 ImaginedFeetHands,0.5 ImaginedOpenCloseFist,0.5 InnerSpeech,0.5 +InsectWingbeat,0.5 JapaneseVowels,0.5 KERAAL-CTK,0.5 KERAAL-CTK-MC,0.5 @@ -70,6 +73,7 @@ KINECAL-GGFV,0.5 KINECAL-QSEC,0.5 KINECAL-QSEO,0.5 LSST,0.5 +LenDB,0.5 Libras,0.5 LiveFuelMoistureContent_disc,0.5 Locust2022,0.5 @@ -92,6 +96,10 @@ PhonemeSpectra,0.5 PhotoStimulation,0.5 PronouncedSpeech,0.5 RacketSports,0.5 +S2Agri-10pc-17,0.5 +S2Agri-10pc-34,0.5 +S2Agri-17,0.5 +S2Agri-34,0.5 SPHERE-WUS,0.5 STEW,0.5 SelfRegulationSCP1,0.5 diff --git a/results/multiverse/Dummy/Dummy_balacc.csv b/results/multiverse/Dummy/Dummy_balacc.csv index 03cdd2b..8484de8 100644 --- a/results/multiverse/Dummy/Dummy_balacc.csv +++ b/results/multiverse/Dummy/Dummy_balacc.csv @@ -23,6 +23,8 @@ CharacterTrajectories,0.05 CounterMovementJump,0.3333333333333333 Cricket,0.08333333333333333 CrowdSourced,0.5 +DREAMERA,0.5 +DREAMERV,0.5 DuckDuckGeese,0.2 ERing,0.16666666666666666 EigenWorms,0.2 @@ -53,6 +55,7 @@ IRDS-STR,0.5 ImaginedFeetHands,0.5 ImaginedOpenCloseFist,0.5 InnerSpeech,0.25 +InsectWingbeat,0.1 JapaneseVowels,0.1111111111111111 KERAAL-CTK,0.5 KERAAL-CTK-MC,0.3333333333333333 @@ -70,6 +73,7 @@ KINECAL-GGFV,0.5 KINECAL-QSEC,0.5 KINECAL-QSEO,0.5 LSST,0.07142857142857142 +LenDB,0.5 Libras,0.06666666666666667 LiveFuelMoistureContent_disc,0.5 Locust2022,0.5 @@ -92,6 +96,10 @@ PhonemeSpectra,0.02564102564102564 PhotoStimulation,0.3333333333333333 PronouncedSpeech,0.25 RacketSports,0.25 +S2Agri-10pc-17,0.058823529411764705 +S2Agri-10pc-34,0.034482758620689655 +S2Agri-17,0.058823529411764705 +S2Agri-34,0.030303030303030304 SPHERE-WUS,0.5 STEW,0.5 SelfRegulationSCP1,0.5 diff --git a/results/multiverse/Dummy/Dummy_f1.csv b/results/multiverse/Dummy/Dummy_f1.csv index d6a8b3b..6ef25e9 100644 --- a/results/multiverse/Dummy/Dummy_f1.csv +++ b/results/multiverse/Dummy/Dummy_f1.csv @@ -23,6 +23,8 @@ CharacterTrajectories,0.007878326358917932 CounterMovementJump,0.16829901124330893 Cricket,0.012820512820512822 CrowdSourced,0.0 +DREAMERA,0.564704171413336 +DREAMERV,0.0 DuckDuckGeese,0.06666666666666667 ERing,0.047619047619047616 EigenWorms,0.24829680702618406 @@ -53,6 +55,7 @@ IRDS-STR,0.0 ImaginedFeetHands,0.0 ImaginedOpenCloseFist,0.666196189131969 InnerSpeech,0.08943319838056679 +InsectWingbeat,0.01818181818181818 JapaneseVowels,0.01295410123340298 KERAAL-CTK,0.0 KERAAL-CTK-MC,0.3333333333333333 @@ -70,6 +73,7 @@ KINECAL-GGFV,0.0 KINECAL-QSEC,0.0 KINECAL-QSEO,0.0 LSST,0.15098437735628223 +LenDB,0.0 Libras,0.008333333333333333 LiveFuelMoistureContent_disc,0.0 Locust2022,0.0 @@ -92,6 +96,10 @@ PhonemeSpectra,0.0012828065503959901 PhotoStimulation,0.2450980392156863 PronouncedSpeech,0.08601626016260162 RacketSports,0.1247638326585695 +S2Agri-10pc-17,0.4611742864396579 +S2Agri-10pc-34,0.4611742864396579 +S2Agri-17,0.4453792438212535 +S2Agri-34,0.4453792438212535 SPHERE-WUS,0.0 STEW,0.6666666666666666 SelfRegulationSCP1,0.0 diff --git a/results/multiverse/Dummy/Dummy_logloss.csv b/results/multiverse/Dummy/Dummy_logloss.csv index 9c989e5..6502681 100644 --- a/results/multiverse/Dummy/Dummy_logloss.csv +++ b/results/multiverse/Dummy/Dummy_logloss.csv @@ -23,6 +23,8 @@ CharacterTrajectories,2.9873893422378415 CounterMovementJump,1.0986313429481223 Cricket,2.4849066497880004 CrowdSourced,0.6931471282693898 +DREAMERA,0.7178228248184128 +DREAMERV,0.6801109256783351 DuckDuckGeese,1.6094379124341003 ERing,1.791759469228055 EigenWorms,1.4718633766740754 @@ -53,6 +55,7 @@ IRDS-STR,0.29463965020934135 ImaginedFeetHands,0.6931478243680321 ImaginedOpenCloseFist,0.6932899710021728 InnerSpeech,1.3922348857254991 +InsectWingbeat,2.302585092994046 JapaneseVowels,2.197224577336219 KERAAL-CTK,0.7266593439339404 KERAAL-CTK-MC,1.191181400090801 @@ -70,6 +73,7 @@ KINECAL-GGFV,0.27491925815268553 KINECAL-QSEC,0.2146132851891638 KINECAL-QSEO,0.22371807606583374 LSST,2.1321750076663113 +LenDB,0.6930252160849788 Libras,2.70805020110221 LiveFuelMoistureContent_disc,0.43812361168597996 Locust2022,0.2997872434658028 @@ -92,6 +96,10 @@ PhonemeSpectra,3.6635616461296454 PhotoStimulation,1.0833754191909117 PronouncedSpeech,1.3893835552665177 RacketSports,1.3817041078077663 +S2Agri-10pc-17,1.0603060277768492 +S2Agri-10pc-34,1.313611111652872 +S2Agri-17,1.0990464414887853 +S2Agri-34,1.363923798049471 SPHERE-WUS,0.7465166183739191 STEW,0.6931471805599454 SelfRegulationSCP1,0.6931495567879038 diff --git a/results/multiverse/Dummy/Dummy_sensitivity.csv b/results/multiverse/Dummy/Dummy_sensitivity.csv index a808323..ca57d5a 100644 --- a/results/multiverse/Dummy/Dummy_sensitivity.csv +++ b/results/multiverse/Dummy/Dummy_sensitivity.csv @@ -23,6 +23,8 @@ CharacterTrajectories,0.06476323119777158 CounterMovementJump,0.33519553072625696 Cricket,0.08333333333333333 CrowdSourced,0.0 +DREAMERA,1.0 +DREAMERV,0.0 DuckDuckGeese,0.2 ERing,0.16666666666666666 EigenWorms,0.4198473282442748 @@ -53,6 +55,7 @@ IRDS-STR,0.0 ImaginedFeetHands,0.0 ImaginedOpenCloseFist,1.0 InnerSpeech,0.235 +InsectWingbeat,0.1 JapaneseVowels,0.08378378378378379 KERAAL-CTK,0.0 KERAAL-CTK-MC,0.5 @@ -70,6 +73,7 @@ KINECAL-GGFV,0.0 KINECAL-QSEC,0.0 KINECAL-QSEO,0.0 LSST,0.3150851581508516 +LenDB,0.0 Libras,0.06666666666666667 LiveFuelMoistureContent_disc,0.0 Locust2022,0.0 @@ -92,6 +96,10 @@ PhonemeSpectra,0.025648672830301224 PhotoStimulation,0.4166666666666667 PronouncedSpeech,0.23 RacketSports,0.28289473684210525 +S2Agri-10pc-17,0.6091353993993388 +S2Agri-10pc-34,0.6091353993993388 +S2Agri-17,0.5962028198994797 +S2Agri-34,0.5962028198994797 SPHERE-WUS,0.0 STEW,1.0 SelfRegulationSCP1,0.0 diff --git a/results/multiverse/Dummy/Dummy_specificity.csv b/results/multiverse/Dummy/Dummy_specificity.csv index 77e8b39..9a72b86 100644 --- a/results/multiverse/Dummy/Dummy_specificity.csv +++ b/results/multiverse/Dummy/Dummy_specificity.csv @@ -23,6 +23,8 @@ CharacterTrajectories,0.06476323119777158 CounterMovementJump,0.33519553072625696 Cricket,0.08333333333333333 CrowdSourced,1.0 +DREAMERA,0.0 +DREAMERV,1.0 DuckDuckGeese,0.2 ERing,0.16666666666666666 EigenWorms,0.4198473282442748 @@ -53,6 +55,7 @@ IRDS-STR,1.0 ImaginedFeetHands,1.0 ImaginedOpenCloseFist,0.0 InnerSpeech,0.235 +InsectWingbeat,0.1 JapaneseVowels,0.08378378378378379 KERAAL-CTK,1.0 KERAAL-CTK-MC,0.5 @@ -70,6 +73,7 @@ KINECAL-GGFV,1.0 KINECAL-QSEC,1.0 KINECAL-QSEO,1.0 LSST,0.3150851581508516 +LenDB,1.0 Libras,0.06666666666666667 LiveFuelMoistureContent_disc,1.0 Locust2022,1.0 @@ -92,6 +96,10 @@ PhonemeSpectra,0.025648672830301224 PhotoStimulation,0.4166666666666667 PronouncedSpeech,0.23 RacketSports,0.28289473684210525 +S2Agri-10pc-17,0.6091353993993388 +S2Agri-10pc-34,0.6091353993993388 +S2Agri-17,0.5962028198994797 +S2Agri-34,0.5962028198994797 SPHERE-WUS,1.0 STEW,0.0 SelfRegulationSCP1,1.0 diff --git a/results/multiverse/leaderboard.html b/results/multiverse/leaderboard.html index 69d1641..c17cfc0 100644 --- a/results/multiverse/leaderboard.html +++ b/results/multiverse/leaderboard.html @@ -60,7 +60,7 @@ details { margin-top: .6rem; } summary { cursor: pointer; color: var(--accent); } code { font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: .9em; } -

Multiverse-core leaderboard

25 estimators on 57 datasets · 7 metrics · ordered by average accuracy rank · built 2026-09-24

#EstimatorAccuracyBalanced accuracyAUROCF1Log loss ↓SensitivitySpecificity
ScoreRankScoreRankScoreRankScoreRankScoreRankScoreRankScoreRank
1HC20.79387.400.75438.480.89515.950.73788.110.53165.810.75708.500.79877.11
2MRHydra0.78388.770.75298.610.807417.720.73368.657.790921.540.76099.070.78289.27
3RDST0.77529.180.73969.460.796817.940.71799.648.101221.610.728910.290.78919.00
4RIST0.77429.820.743010.460.86778.320.721310.660.61238.120.746911.110.772410.69
5CIF0.777510.160.746310.310.88438.350.729310.180.64149.110.752111.060.776710.39
6DrCIF0.772210.370.740210.900.87348.770.721611.240.64139.090.743711.710.771811.18
7Arsenal0.770910.800.733610.970.846913.250.714511.003.638217.640.734011.180.779210.65
8QUANT0.765110.920.736210.900.86788.320.717710.980.72637.630.750611.000.755412.10
9ROCKET0.770111.040.734510.810.792718.790.713411.288.286222.410.731511.740.778111.31
10LITETime-MV0.748311.240.726010.210.84939.900.685110.201.339513.120.71299.980.766811.25
11STSF0.769911.410.744710.860.870010.230.714511.540.67428.400.737911.980.782712.22
12H-InceptionTime0.737911.890.713711.150.843410.390.683711.001.366213.670.719210.890.738912.55
13Catch220.753912.900.722413.240.868010.830.702713.460.697210.490.729013.720.750713.47
14ConvTran0.744613.120.711012.880.852010.980.684612.740.907010.110.721912.750.735613.95
15PatchMTSC0.744313.310.695913.750.824712.530.668213.300.79059.530.699612.920.738113.59
16DisjointCNN0.722413.640.697512.900.823911.610.661113.162.134415.180.681412.430.729912.82
17STC0.751813.860.710614.430.863711.740.685314.410.64289.770.710414.540.759913.31
18TSF0.741914.150.713813.650.855812.240.692613.880.991410.750.714114.510.749814.26
19TDE0.727214.940.684315.060.837912.710.660114.490.845011.460.691914.140.730414.04
20TS2Vec0.719215.410.679015.450.799515.320.656215.180.786911.580.690715.280.712515.47
21Summary0.693616.520.661416.040.819415.720.639316.160.950513.390.664816.390.697917.08
22XCM0.669116.800.635917.040.791514.940.585916.222.169916.280.629915.330.676915.62
23TimesURL0.697517.230.656517.210.783318.850.620617.590.996215.950.645617.520.699816.65
24TimesNet0.700117.410.664716.950.820916.040.639917.251.295715.300.680416.470.691917.68
25Dummy0.374822.700.307223.290.500023.560.183622.681.381017.090.324820.490.377419.34

Average score and average rank over the 57 datasets with results for every estimator on every metric. Best in each column is highlighted. Metrics marked ↓ are better when lower.

Missing results

Scoring uses the 57 datasets every estimator completed, so a dataset any one of them is missing is left out for all. Reasons are from the job logs of these runs.

Notes on listed estimators

Datasets not included

Held out of the collection rather than reported as missing, because no scheduling closes them. Results that do exist for them remain in the repository.

Estimators not listed

Their results remain in the repository under results/multiverse/. Removing an estimator that cannot finish the archive returns the datasets it alone was missing to every other estimator, which is why the scored count above is larger than the number of datasets any single run completed.

Reproducing this page

from multiverse.experiments.tables import leaderboard
+

Multiverse-core leaderboard

25 estimators on 57 datasets · 7 metrics · ordered by average accuracy rank · built 2026-09-24

#EstimatorAccuracyBalanced accuracyAUROCF1Log loss ↓SensitivitySpecificity
ScoreRankScoreRankScoreRankScoreRankScoreRankScoreRankScoreRank
1HC20.79387.400.75438.480.89515.950.73788.110.53165.810.75708.500.79877.11
2MRHydra0.78388.770.75298.610.807417.720.73368.657.790921.540.76099.070.78289.27
3RDST0.77529.180.73969.460.796817.940.71799.648.101221.610.728910.290.78919.00
4RIST0.77429.820.743010.460.86778.320.721310.660.61238.120.746911.110.772410.69
5CIF0.777510.160.746310.310.88438.350.729310.180.64149.110.752111.060.776710.39
6DrCIF0.772210.370.740210.900.87348.770.721611.240.64139.090.743711.710.771811.18
7Arsenal0.770910.800.733610.970.846913.250.714511.003.638217.640.734011.180.779210.65
8QUANT0.765110.920.736210.900.86788.320.717710.980.72637.630.750611.000.755412.10
9ROCKET0.770111.040.734510.810.792718.790.713411.288.286222.410.731511.740.778111.31
10LITETime-MV0.748311.240.726010.210.84939.900.685110.201.339513.120.71299.980.766811.25
11STSF0.769911.410.744710.860.870010.230.714511.540.67428.400.737911.980.782712.22
12H-InceptionTime0.737911.890.713711.150.843410.390.683711.001.366213.670.719210.890.738912.55
13Catch220.753912.900.722413.240.868010.830.702713.460.697210.490.729013.720.750713.47
14ConvTran0.744613.120.711012.880.852010.980.684612.740.907010.110.721912.750.735613.95
15PatchMTSC0.744313.310.695913.750.824712.530.668213.300.79059.530.699612.920.738113.59
16DisjointCNN0.722413.640.697512.900.823911.610.661113.162.134415.180.681412.430.729912.82
17STC0.751813.860.710614.430.863711.740.685314.410.64289.770.710414.540.759913.31
18TSF0.741914.150.713813.650.855812.240.692613.880.991410.750.714114.510.749814.26
19TDE0.727214.940.684315.060.837912.710.660114.490.845011.460.691914.140.730414.04
20TS2Vec0.719215.410.679015.450.799515.320.656215.180.786911.580.690715.280.712515.47
21Summary0.693616.520.661416.040.819415.720.639316.160.950513.390.664816.390.697917.08
22XCM0.669116.800.635917.040.791514.940.585916.222.169916.280.629915.330.676915.62
23TimesURL0.697517.230.656517.210.783318.850.620617.590.996215.950.645617.520.699816.65
24TimesNet0.700117.410.664716.950.820916.040.639917.251.295715.300.680416.470.691917.68
25Dummy0.374822.700.307223.290.500023.560.183622.681.381017.090.324820.490.377419.34

Average score and average rank over the 57 datasets with results for every estimator on every metric. Best in each column is highlighted. Metrics marked ↓ are better when lower.

Missing results

  • MRHydra — Tiselac (LAPACK integer overflow in the RidgeClassifierCV SVD (aeon issue 3738))
  • RDST — Tiselac (LAPACK integer overflow in the RidgeClassifierCV SVD (aeon issue 3738)); USCActivity (Segmentation fault (core dumped))
  • ConvTran — Alzheimers, EigenWorms, PhotoStimulation (CUDA out of memory)
  • TS2Vec — Locust2022, Tiselac, USCActivity (Timed out at 60 hours)

Scoring uses the 57 datasets every estimator completed, so a dataset any one of them is missing is left out for all. Reasons are from the job logs of these runs.

Notes on listed estimators

  • XCM — run at fixed parameters, a single fit at window 0.8 with batch 32, not the per-dataset cross-validated search over window and batch size that the paper describes. The search was run and did not pay: across the 65 shared datasets it was 0.017 mean accuracy worse, 31 wins to 31 with 3 ties, Wilcoxon p = 0.63. On the 14 datasets where the search selected 0.8, the window used here, the two runs still differed by 0.11 mean absolute accuracy and by as much as 0.48, so at one resample XCM's run-to-run variance is larger than the effect the search is tuning for.

Datasets not included

  • AustraliaRainfall_disc — 112186 cases, and three estimators fail on it for reasons compute cannot fix. RDST and ROCKET hit LAPACK integer overflow in RidgeClassifierCV's SVD, aeon issue 3738, after 14 and 12 attempts; MRHydra exhausted 128 GB over 13.
  • PenDigits — the series are length 8 and MRHydra requires at least 9, so the dataset cannot complete while MRHydra is a column.
  • InsectWingbeat — 25,000 training cases of 200 channels. Only the Dummy baseline has a result on it: none of the classifiers in the Multiverse archive paper finished it within the resource limits there, so it cannot enter a table that needs every estimator.
  • BenzeneConcentration_disc — removed from Multiverse-core. The results here are on version 1, whose PT08.S2 channel is a deterministic function of the target; version 2 drops it, on the advice of the original UCI Air Quality donors: https://zenodo.org/records/21871727.

Held out of the collection rather than reported as missing, because no scheduling closes them. Results that do exist for them remain in the repository.

Estimators not listed

  • LiteTIME — LITE is a univariate architecture. The multivariate variant of the same method is listed here as LITETime-MV.
  • FreshPRINCE — cannot complete the archive at the memory available: recorded OOM at 128 GB after eight attempts each on FaceDetection, FordChallenge and Skoda, and 38 on Tiselac.
  • 1NN-DTW — cannot complete the archive within the walltime available: exceeded the limit on BIDMC32HR_disc, with no result recorded for BIDMC32SpO2_disc.
  • MUSE — cannot complete the archive as implemented: on STEW and Skoda its word bag outgrows the 32-bit sparse indices scikit-learn's classifier accepts, which no memory allocation fixes, and on USCActivity its chi2 selection needs 287 GiB. MotorImagery exhausted 16 GB and awaits a rerun at the configuration of its other results.
  • CIF-500 — the 500-tree configuration run for the Multiverse archive paper's full-archive benchmark, where it is reported as CIF. This table reports the default CIF; the full-archive leaderboard reports CIF-500.
  • DrCIF-500 — the 500-tree configuration HC2 uses internally, run for the Multiverse archive paper's full-archive benchmark, where it is reported as DrCIF. This table reports the default DrCIF; the full-archive leaderboard reports DrCIF-500.
  • DisjointCNN-Aeon — aeon's implementation applies a Permute after the final block, so its pooling reduces the wrong axes and the classifier head receives one feature instead of 64 (aeon issue #3775). Held as evidence for that issue; the port of the same method reports as DisjointCNN.

Their results remain in the repository under results/multiverse/. Removing an estimator that cannot finish the archive returns the datasets it alone was missing to every other estimator, which is why the scored count above is larger than the number of datasets any single run completed.

Reproducing this page

from multiverse.experiments.tables import leaderboard
 
 leaderboard(
     datasets=[...]  # 57 datasets,
diff --git a/results/multiverse/leaderboard_uea.html b/results/multiverse/leaderboard_uea.html
index 0a47367..158c19b 100644
--- a/results/multiverse/leaderboard_uea.html
+++ b/results/multiverse/leaderboard_uea.html
@@ -60,7 +60,7 @@
 details { margin-top: .6rem; }
 summary { cursor: pointer; color: var(--accent); }
 code { font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: .9em; }
-

UEA leaderboard

25 estimators on 24 datasets · 7 metrics · ordered by average accuracy rank · built 2026-09-24

#EstimatorAccuracyBalanced accuracyAUROCF1Log loss ↓SensitivitySpecificity
ScoreRankScoreRankScoreRankScoreRankScoreRankScoreRankScoreRank
1HC20.76176.650.74127.560.87526.770.73727.710.66815.880.74298.210.76556.52
2RDST0.74078.270.72508.560.809818.330.71978.549.344820.830.72129.060.75128.29
3MRHydra0.74628.770.73328.620.814518.400.73718.299.148020.880.75268.330.73158.85
4Arsenal0.72829.380.71039.960.837514.600.70849.675.357517.420.708410.310.73808.73
5H-InceptionTime0.72089.560.72149.080.86058.330.69629.791.503711.290.70469.710.732310.08
6ROCKET0.72659.650.710110.040.800419.400.708410.319.859521.960.708410.960.73419.50
7RIST0.737410.040.722610.250.86608.690.726110.440.79339.710.737010.790.729210.46
8CIF0.747510.420.733410.580.87428.750.735610.350.841010.880.755010.670.730710.90
9LITETime-MV0.704810.960.704010.170.85058.580.67699.771.481510.500.68949.670.718110.65
10DrCIF0.732811.100.719911.080.86399.920.719311.980.838611.040.732411.790.725012.10
11QUANT0.724512.540.713612.290.88038.190.716112.420.79788.750.738311.880.703513.52
12DisjointCNN0.694312.620.696111.690.83879.710.662712.541.883911.920.683011.350.704411.79
13STSF0.730912.710.719112.650.870311.190.691413.170.82559.380.698213.500.755612.29
14PatchMTSC0.709613.710.697713.400.854910.600.690613.060.76197.500.722012.940.687414.58
15TS2Vec0.699013.960.683914.270.833413.850.685313.710.882011.710.708813.770.678314.31
16TDE0.702614.190.681814.270.838613.080.674513.941.127712.420.687714.100.701613.79
17ConvTran0.691514.230.679113.810.849311.440.678413.350.80909.750.703213.790.672515.06
18TSF0.718314.500.705214.210.860011.810.690114.040.901810.960.696214.480.732414.10
19STC0.722414.620.700415.040.872211.330.700014.750.805210.250.713815.290.715713.85
20Catch220.694514.980.680015.480.844312.730.683615.190.969013.460.702114.830.676215.44
21XCM0.635615.980.620416.520.818413.790.589916.691.350713.750.617215.440.641015.75
22TimesURL0.674517.040.660016.850.817018.290.647317.121.332617.460.661217.620.670316.27
23TimesNet0.658217.920.650517.210.827516.150.640017.671.148513.540.664817.170.648118.77
24Summary0.643118.060.631417.980.818117.380.617917.791.274516.000.626917.620.652117.75
25Dummy0.228623.150.210623.420.500023.690.104422.711.861517.790.219321.710.219321.62

Average score and average rank over the 24 datasets with results for every estimator on every metric. Best in each column is highlighted. Metrics marked ↓ are better when lower.

Missing results

  • CIF — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • DrCIF — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • DisjointCNN — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • STSF — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • TS2Vec — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • ConvTran — EigenWorms (CUDA out of memory)
  • TSF — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • XCM — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • Summary — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)

Scoring uses the 24 datasets every estimator completed, so a dataset any one of them is missing is left out for all. Reasons are from the job logs of these runs.

1 requested dataset(s) have no results from any estimator: InsectWingbeat.

Notes on listed estimators

  • XCM — run at fixed parameters, a single fit at window 0.8 with batch 32, not the per-dataset cross-validated search over window and batch size that the paper describes. The search was run and did not pay: across the 65 shared datasets it was 0.017 mean accuracy worse, 31 wins to 31 with 3 ties, Wilcoxon p = 0.63. On the 14 datasets where the search selected 0.8, the window used here, the two runs still differed by 0.11 mean absolute accuracy and by as much as 0.48, so at one resample XCM's run-to-run variance is larger than the effect the search is tuning for.

Datasets not included

  • AustraliaRainfall_disc — 112186 cases, and three estimators fail on it for reasons compute cannot fix. RDST and ROCKET hit LAPACK integer overflow in RidgeClassifierCV's SVD, aeon issue 3738, after 14 and 12 attempts; MRHydra exhausted 128 GB over 13.
  • PenDigits — the series are length 8 and MRHydra requires at least 9, so the dataset cannot complete while MRHydra is a column.
  • BenzeneConcentration_disc — removed from Multiverse-core. The results here are on version 1, whose PT08.S2 channel is a deterministic function of the target; version 2 drops it, on the advice of the original UCI Air Quality donors: https://zenodo.org/records/21871727.

Held out of the collection rather than reported as missing, because no scheduling closes them. Results that do exist for them remain in the repository.

Estimators not listed

  • LiteTIME — LITE is a univariate architecture. The multivariate variant of the same method is listed here as LITETime-MV.
  • FreshPRINCE — cannot complete the archive at the memory available: recorded OOM at 128 GB after eight attempts each on FaceDetection, FordChallenge and Skoda, and 38 on Tiselac.
  • 1NN-DTW — cannot complete the archive within the walltime available: exceeded the limit on BIDMC32HR_disc, with no result recorded for BIDMC32SpO2_disc.
  • MUSE — cannot complete the archive as implemented: on STEW and Skoda its word bag outgrows the 32-bit sparse indices scikit-learn's classifier accepts, which no memory allocation fixes, and on USCActivity its chi2 selection needs 287 GiB. MotorImagery exhausted 16 GB and awaits a rerun at the configuration of its other results.
  • CIF-500 — the 500-tree configuration run for the Multiverse archive paper's full-archive benchmark, where it is reported as CIF. This table reports the default CIF; the full-archive leaderboard reports CIF-500.
  • DrCIF-500 — the 500-tree configuration HC2 uses internally, run for the Multiverse archive paper's full-archive benchmark, where it is reported as DrCIF. This table reports the default DrCIF; the full-archive leaderboard reports DrCIF-500.
  • DisjointCNN-Aeon — aeon's implementation applies a Permute after the final block, so its pooling reduces the wrong axes and the classifier head receives one feature instead of 64 (aeon issue #3775). Held as evidence for that issue; the port of the same method reports as DisjointCNN.

Their results remain in the repository under results/multiverse/. Removing an estimator that cannot finish the archive returns the datasets it alone was missing to every other estimator, which is why the scored count above is larger than the number of datasets any single run completed.

Reproducing this page

from multiverse.experiments.tables import leaderboard
+

UEA leaderboard

25 estimators on 24 datasets · 7 metrics · ordered by average accuracy rank · built 2026-09-24

#EstimatorAccuracyBalanced accuracyAUROCF1Log loss ↓SensitivitySpecificity
ScoreRankScoreRankScoreRankScoreRankScoreRankScoreRankScoreRank
1HC20.76176.650.74127.560.87526.770.73727.710.66815.880.74298.210.76556.52
2RDST0.74078.270.72508.560.809818.330.71978.549.344820.830.72129.060.75128.29
3MRHydra0.74628.770.73328.620.814518.400.73718.299.148020.880.75268.330.73158.85
4Arsenal0.72829.380.71039.960.837514.600.70849.675.357517.420.708410.310.73808.73
5H-InceptionTime0.72089.560.72149.080.86058.330.69629.791.503711.290.70469.710.732310.08
6ROCKET0.72659.650.710110.040.800419.400.708410.319.859521.960.708410.960.73419.50
7RIST0.737410.040.722610.250.86608.690.726110.440.79339.710.737010.790.729210.46
8CIF0.747510.420.733410.580.87428.750.735610.350.841010.880.755010.670.730710.90
9LITETime-MV0.704810.960.704010.170.85058.580.67699.771.481510.500.68949.670.718110.65
10DrCIF0.732811.100.719911.080.86399.920.719311.980.838611.040.732411.790.725012.10
11QUANT0.724512.540.713612.290.88038.190.716112.420.79788.750.738311.880.703513.52
12DisjointCNN0.694312.620.696111.690.83879.710.662712.541.883911.920.683011.350.704411.79
13STSF0.730912.710.719112.650.870311.190.691413.170.82559.380.698213.500.755612.29
14PatchMTSC0.709613.710.697713.400.854910.600.690613.060.76197.500.722012.940.687414.58
15TS2Vec0.699013.960.683914.270.833413.850.685313.710.882011.710.708813.770.678314.31
16TDE0.702614.190.681814.270.838613.080.674513.941.127712.420.687714.100.701613.79
17ConvTran0.691514.230.679113.810.849311.440.678413.350.80909.750.703213.790.672515.06
18TSF0.718314.500.705214.210.860011.810.690114.040.901810.960.696214.480.732414.10
19STC0.722414.620.700415.040.872211.330.700014.750.805210.250.713815.290.715713.85
20Catch220.694514.980.680015.480.844312.730.683615.190.969013.460.702114.830.676215.44
21XCM0.635615.980.620416.520.818413.790.589916.691.350713.750.617215.440.641015.75
22TimesURL0.674517.040.660016.850.817018.290.647317.121.332617.460.661217.620.670316.27
23TimesNet0.658217.920.650517.210.827516.150.640017.671.148513.540.664817.170.648118.77
24Summary0.643118.060.631417.980.818117.380.617917.791.274516.000.626917.620.652117.75
25Dummy0.228623.150.210623.420.500023.690.104422.711.861517.790.219321.710.219321.62

Average score and average rank over the 24 datasets with results for every estimator on every metric. Best in each column is highlighted. Metrics marked ↓ are better when lower.

Missing results

  • CIF — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • DrCIF — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • DisjointCNN — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • STSF — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • TS2Vec — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • ConvTran — EigenWorms (CUDA out of memory)
  • TSF — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • XCM — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)
  • Summary — BasicMotions, FingerMovements, SelfRegulationSCP2 (not run outside Multiverse-core)

Scoring uses the 24 datasets every estimator completed, so a dataset any one of them is missing is left out for all. Reasons are from the job logs of these runs.

Notes on listed estimators

  • XCM — run at fixed parameters, a single fit at window 0.8 with batch 32, not the per-dataset cross-validated search over window and batch size that the paper describes. The search was run and did not pay: across the 65 shared datasets it was 0.017 mean accuracy worse, 31 wins to 31 with 3 ties, Wilcoxon p = 0.63. On the 14 datasets where the search selected 0.8, the window used here, the two runs still differed by 0.11 mean absolute accuracy and by as much as 0.48, so at one resample XCM's run-to-run variance is larger than the effect the search is tuning for.

Datasets not included

  • AustraliaRainfall_disc — 112186 cases, and three estimators fail on it for reasons compute cannot fix. RDST and ROCKET hit LAPACK integer overflow in RidgeClassifierCV's SVD, aeon issue 3738, after 14 and 12 attempts; MRHydra exhausted 128 GB over 13.
  • PenDigits — the series are length 8 and MRHydra requires at least 9, so the dataset cannot complete while MRHydra is a column.
  • InsectWingbeat — 25,000 training cases of 200 channels. Only the Dummy baseline has a result on it: none of the classifiers in the Multiverse archive paper finished it within the resource limits there, so it cannot enter a table that needs every estimator.
  • BenzeneConcentration_disc — removed from Multiverse-core. The results here are on version 1, whose PT08.S2 channel is a deterministic function of the target; version 2 drops it, on the advice of the original UCI Air Quality donors: https://zenodo.org/records/21871727.

Held out of the collection rather than reported as missing, because no scheduling closes them. Results that do exist for them remain in the repository.

Estimators not listed

  • LiteTIME — LITE is a univariate architecture. The multivariate variant of the same method is listed here as LITETime-MV.
  • FreshPRINCE — cannot complete the archive at the memory available: recorded OOM at 128 GB after eight attempts each on FaceDetection, FordChallenge and Skoda, and 38 on Tiselac.
  • 1NN-DTW — cannot complete the archive within the walltime available: exceeded the limit on BIDMC32HR_disc, with no result recorded for BIDMC32SpO2_disc.
  • MUSE — cannot complete the archive as implemented: on STEW and Skoda its word bag outgrows the 32-bit sparse indices scikit-learn's classifier accepts, which no memory allocation fixes, and on USCActivity its chi2 selection needs 287 GiB. MotorImagery exhausted 16 GB and awaits a rerun at the configuration of its other results.
  • CIF-500 — the 500-tree configuration run for the Multiverse archive paper's full-archive benchmark, where it is reported as CIF. This table reports the default CIF; the full-archive leaderboard reports CIF-500.
  • DrCIF-500 — the 500-tree configuration HC2 uses internally, run for the Multiverse archive paper's full-archive benchmark, where it is reported as DrCIF. This table reports the default DrCIF; the full-archive leaderboard reports DrCIF-500.
  • DisjointCNN-Aeon — aeon's implementation applies a Permute after the final block, so its pooling reduces the wrong axes and the classifier head receives one feature instead of 64 (aeon issue #3775). Held as evidence for that issue; the port of the same method reports as DisjointCNN.

Their results remain in the repository under results/multiverse/. Removing an estimator that cannot finish the archive returns the datasets it alone was missing to every other estimator, which is why the scored count above is larger than the number of datasets any single run completed.

Reproducing this page

from multiverse.experiments.tables import leaderboard
 
 leaderboard(
     datasets=[...]  # 24 datasets,