Index _ | A | B | C | D | E | F | G | I | K | L | M | N | P | R | S | T | V | W _ __commit_id__ (in module evaluma._version) __getitem__() (evaluma.benchmark.BenchmarkGroup method) (evaluma.BenchmarkGroup method) __version__ (in module evaluma) (in module evaluma._version) __version_tuple__ (in module evaluma._version) _AGG_MODES (in module evaluma.methods.aggregate) _aggregate_scores() (in module evaluma.methods.aggregate) _benchmarks (evaluma.benchmark.BenchmarkGroup attribute) (evaluma.BenchmarkGroup attribute) _common_options() (in module evaluma.cli) _compute_error_matrix() (in module evaluma.methods.improvability) _dataset_metric_map (evaluma.Benchmark attribute) (evaluma.benchmark.Benchmark attribute) _delta_table() (in module evaluma.methods.rank_sensitivity) _fit_elo() (in module evaluma.methods.elo) _holm_correction() (in module evaluma.methods.frequentist) _iterative_elo() (in module evaluma.methods.elo) _KNOWN_LIST (in module evaluma.metric_registry) _load_bench() (in module evaluma.cli) _metric_direction (evaluma.Benchmark attribute) (evaluma.benchmark.Benchmark attribute) _METRIC_ERROR_OPTIMUM (in module evaluma.metric_registry) _new() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) _norm_ref_high (evaluma.Benchmark attribute) (evaluma.benchmark.Benchmark attribute) _norm_ref_low (evaluma.Benchmark attribute) (evaluma.benchmark.Benchmark attribute) _normalize() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) _parse_metric_direction() (in module evaluma.cli) _rank_vector() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) _ranks_from_scores() (in module evaluma.methods.rank_sensitivity) _raw (evaluma.Benchmark attribute) (evaluma.benchmark.Benchmark attribute) _raw_runs (evaluma.Benchmark attribute) (evaluma.benchmark.Benchmark attribute) _resolve_bound() (in module evaluma.normalize) _resolve_metric_type_bounds() (in module evaluma) _save() (in module evaluma.cli) _validate_aligned_axes() (in module evaluma.methods.rank_sensitivity) _validate_callable_rank_vector() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) A agg (evaluma.RankSensitivityResult attribute) (evaluma.results.RankSensitivityResult attribute) aggregate() (in module evaluma.cli) aggregate_ranking() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) AggregateResult (class in evaluma.results) alpha (evaluma.FrequentistResult attribute) (evaluma.results.FrequentistResult attribute) aup (evaluma.results.ProfileResult property) avg_ranks (evaluma.FrequentistResult attribute) (evaluma.results.FrequentistResult attribute) B bayesian_comparison() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) BayesianResult (class in evaluma.results) Benchmark (class in evaluma) (class in evaluma.benchmark) BenchmarkGroup (class in evaluma) (class in evaluma.benchmark) C cd (evaluma.FrequentistResult attribute) (evaluma.results.FrequentistResult attribute) commit_id (in module evaluma._version) compare() (in module evaluma.cli) compute_aggregate() (in module evaluma.methods.aggregate) compute_battles() (in module evaluma.methods.elo) compute_battles_from_runs() (in module evaluma.methods.elo) compute_bayesian() (in module evaluma.methods.bayesian) compute_elo() (in module evaluma.methods.elo) compute_frequentist() (in module evaluma.methods.frequentist) compute_improvability() (in module evaluma.methods.improvability) compute_iqm() (in module evaluma.methods.iqm) compute_profiles() (in module evaluma.methods.profiles) compute_rank_sensitivity() (in module evaluma.methods.rank_sensitivity) compute_rank_sensitivity_from_ranks() (in module evaluma.methods.rank_sensitivity) compute_winrate_matrix() (in module evaluma.methods.elo) cond_a (evaluma.RankSensitivityResult attribute) (evaluma.results.RankSensitivityResult attribute) cond_b (evaluma.RankSensitivityResult attribute) (evaluma.results.RankSensitivityResult attribute) D datasets_ (evaluma.Benchmark property) (evaluma.benchmark.Benchmark property) drop_datasets() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) (evaluma.benchmark.BenchmarkGroup method) (evaluma.BenchmarkGroup method) drop_incomplete() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) drop_models() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) (evaluma.benchmark.BenchmarkGroup method) (evaluma.BenchmarkGroup method) E elo_ranking() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) EloResult (class in evaluma) (class in evaluma.results) evaluma module evaluma._version module evaluma.benchmark module evaluma.cli module evaluma.methods module evaluma.methods.aggregate module evaluma.methods.bayesian module evaluma.methods.elo module evaluma.methods.frequentist module evaluma.methods.improvability module evaluma.methods.iqm module evaluma.methods.profiles module evaluma.methods.rank_sensitivity module evaluma.metric_registry module evaluma.normalize module evaluma.plot module evaluma.results module F frequentist() (in module evaluma.cli) frequentist_comparison() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) FrequentistResult (class in evaluma) (class in evaluma.results) friedman_p_value (evaluma.FrequentistResult attribute) (evaluma.results.FrequentistResult attribute) friedman_statistic (evaluma.FrequentistResult attribute) (evaluma.results.FrequentistResult attribute) G get_direction() (in module evaluma.metric_registry) get_error_optimum() (in module evaluma.metric_registry) get_natural_bounds() (in module evaluma.metric_registry) I improvability_ranking() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) ImprovabilityResult (class in evaluma) (class in evaluma.results) iqm_ranking() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) IQMResult (class in evaluma.results) K KNOWN_METRICS (in module evaluma.metric_registry) L load_csv() (in module evaluma) load_df() (in module evaluma) logger (in module evaluma.methods.elo) M main() (in module evaluma.cli) models_ (evaluma.Benchmark property) (evaluma.benchmark.Benchmark property) module evaluma evaluma._version evaluma.benchmark evaluma.cli evaluma.methods evaluma.methods.aggregate evaluma.methods.bayesian evaluma.methods.elo evaluma.methods.frequentist evaluma.methods.improvability evaluma.methods.iqm evaluma.methods.profiles evaluma.methods.rank_sensitivity evaluma.metric_registry evaluma.normalize evaluma.plot evaluma.results N normalize() (in module evaluma.normalize) P per_dataset (evaluma.ImprovabilityResult attribute) (evaluma.results.ImprovabilityResult attribute) performance_profiles() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) plot() (evaluma.EloResult method) (evaluma.FrequentistResult method) (evaluma.ImprovabilityResult method) (evaluma.RankSensitivityResult method) (evaluma.results.AggregateResult method) (evaluma.results.BayesianResult method) (evaluma.results.EloResult method) (evaluma.results.FrequentistResult method) (evaluma.results.ImprovabilityResult method) (evaluma.results.IQMResult method) (evaluma.results.ProfileResult method) (evaluma.results.RankSensitivityResult method) plot_aggregate_ranking() (in module evaluma.plot) plot_bayesian_heatmap() (in module evaluma.plot) plot_bayesian_reference_bars() (in module evaluma.plot) plot_cd_diagram() (in module evaluma.plot) plot_elo_ranking() (in module evaluma.plot) plot_frequentist_reference_bars() (in module evaluma.plot) plot_improvability_ranking() (in module evaluma.plot) plot_iqm_ranking() (in module evaluma.plot) plot_performance_profiles() (in module evaluma.plot) plot_rank_sensitivity() (in module evaluma.plot) plot_winrate() (evaluma.EloResult method) (evaluma.results.EloResult method) plot_winrate_matrix() (in module evaluma.plot) ProfileResult (class in evaluma.results) profiles() (in module evaluma.cli) R rank() (in module evaluma.cli) rank_sensitivity() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) (evaluma.benchmark.BenchmarkGroup method) (evaluma.BenchmarkGroup method) RankSensitivityResult (class in evaluma) (class in evaluma.results) reference (evaluma.FrequentistResult attribute) (evaluma.results.BayesianResult attribute) (evaluma.results.FrequentistResult attribute) report() (in module evaluma.cli) rho (evaluma.RankSensitivityResult attribute) (evaluma.results.RankSensitivityResult attribute) S scores_ (evaluma.Benchmark property) (evaluma.benchmark.Benchmark property) select_datasets() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) (evaluma.benchmark.BenchmarkGroup method) (evaluma.BenchmarkGroup method) select_models() (evaluma.Benchmark method) (evaluma.benchmark.Benchmark method) (evaluma.benchmark.BenchmarkGroup method) (evaluma.BenchmarkGroup method) T table (evaluma.EloResult attribute) (evaluma.FrequentistResult attribute) (evaluma.ImprovabilityResult attribute) (evaluma.RankSensitivityResult attribute) (evaluma.results.AggregateResult attribute) (evaluma.results.BayesianResult attribute) (evaluma.results.EloResult attribute) (evaluma.results.FrequentistResult attribute) (evaluma.results.ImprovabilityResult attribute) (evaluma.results.IQMResult attribute) (evaluma.results.ProfileResult attribute) (evaluma.results.RankSensitivityResult attribute) tau (evaluma.RankSensitivityResult attribute) (evaluma.results.RankSensitivityResult attribute) tau_ci (evaluma.RankSensitivityResult attribute) (evaluma.results.RankSensitivityResult attribute) V version (in module evaluma._version) version_tuple (in module evaluma._version) W winrate_matrix (evaluma.EloResult attribute) (evaluma.results.EloResult attribute)