import sys from pathlib import Path sys.path.append(str(Path(__file__).parent)) import pandas as pd import streamlit as st from src.strings import ( CITATION_FEV, CITATION_HEADER, FEV_BENCHMARK_DETAILS, PAIRWISE_BENCHMARK_DETAILS, get_pivot_legend, ) from src.task_groups import ( ALL_TASKS, DOMAIN_GROUPS, FREQUENCY_GROUPS, MINI_TASKS, TASK_TYPE_GROUPS, ) from src.utils import ( COLORS, DEFAULT_VISIBLE_MODEL_TYPES, MODEL_TYPE_LABELS, TAG_COLORS, construct_pairwise_chart, format_leaderboard, format_metric_name, get_metric_description, get_model_display_name, select_model_names, validate_model_metadata, ) # Leaderboard computation vendored from the fev repo by save_tables.py, so the numbers shown here are # produced by the same code as the paper figures. from src.vendor.fev_bench_compute import ( AVAILABLE_METRICS, BASELINE_MODEL, LEAKAGE_IMPUTATION_MODEL, SORT_COL, TOP_K_MODELS_TO_PLOT, compute_leaderboard, compute_pairwise, ) from streamlit.elements.lib.column_types import ColumnConfig st.set_page_config( layout="wide", page_title="fev leaderboard", page_icon=":material/trophy:" ) TITLE = "

fev-bench

" MODEL_TYPES_HELP = ( "Win rates are computed among the included models. " f"{BASELINE_MODEL} and {LEAKAGE_IMPUTATION_MODEL} are always included, as they are used for imputation. " 'Click "See details" below the leaderboard for a description of each model type.' ) NON_COMMERCIAL_HELP = "Include models whose license does not permit commercial use." # Group type options GROUP_TYPES = [ "Full (100 tasks)", "Mini (20 tasks)", "By frequency", "By domain", "By task type", ] FREQUENCY_OPTIONS = list(FREQUENCY_GROUPS.keys()) DOMAIN_OPTIONS = list(DOMAIN_GROUPS.keys()) TASK_TYPE_OPTIONS = list(TASK_TYPE_GROUPS.keys()) def get_subset_description( group_type: str, subgroup: str | None, num_tasks: int, metric_name: str ) -> str: """Generate a description of the current subset.""" base = f"Results for various forecasting models, measured using the **{metric_name}** metric, on **{num_tasks} tasks**" if group_type == "Full (100 tasks)": subset_desc = "from the full **fev-bench** benchmark" elif group_type == "Mini (20 tasks)": subset_desc = "from the **fev-bench-mini** subset" elif group_type == "By frequency": subset_desc = f"with **{subgroup.lower()}** frequency" elif group_type == "By task type": subset_desc = f"of type **{subgroup.lower()}**" else: # By domain subset_desc = f"from the **{subgroup}** domain" paper_link = "[fev-bench: A Realistic Benchmark for Time Series Forecasting](https://arxiv.org/abs/2509.26468)" return f"{base} {subset_desc}, as described in {paper_link}." @st.cache_data() def get_summaries() -> pd.DataFrame: return pd.read_csv("tables/summaries.csv") @st.cache_data() def get_leaderboard( metric_name: str, task_list: tuple[str, ...], model_names: tuple[str, ...] ) -> pd.DataFrame: """Leaderboard for the selected tasks and models. Win rate and skill score depend on which models are included, so these are computed here rather than precomputed: including an extra model type changes every row. """ return compute_leaderboard(_filter_summaries(metric_name, task_list, model_names), metric_name) @st.cache_data() def get_pairwise( metric_name: str, task_list: tuple[str, ...], model_names: tuple[str, ...] ) -> pd.DataFrame: summaries = _filter_summaries(metric_name, task_list, model_names) leaderboard_df = get_leaderboard(metric_name, task_list, model_names) top_k_models = ( leaderboard_df.sort_values(by=SORT_COL, ascending=False) .head(TOP_K_MODELS_TO_PLOT)["model_name"] .tolist() ) return compute_pairwise(summaries, metric_name, top_k_models) def _filter_summaries( metric_name: str, task_list: tuple[str, ...], model_names: tuple[str, ...] ) -> pd.DataFrame: summaries = get_summaries() # The baseline and leakage-imputation models must stay in the data for imputation to work, even # if the reader excluded their type. included = set(model_names) | {BASELINE_MODEL, LEAKAGE_IMPUTATION_MODEL} return summaries[ summaries["task_name"].isin(task_list) & summaries["model_name"].isin(included) ] @st.cache_data() def get_pivot_table( metric_name: str, ) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]: pivot_df = pd.read_csv(f"tables/pivot_{metric_name}.csv") baseline_imputed = pd.read_csv(f"tables/pivot_{metric_name}_baseline_imputed.csv") leakage_imputed = pd.read_csv(f"tables/pivot_{metric_name}_leakage_imputed.csv") return pivot_df, baseline_imputed, leakage_imputed with st.sidebar: # Task group selection selected_group_type = st.selectbox("Subset", options=GROUP_TYPES) # Conditional sub-selection for frequency/domain selected_subgroup = None if selected_group_type == "By frequency": selected_subgroup = st.selectbox("Frequency", options=FREQUENCY_OPTIONS) elif selected_group_type == "By domain": selected_subgroup = st.selectbox("Domain", options=DOMAIN_OPTIONS) elif selected_group_type == "By task type": selected_subgroup = st.selectbox("Task type", options=TASK_TYPE_OPTIONS) # Determine the tasks making up the selected subset if selected_group_type in ["Full (100 tasks)", "Mini (20 tasks)"]: task_list = ( ALL_TASKS if selected_group_type == "Full (100 tasks)" else MINI_TASKS ) else: if selected_group_type == "By frequency": task_list = FREQUENCY_GROUPS[selected_subgroup] elif selected_group_type == "By task type": task_list = TASK_TYPE_GROUPS[selected_subgroup] else: task_list = DOMAIN_GROUPS[selected_subgroup] st.caption(f"{len(task_list)} tasks") st.divider() selected_metric = st.selectbox( "Evaluation Metric", options=AVAILABLE_METRICS, format_func=format_metric_name ) st.caption(get_metric_description(selected_metric)) st.divider() included_model_types = st.pills( "Model types", options=list(MODEL_TYPE_LABELS), format_func=MODEL_TYPE_LABELS.get, selection_mode="multi", default=list(DEFAULT_VISIBLE_MODEL_TYPES), help=MODEL_TYPES_HELP, ) include_non_commercial = st.toggle( "Include non-commercial models", value=True, help=NON_COMMERCIAL_HELP ) included_models = tuple( select_model_names(included_model_types, include_non_commercial) ) if included_models: # The imputation models are always part of the computation, so count them too num_models = len( set(included_models) | {BASELINE_MODEL, LEAKAGE_IMPUTATION_MODEL} ) st.caption(f"{num_models} models") cols = st.columns(spec=[0.025, 0.95, 0.025]) with cols[1] as main_container: st.markdown(TITLE, unsafe_allow_html=True) if not included_models: st.warning("No models match the current selection. Select at least one model type in the sidebar.") st.stop() task_names = tuple(task_list) metric_df = get_leaderboard(selected_metric, task_names, included_models).sort_values( by=SORT_COL, ascending=False ) validate_model_metadata(metric_df["model_name"]) pairwise_df = get_pairwise(selected_metric, task_names, included_models) st.markdown("## :material/trophy: Leaderboard", unsafe_allow_html=True) st.markdown( get_subset_description( selected_group_type, selected_subgroup, len(task_list), selected_metric ), unsafe_allow_html=True, ) df_styled = format_leaderboard(metric_df) st.dataframe( df_styled, width="stretch", hide_index=True, column_config={ "model_name": st.column_config.LinkColumn( label="Model", display_text=r"#(.*)$" ), "win_rate": st.column_config.NumberColumn( label="Win rate (%)", format="%.1f" ), "skill_score": st.column_config.NumberColumn( label="Skill score (%)", format="%.1f" ), "median_inference_time_s_per100": st.column_config.NumberColumn( label="Runtime (s / 100 series)", format="%.1f", help="Median end-to-end runtime per 100 time series.", ), "training_corpus_overlap": st.column_config.NumberColumn( label="Leakage (%)", format="%d" ), "num_failures": st.column_config.NumberColumn( label="Failed (%)", format="%.0f", help="Percentage of tasks where the model failed to produce a forecast.", ), "tags": st.column_config.MultiselectColumn( label="Tags", options=list(TAG_COLORS), color=list(TAG_COLORS.values()), help="Model type; Zero-shot = no training on the task; Non-commercial = license does not permit commercial use.", ), "org": ColumnConfig(label="Organization", alignment="left"), }, ) with st.expander("See details"): st.markdown(FEV_BENCHMARK_DETAILS, unsafe_allow_html=True) st.markdown("## :material/bar_chart: Pairwise comparison", unsafe_allow_html=True) chart_col_1, _, chart_col_2 = st.columns(spec=[0.45, 0.1, 0.45]) with chart_col_1: st.altair_chart( construct_pairwise_chart( pairwise_df, col="win_rate", metric_name=selected_metric ), use_container_width=True, ) with chart_col_2: st.altair_chart( construct_pairwise_chart( pairwise_df, col="skill_score", metric_name=selected_metric ), use_container_width=True, ) with st.expander("See details"): st.markdown(PAIRWISE_BENCHMARK_DETAILS, unsafe_allow_html=True) st.markdown( "## :material/table_chart: Results for individual tasks", unsafe_allow_html=True ) with st.expander("Show detailed results"): st.markdown( get_pivot_legend("Seasonal Naive", "Chronos-Bolt"), unsafe_allow_html=True ) pivot_df, baseline_imputed, leakage_imputed = get_pivot_table(selected_metric) pivot_df = pivot_df.set_index("Task name") baseline_imputed = baseline_imputed.set_index("Task name") leakage_imputed = leakage_imputed.set_index("Task name") # Keep only the models included by the sidebar toggles excluded_cols = [c for c in pivot_df.columns if c not in included_models] pivot_df = pivot_df.drop(columns=excluded_cols) baseline_imputed = baseline_imputed.drop(columns=excluded_cols) leakage_imputed = leakage_imputed.drop(columns=excluded_cols) # Filter pivot table to only show tasks in the selected group available_tasks = [t for t in task_list if t in pivot_df.index] pivot_df = pivot_df.loc[available_tasks] baseline_imputed = baseline_imputed.loc[available_tasks] leakage_imputed = leakage_imputed.loc[available_tasks] # Apply display-name overrides to model columns col_rename = {c: get_model_display_name(c) for c in pivot_df.columns} pivot_df = pivot_df.rename(columns=col_rename) baseline_imputed = baseline_imputed.rename(columns=col_rename) leakage_imputed = leakage_imputed.rename(columns=col_rename) def style_pivot_table(errors, is_baseline_imputed, is_leakage_imputed): rank_colors = {1: COLORS["gold"], 2: COLORS["silver"], 3: COLORS["bronze"]} def highlight_by_position(styler): for row_idx in errors.index: row_ranks = errors.loc[row_idx].rank(method="min") for col_idx in errors.columns: rank = row_ranks[col_idx] style_parts = [] if rank <= 3: style_parts.append(f"background-color: {rank_colors[rank]}") if is_leakage_imputed.loc[row_idx, col_idx]: style_parts.append(f"color: {COLORS['leakage_impute']}") elif is_baseline_imputed.loc[row_idx, col_idx]: style_parts.append(f"color: {COLORS['failure_impute']}") elif not style_parts: style_parts.append(f"color: {COLORS['text_default']}") if style_parts: styler = styler.map( lambda x, s="; ".join(style_parts): s, subset=pd.IndexSlice[row_idx:row_idx, col_idx:col_idx], ) return styler return highlight_by_position(errors.style).format(precision=3) st.dataframe(style_pivot_table(pivot_df, baseline_imputed, leakage_imputed)) st.divider() st.markdown("### :material/format_quote: Citation", unsafe_allow_html=True) st.markdown(CITATION_HEADER) st.markdown(CITATION_FEV)