import sys
from pathlib import Path
sys.path.append(str(Path(__file__).parent))
import pandas as pd
import streamlit as st
from src.strings import (
CITATION_FEV,
CITATION_HEADER,
FEV_BENCHMARK_DETAILS,
PAIRWISE_BENCHMARK_DETAILS,
get_pivot_legend,
)
from src.task_groups import (
ALL_TASKS,
DOMAIN_GROUPS,
FREQUENCY_GROUPS,
MINI_TASKS,
TASK_TYPE_GROUPS,
)
from src.utils import (
COLORS,
DEFAULT_VISIBLE_MODEL_TYPES,
MODEL_TYPE_LABELS,
TAG_COLORS,
construct_pairwise_chart,
format_leaderboard,
format_metric_name,
get_metric_description,
get_model_display_name,
select_model_names,
validate_model_metadata,
)
# Leaderboard computation vendored from the fev repo by save_tables.py, so the numbers shown here are
# produced by the same code as the paper figures.
from src.vendor.fev_bench_compute import (
AVAILABLE_METRICS,
BASELINE_MODEL,
LEAKAGE_IMPUTATION_MODEL,
SORT_COL,
TOP_K_MODELS_TO_PLOT,
compute_leaderboard,
compute_pairwise,
)
from streamlit.elements.lib.column_types import ColumnConfig
st.set_page_config(
layout="wide", page_title="fev leaderboard", page_icon=":material/trophy:"
)
TITLE = "
fev-bench
"
MODEL_TYPES_HELP = (
"Win rates are computed among the included models. "
f"{BASELINE_MODEL} and {LEAKAGE_IMPUTATION_MODEL} are always included, as they are used for imputation. "
'Click "See details" below the leaderboard for a description of each model type.'
)
NON_COMMERCIAL_HELP = "Include models whose license does not permit commercial use."
# Group type options
GROUP_TYPES = [
"Full (100 tasks)",
"Mini (20 tasks)",
"By frequency",
"By domain",
"By task type",
]
FREQUENCY_OPTIONS = list(FREQUENCY_GROUPS.keys())
DOMAIN_OPTIONS = list(DOMAIN_GROUPS.keys())
TASK_TYPE_OPTIONS = list(TASK_TYPE_GROUPS.keys())
def get_subset_description(
group_type: str, subgroup: str | None, num_tasks: int, metric_name: str
) -> str:
"""Generate a description of the current subset."""
base = f"Results for various forecasting models, measured using the **{metric_name}** metric, on **{num_tasks} tasks**"
if group_type == "Full (100 tasks)":
subset_desc = "from the full **fev-bench** benchmark"
elif group_type == "Mini (20 tasks)":
subset_desc = "from the **fev-bench-mini** subset"
elif group_type == "By frequency":
subset_desc = f"with **{subgroup.lower()}** frequency"
elif group_type == "By task type":
subset_desc = f"of type **{subgroup.lower()}**"
else: # By domain
subset_desc = f"from the **{subgroup}** domain"
paper_link = "[fev-bench: A Realistic Benchmark for Time Series Forecasting](https://arxiv.org/abs/2509.26468)"
return f"{base} {subset_desc}, as described in {paper_link}."
@st.cache_data()
def get_summaries() -> pd.DataFrame:
return pd.read_csv("tables/summaries.csv")
@st.cache_data()
def get_leaderboard(
metric_name: str, task_list: tuple[str, ...], model_names: tuple[str, ...]
) -> pd.DataFrame:
"""Leaderboard for the selected tasks and models.
Win rate and skill score depend on which models are included, so these are computed here rather
than precomputed: including an extra model type changes every row.
"""
return compute_leaderboard(_filter_summaries(metric_name, task_list, model_names), metric_name)
@st.cache_data()
def get_pairwise(
metric_name: str, task_list: tuple[str, ...], model_names: tuple[str, ...]
) -> pd.DataFrame:
summaries = _filter_summaries(metric_name, task_list, model_names)
leaderboard_df = get_leaderboard(metric_name, task_list, model_names)
top_k_models = (
leaderboard_df.sort_values(by=SORT_COL, ascending=False)
.head(TOP_K_MODELS_TO_PLOT)["model_name"]
.tolist()
)
return compute_pairwise(summaries, metric_name, top_k_models)
def _filter_summaries(
metric_name: str, task_list: tuple[str, ...], model_names: tuple[str, ...]
) -> pd.DataFrame:
summaries = get_summaries()
# The baseline and leakage-imputation models must stay in the data for imputation to work, even
# if the reader excluded their type.
included = set(model_names) | {BASELINE_MODEL, LEAKAGE_IMPUTATION_MODEL}
return summaries[
summaries["task_name"].isin(task_list) & summaries["model_name"].isin(included)
]
@st.cache_data()
def get_pivot_table(
metric_name: str,
) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]:
pivot_df = pd.read_csv(f"tables/pivot_{metric_name}.csv")
baseline_imputed = pd.read_csv(f"tables/pivot_{metric_name}_baseline_imputed.csv")
leakage_imputed = pd.read_csv(f"tables/pivot_{metric_name}_leakage_imputed.csv")
return pivot_df, baseline_imputed, leakage_imputed
with st.sidebar:
# Task group selection
selected_group_type = st.selectbox("Subset", options=GROUP_TYPES)
# Conditional sub-selection for frequency/domain
selected_subgroup = None
if selected_group_type == "By frequency":
selected_subgroup = st.selectbox("Frequency", options=FREQUENCY_OPTIONS)
elif selected_group_type == "By domain":
selected_subgroup = st.selectbox("Domain", options=DOMAIN_OPTIONS)
elif selected_group_type == "By task type":
selected_subgroup = st.selectbox("Task type", options=TASK_TYPE_OPTIONS)
# Determine the tasks making up the selected subset
if selected_group_type in ["Full (100 tasks)", "Mini (20 tasks)"]:
task_list = (
ALL_TASKS if selected_group_type == "Full (100 tasks)" else MINI_TASKS
)
else:
if selected_group_type == "By frequency":
task_list = FREQUENCY_GROUPS[selected_subgroup]
elif selected_group_type == "By task type":
task_list = TASK_TYPE_GROUPS[selected_subgroup]
else:
task_list = DOMAIN_GROUPS[selected_subgroup]
st.caption(f"{len(task_list)} tasks")
st.divider()
selected_metric = st.selectbox(
"Evaluation Metric", options=AVAILABLE_METRICS, format_func=format_metric_name
)
st.caption(get_metric_description(selected_metric))
st.divider()
included_model_types = st.pills(
"Model types",
options=list(MODEL_TYPE_LABELS),
format_func=MODEL_TYPE_LABELS.get,
selection_mode="multi",
default=list(DEFAULT_VISIBLE_MODEL_TYPES),
help=MODEL_TYPES_HELP,
)
include_non_commercial = st.toggle(
"Include non-commercial models", value=True, help=NON_COMMERCIAL_HELP
)
included_models = tuple(
select_model_names(included_model_types, include_non_commercial)
)
if included_models:
# The imputation models are always part of the computation, so count them too
num_models = len(
set(included_models) | {BASELINE_MODEL, LEAKAGE_IMPUTATION_MODEL}
)
st.caption(f"{num_models} models")
cols = st.columns(spec=[0.025, 0.95, 0.025])
with cols[1] as main_container:
st.markdown(TITLE, unsafe_allow_html=True)
if not included_models:
st.warning("No models match the current selection. Select at least one model type in the sidebar.")
st.stop()
task_names = tuple(task_list)
metric_df = get_leaderboard(selected_metric, task_names, included_models).sort_values(
by=SORT_COL, ascending=False
)
validate_model_metadata(metric_df["model_name"])
pairwise_df = get_pairwise(selected_metric, task_names, included_models)
st.markdown("## :material/trophy: Leaderboard", unsafe_allow_html=True)
st.markdown(
get_subset_description(
selected_group_type, selected_subgroup, len(task_list), selected_metric
),
unsafe_allow_html=True,
)
df_styled = format_leaderboard(metric_df)
st.dataframe(
df_styled,
width="stretch",
hide_index=True,
column_config={
"model_name": st.column_config.LinkColumn(
label="Model", display_text=r"#(.*)$"
),
"win_rate": st.column_config.NumberColumn(
label="Win rate (%)", format="%.1f"
),
"skill_score": st.column_config.NumberColumn(
label="Skill score (%)", format="%.1f"
),
"median_inference_time_s_per100": st.column_config.NumberColumn(
label="Runtime (s / 100 series)",
format="%.1f",
help="Median end-to-end runtime per 100 time series.",
),
"training_corpus_overlap": st.column_config.NumberColumn(
label="Leakage (%)", format="%d"
),
"num_failures": st.column_config.NumberColumn(
label="Failed (%)",
format="%.0f",
help="Percentage of tasks where the model failed to produce a forecast.",
),
"tags": st.column_config.MultiselectColumn(
label="Tags",
options=list(TAG_COLORS),
color=list(TAG_COLORS.values()),
help="Model type; Zero-shot = no training on the task; Non-commercial = license does not permit commercial use.",
),
"org": ColumnConfig(label="Organization", alignment="left"),
},
)
with st.expander("See details"):
st.markdown(FEV_BENCHMARK_DETAILS, unsafe_allow_html=True)
st.markdown("## :material/bar_chart: Pairwise comparison", unsafe_allow_html=True)
chart_col_1, _, chart_col_2 = st.columns(spec=[0.45, 0.1, 0.45])
with chart_col_1:
st.altair_chart(
construct_pairwise_chart(
pairwise_df, col="win_rate", metric_name=selected_metric
),
use_container_width=True,
)
with chart_col_2:
st.altair_chart(
construct_pairwise_chart(
pairwise_df, col="skill_score", metric_name=selected_metric
),
use_container_width=True,
)
with st.expander("See details"):
st.markdown(PAIRWISE_BENCHMARK_DETAILS, unsafe_allow_html=True)
st.markdown(
"## :material/table_chart: Results for individual tasks", unsafe_allow_html=True
)
with st.expander("Show detailed results"):
st.markdown(
get_pivot_legend("Seasonal Naive", "Chronos-Bolt"), unsafe_allow_html=True
)
pivot_df, baseline_imputed, leakage_imputed = get_pivot_table(selected_metric)
pivot_df = pivot_df.set_index("Task name")
baseline_imputed = baseline_imputed.set_index("Task name")
leakage_imputed = leakage_imputed.set_index("Task name")
# Keep only the models included by the sidebar toggles
excluded_cols = [c for c in pivot_df.columns if c not in included_models]
pivot_df = pivot_df.drop(columns=excluded_cols)
baseline_imputed = baseline_imputed.drop(columns=excluded_cols)
leakage_imputed = leakage_imputed.drop(columns=excluded_cols)
# Filter pivot table to only show tasks in the selected group
available_tasks = [t for t in task_list if t in pivot_df.index]
pivot_df = pivot_df.loc[available_tasks]
baseline_imputed = baseline_imputed.loc[available_tasks]
leakage_imputed = leakage_imputed.loc[available_tasks]
# Apply display-name overrides to model columns
col_rename = {c: get_model_display_name(c) for c in pivot_df.columns}
pivot_df = pivot_df.rename(columns=col_rename)
baseline_imputed = baseline_imputed.rename(columns=col_rename)
leakage_imputed = leakage_imputed.rename(columns=col_rename)
def style_pivot_table(errors, is_baseline_imputed, is_leakage_imputed):
rank_colors = {1: COLORS["gold"], 2: COLORS["silver"], 3: COLORS["bronze"]}
def highlight_by_position(styler):
for row_idx in errors.index:
row_ranks = errors.loc[row_idx].rank(method="min")
for col_idx in errors.columns:
rank = row_ranks[col_idx]
style_parts = []
if rank <= 3:
style_parts.append(f"background-color: {rank_colors[rank]}")
if is_leakage_imputed.loc[row_idx, col_idx]:
style_parts.append(f"color: {COLORS['leakage_impute']}")
elif is_baseline_imputed.loc[row_idx, col_idx]:
style_parts.append(f"color: {COLORS['failure_impute']}")
elif not style_parts:
style_parts.append(f"color: {COLORS['text_default']}")
if style_parts:
styler = styler.map(
lambda x, s="; ".join(style_parts): s,
subset=pd.IndexSlice[row_idx:row_idx, col_idx:col_idx],
)
return styler
return highlight_by_position(errors.style).format(precision=3)
st.dataframe(style_pivot_table(pivot_df, baseline_imputed, leakage_imputed))
st.divider()
st.markdown("### :material/format_quote: Citation", unsafe_allow_html=True)
st.markdown(CITATION_HEADER)
st.markdown(CITATION_FEV)