From 42157e43378eca8e6b2c1fe4fa4b9f7aa66f6aa9 Mon Sep 17 00:00:00 2001 From: aas008 Date: Mon, 21 Sep 2026 15:31:20 -0400 Subject: [PATCH 01/17] Filter out per-turn breakdown rows from multi-turn benchmarks MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Forge PR #215 adds per-turn CSV rows alongside aggregate rows for multi-turn benchmarks. The `turn` column is empty on aggregate rows and 0/1/2/… on per-turn breakdown rows. Without this filter, per-turn rows would be mixed into _combos() grouping and produce duplicate concurrency points, inflated request counts, and skewed percentiles. Filter keeps only aggregate rows (turn is empty/NaN). CSVs without a turn column are unaffected. Co-Authored-By: Claude Opus 4.6 --- dashboard.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/dashboard.py b/dashboard.py index d188882..0ce4f88 100644 --- a/dashboard.py +++ b/dashboard.py @@ -11597,6 +11597,9 @@ def decode_filters_from_url(): df["prefix_caching"] = df["prefix_caching"].fillna("").astype(str) df["prefix_caching"] = df["prefix_caching"].replace("", "no") + if "turn" in df.columns: + df = df[df["turn"].isna() | (df["turn"].astype(str).str.strip() == "")].copy() + if "turns" not in df.columns: df["turns"] = 1 df["turns"] = df["turns"].fillna(1).astype(int) From ba2f41953dfe23c913b3f798a45e9c4f0b184fa4 Mon Sep 17 00:00:00 2001 From: aas008 Date: Mon, 21 Sep 2026 15:34:06 -0400 Subject: [PATCH 02/17] Rename turn to turn_index to avoid confusion with turns column Co-Authored-By: Claude Opus 4.6 --- dashboard.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/dashboard.py b/dashboard.py index 0ce4f88..037126c 100644 --- a/dashboard.py +++ b/dashboard.py @@ -11597,8 +11597,10 @@ def decode_filters_from_url(): df["prefix_caching"] = df["prefix_caching"].fillna("").astype(str) df["prefix_caching"] = df["prefix_caching"].replace("", "no") - if "turn" in df.columns: - df = df[df["turn"].isna() | (df["turn"].astype(str).str.strip() == "")].copy() + if "turn_index" in df.columns: + df = df[ + df["turn_index"].isna() | (df["turn_index"].astype(str).str.strip() == "") + ].copy() if "turns" not in df.columns: df["turns"] = 1 From 7196695a2b215c6b746b4704096c9faecaf9c017 Mon Sep 17 00:00:00 2001 From: aas008 Date: Wed, 30 Sep 2026 00:24:34 -0400 Subject: [PATCH 03/17] Add Turn x-axis option to Performance Plots for multi-turn benchmarks When per-turn rows are present (from Forge PR #215), a "Turn (Multi-turn)" option appears in the Select X-Axis dropdown. Selecting it plots turn_index on x-axis vs any Y-axis metric, with one line per concurrency level. "Show concurrency up to" filters which lines appear, same as for Concurrency. Replaces the earlier render_per_turn_breakdown expander with native integration into the existing plot controls. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 187 ++++++++++++++++++++++++++++++++++++++------------- 1 file changed, 142 insertions(+), 45 deletions(-) diff --git a/dashboard.py b/dashboard.py index 037126c..c2c680c 100644 --- a/dashboard.py +++ b/dashboard.py @@ -3808,7 +3808,7 @@ def render_dataset_representation_section(selected_profile, use_expander=True): @st.fragment -def render_performance_plots_section(filtered_df, use_expander=True): +def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander=True): """📊 Performance Plots Section - Complete functionality from original.""" if use_expander: if "performance_plots_expanded" not in st.session_state: @@ -3871,12 +3871,30 @@ def render_performance_plots_section(filtered_df, use_expander=True): _sort_cols.append("DP") filtered_df_sorted = filtered_df.sort_values(_sort_cols).copy() + # Build per-turn plot df with run_identifier + per_turn_plot_df = pd.DataFrame() + if per_turn_df is not None and not per_turn_df.empty: + per_turn_plot_df = per_turn_df.copy() + per_turn_plot_df["run_identifier"] = ( + per_turn_plot_df["accelerator"] + + " | " + + per_turn_plot_df["model"] + + " | " + + per_turn_plot_df["version"] + + " | TP=" + + per_turn_plot_df["TP"].apply( + lambda x: str(int(x)) if pd.notna(x) else "N/A" + ) + ) + col1, col2, col3 = st.columns(3) with col1: x_axis_options = { "Concurrency": "intended concurrency", "Throughput (Output Tok/s)": "output_tok/sec", } + if not per_turn_plot_df.empty: + x_axis_options["Turn (Multi-turn)"] = "turn_index" x_axis_label = st.selectbox( "Select X-Axis", options=list(x_axis_options.keys()), @@ -3912,10 +3930,13 @@ def render_performance_plots_section(filtered_df, use_expander=True): y_axis = y_axis_options[y_axis_label] with col3: - if x_axis == "intended concurrency": + if x_axis in ("intended concurrency", "turn_index"): + _conc_source = ( + per_turn_plot_df if x_axis == "turn_index" else filtered_df_sorted + ) concurrency_values = sorted( int(x) - for x in filtered_df_sorted["intended concurrency"] + for x in _conc_source["intended concurrency"] .dropna() .unique() .tolist() @@ -3935,9 +3956,14 @@ def render_performance_plots_section(filtered_df, use_expander=True): on_change=keep_expander_open, args=("performance_plots_expanded",), ) - filtered_df_sorted = filtered_df_sorted[ - filtered_df_sorted["intended concurrency"] <= max_conc - ] + if x_axis == "intended concurrency": + filtered_df_sorted = filtered_df_sorted[ + filtered_df_sorted["intended concurrency"] <= max_conc + ] + else: + per_turn_plot_df = per_turn_plot_df[ + per_turn_plot_df["intended concurrency"] <= max_conc + ] # Add units to y-axis label for certain metrics y_axis_display_label = y_axis_label @@ -3948,14 +3974,18 @@ def render_performance_plots_section(filtered_df, use_expander=True): elif y_axis == "request_latency_median" or y_axis == "request_latency_max": y_axis_display_label = f"{y_axis_label} (s)" - # Build ISL/OSL subtitle from unique values in the filtered data + # Build ISL/OSL subtitle from the active data source + _subtitle_src = ( + per_turn_plot_df if x_axis == "turn_index" and not per_turn_plot_df.empty + else filtered_df_sorted + ) _isl_osl_subtitle = "" if ( - "prompt toks" in filtered_df_sorted.columns - and "output toks" in filtered_df_sorted.columns + "prompt toks" in _subtitle_src.columns + and "output toks" in _subtitle_src.columns ): isl_osl_pairs = ( - filtered_df_sorted[["prompt toks", "output toks"]] + _subtitle_src[["prompt toks", "output toks"]] .dropna() .drop_duplicates() ) @@ -3964,11 +3994,11 @@ def render_performance_plots_section(filtered_df, use_expander=True): for _, r in isl_osl_pairs.iterrows(): isl, osl = int(r["prompt toks"]), int(r["output toks"]) if isl == 0 and osl == 0: - if "dataset" in filtered_df_sorted.columns: + if "dataset" in _subtitle_src.columns: ds_names = ( - filtered_df_sorted.loc[ - (filtered_df_sorted["prompt toks"] == 0) - & (filtered_df_sorted["output toks"] == 0), + _subtitle_src.loc[ + (_subtitle_src["prompt toks"] == 0) + & (_subtitle_src["output toks"] == 0), "dataset", ] .dropna() @@ -3981,33 +4011,71 @@ def render_performance_plots_section(filtered_df, use_expander=True): if pair_labels: _isl_osl_subtitle = f"
ISL/OSL: {', '.join(sorted(set(pair_labels)))}" - fig = px.line( - filtered_df_sorted.sort_values(by=x_axis), - x=x_axis, - y=y_axis, - color="run_identifier", - markers=True, - title=f"{x_axis_label} vs. {y_axis_label}{_isl_osl_subtitle}", - labels={ - x_axis: x_axis_label, - y_axis: y_axis_display_label, - "run_identifier": "Run", - }, - template="plotly_white_light", - category_orders={ - "run_identifier": filtered_df_sorted["run_identifier"].unique().tolist() - }, - ) - _legend_parts = "Accelerator | Model | Version | TP" - if _has_dp_data: - _legend_parts += " | DP" - if (filtered_df_sorted["turns"] > 1).any(): - _legend_parts += " | Turns/PrefixTokens/PrefixCount" - fig.update_layout( - legend_title_text=f"Run Details ({_legend_parts})", - legend={"font": {"size": 14}}, - ) - st.plotly_chart(fig, use_container_width=True, theme=None) + if x_axis == "turn_index" and not per_turn_plot_df.empty: + per_turn_plot_df = per_turn_plot_df.copy() + per_turn_plot_df["_plot_label"] = ( + per_turn_plot_df["run_identifier"] + + " | conc=" + + per_turn_plot_df["intended concurrency"].apply( + lambda x: str(int(x)) if pd.notna(x) else "?" + ) + ) + _pt_valid = per_turn_plot_df.dropna(subset=[y_axis]).copy() if y_axis in per_turn_plot_df.columns else pd.DataFrame() + if _pt_valid.empty: + st.info(f"No per-turn data available for '{y_axis_label}'.") + fig = None + else: + fig = px.line( + _pt_valid.sort_values("turn_index"), + x="turn_index", + y=y_axis, + color="_plot_label", + markers=True, + title=f"Turn vs. {y_axis_label}{_isl_osl_subtitle}", + labels={ + "turn_index": "Turn", + y_axis: y_axis_display_label, + "_plot_label": "Run | Concurrency", + }, + template="plotly_white_light", + ) + fig.update_xaxes( + tickmode="array", + tickvals=sorted(_pt_valid["turn_index"].dropna().unique()), + ) + fig.update_layout( + legend_title_text="Run Details (Accelerator | Model | Version | TP | Concurrency)", + legend={"font": {"size": 14}}, + ) + else: + fig = px.line( + filtered_df_sorted.sort_values(by=x_axis), + x=x_axis, + y=y_axis, + color="run_identifier", + markers=True, + title=f"{x_axis_label} vs. {y_axis_label}{_isl_osl_subtitle}", + labels={ + x_axis: x_axis_label, + y_axis: y_axis_display_label, + "run_identifier": "Run", + }, + template="plotly_white_light", + category_orders={ + "run_identifier": filtered_df_sorted["run_identifier"].unique().tolist() + }, + ) + _legend_parts = "Accelerator | Model | Version | TP" + if _has_dp_data: + _legend_parts += " | DP" + if (filtered_df_sorted["turns"] > 1).any(): + _legend_parts += " | Turns/PrefixTokens/PrefixCount" + fig.update_layout( + legend_title_text=f"Run Details ({_legend_parts})", + legend={"font": {"size": 14}}, + ) + if fig is not None: + st.plotly_chart(fig, use_container_width=True, theme=None) # Right-align the legend caption caption_col1, caption_col2 = st.columns([3, 1]) @@ -11598,9 +11666,27 @@ def decode_filters_from_url(): df["prefix_caching"] = df["prefix_caching"].replace("", "no") if "turn_index" in df.columns: - df = df[ - df["turn_index"].isna() | (df["turn_index"].astype(str).str.strip() == "") - ].copy() + _ti = df["turn_index"].astype(str).str.strip() + per_turn_df = df[df["turn_index"].notna() & (_ti != "") & (_ti != "nan")].copy() + if not per_turn_df.empty: + per_turn_df["turn_index"] = per_turn_df["turn_index"].astype(float).astype(int) + per_turn_df["error_rate"] = ( + per_turn_df["errored_requests"] + / (per_turn_df["successful_requests"] + per_turn_df["errored_requests"]) + * 100 + ).fillna(0) + per_turn_df["efficiency_ratio"] = per_turn_df["output_tok/sec"] / per_turn_df["TP"] + per_turn_df["ttft_p95_s"] = ( + per_turn_df["ttft_p95"] / 1000 if "ttft_p95" in per_turn_df.columns else np.nan + ) + per_turn_df["ttft_median_s"] = ( + per_turn_df["ttft_median"] / 1000 + if "ttft_median" in per_turn_df.columns + else np.nan + ) + df = df[df["turn_index"].isna() | (_ti == "") | (_ti == "nan")].copy() + else: + per_turn_df = pd.DataFrame() if "turns" not in df.columns: df["turns"] = 1 @@ -12775,6 +12861,15 @@ def decode_filters_from_url(): & dp_mask ].copy() + if not per_turn_df.empty: + filtered_per_turn_df = per_turn_df[ + per_turn_df["accelerator"].isin(selected_accelerators) + & per_turn_df["model"].isin(selected_models) + & per_turn_df["version"].isin(selected_versions) + ].copy() + else: + filtered_per_turn_df = pd.DataFrame() + # Detect if filters have changed and close expanders current_filter_state = { "accelerators": tuple(sorted(selected_accelerators)), @@ -12916,7 +13011,9 @@ def _render_selected_section(sel): elif sel == "🔍 Competitive Analysis": render_competitive_analysis_section(df) elif sel == "📊 Performance Plots": - render_performance_plots_section(filtered_df, use_expander=False) + render_performance_plots_section( + filtered_df, filtered_per_turn_df, use_expander=False + ) elif sel == "📈 Dataset Representation": render_dataset_representation_section( selected_profile, use_expander=False From ebced9c02dd0c9842709a36a1a9c5106b39d746e Mon Sep 17 00:00:00 2001 From: aas008 Date: Wed, 30 Sep 2026 00:57:54 -0400 Subject: [PATCH 04/17] Fix code-review findings in per-turn dashboard integration Correctness: - filtered_per_turn_df now uses a shared _apply_filters() function that applies all 10 sidebar filters (was missing TP, profile, spec_decoding, DP, dataset, prefix_caching, and multi-turn masks) - per_turn_plot_df run_identifier now includes DP/spec_decoding/prefix_caching suffixes so per-turn legend labels match the main-chart labels - errored_requests/successful_requests column access now guarded; KeyError no longer crashes data prep for older CSV schemas - turn_index astype(float) replaced with pd.to_numeric(errors='coerce') to handle non-numeric strings like 'N/A' without crashing - efficiency_ratio inf guard: replace [inf, -inf] with NaN for TP=0 rows - output_tok/sec column access now guarded for latency-only schemas Cleanup: - Extract _is_turn_view boolean; evaluated once before all branches use it Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 144 +++++++++++++++++++++++++++++++++++++-------------- 1 file changed, 106 insertions(+), 38 deletions(-) diff --git a/dashboard.py b/dashboard.py index c2c680c..f19b185 100644 --- a/dashboard.py +++ b/dashboard.py @@ -3871,7 +3871,7 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander _sort_cols.append("DP") filtered_df_sorted = filtered_df.sort_values(_sort_cols).copy() - # Build per-turn plot df with run_identifier + # Build per-turn plot df with run_identifier (same suffixes as filtered_df) per_turn_plot_df = pd.DataFrame() if per_turn_df is not None and not per_turn_df.empty: per_turn_plot_df = per_turn_df.copy() @@ -3886,6 +3886,18 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander lambda x: str(int(x)) if pd.notna(x) else "N/A" ) ) + if _has_dp_data and "DP" in per_turn_plot_df.columns: + per_turn_plot_df["run_identifier"] += per_turn_plot_df["DP"].apply( + lambda x: f" | DP={int(x)}" if pd.notna(x) else "" + ) + if "spec_decoding" in per_turn_plot_df.columns and per_turn_plot_df["spec_decoding"].any(): + per_turn_plot_df["run_identifier"] += per_turn_plot_df["spec_decoding"].apply( + lambda x: f" | SD={x}" if x else "" + ) + if "prefix_caching" in per_turn_plot_df.columns and per_turn_plot_df["prefix_caching"].any(): + per_turn_plot_df["run_identifier"] += per_turn_plot_df["prefix_caching"].apply( + lambda x: f" | PC={x}" if x else "" + ) col1, col2, col3 = st.columns(3) with col1: @@ -3965,6 +3977,9 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander per_turn_plot_df["intended concurrency"] <= max_conc ] + # Evaluate once so all branches below stay consistent + _is_turn_view = x_axis == "turn_index" and not per_turn_plot_df.empty + # Add units to y-axis label for certain metrics y_axis_display_label = y_axis_label if y_axis in ("ttft_p95_s", "ttft_median_s"): @@ -3975,10 +3990,7 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander y_axis_display_label = f"{y_axis_label} (s)" # Build ISL/OSL subtitle from the active data source - _subtitle_src = ( - per_turn_plot_df if x_axis == "turn_index" and not per_turn_plot_df.empty - else filtered_df_sorted - ) + _subtitle_src = per_turn_plot_df if _is_turn_view else filtered_df_sorted _isl_osl_subtitle = "" if ( "prompt toks" in _subtitle_src.columns @@ -4011,7 +4023,7 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander if pair_labels: _isl_osl_subtitle = f"
ISL/OSL: {', '.join(sorted(set(pair_labels)))}" - if x_axis == "turn_index" and not per_turn_plot_df.empty: + if _is_turn_view: per_turn_plot_df = per_turn_plot_df.copy() per_turn_plot_df["_plot_label"] = ( per_turn_plot_df["run_identifier"] @@ -11669,13 +11681,30 @@ def decode_filters_from_url(): _ti = df["turn_index"].astype(str).str.strip() per_turn_df = df[df["turn_index"].notna() & (_ti != "") & (_ti != "nan")].copy() if not per_turn_df.empty: - per_turn_df["turn_index"] = per_turn_df["turn_index"].astype(float).astype(int) - per_turn_df["error_rate"] = ( - per_turn_df["errored_requests"] - / (per_turn_df["successful_requests"] + per_turn_df["errored_requests"]) - * 100 - ).fillna(0) - per_turn_df["efficiency_ratio"] = per_turn_df["output_tok/sec"] / per_turn_df["TP"] + per_turn_df["turn_index"] = ( + pd.to_numeric(per_turn_df["turn_index"], errors="coerce") + .dropna() + .astype(int) + .reindex(per_turn_df.index) + ) + per_turn_df = per_turn_df[per_turn_df["turn_index"].notna()].copy() + if ( + "errored_requests" in per_turn_df.columns + and "successful_requests" in per_turn_df.columns + ): + per_turn_df["error_rate"] = ( + per_turn_df["errored_requests"] + / (per_turn_df["successful_requests"] + per_turn_df["errored_requests"]) + * 100 + ).fillna(0) + else: + per_turn_df["error_rate"] = np.nan + if "output_tok/sec" in per_turn_df.columns and "TP" in per_turn_df.columns: + per_turn_df["efficiency_ratio"] = ( + per_turn_df["output_tok/sec"] / per_turn_df["TP"] + ).replace([np.inf, -np.inf], np.nan) + else: + per_turn_df["efficiency_ratio"] = np.nan per_turn_df["ttft_p95_s"] = ( per_turn_df["ttft_p95"] / 1000 if "ttft_p95" in per_turn_df.columns else np.nan ) @@ -12843,32 +12872,71 @@ def decode_filters_from_url(): if selected_mt_prefix_count is not None else True ) - tp_mask = df["TP"].isin(selected_tp) | df["TP"].isna() - filtered_df = df[ - df["accelerator"].isin(selected_accelerators) - & df["model"].isin(selected_models) - & df["version"].isin(selected_versions) - & (df["profile"].isin(selected_profiles) if selected_profiles else True) - & tp_mask - & custom_mask - & dataset_mask - & spec_decoding_mask - & prefix_caching_mask - & multiturn_isl_osl_mask - & mt_turns_mask - & mt_prefix_tokens_mask - & mt_prefix_count_mask - & dp_mask - ].copy() - - if not per_turn_df.empty: - filtered_per_turn_df = per_turn_df[ - per_turn_df["accelerator"].isin(selected_accelerators) - & per_turn_df["model"].isin(selected_models) - & per_turn_df["version"].isin(selected_versions) + def _apply_filters(d): + """Apply all sidebar filters to any DataFrame with the same schema.""" + _tp = d["TP"].isin(selected_tp) | d["TP"].isna() + _dp = ( + (d["DP"].isin(selected_dp) | d["DP"].isna()) + if st.session_state.get("show_advanced_filters", False) and _has_dp and selected_dp + else True + ) + _custom = ( + (d["custom_isl_osl"] == selected_custom_isl_osl) + if selected_profile == "Custom ISL/OSL" and selected_custom_isl_osl + else True + ) + _dataset = ( + (d["dataset"] == selected_dataset_filter) + if selected_dataset_filter is not None + else True + ) + _sd = ( + d["spec_decoding"].isin(selected_spec_decoding_filter) + if selected_spec_decoding_filter + else True + ) + _pc = ( + d["prefix_caching"].isin(selected_prefix_caching_filter) + if selected_prefix_caching_filter + else True + ) + _mt_isl_osl = ( + (d["multiturn_isl_osl"] == selected_multiturn_isl_osl) + if selected_profile == "Multi-turn" and selected_multiturn_isl_osl + else True + ) + _mt_turns = ( + d["turns"].isin(selected_mt_turns) if selected_mt_turns is not None else True + ) + _mt_pt = ( + d["prefix_tokens"].isin(selected_mt_prefix_tokens) + if selected_mt_prefix_tokens is not None + else True + ) + _mt_pc = ( + d["prefix_count"].isin(selected_mt_prefix_count) + if selected_mt_prefix_count is not None + else True + ) + return d[ + d["accelerator"].isin(selected_accelerators) + & d["model"].isin(selected_models) + & d["version"].isin(selected_versions) + & (d["profile"].isin(selected_profiles) if selected_profiles else True) + & _tp + & _custom + & _dataset + & _sd + & _pc + & _mt_isl_osl + & _mt_turns + & _mt_pt + & _mt_pc + & _dp ].copy() - else: - filtered_per_turn_df = pd.DataFrame() + + filtered_df = _apply_filters(df) + filtered_per_turn_df = _apply_filters(per_turn_df) if not per_turn_df.empty else pd.DataFrame() # Detect if filters have changed and close expanders current_filter_state = { From 930f1af81749743fad02c872b8f40e480a5ce9d6 Mon Sep 17 00:00:00 2001 From: aas008 Date: Wed, 30 Sep 2026 11:56:02 -0400 Subject: [PATCH 05/17] bazinga: fix CI failure on PR #101 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Remove orphaned mask variables (tp_mask, dp_mask, custom_mask, etc.) left behind after refactoring to _apply_filters() — ruff F841 flagged all 9 as assigned but unused. Run ruff format to fix formatting. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 106 +++++++++++++++++++-------------------------------- 1 file changed, 39 insertions(+), 67 deletions(-) diff --git a/dashboard.py b/dashboard.py index f19b185..7f1e52f 100644 --- a/dashboard.py +++ b/dashboard.py @@ -3890,14 +3890,20 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander per_turn_plot_df["run_identifier"] += per_turn_plot_df["DP"].apply( lambda x: f" | DP={int(x)}" if pd.notna(x) else "" ) - if "spec_decoding" in per_turn_plot_df.columns and per_turn_plot_df["spec_decoding"].any(): - per_turn_plot_df["run_identifier"] += per_turn_plot_df["spec_decoding"].apply( - lambda x: f" | SD={x}" if x else "" - ) - if "prefix_caching" in per_turn_plot_df.columns and per_turn_plot_df["prefix_caching"].any(): - per_turn_plot_df["run_identifier"] += per_turn_plot_df["prefix_caching"].apply( - lambda x: f" | PC={x}" if x else "" - ) + if ( + "spec_decoding" in per_turn_plot_df.columns + and per_turn_plot_df["spec_decoding"].any() + ): + per_turn_plot_df["run_identifier"] += per_turn_plot_df[ + "spec_decoding" + ].apply(lambda x: f" | SD={x}" if x else "") + if ( + "prefix_caching" in per_turn_plot_df.columns + and per_turn_plot_df["prefix_caching"].any() + ): + per_turn_plot_df["run_identifier"] += per_turn_plot_df[ + "prefix_caching" + ].apply(lambda x: f" | PC={x}" if x else "") col1, col2, col3 = st.columns(3) with col1: @@ -3997,9 +4003,7 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander and "output toks" in _subtitle_src.columns ): isl_osl_pairs = ( - _subtitle_src[["prompt toks", "output toks"]] - .dropna() - .drop_duplicates() + _subtitle_src[["prompt toks", "output toks"]].dropna().drop_duplicates() ) if not isl_osl_pairs.empty: pair_labels = [] @@ -4032,7 +4036,11 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander lambda x: str(int(x)) if pd.notna(x) else "?" ) ) - _pt_valid = per_turn_plot_df.dropna(subset=[y_axis]).copy() if y_axis in per_turn_plot_df.columns else pd.DataFrame() + _pt_valid = ( + per_turn_plot_df.dropna(subset=[y_axis]).copy() + if y_axis in per_turn_plot_df.columns + else pd.DataFrame() + ) if _pt_valid.empty: st.info(f"No per-turn data available for '{y_axis_label}'.") fig = None @@ -4074,7 +4082,9 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander }, template="plotly_white_light", category_orders={ - "run_identifier": filtered_df_sorted["run_identifier"].unique().tolist() + "run_identifier": filtered_df_sorted["run_identifier"] + .unique() + .tolist() }, ) _legend_parts = "Accelerator | Model | Version | TP" @@ -11694,7 +11704,10 @@ def decode_filters_from_url(): ): per_turn_df["error_rate"] = ( per_turn_df["errored_requests"] - / (per_turn_df["successful_requests"] + per_turn_df["errored_requests"]) + / ( + per_turn_df["successful_requests"] + + per_turn_df["errored_requests"] + ) * 100 ).fillna(0) else: @@ -11706,7 +11719,9 @@ def decode_filters_from_url(): else: per_turn_df["efficiency_ratio"] = np.nan per_turn_df["ttft_p95_s"] = ( - per_turn_df["ttft_p95"] / 1000 if "ttft_p95" in per_turn_df.columns else np.nan + per_turn_df["ttft_p95"] / 1000 + if "ttft_p95" in per_turn_df.columns + else np.nan ) per_turn_df["ttft_median_s"] = ( per_turn_df["ttft_median"] / 1000 @@ -12823,61 +12838,14 @@ def decode_filters_from_url(): else: st.caption("No DP data available") - dp_mask = ( - (df["DP"].isin(selected_dp) | df["DP"].isna()) - if st.session_state.get("show_advanced_filters", False) - and _has_dp - and selected_dp - else True - ) - - custom_mask = ( - (df["custom_isl_osl"] == selected_custom_isl_osl) - if selected_profile == "Custom ISL/OSL" and selected_custom_isl_osl - else True - ) - dataset_mask = ( - (df["dataset"] == selected_dataset_filter) - if selected_dataset_filter is not None - else True - ) - spec_decoding_mask = ( - df["spec_decoding"].isin(selected_spec_decoding_filter) - if selected_spec_decoding_filter - else True - ) - prefix_caching_mask = ( - df["prefix_caching"].isin(selected_prefix_caching_filter) - if selected_prefix_caching_filter - else True - ) - # Multi-turn masks - multiturn_isl_osl_mask = ( - (df["multiturn_isl_osl"] == selected_multiturn_isl_osl) - if selected_profile == "Multi-turn" and selected_multiturn_isl_osl - else True - ) - mt_turns_mask = ( - df["turns"].isin(selected_mt_turns) - if selected_mt_turns is not None - else True - ) - mt_prefix_tokens_mask = ( - df["prefix_tokens"].isin(selected_mt_prefix_tokens) - if selected_mt_prefix_tokens is not None - else True - ) - mt_prefix_count_mask = ( - df["prefix_count"].isin(selected_mt_prefix_count) - if selected_mt_prefix_count is not None - else True - ) def _apply_filters(d): """Apply all sidebar filters to any DataFrame with the same schema.""" _tp = d["TP"].isin(selected_tp) | d["TP"].isna() _dp = ( (d["DP"].isin(selected_dp) | d["DP"].isna()) - if st.session_state.get("show_advanced_filters", False) and _has_dp and selected_dp + if st.session_state.get("show_advanced_filters", False) + and _has_dp + and selected_dp else True ) _custom = ( @@ -12906,7 +12874,9 @@ def _apply_filters(d): else True ) _mt_turns = ( - d["turns"].isin(selected_mt_turns) if selected_mt_turns is not None else True + d["turns"].isin(selected_mt_turns) + if selected_mt_turns is not None + else True ) _mt_pt = ( d["prefix_tokens"].isin(selected_mt_prefix_tokens) @@ -12936,7 +12906,9 @@ def _apply_filters(d): ].copy() filtered_df = _apply_filters(df) - filtered_per_turn_df = _apply_filters(per_turn_df) if not per_turn_df.empty else pd.DataFrame() + filtered_per_turn_df = ( + _apply_filters(per_turn_df) if not per_turn_df.empty else pd.DataFrame() + ) # Detect if filters have changed and close expanders current_filter_state = { From 1f4d8ce72f898b0800bce62f5f695df8c53b067f Mon Sep 17 00:00:00 2001 From: aas008 Date: Wed, 30 Sep 2026 11:57:20 -0400 Subject: [PATCH 06/17] bazinga: address PR review feedback (#101) Add turns/prefix_tokens/prefix_count suffixes to per_turn_plot_df run_identifier so it fully matches filtered_df's run_identifier format. All six configuration dimensions now included: DP, spec_decoding, prefix_caching, turns, prefix_tokens, prefix_count. Addresses coderabbitai comment at line 3845. The filtered_per_turn_df filter comment is already fixed by _apply_filters(). Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/dashboard.py b/dashboard.py index 7f1e52f..580cc6d 100644 --- a/dashboard.py +++ b/dashboard.py @@ -3904,6 +3904,20 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander per_turn_plot_df["run_identifier"] += per_turn_plot_df[ "prefix_caching" ].apply(lambda x: f" | PC={x}" if x else "") + if ( + "turns" in per_turn_plot_df.columns + and (per_turn_plot_df["turns"] > 1).any() + ): + per_turn_plot_df["run_identifier"] += per_turn_plot_df.apply( + lambda r: ( + f" | {r['turns']}T" + + (f"/{r['prefix_tokens']}pt" if r.get("prefix_tokens") else "") + + (f"/{r['prefix_count']}pc" if r.get("prefix_count") else "") + if r["turns"] > 1 + else "" + ), + axis=1, + ) col1, col2, col3 = st.columns(3) with col1: From d1413cad9fdd3c8c13c2aecdf1fc8d59c81d00bd Mon Sep 17 00:00:00 2001 From: aas008 Date: Wed, 30 Sep 2026 14:12:41 -0400 Subject: [PATCH 07/17] bazinga: address PR review feedback (#101) Fix turn_index staying float64: filter invalid rows before astype(int) instead of using reindex() which re-introduces NaN and forces float column. Turn labels now show 0/1/2 instead of 0.0/1.0/2.0. Suggested by coderabbitai (comment 4146691396). Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 93 +++++++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 84 insertions(+), 9 deletions(-) diff --git a/dashboard.py b/dashboard.py index 580cc6d..f8ab1c3 100644 --- a/dashboard.py +++ b/dashboard.py @@ -10588,7 +10588,7 @@ def render_view_logs_section(filtered_df, use_expander=True): @st.fragment -def render_filtered_data_section(filtered_df, use_expander=True): +def render_filtered_data_section(filtered_df, per_turn_df=None, use_expander=True): """📄 Filtered Data Display Section - View only, no download functionality.""" if use_expander: ctx = st.expander("📄 Filtered Data from the above filters", expanded=False) @@ -10597,11 +10597,42 @@ def render_filtered_data_section(filtered_df, use_expander=True): with ctx: if not use_expander: st.subheader("📄 Filtered Data") + + _has_per_turn = per_turn_df is not None and not per_turn_df.empty + if _has_per_turn: + _show_per_turn = st.toggle( + "🔄 Show per-turn rows", + value=False, + key="filtered_data_show_per_turn", + help="Switch between aggregate rows (one per concurrency level) and per-turn breakdown rows", + ) + else: + _show_per_turn = False + st.info( "💡 **Tips**: Hover over column headers to see detailed descriptions of each field. " "Select a row to view its server log." + + ( + " Per-turn rows (turn_index 0, 1, 2…) are shown below each aggregate row." + if _show_per_turn and _has_per_turn + else "" + ) ) - display_filtered_df = filtered_df.copy() + if _show_per_turn and _has_per_turn: + display_filtered_df = ( + pd.concat([filtered_df, per_turn_df], ignore_index=True) + .sort_values( + ["model", "intended concurrency", "turn_index"], + key=lambda s: pd.to_numeric(s, errors="coerce").fillna( + -1 if s.name == "turn_index" else s.rank(method="dense") + ), + ) + .reset_index(drop=True) + ) + else: + display_filtered_df = filtered_df.copy() + if "turn_index" in display_filtered_df.columns: + display_filtered_df = display_filtered_df.drop(columns=["turn_index"]) display_filtered_df.reset_index(drop=True, inplace=True) display_filtered_df.insert(0, "Row #", range(1, len(display_filtered_df) + 1)) @@ -11422,7 +11453,7 @@ def main(): } def encode_filters_to_url(accelerators, models, versions, profile, tp_sizes): - """Encode main filter state to URL parameters.""" + """Encode main filter state and active section widget state to URL parameters.""" url_params = {} if accelerators: @@ -11436,6 +11467,21 @@ def encode_filters_to_url(accelerators, models, versions, profile, tp_sizes): if tp_sizes: url_params["tp_sizes"] = ",".join(map(str, tp_sizes)) + # Also push active section's widget state so the bare URL bar is shareable + active_slug = SECTION_TO_SLUG.get( + st.session_state.get("active_section", ""), "" + ) + if active_slug: + url_params["section"] = active_slug + for url_key, ss_key in SECTION_FILTER_KEYS.get(active_slug, {}).items(): + val = st.session_state.get(ss_key) + if val is not None: + url_params[url_key] = ( + ",".join(map(str, val)) + if isinstance(val, list) + else str(val) + ) + st.query_params.update(url_params) def build_share_url(): @@ -11705,13 +11751,11 @@ def decode_filters_from_url(): _ti = df["turn_index"].astype(str).str.strip() per_turn_df = df[df["turn_index"].notna() & (_ti != "") & (_ti != "nan")].copy() if not per_turn_df.empty: - per_turn_df["turn_index"] = ( - pd.to_numeric(per_turn_df["turn_index"], errors="coerce") - .dropna() - .astype(int) - .reindex(per_turn_df.index) + per_turn_df["turn_index"] = pd.to_numeric( + per_turn_df["turn_index"], errors="coerce" ) per_turn_df = per_turn_df[per_turn_df["turn_index"].notna()].copy() + per_turn_df["turn_index"] = per_turn_df["turn_index"].astype(int) if ( "errored_requests" in per_turn_df.columns and "successful_requests" in per_turn_df.columns @@ -11742,6 +11786,35 @@ def decode_filters_from_url(): if "ttft_median" in per_turn_df.columns else np.nan ) + # Mirror the same normalizations applied to df below so _apply_filters + # types match the sidebar multiselect values (which are sourced from df). + per_turn_df["turns"] = per_turn_df["turns"].fillna(1).astype(int) + _str_norm = lambda v: ( # noqa: E731 + str(int(float(v))) if v != "" and str(v) not in ("", "nan") else "" + ) + per_turn_df["prefix_tokens"] = ( + per_turn_df.get("prefix_tokens", pd.Series("", index=per_turn_df.index)) + .fillna("") + .apply(_str_norm) + ) + per_turn_df["prefix_count"] = ( + per_turn_df.get("prefix_count", pd.Series("", index=per_turn_df.index)) + .fillna("") + .apply(_str_norm) + ) + per_turn_df["spec_decoding"] = ( + per_turn_df.get("spec_decoding", pd.Series("", index=per_turn_df.index)) + .fillna("") + .astype(str) + ) + per_turn_df["prefix_caching"] = ( + per_turn_df.get( + "prefix_caching", pd.Series("", index=per_turn_df.index) + ) + .fillna("") + .astype(str) + .replace("no", "") + ) df = df[df["turn_index"].isna() | (_ti == "") | (_ti == "nan")].copy() else: per_turn_df = pd.DataFrame() @@ -13099,7 +13172,9 @@ def _render_selected_section(sel): elif sel == "📋 View Logs": render_view_logs_section(filtered_df, use_expander=False) elif sel == "📄 Filtered Data": - render_filtered_data_section(filtered_df, use_expander=False) + render_filtered_data_section( + filtered_df, filtered_per_turn_df, use_expander=False + ) _render_selected_section(current_section) From 9196d33ac3677ffb8dfae52c920634528652d8fc Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 14:18:18 -0400 Subject: [PATCH 08/17] feat: replace Turn x-axis concurrency filter with multiselect MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When Turn (Multi-turn) is selected as x-axis, col3 now shows a Concurrency multiselect (defaulting to the highest value) instead of the "Show concurrency up to" selectbox. This lets engineers study one concurrency at a time while making it easy to add more for comparison. Also fix empty-list isin() bug for mt_turns/mt_prefix_tokens/ mt_prefix_count sidebar filters — empty selection now means no filter (pass-through) instead of filtering out all data. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 63 +++++++++++++++++++++++++++++++++++----------------- 1 file changed, 43 insertions(+), 20 deletions(-) diff --git a/dashboard.py b/dashboard.py index f8ab1c3..7886ab5 100644 --- a/dashboard.py +++ b/dashboard.py @@ -3962,13 +3962,10 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander y_axis = y_axis_options[y_axis_label] with col3: - if x_axis in ("intended concurrency", "turn_index"): - _conc_source = ( - per_turn_plot_df if x_axis == "turn_index" else filtered_df_sorted - ) + if x_axis == "intended concurrency": concurrency_values = sorted( int(x) - for x in _conc_source["intended concurrency"] + for x in filtered_df_sorted["intended concurrency"] .dropna() .unique() .tolist() @@ -3988,13 +3985,39 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander on_change=keep_expander_open, args=("performance_plots_expanded",), ) - if x_axis == "intended concurrency": - filtered_df_sorted = filtered_df_sorted[ - filtered_df_sorted["intended concurrency"] <= max_conc - ] - else: + filtered_df_sorted = filtered_df_sorted[ + filtered_df_sorted["intended concurrency"] <= max_conc + ] + elif x_axis == "turn_index" and not per_turn_plot_df.empty: + concurrency_values = sorted( + int(x) + for x in per_turn_plot_df["intended concurrency"] + .dropna() + .unique() + .tolist() + ) + if concurrency_values: + _turn_conc_key = "perf_plots_turn_concurrency" + if _turn_conc_key not in st.session_state or not any( + c in concurrency_values + for c in (st.session_state.get(_turn_conc_key) or []) + ): + st.session_state[_turn_conc_key] = [max(concurrency_values)] + selected_concs = st.multiselect( + "Concurrency", + options=concurrency_values, + key=_turn_conc_key, + on_change=keep_expander_open, + args=("performance_plots_expanded",), + ) + st.caption( + "💡 Select multiple concurrency levels to compare turn curves side by side." + ) + if selected_concs: per_turn_plot_df = per_turn_plot_df[ - per_turn_plot_df["intended concurrency"] <= max_conc + per_turn_plot_df["intended concurrency"].isin( + selected_concs + ) ] # Evaluate once so all branches below stay consistent @@ -12558,13 +12581,13 @@ def decode_filters_from_url(): temp_df = temp_df[ temp_df["multiturn_isl_osl"] == selected_multiturn_isl_osl ] - if selected_mt_turns is not None: + if selected_mt_turns: temp_df = temp_df[temp_df["turns"].isin(selected_mt_turns)] - if selected_mt_prefix_tokens is not None: + if selected_mt_prefix_tokens: temp_df = temp_df[ temp_df["prefix_tokens"].isin(selected_mt_prefix_tokens) ] - if selected_mt_prefix_count is not None: + if selected_mt_prefix_count: temp_df = temp_df[ temp_df["prefix_count"].isin(selected_mt_prefix_count) ] @@ -12735,13 +12758,13 @@ def decode_filters_from_url(): temp_df = temp_df[ temp_df["multiturn_isl_osl"] == selected_multiturn_isl_osl ] - if selected_mt_turns is not None: + if selected_mt_turns: temp_df = temp_df[temp_df["turns"].isin(selected_mt_turns)] - if selected_mt_prefix_tokens is not None: + if selected_mt_prefix_tokens: temp_df = temp_df[ temp_df["prefix_tokens"].isin(selected_mt_prefix_tokens) ] - if selected_mt_prefix_count is not None: + if selected_mt_prefix_count: temp_df = temp_df[ temp_df["prefix_count"].isin(selected_mt_prefix_count) ] @@ -13217,13 +13240,13 @@ def _fmt(v): str(int(v)) if isinstance(v, float) and v == int(v) else str(v) ) - if selected_mt_turns is not None: + if selected_mt_turns: desired_params["mt_turns"] = ",".join(map(_fmt, selected_mt_turns)) - if selected_mt_prefix_tokens is not None: + if selected_mt_prefix_tokens: desired_params["mt_prefix_tokens"] = ",".join( map(_fmt, selected_mt_prefix_tokens) ) - if selected_mt_prefix_count is not None: + if selected_mt_prefix_count: desired_params["mt_prefix_count"] = ",".join( map(_fmt, selected_mt_prefix_count) ) From 3685c003eec5ae6dbde42dcf3c3844dab0c15135 Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 14:19:49 -0400 Subject: [PATCH 09/17] bazinga: address PR review feedback (#101) Guard per_turn_df["turns"] against missing column using .get() with default Series(1), matching the pattern already used for prefix_tokens, prefix_count, spec_decoding, and prefix_caching. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/dashboard.py b/dashboard.py index 7886ab5..5f538de 100644 --- a/dashboard.py +++ b/dashboard.py @@ -11811,7 +11811,11 @@ def decode_filters_from_url(): ) # Mirror the same normalizations applied to df below so _apply_filters # types match the sidebar multiselect values (which are sourced from df). - per_turn_df["turns"] = per_turn_df["turns"].fillna(1).astype(int) + per_turn_df["turns"] = ( + per_turn_df.get("turns", pd.Series(1, index=per_turn_df.index)) + .fillna(1) + .astype(int) + ) _str_norm = lambda v: ( # noqa: E731 str(int(float(v))) if v != "" and str(v) not in ("", "nan") else "" ) From 409da3fd227b9dfb64720c4d54c7c4928b2e0051 Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 16:11:05 -0400 Subject: [PATCH 10/17] fix: include Turn concurrency multiselect in shared URL MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add pp_turn_conc → perf_plots_turn_concurrency to SECTION_FILTER_KEYS and MULTISELECT_SESSION_KEYS (with NUMERIC_LIST for int parsing) so the selected concurrency level(s) survive copy-paste URL sharing. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/dashboard.py b/dashboard.py index 5f538de..0f00159 100644 --- a/dashboard.py +++ b/dashboard.py @@ -11437,6 +11437,7 @@ def main(): "pp_x": "perf_plots_x_axis", "pp_y": "perf_plots_y_axis", "pp_conc": "perf_plots_max_concurrency", + "pp_turn_conc": "perf_plots_turn_concurrency", }, "pareto": { "par_model": "pareto_model_select", @@ -11684,8 +11685,12 @@ def decode_filters_from_url(): "trends_tp_multi", "energy_accelerator_filter", "energy_model_filter", + "perf_plots_turn_concurrency", + } + NUMERIC_LIST_SESSION_KEYS = { + "trends_tp_multi", + "perf_plots_turn_concurrency", } - NUMERIC_LIST_SESSION_KEYS = {"trends_tp_multi"} INT_SESSION_KEYS = { "perf_plots_max_concurrency", "model_comparison_concurrency", From 56098740a6c589e5a5be9f60d3b2caf03481d3ab Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 16:15:01 -0400 Subject: [PATCH 11/17] fix: decode pp_turn_conc as int not float to match multiselect options NUMERIC_LIST_SESSION_KEYS converts to float, but the Concurrency multiselect options are integers. Add INT_LIST_SESSION_KEYS and move perf_plots_turn_concurrency there so URL-restored selections correctly match the int options in the widget. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/dashboard.py b/dashboard.py index 0f00159..d989af0 100644 --- a/dashboard.py +++ b/dashboard.py @@ -11687,10 +11687,8 @@ def decode_filters_from_url(): "energy_model_filter", "perf_plots_turn_concurrency", } - NUMERIC_LIST_SESSION_KEYS = { - "trends_tp_multi", - "perf_plots_turn_concurrency", - } + NUMERIC_LIST_SESSION_KEYS = {"trends_tp_multi"} + INT_LIST_SESSION_KEYS = {"perf_plots_turn_concurrency"} INT_SESSION_KEYS = { "perf_plots_max_concurrency", "model_comparison_concurrency", @@ -11716,6 +11714,14 @@ def decode_filters_from_url(): ): converted.append(float(v)) parts = converted + elif ss_key in INT_LIST_SESSION_KEYS: + converted = [] + for v in parts: + with contextlib.suppress( + ValueError, OverflowError + ): + converted.append(int(v)) + parts = converted url_section_filters[ss_key] = parts elif ss_key in INT_SESSION_KEYS: with contextlib.suppress(ValueError): From a1f4749e8eeaae70911f4ea185473e5870eba64e Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 16:16:31 -0400 Subject: [PATCH 12/17] fix: push pp_turn_conc to URL directly from fragment @st.fragment reruns dont trigger the parent encode_filters_to_url, so the concurrency selection was never written to st.query_params. Update it inline after the multiselect renders. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/dashboard.py b/dashboard.py index d989af0..b9f2b71 100644 --- a/dashboard.py +++ b/dashboard.py @@ -4013,6 +4013,13 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander st.caption( "💡 Select multiple concurrency levels to compare turn curves side by side." ) + # Push directly to URL — fragment reruns don't trigger the main + # encode_filters_to_url call, so we update query_params here. + with contextlib.suppress(Exception): + if selected_concs: + st.query_params["pp_turn_conc"] = ",".join( + map(str, selected_concs) + ) if selected_concs: per_turn_plot_df = per_turn_plot_df[ per_turn_plot_df["intended concurrency"].isin( From 857f741b80eedabfff63b16a16e608b723d23812 Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 16:21:18 -0400 Subject: [PATCH 13/17] fix: push all perf-plots URL params from fragment pp_x, pp_y, pp_conc, and pp_turn_conc are all written directly to st.query_params after the chart renders, since @st.fragment reruns do not fire the parent from_dict call. This ensures the URL bar stays accurate for copy-paste sharing when any of these widgets change. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 22 +++++++++++++++------- 1 file changed, 15 insertions(+), 7 deletions(-) diff --git a/dashboard.py b/dashboard.py index b9f2b71..d3b8120 100644 --- a/dashboard.py +++ b/dashboard.py @@ -4013,13 +4013,6 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander st.caption( "💡 Select multiple concurrency levels to compare turn curves side by side." ) - # Push directly to URL — fragment reruns don't trigger the main - # encode_filters_to_url call, so we update query_params here. - with contextlib.suppress(Exception): - if selected_concs: - st.query_params["pp_turn_conc"] = ",".join( - map(str, selected_concs) - ) if selected_concs: per_turn_plot_df = per_turn_plot_df[ per_turn_plot_df["intended concurrency"].isin( @@ -4148,6 +4141,21 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander with caption_col2: st.caption("📜 **Tip**: Scroll within the legend box to see all runs") + # Push all perf-plots URL params directly — fragment reruns don't trigger + # the parent encode_filters_to_url / from_dict, so widgets changed inside + # the fragment would otherwise leave the URL stale. + with contextlib.suppress(Exception): + st.query_params["pp_x"] = x_axis_label + st.query_params["pp_y"] = y_axis_label + if x_axis == "intended concurrency": + conc_val = st.session_state.get("perf_plots_max_concurrency") + if conc_val is not None: + st.query_params["pp_conc"] = str(conc_val) + elif x_axis == "turn_index": + turn_conc = st.session_state.get("perf_plots_turn_concurrency") + if turn_conc: + st.query_params["pp_turn_conc"] = ",".join(map(str, turn_conc)) + def load_pareto_data(csv_file_path, preloaded_df=None): """Load benchmark results from CSV file or S3 for Pareto analysis. From b8f12e5c7355d35e74c66e25146b7d856255fc23 Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 16:30:53 -0400 Subject: [PATCH 14/17] feat(manual-import): add turn_index column to manual run CSV output MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add turn_index as empty string for aggregate rows so manually-imported CSVs are schema-compatible with the dashboard's per-turn logic (PR #101). Per-turn rows (turn_index=0/1/2) are generated by the Forge pipeline when request-level turn data is preserved in the benchmark JSON. Update README: 52 → 53 columns, add turn_index to column table (#7), clarify that manual import produces aggregate rows only. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- manual_runs/scripts/vllm/README.md | 97 ++++++++++--------- .../vllm/import_manual_runs_json_v2.py | 1 + 2 files changed, 50 insertions(+), 48 deletions(-) diff --git a/manual_runs/scripts/vllm/README.md b/manual_runs/scripts/vllm/README.md index e5e17d7..e20a24d 100644 --- a/manual_runs/scripts/vllm/README.md +++ b/manual_runs/scripts/vllm/README.md @@ -112,7 +112,7 @@ python import_manual_runs_json_v2.py \ --csv-file "gemma-multiturn.csv" ``` -> **Note:** `turns`, `prefix_tokens`, `prefix_count`, and `request_type` are auto-detected from the guidellm JSON. +> **Note:** `turns`, `prefix_tokens`, `prefix_count`, and `request_type` are auto-detected from the guidellm JSON. `turn_index` is always empty in manual-import output (aggregate rows only) — per-turn rows are generated by the Forge pipeline when `turn_index` data is available in the benchmark JSON. ## Appending to Consolidated Dashboard @@ -124,7 +124,7 @@ tail -n +2 my-benchmark.csv >> ../../../consolidated_dashboard.csv ## Output CSV Columns -The script outputs 52 columns compatible with the performance dashboard: +The script outputs 53 columns compatible with the performance dashboard: | # | Column | Description | | --- | ------------------------- | ---------------------------------------------------- | @@ -134,52 +134,53 @@ The script outputs 52 columns compatible with the performance dashboard: | 4 | `version` | Framework version | | 5 | `prompt toks` | Configured prompt token count | | 6 | `output toks` | Configured output token count | -| 7 | `TP` | Tensor parallelism size | -| 8 | `measured concurrency` | Actual measured concurrency | -| 9 | `intended concurrency` | Requested concurrency (streams) | -| 10 | `measured rps` | Requests per second | -| 11 | `output_tok/sec` | Output tokens per second | -| 12 | `total_tok/sec` | Total tokens per second | -| 13 | `prompt_token_count_mean` | Mean prompt token count | -| 14 | `prompt_token_count_p99` | P99 prompt token count | -| 15 | `output_token_count_mean` | Mean output token count | -| 16 | `output_token_count_p99` | P99 output token count | -| 17 | `ttft_median` | Time to first token - median (ms) | -| 18 | `ttft_p95` | Time to first token - P95 (ms) | -| 19 | `ttft_p1` | Time to first token - P1 (ms) | -| 20 | `ttft_p999` | Time to first token - P99.9 (ms) | -| 21 | `tpot_median` | Time per output token - median (ms) | -| 22 | `tpot_p95` | Time per output token - P95 (ms) | -| 23 | `tpot_p99` | Time per output token - P99 (ms) | -| 24 | `tpot_p999` | Time per output token - P99.9 (ms) | -| 25 | `tpot_p1` | Time per output token - P1 (ms) | -| 26 | `itl_median` | Inter-token latency - median (ms) | -| 27 | `itl_p95` | Inter-token latency - P95 (ms) | -| 28 | `itl_p999` | Inter-token latency - P99.9 (ms) | -| 29 | `itl_p1` | Inter-token latency - P1 (ms) | -| 30 | `request_latency_median` | End-to-end request latency - median (s) | -| 31 | `request_latency_min` | End-to-end request latency - minimum (s) | -| 32 | `request_latency_max` | End-to-end request latency - maximum (s) | -| 33 | `successful_requests` | Number of successful requests | -| 34 | `errored_requests` | Number of errored requests | -| 35 | `uuid` | Unique benchmark run ID | -| 36 | `ttft_mean` | Time to first token - mean (ms) | -| 37 | `ttft_p99` | Time to first token - P99 (ms) | -| 38 | `itl_mean` | Inter-token latency - mean (ms) | -| 39 | `itl_p99` | Inter-token latency - P99 (ms) | -| 40 | `runtime_args` | Server configuration arguments | -| 41 | `guidellm_start_time_ms` | Benchmark start time (epoch ms) | -| 42 | `guidellm_end_time_ms` | Benchmark end time (epoch ms) | -| 43 | `image_tag` | Container image used | -| 44 | `guidellm_version` | guidellm version used | -| 45 | `DP` | Data parallelism size (empty for TP runs) | -| 46 | `dataset` | Real dataset name (empty for synthetic runs) | -| 47 | `spec_decoding` | Speculative decoding method (empty if none) | -| 48 | `prefix_caching` | Prefix caching status (`yes`, `no`, or empty) | -| 49 | `turns` | Conversation turns for multiturn benchmarks | -| 50 | `prefix_tokens` | Prefix token count (auto-detected from JSON) | -| 51 | `prefix_count` | Prefix count (auto-detected from JSON) | -| 52 | `request_type` | GuideLLM API endpoint type (auto-detected from JSON) | +| 7 | `turn_index` | Per-turn breakdown index (empty for aggregate rows) | +| 9 | `TP` | Tensor parallelism size | +| 9 | `measured concurrency` | Actual measured concurrency | +| 10 | `intended concurrency` | Requested concurrency (streams) | +| 11 | `measured rps` | Requests per second | +| 12 | `output_tok/sec` | Output tokens per second | +| 13 | `total_tok/sec` | Total tokens per second | +| 14 | `prompt_token_count_mean` | Mean prompt token count | +| 15 | `prompt_token_count_p99` | P99 prompt token count | +| 16 | `output_token_count_mean` | Mean output token count | +| 17 | `output_token_count_p99` | P99 output token count | +| 18 | `ttft_median` | Time to first token - median (ms) | +| 19 | `ttft_p95` | Time to first token - P95 (ms) | +| 20 | `ttft_p1` | Time to first token - P1 (ms) | +| 21 | `ttft_p999` | Time to first token - P99.9 (ms) | +| 22 | `tpot_median` | Time per output token - median (ms) | +| 23 | `tpot_p95` | Time per output token - P95 (ms) | +| 24 | `tpot_p99` | Time per output token - P99 (ms) | +| 25 | `tpot_p999` | Time per output token - P99.9 (ms) | +| 26 | `tpot_p1` | Time per output token - P1 (ms) | +| 27 | `itl_median` | Inter-token latency - median (ms) | +| 28 | `itl_p95` | Inter-token latency - P95 (ms) | +| 29 | `itl_p999` | Inter-token latency - P99.9 (ms) | +| 30 | `itl_p1` | Inter-token latency - P1 (ms) | +| 31 | `request_latency_median` | End-to-end request latency - median (s) | +| 32 | `request_latency_min` | End-to-end request latency - minimum (s) | +| 33 | `request_latency_max` | End-to-end request latency - maximum (s) | +| 34 | `successful_requests` | Number of successful requests | +| 35 | `errored_requests` | Number of errored requests | +| 36 | `uuid` | Unique benchmark run ID | +| 37 | `ttft_mean` | Time to first token - mean (ms) | +| 38 | `ttft_p99` | Time to first token - P99 (ms) | +| 39 | `itl_mean` | Inter-token latency - mean (ms) | +| 40 | `itl_p99` | Inter-token latency - P99 (ms) | +| 41 | `runtime_args` | Server configuration arguments | +| 42 | `guidellm_start_time_ms` | Benchmark start time (epoch ms) | +| 43 | `guidellm_end_time_ms` | Benchmark end time (epoch ms) | +| 44 | `image_tag` | Container image used | +| 45 | `guidellm_version` | guidellm version used | +| 46 | `DP` | Data parallelism size (empty for TP runs) | +| 47 | `dataset` | Real dataset name (empty for synthetic runs) | +| 48 | `spec_decoding` | Speculative decoding method (empty if none) | +| 49 | `prefix_caching` | Prefix caching status (`yes`, `no`, or empty) | +| 50 | `turns` | Conversation turns for multiturn benchmarks | +| 51 | `prefix_tokens` | Prefix token count (auto-detected from JSON) | +| 52 | `prefix_count` | Prefix count (auto-detected from JSON) | +| 53 | `request_type` | GuideLLM API endpoint type (auto-detected from JSON) | ## Notes diff --git a/manual_runs/scripts/vllm/import_manual_runs_json_v2.py b/manual_runs/scripts/vllm/import_manual_runs_json_v2.py index de3e369..d9157da 100644 --- a/manual_runs/scripts/vllm/import_manual_runs_json_v2.py +++ b/manual_runs/scripts/vllm/import_manual_runs_json_v2.py @@ -208,6 +208,7 @@ def get_percentile(metrics_dict, key): "dataset": dataset, "spec_decoding": spec_decoding, "prefix_caching": prefix_caching, + "turn_index": "", "turns": turns, "prefix_tokens": detected_prefix_tokens if detected_prefix_tokens is not None From e2ac06130b60d37f2193cbecf8d66696542dc853 Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 16:32:45 -0400 Subject: [PATCH 15/17] fix: correct turn_index position in README column table MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Row numbering was corrupted by the previous renumber script — TP and measured concurrency both landed at row 9. Fix: turn_index=7, TP=8, measured concurrency=9, continuing sequentially to request_type=53. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- manual_runs/scripts/vllm/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/manual_runs/scripts/vllm/README.md b/manual_runs/scripts/vllm/README.md index e20a24d..949dd45 100644 --- a/manual_runs/scripts/vllm/README.md +++ b/manual_runs/scripts/vllm/README.md @@ -135,7 +135,7 @@ The script outputs 53 columns compatible with the performance dashboard: | 5 | `prompt toks` | Configured prompt token count | | 6 | `output toks` | Configured output token count | | 7 | `turn_index` | Per-turn breakdown index (empty for aggregate rows) | -| 9 | `TP` | Tensor parallelism size | +| 8 | `TP` | Tensor parallelism size | | 9 | `measured concurrency` | Actual measured concurrency | | 10 | `intended concurrency` | Requested concurrency (streams) | | 11 | `measured rps` | Requests per second | From 8e7805ce66554d9250310774d4681dce04468ac5 Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 16:35:09 -0400 Subject: [PATCH 16/17] fix: move turn_index adjacent to turns in README column table The script emits turn_index right before turns in the output dict. The README should reflect actual output order, not the RHAIIS schema field position (which pandas aligns by name on concat anyway). turn_index: #49, turns: #50, prefix_tokens: #51, prefix_count: #52, request_type: #53. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- manual_runs/scripts/vllm/README.md | 86 +++++++++++++++--------------- 1 file changed, 43 insertions(+), 43 deletions(-) diff --git a/manual_runs/scripts/vllm/README.md b/manual_runs/scripts/vllm/README.md index 949dd45..11a9448 100644 --- a/manual_runs/scripts/vllm/README.md +++ b/manual_runs/scripts/vllm/README.md @@ -134,49 +134,49 @@ The script outputs 53 columns compatible with the performance dashboard: | 4 | `version` | Framework version | | 5 | `prompt toks` | Configured prompt token count | | 6 | `output toks` | Configured output token count | -| 7 | `turn_index` | Per-turn breakdown index (empty for aggregate rows) | -| 8 | `TP` | Tensor parallelism size | -| 9 | `measured concurrency` | Actual measured concurrency | -| 10 | `intended concurrency` | Requested concurrency (streams) | -| 11 | `measured rps` | Requests per second | -| 12 | `output_tok/sec` | Output tokens per second | -| 13 | `total_tok/sec` | Total tokens per second | -| 14 | `prompt_token_count_mean` | Mean prompt token count | -| 15 | `prompt_token_count_p99` | P99 prompt token count | -| 16 | `output_token_count_mean` | Mean output token count | -| 17 | `output_token_count_p99` | P99 output token count | -| 18 | `ttft_median` | Time to first token - median (ms) | -| 19 | `ttft_p95` | Time to first token - P95 (ms) | -| 20 | `ttft_p1` | Time to first token - P1 (ms) | -| 21 | `ttft_p999` | Time to first token - P99.9 (ms) | -| 22 | `tpot_median` | Time per output token - median (ms) | -| 23 | `tpot_p95` | Time per output token - P95 (ms) | -| 24 | `tpot_p99` | Time per output token - P99 (ms) | -| 25 | `tpot_p999` | Time per output token - P99.9 (ms) | -| 26 | `tpot_p1` | Time per output token - P1 (ms) | -| 27 | `itl_median` | Inter-token latency - median (ms) | -| 28 | `itl_p95` | Inter-token latency - P95 (ms) | -| 29 | `itl_p999` | Inter-token latency - P99.9 (ms) | -| 30 | `itl_p1` | Inter-token latency - P1 (ms) | -| 31 | `request_latency_median` | End-to-end request latency - median (s) | -| 32 | `request_latency_min` | End-to-end request latency - minimum (s) | -| 33 | `request_latency_max` | End-to-end request latency - maximum (s) | -| 34 | `successful_requests` | Number of successful requests | -| 35 | `errored_requests` | Number of errored requests | -| 36 | `uuid` | Unique benchmark run ID | -| 37 | `ttft_mean` | Time to first token - mean (ms) | -| 38 | `ttft_p99` | Time to first token - P99 (ms) | -| 39 | `itl_mean` | Inter-token latency - mean (ms) | -| 40 | `itl_p99` | Inter-token latency - P99 (ms) | -| 41 | `runtime_args` | Server configuration arguments | -| 42 | `guidellm_start_time_ms` | Benchmark start time (epoch ms) | -| 43 | `guidellm_end_time_ms` | Benchmark end time (epoch ms) | -| 44 | `image_tag` | Container image used | -| 45 | `guidellm_version` | guidellm version used | -| 46 | `DP` | Data parallelism size (empty for TP runs) | -| 47 | `dataset` | Real dataset name (empty for synthetic runs) | -| 48 | `spec_decoding` | Speculative decoding method (empty if none) | -| 49 | `prefix_caching` | Prefix caching status (`yes`, `no`, or empty) | +| 7 | `TP` | Tensor parallelism size | +| 8 | `measured concurrency` | Actual measured concurrency | +| 9 | `intended concurrency` | Requested concurrency (streams) | +| 10 | `measured rps` | Requests per second | +| 11 | `output_tok/sec` | Output tokens per second | +| 12 | `total_tok/sec` | Total tokens per second | +| 13 | `prompt_token_count_mean` | Mean prompt token count | +| 14 | `prompt_token_count_p99` | P99 prompt token count | +| 15 | `output_token_count_mean` | Mean output token count | +| 16 | `output_token_count_p99` | P99 output token count | +| 17 | `ttft_median` | Time to first token - median (ms) | +| 18 | `ttft_p95` | Time to first token - P95 (ms) | +| 19 | `ttft_p1` | Time to first token - P1 (ms) | +| 20 | `ttft_p999` | Time to first token - P99.9 (ms) | +| 21 | `tpot_median` | Time per output token - median (ms) | +| 22 | `tpot_p95` | Time per output token - P95 (ms) | +| 23 | `tpot_p99` | Time per output token - P99 (ms) | +| 24 | `tpot_p999` | Time per output token - P99.9 (ms) | +| 25 | `tpot_p1` | Time per output token - P1 (ms) | +| 26 | `itl_median` | Inter-token latency - median (ms) | +| 27 | `itl_p95` | Inter-token latency - P95 (ms) | +| 28 | `itl_p999` | Inter-token latency - P99.9 (ms) | +| 29 | `itl_p1` | Inter-token latency - P1 (ms) | +| 30 | `request_latency_median` | End-to-end request latency - median (s) | +| 31 | `request_latency_min` | End-to-end request latency - minimum (s) | +| 32 | `request_latency_max` | End-to-end request latency - maximum (s) | +| 33 | `successful_requests` | Number of successful requests | +| 34 | `errored_requests` | Number of errored requests | +| 35 | `uuid` | Unique benchmark run ID | +| 36 | `ttft_mean` | Time to first token - mean (ms) | +| 37 | `ttft_p99` | Time to first token - P99 (ms) | +| 38 | `itl_mean` | Inter-token latency - mean (ms) | +| 39 | `itl_p99` | Inter-token latency - P99 (ms) | +| 40 | `runtime_args` | Server configuration arguments | +| 41 | `guidellm_start_time_ms` | Benchmark start time (epoch ms) | +| 42 | `guidellm_end_time_ms` | Benchmark end time (epoch ms) | +| 43 | `image_tag` | Container image used | +| 44 | `guidellm_version` | guidellm version used | +| 45 | `DP` | Data parallelism size (empty for TP runs) | +| 46 | `dataset` | Real dataset name (empty for synthetic runs) | +| 47 | `spec_decoding` | Speculative decoding method (empty if none) | +| 48 | `prefix_caching` | Prefix caching status (`yes`, `no`, or empty) | +| 49 | `turn_index` | Per-turn breakdown index (empty for aggregate rows) | | 50 | `turns` | Conversation turns for multiturn benchmarks | | 51 | `prefix_tokens` | Prefix token count (auto-detected from JSON) | | 52 | `prefix_count` | Prefix count (auto-detected from JSON) | From ea9f941d79bd713192b9ba1975a7d90e62497eec Mon Sep 17 00:00:00 2001 From: aas008 Date: Tue, 6 Oct 2026 17:59:58 -0400 Subject: [PATCH 17/17] fix: address code-review findings in per-turn dashboard integration MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Correctness: - Guard stale perf_plots_turn_concurrency session state: filter stored list to only still-valid values, fall back to max if all stale (previously raised StreamlitInvalidValueError when data reloads) - Add .fillna("?") to run_identifier string concat for accelerator/ model/version, preventing NaN color labels in plotly turn chart - Guard else-branch when x_axis='turn_index' but per_turn_plot_df is empty — show info message instead of sorting aggregate df by NaN turn_index column - Narrow broad contextlib.suppress(Exception) in URL push block to a try/except with a pass, keeping the intent but letting TypeError/ AttributeError from bad values surface during development Simplification: - Factor _str_norm into a proper function shared by both per_turn_df and df preprocessing paths — eliminates duplicated logic - Remove redundant reset_index(drop=True, inplace=True) in filtered data section (concat branch already ends with .reset_index) - Remove double per_turn_plot_df.copy() — plot_label can be assigned directly (per_turn_df.copy() earlier in the block is sufficient) Co-Authored-By: Claude Sonnet 4.6 (1M context) --- dashboard.py | 70 +++++++++++++++++++++++++++------------------------- 1 file changed, 36 insertions(+), 34 deletions(-) diff --git a/dashboard.py b/dashboard.py index d3b8120..7b9ec8c 100644 --- a/dashboard.py +++ b/dashboard.py @@ -3876,11 +3876,11 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander if per_turn_df is not None and not per_turn_df.empty: per_turn_plot_df = per_turn_df.copy() per_turn_plot_df["run_identifier"] = ( - per_turn_plot_df["accelerator"] + per_turn_plot_df["accelerator"].fillna("?") + " | " - + per_turn_plot_df["model"] + + per_turn_plot_df["model"].fillna("?") + " | " - + per_turn_plot_df["version"] + + per_turn_plot_df["version"].fillna("?") + " | TP=" + per_turn_plot_df["TP"].apply( lambda x: str(int(x)) if pd.notna(x) else "N/A" @@ -3998,11 +3998,18 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander ) if concurrency_values: _turn_conc_key = "perf_plots_turn_concurrency" - if _turn_conc_key not in st.session_state or not any( - c in concurrency_values - for c in (st.session_state.get(_turn_conc_key) or []) - ): + if _turn_conc_key not in st.session_state: st.session_state[_turn_conc_key] = [max(concurrency_values)] + else: + # Filter stored list to only still-valid values; reset to max if all stale + _valid = [ + c + for c in (st.session_state[_turn_conc_key] or []) + if c in concurrency_values + ] + st.session_state[_turn_conc_key] = _valid or [ + max(concurrency_values) + ] selected_concs = st.multiselect( "Concurrency", options=concurrency_values, @@ -4065,7 +4072,6 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander _isl_osl_subtitle = f"
ISL/OSL: {', '.join(sorted(set(pair_labels)))}" if _is_turn_view: - per_turn_plot_df = per_turn_plot_df.copy() per_turn_plot_df["_plot_label"] = ( per_turn_plot_df["run_identifier"] + " | conc=" @@ -4104,6 +4110,11 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander legend_title_text="Run Details (Accelerator | Model | Version | TP | Concurrency)", legend={"font": {"size": 14}}, ) + elif x_axis == "turn_index": + # Turn x-axis selected but per_turn_plot_df is empty — don't fall through + # to the aggregate sort which would sort by NaN turn_index values. + st.info("No per-turn data available for the current filter selection.") + fig = None else: fig = px.line( filtered_df_sorted.sort_values(by=x_axis), @@ -4144,17 +4155,23 @@ def render_performance_plots_section(filtered_df, per_turn_df=None, use_expander # Push all perf-plots URL params directly — fragment reruns don't trigger # the parent encode_filters_to_url / from_dict, so widgets changed inside # the fragment would otherwise leave the URL stale. - with contextlib.suppress(Exception): + # Suppress only Streamlit API errors (e.g. no session context outside a run); + # let TypeError/AttributeError from bad values surface for debugging. + try: st.query_params["pp_x"] = x_axis_label st.query_params["pp_y"] = y_axis_label if x_axis == "intended concurrency": conc_val = st.session_state.get("perf_plots_max_concurrency") if conc_val is not None: - st.query_params["pp_conc"] = str(conc_val) + st.query_params["pp_conc"] = str(int(conc_val)) elif x_axis == "turn_index": turn_conc = st.session_state.get("perf_plots_turn_concurrency") if turn_conc: - st.query_params["pp_turn_conc"] = ",".join(map(str, turn_conc)) + st.query_params["pp_turn_conc"] = ",".join( + str(int(c)) for c in turn_conc + ) + except Exception: # noqa: BLE001 + pass # Outside Streamlit session context — no-op def load_pareto_data(csv_file_path, preloaded_df=None): @@ -10668,10 +10685,9 @@ def render_filtered_data_section(filtered_df, per_turn_df=None, use_expander=Tru .reset_index(drop=True) ) else: - display_filtered_df = filtered_df.copy() + display_filtered_df = filtered_df.copy().reset_index(drop=True) if "turn_index" in display_filtered_df.columns: display_filtered_df = display_filtered_df.drop(columns=["turn_index"]) - display_filtered_df.reset_index(drop=True, inplace=True) display_filtered_df.insert(0, "Row #", range(1, len(display_filtered_df) + 1)) # Add Run Date column from guidellm_start_time_ms (epoch milliseconds) @@ -11796,6 +11812,11 @@ def decode_filters_from_url(): df["prefix_caching"] = df["prefix_caching"].fillna("").astype(str) df["prefix_caching"] = df["prefix_caching"].replace("", "no") + # Shared normalizer for numeric-string columns (prefix_tokens, prefix_count) + # used by both the per_turn_df and df preprocessing paths below. + def _str_norm(v): + return str(int(float(v))) if v != "" and str(v) not in ("", "nan") else "" + if "turn_index" in df.columns: _ti = df["turn_index"].astype(str).str.strip() per_turn_df = df[df["turn_index"].notna() & (_ti != "") & (_ti != "nan")].copy() @@ -11842,9 +11863,6 @@ def decode_filters_from_url(): .fillna(1) .astype(int) ) - _str_norm = lambda v: ( # noqa: E731 - str(int(float(v))) if v != "" and str(v) not in ("", "nan") else "" - ) per_turn_df["prefix_tokens"] = ( per_turn_df.get("prefix_tokens", pd.Series("", index=per_turn_df.index)) .fillna("") @@ -11878,27 +11896,11 @@ def decode_filters_from_url(): if "prefix_tokens" not in df.columns: df["prefix_tokens"] = "" - df["prefix_tokens"] = ( - df["prefix_tokens"] - .fillna("") - .apply( - lambda v: ( - str(int(float(v))) if v != "" and str(v) not in ("", "nan") else "" - ) - ) - ) + df["prefix_tokens"] = df["prefix_tokens"].fillna("").apply(_str_norm) if "prefix_count" not in df.columns: df["prefix_count"] = "" - df["prefix_count"] = ( - df["prefix_count"] - .fillna("") - .apply( - lambda v: ( - str(int(float(v))) if v != "" and str(v) not in ("", "nan") else "" - ) - ) - ) + df["prefix_count"] = df["prefix_count"].fillna("").apply(_str_norm) if "request_type" not in df.columns: df["request_type"] = ""