antoineedy commited on
Commit
8c6ed3c
·
verified ·
1 Parent(s): c16c537

Upload app.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. app.py +105 -16
app.py CHANGED
@@ -1,4 +1,8 @@
 
 
1
  import gradio as gr
 
 
2
 
3
  from app.utils import (
4
  add_rank_and_format,
@@ -161,25 +165,11 @@ def main():
161
  ></iframe>
162
  """
163
  )
 
164
  with gr.TabItem("ViDoRe V3 (Pipeline)", id="vidore-v3-pipeline"):
165
  gr.Markdown("# ViDoRe V3 (Pipeline Evaluation): Retrieval Performance for Complex Pipelines ⚙️")
166
  gr.Markdown("### Assessing retrieval performance, latency, and compute costs of complex retrieval pipelines")
167
 
168
- # gr.Markdown(
169
- # """
170
- # This leaderboard displays abstract pipeline evaluation results for ViDoRe V3 on **english-only queries**.
171
- # Unlike model-only evaluations, these results showcase what full retrieval pipelines can achieve.
172
-
173
- # Metrics regarding time complexity are also provided in the form of ("Indexing latency (second/doc)" and "Search latency (second/query)").
174
- # Those are self-reported, highly hardware dependent, and should not be considered absolute.
175
- # Nevertheless we felt it was quite important to ensure transparency and provide users with a rough understanding of the computational demands of each pipeline.
176
-
177
- # Results are sourced from the [vidore-benchmark repository](https://github.com/illuin-tech/vidore-benchmark/tree/vidore_v3_pipeline/results).
178
-
179
- # Those results are not directly comparable to the base ViDoRe v3 results, since they are computed on a different set of queries (english-only queries).
180
- # """
181
- # )
182
-
183
  gr.Markdown(
184
  """
185
  This leaderboard ranks full retrieval pipelines on **English-only queries** for **ViDoRe V3**. Instead of just testing standalone models, we evaluate real-world, multi-step retrieval systems. This includes everything from basic retrievers to advanced setups using AI agents, query reformulation, hybrid search, and any other creative retrieval pipeline one can imagine.
@@ -223,6 +213,79 @@ def main():
223
  datatype_pipeline = ["number", "markdown", "number", "number", "number"] + ["number"] * len(datasets_columns_pipeline)
224
  dataframe_pipeline = gr.Dataframe(data_pipeline, datatype=datatype_pipeline, type="pandas", elem_id="pipeline-table")
225
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
226
  def update_data_pipeline(metric, search_term, selected_columns):
227
  pipeline_handler.get_pipeline_data()
228
  data = pipeline_handler.render_df(metric, "english")
@@ -247,6 +310,10 @@ def main():
247
  inputs=[metric_dropdown_pipeline],
248
  outputs=dataframe_pipeline,
249
  concurrency_limit=20,
 
 
 
 
250
  )
251
 
252
  with gr.Row():
@@ -264,20 +331,42 @@ def main():
264
  df = pipeline_handler.render_df(metric, "english")
265
  return add_rank_and_format(df, benchmark_version=3, is_pipeline=True)
266
 
 
267
  metric_dropdown_pipeline.change(
268
  refresh_pipeline_data,
269
  inputs=[metric_dropdown_pipeline],
270
  outputs=dataframe_pipeline,
 
 
 
 
271
  )
 
272
  research_textbox_pipeline.submit(
273
  lambda metric, search_term, selected_columns: update_data_pipeline(metric, search_term, selected_columns),
274
  inputs=[metric_dropdown_pipeline, research_textbox_pipeline, column_checkboxes_pipeline],
275
  outputs=dataframe_pipeline,
 
 
 
 
276
  )
 
277
  column_checkboxes_pipeline.change(
278
  lambda metric, search_term, selected_columns: update_data_pipeline(metric, search_term, selected_columns),
279
  inputs=[metric_dropdown_pipeline, research_textbox_pipeline, column_checkboxes_pipeline],
280
  outputs=dataframe_pipeline,
 
 
 
 
 
 
 
 
 
 
 
281
  )
282
 
283
  gr.Markdown(
@@ -297,7 +386,7 @@ def main():
297
  eprint={2407.01449},
298
  archivePrefix={arXiv},
299
  primaryClass={cs.IR},
300
- url={https://arxiv.org/abs/2407.01449},
301
  }
302
  @misc{loison2026vidore,
303
  title={ViDoRe V3: A Comprehensive Evaluation of Retrieval Augmented Generation in Complex Real-World Scenarios},
 
1
+ import re
2
+
3
  import gradio as gr
4
+ import pandas as pd
5
+ import plotly.express as px
6
 
7
  from app.utils import (
8
  add_rank_and_format,
 
165
  ></iframe>
166
  """
167
  )
168
+
169
  with gr.TabItem("ViDoRe V3 (Pipeline)", id="vidore-v3-pipeline"):
170
  gr.Markdown("# ViDoRe V3 (Pipeline Evaluation): Retrieval Performance for Complex Pipelines ⚙️")
171
  gr.Markdown("### Assessing retrieval performance, latency, and compute costs of complex retrieval pipelines")
172
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
173
  gr.Markdown(
174
  """
175
  This leaderboard ranks full retrieval pipelines on **English-only queries** for **ViDoRe V3**. Instead of just testing standalone models, we evaluate real-world, multi-step retrieval systems. This includes everything from basic retrievers to advanced setups using AI agents, query reformulation, hybrid search, and any other creative retrieval pipeline one can imagine.
 
213
  datatype_pipeline = ["number", "markdown", "number", "number", "number"] + ["number"] * len(datasets_columns_pipeline)
214
  dataframe_pipeline = gr.Dataframe(data_pipeline, datatype=datatype_pipeline, type="pandas", elem_id="pipeline-table")
215
 
216
+ def clean_pipeline_name(name):
217
+ if not isinstance(name, str):
218
+ return str(name)
219
+ # Remove Markdown links [text](url) -> text
220
+ name = re.sub(r'\[([^\]]+)\]\([^\)]+\)', r'\1', name)
221
+ # Remove HTML tags <a href="...">text</a> -> text
222
+ name = re.sub(r'<[^>]+>', '', name)
223
+ return name.strip()
224
+
225
+ def create_pipeline_plot(df, latency_col):
226
+ if df is None or len(df) == 0:
227
+ return None
228
+
229
+ # Ensure expected columns exist
230
+ if latency_col not in df.columns or "Average Score" not in df.columns or "Pipeline" not in df.columns:
231
+ return None
232
+
233
+ # Clean the dataframe for plotting
234
+ plot_df = df.copy()
235
+
236
+ # Strip HTML and Markdown for clean hover text
237
+ plot_df["Cleaned Pipeline"] = plot_df["Pipeline"].apply(clean_pipeline_name)
238
+
239
+ # Force numerical conversion (coerces weird strings to NaN)
240
+ plot_df[latency_col] = pd.to_numeric(plot_df[latency_col], errors='coerce')
241
+ plot_df["Average Score"] = pd.to_numeric(plot_df["Average Score"], errors='coerce')
242
+
243
+ # Drop NaNs
244
+ plot_df = plot_df.dropna(subset=[latency_col, "Average Score"])
245
+
246
+ # Filter out non-positive values to prevent log-scale math errors
247
+ plot_df = plot_df[plot_df[latency_col] > 0]
248
+
249
+ # Sort by the x-axis to prevent Plotly from drawing out-of-order categorical ticks
250
+ plot_df = plot_df.sort_values(by=latency_col)
251
+
252
+ if len(plot_df) == 0:
253
+ return None
254
+
255
+ fig = px.scatter(
256
+ plot_df,
257
+ x=latency_col,
258
+ y="Average Score",
259
+ hover_name="Cleaned Pipeline", # Use the clean text!
260
+ title=f"Mean Performance vs {latency_col}",
261
+ color="orange",
262
+ )
263
+ fig.update_layout(
264
+ plot_bgcolor='white'
265
+ )
266
+
267
+ # Explicitly force the axis to be log type so Plotly handles the spacing natively
268
+ fig.update_layout(
269
+ xaxis_type="log",
270
+ xaxis_title=latency_col,
271
+ yaxis_title="Average Score"
272
+ )
273
+
274
+ # Style the points
275
+ fig.update_traces(marker=dict(size=12, opacity=0.8, line=dict(width=1, color='DarkSlateGrey')))
276
+ return fig
277
+
278
+ with gr.Row():
279
+ latency_radio = gr.Radio(
280
+ choices=["Search latency (s/query)", "Indexing latency (s/doc)"],
281
+ value="Search latency (s/query)",
282
+ label="Select Latency Metric for X-Axis"
283
+ )
284
+
285
+ with gr.Row():
286
+ initial_fig = create_pipeline_plot(data_pipeline, "Search latency (s/query)")
287
+ performance_plot = gr.Plot(value=initial_fig)
288
+
289
  def update_data_pipeline(metric, search_term, selected_columns):
290
  pipeline_handler.get_pipeline_data()
291
  data = pipeline_handler.render_df(metric, "english")
 
310
  inputs=[metric_dropdown_pipeline],
311
  outputs=dataframe_pipeline,
312
  concurrency_limit=20,
313
+ ).then(
314
+ fn=create_pipeline_plot,
315
+ inputs=[dataframe_pipeline, latency_radio],
316
+ outputs=performance_plot
317
  )
318
 
319
  with gr.Row():
 
331
  df = pipeline_handler.render_df(metric, "english")
332
  return add_rank_and_format(df, benchmark_version=3, is_pipeline=True)
333
 
334
+ # Update dataframe and then update the plot
335
  metric_dropdown_pipeline.change(
336
  refresh_pipeline_data,
337
  inputs=[metric_dropdown_pipeline],
338
  outputs=dataframe_pipeline,
339
+ ).then(
340
+ fn=create_pipeline_plot,
341
+ inputs=[dataframe_pipeline, latency_radio],
342
+ outputs=performance_plot
343
  )
344
+
345
  research_textbox_pipeline.submit(
346
  lambda metric, search_term, selected_columns: update_data_pipeline(metric, search_term, selected_columns),
347
  inputs=[metric_dropdown_pipeline, research_textbox_pipeline, column_checkboxes_pipeline],
348
  outputs=dataframe_pipeline,
349
+ ).then(
350
+ fn=create_pipeline_plot,
351
+ inputs=[dataframe_pipeline, latency_radio],
352
+ outputs=performance_plot
353
  )
354
+
355
  column_checkboxes_pipeline.change(
356
  lambda metric, search_term, selected_columns: update_data_pipeline(metric, search_term, selected_columns),
357
  inputs=[metric_dropdown_pipeline, research_textbox_pipeline, column_checkboxes_pipeline],
358
  outputs=dataframe_pipeline,
359
+ ).then(
360
+ fn=create_pipeline_plot,
361
+ inputs=[dataframe_pipeline, latency_radio],
362
+ outputs=performance_plot
363
+ )
364
+
365
+ # Update plot when the radio button changes
366
+ latency_radio.change(
367
+ fn=create_pipeline_plot,
368
+ inputs=[dataframe_pipeline, latency_radio],
369
+ outputs=performance_plot
370
  )
371
 
372
  gr.Markdown(
 
386
  eprint={2407.01449},
387
  archivePrefix={arXiv},
388
  primaryClass={cs.IR},
389
+ url={[https://arxiv.org/abs/2407.01449](https://arxiv.org/abs/2407.01449)},
390
  }
391
  @misc{loison2026vidore,
392
  title={ViDoRe V3: A Comprehensive Evaluation of Retrieval Augmented Generation in Complex Real-World Scenarios},