add inference / elapsed time
Browse files- app.py +1 -1
- src/display/utils.py +6 -0
- src/populate.py +87 -11
app.py
CHANGED
|
@@ -526,7 +526,7 @@ with gr.Blocks() as demo_submission:
|
|
| 526 |
)
|
| 527 |
with gr.Accordion(
|
| 528 |
f"🔄 Running Evaluation Queue ({len(RUNNING_EVAL_QUEUE_DF)})",
|
| 529 |
-
open=
|
| 530 |
):
|
| 531 |
with gr.Row():
|
| 532 |
running_eval_table = gr.Dataframe(
|
|
|
|
| 526 |
)
|
| 527 |
with gr.Accordion(
|
| 528 |
f"🔄 Running Evaluation Queue ({len(RUNNING_EVAL_QUEUE_DF)})",
|
| 529 |
+
open=True,
|
| 530 |
):
|
| 531 |
with gr.Row():
|
| 532 |
running_eval_table = gr.Dataframe(
|
src/display/utils.py
CHANGED
|
@@ -56,6 +56,8 @@ auto_eval_column_dict.append(["revision", ColumnContent, ColumnContent("Revision
|
|
| 56 |
auto_eval_column_dict.append(["add_special_tokens", ColumnContent, ColumnContent("Add Special Tokens", "bool", False)])
|
| 57 |
auto_eval_column_dict.append(["enable_thinking", ColumnContent, ColumnContent("Enable Thinking", "bool", False)])
|
| 58 |
auto_eval_column_dict.append(["apply_chat_template", ColumnContent, ColumnContent("Apply Chat Template", "bool", False)])
|
|
|
|
|
|
|
| 59 |
auto_eval_column_dict.append(["dummy", ColumnContent, ColumnContent("model_name_for_query", "str", False, dummy=True)])
|
| 60 |
auto_eval_column_dict.append(["row_id", ColumnContent, ColumnContent("ID", "number", False, dummy=True)])
|
| 61 |
|
|
@@ -72,6 +74,10 @@ class EvalQueueColumn: # Queue column (not a dataclass - used as class attribut
|
|
| 72 |
add_special_tokens = ColumnContent("add_special_tokens", "str", True)
|
| 73 |
status = ColumnContent("status", "str", True)
|
| 74 |
submitted_by = ColumnContent("submitted_by", "str", True)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 75 |
apply_chat_template = ColumnContent("apply_chat_template", "bool", False)
|
| 76 |
enable_thinking = ColumnContent("enable_thinking", "bool", False)
|
| 77 |
reasoning_parser = ColumnContent("reasoning_parser", "str", False)
|
|
|
|
| 56 |
auto_eval_column_dict.append(["add_special_tokens", ColumnContent, ColumnContent("Add Special Tokens", "bool", False)])
|
| 57 |
auto_eval_column_dict.append(["enable_thinking", ColumnContent, ColumnContent("Enable Thinking", "bool", False)])
|
| 58 |
auto_eval_column_dict.append(["apply_chat_template", ColumnContent, ColumnContent("Apply Chat Template", "bool", False)])
|
| 59 |
+
auto_eval_column_dict.append(["inference_time", ColumnContent, ColumnContent("Inference Time (s)", "number", False)])
|
| 60 |
+
auto_eval_column_dict.append(["co2_emission", ColumnContent, ColumnContent("CO2 (kg)", "number", False)])
|
| 61 |
auto_eval_column_dict.append(["dummy", ColumnContent, ColumnContent("model_name_for_query", "str", False, dummy=True)])
|
| 62 |
auto_eval_column_dict.append(["row_id", ColumnContent, ColumnContent("ID", "number", False, dummy=True)])
|
| 63 |
|
|
|
|
| 74 |
add_special_tokens = ColumnContent("add_special_tokens", "str", True)
|
| 75 |
status = ColumnContent("status", "str", True)
|
| 76 |
submitted_by = ColumnContent("submitted_by", "str", True)
|
| 77 |
+
submitted_time = ColumnContent("submitted_time", "str", True)
|
| 78 |
+
started_time = ColumnContent("started_time", "str", True)
|
| 79 |
+
elapsed_time = ColumnContent("elapsed_time", "str", True)
|
| 80 |
+
queue_position = ColumnContent("queue_position", "number", True)
|
| 81 |
apply_chat_template = ColumnContent("apply_chat_template", "bool", False)
|
| 82 |
enable_thinking = ColumnContent("enable_thinking", "bool", False)
|
| 83 |
reasoning_parser = ColumnContent("reasoning_parser", "str", False)
|
src/populate.py
CHANGED
|
@@ -1,5 +1,6 @@
|
|
| 1 |
import json
|
| 2 |
import os
|
|
|
|
| 3 |
|
| 4 |
import pandas as pd
|
| 5 |
from huggingface_hub import hf_hub_download
|
|
@@ -49,6 +50,8 @@ def get_leaderboard_df(contents_repo: str, cols: list[str], benchmark_cols: list
|
|
| 49 |
"apply_chat_template": "Apply Chat Template",
|
| 50 |
"model_type": "Type",
|
| 51 |
"model": "model_name_for_query",
|
|
|
|
|
|
|
| 52 |
}
|
| 53 |
# Only rename columns that exist in the dataframe
|
| 54 |
rename_dict = {k: v for k, v in rename_dict.items() if k in df.columns}
|
|
@@ -81,8 +84,76 @@ def get_leaderboard_df(contents_repo: str, cols: list[str], benchmark_cols: list
|
|
| 81 |
return df
|
| 82 |
|
| 83 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 84 |
def get_evaluation_queue_df(save_path: str, cols: list[str]) -> list[pd.DataFrame]:
|
| 85 |
-
"""Creates the different dataframes for the evaluation queues
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 86 |
entries = [entry for entry in os.listdir(save_path) if not entry.startswith(".")]
|
| 87 |
all_evals = []
|
| 88 |
|
|
@@ -91,27 +162,32 @@ def get_evaluation_queue_df(save_path: str, cols: list[str]) -> list[pd.DataFram
|
|
| 91 |
file_path = os.path.join(save_path, entry)
|
| 92 |
with open(file_path) as fp:
|
| 93 |
data = json.load(fp)
|
| 94 |
-
|
| 95 |
-
data[EvalQueueColumn.model.name] = make_clickable_model(data["model"])
|
| 96 |
-
data[EvalQueueColumn.revision.name] = data.get("revision", "main")
|
| 97 |
-
|
| 98 |
-
all_evals.append(data)
|
| 99 |
elif ".md" not in entry:
|
| 100 |
-
# this is a folder
|
| 101 |
sub_entries = [e for e in os.listdir(f"{save_path}/{entry}") if not e.startswith(".")]
|
| 102 |
for sub_entry in sub_entries:
|
| 103 |
file_path = os.path.join(save_path, entry, sub_entry)
|
| 104 |
with open(file_path) as fp:
|
| 105 |
data = json.load(fp)
|
| 106 |
-
|
| 107 |
-
data[EvalQueueColumn.model.name] = make_clickable_model(data["model"])
|
| 108 |
-
data[EvalQueueColumn.revision.name] = data.get("revision", "main")
|
| 109 |
-
all_evals.append(data)
|
| 110 |
|
| 111 |
pending_list = [e for e in all_evals if e["status"] in ["PENDING", "RERUN"]]
|
| 112 |
running_list = [e for e in all_evals if e["status"] == "RUNNING"]
|
| 113 |
finished_list = [e for e in all_evals if e["status"].startswith("FINISHED") or e["status"] == "PENDING_NEW_EVAL"]
|
| 114 |
failed_list = [e for e in all_evals if e["status"] == "FAILED"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 115 |
df_pending = pd.DataFrame.from_records(pending_list, columns=cols)
|
| 116 |
df_running = pd.DataFrame.from_records(running_list, columns=cols)
|
| 117 |
df_finished = pd.DataFrame.from_records(finished_list, columns=cols)
|
|
|
|
| 1 |
import json
|
| 2 |
import os
|
| 3 |
+
from datetime import datetime, timezone
|
| 4 |
|
| 5 |
import pandas as pd
|
| 6 |
from huggingface_hub import hf_hub_download
|
|
|
|
| 50 |
"apply_chat_template": "Apply Chat Template",
|
| 51 |
"model_type": "Type",
|
| 52 |
"model": "model_name_for_query",
|
| 53 |
+
"inference_time_seconds": "Inference Time (s)",
|
| 54 |
+
"co2_emission_kg": "CO2 (kg)",
|
| 55 |
}
|
| 56 |
# Only rename columns that exist in the dataframe
|
| 57 |
rename_dict = {k: v for k, v in rename_dict.items() if k in df.columns}
|
|
|
|
| 84 |
return df
|
| 85 |
|
| 86 |
|
| 87 |
+
def _compute_elapsed_time(time_str: str | None) -> str:
|
| 88 |
+
"""Compute human-readable elapsed time from an ISO timestamp.
|
| 89 |
+
|
| 90 |
+
Args:
|
| 91 |
+
time_str: ISO 8601 timestamp string, or None.
|
| 92 |
+
|
| 93 |
+
Returns:
|
| 94 |
+
Human-readable elapsed duration (e.g. "3h 25m"), or "-" if unavailable.
|
| 95 |
+
"""
|
| 96 |
+
if not time_str:
|
| 97 |
+
return "-"
|
| 98 |
+
try:
|
| 99 |
+
submitted = datetime.fromisoformat(time_str.replace("Z", "+00:00"))
|
| 100 |
+
delta = datetime.now(timezone.utc) - submitted
|
| 101 |
+
total_seconds = int(delta.total_seconds())
|
| 102 |
+
if total_seconds < 0:
|
| 103 |
+
return "-"
|
| 104 |
+
days, remainder = divmod(total_seconds, 86400)
|
| 105 |
+
hours, remainder = divmod(remainder, 3600)
|
| 106 |
+
minutes, _ = divmod(remainder, 60)
|
| 107 |
+
if days > 0:
|
| 108 |
+
return f"{days}d {hours}h {minutes}m"
|
| 109 |
+
if hours > 0:
|
| 110 |
+
return f"{hours}h {minutes}m"
|
| 111 |
+
return f"{minutes}m"
|
| 112 |
+
except (ValueError, TypeError):
|
| 113 |
+
return "-"
|
| 114 |
+
|
| 115 |
+
|
| 116 |
+
def _enrich_queue_entry(data: dict) -> dict:
|
| 117 |
+
"""Add timing fields to a queue entry.
|
| 118 |
+
|
| 119 |
+
Elapsed time is only meaningful for active entries:
|
| 120 |
+
- RUNNING: time since evaluation started (or submitted, as fallback)
|
| 121 |
+
- PENDING/RERUN: time since submission (waiting time)
|
| 122 |
+
- FINISHED/FAILED: shows "-" since the evaluation is already done
|
| 123 |
+
|
| 124 |
+
Args:
|
| 125 |
+
data: Raw queue entry dict loaded from JSON.
|
| 126 |
+
|
| 127 |
+
Returns:
|
| 128 |
+
The same dict with elapsed_time, submitted_time, and started_time populated.
|
| 129 |
+
"""
|
| 130 |
+
data[EvalQueueColumn.model.name] = make_clickable_model(data["model"])
|
| 131 |
+
data[EvalQueueColumn.revision.name] = data.get("revision", "main")
|
| 132 |
+
data["submitted_time"] = data.get("submitted_time", "-")
|
| 133 |
+
data["started_time"] = data.get("started_time", "-")
|
| 134 |
+
|
| 135 |
+
status = data.get("status", "")
|
| 136 |
+
if status == "RUNNING":
|
| 137 |
+
ref_time = data["started_time"] if data["started_time"] != "-" else data["submitted_time"]
|
| 138 |
+
data["elapsed_time"] = _compute_elapsed_time(ref_time)
|
| 139 |
+
elif status in ("PENDING", "RERUN"):
|
| 140 |
+
data["elapsed_time"] = _compute_elapsed_time(data["submitted_time"])
|
| 141 |
+
else:
|
| 142 |
+
data["elapsed_time"] = "-"
|
| 143 |
+
|
| 144 |
+
return data
|
| 145 |
+
|
| 146 |
+
|
| 147 |
def get_evaluation_queue_df(save_path: str, cols: list[str]) -> list[pd.DataFrame]:
|
| 148 |
+
"""Creates the different dataframes for the evaluation queues requests.
|
| 149 |
+
|
| 150 |
+
Args:
|
| 151 |
+
save_path: Path to the directory containing queue JSON files.
|
| 152 |
+
cols: List of column names to include in the output DataFrames.
|
| 153 |
+
|
| 154 |
+
Returns:
|
| 155 |
+
A list of four DataFrames: [finished, running, pending, failed].
|
| 156 |
+
"""
|
| 157 |
entries = [entry for entry in os.listdir(save_path) if not entry.startswith(".")]
|
| 158 |
all_evals = []
|
| 159 |
|
|
|
|
| 162 |
file_path = os.path.join(save_path, entry)
|
| 163 |
with open(file_path) as fp:
|
| 164 |
data = json.load(fp)
|
| 165 |
+
all_evals.append(_enrich_queue_entry(data))
|
|
|
|
|
|
|
|
|
|
|
|
|
| 166 |
elif ".md" not in entry:
|
|
|
|
| 167 |
sub_entries = [e for e in os.listdir(f"{save_path}/{entry}") if not e.startswith(".")]
|
| 168 |
for sub_entry in sub_entries:
|
| 169 |
file_path = os.path.join(save_path, entry, sub_entry)
|
| 170 |
with open(file_path) as fp:
|
| 171 |
data = json.load(fp)
|
| 172 |
+
all_evals.append(_enrich_queue_entry(data))
|
|
|
|
|
|
|
|
|
|
| 173 |
|
| 174 |
pending_list = [e for e in all_evals if e["status"] in ["PENDING", "RERUN"]]
|
| 175 |
running_list = [e for e in all_evals if e["status"] == "RUNNING"]
|
| 176 |
finished_list = [e for e in all_evals if e["status"].startswith("FINISHED") or e["status"] == "PENDING_NEW_EVAL"]
|
| 177 |
failed_list = [e for e in all_evals if e["status"] == "FAILED"]
|
| 178 |
+
|
| 179 |
+
# Add queue position to pending list (sorted by submitted_time)
|
| 180 |
+
pending_list = sorted(pending_list, key=lambda x: x.get("submitted_time", ""))
|
| 181 |
+
for i, entry in enumerate(pending_list):
|
| 182 |
+
entry["queue_position"] = i + 1
|
| 183 |
+
|
| 184 |
+
# Running models: position 0 means currently being processed
|
| 185 |
+
for entry in running_list:
|
| 186 |
+
entry["queue_position"] = 0
|
| 187 |
+
|
| 188 |
+
for entry in finished_list + failed_list:
|
| 189 |
+
entry["queue_position"] = None
|
| 190 |
+
|
| 191 |
df_pending = pd.DataFrame.from_records(pending_list, columns=cols)
|
| 192 |
df_running = pd.DataFrame.from_records(running_list, columns=cols)
|
| 193 |
df_finished = pd.DataFrame.from_records(finished_list, columns=cols)
|