add chat template option
Browse files- app.py +16 -0
- src/display/utils.py +6 -0
- src/populate.py +5 -0
app.py
CHANGED
|
@@ -29,6 +29,7 @@ from src.display.utils import (
|
|
| 29 |
NUMERIC_INTERVALS,
|
| 30 |
TYPES,
|
| 31 |
AddSpecialTokens,
|
|
|
|
| 32 |
AutoEvalColumn,
|
| 33 |
EnableThinking,
|
| 34 |
LLMJpEvalVersion,
|
|
@@ -100,6 +101,7 @@ def filter_models(
|
|
| 100 |
version_query: list[str],
|
| 101 |
vllm_query: list[str],
|
| 102 |
enable_thinking_query: list[str],
|
|
|
|
| 103 |
) -> pd.DataFrame:
|
| 104 |
# Filter by model type
|
| 105 |
type_emoji = [t.split()[0] for t in type_query]
|
|
@@ -134,6 +136,9 @@ def filter_models(
|
|
| 134 |
# Filter by enable_thinking
|
| 135 |
df = df[df["Enable Thinking"].isin(enable_thinking_query)]
|
| 136 |
|
|
|
|
|
|
|
|
|
|
| 137 |
return df
|
| 138 |
|
| 139 |
|
|
@@ -183,6 +188,7 @@ def update_table(
|
|
| 183 |
version_query: list[str],
|
| 184 |
vllm_query: list[str],
|
| 185 |
enable_thinking_query: list[str],
|
|
|
|
| 186 |
query: str,
|
| 187 |
*columns,
|
| 188 |
) -> pd.DataFrame:
|
|
@@ -197,6 +203,7 @@ def update_table(
|
|
| 197 |
version_query,
|
| 198 |
vllm_query,
|
| 199 |
enable_thinking_query,
|
|
|
|
| 200 |
)
|
| 201 |
df = search_models_by_multiple_names(df, query)
|
| 202 |
df = select_columns(df, columns)
|
|
@@ -221,6 +228,7 @@ if len(leaderboard_df) > 0:
|
|
| 221 |
[i.value.name for i in LLMJpEvalVersion],
|
| 222 |
[i.value.name for i in VllmVersion],
|
| 223 |
[i.value.name for i in EnableThinking],
|
|
|
|
| 224 |
)
|
| 225 |
leaderboard_df = select_columns(leaderboard_df, INITIAL_COLUMNS)
|
| 226 |
else:
|
|
@@ -445,6 +453,12 @@ with gr.Blocks() as demo_leaderboard:
|
|
| 445 |
value=[i.value.name for i in EnableThinking],
|
| 446 |
elem_id="filter-columns-enable-thinking",
|
| 447 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 448 |
|
| 449 |
leaderboard_table = gr.Dataframe(
|
| 450 |
value=leaderboard_df,
|
|
@@ -487,6 +501,7 @@ with gr.Blocks() as demo_leaderboard:
|
|
| 487 |
filter_columns_version.change,
|
| 488 |
filter_columns_vllm.change,
|
| 489 |
filter_columns_enable_thinking.change,
|
|
|
|
| 490 |
search_bar.submit,
|
| 491 |
]
|
| 492 |
+ [shown_columns.change for shown_columns in shown_columns_dict.values()],
|
|
@@ -500,6 +515,7 @@ with gr.Blocks() as demo_leaderboard:
|
|
| 500 |
filter_columns_version,
|
| 501 |
filter_columns_vllm,
|
| 502 |
filter_columns_enable_thinking,
|
|
|
|
| 503 |
search_bar,
|
| 504 |
]
|
| 505 |
+ list(shown_columns_dict.values()),
|
|
|
|
| 29 |
NUMERIC_INTERVALS,
|
| 30 |
TYPES,
|
| 31 |
AddSpecialTokens,
|
| 32 |
+
ApplyChatTemplate,
|
| 33 |
AutoEvalColumn,
|
| 34 |
EnableThinking,
|
| 35 |
LLMJpEvalVersion,
|
|
|
|
| 101 |
version_query: list[str],
|
| 102 |
vllm_query: list[str],
|
| 103 |
enable_thinking_query: list[str],
|
| 104 |
+
apply_chat_template_query: list[str],
|
| 105 |
) -> pd.DataFrame:
|
| 106 |
# Filter by model type
|
| 107 |
type_emoji = [t.split()[0] for t in type_query]
|
|
|
|
| 136 |
# Filter by enable_thinking
|
| 137 |
df = df[df["Enable Thinking"].isin(enable_thinking_query)]
|
| 138 |
|
| 139 |
+
# Filter by apply_chat_template
|
| 140 |
+
df = df[df["Apply Chat Template"].isin(apply_chat_template_query)]
|
| 141 |
+
|
| 142 |
return df
|
| 143 |
|
| 144 |
|
|
|
|
| 188 |
version_query: list[str],
|
| 189 |
vllm_query: list[str],
|
| 190 |
enable_thinking_query: list[str],
|
| 191 |
+
apply_chat_template_query: list[str],
|
| 192 |
query: str,
|
| 193 |
*columns,
|
| 194 |
) -> pd.DataFrame:
|
|
|
|
| 203 |
version_query,
|
| 204 |
vllm_query,
|
| 205 |
enable_thinking_query,
|
| 206 |
+
apply_chat_template_query,
|
| 207 |
)
|
| 208 |
df = search_models_by_multiple_names(df, query)
|
| 209 |
df = select_columns(df, columns)
|
|
|
|
| 228 |
[i.value.name for i in LLMJpEvalVersion],
|
| 229 |
[i.value.name for i in VllmVersion],
|
| 230 |
[i.value.name for i in EnableThinking],
|
| 231 |
+
[i.value.name for i in ApplyChatTemplate],
|
| 232 |
)
|
| 233 |
leaderboard_df = select_columns(leaderboard_df, INITIAL_COLUMNS)
|
| 234 |
else:
|
|
|
|
| 453 |
value=[i.value.name for i in EnableThinking],
|
| 454 |
elem_id="filter-columns-enable-thinking",
|
| 455 |
)
|
| 456 |
+
filter_columns_apply_chat_template = gr.CheckboxGroup(
|
| 457 |
+
label="Apply Chat Template",
|
| 458 |
+
choices=[i.value.name for i in ApplyChatTemplate],
|
| 459 |
+
value=[i.value.name for i in ApplyChatTemplate],
|
| 460 |
+
elem_id="filter-columns-apply-chat-template",
|
| 461 |
+
)
|
| 462 |
|
| 463 |
leaderboard_table = gr.Dataframe(
|
| 464 |
value=leaderboard_df,
|
|
|
|
| 501 |
filter_columns_version.change,
|
| 502 |
filter_columns_vllm.change,
|
| 503 |
filter_columns_enable_thinking.change,
|
| 504 |
+
filter_columns_apply_chat_template.change,
|
| 505 |
search_bar.submit,
|
| 506 |
]
|
| 507 |
+ [shown_columns.change for shown_columns in shown_columns_dict.values()],
|
|
|
|
| 515 |
filter_columns_version,
|
| 516 |
filter_columns_vllm,
|
| 517 |
filter_columns_enable_thinking,
|
| 518 |
+
filter_columns_apply_chat_template,
|
| 519 |
search_bar,
|
| 520 |
]
|
| 521 |
+ list(shown_columns_dict.values()),
|
src/display/utils.py
CHANGED
|
@@ -62,6 +62,7 @@ auto_eval_column_dict.append(
|
|
| 62 |
)
|
| 63 |
auto_eval_column_dict.append(["vllm_version", ColumnContent, ColumnContent("vllm version", "str", False)])
|
| 64 |
auto_eval_column_dict.append(["enable_thinking", ColumnContent, ColumnContent("Enable Thinking", "bool", False)])
|
|
|
|
| 65 |
auto_eval_column_dict.append(["dummy", ColumnContent, ColumnContent("model_name_for_query", "str", False, dummy=True)])
|
| 66 |
auto_eval_column_dict.append(["row_id", ColumnContent, ColumnContent("ID", "number", False, dummy=True)])
|
| 67 |
|
|
@@ -170,6 +171,11 @@ class EnableThinking(Enum):
|
|
| 170 |
false = ModelDetails("False")
|
| 171 |
|
| 172 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 173 |
class NumFewShots(Enum):
|
| 174 |
shots_0 = 0
|
| 175 |
shots_4 = 4
|
|
|
|
| 62 |
)
|
| 63 |
auto_eval_column_dict.append(["vllm_version", ColumnContent, ColumnContent("vllm version", "str", False)])
|
| 64 |
auto_eval_column_dict.append(["enable_thinking", ColumnContent, ColumnContent("Enable Thinking", "bool", False)])
|
| 65 |
+
auto_eval_column_dict.append(["apply_chat_template", ColumnContent, ColumnContent("Apply Chat Template", "bool", False)])
|
| 66 |
auto_eval_column_dict.append(["dummy", ColumnContent, ColumnContent("model_name_for_query", "str", False, dummy=True)])
|
| 67 |
auto_eval_column_dict.append(["row_id", ColumnContent, ColumnContent("ID", "number", False, dummy=True)])
|
| 68 |
|
|
|
|
| 171 |
false = ModelDetails("False")
|
| 172 |
|
| 173 |
|
| 174 |
+
class ApplyChatTemplate(Enum):
|
| 175 |
+
true = ModelDetails("True")
|
| 176 |
+
false = ModelDetails("False")
|
| 177 |
+
|
| 178 |
+
|
| 179 |
class NumFewShots(Enum):
|
| 180 |
shots_0 = 0
|
| 181 |
shots_4 = 4
|
src/populate.py
CHANGED
|
@@ -42,6 +42,7 @@ def get_leaderboard_df(contents_repo: str, cols: list[str], benchmark_cols: list
|
|
| 42 |
"llm_jp_eval_version": "llm-jp-eval version",
|
| 43 |
"vllm_version": "vllm version",
|
| 44 |
"enable_thinking": "Enable Thinking",
|
|
|
|
| 45 |
"model_type": "Type",
|
| 46 |
"model": "model_name_for_query",
|
| 47 |
}
|
|
@@ -53,6 +54,10 @@ def get_leaderboard_df(contents_repo: str, cols: list[str], benchmark_cols: list
|
|
| 53 |
if "Enable Thinking" not in df.columns:
|
| 54 |
df["Enable Thinking"] = "False"
|
| 55 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
# Add a row ID column
|
| 57 |
df[AutoEvalColumn.row_id.name] = range(len(df))
|
| 58 |
|
|
|
|
| 42 |
"llm_jp_eval_version": "llm-jp-eval version",
|
| 43 |
"vllm_version": "vllm version",
|
| 44 |
"enable_thinking": "Enable Thinking",
|
| 45 |
+
"apply_chat_template": "Apply Chat Template",
|
| 46 |
"model_type": "Type",
|
| 47 |
"model": "model_name_for_query",
|
| 48 |
}
|
|
|
|
| 54 |
if "Enable Thinking" not in df.columns:
|
| 55 |
df["Enable Thinking"] = "False"
|
| 56 |
|
| 57 |
+
# Add Apply Chat Template column with default value False if it doesn't exist
|
| 58 |
+
if "Apply Chat Template" not in df.columns:
|
| 59 |
+
df["Apply Chat Template"] = "False"
|
| 60 |
+
|
| 61 |
# Add a row ID column
|
| 62 |
df[AutoEvalColumn.row_id.name] = range(len(df))
|
| 63 |
|