e-mon commited on
Commit
0f2c31b
·
1 Parent(s): dacc1a4
Files changed (7) hide show
  1. .python-version +0 -1
  2. README.md +1 -1
  3. app.py +0 -32
  4. pyproject.toml +1 -1
  5. requirements.txt +23 -242
  6. src/display/utils.py +15 -20
  7. uv.lock +0 -0
.python-version DELETED
@@ -1 +0,0 @@
1
- 3.10.15
 
 
README.md CHANGED
@@ -29,4 +29,4 @@ tags:
29
  This repository has been cleaned to properly manage binary files with Git-LFS.
30
 
31
  For the original repository with full history, please visit:
32
- https://huggingface.co/spaces/llm-jp/open-japanese-llm-leaderboard
 
29
  This repository has been cleaned to properly manage binary files with Git-LFS.
30
 
31
  For the original repository with full history, please visit:
32
+ https://huggingface.co/spaces/llm-jp/open-japanese-llm-leaderboard-v2
app.py CHANGED
@@ -32,10 +32,8 @@ from src.display.utils import (
32
  ApplyChatTemplate,
33
  AutoEvalColumn,
34
  EnableThinking,
35
- LLMJpEvalVersion,
36
  ModelType,
37
  Precision,
38
- VllmVersion,
39
  fields,
40
  )
41
  from src.envs import API, CONTENTS_REPO, EVAL_REQUESTS_PATH, QUEUE_REPO, REPO_ID
@@ -96,8 +94,6 @@ def filter_models(
96
  size_query: list[str],
97
  precision_query: list[str],
98
  add_special_tokens_query: list[str],
99
- version_query: list[str],
100
- vllm_query: list[str],
101
  enable_thinking_query: list[str],
102
  apply_chat_template_query: list[str],
103
  ) -> pd.DataFrame:
@@ -122,12 +118,6 @@ def filter_models(
122
  # Filter by special tokens setting
123
  df = df[df["Add Special Tokens"].isin(add_special_tokens_query)]
124
 
125
- # Filter by evaluator version
126
- df = df[df["llm-jp-eval version"].isin(version_query)]
127
-
128
- # Filter by vLLM version
129
- df = df[df["vllm version"].isin(vllm_query)]
130
-
131
  # Filter by enable_thinking
132
  df = df[df["Enable Thinking"].isin(enable_thinking_query)]
133
 
@@ -179,8 +169,6 @@ def update_table(
179
  precision_query: list[str],
180
  size_query: list[str],
181
  add_special_tokens_query: list[str],
182
- version_query: list[str],
183
- vllm_query: list[str],
184
  enable_thinking_query: list[str],
185
  apply_chat_template_query: list[str],
186
  query: str,
@@ -193,8 +181,6 @@ def update_table(
193
  size_query,
194
  precision_query,
195
  add_special_tokens_query,
196
- version_query,
197
- vllm_query,
198
  enable_thinking_query,
199
  apply_chat_template_query,
200
  )
@@ -217,8 +203,6 @@ if len(leaderboard_df) > 0:
217
  list(NUMERIC_INTERVALS.keys()),
218
  [i.value.name for i in Precision],
219
  [i.value.name for i in AddSpecialTokens],
220
- [i.value.name for i in LLMJpEvalVersion],
221
- [i.value.name for i in VllmVersion],
222
  [i.value.name for i in EnableThinking],
223
  [i.value.name for i in ApplyChatTemplate],
224
  )
@@ -425,18 +409,6 @@ with gr.Blocks() as demo_leaderboard:
425
  value=[i.value.name for i in AddSpecialTokens],
426
  elem_id="filter-columns-add-special-tokens",
427
  )
428
- filter_columns_version = gr.CheckboxGroup(
429
- label="llm-jp-eval version",
430
- choices=[i.value.name for i in LLMJpEvalVersion],
431
- value=[i.value.name for i in LLMJpEvalVersion],
432
- elem_id="filter-columns-version",
433
- )
434
- filter_columns_vllm = gr.CheckboxGroup(
435
- label="vllm version",
436
- choices=[i.value.name for i in VllmVersion],
437
- value=[i.value.name for i in VllmVersion],
438
- elem_id="filter-columns-vllm",
439
- )
440
  filter_columns_enable_thinking = gr.CheckboxGroup(
441
  label="Enable Thinking",
442
  choices=[i.value.name for i in EnableThinking],
@@ -489,8 +461,6 @@ with gr.Blocks() as demo_leaderboard:
489
  filter_columns_precision.change,
490
  filter_columns_size.change,
491
  filter_columns_add_special_tokens.change,
492
- filter_columns_version.change,
493
- filter_columns_vllm.change,
494
  filter_columns_enable_thinking.change,
495
  filter_columns_apply_chat_template.change,
496
  search_bar.submit,
@@ -502,8 +472,6 @@ with gr.Blocks() as demo_leaderboard:
502
  filter_columns_precision,
503
  filter_columns_size,
504
  filter_columns_add_special_tokens,
505
- filter_columns_version,
506
- filter_columns_vllm,
507
  filter_columns_enable_thinking,
508
  filter_columns_apply_chat_template,
509
  search_bar,
 
32
  ApplyChatTemplate,
33
  AutoEvalColumn,
34
  EnableThinking,
 
35
  ModelType,
36
  Precision,
 
37
  fields,
38
  )
39
  from src.envs import API, CONTENTS_REPO, EVAL_REQUESTS_PATH, QUEUE_REPO, REPO_ID
 
94
  size_query: list[str],
95
  precision_query: list[str],
96
  add_special_tokens_query: list[str],
 
 
97
  enable_thinking_query: list[str],
98
  apply_chat_template_query: list[str],
99
  ) -> pd.DataFrame:
 
118
  # Filter by special tokens setting
119
  df = df[df["Add Special Tokens"].isin(add_special_tokens_query)]
120
 
 
 
 
 
 
 
121
  # Filter by enable_thinking
122
  df = df[df["Enable Thinking"].isin(enable_thinking_query)]
123
 
 
169
  precision_query: list[str],
170
  size_query: list[str],
171
  add_special_tokens_query: list[str],
 
 
172
  enable_thinking_query: list[str],
173
  apply_chat_template_query: list[str],
174
  query: str,
 
181
  size_query,
182
  precision_query,
183
  add_special_tokens_query,
 
 
184
  enable_thinking_query,
185
  apply_chat_template_query,
186
  )
 
203
  list(NUMERIC_INTERVALS.keys()),
204
  [i.value.name for i in Precision],
205
  [i.value.name for i in AddSpecialTokens],
 
 
206
  [i.value.name for i in EnableThinking],
207
  [i.value.name for i in ApplyChatTemplate],
208
  )
 
409
  value=[i.value.name for i in AddSpecialTokens],
410
  elem_id="filter-columns-add-special-tokens",
411
  )
 
 
 
 
 
 
 
 
 
 
 
 
412
  filter_columns_enable_thinking = gr.CheckboxGroup(
413
  label="Enable Thinking",
414
  choices=[i.value.name for i in EnableThinking],
 
461
  filter_columns_precision.change,
462
  filter_columns_size.change,
463
  filter_columns_add_special_tokens.change,
 
 
464
  filter_columns_enable_thinking.change,
465
  filter_columns_apply_chat_template.change,
466
  search_bar.submit,
 
472
  filter_columns_precision,
473
  filter_columns_size,
474
  filter_columns_add_special_tokens,
 
 
475
  filter_columns_enable_thinking,
476
  filter_columns_apply_chat_template,
477
  search_bar,
pyproject.toml CHANGED
@@ -3,7 +3,7 @@ name = "open-japanese-llm-leaderboard"
3
  version = "0.1.0"
4
  description = ""
5
  readme = "README.md"
6
- requires-python = ">=3.10"
7
  dependencies = [
8
  "apscheduler>=3.10.4",
9
  "datasets>=3.2.0",
 
3
  version = "0.1.0"
4
  description = ""
5
  readme = "README.md"
6
+ requires-python = ">=3.13, <3.14"
7
  dependencies = [
8
  "apscheduler>=3.10.4",
9
  "datasets>=3.2.0",
requirements.txt CHANGED
@@ -1,324 +1,105 @@
1
- # This file was autogenerated by uv via the following command:
2
- # uv export --no-hashes --frozen
3
  aiofiles==24.1.0
4
- # via gradio
5
  aiohappyeyeballs==2.6.1
6
- # via aiohttp
7
  aiohttp==3.13.2
8
- # via fsspec
9
  aiosignal==1.4.0
10
- # via aiohttp
11
  annotated-doc==0.0.3
12
- # via fastapi
13
  annotated-types==0.7.0
14
- # via pydantic
15
  anyio==4.11.0
16
- # via
17
- # gradio
18
- # httpx
19
- # starlette
20
  apscheduler==3.11.1
21
- # via open-japanese-llm-leaderboard
22
- async-timeout==5.0.1 ; python_full_version < '3.11'
23
- # via aiohttp
24
  attrs==25.4.0
25
- # via aiohttp
26
- audioop-lts==0.2.2 ; python_full_version >= '3.13'
27
- # via gradio
28
  authlib==1.6.5
29
- # via gradio
30
  brotli==1.2.0
31
- # via gradio
32
  certifi==2025.10.5
33
- # via
34
- # httpcore
35
- # httpx
36
- # requests
37
- cffi==2.0.0 ; platform_python_implementation != 'PyPy'
38
- # via cryptography
39
  charset-normalizer==3.4.4
40
- # via requests
41
  click==8.3.0
42
- # via
43
- # typer
44
- # uvicorn
45
- colorama==0.4.6 ; sys_platform == 'win32'
46
- # via
47
- # click
48
- # tqdm
49
  cryptography==46.0.3
50
- # via authlib
51
  datasets==4.4.1
52
- # via open-japanese-llm-leaderboard
53
  dill==0.4.0
54
- # via
55
- # datasets
56
- # multiprocess
57
- exceptiongroup==1.3.0 ; python_full_version < '3.11'
58
- # via anyio
59
  fastapi==0.121.1
60
- # via gradio
61
  ffmpy==0.6.4
62
- # via gradio
63
  filelock==3.20.0
64
- # via
65
- # datasets
66
- # huggingface-hub
67
- # torch
68
- # transformers
69
  frozenlist==1.8.0
70
- # via
71
- # aiohttp
72
- # aiosignal
73
  fsspec==2025.10.0
74
- # via
75
- # datasets
76
- # gradio-client
77
- # huggingface-hub
78
- # torch
79
  gradio==5.49.1
80
- # via open-japanese-llm-leaderboard
81
  gradio-client==1.13.3
82
- # via gradio
83
  groovy==0.1.2
84
- # via gradio
85
  h11==0.16.0
86
- # via
87
- # httpcore
88
- # uvicorn
89
  hf-transfer==0.1.9
90
- # via open-japanese-llm-leaderboard
91
- hf-xet==1.2.0 ; platform_machine == 'aarch64' or platform_machine == 'amd64' or platform_machine == 'arm64' or platform_machine == 'x86_64'
92
- # via huggingface-hub
93
  httpcore==1.0.9
94
- # via httpx
95
  httpx==0.28.1
96
- # via
97
- # datasets
98
- # gradio
99
- # gradio-client
100
- # safehttpx
101
  huggingface-hub==0.36.0
102
- # via
103
- # datasets
104
- # gradio
105
- # gradio-client
106
- # tokenizers
107
- # transformers
108
  idna==3.11
109
- # via
110
- # anyio
111
- # httpx
112
- # requests
113
- # yarl
114
  itsdangerous==2.2.0
115
- # via gradio
116
  jinja2==3.1.6
117
- # via
118
- # gradio
119
- # torch
120
  markdown-it-py==4.0.0
121
- # via rich
122
  markupsafe==3.0.3
123
- # via
124
- # gradio
125
- # jinja2
126
  mdurl==0.1.2
127
- # via markdown-it-py
128
  mpmath==1.3.0
129
- # via sympy
130
  multidict==6.7.0
131
- # via
132
- # aiohttp
133
- # yarl
134
  multiprocess==0.70.18
135
- # via datasets
136
- networkx==3.4.2 ; python_full_version < '3.11'
137
- # via torch
138
- networkx==3.5 ; python_full_version >= '3.11'
139
- # via torch
140
- numpy==2.2.6 ; python_full_version < '3.11'
141
- # via
142
- # datasets
143
- # gradio
144
- # pandas
145
- # transformers
146
- numpy==2.3.4 ; python_full_version >= '3.11'
147
- # via
148
- # datasets
149
- # gradio
150
- # pandas
151
- # transformers
152
- nvidia-cublas-cu12==12.8.4.1 ; platform_machine == 'x86_64' and sys_platform == 'linux'
153
- # via
154
- # nvidia-cudnn-cu12
155
- # nvidia-cusolver-cu12
156
- # torch
157
- nvidia-cuda-cupti-cu12==12.8.90 ; platform_machine == 'x86_64' and sys_platform == 'linux'
158
- # via torch
159
- nvidia-cuda-nvrtc-cu12==12.8.93 ; platform_machine == 'x86_64' and sys_platform == 'linux'
160
- # via torch
161
- nvidia-cuda-runtime-cu12==12.8.90 ; platform_machine == 'x86_64' and sys_platform == 'linux'
162
- # via torch
163
- nvidia-cudnn-cu12==9.10.2.21 ; platform_machine == 'x86_64' and sys_platform == 'linux'
164
- # via torch
165
- nvidia-cufft-cu12==11.3.3.83 ; platform_machine == 'x86_64' and sys_platform == 'linux'
166
- # via torch
167
- nvidia-cufile-cu12==1.13.1.3 ; platform_machine == 'x86_64' and sys_platform == 'linux'
168
- # via torch
169
- nvidia-curand-cu12==10.3.9.90 ; platform_machine == 'x86_64' and sys_platform == 'linux'
170
- # via torch
171
- nvidia-cusolver-cu12==11.7.3.90 ; platform_machine == 'x86_64' and sys_platform == 'linux'
172
- # via torch
173
- nvidia-cusparse-cu12==12.5.8.93 ; platform_machine == 'x86_64' and sys_platform == 'linux'
174
- # via
175
- # nvidia-cusolver-cu12
176
- # torch
177
- nvidia-cusparselt-cu12==0.7.1 ; platform_machine == 'x86_64' and sys_platform == 'linux'
178
- # via torch
179
- nvidia-nccl-cu12==2.27.5 ; platform_machine == 'x86_64' and sys_platform == 'linux'
180
- # via torch
181
- nvidia-nvjitlink-cu12==12.8.93 ; platform_machine == 'x86_64' and sys_platform == 'linux'
182
- # via
183
- # nvidia-cufft-cu12
184
- # nvidia-cusolver-cu12
185
- # nvidia-cusparse-cu12
186
- # torch
187
- nvidia-nvshmem-cu12==3.3.20 ; platform_machine == 'x86_64' and sys_platform == 'linux'
188
- # via torch
189
- nvidia-nvtx-cu12==12.8.90 ; platform_machine == 'x86_64' and sys_platform == 'linux'
190
- # via torch
191
  orjson==3.11.4
192
- # via gradio
193
  packaging==25.0
194
- # via
195
- # datasets
196
- # gradio
197
- # gradio-client
198
- # huggingface-hub
199
- # plotly
200
- # transformers
201
  pandas==2.3.3
202
- # via
203
- # datasets
204
- # gradio
205
  pillow==11.3.0
206
- # via gradio
207
  plotly==5.24.1
208
- # via open-japanese-llm-leaderboard
209
  propcache==0.4.1
210
- # via
211
- # aiohttp
212
- # yarl
213
  pyarrow==22.0.0
214
- # via datasets
215
- pycparser==2.23 ; implementation_name != 'PyPy' and platform_python_implementation != 'PyPy'
216
- # via cffi
217
  pydantic==2.11.10
218
- # via
219
- # fastapi
220
- # gradio
221
  pydantic-core==2.33.2
222
- # via pydantic
223
  pydub==0.25.1
224
- # via gradio
225
  pygments==2.19.2
226
- # via rich
227
  python-dateutil==2.9.0.post0
228
- # via pandas
229
  python-multipart==0.0.20
230
- # via gradio
231
  pytz==2025.2
232
- # via pandas
233
  pyyaml==6.0.3
234
- # via
235
- # datasets
236
- # gradio
237
- # huggingface-hub
238
- # transformers
239
  regex==2025.11.3
240
- # via transformers
241
  requests==2.32.5
242
- # via
243
- # datasets
244
- # huggingface-hub
245
- # transformers
246
  rich==14.2.0
247
- # via typer
248
  ruff==0.14.4
249
- # via gradio
250
  safehttpx==0.1.7
251
- # via gradio
252
  safetensors==0.6.2
253
- # via transformers
254
  semantic-version==2.10.0
255
- # via gradio
256
- setuptools==80.9.0 ; python_full_version >= '3.12'
257
- # via torch
258
  shellingham==1.5.4
259
- # via typer
260
  six==1.17.0
261
- # via python-dateutil
262
  sniffio==1.3.1
263
- # via anyio
264
  starlette==0.49.3
265
- # via
266
- # fastapi
267
- # gradio
268
  sympy==1.14.0
269
- # via torch
270
  tenacity==9.1.2
271
- # via plotly
272
  tokenizers==0.22.1
273
- # via transformers
274
  tomlkit==0.13.3
275
- # via gradio
276
  torch==2.9.0
277
- # via open-japanese-llm-leaderboard
278
  tqdm==4.67.1
279
- # via
280
- # datasets
281
- # huggingface-hub
282
- # transformers
283
  transformers==4.57.6
284
- # via open-japanese-llm-leaderboard
285
- triton==3.5.0 ; platform_machine == 'x86_64' and sys_platform == 'linux'
286
- # via torch
287
  typer==0.20.0
288
- # via gradio
289
  typing-extensions==4.15.0
290
- # via
291
- # aiosignal
292
- # anyio
293
- # cryptography
294
- # exceptiongroup
295
- # fastapi
296
- # gradio
297
- # gradio-client
298
- # huggingface-hub
299
- # multidict
300
- # pydantic
301
- # pydantic-core
302
- # starlette
303
- # torch
304
- # typer
305
- # typing-inspection
306
- # uvicorn
307
  typing-inspection==0.4.2
308
- # via pydantic
309
  tzdata==2025.2
310
- # via
311
- # pandas
312
- # tzlocal
313
  tzlocal==5.3.1
314
- # via apscheduler
315
  urllib3==2.5.0
316
- # via requests
317
  uvicorn==0.38.0
318
- # via gradio
319
  websockets==15.0.1
320
- # via gradio-client
321
  xxhash==3.6.0
322
- # via datasets
323
  yarl==1.22.0
324
- # via aiohttp
 
 
 
1
  aiofiles==24.1.0
 
2
  aiohappyeyeballs==2.6.1
 
3
  aiohttp==3.13.2
 
4
  aiosignal==1.4.0
 
5
  annotated-doc==0.0.3
 
6
  annotated-types==0.7.0
 
7
  anyio==4.11.0
 
 
 
 
8
  apscheduler==3.11.1
 
 
 
9
  attrs==25.4.0
10
+ audioop-lts==0.2.2
 
 
11
  authlib==1.6.5
 
12
  brotli==1.2.0
 
13
  certifi==2025.10.5
14
+ cffi==2.0.0
 
 
 
 
 
15
  charset-normalizer==3.4.4
 
16
  click==8.3.0
 
 
 
 
 
 
 
17
  cryptography==46.0.3
 
18
  datasets==4.4.1
 
19
  dill==0.4.0
 
 
 
 
 
20
  fastapi==0.121.1
 
21
  ffmpy==0.6.4
 
22
  filelock==3.20.0
 
 
 
 
 
23
  frozenlist==1.8.0
 
 
 
24
  fsspec==2025.10.0
 
 
 
 
 
25
  gradio==5.49.1
 
26
  gradio-client==1.13.3
 
27
  groovy==0.1.2
 
28
  h11==0.16.0
 
 
 
29
  hf-transfer==0.1.9
30
+ hf-xet==1.2.0
 
 
31
  httpcore==1.0.9
 
32
  httpx==0.28.1
 
 
 
 
 
33
  huggingface-hub==0.36.0
 
 
 
 
 
 
34
  idna==3.11
 
 
 
 
 
35
  itsdangerous==2.2.0
 
36
  jinja2==3.1.6
 
 
 
37
  markdown-it-py==4.0.0
 
38
  markupsafe==3.0.3
 
 
 
39
  mdurl==0.1.2
 
40
  mpmath==1.3.0
 
41
  multidict==6.7.0
 
 
 
42
  multiprocess==0.70.18
43
+ networkx==3.5
44
+ numpy==2.3.4
45
+ nvidia-cublas-cu12==12.8.4.1
46
+ nvidia-cuda-cupti-cu12==12.8.90
47
+ nvidia-cuda-nvrtc-cu12==12.8.93
48
+ nvidia-cuda-runtime-cu12==12.8.90
49
+ nvidia-cudnn-cu12==9.10.2.21
50
+ nvidia-cufft-cu12==11.3.3.83
51
+ nvidia-cufile-cu12==1.13.1.3
52
+ nvidia-curand-cu12==10.3.9.90
53
+ nvidia-cusolver-cu12==11.7.3.90
54
+ nvidia-cusparse-cu12==12.5.8.93
55
+ nvidia-cusparselt-cu12==0.7.1
56
+ nvidia-nccl-cu12==2.27.5
57
+ nvidia-nvjitlink-cu12==12.8.93
58
+ nvidia-nvshmem-cu12==3.3.20
59
+ nvidia-nvtx-cu12==12.8.90
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
60
  orjson==3.11.4
 
61
  packaging==25.0
 
 
 
 
 
 
 
62
  pandas==2.3.3
 
 
 
63
  pillow==11.3.0
 
64
  plotly==5.24.1
 
65
  propcache==0.4.1
 
 
 
66
  pyarrow==22.0.0
67
+ pycparser==2.23
 
 
68
  pydantic==2.11.10
 
 
 
69
  pydantic-core==2.33.2
 
70
  pydub==0.25.1
 
71
  pygments==2.19.2
 
72
  python-dateutil==2.9.0.post0
 
73
  python-multipart==0.0.20
 
74
  pytz==2025.2
 
75
  pyyaml==6.0.3
 
 
 
 
 
76
  regex==2025.11.3
 
77
  requests==2.32.5
 
 
 
 
78
  rich==14.2.0
 
79
  ruff==0.14.4
 
80
  safehttpx==0.1.7
 
81
  safetensors==0.6.2
 
82
  semantic-version==2.10.0
83
+ setuptools==80.9.0
 
 
84
  shellingham==1.5.4
 
85
  six==1.17.0
 
86
  sniffio==1.3.1
 
87
  starlette==0.49.3
 
 
 
88
  sympy==1.14.0
 
89
  tenacity==9.1.2
 
90
  tokenizers==0.22.1
 
91
  tomlkit==0.13.3
 
92
  torch==2.9.0
 
93
  tqdm==4.67.1
 
 
 
 
94
  transformers==4.57.6
95
+ triton==3.5.0
 
 
96
  typer==0.20.0
 
97
  typing-extensions==4.15.0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
98
  typing-inspection==0.4.2
 
99
  tzdata==2025.2
 
 
 
100
  tzlocal==5.3.1
 
101
  urllib3==2.5.0
 
102
  uvicorn==0.38.0
 
103
  websockets==15.0.1
 
104
  xxhash==3.6.0
 
105
  yarl==1.22.0
 
src/display/utils.py CHANGED
@@ -13,7 +13,7 @@ def fields(raw_class):
13
  # These classes are for user facing column names,
14
  # to avoid having to change them all around the code
15
  # when a modif is needed
16
- @dataclass
17
  class ColumnContent:
18
  name: str
19
  type: str
@@ -33,19 +33,17 @@ auto_eval_column_dict.append(["model", ColumnContent, ColumnContent("Model", "ma
33
  # Scores
34
  # auto_eval_column_dict.append(["average", ColumnContent, ColumnContent("Average ⬆️", "number", True)])
35
  for task in Tasks:
36
- auto_eval_column_dict.append(
37
- [
38
- task.name,
39
- ColumnContent,
40
- ColumnContent(
41
- task.value.col_name,
42
- "number",
43
- displayed_by_default=(task.value.task_type == TaskType.AVG or task.value.average),
44
- task_type=task.value.task_type,
45
- average=task.value.average,
46
- ),
47
- ]
48
- )
49
  # Model information
50
  auto_eval_column_dict.append(["model_type", ColumnContent, ColumnContent("Type", "str", False)])
51
  auto_eval_column_dict.append(["architecture", ColumnContent, ColumnContent("Architecture", "str", False)])
@@ -57,19 +55,16 @@ auto_eval_column_dict.append(["likes", ColumnContent, ColumnContent("Hub ❤️"
57
  auto_eval_column_dict.append(["revision", ColumnContent, ColumnContent("Revision", "str", False, False)])
58
  auto_eval_column_dict.append(["add_special_tokens", ColumnContent, ColumnContent("Add Special Tokens", "bool", False)])
59
  auto_eval_column_dict.append(["enable_thinking", ColumnContent, ColumnContent("Enable Thinking", "bool", False)])
60
- auto_eval_column_dict.append(
61
- ["apply_chat_template", ColumnContent, ColumnContent("Apply Chat Template", "bool", False)]
62
- )
63
  auto_eval_column_dict.append(["dummy", ColumnContent, ColumnContent("model_name_for_query", "str", False, dummy=True)])
64
  auto_eval_column_dict.append(["row_id", ColumnContent, ColumnContent("ID", "number", False, dummy=True)])
65
 
66
  # We use make dataclass to dynamically fill the scores from Tasks
67
- AutoEvalColumn = make_dataclass("AutoEvalColumn", auto_eval_column_dict, frozen=True)
68
 
69
 
70
  ## For the queue columns in the submission tab
71
- @dataclass(frozen=True)
72
- class EvalQueueColumn: # Queue column
73
  model = ColumnContent("model", "markdown", True)
74
  revision = ColumnContent("revision", "str", True)
75
  model_type = ColumnContent("model_type", "str", True)
 
13
  # These classes are for user facing column names,
14
  # to avoid having to change them all around the code
15
  # when a modif is needed
16
+ @dataclass(frozen=True)
17
  class ColumnContent:
18
  name: str
19
  type: str
 
33
  # Scores
34
  # auto_eval_column_dict.append(["average", ColumnContent, ColumnContent("Average ⬆️", "number", True)])
35
  for task in Tasks:
36
+ auto_eval_column_dict.append([
37
+ task.name,
38
+ ColumnContent,
39
+ ColumnContent(
40
+ task.value.col_name,
41
+ "number",
42
+ displayed_by_default=(task.value.task_type == TaskType.AVG or task.value.average),
43
+ task_type=task.value.task_type,
44
+ average=task.value.average,
45
+ ),
46
+ ])
 
 
47
  # Model information
48
  auto_eval_column_dict.append(["model_type", ColumnContent, ColumnContent("Type", "str", False)])
49
  auto_eval_column_dict.append(["architecture", ColumnContent, ColumnContent("Architecture", "str", False)])
 
55
  auto_eval_column_dict.append(["revision", ColumnContent, ColumnContent("Revision", "str", False, False)])
56
  auto_eval_column_dict.append(["add_special_tokens", ColumnContent, ColumnContent("Add Special Tokens", "bool", False)])
57
  auto_eval_column_dict.append(["enable_thinking", ColumnContent, ColumnContent("Enable Thinking", "bool", False)])
58
+ auto_eval_column_dict.append(["apply_chat_template", ColumnContent, ColumnContent("Apply Chat Template", "bool", False)])
 
 
59
  auto_eval_column_dict.append(["dummy", ColumnContent, ColumnContent("model_name_for_query", "str", False, dummy=True)])
60
  auto_eval_column_dict.append(["row_id", ColumnContent, ColumnContent("ID", "number", False, dummy=True)])
61
 
62
  # We use make dataclass to dynamically fill the scores from Tasks
63
+ AutoEvalColumn = make_dataclass("AutoEvalColumn", auto_eval_column_dict)
64
 
65
 
66
  ## For the queue columns in the submission tab
67
+ class EvalQueueColumn: # Queue column (not a dataclass - used as class attributes)
 
68
  model = ColumnContent("model", "markdown", True)
69
  revision = ColumnContent("revision", "str", True)
70
  model_type = ColumnContent("model_type", "str", True)
uv.lock CHANGED
The diff for this file is too large to render. See raw diff