eval-leaderboard

Running

App Files Files Community

xeon27 commited on Jan 27

Commit

9c55d6d

1 Parent(s): 84a3b7a

Add model name links and change single-turn to base

Browse files

Files changed (5) hide show

app.py +2 -2
refactor_eval_results.py +1 -0
src/about.py +18 -18
src/display/formatting.py +2 -3
src/populate.py +2 -2

app.py CHANGED Viewed

@@ -78,8 +78,8 @@ with demo:
     gr.Markdown(INTRODUCTION_TEXT, elem_classes="markdown-text")
     with gr.Tabs(elem_classes="tab-buttons") as tabs:
-        with gr.TabItem("Single-turn Benchmark", elem_id="llm-benchmark-tab-table", id=0):
-            leaderboard = init_leaderboard(ST_LEADERBOARD_DF, "single-turn")
         with gr.TabItem("Agentic Benchmark", elem_id="llm-benchmark-tab-table", id=1):
             leaderboard = init_leaderboard(AGENTIC_LEADERBOARD_DF, "agentic")

     gr.Markdown(INTRODUCTION_TEXT, elem_classes="markdown-text")
     with gr.Tabs(elem_classes="tab-buttons") as tabs:
+        with gr.TabItem("Base Benchmark", elem_id="llm-benchmark-tab-table", id=0):
+            leaderboard = init_leaderboard(ST_LEADERBOARD_DF, "base")
         with gr.TabItem("Agentic Benchmark", elem_id="llm-benchmark-tab-table", id=1):
             leaderboard = init_leaderboard(AGENTIC_LEADERBOARD_DF, "agentic")

refactor_eval_results.py CHANGED Viewed

@@ -106,6 +106,7 @@ def main():
         # Create dummy requests file
         requests = {
             "model": model_name,
             "base_model": "",
             "revision": "main",
             "private": False,

         # Create dummy requests file
         requests = {
             "model": model_name,
+            "model_sha": MODEL_SHA_MAP[model_name],
             "base_model": "",
             "revision": "main",
             "private": False,

src/about.py CHANGED Viewed

@@ -15,21 +15,21 @@ class Task:
 class Tasks(Enum):
     # task_key in the json file, metric_key in the json file, name to display in the leaderboard
-    # single-turn
-    task0 = Task("arc_easy", "accuracy", "ARC-Easy", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/arc")
-    task1 = Task("arc_challenge", "accuracy", "ARC-Challenge", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/arc")
-    task2 = Task("drop", "mean", "DROP", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/drop")
-    task3 = Task("winogrande", "accuracy", "WinoGrande", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/winogrande")
-    task4 = Task("gsm8k", "accuracy", "GSM8K", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/gsm8k")
-    task5 = Task("hellaswag", "accuracy", "HellaSwag", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/hellaswag")
-    task6 = Task("humaneval", "mean", "HumanEval", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/humaneval")
-    task7 = Task("ifeval", "final_acc", "IFEval", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/ifeval")
-    task8 = Task("math", "accuracy", "MATH", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mathematics")
-    task9 = Task("mmlu", "accuracy", "MMLU", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mmlu")
-    task10 = Task("mmlu_pro", "accuracy", "MMLU-Pro", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mmlu_pro")
-    task11 = Task("gpqa_diamond", "accuracy", "GPQA-Diamond", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/gpqa")
-    task12 = Task("mmmu_multiple_choice", "accuracy", "MMMU-Multiple-Choice", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mmmu")
-    task13 = Task("mmmu_open", "accuracy", "MMMU-Open-Ended", "single-turn", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mmmu")
     # agentic
     task14 = Task("gaia", "mean", "GAIA", "agentic", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/gaia")
@@ -44,19 +44,19 @@ NUM_FEWSHOT = 0 # Change with your few shot
 # Your leaderboard name
 TITLE = """<h1 align="center" id="space-title">Vector State of Evaluation Leaderboard</h1>"""
-SINGLE_TURN_TASK_NAMES = ", ".join([f"[{task.value.col_name}]({task.value.source})" for task in Tasks if task.value.type == "single-turn"])
 AGENTIC_TASK_NAMES = ", ".join([f"[{task.value.col_name}]({task.value.source})" for task in Tasks if task.value.type == "agentic"])
 # What does your leaderboard evaluate?
 INTRODUCTION_TEXT = f"""
-This leaderboard presents the performance of selected LLM models on a set of tasks. The tasks are divided into two categories: single-turn and agentic. The single-turn tasks are: {SINGLE_TURN_TASK_NAMES}. The agentic tasks are: {AGENTIC_TASK_NAMES}."""
 # Which evaluations are you running? how can people reproduce what you have?
 LLM_BENCHMARKS_TEXT = f"""
 ## How it works
 The following benchmarks are included:
-Single-turn: {SINGLE_TURN_TASK_NAMES}
 Agentic: {AGENTIC_TASK_NAMES}

 class Tasks(Enum):
     # task_key in the json file, metric_key in the json file, name to display in the leaderboard
+    # base
+    task0 = Task("arc_easy", "accuracy", "ARC-Easy", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/arc")
+    task1 = Task("arc_challenge", "accuracy", "ARC-Challenge", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/arc")
+    task2 = Task("drop", "mean", "DROP", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/drop")
+    task3 = Task("winogrande", "accuracy", "WinoGrande", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/winogrande")
+    task4 = Task("gsm8k", "accuracy", "GSM8K", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/gsm8k")
+    task5 = Task("hellaswag", "accuracy", "HellaSwag", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/hellaswag")
+    task6 = Task("humaneval", "mean", "HumanEval", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/humaneval")
+    task7 = Task("ifeval", "final_acc", "IFEval", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/ifeval")
+    task8 = Task("math", "accuracy", "MATH", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mathematics")
+    task9 = Task("mmlu", "accuracy", "MMLU", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mmlu")
+    task10 = Task("mmlu_pro", "accuracy", "MMLU-Pro", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mmlu_pro")
+    task11 = Task("gpqa_diamond", "accuracy", "GPQA-Diamond", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/gpqa")
+    task12 = Task("mmmu_multiple_choice", "accuracy", "MMMU-Multiple-Choice", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mmmu")
+    task13 = Task("mmmu_open", "accuracy", "MMMU-Open-Ended", "base", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/mmmu")
     # agentic
     task14 = Task("gaia", "mean", "GAIA", "agentic", "https://github.com/UKGovernmentBEIS/inspect_evals/tree/main/src/inspect_evals/gaia")
 # Your leaderboard name
 TITLE = """<h1 align="center" id="space-title">Vector State of Evaluation Leaderboard</h1>"""
+SINGLE_TURN_TASK_NAMES = ", ".join([f"[{task.value.col_name}]({task.value.source})" for task in Tasks if task.value.type == "base"])
 AGENTIC_TASK_NAMES = ", ".join([f"[{task.value.col_name}]({task.value.source})" for task in Tasks if task.value.type == "agentic"])
 # What does your leaderboard evaluate?
 INTRODUCTION_TEXT = f"""
+This leaderboard presents the performance of selected LLM models on a set of tasks. The tasks are divided into two categories: base and agentic. The base tasks are: {SINGLE_TURN_TASK_NAMES}. The agentic tasks are: {AGENTIC_TASK_NAMES}."""
 # Which evaluations are you running? how can people reproduce what you have?
 LLM_BENCHMARKS_TEXT = f"""
 ## How it works
 The following benchmarks are included:
+Base: {SINGLE_TURN_TASK_NAMES}
 Agentic: {AGENTIC_TASK_NAMES}

src/display/formatting.py CHANGED Viewed

@@ -2,9 +2,8 @@ def model_hyperlink(link, model_name):
     return f'<a target="_blank" href="{link}" style="color: var(--link-text-color); text-decoration: underline;text-decoration-style: dotted;">{model_name}</a>'
-def make_clickable_model(model_name):
-    link = f"https://huggingface.co/{model_name}"
-    return model_hyperlink(link, model_name)
 def styled_error(error):

     return f'<a target="_blank" href="{link}" style="color: var(--link-text-color); text-decoration: underline;text-decoration-style: dotted;">{model_name}</a>'
+def make_clickable_model(model_name, model_sha):
+    return model_hyperlink(model_sha, model_name)
 def styled_error(error):

src/populate.py CHANGED Viewed

@@ -66,7 +66,7 @@ def get_evaluation_queue_df(save_path: str, cols: list) -> list[pd.DataFrame]:
             with open(file_path) as fp:
                 data = json.load(fp)
-            data[EvalQueueColumn.model.name] = make_clickable_model(data["model"])
             data[EvalQueueColumn.revision.name] = data.get("revision", "main")
             all_evals.append(data)
@@ -78,7 +78,7 @@ def get_evaluation_queue_df(save_path: str, cols: list) -> list[pd.DataFrame]:
                 with open(file_path) as fp:
                     data = json.load(fp)
-                data[EvalQueueColumn.model.name] = make_clickable_model(data["model"])
                 data[EvalQueueColumn.revision.name] = data.get("revision", "main")
                 all_evals.append(data)

             with open(file_path) as fp:
                 data = json.load(fp)
+            data[EvalQueueColumn.model.name] = make_clickable_model(data["model_name"], data["model_sha"])
             data[EvalQueueColumn.revision.name] = data.get("revision", "main")
             all_evals.append(data)
                 with open(file_path) as fp:
                     data = json.load(fp)
+                data[EvalQueueColumn.model.name] = make_clickable_model(data["model_name"], data["model_sha"])
                 data[EvalQueueColumn.revision.name] = data.get("revision", "main")
                 all_evals.append(data)