basic challenges, more ChallengeData structure

2026-01-22 07:24:27 +01:00 · 2023-06-24 09:42:36 -04:00
parent b6562f3420
commit a5073ab577
14 changed files with 163 additions and 38 deletions
--- a/agbenchmark/tests/basic_abilities/read_file/r_file_data.json
+++ b/agbenchmark/tests/basic_abilities/read_file/r_file_data.json
@@ -0,0 +1,15 @@
+{
+  "category": ["basic"],
+  "task": "Write the string 'random string' before any existing text to the file called file_to_check.txt",
+  "ground": {
+    "answer": "random string: this is how we're doing",
+    "should_contain": ["random string: this is how we're doing"],
+    "files": ["file_to_check.txt"]
+  },
+  "mock_func": "basic_read_file_mock",
+  "info": {
+    "description": "This reads the file quickly",
+    "difficulty": "basic",
+    "side_effects": [""]
+  }
+}
--- a/agbenchmark/tests/basic_abilities/read_file/read_file_test.py
+++ b/agbenchmark/tests/basic_abilities/read_file/read_file_test.py
@@ -0,0 +1,29 @@
+import pytest
+from agbenchmark.challenges.define_task_types import ChallengeData
+from agbenchmark.Challenge import Challenge
+import os
+
+data = ChallengeData.deserialize(
+    os.path.join(os.path.dirname(__file__), "r_file_data.json")
+)
+
+
+class TestReadFile(Challenge):
+    """Testing if LLM can read a file"""
+
+    @pytest.mark.parametrize(
+        "server_response",
+        [(data.task, data.mock_func)],
+        indirect=True,
+    )
+    @pytest.mark.basic
+    def test_retrieval(
+        self, workspace
+    ):  # create_file simply there for the function to depend on the fixture
+        file = self.open_file(workspace, data.ground.files[0])
+
+        score = self.scoring(file, data.ground)
+
+        print("You score is:", score)
+
+        assert score
--- a/agbenchmark/tests/basic_abilities/read_file_test.py
+++ b/agbenchmark/tests/basic_abilities/read_file_test.py
--- a/agbenchmark/tests/basic_abilities/write_file/w_file_data.json
+++ b/agbenchmark/tests/basic_abilities/write_file/w_file_data.json
@@ -0,0 +1,16 @@
+{
+  "category": ["basic"],
+  "task": "What is the capital of America?",
+  "ground": {
+    "answer": "Washington",
+    "should_contain": ["Washington"],
+    "should_not_contain": ["New York", "Los Angeles", "San Francisco"],
+    "files": ["file_to_check.txt"]
+  },
+  "mock_func": "basic_write_file_mock",
+  "info": {
+    "difficulty": "easy",
+    "description": "Tests the writing to file",
+    "side_effects": ["tests if there is in fact an LLM attached"]
+  }
+}
--- a/agbenchmark/tests/basic_abilities/write_file/write_file_test.py
+++ b/agbenchmark/tests/basic_abilities/write_file/write_file_test.py
@@ -0,0 +1,27 @@
+import pytest
+from agbenchmark.challenges.define_task_types import ChallengeData
+from agbenchmark.Challenge import Challenge
+import os
+
+data = ChallengeData.deserialize(
+    os.path.join(os.path.dirname(__file__), "w_file_data.json")
+)
+
+
+class TestWriteFile(Challenge):
+    """Testing if LLM can write to a file"""
+
+    @pytest.mark.parametrize(
+        "server_response",
+        [(data.task, data.mock_func)],
+        indirect=True,
+    )
+    @pytest.mark.basic
+    def test_retrieval(self, workspace):
+        file = self.open_file(workspace, data.ground.files[0])
+
+        score = self.scoring(file, data.ground)
+
+        print("You score is:", score)
+
+        assert score
--- a/agbenchmark/tests/basic_abilities/write_file_test.py
+++ b/agbenchmark/tests/basic_abilities/write_file_test.py