policy stringclasses 1
value | wm stringclasses 1
value | task_name stringlengths 7 32 | task_category stringlengths 8 20 | task_difficulty stringclasses 3
values | task_tags listlengths 0 5 | task_instruction stringlengths 110 212 | num_turns int64 3 34 | messages listlengths 8 120 | wm_prediction_correct listlengths 3 34 | wm_pred_token_f1 listlengths 3 34 | test_names listlengths 1 9 | test_passed listlengths 1 9 | tests_passed int64 0 5 | tests_total int64 1 9 | solved int64 0 0 | history_rewritten bool 2
classes |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
qwen3-8b | summarysrc | bn-fit-modify | scientific-computing | hard | [
"bayesian-network",
"stats"
] | "Evaluates the ability to recover a Bayesian Network DAG structure from data, perform causal interve(...TRUNCATED) | 30 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [
1,
1,
1,
1,
1,
1,
0,
0,
0,
0,
0,
1,
1,
0,
0,
0,
1,
0,
1,
0,
1,
1,
0,
1,
0,
1,
0,
1,
0,
null
] | [0.0,0.0,0.0,0.0,0.0,0.0,null,null,null,null,null,0.9,0.0,null,null,null,null,null,null,null,null,nu(...TRUNCATED) | ["test_outputs.py::test_bn_sample_exists","test_outputs.py::test_learned_dag_structure_exists","test(...TRUNCATED) | [
0,
1,
1,
0,
0,
0,
0,
0,
0
] | 2 | 9 | 0 | false |
qwen3-8b | summarysrc | cancel-async-tasks | software-engineering | hard | [
"async",
"concurrency",
"python"
] | "Evaluates the ability to implement async task concurrency control with proper cleanup on cancellati(...TRUNCATED) | 3 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [
1,
1,
null
] | [
0,
3.1,
null
] | ["test_outputs.py::test_run_py_file_exists","test_outputs.py::test_tasks_run_concurrently","test_out(...TRUNCATED) | [
1,
1,
1,
1,
1,
0
] | 5 | 6 | 0 | false |
qwen3-8b | summarysrc | circuit-fibsqrt | software-engineering | hard | [
"software-engineering"
] | "Evaluates the agent's ability to implement complex mathematical functions (Fibonacci of integer squ(...TRUNCATED) | 11 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [
1,
1,
1,
1,
1,
1,
1,
1,
1,
1,
null
] | [
5,
0,
0,
0,
0,
0,
0,
0,
0,
0,
null
] | ["test_outputs.py::test_gates_file_exists","test_outputs.py::test_gates_file_size","test_outputs.py:(...TRUNCATED) | [
1,
0,
0
] | 1 | 3 | 0 | false |
qwen3-8b | summarysrc | cobol-modernization | software-engineering | easy | [
"coding"
] | "Evaluates the ability to reverse-engineer and reimplement a COBOL program's business logic in Pytho(...TRUNCATED) | 31 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [1,1,1,1,1,1,1,1,1,0,0,1,0,0,1,1,null,null,null,null,null,null,null,null,null,null,null,null,null,nu(...TRUNCATED) | [0.0,0.0,8.0,0.0,0.0,0.0,0.0,10.0,0.0,null,null,0.0,null,null,7.4,5.4,null,null,null,null,null,null,(...TRUNCATED) | ["test_outputs.py::test_required_files_exist","test_outputs.py::test_data_files_exist","test_outputs(...TRUNCATED) | [
1,
1,
0
] | 2 | 3 | 0 | true |
qwen3-8b | summarysrc | code-from-image | software-engineering | medium | [
"ocr"
] | "Evaluates an agent's ability to extract code from an image using OCR or vision models, implement th(...TRUNCATED) | 4 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [
0,
1,
1,
null
] | [
null,
0,
0,
null
] | [
"test_outputs.py::test_output_file_exists",
"test_outputs.py::test_output_is_correct"
] | [
1,
0
] | 1 | 2 | 0 | false |
qwen3-8b | summarysrc | constraints-scheduling | personal-assistant | medium | [
"calendar",
"scheduling",
"constraint-satisfaction",
"ics-parsing",
"temporal-reasoning"
] | "Find an optimal 1-hour meeting slot for three people with complex availability constraints by parsi(...TRUNCATED) | 34 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [1,1,1,1,0,1,1,0,null,null,0,0,null,null,1,0,1,1,1,1,null,null,null,null,null,null,null,null,null,nu(...TRUNCATED) | [0.0,0.0,0.0,0.0,null,9.3,0.0,null,null,null,null,20.5,null,null,0.0,null,0.0,0.0,0.0,0.0,null,null,(...TRUNCATED) | ["test_outputs.py::test_structure_event_title_attendees_duration_window","test_outputs.py::test_cons(...TRUNCATED) | [
0,
0,
0
] | 0 | 3 | 0 | true |
qwen3-8b | summarysrc | custom-memory-heap-crash | debugging | medium | [
"cpp",
"memory-management",
"debugging"
] | "Evaluates the ability to debug and fix a C++ program that crashes in release mode due to a static i(...TRUNCATED) | 9 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [
1,
0,
1,
0,
0,
0,
0,
0,
null
] | [
0,
null,
5,
0,
0,
0,
0,
null,
null
] | ["test_outputs.py::test_protected_files_not_modified","test_outputs.py::test_program_compiles_debug"(...TRUNCATED) | [
1,
0,
0,
0,
0,
0
] | 1 | 6 | 0 | false |
qwen3-8b | summarysrc | db-wal-recovery | file-operations | medium | [
"database",
"encryption",
"recovery"
] | "Tests the ability to decrypt an XOR-encrypted SQLite WAL file and recover complete database content(...TRUNCATED) | 9 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [
1,
1,
0,
0,
1,
1,
1,
1,
null
] | [
0,
null,
null,
19.4,
6.5,
3.7,
3.4,
5.1,
null
] | ["test_outputs.py::test_recovered_json_exists","test_outputs.py::test_recovered_json_is_valid","test(...TRUNCATED) | [
1,
0,
0,
0,
0,
0,
0
] | 1 | 7 | 0 | false |
qwen3-8b | summarysrc | dna-assembly | scientific-computing | hard | [
"biology",
"cloning"
] | "Evaluates the ability to design PCR primers for Golden Gate assembly by applying molecular biology (...TRUNCATED) | 30 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [
0,
0,
null,
null,
0,
0,
1,
1,
0,
1,
0,
1,
0,
1,
1,
1,
1,
1,
0,
1,
1,
1,
0,
1,
0,
0,
0,
1,
0,
null
] | [null,null,null,null,null,null,null,null,null,null,null,null,8.5,null,null,null,null,null,null,null,(...TRUNCATED) | [
"test_outputs.py::test_primers"
] | [
0
] | 0 | 1 | 0 | false |
qwen3-8b | summarysrc | dna-insert | scientific-computing | medium | [
"biology",
"cloning"
] | "Evaluates the ability to design PCR primers for site-directed mutagenesis by analyzing plasmid sequ(...TRUNCATED) | 30 | [{"role":"user","source":"env","phase":"prompt","content":"You are an AI assistant tasked with solvi(...TRUNCATED) | [
1,
0,
1,
1,
1,
1,
1,
1,
1,
0,
1,
0,
1,
1,
1,
0,
0,
1,
1,
1,
1,
1,
0,
0,
1,
0,
1,
1,
1,
null
] | [4.3,null,5.6,5.1,0.0,5.7,7.7,4.5,0.0,8.2,0.0,5.9,0.0,0.0,0.0,null,6.5,0.0,8.0,0.0,0.0,0.0,6.6,2.3,0(...TRUNCATED) | [
"test_outputs.py::test_primers"
] | [
0
] | 0 | 1 | 0 | false |
End of preview. Expand in Data Studio
No dataset card yet
- Downloads last month
- 19