{"kind":"task","effective_mode":"full","benchmark":{"kind":"benchmark","effective_mode":"full","slug":"assistantbench","formal_name":"AssistantBench","introduction":"AssistantBench collects realistic, time-consuming tasks that can only be answered by working through the live web, written so that each has one determinable answer. It measures whether a web agent is actually useful.","introduction_ja":"","introduction_en":"","category":"Category not supplied","task_count":null,"acquisition_status":"Acquisition status not supplied","official_url":"https://assistantbench.github.io/","indexing_mode":"noindex","profile":{"resources":[],"task_format":"","scoring":"","metric":"","size":"","answer_access":"","license":"","citation":"","maintainer":"","released":"","why_hard":"","related":[]}},"task_id":"27f859f9-ec97-5277-bbec-090fc1cba254","task_key":"default--test--3940b9ef19e15c72b9f833ad4121c789ae07bf2bdc49bc2197894cc7af6a43d2","task_revision_id":"2","upstream_id":"3940b9ef19e15c72b9f833ad4121c789ae07bf2bdc49bc2197894cc7af6a43d2","short_description":"AssistantBench test 3940b9ef19e15c72b9f833ad4121c789ae07bf2bdc49bc2197894cc7af6a43d2","config":"default","split":"test","body":"{\"task\":\"What was the official state-of-the-art F1 accuracy for HotpotQA in the full-wiki setting and NQ in the short-answer setting on July 2020? (Provide the answer as a list of jsons whose keys are \\\"dataset\\\" and \\\"F1 score\\\")\"}","display_format":"text","language":"","answer_status":"unavailable","assets":[],"source_url":"https://assistantbench.github.io/","history":"initial import","indexing_mode":"noindex","subproblems":[],"grids":[]}