{"kind":"task","effective_mode":"full","benchmark":{"kind":"benchmark","effective_mode":"full","slug":"assistantbench","formal_name":"AssistantBench","introduction":"AssistantBench collects realistic, time-consuming tasks that can only be answered by working through the live web, written so that each has one determinable answer. It measures whether a web agent is actually useful.","introduction_ja":"","introduction_en":"","category":"Category not supplied","task_count":null,"acquisition_status":"Acquisition status not supplied","official_url":"https://assistantbench.github.io/","indexing_mode":"noindex","profile":{"resources":[],"task_format":"","scoring":"","metric":"","size":"","answer_access":"","license":"","citation":"","maintainer":"","released":"","why_hard":"","related":[]}},"task_id":"57f943b6-a89f-5cf2-9992-169926e435ab","task_key":"default--test--f94fa0a81aedde2ba39236a7f64988dbfa92a41a19f26f0caca81f55404de8ce","task_revision_id":"2","upstream_id":"f94fa0a81aedde2ba39236a7f64988dbfa92a41a19f26f0caca81f55404de8ce","short_description":"AssistantBench test f94fa0a81aedde2ba39236a7f64988dbfa92a41a19f26f0caca81f55404de8ce","config":"default","split":"test","body":"{\"task\":\"Which papers published in 2023 have used RNNs to process extremely long texts (tens of thousands of tokens), and evaluated on long-text benchmarks instead of just evaluating perplexity?\"}","display_format":"text","language":"","answer_status":"unavailable","assets":[],"source_url":"https://assistantbench.github.io/","history":"initial import","indexing_mode":"noindex","subproblems":[],"grids":[]}