[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-detail-188016-en":3,"doc-seo-188016-105":30,"detail-sidebar-cat-1-en-105":92},{"code":4,"msg":5,"data":6},0,"success",{"doc_id":7,"user_id":8,"nickname":9,"user_avatar":10,"doc_module":11,"category_id":12,"category_name":13,"doc_title":14,"doc_description":15,"doc_content":16,"file_id":17,"file_url":18,"file_type":19,"file_size":20,"view_count":4,"is_deleted":4,"is_public":11,"is_downloadable":11,"audit_status":11,"page_count":21,"language":22,"language_code":23,"site_id":24,"html_lang":23,"table_of_contents":25,"faqs":26,"seo_title":27,"seo_description":15,"update_tm":28,"read_time":29},188016,1099523882182,"Alex Sinclair","https://ap-avatar.wpscdn.com/davatar_6f874abed73319feea01a86fa6f0fab8",1,158,"General","2025.acl-long.1355","The document presents comparative evaluation results for multiple agent frameworks and models across a set of web, text, embodied, and tool-use environments. It lists environment-level settings such as scenario type, task count, instance size, evaluation size, trajectory length, and number of rounds, then reports performance metrics like success rate and reward. Additional tables compare closed-sourced models, open-source agents, and methods derived from Llama2-Chat-7B, including AGENTTRAJ-SFT, AGENTTRAJ-L-SFT, and AGENTSTAR, as well as out-of-distribution settings and cross-model comparisons on tasks like WebArena, ALF, and related benchmarks.","| Frameworks | Env. | Inter. Fra. | Traj. | Exploration |\n| --- | --- | --- | --- | --- |\n| AgentBench (Liu et al., 2023a) | 8 | Eval | No | No |\n| AgentBoard (Ma et al., 2024) | 12 | Eval | No | No |\n| AgentOhana (Zhang et al., 2024) | 10 | No | Yes | No |\n| Pangu-Agent (Christianos et al., 2023) | 6 | No | Yes | Single-Env |\n| AGENTGYM (Ours) | 14 | Eval & Train | Yes | Multi-Env |\n\n| Env. | Scenario | Task Num. | Eval. Metric | Inst. Size | Eval. Size | Traj. Size | Traj-L Size | Rounds |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- |\n| WebArena (WA, Zhou et al. 2023a) | Web Navigating | 3 | Success rate | 812 | 20 | 0 | 0 | − |\n| WebShop (WS, Yao et al. 2022) | Web Navigating | 1 | Success rate | 6910 | 200 | 1000 | 3930 | 5.1 |\n| MAZE (MZ, Abdulhai et al. 2023b) | Text Game | 1 | Success rate | 240 | 25 | 100 | 215 | 4.3 |\n| Wordle (WD, Abdulhai et al. 2023b) | Text Game | 1 | Success rate | 980 | 25 | 500 | 955 | 4.3 |\n| ALFWorld (ALF, Shridhar et al. 2021) | House-holding | 6 | Success rate | 3827 | 200 | 500 | 2420 | 13.3 |\n| SciWorld (Sci, Wang et al. 2022) | Embodied Tasks | 30 | Reward | 2320 | 200 | 1000 | 2120 | 19.9 |\n| BabyAI (Baby, Chevalier-Boisvert et al. 2019) | Embodied Tasks | 40 | Reward | 900 | 90 | 400 | 810 | 5.7 |\n| TextCraft (TC, Prasad et al. 2023) | Digital Game | 1 | Success rate | 544 | 100 | 300 | 374 | 8.0 |\n| Tool-Weather (WT, Ma et al. 2024) | Tool Use | 1 | Success rate | 331 | 20 | 160 | 311 | 5.5 |\n| Tool-Movie (MV, Ma et al. 2024) | Tool Use | 1 | Success rate | 235 | 20 | 100 | 215 | 4.0 |\n| Tool-Academia (AM, Ma et al. 2024) | Tool Use | 1 | Success rate | 20 | 20 | 0 | 0 | − |\n| Tool-Sheet (ST, Ma et al. 2024) | Tool Use | 1 | Reward | 20 | 20 | 0 | 0 | − |\n| Tool-TODOList (TL, Ma et al. 2024) | Tool Use | 1 | Success rate | 155 | 20 | 70 | 135 | 5.6 |\n| BIRD (BD, Zheng et al. 2023a) | Programming | 1 | Success rate | 3200 | 200 | 2000 | 3000 | 1.0 |\n| Total | − | 89 | − | 20494 | 1160 | 6130 | 14485 | − |\n\n\n| Method | WS | ALF | TC Sci Baby MZ WD | WT | MV | TL | BD |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n|  |  |  | Closed-sourced Models & Agents |  |  |  |  |\n| DeepSeek-Chat | 11.00 | 51.00 | 23.00 16.80 45.67 4.00 24.00 | 70.00 | 70.00 | 75.00 | 13.50 |\n| Claude-3-Haiku | 5.50 | 0.00 | 0.00 0.83 1.93 4.00 16.00 | 55.00 | 50.00 | 65.00 | 13.50 |\n| Claude-3-Sonnet | 1.50 | 13.00 | 38.00 2.78 79.25 0.00 36.00 | 65.00 | 80.00 | 80.00 | 17.00 |\n| GPT-3.5-Turbo | 12.50 | 26.00 | 47.00 7.64 71.36 4.00 20.00 | 25.00 | 70.00 | 40.00 | 12.50 |\n| GPT-4-Turbo | 15.50 | 67.50 | 77.00 14.38 72.83 68.00 88.00 | 80.00 | 95.00 | 95.00 | 16.00 |\n|  |  |  | Open-source Models & Agents |  |  |  |  |\n| Llama2-Chat-7B | 0.50 | 2.00 | 0.00 0.83 0.23 0.00 0.00 | 0.00 | 0.00 | 0.00 | 1.50 |\n| Llama2-Chat-13B | 1.00 | 3.50 | 0.00 0.83 0.10 0.00 0.00 | 0.00 | 0.00 | 0.00 | 1.50 |\n| AgentLM-7B | 36.50 | 71.00 | 4.00 1.63 0.49 12.00 4.00 | 0.00 | 5.00 | 15.00 | 5.00 |\n| AgentLM-13B | 39.50 | 73.00 | 0.00 2.75 0.45 8.00 0.00 | 10.00 | 5.00 | 5.00 | 3.00 |\n| AgentLM-70B | 49.50 | 67.00 | 4.00 10.68 0.66 8.00 4.00 | 0.00 | 0.00 | 40.00 | 7.50 |\n|  |  |  | Ours (Based on Llama2-Chat-7B) |  |  |  |  |\n| AGENTTRAJ-SFT | 66.50 | 77.50 | 44.00 26.42 69.31 12.00 12.00 | 25.00 | 5.00 | 45.00 | 8.00 |\n| AGENTTRAJ-L-SFT | 73.50 | 83.00 | 60.00 74.47 74.19 12.00 36.00 | 45.00 | 5.00 | 65.00 | 8.50 |\n| AGENTSTAR | 76.50 | 88.00 | 64.00 38.00 82.70 12.00 12.00 | 25.00 | 60.00 | 70.00 | 9.00 |\n\n\n| \u003Cbr>\u003Cbr>\u003Cbr>\u003Cbr>\u003Cbr>76.5 73.5\u003Cbr>68.0\u003Cbr>18.9\u003Cbr>15.5 | | |\n| --- | --- | --- |\n\n| Method | ALF-OOD | Baby-OOD | AM | ST |\n| --- | --- | --- | --- | --- |\n| Llama2-Chat-7B | 0.0 | 2.2 | 0.0 | 0.0 |\n| AgentLM-7B | 57.7 | 4.4 | 10.0 | 14.3 |\n| AGENTTRAJ-SFT | 60.8 | 6.2 | 20.0 | 24.3 |\n| AGENTTRAJ-L-SFT | 64.9 | 6.1 | 20.0 | 25.2 |\n| AGENTSTAR | 67.5 | 6.2 | 25.0 | 26.2 |\n\n\n| Model | WS | ALF | TC | Baby | MZ | WD |\n| --- | --- | --- | --- | --- | --- | --- |\n| Qwen2.5-Max | 35.00 | 16.00 | 34.00 | 74.20 | 68.00 | ","cbCaiqSciRrHVYGi","https://ap.wps.com/l/cbCaiqSciRrHVYGi","pdf",1953398,48,"English","en",105,"# Framework Comparison\n## Environments and Metrics\n## Method-Level Results\n## Out-of-Distribution and Cross-Model Comparisons","[{\"question\":\"Which environments and scenarios are used to evaluate the agents?\",\"answer\":\"The document evaluates across WebArena/WebShop (web navigation), MAZE/Wordle (text games), ALFWorld and SciWorld/BabyAI (embodied tasks), and tool-related scenarios like Tool-Weather, Tool-Movie, and Tool-TODOList.\"},{\"question\":\"What evaluation metrics are reported for different scenarios?\",\"answer\":\"Performance is summarized using success rate for many web/text/tool tasks and reward for several environments such as ALFWorld/SciWorld and related benchmarks.\"},{\"question\":\"How do the proposed methods compare against closed-source and open-source baselines?\",\"answer\":\"Tables contrast closed-sourced agents (e.g., DeepSeek-Chat, Claude variants, GPT variants) and open-source models (e.g., Llama2-Chat) with proposed variants based on Llama2-Chat-7B, including AGENTTRAJ-SFT, AGENTTRAJ-L-SFT, and AGENTSTAR.\"}]","2025.acl-long.1355 | PDF",1788386821,17,{"code":4,"msg":31,"data":32},"ok",{"site_id":24,"language":23,"slug":33,"title":14,"keywords":34,"description":15,"schema_data":35,"social_meta":87,"head_meta":89,"extra_data":91,"updated_unix":28},"2025acl-long1355","",{"@graph":36,"@context":86},[37,54,69],{"@type":38,"itemListElement":39},"BreadcrumbList",[40,44,48,51],{"item":41,"name":42,"@type":43,"position":11},"https://docshare.wps.com","Home","ListItem",{"item":45,"name":46,"@type":43,"position":47},"https://docshare.wps.com/template/","Template",2,{"item":49,"name":13,"@type":43,"position":50},"https://docshare.wps.com/template/general/",3,{"item":52,"name":14,"@type":43,"position":53},"https://docshare.wps.com/template/2025acl-long1355/188016/",4,{"url":52,"name":14,"@type":55,"author":56,"headline":14,"publisher":58,"fileFormat":61,"inLanguage":23,"description":15,"dateModified":62,"datePublished":63,"encodingFormat":61,"isAccessibleForFree":64,"interactionStatistic":65},"DigitalDocument",{"name":9,"@type":57},"Person",{"url":41,"name":59,"@type":60},"DocShare","Organization","application/pdf","2026-09-04","2026-09-02",true,{"@type":66,"interactionType":67,"userInteractionCount":11},"InteractionCounter",{"@type":68},"ViewAction",{"@type":70,"mainEntity":71},"FAQPage",[72,78,82],{"name":73,"@type":74,"acceptedAnswer":75},"Which environments and scenarios are used to evaluate the agents?","Question",{"text":76,"@type":77},"The document evaluates across WebArena/WebShop (web navigation), MAZE/Wordle (text games), ALFWorld and SciWorld/BabyAI (embodied tasks), and tool-related scenarios like Tool-Weather, Tool-Movie, and Tool-TODOList.","Answer",{"name":79,"@type":74,"acceptedAnswer":80},"What evaluation metrics are reported for different scenarios?",{"text":81,"@type":77},"Performance is summarized using success rate for many web/text/tool tasks and reward for several environments such as ALFWorld/SciWorld and related benchmarks.",{"name":83,"@type":74,"acceptedAnswer":84},"How do the proposed methods compare against closed-source and open-source baselines?",{"text":85,"@type":77},"Tables contrast closed-sourced agents (e.g., DeepSeek-Chat, Claude variants, GPT variants) and open-source models (e.g., Llama2-Chat) with proposed variants based on Llama2-Chat-7B, including AGENTTRAJ-SFT, AGENTTRAJ-L-SFT, and AGENTSTAR.","https://schema.org",{"og:url":52,"og:type":88,"og:title":14,"og:site_name":59,"og:description":15},"article",{"robots":90,"canonical":52},"index,follow",{"doc_id":7,"site_id":24},{"code":4,"msg":5,"data":93},[94,99,104,109,114,119,123,128,133],{"id":95,"doc_module":11,"doc_module_name":46,"category_name":96,"show_sort_weight":97,"slug":98},11,"Presentations",90,"presentations",{"id":100,"doc_module":11,"doc_module_name":46,"category_name":101,"show_sort_weight":102,"slug":103},12,"Resumes",80,"resumes",{"id":105,"doc_module":11,"doc_module_name":46,"category_name":106,"show_sort_weight":107,"slug":108},14,"Invoices",70,"invoices",{"id":110,"doc_module":11,"doc_module_name":46,"category_name":111,"show_sort_weight":112,"slug":113},15,"Posters",60,"posters",{"id":115,"doc_module":11,"doc_module_name":46,"category_name":116,"show_sort_weight":117,"slug":118},16,"Social Media",50,"social-media",{"id":29,"doc_module":11,"doc_module_name":46,"category_name":120,"show_sort_weight":121,"slug":122},"Forms",40,"forms",{"id":124,"doc_module":11,"doc_module_name":46,"category_name":125,"show_sort_weight":126,"slug":127},18,"Letters",30,"letters",{"id":129,"doc_module":11,"doc_module_name":46,"category_name":130,"show_sort_weight":131,"slug":132},21,"Paper Templates",5,"papers-templates",{"id":12,"doc_module":11,"doc_module_name":46,"category_name":13,"show_sort_weight":4,"slug":134},"general-158"]