[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-191384-105":53,"doc-detail-191384-en":126},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":119,"head_meta":121,"extra_data":123,"updated_unix":125},105,"en","2024-naacl-long-6","2024 NAACL Long 6","","Experiments evaluate dialogue state tracking models on MultiWOZ datasets, reporting joint goal accuracy (JGA) and F1 under different settings and model variants. Example conversations illustrate how users request taxis and restaurant bookings, specifying locations, cuisine, and price range, while the system confirms options and booking outcomes. Results across multiple model sizes and reranking/sum-based combinations compare structured state prediction quality for both dataset versions.",{"@graph":63,"@context":118},[64,80,101],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":11,"@type":70,"position":76},"https://docshare.wps.com/template/presentations/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/2024-naacl-long-6/191384/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/2024-naacl-long-6/191384.png","ImageObject",442,249,{"name":88,"@type":89},"Seraphina","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-10-02","2026-09-03",true,{"@type":98,"interactionType":99,"userInteractionCount":76},"InteractionCounter",{"@type":100},"ViewAction",{"@type":102,"mainEntity":103},"FAQPage",[104,110,114],{"name":105,"@type":106,"acceptedAnswer":107},"What do JGA and F1 measure in the experiments?","Question",{"text":108,"@type":109},"JGA measures joint goal accuracy, and F1 measures prediction quality for dialogue states. Both are reported for MultiWOZ 2.1 and MultiWOZ 2.4.","Answer",{"name":111,"@type":106,"acceptedAnswer":112},"How do the conversations typically specify user goals?",{"text":113,"@type":109},"Users specify constraints such as pickup and drop-off locations for taxis, and restaurant attributes like cuisine (e.g., Indian/Italian) and price range (e.g., expensive/moderate).",{"name":115,"@type":106,"acceptedAnswer":116},"What comparison is performed across models and dataset versions?",{"text":117,"@type":109},"The document compares multiple dialogue state tracking model variants on MultiWOZ 2.1 and MultiWOZ 2.4, using reported JGA and F1 scores to assess performance differences.","https://schema.org",{"og:url":78,"og:type":120,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":122,"canonical":78},"index,follow",{"doc_id":124,"site_id":56},191384,1788408313,{"code":4,"msg":5,"data":127},{"doc_id":124,"user_id":128,"nickname":88,"user_avatar":129,"doc_module":9,"category_id":8,"category_name":11,"doc_title":59,"doc_description":61,"doc_content":130,"file_id":131,"file_url":132,"file_type":133,"file_size":134,"view_count":135,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":30,"language":136,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":137,"faqs":138,"seo_title":139,"seo_description":61,"update_tm":125,"read_time":140},962075114101,"https://ap-avatar.wpscdn.com/avatar/e000253a75eb197efd?x-image-process=image/resize,m_fixed,w_180,h_180&k=1780044092746381165","|  | Summary: The user wants a taxi . |  |\n| --- | --- | --- |\n| | Summary : The user wants an\u003Cbr>|  |\n|  | expensive Indian restaurant . |  |\n\n|  | User : I want to book a taxi from the hotel to  |  |\n| --- | --- | --- |\n|  | User : I am looking to eat somewhere expensive. \u003Cbr>System : There are 2 Chinese, 1 Indian, and 1 Mexican .  Which of those you want?  |  |\n| \u003Cbr>User : I would like the Indian place please. |  |  |\n\n| Model | MultiWOZ 2.1 |  | MultiWOZ 2.4 |  |\n| --- | --- | --- | --- | --- |\n|  | JGA | F1 | JGA | F1 |\n| GPT-Neo 2 .7B (Black et al., 2022) |  |  |  |  |\n| IC-DST (SBERT) | 6.76±0 .87 | 42.91±2 .87 | 6.81±1 .05 | 43.42±3 .18 |\n| IC-DST (LinkBERT) | 6.39±1 .72 | 40.11±3 .30 | 6.35±1 .14 | 40.78±3 .10 |\n| SM2 | 5.44±0 .27 | 35.15±1 .80 | 5.33±0 .76 | 35.03±1 .42 |\n| GTR-T5 | 4.77±0 .66 | 28.58±0 .79 | 4.66±0 .57 | 28.50±0 .84 |\n| Jina | 5.11±0 .18 | 30.93±1 .29 | 5.16±0 .40 | 30.84±1 .33 |\n| Sum. + GTR-T5 | 6.16±0 .54 | 40.60±2 .51 | 6.01±0 .60 | 40.40±2 .34 |\n| Sum. + Jina | 6.09±0 .71 | 40.48±2 .62 | 6.13±0 .77 | 40.84±2 .95 |\n| CONVERSE | 8.07±0 .62 | 44.11±2 .45 | 7.85±0 .65 | 44.92±2 .16 |\n| LLaMA-7B (Touvron et al., 2023) |  |  |  |  |\n| IC-DST (SBERT) | 18.30±2 .81 | 69.51±3 .36 | 18.57±3 .17 | 70.37±3 .54 |\n| IC-DST (LinkBERT) | 18.09±0 .08 | 69.41±0 .65 | 18.97±0 .53 | 70.29±0 .59 |\n| SM2 | 15.23±1 .56 | 64.36±2 .36 | 15.01±1 .72 | 65.12±2 .36 |\n| GTR-T5 | 13.64±0 .16 | 57.95±0 .46 | 13.61±0 .43 | 58.26±0 .44 |\n| Jina | 15.58±0 .58 | 60.89±0 .41 | 15.50±1 .02 | 61.48±0 .34 |\n| Sum. + GTR-T5 | 17.54±0 .34 | 68.36±0 .48 | 17.74±0 .68 | 69.14±0 .77 |\n| Sum. + Jina | 17.85±0 .41 | 68.70±0 .46 | 18.37±0 .61 | 69.65±0 .87 |\n\n\n| Table 1: JGA and F1 using labeled 100 conversations with GPT-Neo-2 .7B and LLaMA-7B. |  |  |  |  |\n| --- | --- | --- | --- | --- |\n| Model | MultiWOZ 2.1 |  | MultiWOZ 2.4 |  |\n|  | JGA | F1 | JGA | F1 |\n| LLaMA-30B (Touvron et al., 2023) |  |  |  |  |\n| IC-DST (SBERT) | 25.41±1 .82 | 77.82±2 .16 | 26.01±2 .17 | 79.01±2 .52 |\n| SM2 | 22.86±1 .35 | 74.73±1 .95 | 23.46±1 .80 | 75.78±2 .41 |\n| GTR-T5 | 25.10±0 .33 | 68.42±1 .93 | 19.94±2 .40 | 68.90±2 .22 |\n| Jina | 22.51±0 .92 | 72.31±1 .01 | 22.42±1 .18 | 72.95±0 .93 |\n| Sum. + GTR-T5 | 26.06±0 .47 | 78.55±0 .35 | 26.75±0 .93 | 78.55±0 .35 |\n| Sum. + Jina | 25.10±0 .33 | 78.07±0 .54 | 25.81±1 .02 | 78.98±0 .66 |\n| CONVERSE | 27.35±0 .77 | 79.75±0 .95 | 28.23±1 .58 | 80.45±0 .55 |\n\n\n| JGA |  |  |\n| --- | --- | --- |\n| Model | MWZ-2.1 | MWZ-2.4 |\n| DS2 + BART-Large | 7.60±2 .17 | 5.86±4 .52 |\n| DS2 + T5-Large | 17.71±1 .84 | 19.08±1 .23 |\n| CONVERSE + LLaMA-7B | 19.33±0 .91 | 20.35±1 .03 |\n| CONVERSE + LLaMA-30B | 27.35±0 .77 | 28.23±1 .58 |\n\n\n| Model | MultiWOZ 2.1 |  | MultiWOZ 2.4 |  |\n| --- | --- | --- | --- | --- |\n|  | JGA | F1 | JGA | F1 |\n| LLaMA-7B (Touvron et al., 2023) |  |  |  |  |\n| IC-DST (SBERT) | 12.52±0 .68 | 62.11±0 .38 | 12.43±0 .09 | 62.45±0 .87 |\n| CONVERSE | 14.05±0 .58 | 63.37±1 .53 | 14.23±0 .48 | 64.18±1 .47 |\n\n| JGA |  |  |\n| --- | --- | --- |\n| Model | MultiWOZ 2.1 | MultiWOZ 2.4 |\n| CONVERSE + Rerank | 19.86 ± 1 .22 | 20.65 ± 1 .28 |\n| CONVERSE | 19.33 ± 0.91 | 20.35 ± 1 .03 |\n\n\n| Conversation |\n| --- |\n| USER: I need some tourist information please. I need to know about a hotel called the Arbury lodge guest house. SYSTEM: The Arbury lodge guest house is in the north area and has a moderate price range. ···\u003Cbr>USER: I would like to book a stay for 3 people for\u003Cbr>2 nights starting from Tuesday.\u003Cbr>USER: I am also looking to eat somewhere expensive, in the south area of town.\u003Cbr>.\u003Cbr>.\u003Cbr>.\u003Cbr>USER: I will also need a taxi , please.\u003Cbr>SYSTEM: Where would you like your taxi to pick you up and drop you off?\u003Cbr>USER: I want to be picked up at the hotel and dropped off at the restaurant. |\n| Summary: The user wants to book a taxi to be\u003Cbr>picked up at a specific location and dropped off at another. |\n\n| Conversation |\n| --- |\n| .\u003Cbr>.\u003Cbr>.\u003Cbr>SYSTEM: Booking was successful.\u003Cbr>The table will be reserved for 15 ","cbCaio2rbQRf6D99","https://ap.wps.com/l/cbCaio2rbQRf6D99","pdf",759399,7,"English","# Dialogue State Tracking on MultiWOZ\n## Taxi and Restaurant Booking Examples\n## Evaluation Metrics (JGA, F1)\n## Model Comparisons Across MultiWOZ 2.1 and 2.4","[{\"question\":\"What do JGA and F1 measure in the experiments?\",\"answer\":\"JGA measures joint goal accuracy, and F1 measures prediction quality for dialogue states. Both are reported for MultiWOZ 2.1 and MultiWOZ 2.4.\"},{\"question\":\"How do the conversations typically specify user goals?\",\"answer\":\"Users specify constraints such as pickup and drop-off locations for taxis, and restaurant attributes like cuisine (e.g., Indian/Italian) and price range (e.g., expensive/moderate).\"},{\"question\":\"What comparison is performed across models and dataset versions?\",\"answer\":\"The document compares multiple dialogue state tracking model variants on MultiWOZ 2.1 and MultiWOZ 2.4, using reported JGA and F1 scores to assess performance differences.\"}]","2024 NAACL Long 6 | PDF",6]