[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-138792-105":59,"doc-detail-138792-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","anchorcot-anchors-pave-the-way-for-multi-hop-reasoning","AnchorCoT - Anchors Pave the Way for Multi-hop Reasoning","","Large Language Models (LLMs) have advanced rapidly across many natural language tasks, yet multi-hop question answering remains difficult because generated reasoning chains can be unreliable and prone to hallucinations. AnchorCoT is introduced to improve multi-hop reasoning by first predicting key entities as “anchors” and then using a ranking algorithm to enforce a coherent logical sequence. Experiments on HotpotQA, 2WikiMultiHopQA, and MuSiQue-Ans show that AnchorCoT outperforms existing methods and yields more accurate reasoning results.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/anchorcot-anchors-pave-the-way-for-multi-hop-reasoning/138792/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/anchorcot-anchors-pave-the-way-for-multi-hop-reasoning/138792.png","ImageObject",300,407,{"name":92,"@type":93},"Finn","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-18","2026-08-23",true,{"@type":102,"interactionType":103,"userInteractionCount":39},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What is the core idea behind AnchorCoT?","Question",{"text":112,"@type":113},"AnchorCoT first predicts key entities as anchors to guide reasoning, then applies a ranking algorithm to order anchors logically, and finally uses anchors to generate step-wise Chain-of-Thought reasoning.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"Why do current LLMs struggle with multi-hop question answering?",{"text":117,"@type":113},"They often generate hallucinations during answer generation and may produce incorrect reasoning steps; errors can propagate and cascade across subsequent steps in multi-hop settings.",{"name":119,"@type":110,"acceptedAnswer":120},"Which datasets and models are used to evaluate AnchorCoT?",{"text":121,"@type":113},"AnchorCoT is evaluated on HotpotQA, 2WikiMultiHopQA, and MuSiQue-Ans, using Qwen2.5-7B/14B and GPT-4o for experiments.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},138792,1787488350,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":39,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},34359740700684,"https://ap-avatar.wpscdn.com/avatar/1f400023980c374ae676?_k=1777273430885731487","AnchorCoT: Anchors Pave the Way for Multi-hop Reasoning Tianshi Ming1 , Xian Wu2 * , Yingying Zhang2 , Zichuan Fu3 , Dawei Cheng1  \n1Tongji University, 2Tencent Jarvis Lab, 3 City University of Hong Kong  \n[2151569@tongji.edu.cn](2151569@tongji.edu.cn), [kevinxwu@tencent.com](kevinxwu@tencent.com), [dcheng@tongji.edu.cn](dcheng@tongji.edu.cn),  \nAbstract  \nLarge Language Models (LLMs) have made substantial strides in a broad array of natural language tasks. Recently, LLMs have demonstrated potential reasoning capabilities through prompt design, such as the Chain of Thought (CoT) . Despite their superiority in question answering, LLMs still face challenges in answering questions that require multi-hop reasoning, often generating unreliable reasoning chains during answer generation. To improve LLMs’performance in multi-hop reasoning, we introduce a novel reasoning approach, AnchorCoT, designed to assist LLMs in answering questions involving complex logical reasoning steps. AnchorCoT first predicts key entities which work as important “anchors” to guide the reasoning process and then employs a novel ranking algorithm to ensure the logical sequence of the predicted answers. We implement AnchorCoTon Qwen2.5-7B/14B and GPT-4o and evaluate our method on widely used multi-hop reasoning datasets, including HotpotQA, 2WikiMultiHopQA, and MuSiQue-Ans. The experimental results show that AnchorCoT outperforms existing methods in multi-hop question reasoning and provides more accurate reasoning results in multi-hop question answering tasks.  \n1 Introduction  \nLarge language models (LLMs) have demonstrated impressive in-context learning abilities across a range of natural language processing (NLP) tasks (Grattafiori et al., 2024 ; OpenAI, 2024 ; Hoffmann et al., 2022 ; Chowdhery et al., 2022), such as information retrieval (Schlatt et al., 2024 ; Guo et al., 2024), relation extraction (Zaratiana et al., 2024 ; Efeoglu and Paschke, 2024) and question answering (Sohn et al., 2024 ; Zhao et al., 2024) . Recently, the reasoning ability of LLMs (Giadikiaroglou et al., 2024) has garnered growing attention due  \n* Corresponding author.  \nto its critical role in solving problems with complex logic.  \nFigure 1: Human, CoT and AnchorCoT approaches in solving multi-hop question (using 3-hop question as example) . Human first split multi-hop questions into sub-questions and get answers of sub-questions by order, then summarize the answer. Chain of Thought(CoT) solve multi-hop question step by step, but get wrong answer without guidance. Our approach first predict intermediate answers as anchors, then rerank the anchorsin logical order. Finally, AnchorCoT uses anchors as guidance in CoT generation. The example use Qwen2.5- 7B-Instruct model.  \nDespite the significant success of standard large language models (LLMs) in tackling questionanswering tasks as demonstrated in various stud-  \n15522  \nFindings of the Association for Computational Linguistics: ACL 2025 , pages 15522–15536 July 27-August 1, 2025 ©2025 Association for Computational Linguistics  \nies (Brown et al., 2020 ; Liu et al., 2023 ; Bao et al., 2023 ; Creswell et al., 2022), their performance tends to falter on complex reasoning tasks that necessitate multiple logical reasoning steps (Wei et al., 2023) . LLMs have a tendency to generate hallucinations during the answer generation process in multi-hop reasoning tasks, often yielding responses that seem plausible but lack factual substantiation (Huang et al., 2025) .  \nCurrent LLMs in multi-hop reasoning task generally follows Chain-of-Thought strategy. Common strategies include instruction prompting (Wei et al., 2023 ; Wang et al., 2023a ; Xu et al., 2024), reasoning path searching (Yao et al., 2023 ; Besta et al., 2024 ; Chu et al., 2024), and majority voting (Wang et al., 2023b ; Chen et al., 2023) . However, the generated reasoning chains often still exhibit hallucinations due to insufficient or weak supervision during the generation proces","cbCaif6B0BYDMlgj","https://ap.wps.com/l/cbCaif6B0BYDMlgj","pdf",777291,15,"English","# Introduction\n## Multi-hop reasoning challenges\n## Existing Chain-of-Thought related strategies\n## Proposed approach: AnchorCoT","[{\"question\":\"What is the core idea behind AnchorCoT?\",\"answer\":\"AnchorCoT first predicts key entities as anchors to guide reasoning, then applies a ranking algorithm to order anchors logically, and finally uses anchors to generate step-wise Chain-of-Thought reasoning.\"},{\"question\":\"Why do current LLMs struggle with multi-hop question answering?\",\"answer\":\"They often generate hallucinations during answer generation and may produce incorrect reasoning steps; errors can propagate and cascade across subsequent steps in multi-hop settings.\"},{\"question\":\"Which datasets and models are used to evaluate AnchorCoT?\",\"answer\":\"AnchorCoT is evaluated on HotpotQA, 2WikiMultiHopQA, and MuSiQue-Ans, using Qwen2.5-7B/14B and GPT-4o for experiments.\"}]","AnchorCoT - Anchors Pave the Way for Multi-hop Reasoning | PDF",38]