[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-138061-105":59,"doc-detail-138061-en":131},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":124,"head_meta":126,"extra_data":128,"updated_unix":130},105,"en","adversarial-paraphrasing-a-universal-attack-for-humanizing-ai-generated-text","Adversarial Paraphrasing - A Universal Attack for Humanizing AI-Generated Text","","Large Language Models (LLMs) are increasingly misused for AI-generated plagiarism and social engineering, motivating the development of AI text detectors. Many detectors remain vulnerable to evasion via paraphrasing, even though newer detectors show improved robustness. This work proposes Adversarial Paraphrasing, a training-free attack framework that universally humanizes AI text using an off-the-shelf instruction-following LLM guided by a detector to produce optimized adversarial examples. Experiments show strong transfer across detector types, sharply reducing T@1%F while only slightly degrading text quality.",{"@graph":69,"@context":123},[70,84,106],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/adversarial-paraphrasing-a-universal-attack-for-humanizing-ai-generated-text/138061/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/adversarial-paraphrasing-a-universal-attack-for-humanizing-ai-generated-text/138061.png","ImageObject",300,407,{"name":92,"@type":93},"Ophelia","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-18","2026-08-23",true,{"@type":102,"interactionType":103,"userInteractionCount":105},"InteractionCounter",{"@type":104},"ViewAction",11,{"@type":107,"mainEntity":108},"FAQPage",[109,115,119],{"name":110,"@type":111,"acceptedAnswer":112},"What problem does Adversarial Paraphrasing target?","Question",{"text":113,"@type":114},"It targets the misuse of AI-generated text and the weaknesses of AI text detectors, especially vulnerabilities to paraphrasing-based evasion.","Answer",{"name":116,"@type":111,"acceptedAnswer":117},"How does the proposed attack work?",{"text":118,"@type":114},"It uses a training-free framework that guides an instruction-following paraphrasing LLM using an AI text detector to generate adversarial examples that bypass detection more effectively.",{"name":120,"@type":111,"acceptedAnswer":121},"How effective is the attack and does it transfer across detectors?",{"text":122,"@type":114},"Experiments show broad effectiveness and high transferability across multiple detection systems, with large reductions in T@1%F under guidance from a specified detector model.","https://schema.org",{"og:url":83,"og:type":125,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":127,"canonical":83},"index,follow",{"doc_id":129,"site_id":62},138061,1787473868,{"code":4,"msg":5,"data":132},{"doc_id":129,"user_id":133,"nickname":92,"user_avatar":134,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":135,"file_id":136,"file_url":137,"file_type":138,"file_size":139,"view_count":105,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":140,"language":141,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":142,"faqs":143,"seo_title":144,"seo_description":67,"update_tm":130,"read_time":145},7971461741311,"https://ap-avatar.wpscdn.com/avatar/74000253aff267980c6?x-image-process=image/resize,m_fixed,w_180,h_180&k=1779345379180704826","arXiv :2506 .07001v2 [ cs .CL] 29 Oct 2025  \nAdversarial Paraphrasing: A Universal Attack for Humanizing AI-Generated Text  \nYize Cheng∗ Vinu Sankar Sadasivan∗ Mehrdad Saberi† Shoumik Saha† Soheil Feizi  \nUniversity of Maryland, College Park  \n{yzcheng, vinu, msaberi, smksaha, [sfeizi}@cs.umd.edu](sfeizi}@cs.umd.edu)[ ](sfeizi}@cs.umd.edu)􀂇 Project: [https://github.com/chengez/Adversarial-Paraphrasing](https://github.com/chengez/Adversarial-Paraphrasing)  \nAbstract  \nThe increasing capabilities of Large Language Models (LLMs) have raised concerns about their misuse in AI-generated plagiarism and social engineering. While various AI-generated text detectors have been proposed to mitigate these risks, many remain vulnerable to simple evasion techniques such as paraphrasing. However, recent detectors have shown greater robustness against such basic attacks. In this work, we introduce Adversarial Paraphrasing, a training-free attack framework that universally humanizes any AI-generated text to evade detection more effectively. Our approach leverages an off-the-shelf instruction-following LLM to paraphrase AI-generated content under the guidance of an AI text detector, producing adversarial examples that are specifically optimized to bypass detection.  \nExtensive experiments show that our attack is both broadly effective and highly transferable across several detection systems. For instance, compared to simple paraphrasing attack—which, ironically, increases the true positive at 1% false positive (T@1%F) by 8.57% on RADAR and 15.03% on Fast-DetectGPT—adversarial paraphrasing, guided by OpenAI-RoBERTa-Large, reduces T@1%F by 64.49% on RADAR and a striking 98.96% on Fast-DetectGPT. Across a diverse set of detectors—including neural network-based, watermark-based, and zero-shot approaches—our attack achieves an average T@1%F reduction of 87.88% under the guidance of OpenAI-RoBERTa-Large. We also analyze the tradeoff between text quality and attack success to find that our method can significantly reduce detection rates, with mostly a slight degradation in text quality. Our adversarial setup highlights the need for more robust and resilient detection strategies in the light of increasingly sophisticated evasion techniques.  \n1 Introduction  \nRecent advancements in natural language generation have given rise to transformer-based Large Language Models (LLMs) such as GPT [23], Gemini [7], and LLaMA [21], which have demonstrated remarkable capabilities across a wide range of tasks, such as email composition and code generation. These models are capable of producing fluent, coherent text that can be difficult to distinguish from that written by humans. However, despite their impressive performance, LLMs also raise significant security and ethical concerns, including risks related to plagiarism and social engineering. To counter these risks, the development of reliable AI-generated text detection tools has become a critical research problem. Several works have proposed training neural network-based classifiersto address this challenge [10, 8, 26, 30, 34, 18] . Although typically weaker than trained detectors,  \n∗Equal contribution †Equal contribution  \n39th Conference on Neural Information Processing Systems (NeurIPS 2025) .  \n\n| iteration 􀢏 |\n| --- |\n| \u003CSystem>: You are a rephraser. Given any input\u003Cbr>\u003Cbr>text, you are supposed to rephrase the text…\u003Cbr>\u003CUser>: Bielsa, 60, a former Argentine and Chile boss, resigned from French club Marseille…\u003Cbr>\u003CAssistant>: 60-year-old |\n\n| iteration 􀢏 + 􀫚 |\n| --- |\n| \u003CSystem>: You are a rephraser. Given any input\u003Cbr>\u003Cbr>text, you are supposed to rephrase the text…\u003Cbr>\u003CUser>: Bielsa, 60, a former Argentine and Chile boss, resigned from French club Marseille…\u003Cbr>\u003CAssistant>: 60-year-old man |\n\nFigure 1: An overview of our universal and training-free framework for humanizing AI text. At every auto-regressive step of adversarial paraphrasing, using the guidance from an AI text detector, we search for the toke","cbCaitQD6Xv0JeW5","https://ap.wps.com/l/cbCaitQD6Xv0JeW5","pdf",1003376,25,"English","# Abstract\n# 1 Introduction\n## Background and motivation\n## Related work and open question","[{\"question\":\"What problem does Adversarial Paraphrasing target?\",\"answer\":\"It targets the misuse of AI-generated text and the weaknesses of AI text detectors, especially vulnerabilities to paraphrasing-based evasion.\"},{\"question\":\"How does the proposed attack work?\",\"answer\":\"It uses a training-free framework that guides an instruction-following paraphrasing LLM using an AI text detector to generate adversarial examples that bypass detection more effectively.\"},{\"question\":\"How effective is the attack and does it transfer across detectors?\",\"answer\":\"Experiments show broad effectiveness and high transferability across multiple detection systems, with large reductions in T@1%F under guidance from a specified detector model.\"}]","Adversarial Paraphrasing - A Universal Attack for Humanizing AI-Generated Text | PDF",63]