[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-327878-105":59,"doc-detail-327878-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","unsupervised-features-mining-via-activation-geometry-abstract","Unsupervised Features Mining via Activation Geometry - Abstract","","Interpretability methods seek to reveal the internal features represented in large language models (LLMs), but many approaches rely on human-labeled concepts that can encode human biases. This work proposes Mining via Activation Geometry (MAG), an unsupervised framework that prepends the same natural-language instruction Q to every input p, where Q specifies the reasoning feature of interest. The method measures how Q shifts a single residual-stream readout and studies eight MAG variants, showing that extracted reasoning features predict the model’s own world understanding and judgment, can be approximated by single activation directions, and enable activation steering to change decisions.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/unsupervised-features-mining-via-activation-geometry-abstract/327878/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/unsupervised-features-mining-via-activation-geometry-abstract/327878.png","ImageObject",300,407,{"name":92,"@type":93},"Levi","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-23","2026-09-21",true,{"@type":102,"interactionType":103,"userInteractionCount":8},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What is Mining via Activation Geometry (MAG)?","Question",{"text":112,"@type":113},"MAG is an unsupervised framework that extracts reasoning features from model activations by prepending the same natural-language instruction Q to every input p and measuring the resulting activation shift at a specific readout point.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How does MAG quantify the effect of the instruction Q?",{"text":117,"@type":113},"MAG computes an activation change at a single readout location using the difference m(Q | p) − m(p), where m(x) denotes the residual-stream readout at the last token of the final block.",{"name":119,"@type":110,"acceptedAnswer":120},"What do the extracted reasoning features enable?",{"text":121,"@type":113},"The extracted reasoning features predict the models’ own world understanding and judgment, can be approximated as linear activation directions, and these directions can be used for activation steering to change LLM decisions.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},327878,1790156123,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":8,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":144,"read_time":145},7971461740909,"https://ap-avatar.wpscdn.com/davatar_155a257f0dc6eb9ab79c44ca47cae57d","Unsupervised Features Mining via Activation Geometry  \nAmit LeVi 1 2 Elad David 1 Max Fomin 1  \narXiv :2607 .04222v 1 [ cs .AI ] 5 Jul 2026  \nAbstract  \nInterpretability methods aim to reveal the features represented inside large language models (LLMs) .  \nMany existing methods begin with labeled examples of a human-defined concept that may reflect human biases, and then identify how that concept is represented within the model, for example in its activation space or through other decomposition methods. We introduce Mining via Activation Geometry (MAG), a simple unsupervised framework for extracting reasoning features from model activations by prepending the same naturallanguage instruction Q to every input p, where Q defines the reasoning feature of interest, such as “Can this object be found in the desert?” or“Is this prompt malicious?” We measure how the instruction changes the model’s internal representation using m(Q | p) − m(p) at a single readout point. We explore eight different MAGs. The extracted reasoning features predict the models’own world understanding and judgment, can be approximated into a single activation direction, we found that some features are more linearly represented and some less, this linear representation, which is vector steering, can change the LLMs’ decisions through activation steering by injecting reasoning features. Finally, we use the same method to select the best training datasets for prompt-injection classifier probes: while similarity between ordinary activations is almost un  \nrelated to downstream performance, RFD-based similarity achieves 94 .7% Top-1 and 100% Top-2 accuracy.  \n1. Introduction  \nLLMs are increasingly used as general-purpose systems insensitive settings, raising important safety concerns (Bommasani et al., 2021 ; Weidinger et al., 2022 ; Kordonsky  \n1Zenity, Tel Aviv, Israel 2Technion—Israel Institute of Technology. Correspondence to: Amit LeVi \u003C[amitlevi@campus.technion.ac.il](amitlevi@campus.technion.ac.il) >.  \nPublished at the Workshop on Failure Modes in Agentic AI (FA GEN) at ICML 2026. Copyright 2026 by the author(s) .  \net al., 2026) . Safety alignment aims to reduce these risks by encouraging models to follow safety guidelines while remaining consistent with user intent and human preferences (Ouyang et al., 2022) . As models become more advanced, alignment and evaluation failures may become harder to detect from outputs alone, therefore, many internal methods of analysis have been developed for evaluation and finding these failure modes (David et al. ; Ben-Levi et al., 2026 ; Fomin et al., 2026) .  \nA major evaluation failure scenario can be caused by reward hacking, a model may appear aligned during ordinary interactions while relying on internal representations or strategies that are not visible in its final response. Alignment faking and evaluation awareness are related failures in which a model changes its behavior when it recognizes that it is being trained or evaluated. In alignment faking, the model selectively complies with the training objective to avoid being modified, but later refuses to comply or behaves differently at inference time (Greenblatt et al., 2024) . Evaluation awareness occurs when a model changes its behavior because it detects that it is being tested, while sabotage occurs when the model actively interferes with oversight, evaluation, or task performance in ways that are difficult to detect from its outputs (Benton et al., 2024 ; Hua et al., 2025) . In fairness evaluation, safety refusals may hide biases and create a false impression of fairness on standard benchmarks (Himelstein et al., 2026) . Subliminal learning extends this concern by showing that during knowledge distillation, a teacher model can transfer behavioral traits to a student through training data whose visible content is unrelated to those traits (Cloud et al., 2025) . Together, these examples show why safety research must directly expose and understand the internal f","cbCaio1Dr20HXt3A","https://ap.wps.com/l/cbCaio1Dr20HXt3A","pdf",473295,17,"English","# Introduction\n## Motivation: safety and evaluation failures in LLMs\n## Interpretability goals for safety, control, and understanding\n## MAG: Mining via Activation Geometry\n## Activation shifts and linear feature directions\n## Vector steering and dataset selection for probe classifiers","[{\"question\":\"What is Mining via Activation Geometry (MAG)?\",\"answer\":\"MAG is an unsupervised framework that extracts reasoning features from model activations by prepending the same natural-language instruction Q to every input p and measuring the resulting activation shift at a specific readout point.\"},{\"question\":\"How does MAG quantify the effect of the instruction Q?\",\"answer\":\"MAG computes an activation change at a single readout location using the difference m(Q | p) − m(p), where m(x) denotes the residual-stream readout at the last token of the final block.\"},{\"question\":\"What do the extracted reasoning features enable?\",\"answer\":\"The extracted reasoning features predict the models’ own world understanding and judgment, can be approximated as linear activation directions, and these directions can be used for activation steering to change LLM decisions.\"}]","Unsupervised Features Mining via Activation Geometry - Abstract | PDF",1789978249,43]