[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-194475-105":53,"doc-detail-194475-en":126},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":119,"head_meta":121,"extra_data":123,"updated_unix":125},105,"en","2025-issta-decoma","2025-ISSTA-DeCoMa","","DeCoMa focuses on dual-channel abstraction for code datasets by mapping concrete program elements into abstract identifiers, abstract expressions, and abstract comments. The method segments identifiers using camelCase and snake_case rules, replaces subexpressions with a normalized token, and captures structural properties from the syntax tree. For numbers and strings, it generalizes values (e.g., num, str) and reconstructs abstract representations by parsing code and pairing code with comments. The algorithm ultimately outputs the AI, AE, and AC sets for downstream analysis.",{"@graph":63,"@context":118},[64,80,101],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/2025-issta-decoma/194475/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/2025-issta-decoma/194475.png","ImageObject",442,249,{"name":88,"@type":89},"Ophelia","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-10-02","2026-09-03",true,{"@type":98,"interactionType":99,"userInteractionCount":47},"InteractionCounter",{"@type":100},"ViewAction",{"@type":102,"mainEntity":103},"FAQPage",[104,110,114],{"name":105,"@type":106,"acceptedAnswer":107},"What does DeCoMa abstract from code datasets?","Question",{"text":108,"@type":109},"DeCoMa abstracts identifiers, expressions, and comments into normalized forms (AI, AE, and AC) using a structured mapping algorithm.","Answer",{"name":111,"@type":106,"acceptedAnswer":112},"How are identifier names generalized in the mapping?",{"text":113,"@type":109},"Identifiers are segmented according to camelCase and snake_case conventions, then aggregated into the abstract identifier set.",{"name":115,"@type":106,"acceptedAnswer":116},"How does DeCoMa handle different literal types like numbers and strings?",{"text":117,"@type":109},"Numbers and strings are replaced with generalized tokens (e.g., num, str). The approach then builds corresponding abstract identifiers or expressions during tree parsing and mapping.","https://schema.org",{"og:url":78,"og:type":120,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":122,"canonical":78},"index,follow",{"doc_id":124,"site_id":56},194475,1789981540,{"code":4,"msg":5,"data":127},{"doc_id":124,"user_id":128,"nickname":88,"user_avatar":129,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":130,"file_id":131,"file_url":132,"file_type":133,"file_size":134,"view_count":79,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":135,"language":136,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":60,"faqs":137,"seo_title":138,"seo_description":61,"update_tm":139,"read_time":140},7971461741311,"https://ap-avatar.wpscdn.com/avatar/74000253aff267980c6?x-image-process=image/resize,m_fixed,w_180,h_180&k=1779345379180704826","| Language | Tree Node Type | Generalization |\n| --- | --- | --- |\n| Java | decimal_integer_literal, decimal_floating_point_literal |   num  |\n|  | character_literal, string_literal |   str  |\n|  | variable_declarator.identifier, formal_parameter.identifier, enhanced_for_statement.identifier |   identifier  |\n|  | binary_expression, assignment_expression, method_invocation, local_variable_declaration, literal, return_statement, object_creation_expression, field_access, array_creation_expression |   subexpression  |\n| Python | integer, float |   num  |\n|  | string |   str  |\n|  | assignment.identifier, argument_list.identifier |   identifier  |\n|  | binary_expression, assignment_expression, call, literal, expression_statement, return_statement, attribute, keyword_argument |   subexpression  |\n\n| Abstract Identifiers:\u003Cbr>1. check\u003Cbr>2. list\u003Cbr>[3.](3. is)[ is](3. is)\u003Cbr>4. Empty |\n| --- |\n| Abstract Expressions:\u003Cbr>1. public static void   identifier  ( ArrayList\u003C?>   identifier ) {\u003Cbr>2. boolean   identifier  =   subexpression ;\u003Cbr>3.   subexpression .isEmpty();\u003Cbr>4. list\u003Cbr>4.   subexpression .println ( str );\u003Cbr>5. System.out\u003Cbr>6.   subexpression .println ( str );\u003Cbr>7. System.out |\n\n\n| Algorithm 1 Dual-Channel Abstraction Mapping |  |  |\n| --- | --- | --- |\n| Input: 􀀙􀀻 | code dataset programming | language |\n|  Output: AI , AE , AC abstract identifiers, abstract expressions, abstract comments \u003Cbr>1: function AbstractIdentifier(􀁁 ) 25: for each child 􀀽 􀀲 in 􀀽 do\u003Cbr>2: 􀀞 ← ∅ 26: if 􀀽 􀀲.type is expression then\u003Cbr>3: for each node 􀀽 in 􀁁 do 27: 􀀽.child.text ← “__subexpression__”\u003Cbr>4: if 􀀽.type is identifier then 28: end if\u003Cbr>5: 􀀸 ← segment identifier 􀀽.text based on camelCase 29: end for\u003Cbr>and snake_case conventions 30: 􀀚 ← 􀀚 ∪ 􀀽 .text\u003Cbr>6: 􀀞 ← 􀀞 ∪ 􀀸 31: end if\u003Cbr>7: end if 32: end for\u003Cbr>8: end for 33: return 􀀚\u003Cbr>9: return 􀀞 34: end function\u003Cbr>10: end function 35:\u003Cbr>11: 36: function AbstractComment(􀁂)\u003Cbr>12: function AbstractExpression(􀁁 ) 37: 􀀨 ← segment comment 􀁂 based on whitespace\u003Cbr>13: 􀀚 ← ∅ 38: return 􀀨\u003Cbr>14: for each node 􀀽 in 􀁁 do 39: end function\u003Cbr>15: if 􀀽.type is identifier then 40:\u003Cbr>16: 􀁁 . 􀀽 .text ← “  identifier ” 41: AI , AE , AC ← ∅, ∅, ∅\u003Cbr>17: else if 􀀽.type is number then 42: for each code-comment pair (􀀲, 􀁂 ) in 􀀙 do\u003Cbr>18: 􀁁 . 􀀽.text ← “  num ” 43: 􀁁 ← tree_sitter.parser(􀀻 , 􀀲) ▷ parse code 􀀲 using Tree-sitter\u003Cbr>19: else if 􀀽.type is string then 44: AI ← AI∪ AbstractIdentifier(􀁁 )\u003Cbr>20: 􀁁 . 􀀽 .text ← “  str ” 45: AE ← AE ∪ AbstractExpression(􀁁 )\u003Cbr>21: end if 46: AC ← AC∪ AbstractComment(􀁂)\u003Cbr>22: end for 47: end for\u003Cbr>23: for each node 􀀽 in 􀁁 do 48:\u003Cbr>24: if 􀀽.type is expression then 49: Output AI , AE , AC |  |  |","cbCaidkqHIK2Fs1C","https://ap.wps.com/l/cbCaidkqHIK2Fs1C","pdf",1635989,24,"English","[{\"question\":\"What does DeCoMa abstract from code datasets?\",\"answer\":\"DeCoMa abstracts identifiers, expressions, and comments into normalized forms (AI, AE, and AC) using a structured mapping algorithm.\"},{\"question\":\"How are identifier names generalized in the mapping?\",\"answer\":\"Identifiers are segmented according to camelCase and snake_case conventions, then aggregated into the abstract identifier set.\"},{\"question\":\"How does DeCoMa handle different literal types like numbers and strings?\",\"answer\":\"Numbers and strings are replaced with generalized tokens (e.g., num, str). The approach then builds corresponding abstract identifiers or expressions during tree parsing and mapping.\"}]","2025-ISSTA-DeCoMa | PDF",1788439342,8]