[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-189876-105":53,"doc-detail-189876-en":127},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":120,"head_meta":122,"extra_data":124,"updated_unix":126},105,"en","log-template-annotation-selection-algorithm-llmlog","Log Template Annotation Selection Algorithm - LLMLog","","Log template annotation is improved through an algorithm that selects which unlabeled logs to label at each round under a limited annotation budget. The method computes semantic-based edit distance and cosine similarity, then estimates an informative score and a confidence measure using LLM template predictions. A greedy round-by-round selection picks logs maximizing the marginal informative gain, aiming to increase prediction consistency while reducing labeling effort. Evaluation reports template and log datasets and compares accuracy, generation time, and API cost across baseline approaches.",{"@graph":63,"@context":119},[64,80,102],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/log-template-annotation-selection-algorithm-llmlog/189876/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/log-template-annotation-selection-algorithm-llmlog/189876.png","ImageObject",442,249,{"name":88,"@type":89},"Genevieve","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-28","2026-09-03",true,{"@type":98,"interactionType":99,"userInteractionCount":101},"InteractionCounter",{"@type":100},"ViewAction",8,{"@type":103,"mainEntity":104},"FAQPage",[105,111,115],{"name":106,"@type":107,"acceptedAnswer":108},"What problem does the annotation selection algorithm address?","Question",{"text":109,"@type":110},"It selects a subset of unlabeled logs for annotation at each round, constrained by a fixed annotation budget, using signals derived from similarity and LLM predictions.","Answer",{"name":112,"@type":107,"acceptedAnswer":113},"Which similarity measures are used when scoring logs?",{"text":114,"@type":110},"Cosine similarity score and semantic-based edit distance between logs are computed to derive informative scores for candidate selection.",{"name":116,"@type":107,"acceptedAnswer":117},"How does the method choose logs during each round?",{"text":118,"@type":110},"It greedily adds the log that maximizes the marginal increase in the informative score until the budget for that round is filled, then moves to the next round.","https://schema.org",{"og:url":78,"og:type":121,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":123,"canonical":78},"index,follow",{"doc_id":125,"site_id":56},189876,1788399762,{"code":4,"msg":5,"data":128},{"doc_id":125,"user_id":129,"nickname":88,"user_avatar":130,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":131,"file_id":132,"file_url":133,"file_type":134,"file_size":135,"view_count":101,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":25,"language":136,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":137,"faqs":138,"seo_title":139,"seo_description":61,"update_tm":126,"read_time":47},1374391974585,"https://ap-avatar.wpscdn.com/davatar_276721f389ce27ea32af1340a28f341c","| \u003Cbr>2024-11-15 [10.0.0.5](10.0.0.5) POST /api/ausers/login 401\u003Cbr>\u003Cbr>2024-11-25 [cse.ust.edu](cse.ust.edu) POST /adam/index 200 |\n| --- |\n| \u003Cbr>2024-11-26 [cuhk.edu](cuhk.edu) GET /api/ausers/login 404 |\n\n\n| \u003Cbr>[DATE] [IP] \u003CPOST> [RESOURCE] [STATUS] |\n| --- |\n| \u003Cbr>[DATE] [IP] \u003CGET> [RESOURCE] [STATUS] |\n\n\n| Notation | Definition |\n| --- | --- |\n| 􀁂,􀁃 | A log 􀁂 ∈ 􀀨 and its template 􀁃 ∈ 􀀩 |\n| ˆ􀁃 | Predicted template ˆ􀁃 of log 􀁂 , 􀁃 ∈ ˆ􀀩 |\n| 􀁆􀁂􀀸 | The 􀀸-th word in the log 􀁂 |\n| 􀁆􀁃􀀸 | The 􀀸-th type in template 􀁃 |\n| K | Keyword set |\n| T | Candidate word type set |\n| 􀀜 = (􀀨,􀀬 ) | Bipartite graph between log set 􀀨 and words 􀀬 |\n| 􀁁 | The 􀁁-th round |\n| 􀀡􀁁 | The annotated logs at the 􀁁-th round |\n| 􀀡 | Annotated logs |\n| 􀀪 | Unlabeled logs |\n| 􀀲􀀾􀁂􀀸􀀽􀀴 (·) | Cosine similarity score |\n| 􀀨􀀚􀀙 (·) | Semantic-based edit distance between two logs |\n| 􀀞􀁂 􀀸 | Representative score of 􀁂 􀀸 |\n| 􀀥 (􀁂 􀀸 , ˆ􀁃􀀸) | Average probability of predicted template ˆ􀁃􀀸 |\n| I (􀁂 􀀸 , ˆ􀁂􀀸) | Prediction consistency indicator |\n| 􀀘 (􀁂 􀀸 , ˆ􀁃􀀸 , ˆ􀁂􀀸) | Confidence score |\n| 􀀗 􀁁 | Annotation budget at the 􀁁 -th round |\n| 􀀬􀁁 | Identified words at the 􀁁 -th round |\n| 􀀞􀀨 (·) | The informative score in Equation (8) |\n| 􀁟 | Trade-off parameter in Equation (8) |\n| 􀀙 􀁂 | Demonstrative logs of 􀁂 |\n| 􀀪􀀬 (􀀙 􀁂) | Word set in 􀀙 􀁂 |\n\n| Algorithm 1: Annotation selection at the 􀁁 -th round |\n| --- |\n| Input: Annotation budget 􀀗 􀁁, unlabelled logs 􀀪 , previous LLM prediction ˆ􀀩􀁁1 and LLM 􀀵 􀁜\u003Cbr>Output: Selected logs 􀀡 􀁁 for annotation\u003Cbr>1 􀀡 􀁁 ← ∅\u003Cbr>2 for 􀁂 ∈ 􀀪 do\u003Cbr>3 ∀􀁂 􀀸 ∈ 􀀪 \\ 􀁂 , 􀀨􀀚􀀙 (􀁂, 􀁂 􀀸) ←Equation (2)\u003Cbr>4 􀀞􀁂 ← Equation (4)\u003Cbr>5   􀀘 (􀁂, ˆ􀁃, ˆ􀁂) ← Equation (7)\u003Cbr>6 while |􀀡 􀁁| \u003C 􀀗 􀁁 do\u003Cbr>7 for 􀁂 ∈ 􀀪 do\u003Cbr>8 􀀞􀀨 (􀀡 􀁁 ∪ {􀁂}) ←Equation (8)\u003Cbr>9   △􀀞􀀨 (􀁂 |􀀡 􀁁) = 􀀞􀀨 (􀀡 􀁁 ∪ {􀁂}) − 􀀞􀀨 (􀀡 􀁁)\u003Cbr>10 􀁂 ∗ = 􀀰􀁁􀀶􀀼􀀰􀁇 􀁂∈􀀪△􀀞􀀨 (􀁂 |􀀡 􀁁)\u003Cbr>11 􀀡 􀁁 = 􀀡 􀁁 ∪ 􀁂 ∗\u003Cbr>12   􀀪 = 􀀪 \\ 􀁂 ∗\u003Cbr>13 return 􀀡 􀁁\u003Cbr>|\n\n| Dataset | Templates\\# | Logs\\# | Words\\# |\n| --- | --- | --- | --- |\n| Android | 165 | 437 | 857 |\n| BGL | 120 | 1367 | 2008 |\n| Hadoop | 114 | 734 | 979 |\n| HDFS | 14 | 2000 | 2960 |\n| Linux | 118 | 290 | 667 |\n| Mac | 341 | 1185 | 3136 |\n| Thunderbird | 149 | 339 | 676 |\n| Zookeeper | 50 | 693 | 959 |\n| HealthApp | 75 | 1179 | 1682 |\n| Spark | 36 | 1699 | 1360 |\n| Windows | 50 | 963 | 1185 |\n| OpenSSH | 27 | 729 | 692 |\n| OpenStack | 43 | 1548 | 1484 |\n| Proxifier | 8 | 1056 | 2284 |\n| HPC | 46 | 381 | 485 |\n| Apache | 5 | 886 | 907 |\n\n\n| Dataset | Drain |  |  | LogPPT |  |  | DivLog |  |  | AdaICL |  |  | LLMLog (Ours) |  |  |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n|  | MLA | PTA | RTA | MLA | PTA | RTA | MLA | PTA | RTA | MLA | PTA | RTA | MLA | PTA | RTA |\n| Android | 73.0 | 56.6 | 62.0 | 76.7 | 58.4 | 68.4 | 63.8 | 58.9 | 68.4 | 97.8 | 89.4 | 92.1 | 99.6 | 94.6 | 96.4 |\n| BGL | 44.4 | 33.9 | 30.8 | 97.0 | 68.6 | 78.3 | 94.0 | 68.4 | 77.5 | 99.4 | 93.5 | 95.8 | 99.9 | 95.1 | 98.3 |\n| Hadoop | 43.9 | 36.8 | 34.2 | 89.5 | 54.0 | 58.8 | 89.0 | 69.3 | 85.1 | 99.4 | 92.2 | 97.4 | 100.0 | 100.0 | 100.0 |\n| HDFS | 95.9 | 81.3 | 92.9 | 90.2 | 85.7 | 85.7 | 100.0 | 100.0 | 100.0 | 99.9 | 86.7 | 92.9 | 100.0 | 100.0 | 100.0 |\n| Linux | 19.4 | 43.4 | 42.2 | 94.9 | 47.5 | 49.1 | 97.3 | 92.4 | 93.2 | 99.7 | 96.6 | 96.6 | 99.8 | 96.6 | 98.3 |\n| Mac | 27.2 | 21.2 | 24.9 | 67.3 | 43.6 | 53.4 | 62.4 | 48.3 | 64.5 | 93.2 | 74.4 | 82.1 | 96.0 | 77.1 | 85.9 |\n| Thunderbird | 19.1 | 29.9 | 36.9 | 92.6 | 50.6 | 59.1 | 88.9 | 86.8 | 92.6 | 98.9 | 83.3 | 90.6 | 99.9 | 93.3 | 98.7 |\n| Zookeeper | 49.8 | 39.1 | 36.0 | 99.0 | 74.1 | 86.0 | 100.0 | 100.0 | 100.0 | 100.0 | 100.0 | 100.0 | 100.0 | 100.0 | 100.0 |\n| HealthApp | 24.1 | 8.3 | 34.7 | 78.9 | 85.3 | 85.3 | 99.9 | 98.7 | 98.7 | 99.9 | 98.7 | 98.7 | 100.0 | 100.0 | 100.0 |\n| Spark | 37.6 | 50.0 | 41.7 | 99.1 | 60.0 | 58.3 | 82.1 | 48.3 | 77.8 | 99.9 | 97.2 | 97.2 | 100.0 | 100.0 | 100.0 |\n| Windows | 69.6 | 46.3 | 50.0 | 98.3 | 55.4 | 72.0 | 97.6 | 55.9 | 76.0 | 99.9 | 92.3 | 96.0 | 100.0 | 100.0 |","cbCaieojof5ySIWj","https://ap.wps.com/l/cbCaieojof5ySIWj","pdf",1288771,"English","# Notation\n## Variables and scoring functions\n# Algorithm 1: Annotation selection\n## Inputs and outputs\n## Greedy selection loop\n# Datasets and evaluation results\n## Template and log statistics\n## Accuracy comparison\n## Generation time and API cost\n# References to equations and confidence scoring","[{\"question\":\"What problem does the annotation selection algorithm address?\",\"answer\":\"It selects a subset of unlabeled logs for annotation at each round, constrained by a fixed annotation budget, using signals derived from similarity and LLM predictions.\"},{\"question\":\"Which similarity measures are used when scoring logs?\",\"answer\":\"Cosine similarity score and semantic-based edit distance between logs are computed to derive informative scores for candidate selection.\"},{\"question\":\"How does the method choose logs during each round?\",\"answer\":\"It greedily adds the log that maximizes the marginal increase in the informative score until the budget for that round is filled, then moves to the next round.\"}]","Log Template Annotation Selection Algorithm - LLMLog | PDF"]