[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-327886-105":59,"doc-detail-327886-en":129},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":122,"head_meta":124,"extra_data":126,"updated_unix":128},105,"en","hcsu-a-dataset-and-benchmark-for-fine-grained-historical-calligraphy-style-understanding","HCSU: A Dataset and Benchmark for Fine-Grained Historical Calligraphy Style Understanding","","Automated fine-grained perception of calligraphy styles—crucial for cultural heritage preservation—remains difficult for Large Vision-Language Models due to dataset constraints such as modal mixture and label flattening. The HCSU dataset introduces 39,307 curated character images from 49 prominent calligraphers spanning 10 dynasties, separating true ink manuscripts from stone rubbings. It provides hierarchical, expert-written aesthetic descriptions and supports two evaluations: fine-grained style discrimination and interpretable aesthetic reasoning.",{"@graph":69,"@context":121},[70,84,104],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/hcsu-a-dataset-and-benchmark-for-fine-grained-historical-calligraphy-style-understanding/327886/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":98,"encodingFormat":97,"isAccessibleForFree":99,"interactionStatistic":100},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/hcsu-a-dataset-and-benchmark-for-fine-grained-historical-calligraphy-style-understanding/327886.png","ImageObject",300,407,{"name":92,"@type":93},"Levi","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-21",true,{"@type":101,"interactionType":102,"userInteractionCount":4},"InteractionCounter",{"@type":103},"ViewAction",{"@type":105,"mainEntity":106},"FAQPage",[107,113,117],{"name":108,"@type":109,"acceptedAnswer":110},"What problem does HCSU address in fine-grained historical calligraphy style understanding?","Question",{"text":111,"@type":112},"HCSU addresses the limitation that existing datasets and benchmarks often suffer from modal mixture and flattened labels, which makes it hard for LVLMs to perceive fine-grained stylistic traits grounded in visual brushwork.","Answer",{"name":114,"@type":109,"acceptedAnswer":115},"How is the HCSU dataset constructed and what key design helps resolve modal mixture?",{"text":116,"@type":112},"HCSU contains 39,307 curated character images from 49 calligraphers across 10 dynasties, systematically decoupling authentic ink manuscripts (Tie) from stone rubbings (Bei) to reduce the modal mixture problem.",{"name":118,"@type":109,"acceptedAnswer":119},"What evaluation protocols does HCSU support?",{"text":120,"@type":112},"HCSU enables two rigorous evaluations: fine-grained style discrimination and interpretable aesthetic reasoning using hierarchical expert-written aesthetic descriptions.","https://schema.org",{"og:url":83,"og:type":123,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":125,"canonical":83},"index,follow",{"doc_id":127,"site_id":62},327886,1789978270,{"code":4,"msg":5,"data":130},{"doc_id":127,"user_id":131,"nickname":92,"user_avatar":132,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":133,"file_id":134,"file_url":135,"file_type":136,"file_size":137,"view_count":4,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":138,"language":139,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":67,"update_tm":128,"read_time":143},7971461740909,"https://ap-avatar.wpscdn.com/davatar_155a257f0dc6eb9ab79c44ca47cae57d","arXiv :2607 .04 147v 1 [ cs .CV] 5 Jul 2026  \nHCSU: A Dataset and Benchmark for Fine-Grained Historical Calligraphy Style Understanding  \nYinsheng Yao 1 *, Yan Liu 1 *, and Chen Ye 1 ,2†  \n1 School of Computer Science and Technology, Tongji University, Shanghai, China  \n2 The Key Laboratory of Embedded System and Service Computing, Ministry of Education, Shanghai, China  \n{2251929,y_an,[yechen}@tongji.edu.cn](yechen}@tongji.edu.cn)  \nAbstract. Automated fine-grained perception of calligraphy styles—atask vital to cultural heritage preservation—remains a critical challenge for Large Vision-Language Models (LVLMs), largely constrained by existing datasets that suffer from modal mixture and flattened labels. To bridge this gap, we introduce HCSU, the first comprehensive dataset tailored for fine-grained Historical Calligraphy Style Understanding. HCSU comprises 39,307 meticulously curated character images from 49 historically prominent calligraphers across 10 dynasties, systematically decoupling authentic ink manuscripts (Tie) from stone rubbings (Bei) to resolve the long-standing modal mixture problem. Moving beyond conventional flattened labels, HCSU provides hierarchical expert-written aesthetic descriptions, enabling two rigorous evaluation protocols: finegrained style discrimination and interpretable aesthetic reasoning. Extensive evaluations reveal a persistent gap between calligraphy-related knowledge and visually grounded style perception: state-of-the-art LVLMs show non-trivial performance but remain sensitive to script-level, textual, and source-specific cues, and often struggle to ground aesthetic judgments in fine-grained brushwork evidence. Ultimately, the HCSU benchmark exposes fundamental limitations in current multimodal architectures, aiming to inspire the evolution of expert-level visual reasoning for cultural heritage preservation. The dataset is available at [https://huggingface.co/datasets/Tongji209/HCSU](https://huggingface.co/datasets/Tongji209/HCSU).  \nKeywords: Calligraphy Styles · Fine-grained Perception · LVLMs  \n1 Introduction  \nRecent advances in Large Vision-Language Models (LVLMs) have greatly expanded the scope of general-purpose visual understanding. Representative models include LLaVA [15] and Qwen-VL [2] . However, their capability often declines  \n*  \n†  \nBoth authors contributed equally to this research.  \nChen Ye is the corresponding author.  \nAccepted at the European Conference on Computer Vision (ECCV) 2026 .  \n2 Y. Yao et al.  \n(a) Instance A (Ren) (b) Instance B (Ren) (c) Instance C (Yi)  \nFig. 1: The challenge of disentangling style from content. (a) and (b) depict the same character identity with distinct artistic executions, whereas (a) and (c) exhibit a consistent authorial signature despite distinct structural forms.  \nwhen moving from coarse object recognition to tasks requiring fine-grained perception, a limitation highlighted by benchmarks such as FG-BMK [29] . For example, evaluations on standard fine-grained datasets like CUB-200-2011 [22] show that LVLMs significantly underperform specialist models when distinguishing visually similar bird sub-species. Despite their broad knowledge base, these models often fail to ground subtle discriminative cues—such as beak shapes or feather patterns—into accurate classifications [7, 31] . Among such challenges, Chinese calligraphic style analysis 1 presents a particularly demanding scenario. While conventional fine-grained tasks such as bird identification require distinguishing subtle but consistent inter-class differences, the underlying visual prototypes remain relatively stable.  \nCalligraphic style analysis fundamentally inverts this paradigm. The core task is to recognize a consistent, abstract authorial signature across visually diverse character instances. For example, a model must identify that Figures 1(a) and 1(c), despite their drastically different structures and stroke counts, belong to the same stylistic class (same calligrap","cbCaihQNGpRnCaQr","https://ap.wps.com/l/cbCaihQNGpRnCaQr","pdf",11248150,17,"English","# Introduction\n## Fine-grained perception limits in LVLMs\n## Calligraphic style analysis as content-style disentanglement\n## Modal mixture issues in existing benchmarks\n## HCSU dataset overview","[{\"question\":\"What problem does HCSU address in fine-grained historical calligraphy style understanding?\",\"answer\":\"HCSU addresses the limitation that existing datasets and benchmarks often suffer from modal mixture and flattened labels, which makes it hard for LVLMs to perceive fine-grained stylistic traits grounded in visual brushwork.\"},{\"question\":\"How is the HCSU dataset constructed and what key design helps resolve modal mixture?\",\"answer\":\"HCSU contains 39,307 curated character images from 49 calligraphers across 10 dynasties, systematically decoupling authentic ink manuscripts (Tie) from stone rubbings (Bei) to reduce the modal mixture problem.\"},{\"question\":\"What evaluation protocols does HCSU support?\",\"answer\":\"HCSU enables two rigorous evaluations: fine-grained style discrimination and interpretable aesthetic reasoning using hierarchical expert-written aesthetic descriptions.\"}]","HCSU: A Dataset and Benchmark for Fine-Grained Historical Calligraphy Style Understanding | PDF",43]