[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-124993-105":59,"doc-detail-124993-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","leveraging-machine-learning-for-enhanced-population-health-monitoring-in-lmics-linking-demographic-surveillance-and-clinic-records","Leveraging Machine Learning for Enhanced Population Health Monitoring in LMICs - Linking Demographic Surveillance and Clinic Records","","Workshop slides present machine learning methods to enhance population health monitoring in LMICs by linking demographic surveillance data with clinic records. The material explains record linkage concepts, contrasts supervised and unsupervised ML approaches, and describes synthetic data generation to enable privacy-preserving model training and algorithm testing. A Python-based unsupervised use case uses KMeans clustering for matching records across datasets. Experiments compare clean and forced-error synthetic clinic data, evaluating changes in match counts, precision, recall, and silhouette scores to assess robustness against real-world data imperfections.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/leveraging-machine-learning-for-enhanced-population-health-monitoring-in-lmics-linking-demographic-surveillance-and-clinic-records/124993/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/leveraging-machine-learning-for-enhanced-population-health-monitoring-in-lmics-linking-demographic-surveillance-and-clinic-records/124993.png","ImageObject",300,407,{"name":92,"@type":93},"Noah","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-26","2026-08-05",true,{"@type":102,"interactionType":103,"userInteractionCount":34},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What is record linkage in the context of population health monitoring?","Question",{"text":112,"@type":113},"Record linkage connects records from different datasets that refer to the same individual or entity, enabling integration for a more comprehensive analysis in research and healthcare applications.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How do supervised and unsupervised machine learning approaches differ for record linkage?",{"text":117,"@type":113},"Supervised ML uses labeled matched and unmatched pairs to learn patterns for future predictions. Unsupervised ML uses no labeled data and groups records based on similarities, such as via clustering.",{"name":119,"@type":110,"acceptedAnswer":120},"Why use synthetic data, and how does it support machine learning development?",{"text":121,"@type":113},"Synthetic data mimics real data while protecting privacy, allowing researchers to train models and test algorithms without exposing sensitive information. It also facilitates secure sharing and collaboration.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},124993,1785895931,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":34,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":8,"language":139,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":67,"update_tm":129,"read_time":81},137451207643,"https://ap-avatar.wpscdn.com/davatar_3d24733baf745e90a7e4bdd5f77d97b2","Leveraging Machine Learning for Enhanced Population Health Monitoring in LMICs: Linking Demographic Surveillance and  \nClinic Records  \nTathagata Bhattacharjee 1 , Emma Slaymaker 1 , Chodziwadziwa Kabudula 2 , Jim Todd 1  \n1 Department of Population Health, London School of Hygiene and Tropical Medicine, University of London, London, United Kingdom  \n2 Medical Research Council/Wits Rural Public Health and Health Transitions Research Unit (Agincourt), South Africa  \nMax Planck Institute for Demographic Research workshop: Demystifying machine learning for population researchers  \nNovember 5th-6th, 2024, Rostock, Germany  \nSynthetic Data  \nSynthetic data mimics real-world data while protecting privacy, enabling researchers to train models and test algorithms without exposing sensitive information . It supports secure data sharing and collaboration, making it essential for advancing population health research.  \nThe Kisesa Health and Demographic Surveillance System ( Kisesa HDSS) in North-Western Tanzania.  \nRecord Linkage  \nRecord linkage is the process of connecting records from different datasets that refer to the same individual or entity. It helps integrate data for a comprehensive view, improving analysis in research, healthcare, and administrative applications.  \nSupervised ML for record linkage, the model is trained on labeled data, where matched and unmatched pairs are known, allowing it to learn patterns for future predictions.  \nUnsupervised ML, no labeled data is provided, and the model groups records based on similarities (e.g. , clustering) without prior knowledge of which records match.  \nGeneration of Synthetic Dataset  \nUse case: Unsupervised machine learning ( KMeans clustering) for record linkage (RL), matching records from two datasets without predefined labels using Python programming.  \nHDSS Dataset: 125852 individuals. Clinic Dataset: 25170 individuals  \nRecord Linkage Cases with Different Levels of Errors: Scenarios where data records from the two sources were compared and matched despite varying degrees of errors or inconsistencies. These errors could arise from spelling variations, missing information, or formatting differences in names, dates of birth, or addresses . By examining record linkage cases with different levels of such induced errors, we can assess the robustness of matching algorithms, ensuring they can handle real-world data imperfections effectively  \nComparison Between Source and Synthetic Datasets  \nGenerating Clean Synthetic Clinic Data Generating Forced Error Synthetic Clinic Data  \nRecord Linkage with less errors in datasets  \nMatches: 18188  \nRecord Linkage with more errors in datasets  \nMatches: 6307  \nConclusion: The comparison between the two datasets highlights the effect of increased data errors on record linkage performance. In the cleaner dataset, the algorithm showed better precision but missed many matches. With the noisier dataset, precision for non-matches dropped drastically, and recall for matches decreased, indicating more difficulty in handling errors . The silhouette score also slightly worsened (from 0.17 to 0.16), reflecting less clear clustering. Additionally, the mismatch between HDSS and clinic records increased, showing the challenge of linking data with higher errors . This underscores the need for better preprocessing or robust methods to handle noisy data .","cbCair0W4Rnw2zoe","https://ap.wps.com/l/cbCair0W4Rnw2zoe","pdf",879888,"English","# Synthetic Data\n## Record Linkage\n# Supervised vs Unsupervised ML for Linkage\n# Generation of Synthetic Dataset\n## Use Case and Datasets\n# Record Linkage with Induced Errors\n## Error Sources and Robustness Assessment\n# Comparison of Source vs Synthetic Datasets\n## Results and Conclusion","[{\"question\":\"What is record linkage in the context of population health monitoring?\",\"answer\":\"Record linkage connects records from different datasets that refer to the same individual or entity, enabling integration for a more comprehensive analysis in research and healthcare applications.\"},{\"question\":\"How do supervised and unsupervised machine learning approaches differ for record linkage?\",\"answer\":\"Supervised ML uses labeled matched and unmatched pairs to learn patterns for future predictions. Unsupervised ML uses no labeled data and groups records based on similarities, such as via clustering.\"},{\"question\":\"Why use synthetic data, and how does it support machine learning development?\",\"answer\":\"Synthetic data mimics real data while protecting privacy, allowing researchers to train models and test algorithms without exposing sensitive information. It also facilitates secure sharing and collaboration.\"}]","Leveraging Machine Learning for Enhanced Population Health Monitoring in LMICs - Linking Demographic Surveillance and Clinic Records | PDF"]