[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-445112-105":59,"doc-detail-445112-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","development-and-validation-of-a-generative-artificial-intelligence-based-pipeline-for-automated-clinical-data-extraction-from-electronic-health-records-technical-implementation-study","Development and Validation of a Generative Artificial Intelligence-Based Pipeline for Automated Clinical Data Extraction From Electronic Health Records - Technical Implementation Study","","Manual abstraction of unstructured clinical data enables granular outcomes research but is time-consuming and varies in quality. Large language models offer promise for medical extraction, yet their integration into research workflows remains underdescribed. This study develops and integrates an LLM-based system for automated extraction from unstructured EHR text reports within an established clinical outcomes database, evaluating performance, reliability, validation, and cost across multiple MRI batches.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/development-and-validation-of-a-generative-artificial-intelligence-based-pipeline-for-automated-clinical-data-extraction-from-electronic-health-records-technical-implementation-study/445112/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/development-and-validation-of-a-generative-artificial-intelligence-based-pipeline-for-automated-clinical-data-extraction-from-electronic-health-records-technical-implementation-study/445112.png","ImageObject",300,407,{"name":92,"@type":93},"LangkahRina","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-10-01","2026-09-29",true,{"@type":102,"interactionType":103,"userInteractionCount":19},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does the study address in clinical outcomes research?","Question",{"text":112,"@type":113},"The study addresses the need to extract granular data from unstructured EHR text reports, which is often labor-intensive and can be variable in quality.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How was the LLM-based system implemented and evaluated?",{"text":117,"@type":113},"A generative AI pipeline was implemented using a flexible language model interface, structured XML prompts, and database connectivity to generate structured data from EHR documentation, then evaluated on completion rate, processing time, and extraction across multiple data elements using MRI reports.",{"name":119,"@type":110,"acceptedAnswer":120},"What were the main performance results and cost findings?",{"text":121,"@type":113},"When piloted on MRI reports, the system processed 1800 documents with a 100% completion rate, an average processing time of 8.90 seconds per report, and an estimated processing cost of US $0.009 per report.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},445112,1790742105,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":19,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":34,"language":139,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":67,"update_tm":143,"read_time":144},962090893776,"https://ap-avatar.wpscdn.com/davatar_155a257f0dc6eb9ab79c44ca47cae57d","JMIR BIOINFORMATICS AND BIOTECHNOLOGY Carlisle et al  \nOriginal Paper  \nDevelopment and Validation of a Generative Artificial Intelligence-Based Pipeline for Automated Clinical Data Extraction From Electronic Health Records: Technical Implementation Study  \n\n| Marvin N Carlisle 1 , BS; William A Pace 1 , BS; Andrew W Liu 1,2 , BA; Robert Krumm 1 , BA; Janet E Cowan1 , MA; Peter R Carroll 1,3 , MD, MPH; Matthew R Cooperberg 1,3,4 , MD, MPH; Anobel Y Odisho 1,3,4,5 , MD, MPH |\n| --- |\n| 1Department of Urology, University of California, San Francisco, San Francisco, CA, United States 2Chan Medical School, University of Massachusetts, Worcester, MA, United States\u003Cbr>3Helen Diller Family Comprehensive Cancer Center, University of California, San Francisco, San Francisco, CA, United States\u003Cbr>4Department of Epidemiology and Biostatistics, School of Medicine, University of California, San Francisco, San Francisco, CA, United States 5Department of Medicine, Division of Clinical Informatics and Transformation, School of Medicine, University of California, San Francisco, San Francisco, CA, United States\u003Cbr>Corresponding Author:\u003Cbr>Anobel Y Odisho, MD, MPH Department of Urology\u003Cbr>University of California, San Francisco 550 16th Street, Box 1695 San Francisco, CA 94158\u003Cbr>United States\u003Cbr>Phone: 1 5109126645\u003Cbr>Email: [anobel.odisho@ucsf.edu](anobel.odisho@ucsf.edu)\u003Cbr>Abstract |\n\nBackground: The manual abstraction of unstructured clinical data is often necessary for granular clinical outcomes research but is time consuming and can be of variable quality. Large language models (LLMs) show promise in medical data extraction yet integrating them into research workflows remains challenging and poorly described.  \nObjective: This study aimed to develop and integrate an LLM-based system for automated data extraction from unstructured electronic health record (EHR) text reports within an established clinical outcomes database.  \nMethods: We implemented a generative artificial intelligence pipeline (UODBLLM) utilizing a flexible language model interface that supports various LLM implementations, including Health Insurance Portability and Accountability Act-compliant cloud services and local open-source models. We used extensible markup language (XML)-structured prompts and integrated using an open database connectivity interface to generate structured data from clinical documentation in the EHR. We evaluated the UODBLLM’s performance on the completion rate, processing time, and extraction capabilities across multiple clinical data elements, including quantitative measurements, categorical assessments, and anatomical descriptions, using sample magnetic resonance imaging (MRI) reports as test cases . System reliability was tested across multiple batches to assess scalability and consistency.  \nResults: Piloted against MRI reports, UODBLLM processed 1800 clinical documents with a 100% completion rate and an average processing time of 8.90 seconds per report. The token utilization averaged 2692 tokens per report, with an input-to-output ratio of approximately 13:2, resulting in a processing cost of US $0.009 per report. UODBLLM had consistent performance across 18 batches of 100 reports each and completed all processing in 4.45 hours. From each report, UODBLLM extracted 16 structured clinical elements, including prostate volume, prostate-specific antigen values, Prostate Imaging Reporting and Data System scores, clinical staging, and anatomical assessments. All extracted data were automatically validated against predefined schemas and stored in standardized JSON format.  \nConclusions: We demonstrated the successful integration of an LLM-based extraction system within an existing clinical outcomes database, achieving rapid, comprehensive data extraction at minimal cost. UODBLLM provides a scalable, efficient  \n[https://bioinform.jmir.org/2026/1/e70708](https://bioinform.jmir.org/2026/1/e70708) JMIR Bioinform Biotech 2026 | vol. 7 | e70708 | p. 1  \n(","cbCaidXVB5JTs7Fx","https://ap.wps.com/l/cbCaidXVB5JTs7Fx","pdf",392089,"English","# Abstract\n## Background\n## Objective\n## Methods\n## Results\n## Conclusions\n# Introduction\n## Background","[{\"question\":\"What problem does the study address in clinical outcomes research?\",\"answer\":\"The study addresses the need to extract granular data from unstructured EHR text reports, which is often labor-intensive and can be variable in quality.\"},{\"question\":\"How was the LLM-based system implemented and evaluated?\",\"answer\":\"A generative AI pipeline was implemented using a flexible language model interface, structured XML prompts, and database connectivity to generate structured data from EHR documentation, then evaluated on completion rate, processing time, and extraction across multiple data elements using MRI reports.\"},{\"question\":\"What were the main performance results and cost findings?\",\"answer\":\"When piloted on MRI reports, the system processed 1800 documents with a 100% completion rate, an average processing time of 8.90 seconds per report, and an estimated processing cost of US $0.009 per report.\"}]","Development and Validation of a Generative Artificial Intelligence-Based Pipeline for Automated Clinical Data Extraction From Electronic Health Records - Technical Implementation Study | PDF",1790710323,18]