<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Nursing</journal-id><journal-id journal-id-type="publisher-id">nursing</journal-id><journal-id journal-id-type="index">33</journal-id><journal-title>JMIR Nursing</journal-title><abbrev-journal-title>JMIR Nursing</abbrev-journal-title><issn pub-type="epub">2562-7600</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v9i1e106133</article-id><article-id pub-id-type="doi">10.2196/106133</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>AI-Based Structured Information Extraction From Synthetic Nursing Handover Transcripts: Comparative Evaluation of Large Language Models</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Conway</surname><given-names>Aaron</given-names></name><degrees>BN, RN, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hada</surname><given-names>Adriana</given-names></name><degrees>RN, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Schluter</surname><given-names>Jessica</given-names></name><degrees>RN, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xu</surname><given-names>Hui (Grace)</given-names></name><degrees>NP, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lowden</surname><given-names>Dan</given-names></name><degrees>MBBS</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Miller</surname><given-names>Tim</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Donald</surname><given-names>Ken</given-names></name><degrees>MBBS, PhD</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Teodorczuk</surname><given-names>Andrew</given-names></name><degrees>BSc, MD, MBChB</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref></contrib></contrib-group><aff id="aff1"><institution>Centre for Healthcare Transformation, Queensland University of Technology</institution><addr-line>Park Rd, Kelvin Grove</addr-line><addr-line>Brisbane</addr-line><addr-line>QLD</addr-line><country>Australia</country></aff><aff id="aff2"><institution>School of Nursing, Queensland University of Technology</institution><addr-line>Brisbane</addr-line><country>Australia</country></aff><aff id="aff3"><institution>The Prince Charles Hospital, Metro North Health</institution><addr-line>Brisbane</addr-line><country>Australia</country></aff><aff id="aff4"><institution>Caboolture Hospital</institution><addr-line>Brisbane</addr-line><country>Australia</country></aff><aff id="aff5"><institution>School of Electrical Engineering and Computer Science, The University of Queensland</institution><addr-line>Brisbane</addr-line><country>Australia</country></aff><aff id="aff6"><institution>School of Medicine &#x0026; Dentistry, Griffith University</institution><addr-line>Gold Coast</addr-line><addr-line>Queensland</addr-line><country>Australia</country></aff><aff id="aff7"><institution>The University of Queensland Northside Clinical Unit, The University of Queensland</institution><addr-line>Brisbane</addr-line><country>Australia</country></aff><aff id="aff8"><institution>Royal Australian and New Zealand College of Psychiatrists</institution><addr-line>Melbourne</addr-line><country>Australia</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Borycki</surname><given-names>Elizabeth</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Galatzan</surname><given-names>Benjamin</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Zhang</surname><given-names>Chengzhi</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Aaron Conway, BN, RN, PhD, Centre for Healthcare Transformation, Queensland University of Technology, Park Rd, Kelvin Grove, Brisbane, QLD, 4059, Australia, 61 731383887; <email>aaron.conway@qut.edu.au</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>5</day><month>10</month><year>2026</year></pub-date><volume>9</volume><elocation-id>e106133</elocation-id><history><date date-type="received"><day>02</day><month>07</month><year>2026</year></date><date date-type="rev-recd"><day>03</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>07</day><month>09</month><year>2026</year></date></history><copyright-statement>&#x00A9; Aaron Conway, Adriana Hada, Jessica Schluter, Hui (Grace) Xu, Dan Lowden, Tim Miller, Ken Donald, Andrew Teodorczuk. Originally published in JMIR Nursing (<ext-link ext-link-type="uri" xlink:href="https://nursing.jmir.org">https://nursing.jmir.org</ext-link>), 5.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Nursing, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://nursing.jmir.org/">https://nursing.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://nursing.jmir.org/2026/1/e106133"/><abstract><sec><title>Background</title><p>Clinical handover is the process during which responsibility and accountability for care are transferred between clinicians. AI has the potential to improve the reliability and completeness of clinical handover by helping clinicians detect predefined content areas that have been communicated, identify explicit information gaps, and prompt clarification before responsibility is transferred.</p></sec><sec><title>Objective</title><p>This study evaluated the performance of several large language models and prompt optimization strategies for structured information extraction of synthetic nursing handover transcripts.</p></sec><sec sec-type="methods"><title>Methods</title><p>Two registered nurses independently annotated a dataset of 203 synthetic handover transcripts to produce consensus labels for information extraction tasks. Tasks included (1) labeling spans of text into SBAR (Situation, Background, Assessment, Recommendation) categories, (2) content detection to determine if specific pieces of information were communicated, and (3) labeling spans of text that communicated information using uncertain terms that included a subtask for identifying unknown facts. Baseline and Genetic-Pareto (GEPA)&#x2013;optimized prompts were compared for the GPT-5.2, GPT-5-nano, and MedGemma 27B large language models. Additionally, the LangExtract framework was evaluated for span-extraction tasks.</p></sec><sec sec-type="results"><title>Results</title><p>The GPT-5.2&#x2013;optimized model achieved a micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.85 (95% CI 0.83&#x2010;0.88) for content detection, an absolute improvement of +0.08 compared with the matched baseline. GPT-5-nano also performed better after optimization for content detection (micro<italic>&#x2013;F</italic><sub>1</sub>-score 0.81, 95% CI 0.78&#x2010;0.84), suggesting that this structured task was not limited to the highest-capacity model. For SBAR span extraction, GPT-5.2 with prompt optimization achieved a micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.76 (95% CI 0.72&#x2010;0.79), improving by +0.24 compared with baseline and exceeding LangExtract; GPT-5-nano also improved to a micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.69 (95% CI 0.66&#x2010;0.72). Broad uncertainty-span extraction remained comparatively weak despite prompt optimization (micro<italic>&#x2013;F</italic><sub>1</sub>-score 0.41, 95% CI 0.33&#x2010;0.48; absolute improvement +0.06). In contrast, explicit unknown-fact extraction was more accurate with GPT-5.2 (micro<italic>&#x2013;F</italic><sub>1</sub>-score 0.84, 95% CI 0.63&#x2010;1.00), GPT-5-nano (micro<italic>&#x2013;F</italic><sub>1</sub>-score, 0.84 95% CI 0.63&#x2010;1.00), and MedGemma 27B (micro<italic>&#x2013;F</italic><sub>1</sub>-score 0.80, 95% CI 0.63&#x2010;1.00). Genetic-Pareto&#x2013;optimized prompts outperformed the LangExtract approach across each span-extraction task.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Prompt optimization improved matched-model point estimates, with the highest performance observed for predefined content detection and SBAR span extraction. Broad uncertainty extraction remained less accurate than the narrower unknown-fact task. These technical results do not establish clinical effectiveness, safety, or readiness for real-time use. Validation using authentic nursing handover communication and prospective evaluation in clinical workflows are required before clinical application.</p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>clinical handover</kwd><kwd>nursing</kwd><kwd>patient handoff</kwd><kwd>natural language processing</kwd><kwd>large language models</kwd><kwd>prompt engineering</kwd><kwd>information extraction</kwd><kwd>patient safety</kwd><kwd>machine learning</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Clinical handover is the process during which responsibility and accountability for some or all aspects of a patient&#x2019;s care are transferred to another clinician or clinical team [<xref ref-type="bibr" rid="ref1">1</xref>]. In nursing, shift-to-shift and transfer handovers support continuity of surveillance, care priorities, pending actions, and the inclusion of patient and family concerns. Inaccurate, incomplete, or misinterpreted communication at this transition can contribute to delays, duplicated work, and preventable harm [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. International and Australian patient-safety standards therefore prioritize structured clinical handover, particularly at shift changes, transfers, and discharge [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>Structured handover processes aim to standardize minimum content and the format of exchange so that critical information is predictably conveyed, acted upon, and auditable. In Australia, the Communicating for Safety Standard emphasizes structured clinical handover while retaining opportunities for questions, clarification, and confirmation [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. One widely used framework is SBAR (Situation, Background, Assessment, Recommendation), which organizes clinician-to-clinician handover content into 4 information categories [<xref ref-type="bibr" rid="ref6">6</xref>]. Recent systematic review evidence suggests that structured handoff protocols may improve some safety outcomes, but the certainty and implementation fidelity vary by protocol and setting; evidence specific to SBAR remains low certainty [<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>Even where structured tools are mandated or encouraged, their enactment in nursing handovers remains shaped by ward-level organizational and cultural conditions, time demands, interruptions, and the need to communicate patient-specific information for continuity of care [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. This variability is amplified in bedside nursing handover where patient/family involvement is increasingly emphasized, yet research highlights tensions between standardization (predictability) and tailoring (patient-centeredness), along with barriers related to confidentiality and clinician concerns [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. Reviews of handover mnemonics similarly emphasize local validation, clarification, and readback rather than assuming that every element is universally relevant [<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Traditional supervised information-extraction methods can be highly effective for stable tasks with sufficiently large, task-specific labeled datasets. However, there are several potential advantages of using large language models (LLMs) in this context. For example, the same instruction-driven model can be configured for heterogeneous, context-dependent tasks. This may be useful where annotated nursing handover data are limited and communication is conversational and nonlinear. Prior work has demonstrated that few-shot clinical information extraction with LLMs can be accurate, while also showing that performance depends materially on prompt design and task framing [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>Recent advances in LLMs create an opportunity to support clinical handover by analyzing information as it is communicated. Other studies have examined AI-based approaches to support clinical handover, although evaluations of these technologies remain limited [<xref ref-type="bibr" rid="ref14">14</xref>]. For example, a recent multihospital study used an LLM to pregenerate content to support the preparation for handover [<xref ref-type="bibr" rid="ref15">15</xref>]. An alternative application of AI yet to be investigated is to support the verbal clinician-to-clinician exchange itself. This spoken exchange remains central to transferring responsibility of care in clinical settings and provides opportunities to question, clarify, and confirm information [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Accurate extraction of structured information from the spoken exchange is a prerequisite for developing this form of communication support. The aim of this study was to evaluate the accuracy of LLM structured information extraction from synthetic clinical handover transcripts.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This study used a model-evaluation design, corresponding to the &#x201C;LLM evaluation&#x201D; research-design category in TRIPOD-LLM (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis&#x2013;Large Language Models) [<xref ref-type="bibr" rid="ref16">16</xref>] (<xref ref-type="supplementary-material" rid="app1">Checklist 1</xref>). This category covers studies that assess existing LLMs for their accuracy or suitability for a specific health care task. We compared selected LLMs and prompt configurations on predefined multilabel classification and span-extraction tasks using completed synthetic nursing handover transcripts and registered nurse annotations. The study was not an evaluation in a health care setting because it did not test workflow integration or clinical, administrative, or workforce outcomes.</p><p>Tasks related to clinical handover communication that were considered potentially augmentable with AI assistance were identified by the researchers using a co-design process with clinicians and consumers, which will be reported separately. The tasks were as follows:</p><list list-type="bullet"><list-item><p>Extracting spans of text from transcripts that aligned with the SBAR framework for structuring clinical handover. In this study, SBAR span extraction was operationalized as a sequence-labeling task to identify contiguous spans of text in handover transcripts that corresponded to each SBAR category. This task evaluated whether models could map conversational handover text to SBAR-labeled spans, rather than whether the transcripts themselves followed a clean sequential SBAR structure.</p></list-item><list-item><p>Identifying if key elements were communicated in handover transcripts as a content detection task. In this study, this task was operationalized as a checklist of items that are recommended to be addressed during nursing clinical handover, which were developed as part of quality-improvement processes at the researchers&#x2019; institution. The checklist items are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p></list-item><list-item><p>Identifying spans of text in transcripts that were communicated using uncertain terms. In this study, this task was operationalized as span annotation of utterances during handover that conveyed incomplete knowledge, vague or hedged wording, imprecise timing, second-hand sourcing, unclear procedures, or unclear responsibility for follow-up actions. These forms of uncertainty were treated as potentially clinically important because they may indicate information that requires clarification or verification by the receiving clinician. The uncertainty categories and example guidance provided to annotators are summarized in <xref ref-type="table" rid="table2">Table 2</xref>.</p></list-item></list><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Checklist items used to operationalize identification of key concepts and entities in handover transcripts.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Category</td><td align="left" valign="bottom">Checklist item</td></tr></thead><tbody><tr><td align="left" valign="top">Patient involvement</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Clinician introductions</p></list-item><list-item><p>Introduction of clinicians to the patient or carer</p></list-item><list-item><p>Invitation for the patient or carer to participate in handover</p></list-item></list></td></tr><tr><td align="left" valign="top">Identification</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Verification of 3 patient identifiers</p></list-item></list></td></tr><tr><td align="left" valign="top">Situation</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Primary diagnosis or reason for admission</p></list-item><list-item><p>Significant events or complications</p></list-item><list-item><p>Current status, including pending tests/procedures and interim plans/orders</p></list-item></list></td></tr><tr><td align="left" valign="top">Background</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Relevant clinical and social history, including comorbidities</p></list-item><list-item><p>Falls risk</p></list-item><list-item><p>Pressure injury risk</p></list-item><list-item><p>Allergies</p></list-item><list-item><p>Advance care planning</p></list-item></list></td></tr><tr><td align="left" valign="top">Assessment</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Observations, deterioration score, and recent escalations</p></list-item><list-item><p>Pain management</p></list-item><list-item><p>Devices, lines, and vascular access</p></list-item><list-item><p>Critical monitoring and alarms</p></list-item><list-item><p>Nutrition and dietary restrictions</p></list-item><list-item><p>Fluid balance and fluid restrictions</p></list-item><list-item><p>Infusions</p></list-item><list-item><p>Medication chart review, including high-risk medicines</p></list-item><list-item><p>Pathology results or pending investigations</p></list-item><list-item><p>Mobility and use of aids</p></list-item><list-item><p>Skin integrity and related interventions</p></list-item></list></td></tr><tr><td align="left" valign="top">Recommendation</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Discharge plan</p></list-item><list-item><p>Critical actions required</p></list-item><list-item><p>Follow-up care plan or pathway actions</p></list-item><list-item><p>Patient or carer goals and preferences</p></list-item></list></td></tr></tbody></table></table-wrap><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Uncertainty categories used to support annotator identification of uncertainty-related spans in handover transcripts.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Category</td><td align="left" valign="bottom">Definition/when to use</td><td align="left" valign="bottom">Example from handover speech</td></tr></thead><tbody><tr><td align="left" valign="top">Hedge/probability language</td><td align="left" valign="top">The speaker indicates partial confidence or doubt about information.</td><td align="left" valign="top">&#x201C;I think ENT reviewed him&#x201D;; &#x201C;He should be going to theatre soon.&#x201D;</td></tr><tr><td align="left" valign="top">Vague/qualitative expression</td><td align="left" valign="top">Information is described using imprecise or subjective language.</td><td align="left" valign="top">&#x201C;He looks fine now&#x201D;; &#x201C;Seems okay.&#x201D;</td></tr><tr><td align="left" valign="top">Unknown fact/explicit lack of knowledge</td><td align="left" valign="top">The speaker openly states missing knowledge or incomplete data.</td><td align="left" valign="top">&#x201C;Not sure if consent&#x2019;s been signed&#x201D;; &#x201C;I don&#x2019;t know his allergies.&#x201D;</td></tr><tr><td align="left" valign="top">Indefinite timing</td><td align="left" valign="top">Timing or schedule for an event is vague or lacks precision.</td><td align="left" valign="top">&#x201C;Later today&#x201D;; &#x201C;After the round.&#x201D;</td></tr><tr><td align="left" valign="top">Source uncertainty</td><td align="left" valign="top">Information relies on a second-hand or unverifiable source.</td><td align="left" valign="top">&#x201C;ENT said he&#x2019;s on the list&#x201D;; &#x201C;Night nurse told me.&#x201D;</td></tr><tr><td align="left" valign="top">Procedural uncertainty</td><td align="left" valign="top">The next step in care is unclear or the plan is not explicitly stated.</td><td align="left" valign="top">&#x201C;You might want to check his IV.&#x201D;</td></tr><tr><td align="left" valign="top">Responsibility uncertainty</td><td align="left" valign="top">A required task or follow-up is mentioned, but it is unclear who is responsible for performing it.</td><td align="left" valign="top">&#x201C;Bloods to be checked later&#x201D;; &#x201C;Needs review this afternoon.&#x201D;</td></tr></tbody></table></table-wrap></sec><sec id="s2-2"><title>Data Sources</title><p>We used the publicly available National Information and Communications Technology Australia (NICTA) Synthetic Nursing Handover Dataset, which contains synthetic recordings of clinical handovers delivered by a registered nurse based on patient profiles with cardiovascular, neurological, renal, and respiratory conditions [<xref ref-type="bibr" rid="ref17">17</xref>]. In this dataset, handover monologs were generated from comprehensive patient profiles that included information such as the patient&#x2019;s name, age, admission history, inpatient duration, and the familiarity between the nurses giving and receiving the handover. The nurse was instructed to simulate a bedside shift-to-shift handover within a medical ward setting [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>For our study, we used 100 handover samples from the training partition of the NICTA dataset. First, audio recordings from the NICTA dataset were transcribed using the OpenAI Whisper speech-to-text model. Second, 3 videos depicting conversational nursing shift-to-shift handovers were transcribed in a similar manner to provide examples of interactive handover dialogue. These videos were developed at the authors&#x2019; institution for educational purposes to demonstrate best practices for clinical handovers that are used in the undergraduate nursing program. Transcripts from the educational videos were then used as few-shot examples within the DSPy framework [<xref ref-type="bibr" rid="ref18">18</xref>] using the BootstrapFewShot optimizer to guide the transformation of the 100 NICTA monolog transcripts into 2-sided conversational handovers so that the dataset would better reflect real-world contemporary clinical handover interactions. To broaden the range of clinical contexts represented in the dataset, we further synthesized 103 additional handover examples using the GPT-5 model interactively in a chat interface. These scenarios included inter- and intrahospital transfers, postprocedural handovers, emergency department transitions, and handovers that involved patients with complex mental health care needs. The final dataset comprised 203 synthetic handover transcripts used for subsequent annotation and model development.</p></sec><sec id="s2-3"><title>Ethical Considerations</title><p>Ethics approval was not sought because this was not human research as defined by the National Statement on Ethical Conduct in Human Research [<xref ref-type="bibr" rid="ref19">19</xref>]. The study involved only synthetic handover transcripts generated for model evaluation, with no recruitment, observation, or testing of human participants, and no use of patient, clinician, clinical-record, personal, identifiable, or potentially reidentifiable data. Accordingly, the study did not require human ethics review, and a formal exemption from ethics review was not applicable.</p></sec><sec id="s2-4"><title>Annotation</title><p>Two annotators, who are experienced clinically active registered nurses, independently annotated transcripts using a structured rubric. Labeled annotations that met consensus between reviewers were used as the reference labels for downstream model development and evaluation.</p><p>Annotation was performed using the Prodigy annotation software using a custom interface that presented each transcript in 3 components [<xref ref-type="bibr" rid="ref20">20</xref>]. First, annotators highlighted relevant spans of text and assigned labels corresponding to the SBAR framework together with markers of communicative uncertainty, including vagueness, hedging, unknown facts, indefinite timing, source uncertainty, procedural uncertainty, and uncertainty regarding responsibility. Overlapping span labels were permitted where a passage served more than 1 communicative function. Second, annotators completed a multiple-response checklist indicating whether predefined handover elements were present in the transcript.</p><p>Both annotators reviewed all transcripts independently within the same annotation environment and were supported by written guidance and examples to promote consistent interpretation of the coding framework. For the creation of the reference standard, a consensus dataset was derived by retaining only those span annotations and checklist items for which both annotators agreed. In practical terms, this meant that only text segments assigned the same label by both reviewers, and only checklist items selected by both reviewers, were carried forward for downstream model development and evaluation.</p></sec><sec id="s2-5"><title>LLM Prompt Optimization Methods</title><p>Annotated handover transcripts were first partitioned deterministically into optimization (75%) and evaluation (25%) subsets using a fixed random split. Prompt optimization was performed within DSPy [<xref ref-type="bibr" rid="ref18">18</xref>], which is a Python framework that can be used for prompt optimization using feedback from model outputs to improve task performance. For each task, we defined a task-specific DSPy signature that specified the transcript as input and a constrained structured output. Checklist prediction was formulated as multilabel classification, where the task was to return a list of the items from the checklist that were covered in the transcript. The SBAR and uncertainty-related tasks were formulated as span extraction requiring the model to return verbatim text segments from the transcript together with the appropriate label. The uncertainty-related tasks were further subdivided into a broad uncertainty-span extraction task (which included all uncertainty categories) and a more specific unknown-fact extraction task (which included only spans labeled as unknown facts).</p><p>Across the baseline DSPy evaluations, Genetic-Pareto (GEPA)&#x2013;based DSPy optimization experiments, and LangExtract experiments, we purposively selected 3 underlying language models with different deployment profiles: OpenAI GPT-5.2 as the higher-capacity proprietary model, OpenAI GPT-5-nano as a smaller lower-cost proprietary model [<xref ref-type="bibr" rid="ref21">21</xref>], and MedGemma 27B as a 27-billion-parameter open-weight model developed for medical tasks [<xref ref-type="bibr" rid="ref22">22</xref>]. This panel was intended to examine matched prompt effects across contrasting model profiles, not to provide an exhaustive leaderboard of all contemporary LLMs. For DSPy baseline and GEPA runs, the same task model was used before and after optimization so that differences reflected the prompt configuration rather than a change in the underlying model. GPT-5.2 and GPT-5-nano were accessed through cloud-hosted OpenAI API end points; MedGemma 27B was run locally on institutional high-performance computing infrastructure.</p><p>GEPA was used to optimize the task prompts [<xref ref-type="bibr" rid="ref23">23</xref>]. It iteratively evaluated candidate task instructions, combined numeric task scores with natural-language error feedback, and used GPT-5.2 as a separate reflection model to propose revisions. Optimization used 576 scoring calls. Checklist optimization targeted multilabel agreement, while span-task optimization rewarded same-label text overlap using intersection over union (IoU) so that closer boundaries received higher scores. A transcript with neither a reference span nor a predicted span for a target label was treated as a correct negative. The final compiled prompts were fixed before evaluation on the held-out partition. For reporting, span detection performance and boundary overlap were presented separately.</p><p>In addition to DSPy prompt optimizations, we conducted separate evaluations using the LangExtract framework, as a prompt-based few-shot structured information extraction approach for the SBAR, uncertainty-span, and unknown-fact span tasks [<xref ref-type="bibr" rid="ref24">24</xref>]. These experiments used task-specific prompt descriptions together with annotated in-context examples derived from the reference data. We used 10 annotated examples from the training partition as few-shot exemplars, and inference was then performed on the full held-out test partition for each task.</p></sec><sec id="s2-6"><title>Data Analysis</title><p>Interrater agreement was assessed with Cohen &#x03BA; and mean IoU among matched spans. Agreement CIs were calculated with 2000 bootstrap resamples of transcript pairs using a fixed random seed.</p><p>Performance was measured for the checklist content detection task at the level of individual labels with counts of true positives, false positives, false negatives, and true negatives, together with precision, recall, and <italic>F</italic><sub>1</sub>-score. Aggregate performance was summarized using micro-averaged (pooled), macro-averaged (unweighted mean), and support-weighted precision, recall, and <italic>F</italic><sub>1</sub>-score across labels.</p><p>For span-extraction tasks, precision, recall, and <italic>F</italic><sub>1</sub>-score were calculated from predicted spans as binary detection measures, so that these statistics reflected the model&#x2019;s ability to identify the correct labeled spans. Span-boundary agreement was reported separately using the mean IoU across matched pairs. We additionally calculated per-label descriptive metrics including the number of reference spans, the number of predicted spans, matched-span precision, recall, <italic>F</italic><sub>1</sub>-score, and mean IoU.</p><p>Sampling uncertainty was summarized with 95% CIs calculated using nonparametric bootstrap resampling. For each result, we resampled transcripts with replacement 2000 times using a fixed random seed and recalculated the relevant metric from the pooled counts in each resample. Confidence limits are reported as the 2.5th and 97.5th percentiles of the bootstrap distribution.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Preconsensus Interrater Agreement</title><p>Cohen &#x03BA; was 0.75 (95% CI 0.73&#x2010;0.76) for checklist decisions, 0.70 (95% CI 0.68&#x2010;0.72) for pooled SBAR token-by-label decisions, 0.12 (95% CI 0.08&#x2010;0.15) for broad uncertainty, and 0.40 (95% CI 0.18&#x2010;0.64) for unknown facts. Among overlapping same-label spans, the mean IoU was 0.86 (95% CI 0.85&#x2010;0.87) for SBAR, 0.71 (95% CI 0.61&#x2010;0.81) for broad uncertainty, and 0.78 (95% CI 0.56&#x2010;0.98) for unknown facts. The relatively high matched-span IoU indicates similar boundaries when both nurses identified the same span type, with disagreement in uncertainty annotations arising mainly over whether and how to label an expression.</p></sec><sec id="s3-2"><title>Key Findings</title><p>Within-model comparisons showed consistent point-estimate gains with DSPy/GEPA over matched unoptimized-prompt baselines. For GPT-5.2, micro<italic>&#x2013;F</italic><sub>1</sub>-score increased from 0.77 (95% CI 0.74&#x2010;0.80) to 0.85 (95% CI 0.83&#x2010;0.88) for checklist prediction, from 0.51 (95% CI 0.47&#x2010;0.55) to 0.76 (95% CI 0.72&#x2010;0.79) for SBAR span extraction, from 0.35 (95% CI 0.28&#x2010;0.41) to 0.41 (95% CI 0.33&#x2010;0.48) for uncertainty span extraction, and from 0.76 (95% CI 0.50&#x2010;1.00) to 0.84 (95% CI 0.63&#x2010;1.00) for unknown-fact extraction. Across span tasks, LangExtract generally had lower point estimates than the corresponding highest DSPy/GEPA configuration. Among matched span predictions, the mean IoU exceeded 0.8 for the highest-performing GPT-5.2 span-extraction configurations. <xref ref-type="fig" rid="figure1">Figure 1</xref> summarizes matched within-model comparisons across tasks for the 3 evaluated models.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Micro<italic>&#x2013;F</italic><sub>1</sub>-scores for baseline, Genetic-Pareto (GEPA)&#x2013;optimized, and LangExtract strategies for each model within each task.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="nursing_v9i1e106133_fig01.png"/></fig></sec><sec id="s3-3"><title>SBAR Span Extraction</title><p>Among SBAR configurations, the highest overall score was achieved by DSPy/GEPA-optimized GPT-5.2 (micro<italic>&#x2013;F</italic><sub>1</sub>-score 0.76, 95% CI 0.72&#x2010;0.79), followed by DSPy/GEPA-optimized GPT-5-nano (micro<italic>&#x2013;F</italic><sub>1</sub>-score 0.69, 95% CI 0.66&#x2010;0.72) and LangExtract GPT-5.2 (micro<italic>&#x2013;F</italic><sub>1</sub>-score 0.59, 95% CI 0.56&#x2010;0.62). <xref ref-type="table" rid="table3">Table 3</xref> provides label-level results for the best-performing GPT-5.2 DSPy/GEPA SBAR configuration.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Per-label SBAR (Situation, Background, Assessment, Recommendation) metrics for the best-performing GPT-5.2 DSPy/Genetic-Pareto (GEPA) model.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Label</td><td align="left" valign="bottom">Reference spans</td><td align="left" valign="bottom">Predicted spans<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom">Recall (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Mean IoU<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>F<sub>1</sub></italic>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">ASSESSMENT</td><td align="left" valign="top">194</td><td align="left" valign="top">192</td><td align="left" valign="top">0.77 (0.72&#x2010;0.82)</td><td align="left" valign="top">0.78 (0.72&#x2010;0.83)</td><td align="left" valign="top">0.77 (0.72&#x2010;0.81)</td><td align="left" valign="top">0.77 (0.73&#x2010;0.81)</td></tr><tr><td align="left" valign="top">BACKGROUND</td><td align="left" valign="top">47</td><td align="left" valign="top">50</td><td align="left" valign="top">0.72 (0.61&#x2010;0.84)</td><td align="left" valign="top">0.68 (0.57&#x2010;0.79)</td><td align="left" valign="top">0.66 (0.56&#x2010;0.75)</td><td align="left" valign="top">0.70 (0.60&#x2010;0.80)</td></tr><tr><td align="left" valign="top">RECOMMENDATION</td><td align="left" valign="top">113</td><td align="left" valign="top">124</td><td align="left" valign="top">0.76 (0.69&#x2010;0.84)</td><td align="left" valign="top">0.69 (0.62&#x2010;0.78)</td><td align="left" valign="top">0.78 (0.73&#x2010;0.83)</td><td align="left" valign="top">0.73 (0.67&#x2010;0.78)</td></tr><tr><td align="left" valign="top">SITUATION</td><td align="left" valign="top">73</td><td align="left" valign="top">52</td><td align="left" valign="top">0.68 (0.57&#x2010;0.81)</td><td align="left" valign="top">0.96 (0.88&#x2010;1.00)</td><td align="left" valign="top">0.82 (0.75&#x2010;0.88)</td><td align="left" valign="top">0.80 (0.72&#x2010;0.89)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Predicted spans: model-generated spans mapped back to the transcript.</p></fn><fn id="table3fn2"><p><sup>b</sup>IoU: intersection over union for matched span boundaries. </p></fn></table-wrap-foot></table-wrap><p>Within this GPT-5.2 SBAR comparison, macro-precision increased from 0.41 (95% CI 0.37&#x2010;0.44) to 0.78 (95% CI 0.73&#x2010;0.82), macro-recall from 0.69 (95% CI 0.63&#x2010;0.75) to 0.73 (95% CI 0.69&#x2010;0.78), and macro<italic>&#x2013;F</italic><sub>1</sub>-score from 0.49 (95% CI 0.46&#x2010;0.53) to 0.75 (95% CI 0.71&#x2010;0.79). Span-boundary agreement among matched predictions was strongest for SITUATION and RECOMMENDATION, as shown by the label-level mean IoU estimates in <xref ref-type="table" rid="table3">Table 3</xref>.</p></sec><sec id="s3-4"><title>Checklist Task</title><p>For checklist prediction, the best overall result was achieved by DSPy/GEPA-optimized GPT-5.2 (micro<italic>&#x2013;F</italic><sub>1</sub>-score 0.85, 95% CI 0.83&#x2010;0.88; macro<italic>&#x2013;F</italic><sub>1</sub>-score 0.73, 95% CI 0.63&#x2010;0.76; and support-weighted <italic>F</italic><sub>1</sub>-score 0.85, 95% CI 0.82&#x2010;0.88). DSPy/GEPA-optimized GPT-5-nano also performed competitively (micro<italic>&#x2013;F</italic><sub>1</sub>-score 0.81, 95% CI 0.78&#x2010;0.84), while MedGemma 27B reached a micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.76 (95% CI 0.74&#x2010;0.79). <xref ref-type="table" rid="table4">Tables 4</xref> and <xref ref-type="table" rid="table5">5</xref> present grouped per-label estimates for accuracy, precision, recall, and <italic>F</italic><sub>1</sub>-score for the best-performing GPT-5.2 checklist model.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Grouped per-label checklist performance for the best-performing GPT-5.2 DSPy/Genetic-Pareto (GEPA) model.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Checklist item</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">Identification</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ID check of 3 patient identifiers</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td></tr><tr><td align="left" valign="top">Situation</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Primary diagnosis | reason for admission</td><td align="left" valign="top">0.96 (0.90&#x2010;1.00)</td><td align="left" valign="top">0.96 (0.89&#x2010;1.00)</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.98 (0.94&#x2010;1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Current status (awaiting tests/procedures, on interim orders/plan)</td><td align="left" valign="top">0.65 (0.51&#x2010;0.78)</td><td align="left" valign="top">0.73 (0.57&#x2010;0.88)</td><td align="left" valign="top">0.71 (0.55&#x2010;0.87)</td><td align="left" valign="top">0.72 (0.58&#x2010;0.84)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Significant events or complications</td><td align="left" valign="top">0.84 (0.73&#x2010;0.94)</td><td align="left" valign="top">0.62 (0.25&#x2010;1.00)</td><td align="left" valign="top">0.50 (0.17&#x2010;0.83)</td><td align="left" valign="top">0.56 (0.20&#x2010;0.80)</td></tr><tr><td align="left" valign="top" colspan="5">Background</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Alerts-allergies</td><td align="left" valign="top">0.92 (0.84&#x2010;0.98)</td><td align="left" valign="top">0.89 (0.73&#x2010;1.00)</td><td align="left" valign="top">0.89 (0.71&#x2010;1.00)</td><td align="left" valign="top">0.89 (0.76&#x2010;0.98)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Relevant clinical and social history | comorbidities</td><td align="left" valign="top">0.98 (0.94&#x2010;1.00)</td><td align="left" valign="top">0.94 (0.81&#x2010;1.00)</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.97 (0.90&#x2010;1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Alerts-falls risk</td><td align="left" valign="top">0.98 (0.94&#x2010;1.00)</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.80</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Alerts-pressure injury risk</td><td align="left" valign="top">0.98 (0.94&#x2010;1.00)</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.67</td><td align="left" valign="top">0.80</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Advanced care planning</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td><td align="left" valign="top">1.00</td></tr><tr><td align="left" valign="top" colspan="5">Assessment</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Observations | Q-ADDS | recent escalations</td><td align="left" valign="top">0.92 (0.84&#x2010;0.98)</td><td align="left" valign="top">0.95 (0.87&#x2010;1.00)</td><td align="left" valign="top">0.95 (0.87&#x2010;1.00)</td><td align="left" valign="top">0.95 (0.89&#x2010;0.99)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication chart | flag high risk meds</td><td align="left" valign="top">0.90 (0.82&#x2010;0.98)</td><td align="left" valign="top">0.86 (0.71&#x2010;0.97)</td><td align="left" valign="top">0.96 (0.87&#x2010;1.00)</td><td align="left" valign="top">0.91 (0.81&#x2010;0.98)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Devices | lines | vascular access</td><td align="left" valign="top">0.94 (0.88&#x2010;1.00)</td><td align="left" valign="top">0.88 (0.75&#x2010;1.00)</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.94 (0.86&#x2010;1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mobility | aids</td><td align="left" valign="top">0.90 (0.80&#x2010;0.98)</td><td align="left" valign="top">0.80 (0.61&#x2010;0.95)</td><td align="left" valign="top">0.94 (0.80&#x2010;1.00)</td><td align="left" valign="top">0.86 (0.72&#x2010;0.97)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pain management</td><td align="left" valign="top">0.94 (0.86&#x2010;1.00)</td><td align="left" valign="top">0.89 (0.71&#x2010;1.00)</td><td align="left" valign="top">0.94 (0.81&#x2010;1.00)</td><td align="left" valign="top">0.91 (0.80&#x2010;1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Infusions</td><td align="left" valign="top">0.84 (0.73&#x2010;0.94)</td><td align="left" valign="top">0.89 (0.67&#x2010;1.00)</td><td align="left" valign="top">0.53 (0.27&#x2010;0.79)</td><td align="left" valign="top">0.67 (0.40&#x2010;0.86)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pathology</td><td align="left" valign="top">0.88 (0.78&#x2010;0.96)</td><td align="left" valign="top">0.74 (0.53&#x2010;0.93)</td><td align="left" valign="top">0.93 (0.79&#x2010;1.00)</td><td align="left" valign="top">0.82 (0.67&#x2010;0.94)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Nutrition | restrictions</td><td align="left" valign="top">0.88 (0.78&#x2010;0.96)</td><td align="left" valign="top">0.70 (0.50&#x2010;0.89)</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.82 (0.67&#x2010;0.94)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Fluid balance | restrictions</td><td align="left" valign="top">0.90 (0.80&#x2010;0.98)</td><td align="left" valign="top">0.70 (0.38&#x2010;1.00)</td><td align="left" valign="top">0.78 (0.44&#x2010;1.00)</td><td align="left" valign="top">0.74 (0.44&#x2010;0.94)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Skin integrity | interventions</td><td align="left" valign="top">0.90 (0.80&#x2010;0.98)</td><td align="left" valign="top">0.55 (0.22&#x2010;0.83)</td><td align="left" valign="top">1.00</td><td align="left" valign="top">0.71 (0.36&#x2010;0.91)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Critical monitoring | alarms</td><td align="left" valign="top">0.96 (0.90&#x2010;1.00)</td><td align="left" valign="top">0.00 (0.00&#x2010;0.00)</td><td align="left" valign="top">0.00 (0.00&#x2010;0.00)</td><td align="left" valign="top">0.00 (0.00&#x2010;0.00)</td></tr></tbody></table></table-wrap><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Grouped per-label checklist performance for the best-performing GPT-5.2 DSPy/Genetic-Pareto (GEPA) model (continued)<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Checklist item</td><td align="left" valign="bottom">Accuracy<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup> (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">Recommendation</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Care plan/pathway actions to follow up</td><td align="char" char="." valign="top">0.90 (0.80&#x2010;0.98)</td><td align="char" char="." valign="top">0.90 (0.80&#x2010;0.98)</td><td align="left" valign="top">1.00</td><td align="char" char="." valign="top">0.95 (0.89&#x2010;0.99)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Asked patient/carer about goals and preferences</td><td align="char" char="." valign="top">0.80 (0.67&#x2010;0.90)</td><td align="char" char="." valign="top">0.00 (0.00&#x2010;0.00)</td><td align="left" valign="top">0.00 (0.00&#x2010;0.00)</td><td align="char" char="." valign="top">0.00 (0.00&#x2010;0.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Discharge plan</td><td align="char" char="." valign="top">0.98 (0.94&#x2010;1.00)</td><td align="char" char="." valign="top">0.80 (0.33&#x2010;1.00)</td><td align="left" valign="top">1.00</td><td align="char" char="." valign="top">0.89 (0.50&#x2010;1.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Critical actions required</td><td align="char" char="." valign="top">0.92 (0.84&#x2010;0.98)</td><td align="char" char="." valign="top">0.20 (0.00&#x2010;0.67)</td><td align="left" valign="top">1.00</td><td align="char" char="." valign="top">0.33 (0.00&#x2010;0.80)</td></tr><tr><td align="left" valign="top" colspan="5">Patient involvement</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Introduction of clinicians involved in handover to patient/carer</td><td align="char" char="." valign="top">0.84 (0.73&#x2010;0.94)</td><td align="char" char="." valign="top">0.71 (0.55&#x2010;0.87)</td><td align="left" valign="top">1.00</td><td align="char" char="." valign="top">0.83 (0.71&#x2010;0.93)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Invitation for patient/carer to participate in handover</td><td align="char" char="." valign="top">0.82 (0.69&#x2010;0.92)</td><td align="char" char="." valign="top">0.65 (0.44&#x2010;0.83)</td><td align="left" valign="top">0.94 (0.78&#x2010;1.00)</td><td align="char" char="." valign="top">0.77 (0.59&#x2010;0.89)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>95% CIs are not provided for labels with few positive examples in the test set.</p></fn><fn id="table5fn2"><p><sup>b</sup>Accuracy: (true positives + true negatives)/all evaluated transcripts for that checklist item.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Uncertainty and Unknown-Fact Span Extraction</title><p>Broad uncertainty-span extraction included all annotated uncertainty categories (hedging, vagueness, unknown facts, indefinite timing, source uncertainty, procedural uncertainty, and responsibility uncertainty) and had the lowest micro<italic>&#x2013;F</italic><sub>1</sub>-score of the evaluated tasks. The highest DSPy/GEPA point estimate used GPT-5.2 and achieved a precision of 0.32 (95% CI 0.26&#x2010;0.39), a recall of 0.56 (95% CI 0.44&#x2010;0.67), a micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.41 (95% CI 0.33&#x2010;0.48), and a mean IoU of 0.83 (95% CI 0.77&#x2010;0.90). The highest unoptimized-prompt baseline had a micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.35 (95% CI 0.28&#x2010;0.41), and the highest LangExtract uncertainty result had a micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.24 (95% CI 0.17&#x2010;0.31).</p><p>The narrower unknown-fact subtask had higher point estimates. DSPy/GEPA reached a micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.84 (95% CI 0.63&#x2010;1.00) with GPT-5.2, compared with the highest baseline micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.76 (95% CI 0.50&#x2010;1.00) and the LangExtract GPT-5.2 micro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.56 (95% CI 0.29&#x2010;0.72). These results concern explicitly stated lack of knowledge, not facts absent from the transcript. <xref ref-type="table" rid="table6">Table 6</xref> summarizes the uncertainty and unknown-fact span results used for these comparisons.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Uncertainty and unknown-fact span extraction.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Task, model, and approach</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Mean IoU<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup> (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">Uncertainty</td></tr><tr><td align="left" valign="top" colspan="5"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5.2</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Baseline</td><td align="left" valign="top">0.24 (0.19&#x2010;0.29)</td><td align="left" valign="top">0.65 (0.53&#x2010;0.75)</td><td align="left" valign="top">0.35 (0.28&#x2010;0.41)</td><td align="left" valign="top">0.62 (0.53&#x2010;0.70)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DSPy<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup>/GEPA<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup></td><td align="left" valign="top">0.32 (0.26&#x2010;0.39)</td><td align="left" valign="top">0.56 (0.44&#x2010;0.67)</td><td align="left" valign="top">0.41 (0.33&#x2010;0.48)</td><td align="left" valign="top">0.83 (0.77&#x2010;0.90)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LangExtract</td><td align="left" valign="top">0.35 (0.21&#x2010;0.55)</td><td align="left" valign="top">0.07 (0.03&#x2010;0.13)</td><td align="left" valign="top">0.12 (0.05&#x2010;0.20)</td><td align="left" valign="top">0.35 (0.28&#x2010;0.43)</td></tr><tr><td align="left" valign="top" colspan="5"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MedGemma 27B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Baseline</td><td align="left" valign="top">0.12 (0.07&#x2010;0.18)</td><td align="left" valign="top">0.17 (0.10&#x2010;0.23)</td><td align="left" valign="top">0.14 (0.08&#x2010;0.20)</td><td align="left" valign="top">0.56 (0.39&#x2010;0.70)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DSPy/GEPA</td><td align="left" valign="top">0.19 (0.12&#x2010;0.31)</td><td align="left" valign="top">0.20 (0.13&#x2010;0.28)</td><td align="left" valign="top">0.20 (0.13&#x2010;0.27)</td><td align="left" valign="top">0.72 (0.55&#x2010;0.88)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LangExtract</td><td align="left" valign="top">0.22 (0.15&#x2010;0.29)</td><td align="left" valign="top">0.26 (0.18&#x2010;0.36)</td><td align="left" valign="top">0.24 (0.17&#x2010;0.31)</td><td align="left" valign="top">0.69 (0.57&#x2010;0.80)</td></tr><tr><td align="left" valign="top" colspan="5">Unknown fact</td></tr><tr><td align="left" valign="top" colspan="5"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5.2</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Baseline</td><td align="left" valign="top">0.73 (0.44&#x2010;1.00)</td><td align="left" valign="top">0.80 (0.56&#x2010;1.00)</td><td align="left" valign="top">0.76 (0.50&#x2010;1.00)</td><td align="left" valign="top">0.86 (0.66&#x2010;0.99)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DSPy/GEPA</td><td align="left" valign="top">0.89 (0.71&#x2010;1.00)</td><td align="left" valign="top">0.80 (0.56&#x2010;1.00)</td><td align="left" valign="top">0.84 (0.63&#x2010;1.00)</td><td align="left" valign="top">0.92 (0.74&#x2010;0.99)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LangExtract</td><td align="left" valign="top">0.41 (0.17&#x2010;0.59)</td><td align="left" valign="top">0.90 (0.79&#x2010;1.00)</td><td align="left" valign="top">0.56 (0.29&#x2010;0.72)</td><td align="left" valign="top">0.68 (0.45&#x2010;0.84)</td></tr><tr><td align="left" valign="top" colspan="5"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-5-nano</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Baseline</td><td align="left" valign="top">0.08 (0.00&#x2010;0.17)</td><td align="left" valign="top">0.40 (0.00&#x2010;1.00)</td><td align="left" valign="top">0.13 (0.00&#x2010;0.27)</td><td align="left" valign="top">0.74 (0.00&#x2010;0.97)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DSPy/GEPA</td><td align="left" valign="top">0.89 (0.71&#x2010;1.00)</td><td align="left" valign="top">0.80 (0.56&#x2010;1.00)</td><td align="left" valign="top">0.84 (0.63&#x2010;1.00)</td><td align="left" valign="top">0.89 (0.79&#x2010;0.97)</td></tr><tr><td align="left" valign="top" colspan="5"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MedGemma 27B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Baseline</td><td align="left" valign="top">0.30 (0.09&#x2010;0.48)</td><td align="left" valign="top">0.80 (0.56&#x2010;1.00)</td><td align="left" valign="top">0.43 (0.16&#x2010;0.61)</td><td align="left" valign="top">0.90 (0.72&#x2010;0.98)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DSPy/GEPA</td><td align="left" valign="top">0.80 (0.67&#x2010;1.00)</td><td align="left" valign="top">0.80 (0.56&#x2010;1.00)</td><td align="left" valign="top">0.80 (0.63&#x2010;1.00)</td><td align="left" valign="top">0.90 (0.72&#x2010;0.98)</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>IoU: intersection over union for matched span boundaries. </p></fn><fn id="table6fn2"><p><sup>b</sup>DSPy: framework for declarative language model programming.</p></fn><fn id="table6fn3"><p><sup>c</sup>GEPA: Genetic-Pareto.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>This study identified that prompt optimization with GEPA outperformed baseline evaluations across all of the comparisons, with higher aggregate performance for checklist prediction than for SBAR span extraction. This pattern is expected because using generative AI to perform clinical natural language processing tasks is sensitive to framing, label definitions, and example selection. GEPA was explicitly supplied with task-specific scores and error feedback that could align instructions more closely with the annotation scheme [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Optimization was particularly useful for improving targeted span extraction across the SBAR task.</p><p>The difference in accuracy that we identified between checklist prediction and SBAR span extraction provides an important insight to consider for designing AI applications to support clinical handover. Checklist prediction is a simpler structured-output task where each item is a bounded present/absent judgment. By contrast, SBAR extraction requires the model to locate clinically relevant text, assign a communicative category, and reproduce appropriate span boundaries. Structured handover tools aim to make minimum content predictable and reduce omissions, while still requiring adaptation to local clinical workflow [<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>]. As such, bounded present/absent outputs may be useful when the intended function is to prompt review of missing elements or support audit and quality monitoring. It is fortunate, then, that the task with the strongest performance in our study was the one most closely aligned with a recurring mechanism of handover-related harm, namely omitted or incomplete information [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. Although checklist prediction achieved high accuracy, the consequences of both false positives and false negatives should be considered. A false negative would usually create review burden by prompting clarification of an item that was already communicated, but a false positive could create more serious false reassurance that a safety-critical element was conveyed when it was not. In addition, a checklist item not detected in the captured transcript is not equivalent to clinically necessary information having been omitted. A safer near-term design could therefore present AI outputs as source-linked prompts for clinician review, with clear distinctions between detected in transcript, not detected, and requires verification. Evidence has indicated that effective clinical decision support is most useful when integrated into a workflow as actionable and readily available support at the time and place of decision-making, while also recognizing the risk of automation bias when clinicians overrely on system outputs [<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>].</p><p>Registered nurse annotation agreement was comparatively high for checklist and SBAR annotations but very low for the broad uncertainty taxonomy. The latter finding indicates that vague, hedged, or context-dependent communication was difficult to operationalize consistently even with written guidance. However, the combination of low &#x03BA; and relatively high matched IoU scores for broad uncertainty indicates that the principal difficulty was deciding whether uncertainty was present, rather than locating its boundaries once identified. As all model optimization and evaluation used the final consensus ratings between annotators, the results should therefore be interpreted as being conservative.</p><p>This study did not compare LLMs with non&#x2013;LLM-based methods of structured information extraction. It therefore provides evidence about relative prompt optimization approaches and model configurations, not the superiority of LLMs over conventional information extraction. Nevertheless, evaluating LLM-based approaches is a logical next step in this handover-specific research program. In the original NICTA benchmark, a feature-engineered conditional random field achieved a macro<italic>&#x2013;F</italic><sub>1</sub>-score of 0.702 across 35 handover-form categories, but performance was markedly uneven. <italic>F</italic><sub>1</sub>-score was 0.217 for the more abstract &#x201C;other observations&#x201D; category and 0.496 for future-care goals, tasks, and expected outcomes, and the error analysis identified clinically relevant missed and misclassified information [<xref ref-type="bibr" rid="ref17">17</xref>]. These limitations were concentrated in categories requiring greater contextual differentiation. Instruction-tuned LLMs therefore warranted evaluation as a potentially more flexible approach to context-dependent handover information, particularly where prompt optimization can adapt extraction behavior using a modest labeled set.</p><p>The optimized SBAR extraction model achieved high span-boundary overlap among matched spans in our study. It should be considered, though, that SBAR is an idealized communication framework, while real clinical handovers often move nonlinearly, revisit information, distribute relevant details across the conversation, or place content between categories. For implementation, this suggests that extracted spans could be displayed with a link to their source transcript context and should support clinician review of what was said, rather than automatically transforming the handover into a definitive structured note. This preserves the benefits of structured communication while avoiding over-compression of clinically meaningful narrative context [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>].</p></sec><sec id="s4-2"><title>Implementation Considerations</title><p>The appropriate balance between recall and precision may differ by use case when considering the implementation of the tasks we evaluated in this study into an AI tool to support clinical handover. For real-time clinician support, the intervention should be framed as shaping safe handover dialogue rather than only producing a structured output after the exchange has ended. Higher recall may be appropriate where the system displays nondetected checklist items under an &#x201C;Items to check&#x201D; heading and allows the receiving clinician to mark each item as addressed, not relevant, or requiring clarification. This presentation would avoid implying that the information was definitely omitted while supporting clarification before responsibility transfers. For retrospective audit or compliance monitoring, precision becomes more important because false positives could overestimate handover quality and obscure residual safety risks. The present checklist results, with high recall but nontrivial false positives in several categories, therefore support cautious separation of 2 implementation pathways: real-time clinician-facing gap prompts and separately validated audit/reporting workflows [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref34">34</xref>].</p><p>Gold standard handovers include active verification and shared situational awareness rather than passive transfer of uncertain information [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref37">37</xref>]. Evaluation results of the broad uncertainty-span performance indicate that even a GEPA-optimized prompt with a state-of-the-art proprietary LLM is not yet reliable for detecting the full range of ambiguous, hedged, or context-dependent communication. This should be interpreted against the clinical reality that uncertainty is also difficult for humans to recognize and act on consistently, particularly under time pressure or when concern is communicated indirectly. Many forms of clinically important uncertainty are implicit and embedded in shared team understanding: phrases such as &#x201C;he seems off today&#x201D; or &#x201C;not quite themself&#x201D; can convey concern through context, trajectory, and prior knowledge rather than through explicit wording that a model can reliably extract. By contrast, the comparatively more accurate unknown-fact extraction results suggest a potentially useful role for identifying explicit information gaps that can be converted into clarification prompts for the receiving clinician. As foundation models continue to improve, accuracy on this task may also improve, although reliable use in safety-critical handover settings will still require ongoing empirical validation.</p><p>Relatedly, it is highly likely that further gains could be realized after implementation independent of general improvements in underlying LLM capability. If clinicians review AI-generated handover outputs and corrections are retained as labeled examples, subsequent optimization cycles using the same GEPA prompt optimization pipeline we used in this study would plausibly improve performance over time. Prior clinical natural language processing active-learning studies have demonstrated that selectively labeled examples can improve data efficiency for both clinical text classification and clinical named-entity recognition [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. In practice, any such learning cycle should be treated as a controlled quality-improvement process, with monitoring and re-evaluation before revised prompts or models are released into clinical use [<xref ref-type="bibr" rid="ref40">40</xref>].</p><p>Finally, it should be noted that this study measured extraction and classification performance against annotated labels but did not evaluate whether AI outputs changed clinician questioning, closed-loop communication, task completion, escalation, near-miss detection, or patient outcomes. Technical metrics are necessary but insufficient for determining whether a clinical AI tool improves safety in practice [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]. Recent commentary and discursive work on nursing automation similarly argue that evaluation should move beyond time saved to examine how AI redistributes nursing work toward review and verification, whether tools meet end user&#x2013;defined use thresholds, whether performance is equitable across linguistic and workforce groups, and how automation can be integrated without undermining professional values or relational care [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>]. Even accurate prompts may create new risks if clinicians overtrust them, ignore nonhighlighted issues, experience alert fatigue, or redirect attention away from direct patient/carer engagement. Human factors testing should therefore assess reliance, trust calibration, interruption burden, and the usability of source-linked evidence before deployment [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref44">44</xref>].</p></sec><sec id="s4-3"><title>Limitations</title><p>The synthetic conversational handovers used in this study may not fully represent those performed in actual clinical practice where interruptions, time pressure, environmental noise, nonverbal cues, and local team dynamics can affect what is communicated and how it is interpreted. External validation on real handover audio/transcripts from intended deployment settings is therefore required [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. External validity may be particularly limited for complex areas such as mental health, multimorbidity, and other contexts where clinical risk is often communicated through narrative nuance, formulation, behavioral change, staff concern, or tone of interaction rather than discrete data points. These contexts strengthen the need for AI outputs to remain clinician-reviewed, source-linked supports rather than autonomous clinical documentation.</p><p>It should be noted that several checklist items had low prevalence in the evaluation partition. Future evaluation should oversample or otherwise specifically test these items [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Furthermore, the reference standard retained only annotations agreed by both reviewers. This creates a conservative and reproducible benchmark but may exclude ambiguous or contested communication.</p><p>The evaluation corpus included synthetic transcripts generated in part with GPT-5. Although the uses were distinct, information extraction tasks were performed on synthesized language from a model from the same LLM provider, which is potentially susceptible to model familiarity. Synthetic transcripts may also be more predictable than authentic handover communication. In addition, the highest-performing models evaluated in this study were hosted proprietary systems that are updated and managed externally. Model updates, infrastructure variability, or configuration changes could alter performance over time in ways that are not fully transparent or controllable, with implications for reproducibility, governance, and consistency in clinical settings where reliability is critical. Locally hosted models may offer greater control over model versioning and deployment conditions but can require substantial compute and may be less responsive to rapid improvements in frontier model capability.</p><p>Our evaluation focused on downstream extraction from transcripts and did not separately quantify transcription accuracy, speaker attribution, or the effect of noisy audio. In a real-time handover system, errors introduced before the LLM stage could alter checklist detection, SBAR span extraction, and uncertainty identification. End-to-end evaluation should therefore include audio capture, transcription, diarization, and transcript-to-output performance under realistic clinical conditions [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. We used a checklist that was developed at a local institution based on local priorities for communication during clinical nursing handover. Other wards, specialties, transfer types, or jurisdictions may prioritize different minimum datasets or use different terminology. Implementation would therefore require local mapping of labels to existing handover policy, audit tools, escalation pathways, and documentation workflows, followed by local validation rather than direct transfer of the reported performance estimates [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref44">44</xref>].</p></sec><sec id="s4-4"><title>Conclusions</title><p>In this evaluation of AI-assisted clinical handover tasks, prompt optimization consistently improved matched model performance. Accuracy was the highest for bounded checklist prediction and SBAR span extraction. Broad uncertainty detection remained insufficiently reliable, while explicit unknown-fact extraction suggests a narrower but important role for prompting clarification when missing knowledge is directly expressed. Before clinical deployment, these approaches require external validation on real handover audio and transcripts; end-to-end testing of transcription and diarization effects; local calibration of checklist definitions, human-factors evaluation of trust, reliance, workflow burden, and downstream safety outcomes; and structured assessment of organizational readiness for AI implementation in nursing care [<xref ref-type="bibr" rid="ref45">45</xref>].</p></sec></sec></body><back><ack><p>The authors used OpenAI ChatGPT/Codex during manuscript preparation to assist with editing, formatting, and generation of tables and figures to present results. The authors reviewed, verified, and approved all analytic decisions, references, and final manuscript text.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the Prince Charles Hospital Foundation Collaboration Grant. The funder had no role in study design, data collection, analysis, interpretation, manuscript preparation, and the decision to submit.</p></sec><sec><title>Data Availability</title><p>The synthetic handover transcripts, annotation schema, evaluation outputs, and analysis code are available in a public repository on figshare [<xref ref-type="bibr" rid="ref46">46</xref>]. The NICTA Synthetic Nursing Handover Dataset is publicly available from its original source.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: AC, AT</p><p>Data curation: AC</p><p>Formal analysis: AC</p><p>Investigation: AC, AH, JS</p><p>Methodology: AC, AH, AT, DL, KD</p><p>Project administration: AC</p><p>Software: AC</p><p>Visualization: AC</p><p>Writing &#x2013; original draft: AC</p><p>Writing &#x2013; review and editing: AC, AH, JS, GX, DL, TM, KD, AT</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">GEPA </term><def><p>Genetic-Pareto</p></def></def-item><def-item><term id="abb2">IoU </term><def><p>intersection over union</p></def></def-item><def-item><term id="abb3">LLM </term><def><p>large language model</p></def></def-item><def-item><term id="abb4">NICTA </term><def><p>National Information and Communications Technology Australia</p></def></def-item><def-item><term id="abb5">SBAR </term><def><p>Situation, Background, Assessment, Recommendation</p></def></def-item><def-item><term id="abb6">TRIPOD-LLM</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis&#x2013;Large Language Models</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>Further information on communicating for safety</article-title><source>Australian Commission on Safety and Quality in Health Care</source><year>2026</year><access-date>2026-09-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.safetyandquality.gov.au/national-standards/nsqhs-standards/communicating-safety-standard/further-information-communicating-safety">https://www.safetyandquality.gov.au/national-standards/nsqhs-standards/communicating-safety-standard/further-information-communicating-safety</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ong</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Coiera</surname><given-names>E</given-names> </name></person-group><article-title>A systematic review of failures in handoff communication during intrahospital transfers</article-title><source>Jt Comm J Qual Patient Saf</source><year>2011</year><month>06</month><volume>37</volume><issue>6</issue><fpage>274</fpage><lpage>284</lpage><pub-id pub-id-type="doi">10.1016/s1553-7250(11)37035-3</pub-id><pub-id pub-id-type="medline">21706987</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Manias</surname><given-names>E</given-names> </name><name name-style="western"><surname>Geddes</surname><given-names>F</given-names> </name><name name-style="western"><surname>Watson</surname><given-names>B</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>D</given-names> </name><name name-style="western"><surname>Della</surname><given-names>P</given-names> </name></person-group><article-title>Perspectives of clinical handover processes: a multi-site survey across different health professionals</article-title><source>J Clin Nurs</source><year>2016</year><month>01</month><volume>25</volume><issue>1-2</issue><fpage>80</fpage><lpage>91</lpage><pub-id pub-id-type="doi">10.1111/jocn.12986</pub-id><pub-id pub-id-type="medline">26415923</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="report"><person-group person-group-type="author"><collab>The Joint Commission, and Joint Commission International</collab></person-group><article-title>Communication during patient hand-overs</article-title><year>2007</year><access-date>2026-09-22</access-date><publisher-name>World Health Organization</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://cdn.who.int/media/docs/default-source/patient-safety/patient-safety-solutions/ps-solution3-communication-during-patient-handovers.pdf">https://cdn.who.int/media/docs/default-source/patient-safety/patient-safety-solutions/ps-solution3-communication-during-patient-handovers.pdf</ext-link></comment></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hada</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>LV</given-names> </name><name name-style="western"><surname>Jack</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Coyer</surname><given-names>F</given-names> </name></person-group><article-title>Translating evidence-based nursing clinical handover practice in an acute care setting: a quasi-experimental study</article-title><source>Nurs Health Sci</source><year>2021</year><month>06</month><volume>23</volume><issue>2</issue><fpage>466</fpage><lpage>476</lpage><pub-id pub-id-type="doi">10.1111/nhs.12836</pub-id><pub-id pub-id-type="medline">33797197</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yung</surname><given-names>AHW</given-names> </name><name name-style="western"><surname>Pak</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Watson</surname><given-names>B</given-names> </name></person-group><article-title>A scoping review of clinical handover mnemonic devices</article-title><source>Int J Qual Health Care</source><year>2023</year><month>09</month><day>8</day><volume>35</volume><issue>3</issue><fpage>mzad065</fpage><pub-id pub-id-type="doi">10.1093/intqhc/mzad065</pub-id><pub-id pub-id-type="medline">37616494</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCarthy</surname><given-names>S</given-names> </name><name name-style="western"><surname>Motala</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lawson</surname><given-names>E</given-names> </name><name name-style="western"><surname>Shekelle</surname><given-names>PG</given-names> </name></person-group><article-title>Use of structured handoff protocols for within-hospital unit transitions: a systematic review from Making Healthcare Safer IV</article-title><source>BMJ Qual Saf</source><year>2025</year><month>09</month><day>18</day><volume>34</volume><issue>10</issue><fpage>680</fpage><lpage>690</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2024-018385</pub-id><pub-id pub-id-type="medline">40306923</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moyo</surname><given-names>P</given-names> </name><name name-style="western"><surname>Anderson</surname><given-names>J</given-names> </name><name name-style="western"><surname>Francis</surname><given-names>K</given-names> </name><name name-style="western"><surname>Biles</surname><given-names>J</given-names> </name></person-group><article-title>Exploring the experiences and perceptions of the utilisation of structured clinical handover frameworks by nurses working in acute care settings: a scoping review</article-title><source>J Clin Nurs</source><year>2024</year><month>11</month><volume>33</volume><issue>11</issue><fpage>4297</fpage><lpage>4313</lpage><pub-id pub-id-type="doi">10.1111/jocn.17430</pub-id><pub-id pub-id-type="medline">39287216</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chien</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Slade</surname><given-names>D</given-names> </name><name name-style="western"><surname>Goncharov</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Implementing a ward-level intervention to improve nursing handover communication with a focus on bedside handover-a qualitative study</article-title><source>J Clin Nurs</source><year>2024</year><month>07</month><volume>33</volume><issue>7</issue><fpage>2688</fpage><lpage>2706</lpage><pub-id pub-id-type="doi">10.1111/jocn.17107</pub-id><pub-id pub-id-type="medline">38528438</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tobiano</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bucknall</surname><given-names>T</given-names> </name><name name-style="western"><surname>Sladdin</surname><given-names>I</given-names> </name><name name-style="western"><surname>Whitty</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Chaboyer</surname><given-names>W</given-names> </name></person-group><article-title>Patient participation in nursing bedside handover: a systematic mixed-methods review</article-title><source>Int J Nurs Stud</source><year>2018</year><month>01</month><volume>77</volume><fpage>243</fpage><lpage>258</lpage><pub-id pub-id-type="doi">10.1016/j.ijnurstu.2017.10.014</pub-id><pub-id pub-id-type="medline">29149634</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Anshasi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Almayasi</surname><given-names>ZA</given-names> </name></person-group><article-title>Perceptions of patients and nurses about bedside nursing handover: a qualitative systematic review and meta-synthesis</article-title><source>Nurs Res Pract</source><year>2024</year><volume>2024</volume><fpage>3208747</fpage><pub-id pub-id-type="doi">10.1155/2024/3208747</pub-id><pub-id pub-id-type="medline">38716049</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Agrawal</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hegselmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sontag</surname><given-names>D</given-names> </name></person-group><article-title>Large language models are few-shot clinical information extractors</article-title><source>Proc 2022 Conf Empir Methods Nat Lang Process</source><year>2022</year><fpage>1998</fpage><lpage>2022</lpage><pub-id pub-id-type="doi">10.18653/v1/2022.emnlp-main.130</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kelley</surname><given-names>M</given-names> </name><name name-style="western"><surname>Samolyk-Mazzanti</surname><given-names>A</given-names> </name><name name-style="western"><surname>Visweswaran</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name></person-group><article-title>An empirical evaluation of prompting strategies for large language models in zero-shot clinical natural language processing: algorithm development and validation study</article-title><source>JMIR Med Inform</source><year>2024</year><month>04</month><day>8</day><volume>12</volume><fpage>e55318</fpage><pub-id pub-id-type="doi">10.2196/55318</pub-id><pub-id pub-id-type="medline">38587879</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Agha-Mir-Salim</surname><given-names>L</given-names> </name><name name-style="western"><surname>Alberto</surname><given-names>IR</given-names> </name><name name-style="western"><surname>Alberto</surname><given-names>NR</given-names> </name><etal/></person-group><article-title>Technological solutions to improve inpatient handover in the era of artificial intelligence: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>07</month><day>31</day><volume>27</volume><issue>July</issue><fpage>e70358</fpage><pub-id pub-id-type="doi">10.2196/70358</pub-id><pub-id pub-id-type="medline">40743446</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Tsai</surname><given-names>LW</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Shen Hsiao</surname><given-names>ST</given-names> </name><name name-style="western"><surname>Lo</surname><given-names>YS</given-names> </name></person-group><article-title>Integrating a large language model to streamline nursing handover documentation across multiple hospitals in Taiwan: development and implementation study</article-title><source>J Med Internet Res</source><year>2026</year><month>03</month><day>12</day><volume>28</volume><fpage>e81604</fpage><pub-id pub-id-type="doi">10.2196/81604</pub-id><pub-id pub-id-type="medline">41819121</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gallifant</surname><given-names>J</given-names> </name><name name-style="western"><surname>Afshar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ameen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The TRIPOD-LLM reporting guideline for studies using large language models</article-title><source>Nat Med</source><year>2025</year><month>01</month><volume>31</volume><issue>1</issue><fpage>60</fpage><lpage>69</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03425-5</pub-id><pub-id pub-id-type="medline">39779929</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Suominen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>L</given-names> </name><name name-style="western"><surname>Hanlen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ferraro</surname><given-names>G</given-names> </name></person-group><article-title>Benchmarking clinical speech recognition and information extraction: new data, methods, and evaluations</article-title><source>JMIR Med Inform</source><year>2015</year><month>04</month><day>27</day><volume>3</volume><issue>2</issue><fpage>e19</fpage><pub-id pub-id-type="doi">10.2196/medinform.4321</pub-id><pub-id pub-id-type="medline">25917752</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Khattab</surname><given-names>O</given-names> </name><name name-style="western"><surname>Singhvi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Maheshwari</surname><given-names>P</given-names> </name><etal/></person-group><article-title>DSPy: compiling declarative language model calls into self-improving pipelines</article-title><access-date>2026-09-22</access-date><conf-name>The 12th International Conference on Learning Representations (ICLR 2024)</conf-name><conf-date>May 7-11, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=sY5N0zY5Od">https://openreview.net/pdf?id=sY5N0zY5Od</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="report"><person-group person-group-type="author"><collab>Australian Research Council (ARC); Universities Australia</collab></person-group><article-title>National statement on ethical conduct in human research 2025</article-title><year>2025</year><access-date>2026-09-22</access-date><publisher-name>National Health and Medical Research Council (NHMRC)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.nhmrc.gov.au/sites/default/files/documents/attachments/publications/National-Statement-on-Ethical-Conduct-Human-Research-2025.pdf">https://www.nhmrc.gov.au/sites/default/files/documents/attachments/publications/National-Statement-on-Ethical-Conduct-Human-Research-2025.pdf</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="web"><source>Prodigy</source><access-date>2026-09-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://prodi.gy">https://prodi.gy</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Singh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fry</surname><given-names>A</given-names> </name><name name-style="western"><surname>Perelman</surname><given-names>A</given-names> </name><etal/></person-group><article-title>OpenAI GPT-5 system card</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 19, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2601.03267</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sellergren</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kazemzadeh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jaroensri</surname><given-names>T</given-names> </name><etal/></person-group><article-title>MedGemma technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 7, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2507.05201</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Agrawal</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Soylu</surname><given-names>D</given-names> </name><etal/></person-group><article-title>GEPA: reflective prompt evolution can outperform reinforcement learning</article-title><access-date>2026-09-22</access-date><conf-name>14th International Conference on Learning Representations (ICLR 2026)</conf-name><conf-date>Apr 23-27, 2026</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=RQm2KQTM5r">https://openreview.net/pdf?id=RQm2KQTM5r</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Goel</surname><given-names>A</given-names> </name></person-group><article-title>LangExtract v1.7.0</article-title><source>Zenodo</source><year>2026</year><access-date>2026-09-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.17015089">https://doi.org/10.5281/zenodo.17015089</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haig</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Sutton</surname><given-names>S</given-names> </name><name name-style="western"><surname>Whittington</surname><given-names>J</given-names> </name></person-group><article-title>SBAR: a shared mental model for improving communication between clinicians</article-title><source>Jt Comm J Qual Patient Saf</source><year>2006</year><month>03</month><volume>32</volume><issue>3</issue><fpage>167</fpage><lpage>175</lpage><pub-id pub-id-type="doi">10.1016/s1553-7250(06)32022-3</pub-id><pub-id pub-id-type="medline">16617948</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Riesenberg</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Leitzsch</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cunningham</surname><given-names>JM</given-names> </name></person-group><article-title>Nursing handoffs: a systematic review of the literature</article-title><source>Am J Nurs</source><year>2010</year><month>04</month><volume>110</volume><issue>4</issue><fpage>24</fpage><lpage>34</lpage><pub-id pub-id-type="doi">10.1097/01.NAJ.0000370154.79857.09</pub-id><pub-id pub-id-type="medline">20335686</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bukoh</surname><given-names>MX</given-names> </name><name name-style="western"><surname>Siah</surname><given-names>CJR</given-names> </name></person-group><article-title>A systematic review on the structured handover interventions between nurses in improving patient safety outcomes</article-title><source>J Nurs Manag</source><year>2020</year><month>04</month><volume>28</volume><issue>3</issue><fpage>744</fpage><lpage>755</lpage><pub-id pub-id-type="doi">10.1111/jonm.12936</pub-id><pub-id pub-id-type="medline">31859377</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Manser</surname><given-names>T</given-names> </name><name name-style="western"><surname>Foster</surname><given-names>S</given-names> </name></person-group><article-title>Effective handover communication: an overview of research and improvement efforts</article-title><source>Best Pract Res Clin Anaesthesiol</source><year>2011</year><month>06</month><volume>25</volume><issue>2</issue><fpage>181</fpage><lpage>191</lpage><pub-id pub-id-type="doi">10.1016/j.bpa.2011.02.006</pub-id><pub-id pub-id-type="medline">21550543</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kawamoto</surname><given-names>K</given-names> </name><name name-style="western"><surname>Houlihan</surname><given-names>CA</given-names> </name><name name-style="western"><surname>Balas</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Lobach</surname><given-names>DF</given-names> </name></person-group><article-title>Improving clinical practice using clinical decision support systems: a systematic review of trials to identify features critical to success</article-title><source>BMJ</source><year>2005</year><month>04</month><day>2</day><volume>330</volume><issue>7494</issue><fpage>765</fpage><pub-id pub-id-type="doi">10.1136/bmj.38398.500764.8F</pub-id><pub-id pub-id-type="medline">15767266</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goddard</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roudsari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wyatt</surname><given-names>JC</given-names> </name></person-group><article-title>Automation bias: a systematic review of frequency, effect mediators, and mitigators</article-title><source>J Am Med Inform Assoc</source><year>2012</year><volume>19</volume><issue>1</issue><fpage>121</fpage><lpage>127</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2011-000089</pub-id><pub-id pub-id-type="medline">21685142</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Challen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Denny</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pitt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gompels</surname><given-names>L</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>T</given-names> </name><name name-style="western"><surname>Tsaneva-Atanasova</surname><given-names>K</given-names> </name></person-group><article-title>Artificial intelligence, bias and clinical safety</article-title><source>BMJ Qual Saf</source><year>2019</year><month>03</month><volume>28</volume><issue>3</issue><fpage>231</fpage><lpage>237</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2018-008370</pub-id><pub-id pub-id-type="medline">30636200</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rosenbloom</surname><given-names>ST</given-names> </name><name name-style="western"><surname>Denny</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lorenzi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Stead</surname><given-names>WW</given-names> </name><name name-style="western"><surname>Johnson</surname><given-names>KB</given-names> </name></person-group><article-title>Data from clinical notes: a perspective on the tension between structure and flexible documentation</article-title><source>J Am Med Inform Assoc</source><year>2011</year><volume>18</volume><issue>2</issue><fpage>181</fpage><lpage>186</lpage><pub-id pub-id-type="doi">10.1136/jamia.2010.007237</pub-id><pub-id pub-id-type="medline">21233086</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Hilligoss</surname><given-names>PB</given-names> </name></person-group><article-title>The published literature on handoffs in hospitals: deficiencies identified in an extensive review</article-title><source>Qual Saf Health Care</source><year>2010</year><month>12</month><volume>19</volume><issue>6</issue><fpage>493</fpage><lpage>497</lpage><pub-id pub-id-type="doi">10.1136/qshc.2009.033480</pub-id><pub-id pub-id-type="medline">20378628</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Redley</surname><given-names>B</given-names> </name><name name-style="western"><surname>Waugh</surname><given-names>R</given-names> </name></person-group><article-title>Mixed methods evaluation of a quality improvement and audit tool for nurse-to-nurse bedside clinical handover in ward settings</article-title><source>Appl Nurs Res</source><year>2018</year><month>04</month><volume>40</volume><issue>April</issue><fpage>80</fpage><lpage>89</lpage><pub-id pub-id-type="doi">10.1016/j.apnr.2017.12.013</pub-id><pub-id pub-id-type="medline">29579504</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patterson</surname><given-names>ES</given-names> </name><name name-style="western"><surname>Roth</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Woods</surname><given-names>DD</given-names> </name><name name-style="western"><surname>Chow</surname><given-names>R</given-names> </name><name name-style="western"><surname>Gomes</surname><given-names>JO</given-names> </name></person-group><article-title>Handoff strategies in settings with high consequences for failure: lessons for health care operations</article-title><source>Int J Qual Health Care</source><year>2004</year><month>04</month><volume>16</volume><issue>2</issue><fpage>125</fpage><lpage>132</lpage><pub-id pub-id-type="doi">10.1093/intqhc/mzh026</pub-id><pub-id pub-id-type="medline">15051706</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leonard</surname><given-names>M</given-names> </name><name name-style="western"><surname>Graham</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bonacum</surname><given-names>D</given-names> </name></person-group><article-title>The human factor: the critical importance of effective teamwork and communication in providing safe care</article-title><source>Qual Saf Health Care</source><year>2004</year><month>10</month><volume>13</volume><issue>Suppl 1</issue><fpage>i85</fpage><lpage>i90</lpage><pub-id pub-id-type="doi">10.1136/qhc.13.suppl_1.i85</pub-id><pub-id pub-id-type="medline">15465961</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Starmer</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Spector</surname><given-names>ND</given-names> </name><name name-style="western"><surname>Srivastava</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Changes in medical errors after implementation of a handoff program</article-title><source>N Engl J Med</source><year>2014</year><month>11</month><day>6</day><volume>371</volume><issue>19</issue><fpage>1803</fpage><lpage>1812</lpage><pub-id pub-id-type="doi">10.1056/NEJMsa1405556</pub-id><pub-id pub-id-type="medline">25372088</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Figueroa</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Zeng-Treitler</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Ngo</surname><given-names>LH</given-names> </name><name name-style="western"><surname>Goryachev</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wiechmann</surname><given-names>EP</given-names> </name></person-group><article-title>Active learning for clinical text classification: is it better than random sampling?</article-title><source>J Am Med Inform Assoc</source><year>2012</year><volume>19</volume><issue>5</issue><fpage>809</fpage><lpage>816</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2011-000648</pub-id><pub-id pub-id-type="medline">22707743</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lasko</surname><given-names>TA</given-names> </name><name name-style="western"><surname>Mei</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Denny</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name></person-group><article-title>A study of active learning methods for named entity recognition in clinical text</article-title><source>J Biomed Inform</source><year>2015</year><month>12</month><volume>58</volume><fpage>11</fpage><lpage>18</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2015.09.010</pub-id><pub-id pub-id-type="medline">26385377</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Phillips</surname><given-names>RV</given-names> </name><name name-style="western"><surname>Malenica</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Clinical artificial intelligence quality improvement: towards continual monitoring and updating of AI algorithms in healthcare</article-title><source>NPJ Digit Med</source><year>2022</year><month>05</month><day>31</day><volume>5</volume><issue>1</issue><fpage>66</fpage><pub-id pub-id-type="doi">10.1038/s41746-022-00611-y</pub-id><pub-id pub-id-type="medline">35641814</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kelly</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Karthikesalingam</surname><given-names>A</given-names> </name><name name-style="western"><surname>Suleyman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Corrado</surname><given-names>G</given-names> </name><name name-style="western"><surname>King</surname><given-names>D</given-names> </name></person-group><article-title>Key challenges for delivering clinical impact with artificial intelligence</article-title><source>BMC Med</source><year>2019</year><month>10</month><day>29</day><volume>17</volume><issue>1</issue><fpage>195</fpage><pub-id pub-id-type="doi">10.1186/s12916-019-1426-2</pub-id><pub-id pub-id-type="medline">31665002</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ronquillo</surname><given-names>CE</given-names> </name></person-group><article-title>Beyond time saved: implementation, equity, and the utility threshold for nursing AI scribes</article-title><source>J Med Internet Res</source><year>2026</year><month>05</month><day>27</day><volume>28</volume><fpage>e101190</fpage><pub-id pub-id-type="doi">10.2196/101190</pub-id><pub-id pub-id-type="medline">42202251</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pepito</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Acaso</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Merioles</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ismael</surname><given-names>J</given-names> </name></person-group><article-title>Opportunities, challenges, and future directions for the integration of automation in nursing practice: discursive study</article-title><source>JMIR Nurs</source><year>2025</year><month>08</month><day>14</day><volume>8</volume><fpage>e72674</fpage><pub-id pub-id-type="doi">10.2196/72674</pub-id><pub-id pub-id-type="medline">40811767</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sittig</surname><given-names>DF</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>H</given-names> </name></person-group><article-title>A new sociotechnical model for studying health information technology in complex adaptive healthcare systems</article-title><source>Qual Saf Health Care</source><year>2010</year><month>10</month><volume>19 Suppl 3</volume><issue>Suppl 3</issue><fpage>i68</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1136/qshc.2010.042085</pub-id><pub-id pub-id-type="medline">20959322</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seibert</surname><given-names>K</given-names> </name><name name-style="western"><surname>Domhoff</surname><given-names>D</given-names> </name><name name-style="western"><surname>Altona</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Readiness assessment for AI in nursing care projects: multimethods study</article-title><source>JMIR Nurs</source><year>2026</year><month>06</month><day>2</day><volume>9</volume><fpage>e84148</fpage><pub-id pub-id-type="doi">10.2196/84148</pub-id><pub-id pub-id-type="medline">42228895</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="web"><article-title>Large language models for structured information extraction in artificial intelligence-assisted clinical handover: evaluation study</article-title><source>Figshare</source><access-date>2026-09-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://figshare.com/articles/dataset/Large_Language_Models_for_Structured_Information_Extraction_in_Artificial_Intelligence-Assisted_Clinical_Handover_Evaluation_Study/32658084?file=65497158">https://figshare.com/articles/dataset/Large_Language_Models_for_Structured_Information_Extraction_in_Artificial_Intelligence-Assisted_Clinical_Handover_Evaluation_Study/32658084?file=65497158</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Checklist 1</label><p>TRIPOD-LLM checklist.</p><media xlink:href="nursing_v9i1e106133_app1.pdf" xlink:title="PDF File, 93 KB"/></supplementary-material></app-group></back></article>