<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Nursing</journal-id><journal-id journal-id-type="publisher-id">nursing</journal-id><journal-id journal-id-type="index">33</journal-id><journal-title>JMIR Nursing</journal-title><abbrev-journal-title>JMIR Nursing</abbrev-journal-title><issn pub-type="epub">2562-7600</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v9i1e100775</article-id><article-id pub-id-type="doi">10.2196/100775</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Nurse-Led Ambient AI Scribe for Patient Safety Incident Investigation Reports (Project NARRATE): Retrospective Pre-Post Comparative Document-Quality Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Teo</surname><given-names>Kai Yunn</given-names></name><degrees>MPSHQ</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Huang</surname><given-names>Liwen</given-names></name><degrees>MPSHQ</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Woh</surname><given-names>Kelly Chai Yuen</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Ang</surname><given-names>Shin Yuh</given-names></name><degrees>MBA</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ng</surname><given-names>Jade Gaik Nai</given-names></name><degrees>MHlthServMgnt</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>Department of Nursing Administration, Singapore General Hospital</institution><addr-line>Outram Road, Singapore</addr-line><addr-line>Singapore</addr-line><country>Singapore</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Borycki</surname><given-names>Elizabeth</given-names></name></contrib><contrib contrib-type="editor"><name name-style="western"><surname>Sarvestan</surname><given-names>Javad</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Rosa</surname><given-names>Delaney La</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Yousefi</surname><given-names>Farzaneh</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Wolfe</surname><given-names>Judith</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Kai Yunn Teo, MPSHQ, Department of Nursing Administration, Singapore General Hospital, Outram Road, Singapore, Singapore, 169608, Singapore, 65 63265843; <email>teo.kai.yunn@sgh.com.sg</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>10</day><month>8</month><year>2026</year></pub-date><volume>9</volume><elocation-id>e100775</elocation-id><history><date date-type="received"><day>08</day><month>05</month><year>2026</year></date><date date-type="rev-recd"><day>18</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>20</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Kai Yunn Teo, Liwen Huang, Kelly Chai Yuen Woh, Shin Yuh Ang, Jade Gaik Nai Ng. Originally published in JMIR Nursing (<ext-link ext-link-type="uri" xlink:href="https://nursing.jmir.org">https://nursing.jmir.org</ext-link>), 10.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Nursing, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://nursing.jmir.org/">https://nursing.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://nursing.jmir.org/2026/1/e100775"/><abstract><sec><title>Background</title><p>Patient safety investigation reports support organizational learning only when they are complete, usable, and sufficiently detailed. Conventional free-text reports are often inconsistent and may omit information needed for review and learning. Project NARRATE (Nursing AI-Refined for Accurate Transcription of Events) is a nursing-led ambient artificial intelligence workflow that uses prompts aligned with the World Health Organization Minimal Information Model for Patient Safety Incident Reporting and Learning Systems, Situation-Background-Assessment-Recommendation output, and visible missing-information cues to support structured supervisor reporting.</p></sec><sec><title>Objective</title><p>This study aimed to compare the completeness and narrative quality of conventional and NARRATE-period supervisor investigation reports for falls and medication administration-related incidents.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a retrospective pre-post document-quality study at a tertiary academic medical center in Singapore. We reviewed 150 deidentified completed supervisor investigation reports: 75 conventional reports from June to August 2025 and 75 confirmed NARRATE reports from January to March 2026. NARRATE use was voluntary, and recorded use represented approximately 40% of eligible postimplementation reports. Two blinded reviewers rated reports using a World Health Organization (WHO)&#x2013;aligned completeness checklist and an adapted 8-domain Physician Documentation Quality Instrument (PDQI). Report-level comparisons were adjusted for repeated reports by the same supervisor using random-intercept linear mixed-effects models. A stratified 60-report plain-paragraph rerating examined whether visible structure influenced ratings.</p></sec><sec sec-type="results"><title>Results</title><p>All 150 reports were analyzed. Unadjusted mean WHO total completeness was 11.81 (SD 3.39) for conventional reports and 13.61 (SD 2.54) for NARRATE reports; the unadjusted difference was 1.80 points, and the cluster-adjusted mean difference was 1.95 (95% CI 0.91&#x2010;3.00; <italic>P</italic>&#x003C;.001). The adapted PDQI mean was 3.61 (SD 0.52) and 4.13 (SD 0.34), respectively; the unadjusted difference was 0.52 points, and the cluster-adjusted mean difference was 0.53 (95% CI 0.37&#x2010;0.69; <italic>P</italic>&#x003C;.001). In the plain-paragraph sensitivity analysis, the completeness advantage remained (adjusted mean difference 1.70, 95% CI 0.27&#x2010;3.14; <italic>P</italic>=.02), as did the adapted PDQI mean advantage (adjusted mean difference 0.25, 95% CI 0.06&#x2010;0.43; <italic>P</italic>=.009). Explanation, organization, and comprehensibility remained significantly higher after deformatting; actions were borderline (<italic>P</italic>=.05), and synthesis, internal consistency, and fairness/balance were not statistically significant.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Among voluntary early adopters, NARRATE use was associated with more complete reports and higher adapted PDQI mean scores after accounting for supervisor clustering. Because recorded use represented approximately 40% of eligible postimplementation reports and users self-selected, findings may reflect adopter and supervisor characteristics. Results support the structured workflow as a whole, not any single AI component, and do not demonstrate downstream patient-safety effects. Confirmatory evaluation under broader adoption with a concurrent, reliably classified comparison group is needed.</p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>ambient AI scribe</kwd><kwd>nursing</kwd><kwd>patient safety</kwd><kwd>incident reporting</kwd><kwd>documentation quality</kwd><kwd>learning health system</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Patient safety incident reporting systems support learning only when reports contain enough structured, interpretable information to guide investigation and improvement [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. The World Health Organization (WHO) Minimal Information Model for Patient Safety Incident Reporting and Learning Systems (MIM PS) specifies the elements needed for this purpose, including what happened, why it happened, and what was done in response [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>In practice, many incident reports fall short. Prior studies have found incomplete capture of causal factors, latent conditions, patient impact, and follow-up actions, limiting both case review and system-level learning [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. These shortcomings are not merely documentation defects; they weaken the ability of reporting systems to generate meaningful analysis and improvement [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>Reporting alone therefore does not guarantee learning [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. Interventions that improve the completeness, structure, and usability of investigation narratives may strengthen the value of incident reporting for review and action.</p></sec><sec id="s1-2"><title>Rationale for the Present Intervention</title><p>Ambient AI documentation tools offer one possible response. These scribes can convert spoken narrative into structured draft notes and may reduce documentation burden while improving consistency. Early evaluations of the in-house SingHealth ambient scribe program suggest acceptable note quality and reduced documentation time in outpatient care [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Whether similar tools improve nursing incident investigation reports remains unclear. To our knowledge, no prior peer-reviewed study has evaluated an ambient AI scribe explicitly aligned with the advanced WHO MIM PS for this purpose.</p></sec><sec id="s1-3"><title>Study Context</title><p>Project NARRATE (Nursing AI-Refined for Accurate Transcription of Events) was developed to improve the quality of completed nursing incident-investigation reports. Built on SingHealth Note Buddy, it prompts supervisors to narrate key WHO MIM PS elements and generates a Situation-Background-Assessment-Recommendation (SBAR)&#x2013;structured draft. When a safety-critical element is missing, the draft retains a visible missing-information cue for supervisor completion before final submission. The workflow was intended to support more complete, organized, and review-ready documentation while reducing manual reformatting.</p><p>We focused on completed supervisor investigation reports rather than initial frontline incident narratives because they contain the postinvestigation information most relevant to learning, including contributing factors, findings, and proposed actions [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref11">11</xref>].</p></sec><sec id="s1-4"><title>Study Objectives and Hypothesis</title><p>This study compared the completeness and narrative quality of completed nursing supervisor investigation reports before and after implementation of the NARRATE workflow. We hypothesized that NARRATE-period reports would score higher on WHO MIM PS&#x2013;aligned completeness and on an adapted 8-domain Physician Documentation Quality Instrument (PDQI)&#x2013;based narrative-quality assessment [<xref ref-type="bibr" rid="ref12">12</xref>]. The primary objective was WHO total completeness; the secondary objective was the adapted PDQI mean; additional objectives included domain-specific and interrater reliability analyses.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>We conducted a retrospective pre-post comparative document-quality study, supplemented by a format-blinded plain-text rerating sensitivity analysis in a stratified subsample to assess whether recognizable SBAR structure influenced ratings. The unit of analysis was the completed supervisor investigation report for a nursing-originated patient safety incident. The exposure was documentation method and period: conventional manual documentation before implementation versus NARRATE-supported documentation after implementation. The primary outcome was WHO total completeness, and the secondary outcome was the adapted PDQI mean. We report the study using the SQUIRE (Standards for Quality Improvement Reporting Excellence) 2.0, supplemented by relevant STROBE (Strengthening the Reporting of Observational Studies in Epidemiology) elements.</p></sec><sec id="s2-2"><title>Setting</title><p>The study was conducted at Singapore General Hospital, a tertiary academic medical center in Singapore, within a nursing-led patient safety documentation improvement initiative.</p></sec><sec id="s2-3"><title>Intervention: NARRATE Supervisor Workflow</title><p>NARRATE stands for Nursing AI-Refined for Accurate Transcription of Events. The intervention used SingHealth Note Buddy, an in-house ambient AI scribe operating within approved institutional digital infrastructure. Note Buddy combines automatic speech recognition with a generative large language model, processes audio in real time, and does not retain audio recordings [<xref ref-type="bibr" rid="ref10">10</xref>]. Users review and edit draft notes before manually transferring verified content into downstream systems [<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>For Project NARRATE, the team configured a supervisor prompt template aligned with the advanced WHO MIM PS. Supervisors narrated incident context, sequence, outcome, causes, contributing and mitigating factors, and actions; the system generated an SBAR-structured draft [<xref ref-type="bibr" rid="ref13">13</xref>]. When key elements were missing, the draft retained a visible missing-information placeholder. Supervisors were required to verify, edit, and complete the draft before transfer into the incident reporting system. The workflow treated AI output as documentation support rather than an autonomous investigation conclusion.</p></sec><sec id="s2-4"><title>AI Governance and Safety Checks</title><p>During the study period, NARRATE operated as a stable, postlaunch workflow. The supervisor prompt template was fixed before the study and unchanged throughout, and the underlying ambient-scribe model did not change during the January to March 2026 window. The underlying platform maintained system-level audit logs of user access and draft-generation activity; these platform logs were not analyzed for this study. Every draft required mandatory supervisor verification and editing, including correction of transcription errors and resolution of all missing-information cues, before transfer into the institutional incident-reporting system.</p><p><xref ref-type="fig" rid="figure1">Figure 1</xref> illustrates the supervisor workflow from verbal narration through human verification to final report submission. <xref ref-type="fig" rid="figure2">Figure 2</xref> provides an illustrative example of NARRATE-generated SBAR output with missing-information placeholders.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Project NARRATE (Nursing AI-Refined for Accurate Transcription of Events) AI-assisted supervisor investigation documentation workflow. The workflow moves from verbal narration to World Health Organization Minimal Information Model for Patient Safety Incident Reporting and Learning Systems (MIM PS)&#x2013;aligned prompts, AI-generated Situation-Background-Assessment-Recommendation (SBAR) draft output, human verification, and transfer to the institutional incident reporting system. Supervisors verify all AI-generated content and resolve any missing-information placeholders before final submission.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="nursing_v9i1e100775_fig01.png"/></fig><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Illustrative sample of NARRATE (Nursing AI-Refined for Accurate Transcription of Events)-generated Situation-Background-Assessment-Recommendation (SBAR) output with missing-information placeholders. This example, based on a fictitious fall incident, shows how the AI-generated draft note retains explicit missing-information markers for safety-critical elements requiring supervisor verification. Supervisors are expected to review and address each placeholder before transferring the verified report into the institutional incident-reporting system. All patient and unit details are fictitious.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="nursing_v9i1e100775_fig02.png"/></fig></sec><sec id="s2-5"><title>Comparator: Conventional Documentation</title><p>In the conventional process, supervisors documented investigation findings manually in free text after reviewing the incident and gathering information from relevant staff or records. Conventional documentation did not include embedded voice-to-text prompts, SBAR-structured AI output, or visible missing-information cues.</p></sec><sec id="s2-6"><title>Sample, Eligibility, and Selection</title><p>We analyzed 150 deidentified completed supervisor investigation reports: 75 conventional reports from June to August 2025 and 75 NARRATE-period reports from January to March 2026. Eligible reports were nursing-originated investigation reports with sufficient narrative text to support scoring. The study focused on falls and medication administration-related incidents.</p><p>We excluded duplicates, administrative entries without substantive investigation narrative, and reports outside the study windows. NARRATE operated through the institutional ambient-scribe platform, which is technically separate from the incident-reporting system, and its use could not be mandated within the existing reporting workflow. Adoption during the postimplementation period was therefore voluntary (opt-in), and recorded NARRATE use represented approximately 40% of eligible reports. Supervisors who used NARRATE recorded the corresponding report number in a brief operational record maintained for a separate reporting-time initiative rather than for the present analysis. Entries in this record allowed confirmed NARRATE-generated reports to be identified, but the record was not designed as an exhaustive NARRATE-use registry. Absence from the record therefore could not confirm nonuse. We did not construct a postimplementation non-NARRATE comparison group because a complement-defined group could have contained unrecorded NARRATE reports and introduced exposure misclassification.</p><p>From the confirmed NARRATE-documented reports, we drew 75 at random, and we drew 75 conventional reports at random from eligible preimplementation reports. Random selection was applied within each pool to avoid investigator selection of individual reports, but it cannot remove adopter self-selection. Early adopters may differ from nonadopters in documentation practice or motivation. We therefore interpret the comparison as the NARRATE workflow as delivered under real-world voluntary uptake rather than under randomized assignment.</p><p>All 150 selected reports were analyzable, with no missing report-level or item-level data.</p><p>Sample size was not determined a priori; it was fixed by the number of eligible reports available within the institutional study windows. Accordingly, we frame this power calculation as a minimum detectable effect analysis for a fixed sample rather than as a sample-size justification. Using G*Power 3.1, a sample of 150 reports (75 per group) was sufficient to detect a standardized mean difference of <italic>d</italic>=0.46 with approximately 80% power at a 2-sided &#x03B1; of .05 [<xref ref-type="bibr" rid="ref14">14</xref>]; both primary outcomes exceeded this minimum detectable effect. Domain-level and subgroup analyses were exploratory.</p></sec><sec id="s2-7"><title>Outcomes and Measurement</title><sec id="s2-7-1"><title>Completeness Checklist Development and Content Validity</title><p>We assessed the primary outcome, report completeness, using a graded checklist derived from WHO technical guidance and aligned with the advanced WHO MIM PS [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. The checklist grouped items into description (what happened and patient impact), explanation (suspected causes, contributing factors, and mitigating factors), and actions (immediate, corrective, and proposed follow-up actions).</p><p>Each item was rated 0=&#x201C;absent,&#x201D; 1=&#x201C;partial or insufficient,&#x201D; 2=&#x201C;clear and sufficient for review and learning,&#x201D; or 9=&#x201C;not applicable.&#x201D; Not-applicable items were excluded from the relevant denominator. We calculated WHO total completeness from applicable items and calculated domain subscores as the mean of items within each domain. The full rubric is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>Content validity was established through structured review by 3 experts with operational responsibility in patient safety incident review: a Nursing Division patient safety officer, an Office of Patient Safety and Quality manager, and a senior principal pharmacist with medication safety expertise.</p><p>Across 2 rounds, panelists evaluated item relevance and contextual appropriateness, and items were refined between rounds. After the second round, all retained items achieved an item-level content validity index of 1.00 [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. The checklist was finalized after the NARRATE prompt template had been locked; the resulting intervention-outcome alignment is addressed in the <italic>Limitations</italic>.</p></sec><sec id="s2-7-2"><title>Adapted 8-Domain PDQI-Based Narrative Quality Assessment</title><p>We assessed narrative quality using an adapted PDQI approach derived from the PDQI [<xref ref-type="bibr" rid="ref12">12</xref>]. For this incident-investigation context, 8 domains were scored: thoroughness, usefulness, organization, comprehensibility, succinctness, synthesis, internal consistency, and fairness/balance. Each domain was rated from 1 to 5, and the adapted PDQI mean was calculated as the average of the 8 domain scores. Accuracy was excluded a priori because the ambient AI workflow did not retain an audio record or other source transcript for independent audit, so reviewers could not compare the final report with the original verbal narration or otherwise verify factual fidelity beyond the report text itself. This exclusion matters because ambient AI outputs can contain transcription errors that may propagate into the final note, occasionally with potential for harm [<xref ref-type="bibr" rid="ref17">17</xref>]; reviewers therefore judged the quality of the final written narrative, not the truth of the underlying investigation. NARRATE drafts were always reviewed and edited by a supervisor before submission. The full rubric is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s2-7-3"><title>Reviewer Training, Blinding, and Scoring</title><p>Two reviewers from the Nursing Safety and Quality team independently assessed all 150 deidentified reports. Neither reviewer was involved in NARRATE design, prompt authoring, user training, or implementation. Reports were deidentified and masked by an independent data handler. Direct identifiers were removed from the reviewer-facing reports, while a pseudonymous supervisor code was retained separately for clustered analysis. The 2 reviewers had no role in data preparation or access to the linkage information.</p><p>Before formal scoring, the reviewers received written rubric guidance and completed iterative calibration on 10 reports outside the study sample; this process also informed refinement of the rubrics in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendices 1</xref> and <xref ref-type="supplementary-material" rid="app2">2</xref>. All calibration-informed rubric refinement, including finalization of the checklist and its item-level content validity index, was completed before any of the 150 analytic reports were rated, so postcalibration rubric drift did not affect the rated sample.</p><p>Residual unblinding remained possible because NARRATE-period reports could retain recognizable SBAR structure. This risk was addressed in the plain-text sensitivity analysis and discussed in the <italic>Limitations</italic>. For the primary analysis, each report was scored independently by both reviewers, and composite scores were calculated as the mean of the 2 ratings after interrater reliability assessment. Individual discrepancies were not adjudicated before the main analysis so that ICC estimates reflected independent ratings.</p></sec></sec><sec id="s2-8"><title>Statistical Analysis</title><p>We prespecified comparisons of preimplementation versus postimplementation reports on the primary outcome (WHO total completeness) and secondary outcome (adapted PDQI [8-domain] mean), with domain-level comparisons treated as exploratory. We reported means and SDs for continuous scores and frequencies with percentages for categorical characteristics.</p><p>Interrater reliability for composite scores was assessed using 2-way random-effects, absolute-agreement intraclass correlation coefficients (ICCs), following the framework described by McGraw and Wong [<xref ref-type="bibr" rid="ref18">18</xref>]. Single-measure ICCs represented the reliability of 1 reviewer&#x2019;s score, and average-measure ICCs represented the reliability of the mean across the 2 reviewers.</p><p>To account for supervisors contributing multiple reports, all report-level comparisons were refitted using linear mixed-effects models with documentation method as a fixed effect and a random intercept for the documenting supervisor. Models used restricted maximum likelihood estimation; 95% CIs and 2-sided <italic>P</italic> values used Wald normal-theory inference. Supervisors were nested within documentation method because none contributed reports in both study periods. Adjusted mean differences are reported as NARRATE minus conventional. Unadjusted group means, SDs, and Cohen <italic>d</italic> values calculated using the pooled SD were retained as descriptive standardized effect-size summaries, with approximate 95% CIs derived from the independent-groups variance estimator [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>Because Levene tests had indicated unequal variance for the 2 headline outcomes, exchangeable generalized estimating equations with robust SEs were also fitted as robustness checks; conclusions were unchanged. Statistical significance was evaluated at a 2-sided &#x03B1; of .05. We performed 11 prespecified exploratory domain comparisons in the full-sample analysis (3 WHO subscores and 8 adapted PDQI domains). These comparisons were not adjusted for multiplicity, so the family-wise risk of type I error is increased, and domain-level findings should be interpreted as hypothesis-generating. Original reliability and descriptive analyses were conducted using IBM SPSS Statistics for Windows (version 26.0; IBM Corp); the clustered reanalysis was conducted in Python 3.12 using statsmodels 0.14.6.</p></sec><sec id="s2-9"><title>Plain-Text Rerating Sensitivity Analysis</title><p>Because recognizable SBAR structure could partially unblind reviewers, we conducted a format-blinded sensitivity analysis on a stratified subsample of 60 reports (30 conventional and 30 NARRATE, balanced by incident type) converted to a uniform plain-paragraph format. SBAR labels, section headers, horizontal rules, and other structural cues were removed while preserving report content; an automated audit confirmed that no residual structural markers remained. The rerating used the same 2 reviewers as the main analysis. To prevent reviewers from linking reports to their original SBAR-formatted versions, the data handler (who did not participate in rating) prepared the deformatted set, removed all structural cues, and renumbered and reordered the reports before rerating. Both reviewers were blinded to documentation method and independently rescored all 60 reports using the same WHO completeness checklist and adapted PDQI (8-domain) rubric. Periods were compared using averaged rater scores and the same random-intercept mixed-model approach, with documenting supervisor as the clustering unit; Cohen <italic>d</italic> values were retained as unadjusted descriptive effect sizes.</p></sec><sec id="s2-10"><title>Ethical Considerations</title><p>We analyzed deidentified existing incident-report text generated through routine institutional reporting for this retrospective quality and safety evaluation; the analytic dataset contained no patient or staff identifiers. The retrospective analysis was reviewed under the institutional Quality Assurance/Service Improvement pathway and was determined not to require review by the SingHealth Centralized Institutional Review Board (institutional reference SHS-RSH-CIRB-4234); this determination for retrospective analysis of existing deidentified records was endorsed by the Research Office, Singapore General Hospital, on May 8, 2026. Individual consent was not required because the study used deidentified existing records and involved no direct interaction with patients or staff. We reported the project in accordance with the SQUIRE 2.0 guidelines for quality-improvement reporting, supplemented by relevant STROBE elements; a completed SQUIRE 2.0 checklist is provided in <xref ref-type="supplementary-material" rid="app3">Checklist 1</xref>.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Report Sample</title><p>A total of 150 nursing-originated supervisor investigation reports were included: 75 conventional and 75 NARRATE reports, covering falls and medication administration-related incidents across both study periods. No cases were excluded. The operational record and associated discussion history identified the documenting supervisor for each selected report. The 75 conventional reports were contributed by 54 supervisors: 39 contributed 1 report, 10 contributed 2, 4 contributed 3, and 1 contributed 4 reports. The 75 NARRATE reports were contributed by 61 supervisors: 49 contributed 1 report, 10 contributed 2, and 2 contributed 3 reports. No supervisor contributed reports in both periods. Sample characteristics by documentation method are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. The 2 groups were comparable in report length (conventional mean 777, SD 475 words; NARRATE mean 733, SD 254 words; <italic>P</italic>=.49), indicating that NARRATE-period gains were not attributable to longer reports.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Sample characteristics by documentation method (N=150)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Conventional (n=75)</td><td align="left" valign="bottom">NARRATE<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> (n=75)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Incident type, n (%)</td></tr><tr><td align="left" valign="top">&#x2003;Fall</td><td align="left" valign="top">41 (55)</td><td align="left" valign="top">45 (60)</td></tr><tr><td align="left" valign="top">&#x2003;Medication administration-related</td><td align="left" valign="top">34 (45)</td><td align="left" valign="top">30 (40)</td></tr><tr><td align="left" valign="top">Sampling period</td><td align="left" valign="top">June-August 2025</td><td align="left" valign="top">January-March 2026</td></tr><tr><td align="left" valign="top">Report length (words), mean (SD)</td><td align="left" valign="top">777 (475)</td><td align="left" valign="top">733 (254)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Report length was calculated as the word count of the event description. Harm-severity tier, ward type, and reporter seniority were not available in the deidentified dataset. Groups did not differ significantly in report length (<italic>P</italic>=.49).</p></fn><fn id="table1fn2"><p><sup>b</sup>NARRATE: Nursing AI-Refined for Accurate Transcription of Events. </p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Interrater Reliability</title><p><xref ref-type="table" rid="table2">Table 2</xref> summarizes interrater reliability for the composite scores. Agreement was moderate for WHO total completeness and somewhat stronger for the adapted PDQI (8-domain) mean. For WHO total completeness, Cronbach &#x03B1; was 0.829; the single-measure ICC was 0.689 (95% CI 0.577&#x2010;0.772; <italic>P</italic>&#x003C;.001), and the average-measure ICC was 0.816 (95% CI 0.732&#x2010;0.872). For the adapted PDQI (8-domain) mean, Cronbach &#x03B1; was 0.853; the single-measure ICC was 0.744 (95% CI 0.664&#x2010;0.808; <italic>P</italic>&#x003C;.001), and the average-measure ICC was 0.854 (95% CI 0.798&#x2010;0.894). Across the 9 WHO items (1350 ratings), reviewers agreed exactly on 957 (70.9%) ratings, differed by 1 category on 285 (21.1%), and differed by more than 1 category on 108 (8%). Larger disagreements clustered in the explanation domain, especially contributing factors (38/150, 25.3%), mitigating factors (23/150, 15.3%), and suspected cause (15/150, 10%), whereas description items showed high agreement (5/150, 3.3% differing by &#x003E;1 category). To confirm that the explanation-domain advantage was not driven by a single reviewer, we examined each reviewer separately: both scored NARRATE-period explanation content higher than conventional content (reviewer 1, mean difference 0.49, <italic>d</italic>=0.81, <italic>P</italic>&#x003C;.001; reviewer 2, mean difference 0.17, <italic>d</italic>=0.26, <italic>P</italic>=.11). The direction was therefore consistent across reviewers, although the effect was larger and statistically significant for 1 reviewer and smaller and nonsignificant for the other, consistent with the greater interpretive difficulty of explanation items.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Interrater reliability for composite report-quality scores (N=150).</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom">Cronbach &#x03B1;</td><td align="left" valign="bottom">ICC, single-measure (95% CI)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="bottom">ICC, average-measure (95% CI)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">WHO<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> total completeness</td><td align="left" valign="top">0.829</td><td align="left" valign="top">0.689 (0.577&#x2010;0.772)</td><td align="left" valign="top">0.816 (0.732&#x2010;0.872)</td></tr><tr><td align="left" valign="top">Adapted PDQI<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup> (8-domain) mean</td><td align="left" valign="top">0.853</td><td align="left" valign="top">0.744 (0.664&#x2010;0.808)</td><td align="left" valign="top">0.854 (0.798&#x2010;0.894)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>ICC: intraclass correlation coefficient; ICCs are 2-way random-effects, absolute-agreement. <italic>P</italic>&#x003C;.001 for all ICCs.</p></fn><fn id="table2fn2"><p><sup>b</sup>WHO: World Health Organization.</p></fn><fn id="table2fn3"><p><sup>c</sup>PDQI: Physician Documentation Quality Instrument.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Primary and Secondary Outcomes</title><p>NARRATE reports had higher WHO total completeness scores than conventional reports. The observed mean was 13.61 (SD 2.54) for NARRATE reports and 11.81 (SD 3.39) for conventional reports, corresponding to an unadjusted difference of 1.80 points. After accounting for multiple reports contributed by the same supervisor, the cluster-adjusted mean difference was 1.95 points (95% CI 0.91&#x2010;3.00; <italic>P</italic>&#x003C;.001). The corresponding unadjusted descriptive standardized effect size was Cohen <italic>d</italic>=0.60 (95% CI 0.27&#x2010;0.93).</p><p>NARRATE reports had higher adapted PDQI (8-domain) mean scores. The observed mean was 4.13 (SD 0.34) for NARRATE reports and 3.61 (SD 0.52) for conventional reports, corresponding to an unadjusted difference of 0.52 points. After accounting for multiple reports contributed by the same supervisor, the cluster-adjusted mean difference was 0.53 (95% CI 0.37&#x2010;0.69; <italic>P</italic>&#x003C;.001) points. The corresponding unadjusted descriptive standardized effect size was Cohen <italic>d</italic>=1.18 (95% CI 0.83&#x2010;1.53). Exchangeable generalized estimating equation robustness analyses yielded similar estimates for WHO total completeness (1.99, 95% CI 0.94&#x2010;3.04; <italic>P</italic>&#x003C;.001) and adapted PDQI mean (0.53, 95% CI 0.37&#x2010;0.70; <italic>P</italic>&#x003C;.001). <xref ref-type="table" rid="table3">Table 3</xref> summarizes group-level results.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Overall report completeness and narrative quality by documentation method (N=150).</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom">Conventional, mean (SD)<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom">NARRATE<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>, mean (SD)<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom">Adjusted mean difference (95% CI)<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom"><italic>P</italic> value<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom">Cohen <italic>d</italic><sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup> (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">WHO<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup> total completeness</td><td align="char" char="." valign="top">11.81 (3.39)</td><td align="char" char="." valign="top">13.61 (2.54)</td><td align="char" char="." valign="top">1.95 (0.91-3.00)</td><td align="char" char="." valign="top">&#x003C;.001</td><td align="char" char="." valign="top">0.60 (0.27-0.93)</td></tr><tr><td align="left" valign="top">Adapted PDQI<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup> (8-domain) mean</td><td align="char" char="." valign="top">3.61 (0.52)</td><td align="char" char="." valign="top">4.13 (0.34)</td><td align="char" char="." valign="top">0.53 (0.37-0.69)</td><td align="char" char="." valign="top">&#x003C;.001</td><td align="char" char="." valign="top">1.18 (0.83-1.53)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Mean differences, 95% CIs, and <italic>P</italic> values are from restricted maximum likelihood random-intercept linear mixed-effects models with documentation method as a fixed effect and documenting supervisor as a random intercept.</p></fn><fn id="table3fn2"><p><sup>b</sup>NARRATE: Nursing AI-Refined for Accurate Transcription of Events.</p></fn><fn id="table3fn3"><p><sup>c</sup>Cohen <italic>d</italic> was calculated from unadjusted group means using the pooled SD; 95% CIs for <italic>d</italic> were derived from the independent-groups variance estimator.</p></fn><fn id="table3fn4"><p><sup>d</sup>WHO: World Health Organization.</p></fn><fn id="table3fn5"><p><sup>e</sup>PDQI: Physician Documentation Quality Instrument.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>Exploratory WHO Completeness Domain Analysis</title><p>Exploratory WHO domain analysis suggested that the overall completeness improvement was driven mainly by explanation and actions (<xref ref-type="table" rid="table4">Table 4</xref>). In cluster-adjusted models, the description subscore increased only slightly from 1.86 (SD 0.32) to 1.94 (SD 0.18; adjusted mean difference 0.08, 95% CI &#x2212;0.01 to 0.16; <italic>P</italic>=.07; <italic>d</italic>=0.30), suggesting a ceiling effect. Explanation increased from 1.00 (SD 0.58) to 1.33 (SD 0.51; adjusted mean difference 0.34, 95% CI 0.15&#x2010;0.52; <italic>P</italic>&#x003C;.001; <italic>d</italic>=0.60), and actions increased from 1.27 (SD 0.43) to 1.43 (SD 0.31; adjusted mean difference 0.19, 95% CI 0.06&#x2010;0.32; <italic>P</italic>=.004; <italic>d</italic>=0.43). Cohen <italic>d</italic> values are unadjusted descriptive effects. These subscore findings should be interpreted as exploratory.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>World Health Organization minimal information model for patient safety incident reporting and learning systems&#x2013;aligned completeness subscores by documentation method (N=150)<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">WHO<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> domain<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="bottom">Conventional, mean (SD)<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="bottom">NARRATE<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup>, mean (SD)<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="bottom">Adjusted mean difference (95% CI)<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="bottom"><italic>P</italic> value<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="bottom">Cohen <italic>d</italic><sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup> (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Description</td><td align="left" valign="top">1.86 (0.32)</td><td align="left" valign="top">1.94 (0.18)</td><td align="left" valign="top">0.08 (&#x2212;0.01 to 0.16)</td><td align="left" valign="top">.07</td><td align="left" valign="top">0.30 (&#x2212;0.03 to 0.62)</td></tr><tr><td align="left" valign="top">Explanation</td><td align="left" valign="top">1.00 (0.58)</td><td align="left" valign="top">1.33 (0.51)</td><td align="left" valign="top">0.34 (0.15 to 0.52)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.60 (0.28 to 0.93)</td></tr><tr><td align="left" valign="top">Actions</td><td align="left" valign="top">1.27 (0.43)</td><td align="left" valign="top">1.43 (0.31)</td><td align="left" valign="top">0.19 (0.06 to 0.32)</td><td align="left" valign="top">.004</td><td align="left" valign="top">0.43 (0.11 to 0.76)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Subscore values are means of constituent items within each domain.</p></fn><fn id="table4fn2"><p><sup>b</sup>WHO: World Health Organization.</p></fn><fn id="table4fn3"><p><sup>c</sup>Domain-level analyses were exploratory.</p></fn><fn id="table4fn4"><p><sup>d</sup>Mean differences, 95% CIs, and <italic>P</italic> values are from restricted maximum likelihood random-intercept linear mixed-effects models clustered by documenting supervisor.</p></fn><fn id="table4fn5"><p><sup>e</sup>NARRATE: Nursing AI-Refined for Accurate Transcription of Events.</p></fn><fn id="table4fn6"><p><sup>f</sup>Cohen <italic>d</italic> values are unadjusted descriptive effects. Domain-level analyses were exploratory.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Exploratory Adapted PDQI (8-Domain) Domain Analysis</title><p>Exploratory adapted PDQI (8-domain) analysis showed higher unadjusted scores in the NARRATE group across all 8 measured domains (<xref ref-type="table" rid="table5">Table 5</xref>). In cluster-adjusted models, differences were statistically significant for thoroughness, usefulness, organization, comprehensibility, synthesis, internal consistency, and fairness/balance. Succinctness improved numerically from 4.31 (SD 0.80) to 4.47 (SD 0.57) but was not statistically significant (adjusted mean difference 0.12, 95% CI &#x2212;0.12 to 0.37; <italic>P</italic>=.31; <italic>d</italic>=0.23). The largest adjusted absolute increase was in organization, and the largest unadjusted standardized effects were for organization (<italic>d</italic>=2.10), comprehensibility (<italic>d</italic>=1.33), and fairness/balance (<italic>d</italic>=1.32). Fairness/balance nevertheless remained the lowest-rated adapted PDQI domain in both groups.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Adapted Physician Documentation Quality Instrument (8-domain) narrative quality domain scores by documentation method (N=150).</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Adapted PDQI<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup> (8-domain) domain<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="bottom">Conventional, mean (SD)<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td><td align="left" valign="bottom">NARRATE<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, mean (SD)<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td><td align="left" valign="bottom">Adjusted mean difference (95% CI)<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Cohen <italic>d</italic><sup><xref ref-type="table-fn" rid="table5fn5">e</xref></sup> (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Thorough</td><td align="left" valign="top">3.61 (1.08)</td><td align="left" valign="top">4.21 (0.85)</td><td align="left" valign="top">0.66 (0.32 to 1.00)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.62 (0.30 to 0.95)</td></tr><tr><td align="left" valign="top">Useful</td><td align="left" valign="top">3.70 (1.09)</td><td align="left" valign="top">4.27 (0.88)</td><td align="left" valign="top">0.61 (0.27 to 0.95)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.57 (0.25 to 0.90)</td></tr><tr><td align="left" valign="top">Organized</td><td align="left" valign="top">3.59 (0.52)</td><td align="left" valign="top">4.65 (0.48)</td><td align="left" valign="top">1.04 (0.86 to 1.23)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">2.10 (1.70 to 2.50)</td></tr><tr><td align="left" valign="top">Comprehensible</td><td align="left" valign="top">4.01 (0.49)</td><td align="left" valign="top">4.57 (0.35)</td><td align="left" valign="top">0.56 (0.41 to 0.72)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">1.33 (0.97 to 1.68)</td></tr><tr><td align="left" valign="top">Succinct</td><td align="left" valign="top">4.31 (0.80)</td><td align="left" valign="top">4.47 (0.57)</td><td align="left" valign="top">0.12 (&#x2212;0.12 to 0.37)</td><td align="left" valign="top">.31</td><td align="left" valign="top">0.23 (&#x2212;0.09 to 0.55)</td></tr><tr><td align="left" valign="top">Synthesized</td><td align="left" valign="top">3.43 (1.14)</td><td align="left" valign="top">3.96 (0.87)</td><td align="left" valign="top">0.54 (0.20 to 0.88)</td><td align="left" valign="top">.002</td><td align="left" valign="top">0.53 (0.20 to 0.85)</td></tr><tr><td align="left" valign="top">Internal consistency</td><td align="left" valign="top">3.47 (0.42)</td><td align="left" valign="top">3.63 (0.34)</td><td align="left" valign="top">0.17 (0.04 to 0.31)</td><td align="left" valign="top">.01</td><td align="left" valign="top">0.43 (0.11 to 0.76)</td></tr><tr><td align="left" valign="top">Fair/balanced</td><td align="left" valign="top">2.79 (0.39)</td><td align="left" valign="top">3.31 (0.41)</td><td align="left" valign="top">0.54 (0.40 to 0.67)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">1.32 (0.97 to 1.68)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>PDQI: Physician Documentation Quality Instrument.</p></fn><fn id="table5fn2"><p><sup>b</sup>Domain-level analyses were exploratory.</p></fn><fn id="table5fn3"><p><sup>c</sup>Mean differences, 95% CIs, and <italic>P</italic> values are from restricted maximum likelihood random-intercept linear mixed-effects models clustered by documenting supervisor.</p></fn><fn id="table5fn4"><p><sup>d</sup>NARRATE: Nursing AI-Refined for Accurate Transcription of Events.</p></fn><fn id="table5fn5"><p><sup>e</sup>Cohen <italic>d</italic> was calculated from unadjusted group means using the pooled SD; 95% CIs were derived from the independent-groups variance estimator.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6"><title>Plain-Text Rerating Sensitivity Analysis</title><p>After reports were re-rendered into a uniform plain-paragraph format and a stratified subsample was rerated, the completeness advantage remained in the cluster-adjusted analysis (adjusted mean difference 1.70, 95% CI 0.27&#x2010;3.14; <italic>P</italic>=.02; unadjusted <italic>d</italic>=0.59, 95% CI 0.08&#x2010;1.11), as did the explanation subscore (adjusted mean difference 0.25, 95% CI 0.05&#x2010;0.45; <italic>P</italic>=.02; <italic>d</italic>=0.63). The actions subscore still favored NARRATE but was borderline (adjusted mean difference 0.16, 95% CI &#x2212;0.0007 to 0.32; <italic>P</italic>=.05; <italic>d</italic>=0.50). The adapted PDQI (8-domain) mean remained higher in NARRATE reports but attenuated to a moderate effect (adjusted mean difference 0.25, 95% CI 0.06&#x2010;0.43; <italic>P</italic>=.009; <italic>d</italic>=0.64, 95% CI 0.12&#x2010;1.16). Among PDQI domains, organization (adjusted mean difference 0.43, 95% CI 0.19&#x2010;0.67; <italic>P</italic>&#x003C;.001; <italic>d</italic>=0.92) and comprehensibility (adjusted mean difference 0.45, 95% CI 0.21&#x2010;0.69; <italic>P</italic>&#x003C;.001; <italic>d</italic>=0.93) remained higher after deformatting, whereas synthesis, internal consistency, and fairness/balance were not statistically significant. These results suggest that the completeness finding and part of the organization and comprehensibility advantage were robust to removal of structural formatting, whereas several other narrative-quality differences were more format-dependent. Cluster-adjusted mean differences, clustered <italic>P</italic> values, and unadjusted effect sizes are summarized in <xref ref-type="table" rid="table6">Table 6</xref>; corresponding full-sample results are reported in <xref ref-type="table" rid="table3">Tables 3</xref><xref ref-type="table" rid="table4"/>-<xref ref-type="table" rid="table5">5</xref>.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Format-blinded plain-text sensitivity-analysis results (n=60)<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup>.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom">Adjusted mean difference (95% CI)</td><td align="left" valign="bottom">Cohen <italic>d</italic><sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">WHO<sup><xref ref-type="table-fn" rid="table6fn4">d</xref></sup> total completeness</td><td align="left" valign="top">1.70 (0.27 to 3.14)</td><td align="left" valign="top">0.59 (0.08 to 1.11)</td><td align="left" valign="top">.02</td></tr><tr><td align="left" valign="top">&#x2003;Description subscore</td><td align="left" valign="top">0.13 (&#x2212;0.004 to 0.26)</td><td align="left" valign="top">0.48 (&#x2212;0.03 to 0.99)</td><td align="left" valign="top">.06</td></tr><tr><td align="left" valign="top">&#x2003;Explanation subscore</td><td align="left" valign="top">0.25 (0.05 to 0.45)</td><td align="left" valign="top">0.63 (0.11 to 1.15)</td><td align="left" valign="top">.02</td></tr><tr><td align="left" valign="top">&#x2003;Actions subscore</td><td align="left" valign="top">0.16 (&#x2212;0.001 to 0.32)</td><td align="left" valign="top">0.50 (&#x2212;0.01 to 1.01)</td><td align="left" valign="top">.05</td></tr><tr><td align="left" valign="top">Adapted PDQI<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup> (8-domain) mean</td><td align="left" valign="top">0.25 (0.06 to 0.43)</td><td align="left" valign="top">0.64 (0.12 to 1.16)</td><td align="left" valign="top">.009</td></tr><tr><td align="left" valign="top">&#x2003;Thorough</td><td align="left" valign="top">0.26 (&#x2212;0.13 to 0.65)</td><td align="left" valign="top">0.30 (&#x2212;0.21 to 0.81)</td><td align="left" valign="top">.19</td></tr><tr><td align="left" valign="top">&#x2003;Useful</td><td align="left" valign="top">0.26 (&#x2212;0.13 to 0.65)</td><td align="left" valign="top">0.30 (&#x2212;0.21 to 0.81)</td><td align="left" valign="top">.19</td></tr><tr><td align="left" valign="top">&#x2003;Organized</td><td align="left" valign="top">0.43 (0.19 to 0.67)</td><td align="left" valign="top">0.92 (0.39 to 1.45)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">&#x2003;Comprehensible</td><td align="left" valign="top">0.45 (0.21 to 0.69)</td><td align="left" valign="top">0.93 (0.40 to 1.46)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">&#x2003;Succinct</td><td align="left" valign="top">0.13 (&#x2212;0.16 to 0.41)</td><td align="left" valign="top">0.18 (&#x2212;0.32 to 0.69)</td><td align="left" valign="top">.38</td></tr><tr><td align="left" valign="top">&#x2003;Synthesized</td><td align="left" valign="top">0.21 (&#x2212;0.03 to 0.45)</td><td align="left" valign="top">0.39 (&#x2212;0.12 to 0.90)</td><td align="left" valign="top">.09</td></tr><tr><td align="left" valign="top">&#x2003;Internal consistency</td><td align="left" valign="top">0.07 (&#x2212;0.14 to 0.29)</td><td align="left" valign="top">0.20 (&#x2212;0.30 to 0.71)</td><td align="left" valign="top">.51</td></tr><tr><td align="left" valign="top">&#x2003;Fair/balanced</td><td align="left" valign="top">0.13 (&#x2212;0.09 to 0.36)</td><td align="left" valign="top">0.30 (&#x2212;0.21 to 0.81)</td><td align="left" valign="top">.25</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>Positive values favor NARRATE. Corresponding full-sample results are reported in <xref ref-type="table" rid="table3">Tables 3</xref><xref ref-type="table" rid="table4"/>-<xref ref-type="table" rid="table5">5</xref>.</p></fn><fn id="table6fn2"><p><sup>b</sup>Cohen <italic>d</italic> was calculated from unadjusted group means using the pooled SD; 95% CIs were derived from the independent-groups variance estimator.</p></fn><fn id="table6fn3"><p><sup>c</sup><italic>P</italic> values are from restricted maximum likelihood random-intercept linear mixed-effects models clustered by documenting supervisor.</p></fn><fn id="table6fn4"><p><sup>d</sup>WHO: World Health Organization.</p></fn><fn id="table6fn5"><p><sup>e</sup>PDQI: Physician Documentation Quality Instrument.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this retrospective pre-post comparative document-quality study of 150 completed supervisor investigation reports, NARRATE-period reports were more complete and had higher adapted PDQI (8-domain) mean scores than conventional reports. The plain-paragraph sensitivity analysis refined that picture: the completeness advantage and explanation-domain gain persisted after structural formatting was removed, the actions-domain gain attenuated, and organization and comprehensibility remained higher but with smaller effects. By contrast, synthesis, internal consistency, and fairness/balance no longer differed significantly. The most defensible interpretation is therefore improved completeness, explanation content, and some aspects of narrative organization and clarity, with several other narrative-quality differences partly mediated by visible format.</p><p>Description was already well documented in conventional reports, which likely limited room for improvement. By contrast, explanation and action content require supervisors to organize findings, identify contributing factors, and specify follow-up, making them more plausible targets for structured prompting [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>].</p></sec><sec id="s4-2"><title>Interpretation of WHO Completeness Findings</title><p>The completeness findings support prompt-guided documentation aligned to the advanced WHO MIM PS. NARRATE did not appear merely to lengthen or polish reports; it improved the explanatory and action-oriented elements most relevant to review and learning [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>This matters because reports that omit causes or contributing factors are weaker starting points for investigation and root-cause analysis [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. By prompting these elements and making omissions visible through missing-information cues, NARRATE may help turn routine supervisor reports into more usable review documents.</p><p>The actions-domain gain should still be interpreted cautiously. The effect was modest in the full sample and only borderline in the plain-text analysis, suggesting improved documentation of actions more than proof of stronger action planning itself. Future iterations could prompt more explicitly for system-level actions, accountability, and timelines.</p></sec><sec id="s4-3"><title>Interpretation of Adapted PDQI (8-Domain) Findings</title><p>The adapted PDQI findings require domain-specific interpretation rather than a single claim of globally improved narrative quality [<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>In the full sample, organization, comprehensibility, and fairness/balance showed the largest effects, consistent with the visible SBAR format, which may reduce reviewer cognitive load [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. The plain-paragraph sensitivity analysis showed that this pattern was only partly format-driven: organization and comprehensibility remained significantly higher after SBAR labels and structural cues were removed, although both attenuated substantially. By contrast, synthesis, internal consistency, and fairness/balance were no longer significant, and the adapted PDQI (8-domain) mean decreased from a large to a moderate effect.</p><p>The narrative-quality findings therefore appear to reflect a mixed mechanism. Some gains, particularly in organization and comprehensibility, seem to extend beyond recognition of the formatted template, whereas a substantial portion of the synthesis, internal consistency, and fairness/balance advantage is likely format-mediated. This does not make structure unimportant in practice; structured presentation may still improve readability and navigation even if it should not be interpreted here as an independent content gain.</p><p>Succinctness was the only PDQI domain without a significant full-sample gain, which is plausible because SBAR summaries and completion prompts can lengthen reports. Even so, succinctness remained high in the NARRATE period, suggesting that improved completeness did not come at the cost of markedly poorer brevity.</p><p>Fairness/balance had the lowest mean PDQI (8-domain) domain score in both groups and became non-significant after deformatting. This suggests that fairness in incident reporting depends less on formatting than on investigation depth, avoidance of blame, and adequate attention to system factors [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. The lower variance of adapted PDQI (8-domain) mean scores in the NARRATE group (SD 0.34 vs 0.52) is also a mixed finding: more standardized documentation may improve consistency, but it may also compress idiosyncratic detail that remains useful for review.</p></sec><sec id="s4-4"><title>Interrater Reliability and Measurement Implications</title><p>Interrater reliability was somewhat stronger for adapted PDQI (8-domain) mean than for WHO total completeness, but both composites were within an acceptable range for group-level comparisons. Larger WHO disagreements clustered in the explanation items, which require more interpretive judgment than the more factual description items. These findings support continued use of detailed anchors and suggest that future studies could consider adjudication or consensus scoring for borderline causal items.</p></sec><sec id="s4-5"><title>Comparison With Prior Work</title><p>These findings extend prior concerns that incomplete incident reports limit patient safety learning [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Rather than evaluating a generic transcription tool, this study examined a workflow that combined voice-to-text capture, WHO MIM PS-aligned prompts, visible omission cues, and SBAR output. The results suggest that prompt-guided narration may improve the capture of learning-relevant elements, although they remain documentation-quality outcomes rather than evidence of reduced harm.</p></sec><sec id="s4-6"><title>Implications for Nursing Practice, Leadership, and Safety Governance</title><p>For nursing leaders, the practical implication is that AI-enabled documentation may be more valuable as a way to improve the quality of incident-review inputs than as a pure efficiency tool [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. When designed to improve the structure, completeness, and usability of investigation narratives, workflows like NARRATE may help supervisors produce reports that are easier to review and more useful for learning.</p><p>Implementation should nevertheless preserve professional accountability. Supervisors must verify AI-generated content, correct transcription errors, resolve missing-information placeholders, and ensure that recommendations are clinically and operationally appropriate. Governance should also address confidentiality, responsible AI use, prompt versioning, audit trails, user training, and equity of use across roles and settings [<xref ref-type="bibr" rid="ref29">29</xref>]. Because accuracy was not assessed, these findings indicate more complete and better-organized reports, not more factually correct ones. No draft-to-final accuracy audit was performed, and accuracy assurance in this workflow rests on mandatory supervisor verification of every draft. Ambient-scribe transcription errors can propagate to the final note, sometimes with potential for moderate-to-severe harm [<xref ref-type="bibr" rid="ref17">17</xref>]. Completeness and organization gains therefore do not substitute for accuracy assurance and should not be taken as evidence that AI-generated incident narratives can be relied upon without source-data verification in higher-stakes investigations.</p></sec><sec id="s4-7"><title>Strengths</title><p>This study has several strengths: a guidance-informed completeness framework aligned to WHO reporting principles, expert content-validity review of the completeness checklist, a 150-report analytic sample, blinded dual-reviewer scoring with quantified interrater reliability, and parallel examination of both total scores and domain-specific patterns.</p></sec><sec id="s4-8"><title>Limitations</title><p>This study has several limitations. The retrospective pre-post design cannot establish causality, and secular changes cannot be entirely excluded. No additional reporting-quality training or audit-feedback initiative was conducted in the interval, and the institutional incident-reporting system was unchanged; the postimplementation window represented a stable postlaunch period with a fixed prompt template.</p><p>NARRATE use was voluntary, and recorded use represented approximately 40% of eligible postimplementation reports. Because the operational record was maintained for another purpose and was not exhaustive, a reliably classified postimplementation non-NARRATE group could not be constructed: absence from the record could not confirm nonuse, and a complement-defined group could have included unrecorded NARRATE reports. The NARRATE group therefore reflects self-selected early adopters, who may differ from nonadopters in motivation, experience, or documentation practice; observed differences may consequently overestimate effects under mandated or universal adoption.</p><p>Although the mixed models accounted for correlation among multiple reports from the same supervisor, no supervisor contributed reports in both periods. Documentation method was therefore completely nested within supervisor pool, and clustering adjustment cannot eliminate confounding by different supervisor composition. The comparison cannot fully separate workflow effects from adopter or supervisor characteristics and should be interpreted as performance under real-world voluntary uptake rather than as an unbiased causal contrast. Residual unblinding was possible because NARRATE reports could retain recognizable SBAR structure; the plain-text analysis suggests that some narrative-quality effects were format-sensitive. Because we assessed final supervisor-edited text, we cannot separate the independent effects of transcription, prompting, missing-information cues, and human editing; effects should be attributed to the workflow as a whole.</p><p>The prompt template and completeness checklist were both aligned with the advanced WHO MIM PS, creating intervention-outcome alignment. Because the checklist was finalized after the NARRATE prompt template was locked, this criterion-contamination risk is directional: the checklist could in principle have been shaped by what the prompt already produced. To limit this risk, the content-validity panel evaluated items against WHO MIM PS source guidance rather than the specific NARRATE prompt wording, the checklist graded sufficiency rather than simple field completion, and the adapted PDQI findings partly persisted after deformatting. Nonetheless, confirmatory work using a completeness instrument developed independently of the NARRATE prompts is needed.</p><p>The adapted PDQI underwent content adaptation but not full construct validation, and the study assessed documentation quality rather than action quality, timeliness, recurrence, factual accuracy, or patient-safety outcomes. No draft-to-final accuracy audit was performed; accuracy assurance rested on mandatory supervisor verification of every draft. Granular case-mix variables were unavailable, and the study was conducted at a single Singapore tertiary center, limiting causal inference and generalizability.</p></sec><sec id="s4-9"><title>Future Work</title><p>Future work should test whether improved completeness translates into stronger action planning, faster review, better aggregation of contributing factors, and measurable organizational learning. Confirmatory studies should include concurrent, reliably classified nonadopter comparison groups or phased implementation designs that permit within-supervisor comparisons, helping distinguish workflow effects from adopter and supervisor characteristics. Larger multicenter studies could assess implementation fidelity, adoption, and equity across roles and settings. Prospective studies should retain auditable source material so that documentation quality and draft-to-final accuracy can be evaluated together.</p></sec><sec id="s4-10"><title>Conclusions</title><p>Project NARRATE is a nursing-led application of ambient AI documentation support to incident-investigation reporting. Among voluntary early adopters in this pre-post study, NARRATE-supported reports were associated with greater completeness and higher adapted PDQI (8-domain) mean scores after accounting for repeated reports by supervisors, and completeness gains persisted after structural formatting was removed. The findings do not establish that NARRATE alone caused these differences because adopter self-selection and nonoverlapping supervisor pools remain potential explanations. The strongest evidence supports gains in completeness, explanation content, and selected aspects of narrative organization and clarity, whereas several other narrative-quality differences were format-sensitive. Broader prospective evaluation with concurrent controls is required, with AI-generated documentation remaining subject to human verification and governance.</p></sec></sec></body><back><ack><p>The authors acknowledge the nursing supervisors and frontline nurses who contributed to the implementation of Project NARRATE (Nursing AI-Refined for Accurate Transcription of Events) and provided feedback that informed iterative workflow refinement. The authors also thank the expert panel members from Nursing Safety and Quality, the Office of Patient Safety and Quality, and the Medication Safety Committee for their review of the incident report completeness checklist. The authors further acknowledge the support of the SingHealth Digital Empowerment Office, SGH Artificial Intelligence and Automation, and the SingHealth Office of Insights and Analytics in enabling the broader implementation and evaluation of the NARRATE workflow.</p><p>The authors used generative AI tools for language editing, organization, formatting support, and manuscript-review suggestions during manuscript preparation. ChatGPT (GPT-5.6 Thinking; OpenAI) supported drafting, editing, and formatting; Claude (Sonnet 5; Anthropic) supported independent review suggestions. The authors reviewed, revised, and verified all scientific content and take full responsibility for the final manuscript.</p></ack><notes><sec><title>Funding</title><p>This work received no external funding.</p></sec><sec><title>Data Availability</title><p>The data analyzed in this study were derived from deidentified patient safety incident-investigation reports and are not publicly available because of confidentiality and patient safety governance restrictions. Aggregated, nonidentifiable data may be made available from the corresponding author on reasonable request and subject to institutional approval.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: KYT</p><p>Data interpretation: LH</p><p>Intervention development: KCYW</p><p>Intervention implementation: LH</p><p>Investigation: KYT</p><p>Methodology: KYT, KCYW, LH</p><p>Project administration: KYT, KCYW</p><p>Project governance: SYA, JGNN</p><p>Supervision: KYT, SYA, JGNN</p><p>Writing &#x2013; original draft: KYT</p><p>Writing &#x2013; review &#x0026; editing: KYT, KCYW, LH, SYA, JGNN</p><p>All authors reviewed and approved the final manuscript and agree to be accountable for all aspects of the work.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb2">MIM PS</term><def><p>Minimal Information Model for Patient Safety Incident Reporting and Learning Systems</p></def></def-item><def-item><term id="abb3">NARRATE</term><def><p>Nursing AI-Refined for Accurate Transcription of Events</p></def></def-item><def-item><term id="abb4">PDQI</term><def><p>Physician Documentation Quality Instrument</p></def></def-item><def-item><term id="abb5">SBAR</term><def><p>Situation-Background-Assessment-Recommendation</p></def></def-item><def-item><term id="abb6">SQUIRE</term><def><p>Standards for Quality Improvement Reporting Excellence</p></def></def-item><def-item><term id="abb7">STROBE</term><def><p>Strengthening the Reporting of Observational Studies in Epidemiology</p></def></def-item><def-item><term id="abb8">WHO</term><def><p>World Health Organization</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Griffeth</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Gajic</surname><given-names>O</given-names> </name><name name-style="western"><surname>Schueler</surname><given-names>N</given-names> </name><name name-style="western"><surname>Todd</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ramar</surname><given-names>K</given-names> </name></person-group><article-title>Multifaceted intervention to improve patient safety incident reporting in intensive care units</article-title><source>J Patient Saf</source><year>2023</year><month>10</month><day>1</day><volume>19</volume><issue>7</issue><fpage>422</fpage><lpage>428</lpage><pub-id pub-id-type="doi">10.1097/PTS.0000000000001151</pub-id><pub-id pub-id-type="medline">37466643</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="report"><article-title>Patient safety incident reporting and learning systems: technical report and guidance</article-title><year>2020</year><access-date>2026-07-25</access-date><publisher-name>World Health Organization</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://iris.who.int/server/api/core/bitstreams/49947589-0a06-4708-8cc9-379bc8d2f9f2/content">https://iris.who.int/server/api/core/bitstreams/49947589-0a06-4708-8cc9-379bc8d2f9f2/content</ext-link></comment></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="report"><article-title>Minimal information model for patient safety incident reporting and learning systems: user guide</article-title><year>2016</year><access-date>2026-07-25</access-date><publisher-name>World Health Organization</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://iris.who.int/server/api/core/bitstreams/a13b925e-fe8e-49d3-a31e-7e7a7d88e646/content">https://iris.who.int/server/api/core/bitstreams/a13b925e-fe8e-49d3-a31e-7e7a7d88e646/content</ext-link></comment></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gong</surname><given-names>Y</given-names> </name></person-group><article-title>Toward reporting support and quality assessment for learning from reporting: a necessary data elements model for narrative medication error reports</article-title><source>AMIA Annu Symp Proc</source><year>2018</year><volume>2018</volume><fpage>1581</fpage><lpage>1590</lpage><pub-id pub-id-type="medline">30815204</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scott</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dawson</surname><given-names>P</given-names> </name><name name-style="western"><surname>Heavey</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Content analysis of patient safety incident reports for older adult patient transfers, handovers, and discharges: do they serve organizations, staff, or patients?</article-title><source>J Patient Saf</source><year>2021</year><month>12</month><day>1</day><volume>17</volume><issue>8</issue><fpage>e1744</fpage><lpage>e1758</lpage><pub-id pub-id-type="doi">10.1097/PTS.0000000000000654</pub-id><pub-id pub-id-type="medline">31790011</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stavropoulou</surname><given-names>C</given-names> </name><name name-style="western"><surname>Doherty</surname><given-names>C</given-names> </name><name name-style="western"><surname>Tosey</surname><given-names>P</given-names> </name></person-group><article-title>How effective are incident&#x2010;reporting systems for improving patient safety? A systematic literature review</article-title><source>Milbank Q</source><year>2015</year><month>12</month><volume>93</volume><issue>4</issue><fpage>826</fpage><lpage>866</lpage><pub-id pub-id-type="doi">10.1111/1468-0009.12166</pub-id><pub-id pub-id-type="medline">26626987</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mitchell</surname><given-names>I</given-names> </name><name name-style="western"><surname>Schuster</surname><given-names>A</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>K</given-names> </name><name name-style="western"><surname>Pronovost</surname><given-names>P</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>A</given-names> </name></person-group><article-title>Patient safety incident reporting: a qualitative study of thoughts and perceptions of experts 15&#x2005;years after &#x201C;To Err is Human&#x201D;</article-title><source>BMJ Qual Saf</source><year>2016</year><month>02</month><volume>25</volume><issue>2</issue><fpage>92</fpage><lpage>99</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2015-004405</pub-id><pub-id pub-id-type="medline">26217037</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Macrae</surname><given-names>C</given-names> </name></person-group><article-title>The problem with incident reporting</article-title><source>BMJ Qual Saf</source><year>2016</year><month>02</month><volume>25</volume><issue>2</issue><fpage>71</fpage><lpage>75</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2015-004732</pub-id><pub-id pub-id-type="medline">26347519</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ong</surname><given-names>JHW</given-names> </name><name name-style="western"><surname>Tung</surname><given-names>JYM</given-names> </name><name name-style="western"><surname>Sng</surname><given-names>GGR</given-names> </name><etal/></person-group><article-title>A pilot study using ambient artificial intelligence scribes in clinical documentation in a urology outpatient clinic</article-title><source>BJU Int</source><year>2025</year><month>05</month><day>20</day><volume>136</volume><issue>3</issue><fpage>415</fpage><lpage>417</lpage><pub-id pub-id-type="doi">10.1111/bju.16784</pub-id><pub-id pub-id-type="medline">40390678</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>JYE</given-names> </name><name name-style="western"><surname>Rafi</surname><given-names>IBM</given-names> </name><name name-style="western"><surname>Sng</surname><given-names>GGR</given-names> </name><etal/></person-group><article-title>Impact of an ambient AI scribe among clinicians and patients: real-world prospective observational time-motion study</article-title><source>JMIR Med Inform</source><year>2026</year><month>03</month><day>31</day><volume>14</volume><fpage>e85580</fpage><pub-id pub-id-type="doi">10.2196/85580</pub-id><pub-id pub-id-type="medline">41915701</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lea</surname><given-names>W</given-names> </name><name name-style="western"><surname>Budworth</surname><given-names>L</given-names> </name><name name-style="western"><surname>O&#x2019;Hara</surname><given-names>J</given-names> </name><name name-style="western"><surname>Vincent</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lawton</surname><given-names>R</given-names> </name></person-group><article-title>Investigators are human too: outcome bias and perceptions of individual culpability in patient safety incident investigations</article-title><source>BMJ Qual Saf</source><year>2026</year><month>02</month><day>19</day><volume>35</volume><issue>3</issue><fpage>159</fpage><lpage>168</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2024-017926</pub-id><pub-id pub-id-type="medline">39929715</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stetson</surname><given-names>PD</given-names> </name><name name-style="western"><surname>Bakken</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wrenn</surname><given-names>JO</given-names> </name><name name-style="western"><surname>Siegler</surname><given-names>EL</given-names> </name></person-group><article-title>Assessing electronic note quality using the Physician Documentation Quality Instrument (PDQI-9)</article-title><source>Appl Clin Inform</source><year>2012</year><volume>3</volume><issue>2</issue><fpage>164</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.4338/aci-2011-11-ra-0070</pub-id><pub-id pub-id-type="medline">22577483</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haig</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Sutton</surname><given-names>S</given-names> </name><name name-style="western"><surname>Whittington</surname><given-names>J</given-names> </name></person-group><article-title>SBAR: a shared mental model for improving communication between clinicians</article-title><source>Jt Comm J Qual Patient Saf</source><year>2006</year><month>03</month><volume>32</volume><issue>3</issue><fpage>167</fpage><lpage>175</lpage><pub-id pub-id-type="doi">10.1016/s1553-7250(06)32022-3</pub-id><pub-id pub-id-type="medline">16617948</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Faul</surname><given-names>F</given-names> </name><name name-style="western"><surname>Erdfelder</surname><given-names>E</given-names> </name><name name-style="western"><surname>Buchner</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lang</surname><given-names>AG</given-names> </name></person-group><article-title>Statistical power analyses using G*Power 3.1: tests for correlation and regression analyses</article-title><source>Behav Res Methods</source><year>2009</year><month>11</month><volume>41</volume><issue>4</issue><fpage>1149</fpage><lpage>1160</lpage><pub-id pub-id-type="doi">10.3758/BRM.41.4.1149</pub-id><pub-id pub-id-type="medline">19897823</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lynn</surname><given-names>MR</given-names> </name></person-group><article-title>Determination and quantification of content validity</article-title><source>Nurs Res</source><year>1986</year><volume>35</volume><issue>6</issue><fpage>382</fpage><lpage>385</lpage><pub-id pub-id-type="medline">3640358</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Polit</surname><given-names>DF</given-names> </name><name name-style="western"><surname>Beck</surname><given-names>CT</given-names> </name><name name-style="western"><surname>Owen</surname><given-names>SV</given-names> </name></person-group><article-title>Is the CVI an acceptable indicator of content validity? Appraisal and recommendations</article-title><source>Res Nurs Health</source><year>2007</year><month>08</month><volume>30</volume><issue>4</issue><fpage>459</fpage><lpage>467</lpage><pub-id pub-id-type="doi">10.1002/nur.20199</pub-id><pub-id pub-id-type="medline">17654487</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Anderson</surname><given-names>TN</given-names> </name><name name-style="western"><surname>Mohan</surname><given-names>V</given-names> </name><name name-style="western"><surname>Dorr</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Ratwani</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Biro</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Gold</surname><given-names>JA</given-names> </name></person-group><article-title>Evaluating the quality and safety of ambient digital scribe platforms using simulated ambulatory encounters</article-title><source>Mayo Clin Proc Digit Health</source><year>2025</year><month>12</month><volume>3</volume><issue>4</issue><fpage>100292</fpage><pub-id pub-id-type="doi">10.1016/j.mcpdig.2025.100292</pub-id><pub-id pub-id-type="medline">41234546</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McGraw</surname><given-names>KO</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>SP</given-names> </name></person-group><article-title>Forming inferences about some intraclass correlation coefficients</article-title><source>Psychol Methods</source><year>1996</year><volume>1</volume><issue>1</issue><fpage>30</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1037/1082-989X.1.1.30</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><source>Statistical Power Analysis for the Behavioral Sciences</source><year>1988</year><edition>2</edition><publisher-name>Lawrence Erlbaum Associates</publisher-name><pub-id pub-id-type="doi">10.4324/9780203771587</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Hedges</surname><given-names>LV</given-names> </name><name name-style="western"><surname>Olkin</surname><given-names>I</given-names> </name></person-group><source>Statistical Methods for Meta-Analysis</source><year>1985</year><publisher-name>Academic Press</publisher-name><pub-id pub-id-type="other">9780123363800</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peerally</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Carr</surname><given-names>S</given-names> </name><name name-style="western"><surname>Waring</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dixon-Woods</surname><given-names>M</given-names> </name></person-group><article-title>The problem with root cause analysis</article-title><source>BMJ Qual Saf</source><year>2017</year><month>05</month><volume>26</volume><issue>5</issue><fpage>417</fpage><lpage>422</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2016-005511</pub-id><pub-id pub-id-type="medline">27340202</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hibbert</surname><given-names>PD</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>MJW</given-names> </name><name name-style="western"><surname>Deakin</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Are root cause analyses recommendations effective and sustainable? An observational study</article-title><source>Int J Qual Health Care</source><year>2018</year><month>03</month><day>1</day><volume>30</volume><issue>2</issue><fpage>124</fpage><lpage>131</lpage><pub-id pub-id-type="doi">10.1093/intqhc/mzx181</pub-id><pub-id pub-id-type="medline">29346587</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Gong</surname><given-names>Y</given-names> </name></person-group><article-title>Data quality assessment of narrative medication error reports</article-title><source>Stud Health Technol Inform</source><year>2019</year><month>08</month><day>9</day><volume>265</volume><fpage>101</fpage><lpage>106</lpage><pub-id pub-id-type="doi">10.3233/SHTI190146</pub-id><pub-id pub-id-type="medline">31431584</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sweller</surname><given-names>J</given-names> </name></person-group><article-title>Cognitive load during problem solving: effects on learning</article-title><source>Cogn Sci</source><year>1988</year><month>04</month><volume>12</volume><issue>2</issue><fpage>257</fpage><lpage>285</lpage><pub-id pub-id-type="doi">10.1207/s15516709cog1202_4</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paas</surname><given-names>F</given-names> </name><name name-style="western"><surname>Renkl</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sweller</surname><given-names>J</given-names> </name></person-group><article-title>Cognitive load theory and instructional design: recent developments</article-title><source>Educ Psychol</source><year>2003</year><month>01</month><day>1</day><volume>38</volume><issue>1</issue><fpage>1</fpage><lpage>4</lpage><pub-id pub-id-type="doi">10.1207/S15326985EP3801_1</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boysen</surname><given-names>PG</given-names> </name></person-group><article-title>Just culture: a foundation for balanced accountability and patient safety</article-title><source>Ochsner J</source><year>2013</year><volume>13</volume><issue>3</issue><fpage>400</fpage><lpage>406</lpage><pub-id pub-id-type="medline">24052772</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cooper</surname><given-names>J</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>A</given-names> </name><name name-style="western"><surname>Williams</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Nature of blame in patient safety incident reports: mixed methods analysis of a national database</article-title><source>Ann Fam Med</source><year>2017</year><month>09</month><volume>15</volume><issue>5</issue><fpage>455</fpage><lpage>461</lpage><pub-id pub-id-type="doi">10.1370/afm.2123</pub-id><pub-id pub-id-type="medline">28893816</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Perkins</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Muste</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Alam</surname><given-names>T</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>RP</given-names> </name></person-group><article-title>Improving clinical documentation with artificial intelligence: a systematic review</article-title><source>Perspect Health Inf Manag</source><year>2024</year><volume>21</volume><issue>2</issue><fpage>1d</fpage><pub-id pub-id-type="medline">40134899</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="web"><article-title>Guidance on the use of AI-enabled ambient scribing products in health and care settings</article-title><source>NHS England</source><year>2025</year><access-date>2026-07-18</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.england.nhs.uk/long-read/guidance-on-the-use-of-ai-enabled-ambient-scribing-products-in-health-and-care-settings/">https://www.england.nhs.uk/long-read/guidance-on-the-use-of-ai-enabled-ambient-scribing-products-in-health-and-care-settings/</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Advanced WHO Minimal Information Model for Patient Safety Incident Reporting and Learning Systems&#x2013;aligned graded completeness checklist, adapted from WHO technical guidance and the MIM PS User Guide [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>].</p><media xlink:href="nursing_v9i1e100775_app1.docx" xlink:title="DOCX File, 43 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Eight-domain narrative-quality rubric adapted from the Physician Documentation Quality Instrument-9 described by Stetson et al [<xref ref-type="bibr" rid="ref12">12</xref>].</p><media xlink:href="nursing_v9i1e100775_app2.docx" xlink:title="DOCX File, 44 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 1</label><p>SQUIRE 2.0 checklist.</p><media xlink:href="nursing_v9i1e100775_app3.docx" xlink:title="DOCX File, 30 KB"/></supplementary-material></app-group></back></article>