<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Nursing</journal-id><journal-id journal-id-type="publisher-id">nursing</journal-id><journal-id journal-id-type="index">33</journal-id><journal-title>JMIR Nursing</journal-title><abbrev-journal-title>JMIR Nursing</abbrev-journal-title><issn pub-type="epub">2562-7600</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v9i1e93638</article-id><article-id pub-id-type="doi">10.2196/93638</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Nursing Process Data for Health Care Cost Prediction Using Machine Learning: Longitudinal Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Company-Sancho</surname><given-names>Mar&#x00ED;a Consuelo</given-names></name><degrees>RN, MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Gonz&#x00E1;lez-Chord&#x00E1;</surname><given-names>V&#x00ED;ctor M</given-names></name><degrees>RN, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Orts-Cort&#x00E9;s</surname><given-names>Mar&#x00ED;a Isabel</given-names></name><degrees>RN, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib></contrib-group><aff id="aff1"><institution>Health Promotion Service, Directorate General for Public Health, Canary Islands Health Service</institution><addr-line>Las Palmas de Gran Canaria</addr-line><addr-line>Canary Islands</addr-line><country>Spain</country></aff><aff id="aff2"><institution>Nursing and Healthcare Research Unit (Invest&#x00E9;n-isciii), Research Network on Chronicity, Primary Care and Health Promotion (RICAPPS), Carlos III Health Institute</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff3"><institution>Nursing and Healthcare Research Unit (Invest&#x00E9;n-isciii), CIBER of Frailty and Healthy Ageing (CIBERFES), Carlos III Health Institute</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff4"><institution>Nursing Research Group GIENF-241, Nursing Department, Universitat Jaume I</institution><addr-line>Castell&#x00F3;n de la Plana</addr-line><addr-line>Castell&#x00F3;n</addr-line><country>Spain</country></aff><aff id="aff5"><institution>Department of Nursing (BALMIS), Institute for Health and Biomedical Research (ISABIAL, Group 23), University of Alicante</institution><addr-line>Alicante</addr-line><country>Spain</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Borycki</surname><given-names>Elizabeth</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Rashidul Hasan</surname><given-names>S M</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Chai</surname><given-names>Soo See</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to V&#x00ED;ctor M Gonz&#x00E1;lez-Chord&#x00E1;, RN, PhD, Nursing Research Group GIENF-241, Nursing Department, Universitat Jaume I, Castell&#x00F3;n de la PlanaCastell&#x00F3;n, 12071, Spain, 34 964 387744; <email>victor.gonzalez@uji.es</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>all authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>4</day><month>9</month><year>2026</year></pub-date><volume>9</volume><elocation-id>e93638</elocation-id><history><date date-type="received"><day>16</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>11</day><month>05</month><year>2026</year></date><date date-type="accepted"><day>17</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Mar&#x00ED;a Consuelo Company-Sancho, V&#x00ED;ctor M Gonz&#x00E1;lez-Chord&#x00E1;, Mar&#x00ED;a Isabel Orts-Cort&#x00E9;s. Originally published in JMIR Nursing (<ext-link ext-link-type="uri" xlink:href="https://nursing.jmir.org">https://nursing.jmir.org</ext-link>), 4.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Nursing, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://nursing.jmir.org/">https://nursing.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://nursing.jmir.org/2026/1/e93638"/><abstract><sec><title>Background</title><p>Machine learning (ML) has been demonstrated to enhance health care cost prediction by handling high-dimensional data and identifying complex patterns. However, current risk-adjustment models rarely incorporate structured nursing information derived from the nursing process. This information captures care needs and human responses to health problems.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the impact of integrating nursing process data into ML-based predictive models of individual health care costs, including cost component analyses, compared with models based solely on sociodemographic, clinical, and morbidity-related variables.</p></sec><sec sec-type="methods"><title>Methods</title><p>A retrospective observational study was conducted using a population-based cohort of 1,691,075 individuals aged 15 years or younger who were registered with the Canary Islands Health Service. Predictors were derived from data available up to 2017 and included sociodemographic and clinical variables, Adjusted Morbidity Groups, health care use, and structured nursing records (Functional Health Patterns [FHP], North American Nursing Diagnosis Association [NANDA], Nursing Outcomes Classification [NOC], and Nursing Interventions Classification [NIC]). Predictive models were developed using feedforward neural networks and extreme gradient boosting; predictions were combined using an ensemble approach. An autoencoder was applied as a dimensionality-reduction technique for the nursing variables. Model performance with and without nursing variables was compared on total cost and individual cost components, and the coefficient of determination (<italic>R</italic>&#x00B2;) was used on the test set.</p></sec><sec sec-type="results"><title>Results</title><p>Including the nursing methodology yielded small numerical increases in predictive performance. With respect to total cost, the ensemble model improved the <italic>R</italic>&#x00B2; from 0.5023 to 0.5058 when the nursing variables were added. Although directionally consistent, these gains were limited in magnitude. In the component-level analyses, performance gains were observed in hospital care (<italic>R</italic>&#x00B2;=0.2396) and pharmaceutical costs (<italic>R</italic>&#x00B2;=0.6631). Reducing 789 nursing variables to 16 latent dimensions using an autoencoder substantially simplified the feature space, with predictive performance remaining broadly comparable but without a substantial additional gain.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Integrating structured information from the nursing process is associated with small incremental improvements in ML-based predictive models and complements commonly used sociodemographic, clinical, and morbidity variables. The systematic incorporation of nursing data into predictive tools may contribute to more accurate health care cost prediction and support more holistic, person-centered approaches.</p></sec></abstract><kwd-group><kwd>standardized nursing terminology</kwd><kwd>nursing process</kwd><kwd>nursing diagnosis</kwd><kwd>machine learning</kwd><kwd>predictive modeling</kwd><kwd>health care costs</kwd><kwd>risk adjustment</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Efficient allocation of resources in health care systems requires accurate and reliable estimates of future expenditure, particularly in a context characterized by budgetary constraints and increasing care complexity. Inaccurate cost estimation may lead to inefficient decision-making and compromise the financial sustainability of health services [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. In this context, machine learning (ML) has gained relevance because it provides tools capable of identifying complex patterns, handling missing values, and improving predictive accuracy compared with traditional statistical methods [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>Within the field of cost prediction, risk-adjustment systems such as Adjusted Clinical Groups (ACGs) [<xref ref-type="bibr" rid="ref10">10</xref>], Clinical Risk Groups (CRGs) [<xref ref-type="bibr" rid="ref11">11</xref>], and Adjusted Morbidity Groups (AMGs) [<xref ref-type="bibr" rid="ref12">12</xref>] play key roles in population risk stratification by anticipating care needs and budgetary requirements. These models have been widely used to estimate costs associated with different patient profiles, and their performance is typically assessed using metrics such as the coefficient of determination (<italic>R</italic>&#x00B2;) [<xref ref-type="bibr" rid="ref13">13</xref>]. In Spain, the AMG risk-adjustment system explains approximately 31% of the variability in total health care costs, suggesting substantial room for improvement [<xref ref-type="bibr" rid="ref14">14</xref>]. In response to these limitations, ML techniques have increasingly been applied to risk-adjustment frameworks because of their greater capacity to identify complex patterns in large datasets. The use of neural networks, boosting models, and penalized regression approaches has been shown to improve predictive accuracy [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Methods such as neural networks and ridge regression have shown good performance in cost estimation using historical data [<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Moreover, the types of data incorporated into risk-adjustment models have evolved. Electronic health records, which capture sociodemographic data and multiple dimensions of health status, are a rich source of information. Supervised learning models are particularly well suited for analyzing these data, as they can detect latent interactions and manage nonnormally distributed variables, such as health care costs. Leveraging these data allows more accurate estimation of health care resource use and identification of relevant patterns of association [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>In this context, Bertsimas et al [<xref ref-type="bibr" rid="ref17">17</xref>] reported that the accuracy of cost prediction models depends largely on the quality, completeness, and standardization of the data used and suggested that incorporating new variables that more faithfully reflect actual care delivery may enhance predictive performance. However, most predictive models, including those based on variables commonly used in ACG, CRG, and AMG risk-adjustment systems, rely primarily on sociodemographic and clinical variables such as gender, age, comorbidities, procedures, and pharmaceutical expenditures [<xref ref-type="bibr" rid="ref11">11</xref>]. The inclusion of social determinants has been shown to significantly improve the predictive capacity of these models. For example, a recent study reported that incorporating such factors increased the <italic>R</italic>&#x00B2; from 0.327 to 0.388, reducing estimation errors by 3.5 million dollars per 10,000 individuals added [<xref ref-type="bibr" rid="ref18">18</xref>]. Similarly, factors such as low income, educational level, food insecurity, or housing instability have been significantly associated with high-cost patient profiles [<xref ref-type="bibr" rid="ref19">19</xref>], although these variables are rarely included in cost prediction models. Harnessing the richness of electronic health record data, together with big data analytics in health care, offers opportunities to improve outcomes and reduce costs [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>Despite these advances, most predictive models continue to exclude information derived from the nursing process, even though it has the potential to capture essential aspects of health status and service use [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. The incorporation of standardized nursing terminologies such as the North American Nursing Diagnosis Association (NANDA), the Nursing Outcomes Classification (NOC), and the Nursing Interventions Classification (NIC) could substantially enhance predictive models by capturing the impact of nursing interventions. These terminologies, which have been progressively integrated into electronic health records, enable systematic characterization of nursing practice, facilitate assessment of its impact on health outcomes, and support the development of robust predictive models [<xref ref-type="bibr" rid="ref21">21</xref>]. Understanding and quantifying nursing care in a systematic and effective manner may contribute to optimizing resource allocation and improving health system sustainability [<xref ref-type="bibr" rid="ref22">22</xref>]. In this context, Functional Health Patterns (FHP) described by Gordon provide a nursing assessment framework that offers a holistic view of patient health, encompassing physical, emotional, and social dimensions. Although this framework may add value in terms of efficiency and resource management, its impact on health care cost prediction has rarely been explored [<xref ref-type="bibr" rid="ref23">23</xref>]. Nevertheless, a previous study revealed that more than 21% of the variability in total health care costs could be explained by a model including gender, age, and variables derived from the nursing process (NANDA, NOC, NIC, and FHP) [<xref ref-type="bibr" rid="ref24">24</xref>]. This study adds a comparison between nursing process data and the clinical, sociodemographic, and morbidity-related variables commonly used in cost prediction. It also extends the analysis to partial cost components and explores dimensionality reduction.</p><p>Recent literature has highlighted that incorporating additional layers of information, such as social determinants or measures of well-being, significantly improves the performance of health care cost prediction models and helps correct biases affecting vulnerable populations [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. In parallel, the inclusion of structured nursing data has been shown to provide valuable insights into the impact of nursing practice on health outcomes and patient safety [<xref ref-type="bibr" rid="ref21">21</xref>]. However, to date, no studies have systematically integrated information derived from the nursing process into ML-based models specifically designed for health care cost prediction.</p><p>Against this background, leveraging the potential of nursing big data and integrating it with advanced ML techniques represents an opportunity to improve risk-adjustment systems, enhance population stratification, and support more efficient and equitable resource allocation. Accordingly, the aim of this study was to evaluate the impact of integrating nursing process data into ML-based predictive models of individual health care costs, including cost component analyses, compared with models based solely on sociodemographic, clinical, and morbidity-related variables.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Setting</title><p>The present study forms part of a longitudinal research program based on a single population cohort that aims to understand the determinants of health care costs from a comprehensive perspective. In the first phase, individual health care costs were estimated, and their variability was analyzed according to AMG using multiple linear regression [<xref ref-type="bibr" rid="ref14">14</xref>]. In the second phase, ML-based predictive models were developed to explain the variability in total health care costs, incorporating information derived from the nursing process [<xref ref-type="bibr" rid="ref24">24</xref>].</p><p>The present (third) phase extends this analytical framework by focusing specifically on the contribution of nursing process variables (NANDA, NOC, NIC, and FHP) to health care cost prediction. To this end, a retrospective, observational, and analytical study was conducted using data from multiple health care information systems. The study was carried out in the Autonomous Community of the Canary Islands (Spain) using data from 2017 and 2018. The predictor variables were derived from data available up to 2017, whereas health care costs were measured for 2018, thus ensuring appropriate temporal ordering for predictive modeling.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>In accordance with the Declaration of Helsinki and Spanish legislation, this study was approved by the Clinical Research Ethics Committee of the Dr. Negr&#x00ED;n University Hospital of Gran Canaria (date: January 29, 2021; code: 2021-037-1). All applicable ethical and regulatory principles for studies using secondary health data were observed.</p></sec><sec id="s2-3"><title>Study Population</title><p>The study population included all individuals aged 15 years or older who were registered in the Canary Islands Health Service health care card database as of December 31, 2017 (N=1,691,075). The entire target population was included. Individuals covered by mutual insurance schemes were excluded; these correspond to public-sector employees whose health care and social care are managed outside the Canary Islands Health Service.</p></sec><sec id="s2-4"><title>Variables</title><p>Predictor variables were derived from data available up to 2017 and included sociodemographic, clinical, morbidity-related, and nursing process information. The sociodemographic variables comprised age, gender, type of affiliation with the social security system, and the percentage of pharmaceutical copayment. Clinical variables included a Barthel Index score greater than 60, a Pfeiffer test score of 5 or lower, 2 or more hospital admissions in the previous 12 months, and the number of prescribed active pharmaceutical ingredients. Morbidity-related variables were obtained from the AMG system and included the complexity weight and morbidity group.</p><p>Nursing process variables were also incorporated and consisted of 44 variables derived from the assessment of the 11 FHP (with 4 possible outcomes per pattern), as well as 211 NANDA nursing diagnoses, 388 NOC outcomes, and 558 NIC interventions. The large number of variables reflects the conceptual structure of nursing data, which captures complementary dimensions of human responses, outcomes, and interventions; therefore, regularization and hyperparameter tuning were applied to mitigate the dimensional burden.</p><p>The primary outcome was total health care cost per individual in 2018. This was calculated as the sum of the costs associated with primary care and specialist consultations, emergency department visits, hospital admissions, major ambulatory surgery in public and contracted centers, and pharmaceutical dispensing (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). In addition, a cost component analysis was conducted for different care settings (primary care, hospital care, and contracted care) and service types (hospital admissions, outpatient consultations, and emergency department visits).</p></sec><sec id="s2-5"><title>Data Sources</title><p>Data were obtained from the following information systems: the Canary Islands Health Service health care card database (sociodemographic variables), primary care electronic health records (medical and nursing consultations and nursing process variables), hospital electronic health records (hospital consultations and emergency department visits), the Minimum Basic Dataset of the Hospital Activity Registry (hospital admissions and surgical procedures), the Hospital Contracted Care Information System (activity in contracted centers), and the electronic prescribing system of the Canary Islands Health Service (pharmaceutical dispensing). All databases were linked using a unique, encrypted clinical identifier for each individual.</p></sec><sec id="s2-6"><title>Data Preprocessing and Analytical Strategy</title><p>Prior to analysis, numerical variables (age and AMG complexity weight) were standardized. Cost variables were transformed using natural logarithms to approximate normal distributions. Nursing diagnoses (NANDA), outcomes (NOC), and interventions (NIC) with a prevalence lower than 1 per 10,000 individuals were excluded, as they provided limited information, had a high proportion of zero or missing values, and unnecessarily increased model dimensionality. This threshold was applied pragmatically, as no comparable reference studies were identified, with the aim of filtering infrequent variables and thereby reducing sparsity and computational burden.</p><p>After the data were cleaned, a descriptive analysis of the study population was performed. As the entire population was included, no statistical inference techniques were applied. The analytical focus was exclusively on the development and evaluation of predictive models.</p><p>Two ML algorithms capable of handling large-scale, high-dimensional health care data were evaluated using complementary modeling approaches. First, a feedforward neural network was implemented [<xref ref-type="bibr" rid="ref25">25</xref>], which consisted of interconnected layers designed to capture nonlinear relationships and latent patterns. Second, extreme gradient boosting (XGBoost) [<xref ref-type="bibr" rid="ref26">26</xref>], a gradient-boosted decision tree algorithm recognized for its efficiency, accuracy, and ability to handle missing values and skewed distributions, was applied. Predictions from both models were combined using an ensemble approach based on their averaged outputs to improve robustness [<xref ref-type="bibr" rid="ref27">27</xref>].</p><p>The dataset was randomly split into training (80%), validation (10%), and test (10%) subsets. During model training, the mean squared error (MSE) was used as the optimization metric, while overall performance was assessed using the coefficient of determination (<italic>R</italic>&#x00B2;). Because health care cost data were highly skewed, the outcome was log-transformed before modeling, and both model fitting and performance metrics, including <italic>R</italic>&#x00B2;, were calculated on that log-transformed scale. <italic>R</italic>&#x00B2; was used as the primary metric to compare relative model performance across the different predictor sets. Given the very large cohort size, a single training, validation, or test split was considered sufficient for initial model development and evaluation, as both the validation and test subsets each included more than 160,000 individuals. For the feedforward neural network, the final architecture consisted of 4 hidden dense layers (128, 64, 32, and 16 units) with rectified linear unit activation and L1 regularization (l1=0.0000625; l2=0). The model was trained using the root mean square propagation optimizer with a learning rate of 0.0005, a batch size of 64, and a maximum of 60 epochs. Early stopping was applied based on the validation <italic>R</italic>&#x00B2; (monitor=val_r_square, mode=max, and patience=5), and the best model weights were restored. For XGBoost, the final model was trained with a learning rate of 0.0960, a maximum tree depth of 10,500 estimators, a subsample of 0.9413, full column sampling per tree (colsample_bytree=1.0), L1 regularization (reg_alpha=10.0), and L2 regularization (reg_lambda=0.001). Hyperparameter optimization was conducted on the validation set, and the final model performance was evaluated exclusively on the test set. A fixed random seed of 42 was used to ensure reproducibility.</p><p>The analytical strategy involved 3 stages. First, baseline models excluding the nursing process variables and extended models including these variables were developed and compared for total health care cost prediction. Second, cost component analyses were conducted at both the care setting level (primary care, hospital care, and contracted care) and the service level (hospital admissions, outpatient consultations, and emergency department visits), and model performance was compared with and without the nursing process variables. Third, an autoencoder was applied to reduce the dimensionality of the nursing process variables, as the feature block was high-dimensional and predominantly binary, making a more flexible nonlinear approach preferable to the conventional principal component analysis. This neural network architecture compresses high-dimensional input data into a lower-dimensional latent representation and reconstructs it, thereby enabling the identification of unsupervised latent patterns [<xref ref-type="bibr" rid="ref28">28</xref>]. Autoencoders have previously been applied in hospital cost prediction studies [<xref ref-type="bibr" rid="ref29">29</xref>]. All analyses were performed using Python version 3.10.8, with the <italic>pandas</italic>, <italic>NumPy</italic>, <italic>scikit-learn</italic>, <italic>TensorFlow</italic>, and <italic>XGBoost</italic> libraries.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Characteristics of the Study Population</title><p>The study included 1,691,075 individuals aged over 15 years, including 863,071 (51.0%) women; the mean age was 46.6 (SD 17.9) years, and 17.4% (n=294,077) were older than 65 years. The mean AMG index weight was 5.82 (SD 5.89) points, which was higher in women (6.25, SD 6.06) than in men (5.10, SD 5.62) and increased with age (Spearman &#x03C1;=0.47). A total of 216,630 (12.8) women and 175,536 (10.4) men were pensioners. The mean number of prescribed medications was 4.35 (SD 5.10), with higher values observed in women.</p><p>Overall, 22,800 (1.3%) individuals had 2 or more hospital admissions in the previous year, 6308 (0.4%) had a Barthel Index score greater than 60, and 7271 (0.4%) demonstrated cognitive impairment (Pfeiffer score&#x2265;5). Nursing records (FHP, NANDA, NOC, and NIC) were present for 57.9% (980,437/1,691,075) of the population, although only 2.7% (46,380/1,691,075) had completed the full nursing process. On average, 3.84 (SD 3.58) FHP were assessed per person, together with 3.07 (SD 2.9) NANDA nursing diagnoses, 2.77 (SD 2.96) NOC outcomes, and 4.12 (7.04) NIC interventions.</p><p>The mean cost per patient was &#x20AC;1283.08 (SD &#x20AC;3007; &#x20AC;1=US $1.19 as of December 31, 2017), with higher values observed among women. Hospital admissions and consultations accounted for the highest cost components. Costs were heterogeneously distributed across care settings, with those for consultations (mean &#x20AC;355.07, SD &#x20AC;586.87) and pharmaceutical dispensing (mean &#x20AC;345.53, SD &#x20AC;946.09) being especially notable. <xref ref-type="table" rid="table1">Table 1</xref> presents details of the variables included in the models; the complete cohort profile can be found in previous publications [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref24">24</xref>].</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics of the study population and variables included in the predictive models.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variables</td><td align="left" valign="bottom">Total<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="bottom">Women</td><td align="left" valign="bottom">Men</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Sociodemographic and clinical, n (%)</td></tr><tr><td align="left" valign="top">&#x2003;Pensioners</td><td align="left" valign="top">392,166 (23.2)</td><td align="left" valign="top">216,630 (12.8)</td><td align="left" valign="top">175,536 (10.4)</td></tr><tr><td align="left" valign="top">&#x2003;&#x2265;2 hospital admissions</td><td align="left" valign="top">22,846 (1.4)</td><td align="left" valign="top">12,082 (0.7)</td><td align="left" valign="top">10,764 (0.6)</td></tr><tr><td align="left" valign="top">&#x2003;Barthel Index &#x003C;60</td><td align="left" valign="top">5971 (0.4)</td><td align="left" valign="top">4315 (0.3)</td><td align="left" valign="top">1656 (0.1)</td></tr><tr><td align="left" valign="top">&#x2003;Pfeiffer score &#x2265;5</td><td align="left" valign="top">6834 (0.4)</td><td align="left" valign="top">5178 (0.3)</td><td align="left" valign="top">1656 (0.1)</td></tr><tr><td align="left" valign="top" colspan="4">AMG<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> system, mean (SD)</td></tr><tr><td align="left" valign="top">&#x2003;Complexity weight</td><td align="left" valign="top">5.82 (5.89)</td><td align="left" valign="top">6.25 (6.06)</td><td align="left" valign="top">5.10 (5.62)</td></tr><tr><td align="left" valign="top">&#x2003;Mean number medications</td><td align="left" valign="top">4.35 (5.10)</td><td align="left" valign="top">5.19 (5.47)</td><td align="left" valign="top">3.48 (4.55)</td></tr><tr><td align="left" valign="top" colspan="4">Nursing process, n (%)</td></tr><tr><td align="left" valign="top">&#x2003;Nursing process records</td><td align="left" valign="top">980,437 (57.9)</td><td align="left" valign="top">533,231 (61.8)</td><td align="left" valign="top">447,206 (54.0)</td></tr><tr><td align="left" valign="top">&#x2003;Completed nursing process</td><td align="left" valign="top">46,380 (2.7)</td><td align="left" valign="top">29,405 (3.4)</td><td align="left" valign="top">16,975 (2.1)</td></tr><tr><td align="left" valign="top">&#x2003;At least 1 FHP<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> assessed</td><td align="left" valign="top">415,068 (24.5)</td><td align="left" valign="top">247,162 (28.6)</td><td align="left" valign="top">167,906 (20.3)</td></tr><tr><td align="left" valign="top">&#x2003;At least 1 active NANDA<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">916,024 (54.2)</td><td align="left" valign="top">489,738 (56.7)</td><td align="left" valign="top">426,286 (51.5)</td></tr><tr><td align="left" valign="top">&#x2003;At least 1 active NOC<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">772,730 (45.7)</td><td align="left" valign="top">419,496 (48.6)</td><td align="left" valign="top">353,234 (42.7)</td></tr><tr><td align="left" valign="top">&#x2003;At least 1 active NIC<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="left" valign="top">787,510 (46.6)</td><td align="left" valign="top">427,370 (49.5)</td><td align="left" valign="top">360,140 (43.5)</td></tr><tr><td align="left" valign="top" colspan="4">Cost (&#x20AC;; &#x20AC;1=US $1.19)<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup>, mean (SD)</td></tr><tr><td align="left" valign="top">&#x2003;Mean total health care cost</td><td align="left" valign="top">1283.08 (3007)</td><td align="left" valign="top">1377.04 (2868.27)</td><td align="left" valign="top">1185.14 (3142.05)</td></tr><tr><td align="left" valign="top">&#x2003;Hospital admissions</td><td align="left" valign="top">463.11 (2207.39)</td><td align="left" valign="top">461.55 (2060.64)</td><td align="left" valign="top">464.72 (2350.62)</td></tr><tr><td align="left" valign="top">&#x2003;Outpatient consultations</td><td align="left" valign="top">355.07 (586.87)</td><td align="left" valign="top">418.80 (623.38)</td><td align="left" valign="top">288.64 (538.24)</td></tr><tr><td align="left" valign="top">&#x2003;Emergency department visits</td><td align="left" valign="top">119.35 (324.68)</td><td align="left" valign="top">131.10 (332.68)</td><td align="left" valign="top">107.09 (315.67)</td></tr><tr><td align="left" valign="top">&#x2003;Pharmaceutical dispensing</td><td align="left" valign="top">345.53 (946.09)</td><td align="left" valign="top">365.55 (914.18)</td><td align="left" valign="top">324.67 (977.81)</td></tr><tr><td align="left" valign="top">&#x2003;Hospital care</td><td align="left" valign="top">571.90 (2274)</td><td align="left" valign="top">597.73 (2118.51)</td><td align="left" valign="top">544.99 (2425.37)</td></tr><tr><td align="left" valign="top">&#x2003;Primary care</td><td align="left" valign="top">246.85 (421)</td><td align="left" valign="top">290.96 (454.43)</td><td align="left" valign="top">200.87 (378.69)</td></tr><tr><td align="left" valign="top">&#x2003;Contracted care</td><td align="left" valign="top">118.8 (808)</td><td align="left" valign="top">122.76 (827.41)</td><td align="left" valign="top">114.59 (786.31)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>The cohort included 1,691,075 individuals (863,071 women and 828,004 men).</p></fn><fn id="table1fn2"><p><sup>b</sup>AMG: Adjusted Morbidity Groups. </p></fn><fn id="table1fn3"><p><sup>c</sup>FHP: Functional Health Pattern.</p></fn><fn id="table1fn4"><p><sup>d</sup>NANDA: North American Nursing Diagnosis Association.</p></fn><fn id="table1fn5"><p><sup>e</sup>NOC: Nursing Outcomes Classification.</p></fn><fn id="table1fn6"><p><sup>f</sup>NIC: Nursing Interventions Classification.</p></fn><fn id="table1fn7"><p><sup>g</sup>Health care cost components are presented by care setting (primary care, hospital care, and contracted care) and by service type (hospital admissions, outpatient consultations, and emergency department visits), as well as pharmaceutical dispensing.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Predictive Performance of the Baseline and Extended Models for Health Care Cost Prediction</title><p>The neural network architecture with the best performance was a feedforward network with 4 hidden layers (128, 64, 32, and 16 nodes) and an output corresponding to the logarithm of the total cost. For the XGBoost model, Bayesian hyperparameter optimization was applied to identify the combinations with the best performance, and an ensemble model was subsequently generated by averaging the predictions.</p><p>The initial feature set included sociodemographic and clinical variables, together with 1201 nursing process variables. After those with a positive prevalence of fewer than one per 10,000 patients were excluded, 789 nursing variables and 14 sociodemographic, clinical, and AMG system variables were retained (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Workflow of nursing-variable selection and dimensionality reduction for the neural network and XGBoost models. AMG: Adjusted Morbidity Group; NNP: nonnursing process; NP: nursing process; XGBoost: extreme gradient boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="nursing_v9i1e93638_fig01.png"/></fig><p>In the baseline models, excluding the nursing variables, the <italic>R</italic>&#x00B2; was 0.4996 for the neural network, 0.5030 for XGBoost, and 0.5023 for the ensemble model. After the nursing process variables were incorporated, the predictive performance increased to 0.5026 (MSE 3.8561), 0.5065 (MSE 3.7936), and 0.5058 (MSE 3.7981), respectively, indicating improvements across all 3 algorithms.</p></sec><sec id="s3-3"><title>Predictive Performance by Cost Component According to Care Setting and Service Type</title><p>Across most cost components, the inclusion of the nursing process variables improved predictive performance, with variations depending on the care setting and model type (<xref ref-type="table" rid="table2">Table 2</xref>). For hospital care, the greatest increase was observed in the ensemble model, with <italic>R</italic>&#x00B2; increasing from 0.2161 to 0.2396 (+0.0235), while the improvements in the neural network and XGBoost models were similar (+0.0230 and +0.0223, respectively). For primary care, improvements were more moderate but consistent; the ensemble model increased from 0.4388 to 0.4442 (+0.0054), and comparable gains were observed for the neural network (+0.0050) and XGBoost (+0.0054). For emergency department visits, the predictive performance also increased, with that of the ensemble model increasing from 0.1431 to 0.1506 (+0.0075) and similar gains observed for XGBoost (+0.0074). For contracted care, improvements were more limited in magnitude but were observed across all models, with the ensemble model&#x2019;s performance increasing from 0.0389 to 0.0443 (+0.0054).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Predictive performance (<italic>R</italic>&#x00B2;) of machine learning&#x2013;models with and without nursing process variables across health care cost components.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Health care cost components and models</td><td align="left" valign="bottom">No (<italic>R</italic>&#x00B2;)</td><td align="left" valign="bottom">Yes (<italic>R</italic>&#x00B2;)</td><td align="left" valign="bottom">Differences</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Care settings</td></tr><tr><td align="left" valign="top" colspan="4"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Primary care</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neural network <italic>R</italic>&#x00B2;</td><td align="char" char="." valign="top">0.4359</td><td align="char" char="." valign="top">0.4409</td><td align="char" char="." valign="top">+0.0050</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGB<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.4394</td><td align="left" valign="top">0.4448</td><td align="left" valign="top">+0.0054</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ensemble <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.4388</td><td align="left" valign="top">0.4442</td><td align="left" valign="top">+0.0054</td></tr><tr><td align="left" valign="top" colspan="4"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hospital care</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neural network <italic>R</italic>&#x00B2;</td><td align="char" char="." valign="top">0.2133</td><td align="char" char="." valign="top">0.2363</td><td align="char" char="." valign="top">+0.0230</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGB <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.2163</td><td align="left" valign="top">0.2386</td><td align="left" valign="top">+0.0223</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ensemble <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.2161</td><td align="left" valign="top">0.2396</td><td align="left" valign="top">+0.0235</td></tr><tr><td align="left" valign="top" colspan="4"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Contracted care</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neural network <italic>R</italic>&#x00B2;</td><td align="char" char="." valign="top">0.0382</td><td align="char" char="." valign="top">0.0402</td><td align="char" char="." valign="top">+0.0020</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGB <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.0380</td><td align="left" valign="top">0.0452</td><td align="left" valign="top">+0.0072</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ensemble <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.0389</td><td align="left" valign="top">0.0443</td><td align="left" valign="top">+0.0054</td></tr><tr><td align="left" valign="top" colspan="4">Service types</td></tr><tr><td align="left" valign="top" colspan="4"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hospital admissions</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neural network <italic>R</italic>&#x00B2;</td><td align="char" char="." valign="top">0.0624</td><td align="char" char="." valign="top">0.0618</td><td align="char" char="." valign="top">&#x2212;0.0006</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGB <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.0636</td><td align="left" valign="top">0.0652</td><td align="left" valign="top">+0.0016</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ensemble <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.0641</td><td align="left" valign="top">0.0650</td><td align="left" valign="top">+0.0009</td></tr><tr><td align="left" valign="top" colspan="4"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Outpatient consultations</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neural network <italic>R</italic>&#x00B2;</td><td align="char" char="." valign="top">0.4568</td><td align="char" char="." valign="top">0.4608</td><td align="char" char="." valign="top">+0.0040</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGB <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.4606</td><td align="left" valign="top">0.4652</td><td align="left" valign="top">+0.0046</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ensemble <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.4598</td><td align="left" valign="top">0.4642</td><td align="left" valign="top">+0.0044</td></tr><tr><td align="left" valign="top" colspan="4"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Emergency department visits</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neural network <italic>R</italic>&#x00B2;</td><td align="char" char="." valign="top">0.1407</td><td align="char" char="." valign="top">0.1473</td><td align="char" char="." valign="top">+0.0066</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGB <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.1434</td><td align="left" valign="top">0.1508</td><td align="left" valign="top">+0.0074</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ensemble <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.1431</td><td align="left" valign="top">0.1506</td><td align="left" valign="top">+0.0075</td></tr><tr><td align="left" valign="top" colspan="4">Other cost components</td></tr><tr><td align="left" valign="top" colspan="4"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pharmaceutical costs</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Neural network <italic>R</italic>&#x00B2;</td><td align="char" char="." valign="top">0.6565</td><td align="char" char="." valign="top">0.6612</td><td align="char" char="." valign="top">+0.0047</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGB <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.6588</td><td align="left" valign="top">0.6631</td><td align="left" valign="top">+0.0043</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ensemble <italic>R</italic>&#x00B2;</td><td align="left" valign="top">0.6584</td><td align="left" valign="top">0.6631</td><td align="left" valign="top">+0.0047</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>XGB: extreme gradient boosting.</p></fn></table-wrap-foot></table-wrap><p>When analyzed by service type, outpatient consultations and emergency department visits showed the most consistent improvements. For outpatient consultations, the <italic>R</italic>&#x00B2; of the ensemble model increased from 0.4598 to 0.4642 (+0.0044), whereas for emergency department visits, it increased from 0.1431 to 0.1506 (+0.0075). For hospital admissions, the changes were minimal; the ensemble model <italic>R</italic>&#x00B2; increased slightly from 0.0641 to 0.0650 (+0.0009), whereas that of the neural network marginally decreased (&#x2212;0.0006).</p></sec><sec id="s3-4"><title>Predictive Models Using Latent Variables Derived From the Autoencoder</title><p>Finally, an autoencoder was applied to reduce the 789 nursing variables to 16 latent dimensions while preserving relevant data patterns. These compressed variables were incorporated with the sociodemographic, clinical, and AMG-related variables. With this combination, the models achieved <italic>R</italic>&#x00B2; values of 0.5013 for the neural network, 0.5062 for XGBoost, and 0.5052 for the ensemble model. Thus, dimensionality reduction markedly simplified the nursing feature space while yielding predictive performance that was broadly comparable to that of the models using the larger nursing-variable set, but without a substantial performance improvement.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>The integration of variables derived from nursing methodology into ML-based prediction of total and component-level health care costs resulted in small numerical improvements in total cost prediction, whereas changes observed across cost components were heterogeneous and, in many cases, of small magnitude. These results suggest that nursing information may contain an additional predictive signal, although the magnitude of this contribution was limited, thereby reflecting differences in care delivery structures and in the relationship between care processes and resource use. Accordingly, the balance between added model complexity and predictive benefit should be interpreted with caution, as the contribution of nursing information varies according to the care setting and service type considered.</p><p>To date, the predominant methodologies used in health care cost analysis have systematically excluded data related to nursing interventions and outcomes, despite evidence that nursing terminologies can improve both the quality and the efficiency of resource management [<xref ref-type="bibr" rid="ref30">30</xref>]. Their standardization enables the documentation and quantification of nursing work and provides a holistic view of the patient that is not usually captured by traditional biomedical data. Recent studies have shown that incorporating nursing information into clinical databases facilitates a better understanding of the impact of nursing practice on patient outcomes [<xref ref-type="bibr" rid="ref21">21</xref>] and contributes to improving patient safety [<xref ref-type="bibr" rid="ref31">31</xref>], with a potential indirect effect on cost reduction.</p><p>However, the literature on the application of ML to nursing data has focused primarily on the development of intelligent clinical tools, such as virtual assistants, fall prediction models, and patient monitoring systems, rather than on economic outcomes [<xref ref-type="bibr" rid="ref32">32</xref>]. In this context, the present study represents, to our knowledge, a novel approach by systematically integrating nursing process information into cost prediction models using ML techniques, thereby extending previous research in which the added economic value of nursing data has rarely been evaluated.</p><p>The use of ML algorithms has demonstrated usefulness in health care cost prediction by overcoming limitations inherent in traditional statistical models, such as data skewness and distributional bias [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. In general, the variables most commonly used to predict health care expenditure are medical diagnoses and prescribed medications [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. For example, in the study by Drewe-Boss et al [<xref ref-type="bibr" rid="ref2">2</xref>], neural network&#x2013;based models relied primarily on medical diagnoses coded according to the <italic>International Classification of Diseases</italic>, 10th Revision (ICD-10) and the Anatomical Therapeutic Chemical (ATC) classification system. In contrast, models built using ridge regression placed greater weight on Geb&#x00FC;hrenordnungspositionen codes, which are broadly comparable to the diagnosis-related groups used in Spain and other European countries.</p><p>Similarly, comorbidity may be a better predictor of pharmaceutical consumption and other health care products than clinical complexity, which encompasses psychological, social, and functional dimensions [<xref ref-type="bibr" rid="ref35">35</xref>]. The findings of the present study are partially consistent with this pattern because the inclusion of nursing variables increased the predictive capacity for total health care costs, although the magnitude of this increase was modest. Nevertheless, given the limited magnitude of the observed improvement, its operational relevance for population risk adjustment or resource planning should be interpreted with caution.</p><p>Health care costs are complex and influenced by multiple factors. The incorporation of measures of well-being into prediction models has been shown to improve the estimation of future health care costs by 5.7% to 13% [<xref ref-type="bibr" rid="ref36">36</xref>]. Another study revealed that each one-point increase in well-being indices was associated with a 1% reduction in health care expenditure [<xref ref-type="bibr" rid="ref37">37</xref>]. Similarly, the inclusion of social determinants in predictive models has demonstrated additional benefits: in populations characterized by high levels of poverty, inequality, or lack of coverage, social determinants reduced cost underestimation and improved predictive performance by 3%, corresponding to approximately US $200 per person per year [<xref ref-type="bibr" rid="ref18">18</xref>].</p><p>The integration of this type of data into ML models represents a promising avenue for improving the efficiency and sustainability of health care systems [<xref ref-type="bibr" rid="ref38">38</xref>]. Along these lines, nursing data capture dimensions related to functional status, care needs, and human responses to health problems that are traditionally absent from cost models and could complement the social and clinical determinants commonly used.</p><p>In the analyses performed, pharmaceutical costs showed the highest predictive accuracy, with <italic>R</italic>&#x00B2; values increasing from 0.6584 to 0.6631 following the inclusion of nursing variables. This finding suggests that these variables may capture information related to therapeutic adherence, medication management, and educational processes, which are consistent with the role of nursing practice. Although the observed improvement was modest, its limited magnitude means that any interpretation regarding its practical usefulness should remain cautious. In contrast, in other areas, such as outpatient consultations or primary care, changes were small and, in some cases, showed a downward trend, indicating a limited effect of nursing variables in these specific settings. Overall, these patterns indicate that the contribution of nursing methodology is not uniform and may be more relevant in functions closely linked to clinical follow-up and pharmacological management.</p><p>In primary care, a setting in which nursing plays a central role, particularly in the management of chronic conditions, an improvement in predictive performance was also observed (<italic>R</italic>&#x00B2; increasing from 0.4388 to 0.4442). This result suggests that long-term care and continuous follow-up, areas with greater nursing involvement, may benefit more from the systematic incorporation of the nursing process into predictive models.</p><p>In contrast, the XGBoost models consistently showed better predictive performance than the neural networks in the present analysis, although the literature has reported mixed findings. For example, in a study on pharmaceutical cost prediction, boosted tree&#x2013;based models outperformed neural networks (area under the curve [AUC]=0.74) [<xref ref-type="bibr" rid="ref39">39</xref>], whereas another study on hospital cost prediction reported better performance for neural networks (<italic>R</italic>&#x00B2;=0.813) than for decision trees (<italic>R</italic>&#x00B2;=0.713) [<xref ref-type="bibr" rid="ref40">40</xref>]. Nevertheless, XGBoost is widely regarded as one of the most robust algorithms for tabular data and for scenarios with missing values [<xref ref-type="bibr" rid="ref41">41</xref>]. The consistency of its performance in the present study further supports its suitability for complex health care data with multiple categorical and sparse variables, such as those derived from the nursing process.</p><p>One of the main uses of predictive models lies in identifying relevant features within large datasets and in understanding their contribution to outcome prediction. However, when the number of variables is high, the presence of irrelevant or redundant dimensions may adversely affect model performance by introducing noise and unnecessarily increasing complexity. In the present study, the modeling strategy was designed to assess the incremental predictive value of the nursing data block as a whole, rather than to determine the individual relevance of specific nursing diagnoses, outcomes, interventions, or functional patterns.</p><p>In this context, autoencoders have become established as effective tools for dimensionality reduction, redundancy elimination, and the generation of latent representations that preserve essential information [<xref ref-type="bibr" rid="ref42">42</xref>]. In the present study, autoencoder implementation enabled the condensing of 789 nursing methodology variables into 16 latent variables. Although this reduction preserved relevant patterns and substantially simplified model complexity, it did not translate into a substantial improvement in predictive performance, indicating that dimensionality reduction alone was not sufficient to strengthen predictive performance in a meaningful way. Therefore, the autoencoder should be interpreted here as a strategy for representation simplification rather than as a predictive advantage. Future studies could explore which of these latent dimensions are most informative and whether they capture distinct patient profiles.</p><p>Among the most widely used health care management tools are risk-adjustment models such as ACGs, Diagnosis-Related Groups, and AMGs, which are extensively applied in Spain and other countries. These models are integrated into electronic health records with the aim of classifying morbidity, predicting hospital admissions or mortality, and estimating health care costs, thereby supporting efficient resource allocation and informing managerial decision-making. However, because they are largely based on linear statistical approaches, they present important limitations when applied in settings characterized by complex, nonlinear, and high-dimensional data, as is common in clinical information systems [<xref ref-type="bibr" rid="ref15">15</xref>]. These limitations may result in the underestimation of actual health care costs and reduced predictive accuracy in scenarios of high care complexity.</p><p>In contrast, ML algorithms have been shown to outperform traditional models across several key metrics. For example, in mortality prediction, they achieve AUC values ranging from 0.92 to 0.94, compared with 0.91 for classical models, and in the prediction of hospital readmissions, performance improves from 0.69 to 0.75-0.76 [<xref ref-type="bibr" rid="ref43">43</xref>]. In light of these findings, the present results suggest that combining ML algorithms with variables derived from the nursing process could increase the precision and practical use of current risk adjustment models, particularly in settings characterized by a high nursing workload.</p></sec><sec id="s4-2"><title>Implications for Practice</title><p>The findings of the present study indicate that the systematic incorporation of variables derived from the nursing process into clinical information systems may contribute to improving the accuracy of health care cost prediction models and could enhance population stratification tools. Integrating these data may enable more accurate identification of patient groups with greater resource needs and facilitate the planning of more efficient, evidence-based interventions.</p><p>In addition, recognizing the predictive value of nursing documentation highlights its contribution to health care system management and underscores the importance of promoting structured, interoperable, and high-quality nursing records in clinical practice.</p></sec><sec id="s4-3"><title>Strengths and Limitations</title><p>The present study has several limitations that should be considered when interpreting the findings. First, the observation period was limited to a single year, which restricts the ability to analyze long-term cost trajectories. Although previous studies have indicated that predictive performance improves with longer follow-up periods [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref44">44</xref>], the use of a large population-based cohort (N=1,691,075) provides a robust foundation and reduces the risk of overfitting, thereby strengthening the external validity of the analysis.</p><p>Second, comparability with other studies is limited because of the heterogeneity in the algorithms, validation strategies, and performance metrics reported in the literature. In addition, no studies have been identified that have systematically integrated nursing process variables into health care cost prediction models, which hinders direct comparison but also highlights the innovative nature of this work.</p><p>Another limitation relates to dimensionality reduction using autoencoders. Although this approach enabled the synthesis of 789 nursing variables into 16 latent components, the clinical interpretability of these components was lower, and no substantial improvement in predictive performance was observed. Nevertheless, these latent representations offer opportunities for future research aimed at identifying patient profiles and patterns associated with resource use.</p><p>Finally, the rapid evolution of AI techniques means that some approaches may become outdated over time. However, the methods applied represent well-established standards in ML and are appropriate for the stated objectives. Replicating the present study in other health care systems would help strengthen the generalizability of the findings and broaden their applicability.</p></sec><sec id="s4-4"><title>Conclusions</title><p>The incorporation of variables derived from the nursing process into ML-based health care cost prediction models was associated with small incremental improvements in the explanatory capacity for total health care costs, whereas effects observed across cost components were heterogeneous and of smaller magnitude. These findings suggest that structured nursing information may complement the sociodemographic, clinical, and morbidity-related variables commonly used in health care expenditure estimation.</p><p>The results highlight the relevance of moving toward the integration of nursing process information into population-stratification systems and risk-adjustment models, which are key tools for health care planning and efficient resource allocation. The inclusion of dimensions related to functional status, care needs, and human responses broadens the traditional perspective based on clinical and administrative data, thereby supporting more comprehensive approaches to cost prediction.</p><p>The potential of ML to improve the estimation of health care resource use should not be limited exclusively to conventional biomedical predictors. The incorporation of structured information derived from nursing practice may enhance the understanding of the determinants of health care costs and support the development of more holistic, person-centered predictive models aligned with contemporary approaches to integrated care.</p></sec></sec></body><back><ack><p>No generative AI tools were used in the preparation of this manuscript. Authors remain fully responsible for the accuracy, originality, and integrity of all the content in the manuscript, including all the references and their citations.</p></ack><notes><sec><title>Funding</title><p>The authors gratefully acknowledge the financial support provided by the Instituto de Investigaci&#x00F3;n Sanitaria y Biom&#x00E9;dica de Alicante (ISABIAL) to cover the open access publication costs of this article within the programme 2026-0243 // UGP-26-05</p></sec><sec><title>Data Availability</title><p>The datasets generated and/or analyzed during the study are not publicly available because of the inclusion of sensitive and confidential data but are available from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: MCC-S, MIO-C, VMG-C</p><p>Investigation: MCC-S, MIO-C</p><p>Formal analysis: MCC-S, VMG-C</p><p>Supervision: MCC-S, MIO-C, VMG-C</p><p>Writing &#x2013; original draft: MCC-S, MIO-C, VMG-C</p><p>Writing &#x2013; review &#x0026; editing: MCC-S, VMG-C, MIO-C</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ACG</term><def><p>Adjusted Clinical Group</p></def></def-item><def-item><term id="abb2">AMG</term><def><p>Adjusted Morbidity Group</p></def></def-item><def-item><term id="abb3">ATC</term><def><p>Anatomical Therapeutic Chemical</p></def></def-item><def-item><term id="abb4">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb5">CRG</term><def><p>Clinical Risk Group</p></def></def-item><def-item><term id="abb6">FHP</term><def><p>Functional Health Patterns</p></def></def-item><def-item><term id="abb7">ICD-10</term><def><p><italic>International Classification of Diseases, 10th Revision</italic></p></def></def-item><def-item><term id="abb8">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb9">MSE</term><def><p>mean squared error</p></def></def-item><def-item><term id="abb10">NANDA</term><def><p>North American Nursing Diagnosis Association</p></def></def-item><def-item><term id="abb11">NIC</term><def><p>Nursing Interventions Classification</p></def></def-item><def-item><term id="abb12">NOC</term><def><p>Nursing Outcomes Classification</p></def></def-item><def-item><term id="abb13">XGBoost</term><def><p>extreme gradient boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Santamar&#x00ED;a Benhumea</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Herrera Villalobos</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Sil Jaimes</surname><given-names>PA</given-names> </name><name name-style="western"><surname>Santamar&#x00ED;a Benhumea</surname><given-names>NH</given-names> </name><name name-style="western"><surname>Flores Manzur</surname><given-names>M&#x00C1;</given-names> </name><name name-style="western"><surname>del Arco Ortiz</surname><given-names>A</given-names> </name></person-group><article-title>Estructura, sistemas y an&#x00E1;lisis de costos de la atenci&#x00F3;n m&#x00E9;dica hospitalaria [Article in Spanish]</article-title><source>Medicina E Investigaci&#x00F3;n</source><year>2015</year><month>07</month><volume>3</volume><issue>2</issue><fpage>134</fpage><lpage>140</lpage><pub-id pub-id-type="doi">10.1016/j.mei.2015.06.001</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Drewe-Boss</surname><given-names>P</given-names> </name><name name-style="western"><surname>Enders</surname><given-names>D</given-names> </name><name name-style="western"><surname>Walker</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ohler</surname><given-names>U</given-names> </name></person-group><article-title>Deep learning for prediction of population health costs</article-title><source>BMC Med Inform Decis Mak</source><year>2022</year><month>02</month><day>3</day><volume>22</volume><issue>1</issue><fpage>32</fpage><pub-id pub-id-type="doi">10.1186/s12911-021-01743-z</pub-id><pub-id pub-id-type="medline">35114978</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaplan</surname><given-names>RS</given-names> </name><name name-style="western"><surname>Porter</surname><given-names>ME</given-names> </name></person-group><article-title>How to solve the cost crisis in health care</article-title><source>Harv Bus Rev</source><year>2011</year><month>09</month><volume>89</volume><issue>9</issue><fpage>46</fpage><lpage>52</lpage><pub-id pub-id-type="medline">21939127</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Badawy</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ramadan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hefny</surname><given-names>HA</given-names> </name></person-group><article-title>Healthcare predictive analytics using machine learning and deep learning techniques: a survey</article-title><source>J Electr Syst Inf Technol</source><year>2023</year><volume>10</volume><issue>1</issue><fpage>40</fpage><pub-id pub-id-type="doi">10.1186/s43067-023-00108-y</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Esteva</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kuprel</surname><given-names>B</given-names> </name><name name-style="western"><surname>Novoa</surname><given-names>RA</given-names> </name><etal/></person-group><article-title>Dermatologist-level classification of skin cancer with deep neural networks</article-title><source>Nature</source><year>2017</year><month>02</month><day>2</day><volume>542</volume><issue>7639</issue><fpage>115</fpage><lpage>118</lpage><pub-id pub-id-type="doi">10.1038/nature21056</pub-id><pub-id pub-id-type="medline">28117445</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Morid</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Kawamoto</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ault</surname><given-names>T</given-names> </name><name name-style="western"><surname>Dorius</surname><given-names>J</given-names> </name><name name-style="western"><surname>Abdelrahman</surname><given-names>S</given-names> </name></person-group><article-title>Supervised learning methods for predicting healthcare costs: systematic literature review and empirical evaluation</article-title><source>AMIA Annu Symp Proc</source><year>2018</year><volume>2017</volume><fpage>1312</fpage><lpage>1321</lpage><pub-id pub-id-type="medline">29854200</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Perez-Lebel</surname><given-names>A</given-names> </name><name name-style="western"><surname>Varoquaux</surname><given-names>G</given-names> </name><name name-style="western"><surname>Le Morvan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Josse</surname><given-names>J</given-names> </name><name name-style="western"><surname>Poline</surname><given-names>JB</given-names> </name></person-group><article-title>Benchmarking missing-values approaches for predictive models on health databases</article-title><source>Gigascience</source><year>2022</year><volume>11</volume><fpage>giac013</fpage><pub-id pub-id-type="doi">10.1093/gigascience/giac013</pub-id><pub-id pub-id-type="medline">35426912</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ru</surname><given-names>B</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Comparison of machine learning algorithms for predicting hospital readmissions and worsening heart failure events in patients with heart failure with reduced ejection fraction: modeling study</article-title><source>JMIR Form Res</source><year>2023</year><month>04</month><day>17</day><volume>7</volume><fpage>e41775</fpage><pub-id pub-id-type="doi">10.2196/41775</pub-id><pub-id pub-id-type="medline">37067873</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Wemmert</surname><given-names>C</given-names> </name><name name-style="western"><surname>Weber</surname><given-names>J</given-names> </name><name name-style="western"><surname>Feuerhake</surname><given-names>F</given-names> </name><name name-style="western"><surname>Forestier</surname><given-names>G</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Elloumi</surname><given-names>M</given-names> </name></person-group><article-title>Deep learning for histopathological image analysis</article-title><source>Deep Learning for Biomedical Data Analysis</source><year>2021</year><publisher-name>Springer</publisher-name><fpage>153</fpage><lpage>169</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-71676-9_7</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="report"><article-title>The Johns Hopkins ACG&#x00AE; system: excerpt from version 11.0 technical reference guide</article-title><year>2014</year><access-date>2026-08-25</access-date><publisher-name>Johns Hopkins University, Bloomberg School of Public Health</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.hopkinsacg.org/document/acg-system-version-11-technical-reference-guide/">https://www.hopkinsacg.org/document/acg-system-version-11-technical-reference-guide/</ext-link></comment></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="report"><article-title>3M&#x2122; Clinical Risk Groups: measuring risk, managing care</article-title><year>2016</year><access-date>2025-02-22</access-date><publisher-name>3M Health Information Systems / 3M United Kingdom PLC</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://multimedia.3m.com/mws/media/1356109O/crg-measuring-risk-managing-care-ukv1.pdf">https://multimedia.3m.com/mws/media/1356109O/crg-measuring-risk-managing-care-ukv1.pdf</ext-link></comment></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Monterde</surname><given-names>D</given-names> </name><name name-style="western"><surname>Vela</surname><given-names>E</given-names> </name><name name-style="western"><surname>Cl&#x00E8;ries</surname><given-names>M</given-names> </name><collab>grupo colaborativo GMA</collab></person-group><article-title>Adjusted morbidity groups: a new multiple morbidity measurement of use in primary care [Article in Spanish]</article-title><source>Aten Primaria</source><year>2016</year><month>12</month><volume>48</volume><issue>10</issue><fpage>674</fpage><lpage>682</lpage><pub-id pub-id-type="doi">10.1016/j.aprim.2016.06.003</pub-id><pub-id pub-id-type="medline">27495004</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rose</surname><given-names>S</given-names> </name></person-group><article-title>A machine learning framework for plan payment risk adjustment</article-title><source>Health Serv Res</source><year>2016</year><month>12</month><volume>51</volume><issue>6</issue><fpage>2358</fpage><lpage>2374</lpage><pub-id pub-id-type="doi">10.1111/1475-6773.12464</pub-id><pub-id pub-id-type="medline">26891974</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Company-Sancho</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Gonz&#x00E1;lez-Chord&#x00E1;</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Orts-Cort&#x00E9;s</surname><given-names>MI</given-names> </name></person-group><article-title>Variability in healthcare expenditure according to the stratification of adjusted morbidity groups in the Canary Islands (Spain)</article-title><source>Int J Environ Res Public Health</source><year>2022</year><month>04</month><day>1</day><volume>19</volume><issue>7</issue><fpage>4219</fpage><pub-id pub-id-type="doi">10.3390/ijerph19074219</pub-id><pub-id pub-id-type="medline">35409900</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kan</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Kharrazi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>HY</given-names> </name><name name-style="western"><surname>Bodycombe</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lemke</surname><given-names>K</given-names> </name><name name-style="western"><surname>Weiner</surname><given-names>JP</given-names> </name></person-group><article-title>Exploring the use of machine learning for risk adjustment: acomparison of standard and penalized linear regression models in predicting health care costs in older adults</article-title><source>PLoS One</source><year>2019</year><volume>14</volume><issue>3</issue><fpage>e0213258</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0213258</pub-id><pub-id pub-id-type="medline">30840682</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jensen</surname><given-names>PB</given-names> </name><name name-style="western"><surname>Jensen</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Brunak</surname><given-names>S</given-names> </name></person-group><article-title>Mining electronic health records: towards better research applications and clinical care</article-title><source>Nat Rev Genet</source><year>2012</year><month>05</month><day>2</day><volume>13</volume><issue>6</issue><fpage>395</fpage><lpage>405</lpage><pub-id pub-id-type="doi">10.1038/nrg3208</pub-id><pub-id pub-id-type="medline">22549152</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bertsimas</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bjarnad&#x00F3;ttir</surname><given-names>MV</given-names> </name><name name-style="western"><surname>Kane</surname><given-names>MA</given-names> </name><etal/></person-group><article-title>Algorithmic prediction of health-care costs</article-title><source>Oper Res</source><year>2008</year><month>12</month><volume>56</volume><issue>6</issue><fpage>1382</fpage><lpage>1392</lpage><pub-id pub-id-type="doi">10.1287/opre.1080.0619</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Irvin</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Kondrich</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Ko</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Incorporating machine learning and social determinants of health indicators into prospective risk adjustment for health plan payments</article-title><source>BMC Public Health</source><year>2020</year><month>05</month><day>1</day><volume>20</volume><issue>1</issue><fpage>608</fpage><pub-id pub-id-type="doi">10.1186/s12889-020-08735-0</pub-id><pub-id pub-id-type="medline">32357871</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fitzpatrick</surname><given-names>T</given-names> </name><name name-style="western"><surname>Rosella</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Calzavara</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Looking beyond income and education: socioeconomic status gradients among future high-cost users of health care</article-title><source>Am J Prev Med</source><year>2015</year><month>08</month><volume>49</volume><issue>2</issue><fpage>161</fpage><lpage>171</lpage><pub-id pub-id-type="doi">10.1016/j.amepre.2015.02.018</pub-id><pub-id pub-id-type="medline">25960393</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raghupathi</surname><given-names>W</given-names> </name><name name-style="western"><surname>Raghupathi</surname><given-names>V</given-names> </name></person-group><article-title>Big data analytics in healthcare: promise and potential</article-title><source>Health Inf Sci Syst</source><year>2014</year><volume>2</volume><fpage>3</fpage><pub-id pub-id-type="doi">10.1186/2047-2501-2-3</pub-id><pub-id pub-id-type="medline">25825667</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Macieira</surname><given-names>TGR</given-names> </name><name name-style="western"><surname>Chianca</surname><given-names>TCM</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>MB</given-names> </name><etal/></person-group><article-title>Secondary use of standardized nursing care data for advancing nursing science and practice:a systematic review</article-title><source>J Am Med Inform Assoc</source><year>2019</year><month>11</month><day>1</day><volume>26</volume><issue>11</issue><fpage>1401</fpage><lpage>1411</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocz086</pub-id><pub-id pub-id-type="medline">31188439</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Welton</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Harper</surname><given-names>EM</given-names> </name></person-group><article-title>Measuring nursing care value</article-title><source>Nurs Econ</source><year>2016</year><volume>34</volume><issue>1</issue><fpage>7</fpage><lpage>14</lpage><pub-id pub-id-type="medline">27055306</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gengo E Silva Butcher</surname><given-names>RC</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>DA</given-names> </name></person-group><article-title>An integrative review of comprehensive nursing assessment tools developed based on Gordon&#x2019;s eleven functional health patterns</article-title><source>Int J Nurs Knowl</source><year>2021</year><month>10</month><volume>32</volume><issue>4</issue><fpage>294</fpage><lpage>307</lpage><pub-id pub-id-type="doi">10.1111/2047-3095.12321</pub-id><pub-id pub-id-type="medline">33620162</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Company-Sancho</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Gonz&#x00E1;lez-Chord&#x00E1;</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Isabel Orts-Cort&#x00E9;s</surname><given-names>M</given-names> </name></person-group><article-title>The nursing process and total health cost variability: an analysis using machine learning</article-title><source>BMC Nurs</source><year>2025</year><month>07</month><day>1</day><volume>24</volume><issue>1</issue><fpage>738</fpage><pub-id pub-id-type="doi">10.1186/s12912-025-03304-5</pub-id><pub-id pub-id-type="medline">40598233</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Goodfellow</surname><given-names>I</given-names> </name><name name-style="western"><surname>Bengio</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Courville</surname><given-names>A</given-names> </name></person-group><source>Deep Learning</source><year>2016</year><publisher-name>MIT Press</publisher-name><pub-id pub-id-type="other">9780262337373</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Guestrin</surname><given-names>C</given-names> </name></person-group><article-title>XGBoost: a scalable tree boosting system</article-title><source>KDD &#x2019;16: Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source><year>2016</year><publisher-name>Association for Computing Machinery</publisher-name><fpage>785</fpage><lpage>794</lpage><pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mahajan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Uddin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hajati</surname><given-names>F</given-names> </name><name name-style="western"><surname>Moni</surname><given-names>MA</given-names> </name></person-group><article-title>Ensemble learning for disease prediction: a review</article-title><source>Healthcare (Basel)</source><year>2023</year><month>06</month><day>20</day><volume>11</volume><issue>12</issue><fpage>1808</fpage><pub-id pub-id-type="doi">10.3390/healthcare11121808</pub-id><pub-id pub-id-type="medline">37372925</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Berahmand</surname><given-names>K</given-names> </name><name name-style="western"><surname>Daneshfar</surname><given-names>F</given-names> </name><name name-style="western"><surname>Salehi</surname><given-names>ES</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name></person-group><article-title>Autoencoders and their applications in machine learning: a survey</article-title><source>Artif Intell Rev</source><year>2024</year><volume>57</volume><issue>2</issue><fpage>28</fpage><pub-id pub-id-type="doi">10.1007/s10462-023-10662-6</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhatti</surname><given-names>MHR</given-names> </name><name name-style="western"><surname>Javaid</surname><given-names>N</given-names> </name><name name-style="western"><surname>Mansoor</surname><given-names>B</given-names> </name><name name-style="western"><surname>Alrajeh</surname><given-names>N</given-names> </name><name name-style="western"><surname>Aslam</surname><given-names>M</given-names> </name><name name-style="western"><surname>Asad</surname><given-names>M</given-names> </name></person-group><article-title>New hybrid deep learning models to predict cost from healthcare providers in smart hospitals</article-title><source>IEEE Access</source><year>2023</year><volume>11</volume><fpage>136988</fpage><lpage>137010</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2023.3336424</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>M&#x00FC;ller-Staub</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lavin</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Needham</surname><given-names>I</given-names> </name><name name-style="western"><surname>van Achterberg</surname><given-names>T</given-names> </name></person-group><article-title>Nursing diagnoses, interventions and outcomes - application and impact on nursing practice: systematic review</article-title><source>J Adv Nurs</source><year>2006</year><month>12</month><volume>56</volume><issue>5</issue><fpage>514</fpage><lpage>531</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2648.2006.04012.x</pub-id><pub-id pub-id-type="medline">17078827</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Douma</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Rejeb</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Zardoub</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Impact of implementing electronic nursing records on quality and safety indicators in care</article-title><source>Libyan J Med</source><year>2024</year><month>12</month><day>31</day><volume>19</volume><issue>1</issue><fpage>2421625</fpage><pub-id pub-id-type="doi">10.1080/19932820.2024.2421625</pub-id><pub-id pub-id-type="medline">39570988</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yelne</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chaudhary</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dod</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sayyad</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>R</given-names> </name></person-group><article-title>Harnessing the power of AI: a comprehensive review of its impact and challenges in nursing science and healthcare</article-title><source>Cureus</source><year>2023</year><month>11</month><volume>15</volume><issue>11</issue><fpage>e49252</fpage><pub-id pub-id-type="doi">10.7759/cureus.49252</pub-id><pub-id pub-id-type="medline">38143615</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Duncan</surname><given-names>I</given-names> </name><name name-style="western"><surname>Loginov</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ludkovski</surname><given-names>M</given-names> </name></person-group><article-title>Testing alternative regression frameworks for predictive modeling of health care costs</article-title><source>N Am Actuar J</source><year>2016</year><month>01</month><volume>20</volume><issue>1</issue><fpage>65</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1080/10920277.2015.1110491</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Powers</surname><given-names>CA</given-names> </name><name name-style="western"><surname>Meyer</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Roebuck</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Vaziri</surname><given-names>B</given-names> </name></person-group><article-title>Predictive modeling of total healthcare costs using pharmacy claims data: a comparison of alternative econometric cost modeling techniques</article-title><source>Med Care</source><year>2005</year><month>11</month><volume>43</volume><issue>11</issue><fpage>1065</fpage><lpage>1072</lpage><pub-id pub-id-type="doi">10.1097/01.mlr.0000182408.54390.00</pub-id><pub-id pub-id-type="medline">16224298</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mera Flores</surname><given-names>AM</given-names> </name><name name-style="western"><surname>del Busto Bonifaz</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bernal Sobrino</surname><given-names>JL</given-names> </name></person-group><article-title>Evaluaci&#x00F3;n de tres sistemas de ajuste de riesgo como predictores del consumo de medicamentos y productos sanitarios en unidades polivalentes de hospitalizaci&#x00F3;n [Article in Spanish]</article-title><source>Rev Esp Salud P&#x00FA;blica</source><year>2016</year><access-date>2026-08-13</access-date><volume>90</volume><fpage>e40018</fpage><comment><ext-link ext-link-type="uri" xlink:href="https://scielo.isciii.es/scielo.php?script=sci_abstract&#x0026;pid=S1135-57272016000100418">https://scielo.isciii.es/scielo.php?script=sci_abstract&#x0026;pid=S1135-57272016000100418</ext-link></comment></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wells</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>X</given-names> </name><name name-style="western"><surname>Coberley</surname><given-names>CR</given-names> </name><name name-style="western"><surname>Pope</surname><given-names>JE</given-names> </name></person-group><article-title>Integrating well-being information and the multidimensional adaptive prediction process to estimate individual-level future health care expenditure levels</article-title><source>Popul Health Manag</source><year>2016</year><month>12</month><volume>19</volume><issue>6</issue><fpage>429</fpage><lpage>438</lpage><pub-id pub-id-type="doi">10.1089/pop.2015.0184</pub-id><pub-id pub-id-type="medline">27267664</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harrison</surname><given-names>PL</given-names> </name><name name-style="western"><surname>Pope</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Coberley</surname><given-names>CR</given-names> </name><name name-style="western"><surname>Rula</surname><given-names>EY</given-names> </name></person-group><article-title>Evaluation of the relationship between individual well-being and future health care utilization and cost</article-title><source>Popul Health Manag</source><year>2012</year><month>12</month><volume>15</volume><issue>6</issue><fpage>325</fpage><lpage>330</lpage><pub-id pub-id-type="doi">10.1089/pop.2011.0089</pub-id><pub-id pub-id-type="medline">22356589</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Orji</surname><given-names>U</given-names> </name><name name-style="western"><surname>Ukwandu</surname><given-names>E</given-names> </name></person-group><article-title>Machine learning for an explainable cost prediction of medical insurance</article-title><source>Mach Learn Appl</source><year>2024</year><month>03</month><volume>15</volume><fpage>100516</fpage><pub-id pub-id-type="doi">10.1016/j.mlwa.2023.100516</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>J&#x00F6;dicke</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Zellweger</surname><given-names>U</given-names> </name><name name-style="western"><surname>Tomka</surname><given-names>IT</given-names> </name><etal/></person-group><article-title>Prediction of health care expenditure increase: how does pharmacotherapy contribute?</article-title><source>BMC Health Serv Res</source><year>2019</year><month>12</month><day>11</day><volume>19</volume><issue>1</issue><fpage>953</fpage><pub-id pub-id-type="doi">10.1186/s12913-019-4616-x</pub-id><pub-id pub-id-type="medline">31829224</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>JO</given-names> </name><name name-style="western"><surname>Suh</surname><given-names>YM</given-names> </name></person-group><article-title>Comparison of hospital charge prediction models for colorectal cancer patients: Neural network vs. decision tree models</article-title><source>J Korean Med Sci</source><year>2004</year><month>10</month><volume>19</volume><issue>5</issue><fpage>677</fpage><lpage>681</lpage><pub-id pub-id-type="doi">10.3346/jkms.2004.19.5.677</pub-id><pub-id pub-id-type="medline">15483343</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jolicoeur-Martineau</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fatras</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kachman</surname><given-names>T</given-names> </name></person-group><article-title>Generating and imputing tabular data via diffusion and flow-based gradient-boosted trees</article-title><access-date>2026-08-13</access-date><conf-name>27th International Conference on Artificial Intelligence and Statistics (AISTATS 2024)</conf-name><conf-date>May 2-4, 2024</conf-date><conf-loc>Valencia, Spain</conf-loc><fpage>1288</fpage><lpage>1296</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v238/jolicoeur-martineau24a/jolicoeur-martineau24a.pdf">https://proceedings.mlr.press/v238/jolicoeur-martineau24a/jolicoeur-martineau24a.pdf</ext-link></comment></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Garzon</surname><given-names>M</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>LY</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>N</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Garzon</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>CC</given-names> </name><name name-style="western"><surname>Venugopal</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>N</given-names> </name><name name-style="western"><surname>Jana</surname><given-names>K</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>LY</given-names> </name></person-group><article-title>What is dimensionality reduction (DR)?</article-title><source>Dimensionality Reduction in Data Science</source><year>2022</year><publisher-name>Springer</publisher-name><fpage>67</fpage><lpage>77</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-05371-9_3</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rajkomar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Oren</surname><given-names>E</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Scalable and accurate deep learning with electronic health records</article-title><source>NPJ Digit Med</source><year>2018</year><volume>1</volume><fpage>18</fpage><pub-id pub-id-type="doi">10.1038/s41746-018-0029-1</pub-id><pub-id pub-id-type="medline">31304302</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tamang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Milstein</surname><given-names>A</given-names> </name><name name-style="western"><surname>S&#x00F8;rensen</surname><given-names>HT</given-names> </name><etal/></person-group><article-title>Predicting patient &#x201C;cost blooms&#x201D; in Denmark: a longitudinal population-based study</article-title><source>BMJ Open</source><year>2017</year><month>01</month><day>11</day><volume>7</volume><issue>1</issue><fpage>e011580</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2016-011580</pub-id><pub-id pub-id-type="medline">28077408</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Cost variables used to calculate total costs.</p><media xlink:href="nursing_v9i1e93638_app1.docx" xlink:title="DOCX File, 14 KB"/></supplementary-material></app-group></back></article>