<?xml version="1.1" encoding="utf-8"?>
<article xsi:noNamespaceSchemaLocation="http://jats.nlm.nih.gov/publishing/1.1/xsd/JATS-journalpublishing1-mathml3.xsd" dtd-version="1.1" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"><front><journal-meta><journal-id journal-id-type="publisher-id">JCNR</journal-id><journal-title-group><journal-title>Journal of Clinical and Nursing Research</journal-title></journal-title-group><issn>2208-3685</issn><eissn>2208-3693</eissn><publisher><publisher-name>Bio-Byword Scientific Publishing Pty. Ltd.</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.26689/jcnr.v7i3.4957</article-id><article-categories><subj-group subj-group-type="heading"><subject>Article</subject></subj-group></article-categories><title>An Interpretable Prediction Model for Stroke Based on XGBoost and SHAP</title><url>https://artdesignp.com/journal/JCNR/7/3/10.26689/jcnr.v7i3.4957</url><author>FangTianshu,DengJiacheng</author><pub-date pub-type="publication-year"><year>2023</year></pub-date><volume>7</volume><issue>3</issue><history><date date-type="pub"><published-time>2023-05-30</published-time></date></history><abstract>Objective: To establish a stroke prediction and feature analysis model integrating XGBoost and SHAP to aid the clinical diagnosis and prevention of stroke. Methods: Based on the open data set on Kaggle, with the help of data preprocessing and grid parameter optimization, an interpretable stroke risk prediction model was established by integrating XGBoost and SHAP and an explanatory analysis of risk factors was performed. Results: The XGBoost model’s accuracy, sensitivity, specificity, and area under the receiver operating characteristic (ROC) curve (AUC) were 96.71%, 93.83%, 99.59%, and 99.19%, respectively. Our explanatory analysis showed that age, type of residence, and history of hypertension were key factors affecting the incidence of stroke. Conclusion: Based on the data set, our analysis showed that the established model can be used to identify stroke, and our explanatory analysis based on SHAP increases the transparency of the model and facilitates medical practitioners to analyze the reliability of the model.</abstract><keywords/></article-meta></front><body/><back><ref-list><ref id="B1" content-type="article"><label>1</label><element-citation publication-type="journal"><p>The Writing Group of “China Stroke Prevention Report”, 2022, Summary of “China Stroke Prevention Report 2020”. Chinese Journal of Cerebrovascular Diseases, 19(2): 136–144.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B2" content-type="article"><label>2</label><element-citation publication-type="journal"><p>Pandian JD, Gall SL, Kate MP, et al., 2018, Prevention of Stroke: A Global Perspective. The Lancet, 392(10154): 1269–1278.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B3" content-type="article"><label>3</label><element-citation publication-type="journal"><p>Wang Y, 2019, Research on Influencing Factors and Risk Prediction Model of Stroke Based on Big Data, thesis, Guangdong University of Technology.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B4" content-type="article"><label>4</label><element-citation publication-type="journal"><p>Barra S, Almeida I, Caetano F, et al., 2013, Stroke Prediction with an Adjusted R-CHA2DS2VASc Score in a Cohort of Patients with a Myocardial Infarction. Thrombosis Research, 132(2): 293–299.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B5" content-type="article"><label>5</label><element-citation publication-type="journal"><p>Vartiainen E, Laatikainen T, Peltonen M, et al., 2016, Predicting Coronary Heart Disease and Stroke: The FINRISK Calculator. Global Heart, 11(2): 213–216.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B6" content-type="article"><label>6</label><element-citation publication-type="journal"><p>Hou Y, Zhang C, Su Y, 2019, Risk Prediction of Ischemic Stroke Based on Support Vector Machine. Modern Preventive Medicine, 46(15): 2692–2695 + 2700.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B7" content-type="article"><label>7</label><element-citation publication-type="journal"><p>Luo Y, Shao Y, Chen D, 2021, Prediction of Annual Stroke Risk of Ischemic Stroke Based on BiLSTM-Attention Model. Journal of Donghua University (Natural Science Edition), 47(4): 62–68.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B8" content-type="article"><label>8</label><element-citation publication-type="journal"><p>Chen J, Chen Y, Li J, et al., 2022, Stroke Risk Prediction with Hybrid Deep Transfer Learning Framework. IEEE Journal of Biomedical and Health Informatics, 26(1): 411–422.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B9" content-type="article"><label>9</label><element-citation publication-type="journal"><p>Chawla NV, Bowyer KW, Hall LO, et al., 2002, SMOTE: Synthetic Minority Over-Sampling Technique. Journal of Artificial Intelligence Research, 16: 321–357.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B10" content-type="article"><label>10</label><element-citation publication-type="journal"><p>Zhou Z-H, 2021, Ensemble Learning, in Machine Learning, Springer, Singapore, 181–210.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B11" content-type="article"><label>11</label><element-citation publication-type="journal"><p>Chen T, Guestrin C, 2016, Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, August 13–17, 2016: XGBoost: A Scalable Tree Boosting System, ACM, San Francisco California USA, 785–794.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B12" content-type="article"><label>12</label><element-citation publication-type="journal"><p>Gomolin A, Netchiporouk E, Gniadecki R, et al., 2020, Artificial Intelligence Applications in Dermatology: Where Do We Stand?. Frontiers in Medicine, 7: 100.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B13" content-type="article"><label>13</label><element-citation publication-type="journal"><p>Lundberg SM, Lee S-I, 2017, A Unified Approach to Interpreting Model Predictions, in Advances in Neural Information Processing Systems 30, Curran Associates, Inc., 4765–4774.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B14" content-type="article"><label>14</label><element-citation publication-type="journal"><p>Shapley LS, 1952, A Value for N-Person Games, RAND Corporation, Santa Monica, CA.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B15" content-type="article"><label>15</label><element-citation publication-type="journal"><p>Li M, Wang C, Xia B, et al., 2017, Stroke Risk Prediction Model for Health Management Population. Journal of Shandong University (Medical Science), 55(6): 93–97 + 103.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B16" content-type="article"><label>16</label><element-citation publication-type="journal"><p>Boehme AK, Esenwa C, Elkind MSV, 2017, Stroke Risk Factors, Genetics, and Prevention. Circulation Research, 120(3): 472–495.</p><pub-id pub-id-type="doi"/></element-citation></ref><ref id="B17" content-type="article"><label>17</label><element-citation publication-type="journal"><p>Murphy SJX, Werring DJ, 2020, Stroke: Causes and Clinical Features. Medicine, 48(9): 561–566.</p><pub-id pub-id-type="doi"/></element-citation></ref></ref-list></back></article>
