<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Int. J. Public Health</journal-id>
<journal-title-group>
<journal-title>International Journal of Public Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Int. J. Public Health</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">1661-8564</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1609472</article-id>
<article-id pub-id-type="doi">10.3389/ijph.2026.1609472</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>AI powered patient records analysis for injury surveillance in children and adolescents &#x2013; a feasibility study from Switzerland</article-title>
<alt-title alt-title-type="left-running-head">Feer et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/ijph.2026.1609472">10.3389/ijph.2026.1609472</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" equal-contrib="yes">
<name>
<surname>Feer</surname>
<given-names>Sonja</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3225019"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Saaro</surname>
<given-names>Felix Matthias</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zysset</surname>
<given-names>Annina</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>B&#xe4;chli</surname>
<given-names>Mirjam</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nieman</surname>
<given-names>Steffen</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Meier</surname>
<given-names>Delphine</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Seiler</surname>
<given-names>Michelle</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Bogojeska</surname>
<given-names> Jasmina</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn002">
<sup>&#x2021;</sup>
</xref>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Dratva</surname>
<given-names>Julia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="author-notes" rid="fn002">
<sup>&#x2021;</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>Institute of Public Health, Zurich University of Applied Sciences</institution>, <city>Winterthur</city>, <country country="CH">Switzerland</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>Centre for Artificial Intelligence, Zurich University of Applied Sciences</institution>, <city>Winterthur</city>, <country country="CH">Switzerland</country>
</aff>
<aff id="aff3">
<label>3</label>
<institution>Swiss Council for Accident Prevention BFU</institution>, <city>Bern</city>, <country country="CH">Switzerland</country>
</aff>
<aff id="aff4">
<label>4</label>
<institution>Pediatric Emergency Department and Children&#x2019;s Research Centre, University Children&#x2019;s Hospital Zurich</institution>, <city>Zurich</city>, <country country="CH">Switzerland</country>
</aff>
<aff id="aff5">
<label>5</label>
<institution>Medical Faculty, University of Basel</institution>, <city>Basel</city>, <country country="CH">Switzerland</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Sonja Feer, <email xlink:href="mailto:sonja.feer@zhaw.ch">sonja.feer@zhaw.ch</email>
</corresp>
<fn fn-type="other" id="fn003">
<p>This Original Article is part of the IJPH Special Issue &#x201c;Artificial Intelligence (AI) and Public Health&#x201d;</p>
</fn>
<fn fn-type="equal" id="fn001">
<label>&#x2020;</label>
<p>These authors share first authorship</p>
</fn>
<fn fn-type="equal" id="fn002">
<label>&#x2021;</label>
<p>These authors share last authorship</p>
</fn>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-08-19">
<day>19</day>
<month>08</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2026</year>
</pub-date>
<volume>71</volume>
<elocation-id>1609472</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>22</day>
<month>06</month>
<year>2026</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>07</month>
<year>2026</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2026 Feer, Saaro, Zysset, B&#xe4;chli, Nieman, Meier, Seiler, Bogojeska and Dratva.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Feer, Saaro, Zysset, B&#xe4;chli, Nieman, Meier, Seiler, Bogojeska and Dratva</copyright-holder>
<license>
<ali:license_ref start_date="2026-08-19">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Objective</title>
<p>This study investigates the feasibility of using an automated approach to extract injury-related information from narrative text in a paediatric emergency department to improve data basis on injuries among children and adolescents in Switzerland.</p>
</sec>
<sec>
<title>Methods</title>
<p>The dataset comprises paediatric injury cases treated between 2018 and 2022&#xa0;at the University Children&#x2019;s Hospital Z&#xfc;rich emergency department (N &#x3d; 30,876). Model development involved (1) adapting EU-IDB, a domain-specific hierarchical <italic>coding-system</italic>; (2) <italic>manual data annotation</italic>; and (3) <italic>fine tuning</italic> and (4) <italic>evaluation</italic> of a transformer-based model for multi-label text span classification.</p>
</sec>
<sec>
<title>Results</title>
<p>Interrater reliability improved from moderate (&#x3ba; &#x3d; 0.45) to substantial (&#x3ba; &#x3d; 0.73) following adaptation of the coding system. Variable level classification performance was encouraging across three training variants (macro F1: ALL &#x3d; 0.57, IND &#x3d; 0.64, VAR &#x3d; 0.63). Performance declined at lower levels, particularly at level 3 (macro F1: ALL &#x3d; 0.08, IND &#x3d; 0.23). Automated prevalence estimates correlated strongly with manual annotations (&#x3c1; &#x3d; 0.948).</p>
</sec>
<sec>
<title>Conclusion</title>
<p>Automated text span classification of injury-relevant information from electronic patient records shows promising first results to provide valuable information for injury surveillance and prevention, missing in Switzerland. However, more annotated data and detailed validation are needed to draw robust conclusions.</p>
</sec>
</abstract>
<kwd-group>
<kwd>artificial intelligence</kwd>
<kwd>children and adolescents</kwd>
<kwd>injury</kwd>
<kwd>prevention</kwd>
<kwd>surveillance</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declared that financial support was received for this work and/or its publication. The authors declare that this study was funded by the Swiss Council for Accident Prevention BFU, under contract number 23-3901.</funding-statement>
</funding-group>
<counts>
<fig-count count="2"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="31"/>
<page-count count="10"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Injuries in the age group of children and adolescents are a major public health concern. Unintentional injuries cause nearly 90% of total injury cases and are a primary reason for death and disability among the age group of children and adolescents [<xref ref-type="bibr" rid="B1">1</xref>]. Many injured children suffer from lifelong consequences such as disabilities or scarring [<xref ref-type="bibr" rid="B1">1</xref>]. Accidents also demand substantial healthcare resources, as injured children often require acute medical care [<xref ref-type="bibr" rid="B2">2</xref>]. Therefore, prevention of injuries must be a major public health initiative to enhance health and wellbeing of children and reduce the healthcare burden.</p>
<p>In Switzerland, data on child injuries is both incomplete and fragmented [<xref ref-type="bibr" rid="B3">3</xref>]. Existing data from various national reporting systems &#x2013; such as road traffic accident reports, mortality statistics, or poison control services &#x2013; provide only limited insight into the actual extent and causes of accidents and injuries among children and adolescents. Representative data on non-fatal injuries in the home and leisure sector and their causes are insufficiently available or not accessible in Switzerland [<xref ref-type="bibr" rid="B4">4</xref>]. Moreover, existing data sources also vary considerably in the level of detail provided. However, detailed information on injuries, in particular injury circumstances, e.g., place of occurrence, activity when injured or mechanism of injury, are important information for evidence-based injury prevention [<xref ref-type="bibr" rid="B5">5</xref>]. In addition to repeated national surveys, systematically collected clinical data on injuries would provide relevant and timely information.</p>
<p>In many countries, clinical data is used for injury surveillance in children and adolescents. A well-documented example is the Styrian Injury Surveillance System (StLSS) in Austria, which has been collecting detailed data on injuries among child and adolescents [<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B7">7</xref>]. Clinical data is also utilized in the European Injury Database (EU-IDB), to record injuries among children and adolescents in a standardized format across participating countries, based on emergency department (ED) visits [<xref ref-type="bibr" rid="B8">8</xref>]. Similarly, both the Canadian Hospitals Injury Reporting and Prevention Program (CHIRPP) and the National Electronic Injury Surveillance (NEISS) in the USA collect injury data on children and adolescents in EDs [<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>]. These injury surveillance systems based on clinical data from hospitals have traditionally depended on manual data entry and interpretation [<xref ref-type="bibr" rid="B11">11</xref>].</p>
<p>Recent advancements in Machine Learning (ML), namely, Large Language Models (LLMs), offer a promising approach to enhance injury insights by enabling the automatic classification of injury-relevant information from already existing narrative (unstructured) clinical text data. Vallmuur et al. concluded that applying ML techniques, to narrative text data can improve the completeness and timeliness of injury surveillance and thereby support more effective injury prevention policies and practices [<xref ref-type="bibr" rid="B12">12</xref>]. Recently, Azzolina et al. (2023) trained a model based on a random sample of manually annotated narrative diagnosis texts and concluded that ML methods are promising for improving injury surveillance by automatically classifying paediatric ED diagnoses [<xref ref-type="bibr" rid="B13">13</xref>]. Moreover, automated approaches have the potential to identify risk factors as basis for tailored health prevention strategies [<xref ref-type="bibr" rid="B14">14</xref>]. International efforts demonstrate that these automated approaches are highly efficient [<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B15">15</xref>&#x2013;<xref ref-type="bibr" rid="B17">17</xref>].</p>
<p>Automated approaches have been successfully applied for screening or classification of a single injury-related information from narrative clinical text data for surveillance tasks. To our knowledge, so far only one study by Choi et al. has managed to successfully classify a comprehensive set of injury-related information from ED narrative clinical text data [<xref ref-type="bibr" rid="B17">17</xref>]. They fine-tuned an LLM (Llama-2 12B parameters) on a multi-label task to classify five variables with 24 sub-labels from a large injury dataset, covering ED cases from two adult cohorts. However, an attempt at multi-label text-span classification to obtain injury-relevant information from unstructured ED narrative text data using automated approaches has not been published to date.</p>
<p>This study investigates the feasibility and necessary validation steps of this approach, applying an automated approach to classify a comprehensive set of injury-related information, using narrative (unstructured) text data recorded by medical staff in a paediatric ED. It is assumed that potentially valuable injury-related information&#x2013;such as place of occurrence, mechanism of injury, and objects involved - is contained within these narrative text data. Combined with already structured data from electronic patient records (e.g., age and sex), this approach aims to not only improve the surveillance data on injuries in children and adolescents in Switzerland but also to support evidence-based injury prevention measures for children and adolescents.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>Methods</title>
<sec id="s2-1">
<title>Study population and data</title>
<p>Electronic health record data were accessed from ED visits from the University Children&#x2019;s Hospital Zurich (UCHZ). All injury cases of children and adolescents aged 0 to 18, who were treated for injury between 1.1.2018 and 31.12.2022 at the UCHZ ED and for whom a general consent was obtained were included in the study. Data was anonymized by the UCHZ prior to the data transfer, ensuring that no personally identifiable data or identification codes, allowing data to be traced back to a patient, were included in the dataset. Ethics application was submitted to the Ethics Committee of the canton of Zurich (BASEC-Nr.: 2023-01947) and the project was approved.</p>
<p>Data include: Reason for treatment (injury vs. illness) recorded as a structured variable upon admission in the ED, further variables such as age (in years), sex, and month/year of arrival, as well as unstructured narrative data, such as a diagnosis and medical history text. To ensure data accuracy and consistency, a data cleaning process was conducted prior to analysis. The data cleaning involved exclusion of cases with repeated visits for the same injury (n &#x3d; 4,270), missing key variables such as age, sex and month/year of treatment (n &#x3d; 4), fell outside the observation period (n &#x3d; 7), outside age range (n &#x3d; 2) or duplicates (n &#x3d; 6). The final dataset comprised 30,876 injury cases of children and adolescents aged 0&#x2013;18 years, with a mean age of 6.9 years (SD 4.5).</p>
<p>This study followed a structured iterative process for the model development that involved various steps (see <xref ref-type="fig" rid="F1">Figure 1</xref>) and adopted a cyclical workflow, enabling stepwise refinement through continuous learning and evaluation. The iterative process facilitated dynamic adaption to new insights based on evaluation and testing, addressed challenges as they emerged, and progressively enhanced study outcome.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Steps of model development and application. Switzerland, 2018&#x2013;2022.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ijph-71-1609472-g001.tif">
<alt-text content-type="machine-generated">Flowchart illustrating the process of model development and application, including four stages: coding system adaption, annotation, AI model desing and training, and performance evaluation, leading to automatic annotation of the compelte data to determine code frequency within the dataset.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2-2">
<title>Coding system</title>
<p>The internationally established injury coding system EU-IDB was chosen to label injury-relevant information [<xref ref-type="bibr" rid="B18">18</xref>]. It has a hierarchical structure, in which each variable can be further specified in up to two or three levels of labels (e.g., variable <italic>place of occurrence</italic> - Level 1 <italic>private living area</italic> - Level 2 <italic>kitchen</italic>) [<xref ref-type="bibr" rid="B18">18</xref>]. It covers all age groups including relevant injury labels for children and adolescents. After the initial manual annotation phase and preliminary classification results, the coding system was substantially reduced and adapted to better reflect injury patterns in children and adolescents, such as reducing the number of labels and adding child-specific labels, reducing the labels by a third from a total of 1187 to 574. The final coding system comprised the variables <italic>place of occurrence</italic>, <italic>mechanism of injury</italic>, <italic>activity when injured</italic>, <italic>products involved</italic>, <italic>type of sport</italic> and <italic>mode of transportation,</italic> as well as <italic>height,</italic> added by the authors to obtain additional information about the accident circumstances.</p>
</sec>
<sec id="s2-3">
<title>Manual annotation</title>
<p>To create a dataset for the model training and evaluation, a randomly chosen subset of the narrative medical history texts was manually annotated. The variables and their corresponding labels were assigned to identified text spans which provided relevant information for a specific variable of interest (see <xref ref-type="table" rid="T1">Table 1</xref>). For the annotation process, the tool Prodigy (<ext-link ext-link-type="uri" xlink:href="https://prodi.gy/">https://prodi.gy/</ext-link>) was used. Manual annotation was done in two phases by a total of five individuals (annotators): three members of the project team, and two scientific assistants. All annotators received training that included an introduction to the coding tree and coding rules, followed by practical exercises in which randomly selected cases were annotated and discussed in the group.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Example manual annotation narrative medical history text. Switzerland, 2018&#x2013;2022.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="5" align="left">Narrative medical history text</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="5" align="left" style="background-color:#FFFFFF">Child was playing with her sibling on a loft bed in the bedroom and suddenly fell down (approx. 110&#xa0;cm). She suffered RQW on the left eyelid. Cried straight away, was not unconscious, no vomiting so far, played at home</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th colspan="5" align="left">Annotations (identification of text spans and assignment of labels)</th>
</tr>
<tr>
<th align="left">Text span</th>
<th align="left">Label - variable</th>
<th align="left">Lable - level 1</th>
<th align="left">Lable - level 2</th>
<th align="left">Lable - level 3</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Playing</td>
<td align="left">Activity when injured</td>
<td align="left">Leisure activity</td>
<td align="left">Playing</td>
<td align="left">&#x200b;</td>
</tr>
<tr>
<td align="left">Loft bed</td>
<td align="left">Product</td>
<td align="left">Furniture and furnishings</td>
<td align="left">Bed</td>
<td align="left">&#x200b;</td>
</tr>
<tr>
<td align="left">Bedroom</td>
<td align="left">Place of occurrence</td>
<td align="left">Home</td>
<td align="left">Living room, bedroom, children&#x2019;s room</td>
<td align="left">N/A</td>
</tr>
<tr>
<td align="left">Fell down</td>
<td align="left">Mechanism of injury</td>
<td align="left">Falling, stumbling, jumping, pushed</td>
<td align="left">N/A</td>
<td align="left">&#x200b;</td>
</tr>
<tr>
<td align="left">110&#xa0;cm</td>
<td align="left">Height</td>
<td align="left">Falling, stumbling, jumping, pushed<break/>(&#x3e;0.5&#xa0;m and &#x3c;1.5&#xa0;m)</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The degree of agreement between the annotators was analysed twice during the feasibility study. For each test, a sample of 50 cases was randomly selected from the entire dataset and independently manually annotated by each annotator. Cohens &#x3ba; was used to assess interrater reliability, where agreement is defined as both annotators selecting the same variable and producing overlapping text spans. In the first annotation phase, two annotators manually annotated n &#x3d; 1,000 cases using the original IDB coding tree. The first interrater test involved three annotators, of which one was considered as the reference annotator because of his established experience in applying the IDB coding tree. The first interrater reliability showed a moderate agreement (&#x3ba; &#x3d; 0.45). Given the changes to the coding system, two new annotators were trained and involved in the second manual annotation phase, which comprised the manual annotation of an additional n &#x3d; 1,000 cases and re-annotation of the initial n &#x3d; 1,000 cases to reflect updated coding rules. This resulted in a final dataset (n &#x3d; 2,000) single-annotated according to the same coding system. The new annotators performed a second interrater test using the adapted coding tree, resulting in substantial agreement between annotators (&#x3ba; &#x3d; 0.73).</p>
<p>To evaluate the performance of the trained models the manually annotated dataset was randomly divided into a training set (n &#x3d; 1,800) and a test set (n &#x3d; 200).</p>
</sec>
<sec id="s2-4">
<title>Model design and training for classification</title>
<p>The base model, an instruction fine-tuned version of the Mistral 7B v0.3 model (released May 22, 2024), was chosen because of its multilingual ability, openly available weights, and low hardware requirements for the fine-tuning. It is a transformer model with approximately seven billion parameters [<xref ref-type="bibr" rid="B19">19</xref>]. Supervised fine-tuning was performed using Parameter-Efficient Fine-Tuning (PEFT) via the Low-Rank Adaptation (LoRA) method [<xref ref-type="bibr" rid="B20">20</xref>]. Details regarding the hyperparameters can be found on the project page (<ext-link ext-link-type="uri" xlink:href="https://github.com/AIPRA-IS/feasibility">https://github.com/AIPRA-IS/feasibility</ext-link>). Fine-tuning was formulated as a multi-label text-span classification task. During training, the model is provided with ED texts together with a semi-structured prompt that highlights the target variable and instructs the model to extract the corresponding spans. The model then produces a structured output that contains both the extracted text span and its associated hierarchical labels (levels 1&#x2013;3) in the same structure as the annotated dataset in <xref ref-type="table" rid="T1">Table 1</xref>. This output format ensures that the model does not only locate the relevant span but also assigns the appropriate variable. Training was conducted on a single NVIDIA Tesla V100 GPU. Three variations of the fine-tuning setup were implemented to compare different prediction targets:<list list-type="roman-upper">
<list-item>
<p>ALL One model to find text spans and assign the correct variable as well as the labels from the hierarchical coding system.</p>
</list-item>
<list-item>
<p>VAR One model to find text spans and only assign the correct variable.</p>
</list-item>
<list-item>
<p>IND A separate model for each variable (variable-specific model) to find the text spans and assign the labels from the hierarchical coding system for the target variable.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2-5">
<title>Performance evaluation</title>
<p>Model performance was assessed in two steps. First, for each ground-truth text span, we checked whether the model produces a text span sharing at least one word with it. Second, for each such overlapping span, we verified whether the predicted Variable and Level 1&#x2013;3 labels are correct. This enables the generation of a confusion matrix for each label at each level, from which an F1 score is computed that is especially suitable for unbalanced class distributions.</p>
<p>To obtain uncertainty estimates for the performance scores without requiring costly model training, we applied non-parametric bootstrapping by resampling the test documents with replacement over 1,000 iterations, computing the F1 score in each iteration to derive 95% confidence intervals.</p>
</sec>
<sec id="s2-6">
<title>Automatic annotation of the complete data</title>
<p>To determine the frequency of each code within the entire dataset, the IND model, chosen because it performed best across the hierarchical levels, was applied to estimate the prevalence of all hierarchical labels for each variable separately. However, applying a classifier directly to estimate prevalence via na&#xef;ve counting is known to produce biased estimates [<xref ref-type="bibr" rid="B21">21</xref>], as asymmetric misclassification rates cause systematic over- or underestimation of class frequencies. To correct for this, we calibrated the raw prediction counts using the confusion matrix derived from the annotated subset and the Bayesian Classify and Count (BCC) method proposed by von D&#xe4;niken et al. (2024) [<xref ref-type="bibr" rid="B22">22</xref>], which adjust prevalence estimates to account for the classifier&#x2019;s error structure. To assess plausibility, we compared the prevalence estimates obtained from the automatic annotations with those from the manually annotated subset.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>Manually annotated data</title>
<p>Our subset of 2,000 manually annotated narrative medical history texts contained a total of 5,326 annotations, each containing a text span with the corresponding variable and labels from the coding system. <xref ref-type="table" rid="T2">Table 2</xref> shows the utilization of the labels per levels (coding hierarchy). Level 1 and 2 show a high utilization with 92.6% respectively 71.3%, although usage is lower for the variables <italic>product</italic> and <italic>sport</italic> for level 2. In contrast, level 3 shows lower utilization, with 53.9% of the available labels being used.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Number of available (Av), utilized (Util), and annotated (Annot) labels in the annotated dataset (n &#x3d; 2,000 narrative medical history texts) for each level grouped by the variable including the utilization across each level. Switzerland, 2018&#x2013;2022.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Variable</th>
<th colspan="3" align="center">Level 1</th>
<th colspan="3" align="center">Level 2</th>
<th colspan="3" align="center">Level 3</th>
</tr>
<tr>
<th align="left">Av</th>
<th align="left">Util</th>
<th align="left">Annot</th>
<th align="left">Av</th>
<th align="left">Util</th>
<th align="left">Annot</th>
<th align="left">Av</th>
<th align="left">Util</th>
<th align="left">Annot</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Place of occurrence</td>
<td align="right">9</td>
<td align="right">9</td>
<td align="right">750</td>
<td align="right">36</td>
<td align="right">32</td>
<td align="right">649</td>
<td align="right">17</td>
<td align="right">14</td>
<td align="right">96</td>
</tr>
<tr>
<td align="left">Mechanism of injury</td>
<td align="right">10</td>
<td align="right">9</td>
<td align="right">1858</td>
<td align="right">33</td>
<td align="right">27</td>
<td align="right">1213</td>
<td align="right">24</td>
<td align="right">10</td>
<td align="right">95</td>
</tr>
<tr>
<td align="left">Activity when injured</td>
<td align="right">8</td>
<td align="right">8</td>
<td align="right">955</td>
<td align="right">28</td>
<td align="right">25</td>
<td align="right">735</td>
<td align="right">0</td>
<td align="right">0</td>
<td align="right">0</td>
</tr>
<tr>
<td align="left">Product</td>
<td align="right">17</td>
<td align="right">17</td>
<td align="right">1090</td>
<td align="right">76</td>
<td align="right">58</td>
<td align="right">918</td>
<td align="right">197</td>
<td align="right">105</td>
<td align="right">734</td>
</tr>
<tr>
<td align="left">Sport</td>
<td align="right">17</td>
<td align="right">13</td>
<td align="right">338</td>
<td align="right">69</td>
<td align="right">31</td>
<td align="right">290</td>
<td align="right">0</td>
<td align="right">0</td>
<td align="right">0</td>
</tr>
<tr>
<td align="left">Mode of transportation</td>
<td align="right">4</td>
<td align="right">4</td>
<td align="right">169</td>
<td align="right">16</td>
<td align="right">11</td>
<td align="right">169</td>
<td align="right">3</td>
<td align="right">1</td>
<td align="right">4</td>
</tr>
<tr>
<td align="left">Height</td>
<td align="right">3</td>
<td align="right">3</td>
<td align="right">161</td>
<td align="right">0</td>
<td align="right">0</td>
<td align="right">0</td>
<td align="right">0</td>
<td align="right">0</td>
<td align="right">0</td>
</tr>
<tr>
<td align="left">Total &#x23;</td>
<td align="right">68</td>
<td align="right">63</td>
<td align="right">5321</td>
<td align="right">258</td>
<td align="right">184</td>
<td align="right">3974</td>
<td align="right">241</td>
<td align="right">130</td>
<td align="right">929</td>
</tr>
<tr>
<td align="left">Utilization %</td>
<td colspan="3" align="center" style="background-color:#FFFFFF">92.6</td>
<td colspan="3" align="center" style="background-color:#FFFFFF">71.3</td>
<td colspan="3" align="center" style="background-color:#FFFFFF">53</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> also shows the annotation counts per level grouped by variables. <italic>Mechanism of injury</italic> is the most often used variable with a total of 1,859 annotations. Of these n &#x003D; 1,858 have an annotation on level 1, n &#x003D; 1,213 on level 2 and n &#x003D; 95 on level 3.</p>
<p>The number of annotations, and the number of labels used indicate that annotations of variables and level 1 are frequent. On level 2 the number of annotations is still high (n &#x3d; 3,790) but spread across a lot of labels (n &#x3d; 184). Level 3 has fewer annotations (n &#x3d; 929) but still comprising many labels (n &#x3d; 130), resulting in a lower number of annotations per code.</p>
</sec>
<sec id="s3-2">
<title>Performance evaluation</title>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> shows the classification performance across the three training variations (ALL, VAR and IND) including the 95% confidence intervals. Prediction performance on <italic>mechanism of injury,</italic> the variable with the most annotations<italic>,</italic> was high across all three variations (ALL &#x3d; 0.65, VAR &#x3d; 0.795, IND, 0.80) with the smallest confidence intervals. The models VAR and IND showed a good performance on <italic>height</italic> (VAR &#x3d; 0.77, IND &#x3d; 0.78) while ALL exhibited the lowest performance (ALL &#x3d; 0.4) compared to all other variables. The confidence intervals on <italic>height</italic> were also the largest and it is the class with the lowest number of annotations. Fine-tuning a model to predict only the variable (VAR) yielded better average performance across most variables (Macro F1 &#x3d; 0.63), however, it seemed to work worse for the variables <italic>sport</italic>, <italic>mode of transportation</italic>, and <italic>activity when injured</italic>. The best performance on average was achieved by separately fine-tuning individual variable-specific models (IND), with an average Macro F1 score of 0.64.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>F1 score and 95% confident interval of the variable classification for the variants ALL (one model trained on all variables and levels), VAR (one model trained on only the variables without the hierarchical labels), and IND (individually trained model for each variable). Switzerland, 2018&#x2013;2022.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ijph-71-1609472-g002.tif">
<alt-text content-type="machine-generated">Bar chart comparing F1 scores for seven variables &#x2010; place of occurrence, mechanism of injury, activity when injured, product, sport, mode of transportation, and height &#x2010; across three model variants: ALL, VAR, and IND. Error bars indicate the 95% confidence interval of the prediction.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="table" rid="T3">Table 3</xref> shows the macro F1 scores for variable-level and hierarchical code-level prediction across the three model configurations (ALL, VAR, IND). At the variable level, the IND and VAR models performed comparably, and both outperformed the ALL model (mean F1: IND &#x3d; 0.64, VAR &#x3d; 0.63, ALL &#x3d; 0.57). Gains were most pronounced for specific variables. The IND model substantially improved prediction of <italic>height</italic> (F1 &#x3d; 0.78 vs. 0.39 for ALL) and <italic>mechanism of injury</italic> (F1 &#x3d; 0.80 vs. 0.65), while the VAR model showed the strongest performance for product (F1 &#x3d; 0.67) and achieved F1 &#x3d; 0.78 for <italic>height</italic>. Across all models, prediction performance declined with each lower level of the label hierarchy. At level 1, the IND model again outperformed ALL (mean F1: 0.40 vs. 0.32). At level 2, performance was considerably lower across all variables (IND mean F &#x3d; 0.26, All mean F1 &#x3d; 0.23), with <italic>mechanism of injury</italic> showing the greatest benefit from the IND configuration (F1 &#x3d; 0.42 vs. 0.23). Level 3 prediction was largely poor or not applicable. <italic>Mechanism of injury</italic> achieved F1 &#x3d; 0.00 under the ALL model and only 0.28 under IND, while <italic>mode of transportation</italic> returned F1 &#x3d; 0.00 under both configurations. Across all three model types, <italic>mechanism of injury</italic> and <italic>mode of transportation</italic> at the variable level, and <italic>height</italic> at level 1, showed the strongest absolute performance, whereas level 3 prediction across all variables remained the most challenging aspect of the classification task.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>F1 scores and 95% confidents intervals for the variants ALL (one model trained on all variables and levels) and IND (individually trained model for each variable) across the Variable and Label 1 to 3 aggregated by variable. Switzerland, 2018&#x2013;2022.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">&#x200b;</th>
<th colspan="3" align="center">Variable</th>
<th colspan="2" align="center">Level 1</th>
<th colspan="2" align="center">Level 2</th>
<th colspan="2" align="center">Level 3</th>
</tr>
<tr>
<th align="left">Variable</th>
<th align="left">ALL</th>
<th align="left">IND</th>
<th align="left">VAR</th>
<th align="left">ALL</th>
<th align="left">IND</th>
<th align="left">ALL</th>
<th align="left">IND</th>
<th align="left">ALL</th>
<th align="left">IND</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">Place of occurrence</td>
<td align="right">0.43</td>
<td align="right">0.40</td>
<td align="right">0.52</td>
<td align="right">0.28</td>
<td align="right">0.25</td>
<td align="right">0.26</td>
<td align="right">0.19</td>
<td align="right">0.17</td>
<td align="right">0.41</td>
</tr>
<tr>
<td align="right">0.31&#x2013;0.55</td>
<td align="right">0.30&#x2013;0.50</td>
<td align="right">0.42&#x2013;0.62</td>
<td align="right">0.16&#x2013;0.43</td>
<td align="right">0.16&#x2013;0.36</td>
<td align="right">0.15&#x2013;0.36</td>
<td align="right">0.10&#x2013;0.28</td>
<td align="right">0.00&#x2013;0.50</td>
<td align="right">0.00&#x2013;0.83</td>
</tr>
<tr>
<td rowspan="2" align="left">Mechanism of injury</td>
<td align="right">0.65</td>
<td align="right">0.80</td>
<td align="right">0.80</td>
<td align="right">0.44</td>
<td align="right">0.57</td>
<td align="right">0.23</td>
<td align="right">0.42</td>
<td align="right">0.00</td>
<td align="right">0.28</td>
</tr>
<tr>
<td align="right">0.58&#x2013;0.71</td>
<td align="right">0.75&#x2013;0.86</td>
<td align="right">0.74&#x2013;0.85</td>
<td align="right">0.31&#x2013;0.60</td>
<td align="right">0.42&#x2013;0.79</td>
<td align="right">0.18&#x2013;0.29</td>
<td align="right">0.32&#x2013;0.52</td>
<td align="right">0.00&#x2013;0.00</td>
<td align="right">0.00&#x2013;0.59</td>
</tr>
<tr>
<td rowspan="2" align="left">Activity when injured</td>
<td align="right">0.65</td>
<td align="right">0.66</td>
<td align="right">0.58</td>
<td align="right">0.3</td>
<td align="right">0.33</td>
<td align="right">0.23</td>
<td align="right">0.26</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
</tr>
<tr>
<td align="right">0.56&#x2013;0.73</td>
<td align="right">0.57&#x2013;0.74</td>
<td align="right">0.48&#x2013;0.67</td>
<td align="right">0.22&#x2013;0.41</td>
<td align="right">0.23&#x2013;0.45</td>
<td align="right">0.15&#x2013;0.33</td>
<td align="right">0.17&#x2013;0.37</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
</tr>
<tr>
<td rowspan="2" align="left">Product</td>
<td align="right">0.57</td>
<td align="right">0.60</td>
<td align="right">0.67</td>
<td align="right">0.33</td>
<td align="right">0.36</td>
<td align="right">0.28</td>
<td align="right">0.27</td>
<td align="right">0.14</td>
<td align="right">0.23</td>
</tr>
<tr>
<td align="right">0.48&#x2013;0.66</td>
<td align="right">0.52&#x2013;0.67</td>
<td align="right">0.60&#x2013;0.74</td>
<td align="right">0.25&#x2013;0.42</td>
<td align="right">0.25&#x2013;0.46</td>
<td align="right">0.20&#x2013;0.37</td>
<td align="right">0.19&#x2013;0.35</td>
<td align="right">0.08&#x2013;0.20</td>
<td align="right">0.15&#x2013;0.31</td>
</tr>
<tr>
<td rowspan="2" align="left">Sport</td>
<td align="right">0.62</td>
<td align="right">0.57</td>
<td align="right">0.49</td>
<td align="right">0.22</td>
<td align="right">0.23</td>
<td align="right">0.13</td>
<td align="right">0.13</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
</tr>
<tr>
<td align="right">0.44&#x2013;0.76</td>
<td align="right">0.40&#x2013;0.73</td>
<td align="right">0.33&#x2013;0.63</td>
<td align="right">0.10&#x2013;0.36</td>
<td align="right">0.11&#x2013;0.39</td>
<td align="right">0.09&#x2013;0.21</td>
<td align="right">0.07&#x2013;0.21</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
</tr>
<tr>
<td rowspan="2" align="left">Mode of transportation</td>
<td align="right">0.69</td>
<td align="right">0.69</td>
<td align="right">0.57</td>
<td align="right">0.43</td>
<td align="right">0.36</td>
<td align="right">0.22</td>
<td align="right">0.28</td>
<td align="right">0.00</td>
<td align="right">0.00</td>
</tr>
<tr>
<td align="right">0.50&#x2013;0.85</td>
<td align="right">0.53&#x2013;0.84</td>
<td align="right">0.36&#x2013;0.75</td>
<td align="right">0.29&#x2013;0.83</td>
<td align="right">0.25&#x2013;0.69</td>
<td align="right">0.10&#x2013;0.40</td>
<td align="right">0.17&#x2013;0.45</td>
<td align="right">0.00&#x2013;0.00</td>
<td align="right">0.00&#x2013;0.00</td>
</tr>
<tr>
<td rowspan="2" align="left">Height</td>
<td align="right">0.39</td>
<td align="right">0.78</td>
<td align="right">0.78</td>
<td align="right">0.24</td>
<td align="right">0.68</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
</tr>
<tr>
<td align="right">0.12&#x2013;0.62</td>
<td align="right">0.60&#x2013;0.93</td>
<td align="right">0.57&#x2013;0.93</td>
<td align="right">0.00&#x2013;0.50</td>
<td align="right">0.43&#x2013;0.88</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
</tr>
<tr>
<td align="left">Average</td>
<td align="right">0.57</td>
<td align="right">0.64</td>
<td align="right">0.63</td>
<td align="right">0.32</td>
<td align="right">0.40</td>
<td align="right">0.23</td>
<td align="right">0.26</td>
<td align="right">0.08</td>
<td align="right">0.23</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-3">
<title>Automatic annotation of the complete data</title>
<p>
<xref ref-type="table" rid="T4">Table 4</xref> shows the prevalence (including the estimated standard deviation) of the most relevant labels (prevalence at least 1%) of level 1 and 2 labels grouped by their variable within the annotated subset (n &#x3d; 2,000) and compares it to the entire dataset (n &#x3d; 30,876). The prevalence from the manual annotations is compared to the prevalence derived from the predictions obtained by the IND models on the full dataset. This comparison yields a Spearman correlation coefficient of &#x3c1; &#x3d; 0.948 across the labels listed in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Prevalence estimates of the labels within the annotated dataset (n &#x3d; 2,000) and the predictions of the entire dataset, including their difference, from the ED of the University Children&#x2019;s Hospital Zurich (UCZH) (n &#x3d; 30,876). Switzerland, 2018&#x2013;2022.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Variable</th>
<th colspan="2" align="left">Annotated Dataset</th>
<th colspan="2" align="left">Entire Dataset</th>
<th align="left">&#x200b;</th>
</tr>
<tr>
<th align="left">&#x200b;</th>
<th align="left">Prevalence</th>
<th align="left">Std. Dev.</th>
<th align="left">Prevalence</th>
<th align="left">Std. Dev.</th>
<th align="left">Diff</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<th colspan="6" align="left">Mechanism of injury</th>
</tr>
<tr>
<td align="left">1 &#x2013; Blunt force</td>
<td align="right">30%</td>
<td align="right">&#xb1;1.0%</td>
<td align="right">26%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">4%</td>
</tr>
<tr>
<td align="left">&#x2003;1.2 &#x2013; Contact with object/material/element</td>
<td align="right">16%</td>
<td align="right">&#xb1;0.8%</td>
<td align="right">18%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">&#x2212;2%</td>
</tr>
<tr>
<td align="left">&#x2003;1.3 &#x2013; Contact with person</td>
<td align="right">6%</td>
<td align="right">&#xb1;0.5%</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">4%</td>
</tr>
<tr>
<td align="left">2 &#x2013; Falling, stumbling, jumping, pushed</td>
<td align="right">49%</td>
<td align="right">&#xb1;1.1%</td>
<td align="right">46%</td>
<td align="right">&#xb1;0.3%</td>
<td align="right">3%</td>
</tr>
<tr>
<td align="left">&#x2003;2.1 &#x2013; Tripping, stumbling</td>
<td align="right">6%</td>
<td align="right">&#xb1;0.5%</td>
<td align="right">4%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">2%</td>
</tr>
<tr>
<td align="left">&#x2003;2.2 &#x2013; Slipping, sliding</td>
<td align="right">5%</td>
<td align="right">&#xb1;0.5%</td>
<td align="right">3%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">2%</td>
</tr>
<tr>
<td align="left">&#x2003;2.6 &#x2013; Fall from/with sports/transport equipment</td>
<td align="right">4%</td>
<td align="right">&#xb1;0.4%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">3%</td>
</tr>
<tr>
<td align="left">4 &#x2013; Thermal mechanism</td>
<td align="right">3%</td>
<td align="right">&#xb1;0.4%</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">1%</td>
</tr>
<tr>
<td align="left">&#x2003;4.1 &#x2013; Heat</td>
<td align="right">3%</td>
<td align="right">&#xb1;0.4%</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">1%</td>
</tr>
<tr>
<td align="left">7 &#x2013; Physical overexerting</td>
<td align="right">8%</td>
<td align="right">&#xb1;0.6%</td>
<td align="right">6%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">2%</td>
</tr>
<tr>
<td align="left">&#x2003;7.1 &#x2013; Acute overexertion</td>
<td align="right">8%</td>
<td align="right">&#xb1;0.6%</td>
<td align="right">6%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">2%</td>
</tr>
<tr>
<th colspan="6" align="left">Product</th>
</tr>
<tr>
<td align="left">5 &#x2013; Furniture and furnishings</td>
<td align="right">9%</td>
<td align="right">&#xb1;0.6%</td>
<td align="right">8%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">1%</td>
</tr>
<tr>
<td align="left">&#x2003;5.1 &#x2013; Bed (incl. parts/components)</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.3%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">1%</td>
</tr>
<tr>
<td align="left">&#x2003;5.2 &#x2013; Chair, bench, armchair, sofa</td>
<td align="right">4%</td>
<td align="right">&#xb1;0.4%</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">2%</td>
</tr>
<tr>
<td align="left">6 &#x2013; Baby equipment or children&#x2019;s product</td>
<td align="right">10%</td>
<td align="right">&#xb1;0.7%</td>
<td align="right">16%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">&#x2212;6%</td>
</tr>
<tr>
<td align="left">&#x2003;6.1 &#x2013; Baby or children&#x2019;s item</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.3%</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">0%</td>
</tr>
<tr>
<td align="left">&#x2003;6.2 &#x2013; Toys</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.0%</td>
<td align="right">0%</td>
</tr>
<tr>
<td align="left">&#x2003;6.3 &#x2013; Play equipment</td>
<td align="right">7%</td>
<td align="right">&#xb1;0.6%</td>
<td align="right">8%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">&#x2212;1%</td>
</tr>
<tr>
<td align="left">10 &#x2013; Sports equipment</td>
<td align="right">7%</td>
<td align="right">&#xb1;0.6%</td>
<td align="right">5%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">2%</td>
</tr>
<tr>
<td align="left">14 &#x2013; Building structure or component</td>
<td align="right">9%</td>
<td align="right">&#xb1;0.6%</td>
<td align="right">5%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">4%</td>
</tr>
<tr>
<td align="left">16 &#x2013; Material (n.e.c.)</td>
<td align="right">3%</td>
<td align="right">&#xb1;0.4%</td>
<td align="right">5%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">&#x2212;2%</td>
</tr>
<tr>
<th colspan="6" align="left">Activity when injured</th>
</tr>
<tr>
<td align="left">3 &#x2013; Education (incl. school sport, university sport)</td>
<td align="right">6%</td>
<td align="right">&#xb1;0.5%</td>
<td align="right">3%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">3%</td>
</tr>
<tr>
<td align="left">4 &#x2013; Sport and physical activity</td>
<td align="right">18%</td>
<td align="right">&#xb1;0.9%</td>
<td align="right">12%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">6%</td>
</tr>
<tr>
<td align="left">5 &#x2013; Leisure activity</td>
<td align="right">16%</td>
<td align="right">&#xb1;0.8%</td>
<td align="right">15%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">1%</td>
</tr>
<tr>
<td align="left">&#x2003;5.2 &#x2013; Playing</td>
<td align="right">14%</td>
<td align="right">&#xb1;0.8%</td>
<td align="right">13%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">1%</td>
</tr>
<tr>
<td align="left">8 &#x2013; Being on the move, journeys/transports</td>
<td align="right">3%</td>
<td align="right">&#xb1;0.4%</td>
<td align="right">5%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">&#x2212;2%</td>
</tr>
<tr>
<th colspan="6" align="left">Place of occurrence</th>
</tr>
<tr>
<td align="left">1 &#x2013; Home</td>
<td align="right">13%</td>
<td align="right">&#xb1;0.8%</td>
<td align="right">3%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">10%</td>
</tr>
<tr>
<td align="left">&#x2003;1.2 &#x2013; Living room, bedroom, children&#x2019;s room</td>
<td align="right">6%</td>
<td align="right">&#xb1;0.5%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">5%</td>
</tr>
<tr>
<td align="left">4 &#x2013; Day care, school, educational area</td>
<td align="right">12%</td>
<td align="right">&#xb1;0.7%</td>
<td align="right">6%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">6%</td>
</tr>
<tr>
<td align="left">&#x2003;4.1 &#x2013; School or educational establishment</td>
<td align="right">5%</td>
<td align="right">&#xb1;0.5%</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">3%</td>
</tr>
<tr>
<td align="left">&#x2003;4.2 &#x2013; Day care facility</td>
<td align="right">3%</td>
<td align="right">&#xb1;0.4%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.0%</td>
<td align="right">2%</td>
</tr>
<tr>
<td align="left">5 &#x2013; Sports and athletics area</td>
<td align="right">7%</td>
<td align="right">&#xb1;0.6%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">6%</td>
</tr>
<tr>
<td align="left">6 &#x2013; Transport area</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.3%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.0%</td>
<td align="right">1%</td>
</tr>
<tr>
<td align="left">10 &#x2013; Recreational area</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.3%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.0%</td>
<td align="right">1%</td>
</tr>
<tr>
<th colspan="6" align="left">Sport</th>
</tr>
<tr>
<td align="left">1 &#x2013; Team ball sport</td>
<td align="right">8%</td>
<td align="right">&#xb1;0.6%</td>
<td align="right">6%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">2%</td>
</tr>
<tr>
<td align="left">&#x2003;1.9 &#x2013; Football (soccer)</td>
<td align="right">6%</td>
<td align="right">&#xb1;0.5%</td>
<td align="right">4%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">2%</td>
</tr>
<tr>
<td align="left">8 &#x2013; Artistic gymnastics and sport gymnastics with equipment</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.2%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">0%</td>
</tr>
<tr>
<td align="left">17 &#x2013; Wheeled sport (non-motorised)</td>
<td align="right">4%</td>
<td align="right">&#xb1;0.4%</td>
<td align="right">1%</td>
<td align="right">&#xb1;0.0%</td>
<td align="right">3%</td>
</tr>
<tr>
<th colspan="6" align="left">Height</th>
</tr>
<tr>
<td align="left">1 &#x2013; Falling, stumbling, jumping, pushed (&#x3c;0.5&#xa0;m)</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.3%</td>
<td align="right">4%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">&#x2212;2%</td>
</tr>
<tr>
<td align="left">2 &#x2013; Falling, stumbling, jumping, pushed (&#x2265;0.5&#xa0;m and &#x3c;1.5&#xa0;m)</td>
<td align="right">4%</td>
<td align="right">&#xb1;0.4%</td>
<td align="right">4%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">0%</td>
</tr>
<tr>
<td align="left">3 &#x2013; Falling, stumbling, jumping, pushed (&#x2265;1.5&#xa0;m)</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.3%</td>
<td align="right">2%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">0%</td>
</tr>
<tr>
<th colspan="6" align="left">Mode of transportation</th>
</tr>
<tr>
<td align="left">1 &#x2013; Transport means without motor drive</td>
<td align="right">8%</td>
<td align="right">&#xb1;0.6%</td>
<td align="right">4%</td>
<td align="right">&#xb1;0.1%</td>
<td align="right">4%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>For most labels, the difference between the prevalence estimations obtained from the annotated dataset and those obtained from the entire dataset is small. However, notable absolute discrepancies exist for certain labels, such as <italic>product (baby equipment or children&#x2019;s product: 6%), activity when injured (sport and physical activity: 6%), place of occurrence (home: 10%)</italic>.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>This study investigated the feasibility of utilizing an automated approach to extract multiple injury-related information from narrative (unstructured) medical history texts from a paediatric ED. Three fine-tuning model variants were evaluated: a single model predicting all variables and labels simultaneously (ALL), a model predicting only the variable label (VAR), and separate variable-specific models (IND). Variable-level extraction performance was encouraging across all three variants, demonstrating that the narrative texts contain sufficient injury-relevant information for automated extraction, consistent with our prior utility evaluation [<xref ref-type="bibr" rid="B3">3</xref>].</p>
<p>Applying the IND model to the full dataset produced prevalence estimates that showed a strong rank correlation with the manually annotated subset (&#x3c1; &#x3d; 0.948), and are broadly consistent with national injury projections and international data, for instance, <italic>falls</italic> as the leading mechanism of injury in children and adolescents [<xref ref-type="bibr" rid="B23">23</xref>, <xref ref-type="bibr" rid="B24">24</xref>]. Notable absolute discrepancies exist for certain labels (<italic>product, activity when injured and place of occurrence).</italic> While the precise sources cannot be determined with certainty, these discrepancies may reflect a combination of sampling bias in the annotated subset, systematic under- or over-representation of certain injury contexts in the narrative texts, and the model&#x2019;s tendency to misclassify semantically similar labels, particularly in categories where training annotations were sparse. The consistency with established epidemiological patterns is encouraging as a plausibility check, but larger and more representative annotated datasets are needed before the model&#x2019;s use for surveillance purposes.</p>
<p>The hierarchically structured coding system, with its large number of labels across multiple levels, presented a fundamental challenge for automated classification. Adapting the EU-IDB coding system to the paediatric context, was a key prerequisite for model development. Reducing the breath of labels not relevant to children and adolescents improved interrater reliability. However, performance degraded consistently at lower levels of the coding hierarchy across all model variants, with the sharpest drop observed at level 3 (see <xref ref-type="fig" rid="F2">Figure 2</xref>). Reflecting on one side the structural limitation of the data providing limited level of detail present in the ED narrative texts for some variables (e.g., <italic>injury circumstances</italic>) and greater detail for others (e.g., <italic>product</italic>). Highlighting a fundamental tension between the granularity of information needed for evidence-based injury prevention and the level of detail routinely captured in clinical documentation [<xref ref-type="bibr" rid="B25">25</xref>]. On the other side, the small number of manual annotations (n &#x3d; 2,000) illustrates the challenge of &#x201c;data annotation bottleneck&#x201d; as identified as a key obstacle in NLP-based automatic extraction approaches in other clinical text data more broadly [<xref ref-type="bibr" rid="B26">26</xref>]. Small training sets also increase the risk of overfitting, limiting the model&#x2019;s ability to learn robust patterns from rare labels [<xref ref-type="bibr" rid="B27">27</xref>]. In addition, confidence intervals indicate substantial uncertainty for variables with few annotations. This is an expected consequence of the small test set (n &#x3d; 200) and the uneven label distribution, limiting the strength of conclusions that can be drawn. The number of manual annotations in our feasibility study produced a dataset of sufficient quality for model training to establish important first insights. However, a larger annotated dataset with sufficient examples for lower levels is needed to draw robust conclusions about model performance across the full label space. Underscoring the trade-off between coding granularity, necessary for detailed injury surveillance, and the annotation effort required to achieve reliable model performance at lower levels.</p>
<p>Historically, injury surveillance has relied on manual data entry and interpretation. Applying automated extraction to ED electronic records has the potential to establish comprehensive injury surveillance systems that minimise manual effort while delivering high accuracy [<xref ref-type="bibr" rid="B15">15</xref>&#x2013;<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B28">28</xref>, <xref ref-type="bibr" rid="B29">29</xref>]. Such integration could transform how injury data are collected and used, enabling timely surveillance and more efficient allocation of healthcare resources, ultimately strengthening public health responses and improving patient health outcomes [<xref ref-type="bibr" rid="B30">30</xref>]. International efforts in countries such as Canada and Italy are also attempting to implement AI-based approaches for paediatric injury surveillance [<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B13">13</xref>]. However, unlike Switzerland, those countries already possess substantial historically annotated injury datasets to draw on.</p>
<p>To our knowledge, this is one of the first studies to classify multiple injury-related text spans from paediatric ED records using a hierarchical coding system, extending beyond the single-variable classification approaches that characterise most prior work [<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B17">17</xref>]. Preserving text span context improves interpretability and opens opportunities for automated data enrichment. Future integration of explainable AI techniques for LLMs could further strengthen transparency and clinical trust in model outputs [<xref ref-type="bibr" rid="B31">31</xref>], addressing one of the key barriers to adopting AI in clinical and public health settings [<xref ref-type="bibr" rid="B30">30</xref>].</p>
<sec id="s4-1">
<title>Limitations</title>
<p>Several limitations of this feasibility study should be acknowledged:</p>
<p>First, the complexity of the EU-IDB coding tree posed a substantial challenge for automatic classification. Despite reducing the original coding system, the remaining hierarchical structure proved difficult for the models to navigate, particularly at lower levels where prediction performance declined markedly. Future work should explore whether further reduction of the coding tree, guided by annotation density and surveillance priorities, could improve classification performance without sacrificing the detail need for evidence-based injury prevention.</p>
<p>Second, the small number of manual annotations limited the size of the training and test datasets. To draw robust conclusions about the automated extraction, further evaluation using a larger test dataset with sufficient sample size for all analysed label categories is required. This highlights the necessity of increasing the manually annotated data to provide the basis for further development and enhancements of automated approaches for injury monitoring.</p>
<p>Third, the choice of model architecture and training setup involved trade-offs that may have affected performance. Only a single base model was used for the evaluation. Future work should compare different model sizes and output generation approaches such as reasoning strategies, unavailable at the time of the study. In addition, this baseline should be used to identify which narrative annotations would most improve classification performance. In this way, the manual annotation effort can be optimized so that only as much data is manually annotated as necessary to achieve a useful prediction performance.</p>
<p>Finally, model evaluation was limited in scope. Performance was assessed on a small test set, without cross-validation, which affects the robustness of the reported estimates for labels with a small number of samples in the test set. Further evaluation using human annotations of a blinded sample is needed to distinguish whether limitations stem from insufficient annotations, fine-tuning shortcomings, or inherent text ambiguity.</p>
</sec>
<sec id="s4-2">
<title>Conclusion</title>
<p>Our feasibility study provides important first insights on the automatic identification and multi-label classification of text spans containing paediatric injury-relevant information in Switzerland. This approach has the potential to improve currently insufficient injury surveillance and evidence-based prevention in children and adolescents in Switzerland. Prerequisites for the successful application of automated systems include digital access to ED patient records and detailed narrative text input from medical staff.</p>
<p>The results demonstrate that reducing its complexity and adding child-specific labels to the EU-IDB coding tree, enhanced model performance. The remaining hierarchical structure continued to pose challenges. Future work should expand the annotated dataset, systematically evaluate alternative models, integrate most recent methodological advances, and explore continuous active learning strategies to optimise the annotation effort. Model evaluation using larger test sets, evaluation across multiple representative test sets, cross-validation when feasible, and human validation will be required before it can be considered as ready for implementation in child injury surveillance in Switzerland. Data privacy, ethical considerations and transparency are other challenges that must be carefully addressed to ensure a secure and responsible implementation of the approach in using clinical data. Considering these challenges and once the necessary advancements for implementation are achieved, applying automated methods for text span classification while preserving injury context information could enable timely surveillance of injury trends allowing rapid responses to emerging injury patterns and circumstances and thereby supporting evidence-based prevention strategies aimed at improving public health outcomes in children and adolescents.</p>
</sec>
</sec>
</body>
<back>
<sec sec-type="ethics-statement" id="s5">
<title>Ethics statement</title>
<p>The studies involving humans were approved by the Ethics Committee of the canton of Zurich (BASEC-Nr.: 2023-01947). The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation in this study was provided by the participants&#x2019; legal guardians/next of kin.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>All authors actively participated in different stages of the study. JD and JB acquired the funding and developed the study design. SF, JD and AZ were responsible for the planning and manual labelling of the study, while FS and JB performed the AI model design, training and performance evaluation. DM, SN and MB were involved as injury experts and MS was involved as paediatric ED expert in all stages of the study. SF and FS designed the paper and wrote a first draft of the manuscript. All authors contributed to the final manuscript and approved the submitted version.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that they do not have any conflicts of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declared that generative AI was not used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Peden</surname>
<given-names>MM</given-names>
</name>
</person-group>. <source>World Report on Child Injury Prevention</source>. <publisher-name>Geneva: UNICEF and WHO</publisher-name> (<year>2008</year>).</mixed-citation>
</ref>
<ref id="B2">
<label>2.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zonfrillo</surname>
<given-names>MR</given-names>
</name>
<name>
<surname>Spicer</surname>
<given-names>RS</given-names>
</name>
<name>
<surname>Lawrence</surname>
<given-names>BA</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>TR</given-names>
</name>
</person-group>. <article-title>Incidence and costs of injuries to children and adults in the United States</article-title>. <source>Inj Epidemiol</source> (<year>2018</year>) <volume>5</volume>(<issue>1</issue>):<fpage>37</fpage>. <pub-id pub-id-type="doi">10.1186/s40621-018-0167-6</pub-id>
<pub-id pub-id-type="pmid">30294767</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<label>3.</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Feer</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Zysset</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Dratva</surname>
<given-names>J</given-names>
</name>
</person-group>. <source>Machbarkeitsstudie Datenerhebung Kinder-Und Jugendunf&#xe4;lle</source>. <publisher-loc>Winterthur</publisher-loc>: <publisher-name>Zurich University of Applied Sciences ZHAW</publisher-name> (<year>2023</year>). <pub-id pub-id-type="doi">10.21256/zhaw-2439</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<label>4.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dratva</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zysset</surname>
<given-names>A</given-names>
</name>
</person-group>. <source>K&#xf6;rperliche Gesundheit und Entwicklung</source> (<year>2020</year>) <fpage>77</fpage>&#x2013;<lpage>107</lpage>. </mixed-citation>
</ref>
<ref id="B5">
<label>5.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scott</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Harrison</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Purdie</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Bain</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Najman</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Nixon</surname>
<given-names>J</given-names>
</name>
<etal/>
</person-group> <article-title>The properties of the international classification of the external cause of injury when used as an instrument for injury prevention research</article-title>. <source>Inj Prev</source> (<year>2006</year>) <volume>12</volume>(<issue>4</issue>):<fpage>253</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1136/ip.2006.011510</pub-id>
<pub-id pub-id-type="pmid">16887948</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<label>6.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Till</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Spitzer</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Fanninger</surname>
<given-names>E</given-names>
</name>
</person-group>. <article-title>Verein GROSSE SCH&#xdc;TZEN KLEINE</article-title>. <source>P&#xe4;diatr P&#xe4;dol</source> (<year>2018</year>) <volume>53</volume>(<issue>1</issue>):<fpage>12</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1007/s00608-017-0534-5</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<label>7.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peter</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Gudula</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Kleine</surname>
<given-names>GS</given-names>
</name>
</person-group>. <article-title>STISS&#x2014;The Styrian injury surveillance system: a hospital database on injuries to support safe community work</article-title>. <source>Inj Prev</source> (<year>2012</year>) <volume>18</volume>(<issue>Suppl. 1</issue>):<fpage>A241.2</fpage>&#x2013;<lpage>A241</lpage>. <pub-id pub-id-type="doi">10.1136/injuryprev-2012-040590w.57</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<label>8.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bauer</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Giustini</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Nijman</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Bejko</surname>
<given-names>D</given-names>
</name>
</person-group>. <article-title>Data collection and analysis on the burden of injuries: lessons learned from the IDB network</article-title>. <source>Eur J Public Health</source> (<year>2023</year>) <volume>33</volume>(<issue>Suppl. 2</issue>):<fpage>ckad160</fpage>&#x2013;<lpage>448</lpage>. <pub-id pub-id-type="doi">10.1093/eurpub/ckad160.448</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<label>9.</label>
<mixed-citation publication-type="book">
<collab>Public Health Agency of Canada</collab>. <source>Canadian Hospitals Injury Reporting and Prevention Program (CHIRPP)</source>. <publisher-loc>Ottawa</publisher-loc>: <publisher-name>Government of Canada</publisher-name>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.canada.ca/en/public-health/services/injury-prevention/canadian-hospitals-injury-reporting-prevention-program.html">https://www.canada.ca/en/public-health/services/injury-prevention/canadian-hospitals-injury-reporting-prevention-program.html</ext-link> (Accessed December 18, 2025)</comment>.</mixed-citation>
</ref>
<ref id="B10">
<label>10.</label>
<mixed-citation publication-type="book">
<collab>United States Consumer Product Safety Commission</collab>. In: <source>National Electronic Injury Surveillance System (NEISS) Injury Data</source>. <publisher-loc>Bethesda, MD</publisher-loc>: <publisher-name>U.S. Consumer Product Safety Commission</publisher-name>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cpsc.gov/Research--Statistics/NEISS-Injury-Data">https://www.cpsc.gov/Research--Statistics/NEISS-Injury-Data</ext-link> (Accessed December 18, 2025)</comment>.</mixed-citation>
</ref>
<ref id="B11">
<label>11.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Celik</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>EWJ</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>E</given-names>
</name>
<name>
<surname>Rosenfield</surname>
<given-names>D</given-names>
</name>
</person-group>. <article-title>AI-driven injury reporting in pediatric emergency departments</article-title>. <source>JAMA Netw Open</source> (<year>2025</year>) <volume>8</volume>(<issue>7</issue>):<fpage>e2524154</fpage>. <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.24154</pub-id>
<pub-id pub-id-type="pmid">40742588</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<label>12.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vallmuur</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Marucci-Wellman</surname>
<given-names>HR</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>JA</given-names>
</name>
<name>
<surname>Lehto</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Corns</surname>
<given-names>HL</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>GS</given-names>
</name>
</person-group>. <article-title>Harnessing information from injury narratives in the &#x2018;big data&#x2019; era: understanding and applying machine learning for injury surveillance</article-title>. <source>Inj Prev</source> (<year>2016</year>) <volume>22</volume>(<issue>Suppl. 1</issue>):<fpage>i34</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1136/injuryprev-2015-041813</pub-id>
<pub-id pub-id-type="pmid">26728004</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<label>13.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Azzolina</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Bressan</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Lorenzoni</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Baldan</surname>
<given-names>GA</given-names>
</name>
<name>
<surname>Bartolotta</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Scognamiglio</surname>
<given-names>F</given-names>
</name>
<etal/>
</person-group> <article-title>Pediatric injury surveillance from uncoded emergency department admission records in Italy: machine learning&#x2013;based text-mining approach</article-title>. <source>JMIR Public Health Surveill</source> (<year>2023</year>) <volume>9</volume>:<fpage>e44467</fpage>. <pub-id pub-id-type="doi">10.2196/44467</pub-id>
<pub-id pub-id-type="pmid">37436799</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<label>14.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Araujo-Moura</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Souza</surname>
<given-names>L</given-names>
</name>
<name>
<surname>De Oliveira</surname>
<given-names>TA</given-names>
</name>
<name>
<surname>Rocha</surname>
<given-names>MS</given-names>
</name>
<name>
<surname>De Moraes</surname>
<given-names>ACF</given-names>
</name>
<name>
<surname>Filho</surname>
<given-names>AC</given-names>
</name>
</person-group>. <article-title>Prediction of hypertension in the pediatric population using machine learning and transfer learning: a multicentric analysis of the SAYCARE study</article-title>. <source>Int J Public Health</source> (<year>2025</year>) <volume>70</volume>:<fpage>1607944</fpage>. <pub-id pub-id-type="doi">10.3389/ijph.2025.1607944</pub-id>
<pub-id pub-id-type="pmid">40145015</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<label>15.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Catchpoole</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Niven</surname>
<given-names>C</given-names>
</name>
<name>
<surname>M&#xf6;ller</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Harrison</surname>
<given-names>JE</given-names>
</name>
<name>
<surname>Ivers</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Craig</surname>
<given-names>S</given-names>
</name>
<etal/>
</person-group> <article-title>External causes of emergency department presentations: a missing piece to understanding unintentional childhood injury in Australia</article-title>. <source>Emerg Med Australas</source> (<year>2023</year>) <volume>35</volume>(<issue>6</issue>):<fpage>927</fpage>&#x2013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.1111/1742-6723.14259</pub-id>
<pub-id pub-id-type="pmid">37366326</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<label>16.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Landau</surname>
<given-names>AY</given-names>
</name>
<name>
<surname>Blanchard</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Kulkarni</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Althobaiti</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Idnay</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Patton</surname>
<given-names>DU</given-names>
</name>
<etal/>
</person-group> <article-title>Harnessing the power of machine learning and electronic health records to support child abuse and neglect identification in emergency department settings</article-title>. <source>Stud Health Technol Inform</source> (<year>2024</year>) <volume>316</volume>:<fpage>1652</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.3233/SHTI240740</pub-id>
<pub-id pub-id-type="pmid">39176527</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<label>17.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>DH</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>SW</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>KH</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Shin</surname>
<given-names>SD</given-names>
</name>
</person-group>. <article-title>Using large language models to extract core injury information from emergency department notes</article-title>. <source>J Korean Med Sci</source> (<year>2024</year>) <volume>39</volume>(<issue>46</issue>):<fpage>e291</fpage>. <pub-id pub-id-type="doi">10.3346/jkms.2024.39.e291</pub-id>
<pub-id pub-id-type="pmid">39623965</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<label>18.</label>
<mixed-citation publication-type="book">
<collab>European Commission, Directorate-General for Health and Food Safety</collab>. <source>Injury Data Base (IDB)</source>. <publisher-loc>Luxembourg</publisher-loc>: <publisher-name>Publications Office of the European Union</publisher-name> (<year>2022</year>). <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://health.ec.europa.eu/document/download/657e4de7-d6f8-4933-b237-285e8c5278fd_en">https://health.ec.europa.eu/document/download/657e4de7-d6f8-4933-b237-285e8c5278fd_en (Accessed December 10, 2025)</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B19">
<label>19.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>AQ</given-names>
</name>
<name>
<surname>Sablayrolles</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Mensch</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Bamford</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Chaplot</surname>
<given-names>DS</given-names>
</name>
<name>
<surname>De Las Casas</surname>
<given-names>D</given-names>
</name>
<etal/>
</person-group> <article-title>Mistral 7B</article-title>. <source>arXiv (Cornell University)</source> (<year>2023</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2310.06825</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<label>20.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>EJ</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Wallis</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Allen-Zhu</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S</given-names>
</name>
<etal/>
</person-group> <article-title>LORA: low-rank adaptation of large language models</article-title>. <source>arXiv (Cornell University)</source> (<year>2021</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2106.09685</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<label>21.</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Saaro</surname>
<given-names>FM</given-names>
</name>
<name>
<surname>von D&#xe4;niken</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Cieliebak</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Deriu</surname>
<given-names>JM</given-names>
</name>
</person-group>. <article-title>Do NOT classify and count: hybrid attribute control success evaluation</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Demberg</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Inui</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Marquez</surname>
<given-names>L</given-names>
</name>
</person-group>, editors. <source>Proceedings of the 19th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers); 2026 Mar; Rabat, Morocco</source>. <publisher-loc>Rabat</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name> (<year>2026</year>). p. <fpage>1101</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2026.eacl-long.48</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<label>22.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Von D&#xe4;niken</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Deriu</surname>
<given-names>JM</given-names>
</name>
<name>
<surname>Rodrigo</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Cieliebak</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Improving quantification with minimal in-domain annotations: beyond classify and count</article-title>. <source>Proc Int AAAI Conf Web Soc Media</source> (<year>2024</year>) <volume>18</volume>:<fpage>1585</fpage>&#x2013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.1609/icwsm.v18i1.31411</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<label>23.</label>
<mixed-citation publication-type="other">
<person-group person-group-type="author">
<name>
<surname>Meier</surname>
<given-names>D</given-names>
</name>
<name>
<surname>B&#xe4;chli</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Niemann</surname>
<given-names>S</given-names>
</name>
</person-group>. <article-title>Sicherheitsbarometer 2024: sicherheitsniveau in Haus und Freizeit</article-title>. <source>Beratungsstelle f&#xfc;r Unfallverh&#xfc;tung (BFU)</source>. <pub-id pub-id-type="doi">10.13100/bfu.2.538.01.2024</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<label>24.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alves</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Giustini</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Papadakaki</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Nijmans</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Valkenberg</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>S</given-names>
</name>
<etal/>
</person-group> <article-title>Home and leisure accidents among children and young people up to 19 years old as an event observed in the EU-IDB surveillance system: 2012&#x2013;2021 data</article-title>. <source>Eur J Public Health</source> (<year>2024</year>) <volume>34</volume>(<issue>Suppl. 3</issue>):<fpage>ckae144.061</fpage>. <pub-id pub-id-type="doi">10.1093/eurpub/ckae144.061</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<label>25.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hoffmann</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Morris</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Herrmann</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Heinze</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Wynants</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Van Calster</surname>
<given-names>B</given-names>
</name>
<etal/>
</person-group> <article-title>Using routinely collected data for research purposes: challenges and mitigation strategies</article-title>. <source>BMJ</source> (<year>2026</year>) <volume>393</volume>:<fpage>e087812</fpage>. <pub-id pub-id-type="doi">10.1136/bmj-2025-087812</pub-id>
<pub-id pub-id-type="pmid">42229936</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<label>26.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spasic</surname>
<given-names>I</given-names>
</name>
<name>
<surname>Nenadic</surname>
<given-names>G</given-names>
</name>
</person-group>. <article-title>Clinical text data in machine learning: systematic review</article-title>. <source>JMIR Med Inform</source> (<year>2020</year>) <volume>8</volume>(<issue>3</issue>):<fpage>e17984</fpage>. <pub-id pub-id-type="doi">10.2196/17984</pub-id>
<pub-id pub-id-type="pmid">32229465</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<label>27.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mooney</surname>
<given-names>SJ</given-names>
</name>
<name>
<surname>Pejaver</surname>
<given-names>V</given-names>
</name>
</person-group>. <article-title>Big data in public health: terminology, machine learning, and privacy</article-title>. <source>Annu Rev Public Health</source> (<year>2018</year>) <volume>39</volume>:<fpage>95</fpage>&#x2013;<lpage>112</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-publhealth-040617-014208</pub-id>
<pub-id pub-id-type="pmid">29261408</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<label>28.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaboudi</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Firouzbakht</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Shahir Eftekhar</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Fayazbakhsh</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Joharivarnoosfaderani</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Ghaderi</surname>
<given-names>S</given-names>
</name>
<etal/>
</person-group> <article-title>Diagnostic accuracy of ChatGPT for patients&#x2019; triage: a systematic review and meta-analysis</article-title>. <source>Arch Acad Emerg Med</source> (<year>2024</year>) <volume>12</volume>(<issue>1</issue>):<fpage>e60</fpage>. <pub-id pub-id-type="doi">10.22037/aaem.v12i1.2384</pub-id>
<pub-id pub-id-type="pmid">39290765</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<label>29.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weeks</surname>
<given-names>WB</given-names>
</name>
<name>
<surname>Taliesin</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Lavista</surname>
<given-names>JM</given-names>
</name>
</person-group>. <article-title>Using artificial intelligence to advance public health</article-title>. <source>Int J Public Health</source> (<year>2023</year>) <volume>68</volume>:<fpage>1606716</fpage>. <pub-id pub-id-type="doi">10.3389/ijph.2023.1606716</pub-id>
<pub-id pub-id-type="pmid">38024205</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<label>30.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharwood</surname>
<given-names>LN</given-names>
</name>
</person-group>. <article-title>AI use for injury surveillance in emergency departments</article-title>. <source>JAMA Netw Open</source> (<year>2025</year>) <volume>8</volume>(<issue>7</issue>):<fpage>e2524162</fpage>. <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.24162</pub-id>
<pub-id pub-id-type="pmid">40742598</pub-id>
</mixed-citation>
</ref>
<ref id="B31">
<label>31.</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mienye</surname>
<given-names>ID</given-names>
</name>
<name>
<surname>Obaido</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Jere</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Mienye</surname>
<given-names>E</given-names>
</name>
<name>
<surname>Aruleba</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Emmanuel</surname>
<given-names>ID</given-names>
</name>
<etal/>
</person-group> <article-title>A survey of explainable artificial intelligence in healthcare: concepts, applications, and challenges</article-title>. <source>Inform Med Unlocked</source> (<year>2024</year>) <volume>51</volume>:<fpage>101587</fpage>. <pub-id pub-id-type="doi">10.1016/j.imu.2024.101587</pub-id>
</mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1002545/overview">Gabriel Gulis</ext-link>, University of Southern Denmark, Denmark</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/461897/overview">Lukas Novak</ext-link>, Olomouc University Social Health Institute (OUSHI), Czechia</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3199744/overview">Dragan Stoll</ext-link>, ZHAW Zurcher Hochschule fur Angewandte Wissenschaften Departement Soziale Arbeit, Switzerland</p>
</fn>
</fn-group>
</back>
</article>