<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">Interact J Med Res</journal-id><journal-id journal-id-type="publisher-id">i-jmr</journal-id><journal-id journal-id-type="index">3</journal-id><journal-title>Interactive Journal of Medical Research</journal-title><abbrev-journal-title>Interact J Med Res</abbrev-journal-title><issn pub-type="epub">1929-073X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v15i1e96228</article-id><article-id pub-id-type="doi">10.2196/96228</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Design, Facilitation, and Evaluation of Tabletop Exercises for Prehospital Mass Casualty Preparedness: Scoping Review</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Prashanth</surname><given-names>Paurnami</given-names></name><degrees>MBBS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>AlRahma</surname><given-names>Ali</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Narayanan</surname><given-names>Dayol</given-names></name><degrees>BSN, MBA</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Omayer</surname><given-names>Abu</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Yousif</surname><given-names>Azza</given-names></name><degrees>MBBS, MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hubloue</surname><given-names>Ives</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zary</surname><given-names>Nabil</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>Emergency Department, Dubai Health</institution><addr-line>Umm Hurair Second, Bur Dubai</addr-line><addr-line>Dubai</addr-line><country>United Arab Emirates</country></aff><aff id="aff2"><institution>Dubai Corporation for Ambulance Services</institution><addr-line>Dubai</addr-line><country>United Arab Emirates</country></aff><aff id="aff3"><institution>Institute of Learning, Mohammed Bin Rashid University of Medicine and Health Sciences, Dubai Health</institution><addr-line>Dubai</addr-line><country>United Arab Emirates</country></aff><aff id="aff4"><institution>Emergency Department, UZ Brussel University Hospital</institution><addr-line>Brussels</addr-line><country>Belgium</country></aff><aff id="aff5"><institution>Research Group on Emergency and Disaster Medicine (ReGEDiM) Vrije Universiteit Brussel</institution><addr-line>Brussels</addr-line><country>Belgium</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Cardoso</surname><given-names>Taiane de Azevedo</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Clark</surname><given-names>Cullen</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Sands</surname><given-names>Joann</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Azza Yousif, MBBS, MSc, Emergency Department, Dubai Health, Umm Hurair Second, Bur Dubai, Dubai, United Arab Emirates, 971 800 60 ext 800; <email>AOYousif@dubaihealth.ae</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>5</day><month>10</month><year>2026</year></pub-date><volume>15</volume><elocation-id>e96228</elocation-id><history><date date-type="received"><day>27</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>13</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>14</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Paurnami Prashanth, Ali AlRahma, Dayol Narayanan, Abu Omayer, Azza Yousif, Ives Hubloue, Nabil Zary. Originally published in the Interactive Journal of Medical Research (<ext-link ext-link-type="uri" xlink:href="https://www.i-jmr.org/">https://www.i-jmr.org/</ext-link>), 5.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Interactive Journal of Medical Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.i-jmr.org/">https://www.i-jmr.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.i-jmr.org/2026/1/e96228"/><abstract><sec><title>Background</title><p>Tabletop exercises (TTXs) are commonly used to enhance prehospital readiness for mass casualty incidents (MCIs). However, evidence on their design, facilitation, and evaluation remains scattered. TTXs simulate organized interactions at 3 levels: among individual responders and response protocols, within interdisciplinary teams, and across organizations and systems. While existing reviews cover tabletop simulation generally, they do not specifically focus on the prehospital MCI interface or assess whether evaluation methods match the exercise objectives.</p></sec><sec><title>Objective</title><p>This scoping review aimed to explore how TTXs are designed, facilitated, and evaluated in prehospital MCI preparedness. It classified outcomes using the Kirkpatrick Evaluation Model and analyzed how evaluation methods align with the purpose of each exercise.</p></sec><sec sec-type="methods"><title>Methods</title><p>We performed a scoping review following Arksey and O&#x2019;Malley&#x2019;s framework, with enhancements from Levac et al, along with PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews) and PRISMA-S (PRISMA Extension for Reporting Literature Searches) guidelines. Searches across PubMed, Embase, Scopus, PsycINFO, CINAHL, the Cochrane Library, ClinicalTrials.gov, Google Scholar, and references were limited to English peer-reviewed and gray literature published from 2015 to May 2026. Eligible studies focused on TTXs related to prehospital disaster or MCI preparedness and reported measurable educational, clinical, or system outcomes. The review protocol was registered beforehand on Protocols.io. Two reviewers (PP and AA) independently screened records, extracted data, and classified outcomes using Kirkpatrick levels, with an added level 2+ (Applied Learning) category for structured performance evaluation during exercises.</p></sec><sec sec-type="results"><title>Results</title><p>Thirteen studies from 9 countries were included. They were categorized into a preliminary 3-tier typology called the TTX Design Spectrum, representing increasing levels of interaction: algorithm-rehearsal exercises (n=2) focused on individual responders&#x2019; interaction with triage protocols, scenario-based decision-training exercises (n=7) aimed at multidisciplinary team decision-making under realistic conditions, and systems integration exercises (n=4) centered on interagency coordination and system-level preparedness. Facilitation varied by exercise purpose, from standardized assessment-focused approaches to expert-led and multidisciplinary facilitation. Evaluation primarily targeted Kirkpatrick level 1: Reaction (n=10), level 2: Learning (n=10), and level 2+ (Applied Learning) (n=8), with fewer studies examining level 3: Behavior (n=2) or level 4: Results (n=3). Operational frameworks were reported more consistently than formal educational design or assessment frameworks. Additionally, natural disaster scenarios and evidence from resource-limited settings were underrepresented.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>TTXs for prehospital MCI preparedness should be viewed as a collection of related exercise types rather than a single, uniform intervention. This review introduces an initial typology called the TTX Design Spectrum, along with the level 2+ (Applied Learning) classification, which operationalizes the distinction between in-training performance and behavioral transfer. These tools aim to help align exercise purpose, facilitation, and evaluation strategies. Future research should focus on validating this typology, enhancing follow-up assessments of behavioral transfer, and adapting TTX design for natural disaster and resource-limited settings.</p></sec></abstract><kwd-group><kwd>tabletop exercise</kwd><kwd>mass casualty incident</kwd><kwd>prehospital</kwd><kwd>scoping review</kwd><kwd>Kirkpatrick Evaluation Model</kwd><kwd>simulation-based education</kwd><kwd>disaster preparedness</kwd><kwd>applied learning</kwd><kwd>training transfer</kwd><kwd>exercise design</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Mass casualty incidents (MCIs) and disasters continue to challenge health systems because they demand quick coordination, prioritization, and adaptation amid uncertainty, time constraints, and limited resources. The Sendai Framework for Disaster Risk Reduction emphasizes the importance of preparedness for effective response in modern disaster management [<xref ref-type="bibr" rid="ref1">1</xref>]. Prehospital readiness involves personnel and systems managing triage, incident command, scene coordination, interagency communication, resource management, and treatment prioritization. This preparedness significantly impacts outcomes when systems are overwhelmed. As a result, health care simulation use has increased [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>], and recent tabletop and disaster education research highlights incident command, triage, surge management, communication, and resource allocation as essential skills that require practice before real emergencies [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Tabletop exercises (TTXs) stand out among simulation methods because they are structured, discussion-driven activities where participants can rehearse plans, roles, communication, and operational choices without the logistical challenges of full-scale or functional exercises [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. They are also versatile, being customizable to an organization&#x2019;s specific risks with adaptable goals, scope, and difficulty levels [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref5">5</xref>], making them a popular tool for improving disaster and MCI preparedness [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>The existing literature indicates that TTXs are not a uniform type of intervention. Some primarily concentrate on practicing triage algorithms, while others focus on team decision-making, interagency communication, or organizational coordination [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref6">6</xref>]. This variation is expected because exercises serve different educational and operational goals. The challenge lies in the fact that the term &#x201C;tabletop exercise&#x201D; is often used to describe quite different activities. Without a common framework for describing the purpose of each exercise, educators and planners struggle to determine whether an evaluation method fits the exercise&#x2019;s objectives and find it difficult to compare findings across studies that use the same term for different types of activities.</p><p>At their core, TTXs are organized interactions involving individual responders, the protocols they follow, professionals from various disciplines, and different organizations, agencies, and systems. The specific level of interaction an exercise aims to rehearse determines who participates, how it is facilitated, and what metrics should be used for evaluation. Consequently, a typology of exercise purposes also reflects the types of interactions being trained. This shared vocabulary helps determine whether an evaluation method is appropriate for a particular exercise and allows for comparison of evidence across studies.</p><p>Recent review literature highlights the need for a more targeted synthesis. Fr&#x00E9;geau et al [<xref ref-type="bibr" rid="ref7">7</xref>] conducted a comprehensive scoping review of tabletop simulations in medical emergencies, analyzing 70 studies across various settings, specialties, formats, and learner groups. They found that reported outcomes mainly focused on the Kirkpatrick reaction and learning levels [<xref ref-type="bibr" rid="ref7">7</xref>]. Emaliyawati et al [<xref ref-type="bibr" rid="ref8">8</xref>] reviewed 12 tabletop disaster exercise studies involving health care workers and students, concluding that TTXs enhance knowledge, attitudes, preparedness, confidence, and performance. Although these reviews are valuable, they address TTXs broadly, rather than focusing specifically on the prehospital interface. They do not analyze how prehospital MCI competencies are distributed across different exercise designs, nor do they systematically link exercise purpose with facilitation style and evaluation depth. As far as we know, no published review has specifically synthesized TTXs for prehospital MCI preparedness while examining the alignment among exercise design, facilitation, and evaluation [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>This scoping review aimed to map the existing literature and clarify how TTXs function within prehospital MCI preparedness, considering them as a group of related but distinct interventions. The objectives included systematically investigating how TTXs have been used to evaluate or enhance prehospital readiness for MCIs; classifying reported outcomes according to the Kirkpatrick Evaluation Model, including structured performance within exercises; describing key design, delivery, and facilitation features along with the competencies they target; and pinpointing gaps in the evidence. The review seeks to develop a preliminary typology called the TTX Design Spectrum, which aims to guide purpose-aligned design, facilitation, and evaluation in prehospital disaster preparedness.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Methodological Framework</title><sec id="s2-1-1"><title>Overview</title><p>This scoping review, part of the Mass-Casualty Incident, Prehospital Emergency Response project, systematically maps evidence related to TTXs for prehospital disaster preparedness. Our methodology follows Arksey and O&#x2019;Malley&#x2019;s [<xref ref-type="bibr" rid="ref11">11</xref>] framework, with modifications by Levac et al [<xref ref-type="bibr" rid="ref12">12</xref>], and adheres to PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews) reporting standards [<xref ref-type="bibr" rid="ref13">13</xref>]. The protocol was registered on Protocols.io (July 3, 2025) [<xref ref-type="bibr" rid="ref14">14</xref>] and published as a preprint [<xref ref-type="bibr" rid="ref15">15</xref>], ensuring transparency and reproducibility. Stakeholder consultation, recommended by Levac et al [<xref ref-type="bibr" rid="ref12">12</xref>] for knowledge translation, was not carried out because the study primarily focused on evidence mapping for research synthesis rather than developing practice guidelines.</p></sec><sec id="s2-1-2"><title>Stage 1: Identifying the Research Question</title><p>This scoping review was based on 2 main research questions and 3 subsidiary questions. The primary questions explored how TTXs are used to assess or enhance prehospital preparedness for MCIs, and which Kirkpatrick evaluation levels (Reaction, Learning, Behavior, and Results) are most frequently reported in prehospital MCI TTX studies. The secondary questions investigated the key design and delivery features of these TTXs, the competencies and operational areas they target, and identified gaps in evidence concerning evaluation methods and their effectiveness.</p></sec><sec id="s2-1-3"><title>Stage 2: Identifying Relevant Studies</title><p>Search reporting followed the PRISMA-S (PRISMA Extension for Reporting Literature Searches) extension for literature search reporting [<xref ref-type="bibr" rid="ref16">16</xref>], along with PRISMA-ScR. A completed PRISMA-ScR checklist is included in <xref ref-type="supplementary-material" rid="app6">Checklist 1</xref>. We searched several multidisciplinary databases, including PubMed, Embase, Scopus, PsycINFO (via APA PsycNet), CINAHL, Cochrane Library, and ClinicalTrials.gov, to cover biomedical, educational, and allied health research. Each database and registry was searched individually; no multidatabase search was performed simultaneously. We also reviewed Google Scholar for gray literature and manually screened reference lists of included studies and relevant reviews. No contact was made with study authors, experts, manufacturers, or external contacts to gather additional records or data. Our search methods included only database searches, Google Scholar, and manual reference list screening.</p><p>On June 10, 2025, we established a detailed search strategy based on the Population-Concept-Context framework (<xref ref-type="table" rid="table1">Table 1</xref>), which was subsequently updated on May 15, 2026. The Population targeted prehospital health care workers such as paramedics, emergency medical technicians, emergency physicians, and nurses, as well as emergency system administrators involved in MCI responses. The Concept focused on TTXs used for training and preparedness, with outcomes measurable at the educational, clinical, or system levels. The Context covered prehospital systems that respond to mass-casualty incidents and disasters across various geographic areas.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Search strategy overview using the Population-Concept-Context framework. Representative controlled vocabulary and free-text terms are displayed. Detailed database-specific search strings are available in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Category</td><td align="left" valign="bottom">Keywords</td><td align="left" valign="bottom">Search strategy</td></tr></thead><tbody><tr><td align="left" valign="top">Population</td><td align="left" valign="top">&#x201C;Emergency Medical Services&#x201D; [MeSH]; &#x201C;Emergency Medical Technicians&#x201D; [MeSH]; &#x201C;Allied Health Personnel&#x201D; [MeSH]</td><td align="left" valign="top">paramedic*; &#x201C;first responder*&#x201D;; &#x201C;first responder&#x201D;; &#x201C;first responders&#x201D;; ambulance*; &#x201C;emergency medical service*&#x201D;; &#x201C;emergency medical service&#x201D;; &#x201C;emergency medical services&#x201D;; EMS<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Concept</td><td align="left" valign="top">&#x201C;Simulation Training&#x201D; [MeSH]; &#x201C;Patient Simulation&#x201D; [MeSH]; &#x201C;Computer Simulation&#x201D; [MeSH]; &#x201C;simulation training&#x201D;; &#x201C;tabletop simulation&#x201D;; &#x201C;tabletop exercise&#x201D;</td><td align="left" valign="top">tabletop exercise*; table-top exercise*; tabletop simulation*; tabletop drill*; discussion-based exercise*; scenario-based simulation*; preparedness exercise*; board game*; simulation game*; gamified exercise; paper-based exercise</td></tr><tr><td align="left" valign="top">Context</td><td align="left" valign="top">&#x201C;Mass Casualty Incidents&#x201D; [MeSH]; &#x201C;Disaster Planning&#x201D; [MeSH]; &#x201C;Emergency Preparedness&#x201D; [MeSH]; &#x201C;Triage&#x201D; [MeSH]; &#x201C;mass disaster&#x201D;; &#x201C;field triage&#x201D;; &#x201C;incident command&#x201D;; &#x201C;prehospital care&#x201D;</td><td align="left" valign="top">emergency preparedness; disaster preparedness; mass casualty incident*; MCI<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>; field triage; incident command; prehospital care; disaster simulation; response coordination; crisis; disaster response; multi-casualty; multiple casualty</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>EMS: emergency medical services.</p></fn><fn id="table1fn2"><p><sup>b</sup>MCI: mass casualty incident.</p></fn></table-wrap-foot></table-wrap><p>We used a combination of controlled vocabulary (MeSH) and free-text terms with Boolean operators in search strings tailored to each database, refining them iteratively to improve sensitivity and specificity (<xref ref-type="table" rid="table1">Table 1</xref>). An information specialist or librarian peer-reviewed the search strategy following the Peer Review of Electronic Search Strategies criteria before final implementation [<xref ref-type="bibr" rid="ref17">17</xref>]. No published search filters were used, nor were previous review strategies reused or adapted; instead, prior reviews helped identify relevant references through citation screening. Details about platforms and complete database-specific search strings, including the full PubMed strategy, are available in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-1-4"><title>Stage 3: Study Selection</title><p>Retrieved records from various sources, including citation searches and gray literature, were imported into Covidence [<xref ref-type="bibr" rid="ref18">18</xref>], where they were pooled and deduplicated before screening. Two reviewers, PP and AA, independently examined all records through a 2-stage process: first screening titles and abstracts and then reviewing full texts. Any disagreements were resolved through discussion or, if necessary, by consulting a third reviewer. These disagreements were systematically recorded in Covidence and addressed through structured discussions with documented reasons.</p></sec></sec><sec id="s2-2"><title>Eligibility Criteria</title><sec id="s2-2-1"><title>Overview</title><p>We included studies that assessed TTXs in prehospital disaster or MCI preparedness and reported measurable educational, clinical, or system outcomes. These were peer-reviewed papers or gray literature published in English between January 2015 and May 2026. For this review, &#x201C;prehospital&#x201D; was mainly defined by the medical phase and operational functions of MCI or disaster response, rather than solely by participant role or physical setting. This encompassed EMS (emergency medical services) care, field triage, scene coordination, casualty distribution, transport decisions, and the transition from prehospital to hospital care. Eligible study designs included quantitative, qualitative, mixed methods, quasi-experimental, pre-post, randomized, observational, descriptive, pilot, feasibility, and implementation studies. We excluded systematic reviews, narrative reviews, editorials, commentaries, opinion pieces, conference abstracts without full texts, studies lacking full-text access, non-English studies, studies published before 2015, and studies where TTXs were not the primary focus. Additionally, studies focused on nonmedical TTX applications, such as military, corporate, cybersecurity, or other non&#x2013;health care settings, were excluded unless they included a relevant medical prehospital or MCI response component. Studies involving hospital personnel were included only if they addressed prehospital medical functions, EMS coordination, transport, casualty distribution, or the hospital transition process.</p><p>We chose papers from 2015 onward because initial scoping indicated limited peer-reviewed literature before 2015 on TTX outcomes in prehospital MCI settings. Furthermore, reporting standards for simulation-based education improved significantly during this period [<xref ref-type="bibr" rid="ref19">19</xref>], and the 2015 Sendai Framework for Disaster Risk Reduction [<xref ref-type="bibr" rid="ref1">1</xref>] provided a current policy context for disaster preparedness research.</p></sec><sec id="s2-2-2"><title>Stage 4: Data Charting</title><p>Before starting the formal data extraction, we tested a standardized data-charting form with 2 reviewers (DN and AO) across 5 studies to verify that the extraction fields were clear, comprehensive, and consistently understood. During this pilot, reviewers independently extracted data, compared their entries, discussed any discrepancies, and refined the form before proceeding with full extraction. The finalized form included study details (ID, country, design, population, and sample size), TTX features (targeted competencies, duration, scenario design, and preparatory instruction), facilitation aspects, evaluation methods, Kirkpatrick levels addressed, and the underlying training, assessment, and operational frameworks.</p><p>After the data-charting form was finalized, 2 reviewers (DN and AO) independently extracted data from all included studies. They compared their extractions, resolving any discrepancies through discussion and consensus. If they could not reach an agreement, a third reviewer was consulted. Microsoft Excel was used to manage, review, and organize the extracted data.</p><p>Outcomes were categorized using the Kirkpatrick Evaluation Model [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Beyond the 4 standard levels, we added a level 2+ (Applied Learning) subcategory to differentiate structured performance assessments within the exercise setting from knowledge acquisition and actual behavioral transfer in real-world contexts. This subcategory helps operationalize, for TTX outcome mapping, distinctions outlined in earlier frameworks: the updated (&#x201C;New World&#x201D;) Kirkpatrick model&#x2019;s focus on skills shown during training [<xref ref-type="bibr" rid="ref22">22</xref>], Miller&#x2019;s &#x201C;shows how&#x201D; level of assessment [<xref ref-type="bibr" rid="ref23">23</xref>], and translational science models of simulation-based education outcomes (T1-T3), which differentiate in-simulation performance from later behavior and patient or system results [<xref ref-type="bibr" rid="ref24">24</xref>]. Each outcome was classified independently; thus, a single study could contribute to multiple Kirkpatrick levels. Detailed definitions, decision rules, and examples for classification are available in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s2-2-3"><title>Stage 5: Collating, Summarizing, and Reporting the Results</title><p>We used a combination of descriptive and narrative approaches appropriate for scoping reviews. Quantitative information was summarized using frequencies and counts, while qualitative and contextual data, such as exercise design, scenario features, facilitation processes, targeted skills, and evaluation techniques, were synthesized narratively. The extracted data were categorized into 7 domains: general study details, population traits, TTX features and elements, facilitator responsibilities, data collection methods, training results, and frameworks that inform training design, assessment, and operational practice.</p><p>Through iterative cross-study comparison, we discovered common patterns in exercise design, facilitation, and evaluation. To categorize these patterns, we created the TTX Design Spectrum, which classifies exercises into 3 tiers based on their primary purpose and scope: tier 1 focuses on algorithm rehearsal, tier 2 on scenario-based decision training, and tier 3 on systems integration testing. Facilitation methods and evaluation patterns were summarized descriptively rather than formalized into additional models. This typology was developed inductively during synthesis rather than being predefined; it is an initial organizational tool and not a validated framework. Definitions can be found in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec></sec><sec id="s2-3"><title>Protocol Deviations</title><p>This review was based on a prospectively registered protocol [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Two post hoc analytical refinements were introduced: the creation of the TTX Design Spectrum typology, which provides a descriptive overview of facilitation and evaluation patterns during stage 5, and the addition of the level 2+ (Applied Learning) subcategory in outcome classification during stage 4. Neither of these changes was specified in the protocol; both were developed through iterative comparison and synthesis of data. Importantly, there were no deviations that affected the search strategy or eligibility criteria.</p></sec><sec id="s2-4"><title>Ethical Considerations</title><p>This scoping review summarized publicly accessible, previously published literature without direct involvement of human participants. Therefore, ethical approval was not required. We adhered to established ethical research principles, such as transparent reporting, proper attribution and citation, and preventing plagiarism and duplicate publications.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Literature Search and Study Selection</title><p>Our search identified 2693 records from databases and trial registers, along with 20 additional records from other sources. After removing 272 duplicates, we screened 2441 records by title and abstract, excluding 2353. We attempted to retrieve 88 full-text reports; of these, 19 were conference abstracts without full publications. Metadata for these 19 records (author, year, title, and source) is provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. The publication years and geographic origins of unretrieved records were similar to those of the retrieved records (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). Of the 69 full-text reports assessed for eligibility, 56 were excluded (wrong setting, n=22; wrong intervention, n=21; wrong study design, n=11; and non-English, n=2). Thirteen studies met the inclusion criteria and were included in the final analysis. <xref ref-type="fig" rid="figure1">Figure 1</xref> illustrates the study selection process.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 flow diagram for study selection. Records were identified through 6 databases, 1 register, and other methods. Of these, 13 studies met all inclusion criteria and were categorized into 3 tiers of the Tabletop Exercise Design Spectrum.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="i-jmr_v15i1e96228_fig01.png"/></fig></sec><sec id="s3-2"><title>Study Characteristics</title><p>Thirteen studies published between 2016 and 2026 were included, with most publications in 2020 (n=3). The studies took place in various countries: 4 in the United States, 2 in Thailand, and 1 each in the United Kingdom, Saudi Arabia, South Korea, Qatar, Spain, Sweden, and Japan. The research designs comprised 4 pre-post evaluations, 3 quasi-experimental studies, 2 reliability assessments, 2 mixed methods investigations, 1 randomized controlled trial, and 1 descriptive report. The sample sizes varied from 12 to 230 participants; 1 study did not specify the number of fellows involved (refer to <xref ref-type="table" rid="table2">Table 2</xref>, footnote).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Characteristics of Included Studies (N=13).</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study</td><td align="left" valign="bottom">Country</td><td align="left" valign="bottom">Design</td><td align="left" valign="bottom">Participants, n</td><td align="left" valign="bottom">Population</td><td align="left" valign="bottom">Duration</td><td align="left" valign="bottom">Scenarios</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="7">Tier 1: algorithm rehearsal (n=2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cheng et al (2022) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Reliability study</td><td align="left" valign="top">107</td><td align="left" valign="top">Prehospital EMS<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> professionals</td><td align="left" valign="top">4&#x2010;11 minutes per set</td><td align="left" valign="top">1 (pediatric bus crash, 25 patients; 6 triage algorithms compared)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>McGlynn et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Reliability study</td><td align="left" valign="top">NR<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">Pediatric emergency medicine fellows</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">3 (10-, 100-, and 1000-victim MCI<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup> scenarios; 253 patient cases)</td></tr><tr><td align="left" valign="top" colspan="7">Tier 2: scenario-based decision training (n=7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Kim et al (2021) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">South Korea</td><td align="left" valign="top">Pre-post</td><td align="left" valign="top">40</td><td align="left" valign="top">Physicians, nurses, and public health officers</td><td align="left" valign="top">25 minutes + 10 minutes debrief</td><td align="left" valign="top">1 (hydrofluoric acid explosion at chemical plant)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Phattharapornjaroen et al (2020) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Thailand</td><td align="left" valign="top">Pre-post</td><td align="left" valign="top">52</td><td align="left" valign="top">Emergency physicians</td><td align="left" valign="top">2-day course</td><td align="left" valign="top">3 (building fire, terrorist bombing, and riots or active shooter)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sena et al (2021) [<xref ref-type="bibr" rid="ref4">4</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Pre-post</td><td align="left" valign="top">18</td><td align="left" valign="top">Emergency medicine residents</td><td align="left" valign="top">2 hours</td><td align="left" valign="top">1 (explosion or mass shooting with fire)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Farhat et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">Qatar</td><td align="left" valign="top">Pre-post</td><td align="left" valign="top">12</td><td align="left" valign="top">Physicians, nurses, and paramedics</td><td align="left" valign="top">3 hours</td><td align="left" valign="top">4 (biological, chemical, radiological, and nuclear CBRNE<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup>)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Alakrawi et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">Saudi Arabia</td><td align="left" valign="top">Quasi-experimental</td><td align="left" valign="top">45</td><td align="left" valign="top">Senior paramedic students</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">1 (multivehicle accident, 15 patients)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Chumvanichaya et al (2025) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Thailand</td><td align="left" valign="top">RCT<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">83</td><td align="left" valign="top">Paramedic students</td><td align="left" valign="top">40 minutes</td><td align="left" valign="top">3 (10 simulated victims: SIEVE<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup>, SORT<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup>, and START<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup> protocols)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cuartas-Alvarez et al (2026) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Spain</td><td align="left" valign="top">Quasi-experimental</td><td align="left" valign="top">27</td><td align="left" valign="top">Primary care professionals (doctors and nurses)</td><td align="left" valign="top">2 hours</td><td align="left" valign="top">1 (MCI scenario: collapse with multiple victims; PHCT<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup>-led prehospital response with ECC<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup> or resource coordination)</td></tr><tr><td align="left" valign="top" colspan="7">Tier 3: systems integration testing (n=4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Anan et al (2016) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Japan</td><td align="left" valign="top">Quasi-experimental</td><td align="left" valign="top">230</td><td align="left" valign="top">Physicians, firefighters, police officers, and emergency responders</td><td align="left" valign="top">60&#x2010;85 minutes per scenario</td><td align="left" valign="top">3 (fire &#x2192; chemical recognition &#x2192; decontamination &#x2192; triage)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cicero et al (2019) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Descriptive</td><td align="left" valign="top">27</td><td align="left" valign="top">EMS, nurses, physicians, hospital administrators, and emergency managers</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">1 (school bus rollover, pediatric MCI)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Skryabina et al (2020) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">United Kingdom</td><td align="left" valign="top">Mixed methods</td><td align="left" valign="top">125</td><td align="left" valign="top">Prehospital, hospital staff, and emergency planners</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">1 (suicide bombing + marauding terrorist firearm attack)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zimmerman et al (2026) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">Sweden</td><td align="left" valign="top">Mixed methods</td><td align="left" valign="top">16</td><td align="left" valign="top">Physicians, residents, specialists across emergency medicine, anesthesiology, internal medicine, surgery, psychiatry, radiology, and ophthalmology</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">Not reported</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>EMS: emergency medical services.</p></fn><fn id="table2fn2"><p><sup>b</sup>The participant number is not clearly reported (NR); there were 253 patient cases available for triage across the scenario sets.</p></fn><fn id="table2fn3"><p><sup>c</sup>MCI: mass casualty incident.</p></fn><fn id="table2fn4"><p><sup>d</sup>CBRNE: Chemical, Biological, Radiological, Nuclear, Explosive.</p></fn><fn id="table2fn5"><p><sup>e</sup>RCT: randomized controlled trial.</p></fn><fn id="table2fn6"><p><sup>f</sup>SIEVE: Sieve (UK Major Incident Primary Triage).</p></fn><fn id="table2fn7"><p><sup>g</sup>SORT: Secondary Triage (UK Major Incident).</p></fn><fn id="table2fn8"><p><sup>h</sup>START: Simple Triage and Rapid Treatment.</p></fn><fn id="table2fn9"><p><sup>i</sup>PHCT: primary health care team.</p></fn><fn id="table2fn10"><p><sup>j</sup>ECC: emergency coordination center.</p></fn></table-wrap-foot></table-wrap><p>The study populations were diverse, comprising EMS personnel, paramedics, paramedic students, physicians, emergency medicine residents, pediatric emergency specialists, nurses, and public health officers. Additionally, 3 studies specifically included nonclinical stakeholders: hospital administrators and emergency managers [<xref ref-type="bibr" rid="ref34">34</xref>], firefighters and police officers [<xref ref-type="bibr" rid="ref33">33</xref>], and emergency planners [<xref ref-type="bibr" rid="ref35">35</xref>]. <xref ref-type="table" rid="table2">Table 2</xref> provides a summary of the main study characteristics, while <xref ref-type="fig" rid="figure2">Figure 2A and B</xref> illustrate the distribution of the studies over time and across different regions.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Temporal and geographic distribution of included studies (N=13). Panel A shows the count of included studies by publication year from 2015 to 2026, while panel B illustrates the geographic distribution across countries. In both panels, the bar segments are color-coded to represent the Tabletop Exercise Design Spectrum tiers: tier 1 for algorithm rehearsal, tier 2 for scenario-based decision training, and tier 3 for systems integration testing.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="i-jmr_v15i1e96228_fig02.png"/></fig></sec><sec id="s3-3"><title>Organization of the Synthesis</title><p><xref ref-type="fig" rid="figure3">Figure 3</xref> offers a visual overview that clarifies the relationship between the classification systems and how the results are organized. In this review, &#x201C;tiers&#x201D; specifically refer to the TTX Design Spectrum, which relates to exercise purpose and scope, while &#x201C;levels&#x201D; exclusively denote Kirkpatrick outcome levels, indicating evaluation depth. Facilitation modes describe how facilitators engaged participants. In the discussion, we use descriptive names such as algorithm rehearsal, scenario-based decision training, and systems integration testing to refer to tiers, thereby keeping the 2 numbering systems distinct.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Study-level matrix of tabletop exercise (TTX) tier, primary facilitation mode, and Kirkpatrick levels assessed (N=13). Each row represents a study included, color-coded by TTX Design Spectrum tier, highlighting its main facilitation mode and the Kirkpatrick outcome levels evaluated (level 1, Reaction; level 2, Learning; level 2+, Applied Learning; level 3, Behavior; and level 4, Results). Filled circles denote assessed levels; column totals match those in Table 3. Each study is assigned to 1 tier and 1 primary facilitation mode but can contribute to multiple outcome levels [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref36">36</xref>].</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="i-jmr_v15i1e96228_fig03.png"/></fig></sec><sec id="s3-4"><title>The TTX Design Spectrum: 3 Tiers of Exercise Purpose</title><sec id="s3-4-1"><title>Overview</title><p>Through iterative comparison across studies, we grouped the 13 included studies into 3 initial tiers based on each TTX&#x2019;s primary goal and scope. This grouping was consistent across 3 key dimensions: learning objectives, participant makeup, and outcome focus. Tier 1 involved proficiency with single-skill triage algorithms among relatively uniform prehospital groups, tier 2 covered integrated decision-making involving multiple competencies within multidisciplinary clinical teams, and tier 3 included nonclinical stakeholders, focusing on system readiness, organizational coordination, and policy changes. These tiers are preliminary patterns derived inductively and should be validated with larger samples.</p></sec><sec id="s3-4-2"><title>Tier 1: Algorithm Rehearsal</title><p>Two studies examined how individual responders interacted with a triage protocol by using TTXs to practice and assess specific procedures in controlled settings [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. These exercises evaluated triage accuracy, used standardized scoring, and tested performance within short time frames. McGlynn et al [<xref ref-type="bibr" rid="ref26">26</xref>] used patient groups of 10, 100, and 1000 casualties, while Cheng et al [<xref ref-type="bibr" rid="ref25">25</xref>] compared 6 pediatric triage algorithms. Participants were mainly prehospital professionals trained in the relevant protocols. This phase focused on measuring individual accuracy and the correct application of algorithms rather than on team coordination or broader decision-making contexts.</p></sec><sec id="s3-4-3"><title>Tier 2: Scenario-Based Decision Training</title><p>Seven studies focused on interaction within interdisciplinary teams, using TTXs to enhance multicompetency decision-making under realistic operational constraints [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref32">32</xref>]. These exercises went beyond triage to involve incident command, team coordination, resource management, CBRNE (Chemical, Biological, Radiological, Nuclear, Explosive) response, leadership, and ethical reasoning. Duration varied from 25 minutes to 2 days, with 1&#x2010;4 scenarios per session. Scenarios were tailored to specific contexts and included incidents such as a hydrofluoric acid chemical plant explosion [<xref ref-type="bibr" rid="ref27">27</xref>]; fire, terrorism, and active shooter events [<xref ref-type="bibr" rid="ref28">28</xref>]; mass shootings and fires [<xref ref-type="bibr" rid="ref4">4</xref>]; 4 CBRNE situations [<xref ref-type="bibr" rid="ref29">29</xref>]; a multivehicle accident [<xref ref-type="bibr" rid="ref30">30</xref>]; triage across multiple disaster protocols [<xref ref-type="bibr" rid="ref31">31</xref>]; and a building-collapse MCI scenario [<xref ref-type="bibr" rid="ref32">32</xref>]. Participants were typically multidisciplinary, comprising prehospital personnel, emergency department staff, primary care providers, and sometimes public health officials. This tier prioritized integrated team decision-making under time constraints and uncertain information, rather than isolated skill assessment.</p></sec><sec id="s3-4-4"><title>Tier 3: Systems Integration Testing</title><p>Four studies examined interorganizational and agency interactions through TTXs to evaluate coordination, collaboration, and policy response strategies [<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]. These exercises engaged multiple agencies beyond clinical staff, such as hospital leaders, emergency services, firefighters, police, and other government entities. The scenarios included pediatric MCI management after a school bus crash [<xref ref-type="bibr" rid="ref34">34</xref>], a CBRNE response with decontamination and triage [<xref ref-type="bibr" rid="ref33">33</xref>], a combined suicide bombing and terrorist attack involving firearms [<xref ref-type="bibr" rid="ref35">35</xref>], and postgraduate disaster medicine training focused on leadership and system coordination [<xref ref-type="bibr" rid="ref36">36</xref>]. Unlike tiers 1 and 2, this level prioritized organizational preparedness, clarifying agency roles, formal after-action reviews, policy documentation, memoranda of understanding, and ongoing postexercise system improvements over immediate individual training.</p></sec></sec><sec id="s3-5"><title>Facilitation Approaches Across Tiers</title><p>Across the 13 studies, 5 recurring facilitation modes were identified, each associated with particular tiers and learning goals. These modes reflected differences in the facilitator&#x2019;s main role, such as guiding participants through a scenario, providing feedback on performance, ensuring consistent assessment, or supporting postexercise review. The following subsections describe the characteristics of each mode and its use across the 3 tiers.</p><sec id="s3-5-1"><title>Instructional Guide (Tier 2 Predominant)</title><p>Six tier 2 studies incorporated active instructional guidance in exercise scenarios [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref32">32</xref>]. Facilitators managed the scenario flow by providing timed information, posing clarifying questions, and offering prompts when decision-making slowed. They also adjusted scenario elements such as casualty numbers, resource availability, and time constraints to enhance learning. After each scenario, structured debriefings took place, often led by facilitators who asked questions to reinforce essential learning points.</p></sec><sec id="s3-5-2"><title>Performance Coach (Tier 2)</title><p>The study by Phattharapornjaroen et al [<xref ref-type="bibr" rid="ref28">28</xref>] involved placing 2 supervisory personnel in each exercise group, explicitly tasked with observing and giving iterative feedback on team leadership behaviors and decision-making quality, using a combination of live observation and structured feedback.</p></sec><sec id="s3-5-3"><title>Standardization Agent (Tier 1 Predominant)</title><p>McGlynn et al [<xref ref-type="bibr" rid="ref26">26</xref>] and Cheng et al [<xref ref-type="bibr" rid="ref25">25</xref>] emphasized measurement consistency and the reliability of assessments by using standardized facilitation protocols [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Facilitators underwent rater calibration training, used reference sheets and standardized scoring guides, and followed scripted scenario presentations to ensure interrater reliability.</p></sec><sec id="s3-5-4"><title>Expert-Led Facilitation (Tier 3)</title><p>Cicero et al [<xref ref-type="bibr" rid="ref34">34</xref>] and Skryabina et al [<xref ref-type="bibr" rid="ref35">35</xref>] involved senior clinicians and emergency preparedness specialists in conducting exercises and guiding after-action reports. Their role therefore extended beyond scenario delivery to postexercise review, consistent with tier 3&#x2019;s focus on organizational preparedness and interagency coordination.</p></sec><sec id="s3-5-5"><title>Multidisciplinary Facilitation (Tier 3)</title><p>Anan et al [<xref ref-type="bibr" rid="ref33">33</xref>] and Zimmerman et al [<xref ref-type="bibr" rid="ref36">36</xref>] employed multidisciplinary facilitation teams from various agencies to guarantee scenario realism and credibility within each participant&#x2019;s operational setting. In both modes, tier 3 facilitation emphasized interagency collaboration and systems-level learning over individual skill evaluation. <xref ref-type="fig" rid="figure4">Figure 4</xref> shows the alignment between TTX tiers and facilitation modes.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>TTX tier&#x2013;facilitation mode alignment (N=13). Alluvial diagram illustrating the connection between TTX Design Spectrum tiers (on the left) and primary facilitation modes (on the right). Blue represents tier 1 (algorithm rehearsal), brown represents tier 2 (scenario-based decision training), and green represents tier 3 (systems integration testing). The lighter-shaded flows retain the color of their source tier, and flow width corresponds to the number of studies. Each study is linked to 1 specific tier and 1 primary facilitation mode. TTX: tabletop exercise.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="i-jmr_v15i1e96228_fig04.png"/></fig></sec></sec><sec id="s3-6"><title>Evaluation Strategies and Kirkpatrick Mapping</title><sec id="s3-6-1"><title>Data Collection Instruments and Timing</title><p>The 13 studies used diverse data collection methods. Tools for Kirkpatrick levels 1&#x2010;2 included knowledge tests (used in 8 studies), satisfaction ratings (7 studies), confidence assessments (5 studies), and self-reports on knowledge, competency, or preparedness (also 5 studies). Reaction measures extended to perceived usefulness, usability, and relevance ratings. <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> details the specific instruments mapped to each outcome level [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]. For level 2+ (Applied Learning), assessment methods included scoring triage accuracy and speed against reference standards [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref31">31</xref>], structured observation to evaluate team behaviors, task performance, decision quality during exercises [<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref36">36</xref>], and after-action reports assessed across sequential scenarios [<xref ref-type="bibr" rid="ref33">33</xref>].</p><p>Level 3 (Behavior) evidence includes documented use of exercise-based skills in real operational contexts. Two studies reported this level of evidence. Skryabina et al [<xref ref-type="bibr" rid="ref35">35</xref>] observed that participants used exercise-based knowledge and decision-making during the Manchester Arena bombing response. Zimmerman et al [<xref ref-type="bibr" rid="ref36">36</xref>] described postprogram role application, with participants taking on roles such as departmental preparedness coordinator and simulation instructor-candidate, suggesting early signs of transferring practice beyond the exercise environment.</p><p>Level 4 data pertain to organizational change. Cicero et al [<xref ref-type="bibr" rid="ref34">34</xref>] developed hospital preparedness metrics based on a 2-year postexercise follow-up. Skryabina et al [<xref ref-type="bibr" rid="ref35">35</xref>] observed improvements in casualty distribution across the system and enhanced operational efficiency during the actual MCI response. Zimmerman et al [<xref ref-type="bibr" rid="ref36">36</xref>] noted early organizational impacts, such as establishing a structured simulation platform, conducting additional course iterations, holding refresher meetings, engaging in alumni follow-up activities, and continuously refining contingency plans. Most studies used pre- and postassessment timings, with some also including during-exercise measurements such as triage accuracy and incident command compliance. Additionally, 1 study incorporated a 1-week delayed posttest to assess knowledge retention [<xref ref-type="bibr" rid="ref31">31</xref>].</p></sec><sec id="s3-6-2"><title>Kirkpatrick-Tier Mapping</title><p>Across the 13 studies, the distribution of Kirkpatrick levels was as follows: level 1 (Reaction) n=10, level 2 (Learning) n=10, level 2+ (Applied Learning) n=8, level 3 (Behavior) n=2, and level 4 (Results) n=3. The level 2+ category indicates a measurement pattern that the standard 4-level model does not clearly depict. The study-level distribution of Kirkpatrick outcome assessments across tiers is shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>, with detailed outcome mapping by study available in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>. Evaluation depth varied between tiers (<xref ref-type="table" rid="table3">Table 3</xref>). Outcomes for levels 3 and 4 were assessed solely in systems integration (tier 3) studies [<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref36">36</xref>], aligning with this tier&#x2019;s emphasis on systems-level readiness and real-world validation opportunities.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Kirkpatrick outcome assessment by Tabletop Exercise Design Spectrum tier (number of studies assessing each level).<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Tier</td><td align="left" valign="bottom">Studies</td><td align="left" valign="bottom">Level 1: Reaction</td><td align="left" valign="bottom">Level 2: Learning</td><td align="left" valign="bottom">Level 2+ (Applied Learning)</td><td align="left" valign="bottom">Level 3: Behavior</td><td align="left" valign="bottom">Level 4: Results</td></tr></thead><tbody><tr><td align="left" valign="top">1: algorithm rehearsal</td><td align="left" valign="top">2</td><td align="left" valign="top">1</td><td align="left" valign="top">0</td><td align="left" valign="top">2</td><td align="left" valign="top">0</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top">2: scenario-based decision training</td><td align="left" valign="top">7</td><td align="left" valign="top">6</td><td align="left" valign="top">7</td><td align="left" valign="top">4</td><td align="left" valign="top">0</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top">3: systems integration testing</td><td align="left" valign="top">4</td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td></tr><tr><td align="left" valign="top">Total (N=13)</td><td align="left" valign="top">13</td><td align="left" valign="top">10</td><td align="left" valign="top">10</td><td align="left" valign="top">8</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Cell values represent the number of studies evaluating each level; a single study can contribute to multiple levels. Level 2+ (Applied Learning) indicates structured performance assessments conducted within the exercise environment (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>).</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s3-7"><title>Frameworks Guiding Exercise Design and Evaluation</title><sec id="s3-7-1"><title>Overview</title><p><xref ref-type="fig" rid="figure5">Figure 5</xref> provides a summary of how training design, assessment, and operational frameworks are reported across the 13 included studies. Operational frameworks were documented most consistently, while specific educational design and assessment frameworks appeared less frequently.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Framework reporting landscape across included studies (N=13). Matrix chart displaying framework reporting across studies, categorized into 8 groups: formal educational design, domain-specific design, assessment frameworks, and 5 operational areas: incident command (ICS/HICS), triage protocols, CBRNE-specific frameworks, coordination mechanisms (such as memoranda of understanding and after-action reports), and other operational frameworks (such as SIRATTE; MRMI/MACSIM). Columns represent individual studies, color-coded by TTX Design Spectrum tier; filled cells denote reported frameworks, while light gray cells show frameworks not reported [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]. AAR: after-action report; CBRNE: Chemical, Biological, Radiological, Nuclear, Explosive; HICS: Hospital Incident Command System; ICS: Incident Command System; MACSIM: Mass Casualty Simulation System; MOU: Memorandum of Understanding; MRMI: Medical Response to Major Incidents; SIRATTE: Security, Information, Roles, Areas, Triage, Treatment, and Evacuation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="i-jmr_v15i1e96228_fig05.png"/></fig></sec><sec id="s3-7-2"><title>Training Design Frameworks</title><p>Out of 13 studies, only 3 explicitly described formal educational training frameworks. Chumvanichaya et al [<xref ref-type="bibr" rid="ref31">31</xref>] used backward design, focusing the exercise on previously identified preparedness gaps. Sena et al [<xref ref-type="bibr" rid="ref4">4</xref>] applied Kolb&#x2019;s [<xref ref-type="bibr" rid="ref37">37</xref>] experiential learning model and adult learning principles to shape scenario development and debriefings. Zimmerman et al [<xref ref-type="bibr" rid="ref36">36</xref>] integrated disaster medicine education and competency frameworks into postgraduate training. Seven studies did not specify any formal educational design framework [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref34">34</xref>]. Meanwhile, 3 studies used domain-specific frameworks instead of broad educational models: Kim et al [<xref ref-type="bibr" rid="ref27">27</xref>] organized CBRNE response phases around the Chain of Chemical Survival, Phattharapornjaroen et al [<xref ref-type="bibr" rid="ref28">28</xref>] applied the 3LC (3-Level Collaboration) framework with Major Incident Medical Management and Support or Medical Response to Major Incidents triage protocols, and Skryabina et al [<xref ref-type="bibr" rid="ref35">35</xref>] based their approach on the Emergency Preparedness Cycle and Emergo Train System. Overall, most research relied more on disaster response or operational frameworks than on explicit educational design theories.</p></sec><sec id="s3-7-3"><title>Assessment Frameworks</title><p>Assessment frameworks were described more clearly than training design frameworks. Phattharapornjaroen et al [<xref ref-type="bibr" rid="ref28">28</xref>] used the CSCATTT (Command, Safety, Communication, Assessment, Triage, Treatment, Transportation) framework to organize and assess incident command performance. Cicero et al [<xref ref-type="bibr" rid="ref34">34</xref>] used the Hospital Pediatric Disaster Preparedness Checklist and after-action reports to evaluate organizational readiness. Skryabina et al [<xref ref-type="bibr" rid="ref35">35</xref>] integrated Kirkpatrick levels with the Promoting Action on Research Implementation in Health Services framework to analyze both training outcomes and implementation processes. Other studies used specific structured tools for assessments: Chumvanichaya et al [<xref ref-type="bibr" rid="ref31">31</xref>] applied the Attention, Relevance, Confidence, Satisfaction Motivation Model to measure participant engagement, Alakrawi et al [<xref ref-type="bibr" rid="ref30">30</xref>] adapted the Competency Level Use in Triage Scale to evaluate triage competency, and Zimmerman et al [<xref ref-type="bibr" rid="ref36">36</xref>] used a 29-item questionnaire along with CSCATTT-anchored instructor scores to assess applied performance during simulations.</p></sec><sec id="s3-7-4"><title>Operational Frameworks</title><p>Operational frameworks were reported more consistently than educational design frameworks. Six studies mentioned the Incident Command System or Hospital Incident Command System [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. Triage protocols, such as Simple Triage and Rapid Treatment; Sort, Assess, Lifesaving Interventions, Treatment/Transport; Sieve (UK Major Incident Primary Triage); Secondary Triage (UK Major Incident); JumpSTART; CareFlight; Pediatric Triage Tape; and Sacco triage, were the most common operational frameworks used across different tiers [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. Three studies used CBRNE-specific response frameworks: the Chain of Chemical Survival [<xref ref-type="bibr" rid="ref27">27</xref>], CBRNE workshop competencies [<xref ref-type="bibr" rid="ref29">29</xref>], and the MCLS-CBRNE course framework [<xref ref-type="bibr" rid="ref33">33</xref>]. Tier 3 studies more frequently included coordination mechanisms such as memoranda of understanding, centralized dispatch protocols, interagency responsibilities, and after-action reports [<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref35">35</xref>]. Cuartas-Alvarez et al [<xref ref-type="bibr" rid="ref32">32</xref>] used SIRATTE, a prehospital MCI framework that covers Security, Information, Roles, Areas, Triage, Treatment, and Evacuation, in the MassCas tabletop game. Zimmerman et al [<xref ref-type="bibr" rid="ref36">36</xref>] incorporated several operational and collaborative frameworks, such as Incident Command System, CSCATTT, 3LC, Medical Response to Major Incidents, and Mass Casualty Simulation System, into its postgraduate disaster medicine program.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>The main conclusion of this scoping review is that TTXs in prehospital MCI preparedness should be viewed as a group of related exercise types, each serving different educational and operational goals, rather than a single, uniform intervention. These purposes are categorized into 3 tiers: algorithm rehearsal, scenario-based decision training, and systems integration testing. Outcomes are mapped using the Kirkpatrick framework, with level 2+ indicating structured performance during the exercise that lies conceptually between knowledge acquisition and real-world application. These tiers represent increasing levels of interaction: between individual responders and protocols, among interdisciplinary team members, and across organizations and agencies. Facilitation modes involve intentionally designed interactions between facilitators and participants. While diverse exercise designs are common and often suitable, the key message is that the purpose, facilitation, and evaluation of the exercise must be explicitly aligned. Without a shared typology of purpose, evidence from exercises with different goals cannot be meaningfully compared. As shown in <xref ref-type="fig" rid="figure6">Figure 6</xref>, the literature indicates a pattern where exercise purpose, facilitation mode, and evaluation depth progress together.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Integrated overview of TTX purpose, facilitation, and evaluation depth across included studies (N=13). The figure outlines the included studies across 3 interconnected layers: (1) the TTX Design Spectrum tier, (2) the main facilitation modes associated with each tier, and (3) the evaluation depth categorized by Kirkpatrick levels, including level 2+ (Applied Learning). CBRNE: Chemical, Biological, Radiological, Nuclear, Explosive; TTX: tabletop exercise.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="i-jmr_v15i1e96228_fig06.png"/></fig><p>The tier typology should be seen as an initial guideline rather than a confirmed classification. Nonetheless, it clarifies why studies labeled as TTXs often appeared quite different in practice: algorithm-rehearsal exercises focused on standardized practice of specific skills [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>], scenario-based decision training involved integrated clinical and command decisions in discussion settings [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref32">32</xref>], and systems-integration exercises targeted organizational coordination and system readiness [<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]. Exercises aimed at rehearsing an algorithm, enhancing team decision-making, or testing system coordination do not share the same facilitation approach or outcome metrics.</p><p>The Design Spectrum should be aligned with existing exercise doctrine. The HSEEP (Homeland Security Exercise and Evaluation Program) connects exercise goals to structured assessments using capability-based exercise evaluation guides and standardized after-action reports [<xref ref-type="bibr" rid="ref38">38</xref>]. However, HSEEP treats TTXs as a single, uniform discussion-based type. The current typology offers a more detailed classification within that category, identifying different types of TTXs that require varied facilitation and evaluation approaches. This typology is designed to complement HSEEP objectives and not replace them.</p><p>This has a practical implication for anyone designing or assessing a TTX: the evaluation approach should be aligned with the exercise&#x2019;s goal, not judged by a universal standard of assessment thoroughness. A focused protocol rehearsal may rightly prioritize structured performance metrics such as decision accuracy and interrater reliability, without necessarily claiming to measure organizational impact [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Meanwhile, a decision-training exercise might aim to improve knowledge, confidence, communication, and role clarity but should ideally incorporate follow-up methods if it seeks to influence future practice [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. A systems-centered exercise should analyze after-action processes, coordination structures, and readiness outputs, as these are its core focus [<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]. The key point is not whether every TTX achieves the highest Kirkpatrick level but whether the chosen level is appropriate for the capability being tested.</p></sec><sec id="s4-2"><title>Separating Applied Learning From Behavioral Transfer</title><p>The level 2+ subcategory is valuable because it differentiates between applied performance during the exercise and behavioral transfer outside of it. Multiple studies have assessed applied performance during the tabletop, providing evidence of learning in a simulated environment [<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. While this evidence is educationally significant, it does not necessarily reflect practice change in real settings. Making this distinction helps prevent overinterpreting simulation results as proof of real-world transfer. This differentiation is not new; models such as the revised Kirkpatrick framework, Miller&#x2019;s framework, and translational science models of simulation-based education outcomes all recognize it [<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref24">24</xref>]. However, its systematic application to prehospital MCI tabletop evidence has not been reported before. Broader reviews have generally shown clustering at the Reaction and Learning levels without separating structured in-exercise performance from actual operational transfer [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>This distinction also assists in understanding the limited occurrence of specific behaviors and outcome results. Their absence should not be automatically interpreted as failure. In prehospital disaster education, actual events are rare, unpredictable, and difficult to tie to a single training session [<xref ref-type="bibr" rid="ref39">39</xref>], making it challenging to observe behavioral transfer even when learning has occurred. Additionally, follow-up methods in scenario-based decision-training studies were often limited [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], with only a single delayed posttest after 1 week [<xref ref-type="bibr" rid="ref31">31</xref>], which hampers the ability to determine whether these exercises impacted later operational reasoning. This does not imply that such exercises are ineffective but rather that they lacked sufficient follow-up relative to their expected outcomes, highlighting a gap in evaluation data rather than a flaw in design.</p></sec><sec id="s4-3"><title>Timing, Transfer, and Facilitation</title><p>The study by Skryabina et al [<xref ref-type="bibr" rid="ref35">35</xref>] provides rare naturalistic evidence that exercise learning can be applied in real responses when training is close to deployment. It found that participants reported using knowledge from the TTX during the Manchester Arena incident, which occurred soon afterward. Although this single uncontrolled observation cannot establish causality, it aligns with the literature on distributed practice, which shows that the spacing and timing of learning opportunities influence retention and transfer [<xref ref-type="bibr" rid="ref40">40</xref>], and with the advice by Cicero et al [<xref ref-type="bibr" rid="ref34">34</xref>] to exercise every 6 months [<xref ref-type="bibr" rid="ref34">34</xref>]. While single-exposure TTXs may support learning, lasting behavior change likely depends more on repetition and deliberate timing than on a one-time exposure. This remains an open question for future research rather than a definitive conclusion of this review.</p><p>Facilitation findings consistently indicate that it is a strategic design choice rather than a neutral delivery method, and that it varies with the exercise&#x2019;s purpose (<xref ref-type="fig" rid="figure4">Figure 4</xref>). More directive, standardized facilitation is better suited for narrowly focused rehearsal activities, while expert-led, coaching, or multidisciplinary facilitation is typically used in broader scenario-based and system-level exercises [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. There is no single facilitation style that is universally superior; instead, the facilitator&#x2019;s role should be aligned with the exercise&#x2019;s goals, whether to standardize performance, encourage collaborative reasoning, identify coordination gaps, or foster organizational reflection.</p></sec><sec id="s4-4"><title>Frameworks and Implementation</title><p>Operational frameworks were described more consistently than educational or assessment frameworks (<xref ref-type="fig" rid="figure5">Figure 5</xref>). Only a few studies explicitly detailed formal educational frameworks such as experiential learning, backward design, or competency-based disaster medicine education [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. While the field is grounded in operational principles, its pedagogical aspects are less well articulated. Operational doctrine can effectively shape scenarios, but inconsistent reporting of educational and assessment approaches hampers reproducibility and obscures which design choices actually enhance learning or transfer. Notably, the combination of the Kirkpatrick model with the Promoting Action on Research Implementation in Health Services implementation framework by Skryabina et al [<xref ref-type="bibr" rid="ref35">35</xref>] stands out, as it connects training outcomes with implementation processes rather than viewing evaluation as an isolated end point [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. This approach demonstrates how exercise evaluation can more systematically contribute to system-level preparedness improvements.</p><p>These findings generally align with earlier review literature but offer a more specific interpretive perspective. Fr&#x00E9;geau et al [<xref ref-type="bibr" rid="ref7">7</xref>] observed outcomes primarily at lower Kirkpatrick levels in medical emergency contexts, while Emaliyawati et al [<xref ref-type="bibr" rid="ref8">8</xref>] noted that tabletop disaster exercises enhance knowledge, confidence, preparedness, and performance. This review differs by focusing on the prehospital MCI interface, introducing the level 2+ category, and suggesting that the main challenge is purpose-aligned evaluation rather than measurement deficiencies. This distinction is practically valuable, helping educators and planners determine the kind of evidence they can reasonably expect from various exercises.</p><p>The clearest takeaway from practice is that TTXs should be designed starting from the desired capability. If the aim is to rehearse algorithms, then structured in-exercise performance measures might suffice as end points. For scenario-based decision training, evaluation should go beyond participant satisfaction and include follow-up methods such as delayed case-based testing, repeated scenarios, structured interviews, or local practice audits. If the focus is on systems integration, exercises should connect to after-action reviews, policy updates, coordination metrics, or preparedness planning [<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]. This review advocates for a proportional evaluation approach: not every exercise needs maximum evaluation depth, but each should have an evaluation strategy aligned with its purpose. <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref> summarizes these pairings, listing typical objectives, facilitation modes, evaluation targets, example instruments from the studies, and suggested data to record [<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref36">36</xref>].</p></sec><sec id="s4-5"><title>Future Research</title><p>Future research should focus on refining this purpose alignment approach instead of merely questioning whether TTXs &#x201C;work&#x201D; in general. One key area is to develop practical methods for measuring transfer following scenario-based decision-training exercises without relying on the occurrence of a real MCI: options include repeating scenarios, using follow-up decision vignettes, conducting interval audits of triage or command processes, and linking short-format exercises as possible next steps. Another important area is to investigate whether repeated or ongoing exposure to related TTXs enhances retention and transfer more than isolated sessions. This is especially relevant since the current review does not establish a dose-response relationship; however, the training-transfer literature suggests that this is plausible because transfer relies on reinforcement, opportunities for application, and workplace conditions [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>].</p><p>A third priority involves stakeholder-based implementation research, engaging EMS leaders, emergency preparedness officers, public health partners, hospital interface stakeholders, and simulation faculty in designing exercise objectives and follow-up plans. This ensures that evaluations align with real operational needs. It should also include a more structured, comparable collection of evaluation data, such as observation ratings, accuracy scores, and after-action findings, using existing templates such as HSEEP&#x2019;s Exercise Evaluation Guides and standardized reporting [<xref ref-type="bibr" rid="ref38">38</xref>]. A fourth priority is expanding scenarios beyond specific, bounded incidents. Predictable hazards such as storms, wildfires, floods, or hurricanes could be especially valuable, as threat windows can sometimes be forecasted. Advances in satellite hazard assessments, artificial intelligence&#x2013;enabled risk estimation, and forecasting may help identify these periods and enable more realistic timing of training relative to operational demand [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. These designs could also examine whether training closer to deployment improves transfer.</p></sec><sec id="s4-6"><title>Limitations</title><p>Several limitations should inform how these findings are interpreted. First, the evidence base was limited in size and diverse in nature, so the proposed tiers and alignment patterns should be viewed as preliminary, hypothesis-generating concepts rather than confirmed frameworks. Specifically, since the tier definitions partly depended on participant composition and outcome scope, the link between tier and evaluation depth is partly inherent and should not be seen as an independent empirical correlation. Second, this was a scoping review without formal critical appraisal; thus, the findings depict the state of the literature rather than comparing the effectiveness of different TTX designs. Most studies included used pre-post, quasi-experimental, descriptive, or mixed methods approaches, which restrict causal conclusions.</p><p>Third, the study reporting was inconsistent. The definitions of outcomes, facilitation descriptions, and framework reporting varied, leading to interpretive uncertainty when aligning studies across tiers and Kirkpatrick levels. The level 2+ subcategory enhances conceptual clarity, but its assignment still relied on reviewer judgment. Fourth, the literature was mainly from high-income or upper-middle-income settings and focused on human-caused or bounded hazard scenarios, restricting its applicability to resource-limited environments and prolonged natural disasters (<xref ref-type="fig" rid="figure2">Figure 2</xref>). Additionally, limiting the review to English-language sources might have excluded important evidence.</p><p>Two of the review&#x2019;s more speculative points rely on limited evidence. The idea that temporal proximity between training and real events might facilitate transfer primarily derives from a single naturalistic case [<xref ref-type="bibr" rid="ref35">35</xref>]. While interesting, it remains unconfirmed. Similarly, the notion that short, time-limited tabletop formats may lack the functional fidelity, according to the definition by Hamstra et al [<xref ref-type="bibr" rid="ref46">46</xref>], to simulate long-term, evolving disasters that disrupt infrastructure is speculative. The existing literature primarily involves bounded incidents and does not directly test this idea. Both points are presented as suggestions for future research rather than confirmed explanations.</p></sec><sec id="s4-7"><title>Conclusions</title><p>TTXs for prehospital MCI preparedness should not be seen as a single, uniform approach. Their success mainly depends on how clearly the designers define the expected level of interaction. Whether it is an individual following protocols, a multidisciplinary team working together, or organizations coordinating efforts, facilitation and evaluation should be tailored accordingly. By concentrating on the prehospital MCI literature and differentiating between applied learning during exercises (level 2+) and actual behavioral transfer, while considering variations through the lens of design&#x2013;evaluation alignment, this review offers a targeted synthesis of the prehospital MCI interface that complements earlier, broader tabletop reviews [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. When used carefully, TTXs can remain a scalable and strategic component of prehospital disaster preparedness, with further research needed to determine when and how they promote lasting practice change and system readiness.</p></sec></sec></body><back><ack><p>The authors express gratitude to Dr Sarah Kazim, chair of emergency medicine at Dubai Health, for her support and for creating a departmental environment conducive to this work. They also appreciate Mohammed Bin Rashid University of Medicine and Health Sciences for its support and collaboration, Graduate Medical Education for assisting the residents, Shakeel Tegginmani of the Al Maktoum Medical Library for helping with the literature search, and the Institute of Learning for providing research support. Generative AI tools, such as ChatGPT and Perplexity, were used solely for editorial tasks, such as enhancing language clarity, readability, grammar, and organization, as well as brainstorming visual presentation ideas for data already extracted and analyzed by the authors. AI was not used for data generation, literature searches, study screening, data extraction, outcome classification, analysis, interpretation, or forming scientific conclusions. All AI-suggested edits were carefully reviewed, revised, and approved by the authors. The authors assume full responsibility for the accuracy, integrity, and final content of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This research was supported by the Mohammed Bin Rashid University of Medicine and Health Sciences via the Dubai Health Collaborative Stimulus Research Grant for the MCIPHER project. The funding body had no role in the study design, data analysis, or manuscript writing.</p></sec><sec><title>Data Availability</title><p>All data produced or examined in this study are included within this published paper and its multimedia appendices. The full data charting form can be obtained from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: AO, AY, NZ</p><p>Data curation: PP, AA, DN, AO</p><p>Formal analysis: AO</p><p>Investigation: PP, AA, DN</p><p>Methodology: PP, AA, DN, AO, IH, NZ</p><p>Project administration: AY</p><p>Resources: AY</p><p>Supervision: AO, AY, IH, NZ</p><p>Visualization: NZ</p><p>Writing &#x2013; original draft: PP, DN, AO</p><p>Writing &#x2013; review &#x0026; editing: PP, AA, AY, IH, NZ</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CBRNE</term><def><p>Chemical, Biological, Radiological, Nuclear, Explosive</p></def></def-item><def-item><term id="abb2">CSCATTT</term><def><p>Command, Safety, Communication, Assessment, Triage, Treatment, Transportation</p></def></def-item><def-item><term id="abb3">EMS</term><def><p>emergency medical services</p></def></def-item><def-item><term id="abb4">HSEEP</term><def><p>Homeland Security Exercise and Evaluation Program</p></def></def-item><def-item><term id="abb5">MCI</term><def><p>mass casualty incident</p></def></def-item><def-item><term id="abb6">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb7">PRISMA-S</term><def><p>PRISMA Extension for Reporting Literature Searches</p></def></def-item><def-item><term id="abb8">PRISMA-ScR</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews</p></def></def-item><def-item><term id="abb9">3LC</term><def><p>3-Level Collaboration</p></def></def-item><def-item><term id="abb10">TTX</term><def><p>tabletop exercise</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>Sendai Framework for Disaster Risk Reduction 2015-2030</article-title><source>United Nations Office for Disaster Risk Reduction</source><year>2015</year><access-date>2026-09-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.undrr.org/publication/sendai-framework-disaster-risk-reduction-2015-2030">https://www.undrr.org/publication/sendai-framework-disaster-risk-reduction-2015-2030</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bissmeyer</surname><given-names>H</given-names> </name><name name-style="western"><surname>Gallegos</surname><given-names>C</given-names> </name><name name-style="western"><surname>Randazzo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Scheffler</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sullivan Lee</surname><given-names>L</given-names> </name><name name-style="western"><surname>Powell</surname><given-names>L</given-names> </name></person-group><article-title>Tabletop simulation as an innovative tool for clinical workflow testing</article-title><source>Nurs Adm Q</source><year>2025</year><volume>49</volume><issue>1</issue><fpage>E1</fpage><lpage>E7</lpage><pub-id pub-id-type="doi">10.1097/NAQ.0000000000000664</pub-id><pub-id pub-id-type="medline">39622034</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Castro Delgado</surname><given-names>R</given-names> </name><name name-style="western"><surname>Fern&#x00E1;ndez Garc&#x00ED;a</surname><given-names>L</given-names> </name><name name-style="western"><surname>Cernuda Mart&#x00ED;nez</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Cuartas &#x00C1;lvarez</surname><given-names>T</given-names> </name><name name-style="western"><surname>Arcos Gonz&#x00E1;lez</surname><given-names>P</given-names> </name></person-group><article-title>Training of medical students for mass casualty incidents using table-top gamification</article-title><source>Disaster Med Public Health Prep</source><year>2022</year><month>09</month><day>21</day><volume>17</volume><fpage>e255</fpage><pub-id pub-id-type="doi">10.1017/dmp.2022.206</pub-id><pub-id pub-id-type="medline">36128647</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sena</surname><given-names>A</given-names> </name><name name-style="western"><surname>Forde</surname><given-names>F</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sule</surname><given-names>H</given-names> </name><name name-style="western"><surname>Masters</surname><given-names>MM</given-names> </name></person-group><article-title>Disaster preparedness training for emergency medicine residents using a tabletop exercise</article-title><source>MedEdPORTAL</source><year>2021</year><month>03</month><day>12</day><volume>17</volume><fpage>11119</fpage><pub-id pub-id-type="doi">10.15766/mep_2374-8265.11119</pub-id><pub-id pub-id-type="medline">33768151</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>B</given-names> </name><name name-style="western"><surname>Gui</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>L</given-names> </name></person-group><article-title>Development and evaluation of innovative and practical table-top exercises based on a real mass-casualty incident</article-title><source>Disaster Med Public Health Prep</source><year>2022</year><month>05</month><day>16</day><volume>17</volume><fpage>e200</fpage><pub-id pub-id-type="doi">10.1017/dmp.2022.95</pub-id><pub-id pub-id-type="medline">35575292</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Torpan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Orru</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hansson</surname><given-names>S</given-names> </name><name name-style="western"><surname>Klaos</surname><given-names>M</given-names> </name></person-group><article-title>Using a table-top exercise to identify communication-related vulnerability to disasters</article-title><source>Int J Disaster Risk Reduct</source><year>2025</year><month>03</month><volume>119</volume><fpage>105264</fpage><pub-id pub-id-type="doi">10.1016/j.ijdrr.2025.105264</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fr&#x00E9;geau</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vinette</surname><given-names>B</given-names> </name><name name-style="western"><surname>Lapierre</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Tabletop simulations in medical emergencies: a scoping review</article-title><source>Simul Healthc</source><year>2025</year><month>08</month><day>1</day><volume>20</volume><issue>4</issue><fpage>223</fpage><lpage>228</lpage><pub-id pub-id-type="doi">10.1097/SIH.0000000000000838</pub-id><pub-id pub-id-type="medline">39625695</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Emaliyawati</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ibrahim</surname><given-names>K</given-names> </name><name name-style="western"><surname>Trisyani</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Enhancing disaster preparedness through tabletop disaster exercises: a scoping review of benefits for health workers and students</article-title><source>Adv Med Educ Pract</source><year>2025</year><volume>16</volume><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.2147/AMEP.S504705</pub-id><pub-id pub-id-type="medline">39807178</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mutiarasari</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zulkifli</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rivai</surname><given-names>F</given-names> </name><name name-style="western"><surname>Thamrin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Mallongi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Miranti</surname><given-names>M</given-names> </name></person-group><article-title>The effectiveness of table-top exercise simulation for hospital disaster preparedness training a systematic review</article-title><source>J Neonatal Surg</source><year>2025</year><volume>14</volume><issue>9S</issue><fpage>102</fpage><lpage>110</lpage><pub-id pub-id-type="doi">10.52783/jns.v14.2634</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khirekar</surname><given-names>J</given-names> </name><name name-style="western"><surname>Badge</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bandre</surname><given-names>GR</given-names> </name><name name-style="western"><surname>Shahu</surname><given-names>S</given-names> </name></person-group><article-title>Disaster preparedness in hospitals</article-title><source>Cureus</source><year>2023</year><month>12</month><volume>15</volume><issue>12</issue><fpage>e50073</fpage><pub-id pub-id-type="doi">10.7759/cureus.50073</pub-id><pub-id pub-id-type="medline">38192940</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arksey</surname><given-names>H</given-names> </name><name name-style="western"><surname>O&#x2019;Malley</surname><given-names>L</given-names> </name></person-group><article-title>Scoping studies: towards a methodological framework</article-title><source>Int J Soc Res Methodol</source><year>2005</year><month>02</month><volume>8</volume><issue>1</issue><fpage>19</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1080/1364557032000119616</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levac</surname><given-names>D</given-names> </name><name name-style="western"><surname>Colquhoun</surname><given-names>H</given-names> </name><name name-style="western"><surname>O&#x2019;Brien</surname><given-names>KK</given-names> </name></person-group><article-title>Scoping studies: advancing the methodology</article-title><source>Implement Sci</source><year>2010</year><month>09</month><day>20</day><volume>5</volume><issue>1</issue><fpage>69</fpage><pub-id pub-id-type="doi">10.1186/1748-5908-5-69</pub-id><pub-id pub-id-type="medline">20854677</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tricco</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Lillie</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zarin</surname><given-names>W</given-names> </name><etal/></person-group><article-title>PRISMA extension for scoping reviews (PRISMA-ScR): checklist and explanation</article-title><source>Ann Intern Med</source><year>2018</year><month>10</month><day>2</day><volume>169</volume><issue>7</issue><fpage>467</fpage><lpage>473</lpage><pub-id pub-id-type="doi">10.7326/M18-0850</pub-id><pub-id pub-id-type="medline">30178033</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Kazim</surname><given-names>M</given-names> </name><name name-style="western"><surname>Prashanth</surname><given-names>P</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Tabletop exercises for prehospital preparedness during emergencies: a scoping review protocol</article-title><source>protocols.io</source><year>2025</year><access-date>2026-09-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.protocols.io/view/tabletop-exercises-for-prehospital-preparedness-du-g4pebyvjf">https://www.protocols.io/view/tabletop-exercises-for-prehospital-preparedness-du-g4pebyvjf</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kazim</surname><given-names>M</given-names> </name><name name-style="western"><surname>Prashanth</surname><given-names>P</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Tabletop exercises to assess prehospital preparedness: a scoping review protocol</article-title><source>Preprints.org</source><comment>Preprint posted online on  Oct 5, 2025</comment><pub-id pub-id-type="doi">10.20944/preprints202510.0457.v1</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rethlefsen</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Kirtley</surname><given-names>S</given-names> </name><name name-style="western"><surname>Waffenschmidt</surname><given-names>S</given-names> </name><etal/></person-group><article-title>PRISMA-S: an extension to the PRISMA statement for reporting literature searches in systematic reviews</article-title><source>Syst Rev</source><year>2021</year><month>01</month><day>26</day><volume>10</volume><issue>1</issue><fpage>39</fpage><pub-id pub-id-type="doi">10.1186/s13643-020-01542-z</pub-id><pub-id pub-id-type="medline">33499930</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McGowan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sampson</surname><given-names>M</given-names> </name><name name-style="western"><surname>Salzwedel</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Cogo</surname><given-names>E</given-names> </name><name name-style="western"><surname>Foerster</surname><given-names>V</given-names> </name><name name-style="western"><surname>Lefebvre</surname><given-names>C</given-names> </name></person-group><article-title>PRESS Peer Review of Electronic Search Strategies: 2015 guideline statement</article-title><source>J Clin Epidemiol</source><year>2016</year><month>07</month><volume>75</volume><fpage>40</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2016.01.021</pub-id><pub-id pub-id-type="medline">27005575</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="web"><article-title>Covidence systematic review software</article-title><source>Veritas Health Innovation</source><access-date>2026-09-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.covidence.org">https://www.covidence.org</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cheng</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kessler</surname><given-names>D</given-names> </name><name name-style="western"><surname>Mackinnon</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Reporting guidelines for health care simulation research: extensions to the CONSORT and STROBE statements</article-title><source>Adv Simul</source><year>2016</year><month>01</month><volume>1</volume><issue>1</issue><fpage>25</fpage><pub-id pub-id-type="doi">10.1186/s41077-016-0025-y</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kirkpatrick</surname><given-names>DL</given-names> </name></person-group><article-title>Techniques for evaluating training programs</article-title><source>J Am Soc Train Dir</source><year>1959</year><access-date>2025-07-12</access-date><volume>13</volume><issue>11</issue><fpage>3</fpage><lpage>9</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://assets.td.org/m/486fb05fce23e065/original/Techniques-For-Evaluating-Training-Programs-January-1960.pdf">https://assets.td.org/m/486fb05fce23e065/original/Techniques-For-Evaluating-Training-Programs-January-1960.pdf</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Kirkpatrick</surname><given-names>DL</given-names> </name><name name-style="western"><surname>Kirkpatrick</surname><given-names>JD</given-names> </name></person-group><source>Evaluating Training Programs: The Four Levels</source><year>2006</year><edition>3</edition><publisher-name>Berrett-Koehler</publisher-name><pub-id pub-id-type="other">1576753484</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Kirkpatrick</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Kirkpatrick</surname><given-names>WK</given-names> </name></person-group><source>Kirkpatrick&#x2019;s Four Levels of Training Evaluation</source><year>2016</year><publisher-name>ATD Press</publisher-name><pub-id pub-id-type="other">9781607280088</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miller</surname><given-names>GE</given-names> </name></person-group><article-title>The assessment of clinical skills/competence/performance</article-title><source>Acad Med</source><year>1990</year><month>09</month><volume>65</volume><issue>9 Suppl</issue><fpage>S63</fpage><lpage>S67</lpage><pub-id pub-id-type="doi">10.1097/00001888-199009000-00045</pub-id><pub-id pub-id-type="medline">2400509</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McGaghie</surname><given-names>WC</given-names> </name><name name-style="western"><surname>Issenberg</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>ER</given-names> </name><name name-style="western"><surname>Barsuk</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Wayne</surname><given-names>DB</given-names> </name></person-group><article-title>Translational educational research: a necessity for effective health-care improvement</article-title><source>Chest</source><year>2012</year><month>11</month><volume>142</volume><issue>5</issue><fpage>1097</fpage><lpage>1103</lpage><pub-id pub-id-type="doi">10.1378/chest.12-0148</pub-id><pub-id pub-id-type="medline">23138127</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cheng</surname><given-names>T</given-names> </name><name name-style="western"><surname>Staats</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kaji</surname><given-names>AH</given-names> </name><name name-style="western"><surname>D&#x2019;Arcy</surname><given-names>N</given-names> </name><name name-style="western"><surname>Niknam</surname><given-names>K</given-names> </name><name name-style="western"><surname>Donofrio-Odmann</surname><given-names>JJ</given-names> </name></person-group><article-title>Comparison of prehospital professional accuracy, speed, and interrater reliability of six pediatric triage algorithms</article-title><source>J Am Coll Emerg Physicians Open</source><year>2022</year><month>02</month><volume>3</volume><issue>1</issue><fpage>e12613</fpage><pub-id pub-id-type="doi">10.1002/emp2.12613</pub-id><pub-id pub-id-type="medline">35059689</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McGlynn</surname><given-names>N</given-names> </name><name name-style="western"><surname>Claudius</surname><given-names>I</given-names> </name><name name-style="western"><surname>Kaji</surname><given-names>AH</given-names> </name><etal/></person-group><article-title>Tabletop application of SALT triage to 10, 100, and 1000 pediatric victims</article-title><source>Prehosp Disaster Med</source><year>2020</year><month>04</month><volume>35</volume><issue>2</issue><fpage>165</fpage><lpage>169</lpage><pub-id pub-id-type="doi">10.1017/S1049023X20000163</pub-id><pub-id pub-id-type="medline">32054549</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Evaluating the Effectiveness of the Chemical-Mass Casualty Incident Response Education Module (C-MCIREM): a pilot simulation study with a before and after design</article-title><source>Cureus</source><year>2021</year><month>09</month><volume>13</volume><issue>9</issue><fpage>e17980</fpage><pub-id pub-id-type="doi">10.7759/cureus.17980</pub-id><pub-id pub-id-type="medline">34667664</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Phattharapornjaroen</surname><given-names>P</given-names> </name><name name-style="western"><surname>Glantz</surname><given-names>V</given-names> </name><name name-style="western"><surname>Carlstr&#x00F6;m</surname><given-names>E</given-names> </name><name name-style="western"><surname>Dahl&#x00E9;n Holmqvist</surname><given-names>L</given-names> </name><name name-style="western"><surname>Khorram-Manesh</surname><given-names>A</given-names> </name></person-group><article-title>Alternative leadership in flexible surge capacity&#x2014;the perceived impact of tabletop simulation exercises on Thai emergency physicians capability to manage a major incident</article-title><source>Sustainability</source><year>2020</year><volume>12</volume><issue>15</issue><fpage>6216</fpage><pub-id pub-id-type="doi">10.3390/su12156216</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Farhat</surname><given-names>H</given-names> </name><name name-style="western"><surname>Laughton</surname><given-names>J</given-names> </name><name name-style="western"><surname>Joseph</surname><given-names>A</given-names> </name><name name-style="western"><surname>Abougalala</surname><given-names>W</given-names> </name><name name-style="western"><surname>Dhiab</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Alinier</surname><given-names>G</given-names> </name></person-group><article-title>The educational outcomes of an online pilot workshop in CBRNe emergencies</article-title><source>Journal of Emergency Medicine, Trauma and Acute Care</source><year>2022</year><month>11</month><day>22</day><volume>2022</volume><issue>5</issue><fpage>38</fpage><pub-id pub-id-type="doi">10.5339/jemtac.2022.38</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alakrawi</surname><given-names>GA</given-names> </name><name name-style="western"><surname>Al-Wathinani</surname><given-names>AM</given-names> </name><name name-style="western"><surname>G&#x00F3;mez-Salgado</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Evaluating the efficacy of full-scale and tabletop exercises in enhancing paramedic preparedness for external disasters: a quasi-experimental study</article-title><source>Medicine (Baltimore)</source><year>2024</year><month>12</month><day>6</day><volume>103</volume><issue>49</issue><fpage>e40777</fpage><pub-id pub-id-type="doi">10.1097/MD.0000000000040777</pub-id><pub-id pub-id-type="medline">39654246</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chumvanichaya</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yuksen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Nuanprom</surname><given-names>P</given-names> </name><name name-style="western"><surname>Aramvanitch</surname><given-names>K</given-names> </name></person-group><article-title>A comparison of SIEVE, SORT, and START triage training effectiveness between immersive interactive 3D learning materials using virtual reality (VR-SSST) and traditional methods in mass casualty incidents</article-title><source>Int J Emerg Med</source><year>2025</year><month>03</month><day>13</day><volume>18</volume><issue>1</issue><fpage>55</fpage><pub-id pub-id-type="doi">10.1186/s12245-025-00850-2</pub-id><pub-id pub-id-type="medline">40082745</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cuartas-Alvarez</surname><given-names>T</given-names> </name><name name-style="western"><surname>Garijo Gonzalo</surname><given-names>G</given-names> </name><name name-style="western"><surname>Vali&#x00F1;o Otero</surname><given-names>E</given-names> </name><etal/></person-group><article-title>MassCas tabletop game as a training tool in mass casualty incidents for primary healthcare doctors and nurses: a pilot study</article-title><source>Prim Health Care Res Dev</source><year>2026</year><month>03</month><day>26</day><volume>27</volume><fpage>e39</fpage><pub-id pub-id-type="doi">10.1017/S1463423626101108</pub-id><pub-id pub-id-type="medline">41883338</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Anan</surname><given-names>H</given-names> </name><name name-style="western"><surname>Otomo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kondo</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Development of Mass-casualty Life Support-CBRNE (MCLS-CBRNE) in Japan</article-title><source>Prehosp Disaster Med</source><year>2016</year><month>10</month><volume>31</volume><issue>5</issue><fpage>547</fpage><lpage>550</lpage><pub-id pub-id-type="doi">10.1017/S1049023X16000686</pub-id><pub-id pub-id-type="medline">27531062</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cicero</surname><given-names>MX</given-names> </name><name name-style="western"><surname>Golloshi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gawel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Parker</surname><given-names>J</given-names> </name><name name-style="western"><surname>Auerbach</surname><given-names>M</given-names> </name><name name-style="western"><surname>Violano</surname><given-names>P</given-names> </name></person-group><article-title>A tabletop school bus rollover: Connecticut-wide drills to build pediatric disaster preparedness and promote a novel hospital disaster readiness checklist</article-title><source>Am J Disaster Med</source><year>2019</year><volume>14</volume><issue>2</issue><fpage>75</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.5055/ajdm.2019.0318</pub-id><pub-id pub-id-type="medline">31637688</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Skryabina</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Betts</surname><given-names>N</given-names> </name><name name-style="western"><surname>Reedy</surname><given-names>G</given-names> </name><name name-style="western"><surname>Riley</surname><given-names>P</given-names> </name><name name-style="western"><surname>Aml&#x00F4;t</surname><given-names>R</given-names> </name></person-group><article-title>The role of emergency preparedness exercises in the response to a mass casualty terrorist incident: a mixed methods study</article-title><source>Int J Disaster Risk Reduct</source><year>2020</year><month>06</month><volume>46</volume><fpage>101503</fpage><pub-id pub-id-type="doi">10.1016/j.ijdrr.2020.101503</pub-id><pub-id pub-id-type="medline">33312855</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zimmerman</surname><given-names>J</given-names> </name><name name-style="western"><surname>Khorram-Manesh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Proactive postgraduate education in disaster medicine and preparedness for enhanced disaster management</article-title><source>BMC Med Educ</source><year>2026</year><month>02</month><day>14</day><volume>26</volume><issue>1</issue><fpage>336</fpage><pub-id pub-id-type="doi">10.1186/s12909-026-08638-5</pub-id><pub-id pub-id-type="medline">41691252</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Kolb</surname><given-names>DA</given-names> </name></person-group><source>Experiential Learning: Experience as the Source of Learning and Development</source><year>1984</year><publisher-name>Prentice Hall</publisher-name><pub-id pub-id-type="other">0132952610</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="web"><article-title>Homeland Security Exercise and Evaluation Program (HSEEP)</article-title><source>United States Department of Homeland Security, Federal Emergency Management Agency</source><year>2020</year><access-date>2026-09-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.fema.gov/emergency-managers/national-preparedness/exercises/hseep">https://www.fema.gov/emergency-managers/national-preparedness/exercises/hseep</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gillespie</surname><given-names>TW</given-names> </name><name name-style="western"><surname>Chu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Frankenberg</surname><given-names>E</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>D</given-names> </name></person-group><article-title>Assessment and prediction of natural hazards from satellite imagery</article-title><source>Prog Phys Geogr</source><year>2007</year><month>10</month><volume>31</volume><issue>5</issue><fpage>459</fpage><lpage>470</lpage><pub-id pub-id-type="doi">10.1177/0309133307083296</pub-id><pub-id pub-id-type="medline">25170186</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cepeda</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Pashler</surname><given-names>H</given-names> </name><name name-style="western"><surname>Vul</surname><given-names>E</given-names> </name><name name-style="western"><surname>Wixted</surname><given-names>JT</given-names> </name><name name-style="western"><surname>Rohrer</surname><given-names>D</given-names> </name></person-group><article-title>Distributed practice in verbal recall tasks: a review and quantitative synthesis</article-title><source>Psychol Bull</source><year>2006</year><month>05</month><volume>132</volume><issue>3</issue><fpage>354</fpage><lpage>380</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.132.3.354</pub-id><pub-id pub-id-type="medline">16719566</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kitson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Harvey</surname><given-names>G</given-names> </name><name name-style="western"><surname>McCormack</surname><given-names>B</given-names> </name></person-group><article-title>Enabling the implementation of evidence based practice: a conceptual framework</article-title><source>Qual Health Care</source><year>1998</year><month>09</month><volume>7</volume><issue>3</issue><fpage>149</fpage><lpage>158</lpage><pub-id pub-id-type="doi">10.1136/qshc.7.3.149</pub-id><pub-id pub-id-type="medline">10185141</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harvey</surname><given-names>G</given-names> </name><name name-style="western"><surname>Kitson</surname><given-names>A</given-names> </name></person-group><article-title>PARIHS revisited: from heuristic to integrated framework for the successful implementation of knowledge into practice</article-title><source>Implement Sci</source><year>2016</year><month>03</month><day>10</day><volume>11</volume><fpage>33</fpage><pub-id pub-id-type="doi">10.1186/s13012-016-0398-2</pub-id><pub-id pub-id-type="medline">27013464</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baldwin</surname><given-names>TT</given-names> </name><name name-style="western"><surname>Ford</surname><given-names>JK</given-names> </name></person-group><article-title>Transfer of training: a review and directions for future research</article-title><source>Pers Psychol</source><year>1988</year><month>03</month><volume>41</volume><issue>1</issue><fpage>63</fpage><lpage>105</lpage><pub-id pub-id-type="doi">10.1111/j.1744-6570.1988.tb00632.x</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grossman</surname><given-names>R</given-names> </name><name name-style="western"><surname>Salas</surname><given-names>E</given-names> </name></person-group><article-title>The transfer of training: what really matters</article-title><source>Int J Training Development</source><year>2011</year><month>06</month><volume>15</volume><issue>2</issue><fpage>103</fpage><lpage>120</lpage><pub-id pub-id-type="doi">10.1111/j.1468-2419.2011.00373.x</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bari</surname><given-names>LF</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>I</given-names> </name><name name-style="western"><surname>Ahamed</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Potential use of artificial intelligence (AI) in disaster risk and emergency health management: a critical appraisal on environmental health</article-title><source>Environ Health Insights</source><year>2023</year><volume>17</volume><fpage>11786302231217808</fpage><pub-id pub-id-type="doi">10.1177/11786302231217808</pub-id><pub-id pub-id-type="medline">38089525</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hamstra</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Brydges</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hatala</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zendejas</surname><given-names>B</given-names> </name><name name-style="western"><surname>Cook</surname><given-names>DA</given-names> </name></person-group><article-title>Reconsidering fidelity in simulation-based training</article-title><source>Acad Med</source><year>2014</year><month>03</month><volume>89</volume><issue>3</issue><fpage>387</fpage><lpage>392</lpage><pub-id pub-id-type="doi">10.1097/ACM.0000000000000130</pub-id><pub-id pub-id-type="medline">24448038</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Complete database-specific search strategies.</p><media xlink:href="i-jmr_v15i1e96228_app1.docx" xlink:title="DOCX File, 28 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Glossary, Kirkpatrick classification rules, and Tabletop Exercise Design Spectrum definitions.</p><media xlink:href="i-jmr_v15i1e96228_app2.docx" xlink:title="DOCX File, 39 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Metadata for unretrieved conference abstracts.</p><media xlink:href="i-jmr_v15i1e96228_app3.docx" xlink:title="DOCX File, 42 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Kirkpatrick outcome mapping by study and exercise tier.</p><media xlink:href="i-jmr_v15i1e96228_app4.docx" xlink:title="DOCX File, 38 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Practitioner summary: purpose-aligned design, facilitation, and evaluation by tabletop exercise tier.</p><media xlink:href="i-jmr_v15i1e96228_app5.docx" xlink:title="DOCX File, 37 KB"/></supplementary-material><supplementary-material id="app6"><label>Checklist 1</label><p>PRISMA-ScR checklist.</p><media xlink:href="i-jmr_v15i1e96228_app6.docx" xlink:title="DOCX File, 39 KB"/></supplementary-material></app-group></back></article>