<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Mhealth Uhealth</journal-id><journal-id journal-id-type="publisher-id">mhealth</journal-id><journal-id journal-id-type="index">13</journal-id><journal-title>JMIR mHealth and uHealth</journal-title><abbrev-journal-title>JMIR Mhealth Uhealth</abbrev-journal-title><issn pub-type="epub">2291-5222</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e76632</article-id><article-id pub-id-type="doi">10.2196/76632</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Machine Learning Frameworks for Wearable-Based Stress Modeling in Naturalistic Settings: Scoping Review</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Sharma</surname><given-names>Shifali</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Janakiraman</surname><given-names>Aswin Kumar</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Chen</surname><given-names>Lujie Karen</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>Department of Information Systems, University of Maryland, Baltimore County</institution><addr-line>1000 Hilltop Cir</addr-line><addr-line>Baltimore</addr-line><addr-line>MD</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Chandrasekaran</surname><given-names>Ranganathan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Gupta</surname><given-names>Ankit</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Rafi</surname><given-names>Hira</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Kallio</surname><given-names>Johanna</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Potla</surname><given-names>Ravi Teja</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Zhang</surname><given-names>Yonggang</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Lujie Karen Chen, PhD, Department of Information Systems, University of Maryland, Baltimore County, 1000 Hilltop Cir, Baltimore, MD, United States, 1 412 657 5305; <email>lujiec@umbc.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>31</day><month>7</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e76632</elocation-id><history><date date-type="received"><day>29</day><month>04</month><year>2025</year></date><date date-type="rev-recd"><day>20</day><month>05</month><year>2026</year></date><date date-type="accepted"><day>22</day><month>05</month><year>2026</year></date></history><copyright-statement>&#x00A9; Shifali Sharma, Aswin Kumar Janakiraman, Lujie Karen Chen. Originally published in JMIR mHealth and uHealth (<ext-link ext-link-type="uri" xlink:href="https://mhealth.jmir.org">https://mhealth.jmir.org</ext-link>), 31.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR mHealth and uHealth, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mhealth.jmir.org/">https://mhealth.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mhealth.jmir.org/2026/1/e76632"/><abstract><sec><title>Background</title><p>Stress, as commonly recognized, is an integral part of modern life and can significantly affect both mental and physical health. While substantial advancements have been made in measuring physical fitness through wearable devices, the detection and assessment of mental stress remain in their early stages.</p></sec><sec><title>Objective</title><p>The objective of this paper is to review recent studies of wearable-based stress detection in naturalistic settings, with a specific focus on characterizing machine learning frameworks inspired by the model card approach.</p></sec><sec sec-type="methods"><title>Methods</title><p>This review was conducted using the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) checklist. A total of 353 articles were identified through searches in databases such as PubMed, MEDLINE, ScienceDirect, IEEE, ACM Digital Library, Web of Science, and Embase. Studies were considered eligible if they collected data from healthy adults in naturalistic settings using wearable devices and used machine learning models for stress detection.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 34 articles met the eligibility criteria, including 11 conference papers, 22 journal articles, and 1 preprint published between 2017 and 2024. From these studies, we analyzed key machine learning modeling decisions such as problem formulation, ground truth determination, and machine learning algorithms. Additionally, we examined the major contributions of each study, focusing on the challenges they addressed and the solutions they proposed. Based on these findings, we proposed a model card framework for reporting machine learning&#x2013;based, wearable-based stress detection.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This scoping review highlights recent trends in machine learning models for stress detection and measurement using wearable signals. It underscores the need for improved standardization in reporting practices for datasets and key machine learning decisions, as well as the importance of addressing critical challenges associated with data collection in real-world settings. We hope this review will support and strengthen ongoing research efforts, promote knowledge sharing, and promote collaboration among researchers&#x2014;ultimately advancing the field as a community.</p></sec></abstract><kwd-group><kwd>mental health</kwd><kwd>stress detection</kwd><kwd>naturalistic setting</kwd><kwd>in the wild</kwd><kwd>field studies</kwd><kwd>wearables</kwd><kwd>physiology</kwd><kwd>machine learning models</kwd><kwd>scoping review</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Stress affects the well-being of many individuals in modern society [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. From an evolutionary perspective, stress is a fight-or-flight response to perceived threat or danger, which prompts necessary survival actions to flee from dangerous situations [<xref ref-type="bibr" rid="ref3">3</xref>]. However, sustained unmanaged stress may have long-term effects on both mental and physical health [<xref ref-type="bibr" rid="ref4">4</xref>]. In extreme conditions, it may lead to serious adverse health outcomes such as depression, cardiovascular diseases [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref7">7</xref>], substance abuse, drugs and alcohol addiction, or self-harm behaviors such as suicide [<xref ref-type="bibr" rid="ref8">8</xref>]. This underscores the importance of timely interventions or management of stress in daily life. Stress detection is fundamental to this type of support system, which permits continuous monitoring of stressful states in naturalistic settings, that is, everyday life. Given their nonintrusive and user-friendly design [<xref ref-type="bibr" rid="ref9">9</xref>], wearables have been explored to monitor physical fitness, using metrics such as step counts or heart rates. In recent years, there have been emerging interests in using wearables to understand individuals&#x2019; psychological fitness or well-being, in which stress detection and measurement are the major driving components.</p><p>Although stress detection from instruments such as wearables is not new, detecting stress in naturalistic settings poses significant new challenges. For studies conducted in controlled settings [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref12">12</xref>] in laboratory settings, researchers prescribe stress-inducing activities that participants engage with, such as the Trier Social Stress Test, the Stroop Color-Word Interference Test, the Montreal Imaging Stress Task, the Cold Pressor Test, and the &#x201C;Sing-a-Song&#x201D; Stress Test [<xref ref-type="bibr" rid="ref13">13</xref>]. Similarly, in studies outside of the laboratory, participants may participate in activities that are known to induce stress for some, such as hackathons, driving, or presentations [<xref ref-type="bibr" rid="ref14">14</xref>]. In contrast, stress detection in naturalistic settings requires individuals to be involved in normal daily activities, often with significant uncertainty regarding the timing, duration, stressor, and stress responses. Moreover, there is additional complexity in dealing with motion artifacts of wearable signals resulting from ambulatory participants [<xref ref-type="bibr" rid="ref13">13</xref>]. Both issues need to be appropriately addressed for reliable stress detection models.</p><p>The massive amounts of high-resolution data collected from wearables provide opportunities to leverage advanced analytics, machine learning (ML), and AI models for stress detection. In recent years, there has been a growing trend in stress detection research using ML as the primary modeling technique. This trend is supported by wearables data collected unobtrusively in naturalistic, real-world settings, driven in part by the availability of large-scale, open-source datasets that attempt to measure psychological constructs such as stress.</p><p>While several literature reviews have evaluated studies on wearable-based stress detection, most have focused on cataloging wearables, including their models, sensor types, and factors such as placement, cost, and usability [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>Several prior reviews have examined aspects of ML-based stress modeling. For example, Namvari et al [<xref ref-type="bibr" rid="ref16">16</xref>] focused on classification models, which discussed feature engineering approaches and summarized model performance. However, only 5 of the 23 studies included in their review were conducted exclusively in real-life or naturalistic settings. Similarly, the scoping review by Bolpagni et al [<xref ref-type="bibr" rid="ref3">3</xref>] evaluated 56 studies, of which only 13 were conducted in real-world contexts. Their review primarily focused on preprocessing pipelines, feature extraction methods, and ML model types. Pinge et al [<xref ref-type="bibr" rid="ref18">18</xref>] reviewed 39 studies in which stressors were predominantly derived from laboratory-induced or controlled stress-inducing stimuli, such as public speaking tasks or exposure to horror movies. Only a small subset of the reviewed studies involved those from free-living conditions. This review summarized preprocessing strategies, feature computation methods, ML techniques (including both classical and deep learning models), and commonly used performance metrics.</p><p>In summary, there are several notable gaps in the wearable-based ML-focused review for stress modeling. (1) None of these reviews focus exclusively on studies using data collected in real-world, naturalistic settings. (2) None of the reviews address methodological rigor explicitly, for example, by evaluating whether models use appropriate experimental setups to avoid data leakage [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. This issue is particularly relevant but easy to overlook in wearable studies with repeated measures from the same participants, and failure to account for it can lead to overestimation of model performance. (3) None of the reviews focus on important problem formulation decisions, such as how to appropriately configure the training dataset so that input and output windows are correctly aligned so that it matches the modeling objectives, for example, whether the goal is to detect stress in the present moment (ie, nowcasting) or predict stress in the future (ie, forecasting). (4) There is no review attempting to explicitly tackle the standardization of reporting, for example, by adopting an existing ML model reporting framework such as model card [<xref ref-type="bibr" rid="ref23">23</xref>]. This scoping review aims to address these critical gaps.</p><p>The main contributions of this paper are as follows:</p><list list-type="order"><list-item><p>Provided a focused review of recent studies on wearable-based stress detection exclusively in naturalistic settings using ML techniques.</p></list-item><list-item><p>Analyzed model performance with a focus on methodological rigor, explicitly examining potential threats to validity (eg, data leakage).</p></list-item><list-item><p>Analyzed key problem formulation decisions such as input and output window configuration, and distinctions between nowcasting and forecasting problems.</p></list-item><list-item><p>Proposed standardized report ML framework in wearable-based stress detection, inspired by the model card model reporting framework.</p></list-item></list><p>This review paper is organized as follows: The &#x201C;Methods&#x201D; section describes the process used to select papers for review, including eligibility criteria and the PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) diagram. The &#x201C;Results&#x201D; section presents the dataset features and ML frameworks used in the selected studies. Finally, the &#x201C;Discussion&#x201D; section analyzes the findings, identifies limitations, and provides recommendations for future research.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>We adopt a scoping review approach to systematically identify gaps in the existing literature. Although stress detection has long been studied in laboratory settings, it remains challenging to translate these findings into real-life contexts. As such, our primary goal is to map current research on stress detection, identify existing gaps, and provide an overview of ML frameworks inspired by the concept of model cards. Specifically, this review examines how ML frameworks are used to detect stress in naturalistic settings using wearable sensors. The review is guided by the methodological framework proposed by Arksey and O&#x2019;Malley [<xref ref-type="bibr" rid="ref24">24</xref>] and adheres to the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) guidelines (<xref ref-type="supplementary-material" rid="app4">Checklist 1</xref>) [<xref ref-type="bibr" rid="ref25">25</xref>].</p></sec><sec id="s2-2"><title>Search Strategy</title><p>The primary search for this scoping review was conducted in April 2024 across 7 databases: PubMed, MEDLINE, ScienceDirect, IEEE Xplore, ACM Digital Library, Web of Science, and Embase. These databases were chosen for their extensive coverage of recent physiology research on stress detection. Our aim was to identify articles examining stress detection in naturalistic settings using biosignals and wearables. We restricted our search to English-language, peer-reviewed studies published between January 2017 and April 2024 that were available in full text. Initially, we tested various search strings&#x2014;for instance, using &#x201C;stress detection&#x201D; alone yielded too many results, whereas &#x201C;stress detection&#x201D; AND (&#x201C;naturalistic setting&#x201D; OR &#x201C;field study&#x201D; OR &#x201C;real life&#x201D;) yielded too few. Ultimately, using more generic terms such as &#x201C;stress detection,&#x201D; &#x201C;wearables,&#x201D; and &#x201C;physiology,&#x201D; combined with AND operators, produced a sufficiently broad yet focused list of results. Research articles involving either primary or secondary data analyses (or both) were included. Primary data analysis refers to studies where data were collected by the authors, whereas secondary data analysis refers to studies using data collected by other researchers, which were either openly available or accessible upon request.</p></sec><sec id="s2-3"><title>Eligibility Criteria</title><p>Studies were deemed eligible for inclusion if they used data collected in naturalistic (real-world or in-the-wild) settings. Here, naturalistic refers to situations in which stress is not experimentally induced or linked to known stress-inducing events (eg, taking a test), but instead reflects stress as it occurs in daily life. This criterion applied even if some studies use multiple datasets, including those from controlled or laboratory environments, provided that at least one dataset was collected in the wild. This decision was motivated by the fact that stress labeling mechanisms in laboratory studies differ fundamentally from those used in real-life or naturalistic contexts, where stress must be explicitly self-reported or otherwise captured from participants during their daily lives, which pose unique challenges. Given the study&#x2019;s focus on acute stress, only research involving healthy populations in everyday contexts was considered. Consequently, the use of wearable devices was a prerequisite, as they facilitate continuous monitoring in daily life. Recognizing the surge in wearable technology adoption in recent years, the review encompassed studies published between 2017 and 2024. The specific eligibility criteria pertinent to this scoping review are detailed in <xref ref-type="other" rid="box1">Textbox 1</xref>.</p><boxed-text id="box1"><title> Inclusion and exclusion criteria.</title><p><bold>Inclusion criteria</bold></p><list list-type="bullet"><list-item><p>Data were collected in naturalistic settings where participants engaged in normal, everyday activities (including work).</p></list-item><list-item><p>Healthy participants were involved. At least one type of data was collected using wearable devices to capture physiological and/or psychological signals.</p></list-item><list-item><p>Machine learning models or other advanced data-driven techniques were used to detect stress.</p></list-item><list-item><p>The study needs to be published in English.</p></list-item></list><p><bold>Exclusion criteria</bold></p><list list-type="bullet"><list-item><p>Data were collected only in controlled settings (eg, in laboratory environments or when participants engaged in known stress-inducing activity such as a hackathon as described in the &#x201C;Introduction&#x201D;).</p></list-item><list-item><p>The participants had been diagnosed with mental health disorders or other chronic conditions.</p></list-item><list-item><p>No wearable devices were used in the data collection process.</p></list-item><list-item><p>No machine learning or other advanced data-driven modeling techniques were used to detect stress.</p></list-item><list-item><p>The study was not published in English.</p></list-item></list></boxed-text></sec><sec id="s2-4"><title>Title and Abstract Screening</title><p>We began by applying basic eligibility filters (language and publication year) within the scientific databases themselves. After eliminating duplicate entries, the first author compiled all the retrieved records into a spreadsheet and conducted the initial title and abstract screening using preagreed eligibility criteria. The third author reviewed and verified the screened records and decisions regarding the inclusion and exclusion in the spreadsheet against the same criteria. During abstract screening, we identified publicly available, on-request datasets containing naturalistic wearable data for stress detection and reviewed their citations, adding any studies that used these datasets to our review. Any ambiguities or discrepancies were resolved through discussion until consensus was reached. Since formal full parallel double-screening was not performed, Cohen &#x03BA; or percentage agreement was not calculated. Finally, the remaining full-text articles were reassessed for eligibility using the specified inclusion and exclusion criteria.</p></sec><sec id="s2-5"><title>Full-Text Review, Data Extraction, and Analysis</title><p>Data extraction involved classifying studies into predefined categories to facilitate evidence synthesis, with a particular focus on the design of ML frameworks inspired by model cards. Extracted data elements included availability of datasets such as indicating whether datasets were open source, available upon request, or collected within the study design, demographic information such as participants' professions, total number of participants, and geographic location, details of wearable devices and biosignals used, duration of data collection, primary research focus such as problem addressed which are either originating from domain-centric or data-centric approaches, and ML models used. Additionally, we examined the formulation of problem statements, including the selection of input and output windows for nowcasting or forecasting tasks. We also assessed validation practices, with particular attention to measures implemented to prevent data leakage, such as appropriate data partitioning and cross-validation techniques.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Literature Search</title><p>As shown in the flowchart in <xref ref-type="fig" rid="figure1">Figure 1</xref>, a total of 353 articles were identified through database searches: 100 from PubMed, 16 from IEEE, 167 from ScienceDirect, 14 from MEDLINE, 4 from Web of Science, 28 from ACM Digital Library, and 24 from Embase. During the title and abstract screening, 258 articles were excluded for not meeting the eligibility criteria.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flowchart for paper screening.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mhealth_v14i1e76632_fig01.png"/></fig><p>During the abstract screening phase, we identified additional datasets that were collected using wearables to model wellness-related constructs in naturalistic settings, for example, TILES-2018 [<xref ref-type="bibr" rid="ref26">26</xref>], Lifesnaps [<xref ref-type="bibr" rid="ref27">27</xref>], Tesserae [<xref ref-type="bibr" rid="ref28">28</xref>], and SWEET [<xref ref-type="bibr" rid="ref29">29</xref>]. In total, we find 227 secondary analysis papers citing those datasets, with 45 of them using TILES-2018, 106 using SWEET, 70 using Tesserae, and 6 using Lifesnaps. After further title and abstract screening, 163 studies were excluded. This left a total of 146 studies considered for full-text review, with a mixture of primary and secondary analyses. Of these, 71 studies were excluded from the primary data analysis category, and 41 studies were excluded from the secondary data analysis category based on the eligibility criteria. In the end, 34 articles were selected for inclusion in this scoping review. For a detailed description of the search and selection strategy using PubMed, please refer to <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s3-2"><title>Study Characteristics</title><sec id="s3-2-1"><title>Overview</title><p>Among the 34 selected papers, 11 studies were conference proceedings (32%), 22 were journal articles (65%), and 1 was a preprint (3%). The studies were published between 2017 and 2024. Of these, 24 out of 34 (71%) studies conducted secondary data analysis using datasets collected by other researchers&#x2014;either open-source or available upon request. The remaining 10 out of 34 (29%) studies performed primary data analysis based on datasets collected by the authors specifically for the reported research. Fitbit was the most commonly used wearable device, followed by Garmin, Empatica E4, and OMsignal garments. The duration of data collection across studies ranged from a single day to several months.</p></sec><sec id="s3-2-2"><title>Dataset Characteristics</title><p>Although the dataset is not the primary focus of this review, we provide a brief overview of the dataset characteristics since it forms the foundation for the ML models. A comprehensive overview of the datasets used in the included studies is presented in <xref ref-type="table" rid="table1">Table 1</xref>. For detailed field descriptions, please refer to <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. The frequently-used datasets as part of the secondary data analysis are TILES-2018, SWEET, LifeSnaps, and Tesserae. The populations represented include clinical providers, office and information workers, police officers, military personnel, university students, and remote workers, while a few studies did not specify the population. The number of participants in the datasets ranged from as few as 3 to as many as 1002. The duration of data collection varied, ranging from 1 day to 10 weeks.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Dataset characteristics.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dataset citation key</td><td align="left" valign="bottom">Dataset name</td><td align="left" valign="bottom">Participant pool</td><td align="left" valign="bottom">Number of participants</td><td align="left" valign="bottom">Data collection duration</td><td align="left" valign="bottom">Wearable device used</td></tr></thead><tbody><tr><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">TILES-2018</td><td align="left" valign="top">Clinical providers</td><td align="left" valign="top">212</td><td align="left" valign="top">10 weeks</td><td align="left" valign="top">Fitbit Charge 2 and OMsignal smart garment</td></tr><tr><td align="left" valign="top">Smets et al (2018) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">SWEET</td><td align="left" valign="top">Staff members in technology-oriented, banking, and public sector companies.</td><td align="left" valign="top">1002</td><td align="left" valign="top">5 days</td><td align="left" valign="top">imec&#x2019;s Chillband and chest patch</td></tr><tr><td align="left" valign="top">Yfantidou et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Lifesnaps</td><td align="left" valign="top">Participants were recruited through university mailing lists, and although not explicitly stated, they were likely university students from Greece, Cyprus, Italy, and Sweden.</td><td align="left" valign="top">71</td><td align="left" valign="top">4 months (2 waves, each lasting 2 months)</td><td align="left" valign="top">Fitbit Sense</td></tr><tr><td align="left" valign="top">Mattingly et al (2019) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Tesserae</td><td align="left" valign="top">Information workers</td><td align="left" valign="top">757</td><td align="left" valign="top">56 days</td><td align="left" valign="top">Garmin Vivosmart 3, a waterproof wristwatch</td></tr><tr><td align="left" valign="top">Boateng and Kotz (2016) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">Unknown</td><td align="left" valign="top">10</td><td align="left" valign="top">1 day</td><td align="left" valign="top">Zephyr chest strap and Amulet</td></tr><tr><td align="left" valign="top">Tervonen et al (2020) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Office workers</td><td align="left" valign="top">74</td><td align="left" valign="top">4 weeks</td><td align="left" valign="top">Polar M600 smartwatch</td></tr><tr><td align="left" valign="top">Bavaresco et al (2020) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">5</td><td align="left" valign="top">8 days</td><td align="left" valign="top">Polar H7 chest strap</td></tr><tr><td align="left" valign="top">Gjoreski et al (2017) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">5</td><td align="left" valign="top">11 days, on average</td><td align="left" valign="top">Empatica E3 and E4 wrist devices</td></tr><tr><td align="left" valign="top">de Vries et al (2022) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Dutch police officers</td><td align="left" valign="top">8</td><td align="left" valign="top">At least 15 weeks, up to 55 weeks.</td><td align="left" valign="top">Oura Ring (Generation 2)</td></tr><tr><td align="left" valign="top">Han et al (2020) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Unknown</td><td align="left" valign="top">3</td><td align="left" valign="top">2 days</td><td align="left" valign="top">Empatica E4 wristband and Shimmer3 ECG chest strap</td></tr><tr><td align="left" valign="top">de Vries et al (2023) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Dutch military</td><td align="left" valign="top">73</td><td align="left" valign="top">8 weeks</td><td align="left" valign="top">Garmin Tactix Charlie, smartwatch</td></tr><tr><td align="left" valign="top">Tump et al (2022) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Remote-working employees of the global insurance company Cigna in 3 locations: the United States, the United Kingdom, and Hong Kong.</td><td align="left" valign="top">198</td><td align="left" valign="top">7 consecutive days</td><td align="left" valign="top">Garmin Vivosmart 4</td></tr><tr><td align="left" valign="top">Mishra et al (2020) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">University students</td><td align="left" valign="top">26</td><td align="left" valign="top">3 days</td><td align="left" valign="top">Polar H7 chest strap, Amulet wrist device, and a custom GSR<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> sensor</td></tr><tr><td align="left" valign="top">Schmidt et al (2019) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">University students</td><td align="left" valign="top">11</td><td align="left" valign="top">Approximately 16 days</td><td align="left" valign="top">Empatica E4 wristband</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Not available.</p></fn><fn id="table1fn2"><p><sup>b</sup>GSR: Galvanic skin response.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2-3"><title>Wearables Characteristic</title><p>The choice of wearables has a direct impact on the quality and diversity of data collected. A wide range of wearables has been used in these studies to unobtrusively collect data, with the most frequently used devices being the Garmin Vivosmart [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>], Fitbit Charge 2 [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref40">40</xref>], OmSignal garment [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>], Empatica E4 [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref37">37</xref>], Unihertz Jelly Pro smartphone [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref43">43</xref>], and the Imec Chillband [<xref ref-type="bibr" rid="ref44">44</xref>]. Wearable sensors enable the capture of diverse data types across multiple modalities, including physiological data, for example, heart rate, temperature, and skin conductance [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref45">45</xref>-<xref ref-type="bibr" rid="ref47">47</xref>]. Additional data types were also recorded, for example, phone usage, sleep duration, circadian rhythm, and movement [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref48">48</xref>-<xref ref-type="bibr" rid="ref50">50</xref>], audio signals [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref49">49</xref>], sociodemographic data [<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref50">50</xref>], and environmental data (eg, temperature and humidity) [<xref ref-type="bibr" rid="ref51">51</xref>]. These modalities provide options for studies to use either unimodal [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref36">36</xref>] or multimodal approaches [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref53">53</xref>].</p></sec><sec id="s3-2-4"><title>ML Framework and Model Performances</title><p>In this section, we organize the analysis around key methodological decisions identified across the reviewed studies on wearable-based stress detection in naturalistic settings. At a high level, this task was framed as learning a mapping between the physiological signals collected from wearables and the psychological construct of stress. This was often formulated as a supervised learning framework (either a classification or regression problem), where the inputs are wearable sensing data and the outputs are stress labels.</p><p>Due to the nature of the data being collected, the reviewed studies varied across several methodological dimensions. We analyze the model frameworks along the following dimensions:</p><list list-type="order"><list-item><p>Ground truth stress labels: how the labels are collected and the types of labels collected.</p></list-item><list-item><p>Input and output windows: how to set up input-output pairs based on corresponding time windows and how the temporal relationships between input and output windows are informed by the problem being addressed&#x2014;whether it involves in-the-moment stress detection (ie, nowcasting) or predicting future stress (ie, forecasting).</p></list-item><list-item><p>ML experiment design: how to appropriately set up ML experiments, for example, split training and testing data appropriately to avoid data leakage.</p></list-item><list-item><p>ML models: the primary ML algorithms used in the studies.</p></list-item><list-item><p>Challenges and solutions: the specific challenges that the reviewed papers aim to tackle and the main solutions proposed.</p></list-item><list-item><p>Analysis of model performance: analysis of model performance with respect to ML and specific solutions to address challenges, with caveats resulting from issues in reporting and methodological rigor.</p></list-item></list></sec><sec id="s3-2-5"><title>Overview of Model Characteristics</title><p>Across the 34 included studies [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref59">59</xref>], the large majority adopted a global, population-level modeling approach, with 28 [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref48">48</xref>-<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>-<xref ref-type="bibr" rid="ref59">59</xref>] out of 34 (82%) studies following this approach, while only 6 [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref51">51</xref>] out of 34 (18%) studies pursued personalization or built individualized models. With respect to model family, 21 [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref58">58</xref>] out of 34 (62%) studies used traditional ML, while 8 [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref57">57</xref>] out of 34 (24%) studies relied exclusively on deep learning. A further 4 [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref55">55</xref>] out of 34 (12%) studies combined traditional ML and deep learning, and 1 [<xref ref-type="bibr" rid="ref59">59</xref>] out of 34 (3%) studies additionally incorporated a large language model (LLM), bringing the total share of studies that used deep learning in some capacity to 13 [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref55">55</xref>-<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref59">59</xref>] out of 34 (38%) studies. Reporting of evaluation metrics was assessed among the 22 classification studies [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>-<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref55">55</xref>-<xref ref-type="bibr" rid="ref58">58</xref>]. <italic>F</italic><sub>1</sub>-score was the most commonly reported metric, appearing in 19 [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>-<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref55">55</xref>-<xref ref-type="bibr" rid="ref58">58</xref>] out of 22 (86%) classification studies, followed by accuracy in 12 [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref56">56</xref>-<xref ref-type="bibr" rid="ref58">58</xref>] out of 22 (55%) studies, precision in 7 [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref58">58</xref>] out of 22 (32%) studies, and recall in 5 [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref58">58</xref>] out of 22 (23%) studies. Only 3 [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref55">55</xref>] out of 22 (14%) studies reported alternative discrimination or balance-aware metrics such as area under the receiver operating characteristic curve (AUC), balanced accuracy, or Matthews correlation coefficient (MCC), suggesting limited adoption of metrics that are robust to class imbalance. Among the 13 regression studies [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref38">38</xref>-<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref59">59</xref>], all reported at least one quality measure such as R-squared or mean-squared error (MSE). Potential data leakage was identified in roughly half of the corpus, with 19 [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref53">53</xref>-<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref59">59</xref>] out of 34 (56%) studies judged to have at least one source of potential leakage and the remaining 15 [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref57">57</xref>] out of 34 (44%) studies judged to be free of obvious leakage. Finally, in terms of temporal problem formulation, the field is dominated by nowcasting, with 27 [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref51">51</xref>-<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref58">58</xref>] out of 34 (79%) studies framing their task as inferring a current or concurrent state; only 4 [<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref57">57</xref>] out of 34 (12%) studies explicitly framed the task as forecasting a future state, and the formulation could not be unambiguously determined for 3 [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref59">59</xref>] out of 34 (9%) studies.</p><p>Details of the key components of the ML framework are presented in <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>, which summarize the reported best model performance; please refer to <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> for additional details.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Machine learning framework characteristics.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Paper citation key</td><td align="left" valign="bottom">Dataset citation key</td><td align="left" valign="bottom">Main problems addressed</td><td align="left" valign="bottom">Problem formulation</td><td align="left" valign="bottom">Input window</td><td align="left" valign="bottom">Output window</td><td align="left" valign="bottom">Model type</td><td align="left" valign="bottom">Sample size (N)</td><td align="left" valign="bottom">Possible data leakage</td></tr></thead><tbody><tr><td align="left" valign="top">Boateng and Kotz 2016 [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">Boateng and Kotz (2016) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">No specific problem was addressed</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">1 minute or 5 minutes</td><td align="left" valign="top">15 minutes</td><td align="left" valign="top">Classification</td><td align="left" valign="top">10</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Parousidou et al 2023 [<xref ref-type="bibr" rid="ref45">45</xref>]</td><td align="left" valign="top">Yfantidou et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Personalization</td><td align="left" valign="top">Unclear</td><td align="left" valign="top">Unclear</td><td align="left" valign="top">Unclear</td><td align="left" valign="top">Classification</td><td align="left" valign="top">71</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Tervonen et al 2020 [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Tervonen et al (2020) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Personalization</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Classification</td><td align="left" valign="top">74</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Bavaresco et al 2020 [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Bavaresco et al (2020) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Context awareness</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">20 minutes</td><td align="left" valign="top">20 minutes</td><td align="left" valign="top">Classification</td><td align="left" valign="top">5</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Gjoreski et al 2017 [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">Gjoreski et al (2017) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">Context awareness</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">A range of windows from 10 minutes to 30 minutes</td><td align="left" valign="top">20 minutes</td><td align="left" valign="top">Classification</td><td align="left" valign="top">5</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">de Vries et al 2022 [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">de Vries et al (2022) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Temporal dependency</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Previous day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Regression (statistical)</td><td align="left" valign="top">8</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Han et al 2020 [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">Han et al (2020) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">No specific problem was addressed</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Classification</td><td align="left" valign="top">17</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">de Vries et al 2023 [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">de Vries et al (2023) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">Temporal dependency</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Last night&#x2019;s sleep duration</td><td align="left" valign="top">at wake up</td><td align="left" valign="top">Regression (statistical)</td><td align="left" valign="top">73</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Tump et al 2022 [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">Tump et al (2022) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">Context awareness</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">3 hours</td><td align="left" valign="top">3 hours</td><td align="left" valign="top">Classification</td><td align="left" valign="top">198</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Mishra et al 2020 [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">Mishra et al (2020) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">No specific problem was addressed</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">1 minute</td><td align="left" valign="top">1 minute</td><td align="left" valign="top">Classification</td><td align="left" valign="top">27</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Schmidt et al 2019 [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Schmidt et al (2019) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">No specific problem was addressed</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">1 minute</td><td align="left" valign="top">1 minute</td><td align="left" valign="top">Classification</td><td align="left" valign="top">12</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Jiang et al 2023 [<xref ref-type="bibr" rid="ref54">54</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Addressing limited supply of training data</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">3 days centered on current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Regression</td><td align="left" valign="top">212</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Hadjiantonis et al 2020 [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Temporal dependency</td><td align="left" valign="top">Forecasting</td><td align="left" valign="top">Previous day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Regression (statistical)</td><td align="left" valign="top">130</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Gaballah et al 2021 [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Temporal dependency</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current shift</td><td align="left" valign="top">Current shift</td><td align="left" valign="top">Classification</td><td align="left" valign="top">212</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Yu and Sano 2022 [<xref ref-type="bibr" rid="ref48">48</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Addressing limited supply of training data</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">2.5 hours after the label was collected</td><td align="left" valign="top">Current shift</td><td align="left" valign="top">Classification</td><td align="left" valign="top">212</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Burghardt et al 2021 [<xref ref-type="bibr" rid="ref52">52</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Temporal dependency</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">3 days centered on the current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Classification</td><td align="left" valign="top">212</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Kao et al 2020 [<xref ref-type="bibr" rid="ref51">51</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Personalization</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">3 days centered on the current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Regression</td><td align="left" valign="top">212</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Paromita et al 2023 [<xref ref-type="bibr" rid="ref43">43</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Personalization</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Regression</td><td align="left" valign="top">212</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Pimentel et al 2021 [<xref ref-type="bibr" rid="ref55">55</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">No specific problem was addressed</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Classification</td><td align="left" valign="top">212</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Yang et al 2022 [<xref ref-type="bibr" rid="ref46">46</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Multiple modalities</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">2 hours before the label was collected</td><td align="left" valign="top">Current shift</td><td align="left" valign="top">Classification</td><td align="left" valign="top">212</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Zanna et al 2022 [<xref ref-type="bibr" rid="ref56">56</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Bias mitigation</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">2 hours before the label was collected</td><td align="left" valign="top">Current shift</td><td align="left" valign="top">Classification</td><td align="left" valign="top">212</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Tiwari and Falk 2021 [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">No specific problem was addressed</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Classification</td><td align="left" valign="top">212</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Feng and Narayanan 2022 [<xref ref-type="bibr" rid="ref49">49</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Context awareness</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Classification (stable)</td><td align="left" valign="top">99</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Feng et al 2021 [<xref ref-type="bibr" rid="ref53">53</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Context awareness</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Regression (stable, statistical)</td><td align="left" valign="top">113</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Ravuri et al 2020 [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">Mundnich et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Personalization</td><td align="left" valign="top">Unclear</td><td align="left" valign="top">Not given</td><td align="left" valign="top">Not given</td><td align="left" valign="top">Regression</td><td align="left" valign="top">154</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Stojchevska et al 2022 [<xref ref-type="bibr" rid="ref44">44</xref>]</td><td align="left" valign="top">Smets et al (2018) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">Context awareness</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">1 hour</td><td align="left" valign="top">1 hour</td><td align="left" valign="top">Classification</td><td align="left" valign="top">1002</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Booth et al 2022 [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">Mattingly et al (2019) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Temporal dependency</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Regression and classification</td><td align="left" valign="top">606</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Martinez et al 2022 [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">Mattingly et al (2019) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">No specific problem was addressed</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">A range of windows from 5 minutes to 24 hours</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Regression (statistical)</td><td align="left" valign="top">657</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Robles-Granda et al 2021 [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">Mattingly et al (2019) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Temporal dependency</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Regression (stable)</td><td align="left" valign="top">757</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Saylam and &#x0130;ncel 2023 [<xref ref-type="bibr" rid="ref50">50</xref>]</td><td align="left" valign="top">Mattingly et al (2019) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Temporal dependency</td><td align="left" valign="top">Forecasting</td><td align="left" valign="top">A range of windows from 1 day to 30 days before</td><td align="left" valign="top">A range of windows from 1 day to 7 days ahead</td><td align="left" valign="top">Regression</td><td align="left" valign="top">757</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Li et al 2024 [<xref ref-type="bibr" rid="ref57">57</xref>]</td><td align="left" valign="top">Mattingly et al (2019) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Temporal dependency</td><td align="left" valign="top">Forecasting</td><td align="left" valign="top">A range of windows from 25 days to 35 days before</td><td align="left" valign="top">A range of windows from 12 to 19 days ahead</td><td align="left" valign="top">Classification</td><td align="left" valign="top">478</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Saylam and Durmaz &#x0130;ncel 2022 [<xref ref-type="bibr" rid="ref58">58</xref>]</td><td align="left" valign="top">Mattingly et al (2019) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Context awareness</td><td align="left" valign="top">Nowcasting</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Current day</td><td align="left" valign="top">Classification</td><td align="left" valign="top">757</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Kim et al 2024 [<xref ref-type="bibr" rid="ref59">59</xref>]</td><td align="left" valign="top">Yfantidou et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Use of LLM<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> in stress detection</td><td align="left" valign="top">Unclear</td><td align="left" valign="top">Unclear</td><td align="left" valign="top">Unclear</td><td align="left" valign="top">Regression</td><td align="left" valign="top">16</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Paraschou et al 2023 [<xref ref-type="bibr" rid="ref47">47</xref>]</td><td align="left" valign="top">Yfantidou et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">No specific problem was addressed</td><td align="left" valign="top">Forecasting</td><td align="left" valign="top">Unclear</td><td align="left" valign="top">Unclear</td><td align="left" valign="top">Classification</td><td align="left" valign="top">71</td><td align="left" valign="top">Yes</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Machine learning model performance.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" rowspan="2">Paper citation key</td><td align="left" valign="bottom" colspan="5">Classification model</td><td align="left" valign="bottom">Regression model</td><td align="left" valign="bottom" rowspan="2">Baseline model reported</td><td align="left" valign="bottom" rowspan="2">Uncertainty measure reported</td></tr><tr><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom">Other measures</td><td align="left" valign="bottom">Quality measure</td></tr></thead><tbody><tr><td align="left" valign="top">Boateng and Kotz 2016 [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">1.00</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Parousidou et al 2023 [<xref ref-type="bibr" rid="ref45">45</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.66</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Yes</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Tervonen et al 2020 [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">0.51</td><td align="left" valign="top">0.62</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Bavaresco et al 2020 [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">0.82</td><td align="left" valign="top">0.75</td><td align="left" valign="top">0.9</td><td align="left" valign="top">0.64</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Gjoreski 2017 et al [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.9</td><td align="left" valign="top">0.95</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">de Vries et al 2022 [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Adjusted <italic>R</italic><sup>2</sup>: 0.01&#x2010;0.23</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top">Han et al 2020 [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">1.00</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">de Vries et al 2023 [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Marginal <italic>R</italic><sup>2</sup>=0.004</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top">Tump et al 2022 [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.47</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Mishra et al 2020 [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.7</td><td align="left" valign="top">0.57</td><td align="left" valign="top">0.91</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Schmidt et al 2019 [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.47</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Jiang et al 2023 [<xref ref-type="bibr" rid="ref54">54</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">MSE<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup>~1.00</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Hadjiantonis et al 2020 [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top"><italic>r</italic>=0.24 (<italic>P</italic>&#x003C;.01)</td><td align="left" valign="top">Yes</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Gaballah et al 2021 [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.64</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Yu and Sano 2022 [<xref ref-type="bibr" rid="ref48">48</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.70</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Burghardt et al 2021 [<xref ref-type="bibr" rid="ref52">52</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.23</td><td align="left" valign="top">0.16</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">AUC<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup>=0.56</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Yes</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Kao et al 2020 [<xref ref-type="bibr" rid="ref51">51</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">RMSE<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup>=0.89; <italic>r</italic>=0.42; <italic>R</italic><sup>2</sup>=0.15</td><td align="left" valign="top">Yes</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Paromita et al 2023 [<xref ref-type="bibr" rid="ref43">43</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">One-tailed paired <italic>t</italic> test (<italic>P</italic>&#x003C;.001)</td><td align="left" valign="top">No</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Pimentel et al 2021 [<xref ref-type="bibr" rid="ref55">55</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.68</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">BACC<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup>=0.65; MCC<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup>=0.30</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Yang et al 2022 [<xref ref-type="bibr" rid="ref46">46</xref>]</td><td align="left" valign="top">0.58</td><td align="left" valign="top">0.72</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Zanna et al 2022 [<xref ref-type="bibr" rid="ref56">56</xref>]</td><td align="left" valign="top">0.43&#x2010;0.54</td><td align="left" valign="top">0.30&#x2010;0.42</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Tiwari and Falk 2021 [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top"/><td align="left" valign="top">0.69</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">BACC=0.66; MCC=0.31</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Feng and Narayanan 2022 [<xref ref-type="bibr" rid="ref49">49</xref>]</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">66</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Feng et al 2021 [<xref ref-type="bibr" rid="ref53">53</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Adjusted <italic>R</italic><sup>2</sup>: (PA<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup> 0.152; NA<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup> 0.034)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top">Ravuri et al 2020 [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">&#x03C1;=0.08</td><td align="left" valign="top">No</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Stojchevska et al 2022 [<xref ref-type="bibr" rid="ref44">44</xref>]</td><td align="left" valign="top">0.42</td><td align="left" valign="top">41</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Booth et al 2022 [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">0.62</td><td align="left" valign="top">0.75</td><td align="left" valign="top">0.65</td><td align="left" valign="top">0.89</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x03C1;=0.25</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Martinez et al 2022 [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">Marginal <italic>R</italic><sup>2</sup>: 0.022 (main); 0.032 (follow-up)</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td></tr><tr><td align="left" valign="top">Robles-Granda et al 2021 [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">SMAPE<sup><xref ref-type="table-fn" rid="table3fn10">j</xref></sup>=0.66</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Saylam and &#x0130;ncel 2023 [<xref ref-type="bibr" rid="ref50">50</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">MAE<sup><xref ref-type="table-fn" rid="table3fn11">k</xref></sup>=0.47</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Li et al 2024 [<xref ref-type="bibr" rid="ref57">57</xref>]</td><td align="left" valign="top">0.74</td><td align="left" valign="top">72</td><td align="left" valign="top">0.73</td><td align="left" valign="top">0.71</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Saylam and Durmaz &#x0130;ncel 2022 [<xref ref-type="bibr" rid="ref58">58</xref>]</td><td align="left" valign="top">0.85</td><td align="left" valign="top">0.85</td><td align="left" valign="top">0.85</td><td align="left" valign="top">0.85</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Kim et al 2024 [<xref ref-type="bibr" rid="ref59">59</xref>]</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">N/A</td><td align="left" valign="top">MAE=0.32</td><td align="left" valign="top">No</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Paraschou et al 2023 [<xref ref-type="bibr" rid="ref47">47</xref>]</td><td align="left" valign="top">0.92</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">N/A</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Not available/not reported.</p></fn><fn id="table3fn2"><p><sup>b</sup>N/A: not applicable.</p></fn><fn id="table3fn3"><p><sup>c</sup>MSE: mean-squared error.</p></fn><fn id="table3fn4"><p><sup>d</sup>AUC: area under the receiver operating characteristic curve.</p></fn><fn id="table3fn5"><p><sup>e</sup>RMSE: root mean square error. </p></fn><fn id="table3fn6"><p><sup>f</sup>BACC: balanced accuracy.</p></fn><fn id="table3fn7"><p><sup>g</sup>MCC: Matthews correlation coefficient.</p></fn><fn id="table3fn8"><p><sup>h</sup>PA: positive affect.</p></fn><fn id="table3fn9"><p><sup>i</sup>NA: negative affect.</p></fn><fn id="table3fn10"><p><sup>j</sup>SMAPE: symmetric mean absolute percentage error.</p></fn><fn id="table3fn11"><p><sup>k</sup>MAE: mean absolute error.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2-6"><title>Ground Truth Stress Labels</title><p>The goal of ML models is to predict or estimate psychological constructs, such as stress [<xref ref-type="bibr" rid="ref43">43</xref>]. These constructs are often referred to as ground truth labels or targets within an ML framework. Since data are collected in naturalistic settings, aligning stress labels with physiological data from wearables poses a challenge, as there are no obvious ground truth labels, unlike in controlled environments where the stressor and stress-inducing periods are known with high levels of certainty. In naturalistic settings, stress labels are typically collected from participants through self-report mechanisms, such as ecological momentary assessment (EMA) [<xref ref-type="bibr" rid="ref45">45</xref>]. EMA involves short surveys delivered through phone-based apps, a method used by most of the studies. In this section, we outline several key decisions involved in collecting ground truth labels using EMA.</p></sec></sec><sec id="s3-3"><title>EMA Prompts Frequency and Delivery Mechanisms</title><p>EMA was most frequently administered daily [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref54">54</xref>], capturing a range of psychological constructs. In fewer instances, more frequent assessments were conducted (eg, every 30 minutes or every 2 hours throughout the day) [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>]. In all studies, EMA prompts were triggered by predefined schedules, with most being delivered via mobile apps, often through custom-developed apps [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]. A smaller number of studies used SMS messages to deliver EMA surveys [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref50">50</xref>].</p></sec><sec id="s3-4"><title>Response Window</title><p>Since the data were collected in real-world contexts, participants were not expected to respond promptly to EMA probes as they may have been busy with other tasks. To control the delay in response and ensure the accurate timing of the label, studies may put an upper bound on the delay, that is, the probe will expire after a certain period of time [<xref ref-type="bibr" rid="ref32">32</xref>] to ensure the relevance of labels.</p></sec><sec id="s3-5"><title>Types of Stress Labels and Instruments Used</title><p>The most commonly collected psychology constructs in the studies were stress, which occurred in 26 [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref57">57</xref>-<xref ref-type="bibr" rid="ref59">59</xref>] out of 34 (76%) studies, anxiety in 9 [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref59">59</xref>] out of 34 (26%) studies, positive or negative affect in 9 [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref49">49</xref>-<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>] out of 34 (26%) studies. Various instruments were used, including the Short State-Trait Anxiety Inventory (S-STAI) [<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref59">59</xref>], which assessed both stress and anxiety, and the Perceived Stress Scale (PSS) [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref48">48</xref>]. Anxiety was also measured using the State-Trait Anxiety Inventory (STAI) [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref56">56</xref>]. The Positive and Negative Affect Schedule (PANAS) short scale [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref53">53</xref>] was used to assess both positive and negative effects.</p></sec><sec id="s3-6"><title>Stress Measurement Scale</title><p>Additionally, different scales were used, such as the Likert scale [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>], Numeric Rating Scale [<xref ref-type="bibr" rid="ref33">33</xref>], and Visual Analog Scale [<xref ref-type="bibr" rid="ref35">35</xref>], which is a gliding scale ranging from &#x201C;not at all&#x201D; to &#x201C;extremely.&#x201D; In a few studies, discrete ordered scales were used to categorize stress levels as low, medium, or high [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref58">58</xref>].</p></sec><sec id="s3-7"><title>Preprocessing of Stress Labels</title><p>There were various preprocessing decisions researchers have made with regard to the stress labels in the reviewed studies. In most studies, the output targets in the ML model maintained the same granularity as the ground truth stress label. However, some studies applied downsampling techniques. For instance, when ground truth labels were collected every 30 minutes, they were aggregated into daily values [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref31">31</xref>] in modeling.</p><p>In classification studies, these outputs collected on a numeric scale were often categorized into bins or classes, such as low, medium, or high levels of stress or anxiety [<xref ref-type="bibr" rid="ref19">19</xref>]. Most studies used binarized output targets for classification, while only a few applied multiclass classification. Due to the ordinal nature of the data, multiple categories were combined into fewer groups. For example, on a 5-point scale, ratings of 0&#x2010;1 were classified as stress, while ratings of 2&#x2010;4 were classified as no stress [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. Another example involves merging the last 3 categories of a 5-point scale into a single &#x201C;high stress&#x201D; group, with the first 2 categories remaining as &#x201C;low&#x201D; and &#x201C;medium stress&#x201D; [<xref ref-type="bibr" rid="ref44">44</xref>]. For nondiscrete values, such as on a 100-point scale, values below 50 were marked as low stress, and values above 50 were marked as high stress [<xref ref-type="bibr" rid="ref35">35</xref>].</p><p>Two primary methods were used to define classification label thresholds: a personalized approach and a global approach. The personalized approach involved setting thresholds specific to individual participants, whereas the global approach applied the same thresholds across all participants. For personalized thresholds, various statistical methods were used, including majority voting, median calculation, and z-score analysis. The z-score method was the most commonly used [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref56">56</xref>], where z-scores were calculated for each individual to account for subjective variability. Data were then categorized into 2 classes; for example, negative classes included those with z-scores below zero, and vice versa. In the global approach, values were discretized using predefined scales or thresholds derived from the entire participant sample. For instance, some studies applied fixed global thresholds to classify stress and anxiety levels across all participants [<xref ref-type="bibr" rid="ref42">42</xref>].</p></sec><sec id="s3-8"><title>Input and Output Windows</title><p>As part of the modeling framework, we extracted the input and output windows reported in the reviewed studies. The input window refers to the time frame from which input features are extracted, typically derived from physiological and behavioral signals. On the other hand, the output window defines the period during which the model predicts target outcomes corresponding to the ground truth labels, as discussed in the &#x201C;Ground Truth Stress Labels&#x201D; section.</p><p>We use the terms nowcasting and forecasting to describe the relationship between input and output windows. Nowcasting (<xref ref-type="fig" rid="figure2">Figure 2A-C</xref>) refers to models that predict outcomes, such as stress, for the present or immediate future, resulting in a very short output window with no gap between input and output windows. In contrast, forecasting (<xref ref-type="fig" rid="figure2">Figure 2D</xref>) predicts stress over future time periods, with short-term forecasting focusing on the immediate next period and long-term forecasting predicting stress at a time point extended further into the future.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>The relationship between input and output windows; nowcasting versus forecasting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mhealth_v14i1e76632_fig02.png"/></fig><p>Several studies have used an exact overlap between input and output windows as shown in <xref ref-type="fig" rid="figure2">Figure 2A</xref>, often with a window size of 24 hours [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]. A smaller number of studies have used shorter input-output windows, such as every 20 minutes [<xref ref-type="bibr" rid="ref32">32</xref>], hourly [<xref ref-type="bibr" rid="ref44">44</xref>], or every 3 hours [<xref ref-type="bibr" rid="ref35">35</xref>]. This exact overlap is a defining characteristic of nowcasting studies, as illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><p>Additionally, many studies included partial overlap, where the input window is either longer or shorter than the output window, as shown in <xref ref-type="fig" rid="figure2">Figure 2B and C</xref>. For example, an input window might include data from the previous day, the current day, and the next day, while the output window focuses only on the current day [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>] or both the previous and current days [<xref ref-type="bibr" rid="ref54">54</xref>]. In other examples, input window is shorter than the output window; for example, the input window covers 1 minute and the output window spans 15 minutes [<xref ref-type="bibr" rid="ref30">30</xref>], or where the input window is 2 hours paired with a 1-day output window [<xref ref-type="bibr" rid="ref56">56</xref>], or an input window that selects an optimal range between 10 and 27.5 minutes while the output window remains fixed at 20 minutes [<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>In the forecasting studies identified, the input window is chosen to be immediately adjacent and trail the output window (<xref ref-type="fig" rid="figure2">Figure 2D</xref>). For example, the input window may consist of data from the previous day, with the output window set for the current day [<xref ref-type="bibr" rid="ref40">40</xref>], in which case the gap is zero. In other instances, the input window might consist of data from the last 35, 30, or 25 days, while the output window could be set for 19, 17, or 12 days [<xref ref-type="bibr" rid="ref57">57</xref>] into the future, starting from the prediction time, in which case the gap is larger than zero.</p><p>It has been observed from the selected studies that the majority of researchers have used a nowcasting relationship between the input and output windows, as illustrated in <xref ref-type="fig" rid="figure3">Figure 3</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Number of studies formulated as nowcasting versus forecasting problem, grouped by year.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mhealth_v14i1e76632_fig03.png"/></fig></sec><sec id="s3-9"><title>Addressing Issues of Data Leakage</title><p>The reviewed studies differed in how their training, test, and validation sets were split. Since the wearable dataset is longitudinally collected, it is typical to see multiple input-output pairs extracted from the same participants. Data leakage can occur where data points in the training set and test set may originate from the same participant. This made participant-level splitting relevant to the assessment of potential data leakage and possible inflation of reported model performance.</p><p>In our review, we observed that 13 [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref59">59</xref>] out of 28 applicable (46%) studies [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref59">59</xref>] did not explicitly address potential data leakage issues as inferred from the description of the split of training, test, or validation set, while the rest addressed data leakage issues by using strategies such as leave-one-subject-out cross-validation.</p></sec><sec id="s3-10"><title>ML Models</title><p>This section provides insights into the types of ML models used. The studies included in this review primarily focused on supervised learning paradigms, addressing both classification and regression tasks. A diverse range of models was used, spanning traditional ML frameworks to end-to-end deep learning architectures. For feature extraction, most studies relied on statistical summaries [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref36">36</xref>], while a smaller subset applied Fourier transformations, particularly for analyzing heart rate variability signals [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref42">42</xref>].</p><p>Traditional ML algorithms, valued for their simplicity and interpretability, such as support vector machines (SVMs), Naive Bayes, and k-nearest neighbors (KNN), were frequently used. Additionally, various ensemble methods, including bagging and boosting techniques, were widely applied [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]. Among these, random forest (RF) was the most commonly used in studies using traditional ML algorithms [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref59">59</xref>].</p><p><xref ref-type="table" rid="table4">Table 4</xref> summarizes the most frequently used traditional ML models across the studies.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Counts of papers using specific traditional machine learning models.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Count of papers using the model</td><td align="left" valign="bottom">Reference numbers</td></tr></thead><tbody><tr><td align="left" valign="top">Random forest</td><td align="left" valign="top">11</td><td align="left" valign="top">Parousidou et al 2023 [<xref ref-type="bibr" rid="ref45">45</xref>], Gjoreski et al 2017 [<xref ref-type="bibr" rid="ref19">19</xref>], Mishra et al 2020 [<xref ref-type="bibr" rid="ref36">36</xref>], Jiang et al 2023 [<xref ref-type="bibr" rid="ref54">54</xref>], Burghardt et al 2021 [<xref ref-type="bibr" rid="ref52">52</xref>], Feng and Narayanan 2022 [<xref ref-type="bibr" rid="ref49">49</xref>], Booth et al 2022 [<xref ref-type="bibr" rid="ref38">38</xref>], Robles-Granda et al 2021 [<xref ref-type="bibr" rid="ref22">22</xref>], Saylam and &#x0130;ncel 2023 [<xref ref-type="bibr" rid="ref50">50</xref>], Saylam and Durmaz &#x0130;ncel 2022 [<xref ref-type="bibr" rid="ref58">58</xref>], Kim et al 2024 [<xref ref-type="bibr" rid="ref59">59</xref>]</td></tr><tr><td align="left" valign="top">SVM<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">9</td><td align="left" valign="top">Boateng and Kotz 2016 [<xref ref-type="bibr" rid="ref30">30</xref>], Bavaresco et al 2020 [<xref ref-type="bibr" rid="ref32">32</xref>], Gjoreski et al 2017 [<xref ref-type="bibr" rid="ref19">19</xref>], Han et al 2020 [<xref ref-type="bibr" rid="ref20">20</xref>], Mishra et al 2020 [<xref ref-type="bibr" rid="ref36">36</xref>], Burghardt et al 2021 [<xref ref-type="bibr" rid="ref52">52</xref>], Pimentel et al 2021 [<xref ref-type="bibr" rid="ref55">55</xref>], Tiwari and Falk 2021 [<xref ref-type="bibr" rid="ref42">42</xref>], Kim et al 2024 [<xref ref-type="bibr" rid="ref59">59</xref>]</td></tr><tr><td align="left" valign="top">Boosting methods</td><td align="left" valign="top">7</td><td align="left" valign="top">Gjoreski et al 2017 [<xref ref-type="bibr" rid="ref19">19</xref>], Stojchevska et al 2022 [<xref ref-type="bibr" rid="ref44">44</xref>], Parousidou et al 2023 [<xref ref-type="bibr" rid="ref45">45</xref>], Paraschou et al 2023 [<xref ref-type="bibr" rid="ref47">47</xref>], Saylam and &#x0130;ncel 2023 [<xref ref-type="bibr" rid="ref50">50</xref>], Kao et al 2020 [<xref ref-type="bibr" rid="ref51">51</xref>], Jiang et al 2023 [<xref ref-type="bibr" rid="ref54">54</xref>]</td></tr><tr><td align="left" valign="top">Naive Bayes</td><td align="left" valign="top">4</td><td align="left" valign="top">Parousidou et al 2023 [<xref ref-type="bibr" rid="ref45">45</xref>], Bavaresco et al 2020 [<xref ref-type="bibr" rid="ref32">32</xref>], Gjoreski et al 2017 [<xref ref-type="bibr" rid="ref19">19</xref>], Han et al 2020 [<xref ref-type="bibr" rid="ref20">20</xref>]</td></tr><tr><td align="left" valign="top">KNN<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">4</td><td align="left" valign="top">Bavaresco et al 2020 [<xref ref-type="bibr" rid="ref32">32</xref>], Gjoreski et al 2017 [<xref ref-type="bibr" rid="ref19">19</xref>], Han et al 2020 [<xref ref-type="bibr" rid="ref20">20</xref>], Jiang et al 2023 [<xref ref-type="bibr" rid="ref54">54</xref>]</td></tr><tr><td align="left" valign="top">Logistic regression</td><td align="left" valign="top">4</td><td align="left" valign="top">Parousidou et al 2023 [<xref ref-type="bibr" rid="ref45">45</xref>], Tump et al 2022 [<xref ref-type="bibr" rid="ref35">35</xref>], Jiang et al 2023 [<xref ref-type="bibr" rid="ref54">54</xref>], Burghardt et al 2021 [<xref ref-type="bibr" rid="ref52">52</xref>]</td></tr><tr><td align="left" valign="top">Decision tree</td><td align="left" valign="top">3</td><td align="left" valign="top">Parousidou et al 2023 [<xref ref-type="bibr" rid="ref45">45</xref>], Gjoreski et al 2017 [<xref ref-type="bibr" rid="ref19">19</xref>], Robles-Granda et al 2021 [<xref ref-type="bibr" rid="ref22">22</xref>]</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>SVM: support vector machine.</p></fn><fn id="table4fn2"><p><sup>b</sup>KNN: k-nearest neighbors.</p></fn></table-wrap-foot></table-wrap><p>In addition to traditional ML, the use of deep learning models has seen a significant rise. Models explored include recurrent neural networks (RNNs) and long short-term memory (LSTM) networks, convolutional neural networks (CNNs), and LLMs. The choice of models is closely tied to the specific objectives and implications of each study, as discussed below.</p></sec><sec id="s3-11"><title>Challenges and Solutions</title><p>The reviewed studies aimed to address specific challenges, each accompanied by proposed solutions. These challenges were either intrinsic to the data or inherent to the domain. Some common challenges and their corresponding solutions are outlined below</p></sec><sec id="s3-12"><title>Personalization</title><p>Personalization was one of the challenges addressed in the reviewed studies, because of intrinsic between-person variability, as individuals may respond differently to the same stressors. In many of the selected studies, personalization was not considered, thus following a one-size-fits-all approach in modeling. However, a notable number of studies addressed personalization as part of the preprocessing phase. Two key personalization approaches were identified: user-based and group-based. While the user-based approach involves creating a unique model for each participant, the group-based approach groups participants based on selected attributes, such as demographics or gender.</p><p>Various clustering algorithms were used in the studies to create cohorts or groups, including k-means, spectral clustering, agglomerative clustering, mean shift, affinity propagation, and Gaussian mixture models (GMM) [<xref ref-type="bibr" rid="ref51">51</xref>]. Some other studies also used additional techniques to enhance cluster quality, such as iterative clustering [<xref ref-type="bibr" rid="ref41">41</xref>], which progressively refined participant groupings by repeating the clustering process multiple times. Another method listed in the reviewed studies was dimensionality reduction prior to clustering, which was achieved through principal component analysis (PCA) [<xref ref-type="bibr" rid="ref43">43</xref>]. PCA simplified the feature space while preserving variance by projecting the data into a lower-dimensional space, whereas self-organizing maps (SOMs) [<xref ref-type="bibr" rid="ref31">31</xref>] facilitated the organization of complex patterns within the data.</p><p>The clusters created during the preprocessing step were further trained using traditional ML techniques (eg, RF, decision trees, and linear regression) [<xref ref-type="bibr" rid="ref45">45</xref>] as well as deep learning approaches, including metric learning with Siamese neural networks (SNNs) [<xref ref-type="bibr" rid="ref43">43</xref>]. SNNs transform input data to ensure that similar data points are projected close together, while dissimilar points are spaced apart in the learned feature space. Additionally, feedforward neural networks (FFNs) [<xref ref-type="bibr" rid="ref41">41</xref>] were used to fine-tune the generalized model by adjusting its weights through iterative training on group-specific data.</p><p>Clustering offers several benefits. For example, it may induce more cohesive groups as grouping users with similar characteristics helps the model detect common stress patterns within each group [<xref ref-type="bibr" rid="ref45">45</xref>]. In another study, the user-based collaborative filtering method is used by leveraging health information from other users. This type of model can fill gaps within groups and reduce the reliance on historical data from individual users [<xref ref-type="bibr" rid="ref51">51</xref>].</p></sec><sec id="s3-13"><title>Temporal Dependency</title><p>Data collected from wearable sensors inherently possess a time-dependent nature due to its intrinsic properties. Consequently, studies have explored methods to explicitly address these temporal dependencies to improve model accuracy, with the hope that those methods may reveal complex relationships that conventional ML models may overlook. Various models specifically designed to capture these temporal dependencies are listed below, along with a brief summary of their usage in the selected studies.</p><list list-type="order"><list-item><p>LSTM networks, designed to capture long-term dependencies in sequential data, were used as bidirectional LSTM (BI-LSTM) in one of the studies [<xref ref-type="bibr" rid="ref21">21</xref>]. In that study, BI-LSTM captured both past and future values to predict current stress levels in a nowcasting setup. An hourly vector was processed in both the forward and backward directions to analyze data from start to finish and vice versa. The outputs from both directions were combined to generate the final predictions, offering a comprehensive understanding of the hourly temporal context. In another study, a lagged version of the data was explored to highlight trends over time, associating each data point with its historical values, allowing the model to learn from past trends [<xref ref-type="bibr" rid="ref38">38</xref>].</p></list-item><list-item><p>Vector autoregression (VAR) [<xref ref-type="bibr" rid="ref33">33</xref>] was used to predict current values based on past data through lag selection. Two approaches were taken: predicting sensor outcomes (eg, total sleep time and heart rate variability) using EMA variables such as stress, and predicting self-reported EMA variables based on sensor data. An impulse response function (IRF) was also applied to visualize how changes in predictor variables affect outcomes over time.</p></list-item><list-item><p>A hidden Markov model was used to capture how different physiological conditions evolve over time through hidden states. This study [<xref ref-type="bibr" rid="ref52">52</xref>] aimed to detect both typical and atypical events by modeling physiological data, helping identify patterns that represent normal daily behaviors as well as deviations caused by stress and anxiety.</p></list-item><list-item><p>Higher-order networks [<xref ref-type="bibr" rid="ref22">22</xref>] were used to model temporal dependencies using ensemble learning frameworks, integrating multiple ML models to predict physical, psychological, and job performance variables. These models leveraged time-dependent patterns in physiological signals by incorporating lagged data.</p></list-item><list-item><p>Conceptual frameworks such as dynamical systems theory [<xref ref-type="bibr" rid="ref40">40</xref>] and chaos theory [<xref ref-type="bibr" rid="ref57">57</xref>] have been applied to capture temporal relationships. One study used a dynamical systems model with linear regression to identify daily patterns of emotional self-regulation and stress spillover in health care professionals, effectively capturing recurring cycles. In another study that used chaos theory, phase space reconstruction was applied to transform low-dimensional data into high-dimensional data by combining stress levels from multiple days as output variables. This approach enabled the model to account for previous days&#x2019; stress levels, ensuring that similar stress levels appeared closer in the reconstructed space. Additionally, integrating chaos theory with LSTM models that incorporated lagged values improved performance compared to using either chaos theory or LSTM models with lagged values independently.</p></list-item></list></sec><sec id="s3-14"><title>Multiple Modalities</title><p>The feasibility of using all training modalities is often limited in real-world scenarios due to the restricted availability of sensors in wearables. To address this, one study focused on reducing the number of modalities during testing while incorporating more during training [<xref ref-type="bibr" rid="ref46">46</xref>]. This was achieved using a technique called knowledge distillation, implemented as the &#x201C;More to Less&#x201D; framework. Knowledge distillation involves transferring knowledge from a stronger network to a weaker one by using classifier networks from different modalities. During training, the model used all available modalities, with each having its own classifier network. These classifiers exchanged knowledge through adaptive mechanisms defined in the &#x201C;More to Less&#x201D; framework, allowing each classifier to learn not only from its own modality but also from those performing better. As a result, testing was conducted with fewer modalities, as the network had already developed robust representations during training. This approach enhanced generalization by leveraging information from multiple modalities, even when fewer input modalities were available during testing.</p></sec><sec id="s3-15"><title>Bias Mitigation</title><p>Bias in relation to demographic groupings was addressed in one reviewed study [<xref ref-type="bibr" rid="ref56">56</xref>]. In that study, an LSTM model was used by incorporating multitask learning, which included both the protected attribute and the target variable. This approach aimed to mitigate bias related to the protected attribute. To achieve this, different loss weights were assigned. For example, if the target variable (eg, anxiety) is given a weight of 4.5 and the protected attribute (eg, gender) a weight of 0.5, the problem is treated as a multitask learning scenario. Each task has a separate loss function, ensuring that the nonprotected label, such as anxiety in this case, is prioritized.</p></sec><sec id="s3-16"><title>Context Awareness</title><p>Contextual information is widely recognized as crucial in studies aimed at identifying the causes or triggers of stress, as learning context helps improve model accuracy. Various types of contextual data considered in these studies are outlined below:</p><list list-type="order"><list-item><p>One study compared 2 methods of context collection: ML-derived data, such as sleep and activity, versus self-reported data collected through EMA [<xref ref-type="bibr" rid="ref44">44</xref>]. The results showed that ML-derived context significantly improved model performance over self-reported data.</p></list-item><list-item><p>Another study focused on key stressors among employees working from home, highlighting the role of environmental factors as contextual elements affecting stress and well-being [<xref ref-type="bibr" rid="ref35">35</xref>]. This approach highlighted the importance of both personal and environmental context in remote work settings.</p></list-item><list-item><p>A separate set of studies integrated circadian cycles as a contextual factor for predicting wellness indicators&#x2014;such as positive affect, negative affect, and life satisfaction&#x2014;by analyzing data such as audio features, heart rate, sleep patterns, activity levels, and location [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref53">53</xref>].</p></list-item><list-item><p>Various models have been used in these studies, including ensemble methods such as gradient-boosted trees (eg, CatBoost), RF, logistic regression (LR), and linear regression.</p></list-item></list></sec><sec id="s3-17"><title>Addressing Limited Supply of Training Data</title><p>Given the complexity of data collection in the naturalistic setting, there is a limited supply of high-quality datasets, along with challenges in capturing ground truth. To address those challenges, a study [<xref ref-type="bibr" rid="ref54">54</xref>] explored zero-shot learning within a meta-learning framework. This framework incorporated various base learners, including RF, AdaBoost (AB), gradient boosting (GB), KNN, and LR. The approach enabled the model to learn from diverse tasks, allowing it to quickly adapt to new, unseen tasks with minimal additional training. By training on different tasks and populations, the model developed effective initialization weights, improving its ability to make accurate predictions on unseen data.</p></sec><sec id="s3-18"><title>Use of LLM in Stress Detection</title><p>Recently, with advancements in AI, emerging technologies such as LLMs have been developed. These models are pretrained on vast amounts of text and can be applied to a wide range of tasks. One study in the review used LLMs along with wearable sensor data through structured prompts that incorporated user context, health knowledge, and temporal information [<xref ref-type="bibr" rid="ref59">59</xref>]. The study focused on three approaches:</p><list list-type="order"><list-item><p>Zero-shot prompting: in this approach, the model received prompts without any prior examples. Models pretrained on task-specific data, such as medical or sensor data, performed better; for instance, Asclepius outperformed larger models.</p></list-item><list-item><p>Few-shot prompting: a small number of examples were included in the prompt to provide context for the model. This resulted in improved performance in larger models such as GPT-3.5 (OpenAI) and GPT-4 (OpenAI).</p></list-item><list-item><p>Fine-tuning: this method involves adjusting some or all parameters of a pretrained model using a dataset specific to the target task. HealAlpaca achieved the highest accuracy in this category.</p></list-item></list><p>Overall, few-shot prompting outperformed zero-shot prompting, while fine-tuning proved to be the most effective approach.</p></sec><sec id="s3-19"><title>Analysis of Model Performance</title><p>In <xref ref-type="table" rid="table3">Table 3</xref>, we presented a detailed analysis of model performance and extracted key evaluation metrics, such as <italic>F</italic><sub>1</sub>-score, accuracy, precision, recall, and AUC for classification tasks, and MSE for regression tasks. We reported the best-performing model in each study. Where applicable, we also documented the corresponding model configurations and comparison baselines. It should be noted that data leakage considerations are only relevant for models designed to predict momentary stress labels. Six studies [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref53">53</xref>] fall outside this scope, including (1) statistical models (often linear or linear mixed-effects models) that fit the entire dataset to examine coefficient significance or overall goodness-of-fit; and (2) models predicting stable psychological constructs derived from one-time baseline surveys, where the unit of analysis is the individual rather than the event or time point. We identified a total of 13 studies [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref59">59</xref>] that had potential data leakage issues out of 28 applicable studies [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref59">59</xref>].</p><p>There are a few caveats in interpreting performance measures:</p><list list-type="order"><list-item><p>Substantial heterogeneity exists across datasets and problem formulations (eg, classification vs regression, input and output window configurations, nowcasting vs forecasting, and different discretization strategies for stress labels), making direct, apples-to-apples comparisons challenging.</p></list-item><list-item><p>Performance reporting is not standardized across studies. Although <italic>F</italic><sub>1</sub>-score is the most commonly reported metric for classification, several studies do not report it; in the case of class imbalance (which is often not reported), those metrics may not adequately reflect true model performance.</p></list-item><list-item><p>A notable proportion of studies (13 [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref59">59</xref>] out of 28 [46%] studies [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref59">59</xref>] among those applicable studies) has potential data leakage, most often due to inappropriate data-splitting strategies, which may inflate reported results. In fact, we note a few studies report unusually high performance (eg, with accuracy&#x003E;85%), many of them with potential data leakage issues.</p></list-item><list-item><p>We note that a substantial proportion of studies (approximately 50%) do not report uncertainty quantification for performance metrics, such as CIs, SEs, SDs, or <italic>P</italic> values, making it difficult to determine whether observed performance differences are statistically meaningful or attributable to random variation.</p></list-item></list><p>There are several major findings:</p><list list-type="order"><list-item><p>Across studies, deep learning models achieved <italic>F</italic><sub>1</sub>-scores ranging from 0.44 to 0.72, whereas traditional ML models exhibited a wider performance range (<italic>F</italic><sub>1</sub>-score=0.47&#x2010;0.90), making it difficult to draw a general conclusion about the absolute superiority of deep learning approaches. A rigorous comparison would require controlled, head-to-head experiments that hold other factors, such as data modality, problem formulation, or data split, constant. As 2 of those examples, Parousidou et al [<xref ref-type="bibr" rid="ref45">45</xref>] show that traditional ML combined with personalization can outperform deep learning, while Pimentel et al [<xref ref-type="bibr" rid="ref55">55</xref>] demonstrate that careful feature engineering enables traditional models to achieve superior performance relative to deep learning methods.</p></list-item><list-item><p>We note that performance also varied by modeling strategy: personalization-based approaches reported <italic>F</italic><sub>1</sub>-scores between 0.62 and 0.66; methods addressing limited data availability achieved <italic>F</italic><sub>1</sub>-scores around 0.70; context-aware models showed a wide range with <italic>F</italic><sub>1</sub>-score from 0.41 to 0.90; multimodal integration approaches reported an <italic>F</italic><sub>1</sub>-score of 0.72; and models accounting for temporal dependencies exhibited wide variability (<italic>F</italic><sub>1</sub>-score=0.23&#x2010;0.71). Finally, studies using LLMs reported accuracy values as high as 92%, although corresponding <italic>F</italic><sub>1</sub>-scores were not provided.</p></list-item></list><p>Ethical Considerations</p><p>Stress detection applications and associated wearable devices often involve the collection of fine-grained longitudinal data, as well as psychological profile information, either in raw form or inferred through computational models. Such data fall within the realm of sensitive private information and, if inadvertently disclosed, may pose significant risks to individuals. For example, employers could potentially misuse stress or negative affect profiles in human resource&#x2013;related decision-making processes. In addition, weak data security policies may increase the likelihood of security breaches, which could enable the reidentification of individuals and lead to the leakage of personally identifiable information. Accordingly, research studies on stress detection, as well as the downstream applications built upon these systems, must adopt deliberate and robust measures to safeguard the entire data collection, storage, and processing pipeline. These measures are essential for providing a high level of assurance and protection to individuals, whether they participate as research subjects or engage with such systems as end users of downstream applications. Some emerging approaches may also be explored, such as federated learning, as suggested in one of the reviewed studies [<xref ref-type="bibr" rid="ref59">59</xref>]. In this approach, raw data remain on the local device where they are recorded, thereby reducing the privacy risks associated with centralized cloud-based data storage and processing.</p><p>Equally important is the need for transparent and meaningful informed consent procedures. Data owners should be clearly informed about what types of data are being collected, how the data will be processed and analyzed, who will have access to the data, how long the data will be retained, and whether the data may be reused for secondary research purposes or commercial applications. Such transparency is critical for establishing trust and supporting individuals&#x2019; autonomy in making informed decisions regarding participation. In addition, participants should be made aware of potential risks, including the possibility of reidentification or unintended disclosure of sensitive psychological or behavioral information. As noted, among the 14 datasets used across the reviewed studies, some based on primary data collection and others relying on secondary datasets, only 11 explicitly acknowledged obtaining proper institutional review board (IRB) approval, whereas the remaining 3 studies did not report such approval [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. This observation highlights the continuing need for stronger ethical oversight and clearer reporting practices in stress detection research involving wearable and behavioral data.</p></sec><sec id="s3-20"><title>Model Card for Wearable-Based Stress Modeling</title><p>In this section, we propose a model card framework tailored for wearable-based stress modeling, adapted from the general model card template introduced by prior work [<xref ref-type="bibr" rid="ref23">23</xref>] in the ML community. The specific focus is on models that enable the momentary estimation of current stress or to forecast stress in the future. The original template includes sections such as model details, intended use, factors, metrics, evaluation data, training data, quantitative analysis, ethical considerations, and caveats and recommendations. Given the unique characteristics of wearable-based stress modeling, particularly the reliance on longitudinal, in-the-wild biosignal data collection and self-reported ground truth, we reorganize and customize the model card into two core components: (1) dataset and (2) modeling decisions. This structure foregrounds the data provenance and methodological choices that most strongly influence model validity, transparency, and reproducibility. A detailed model card for wearables, including its components, dimensions, and reporting requirements, is presented in <xref ref-type="table" rid="table5">Table 5</xref> below.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Proposed model card reporting template for wearable-based stress detection models.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Component and dimension</td><td align="left" valign="bottom">Reporting requirements</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Modeling decisions</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Model metadata</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Model creators and affiliated organization</p></list-item><list-item><p>Model version, date, citation, and license</p></list-item><list-item><p>Point of contacts and link to code repository if applicable</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Dataset</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Participant characteristics and context</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Participant demographics (eg, age, gender, and occupation) and their distributions</p></list-item><list-item><p>Sample size and duration and frequency of data collection</p></list-item><list-item><p>Geographic locations of participants</p></list-item><list-item><p>Naturalistic context: description of participants&#x2019; typical daily life (eg, work schedules and routines) and common stressors relevant to the population under study</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Wearable devices and signals</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Wearable devices used and placement</p></list-item><list-item><p>Biosignals captured (eg, heart rate, HRV<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>, EDA<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup>, and accelerometry) and sampling frequency</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ground truth annotation</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Target variables (eg, perceived stress) and how they are captured</p></list-item><list-item><p>EMA<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup> protocol details, including prompt frequency and scheduling strategy</p></list-item><list-item><p>Response window and compliance requirements</p></list-item><list-item><p>Survey instrument used</p></list-item><list-item><p>Descriptive statistics of target variables, including distributions across demographic subgroups, where applicable</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Metadata</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Dataset versioning, documentation, point of contact, citations, licenses, and so on and any relevant contextual metadata (eg, time zones, device firmware, and protocol deviations)</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Modeling decisions</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Data preparation</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Preprocessing steps, including handling of missing data, signal cleaning, normalization, multimodal data synchronization, and identification of subgroups</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Problem formulation</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Task definition (eg, classification vs regression)</p></list-item><list-item><p>Temporal framing</p></list-item><list-item><p>Task specification</p></list-item><list-item><p>Nowcasting versus forecasting input and output window configurations</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Experimental design</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Modeling setup and validation strategy</p></list-item><list-item><p>Data splitting procedures (training, validation, and test)</p></list-item><list-item><p>Assessment of potential data leakage and use mitigation strategies such as subject-level data-split</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Modeling approach</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Provide detailed configuration of the model, including those do not perform well</p></list-item><list-item><p>Document hyperparameter tuning strategy and process</p></list-item><list-item><p>Report a variety of performance metrics, including uncertainty measures such as CIs, SE, or <italic>P</italic> value from statistical significance test</p></list-item><list-item><p>Report baseline model performance (including default model) for comparison. Decision thresholds (for classification models), if applicable</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Dataset</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ethical considerations</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Ethical approval (eg, IRB<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup> and participant consent procedures)</p></list-item><list-item><p>Recruitment and sampling strategy, and discuss potential of sampling bias</p></list-item><list-item><p>Privacy protection strategy, eg, anonymization</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Modeling decisions</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Caveats and recommendations</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Justification of key modeling choices and the specific challenges they are intended to address</p></list-item><list-item><p>Discussion on factors influencing performance, including reporting of disaggregated performance across relevant subgroups where applicable</p></list-item><list-item><p>Discussion on sampling bias and generalization limitations, especially when transferring models across populations, contexts, or devices</p></list-item><list-item><p>Other known limitations, appropriate use cases, and guidance for interpretation</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>HRV: heart rate variability. </p></fn><fn id="table5fn2"><p><sup>b</sup>EDA: electrodermal activity. </p></fn><fn id="table5fn3"><p><sup>c</sup>EMA: ecological momentary assessment.</p></fn><fn id="table5fn4"><p><sup>d</sup>IRB: institutional review board.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>Stress detection has become increasingly important due to the rising prevalence of stress in individuals&#x2019; daily lives. Advances in wearable technologies have opened new opportunities for detecting stress in real-world settings. This scoping review aims to examine the existing literature on the use of ML applied to data collected from wearables in naturalistic environments for stress detection. In this section, we summarize the principal findings, the identified research gaps, and potential avenues for further investigation, while acknowledging the limitations of this review.</p><p>This review identified a lack of standardization in model reporting, which is one of the biggest barriers to advancing the field beyond existing research, and the generalizability of stress detection in naturalistic settings. This is due to substantial heterogeneity in how problem specifications are defined, with no standard format for model reporting, which limits benchmarking across current studies. The reviewed studies differed considerably in the wearables used, ground-truth labeling strategies, input-output window definitions, and ML approaches. We also identified methodological gaps, including potential data leakage, inconsistent reporting of performance metrics, limited handling of class imbalance, and relatively little explicit attention to motion artifacts. In addition, we noted that the reviewed studies addressed a range of challenges, though not uniformly, including personalization, temporal modeling, bias mitigation, modality-related concerns, challenges specific to in-the-wild settings, and the use of LLMs. These findings also motivated the proposed model card specification to improve comparability, transparency, and reproducibility across studies in the future.</p></sec><sec id="s4-2"><title>Methodological Rigor</title><p>In our analysis of model frameworks, characteristics, and performance summarized in <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>, we observed that 13 [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref59">59</xref>] out of 28 (46%) applicable studies [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref59">59</xref>], specifically, those that developed ML models to predict momentary stress had potential data leakage issues. These issues primarily arise from ignoring the grouped structure of the data, where multiple observations or episodes are contributed by the same participant. When training, validation, and test splits are performed at the observation level rather than the participant level, data points from the same individual may appear in both the training and evaluation sets. This overlap can result in unintended information leakage from training to test data, thereby artificially inflating model performance. A commonly recommended practice to mitigate this issue is to perform data splitting at the subject level, ensuring that all observations from a given participant are assigned to a single split. Given that a substantial proportion of the reviewed studies exhibit this potential methodological flaw, the performance metrics reported in <xref ref-type="table" rid="table3">Table 3</xref> should be interpreted with caution, as they may overestimate true generalization performance.</p><p>Another issue identified in our analysis of model performance is that a large proportion of studies do not explicitly report baseline model performance. This omission is particularly problematic in the presence of class imbalance, where commonly reported metrics such as accuracy may be misleading. For example, a na&#x00EF;ve classifier that always predicts the majority class may achieve deceptively high accuracy, despite providing little meaningful predictive value. Moreover, performance metrics that are more robust to class imbalance, such as the AUC, are rarely reported. While some studies attempt to address this issue by using alternative measures, including balanced accuracy (BACC) or Matthews correlation coefficient (MCC), such practices remain the exception rather than the norm. In addition, we observe that a substantial proportion of studies do not report uncertainty quantification for performance metrics, such as CIs or SDs. The absence of these measures complicates performance comparison across models or approaches, as observed differences may simply reflect random variation rather than meaningful performance gains.</p><p>To address these methodological gaps, we explicitly incorporate these considerations into our proposed model card specification, with the goal of encouraging future research to adopt more rigorous experimental designs, performance measures, and reporting practices.</p></sec><sec id="s4-3"><title>Standardization and Benchmarking</title><p>The complexity and richness of data collected from wearables and other devices in real-world settings create ample opportunities to explore a wide range of ML frameworks. While this flexibility allows researchers to pose a variety of research questions and explore various kinds of models, it also poses significant challenges when comparing the utility and performance of different models in tasks such as stress detection and estimation. As demonstrated in our analysis, the lack of standardization in reporting often makes it difficult to extract key specific details about ML modeling decisions, thus limiting the ability to compare model effectiveness across studies. This lack of standard and benchmarking is likely to slow progress in developing robust ML models for stress detection.</p><p>Future research could focus on the establishment of benchmark datasets and standardized tasks to enable meaningful comparisons. Additionally, the research community would benefit from adopting consistent reporting standards that ensure critical components of ML frameworks are clearly documented. For example, approaches such as model cards [<xref ref-type="bibr" rid="ref23">23</xref>] promote transparency and reproducibility. There should also be a stronger emphasis on model quality control, including the adoption of best practices in experimental design to mitigate issues such as data leakage, which may contribute to the inflation of reported model performance. In this paper, we proposed a model card solution specifically tailored to modeling stress in the wild using wearable data, based on our analysis and understanding of the critical issues and related best practices related to modeling decisions, performance analysis, and reporting.</p></sec><sec id="s4-4"><title>Longitudinal View of Stress: From Nowcasting to Forecasting</title><p>A distinguishing feature of real-life stress studies is the ability to examine stress over extended periods and explore its interactions with contextual factors. Despite these opportunities, we observed that most studies adopt a short-term approach to stress detection, where the input and output windows are closely aligned in time. Only a few recent studies have begun to explore forecasting methods that account for long-term temporal trends in psychological and physiological states.</p><p>Among the 4 forecasting-oriented papers, 2 adopt a conventional ML pipeline focused on next-day stress prediction. Paraschou et al [<xref ref-type="bibr" rid="ref47">47</xref>] report an accuracy of 92.3% from a GB ML model, although the evaluation may be subject to data leakage concerns. Another study [<xref ref-type="bibr" rid="ref50">50</xref>] systematically explores a wide range of input and output window configurations and finds that LSTM-based models generally achieve the strongest performance. In particular, the highest predictive accuracy is obtained when using data from the preceding 15 days (ie, input window=15 days) to predict next-day stress (ie, output window=1 day), although XGBoost (eXtreme Gradient Boosting) models show competitive performance under several configurations. In contrast, the remaining 2 papers adopt dynamic systems modeling approaches [<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref57">57</xref>]. Hadjiantonis et al [<xref ref-type="bibr" rid="ref40">40</xref>] fit a linear dynamical system model that characterizes next-day changes in emotional arousal as a linear function of the current day&#x2019;s perceived stress and emotional arousal, with model parameters estimated via linear regression. Li et al [<xref ref-type="bibr" rid="ref57">57</xref>], however, model day-to-day perceived stress as a nonlinear dynamical system. The study follows a 2-step approach: first, chaos theory is used to identify the most reliable prediction horizon, which shows that an input window of 19 days maximizes predictability for forecasting stress over the subsequent 35 days. Using phase-space embedding and deep neural networks, the proposed method achieves accuracies of 74.4% for binary stress classification and 69.23% for 3-level stress classification under a person-wise 80/20 split.</p><p>We encourage future research to broaden the scope of stress modeling beyond momentary detection, incorporating longitudinal perspectives that capture the ebb and flow of stress over time. Such studies could also investigate the interplay between stress and mediating factors such as sleep, as well as coping strategies such as meditation, to provide a more comprehensive understanding of stress dynamics in daily life. Our review suggests that methodological innovations adopting multivariate and longitudinal perspectives are still emerging. Traditional ML pipelines provide a reasonable starting point by framing stress forecasting as a mapping between multimodal signal streams within predefined input windows and corresponding output windows. However, much of the existing work relies on black-box models that offer limited insight into the underlying mechanisms of stress-related dynamics. A complementary line of research draws on dynamical systems&#x2013;inspired approaches, including both linear and nonlinear models. Within these frameworks, prediction targets often extend beyond a single future time point to a prediction horizon spanning multiple days or weeks. This shift enables more meaningful and theoretically grounded questions to be addressed, such as the evolution trajectory and long-term trends of stress dynamics over time. Chaos theory&#x2013;based approaches, in particular, provide a useful lens for understanding the limits of predictability in forecasting future stress trajectories. Li et al [<xref ref-type="bibr" rid="ref57">57</xref>] represent a promising first step in this direction, and future work could further advance this line of inquiry by incorporating richer, multivariate biosignals from wearable devices in both the input and output spaces.</p><p>For both ML-based and dynamical systems&#x2013;based models, a key challenge in fitting highly complex models lies in data sparsity and limited transparency, as these approaches often require large amounts of high-quality longitudinal data while offering limited interpretability. One promising direction for addressing these challenges is the systematic integration of domain knowledge, such as insights from psychological theory, through principled frameworks such as neurosymbolic systems [<xref ref-type="bibr" rid="ref60">60</xref>,<xref ref-type="bibr" rid="ref61">61</xref>], which combine data-driven learning with theory-informed constraints and representations. Advancing such approaches will likely require close multidisciplinary collaboration between computing researchers and domain experts in fields such as psychology and health care.</p></sec><sec id="s4-5"><title>Addressing Modeling Challenges in the Wild</title><p>Data collection in real-life settings, where participants are moving, poses significant challenges to the validity of physiological signals due to motion artifacts. Additionally, it remains unclear to what extent models trained on laboratory-controlled datasets generalize to real-world scenarios. Our analysis found that only a few studies explicitly acknowledge and address these challenges. To address motion artifacts, a study [<xref ref-type="bibr" rid="ref42">42</xref>] proposed subband HRV features in addition to traditional HRV benchmark features (time-domain and frequency-domain features). The proposed features were obtained by splitting the HRV tachogram into low-frequency (LF) and high-frequency (HF) bands and analyzing each band separately. From these bands, the authors computed nonlinear features, such as transfer entropy between LF and HF components, as well as spectral descriptor features, such as centroid, spread, skewness, kurtosis, crest, and spectral entropy. The study found that its Fuse-All model, which combined the traditional benchmark HRV features with the proposed subband features, gave better performance than the benchmark features alone. Another study [<xref ref-type="bibr" rid="ref36">36</xref>] took a deliberate approach to compare the models developed in the lab versus those in the wild. They used a commercial off-the-shelf heart rate monitor in both laboratory and real-world settings on the same cohort of participants to ensure comparability. Stress labels collected in the wild were obtained through EMA-based self-reports. The results showed that the best-performing pipeline involved removing extreme values through trimming (ie, outlier removal), followed by data standardization using z-score normalization (trim_zscore). Furthermore, excluding periods of high physical activity further improved performance in real-world settings.</p></sec><sec id="s4-6"><title>Model Transparency</title><p>In addition to evaluating model performance using common metrics such as accuracy, an equally important aspect of model quality is transparency. Transparent models not only allow for better diagnosis and refinement but also create opportunities for interdisciplinary collaboration with experts from fields such as psychology, who bring deep knowledge of stress and its mechanisms. However, many current modeling approaches rely on black-box methods&#x2014;particularly those based on deep learning and, more recently, generative AI. While these models may improve performance in certain cases, they also make interpreting results more challenging [<xref ref-type="bibr" rid="ref59">59</xref>]. Future research should explore modeling approaches that achieve a better balance between transparency and accuracy. By integrating domain knowledge from psychology and physiology, such models could not only predict stress but also explain why certain predictions are made&#x2014;uncovering not just correlational relationships, but potentially causal ones that offer insights for designing effective stress interventions. Additionally, developing models and systems that facilitate clear communication of model decisions to end users could open a new line of inquiry focused on ensuring human-in-the-loop approaches in stress modeling as explored in the study by Paraschou [<xref ref-type="bibr" rid="ref47">47</xref>].</p></sec><sec id="s4-7"><title>LLMs in Wearable-Based Stress Detection</title><p>With the rise of LLMs, there is increasing interest in using them for mental health prediction tasks such as stress, anxiety, depression, and sleep disorders. One study included in this review demonstrated promising predictive performance using both larger LLMs, such as GPT-3.5 and GPT-4, and a smaller health-focused fine-tuned model, HealthAlpaca. The study suggested that performance depends not only on model size but also on prompting strategy, contextual information, and domain-specific fine-tuning, with smaller fine-tuned models performing comparably to or better than larger general-purpose LLMs. It also showed that the performance of larger models can be improved by providing examples through few-shot prompting or by adding contextual information to zero-shot prompts. However, several limitations remain. Wearable time-series data are high-dimensional, nonlinear, and continuous, making them more challenging for LLMs than conventional language inputs. In addition, the validity and interpretability of LLM-based predictions remain important concerns due to the lack of standardized evaluation benchmarks for nonlinguistic wearable data. These models also have substantial data and computational demands, as well as the risk of false-positive or hallucinated outputs in health-related settings. Future research should therefore examine explainability, validity, reliability, and whether LLM-based approaches provide meaningful advantages over more conventional models in wearable-based stress prediction.</p></sec><sec id="s4-8"><title>Data Quality and Validity Considerations </title><p>Our review highlights several recurring data quality and validity challenges in wearable-based stress studies conducted in naturalistic settings. In particular, field-collected datasets are often shaped by study design constraints, resulting in nonuniform data distributions, limited sample sizes, short observation periods, and skewed participant characteristics (eg, gender, age, and other demographic attributes). These factors collectively constrain population representativeness and limit the external validity and generalizability of reported findings. In contrast, studies that rely on secondary or publicly available large datasets typically benefit from longer monitoring durations and more heterogeneous participant pools. While such datasets may offer improved statistical power and demographic coverage, they also introduce trade-offs, including reduced control over data collection protocols, labeling procedures, and contextual fidelity. Taken together, these differences underscore the importance of explicitly reporting dataset characteristics and carefully considering how data provenance and study design choices shape model validity and interpretability in real-world stress modeling. In our proposed model solution, we explicitly include sections on dataset characteristics to encourage robust study design, as well as transparency and standardization in reporting.</p></sec><sec id="s4-9"><title>Limitations</title><p>Limitations of this scoping review include the challenge of ensuring that all relevant studies have been included and none have been overlooked due to language constraints, as well as the exclusion of unpublished or ongoing studies. Additionally, we focused on a limited range of databases, which restricts the scope of the studies retrieved. The search criteria and query string used may also limit the outcomes of the scoping review. Rather than implementing full parallel double-screening of all records, we adopted a variant of the screening approach in which one researcher conducted the initial screening, followed by a review by a second reviewer. As a result, interrater reliability metrics, such as Cohen &#x03BA; or percentage agreement, were not calculated. The absence of formal double-screening and corresponding reliability statistics may limit the reproducibility of the screening process.</p></sec><sec id="s4-10"><title>Conclusion</title><p>Wearable devices offer valuable opportunities to measure and understand how stress manifests in real-world settings. ML and other advanced computational and statistical methods have increasingly been adopted to develop models that detect, characterize, and understand stress and related psychological constructs from physiological signals captured by wearables, as well as from self-reported data collected at increasingly fine temporal granularity using methods such as EMA. In this focused review, we examined recent studies that apply ML and advanced computational methods to model stress using data from wearable devices in the wild. This review provides an in-depth assessment of current modeling practices and reported performance in this domain and adds to existing reviews by focusing specifically on problem formulation, detailed analysis of model performance, and methodological rigor. In particular, we highlight key problem formulation decisions, such as input and output window configuration, and make the distinction between nowcasting and forecasting tasks, and conduct a detailed analysis of model performance across varied modeling approaches. Through this analysis, we observe a few general patterns of model performance as a function of modeling approaches. We also identify important methodological issues, such as potential data leakage, which may threaten the validity of reported results and inflate reported model performance. We also noted a lack of standardization in datasets, task definitions, and reporting practices, making cross-study comparisons challenging and potentially slowing progress in wearable-based stress detection research. In response to these findings, we propose a model card framework to guide modeling decisions, experimental design, and reporting practices, with the goal of promoting methodological rigor, enhancing transparency, and reproducibility. We hope this review serves as a roadmap for new researchers entering the field, while also offering a framework for researchers working on similar problems to share findings and collaboratively advance the field as a community.</p></sec></sec></body><back><ack><p>Generative AI tools were not used to draft or write the manuscript. The authors reviewed and approved all content and remain fully responsible for the accuracy, originality, and integrity of the manuscript.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AB</term><def><p>AdaBoost</p></def></def-item><def-item><term id="abb2">AUC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb3">BACC</term><def><p>balanced accuracy</p></def></def-item><def-item><term id="abb4">BI-LSTM</term><def><p>bidirectional long short-term memory network</p></def></def-item><def-item><term id="abb5">CNN</term><def><p>convolutional neural network</p></def></def-item><def-item><term id="abb6">EMA</term><def><p>ecological momentary assessment</p></def></def-item><def-item><term id="abb7">FFN</term><def><p>feedforward neural network</p></def></def-item><def-item><term id="abb8">GB</term><def><p>gradient boosting</p></def></def-item><def-item><term id="abb9">GMM</term><def><p>Gaussian mixture model</p></def></def-item><def-item><term id="abb10">HF</term><def><p>high frequency</p></def></def-item><def-item><term id="abb11">IRB</term><def><p>institutional review board</p></def></def-item><def-item><term id="abb12">IRF</term><def><p>impulse response function</p></def></def-item><def-item><term id="abb13">KNN</term><def><p>k-nearest neighbors</p></def></def-item><def-item><term id="abb14">LF</term><def><p>low frequency</p></def></def-item><def-item><term id="abb15">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb16">LR</term><def><p>logistic regression</p></def></def-item><def-item><term id="abb17">LSTM</term><def><p>long short-term memory</p></def></def-item><def-item><term id="abb18">MCC</term><def><p>Matthews correlation coefficient</p></def></def-item><def-item><term id="abb19">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb20">MSE</term><def><p>mean-squared error</p></def></def-item><def-item><term id="abb21">PANAS</term><def><p>Positive and Negative Affect Schedule</p></def></def-item><def-item><term id="abb22">PCA</term><def><p>principal component analysis</p></def></def-item><def-item><term id="abb23">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb24">PRISMA-ScR</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews</p></def></def-item><def-item><term id="abb25">PSS</term><def><p>Perceived Stress Scale</p></def></def-item><def-item><term id="abb26">RF</term><def><p>random forest</p></def></def-item><def-item><term id="abb27">RNN</term><def><p>recurrent neural network</p></def></def-item><def-item><term id="abb28">S-STAI</term><def><p>Short State-Trait Anxiety Inventory</p></def></def-item><def-item><term id="abb29">SNN</term><def><p>Siamese neural network</p></def></def-item><def-item><term id="abb30">SOM</term><def><p>self-organizing map</p></def></def-item><def-item><term id="abb31">STAI</term><def><p>State-Trait Anxiety Inventory</p></def></def-item><def-item><term id="abb32">STAI</term><def><p>State-Trait Anxiety Inventory</p></def></def-item><def-item><term id="abb33">SVM</term><def><p>support vector machine</p></def></def-item><def-item><term id="abb34">VAR</term><def><p>vector autoregression</p></def></def-item><def-item><term id="abb35">XGBoost</term><def><p>eXtreme Gradient Boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Razavi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ziyadidegan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sasangohar</surname><given-names>F</given-names> </name></person-group><article-title>Machine learning techniques for prediction of stress-related mental disorders: a scoping review</article-title><source>Proc Hum Factors Ergon Soc Annu Meet</source><year>2022</year><month>09</month><volume>66</volume><issue>1</issue><fpage>300</fpage><lpage>304</lpage><pub-id pub-id-type="doi">10.1177/1071181322661298</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Janicki&#x2010;deverts</surname><given-names>D</given-names> </name></person-group><article-title>Who&#x2019;s stressed? Distributions of psychological stress in the United States in probability samples from 1983, 2006, and 2009</article-title><source>J Appl Soc Pyschol</source><year>2012</year><month>06</month><volume>42</volume><issue>6</issue><fpage>1320</fpage><lpage>1334</lpage><pub-id pub-id-type="doi">10.1111/j.1559-1816.2012.00900.x</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bolpagni</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pardini</surname><given-names>S</given-names> </name><name name-style="western"><surname>Dianti</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gabrielli</surname><given-names>S</given-names> </name></person-group><article-title>Personalized stress detection using biosignals from wearables: a scoping review</article-title><source>Sensors (Basel)</source><year>2024</year><month>05</month><day>18</day><volume>24</volume><issue>10</issue><fpage>3221</fpage><pub-id pub-id-type="doi">10.3390/s24103221</pub-id><pub-id pub-id-type="medline">38794074</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Iqbal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Howse</surname><given-names>J</given-names> </name><name name-style="western"><surname>Banu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The effects of stress on health</article-title><source>Psychiatr Ann</source><year>2024</year><month>10</month><volume>54</volume><issue>10</issue><fpage>e272</fpage><lpage>e276</lpage><pub-id pub-id-type="doi">10.3928/00485713-20241004-02</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dimsdale</surname><given-names>JE</given-names> </name></person-group><article-title>Psychological stress and cardiovascular disease</article-title><source>J Am Coll Cardiol</source><year>2008</year><month>04</month><day>1</day><volume>51</volume><issue>13</issue><fpage>1237</fpage><lpage>1246</lpage><pub-id pub-id-type="doi">10.1016/j.jacc.2007.12.024</pub-id><pub-id pub-id-type="medline">18371552</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Genet</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Siemer</surname><given-names>M</given-names> </name></person-group><article-title>Rumination moderates the effects of daily events on negative mood: results from a diary study</article-title><source>Emotion</source><year>2012</year><month>12</month><volume>12</volume><issue>6</issue><fpage>1329</fpage><lpage>1339</lpage><pub-id pub-id-type="doi">10.1037/a0028070</pub-id><pub-id pub-id-type="medline">22775130</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wijsman</surname><given-names>J</given-names> </name><name name-style="western"><surname>Grundlehner</surname><given-names>B</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hermens</surname><given-names>H</given-names> </name><name name-style="western"><surname>Penders</surname><given-names>J</given-names> </name></person-group><article-title>Towards mental stress detection using wearable physiological sensors</article-title><source>Annu Int Conf IEEE Eng Med Biol Soc</source><year>2011</year><volume>2011</volume><fpage>1798</fpage><lpage>1801</lpage><pub-id pub-id-type="doi">10.1109/IEMBS.2011.6090512</pub-id><pub-id pub-id-type="medline">22254677</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dobson</surname><given-names>R</given-names> </name><name name-style="western"><surname>Li</surname><given-names>LL</given-names> </name><name name-style="western"><surname>Garner</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tane</surname><given-names>T</given-names> </name><name name-style="western"><surname>McCool</surname><given-names>J</given-names> </name><name name-style="western"><surname>Whittaker</surname><given-names>R</given-names> </name></person-group><article-title>The use of sensors to detect anxiety for in-the-moment intervention: scoping review</article-title><source>JMIR Ment Health</source><year>2023</year><month>02</month><day>2</day><volume>10</volume><issue>1</issue><fpage>e42611</fpage><pub-id pub-id-type="doi">10.2196/42611</pub-id><pub-id pub-id-type="medline">36729590</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Scalise</surname><given-names>L</given-names> </name><name name-style="western"><surname>Cosoli</surname><given-names>G</given-names> </name></person-group><article-title>Wearables for health and fitness: measurement characteristics and accuracy</article-title><conf-name>2018 IEEE International Instrumentation and Measurement Technology Conference (I2MTC)</conf-name><conf-date>May 14-17, 2018</conf-date><conf-loc>Houston, TX, USA</conf-loc><fpage>1</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1109/I2MTC.2018.8409635</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bakker</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pechenizkiy</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sidorova</surname><given-names>N</given-names> </name></person-group><article-title>What&#x2019;s your current stress level? detection of stress patterns from GSR sensor data</article-title><year>2011</year><conf-name>2011 IEEE 11th International Conference on Data Mining Workshops</conf-name><conf-date>Dec 11-14, 2011</conf-date><conf-loc>Vancouver, Canada</conf-loc><fpage>573</fpage><lpage>580</lpage><pub-id pub-id-type="doi">10.1109/ICDMW.2011.178</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Akmandor</surname><given-names>AO</given-names> </name><name name-style="western"><surname>Jha</surname><given-names>NK</given-names> </name></person-group><article-title>Keep the stress away with SODA: stress detection and alleviation system</article-title><source>IEEE Trans Multi-Scale Comp Syst</source><year>2017</year><volume>3</volume><issue>4</issue><fpage>269</fpage><lpage>282</lpage><pub-id pub-id-type="doi">10.1109/TMSCS.2017.2703613</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>FT</given-names> </name><name name-style="western"><surname>Kuo</surname><given-names>C</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>HT</given-names> </name><name name-style="western"><surname>Buthpitiya</surname><given-names>S</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>P</given-names> </name><name name-style="western"><surname>Griss</surname><given-names>M</given-names> </name></person-group><article-title>Activity-aware mental stress detection using physiological sensors</article-title><source>Mobile Computing, Applications, and Services</source><year>2012</year><publisher-name>Springer</publisher-name><fpage>282</fpage><lpage>301</lpage><pub-id pub-id-type="doi">10.1007/978-3-642-29336-8_16</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Can</surname><given-names>YS</given-names> </name><name name-style="western"><surname>Arnrich</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ersoy</surname><given-names>C</given-names> </name></person-group><article-title>Stress detection in daily life scenarios using smart phones and wearable sensors: a survey</article-title><source>J Biomed Inform</source><year>2019</year><month>04</month><volume>92</volume><fpage>103139</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2019.103139</pub-id><pub-id pub-id-type="medline">30825538</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Can</surname><given-names>YS</given-names> </name><name name-style="western"><surname>Chalabianloo</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ekiz</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ersoy</surname><given-names>C</given-names> </name></person-group><article-title>Continuous stress detection using wearable sensors in real life: algorithmic programming contest case study</article-title><source>Sensors (Basel)</source><year>2019</year><month>04</month><day>18</day><volume>19</volume><issue>8</issue><fpage>1849</fpage><pub-id pub-id-type="doi">10.3390/s19081849</pub-id><pub-id pub-id-type="medline">31003456</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gonz&#x00E1;lez Ram&#x00ED;rez</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Garc&#x00ED;a V&#x00E1;zquez</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Rodr&#x00ED;guez</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Padilla-L&#x00F3;pez</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Galindo-Aldana</surname><given-names>GM</given-names> </name><name name-style="western"><surname>Cuevas-Gonz&#x00E1;lez</surname><given-names>D</given-names> </name></person-group><article-title>Wearables for stress management: a scoping review</article-title><source>Healthcare (Basel)</source><year>2023</year><month>08</month><day>22</day><volume>11</volume><issue>17</issue><fpage>2369</fpage><pub-id pub-id-type="doi">10.3390/healthcare11172369</pub-id><pub-id pub-id-type="medline">37685403</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Namvari</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lipoth</surname><given-names>J</given-names> </name><name name-style="western"><surname>Knight</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Photoplethysmography enabled wearable devices and stress detection: a scoping review</article-title><source>J Pers Med</source><year>2022</year><month>10</month><day>31</day><volume>12</volume><issue>11</issue><fpage>1792</fpage><pub-id pub-id-type="doi">10.3390/jpm12111792</pub-id><pub-id pub-id-type="medline">36579537</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Long</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lei</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mao</surname><given-names>P</given-names> </name></person-group><article-title>A scoping review on monitoring mental health using smart wearable devices</article-title><source>Math Biosci Eng</source><year>2022</year><month>05</month><day>27</day><volume>19</volume><issue>8</issue><fpage>7899</fpage><lpage>7919</lpage><pub-id pub-id-type="doi">10.3934/mbe.2022369</pub-id><pub-id pub-id-type="medline">35801449</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pinge</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gad</surname><given-names>V</given-names> </name><name name-style="western"><surname>Jaisighani</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ghosh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sen</surname><given-names>S</given-names> </name></person-group><article-title>Detection and monitoring of stress using wearables: a systematic review</article-title><source>Front Comput Sci</source><year>2024</year><volume>6</volume><fpage>1478851</fpage><pub-id pub-id-type="doi">10.3389/fcomp.2024.1478851</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gjoreski</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lu&#x0161;trek</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gams</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gjoreski</surname><given-names>H</given-names> </name></person-group><article-title>Monitoring stress with a wrist device using context</article-title><source>J Biomed Inform</source><year>2017</year><month>09</month><volume>73</volume><fpage>159</fpage><lpage>170</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2017.08.006</pub-id><pub-id pub-id-type="medline">28803947</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Labbaf</surname><given-names>S</given-names> </name><name name-style="western"><surname>Borelli</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Dutt</surname><given-names>N</given-names> </name><name name-style="western"><surname>Rahmani</surname><given-names>AM</given-names> </name></person-group><article-title>Objective stress monitoring based on wearable sensors in everyday settings</article-title><source>J Med Eng Technol</source><year>2020</year><month>05</month><volume>44</volume><issue>4</issue><fpage>177</fpage><lpage>189</lpage><pub-id pub-id-type="doi">10.1080/03091902.2020.1759707</pub-id><pub-id pub-id-type="medline">32589065</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gaballah</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tiwari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Falk</surname><given-names>TH</given-names> </name></person-group><article-title>Context-aware speech stress detection in hospital workers using bi-LSTM classifiers</article-title><conf-name>ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</conf-name><conf-date>Jun 6-12, 2021</conf-date><conf-loc>Toronto, ON, Canada</conf-loc><fpage>8348</fpage><lpage>8352</lpage><pub-id pub-id-type="doi">10.1109/ICASSP39728.2021.9414666</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Robles-Granda</surname><given-names>P</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Jointly predicting job performance, personality, cognitive ability, affect, and well-being</article-title><source>IEEE Comput Intell Mag</source><year>2021</year><volume>16</volume><issue>2</issue><fpage>46</fpage><lpage>61</lpage><pub-id pub-id-type="doi">10.1109/MCI.2021.3061877</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mitchell</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zaldivar</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Model cards for model reporting</article-title><conf-name>FAT* &#x2019;19: Proceedings of the Conference on Fairness, Accountability, and Transparency</conf-name><conf-date>Jan 29-31, 2019</conf-date><conf-loc>Atlanta, GA, USA</conf-loc><fpage>220</fpage><lpage>229</lpage><pub-id pub-id-type="doi">10.1145/3287560.3287596</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arksey</surname><given-names>H</given-names> </name><name name-style="western"><surname>O&#x2019;Malley</surname><given-names>L</given-names> </name></person-group><article-title>Scoping studies: towards a methodological framework</article-title><source>Int J Soc Res Methodol</source><year>2005</year><month>02</month><volume>8</volume><issue>1</issue><fpage>19</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1080/1364557032000119616</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tricco</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Lillie</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zarin</surname><given-names>W</given-names> </name><etal/></person-group><article-title>PRISMA Extension for Scoping Reviews (PRISMA-ScR): checklist and explanation</article-title><source>Ann Intern Med</source><year>2018</year><month>10</month><day>2</day><volume>169</volume><issue>7</issue><fpage>467</fpage><lpage>473</lpage><pub-id pub-id-type="doi">10.7326/M18-0850</pub-id><pub-id pub-id-type="medline">30178033</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mundnich</surname><given-names>K</given-names> </name><name name-style="western"><surname>Booth</surname><given-names>BM</given-names> </name><name name-style="western"><surname>L&#x2019;Hommedieu</surname><given-names>M</given-names> </name><etal/></person-group><article-title>TILES-2018, a longitudinal physiologic and behavioral data set of hospital workers</article-title><source>Sci Data</source><year>2020</year><month>10</month><day>16</day><volume>7</volume><issue>1</issue><fpage>354</fpage><pub-id pub-id-type="doi">10.1038/s41597-020-00655-3</pub-id><pub-id pub-id-type="medline">33067468</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yfantidou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Karagianni</surname><given-names>C</given-names> </name><name name-style="western"><surname>Efstathiou</surname><given-names>S</given-names> </name><etal/></person-group><article-title>LifeSnaps, a 4-month multi-modal dataset capturing unobtrusive snapshots of our lives in the wild</article-title><source>Sci Data</source><year>2022</year><month>10</month><day>31</day><volume>9</volume><issue>1</issue><fpage>663</fpage><pub-id pub-id-type="doi">10.1038/s41597-022-01764-x</pub-id><pub-id pub-id-type="medline">36316345</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mattingly</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Gregg</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Audia</surname><given-names>P</given-names> </name><etal/></person-group><article-title>The Tesserae project: large-scale, longitudinal, in situ multimodal sensing of information workers</article-title><conf-name>CHI EA &#x2019;19: Extended Abstracts of the 2019 CHI Conference on Human Factors in Computing Systems</conf-name><conf-date>May 4-9, 2019</conf-date><conf-loc>Glasgow, Scotland, UK</conf-loc><fpage>1</fpage><lpage>8</lpage><pub-id pub-id-type="doi">10.1145/3290607.3299041</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Smets</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rios Velazquez</surname><given-names>E</given-names> </name><name name-style="western"><surname>Schiavone</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Large-scale wearable data reveal digital phenotypes for daily-life stress detection</article-title><source>NPJ Digit Med</source><year>2018</year><volume>1</volume><issue>1</issue><fpage>67</fpage><pub-id pub-id-type="doi">10.1038/s41746-018-0074-9</pub-id><pub-id pub-id-type="medline">31304344</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Boateng</surname><given-names>G</given-names> </name><name name-style="western"><surname>Kotz</surname><given-names>D</given-names> </name></person-group><article-title>StressAware: an app for real-time stress monitoring on the amulet wearable platform</article-title><year>2016</year><conf-name>2016 IEEE MIT Undergraduate Research Technology Conference (URTC)</conf-name><conf-date>Nov 4-6, 2016</conf-date><conf-loc>Cambridge, MA</conf-loc><fpage>1</fpage><lpage>4</lpage><pub-id pub-id-type="doi">10.1109/URTC.2016.8284068</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tervonen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Puttonen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sillanp&#x00E4;&#x00E4;</surname><given-names>MJ</given-names> </name><etal/></person-group><article-title>Personalized mental stress detection with self-organizing map: from laboratory to the field</article-title><source>Comput Biol Med</source><year>2020</year><month>09</month><volume>124</volume><fpage>103935</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2020.103935</pub-id><pub-id pub-id-type="medline">32771674</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bavaresco</surname><given-names>R</given-names> </name><name name-style="western"><surname>Barbosa</surname><given-names>J</given-names> </name><name name-style="western"><surname>Vianna</surname><given-names>H</given-names> </name><name name-style="western"><surname>B&#x00FC;ttenbender</surname><given-names>P</given-names> </name><name name-style="western"><surname>Dias</surname><given-names>L</given-names> </name></person-group><article-title>Design and evaluation of a context-aware model based on psychophysiology</article-title><source>Comput Methods Programs Biomed</source><year>2020</year><month>06</month><volume>189</volume><fpage>105299</fpage><pub-id pub-id-type="doi">10.1016/j.cmpb.2019.105299</pub-id><pub-id pub-id-type="medline">31935581</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Vries</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Pennings</surname><given-names>HJM</given-names> </name><name name-style="western"><surname>van der Schans</surname><given-names>CP</given-names> </name><name name-style="western"><surname>Sanderman</surname><given-names>R</given-names> </name><name name-style="western"><surname>Oldenhuis</surname><given-names>HKE</given-names> </name><name name-style="western"><surname>Kamphuis</surname><given-names>W</given-names> </name></person-group><article-title>Wearable-measured sleep and resting heart rate variability as an outcome of and predictor for subjective stress measures: a multiple N-of-1 observational study</article-title><source>Sensors (Basel)</source><year>2022</year><month>12</month><day>28</day><volume>23</volume><issue>1</issue><fpage>332</fpage><pub-id pub-id-type="doi">10.3390/s23010332</pub-id><pub-id pub-id-type="medline">36616929</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Vries</surname><given-names>H</given-names> </name><name name-style="western"><surname>Oldenhuis</surname><given-names>H</given-names> </name><name name-style="western"><surname>van der Schans</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sanderman</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kamphuis</surname><given-names>W</given-names> </name></person-group><article-title>Does wearable-measured heart rate variability during sleep predict perceived morning mental and physical fitness?</article-title><source>Appl Psychophysiol Biofeedback</source><year>2023</year><month>06</month><volume>48</volume><issue>2</issue><fpage>247</fpage><lpage>257</lpage><pub-id pub-id-type="doi">10.1007/s10484-022-09578-8</pub-id><pub-id pub-id-type="medline">36622531</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tump</surname><given-names>D</given-names> </name><name name-style="western"><surname>Narayan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Verbiest</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Stressors and destressors in working from home based on context and physiology from self-reports and smartwatch measurements: international observational study trial</article-title><source>JMIR Form Res</source><year>2022</year><month>11</month><day>10</day><volume>6</volume><issue>11</issue><fpage>e38562</fpage><pub-id pub-id-type="doi">10.2196/38562</pub-id><pub-id pub-id-type="medline">36265030</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mishra</surname><given-names>V</given-names> </name><name name-style="western"><surname>Pope</surname><given-names>G</given-names> </name><name name-style="western"><surname>Lord</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Continuous detection of physiological stress with commodity hardware</article-title><source>ACM Trans Comput Healthc</source><year>2020</year><month>04</month><volume>1</volume><issue>2</issue><fpage>1</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.1145/3361562</pub-id><pub-id pub-id-type="medline">32832933</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmidt</surname><given-names>P</given-names> </name><name name-style="western"><surname>D&#x00FC;richen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Reiss</surname><given-names>A</given-names> </name><name name-style="western"><surname>Laerhoven</surname><given-names>K</given-names> </name><name name-style="western"><surname>Pl&#x00F6;tz</surname><given-names>T</given-names> </name></person-group><article-title>Multi-target affect detection in the wild: an exploratory study</article-title><source>ISWC &#x2019;19: Proceedings of the 2019 ACM International Symposium on Wearable Computers</source><year>2019</year><fpage>211</fpage><lpage>219</lpage><pub-id pub-id-type="doi">10.1145/3341163.3347741</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Booth</surname><given-names>BM</given-names> </name><name name-style="western"><surname>Vrzakova</surname><given-names>H</given-names> </name><name name-style="western"><surname>Mattingly</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Martinez</surname><given-names>GJ</given-names> </name><name name-style="western"><surname>Faust</surname><given-names>L</given-names> </name><name name-style="western"><surname>D&#x2019;Mello</surname><given-names>SK</given-names> </name></person-group><article-title>Toward robust stress prediction in the age of wearables: modeling perceived stress in a longitudinal study with information workers</article-title><source>IEEE Trans Affective Comput</source><year>2022</year><volume>13</volume><issue>4</issue><fpage>2201</fpage><lpage>2217</lpage><pub-id pub-id-type="doi">10.1109/TAFFC.2022.3188006</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Martinez</surname><given-names>GJ</given-names> </name><name name-style="western"><surname>Grover</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mattingly</surname><given-names>SM</given-names> </name><etal/></person-group><article-title>Alignment between heart rate variability from fitness trackers and perceived stress: perspectives from a large-scale in situ longitudinal study of information workers</article-title><source>JMIR Hum Factors</source><year>2022</year><month>08</month><day>4</day><volume>9</volume><issue>3</issue><fpage>e33754</fpage><pub-id pub-id-type="doi">10.2196/33754</pub-id><pub-id pub-id-type="medline">35925662</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hadjiantonis</surname><given-names>G</given-names> </name><name name-style="western"><surname>Paromita</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mundnich</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Dynamical systems modeling of day-to-day signal-based patterns of emotional self-regulation and stress spillover in highly-demanding health professions</article-title><source>Annu Int Conf IEEE Eng Med Biol Soc</source><year>2020</year><month>07</month><volume>2020</volume><fpage>284</fpage><lpage>287</lpage><pub-id pub-id-type="doi">10.1109/EMBC44109.2020.9175604</pub-id><pub-id pub-id-type="medline">33017984</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ravuri</surname><given-names>V</given-names> </name><name name-style="western"><surname>Paromita</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mundnich</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Group-specific models of healthcare workers&#x2019; well-being using iterative participant clustering</article-title><conf-name>2020 Second International Conference on Transdisciplinary AI (TransAI)</conf-name><conf-date>Sep 21-23, 2020</conf-date><conf-loc>Irvine, CA, USA</conf-loc><fpage>115</fpage><lpage>118</lpage><pub-id pub-id-type="doi">10.1109/TransAI49837.2020.00026</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tiwari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Falk</surname><given-names>TH</given-names> </name></person-group><article-title>New measures of heart rate variability based on subband tachogram complexity and spectral characteristics for improved stress and anxiety monitoring in highly ecological settings</article-title><source>Front Signal Process</source><year>2021</year><volume>1</volume><fpage>737881</fpage><pub-id pub-id-type="doi">10.3389/frsip.2021.737881</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paromita</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mundnich</surname><given-names>K</given-names> </name><name name-style="western"><surname>Nadarajan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Booth</surname><given-names>BM</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Chaspari</surname><given-names>T</given-names> </name></person-group><article-title>Modeling inter-individual differences in ambulatory-based multimodal signals via metric learning: a case study of personalized well-being estimation of healthcare workers</article-title><source>Front Digit Health</source><year>2023</year><volume>5</volume><fpage>1195795</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2023.1195795</pub-id><pub-id pub-id-type="medline">37363272</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stojchevska</surname><given-names>M</given-names> </name><name name-style="western"><surname>Steenwinckel</surname><given-names>B</given-names> </name><name name-style="western"><surname>Van Der Donckt</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Assessing the added value of context during stress detection from wearable data</article-title><source>BMC Med Inform Decis Mak</source><year>2022</year><month>10</month><day>15</day><volume>22</volume><issue>1</issue><fpage>268</fpage><pub-id pub-id-type="doi">10.1186/s12911-022-02010-5</pub-id><pub-id pub-id-type="medline">36243691</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Parousidou</surname><given-names>V</given-names> </name><name name-style="western"><surname>Yfantidou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Karagianni</surname><given-names>C</given-names> </name><name name-style="western"><surname>Vakali</surname><given-names>A</given-names> </name></person-group><article-title>Stress beats: a continuum of learning methods for personalized stress detection</article-title><conf-name>2023 IEEE International Conference on Web Intelligence and Intelligent Agent Technology (WI-IAT)</conf-name><conf-date>Oct 26-29, 2023</conf-date><conf-loc>Venice, Italy</conf-loc><fpage>40</fpage><lpage>47</lpage><pub-id pub-id-type="doi">10.1109/WI-IAT59888.2023.00012</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sridhar</surname><given-names>K</given-names> </name><name name-style="western"><surname>Vaessen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Myin-Germeys</surname><given-names>I</given-names> </name><name name-style="western"><surname>Sano</surname><given-names>A</given-names> </name></person-group><article-title>More to less (M2L): enhanced health recognition in the wild with reduced modality of wearable sensors</article-title><conf-name>2022 44th Annual International Conference of the IEEE Engineering in Medicine &#x0026; Biology Society (EMBC)</conf-name><conf-date>Jul 11-15, 2022</conf-date><conf-loc>Glasgow, Scotland, United Kingdom</conf-loc><fpage>3253</fpage><lpage>3256</lpage><pub-id pub-id-type="doi">10.1109/EMBC48229.2022.9871472</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Paraschou</surname><given-names>E</given-names> </name><name name-style="western"><surname>Yfantidou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Vakali</surname><given-names>A</given-names> </name></person-group><article-title>UnStressMe: explainable stress analytics and self-tracking data visualizations</article-title><conf-name>2023 IEEE International Conference on Pervasive Computing and Communications Workshops and other Affiliated Events (PerCom Workshops)</conf-name><conf-date>Mar 13-17, 2023</conf-date><conf-loc>Atlanta, GA, USA</conf-loc><fpage>340</fpage><lpage>342</lpage><pub-id pub-id-type="doi">10.1109/PerComWorkshops56833.2023.10150274</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sano</surname><given-names>A</given-names> </name></person-group><article-title>Semi-supervised learning and data augmentation in wearable-based momentary stress detection in the wild</article-title><source>Proc ACM Interact Mob Wearable Ubiquitous Technol</source><year>2023</year><month>06</month><day>12</day><volume>7</volume><issue>2</issue><fpage>1</fpage><lpage>23</lpage><pub-id pub-id-type="doi">10.1145/3596246</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>T</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>S</given-names> </name></person-group><article-title>Exploring workplace behaviors through speaking patterns using large-scale multimodal wearable recordings: a study of healthcare providers</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 18, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2212.09090</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Saylam</surname><given-names>B</given-names> </name><name name-style="western"><surname>&#x0130;ncel</surname><given-names>&#x00D6;D</given-names> </name></person-group><article-title>Quantifying digital biomarkers for well-being: stress, anxiety, positive and negative affect via wearable devices and their time-based predictions</article-title><source>Sensors (Basel)</source><year>2023</year><month>11</month><day>5</day><volume>23</volume><issue>21</issue><fpage>8987</fpage><pub-id pub-id-type="doi">10.3390/s23218987</pub-id><pub-id pub-id-type="medline">37960685</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kao</surname><given-names>HT</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hosseinmardi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lerman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ferrara</surname><given-names>E</given-names> </name></person-group><article-title>User-based collaborative filtering mobile health system</article-title><source>Proc ACM Interact Mob Wearable Ubiquitous Technol</source><year>2020</year><month>12</month><day>17</day><volume>4</volume><issue>4</issue><fpage>1</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1145/3432703</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Burghardt</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tavabi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ferrara</surname><given-names>E</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lerman</surname><given-names>K</given-names> </name></person-group><article-title>Having a bad day? detecting the impact of atypical events using wearable sensors</article-title><conf-name>Social, Cultural, and Behavioral Modeling: 14th International Conference, SBP-BRiMS 2021</conf-name><conf-date>Jul 6-9, 2021</conf-date><fpage>257</fpage><lpage>267</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-80387-2_25</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>T</given-names> </name><name name-style="western"><surname>Booth</surname><given-names>BM</given-names> </name><name name-style="western"><surname>Baldwin-Rodr&#x00ED;guez</surname><given-names>B</given-names> </name><name name-style="western"><surname>Osorno</surname><given-names>F</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>S</given-names> </name></person-group><article-title>A multimodal analysis of physical activity, sleep, and work shift in nurses with wearable sensor data</article-title><source>Sci Rep</source><year>2021</year><month>04</month><day>22</day><volume>11</volume><issue>1</issue><fpage>8693</fpage><pub-id pub-id-type="doi">10.1038/s41598-021-87029-w</pub-id><pub-id pub-id-type="medline">33888731</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lerman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ferrara</surname><given-names>E</given-names> </name></person-group><article-title>Zero-shot meta-learning for small-scale data from human subjects</article-title><conf-name>2023 IEEE 11th International Conference on Healthcare Informatics (ICHI)</conf-name><conf-date>Jun 26-29, 2023</conf-date><conf-loc>Houston, TX, USA</conf-loc><fpage>311</fpage><lpage>320</lpage><pub-id pub-id-type="doi">10.1109/ICHI57859.2023.00049</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pimentel</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tiwari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Falk</surname><given-names>TH</given-names> </name></person-group><article-title>Human mental state monitoring in the wild: are we better off with deeper neural networks or improved input features</article-title><year>2021</year><access-date>2026-07-02</access-date><conf-name>CMBEC 44</conf-name><conf-date>May 11, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.cmbes.ca/index.php/proceedings/article/view/941">https://proceedings.cmbes.ca/index.php/proceedings/article/view/941</ext-link></comment></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zanna</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sridhar</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sano</surname><given-names>A</given-names> </name></person-group><article-title>Bias reducing multitask learning on mental health prediction</article-title><conf-name>2022 10th International Conference on Affective Computing and Intelligent Interaction (ACII)</conf-name><conf-date>Oct 17-21, 2022</conf-date><conf-loc>Nara, Japan</conf-loc><fpage>1</fpage><lpage>8</lpage><pub-id pub-id-type="doi">10.1109/ACII55700.2022.9953850</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>N</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name></person-group><article-title>Analyzing and identifying predictable time range for stress prediction based on chaos theory and deep learning</article-title><source>Health Inf Sci Syst</source><year>2024</year><month>12</month><volume>12</volume><issue>1</issue><fpage>16</fpage><pub-id pub-id-type="doi">10.1007/s13755-024-00280-z</pub-id><pub-id pub-id-type="medline">39185396</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Saylam</surname><given-names>B</given-names> </name><name name-style="western"><surname>Durmaz &#x0130;ncel</surname><given-names>&#x00D6;</given-names> </name></person-group><article-title>Extracting digital biomarkers for unobtrusive stress state screening from multimodal wearable data</article-title><conf-name>Smart Technologies for Sustainable and Resilient Ecosystems Edge-IoT SmartGov 2022</conf-name><conf-date>Jun 18, 2023</conf-date><fpage>130</fpage><lpage>151</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-35982-8_10</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>X</given-names> </name><name name-style="western"><surname>McDuff</surname><given-names>D</given-names> </name><name name-style="western"><surname>Breazeal</surname><given-names>C</given-names> </name><name name-style="western"><surname>Park</surname><given-names>HW</given-names> </name></person-group><article-title>Health-LLM: large language models for health prediction via wearable sensor data</article-title><source>arXiv</source><comment>Preprint posted online on 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2401.06866</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garcez</surname><given-names>A d</given-names> </name><name name-style="western"><surname>Lamb</surname><given-names>LC</given-names> </name></person-group><article-title>Neurosymbolic AI: the 3rd wave</article-title><source>Artif Intell Rev</source><year>2023</year><month>11</month><volume>56</volume><issue>11</issue><fpage>12387</fpage><lpage>12406</lpage><pub-id pub-id-type="doi">10.1007/s10462-023-10448-w</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Monroe</surname><given-names>D</given-names> </name></person-group><article-title>Neurosymbolic AI</article-title><source>Commun ACM</source><year>2022</year><month>10</month><volume>65</volume><issue>10</issue><fpage>11</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.1145/3554918</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Search strategy and study selection for PubMed (PRISMA-ScR Item 8).</p><media xlink:href="mhealth_v14i1e76632_app1.docx" xlink:title="DOCX File, 2105 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Dataset characteristics (XLSX file, 9 KB).</p><media xlink:href="mhealth_v14i1e76632_app2.xlsx" xlink:title="XLSX File, 8 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Machine learning framework (XLSX file, 16 KB).</p><media xlink:href="mhealth_v14i1e76632_app3.xlsx" xlink:title="XLSX File, 16 KB"/></supplementary-material><supplementary-material id="app4"><label>Checklist 1</label><p>PRISMA-ScR checklist.</p><media xlink:href="mhealth_v14i1e76632_app4.docx" xlink:title="DOCX File, 47 KB"/></supplementary-material></app-group></back></article>