<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Mhealth Uhealth</journal-id><journal-id journal-id-type="publisher-id">mhealth</journal-id><journal-id journal-id-type="index">13</journal-id><journal-title>JMIR mHealth and uHealth</journal-title><abbrev-journal-title>JMIR Mhealth Uhealth</abbrev-journal-title><issn pub-type="epub">2291-5222</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e90970</article-id><article-id pub-id-type="doi">10.2196/90970</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Incremental Value of Smartphone Sensing for Monitoring Momentary Affect Intensity in Adults Using Transformer-Based Models: Observational Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Zhu</surname><given-names>Yiqin</given-names></name><degrees>MS, MA</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Yuyi</given-names></name><degrees>MPH</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Thompson</surname><given-names>Renee J</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Psychological and Brain Sciences, Washington University in St. Louis</institution><addr-line>1 Brookings Drive, CB 1125</addr-line><addr-line>St. Louis</addr-line><addr-line>MO</addr-line><country>United States</country></aff><aff id="aff2"><institution>Division of Computational and Data Sciences, Washington University in St Louis</institution><addr-line>1 Brookings Drive</addr-line><addr-line>St. Louis</addr-line><addr-line>MO</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Buis</surname><given-names>Lorraine</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Trabassi</surname><given-names>Dante</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Shin</surname><given-names>Daun</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Yiqin Zhu, MS, MA, Department of Psychological and Brain Sciences, Washington University in St. Louis, 1 Brookings Drive, CB 1125, St. Louis, MO, 63130, United States, +1-314-935-3502; <email>yiqin@wustl.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>28</day><month>9</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e90970</elocation-id><history><date date-type="received"><day>07</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>14</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>18</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Yiqin Zhu, Yuyi Yang, Renee J Thompson. Originally published in JMIR mHealth and uHealth (<ext-link ext-link-type="uri" xlink:href="https://mhealth.jmir.org">https://mhealth.jmir.org</ext-link>), 28.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR mHealth and uHealth, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://mhealth.jmir.org/">https://mhealth.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://mhealth.jmir.org/2026/1/e90970"/><abstract><sec><title>Background</title><p>Ubiquitous smartphone access and statistical advances offer opportunities to continuously track affect intensity, which is central to various psychological processes and behaviors. Research demonstrated the potential of personalized predictions of momentary negative affect (NA) and positive affect (PA) using passive sensing. However, studies typically incorporated all available data sources without differentiating their added value, nor did they investigate whether refining location features with self-reported semantic location (eg, workplaces) improved personalized predictions.</p></sec><sec><title>Objective</title><p>We evaluated three specific aims: (1) how different combinations of data sources improved performance compared to personalized baseline models, (2) whether model predictions differed across passive data aggregation timescales, and (3) whether incorporating self-reported semantic locations improved model predictions.</p></sec><sec sec-type="methods"><title>Methods</title><p>Adults (final n=133) completed a 14-day ecological momentary assessment (EMA) protocol reporting emotional experiences 5 times daily alongside smartphone sensing. Testing data (n=532 EMAs) used the last 4 surveys for each individual, with the remaining used for training and validation (n=6805 [NA]/6800 [PA] EMAs). We evaluated whether combinations of personalization, passive sensing, and affect history improved baseline prediction, and how full-information temporal fusion transformers (TFTs) performed across 6 timescales (1, 3, 6, 12, 24, and 48 hours), with or without self-reported semantic location features.</p></sec><sec sec-type="results"><title>Results</title><p>The baseline model, using each individual&#x2019;s mean affect in the training set, demonstrated moderate predictive performance for NA (mean absolute error [MAE]=0.66, 95% CI 0.60&#x2010;0.73; <italic>R</italic>&#x00B2;=40.2%) and PA (MAE=0.71, 95% CI 0.65&#x2010;0.78; <italic>R</italic>&#x00B2;=36.1%). Full-information TFTs improved NA prediction (MAE=0.62, 95% CI 0.56&#x2010;0.69; <italic>R&#x00B2;</italic>=45.2%; &#x0394;MAE=&#x2212;0.04, 95% CI &#x2212;0.06 to &#x2212;0.01; <italic>P</italic> values &#x2264;.004; Cohen <italic>d</italic>=&#x2212;0.27) but not PA prediction (MAE=0.70, 95% CI 0.64&#x2010;0.78; <italic>R</italic>&#x00B2;=32.5%; &#x0394;MAE=&#x2212;0.01, 95% CI -0.03 to 0.02; <italic>P</italic> values &#x003E;.10; Cohen <italic>d</italic>=&#x2212;0.04). No pairwise timescale comparison survived false discovery rate (FDR) correction (NA: <italic>P</italic><sub>FDR</sub>=.05-.98; PA: <italic>P</italic><sub>FDR</sub>=.08-.99). Adding self-reported locations did not improve NA prediction (&#x0394;MAE=0.02, 95% CI &#x2212;0.01 to 0.04; <italic>P</italic> values &#x003E;.20; Cohen <italic>d</italic>=0.11) or PA prediction (&#x0394;MAE=0.01, 95% CI &#x2212;0.01 to 0.03; <italic>P</italic> values&#x003E;.33; Cohen <italic>d</italic>=0.07). However, incorporating self-reported semantic locations changed the composition and relative ranking of important inputs, with these changes varying across NA and PA and between past and future inputs.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Incorporating smartphone features provided a modest and significant improvement in momentary NA prediction, but not PA prediction. Model performance did not vary across passive data aggregation timescales. While adding self-report semantic locations did not improve prediction accuracy, it changed variable-importance patterns and may provide additional context for interpreting digital behavioral markers. Future personalized predictions should incorporate person-mean affect as an essential benchmark. These findings support passive smartphone sensing as a valuable supplement to, rather than a replacement for, active EMA.</p></sec></abstract><kwd-group><kwd>transformer-based models</kwd><kwd>affect</kwd><kwd>smartphone sensing</kwd><kwd>ubiquitous computing</kwd><kwd>affective computing</kwd><kwd>digital phenotyping</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Modern technologies already enable us to unobtrusively and accurately track sleep, physical activity, or health essentials such as heart rate and blood pressure, but what about affect intensities? Like the continuous tracking of human behaviors indicative of health status, continuous tracking of affect intensities holds broad research and health implications, given the central role of affect intensity in various outcomes (eg, attention [<xref ref-type="bibr" rid="ref1">1</xref>], memory scope [<xref ref-type="bibr" rid="ref2">2</xref>], and decision-making [<xref ref-type="bibr" rid="ref3">3</xref>]) and its defining role in the diagnostic criteria of many mental disorders [<xref ref-type="bibr" rid="ref4">4</xref>]. In evaluations for mood and anxiety disorders, for example, diagnostic impression relies on individuals&#x2019; retrospective reports of their affect intensity patterns compared to their own baseline patterns (eg, intense negative affect [NA] or blunted positive affect [PA]) over an extended period in the past (eg, the past several months or years). However, retrospective reports are affected by recall bias, especially in clinical samples [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Assessing the intensity of one&#x2019;s subjective feelings from moment to moment reduces bias but requires constant attention that results in burden and missingness (eg, 30% missingness for data collection less than a month [<xref ref-type="bibr" rid="ref7">7</xref>]). These limitations, bias in recalling experiences in the past, and the burden of subjective momentary assessment could be overcome by personalized models to passively and continuously compute affect intensities. Research suggested the potential of such personalized prediction using data from smartphone sensors [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. However, existing studies combined all data sources in models without differentiating each source&#x2019;s contribution, particularly between an individual&#x2019;s own affective history and passively sensed behaviors. As a result, it remains unclear whether passive sensing meaningfully improves prediction beyond what could already be achieved from a person&#x2019;s own affective history (eg, person-mean affect). The overarching aim of this study was to parse out variances of different data sources from personalized prediction models. In addition, we used linear mixed-effects models and post hoc interpretability measures (ie, attention weights) from machine learning (ML) to generate hypotheses about relations between affect and smartphone-tracked behaviors.</p></sec><sec id="s1-2"><title>Monitoring and Understanding Momentary Affect Intensity Through Smartphones</title><p>Cheap and wide access to smartphones (91% ownership in the United States [<xref ref-type="bibr" rid="ref11">11</xref>]) provides unprecedented opportunities not only for passive monitoring of momentary affect intensity but also for understanding ubiquitous human-smartphone interactions. Five studies incorporated smartphone sensor data and modeled momentary affect intensities. Three of the studies dichotomized affective intensity as outcomes and achieved modest prediction with data from smartphones, smartwatches, and smartrings [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. Research showed that traditional AI algorithms could make moderate predictions on whether an individual&#x2019;s daily affect intensities [<xref ref-type="bibr" rid="ref9">9</xref>] or momentary affect intensities [<xref ref-type="bibr" rid="ref10">10</xref>] were above a certain threshold, and make just-surpassing-chance predictions of the presence of affect in unseen individuals [<xref ref-type="bibr" rid="ref8">8</xref>]. However, dichotomizing affect intensity loses important granular information, as different levels of intensity of the same type of emotion lead to different behavioral outcomes. For example, strong feelings of anger might lead to physical violence, whereas weak feelings of anger might lead to verbal profanity or no overt behavior. Therefore, affect intensity is, by nature, continuous and, therefore, motivates approaches that preserve variation across a continuous scale.</p><p>Research that continuously modeled momentary affect included predictors that require individuals&#x2019; active engagement [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>] and only focused on certain NA without modeling PA. One study predicted momentary depressed feelings with smartwatches and self-reported items [<xref ref-type="bibr" rid="ref13">13</xref>]. Both momentary anxiety and depressed affect were reported at the same time, and anxiety was the most important predictor of momentary depressed affect in the models, above smartwatch predictors, for 11 out of 14 participants. The other study collected heart rate metrics (ie, participants pressed their finger against the rear camera for 30 seconds hourly), along with smartphone sensors, to predict momentary depressed mood [<xref ref-type="bibr" rid="ref12">12</xref>]. Importantly, these models combined all sources and did not test against simple baselines of a person&#x2019;s mean affect. Thus, it remains unclear how much smartphone sensing specifically adds to personalized prediction. Finally, it is equally important to test personalized PA prediction using smartphone sensing. Theoretical models [<xref ref-type="bibr" rid="ref14">14</xref>] and clinical trials [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>] have highlighted the critical role of raising PA in reducing depression, and PA is conceptually independent of NA [<xref ref-type="bibr" rid="ref17">17</xref>] and influenced by different biopsychosocial predictors from NA [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>Smartphones are both a data-collection tool people carry and an object to interact with regularly, but few studies have examined associations between momentary affect intensity and smartphone-tracked behaviors. People engage in social interactions (eg, making phone calls) and interact with phone screens for multiple purposes, including using social media. Revealing relations between everyday affective experiences and behaviors tracked by smartphones could inform intervention strategies to aid emotion regulation. For example, if total screen use time in the prior hour is an important predictor of feeling worse among other smartphone-tracked behaviors (eg, total distance traveled), individuals could practice limiting their screen usage. One challenge for elucidating affect&#x2013;smartphone-based behavior associations (and probably for most psychological research) is the large number of smartphone-tracked behaviors and the dependencies between them. For example, when individuals are traveling (as collected through GPS), they may make fewer phone calls and have fewer screen interactions. On the other hand, when individuals remain at home for an extended time, their likelihood of unlocking their screens may be higher. Thus, smartphone-tracked behaviors can be predictors of each other and work together to shape affective experiences.</p><p>Temporal fusion transformer (TFT) is well-suited to address the challenge brought by multiple and interacting predictors. As a state-of-the-art algorithm for time-series analysis, TFT may predict momentary affect intensity more accurately than traditional AI models while addressing the challenge brought by multiple and interacting predictors, as shown in its performance for other time-series predictions (eg, electricity, traffic, retail, and volatility [<xref ref-type="bibr" rid="ref21">21</xref>]). As a transformer-based model, TFT uses self-attention to capture dependencies across time, while its variable selection networks identify predictors that are most informative for the prediction task [<xref ref-type="bibr" rid="ref21">21</xref>]. Additionally, TFT accommodates 3 types of inputs based on their temporal availability: static inputs (time-invariant inputs), past inputs (predictors before the time point of making predictions), and future inputs (inputs corresponding to prediction time points processed by the decoder [<xref ref-type="bibr" rid="ref21">21</xref>]). TFT&#x2019;s capacity to distinguish the 3 input types is important for predicting momentary affect intensity. Because affect intensity is often viewed as an individual difference characteristic [<xref ref-type="bibr" rid="ref22">22</xref>], adding static inputs may enhance prediction, although there is high intraindividual variability for momentary affect intensity. Additionally, it remained unclear at which lag length momentary affect intensity can be predicted by different smartphone-tracked behaviors. While traditional psychological research testing temporal associations usually adopts one lagging length (eg, lagging one assessment schedule), TFT could integrate sensing features derived across multiple temporal windows, allowing information from both proximal and more distal timescales to contribute to prediction.</p></sec><sec id="s1-3"><title>Timescales and Semantic Locations</title><p>Passive sensing research showed that the timescales to aggregate passive sensing data may influence the interpretation of results on psychological well-being [<xref ref-type="bibr" rid="ref23">23</xref>]. Research on predicting momentary affect with passive sensors has used different timescales [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>], and there is a lack of empirical tests comparing the influence of timescales on model performance and predictor importance. For affect intensity, the same behavioral indicator from smartphone sensors with different timescales could have different clinical implications. Intuitively, staying stationary for most of the time in the past hour might be less influential on momentary affect intensities than staying stationary for most of the time in the past 48 hours. Thus, how TFTs performed differently across timescales of aggregating the smartphone sensor data warrants empirical investigation. The timescales might also change the attention that TFTs put on the mobility markers (as collected by GPS) and social interaction markers (as collected by phone calls). Since social interactions are acute events or stressors [<xref ref-type="bibr" rid="ref24">24</xref>], proxies of social interactions tracked by smartphones may be important predictors for NA and PA intensities when they are aggregated in smaller timescales. By contrast, physical mobility patterns in shorter timescales may be influenced by random factors (eg, weather), whereas they may indicate behavioral routines when aggregated in larger timescales. Therefore, mobility patterns tracked by smartphones may be more important when aggregated on larger timescales.</p><p>Contextual features, such as the semantic meaning of a location to an individual (eg, where they work, or where they entertain, or where they spend time with important others), might improve both model performance and interpretations. Affective experiences are constructed by subjective evaluations of both internal and external cues [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. A model of situation perception [<xref ref-type="bibr" rid="ref27">27</xref>] proposes three situational cues: (1) persons and interactions (who?); (2) objects, events, and activities (what?); and (3) spatial location (where?). Physical locations recorded through GPS can provide information about &#x201C;where&#x201D; but not &#x201C;what,&#x201D; namely, the activities or the motivations for activities. Semantic meaning of a location may offer such information. For example, one may assume the major activities that individuals engage in are work in their self-reported workplaces, or that the major activities that they engage in are recreational in their self-reported leisure places. In prior studies, usually one semantic location feature (time spent at home) was included [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Collecting self-report locations from individuals at baseline may improve the coding accuracy of semantic locations and help create new location features, thereby potentially improving model performance. Additionally, semantic locations could inform intervention strategies, as they may elicit more awareness of what patients are engaging in than where they are. Overall, the role of semantic locations in predicting momentary affect warrants further investigation.</p></sec><sec id="s1-4"><title>Study Aims and Hypotheses</title><p>This study collected data from a sample of unselected adults who completed an ecological momentary assessment (EMA) protocol measuring self-report emotional experiences alongside continuous smartphone behavioral monitoring. For aim 1, we evaluated the incremental predictive value of contributions of personalization (participant identity and affective history) versus smartphone sensing data in momentary affect prediction, by comparing TFT and other ML models that vary in their usage of these data sources against a person-mean affect baseline. We hypothesized that the full TFT model incorporating all data sources would achieve prediction performance for unseen momentary NA and PA above the person-mean-affect baseline (hypothesis 1). For aim 2, we examined how feature extraction across 6 aggregation timescales (1, 3, 6, 12, 24, and 48 hours preceding the report) affected TFT&#x2019;s performance. We explored the optimal aggregate windows (exploratory aim 1) and identified top behavioral predictors across timescales. We hypothesized that social interaction (eg, call logs) and screen activity would demonstrate greater feature importance at proximal timescales (hypothesis 2a), whereas mobility markers might be more important at longer timescales (hypothesis 2b). For aim 3, we examined the impact of integrating participant-annotated semantic location features (eg, work and home of significant other). We hypothesized that refining semantic location features with participant self-reports would improve model performance (hypothesis 3a) and shift variable importance rankings (hypothesis 3b). Because attention weights reflect predictive relevance rather than causal mechanisms, findings from hypotheses 2a, 2b, and 3b are framed as exploratory and hypothesis-generating.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Participants</title><p>A total of 179 adults from the greater area of St. Louis, Missouri, United States, participated in a study to understand daily emotional experiences [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Individuals were eligible if they had a smartphone and at least one active social media profile. They were ineligible if they had any contraindications for peripheral physiological assessment (eg, pregnant and with an implanted cardiac device). Participants with less than 40 valid EMAs (70 possible surveys) on NA or PA (n=46) were excluded from further analysis. As a result, 133 out of 179 (74.3% of original participants) were retained in final analysis. Although there are no recommendations or requirements on the number of time points within each individual, in general, more time points are better. The time points ranged from 90 to 252 in the original TFT paper [<xref ref-type="bibr" rid="ref21">21</xref>]. Considering the sample size to be retained, we decided on the threshold of 40 EMAs such that the rate of sample size to be retained in this study (74.3%) was comparable to or higher than existing studies that predict momentary affect intensity (71.6% [<xref ref-type="bibr" rid="ref8">8</xref>]; 35% [<xref ref-type="bibr" rid="ref9">9</xref>]).</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>The study was approved by the Washington University Institutional Review Board (number 202209063) and was performed in line with the standards of the 1964 Declaration of Helsinki. Informed consent and assent were collected from all participants. To protect participant privacy, all data were deidentified before analysis and stored on secure, password-protected servers. Participants were compensated US $135 for completing the study. There was a US $15 bonus if participants completed 80% (56/70) surveys or more. Thus, participants can receive up to US $150 depending on their survey completion rates.</p></sec><sec id="s2-3"><title>Study Design and Procedure</title><p>Eligibility was determined via a phone screen. Participants were scheduled for an in-person laboratory session. During the laboratory session, participants completed a semistructured EMA tutorial, including a practice EMA survey, administered by a staff member or an undergraduate research assistant. The tutorial included helping participants download and install the SEMA<sup>3</sup> app (developed and hosted by the Melbourne eResearch Group at the University of Melbourne [<xref ref-type="bibr" rid="ref32">32</xref>]) for EMA surveys and the AWARE ([<xref ref-type="bibr" rid="ref33">33</xref>]; University of Oulu) apps for smartphone sensing. EMA surveys occurred during a 15-hour window of the participants&#x2019; choice. Participants were surveyed 5 times a day for 14 days, starting the day after the laboratory session. At the laboratory session, participants also reported anticipated locations they frequently visited (eg, office and home) and their addresses in the next 14 days. The AWARE framework [<xref ref-type="bibr" rid="ref33">33</xref>] is an open-source mobile app to unobtrusively record location coordinates, screen interactions (ie, when the screen status changed to on or off and locked or unlocked), call logs for incoming, outgoing, and missed calls, and Bluetooth connections. The AWARE started to collect deidentified data unobtrusively from the smartphone sensors after installation. By default, location coordinates are sampled once per 180 seconds, and Bluetooth is sampled once per 60 seconds. Calls and screen usage are event-based sensor streams. Bluetooth data were not included in analyses due to high missingness. A total of 156 out of 179 (87.2%) participants had no Bluetooth scans recorded in the hour preceding any EMA prompt, and a total of 103 out of 179 (57.5%) participants had no Bluetooth data at all.</p><p>The original sample completed a total of 8690 out of 12,530 (69.4%) EMA surveys. On average, individuals completed 49 out of 70 (70%; SD 15.1) EMA surveys. The completion rate was higher in the final analytic sample, which on average completed 56 out of 70 (80%) EMAs because of the selection criteria (see the &#x201C;Participants&#x201D; section). Participants could receive up to US $135 for completing the study and a bonus of US $15 if their EMA completion was more than 80%, resulting in a possible compensation of $150.</p></sec><sec id="s2-4"><title>Measures</title><sec id="s2-4-1"><title>EMA: Momentary Affect Intensity</title><p>In each survey, participants were asked to report on how they felt in the hour preceding the EMA survey. They indicated the extent to which they felt 4 PA and 4 NA using a 7-point Likert scale (1=not at all; 7=extremely). Items were chosen to represent a 2-dimensional affective circumplex (ie, valence and arousal). The PA items were &#x201C;During the last hour, I felt CONTENT/CALM/HAPPY/ENTHUSIASTIC,&#x201D; and the NA items were &#x201C;During the last hour, I felt SAD/SLUGGISH/WORRIED/FRUSTRATED.&#x201D; In the final analytic sample, both NA and PA showed acceptable within-person reliability (NA: <italic>&#x03C9;</italic><sub>within</sub>=0.64, PA: <italic>&#x03C9;</italic><sub>within</sub>=0.69) and great to excellent between-person reliability (NA: <italic>&#x03C9;</italic><sub>between</sub>=0.88, PA: <italic>&#x03C9;</italic><sub>between</sub>=0.90).</p></sec><sec id="s2-4-2"><title>Measures Derived From Passive Sensors</title><sec id="s2-4-2-1"><title>Overview</title><p><xref ref-type="fig" rid="figure1">Figure 1</xref> illustrates the pipeline of data collection, cleaning, and analysis.</p><p><xref ref-type="table" rid="table1">Table 1</xref> presents a list of extracted features. We computed features from 3 smartphone sensors (location coordinates, screen interactions<italic>,</italic> and call logs) because of their potential to predict depression severity [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>] or predict dichotomized momentary affect intensity [<xref ref-type="bibr" rid="ref10">10</xref>].</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Data collection, cleaning, and analysis pipeline. EMA: ecological momentary assessment; ElasticNet: elastic net regression; MAE: mean absolute error; NA: negative affect; N-BEATS: neural basis expansion analysis for interpretable time series forecasting; PA: positive affect; TFT: temporal fusion transformer; XGBoost: extreme gradient boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mhealth_v14i1e90970_fig01.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Predictors derived from smartphone sensors and their sources<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Variable type and sensor source</td><td align="left" valign="top">Variables</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Spatial mobility markers</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPS coordinates [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Location variance</p></list-item><list-item><p>Log of location variance</p></list-item><list-item><p>Total distance traveled</p></list-item><list-item><p>Average speed</p></list-item><list-item><p>Variance in speed</p></list-item><list-item><p>Percentage of time being stationary (stationary: momentary speed &#x003C;1 km/h)</p></list-item><list-item><p>Radius of gyration</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPS coordinates clustered by DBSCAN<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Number of significant places</p></list-item><list-item><p>Number of significant transitions</p></list-item><list-item><p>Percentage of time spent in rarely visited locations (ie, outliers by DBSCAN)</p></list-item><list-item><p>Location entropy</p></list-item><list-item><p>Normalized location entropy</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Clusters labeled by OSM<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> or self-report locations</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Percentage of time spent at home (OSM)</p></list-item><list-item><p>Percentage of time spent at leisure places (OSM)</p></list-item><list-item><p>Percentage of time spent at home (OSM + self-report)</p></list-item><list-item><p>Percentage of time spent at leisure places (OSM + self-report)</p></list-item><list-item><p>Percentage of time spent at workplaces (OSM + self-report)</p></list-item><list-item><p>Percentage of time spent at home of significant others (OSM + self-report)</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Proxies of smartphone social interactions</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Call logs [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Number of incoming calls</p></list-item><list-item><p>Number of outgoing calls</p></list-item><list-item><p>Number of missed calls</p></list-item><list-item><p>Number of correspondents called</p></list-item><list-item><p>Duration of incoming calls (average, maximum, minimum, and total duration)</p></list-item><list-item><p>Duration of outgoing calls (average, maximum, minimum, and total duration)</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Screen interactions</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Screen status [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Number of unlocked episodes</p></list-item><list-item><p>Average, maximum, minimum, and total screen-on duration</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Time variables</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Timestamp</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Relative time in hours since the first self-report EMA<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> survey</p></list-item><list-item><p>Weekend or weekday (1=weekend; 0=weekday)</p></list-item><list-item><p>Time of day (night: midnight to 6 AM; morning: 6 AM to noon; afternoon: noon to 6 PM; evening: 6 PM to midnight)</p></list-item><list-item><p>Prompt (the number of the self-report EMA survey)</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Each predictor was computed for each timescale. For example, for the timescale of 3 hours, the location variance of the 3 hours preceding each survey was created.</p></fn><fn id="table1fn2"><p><sup>b</sup>DBSCAN: Density-Based Spatial Clustering of Application with Noise.</p></fn><fn id="table1fn3"><p><sup>c</sup>OSM: OpenStreetMap.</p></fn><fn id="table1fn4"><p><sup>d</sup>EMA: ecological momentary assessment.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-4-2-2"><title>Location Feature Sets</title><p>Location features are derived from the coordinates of latitude and longitude and the timestamp of each coordinate. We first removed duplicated data points for each individual and then extracted the following location features consistent with research [<xref ref-type="bibr" rid="ref28">28</xref>] that detected depression and predicted its onset from smartphone data: (1) location variance (sum of the variance in latitude and longitude coordinates), (2) log of location variance, (3) total distance traveled, (4) average speed, (5) variance in speed, (6) the percentage of time being stationary (stationary was defined as momentary speed being less than 1 km/h), and (7) radius of gyration. When calculating distances and speed, only adjacent data points whose time difference was greater than 1 second and less than twice the sampling rate (ie, 180 seconds) were retained.</p><p>Then each individual&#x2019;s location coordinates were clustered using Density-Based Spatial Clustering of Applications with Noise (DBSCAN [<xref ref-type="bibr" rid="ref37">37</xref>]), a clustering algorithm for large spatial databases with noise. Consistent with prior work [<xref ref-type="bibr" rid="ref28">28</xref>], based on identified clusters, we extracted the (1) number of significant places, (2) the number of location transitions, (3) the percentage of time spent in insignificant or rarely visited locations (ie, outliers identified by DBSCAN), (4) location entropy (higher location entropy occurs when time is spent evenly across significant places), and (5) normalized location entropy across significant places (following previous definitions and operationalizations [<xref ref-type="bibr" rid="ref35">35</xref>]).</p><p>Consistent with prior work [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>], we assumed the place most visited by the participant late at night (between midnight and 6 AM) was their home location. Specifically, the location coordinates of the cluster most frequently used during all nights were assumed to be the participant&#x2019;s home location center. Then, based on the coordinates of longitude and latitude for each individual&#x2019;s core clusters, we labeled their type of location through the OpenStreetMap Nominatim API&#x2019;s geocoding [<xref ref-type="bibr" rid="ref38">38</xref>]. Locations labeled as &#x201C;highway,&#x201D; &#x201C;aeroway,&#x201D; &#x201C;tourism,&#x201D; &#x201C;leisure,&#x201D; and &#x201C;shop&#x201D; by OpenStreetMap were categorized as leisure and recreation. For each of the location labels (including the labels in the next paragraph), when participants&#x2019; GPS coordinates were within 100 meters of one location center cluster, their location at that timestamp was given the label of the location center cluster. For example, a participant was assumed to be at home if they were within 100 m of the home location. With these 2 labels, we extracted two location features: (1) the percentage of time spent at home and (2) the percentage of time spent at leisure and recreation.</p><p>We tested whether adding more semantic location features could improve model prediction and interpretation through their indication of possible activities. These sets of location features were extracted based on participants&#x2019; self-reported locations, which included categories of personal home, home of significant others (eg, parents, partner, and so on), workplace, and places for leisure and recreation. Based on their self-report addresses, the locations of the previously coded home and leisure were corrected, resulting in two extracted features: (1) the percentage of time spent at home (self-report) and (2) the percentage of time spent at leisure and recreation (self-report). Additionally, with 2 additional addresses (workplace and home of significant others), we extracted (3) the percentage of time spent at work (self-report) and (4) the percentage of time spent with significant others (self-report).</p><p>We created 6 sets of location features based on 6 temporal resolutions (ie, 1, 3, 6, 12, 24, and 48 hours) preceding each momentary affect (ie, NA and PA) report. Take the location variance feature as an example. We created 6 location variance features for a given momentary affect report&#x2014;the location variance during 1, 3, 6, 12, 24, and 48 hours preceding the report.</p></sec><sec id="s2-4-2-3"><title>Calls Feature Sets</title><p>Call features were calculated using the smartphone&#x2019;s call logs. Consistent with prior work [<xref ref-type="bibr" rid="ref28">28</xref>], we extracted the following six features: (1) the number of all incoming and outgoing calls, (2) the number of missed calls and the number of correspondents overall, and (3) the duration of all incoming and outgoing calls (in terms of their average, maximum, minimum, and total duration). As with the location features, we created 6 sets of call features based on 6 temporal resolutions (ie, 1, 3, 6, 12, 24, and 48 hours preceding each momentary affect report).</p></sec><sec id="s2-4-2-4"><title>Screen Feature Sets</title><p>Screen features were calculated using the smartphone&#x2019;s screen status sensor, which recorded screen status (ie, on or off) and its timestamp once the screen switched from one status (eg, on) to the other (eg, off). Consistent with prior work [<xref ref-type="bibr" rid="ref28">28</xref>], we extracted the following five phone usage features: (1) the number of unlock episodes, (2) the average, (3) sum, (4) maximum, and (5) minimum duration of the screen-on time. Similar to the location features, we created 6 sets of screen features based on 6 temporal resolutions (ie, 1, 3, 6, 12, 24, and 48 hours preceding each momentary affect report).</p></sec><sec id="s2-4-2-5"><title>Time Variables</title><p>We included several time variables based on the time of the EMA survey completion as predictors of momentary affect because of emotion&#x2019;s time-varying nature. We computed one continuous time variable, (1) relative time (relative time in hours since the first EMA survey), and 2 categorical variables: (2) weekend or not (1=yes; 0=no [ie, weekday]) and (3) time of day (midnight to 6 AM: night, 6 AM to noon: morning, noon to 6 PM: afternoon, 6 PM to midnight: evening; dummy coded). Finally, one count variable, (4) prompt (the EMA survey number), was used as the time indicator required by the time-series deep learning algorithm.</p></sec></sec></sec><sec id="s2-5"><title>Data Preprocessing</title><p>Before imputation, we screened the passive-sensing predictors for near-constant and redundant features: a predictor was dropped if (1) it was near constant, defined as a single value accounted for at least 99% of its nonmissing observations, (2) it was highly correlated with another one, defined as an absolute Spearman correlation over 0.95. We dropped the one with higher missingness in the highly correlated pair. We then handled missing data and extreme values of the smartphone sensor features for each timescale by performing 6 imputations and then 6 winsorizations. For each of the 6 feature sets, we performed single imputation for the missing values in predictors using an iterative imputer (IterativeImputer) that uses the chained equations (MICE [Multivariate Imputation by Chained Equations]) algorithm from the scikit-learn library [<xref ref-type="bibr" rid="ref39">39</xref>] in Python. Instead of performing imputations for each smartphone sensor (locations, calls, and screen), we performed imputations across smartphone sensor data simultaneously, in which IterativeImputer could use interactions between different sensor data to impute missing values. Importantly, we excluded outcome variables (ie, NA and PA) from imputations to avoid data leakage and overfitting of the prediction models. Finally, we winsorized extreme values in the predictors using the quantile-based method, where we replaced values above the 99th percentile with the 99th-percentile value and replaced values smaller than the 1st percentile with the 1st-percentile value.</p><p>To avoid data leakage, the feature screening, feature selection, imputation, and winsorization were all performed in the training set only. The feature list, imputation model, and clipping bounds from the training set were then applied to the validation and test sets. For the TFT, the same principle applied to the participant-level target normalizer used to construct the NA and PA center and scale, which was fit on the training set and reused, rather than refit, in the validation and test sets. A similar training-only-fitted preprocessing pipeline (including standardization for the ElasticNet [elastic net regression]) was also performed for approaches using the ElasticNet and XGBoost (extreme gradient boosting).</p></sec><sec id="s2-6"><title>Training and Testing Split</title><p>We used a temporally ordered split approach, splitting observations within each individual into three subsets: (1) a test set for final model evaluation, comprising the last 4 consecutive EMA surveys (~1 day); (2) a validation set for hyperparameter tuning, early stopping, and, for model selection between ElasticNet, XGBoost, and N-BEATS (neural basis expansion analysis for interpretable time series forecasting; described below), comprising the 4 consecutive EMA surveys immediately preceding the test window; (3) a training set with all earlier entries (~85% of all EMA surveys). Because the validation and test windows followed the training window, this setup prevents temporal leakage and evaluates short-term within-person generalization. This approach resulted in 6273/6269 EMAs in the training set, 532/531 in the validation set, and 532/532 in the test set for NA and PA, respectively. The training set of 6273 (NA) and 6269 (PA) EMAs allowed 6672 (NA) and 6669 (PA) rolling training sequences (encoder length: 8&#x2010;16, prediction length: 1&#x2010;4; window lengths were allowed to be flexible and can overlap, so they can exceed observation counts), across 133 participants. The small NA and PA differences reflect outcome-specific missingness. Because the sliding window allows the same EMA to be reused as parts of multiple different-length windows (eg, first window being 1&#x2010;11 EMAs predicting 12&#x2010;15 EMAs and second window being 2&#x2010;13 EMAs predicting 14&#x2010;16 EMAs), the number of training sequences to fit the TFT exceeds the number of EMAs in the training set.</p><p>For the TFT, we allowed the encoder length to vary between 8 and 16 time points and the prediction length to vary between 1 and 4 time points. As a result, training sequences consist of different combinations of encoder and decoder lengths (eg, 1&#x2010;11 EMAs predicting the next 2 EMAs). The final TFT configuration used an encoder length of 16 observations because this window spans roughly 3 days of recent data, given the study&#x2019;s sampling schedule of approximately 5 EMAs per day. We considered this interval sufficiently long to capture short-term temporal patterns in affect while minimizing data loss that would occur from requiring longer continuous observation histories. A prediction length of 4 was chosen to forecast affect over approximately the subsequent day, directly aligning with the study&#x2019;s objective of predicting near-term affective dynamics which may ultimately be relevant for timely, adaptive interventions. The final TFT configuration used a maximum encoder length of 16 observations and a maximum prediction length of 4 observations.</p></sec><sec id="s2-7"><title>Statistical Analysis</title><sec id="s2-7-1"><title>Model Evaluations</title><p>Model performance metrics were mean absolute error (MAE), root-mean-square error (RMSE), and coefficient of determination (<italic>R</italic><sup>2</sup>). The MAE is the average of all the absolute values of the differences between the predicted values and the actual values. The RMSE is the square root of the mean of the squared differences between predicted values and actual values. <italic>R</italic><sup>2</sup> represents the proportion of variance explained by the model. When <italic>R<sup>2</sup></italic> equals zero, a model explains no more than the average score. When <italic>R</italic><sup>2</sup> equals 1, a model perfectly explains all the outcome variances. Unlike <italic>R</italic><sup>2</sup>, lower values of MAE and RMSE indicate better performance. Both MAE and RMSE are in the original scale of the outcomes and indicate the extent to which predicted values deviate from the actual values. RMSE is more sensitive to extreme deviations. When MAE or RMSE equals zero, the model perfectly predicts the actual values. When RMSE equals the SD of the outcome variable, the model performs no better than the average score.</p></sec><sec id="s2-7-2"><title>Person-Mean Affect: For Baseline</title><p>In this baseline model, the predicted value for each held-out validation or test observation was set to each participant&#x2019;s mean NA and PA intensity calculated across the training set. This benchmark required no model fitting or hyperparameter tuning and operated without smartphone sensing or temporal information.</p></sec><sec id="s2-7-3"><title>ElasticNet and XGBoost Models: For Nonpersonalized Sensing, Personalized Time, and Personalized Sensing Approaches</title><p>ElasticNet regression [<xref ref-type="bibr" rid="ref40">40</xref>], implemented in scikit-learn, and XGBoost (gradient-boosted trees) [<xref ref-type="bibr" rid="ref41">41</xref>] were used to fit three model variants: (1) nonpersonalized sensing model, (2) personalized time (incorporating participant ID and time variables, without smartphone-sensing features), and (3) personalized sensing (incorporating participant ID, time, and smartphone sensing features, without affect history). Both ElasticNet and XGBoost used a standardized preprocessing pipeline (z-scoring of continuous predictors) prior to model fitting. For ElasticNet, we evaluated 21 hyperparameter combinations crossing 7 regularization strengths (&#x03B1;=0.001, 0.003, 0.01, 0.03, 0.1, 0.3, 1.0) with 3 L1 mixing parameters (l1_ratio=0.1, 0.5, 0.9), each fit for up to 20,000 iterations. For XGBoost, we searched 6 candidate configurations varying the number of trees (200-500), maximum tree depth (2-3), learning rate (0.02&#x2010;0.05), row and column subsampling (0.8&#x2010;1.0), minimum child weight (1-5), and L1/L2 regularization strength. Within each of the 6 sensor aggregation timescales and 2 semantic-location label versions, the hyperparameter configuration with the lowest validation-set RMSE was selected. Across test sets, the specific timescale and semantic label specifications that yielded the lowest test-set MAE were selected and reported as the top-performing model under each.</p></sec><sec id="s2-7-4"><title>N-BEATS Model: For Personalized Affective History Approach</title><p>To evaluate the personalized affect-history approach, we implemented N-BEATS, a neural network for time-series forecasting composed of stacks of fully connected blocks with backward (backcast) and forward (forecast) residual connections. Unlike the TFT, N-BEATS relied exclusively on each participant&#x2019;s own historical affect sequence as input, incorporating no smartphone-sensor features. We used fixed-length sequence windows of 16 encoder time points and 4 prediction time points to match the TFT&#x2019;s maximum window. Outcomes were normalized within participants using a group normalizer with a softplus transformation, and missing time steps were permitted. We set the N-BEATS stack widths to 32 and 64 and the backcast loss ratio to 0, thereby weighting the training loss entirely toward the forecast; all other architecture parameters retained the PyTorch Forecasting defaults. Training used a learning rate of 0.001, weight decay of 0.01, batch size of 64, and gradient clipping at 0.01. We monitored validation loss and retained the checkpoint with the lowest validation loss. Training stopped after 5 consecutive validation checks without an improvement of at least 0.0001, with a maximum of 30 epochs. Training was deterministic and used a random seed of 42. This fixed-window procedure yielded 3746 (NA) and 3741 (PA) windowed training sequences derived from the same underlying training observations described above.</p></sec><sec id="s2-7-5"><title>Temporal Fusion Transformer: For Full-Information and Personalized Sensing Approaches</title><p>To test hypothesis 1, we compared TFTs&#x2019; performance (ie, <italic>R</italic><sup>2</sup>, MAE, and RMSE) with that reported in existing studies [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. For exploratory aim 1, we examined TFTs&#x2019; performance across 6 temporal scales (paired n=133). To test hypotheses 2a and 2b, the distal and proximal effects of different types of smartphone-tracked behaviors, we evaluated the top feature importance weights across timescales across encoder and decoder inputs. Because raw TFT variable importance weights reflect nonlinear transformations, we cross-referenced top predictors with statistically significant effects from within-person linear mixed-effects models (detailed below). We used up to 16 consecutive data points (encoder 8&#x2010;16) of smartphone sensor data (past inputs) to predict the following up to 4 time points (decoder 1&#x2010;4; future inputs) of momentary affect. The important predictors in the past inputs represent more distal factors, whereas future inputs represent more proximal factors in predicting momentary affect. In addition, important predictors in larger timescales and past inputs indicate features with more distal effects, whereas those in smaller timescales and future inputs indicate more proximal or immediate features. For hypothesis 3a, which tested whether self-reported semantic-location features changed performance, we used the paired participant-clustered bootstrap and participant-level Wilcoxon procedures described for aim 1. To test hypothesis 3b, we compared the pattern of important past and future inputs before and after incorporating self-report semantic location features.</p><p>All TFT models were implemented in the PyTorch Lightning framework through Google Colab Pro (NVIDIA A100-SXM4-40GB, 40 GB RAM). Model configurations were: (1) initialized the learning rate at 0.001 and then determined it via a systematic finder. The learning rate suggested by the systematic finder was scaled down by a factor of 4 in model training; (2) set the hidden size to 64 and hidden continuous size of 32 to deal with complex features; (3) added 4 multihead attention heads to capture multiscale temporal dependencies; (4) applied a dropout rate of 0.1 and gradient norm clipping threshold of 0.1 in the network to avoid overfitting; (5) used the Quantile loss function for probabilistic interval predictions; and (6) set the batch size to 10 to balance memory footprint and gradient estimation stability.</p><p>Using the validation set described above, we monitored validation loss after each epoch, retained the checkpoint with the lowest validation loss, and applied early stopping after 5 consecutive epochs without a minimum improvement of 0.0001 (up to 100 epochs). The retained checkpoint was used for all subsequent test-set evaluation. To make personalized predictions, the full-information TFT included 4 static covariates: participant ID, sequence encoder length, NA or PA center (individual&#x2019;s mean affect intensities), and NA or PA scale (individual SD of affect intensities). To isolate variance sources for our aim 1, we specified a separate personalized sensing TFT model by refitting the optimal specification for NA and PA after omitting static affect-history statistics (NA or PA center and NA or PA scale). Test-set performance for this model was compared against the ElasticNet and XGBoost baselines to select the best models under the personalized sensing approach.</p></sec><sec id="s2-7-6"><title>CIs and Significance Testing</title><p>We computed 95% CIs for <italic>R</italic><sup>2</sup><italic>,</italic> RMSE, and MAE using a participant-level bootstrap. To account for the nonindependence of repeated observations within individuals, resampling was performed at the participant level rather than the observation (EMA) level. Specifically, participants (not individual rows) were resampled with replacement for 2000 resamples, and all test-set rows of each selected participant were pooled. Model performance (<italic>R</italic><sup>2</sup>, RMSE, and MAE) was computed for each resample, and the 2.5th and 97.5th percentiles of the resulting bootstrap distribution were the CI.</p><p>To test aim 1, which was to evaluate whether each approach&#x2019;s held-out predictive performance differed from the baseline (person-mean affect), we conducted 2 complementary paired statistical tests applied to each approach&#x2019;s best-performing specification (see above). First, we performed a primary bootstrapping comparison by calculating the MAE difference (&#x0394;MAE=MAE<sub>Approach</sub> &#x2013; MAE<sub>Baseline</sub>) for each EMA in the test set. Using 2000 resamples at the participant level (with a fixed random seed of 42 for reproducibility), we derived a 95% CI and a 2-sided bootstrap <italic>P</italic> value (defined as twice the smaller proportion of resampled &#x0394;MAE values falling at or below zero vs at or above zero). Second, as a supplementary test, we conducted a participant-level paired Wilcoxon signed-rank test comparing paired series of mean per-participant MAE<sub>Approach</sub> and MAE<sub>Baseline</sub> values, avoiding the assumption of independent repeated observations. To remain conservative, an approach&#x2019;s improvement over the baseline was considered statistically significant only when both tests met the significance threshold of &#x03B1;=.05. Effect sizes for these paired comparisons were reported as Cohen <italic>d</italic><sub>z</sub> (mean of participant-level &#x0394;MAE divided by its SD, accompanied by approximate 95% CIs).</p><p>For exploratory aim 1 (evaluating whether TFT performance differed across the 6 temporal aggregation windows), we conducted a nonparametric Friedman test on a participant-by-window matrix containing each participant&#x2019;s mean MAE at each timescale (n=133). When the omnibus test indicated significant differences across windows, we conducted a post hoc analysis using all 15 pairwise Wilcoxon signed-rank tests between timescales, adjusting for multiple comparisons using the Benjamini-Hochberg false discovery rate (FDR) procedure. For hypothesis 3a (testing whether incorporating self-reported semantic-location features improved model performance), we applied the same dual-testing procedure as in aim 1, a paired participant-clustered bootstrap alongside a participant-level paired Wilcoxon signed-rank procedure.</p></sec><sec id="s2-7-7"><title>Linear Mixed-Effect Models</title><p>Since the TFTs extract the importance of a feature but not the direction or the magnitude of associations, we further examined the within-person associations with a series of linear mixed-effects models. Models were estimated using restricted maximum likelihood (REML) via the lmer() function in the <italic>lme4</italic> package in R (R Core Team; accessed via pymer4 in Python). All variables were standardized (z-scored) before analysis to yield standardized coefficients &#x03B2;. All predictors were person-mean centered to only investigate within-individual associations. In each linear mixed-effect model, momentary NA or PA intensities were outcomes and were regressed on one smartphone-tracked predictor (eg, person-mean-centered total distance traveled in the 3 hours preceding the survey). In each model, we included a random intercept to handle between-person baseline differences in affect and included a random slope to allow the relationship between the predictor and outcome to vary across individuals.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Participants Characteristics</title><p>The final analytic sample consisted of 133 adults (mean age 36.4, SD 12.0, age range: 19&#x2010;63 years). Participants&#x2019; gender composition was as follows: 71 out of 133 (53.4%) women, 51 out of 133 (38.3%) men, 3 out of 133 (2.3%) gender diverse, and 8 out of 133 (4.9%) unknown. Their race was 91 out of 133 (68.4%) White, 14 out of 133 (11.8%) Asian, 12 out of 133 (6.9%) Black or African American, 7 out of 133 (5.3%) others, and 9 out of 133 (6.8%) preferred not to say. A total of 9 out of 133 (6.8%) of the analytic sample reported being Latinx or Hispanic.</p></sec><sec id="s3-2"><title>Aim 1: Model Performance in Predicting Momentary NA and PA</title><p><xref ref-type="table" rid="table2">Table 2</xref> presents the best-performing model under different approaches (ie, including different sets of predictors and algorithms). Results showed that models that simply used the mean affect intensity for each person (predictors: ID + training NA or PA) showed moderate performance on unseen test sets (NA: <italic>R</italic>&#x00B2;=40.2% [95% CI 31-48.2], RMSE=0.89 [95% CI 0.80-0.97], MAE=0.66 [95% CI 0.60-0.73]); PA: <italic>R</italic>&#x00B2;=36.1% [95% CI 20.4-48.3], RMSE=0.93 [95% CI 0.84-1.03], MAE=0.71 [95% CI 0.65-0.78]). Nonpersonalized sensing models performed significantly worse than the baseline for both outcomes (both <italic>P</italic> values&#x003C;.001; NA Cohen <italic>d</italic>=0.56; PA Cohen <italic>d</italic>=0.37). Among personalized benchmarks, the ID-and-time ElasticNet was significantly worse for NA (&#x0394;MAE=0.020, 95% CI 0.004-0.036; <italic>P</italic><sub>bootstrap</sub>=.01; <italic>P</italic><sub>Wilcoxon</sub>=.04; Cohen <italic>d</italic>=0.21), whereas the affect-history N-BEATS model did not differ for NA. N-BEATS produced a statistically significant but very small improvement for PA (&#x0394;MAE=&#x2212;0.002, 95% CI &#x2212;0.003 to&#x2212;0.000; <italic>P</italic><sub>bootstrap</sub>=.04; <italic>P</italic><sub>Wilcoxon</sub>=.01; Cohen <italic>d</italic>=&#x2212;0.18). The personalized sensing TFT without Affect Center and Scale did not differ from the baseline for either outcome. The full-information TFT significantly improved NA prediction (&#x0394;MAE=&#x2212;0.040, 95% CI &#x2212;0.065 to&#x2212;0.015; <italic>P</italic><sub>bootstrap</sub>=.001; <italic>P</italic><sub>Wilcoxon</sub>=.004; Cohen <italic>d</italic>=&#x2212;0.27) but not PA prediction (&#x0394;MAE=&#x2212;0.006, 95% CI &#x2212;0.031 to 0.020; <italic>P</italic><sub>bootstrap</sub>=.66; <italic>P</italic><sub>Wilcoxon</sub>=.13; Cohen <italic>d</italic>=&#x2212;0.04). The best full-information models used 6-hour features with self-reported semantic locations for NA and 1-hour features with self-reported semantic locations for PA.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Prediction performances of different model approaches when predicting momentary NA<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> and PA<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> on the test dataset (n=133 individuals; total 532 surveys)<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Approach</td><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Feature sets</td><td align="left" valign="bottom"><italic>R</italic>&#x00B2; (%)<break/>(95% CI)</td><td align="left" valign="bottom">RMSE<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup><break/>(95% CI)</td><td align="left" valign="bottom">MAE<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup><break/>(95% CI)</td><td align="left" valign="bottom">&#x0394;MAE vs baseline<break/>(95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value<break/>(bootstrap)</td><td align="left" valign="bottom"><italic>P</italic> value<break/>(Wilcoxon)</td><td align="left" valign="bottom">Cohen <italic>d</italic></td></tr></thead><tbody><tr><td align="left" valign="top">NA</td><td align="left" valign="top" colspan="9"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Person-mean NA</td><td align="left" valign="top">Mean training NA</td><td align="left" valign="top">No passive sensing</td><td align="left" valign="top">40.2<break/>(31.0 to 48.2)</td><td align="left" valign="top">0.89<break/>(0.80 to 0.97)</td><td align="left" valign="top">0.66<break/>(0.60 to 0.73)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Nonpersonalized sensing</td><td align="left" valign="top">XGBoost<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup></td><td align="left" valign="top">48h+OSM<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup></td><td align="left" valign="top">4.6<break/>(&#x2212;2.3 to 10.5)</td><td align="left" valign="top">1.12<break/>(1.01 to 1.23)</td><td align="left" valign="top">0.89<break/>(0.80 to 0.97)</td><td align="left" valign="top">0.222<break/>(0.158 to 0.295)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.56</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Personalized time</td><td align="left" valign="top">ElasticNet<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td><td align="left" valign="top">No passive sensing</td><td align="left" valign="top">38.2<break/>(29.6 to 45.6)</td><td align="left" valign="top">0.90<break/>(0.81 to 0.99)</td><td align="left" valign="top">0.68<break/>(0.62 to 0.75)</td><td align="left" valign="top">0.020<break/>(0.004 to 0.036)</td><td align="left" valign="top">.01</td><td align="left" valign="top">.04</td><td align="left" valign="top">0.21</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Personalized sensing</td><td align="left" valign="top">TFT<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">6h+OSM+self-report locations</td><td align="left" valign="top">36.6<break/>(26.3 to 45.8)</td><td align="left" valign="top">0.91<break/>(0.82 to 1.01)</td><td align="left" valign="top">0.66<break/>(0.59 to 0.73)</td><td align="left" valign="top">&#x2212;0.002<break/>(&#x2212;0.020 to 0.018)</td><td align="left" valign="top">.86</td><td align="left" valign="top">.89</td><td align="left" valign="top">&#x2212;0.01</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Personalized NA history</td><td align="left" valign="top">N-BEATS<sup><xref ref-type="table-fn" rid="table2fn11">k</xref></sup></td><td align="left" valign="top">No passive sensing</td><td align="left" valign="top">35.5<break/>(25.3 to 44.5)</td><td align="left" valign="top">0.92<break/>(0.82 to 1.02)</td><td align="left" valign="top">0.66<break/>(0.59 to 0.74)</td><td align="left" valign="top">&#x2212;0.001<break/>(&#x2212;0.022 to 0.019)</td><td align="left" valign="top">.93</td><td align="left" valign="top">.88</td><td align="left" valign="top">&#x2212;0.01</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Full-information</td><td align="left" valign="top">TFT</td><td align="left" valign="top">6h+OSM+self-report locations</td><td align="left" valign="top">45.2<break/>(33.8 to 54.3)</td><td align="left" valign="top">0.85<break/>(0.76 to 0.94)</td><td align="left" valign="top">0.62<break/>(0.56 to 0.69)</td><td align="left" valign="top">&#x2212;0.040<break/>(&#x2212;0.065 to &#x2212;0.015)</td><td align="left" valign="top">.001</td><td align="left" valign="top">.004</td><td align="left" valign="top">&#x2212;0.27</td></tr><tr><td align="left" valign="top">PA</td><td align="left" valign="top" colspan="9"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Person-mean PA</td><td align="left" valign="top">Mean training PA</td><td align="left" valign="top">No passive sensing</td><td align="left" valign="top">36.1<break/>(20.4 to 48.3)</td><td align="left" valign="top">0.93<break/>(0.84 to 1.03)</td><td align="left" valign="top">0.71<break/>(0.65 to 0.78)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Nonpersonalized sensing</td><td align="left" valign="top">XGBoost</td><td align="left" valign="top">48h+OSM+self-report locations</td><td align="left" valign="top">10.8<break/>(1.4 to 18.7)</td><td align="left" valign="top">1.10<break/>(1.01 to 1.19)</td><td align="left" valign="top">0.87<break/>(0.80 to 0.95)</td><td align="left" valign="top">0.163<break/>(0.091 to 0.239)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.37</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Personalized time</td><td align="left" valign="top">ElasticNet</td><td align="left" valign="top">No passive sensing</td><td align="left" valign="top">35.8<break/>(20.5 to 47.2)</td><td align="left" valign="top">0.93<break/>(0.85 to 1.03)</td><td align="left" valign="top">0.72<break/>(0.66 to 0.79)</td><td align="left" valign="top">0.009<break/>(&#x2212;0.008 to 0.028)</td><td align="left" valign="top">.34</td><td align="left" valign="top">.46</td><td align="left" valign="top">0.09</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Personalized sensing</td><td align="left" valign="top">TFT</td><td align="left" valign="top">1h+OSM+self-report locations</td><td align="left" valign="top">34.8<break/>(17.7 to 48.2)</td><td align="left" valign="top">0.94<break/>(0.84 to 1.05)</td><td align="left" valign="top">0.70<break/>(0.64 to 0.78)</td><td align="left" valign="top">&#x2212;0.007<break/>(&#x2212;0.029 to 0.016)</td><td align="left" valign="top">.59</td><td align="left" valign="top">.21</td><td align="left" valign="top">&#x2212;0.05</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Personalized PA history</td><td align="left" valign="top">N-BEATS</td><td align="left" valign="top">No passive sensing</td><td align="left" valign="top">35.4<break/>(20.2 to 48.1)</td><td align="left" valign="top">0.93<break/>(0.83 to 1.03)</td><td align="left" valign="top">0.71<break/>(0.65 to 0.77)</td><td align="left" valign="top">&#x2212;0.002<break/>(&#x2212;0.003 to &#x2212;0.000)</td><td align="left" valign="top">.04</td><td align="left" valign="top">.01</td><td align="left" valign="top">&#x2212;0.18</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Full-information</td><td align="left" valign="top">TFT</td><td align="left" valign="top">1h+OSM+self-report locations</td><td align="left" valign="top">32.5<break/>(11.6 to 47.8)</td><td align="left" valign="top">0.96<break/>(0.84 to 1.08)</td><td align="left" valign="top">0.70<break/>(0.63 to 0.78)</td><td align="left" valign="top">&#x2212;0.006<break/>(&#x2212;0.031 to 0.020)</td><td align="left" valign="top">.66</td><td align="left" valign="top">.13</td><td align="left" valign="top">&#x2212;0.04</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>NA: negative affect.</p></fn><fn id="table2fn2"><p><sup>b</sup>PA: positive affect.</p></fn><fn id="table2fn3"><p><sup>c</sup><italic>R</italic>&#x00B2;, RMSE, and MAE are computed on the held-out test set; brackets denote the 95% participant-level bootstrap CI (2000 resamples with replacement). &#x0394;MAE vs Baseline=MAE<sub>approach</sub>- MAE<sub>Person-Mean Baseline</sub>, calculated from a paired participant-level bootstrap (positive values indicate worse performance than baseline, negative values indicate improvement). The <italic>P</italic> (bootstrap) value represents the 2-sided bootstrap <italic>P</italic> value for this difference; <italic>P</italic> value (Wilcoxon) is derived from a participant-level paired Wilcoxon signed-rank test on MAE (n=133 participants rather than 532 observations, to avoid pseudoreplication). Cohen <italic>d</italic> indicates the paired-samples effect size (mean difference divided by the SD of the MAE differences). Bold text indicates a statistically significant improvement or decline relative to the person-mean baseline (requiring both bootstrap and Wilcoxon <italic>P</italic>&#x003C;.05). For approaches incorporating passive sensing feature sets (combinations of time windows and semantic-location label sets), the row displays the best-performing specification based on test-set MAE.</p></fn><fn id="table2fn4"><p><sup>d</sup>RMSE: root-mean-square error.</p></fn><fn id="table2fn5"><p><sup>e</sup>MAE: mean absolute error.</p></fn><fn id="table2fn6"><p><sup>f</sup>Not applicable.</p></fn><fn id="table2fn7"><p><sup>g</sup>XGBoost: extreme gradient boosting. </p></fn><fn id="table2fn8"><p><sup>h</sup>OSM: OpenStreetMap.</p></fn><fn id="table2fn9"><p><sup>i</sup>ElasticNet: elastic net regression.</p></fn><fn id="table2fn10"><p><sup>j</sup>TFT: temporal fusion transformer.</p></fn><fn id="table2fn11"><p><sup>k</sup>N-BEATS: neural basis expansion analysis for interpretable time series forecasting.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Aim 2: Role of Timescales in Model Performance and Variable Importance</title><sec id="s3-3-1"><title>Overview</title><p><xref ref-type="fig" rid="figure2">Figure 2</xref> shows MAE across the 6 aggregation windows. In the figure, shaded bands represent 95% participant-level bootstrap CIs. In-panel statistics report the Friedman omnibus test across all 6 time windows and the range of FDR-corrected pairwise Wilcoxon <italic>P</italic> values across all 15 window-pair comparisons for each semantic-location feature set. For NA, neither the OpenStreetMap-only specification (<italic>&#x03C7;</italic>&#x00B2;<sub>5</sub>=5.93, <italic>P</italic>=.31) nor the OpenStreetMap-plus-self-report specification (<italic>&#x03C7;</italic>&#x00B2;<sub>5</sub>=10.10, <italic>P</italic>=.07) differed significantly across windows. FDR-corrected pairwise <italic>P</italic> values ranged from .05 to .98. For PA, the OpenStreetMap-only omnibus test was nonsignificant (<italic>&#x03C7;</italic>&#x00B2;<sub>5</sub>=2.16, <italic>P</italic>=.83), whereas the OpenStreetMap-plus-self-report omnibus test reached significance (<italic>&#x03C7;</italic>&#x00B2;<sub>5</sub>=11.21, <italic>P</italic>=.05), which was driven by the worst performance using the 48-hour OpenStreetMap-plus-self-report specification. However, none of the PA pairwise comparisons survived FDR correction (<italic>P</italic><sub>FDR</sub>=.08-.99). Thus, no specific pair of aggregation windows showed a robust performance difference.</p></sec><sec id="s3-3-2"><title>Variable Importance Across Timescales</title><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Temporal fusion transformer (TFT) mean absolute error (MAE) across time windows (1, 3, 6, 12, 24, and 48 hours) in predicting momentary negative affect (NA, left) and positive affect (PA, right) intensity, for semantic-location features derived from OpenStreetMap versus OpenStreetMap supplemented with self-report in the test set (n=133 individuals; total 532 surveys). FDR: false discovery rate.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mhealth_v14i1e90970_fig02.png"/></fig><p><xref ref-type="fig" rid="figure3">Figure 3</xref> presents a heatmap of important past (left) versus future inputs (right) identified by the TFTs across time windows for predicting NA (upper 2 panels) and PA (lower 2 panels) using location features from OpenStreetMap (blue names) versus from OpenStreetMap supplemented with self-report. We did not find consistent support for hypothesis 2a. Although behavioral indicators of social interactions (ie, calls) and screen interactions appeared among the important future inputs for both NA and PA, this holds only before supplementing OpenStreetMap with self-reported locations to refine location features. After supplementing OpenStreetMap with self-reported locations to refine location features, percent time at leisure places (self-report) was an important future input for both NA and PA, especially in the 3-hour window (NA 0.16, PA 0.17). Partially supporting hypothesis 2b, partial-mobility features contributed as both past and future inputs, with number of location transitions and variance in speed as notable proximal (future) inputs and radius of gyration and number of significant places as distal (past) inputs. Not supporting hypotheses 2a and 2b, there is no clear pattern in variable importance across time windows (eg, consistently darker or lighter from left to right).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Heatmap of important past (left, panels A, C, E, and G) versus future inputs (right, panels B, D, F, and H) identified by temporal fusion transformer (TFT) across time windows for predicting negative affect (NA; upper panels A-D) and positive affect (PA; lower panels E-H) using location features from OpenStreetMap (blue names) versus from OpenStreetMap supplemented with self-report. EMA: ecological momentary assessment.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mhealth_v14i1e90970_fig03.png"/></fig><p>Outside of hypotheses, the individual&#x2019;s affect history was an important predictor across time windows for both NA and PA (left panels in <xref ref-type="fig" rid="figure3">Figure 3</xref>). Across time windows, a time variable (weekend or weekday) is an important future input and, sometimes, a past input across location feature sets for NA and PA, especially at the longer windows (24-hour as a future input and 12-hour as a past input; <xref ref-type="fig" rid="figure3">Figure 3</xref>).</p></sec></sec><sec id="s3-4"><title>Aim 3: The Role of Self-Report Semantic Locations</title><sec id="s3-4-1"><title>Overview</title><p>Overall, both bootstrapping and Wilcoxon significance tests suggested that adding self-report semantic-location features did not significantly change prediction accuracy for either NA or PA. Specifically, for predicting NA intensity, the best TFT using semantic-location features derived from OpenStreetMap only, which was based on 3-hour features (MAE=0.64), did not perform significantly differently from the TFT using semantic-location features from OpenStreetMap supplemented with self-report locations, which was based on 6-hour features (MAE=0.62; &#x0394;MAE [95% CI]=0.016 [&#x2212;0.009 to 0.040]; <italic>P</italic><sub>bootstrap</sub>=0.21; <italic>P</italic><sub>Wilcoxon</sub>=0.35; Cohen <italic>d</italic> [95% CI]=0.11 [&#x2212;0.06 to 0.28]). Results were similar for PA, such that the best TFT before incorporating self-report locations, which was based on 6-hour features (MAE=0.71), did not perform significantly differently from the best TFT after incorporating self-report locations, which was based on 1-hour features (MAE=0.70; &#x0394;MAE [95% CI]=0.007 [&#x2212;0.012 to 0.026]; <italic>P</italic><sub>bootstrap</sub>=0.44; <italic>P</italic><sub>Wilcoxon</sub>=0.34; Cohen <italic>d</italic> [95% CI]=0.07 [&#x2212;0.10 to 0.24]).</p></sec><sec id="s3-4-2"><title>Variable Importance After Adding Self-Report Semantic Locations</title><p><xref ref-type="fig" rid="figure4">Figure 4</xref> presents the attention weights of past (left) versus future inputs (right) from the best-performing TFT for NA intensity (top) and PA intensity (bottom). Blue bars represent attention weights before adding self-report semantic locations, whereas orange bars represent attention weights after adding self-report semantic locations. Consistent with hypothesis 3b, incorporating self-report semantic locations changed the composition and relative ranking of important inputs, although the pattern of change varied across past and future inputs. The percent time spent at leisure places (self-report) emerged as an important input in the TFT attention weights, particularly at the 3-hour window for both NA and PA (<xref ref-type="fig" rid="figure3">Figure 3</xref>; NA: 0.16, PA: 0.17; see detailed attention weights for each variable in Tables S1 and S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Several GPS features indicating spatial mobility (variance in speed, number of location transitions, and radius of gyration) were identified among important inputs across models.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Attention weights of top-5 past (left; panels A and C) versus future inputs (right; panels B and D) from the best-performing temporal fusion transformer (TFT) in predicting momentary negative affect (NA, top, panels A and B) and positive affect (PA, bottom, panels C and D) intensity, for semantic-location features derived from OpenStreetMap (blue) versus OpenStreetMap supplemented with self-report (orange) in the test set (n=133 individuals; total 532 surveys). EMA: ecological momentary assessment.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="mhealth_v14i1e90970_fig04.png"/></fig></sec></sec><sec id="s3-5"><title>Within-Individual Associations in Linear Mixed-Effects Models</title><p>Tables S1 and S2 present standardized coefficients in linear mixed-effects models for NA and PA, respectively (see them in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Smartphone-tracked behaviors showed weak, if any, bivariate associations with momentary affect intensity (all |&#x03B2;|&#x003C;0.06). After FDR correction across 236 tests per outcome, 5 proximal call features were positively associated with NA: the numbers of incoming calls, outgoing calls, and correspondents in the 1-hour window and the numbers of outgoing calls and correspondents in the 3-hour window (&#x03B2;=0.03-0.06, <italic>P</italic><sub>FDR</sub>=.003-.05). Thirteen features were associated with PA after correction. PA was positively associated with location diversity across 3- to 24-hour windows (log location variance, location entropy, normalized location entropy, and number of significant places; &#x03B2;=0.04-0.06) and negatively associated with time at self-reported workplaces in the 1-, 3-, and 6-hour windows and with the number of correspondents in the 1-hour window (&#x03B2;=&#x2013;0.04 to &#x2013;0.05; <italic>P</italic><sub>FDR</sub>=.002-.03).</p></sec><sec id="s3-6"><title>Sensitivity Analyses</title><p>Three sensitivity analyses supported the robustness of the primary findings. First, lowering the minimum number of valid EMA reports from 40 to 20 retained 167 participants (n=8398 NA EMAs and n=8391 PA EMAs), compared with 133 participants in the primary sample (n=7337 NA EMAs and n=7332 PA EMAs). Predictive performance did not differ significantly for either NA (6-hour, OSM + self-report; &#x0394;MAE=0.04, 95% CI &#x2212;0.10 to 0.20; <italic>P</italic><sub>bootstrap</sub>=.624; <italic>P</italic><sub>Wilcoxon</sub>=.54) or PA (1-hour, OSM + self-report; &#x0394;MAE=0.12, 95% CI &#x2212;0.07 to 0.36; <italic>P</italic><sub>bootstrap</sub>=.26; <italic>P</italic><sub>Wilcoxon</sub>=.32). Second, replacing MICE with median imputation did not significantly change performance for NA (6-hour, OSM + self-report; &#x0394;MAE=0.010, 95% CI &#x2212;0.001 to 0.021; <italic>P</italic><sub>bootstrap</sub>=.07; <italic>P</italic><sub>Wilcoxon</sub>=.21) or PA (1-hour, OSM + self-report; &#x0394;MAE=0.0005, 95% CI &#x2212;0.0046 to 0.0052; <italic>P</italic><sub>bootstrap</sub>=.84; <italic>P</italic><sub>Wilcoxon</sub>=.70). Finally, increasing the maximum encoder length from 16 (minimum 8) to 20 (minimum 12), while retaining a maximum decoder length of 4, did not significantly affect performance for NA (6-hour, OSM + self-report; &#x0394;MAE=0.0008, 95% CI &#x2212;0.0089 to 0.0106; <italic>P</italic><sub>bootstrap</sub>=.89, <italic>P</italic><sub>Wilcoxon</sub>=.86) or PA (1-hour, OSM + self-report; &#x0394;MAE=0.0035, 95% CI &#x2212;0.0080 to 0.0137; <italic>P</italic><sub>bootstrap</sub>=.53; <italic>P</italic><sub>Wilcoxon</sub>=.21).</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study parsed the contributions of EMA and smartphone sensing data to personalized predictions of momentary affect. On held-out EMA observations, smartphone sensing within the full-information TFT added statistically significant and modest predictive value for NA but small, nonsignificant value for PA relative to the person-mean baseline. Person-specific mean affect accounted for a substantial share of predictive performance. Although one PA omnibus test differed across aggregation windows, no pairwise time-window comparison survived FDR correction. Adding self-reported semantic locations did not significantly improve NA or PA prediction. Before these locations were added, mobility markers and smartphone social interactions appeared among important past and future inputs. After they were added, mobility markers appeared more often as past inputs, partially supporting hypothesis 2b, whereas screen and smartphone social interactions showed no consistent proximal or distal pattern, providing no support for hypothesis 2a. Overall, our results suggest modest added value of smartphone sensing over EMA for NA prediction following person-specific calibration.</p></sec><sec id="s4-2"><title>Aim 1: Model Performance in Predicting Momentary NA and PA</title><p>Aim 1 tested how much each data source contributed to personalized prediction and whether the TFT improved on a person-mean baseline. Partially consistent with hypothesis 1, smartphone sensing added significant predictive value for NA (but not PA), empowered by transformer-based models. All prior studies made personalized predictions through incorporating all data sources, including both active and passive inputs [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. Ultimately, the primary promise of digital phenotyping is the possibility of generating accurate predictions of momentary affect intensity using passive inputs alone. This is a goal that has attracted increasing interest given its broad clinical and practical implications. The present findings can be situated within a broader affect-dynamics framework in which momentary affect reflects a combination of relatively stable person-specific tendencies, recent affective trajectories, and immediate contextual inputs. Findings showed that participants&#x2019; mean affect and affective history accounted for a substantial proportion of predictive performance, whereas smartphone sensing provided only modest incremental information. This pattern indicates that smartphone sensing should not yet be viewed as a standalone replacement for active self-report, but rather as a complementary source of information to EMA. Future personalized prediction research should consistently evaluate the person-mean affect intensity as an essential benchmark and explore richer multimodal passive signals that capture dimensions of emotional experience beyond smartphone-tracked behaviors (eg, physiological measures and linguistic features).</p><p>Another motivation of the study was to address the relative neglect of PA in the affective computing literature, particularly given its established importance for psychological well-being and depression treatment. Our findings show that the incremental value of smartphone sensing was significant and modest for NA, but small and nonsignificant for PA. Consequently, high predictive performance for NA (or depressed mood, which is more commonly investigated) should not be assumed to automatically generalize to PA. Current sensing modalities may be differentially informative across valence dimensions. Improving PA prediction may require larger sample sizes or alternative digital markers. Future affective computing research focused on PA is clearly warranted.</p></sec><sec id="s4-3"><title>Aims 2 and 3: The Role of Timescales and Self-Reported Semantic Locations</title><p>Aims 2 and 3 examined aggregation timescales and self-reported semantic locations. Across 6 windows, 3 omnibus tests were nonsignificant and 1 PA test reached statistical significance, but no pairwise comparison survived FDR correction. Aggregation-window choices may therefore be guided partly by practical considerations, such as computational efficiency, pending replication. Adding self-reported semantic locations did not improve predictive performance, so the value of collecting these data appears to lie more in contextual interpretation than in accuracy.</p><p>Three broader patterns emerged from the attention-weight analyses. First, incorporating self-reported semantic locations changed the composition and relative ranking of important inputs despite not improving predictive accuracy, suggesting that contextual information may contribute more to interpretation than to predictive accuracy. These changes did not follow a clear pattern across NA and PA or across past and future inputs. Time at self-reported leisure places appeared as an important input after semantic-location features were added. Second, the TFT prioritized different classes of smartphone-tracked behavior across aggregation windows while maintaining broadly comparable predictions. Third, spatial-mobility features appeared more consistently as past than future inputs, suggesting that they may summarize accumulated behavioral context rather than immediate affective triggers. However, these attention weights indicate model-based predictive relevance, not causal effects, and should be treated more as hypothesis-generating.</p></sec><sec id="s4-4"><title>Implications for Research and Applications</title><p>The affective computing approach to personalized monitoring of affect intensity is still in its infancy, and this study addressed three key methodological questions: (1) how much variance did smartphone sensing add to personalized prediction of momentary affect beyond EMA baselines, (2) does the data aggregation timescale impact predictive performance, and (3) what is the value of integrating self-report semantic locations. First, our framework provides a more rigorous benchmark for evaluating passive sensing for affect monitoring. Future work on personalized predictions should benchmark models against person-specific mean affect, demonstrating incremental predictive value rather than relying solely on absolute accuracy metrics. Second, because predictive performance remained stable across the 6 evaluated timescales, feature engineering decisions regarding aggregation windows can be guided by practical considerations, such as computational efficiency. Third, incorporating contextual self-reports may enhance the interpretability of digital behavioral markers, even when they do not directly improve prediction accuracy. Finally, rather than replacing EMA, passive sensing may reduce the frequency of active assessment while maintaining continuous, individualized monitoring between self-reports.</p></sec><sec id="s4-5"><title>Limitations and Future Directions</title><p>This study has several limitations. First, within-person reliability was modest for NA and PA (<italic>&#x03C9;</italic>=0.64 and 0.69), which constrains achievable predictive accuracy. Future studies should balance more reliable measurement of momentary affect against EMA burden. Second, the feature-importance rankings produced by the TFT reflect associational rather than causal relationships. A predictor may receive high attention weight simply because it is correlated with an unobserved driver or acts as a proxy that reduces prediction error. Consequently, these findings should be interpreted as hypotheses regarding potential mechanisms rather than definitive evidence of causal influence. Experimental manipulation or microrandomized intervention studies are necessary to establish causal pathways.</p><p>In addition, our models were good at predicting momentary affect intensity at time points immediately following the training period (about 2 weeks) for the same individuals. However, it is unclear whether our models can make good predictions days, weeks, or even months after the training data time points. Future work could examine whether algorithms can stop assessing self-report experiences after a certain period and then allow ubiquitous and continuous monitoring. If so, it would also be important to clarify the maximum length of the no-report period.</p><p>The present findings illustrate AI&#x2019;s potential to predict momentary affect in daily life (situational generalizability) by forecasting an individual&#x2019;s future states from their own past data (within-person temporal generalizability). A critical remaining question concerns population generalizability and whether such models can generalize to entirely new individuals. We maintain that, at least for predicting momentary affect intensity, developing personalized models is more feasible than building universal models expected to generalize across individuals. In human-to-human interactions, people can readily recognize distinct emotions in a stranger, but accurately gauging another person&#x2019;s emotional intensity typically requires familiarity. Consistent with this distinction, research evaluating whether passive sensor data could classify discrete momentary affect states (eg, angry or not angry) across new individuals found that traditional ML models performed only slightly above chance [<xref ref-type="bibr" rid="ref8">8</xref>]. Although population-level prediction of affect intensity without a priori calibration remains challenging, achieving it would carry substantial public health implications.</p><p>Third, the present work is limited by the number of modalities in data sources. Additional smartphone sensors collecting other social interaction data (eg, text messages or social media) and audio information (eg, volumes and frequency of meaningful conversations around an individual) could add modalities while maintaining the benefits from the cheap and wide access of smartphones. Future affect computing research could also benefit from a larger dataset with more people and more observations, as other AI fields (eg, large language models) have illustrated that &#x201C;a &#x2018;dumb&#x2019; algorithm with lots and lots of data beats a &#x2018;clever&#x2019; one with modest amounts of data&#x201D; [<xref ref-type="bibr" rid="ref42">42</xref>].</p></sec><sec id="s4-6"><title>Conclusion</title><p>This study highlights the incremental value of smartphone sensing for personalized, short-term prediction of momentary affect beyond simple person-specific affective baselines. Notably, this added predictive value was significant only for NA, while participants&#x2019; mean baseline affect accounted for a substantial proportion of overall predictive performance across all models. No pairwise performance difference emerged across the 6 passive-data aggregation windows. Furthermore, incorporating self-reported semantic locations did not improve prediction accuracy, though it provided more interpretable contextual features for generating hypotheses about affect-behavior relations. Overall, these findings suggest that smartphone sensing is currently better positioned to augment rather than replace EMA. Following an initial period of person-specific calibration, passive sensing may estimate short-term affective fluctuations between self-report prompts and support lower-burden or adaptive EMA designs. Replication in larger, more diverse, and clinical samples remains essential before deploying these models in real-world assessment or intervention use.</p></sec></sec></body><back><ack><p>The authors would like to thank Dr Ellen E Fitzsimmons-Craft, Dr Joshua R Oltmanns, Dr Nathaniel S Eckland, and Dr Jocelyn Lai for their review and feedback on the manuscript. The authors thank Anna Leah Davis and Analise Black for their assistance with data collection.</p><p>Generative AI tools were used to assist with code development, debugging, and refinement, including code for data preprocessing and implementation of the temporal fusion transformer and comparison models. They were also used to support language editing and improve the clarity and organization of selected portions of the manuscript. The tools used included ChatGPT-4o and GPT-5.6 (OpenAI), Claude 3.5 Sonnet and Fable (Anthropic), Gemini 1.5 Pro (Google), and Microsoft Copilot (Microsoft Corporation). All AI-assisted code and text were critically reviewed and revised by the authors. The authors retained full responsibility for the study design, analytic decisions, accuracy of the code, interpretation of the results, and final manuscript.</p></ack><notes><sec><title>Funding</title><p>Funding was provided by the Office of the Provost at Washington University in St. Louis (total amount of US $50,000). The funder had no involvement in the study design, data collection, analysis, interpretation, or the manuscript preparation.</p></sec><sec><title>Data Availability</title><p>The datasets generated or analyzed during this study are not publicly available because they contain sensitive smartphone-sensing and location information, and because public data sharing was not covered by participants&#x2019; consent. However, they may be available from the corresponding author on reasonable request, subject to institutional approval, and an appropriate data-use agreement. The analysis code is available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: YZ, RJT</p><p>Data curation: YZ, YY, RJT</p><p>Formal analysis: YZ, YY</p><p>Funding acquisition: RJT</p><p>Investigation: RJT</p><p>Methodology: YZ, YY</p><p>Project administration: RJT</p><p>Resources: RJT</p><p>Software: YY</p><p>Supervision: RJT</p><p>Validation: YY</p><p>Visualization: YZ</p><p>Writing &#x2013; original draft: YZ</p><p>Writing &#x2013; review &#x0026; editing: YZ, YY, RJT</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">DBSCAN</term><def><p>Density-Based Spatial Clustering of Applications with Noise</p></def></def-item><def-item><term id="abb2">ElasticNet</term><def><p>elastic net regression</p></def></def-item><def-item><term id="abb3">EMA</term><def><p>ecological momentary assessment</p></def></def-item><def-item><term id="abb4">FDR</term><def><p>false discovery rate</p></def></def-item><def-item><term id="abb5">MAE</term><def><p>mean absolute error</p></def></def-item><def-item><term id="abb6">MICE</term><def><p>Multivariate Imputation by Chained Equations</p></def></def-item><def-item><term id="abb7">ML</term><def><p> machine learning</p></def></def-item><def-item><term id="abb8">N-BEATS</term><def><p>neural basis expansion analysis for interpretable time series forecasting</p></def></def-item><def-item><term id="abb9">NA</term><def><p>negative affect</p></def></def-item><def-item><term id="abb10">PA</term><def><p>positive affect</p></def></def-item><def-item><term id="abb11">REML</term><def><p>restricted maximum likelihood</p></def></def-item><def-item><term id="abb12">RMSE</term><def><p>root-mean-square error</p></def></def-item><def-item><term id="abb13">TFT</term><def><p>temporal fusion transformer</p></def></def-item><def-item><term id="abb14">XGBoost</term><def><p>extreme gradient boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Strauss</surname><given-names>GP</given-names> </name><name name-style="western"><surname>Allen</surname><given-names>DN</given-names> </name></person-group><article-title>The experience of positive emotion is associated with the automatic processing of positive emotional words</article-title><source>J Posit Psychol</source><year>2006</year><month>07</month><volume>1</volume><issue>3</issue><fpage>150</fpage><lpage>159</lpage><pub-id pub-id-type="doi">10.1080/17439760600566016</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Threadgill</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Gable</surname><given-names>PA</given-names> </name></person-group><article-title>Negative affect varying in motivational intensity influences scope of memory</article-title><source>Cogn Emot</source><year>2019</year><month>03</month><volume>33</volume><issue>2</issue><fpage>332</fpage><lpage>345</lpage><pub-id pub-id-type="doi">10.1080/02699931.2018.1451306</pub-id><pub-id pub-id-type="medline">29621935</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lerner</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Valdesolo</surname><given-names>P</given-names> </name><name name-style="western"><surname>Kassam</surname><given-names>KS</given-names> </name></person-group><article-title>Emotion and decision making</article-title><source>Annu Rev Psychol</source><year>2015</year><month>01</month><day>3</day><volume>66</volume><issue>1</issue><fpage>799</fpage><lpage>823</lpage><pub-id pub-id-type="doi">10.1146/annurev-psych-010213-115043</pub-id><pub-id pub-id-type="medline">25251484</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="book"><source>Diagnostic and Statistical Manual of Mental Disorders: DSM-5</source><year>2013</year><publisher-name>American Psychiatric Association</publisher-name><pub-id pub-id-type="doi">10.1176/appi.books.9780890425596</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gotlib</surname><given-names>IH</given-names> </name><name name-style="western"><surname>Joormann</surname><given-names>J</given-names> </name></person-group><article-title>Cognition and depression: current status and future directions</article-title><source>Annu Rev Clin Psychol</source><year>2010</year><volume>6</volume><issue>1</issue><fpage>285</fpage><lpage>312</lpage><pub-id pub-id-type="doi">10.1146/annurev.clinpsy.121208.131305</pub-id><pub-id pub-id-type="medline">20192795</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Williams</surname><given-names>JMG</given-names> </name><name name-style="western"><surname>Barnhofer</surname><given-names>T</given-names> </name><name name-style="western"><surname>Crane</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Autobiographical memory specificity and emotional disorder</article-title><source>Psychol Bull</source><year>2007</year><month>01</month><volume>133</volume><issue>1</issue><fpage>122</fpage><lpage>148</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.133.1.122</pub-id><pub-id pub-id-type="medline">17201573</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Drexl</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ralisa</surname><given-names>V</given-names> </name><name name-style="western"><surname>Rosselet-Amoussou</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Readdressing the ongoing challenge of missing data in youth ecological momentary assessment studies: meta-analysis update</article-title><source>J Med Internet Res</source><year>2025</year><month>04</month><day>30</day><volume>27</volume><fpage>e65710</fpage><pub-id pub-id-type="doi">10.2196/65710</pub-id><pub-id pub-id-type="medline">40305088</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Akre-Bhide</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>ZD</given-names> </name><name name-style="western"><surname>Welborn</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zbozinek</surname><given-names>TD</given-names> </name><name name-style="western"><surname>Craske</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Bui</surname><given-names>A</given-names> </name></person-group><article-title>Detecting momentary reward and affect with real-time passive digital sensor data</article-title><source>JAMIA Open</source><year>2026</year><month>02</month><volume>9</volume><issue>1</issue><fpage>ooag005</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooag005</pub-id><pub-id pub-id-type="medline">41669161</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jafarlou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Azimi</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Objective prediction of next-day&#x2019;s affect using multimodal physiological and behavioral data: algorithm development and validation study</article-title><source>JMIR Form Res</source><year>2023</year><month>03</month><day>15</day><volume>7</volume><issue>1</issue><fpage>e39425</fpage><pub-id pub-id-type="doi">10.2196/39425</pub-id><pub-id pub-id-type="medline">36920456</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ren</surname><given-names>B</given-names> </name><name name-style="western"><surname>Balkind</surname><given-names>EG</given-names> </name><name name-style="western"><surname>Pastro</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Predicting states of elevated negative affect in adolescents from smartphone sensors: a novel personalized machine learning approach</article-title><source>Psychol Med</source><year>2023</year><month>08</month><volume>53</volume><issue>11</issue><fpage>5146</fpage><lpage>5154</lpage><pub-id pub-id-type="doi">10.1017/S0033291722002161</pub-id><pub-id pub-id-type="medline">35894246</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="web"><article-title>Mobile fact sheet</article-title><source>Pew Research Center</source><year>2025</year><month>11</month><day>20</day><access-date>2026-09-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.pewresearch.org/internet/fact-sheet/mobile/">https://www.pewresearch.org/internet/fact-sheet/mobile/</ext-link></comment></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jacobson</surname><given-names>NC</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>YJ</given-names> </name></person-group><article-title>Passive sensing of prediction of moment-to-moment depressed mood among undergraduates with clinical levels of depression sample using smartphones</article-title><source>Sensors (Basel)</source><year>2020</year><month>06</month><day>24</day><volume>20</volume><issue>12</issue><fpage>3572</fpage><pub-id pub-id-type="doi">10.3390/s20123572</pub-id><pub-id pub-id-type="medline">32599801</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shah</surname><given-names>RV</given-names> </name><name name-style="western"><surname>Grennan</surname><given-names>G</given-names> </name><name name-style="western"><surname>Zafar-Khan</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Personalized machine learning of depressed mood using wearables</article-title><source>Transl Psychiatry</source><year>2021</year><month>06</month><day>9</day><volume>11</volume><issue>1</issue><fpage>338</fpage><pub-id pub-id-type="doi">10.1038/s41398-021-01445-0</pub-id><pub-id pub-id-type="medline">34103481</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Craske</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Dunn</surname><given-names>BD</given-names> </name><name name-style="western"><surname>Meuret</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Rizvi</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>CT</given-names> </name></person-group><article-title>Positive affect and reward processing in the treatment of depression, anxiety and trauma</article-title><source>Nat Rev Psychol</source><year>2024</year><volume>3</volume><issue>10</issue><fpage>665</fpage><lpage>685</lpage><pub-id pub-id-type="doi">10.1038/s44159-024-00355-4</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Craske</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Meuret</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Ritz</surname><given-names>T</given-names> </name><name name-style="western"><surname>Treanor</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dour</surname><given-names>H</given-names> </name><name name-style="western"><surname>Rosenfield</surname><given-names>D</given-names> </name></person-group><article-title>Positive affect treatment for depression and anxiety: a randomized clinical trial for a core feature of anhedonia</article-title><source>J Consult Clin Psychol</source><year>2019</year><month>05</month><volume>87</volume><issue>5</issue><fpage>457</fpage><lpage>471</lpage><pub-id pub-id-type="doi">10.1037/ccp0000396</pub-id><pub-id pub-id-type="medline">30998048</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meuret</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Rosenfield</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>E</given-names> </name><name name-style="western"><surname>Hough</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Ritz</surname><given-names>T</given-names> </name><name name-style="western"><surname>Craske</surname><given-names>MG</given-names> </name></person-group><article-title>Positive affect treatment for depression, anxiety, and low positive affect: a randomized clinical trial</article-title><source>JAMA Netw Open</source><year>2026</year><month>04</month><day>1</day><volume>9</volume><issue>4</issue><fpage>e267403</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2026.7403</pub-id><pub-id pub-id-type="medline">42030048</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Watson</surname><given-names>D</given-names> </name><name name-style="western"><surname>Tellegen</surname><given-names>A</given-names> </name></person-group><article-title>Toward a consensual structure of mood</article-title><source>Psychol Bull</source><year>1985</year><volume>98</volume><issue>2</issue><fpage>219</fpage><lpage>235</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.98.2.219</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Costa</surname><given-names>PT</given-names> </name><name name-style="western"><surname>McCrae</surname><given-names>RR</given-names> </name></person-group><article-title>Influence of extraversion and neuroticism on subjective well-being: happy and unhappy people</article-title><source>J Pers Soc Psychol</source><year>1980</year><volume>38</volume><issue>4</issue><fpage>668</fpage><lpage>678</lpage><pub-id pub-id-type="doi">10.1037/0022-3514.38.4.668</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Davidson</surname><given-names>RJ</given-names> </name></person-group><article-title>What does the prefrontal cortex &#x201C;do&#x201D; in affect: perspectives on frontal EEG asymmetry research</article-title><source>Biol Psychol</source><year>2004</year><month>10</month><volume>67</volume><issue>1-2</issue><fpage>219</fpage><lpage>233</lpage><pub-id pub-id-type="doi">10.1016/j.biopsycho.2004.03.008</pub-id><pub-id pub-id-type="medline">15130532</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Newsom</surname><given-names>JT</given-names> </name><name name-style="western"><surname>Rook</surname><given-names>KS</given-names> </name><name name-style="western"><surname>Nishishiba</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sorkin</surname><given-names>DH</given-names> </name><name name-style="western"><surname>Mahan</surname><given-names>TL</given-names> </name></person-group><article-title>Understanding the relative importance of positive and negative social exchanges: examining specific domains and appraisals</article-title><source>J Gerontol B Psychol Sci Soc Sci</source><year>2005</year><month>11</month><volume>60</volume><issue>6</issue><fpage>304</fpage><lpage>P312</lpage><pub-id pub-id-type="doi">10.1093/geronb/60.6.p304</pub-id><pub-id pub-id-type="medline">16260704</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lim</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ar&#x0131;k</surname><given-names>S&#x00D6;</given-names> </name><name name-style="western"><surname>Loeff</surname><given-names>N</given-names> </name><name name-style="western"><surname>Pfister</surname><given-names>T</given-names> </name></person-group><article-title>Temporal Fusion Transformers for interpretable multi-horizon time series forecasting</article-title><source>Int J Forecast</source><year>2021</year><month>10</month><volume>37</volume><issue>4</issue><fpage>1748</fpage><lpage>1764</lpage><pub-id pub-id-type="doi">10.1016/j.ijforecast.2021.03.012</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Larsen</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Diener</surname><given-names>E</given-names> </name></person-group><article-title>Affect intensity as an individual difference characteristic: a review</article-title><source>J Res Pers</source><year>1987</year><month>03</month><volume>21</volume><issue>1</issue><fpage>1</fpage><lpage>39</lpage><pub-id pub-id-type="doi">10.1016/0092-6566(87)90023-7</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Langener</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Stulp</surname><given-names>G</given-names> </name><name name-style="western"><surname>Jacobson</surname><given-names>NC</given-names> </name></person-group><article-title>It&#x2019;s all about timing: exploring different temporal resolutions for analyzing digital-phenotyping data</article-title><source>Adv Methods Pract Psychol Sci</source><year>2024</year><month>01</month><volume>7</volume><issue>1</issue><pub-id pub-id-type="doi">10.1177/25152459231202677</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dickerson</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Kemeny</surname><given-names>ME</given-names> </name></person-group><article-title>Acute stressors and cortisol responses: a theoretical integration and synthesis of laboratory research</article-title><source>Psychol Bull</source><year>2004</year><month>05</month><volume>130</volume><issue>3</issue><fpage>355</fpage><lpage>391</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.130.3.355</pub-id><pub-id pub-id-type="medline">15122924</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moors</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ellsworth</surname><given-names>PC</given-names> </name><name name-style="western"><surname>Scherer</surname><given-names>KR</given-names> </name><name name-style="western"><surname>Frijda</surname><given-names>NH</given-names> </name></person-group><article-title>Appraisal theories of emotion: state of the art and future development</article-title><source>Emotion Review</source><year>2013</year><month>04</month><volume>5</volume><issue>2</issue><fpage>119</fpage><lpage>124</lpage><pub-id pub-id-type="doi">10.1177/1754073912468165</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Russell</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Barrett</surname><given-names>LF</given-names> </name></person-group><article-title>Core affect, prototypical emotional episodes, and other things called emotion: dissecting the elephant</article-title><source>J Pers Soc Psychol</source><year>1999</year><volume>76</volume><issue>5</issue><fpage>805</fpage><lpage>819</lpage><pub-id pub-id-type="doi">10.1037/0022-3514.76.5.805</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rauthmann</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Gallardo-Pujol</surname><given-names>D</given-names> </name><name name-style="western"><surname>Guillaume</surname><given-names>EM</given-names> </name><etal/></person-group><article-title>The Situational Eight DIAMONDS: a taxonomy of major dimensions of situation characteristics</article-title><source>J Pers Soc Psychol</source><year>2014</year><month>10</month><volume>107</volume><issue>4</issue><fpage>677</fpage><lpage>718</lpage><pub-id pub-id-type="doi">10.1037/a0037250</pub-id><pub-id pub-id-type="medline">25133715</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chikersal</surname><given-names>P</given-names> </name><name name-style="western"><surname>Venkatesh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Masown</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Predicting multiple sclerosis outcomes during the COVID-19 stay-at-home period: observational study using passively sensed behaviors and digital phenotyping</article-title><source>JMIR Ment Health</source><year>2022</year><month>08</month><day>24</day><volume>9</volume><issue>8</issue><fpage>e38495</fpage><pub-id pub-id-type="doi">10.2196/38495</pub-id><pub-id pub-id-type="medline">35849686</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Doryab</surname><given-names>A</given-names> </name><name name-style="western"><surname>Villalba</surname><given-names>DK</given-names> </name><name name-style="western"><surname>Chikersal</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Identifying behavioral phenotypes of loneliness and social isolation with passive sensing: statistical analysis, data mining and machine learning of smartphone and Fitbit data</article-title><source>JMIR Mhealth Uhealth</source><year>2019</year><month>07</month><day>24</day><volume>7</volume><issue>7</issue><fpage>e13209</fpage><pub-id pub-id-type="doi">10.2196/13209</pub-id><pub-id pub-id-type="medline">31342903</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gross</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Eckland</surname><given-names>NS</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>RJ</given-names> </name></person-group><article-title>Interoceptive awareness and clarity of one&#x2019;s emotions and goals: a naturalistic investigation</article-title><source>Emotion</source><year>2025</year><month>09</month><volume>25</volume><issue>6</issue><fpage>1516</fpage><lpage>1530</lpage><pub-id pub-id-type="doi">10.1037/emo0001510</pub-id><pub-id pub-id-type="medline">39977691</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Eckland</surname><given-names>NS</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>RJ</given-names> </name></person-group><article-title>When and why people do NOT regulate their emotions: examining the reasons and contexts</article-title><source>Cogn Emot</source><year>2026</year><month>03</month><volume>40</volume><issue>2</issue><fpage>301</fpage><lpage>315</lpage><pub-id pub-id-type="doi">10.1080/02699931.2025.2504560</pub-id><pub-id pub-id-type="medline">40409276</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>O&#x2019;Brien</surname><given-names>ST</given-names> </name><name name-style="western"><surname>Dozo</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hinton</surname><given-names>JDX</given-names> </name><etal/></person-group><article-title>SEMA3: a free smartphone platform for daily life surveys</article-title><source>Behav Res</source><year>2024</year><volume>56</volume><issue>7</issue><fpage>7691</fpage><lpage>7706</lpage><pub-id pub-id-type="doi">10.3758/s13428-024-02445-w</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferreira</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kostakos</surname><given-names>V</given-names> </name><name name-style="western"><surname>Dey</surname><given-names>AK</given-names> </name></person-group><article-title>AWARE: Mobile context instrumentation framework</article-title><source>Front ICT</source><year>2015</year><volume>2</volume><fpage>6</fpage><pub-id pub-id-type="doi">10.3389/fict.2015.00006</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Canzian</surname><given-names>L</given-names> </name><name name-style="western"><surname>Musolesi</surname><given-names>M</given-names> </name></person-group><article-title>Trajectories of depression: unobtrusive monitoring of depressive states by means of smartphone mobility traces analysis</article-title><access-date>2026-09-21</access-date><conf-name>UbiComp &#x2019;15: The 2015 ACM International Joint Conference on Pervasive and Ubiquitous Computing Association for Computing Machinery</conf-name><conf-date>Sep 7-11, 2015</conf-date><conf-loc>Osaka, Japan</conf-loc><fpage>1293</fpage><lpage>1304</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/2750858">https://dl.acm.org/doi/proceedings/10.1145/2750858</ext-link></comment><pub-id pub-id-type="doi">10.1145/2750858.2805845</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Saeb</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kwasny</surname><given-names>M</given-names> </name><name name-style="western"><surname>Karr</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Kording</surname><given-names>K</given-names> </name><name name-style="western"><surname>Mohr</surname><given-names>DC</given-names> </name></person-group><article-title>The relationship between clinical, momentary, and sensor-based assessment of depression</article-title><access-date>2026-09-04</access-date><conf-name>9th International Conference on Pervasive Computing Technologies for Healthcare</conf-name><conf-date>May 20-23, 2015</conf-date><conf-loc>Istanbul, Turkey</conf-loc><fpage>229</fpage><lpage>232</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://eudl.eu/proceedings/PervasiveHealth/2015">http://eudl.eu/proceedings/PervasiveHealth/2015</ext-link></comment><pub-id pub-id-type="doi">10.4108/icst.pervasivehealth.2015.259034</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="thesis"><person-group person-group-type="author"><name name-style="western"><surname>Ringwald</surname><given-names>WR</given-names> </name></person-group><article-title>Refining behavioral phenotypes for binge drinking from broad liabilities to proximal predictors with multiple raters of personality and smartphone sensor data [Dissertation]</article-title><year>2024</year><access-date>2025-05-08</access-date><publisher-name>University of Pittsburgh</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://d-scholarship.pitt.edu/45398">https://d-scholarship.pitt.edu/45398</ext-link></comment></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ester</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kriegel</surname><given-names>HP</given-names> </name><name name-style="western"><surname>Sander</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>X</given-names> </name></person-group><article-title>A density-based algorithm for discovering clusters in large spatial databases with noise</article-title><access-date>2026-09-20</access-date><conf-name>KDD&#x2019;96: Second International Conference on Knowledge Discovery and Data Mining</conf-name><conf-date>Aug 2-4, 1996</conf-date><conf-loc>Portland, Oregon, USA</conf-loc><fpage>226</fpage><lpage>231</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aaai.org/papers/kdd96-037-a-density-based-algorithm-for-discovering-clusters-in-large-spatial-databases-with-noise/">https://aaai.org/papers/kdd96-037-a-density-based-algorithm-for-discovering-clusters-in-large-spatial-databases-with-noise/</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Clemens</surname><given-names>K</given-names> </name></person-group><article-title>Geocoding with OpenStreetMap data</article-title><access-date>2026-09-20</access-date><conf-name>GEOProcessing</conf-name><conf-date>Feb 22-27, 2015</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.thinkmind.org/download_full.php?instance=GEOProcessing+2015">https://www.thinkmind.org/download_full.php?instance=GEOProcessing+2015</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pedregosa</surname><given-names>F</given-names> </name><name name-style="western"><surname>Varoquaux</surname><given-names>G</given-names> </name><name name-style="western"><surname>Gramfort</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Scikit-learn: machine learning in python</article-title><source>J Mach Learn Res</source><year>2011</year><access-date>2026-09-20</access-date><volume>12</volume><fpage>2825</fpage><lpage>2830</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://jmlr.org/papers/v12/pedregosa11a.html">https://jmlr.org/papers/v12/pedregosa11a.html</ext-link></comment></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zou</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hastie</surname><given-names>T</given-names> </name></person-group><article-title>Regularization and variable selection via the elastic net</article-title><source>J R Stat Soc Series B Stat Methodol</source><year>2005</year><month>04</month><day>1</day><volume>67</volume><issue>2</issue><fpage>301</fpage><lpage>320</lpage><pub-id pub-id-type="doi">10.1111/j.1467-9868.2005.00503.x</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Guestrin</surname><given-names>C</given-names> </name></person-group><article-title>XGBoost: a scalable tree boosting system</article-title><access-date>2026-09-20</access-date><conf-name>KDD &#x2019;16: The 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining Association for Computing Machinery</conf-name><conf-date>Aug 13-17, 2016</conf-date><conf-loc>San Francisco, CA, USA</conf-loc><fpage>785</fpage><lpage>794</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/2939672">https://dl.acm.org/doi/proceedings/10.1145/2939672</ext-link></comment><pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Domingos</surname><given-names>P</given-names> </name></person-group><article-title>A few useful things to know about machine learning</article-title><source>Commun ACM</source><year>2012</year><month>10</month><volume>55</volume><issue>10</issue><fpage>78</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1145/2347736.2347755</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Affect distributions, important predictors across temporal and semantic-location feature sets, and prediction performance for discrete affect states.</p><media xlink:href="mhealth_v14i1e90970_app1.docx" xlink:title="DOCX File, 227 KB"/></supplementary-material></app-group></back></article>