<!DOCTYPE article
PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN" "JATS-archivearticle1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="1.3" xml:lang="en" article-type="research-article"><?properties manuscript?><processing-meta base-tagset="archiving" mathml-version="3.0" table-model="xhtml" tagset-family="jats"><restricted-by>pmc</restricted-by></processing-meta><front><journal-meta><journal-id journal-id-type="nlm-journal-id">101656808</journal-id><journal-id journal-id-type="pubmed-jr-id">43770</journal-id><journal-id journal-id-type="nlm-ta">Sleep Health</journal-id><journal-id journal-id-type="iso-abbrev">Sleep Health</journal-id><journal-title-group><journal-title>Sleep health</journal-title></journal-title-group><issn pub-type="ppub">2352-7218</issn><issn pub-type="epub">2352-7226</issn></journal-meta><article-meta><article-id pub-id-type="pmid">39788836</article-id><article-id pub-id-type="pmc">12712857</article-id><article-id pub-id-type="doi">10.1016/j.sleh.2024.10.003</article-id><article-id pub-id-type="manuscript">NIHMS2047577</article-id><article-categories><subj-group subj-group-type="heading"><subject>Article</subject></subj-group></article-categories><title-group><article-title>Performance of machine learning-based methodology using dynamical features to detect non-wear intervals in actigraphy data in a free-living setting</article-title></title-group><contrib-group><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0003-4068-3988</contrib-id><name><surname>Das</surname><given-names>Jyotirmoy Nirupam</given-names></name><xref rid="A1" ref-type="aff">a</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0003-1908-3718</contrib-id><name><surname>Ji</surname><given-names>Linying</given-names></name><xref rid="A2" ref-type="aff">b</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0009-0004-3047-2209</contrib-id><name><surname>Shen</surname><given-names>Yuqi</given-names></name><xref rid="A1" ref-type="aff">a</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-7941-8818</contrib-id><name><surname>Kumara</surname><given-names>Soundar</given-names></name><xref rid="A1" ref-type="aff">a</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0001-5057-633X</contrib-id><name><surname>Buxton</surname><given-names>Orfeu M.</given-names></name><xref rid="A1" ref-type="aff">a</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0003-1938-027X</contrib-id><name><surname>Chow</surname><given-names>Sy-Miin</given-names></name><xref rid="A1" ref-type="aff">a</xref></contrib></contrib-group><aff id="A1"><label>a</label>The Pennsylvania State University, State College, PA</aff><aff id="A2"><label>b</label>Montana State University, Bozeman, MT</aff><author-notes><corresp id="CR1">
<email>jfd5895@psu.edu</email>
</corresp></author-notes><pub-date pub-type="nihms-submitted"><day>11</day><month>12</month><year>2025</year></pub-date><pub-date pub-type="ppub"><month>4</month><year>2025</year></pub-date><pub-date pub-type="epub"><day>09</day><month>1</month><year>2025</year></pub-date><pub-date pub-type="pmc-release"><day>09</day><month>1</month><year>2026</year></pub-date><volume>11</volume><issue>2</issue><fpage>166</fpage><lpage>173</lpage><abstract id="ABS1"><sec id="S1"><title>Goal and Aims:</title><p id="P1">One challenge using wearable sensors is non-wear time. Without a non-wear (e.g., capacitive) sensor, actigraphy data quality can be biased by subjective determinations confounding sleep/wake classification. We developed and evaluated a machine learning algorithm supplemented by dynamic features to discern wear/non-wear episodes.</p></sec><sec id="S2"><title>Focus Technology:</title><p id="P2">Actigraphy data from wrist actigraph (Spectrum, Philips-Respironics)</p></sec><sec id="S3"><title>Reference Technology:</title><p id="P3">the built-in non-wear sensor as &#x02018;ground truth&#x02019; to classify non-wear periods using other data, mimicking features of Actiwatch 2.</p></sec><sec id="S4"><title>Sample:</title><p id="P4">Data was collected over one week from employed adults (n=853).</p></sec><sec id="S5"><title>Design:</title><p id="P5">Extreme gradient boosting (XGBoost), a tree-based classifier algorithm, was used to classify wear/non-wear, supplemented by dynamic features calculated over various time windows.</p></sec><sec id="S6"><title>Core analytics:</title><p id="P6">The performance of the proposed algorithm was tested over 30-sec epochs. Additional analytics and exploratory analyses: Evaluation of the SHapley Additive exPlanations (SHAP) values to find the effectiveness of the dynamic features.</p></sec><sec id="S7"><title>Core Outcomes:</title><p id="P7">The XGBoost classifier yielded substantial improvements in balanced accuracy, sensitivity, and specificity, including dynamic features and comparison to default actiwatch classification algorithms.</p></sec><sec id="S8"><title>Important supplemental outcomes:</title><p id="P8">The proposed classifier effectively distinguished between valid and invalid days, and the duration of contiguous periods of non-wear correctly identified.</p></sec><sec id="S9"><title>Core conclusion:</title><p id="P9">Our findings highlight the potential of XGBoost using dynamic features of varying activity levels across the time series to provide insights on wear/non-wear classification using a large dataset. The methodology provides an alternative to laborious manual benchmarking of the data for similar devices that do not have a non-wear sensor.</p></sec></abstract><kwd-group><kwd>non-wear detection</kwd><kwd>machine learning</kwd><kwd>actigraphy</kwd><kwd>sleep</kwd></kwd-group></article-meta></front><body><sec id="S10"><title>Introduction</title><p id="P10">Data from wrist motion sensors for classifying sleep-wake patterns has grown exponentially due to ease of implementation and cost-effectiveness<sup><xref rid="R1" ref-type="bibr">1</xref></sup>. Wearable-derived digital biomarkers generally provide activity (movement) and light levels or other data to capture passive data over long periods<sup><xref rid="R2" ref-type="bibr">2</xref></sup> in free-living participants non-obtrusively.</p><p id="P11">One of the crucial challenges of studies using wearable sensors is non-wear time, a form of missing data potentially introducing bias. When not worn, the devices may still register low activity levels during non-wear, which may be confounded with or indistinguishable from sleep or sedentary periods based on visual inspection. In devices without non-wear sensors, manually marking non-wear episodes is time-consuming and could introduce further bias. Common automated non-wear detection algorithms in the current literature use tri-axial accelerometers and other signals, such as skin temperature or respiration rate, to detect non-wear intervals accurately. Another approach is setting a time interval with low activity counts for non-wear detection. For example, using data from a cohort of children aged 11 years, ten different non-wear criteria were evaluated; it was concluded that time intervals of 45&#x02013;60 minutes with zero activity counts were sufficient to detect non-wear intervals<sup><xref rid="R3" ref-type="bibr">3</xref></sup>, consistent with similar time interval thresholds for different cohorts and accelerometers in adults<sup><xref rid="R4" ref-type="bibr">4</xref>,<xref rid="R5" ref-type="bibr">5</xref></sup>. Lok &#x00026; Zeitzer, 2020<sup><xref rid="R5" ref-type="bibr">5</xref></sup> used a temporal threshold to distinguish wear and non-wear periods after manually scoring the periods as wear and non-wear.</p><p id="P12">Machine learning techniques have been used in non-wear detection studies, such as a logistic regression model to distinguish non-wear patterns during sleep and wake from 23 participants from the NHANES study<sup><xref rid="R6" ref-type="bibr">6</xref></sup>. A similar regression tree-based method classified wear/nonwear periods using movement and temperature sensor data<sup><xref rid="R7" ref-type="bibr">7</xref></sup> from different time windows as an input feature. Temperature changes increase their model&#x02019;s performance. Some studies generally set a threshold over activity levels or accelerometry sensors over an interval such as 30, 60, or 90 minutes<sup><xref rid="R8" ref-type="bibr">8</xref>,<xref rid="R9" ref-type="bibr">9</xref></sup>. Manual threshold constraints can be arbitrary and exclude the possibility of detecting shorter non-wear episodes.</p><p id="P13">Other studies have identified additional features from their data to improve the performance of their algorithms. A study using Random Forest for a sleep and non-wear classification study implemented in a clinical setting with 36 features extracted from tri-axial accelerometer data, including 12 statistical measures (e.g., mean, median, standard deviation, t increased model accuracy<sup><xref rid="R10" ref-type="bibr">10</xref></sup>. Additional opportunities to maximize features not used in these studies include the stability of the time series, shifts in the variable level, and variability across multiple time resolutions.</p><p id="P14">In this paper, we propose and evaluate, consistent with best practice for the evaluation of sleep algorithms<sup><xref rid="R11" ref-type="bibr">11</xref></sup>, a machine learning-based methodology to detect non-wear time epochs (30-s) from actigraphy data. An eXtreme Gradient Boosting (XGBoost) model was applied to a dataset containing 853 participants, which led to a generalized machine learning-based approach for non-wear classification, including dynamic features, efficient tuning of hyperparameters, and visualization of results to facilitate interpretation of critical dynamic and time scale information. Due to the minimal data requirements, the method developed is versatile enough to be applied to a wide range of sleep wearable studies. Our methodology also addresses the limitations of thresholding used in earlier work, in which a hard threshold has to be chosen in terms of physical activity level and/or time interval for inactiveness. The optimal values for such fixed thresholds vary greatly from dataset to dataset, and do not capture the complexity of a dataset with varied non-wear intervals.</p></sec><sec id="S11"><title>Methods</title><sec id="S12"><title>Sample</title><p id="P15">Data were collected from employed adults from two different industries, as previously described<sup><xref rid="R12" ref-type="bibr">12</xref>&#x02013;<xref rid="R14" ref-type="bibr">14</xref></sup>, including an extended care provider (nursing home direct patient care workers; n=545 individuals, 92% female, Mean<sub>age</sub>: 38 years), and an information technology company (n=308 participants, 44% female, Mean<sub>age</sub>: 46 years). We excluded other participants from the larger dataset with activity counts during off-wrist periods over-written by an early, defective version of the manufacturer&#x02019;s software and set to &#x02018;n/a,&#x02019; so &#x02018;truth&#x02019; was unavailable. Each participant was instructed to wear a wrist actigraph (Spectrum, Philips-Respironics, Murrysville PA) on their non-dominant wrist. The actiwatch provided wrist actigraphy measurements, including activity levels, white light, red light, blue light, and green light collected in 30-second epochs for a week. Non-wear periods were identified with the built-in wrist capacitive sensor, which, if the device was worn correctly, measured skin contact/proximity or its absence. The capacitive sensor could provide, in rare circumstances, both false on-wrist and false off-wrist values in stereotypical and visually identifiable ways, such as nearing battery failure, visually identified by two trained scorers. Scorers visually reviewed the entire recording for each file to better understand an individual&#x02019;s activity pattern and intensity levels across days and examined points of sudden change in activity levels and light levels in the recordings. After each scorer inspected the file independently, inter-rater reliability was assessed, and any discrepancies were resolved among the scoring team. The total number of off-wrist was 7.4% of epochs, and on-wrist was 92.5%, leading to an imbalanced dataset.</p></sec><sec id="S13"><title>Machine learning approach</title><sec id="S14"><title>Data Preparation.</title><p id="P16">In this paper, we explore the use of a tree-based classifier known as XGBoost to classify wear/non-wear epochs<sup><xref rid="R15" ref-type="bibr">15</xref></sup>. Wear is labeled as 1, and non-wear has been labeled as 0 from the Spectrum data. Mimicking the data of the closely related and commonly used Actiwatch 2 model (which has no non-wear sensor), the two key inputs taken from the actiwatch data are the white light and the activity levels. A correlation study between the different light signals confirmed that the white light correlated at least 0.9 with all other light signals; hence, only the white light level was taken as input. To leverage information concerning short- and long-term time dependencies, dynamical features were extracted using the tsfeatures package in R<sup><xref rid="R16" ref-type="bibr">16</xref>,<xref rid="R17" ref-type="bibr">17</xref></sup>. Dynamical features created using the tsfeature package are spectral entropy, maximum level shift, maximum variance shift, maximum Kulback-Leiber divergence, hurst, spike, heterogeneity, the standard deviation of the first derivative of the time series, and activity lags. Spectral entropy, hurst, heterogeneity, and standard deviation of the first derivative of the time series capture if the time series is random/noisy or if it is stable and has a constant trend. Maximum level shifts and spikes keep track of any sudden changes in the time series. Maximum Kulback-Leiber divergence and maximum variance level shift detect any changes in data distribution in consecutive time windows. Dynamical features were created to provide statistical information to the XGBoost model and improve its ability to discern trends or patterns in the time series. The features have been made on the activity levels with a window size of 20 minutes and 1 hour. The 20-minute window captures shorter naps, and the 1-hour window size captures the average nap duration<sup><xref rid="R18" ref-type="bibr">18</xref></sup>. Furthermore, activity lags were created over 5 minutes, 10 minutes, 15 minutes, and 20 minutes. Next, we processed the time series to generate several temporal features such as minute, hour, quarter, day of the week, and time of the day. This information helps to interpret the timestamp in a useful representation, such as if the time of the day is night or the hour is early morning, then the person is more likely to be asleep. In addition, for each participant, the temporal features were converted into sinusoidal curves to capture prominent periodicities in the data. For each participant, employer information was added as a binary variable (1: extended care and 0: information technology). After processing each participant&#x02019;s data, the data were combined to form a larger dataset. This dataset was split into a 70:30 ratio to create the training and test sets. The random split was formed such that the training set has 597 unique participants and the test set has 256 unique participants.</p></sec><sec id="S15"><title>Hyperparameter tuning and model training.</title><p id="P17">The parameters for the XGBoost model were determined by using hyperparameter optimization and cross-validation. The Hyperopt Python library<sup><xref rid="R19" ref-type="bibr">19</xref></sup> was used to efficiently explore a range of hyperparameter values for the XGBoost model (see <xref rid="F1" ref-type="fig">Fig. 1</xref>). The parameters were optimized on the average balanced accuracy across five folds of cross-validation training data. The initial ranges of hyperparameter values were expanded and re-evaluated when any hyperparameter value was observed near the search space&#x02019;s boundaries. The range of the search space of the parameters is given in <xref rid="T1" ref-type="table">Table 1</xref>. The final optimized hyperparameter values and parameter descriptions are summarized in <xref rid="T2" ref-type="table">Table 2</xref>. We used the histogram tree method, which buckets continuous features into bins to yield approximate tree-splitting solutions to enhance computational speed<sup><xref rid="R20" ref-type="bibr">20</xref></sup>. To further speed up the algorithm gbtree booster and GPU Quadro K4000 were used. Here we tuned the scale_pos_weight as part of the hyperparameter values to be optimized via Hyperopt to allow for heavier weighing of the rare instances of non-wear under imbalanced data (i.e., with the inflated presence of wear data) in the loss function<sup><xref rid="R21" ref-type="bibr">21</xref></sup>. The two regularization parameters helped reduce overfitting. From <xref rid="F1" ref-type="fig">Fig. 1</xref>. We can see that after fitting the XGBoost model to the training set, we used Youden&#x02019;s J statistic<sup><xref rid="R22" ref-type="bibr">22</xref></sup> to select the probability threshold for classifying data as non-wear under imbalanced data optimal threshold was found and computed the final average balanced accuracy across the validation folds. The entire modeling process is shown in <xref rid="F1" ref-type="fig">Fig.1</xref>. Once we got the classification results from the XGBoost model based on the optimum parameters, we manually changed the probability threshold to get the final performance metrics on the training dataset. We manually changed the threshold instead of relying on Youden&#x02019;s J statistic because Yoden&#x02019;s J statistic increased the specificity at the expense of Negative Predictive Value ( NPV). The effects of using different probability thresholds for non-wear classification on the performance metrics are shown in <xref rid="F2" ref-type="fig">Fig 2</xref>. We observe that at probability threshold 0.55, there is a balance between the specificity and NPV value. Increasing or decreasing the probability threshold would lead to unequal tradeoffs in the performance metrics; hence, 0.55 was set as the optimum probability threshold for our dataset.</p><p id="P18">To further improve our model, we employed a temporal smoothing function to address abrupt (and rare/unlikely 1-epoch) changes in the output at an epoch level. This smoothing function checks neighborhood labels and assigns the label that has the majority in that time window. For example, in a window size of 3 epochs, if there are two non-wear labels and one wear label, all epochs are labeled as non-wear. A range of window sizes from 3 to 20 epochs were used to find the optimal window size. The optimal window size was found to be 3 for the training dataset. This window size gave the maximum specificity; as we increased the window size, specificity decreased.</p></sec><sec id="S16"><title>Classification performance evaluation.</title><p id="P19">The main aim of this paper was to distinguish non-wear epochs in actigraphy data. The XGBoost model provided probabilities at an epoch level and we determined the optimal probability threshold level based on different classification metrics. The classification metrics studied in the paper are sensitivity, specificity, balanced accuracy, F1 score, positive predictive value (PPV), and negative predictive value (NPV). Area under the receiver operating characteristic curve (AUC) was computed for the initial set of results (<xref rid="T3" ref-type="table">Table 3</xref>). All other results are based on the optimal probability threshold.</p><p id="P20">To further evaluate our methodology, we compared our result with a published methodology addressing the same general problem<sup><xref rid="R5" ref-type="bibr">5</xref></sup>, which showed that 95% of the inactive episodes are of length shorter than 20 minutes in an Actiwatch 2 with a resolution of data of 60s. To create these non-wear /inactive episodes automatically, they used R studio to label any consecutive epochs with zero activity counts as an episode of non-wear. The episodes were then visually labeled as wear and non-wear. They then generated predictions on the episodes by calculating the length of each episode and classifying it as non-wear if the episode had a length of more than 20 minutes. The optimal time threshold was computed from the receiver operating curve. Following their methodology, we derived an optimal threshold of 1 minute, which gave the highest accuracy in detecting non-wear episodes in our dataset. They used activity levels to classify wear/non-wear, and we followed similar approaches, adding dynamic features and white light to predict non-wear and wear epochs.</p></sec></sec></sec><sec id="S17"><title>Results</title><sec id="S18"><title>Core analytics and main outcome variables</title><sec id="S19"><title>Epoch-by-Epoch Performance</title><p id="P21">In this paper, using data from 853 participants, 597 in the training dataset, and 256 in the test dataset. Each participant contributed ~ 20,000 epochs. Sensitivity refers to correctly predicting wear epochs whereas specificity gives the performance of the model on the non-wear epochs. The results from the optimum XGBoost model (<xref rid="T3" ref-type="table">Table 3</xref>) show that the balanced accuracy in the training is 0.881 and 0.876 in the test set. Due to the class imbalance in the data, we have a high sensitivity of 0.991 as opposed to a specificity of 0.771 in the training set. To balance this, we reclassify the probabilities generated by the XGBoost model using 0.55 as the new threshold and add the temporal smoothing function. The updated results are shown in <xref rid="T4" ref-type="table">Table 4</xref>, here, we can see that our specificity rises to 0.822 in the training data. The specificity is lower in the test at 0.791 but is still comparable to the training accuracy. Thus, we do not overfit or underfit our model. The sensitivity decreases from 0.991 to 0.985 in the training set but is still high enough to be determinantal in classifying wear time points. The NPV comes down when we use the probability threshold, but that is the best tradeoff we found in the model against increasing the specificity. The results of the comparison study can be seen in <xref rid="T4" ref-type="table">Table 4</xref>. We can see that our proposed model outperforms the temporal threshold method on every classification metric. This evaluation supports the claim that machine learning algorithms such as XGBoost can be an automation tool to classify wear and non-wear.</p><p id="P22">To validate the model further, we matched the invalid days in the dataset against the invalid days predicted by our algorithm. A day is defined as invalid if the person does not wear the actiwatch for more than 4 hours. We label invalid days as 1 and valid days as 0. From <xref rid="T5" ref-type="table">Table 5</xref>, we can calculate that we are predicting valid days with an accuracy of 0.997 and invalid days with 0.921 accuracy in the training dataset. We get comparable accuracy in the test set, as can be seen in <xref rid="T5" ref-type="table">Table 5</xref>. <xref rid="T6" ref-type="table">Table 6</xref> shows a comprehensive list of classification metrics for predicting invalid and valid days. Most of the metrics are above 0.9; hence, our proposed methodology does not lead to a mismatch between valid and invalid days.</p></sec></sec><sec id="S20"><title>Additional analytics and exploratory analyses</title><p id="P23">To visualize the predictive roles of the dynamic features, we plotted the SHAP (SHapley Additive exPlanations)<sup><xref rid="R23" ref-type="bibr">23</xref></sup> values of the 20 most important features (<xref rid="F3" ref-type="fig">Fig. 3</xref>) to identify how a change in each observed value of a feature would change the prediction of the probability of wearing the device in the model. A high standard deviation in the first derivative of the activity levels in a time window of one hour leads to a higher probability of wear. That is, higher growth in activity levels was associated with an increase in the probability of wearing the device. In contrast, lower maximum level shift for that hour (denoted as max_level_shift_hr), maximum variance shift for that hour (denoted as max_var_shift_hr), and number of sharp spikes for that hour (spike_hr) were observed to play multiple roles: they were associated in higher probability for wear in some instances, but also higher probability for non-wear in other instances. In other words, while prolonged instances with little spikes, level, and variance shifts in an hourly window might be indicative of non-wear, these dynamical features might also capture episodes of sustained high physical activity. The abbreviation &#x0201c;short&#x0201d; in the feature labels in <xref rid="F3" ref-type="fig">Fig. 3</xref> corresponds to dynamical features created on the activity levels over a time window of 20 minutes. Hence, we have two-time windows of 20 minutes and one hour for each dynamic feature.</p><p id="P24">The standard deviation of the activity levels, spikes, and maximum variance shift over the shorter time window also contributed to the accuracy of the model in a similar pattern as the same features over a one-hour time window. Both time windows work in tandem to give short and long-term information to the model. The shorter time window reflected changes in the recent history of the time series, while the hourly time window captured the general trend of the feature. A high Hurst value suggested high persistence in an individual activity level, the most likely occurrence of which might be during individuals&#x02019; periods of non-wear. In contrast, a low Hurst value, suggesting low persistence, reflected sustained continuous changes in the process of interest, namely ongoing fluctuations in the individual&#x02019;s physical activity during wear. For the Hurst exponent feature, the SHAP value plot gives similar interpretations for both time windows. Volatile activity levels during physical activities are indicated by high entropy levels over the hour (entorpy_hr), whereas low entropy levels suggested no change in activity levels, which in turn, might be suggestive of non-wear. The maximum variance shift, maximum level shift, and spikes capture the sudden changes in the activity levels, in contrast, the Hurst and entropy indicate the flow of the activity levels. The SHAP value plot in <xref rid="F3" ref-type="fig">Fig 3</xref>. shows that the light signal, when compared to the dynamic features, was not important in facilitating non-wear classification . However, we kept the light signal as a feature since it did not significantly affect computational time. The above analysis demonstrates the relevance and insights dynamic features provide to the XGBoost model.</p><p id="P25">To further analyze the results in terms of non-wear intervals of different durations, we created epoch intervals with different numbers of continuous non-wear epochs. An epoch interval length is defined as the length of the continuous set of non-wear epochs predicted by our methodology. Our nonwear epoch interval length varied from 2 to 19970 epochs, corresponding to nonwear intervals from 1 minute to 7 days for 30-second epochs. Next, we grouped the different epoch intervals based on their length with a bin size of 20 epochs (10 mins) as can be seen in <xref rid="F4" ref-type="fig">Fig. 4</xref>, depicting the mean and median accuracy of the predicted non-wear epoch intervals against the actual number of non-wear epochs present in them. From the plot, we can observe that the mean accuracy gradually rises to 1 as the epoch interval length increases to 530 epochs. So for higher epoch interval length, the algorithm is better able to identify a set of continuous non-wear episodes. The median values approach 1 for epoch interval length of 110 epochs (55mins) to 530 epochs (nearly 4.5 hours). This suggests that the proposed methodology can discern between long naps or long periods of inactivity against non-wear epochs. Hence this methodology can be used in sleep studies where the main aim in preprocessing data is to identify valid days or long periods of non-wear in the dataset. To better illustrate the methodology&#x02019;s performance on longer time windows, two data samples of length one week each have been shown in <xref rid="F5" ref-type="fig">Fig. 5</xref>. The black bands represent non-wear episodes, and the white bands show wear episodes. For each sample data, the ground truth from the capacitive sensor and the prediction from the methodology are depicted. The two bands from each sample are not identical, but the non-wear episodes on a longer horizon are effectively captured by the proposed methodology.</p></sec></sec><sec id="S21"><title>Discussion</title><p id="P26">In the current study, we addressed the problem of determining non-wear epochs in a sleep wearable (Actiwatch Spectrum) relevant to data from a different model of the same device (Actiwatch 2.0) that does not include a non-wear sensor feature, a commonly used device/data type in prior sleep studies. Addressing non-wear is particularly important because non-wear intervals near or during actual time in bed periods may be classified as sleep, leading to non-random bias in characterizing sleep and wake patterns. We used a large dataset using the Actiwatch Spectrum device, similar to the Actiwatch 2.0 but including a non-wear sensor that was used as the &#x02018;ground truth&#x02019; for non-wear. We developed and evaluated an XGBoost model to classify wear and non-wear epochs from an actiwatch worn by a participant in a free-living setting. We extracted different dynamical features from the activity level data from the actiwatch and also expanded the time stamps into sinusoidal curves to better utilize the temporal component of the data. This expanded dataset with 37 features was then used as input into the XGBoost model. The optimum parameters for the model were selected by running a cross-validation and hyperparameter tuning algorithm on the XGBoost model. Our systematic evaluation revealed classification performance in classifying epoch level non-wear in the test dataset of AUC: 0.937, Sensitivity: 0.760, Specificity: 0.760, and Balanced Accuracy: 0.875, among other favorable evaluation metrics (<xref rid="T3" ref-type="table">Table 3</xref>). The results were further optimized by the application of manual probability thresholding and the temporal smoothing function, leading to a .887 balanced accuracy score in the test dataset. In comparison to the temporal thresholding methodology described in Lok &#x00026; Zeitzer, 2020<sup><xref rid="R5" ref-type="bibr">5</xref></sup> our methodology performed better in all performance metrics. Since we achieved high balanced accuracy on the test set, the proposed model and dynamic features can be implemented to detect nonwear on other datasets. As a practical outcome, if an &#x02018;invalid day&#x02019; is deemed to require at least 24h of wear time or &#x0003c;4h of non-wear, we calculated that we predicted valid days with an accuracy of 0.997 and invalid days with an accuracy of 0.921. Hence, in studies where the determination of valid days plays a major role, our methodology can be used to identify the days efficiently. Importantly, we further evaluated the threshold of the duration of a non-wear period of consecutive non-wear epochs where classification approached 100% correct at a contiguous duration of ~ 1 hour. This evaluation supports the use of this non-wear detection algorithm for rigorous detection of sleep in the absence of a non-wear sensor on a wearable.</p><sec id="S22"><title>Limitations and Future Perspectives</title><p id="P27">One of the limitations of our study can be observed in <xref rid="F4" ref-type="fig">Fig 4</xref>, where we can see that the median accuracy is near zero for epoch lengths of less than 110 epochs (55 mins). Brief periods of time when the activity level is low, leading to misclassification of non-wear. Due to this, the proposed methodology has an NPV of 0.783. The median accuracy might be low due to short naps, which might be confused with non-wear epochs due to low activity levels. In addition to this short duration, epoch lengths might be inaccurately classified due to a lack of knowledge of the recent history of the time series. Alternative machine learning variations, including long short-term memory networks and other recurrent neural network variations<sup><xref rid="R24" ref-type="bibr">24</xref></sup>, and a wavelet scattering-based model<sup><xref rid="R25" ref-type="bibr">25</xref></sup> might be an approach for future work. Due to the unbalanced nature of the data, a high window length for the RNN model might be necessary to capture non-wear periods with the limitation of high computational resources, which might be infeasible in large-scale studies. In this manuscript, we extracted dynamic features from activity levels. One may further analyze the activity levels regarding the hierarchical organization of human physical activity. Depending on the time scale of a physiological process, it can be characterized in 4 ranges. Each range has specific characteristics that might add insights into the activity pattern<sup><xref rid="R26" ref-type="bibr">26</xref></sup>. Similarly, if the data has sleep periods, one can find patterns in locomotor inactivity to study sleep dynamics<sup><xref rid="R27" ref-type="bibr">27</xref></sup>. Such patterns can be used as additional data for the machine learning model but it might require high computational memory. Another limitations is due to the nature of the capacitive sensor on the wrist which may lead to bried false positives.</p></sec><sec id="S23"><title>Core conclusion</title><p id="P28">An XGBoost model was built on a dataset to classify non-wear and wear epochs of a commonly used sleep wearable without a non-wear sensor using the data from a nearly identical version of the device with a non-wear sensor as a reference. A set of dynamic features was created to augment the dataset and substantially increase the accuracy of the model. Even though the non-wear classification algorithm proposed in this article was applied to data from one specific device, the key innovation resides in the extraction and use of general, device-agnostic dynamic features that summarize differences in the trends and patterns of movement data characterizing wear and non-wear episodes. Thus, the algorithm is also applicable to other accelerometry devices. When compared to a temporal thresholding method, our methodology performed better on five out of the six classification metrics used in this paper. Our methodology was able to detect invalid days in the dataset and contiguous non-wear periods over 55 minutes, meaning night sleep and longer naps can be detected using this algorithm in data from a sleep wearable without a non-wear sensor. Future work may involve using complex neural network models to improve accuracy.</p></sec></sec></body><back><ack id="S24"><title>Funding</title><p id="P29">&#x02022; The source of this data, the Work Family and Health Study (<ext-link xlink:href="http://www.workfamilyhealthnetwork.org/" ext-link-type="uri">www.WorkFamilyHealthNetwork.org</ext-link>), was funded by a cooperative agreement through the National Institutes of Health and the Centers for Disease Control and Prevention: Eunice Kennedy Shriver National Institute of Child Health and Human Development (U01HD051217, U01HD051218, U01HD051256, and U01HD051276); National Institute on Aging (U01AG027669); Office of Behavioral and Social Sciences Research and National Institute for Occupational Safety and Health (U01OH008788 and U01HD059773). Grants from the National Heart, Lung, and Blood Institute (R01HL107240), William T. Grant Foundation, Alfred P. Sloan Foundation, and the Administration for Children and Families have provided additional funding. Specific funding for this project is also provided by the Penn State Social Science Research Institute. The contents of this publication are solely the responsibility of the authors and do not necessarily represent the official views of these institutes and offices.</p><p id="P30">&#x02022; National Institutes of Health grants U24AA027684, OT2 HL161847, and R01 DK134863; the Pennsylvania State University Quantitative Social Sciences Initiative and UL1TR002014&#x02013;06 from the National Center for Advancing Translational Sciences</p></ack><fn-group><fn fn-type="COI-statement" id="FN1"><p id="P31">Conflicts of interest</p><p id="P32">Dr. Orfeu M. Buxton: Outside of the current work, Dr. Orfeu M. Buxton discloses that he received subcontract grants to Penn State from Proactive Life (formerly Mobile Sleep Technologies), doing business as SleepSpace (National Science Foundation grant #1622766 and NIH/National Institute on Aging Small Business Innovation Research Program R43AG056250, R44 AG056250), received honoraria/travel support for lectures from Boston University, Boston College, Tufts School of Dental Medicine, New York University, University of Miami, University of South Florida, University of Utah, University of Arizona, Harvard Chan School of Public Health, Eric H. Angle Society of Orthodontists, consulting fees for SleepNumber, and receives an honorarium for his role as the Editor in Chief of the journal Sleep Health.</p><p id="P33">Dr. Chow: Dr. Chow reports grants from National Institutes of Health, during the conduct of the study.</p></fn><fn id="FN2"><p id="P34">No Disclosures to report: JND, LJ, YS</p></fn><fn id="FN3"><p id="P35" content-type="publisher-disclaimer">This is a PDF file of an unedited manuscript that has been accepted for publication. As a service to our customers we are providing this early version of the manuscript. The manuscript will undergo copyediting, typesetting, and review of the resulting proof before it is published in its final form. Please note that during the production process errors may be discovered which could affect the content, and all legal disclaimers that apply to the journal pertain.</p></fn></fn-group><ref-list><title>References</title><ref id="R1"><label>1.</label><mixed-citation publication-type="journal"><name><surname>Danilenko</surname><given-names>KV</given-names></name>, <name><surname>Stefani</surname><given-names>O</given-names></name>, <name><surname>Voronin</surname><given-names>KA</given-names></name>, <etal/>
<article-title>Wearable Light-and-Motion Dataloggers for Sleep/Wake Research: A Review</article-title>. <source>Appl Sci</source>. <year>2022</year>;<volume>12</volume>(<issue>22</issue>):<fpage>11794</fpage>. doi:<pub-id pub-id-type="doi">10.3390/app122211794</pub-id></mixed-citation></ref><ref id="R2"><label>2.</label><mixed-citation publication-type="journal"><name><surname>Nakagata</surname><given-names>T</given-names></name>, <name><surname>Murakami</surname><given-names>H</given-names></name>, <name><surname>Kawakami</surname><given-names>R</given-names></name>, <etal/>
<article-title>Step-count outcomes of 13 different activity trackers: Results from laboratory and free-living experiments</article-title>. <source>Gait Posture</source>. <year>2022</year>;<volume>98</volume>:<fpage>24</fpage>&#x02013;<lpage>33</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.gaitpost.2022.08.004</pub-id><pub-id pub-id-type="pmid">36030707</pub-id>
</mixed-citation></ref><ref id="R3"><label>3.</label><mixed-citation publication-type="journal"><name><surname>Aadland</surname><given-names>E</given-names></name>, <name><surname>Andersen</surname><given-names>LB</given-names></name>, <name><surname>Anderssen</surname><given-names>SA</given-names></name>, <name><surname>Resaland</surname><given-names>GK</given-names></name>. <article-title>A comparison of 10 accelerometer non-wear time criteria and logbooks in children</article-title>. <source>BMC Public Health</source>. <year>2018</year>;<volume>18</volume>(<issue>1</issue>):<fpage>323</fpage>. doi:<pub-id pub-id-type="doi">10.1186/s12889-018-5212-4</pub-id><pub-id pub-id-type="pmid">29510709</pub-id>
</mixed-citation></ref><ref id="R4"><label>4.</label><mixed-citation publication-type="journal"><name><surname>Berendsen</surname><given-names>BAJ</given-names></name>, <name><surname>Hendriks</surname><given-names>MRC</given-names></name>, <name><surname>Willems</surname><given-names>P</given-names></name>, <name><surname>Meijer</surname><given-names>K</given-names></name>, <name><surname>Schaper</surname><given-names>NC</given-names></name>, <article-title>Savelberg HHCM. A 20 min window is optimal in a non-wear algorithm for tri-axial thigh-worn accelerometry in overweight people</article-title>. <source>Physiol Meas</source>. <year>2014</year>;<volume>35</volume>(<issue>11</issue>):<fpage>2205</fpage>&#x02013;<lpage>2212</lpage>. doi:<pub-id pub-id-type="doi">10.1088/0967-3334/35/11/2205</pub-id><pub-id pub-id-type="pmid">25341022</pub-id>
</mixed-citation></ref><ref id="R5"><label>5.</label><mixed-citation publication-type="journal"><name><surname>Lok</surname><given-names>R</given-names></name>, <name><surname>Zeitzer</surname><given-names>JM</given-names></name>. <article-title>A Temporal Threshold for Distinguishing Off-Wrist from Inactivity Periods: A Retrospective Actigraphy Analysis</article-title>. <source>Clocks Sleep</source>. <year>2020</year>;<volume>2</volume>(<issue>4</issue>):<fpage>466</fpage>&#x02013;<lpage>472</lpage>. doi:<pub-id pub-id-type="doi">10.3390/clockssleep2040034</pub-id><pub-id pub-id-type="pmid">33198122</pub-id>
</mixed-citation></ref><ref id="R6"><label>6.</label><mixed-citation publication-type="journal"><name><surname>Thapa-Chhetry</surname><given-names>B</given-names></name>, <name><surname>Arguello</surname><given-names>DJ</given-names></name>, <name><surname>John</surname><given-names>D</given-names></name>, <name><surname>Intille</surname><given-names>S</given-names></name>. <article-title>Detecting Sleep and Nonwear in 24-h Wrist Accelerometer Data from the National Health and Nutrition Examination Survey</article-title>. <source>Med Sci Sports Exerc</source>. <year>2022</year>;<volume>54</volume>(<issue>11</issue>):<fpage>1936</fpage>&#x02013;<lpage>1946</lpage>. doi:<pub-id pub-id-type="doi">10.1249/MSS.0000000000002973</pub-id><pub-id pub-id-type="pmid">36007161</pub-id>
</mixed-citation></ref><ref id="R7"><label>7.</label><mixed-citation publication-type="journal"><name><surname>Vert</surname><given-names>A</given-names></name>, <name><surname>Weber</surname><given-names>KS</given-names></name>, <name><surname>Thai</surname><given-names>V</given-names></name>, <etal/>
<article-title>Detecting accelerometer non-wear periods using change in acceleration combined with rate-of-change in temperature</article-title>. <source>BMC Med Res Methodol</source>. <year>2022</year>;<volume>22</volume>(<issue>1</issue>):<fpage>147</fpage>. doi:<pub-id pub-id-type="doi">10.1186/s12874-022-01633-6</pub-id><pub-id pub-id-type="pmid">35596151</pub-id>
</mixed-citation></ref><ref id="R8"><label>8.</label><mixed-citation publication-type="journal"><name><surname>Choi</surname><given-names>L</given-names></name>, <name><surname>Liu</surname><given-names>Z</given-names></name>, <name><surname>Matthews</surname><given-names>CE</given-names></name>, <name><surname>Buchowski</surname><given-names>MS</given-names></name>. <article-title>Validation of Accelerometer Wear and Nonwear Time Classification Algorithm</article-title>. <source>Med Sci Sports Exerc</source>. <year>2011</year>;<volume>43</volume>(<issue>2</issue>):<fpage>357</fpage>&#x02013;<lpage>364</lpage>. doi:<pub-id pub-id-type="doi">10.1249/MSS.0b013e3181ed61a3</pub-id><pub-id pub-id-type="pmid">20581716</pub-id>
</mixed-citation></ref><ref id="R9"><label>9.</label><mixed-citation publication-type="journal"><name><surname>Hecht</surname><given-names>A</given-names></name>, <name><surname>Ma</surname><given-names>S</given-names></name>, <name><surname>Porszasz</surname><given-names>J</given-names></name>, <name><surname>Casaburi</surname><given-names>R</given-names></name>, <article-title>For The Copd Clinical Research Netw. Methodology for Using Long-Term Accelerometry Monitoring to Describe Daily Activity Patterns in COPD</article-title>. <source>COPD J Chronic Obstr Pulm Dis</source>. <year>2009</year>;<volume>6</volume>(<issue>2</issue>):<fpage>121</fpage>&#x02013;<lpage>129</lpage>. doi:<pub-id pub-id-type="doi">10.1080/15412550902755044</pub-id></mixed-citation></ref><ref id="R10"><label>10.</label><mixed-citation publication-type="journal"><name><surname>Sundararajan</surname><given-names>K</given-names></name>, <name><surname>Georgievska</surname><given-names>S</given-names></name>, <name><surname>Te Lindert</surname><given-names>BHW</given-names></name>, <etal/>
<article-title>Sleep classification from wrist-worn accelerometer data using random forests</article-title>. <source>Sci Rep</source>. <year>2021</year>;<volume>11</volume>(<issue>1</issue>):<fpage>24</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-020-79217-x</pub-id><pub-id pub-id-type="pmid">33420133</pub-id>
</mixed-citation></ref><ref id="R11"><label>11.</label><mixed-citation publication-type="journal"><name><surname>De Zambotti</surname><given-names>M</given-names></name>, <name><surname>Menghini</surname><given-names>L</given-names></name>, <name><surname>Grandner</surname><given-names>MA</given-names></name>, <etal/>
<article-title>Rigorous performance evaluation (previously, &#x0201c;validation&#x0201d;) for informed use of new technologies for sleep health measurement</article-title>. <source>Sleep Health</source>. <year>2022</year>;<volume>8</volume>(<issue>3</issue>):<fpage>263</fpage>&#x02013;<lpage>269</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.sleh.2022.02.006</pub-id><pub-id pub-id-type="pmid">35513978</pub-id>
</mixed-citation></ref><ref id="R12"><label>12.</label><mixed-citation publication-type="book"><name><surname>Bray</surname><given-names>J</given-names></name>, <name><surname>Kelly</surname><given-names>E</given-names></name>, <name><surname>Hammer</surname><given-names>L</given-names></name>, <etal/>
<source>An Integrative, Multilevel, and Transdisciplinary Research Approach to Challenges of Work, Family, and Health</source>. <publisher-name>RTI Press</publisher-name>; <year>2013</year>. doi:<pub-id pub-id-type="doi">10.3768/rtipress.2013.mr.0024.1303</pub-id></mixed-citation></ref><ref id="R13"><label>13.</label><mixed-citation publication-type="journal"><name><surname>Olson</surname><given-names>R</given-names></name>, <name><surname>Crain</surname><given-names>TL</given-names></name>, <name><surname>Bodner</surname><given-names>TE</given-names></name>, <etal/>
<article-title>A workplace intervention improves sleep: results from the randomized controlled Work, Family, and Health Study</article-title>. <source>Sleep Health</source>. <year>2015</year>;<volume>1</volume>(<issue>1</issue>):<fpage>55</fpage>&#x02013;<lpage>65</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.sleh.2014.11.003</pub-id><pub-id pub-id-type="pmid">29073416</pub-id>
</mixed-citation></ref><ref id="R14"><label>14.</label><mixed-citation publication-type="journal"><name><surname>Berkman</surname><given-names>LF</given-names></name>, <name><surname>Kelly</surname><given-names>EL</given-names></name>, <name><surname>Hammer</surname><given-names>LB</given-names></name>, <etal/>
<article-title>Employee Cardiometabolic Risk Following a Cluster-Randomized Workplace Intervention From the Work, Family and Health Network, 2009&#x02013;2013</article-title>. <source>Am J Public Health</source>. <year>2023</year>;<volume>113</volume>(<issue>12</issue>):<fpage>1322</fpage>&#x02013;<lpage>1331</lpage>. doi:<pub-id pub-id-type="doi">10.2105/AJPH.2023.307413</pub-id><pub-id pub-id-type="pmid">37939328</pub-id>
</mixed-citation></ref><ref id="R15"><label>15.</label><mixed-citation publication-type="confproc"><name><surname>Chen</surname><given-names>T</given-names></name>, <name><surname>Guestrin</surname><given-names>C</given-names></name>. <source>XGBoost: A Scalable Tree Boosting System</source>. In: <conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name>. <publisher-name>ACM</publisher-name>; <year>2016</year>:<fpage>785</fpage>&#x02013;<lpage>794</lpage>. doi:<pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></mixed-citation></ref><ref id="R16"><label>16.</label><mixed-citation publication-type="confproc"><name><surname>Hyndman</surname><given-names>RJ</given-names></name>, <name><surname>Wang</surname><given-names>E</given-names></name>, <name><surname>Laptev</surname><given-names>N</given-names></name>. <source>Large-Scale Unusual Time Series Detection</source>. In: <conf-name>2015 IEEE International Conference on Data Mining Workshop (ICDMW)</conf-name>. <publisher-name>IEEE</publisher-name>; <year>2015</year>:<fpage>1616</fpage>&#x02013;<lpage>1619</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICDMW.2015.104</pub-id></mixed-citation></ref><ref id="R17"><label>17.</label><mixed-citation publication-type="other"><ext-link xlink:href="http://tsfeatures.pdf" ext-link-type="uri">tsfeatures.pdf</ext-link>. Accessed <date-in-citation>April 4, 2024</date-in-citation>. <ext-link xlink:href="https://cran.r-project.org/web/packages/tsfeatures/tsfeatures.pdf" ext-link-type="uri">https://cran.r-project.org/web/packages/tsfeatures/tsfeatures.pdf</ext-link></mixed-citation></ref><ref id="R18"><label>18.</label><mixed-citation publication-type="journal"><name><surname>Milner</surname><given-names>CE</given-names></name>, <name><surname>Cote</surname><given-names>KA</given-names></name>. <article-title>Benefits of napping in healthy adults: impact of nap length, time of day, age, and experience with napping</article-title>. <source>J Sleep Res</source>. <year>2009</year>;<volume>18</volume>(<issue>2</issue>):<fpage>272</fpage>&#x02013;<lpage>281</lpage>. doi:<pub-id pub-id-type="doi">10.1111/j.1365-2869.2008.00718.x</pub-id><pub-id pub-id-type="pmid">19645971</pub-id>
</mixed-citation></ref><ref id="R19"><label>19.</label><mixed-citation publication-type="confproc"><name><surname>Bergstra</surname><given-names>J</given-names></name>, <name><surname>Yamins</surname><given-names>D</given-names></name>, <name><surname>Cox</surname><given-names>DD</given-names></name>. <source>Making a Science of Model Search: Hyperparameter Optimization in Hundreds of Dimensions for Vision Architectures</source>. In <conf-name>International conference on machine learning</conf-name>, pp. <fpage>115</fpage>&#x02013;<lpage>123</lpage>. <publisher-name>PMLR</publisher-name>, <year>2013</year>.</mixed-citation></ref><ref id="R20"><label>20.</label><mixed-citation publication-type="journal"><name><surname>Ke</surname><given-names>G</given-names></name>, <name><surname>Meng</surname><given-names>Q</given-names></name>, <name><surname>Finley</surname><given-names>T</given-names></name>, <etal/>
<article-title>LightGBM: A Highly Efficient Gradient Boosting Decision Tree</article-title>. <source>Advances in neural information processing systems</source>
<volume>30</volume> (<year>2017</year>).</mixed-citation></ref><ref id="R21"><label>21.</label><mixed-citation publication-type="webpage"><source>XGBoost Parameters &#x02014; xgboost 2.1.0-dev documentation</source>. Accessed <date-in-citation>April 4, 2024</date-in-citation>. <ext-link xlink:href="https://xgboost.readthedocs.io/en/latest/parameter.html" ext-link-type="uri">https://xgboost.readthedocs.io/en/latest/parameter.html</ext-link></mixed-citation></ref><ref id="R22"><label>22.</label><mixed-citation publication-type="journal"><name><surname>Youden</surname><given-names>WJ</given-names></name>. <article-title>Index for rating diagnostic tests</article-title>. <source>Cancer</source>. <year>1950</year>;<volume>3</volume>(<issue>1</issue>):<fpage>32</fpage>&#x02013;<lpage>35</lpage>. doi:<pub-id pub-id-type="doi">10.1002/1097-0142(1950)3:1&#x0003c;32::AID-CNCR2820030106&#x0003e;3.0.CO;2-3</pub-id><pub-id pub-id-type="pmid">15405679</pub-id>
</mixed-citation></ref><ref id="R23"><label>23.</label><mixed-citation publication-type="journal"><name><surname>Lundberg</surname><given-names>S</given-names></name>, <name><surname>Lee</surname><given-names>SI</given-names></name>. <source>A Unified Approach to Interpreting Model Predictions</source>. Published online November 24, 2017. Accessed <date-in-citation>February 21, 2024</date-in-citation>. <ext-link xlink:href="http://arxiv.org/abs/1705.07874" ext-link-type="uri">http://arxiv.org/abs/1705.07874</ext-link></mixed-citation></ref><ref id="R24"><label>24.</label><mixed-citation publication-type="journal"><name><surname>Schmidt</surname><given-names>RM</given-names></name>. <source>Recurrent Neural Networks (RNNs): A gentle Introduction and Overview</source>. Published online November 23, 2019. Accessed <date-in-citation>April 23, 2024</date-in-citation>. <ext-link xlink:href="http://arxiv.org/abs/1912.05911" ext-link-type="uri">http://arxiv.org/abs/1912.05911</ext-link></mixed-citation></ref><ref id="R25"><label>25.</label><mixed-citation publication-type="journal"><name><surname>Sharma</surname><given-names>M</given-names></name>, <name><surname>Lodhi</surname><given-names>H</given-names></name>, <name><surname>Yadav</surname><given-names>R</given-names></name>, <name><surname>Acharya</surname><given-names>UR</given-names></name>. <article-title>Sleep disorder identification using wavelet scattering on ECG signals</article-title>. <source>Int J Imaging Syst Technol</source>. <year>2024</year>;<volume>34</volume>(<issue>1</issue>):e22980. doi:<pub-id pub-id-type="doi">10.1002/ima.22980</pub-id></mixed-citation></ref><ref id="R26"><label>26.</label><mixed-citation publication-type="journal"><name><surname>B&#x000fa;z&#x000e1;s</surname><given-names>A</given-names></name>, <name><surname>Makai</surname><given-names>A</given-names></name>, <name><surname>Groma</surname><given-names>GI</given-names></name>, <etal/>
<article-title>Hierarchical organization of human physical activity</article-title>. <source>Sci Rep</source>. <year>2024</year>;<volume>14</volume>(<issue>1</issue>):<fpage>5981</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-024-56185-0</pub-id><pub-id pub-id-type="pmid">38472275</pub-id>
</mixed-citation></ref><ref id="R27"><label>27.</label><mixed-citation publication-type="journal"><name><surname>Winnebeck</surname><given-names>EC</given-names></name>, <name><surname>Fischer</surname><given-names>D</given-names></name>, <name><surname>Leise</surname><given-names>T</given-names></name>, <name><surname>Roenneberg</surname><given-names>T</given-names></name>. <article-title>Dynamics and Ultradian Structure of Human Sleep in Real Life</article-title>. <source>Curr Biol</source>. <year>2018</year>;<volume>28</volume>(<issue>1</issue>):<fpage>49</fpage>&#x02013;<lpage>59.e5</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.cub.2017.11.063</pub-id><pub-id pub-id-type="pmid">29290561</pub-id>
</mixed-citation></ref></ref-list></back><floats-group><fig position="float" id="F1"><label>Fig. 1.</label><caption><p id="P36">Hyperparameter tuning and k-fold cross-validation to optimize XGBoost model parameters.</p></caption><graphic xlink:href="nihms-2047577-f0001" position="float"/></fig><fig position="float" id="F2"><label>Fig. 2.</label><caption><p id="P37">Performance measures associated with different probability thresholds for non-wear classification. The optimal probability threshold (.55) was selected since, at this probability, the best tradeoff between NPV and specificity was achieved.</p></caption><graphic xlink:href="nihms-2047577-f0002" position="float"/></fig><fig position="float" id="F3"><label>Fig. 3.</label><caption><p id="P38">The SHAP value plot is arranged according to the feature importance. Also, the SHAP values show the combined impact of each feature on the probability of wear and non-wear. Higher absolute SHAP values, defined as feature importance, conveying greater changes &#x02013; either increases (in red) or decreases (in blue) in wear classification probability, are arranged in decreasing order from top to bottom according to their absolute SHAP value.</p></caption><graphic xlink:href="nihms-2047577-f0003" position="float"/></fig><fig position="float" id="F4"><label>Fig 4.</label><caption><p id="P39">Each epoch interval represents a non-wear episode of certain time period. Here [10,30) represents non-wear episodes ranging from 5 mins to 15 mins. The Mean and Median provide the model classification accuracy on these epoch interval bins. On the 2<sup>nd</sup> y-axis shows the number of epochs in each epoch intervals.</p></caption><graphic xlink:href="nihms-2047577-f0004" position="float"/></fig><fig position="float" id="F5"><label>Fig. 5.</label><caption><p id="P40">Two data samples over one week showcase the comparison between methodology-based prediction and ground truth.</p></caption><graphic xlink:href="nihms-2047577-f0005" position="float"/></fig><table-wrap position="float" id="T1"><label>Table 1.</label><caption><p id="P41">Parameter Search Space</p></caption><table frame="hsides" rules="none"><colgroup span="1"><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/></colgroup><thead><tr><th align="left" valign="top" rowspan="1" colspan="1">Parameter</th><th align="left" valign="top" rowspan="1" colspan="1">Range</th></tr><tr><th colspan="2" align="left" valign="top" rowspan="1">
<hr/>
</th></tr></thead><tbody><tr><td align="left" valign="top" rowspan="1" colspan="1">Maximum Bin</td><td align="left" valign="top" rowspan="1" colspan="1">10&#x02013;500</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">Colsample_bytree</td><td align="left" valign="top" rowspan="1" colspan="1">0.01&#x02013;1</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">Learning rate</td><td align="left" valign="top" rowspan="1" colspan="1">0.00001&#x02013;1</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">Maximum Depth</td><td align="left" valign="top" rowspan="1" colspan="1">1&#x02013;40</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">Reg_alpha</td><td align="left" valign="top" rowspan="1" colspan="1">0&#x02013;10</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">N_estimators</td><td align="left" valign="top" rowspan="1" colspan="1">0&#x02013;400</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">gamma</td><td align="left" valign="top" rowspan="1" colspan="1">0&#x02013;15</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">eta</td><td align="left" valign="top" rowspan="1" colspan="1">0.01&#x02013;1</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">Reg_lambda</td><td align="left" valign="top" rowspan="1" colspan="1">0&#x02013;10</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">Rate Drop</td><td align="left" valign="top" rowspan="1" colspan="1">0&#x02013;1</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">Scale pos weight</td><td align="left" valign="top" rowspan="1" colspan="1">0.05&#x02013;1</td></tr></tbody></table></table-wrap><table-wrap position="float" id="T2"><label>Table 2.</label><caption><p id="P42">Optimal parameter values</p></caption><table frame="hsides" rules="none"><colgroup span="1"><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/></colgroup><thead><tr><th align="left" valign="top" rowspan="1" colspan="1">Parameters</th><th align="left" valign="top" rowspan="1" colspan="1">Value</th><th align="left" valign="top" rowspan="1" colspan="1">Parameter Purpose</th></tr><tr><th colspan="3" align="left" valign="top" rowspan="1">
<hr/>
</th></tr></thead><tbody><tr><td align="left" valign="top" rowspan="1" colspan="1">tree_method</td><td align="left" valign="top" rowspan="1" colspan="1">Hist</td><td align="left" valign="top" rowspan="1" colspan="1">Can handle larger datasets and has been optimized for speed</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">booster</td><td align="left" valign="top" rowspan="1" colspan="1">gbtree</td><td align="left" valign="top" rowspan="1" colspan="1">The gbtree method is more suitable to fit nonlinear boundaries</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">normalize_type</td><td align="left" valign="top" rowspan="1" colspan="1">Forest</td><td align="left" valign="top" rowspan="1" colspan="1">Stabilizes learning</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">device</td><td align="left" valign="top" rowspan="1" colspan="1">Cuda</td><td align="left" valign="top" rowspan="1" colspan="1">The model can be trained on GPU</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">sampling_method</td><td align="left" valign="top" rowspan="1" colspan="1">uniform</td><td align="left" valign="top" rowspan="1" colspan="1">No bias in selecting the samples for each tree</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">colsample_bytree</td><td align="left" valign="top" rowspan="1" colspan="1">0.2674</td><td align="left" valign="top" rowspan="1" colspan="1">Low value reduces the risk of overfitting</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">gamma</td><td align="left" valign="top" rowspan="1" colspan="1">12.7667</td><td align="left" valign="top" rowspan="1" colspan="1">High value reduces overfitting by reducing tree complexity</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">learning_rate</td><td align="left" valign="top" rowspan="1" colspan="1">0.0083</td><td align="left" valign="top" rowspan="1" colspan="1">Lower learning rate helps in getting optimal loss</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">max_bin</td><td align="left" valign="top" rowspan="1" colspan="1">489</td><td align="left" valign="top" rowspan="1" colspan="1">Buckets continuous features to make a detailed yet computationally efficient representation of features</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">max_depth</td><td align="left" valign="top" rowspan="1" colspan="1">19</td><td align="left" valign="top" rowspan="1" colspan="1">A deeper tree can better learn the hidden patterns</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">n_estimators</td><td align="left" valign="top" rowspan="1" colspan="1">91</td><td align="left" valign="top" rowspan="1" colspan="1">Denotes the number of trees and a higher number can avoid underfitting</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">rate_drop</td><td align="left" valign="top" rowspan="1" colspan="1">0.4`</td><td align="left" valign="top" rowspan="1" colspan="1">Makes the model more robust</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">reg_alpha</td><td align="left" valign="top" rowspan="1" colspan="1">5.44</td><td align="left" valign="top" rowspan="1" colspan="1">Makes the weights sparse</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">reg_lambda</td><td align="left" valign="top" rowspan="1" colspan="1">1.51</td><td align="left" valign="top" rowspan="1" colspan="1">Smoothens the impact of weights</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">scale_pos_weight</td><td align="left" valign="top" rowspan="1" colspan="1">0.157</td><td align="left" valign="top" rowspan="1" colspan="1">Gives more weight for the negative class in the loss function</td></tr></tbody></table></table-wrap><table-wrap position="float" id="T3"><label>Table 3.</label><caption><p id="P43">Classification metrics on the training and test set.</p></caption><table frame="hsides" rules="none"><colgroup span="1"><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/></colgroup><thead><tr><th align="left" valign="top" rowspan="1" colspan="1"/><th align="left" valign="top" rowspan="1" colspan="1">AUC</th><th align="left" valign="top" rowspan="1" colspan="1">Sensitivity</th><th align="left" valign="top" rowspan="1" colspan="1">Specificity</th><th align="left" valign="top" rowspan="1" colspan="1">Balanced Accuracy</th><th align="left" valign="top" rowspan="1" colspan="1">F1 score</th><th align="left" valign="top" rowspan="1" colspan="1">PPV</th><th align="left" valign="top" rowspan="1" colspan="1">NPV</th></tr><tr><th colspan="8" align="left" valign="top" rowspan="1">
<hr/>
</th></tr></thead><tbody><tr><td align="left" valign="top" rowspan="1" colspan="1">Training dataset</td><td align="left" valign="top" rowspan="1" colspan="1">0.981</td><td align="left" valign="top" rowspan="1" colspan="1">0.991</td><td align="left" valign="top" rowspan="1" colspan="1">0.771</td><td align="left" valign="top" rowspan="1" colspan="1">0.881</td><td align="left" valign="top" rowspan="1" colspan="1">0.986</td><td align="left" valign="top" rowspan="1" colspan="1">0.981</td><td align="left" valign="top" rowspan="1" colspan="1">0.879</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">Test dataset</td><td align="left" valign="top" rowspan="1" colspan="1">0.937</td><td align="left" valign="top" rowspan="1" colspan="1">0.990</td><td align="left" valign="top" rowspan="1" colspan="1">0.760</td><td align="left" valign="top" rowspan="1" colspan="1">0.875</td><td align="left" valign="top" rowspan="1" colspan="1">0.985</td><td align="left" valign="top" rowspan="1" colspan="1">0.981</td><td align="left" valign="top" rowspan="1" colspan="1">0.855</td></tr></tbody></table></table-wrap><table-wrap position="float" id="T4"><label>Table 4.</label><caption><p id="P44">Comparison Study on the Improved XGBoost methodology and the threshold Method</p></caption><table frame="hsides" rules="none"><colgroup span="1"><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/></colgroup><thead><tr><th rowspan="2" align="left" valign="middle" colspan="1"/><th colspan="2" align="center" valign="middle" rowspan="1">XGBoost with Probability Threshold</th><th colspan="2" align="center" valign="middle" rowspan="1">Threshold Method</th></tr><tr><th colspan="5" align="left" valign="top" rowspan="1">
<hr/>
</th></tr><tr><th align="left" valign="middle" rowspan="1" colspan="1"/><th align="left" valign="middle" rowspan="1" colspan="1">Train set</th><th align="left" valign="middle" rowspan="1" colspan="1">Test set</th><th align="left" valign="middle" rowspan="1" colspan="1">Train set</th><th align="left" valign="middle" rowspan="1" colspan="1">Test set</th></tr><tr><th colspan="5" align="left" valign="top" rowspan="1">
<hr/>
</th></tr></thead><tbody><tr><td align="left" valign="middle" rowspan="1" colspan="1">Sensitivity</td><td align="left" valign="middle" rowspan="1" colspan="1">0.985</td><td align="left" valign="middle" rowspan="1" colspan="1">0.983</td><td align="left" valign="middle" rowspan="1" colspan="1">0.957</td><td align="left" valign="middle" rowspan="1" colspan="1">0.947</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">Specificity</td><td align="left" valign="middle" rowspan="1" colspan="1">0.822</td><td align="left" valign="middle" rowspan="1" colspan="1">0.791</td><td align="left" valign="middle" rowspan="1" colspan="1">0.104</td><td align="left" valign="middle" rowspan="1" colspan="1">0.117</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">Balanced Accuracy</td><td align="left" valign="middle" rowspan="1" colspan="1">0.904</td><td align="left" valign="middle" rowspan="1" colspan="1">0.887</td><td align="left" valign="middle" rowspan="1" colspan="1">0.526</td><td align="left" valign="middle" rowspan="1" colspan="1">0.532</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">F1 score</td><td align="left" valign="middle" rowspan="1" colspan="1">0.985</td><td align="left" valign="middle" rowspan="1" colspan="1">0.983</td><td align="left" valign="middle" rowspan="1" colspan="1">0.965</td><td align="left" valign="middle" rowspan="1" colspan="1">0.967</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">PPV</td><td align="left" valign="middle" rowspan="1" colspan="1">0.985</td><td align="left" valign="middle" rowspan="1" colspan="1">0.983</td><td align="left" valign="middle" rowspan="1" colspan="1">0.984</td><td align="left" valign="middle" rowspan="1" colspan="1">0.989</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">NPV</td><td align="left" valign="middle" rowspan="1" colspan="1">0.824</td><td align="left" valign="middle" rowspan="1" colspan="1">0.783</td><td align="left" valign="middle" rowspan="1" colspan="1">0.032</td><td align="left" valign="middle" rowspan="1" colspan="1">0.025</td></tr></tbody></table></table-wrap><table-wrap position="float" id="T5"><label>Table 5.</label><caption><p id="P45">Prediction of Invalid days on the train and test set</p></caption><table frame="hsides" rules="none"><colgroup span="1"><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/></colgroup><thead><tr><th align="left" valign="middle" rowspan="1" colspan="1"/><th colspan="2" align="center" valign="middle" rowspan="1">Train set</th><th colspan="2" align="center" valign="middle" rowspan="1">Test Set</th></tr><tr><th align="left" valign="middle" rowspan="1" colspan="1"/><th colspan="4" align="left" valign="top" rowspan="1">
<hr/>
</th></tr><tr><th align="left" valign="middle" rowspan="1" colspan="1"/><th align="left" valign="middle" rowspan="1" colspan="1">Predicted Invalid</th><th align="left" valign="middle" rowspan="1" colspan="1">Predicted Valid</th><th align="left" valign="middle" rowspan="1" colspan="1">Predicted Invalid</th><th align="left" valign="middle" rowspan="1" colspan="1">Predicted Valid</th></tr><tr><th align="left" valign="middle" rowspan="1" colspan="1"/><th colspan="4" align="left" valign="top" rowspan="1">
<hr/>
</th></tr></thead><tbody><tr><td align="left" valign="middle" rowspan="1" colspan="1">True Invalid</td><td align="left" valign="middle" rowspan="1" colspan="1">389</td><td align="left" valign="middle" rowspan="1" colspan="1">33</td><td align="left" valign="middle" rowspan="1" colspan="1">156</td><td align="left" valign="middle" rowspan="1" colspan="1">18</td></tr><tr><td align="left" valign="middle" rowspan="1" colspan="1">True Valid</td><td align="left" valign="middle" rowspan="1" colspan="1">8</td><td align="left" valign="middle" rowspan="1" colspan="1">3457</td><td align="left" valign="middle" rowspan="1" colspan="1">1</td><td align="left" valign="middle" rowspan="1" colspan="1">1505</td></tr></tbody></table></table-wrap><table-wrap position="float" id="T6"><label>Table 6.</label><caption><p id="P46">Classification metrics for predicting invalid days</p></caption><table frame="hsides" rules="none"><colgroup span="1"><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/><col align="left" valign="middle" span="1"/></colgroup><thead><tr><th align="left" valign="top" rowspan="1" colspan="1"/><th align="left" valign="top" rowspan="1" colspan="1">Sensitivity</th><th align="left" valign="top" rowspan="1" colspan="1">Specificity</th><th align="left" valign="top" rowspan="1" colspan="1">Balanced Accuracy</th><th align="left" valign="top" rowspan="1" colspan="1">F1 score</th><th align="left" valign="top" rowspan="1" colspan="1">PPV</th><th align="left" valign="top" rowspan="1" colspan="1">NPV</th></tr><tr><th colspan="7" align="left" valign="top" rowspan="1">
<hr/>
</th></tr></thead><tbody><tr><td align="left" valign="top" rowspan="1" colspan="1">Training dataset</td><td align="left" valign="top" rowspan="1" colspan="1">0.921</td><td align="left" valign="top" rowspan="1" colspan="1">0.997</td><td align="left" valign="top" rowspan="1" colspan="1">0.959</td><td align="left" valign="top" rowspan="1" colspan="1">0.949</td><td align="left" valign="top" rowspan="1" colspan="1">0.979</td><td align="left" valign="top" rowspan="1" colspan="1">0.990</td></tr><tr><td align="left" valign="top" rowspan="1" colspan="1">Test dataset</td><td align="left" valign="top" rowspan="1" colspan="1">0.896</td><td align="left" valign="top" rowspan="1" colspan="1">0.999</td><td align="left" valign="top" rowspan="1" colspan="1">0.944</td><td align="left" valign="top" rowspan="1" colspan="1">0.942</td><td align="left" valign="top" rowspan="1" colspan="1">0.993</td><td align="left" valign="top" rowspan="1" colspan="1">0.988</td></tr></tbody></table></table-wrap></floats-group></article>