diff --git a/main.bib b/main.bib index aef42c3..bb258d6 100644 --- a/main.bib +++ b/main.bib @@ -1,564 +1,229 @@ -@misc{taylor_forecasting_2017, - title = {Forecasting at scale}, - copyright = {http://creativecommons.org/licenses/by/4.0/}, - url = {https://peerj.com/preprints/3190v2}, - doi = {10.7287/peerj.preprints.3190v2}, - abstract = {Forecasting is a common data science task that helps organizations with capacity planning, goal setting, and anomaly detection. Despite its importance, there are serious challenges associated with producing reliable and high quality forecasts –especially when there are a variety of time series and analysts with expertise in time series modeling are relatively rare. To address these challenges, we describe a practical approach to forecasting “at scale” that combines configurable models with analyst-in-the-loop performance analysis. We propose a modular regression model with interpretable parameters that can be intuitively adjusted by analysts with domain knowledge about the time series. We describe performance analyses to compare and evaluate forecasting procedures, and automatically flag forecasts for manual review and adjustment. Tools that help analysts to use their expertise most effectively enable reliable, practical forecasting of business time series.}, - language = {en}, - urldate = {2024-10-14}, - publisher = {PeerJ Preprints}, - author = {Taylor, Sean J and Letham, Benjamin}, - month = sep, - year = {2017}, - file = {PDF:/home/alex/Zotero/storage/GK5AIG2V/Taylor and Letham - 2017 - Forecasting at scale.pdf:application/pdf}, +@article{dunson_day-specific_1999, + title = {Day-specific probabilities of clinical pregnancy based on two studies with imperfect measures of ovulation}, + volume = {14}, + issn = {1460-2350, 0268-1161}, + url = {https://academic.oup.com/humrep/article-lookup/doi/10.1093/humrep/14.7.1835}, + doi = {10.1093/humrep/14.7.1835}, + pages = {1835--1839}, + number = {7}, + journaltitle = {Human Reproduction}, + author = {Dunson, D.B. and Baird, D.D. and Wilcox, A.J. and Weinberg, C.R.}, + urldate = {2025-03-05}, + date = {1999-07}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/8DE3ZLPJ/Dunson et al. - 1999 - Day-specific probabilities of clinical pregnancy based on two studies with imperfect measures of ovu.pdf:application/pdf}, } -@book{hutter_machine_2021, - address = {Cham}, - series = {Lecture {Notes} in {Computer} {Science}}, - title = {Machine {Learning} and {Knowledge} {Discovery} in {Databases}: {European} {Conference}, {ECML} {PKDD} 2020, {Ghent}, {Belgium}, {September} 14–18, 2020, {Proceedings}, {Part} {III}}, - volume = {12459}, - copyright = {https://www.springernature.com/gp/researchers/text-and-data-mining}, - isbn = {978-3-030-67663-6 978-3-030-67664-3}, - shorttitle = {Machine {Learning} and {Knowledge} {Discovery} in {Databases}}, - url = {https://link.springer.com/10.1007/978-3-030-67664-3}, - language = {en}, - urldate = {2024-10-14}, - publisher = {Springer International Publishing}, - editor = {Hutter, Frank and Kersting, Kristian and Lijffijt, Jefrey and Valera, Isabel}, - year = {2021}, - doi = {10.1007/978-3-030-67664-3}, - file = {Submitted Version:/home/alex/Zotero/storage/JMVJMLJ5/Hutter et al. - 2021 - Machine Learning and Knowledge Discovery in Databases European Conference, ECML PKDD 2020, Ghent, B.pdf:application/pdf}, +@article{papacharalampous_predictability_2018, + title = {Predictability of monthly temperature and precipitation using automatic time series forecasting methods}, + volume = {66}, + issn = {1895-7455}, + url = {https://doi.org/10.1007/s11600-018-0120-7}, + doi = {10.1007/s11600-018-0120-7}, + abstract = {We investigate the predictability of monthly temperature and precipitation by applying automatic univariate time series forecasting methods to a sample of 985 40-year-long monthly temperature and 1552 40-year-long monthly precipitation time series. The methods include a naïve one based on the monthly values of the last year, as well as the random walk (with drift), {AutoRegressive} Fractionally Integrated Moving Average ({ARFIMA}), exponential smoothing state-space model with Box–Cox transformation, {ARMA} errors, Trend and Seasonal components ({BATS}), simple exponential smoothing, Theta and Prophet methods. Prophet is a recently introduced model inspired by the nature of time series forecasted at Facebook and has not been applied to hydrometeorological time series before, while the use of random walk, {BATS}, simple exponential smoothing and Theta is rare in hydrology. The methods are tested in performing multi-step ahead forecasts for the last 48 months of the data. We further investigate how different choices of handling the seasonality and non-normality affect the performance of the models. The results indicate that: (a) all the examined methods apart from the naïve and random walk ones are accurate enough to be used in long-term applications; (b) monthly temperature and precipitation can be forecasted to a level of accuracy which can barely be improved using other methods; (c) the externally applied classical seasonal decomposition results mostly in better forecasts compared to the automatic seasonal decomposition used by the {BATS} and Prophet methods; and (d) Prophet is competitive, especially when it is combined with externally applied classical seasonal decomposition.}, + pages = {807--831}, + number = {4}, + journaltitle = {Acta Geophys.}, + author = {Papacharalampous, Georgia and Tyralis, Hristos and Koutsoyiannis, Demetris}, + urldate = {2025-03-04}, + date = {2018-08-01}, + langid = {english}, + keywords = {{ARFIMA}, Multi-step ahead forecasting, Precipitation forecasting, Prophet, Temperature forecasting, Time series forecasting}, } -@incollection{hutter_general_2021, - address = {Cham}, - title = {A {General} {Machine} {Learning} {Framework} for {Survival} {Analysis}}, - volume = {12459}, - isbn = {978-3-030-67663-6 978-3-030-67664-3}, - url = {https://link.springer.com/10.1007/978-3-030-67664-3_10}, - language = {en}, - urldate = {2024-10-14}, - booktitle = {Machine {Learning} and {Knowledge} {Discovery} in {Databases}}, - publisher = {Springer International Publishing}, - author = {Bender, Andreas and Rügamer, David and Scheipl, Fabian and Bischl, Bernd}, - editor = {Hutter, Frank and Kersting, Kristian and Lijffijt, Jefrey and Valera, Isabel}, - year = {2021}, - doi = {10.1007/978-3-030-67664-3_10}, +@inproceedings{habib_n-beats_2025, + location = {Cham}, + title = {N-{BEATS} \& Temporal Fusion Transformer Based Surface Temperature Prediction and Forecasting for Realizing Global Warming Trends}, + isbn = {978-3-031-75167-7}, + doi = {10.1007/978-3-031-75167-7_3}, + abstract = {At the pinnacle of civilization, where the impacts of climate change have been increasingly felt, weather prediction plays a critical role in mitigating the potential disasters that may arise. Moreover, with the gradual change on climate, surface temperature of the earth is increasing. This increasing rate of the surface temperature causing global warming which is a matter of intimidation. To leave off this global warming, weather forecasting can be used as an arsenal. Selecting the appropriate tools and models for weather prediction is a crucial step in ensuring accurate forecasts. In this research paper, the focus was on studying the versatility of three specific architectures for weather prediction: {LSTM}, Temporal Fusion Transformer, and N-{BEATS}. To assess these architectures’ performance, we conducted a number of experiments. With the lowest Mean Absolute Error ({MAE}) and Root Mean Square Error ({RMSE}) of the three, {NBEATS} stood out. This shows that when compared to the other models, the N-{BEATS} architecture had greater prediction accuracy. It's vital to remember, too, that the trials also showed that the Temporal Fusion Transformer and {LSTM} performed well. The only distinction was that these models required larger sizes in terms of parameters and computational complexity to achieve their performance levels. Consequently, considering both performance and model size, the researchers determined that N-{BEATS} was the most optimal and versatile architecture for weather prediction. Its ability to achieve excellent results with a smaller model size makes it a favorable choice for practical applications.}, + pages = {30--41}, + booktitle = {Artificial Intelligence and Speech Technology}, + publisher = {Springer Nature Switzerland}, + author = {Habib, Adria Binte and Ashraf, Faisal Bin and Hossain, Muhammad Iqbal and Alam, Golam Rabiul}, + editor = {Dev, Amita and Sharma, Arun and Agrawal, S. S. and Rani, Ritu}, + date = {2025}, + langid = {english}, + keywords = {{LSTM}, N-{BEATS}, Temporal Fusion Transformer, Time Series Analysis, Weather Prediction}, +} + +@article{wu_interpretable_2022, + title = {Interpretable wind speed prediction with multivariate time series and temporal fusion transformers}, + volume = {252}, + issn = {03605442}, + url = {https://linkinghub.elsevier.com/retrieve/pii/S0360544222008933}, + doi = {10.1016/j.energy.2022.123990}, + abstract = {Wind power has been utilized well in power systems, so steady and successful wind speed forecasting is crucial to security management power grid market economy. To date, most researchers have often discounted the interpretability of prediction models, leading to obscure forecasts. This study puts forward a unique forecasting methodology that incorporates notable decomposition techniques, multifactor interpretable forecasting models, and optimization algorithms. In the proposed model, variational mode decomposition is employed to break down the raw wind speed sequence into a set of intrinsic mode functions. Adaptive differential evolution is then used for optimizing several parameters of temporal fusion transformers ({TFT}) to achieve satisfactory forecasting performance. {TFT} is a new attention-based deep learning model that puts together high-performance multi-horizon prediction and interpretable insights into temporal dynamics. Empirical studies using eight real-world 1-h wind speed data sets in Albert, Canada, and Five Points, {USA} demonstrate that the system using the proposed model outperforms those employing other comparable models in nearly all performance metrics. Examples of {TFT}'s interpretable outputs are the importance ranking of the decomposed wind speed sub-sequences and meteorological data and attention analysis of different step lengths. The findings signify substantial progress for wind speed prediction and aid policymakers.}, + pages = {123990}, + journaltitle = {Energy}, + author = {Wu, Binrong and Wang, Lin and Zeng, Yu-Rong}, + urldate = {2025-02-25}, + date = {2022-08}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/MHAFY926/Wu et al. - 2022 - Interpretable wind speed prediction with multivariate time series and temporal fusion transformers.pdf:application/pdf}, +} + +@article{schwenke_show_nodate, + title = {Show Me What You’re Looking For: Visualizing Abstracted Transformer Attention for Enhancing Their Local Interpretability on Time Series Data}, + abstract = {While Transformers have shown their advantages considering their learning performance, their lack of explainability and interpretability is still a major problem. This specifically relates to the processing of time series, as a specific form of complex data. In this paper, we propose an approach for visualizing abstracted information in order to enable computational sensemaking and local interpretability on the respective Transformer model. Our results demonstrate the efficacy of the proposed abstraction method and visualization, utilizing both synthetic and real world data for evaluation.}, + author = {Schwenke, Leonid and Atzmueller, Martin}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/SLVVAAXA/Schwenke and Atzmueller - Show Me What You’re Looking For Visualizing Abstracted Transformer Attention for Enhancing Their Lo.pdf:application/pdf}, +} + +@inproceedings{schwenke_constructing_2021, + title = {Constructing Global Coherence Representations: Identifying Interpretability and Coherences of Transformer Attention in Time Series Data}, + url = {https://ieeexplore.ieee.org/document/9564126/?arnumber=9564126}, + doi = {10.1109/DSAA53316.2021.9564126}, + shorttitle = {Constructing Global Coherence Representations}, + abstract = {Transformer models have shown significant advances recently based on the general concept of Attention — to focus on specifically important and relevant parts of the input data. However, methods for enhancing their interpretability and explainability are still lacking. This is the problem which we tackle in this paper, to make Multi-Headed Attention more interpretable and explainable for time series classification. We present a method for constructing global coherence representations from Multi-Headed Attention of Transformer architectures. Accordingly, we present abstraction and interpretation methods, leading to intuitive visualizations of the respective attention patterns. We evaluate our proposed approach and the presented methods on several datasets demonstrating their efficacy.}, + eventtitle = {2021 {IEEE} 8th International Conference on Data Science and Advanced Analytics ({DSAA})}, + pages = {1--12}, + booktitle = {2021 {IEEE} 8th International Conference on Data Science and Advanced Analytics ({DSAA})}, + author = {Schwenke, Leonid and Atzmueller, Martin}, + urldate = {2025-02-25}, + date = {2021-10}, + keywords = {Attention, Coherence, Comprehensibility, Conferences, Data science, Data visualization, Deep Learning, Explainability, Global Class Representation, Interpretability, Scalability, Time series analysis, Time Series Classification, Transformer, Transformers, Visualization}, + file = {Full Text PDF:/home/alex/Zotero/storage/VRGIDFY8/Schwenke and Atzmueller - 2021 - Constructing Global Coherence Representations Identifying Interpretability and Coherences of Transf.pdf:application/pdf;IEEE Xplore Abstract Record:/home/alex/Zotero/storage/Z45R3XWF/9564126.html:text/html}, +} + +@incollection{iliadis_temporal_2023, + location = {Cham}, + title = {Temporal Attention Signatures for Interpretable Time-Series Prediction}, + volume = {14259}, + isbn = {978-3-031-44222-3 978-3-031-44223-0}, + url = {https://link.springer.com/10.1007/978-3-031-44223-0_22}, + abstract = {Deep neural networks have become a staple in time-series prediction due to their remarkable accuracy. However, their internal workings often remain elusive. Significant advancements have been made in the interpretability of these networks, with attention mechanisms and feature maps being notably effective for image classification by highlighting the crucial data points. While human observers can readily confirm the significance of features in image classification, the interpretability of time-series data and its modeling remains challenging. To address this, we put forth an innovative approach that unifies temporal attention and visualization as a blend of recurrent neural networks, self-attention, and general attention. This synergy results in the generation of temporal attention signatures, akin to image attention heat maps. Temporal attention not only enhances prediction accuracy beyond that of recurrent networks alone but also demonstrates that varying label classes yield distinct attention signatures. This observation indicates that neural networks focus on different sections of time-series sequences contingent on the prediction target. We conclude with a discussion on the practical implications of this novel approach, including its applicability to model interpretation, sequence length selection, and model validation. This leads to more accurate, robust, and interpretable models, instilling greater confidence in their results.}, + pages = {268--280}, + booktitle = {Artificial Neural Networks and Machine Learning – {ICANN} 2023}, + publisher = {Springer Nature Switzerland}, + author = {Katrompas, Alexander and Metsis, Vangelis}, + editor = {Iliadis, Lazaros and Papaleonidas, Antonios and Angelov, Plamen and Jayne, Chrisina}, + urldate = {2025-02-25}, + date = {2023}, + langid = {english}, + doi = {10.1007/978-3-031-44223-0_22}, note = {Series Title: Lecture Notes in Computer Science}, - pages = {158--173}, - file = {Submitted Version:/home/alex/Zotero/storage/WQIHZ7IP/Bender et al. - 2021 - A General Machine Learning Framework for Survival Analysis.pdf:application/pdf}, + file = {PDF:/home/alex/Zotero/storage/LV7IVKZK/Katrompas and Metsis - 2023 - Temporal Attention Signatures for Interpretable Time-Series Prediction.pdf:application/pdf}, } -@misc{lightningai_pytorch_2024, - title = {{PyTorch} {Lightning}}, - url = {https://www.pytorchlightning.ai}, - urldate = {2024-10-14}, - author = {lightning.ai}, - year = {2024}, +@inproceedings{guo_exploring_2019, + title = {Exploring interpretable {LSTM} neural networks over multi-variable data}, + url = {https://proceedings.mlr.press/v97/guo19b.html}, + abstract = {For recurrent neural networks trained on time series with target and exogenous variables, in addition to accurate prediction, it is also desired to provide interpretable insights into the data. In this paper, we explore the structure of {LSTM} recurrent neural networks to learn variable-wise hidden states, with the aim to capture different dynamics in multi-variable time series and distinguish the contribution of variables to the prediction. With these variable-wise hidden states, a mixture attention mechanism is proposed to model the generative process of the target. Then we develop associated training methods to jointly learn network parameters, variable and temporal importance w.r.t the prediction of the target variable. Extensive experiments on real datasets demonstrate enhanced prediction performance by capturing the dynamics of different variables. Meanwhile, we evaluate the interpretation results both qualitatively and quantitatively. It exhibits the prospect as an end-to-end framework for both forecasting and knowledge extraction over multi-variable data.}, + eventtitle = {International Conference on Machine Learning}, + pages = {2494--2504}, + booktitle = {Proceedings of the 36th International Conference on Machine Learning}, + publisher = {{PMLR}}, + author = {Guo, Tian and Lin, Tao and Antulov-Fantulin, Nino}, + urldate = {2025-02-25}, + date = {2019-05-24}, + langid = {english}, + note = {{ISSN}: 2640-3498}, + file = {Full Text PDF:/home/alex/Zotero/storage/VV3I2T4E/Guo et al. - 2019 - Exploring interpretable LSTM neural networks over multi-variable data.pdf:application/pdf;Supplementary PDF:/home/alex/Zotero/storage/57IK29PA/Guo et al. - 2019 - Exploring interpretable LSTM neural networks over multi-variable data.pdf:application/pdf}, } -@misc{alexandrov_gluonts_2019, - title = {{GluonTS}: {Probabilistic} {Time} {Series} {Models} in {Python}}, - shorttitle = {{GluonTS}}, - url = {http://arxiv.org/abs/1906.05264}, - abstract = {We introduce Gluon Time Series (GluonTS, available at https://gluon-ts.mxnet.io), a library for deep-learning-based time series modeling. GluonTS simplifies the development of and experimentation with time series models for common tasks such as forecasting or anomaly detection. It provides all necessary components and tools that scientists need for quickly building new models, for efficiently running and analyzing experiments and for evaluating model accuracy.}, - urldate = {2024-10-14}, - publisher = {arXiv}, - author = {Alexandrov, Alexander and Benidis, Konstantinos and Bohlke-Schneider, Michael and Flunkert, Valentin and Gasthaus, Jan and Januschowski, Tim and Maddix, Danielle C. and Rangapuram, Syama and Salinas, David and Schulz, Jasper and Stella, Lorenzo and Türkmen, Ali Caner and Wang, Yuyang}, - month = jun, - year = {2019}, - note = {arXiv:1906.05264}, - keywords = {Computer Science - Machine Learning, Statistics - Machine Learning}, - file = {Preprint PDF:/home/alex/Zotero/storage/JP9K74A8/Alexandrov et al. - 2019 - GluonTS Probabilistic Time Series Models in Python.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/RJYSBT29/1906.html:text/html}, -} - -@misc{cho_learning_2014, - title = {Learning {Phrase} {Representations} using {RNN} {Encoder}-{Decoder} for {Statistical} {Machine} {Translation}}, - url = {http://arxiv.org/abs/1406.1078}, - abstract = {In this paper, we propose a novel neural network model called RNN Encoder-Decoder that consists of two recurrent neural networks (RNN). One RNN encodes a sequence of symbols into a fixed-length vector representation, and the other decodes the representation into another sequence of symbols. The encoder and decoder of the proposed model are jointly trained to maximize the conditional probability of a target sequence given a source sequence. The performance of a statistical machine translation system is empirically found to improve by using the conditional probabilities of phrase pairs computed by the RNN Encoder-Decoder as an additional feature in the existing log-linear model. Qualitatively, we show that the proposed model learns a semantically and syntactically meaningful representation of linguistic phrases.}, - urldate = {2024-10-10}, - publisher = {arXiv}, - author = {Cho, Kyunghyun and Merrienboer, Bart van and Gulcehre, Caglar and Bahdanau, Dzmitry and Bougares, Fethi and Schwenk, Holger and Bengio, Yoshua}, - month = sep, - year = {2014}, - note = {arXiv:1406.1078}, - keywords = {Computer Science - Computation and Language, Computer Science - Machine Learning, Statistics - Machine Learning, Computer Science - Neural and Evolutionary Computing}, - file = {Preprint PDF:/home/alex/Zotero/storage/E8WMK2IN/Cho et al. - 2014 - Learning Phrase Representations using RNN Encoder-Decoder for Statistical Machine Translation.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/6PTCL8LW/1406.html:text/html}, -} - -@article{hochreiter_long_1997, - title = {Long {Short}-{Term} {Memory}}, - volume = {9}, - issn = {0899-7667, 1530-888X}, - url = {https://direct.mit.edu/neco/article/9/8/1735-1780/6109}, - doi = {10.1162/neco.1997.9.8.1735}, - abstract = {Learningtostoreinformationoverextendedtimeintervalsviarecurrentbackpropagation takesaverylongtime,mostlyduetoinsu cient,decayingerrorbackow.Webrieyreview Hochreiter's1991analysisofthisproblem,thenaddressitbyintroducinganovel,e cient, gradient-basedmethodcalled{\textbackslash}LongShort-TermMemory"(LSTM).Truncatingthegradient wherethisdoesnotdoharm,LSTMcanlearntobridgeminimaltimelagsinexcessof1000 discretetimestepsbyenforcingconstanterrorowthrough{\textbackslash}constanterrorcarrousels"within specialunits.Multiplicativegateunitslearntoopenandcloseaccesstotheconstanterror ow.LSTMislocalinspaceandtime;itscomputationalcomplexitypertimestepandweight isO(1).Ourexperimentswitharticialdatainvolvelocal,distributed,real-valued,andnoisy patternrepresentations.IncomparisonswithRTRL,BPTT,RecurrentCascade-Correlation, Elmannets,andNeuralSequenceChunking,LSTMleadstomanymoresuccessfulruns,and learnsmuchfaster.LSTMalsosolvescomplex,articiallongtimelagtasksthathavenever beensolvedbypreviousrecurrentnetworkalgorithms.}, - language = {en}, - number = {8}, - urldate = {2024-10-10}, - journal = {Neural Computation}, - author = {Hochreiter, Sepp and Schmidhuber, Jürgen}, - month = nov, - year = {1997}, - pages = {1735--1780}, - file = {PDF:/home/alex/Zotero/storage/CZSV2ASE/Hochreiter and Schmidhuber - 1997 - Long Short-Term Memory.pdf:application/pdf}, -} - -@misc{lim_temporal_2020, - title = {Temporal {Fusion} {Transformers} for {Interpretable} {Multi}-horizon {Time} {Series} {Forecasting}}, - url = {http://arxiv.org/abs/1912.09363}, - abstract = {Multi-horizon forecasting problems often contain a complex mix of inputs -- including static (i.e. time-invariant) covariates, known future inputs, and other exogenous time series that are only observed historically -- without any prior information on how they interact with the target. While several deep learning models have been proposed for multi-step prediction, they typically comprise black-box models which do not account for the full range of inputs present in common scenarios. In this paper, we introduce the Temporal Fusion Transformer (TFT) -- a novel attention-based architecture which combines high-performance multi-horizon forecasting with interpretable insights into temporal dynamics. To learn temporal relationships at different scales, the TFT utilizes recurrent layers for local processing and interpretable self-attention layers for learning long-term dependencies. The TFT also uses specialized components for the judicious selection of relevant features and a series of gating layers to suppress unnecessary components, enabling high performance in a wide range of regimes. On a variety of real-world datasets, we demonstrate significant performance improvements over existing benchmarks, and showcase three practical interpretability use-cases of TFT.}, - urldate = {2024-10-10}, - publisher = {arXiv}, - author = {Lim, Bryan and Arik, Sercan O. and Loeff, Nicolas and Pfister, Tomas}, - month = sep, - year = {2020}, - note = {arXiv:1912.09363}, - keywords = {Computer Science - Machine Learning, Statistics - Machine Learning}, - file = {Preprint PDF:/home/alex/Zotero/storage/2R2H34KB/Lim et al. - 2020 - Temporal Fusion Transformers for Interpretable Multi-horizon Time Series Forecasting.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/ETDAYW36/1912.html:text/html}, -} - -@misc{nie_time_2023, - title = {A {Time} {Series} is {Worth} 64 {Words}: {Long}-term {Forecasting} with {Transformers}}, - shorttitle = {A {Time} {Series} is {Worth} 64 {Words}}, - url = {http://arxiv.org/abs/2211.14730}, - abstract = {We propose an efficient design of Transformer-based models for multivariate time series forecasting and self-supervised representation learning. It is based on two key components: (i) segmentation of time series into subseries-level patches which are served as input tokens to Transformer; (ii) channel-independence where each channel contains a single univariate time series that shares the same embedding and Transformer weights across all the series. Patching design naturally has three-fold benefit: local semantic information is retained in the embedding; computation and memory usage of the attention maps are quadratically reduced given the same look-back window; and the model can attend longer history. Our channel-independent patch time series Transformer (PatchTST) can improve the long-term forecasting accuracy significantly when compared with that of SOTA Transformer-based models. We also apply our model to self-supervised pre-training tasks and attain excellent fine-tuning performance, which outperforms supervised training on large datasets. Transferring of masked pre-trained representation on one dataset to others also produces SOTA forecasting accuracy. Code is available at: https://github.com/yuqinie98/PatchTST.}, - urldate = {2024-10-10}, - publisher = {arXiv}, - author = {Nie, Yuqi and Nguyen, Nam H. and Sinthong, Phanwadee and Kalagnanam, Jayant}, - month = mar, - year = {2023}, - note = {arXiv:2211.14730}, - keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, - file = {Preprint PDF:/home/alex/Zotero/storage/DG4ZJCWV/Nie et al. - 2023 - A Time Series is Worth 64 Words Long-term Forecasting with Transformers.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/H6XGVBY6/2211.html:text/html}, -} - -@misc{shao_exploring_2023, - title = {Exploring {Progress} in {Multivariate} {Time} {Series} {Forecasting}: {Comprehensive} {Benchmarking} and {Heterogeneity} {Analysis}}, - shorttitle = {Exploring {Progress} in {Multivariate} {Time} {Series} {Forecasting}}, - url = {http://arxiv.org/abs/2310.06119}, - abstract = {Multivariate Time Series (MTS) widely exists in real-word complex systems, such as traffic and energy systems, making their forecasting crucial for understanding and influencing these systems. Recently, deep learning-based approaches have gained much popularity for effectively modeling temporal and spatial dependencies in MTS, specifically in Long-term Time Series Forecasting (LTSF) and Spatial-Temporal Forecasting (STF). However, the fair benchmarking issue and the choice of technical approaches have been hotly debated in related work. Such controversies significantly hinder our understanding of progress in this field. Thus, this paper aims to address these controversies to present insights into advancements achieved. To resolve benchmarking issues, we introduce BasicTS, a benchmark designed for fair comparisons in MTS forecasting. BasicTS establishes a unified training pipeline and reasonable evaluation settings, enabling an unbiased evaluation of over 30 popular MTS forecasting models on more than 18 datasets. Furthermore, we highlight the heterogeneity among MTS datasets and classify them based on temporal and spatial characteristics. We further prove that neglecting heterogeneity is the primary reason for generating controversies in technical approaches. Moreover, based on the proposed BasicTS and rich heterogeneous MTS datasets, we conduct an exhaustive and reproducible performance and efficiency comparison of popular models, providing insights for researchers in selecting and designing MTS forecasting models.}, - urldate = {2024-10-10}, - publisher = {arXiv}, - author = {Shao, Zezhi and Wang, Fei and Xu, Yongjun and Wei, Wei and Yu, Chengqing and Zhang, Zhao and Yao, Di and Jin, Guangyin and Cao, Xin and Cong, Gao and Jensen, Christian S. and Cheng, Xueqi}, - month = oct, - year = {2023}, - note = {arXiv:2310.06119}, - keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, - file = {Preprint PDF:/home/alex/Zotero/storage/7EFZ5IT6/Shao et al. - 2023 - Exploring Progress in Multivariate Time Series Forecasting Comprehensive Benchmarking and Heterogen.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/W6RNWBLM/2310.html:text/html}, -} - -@article{zhang_crossformer_2023, - title = {{CROSSFORMER}: {TRANSFORMER} {UTILIZING} {CROSS}- {DIMENSION} {DEPENDENCY} {FOR} {MULTIVARIATE} {TIME} {SERIES} {FORECASTING}}, - abstract = {Recently many deep models have been proposed for multivariate time series (MTS) forecasting. In particular, Transformer-based models have shown great potential because they can capture long-term dependency. However, existing Transformerbased models mainly focus on modeling the temporal dependency (cross-time dependency) yet often omit the dependency among different variables (crossdimension dependency), which is critical for MTS forecasting. To fill the gap, we propose Crossformer, a Transformer-based model utilizing cross-dimension dependency for MTS forecasting. In Crossformer, the input MTS is embedded into a 2D vector array through the Dimension-Segment-Wise (DSW) embedding to preserve time and dimension information. Then the Two-Stage Attention (TSA) layer is proposed to efficiently capture the cross-time and cross-dimension dependency. Utilizing DSW embedding and TSA layer, Crossformer establishes a Hierarchical Encoder-Decoder (HED) to use the information at different scales for the final forecasting. Extensive experimental results on six real-world datasets show the effectiveness of Crossformer against previous state-of-the-arts.}, - language = {en}, - author = {Zhang, Yunhao and Yan, Junchi}, - year = {2023}, - file = {PDF:/home/alex/Zotero/storage/NM9CETJS/Zhang and Yan - 2023 - CROSSFORMER TRANSFORMER UTILIZING CROSS- DIMENSION DEPENDENCY FOR MULTIVARIATE TIME SERIES FORECAST.pdf:application/pdf}, -} - -@misc{noauthor_pytorch-forecasting_2024, - title = {Pytorch-{Forecasting}}, - url = {https://pytorch-forecasting.readthedocs.io/en/stable/}, - urldate = {2024-10-15}, - year = {2024}, -} - -@misc{liang_foundation_2024, - title = {Foundation {Models} for {Time} {Series} {Analysis}: {A} {Tutorial} and {Survey}}, - shorttitle = {Foundation {Models} for {Time} {Series} {Analysis}}, - url = {http://arxiv.org/abs/2403.14735}, - abstract = {Time series analysis stands as a focal point within the data mining community, serving as a cornerstone for extracting valuable insights crucial to a myriad of real-world applications. Recent advances in Foundation Models (FMs) have fundamentally reshaped the paradigm of model design for time series analysis, boosting various downstream tasks in practice. These innovative approaches often leverage pre-trained or fine-tuned FMs to harness generalized knowledge tailored for time series analysis. This survey aims to furnish a comprehensive and up-to-date overview of FMs for time series analysis. While prior surveys have predominantly focused on either application or pipeline aspects of FMs in time series analysis, they have often lacked an in-depth understanding of the underlying mechanisms that elucidate why and how FMs benefit time series analysis. To address this gap, our survey adopts a methodology-centric classification, delineating various pivotal elements of time-series FMs, including model architectures, pre-training techniques, adaptation methods, and data modalities. Overall, this survey serves to consolidate the latest advancements in FMs pertinent to time series analysis, accentuating their theoretical underpinnings, recent strides in development, and avenues for future exploration.}, - urldate = {2024-10-16}, - publisher = {arXiv}, - author = {Liang, Yuxuan and Wen, Haomin and Nie, Yuqi and Jiang, Yushan and Jin, Ming and Song, Dongjin and Pan, Shirui and Wen, Qingsong}, - month = jun, - year = {2024}, - note = {arXiv:2403.14735}, - keywords = {Computer Science - Machine Learning}, - file = {Preprint PDF:/home/alex/Zotero/storage/YEA28FMJ/Liang et al. - 2024 - Foundation Models for Time Series Analysis A Tutorial and Survey.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/9SGH3AIN/2403.html:text/html}, -} - -@misc{goswami_moment_2024, - title = {{MOMENT}: {A} {Family} of {Open} {Time}-series {Foundation} {Models}}, - shorttitle = {{MOMENT}}, - url = {http://arxiv.org/abs/2402.03885}, - abstract = {We introduce MOMENT, a family of open-source foundation models for general-purpose time series analysis. Pre-training large models on time series data is challenging due to (1) the absence of a large and cohesive public time series repository, and (2) diverse time series characteristics which make multi-dataset training onerous. Additionally, (3) experimental benchmarks to evaluate these models, especially in scenarios with limited resources, time, and supervision, are still in their nascent stages. To address these challenges, we compile a large and diverse collection of public time series, called the Time series Pile, and systematically tackle time series-specific challenges to unlock large-scale multi-dataset pre-training. Finally, we build on recent work to design a benchmark to evaluate time series foundation models on diverse tasks and datasets in limited supervision settings. Experiments on this benchmark demonstrate the effectiveness of our pre-trained models with minimal data and task-specific fine-tuning. Finally, we present several interesting empirical observations about large pre-trained time series models. Pre-trained models (AutonLab/MOMENT-1-large) and Time Series Pile (AutonLab/Timeseries-PILE) are available on Huggingface.}, - urldate = {2024-10-16}, - publisher = {arXiv}, - author = {Goswami, Mononito and Szafer, Konrad and Choudhry, Arjun and Cai, Yifu and Li, Shuo and Dubrawski, Artur}, - month = oct, - year = {2024}, - note = {arXiv:2402.03885}, - keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, - file = {Preprint PDF:/home/alex/Zotero/storage/QF6E6J8W/Goswami et al. - 2024 - MOMENT A Family of Open Time-series Foundation Models.pdf:application/pdf}, -} - -@misc{noauthor_decoder-only_nodate, - title = {A decoder-only foundation model for time-series forecasting}, - url = {http://research.google/blog/a-decoder-only-foundation-model-for-time-series-forecasting/}, - abstract = {Posted by Rajat Sen and Yichen Zhou, Google Research Time-series forecasting is ubiquitous in various domains, such as retail, finance, manufacturi...}, - language = {en}, - urldate = {2024-10-16}, - file = {Snapshot:/home/alex/Zotero/storage/JV9JIF73/a-decoder-only-foundation-model-for-time-series-forecasting.html:text/html}, -} - -@misc{shi_time-moe_2024, - title = {Time-{MoE}: {Billion}-{Scale} {Time} {Series} {Foundation} {Models} with {Mixture} of {Experts}}, - shorttitle = {Time-{MoE}}, - url = {http://arxiv.org/abs/2409.16040}, - abstract = {Deep learning for time series forecasting has seen significant advancements over the past decades. However, despite the success of large-scale pre-training in language and vision domains, pre-trained time series models remain limited in scale and operate at a high cost, hindering the development of larger capable forecasting models in real-world applications. In response, we introduce Time-MoE, a scalable and unified architecture designed to pre-train larger, more capable forecasting foundation models while reducing inference costs. By leveraging a sparse mixture-of-experts (MoE) design, Time-MoE enhances computational efficiency by activating only a subset of networks for each prediction, reducing computational load while maintaining high model capacity. This allows Time-MoE to scale effectively without a corresponding increase in inference costs. Time-MoE comprises a family of decoder-only transformer models that operate in an auto-regressive manner and support flexible forecasting horizons with varying input context lengths. We pre-trained these models on our newly introduced large-scale data Time-300B, which spans over 9 domains and encompassing over 300 billion time points. For the first time, we scaled a time series foundation model up to 2.4 billion parameters, achieving significantly improved forecasting precision. Our results validate the applicability of scaling laws for training tokens and model size in the context of time series forecasting. Compared to dense models with the same number of activated parameters or equivalent computation budgets, our models consistently outperform them by large margin. These advancements position Time-MoE as a state-of-the-art solution for tackling real-world time series forecasting challenges with superior capability, efficiency, and flexibility.}, - urldate = {2024-10-16}, - publisher = {arXiv}, - author = {Shi, Xiaoming and Wang, Shiyu and Nie, Yuqi and Li, Dianqi and Ye, Zhou and Wen, Qingsong and Jin, Ming}, - month = oct, - year = {2024}, - note = {arXiv:2409.16040}, - keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, - file = {Preprint PDF:/home/alex/Zotero/storage/49N63CMZ/Shi et al. - 2024 - Time-MoE Billion-Scale Time Series Foundation Models with Mixture of Experts.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/HPHN7WNJ/2409.html:text/html}, -} - -@article{coninck_dianne_2018, - title = {{DIANNE}: a modular framework for designing, training and deploying deep neural networks on heterogeneous distributed infrastructure}, - volume = {141}, - issn = {01641212}, - shorttitle = {{DIANNE}}, - url = {https://linkinghub.elsevier.com/retrieve/pii/S0164121218300487}, - doi = {10.1016/j.jss.2018.03.032}, - language = {en}, - urldate = {2024-10-23}, - journal = {Journal of Systems and Software}, - author = {Coninck, Elias De and Bohez, Steven and Leroux, Sam and Verbelen, Tim and Vankeirsbilck, Bert and Simoens, Pieter and Dhoedt, Bart}, - month = jul, - year = {2018}, - pages = {52--65}, - file = {Full Text:/home/alex/Zotero/storage/SFH656QN/Coninck et al. - 2018 - DIANNE a modular framework for designing, training and deploying deep neural networks on heterogene.pdf:application/pdf}, -} - -@misc{noauthor_keras_nodate, - title = {Keras: {Deep} {Learning} for humans}, - url = {https://keras.io/}, - urldate = {2024-10-23}, - file = {Keras\: Deep Learning for humans:/home/alex/Zotero/storage/MS4QLPRC/keras.io.html:text/html}, -} - -@article{masuda_machine_2025, - title = {Machine learning model for menstrual cycle phase classification and ovulation day detection based on sleeping heart rate under free-living conditions}, - volume = {187}, - issn = {0010-4825}, - url = {https://www.sciencedirect.com/science/article/pii/S0010482525000551}, - doi = {10.1016/j.compbiomed.2025.109705}, - abstract = {The accurate classification of menstrual cycle phases and detection of ovulation is critical for women's health management, particularly in addressing infertility, alleviating premenstrual syndrome, and preventing hormone-related disorders. However, traditional basal body temperature (BBT) measurement methods are susceptible to disruptions in sleep timing and environmental conditions, limiting practical application. This study is aimed to overcome these limitations by introducing a novel feature, heart rate at the circadian rhythm nadir (minHR), for classifying menstrual cycle phases and predicting ovulation. A machine learning model was developed using XGBoost, and data were collected under free-living conditions from 40 healthy women (18–34 years) over a maximum of three menstrual cycles. Three feature combinations— “day,” “day + minHR,” and “day + BBT”—were evaluated, and model performance was assessed using nested leave-one-group-out cross-validation. The feature “day” represents the number of days elapsed since the onset of menstruation. Participants were stratified into groups depending on high variability and low variability in sleep timing. Results demonstrated that adding minHR significantly improved luteal phase classification and ovulation day detection performance compared to “day” only. Furthermore, in participants with high variability in sleep timing, the minHR-based model outperformed the BBT-based model, significantly improving luteal phase recall and reducing ovulation day detection absolute errors by 2 d (p {\textless} 0.05). These findings highlight the robustness and practicality of the minHR-based model for menstrual cycle tracking, particularly in individuals with high variability in sleep timing. The proposed model holds great promise for personalized health management and large-scale epidemiological research.}, - urldate = {2025-02-11}, - journal = {Computers in Biology and Medicine}, - author = {Masuda, Hazuki and Okada, Shima and Shiozawa, Naruhiro and Sakaue, Yusuke and Manno, Masanobu and Makikawa, Masaaki and Isaka, Tadao}, - month = mar, - year = {2025}, - keywords = {Circadian rhythm, Heart rate, Machine learning, Menstrual cycle tracking, Ovulation day detection, Wearable sensor, XGBoost}, - pages = {109705}, - file = {ScienceDirect Snapshot:/home/alex/Zotero/storage/PC6FSQIA/S0010482525000551.html:text/html}, -} - -@article{maman_prediction_2023, - title = {Prediction of ovulation: new insight into an old challenge}, - volume = {13}, - issn = {2045-2322}, - shorttitle = {Prediction of ovulation}, - url = {https://www.ncbi.nlm.nih.gov/pmc/articles/PMC10651856/}, - doi = {10.1038/s41598-023-47241-2}, - abstract = {Ultrasound monitoring and hormonal blood testing are considered by many as an accurate method to predict ovulation time. However, uniform and validated algorithms for predicting ovulation have yet to be defined. Daily hormonal tests and transvaginal ultrasounds were recorded to develop an algorithm for ovulation prediction. The rupture of the leading ovarian follicle was a marker for ovulation day. The model was validated retrospectively on natural cycles frozen embryo transfer cycles with documented ovulation. Circulating levels of LH or its relative variation failed, by themselves, to reliably predict ovulation. Any decrease in estrogen was 100\% associated with ovulation emergence the same day or the next day. Progesterone levels {\textgreater} 2 nmol/L had low specificity to predict ovulation the next day (62.7\%), yet its sensitivity was high (91.5\%). A model for ovulation prediction, combining the three hormone levels and ultrasound was created with an accuracy of 95\% to 100\% depending on the combination of the hormone levels. Model validation showed correct ovulation prediction in 97\% of these cycles. We present an accurate ovulation prediction algorithm. The algorithm is simple and user-friendly so both reproductive endocrinologists and general practitioners can use it to benefit their patients.}, - urldate = {2025-02-11}, - journal = {Scientific Reports}, - author = {Maman, Ettie and Adashi, Eli Y. and Baum, Micha and Hourvitz, Ariel}, - month = nov, - year = {2023}, - pmid = {37968377}, - pmcid = {PMC10651856}, - pages = {20003}, - annote = {also not “really” relevant, as domain is ultrasound and hormone levels - -}, - file = {PubMed Central Full Text PDF:/home/alex/Zotero/storage/MTHDPZ5B/Maman et al. - 2023 - Prediction of ovulation new insight into an old challenge.pdf:application/pdf}, -} - -@article{luo_prediction_2025, - title = {Prediction of the fertile window and menstrual cycles with a wearable device via machine-learning algorithms}, - issn = {1472-6483}, - url = {https://www.sciencedirect.com/science/article/pii/S1472648325000021}, - doi = {10.1016/j.rbmo.2025.104795}, - abstract = {Research question -We aimed to develop fertile window and menstruation prediction algorithms through machine learning based on women's physiological parameters data collected by Huawei Band 6 pro from both regular and irregular menstruators. -Design -This was a prospective observational cohort study conducted at Obstetrics and Gynecology Hospital of Fudan University. Participants were recruited from November 2021 to September 2022. Each participant wore Huawei Band 6 pro to record wrist skin temperature (WST), heart rate (HR), heart rate variability, and respiratory rate. Algorithms were developed to predict the fertile window and menstrual cycle based on WST and HR. -Results -We included data from 270 and 84 qualified cycles with confirmed ovulations from 136 regular and 47 irregular menstruators. For regular menstruators, the prediction algorithm based on WST and HR for the fertile window had an accuracy of 85.47\%, a sensitivity of 70.07\%, a specificity of 89.77\%, and AUC of 0.869. The algorithms for menstrual first day labelling and onset within 3 days gained an accuracy of 83.6\% and 75.0\%. For irregular menstruators, the accuracy, sensitivity, specificity and AUC were 79.85\%, 42.79\%, 87.28\%, and 0.763 respectively, for fertile window prediction. The accuracy of menses labelling and prediction were 61.2\%, and 50.8\% respectively. -Conclusions -Based on WST and HR data from the wearable device, the algorithms demonstrated reliable performance in predicting the fertile window and menstruation day among regular menstruators. These algorithms also showed potential for assisting irregular menstruators in managing their cycles and planning for conception.}, - urldate = {2025-02-11}, - journal = {Reproductive BioMedicine Online}, - author = {Luo, Chuan and Su, Yun-Fei and Ren, Yun-Yun and Zhang, Qin and Li, Ran and Zhang, Qi and Li, Cheng and Hao, Yan-Hui and Zhang, An-Qi and Zhang, Hao and Huang, He-Feng and Wu, Yan-Ting}, - month = jan, - year = {2025}, - keywords = {Machine learning, Fertile window, Menstrual cycle, Natural cycle, Non-invasive wearable device, Wrist skin temperature}, - pages = {104795}, - annote = { - -need closer look, how do they determine the fertile phase? what do they test against? what model do they use? - - -}, - file = {PDF:/home/alex/Zotero/storage/YIV6T8MS/Luo et al. - 2025 - Prediction of the fertile window and menstrual cycles with a wearable device via machine-learning al.pdf:application/pdf;ScienceDirect Snapshot:/home/alex/Zotero/storage/3IJMUH82/S1472648325000021.html:text/html}, -} - -@article{yu_tracking_2022, - title = {Tracking of menstrual cycles and prediction of the fertile window via measurements of basal body temperature and heart rate as well as machine-learning algorithms}, - volume = {20}, - issn = {1477-7827}, - url = {https://doi.org/10.1186/s12958-022-00993-4}, - doi = {10.1186/s12958-022-00993-4}, - abstract = {Fertility awareness and menses prediction are important for improving fecundability and health management. Previous studies have used physiological parameters, such as basal body temperature (BBT) and heart rate (HR), to predict the fertile window and menses. However, their accuracy is far from satisfactory. Additionally, few researchers have examined irregular menstruators. Thus, we aimed to develop fertile window and menstruation prediction algorithms for both regular and irregular menstruators.}, - number = {1}, - urldate = {2025-02-11}, - journal = {Reproductive Biology and Endocrinology}, - author = {Yu, Jia-Le and Su, Yun-Fei and Zhang, Chen and Jin, Li and Lin, Xian-Hua and Chen, Lu-Ting and Huang, He-Feng and Wu, Yan-Ting}, - month = aug, - year = {2022}, - keywords = {Heart rate, Machine learning, Fertile window, Menstrual cycle, Basal body temperature, Wearable device}, - pages = {118}, - annote = { - -potentially interesting - - -also use temperature plus hormone levels and ultrasounds - - -}, - file = {Full Text PDF:/home/alex/Zotero/storage/7Z8P97UF/Yu et al. - 2022 - Tracking of menstrual cycles and prediction of the fertile window via measurements of basal body tem.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/SMRBT6EG/s12958-022-00993-4.html:text/html}, -} - -@article{luz_improved_2024, - title = {Improved clinical pregnancy rates in natural frozen-thawed embryo transfer cycles with machine learning ovulation prediction: insights from a retrospective cohort study}, - volume = {14}, - copyright = {2024 The Author(s)}, - issn = {2045-2322}, - shorttitle = {Improved clinical pregnancy rates in natural frozen-thawed embryo transfer cycles with machine learning ovulation prediction}, - url = {https://www.nature.com/articles/s41598-024-80356-8}, - doi = {10.1038/s41598-024-80356-8}, - abstract = {This study aims to develop physician support software for determining ovulation time and assess its impact on pregnancy outcomes in natural cycle frozen embryo transfers (NC-FET). To develop, assess, and validate an ovulation prediction model, three datasets were used: REI Ovulation Determination dataset (500 cycles) split into training (309), validation (90), and test (101) sets; the Documented Ovulation dataset (101 cycles) with confirmed ovulation (documented follicular rupture and LH surge); and the Clinical Pregnancy Rates dataset (515 NC-FET cycles), categorized into “Matched” and “Mismatched” based on alignment with the model’s ovulation determination. Pregnancy outcomes were compared between the groups. The ovulation prediction model exhibited 93.85\% and 92.89\% matching rates with the REI Ovulation Determination and Documented Ovulation datasets, respectively. In the Clinical Pregnancy Rates dataset, the Matched group (282 cycles) showed significantly higher clinical pregnancy rates than the Mismatched group (34.6\% vs. 25.9\%, p = 0.04) and similar results for patients under 37 (41.1\% vs. 30.7\%, p = 0.04). Logistic regression indicated lower pregnancy rates in Mismatched cases (odds ratio 0.67 for the general population, 0.63 for patients under 37). In conclusion, we introduce a highly accurate AI ovulation prediction model. Treatment cycles aligning with the model’s recommendations had significantly increased clinical pregnancy rates.}, - language = {en}, - number = {1}, - urldate = {2025-02-11}, - journal = {Scientific Reports}, - author = {Luz, Almog and Hourvitz, Ariel and Moran, Eden and Itzhak, Nevo and Reuvenny, Shachar and Hourvitz, Rohi and Youngster, Michal and Baum, Micha and Maman, Ettie}, - month = nov, - year = {2024}, - note = {Publisher: Nature Publishing Group}, - keywords = {Infertility, Outcomes research}, - pages = {29451}, - file = {Full Text PDF:/home/alex/Zotero/storage/BSHNTIFD/Luz et al. - 2024 - Improved clinical pregnancy rates in natural frozen-thawed embryo transfer cycles with machine learn.pdf:application/pdf}, -} - -@article{pratikno_pdf_2024, - title = {({PDF}) {A} novel women's ovulation prediction through salivary ferning using the box counting and deep learning}, - url = {https://www.researchgate.net/publication/379467622_A_novel_women's_ovulation_prediction_through_salivary_ferning_using_the_box_counting_and_deep_learning}, - doi = {10.11591/eei.v13i2.5847}, - abstract = {PDF {\textbar} There are several methods to predict a woman's ovulation time, including using a calendar system, basal body temperature, ovulation prediction... {\textbar} Find, read and cite all the research you need on ResearchGate}, - language = {en}, - urldate = {2025-02-11}, - journal = {ResearchGate}, - author = {Pratikno and Ibrahim and Jusak}, - month = dec, - year = {2024}, - annote = {Not really relevant, as different data domain -{\textgreater} saliva -}, - file = {Full Text:/home/alex/Zotero/storage/MVSBPZQG/2024 - (PDF) A novel women's ovulation prediction through salivary ferning using the box counting and deep.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/8FYLC8YZ/379467622_A_novel_women's_ovulation_prediction_through_salivary_ferning_using_the_box_counting_.html:text/html}, -} - -@article{shkodzik_innovative_2024, - title = {Innovative {Approaches} to {Digital} {Health} in {Ovulation} {Detection}: {A} {Review} of {Current} {Methods} and {Emerging} {Technologies}}, - volume = {42}, - issn = {1526-4564}, - shorttitle = {Innovative {Approaches} to {Digital} {Health} in {Ovulation} {Detection}}, - doi = {10.1055/s-0044-1793829}, - abstract = {Ovulation is a vital sign, as significant as body temperature, heart rate, respiratory rate, and blood pressure, in assessing overall health and identifying potential health issues. Ovulation is a key event of the menstrual cycle that provides insights into the hormonal and reproductive health aspects. Affected by the orchestra of hormones, namely thyroid, prolactin, and androgens, disruptions in ovulation can indicate endocrinological conditions and lead to gynecological problems, such as heavy menstrual bleeding, irregular periods, amenorrhea, dysmenorrhea, and difficulties in getting pregnant. Monitoring ovulation and detecting disruptions can aid in the early detection of health issues, extending beyond reproductive health concerns. It can help identify underlying causes of symptoms like excessive fatigue and abnormal hair growth. The integration of digital health technologies, such as mobile apps using machine learning algorithms, wearables tracking temperature, heart rate, breath rate, and sleep patterns, and devices measuring reproductive hormones in urine or saliva samples, offers a wealth of opportunities in family planning, early health issue diagnosis, treatment adjustment, and tracking menstrual cycles during assisted reproductive techniques. These advancements provide a comprehensive approach to health monitoring, addressing both reproductive and overall health concerns.}, - language = {eng}, +@article{yuan_dcfa-itimenet_2024, + title = {{DCFA}-{iTimeNet}: Dynamic cross-fusion attention network for interpretable time series prediction}, + volume = {55}, + issn = {1573-7497}, + url = {https://doi.org/10.1007/s10489-024-05973-2}, + doi = {10.1007/s10489-024-05973-2}, + shorttitle = {{DCFA}-{iTimeNet}}, + abstract = {Although time series prediction research among engineering and technology has made breakthrough progress in performance, challenges remain in modeling complex dynamic interactions between variables and interpretability. To address these two problems, a novel two-stage strategy framework called {DCFA}-{iTimeNet} is introduced. In the first stage, this paper innovatively proposes a dynamic cross-fusion attention mechanism ({DCFA}) . This module facilitates the model to exchange information between different patches of the time series, thereby capturing the complex interactions between variables across time. In the second stage, we exploit a decomposition-based linear explainable Bidirectional Gated Recurrent Unit ({DeLEBiGRU}), which consists mainly of standard {BiGRU} and tensorized {BiGRU}. It is proposed to analyze each variable’s historical long-term, instantaneous, and future impacts. Such design is crucial for understanding how each variable impacts the overall prediction over time. Extensive experimental results demonstrate that the proposed model can effectively model and interpret complex dynamic relationships of multivariate time series and understand the model’s decision-making process. Moreover, the performance outperforms the state-of-the-art methods.}, + pages = {86}, number = {2}, - journal = {Seminars in Reproductive Medicine}, - author = {Shkodzik, Katerina}, - month = jun, - year = {2024}, - pmid = {39572028}, - keywords = {Humans, Digital Health, Female, Mobile Applications, Ovulation, Ovulation Detection, Telemedicine, Wearable Electronic Devices}, - pages = {81--89}, + journaltitle = {Appl Intell}, + author = {Yuan, Jianjun and Wu, Fujun and Zhao, Luoming and Pan, Dongbo and Yu, Xinyue}, + urldate = {2025-02-25}, + date = {2024-12-06}, + langid = {english}, + keywords = {Interpretability, Artificial Intelligence, Dynamic cross-fusion attention, Dynamic interaction, Time series prediction}, + file = {Full Text PDF:/home/alex/Zotero/storage/JEXYN7BN/Yuan et al. - 2024 - DCFA-iTimeNet Dynamic cross-fusion attention network for interpretable time series prediction.pdf:application/pdf}, } -@article{lin_transformer_2023, - title = {Transformer neural network to predict and interpret pregnancy loss from activity data in {Holstein} dairy cows}, - volume = {205}, - issn = {0168-1699}, - url = {https://www.sciencedirect.com/science/article/pii/S0168169923000261}, - doi = {10.1016/j.compag.2023.107638}, - abstract = {Predicting/detecting pregnancy loss of dairy cows offers the opportunity to shorten the time interval between artificial inseminations. Although several methods of pregnancy detection are being practiced, models with accurate, timely and interpretable detection of pregnancy are still lacking. This study proposed a transformer neural network to predict the probability of pregnancy loss based on continuous activity data, which were collected from activity-monitoring tags attached to 185 Holstein cows from a commercial dairy farm in Cayuga County, NY, USA. Our best model achieved an average accuracy of 0.87, F1 score of 0.87, recall of 0.87 and specificity of 0.90 using 14-day time-series activity windows (90\% overlap) using 5-fold cross-validation, outperforming commonly used classic statistical learning and deep learning models for time-series data. The results indicated that our predictive model gave high probabilities of correctly detecting pregnancy loss prior to the increased activities and veterinary confirmation by transrectal ultrasound. In addition, our model interpretation aligned with the changes in the temporal activity levels, revealing that drastic fluctuations in time-series activity data contributed heavily to the final prediction. To the best of our knowledge, this is the first work on developing transformer models for the prediction of pregnancy loss in dairy cows. In addition to facilitating the development of future precision management on modern farms, our work potentiates an increase in the reproductive efficiency and profitability of dairy farms.}, - urldate = {2025-02-11}, - journal = {Computers and Electronics in Agriculture}, - author = {Lin, Dan and Kenéz, Ákos and McArt, Jessica A. A. and Li, Jun}, - month = feb, - year = {2023}, - keywords = {Dairy cow, Precision livestock farming, Pregnancy loss prediction, Time-series activity}, - pages = {107638}, +@misc{sprang_enforcing_2024, + title = {Enforcing Interpretability in Time Series Transformers: A Concept Bottleneck Framework}, + url = {http://arxiv.org/abs/2410.06070}, + doi = {10.48550/arXiv.2410.06070}, + shorttitle = {Enforcing Interpretability in Time Series Transformers}, + abstract = {There has been a recent push of research on Transformer-based models for long-term time series forecasting, even though they are inherently difficult to interpret and explain. While there is a large body of work on interpretability methods for various domains and architectures, the interpretability of Transformer-based forecasting models remains largely unexplored. To address this gap, we develop a framework based on Concept Bottleneck Models to enforce interpretability of time series Transformers. We modify the training objective to encourage a model to develop representations similar to predefined interpretable concepts. In our experiments, we enforce similarity using Centered Kernel Alignment, and the predefined concepts include time features and an interpretable, autoregressive surrogate model ({AR}). We apply the framework to the Autoformer model, and present an in-depth analysis for a variety of benchmark tasks. We find that the model performance remains mostly unaffected, while the model shows much improved interpretability. Additionally, interpretable concepts become local, which makes the trained model easily intervenable. As a proof of concept, we demonstrate a successful intervention in the scenario of a time shift in the data, which eliminates the need to retrain.}, + number = {{arXiv}:2410.06070}, + publisher = {{arXiv}}, + author = {Sprang, Angela van and Acar, Erman and Zuidema, Willem}, + urldate = {2025-02-25}, + date = {2024-10-08}, + eprinttype = {arxiv}, + eprint = {2410.06070 [cs]}, + keywords = {Computer Science - Machine Learning}, + file = {Preprint PDF:/home/alex/Zotero/storage/HVUXXRXJ/Sprang et al. - 2024 - Enforcing Interpretability in Time Series Transformers A Concept Bottleneck Framework.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/6QS78DZP/2410.html:text/html}, } -@article{luz_p-656_2023, - title = {P-656 {Machine} learning algorithm automatically manages and accurately predicts ovulation in natural frozen-thawed embryo transfer cycles.}, - volume = {38}, - issn = {0268-1161}, - url = {https://doi.org/10.1093/humrep/dead093.983}, - doi = {10.1093/humrep/dead093.983}, - abstract = {Can an Artificial Intelligence (AI) algorithm automatically manage frozen-thawed embryo transfer (NC-FET) treatment cycles and give an accurate prediction of ovulation day.An AI algorithm automatically managed and predicted the ovulation of NC-FET treatment cycles with 94.8\% accuracy using an average of 3.01 test days.Today the preferred method for frozen embryo transfer is natural cycle based on ovulation detection. Currently, there is no software capable of managing the treatment cycle automatically and identifying the time of ovulation to support doctor decisions. The aim of this study is to develop a physician support AI software for determining ovulation time reliably with high accuracy.2083 NC-FET cycles from September 2018 to June 2021 were used to develop the ovulation detection and treatment management algorithms.Each cycle had data from at least 2 visits including: hormonal levels (Estrogen/Progesterone/LH) and follicle sizes.The dataset was divided into a train set and two test sets. In the 1st test set ovulation was determined by experts’ opinions and the 2nd test set included cycles in which follicle rupture was documented in consecutive ultrasounds.Two algorithms were developed, an ovulation prediction algorithm based on an NGBoost model and a treatment management algorithm that used the model to determine if and when to call for a new test or declare the ovulation day.Both algorithms were jointly tuned to reach the highest success rate, defined as providing the correct day of ovulation using the available cycle data, with as few test days as possible.On the first test set, which consisted of 176 cycles in which ovulation was determined through the majority decision of 2 independent experts and the attending physician, the treatment management algorithm required on average 3.01 tests to reach a prediction and successfully predicted the ovulation day in 94.8\% of cycles.In the second test set, which consisted of 29 cycles in which ovulation was determined through the follicular rupture in two consecutive ultrasounds, only the ovulation prediction model was tested. To ensure that the model provides a reliable answer and does not rely solely on the follicular disappearances, examined cycles were tested twice: Once using the ovulation day without the day prior to it, and again using only the day prior to ovulation without the ovulation day itself. The algorithm accurately predicted ovulation in 28 out of 29 instances (96.6\%) using the day of ovulation and in 28 out of 29 instances (96.6\%) using the day before ovulation.The main drawback is this being a retrospective study: while the algorithm was trained to maximize accuracy when it selects the test days, the dataset test days were selected by the attending physicians. Statistical methods were used to overcome this, however further prospective trials are needed to validate the results.This is the first AI algorithm designed to automatically manage NC-FET IVF treatment cycles and predict ovulation. The high accuracy and low average tests count might improve treatment outcomes, reduce the patients’ life disruption, and allow physicians to spend less time monitoring their patients’ treatments.not applicable}, - number = {Supplement\_1}, - urldate = {2025-02-11}, - journal = {Human Reproduction}, - author = {Luz, A and Hourvitz, R and Reuvenny, S and Youngster, M and Baum, M and Hourvitz, A and Maman, E}, - month = jun, - year = {2023}, - pages = {dead093.983}, - file = {Full Text PDF:/home/alex/Zotero/storage/C8YM765Y/Luz et al. - 2023 - P-656 Machine learning algorithm automatically manages and accurately predicts ovulation in natural.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/WANIZX3H/7202977.html:text/html}, +@misc{chefer_transformer_2021, + title = {Transformer Interpretability Beyond Attention Visualization}, + url = {http://arxiv.org/abs/2012.09838}, + doi = {10.48550/arXiv.2012.09838}, + abstract = {Self-attention techniques, and specifically Transformers, are dominating the field of text processing and are becoming increasingly popular in computer vision classification tasks. In order to visualize the parts of the image that led to a certain classification, existing methods either rely on the obtained attention maps or employ heuristic propagation along the attention graph. In this work, we propose a novel way to compute relevancy for Transformer networks. The method assigns local relevance based on the Deep Taylor Decomposition principle and then propagates these relevancy scores through the layers. This propagation involves attention layers and skip connections, which challenge existing methods. Our solution is based on a specific formulation that is shown to maintain the total relevancy across layers. We benchmark our method on very recent visual Transformer networks, as well as on a text classification problem, and demonstrate a clear advantage over the existing explainability methods.}, + number = {{arXiv}:2012.09838}, + publisher = {{arXiv}}, + author = {Chefer, Hila and Gur, Shir and Wolf, Lior}, + urldate = {2025-02-25}, + date = {2021-04-05}, + eprinttype = {arxiv}, + eprint = {2012.09838 [cs]}, + keywords = {Computer Science - Computer Vision and Pattern Recognition}, + file = {Preprint PDF:/home/alex/Zotero/storage/3FRISAP7/Chefer et al. - 2021 - Transformer Interpretability Beyond Attention Visualization.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/QUPBJPC9/2012.html:text/html}, } -@inproceedings{azaria_semi-supervised_2019, - title = {Semi-{Supervised} {Ovulation} {Detection} {Based} on {Multiple} {Properties}}, - url = {https://ieeexplore.ieee.org/document/8995235}, - doi = {10.1109/ICTAI.2019.00039}, - abstract = {Despite being a well-researched problem, ovulation detection in human female remains a difficult task. Most current methods for ovulation detection rely on measurements of a single property (e.g. morning body temperature) or at most on two properties (e.g. both salivary and vaginal electrical resistance). In this paper we present a machine learning based method for detecting the day in which ovulation occurs. Our method considered measurements of five different properties. We crawled a data-set from the web and showed that our method outperforms current state-of-the-art methods for ovulation detection. Our method performs well also when considering measurements of fewer properties. We show that our method's performance can be further improved by using unlabeled data, that is, mensuration cycles without a know ovulation date. Our resulted machine learning model can be very useful for women trying to conceive that have trouble in recognizing their ovulation period, especially when some measurements are missing.}, - urldate = {2025-02-11}, - booktitle = {2019 {IEEE} 31st {International} {Conference} on {Tools} with {Artificial} {Intelligence} ({ICTAI})}, - author = {Azaria, Amos and Azaria, Seagal}, - month = nov, - year = {2019}, - note = {ISSN: 2375-0197}, - keywords = {ovulation detection, semi supervised learning}, - pages = {222--228}, - annote = { - -datapoints - - -basal body temperature - - -salivary electrical resistance - - -vaginal electric resistance - - -ovulation prediction kit -{\textgreater} measures for LH hormone - - -markers such as breast tenderness, cervical mucus and custom - - - - -models - - -cnn - - -LSTM - - -conditional random fields approach? - - -“semi supervised” idea is nuts, using the model to create labels that feed back into the training process (?) - - - - -Use leave one out cross validation, -{\textgreater} questionable results - - -}, - file = {IEEE Xplore Abstract Record:/home/alex/Zotero/storage/JSAEWRGB/8995235.html:text/html;PDF:/home/alex/Zotero/storage/P74SEG9S/Azaria and Azaria - 2019 - Semi-Supervised Ovulation Detection Based on Multiple Properties.pdf:application/pdf}, -} - -@article{fanton_interpretable_2022, - title = {An interpretable machine learning model for predicting the optimal day of trigger during ovarian stimulation}, - volume = {118}, - issn = {0015-0282, 1556-5653}, - url = {https://www.fertstert.org/article/S0015-0282%2822%2900244-8/fulltext}, - doi = {10.1016/j.fertnstert.2022.04.003}, - language = {English}, - number = {1}, - urldate = {2025-02-11}, - journal = {Fertility and Sterility}, - author = {Fanton, Michael and Nutting, Veronica and Solano, Funmi and Maeder-York, Paxton and Hariton, Eduardo and Barash, Oleksii and Weckstein, Louis and Sakkas, Denny and Copperman, Alan B. and Loewke, Kevin}, - month = jul, - year = {2022}, - note = {Publisher: Elsevier}, - keywords = {Artificial intelligence, in vitro fertilization, machine learning, ovarian stimulation, trigger}, - pages = {101--108}, - file = {Full Text PDF:/home/alex/Zotero/storage/UNX7ASLP/Fanton et al. - 2022 - An interpretable machine learning model for predicting the optimal day of trigger during ovarian sti.pdf:application/pdf}, -} - -@article{braude_machine_2024, - title = {Machine learning for predicting elective fertility preservation outcomes}, - volume = {14}, - copyright = {2024 The Author(s)}, - issn = {2045-2322}, - url = {https://www.nature.com/articles/s41598-024-60671-w}, - doi = {10.1038/s41598-024-60671-w}, - abstract = {This retrospective study applied machine-learning models to predict treatment outcomes of women undergoing elective fertility preservation. Two-hundred-fifty women who underwent elective fertility preservation at a tertiary center, 2019–2022 were included. Primary outcome was the number of metaphase II oocytes retrieved. Outcome class was based on oocyte count (OC): Low (≤ 8), Medium (9–15) or High (≥ 16). Machine-learning models and statistical regression were used to predict outcome class, first based on pre-treatment parameters, and then using post-treatment data from ovulation-triggering day. OC was 136 Low, 80 Medium, and 34 High. Random Forest Classifier (RFC) was the most accurate model (pre-treatment receiver operating characteristic (ROC) area under the curve (AUC) was 77\%, and post-treatment ROC AUC was 87\%), followed by XGBoost Classifier (pre-treatment ROC AUC 74\%, post-treatment ROC AUC 86\%). The most important pre-treatment parameters for RFC were basal FSH (22.6\%), basal LH (19.1\%), AFC (18.2\%), and basal estradiol (15.6\%). Post-treatment parameters were estradiol levels on trigger-day (17.7\%), basal FSH (11\%), basal LH (9\%), and AFC (8\%). Machine-learning models trained with clinical data appear to predict fertility preservation treatment outcomes with relatively high accuracy.}, - language = {en}, - number = {1}, - urldate = {2025-02-11}, - journal = {Scientific Reports}, - author = {Braude, Itai and Haikin Herzberger, Einat and Semo, Mor and Soifer, Kim and Goren Gepstein, Nitzan and Wiser, Amir and Miller, Netanella}, - month = may, - year = {2024}, - note = {Publisher: Nature Publishing Group}, - keywords = {Outcomes research, Computational models}, - pages = {10158}, - file = {Full Text PDF:/home/alex/Zotero/storage/URDGBHLV/Braude et al. - 2024 - Machine learning for predicting elective fertility preservation outcomes.pdf:application/pdf}, +@misc{helbling_conceptattention_2025, + title = {{ConceptAttention}: Diffusion Transformers Learn Highly Interpretable Features}, + url = {http://arxiv.org/abs/2502.04320}, + doi = {10.48550/arXiv.2502.04320}, + shorttitle = {{ConceptAttention}}, + abstract = {Do the rich representations of multi-modal diffusion transformers ({DiTs}) exhibit unique properties that enhance their interpretability? We introduce {ConceptAttention}, a novel method that leverages the expressive power of {DiT} attention layers to generate high-quality saliency maps that precisely locate textual concepts within images. Without requiring additional training, {ConceptAttention} repurposes the parameters of {DiT} attention layers to produce highly contextualized concept embeddings, contributing the major discovery that performing linear projections in the output space of {DiT} attention layers yields significantly sharper saliency maps compared to commonly used cross-attention mechanisms. Remarkably, {ConceptAttention} even achieves state-of-the-art performance on zero-shot image segmentation benchmarks, outperforming 11 other zero-shot interpretability methods on the {ImageNet}-Segmentation dataset and on a single-class subset of {PascalVOC}. Our work contributes the first evidence that the representations of multi-modal {DiT} models like Flux are highly transferable to vision tasks like segmentation, even outperforming multi-modal foundation models like {CLIP}.}, + number = {{arXiv}:2502.04320}, + publisher = {{arXiv}}, + author = {Helbling, Alec and Meral, Tuna Han Salih and Hoover, Ben and Yanardag, Pinar and Chau, Duen Horng}, + urldate = {2025-02-25}, + date = {2025-02-06}, + eprinttype = {arxiv}, + eprint = {2502.04320 [cs]}, + keywords = {Computer Science - Machine Learning, Computer Science - Computer Vision and Pattern Recognition}, + file = {Preprint PDF:/home/alex/Zotero/storage/AEEDM4ZW/Helbling et al. - 2025 - ConceptAttention Diffusion Transformers Learn Highly Interpretable Features.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/SAHGAINH/2502.html:text/html}, } @article{noauthor_temporal_2021, - title = {Temporal {Fusion} {Transformers} for interpretable multi-horizon time series forecasting}, + title = {Temporal Fusion Transformers for interpretable multi-horizon time series forecasting}, volume = {37}, issn = {0169-2070}, url = {https://www.sciencedirect.com/science/article/pii/S0169207021000637}, doi = {10.1016/j.ijforecast.2021.03.012}, abstract = {Multi-horizon forecasting often contains a complex mix of inputs – including static (i.e. time-invariant) covariates, known future inputs, and other e…}, - language = {en-US}, - number = {4}, - urldate = {2025-02-24}, - journal = {International Journal of Forecasting}, - month = oct, - year = {2021}, - note = {Publisher: Elsevier}, pages = {1748--1764}, + number = {4}, + journaltitle = {International Journal of Forecasting}, + urldate = {2025-02-24}, + date = {2021-10-01}, + langid = {american}, + note = {Publisher: Elsevier}, file = {Snapshot:/home/alex/Zotero/storage/SFYESIWK/S0169207021000637.html:text/html;Submitted Version:/home/alex/Zotero/storage/A9AYS5UI/2021 - Temporal Fusion Transformers for interpretable multi-horizon time series forecasting.pdf:application/pdf}, } @inproceedings{vaswani_attention_2017, - title = {Attention is {All} you {Need}}, + title = {Attention is All you Need}, volume = {30}, url = {https://proceedings.neurips.cc/paper/2017/hash/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html}, - abstract = {The dominant sequence transduction models are based on complex recurrent orconvolutional neural networks in an encoder and decoder configuration. The best performing such models also connect the encoder and decoder through an attentionm echanisms. We propose a novel, simple network architecture based solely onan attention mechanism, dispensing with recurrence and convolutions entirely.Experiments on two machine translation tasks show these models to be superiorin quality while being more parallelizable and requiring significantly less timeto train. Our single model with 165 million parameters, achieves 27.5 BLEU onEnglish-to-German translation, improving over the existing best ensemble result by over 1 BLEU. On English-to-French translation, we outperform the previoussingle state-of-the-art with model by 0.7 BLEU, achieving a BLEU score of 41.1.}, - urldate = {2025-02-24}, - booktitle = {Advances in {Neural} {Information} {Processing} {Systems}}, + abstract = {The dominant sequence transduction models are based on complex recurrent orconvolutional neural networks in an encoder and decoder configuration. The best performing such models also connect the encoder and decoder through an attentionm echanisms. We propose a novel, simple network architecture based solely onan attention mechanism, dispensing with recurrence and convolutions entirely.Experiments on two machine translation tasks show these models to be superiorin quality while being more parallelizable and requiring significantly less timeto train. Our single model with 165 million parameters, achieves 27.5 {BLEU} {onEnglish}-to-German translation, improving over the existing best ensemble result by over 1 {BLEU}. On English-to-French translation, we outperform the previoussingle state-of-the-art with model by 0.7 {BLEU}, achieving a {BLEU} score of 41.1.}, + booktitle = {Advances in Neural Information Processing Systems}, publisher = {Curran Associates, Inc.}, author = {Vaswani, Ashish and Shazeer, Noam and Parmar, Niki and Uszkoreit, Jakob and Jones, Llion and Gomez, Aidan N and Kaiser, Ł ukasz and Polosukhin, Illia}, - year = {2017}, + urldate = {2025-02-24}, + date = {2017}, file = {Full Text PDF:/home/alex/Zotero/storage/MU7NU9LR/Vaswani et al. - 2017 - Attention is All you Need.pdf:application/pdf}, } -@misc{noauthor_vivosens_nodate, +@online{noauthor_vivosens_nodate, title = {vivosens medical gmbh}, url = {https://www.vivosensmedical.com/}, urldate = {2025-02-24}, @@ -571,119 +236,117 @@ Use leave one out cross validation, -{\textgreater} questionable results issn = {1097-0258}, url = {https://onlinelibrary.wiley.com/doi/abs/10.1002/sim.4780100207}, doi = {10.1002/sim.4780100207}, - abstract = {The identification of the human fertile phase as the time during which a woman or a couple may conceive is elusive. The fertile time depends on many factors in each individual menstrual cycle and may be said to be more of a statistical than a physiological entity. This paper reviews the application of statistical methods to three areas related to conception and the fertile phase. The first is the prediction and detection of ovulation from serial measurements, such as hormones, basal body temperature and cervical mucus, throughout the menstrual cycle. Typically, such variables increase from some baseline level to a peak around ovulation (the most fertile time), then subside to low levels in the postovulatory phase. The statistical challenge is to detect the rise (signalling the onset of potential fertility) and subsequent fall. Analytic methods considered include thresholds, Bayesian change-point models and particularly the cumulative sum (cusum) technique which is both simple to apply and understand, and effective. The second area comprises appropriate methods of analysing and interpreting data from clinical studies of the fertile phase, especially in so-alled natural family planning (NFP) where it is usual for women to observe several indices of potential fertility. Such studies usually try to establish the temporal relationships between markers of the fertile phase and examine the success of different combinations of markers in delineating the fertile time in comparison with a standard ‘defined’ phase, for example, the interval from three days before to two days after the peak of luteinizing hormone. The third area is the assessment of the probability of conception on certain days of the cycle, which is vital to the understanding of the fertile phase and its application to NFP. Direct estimation of such probabilities is impractical; instead, resort must be made to estimation by maximum likelihood of the parameters of specially constructed models. Suitable models are described. Finally, the need for a new prospective study of the probability of conception in relation to the markers of the fertile phase used in the symptothermal method of NFP is discussed.}, - language = {en}, - number = {2}, - urldate = {2025-02-24}, - journal = {Statistics in Medicine}, - author = {Royston, Patrick}, - year = {1991}, - note = {\_eprint: https://onlinelibrary.wiley.com/doi/pdf/10.1002/sim.4780100207}, + abstract = {The identification of the human fertile phase as the time during which a woman or a couple may conceive is elusive. The fertile time depends on many factors in each individual menstrual cycle and may be said to be more of a statistical than a physiological entity. This paper reviews the application of statistical methods to three areas related to conception and the fertile phase. The first is the prediction and detection of ovulation from serial measurements, such as hormones, basal body temperature and cervical mucus, throughout the menstrual cycle. Typically, such variables increase from some baseline level to a peak around ovulation (the most fertile time), then subside to low levels in the postovulatory phase. The statistical challenge is to detect the rise (signalling the onset of potential fertility) and subsequent fall. Analytic methods considered include thresholds, Bayesian change-point models and particularly the cumulative sum (cusum) technique which is both simple to apply and understand, and effective. The second area comprises appropriate methods of analysing and interpreting data from clinical studies of the fertile phase, especially in so-alled natural family planning ({NFP}) where it is usual for women to observe several indices of potential fertility. Such studies usually try to establish the temporal relationships between markers of the fertile phase and examine the success of different combinations of markers in delineating the fertile time in comparison with a standard ‘defined’ phase, for example, the interval from three days before to two days after the peak of luteinizing hormone. The third area is the assessment of the probability of conception on certain days of the cycle, which is vital to the understanding of the fertile phase and its application to {NFP}. Direct estimation of such probabilities is impractical; instead, resort must be made to estimation by maximum likelihood of the parameters of specially constructed models. Suitable models are described. Finally, the need for a new prospective study of the probability of conception in relation to the markers of the fertile phase used in the symptothermal method of {NFP} is discussed.}, pages = {221--240}, + number = {2}, + journaltitle = {Statistics in Medicine}, + author = {Royston, Patrick}, + urldate = {2025-02-24}, + date = {1991}, + langid = {english}, + note = {\_eprint: https://onlinelibrary.wiley.com/doi/pdf/10.1002/sim.4780100207}, file = {PDF:/home/alex/Zotero/storage/CFS45CFD/Royston - 1991 - Identifying the fertile phase of the human menstrual cycle.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/LCN63EBN/sim.html:text/html}, } @inproceedings{rigotti_attention-based_2021, - title = {Attention-based {Interpretability} with {Concept} {Transformers}}, + title = {Attention-based Interpretability with Concept Transformers}, url = {https://openreview.net/forum?id=kAa9eDS0RdO}, - abstract = {Attention is a mechanism that has been instrumental in driving remarkable performance gains of deep neural network models in a host of visual, NLP and multimodal tasks. One additional notable aspect of attention is that it conveniently exposes the ``reasoning'' behind each particular output generated by the model. Specifically, attention scores over input regions or intermediate features have been interpreted as a measure of the contribution of the attended element to the model inference. While the debate in regard to the interpretability of attention is still not settled, researchers have pointed out the existence of architectures and scenarios that afford a meaningful interpretation of the attention mechanism. Here we propose the generalization of attention from low-level input features to high-level concepts as a mechanism to ensure the interpretability of attention scores within a given application domain. In particular, we design the ConceptTransformer, a deep learning module that exposes explanations of the output of a model in which it is embedded in terms of attention over user-defined high-level concepts. Such explanations are {\textbackslash}emph\{plausible\} (i.e.{\textbackslash} convincing to the human user) and {\textbackslash}emph\{faithful\} (i.e.{\textbackslash} truly reflective of the reasoning process of the model). Plausibility of such explanations is obtained by construction by training the attention heads to conform with known relations between inputs, concepts and outputs dictated by domain knowledge. Faithfulness is achieved by design by enforcing a linear relation between the transformer value vectors that represent the concepts and their contribution to the classification log-probabilities. We validate our ConceptTransformer module on established explainability benchmarks and show how it can be used to infuse domain knowledge into classifiers to improve accuracy, and conversely to extract concept-based explanations of classification outputs. Code to reproduce our results is available at: {\textbackslash}url\{https://github.com/ibm/concept\_transformer\}.}, - language = {en}, - urldate = {2025-02-21}, + abstract = {Attention is a mechanism that has been instrumental in driving remarkable performance gains of deep neural network models in a host of visual, {NLP} and multimodal tasks. One additional notable aspect of attention is that it conveniently exposes the ``reasoning'' behind each particular output generated by the model. Specifically, attention scores over input regions or intermediate features have been interpreted as a measure of the contribution of the attended element to the model inference. While the debate in regard to the interpretability of attention is still not settled, researchers have pointed out the existence of architectures and scenarios that afford a meaningful interpretation of the attention mechanism. Here we propose the generalization of attention from low-level input features to high-level concepts as a mechanism to ensure the interpretability of attention scores within a given application domain. In particular, we design the {ConceptTransformer}, a deep learning module that exposes explanations of the output of a model in which it is embedded in terms of attention over user-defined high-level concepts. Such explanations are {\textbackslash}emph\{plausible\} (i.e.{\textbackslash} convincing to the human user) and {\textbackslash}emph\{faithful\} (i.e.{\textbackslash} truly reflective of the reasoning process of the model). Plausibility of such explanations is obtained by construction by training the attention heads to conform with known relations between inputs, concepts and outputs dictated by domain knowledge. Faithfulness is achieved by design by enforcing a linear relation between the transformer value vectors that represent the concepts and their contribution to the classification log-probabilities. We validate our {ConceptTransformer} module on established explainability benchmarks and show how it can be used to infuse domain knowledge into classifiers to improve accuracy, and conversely to extract concept-based explanations of classification outputs. Code to reproduce our results is available at: {\textbackslash}url\{https://github.com/ibm/concept\_transformer\}.}, + eventtitle = {International Conference on Learning Representations}, author = {Rigotti, Mattia and Miksovic, Christoph and Giurgiu, Ioana and Gschwind, Thomas and Scotton, Paolo}, - month = oct, - year = {2021}, + urldate = {2025-02-21}, + date = {2021-10-06}, + langid = {english}, file = {Full Text PDF:/home/alex/Zotero/storage/U2FUGVF6/Rigotti et al. - 2021 - Attention-based Interpretability with Concept Transformers.pdf:application/pdf}, } @article{kitada_attention_2021, - title = {Attention {Meets} {Perturbations}: {Robust} and {Interpretable} {Attention} {With} {Adversarial} {Training}}, + title = {Attention Meets Perturbations: Robust and Interpretable Attention With Adversarial Training}, volume = {9}, issn = {2169-3536}, - shorttitle = {Attention {Meets} {Perturbations}}, url = {https://ieeexplore.ieee.org/abstract/document/9467291}, doi = {10.1109/ACCESS.2021.3093456}, - abstract = {Although attention mechanisms have been applied to a variety of deep learning models and have been shown to improve the prediction performance, it has been reported to be vulnerable to perturbations to the mechanism. To overcome the vulnerability to perturbations in the mechanism, we are inspired by adversarial training (AT), which is a powerful regularization technique for enhancing the robustness of the models. In this paper, we propose a general training technique for natural language processing tasks, including AT for attention (Attention AT) and more interpretable AT for attention (Attention iAT). The proposed techniques improved the prediction performance and the model interpretability by exploiting the mechanisms with AT. In particular, Attention iAT boosts those advantages by introducing adversarial perturbation, which enhances the difference in the attention of the sentences. Evaluation experiments with ten open datasets revealed that AT for attention mechanisms, especially Attention iAT, demonstrated (1) the best performance in nine out of ten tasks and (2) more interpretable attention (i.e., the resulting attention correlated more strongly with gradient-based word importance) for all tasks. Additionally, the proposed techniques are (3) much less dependent on perturbation size in AT.}, - urldate = {2025-02-21}, - journal = {IEEE Access}, - author = {Kitada, Shunsuke and Iyatomi, Hitoshi}, - year = {2021}, - note = {Conference Name: IEEE Access}, - keywords = {adversarial training, attention mechanism, binary classification, interpretability, Knowledge discovery, natural language inference, Natural language processing, Perturbation methods, Predictive models, question answering, Robustness, Solid modeling, Task analysis, Training}, + shorttitle = {Attention Meets Perturbations}, + abstract = {Although attention mechanisms have been applied to a variety of deep learning models and have been shown to improve the prediction performance, it has been reported to be vulnerable to perturbations to the mechanism. To overcome the vulnerability to perturbations in the mechanism, we are inspired by adversarial training ({AT}), which is a powerful regularization technique for enhancing the robustness of the models. In this paper, we propose a general training technique for natural language processing tasks, including {AT} for attention (Attention {AT}) and more interpretable {AT} for attention (Attention {iAT}). The proposed techniques improved the prediction performance and the model interpretability by exploiting the mechanisms with {AT}. In particular, Attention {iAT} boosts those advantages by introducing adversarial perturbation, which enhances the difference in the attention of the sentences. Evaluation experiments with ten open datasets revealed that {AT} for attention mechanisms, especially Attention {iAT}, demonstrated (1) the best performance in nine out of ten tasks and (2) more interpretable attention (i.e., the resulting attention correlated more strongly with gradient-based word importance) for all tasks. Additionally, the proposed techniques are (3) much less dependent on perturbation size in {AT}.}, pages = {92974--92985}, + journaltitle = {{IEEE} Access}, + author = {Kitada, Shunsuke and Iyatomi, Hitoshi}, + urldate = {2025-02-21}, + date = {2021}, + note = {Conference Name: {IEEE} Access}, + keywords = {adversarial training, attention mechanism, binary classification, interpretability, Knowledge discovery, natural language inference, Natural language processing, Perturbation methods, Predictive models, question answering, Robustness, Solid modeling, Task analysis, Training}, file = {Full Text PDF:/home/alex/Zotero/storage/BUXFCXRU/Kitada and Iyatomi - 2021 - Attention Meets Perturbations Robust and Interpretable Attention With Adversarial Training.pdf:application/pdf;IEEE Xplore Abstract Record:/home/alex/Zotero/storage/ZNQD3FRW/9467291.html:text/html}, } @inproceedings{choi_retain_2016, - title = {{RETAIN}: {An} {Interpretable} {Predictive} {Model} for {Healthcare} using {Reverse} {Time} {Attention} {Mechanism}}, + title = {{RETAIN}: An Interpretable Predictive Model for Healthcare using Reverse Time Attention Mechanism}, volume = {29}, - shorttitle = {{RETAIN}}, url = {https://proceedings.neurips.cc/paper/2016/hash/231141b34c82aa95e48810a9d1b33a79-Abstract.html}, - abstract = {Accuracy and interpretability are two dominant features of successful predictive models. Typically, a choice must be made in favor of complex black box models such as recurrent neural networks (RNN) for accuracy versus less accurate but more interpretable traditional models such as logistic regression. This tradeoff poses challenges in medicine where both accuracy and interpretability are important. We addressed this challenge by developing the REverse Time AttentIoN model (RETAIN) for application to Electronic Health Records (EHR) data. RETAIN achieves high accuracy while remaining clinically interpretable and is based on a two-level neural attention model that detects influential past visits and significant clinical variables within those visits (e.g. key diagnoses). RETAIN mimics physician practice by attending the EHR data in a reverse time order so that recent clinical visits are likely to receive higher attention. RETAIN was tested on a large health system EHR dataset with 14 million visits completed by 263K patients over an 8 year period and demonstrated predictive accuracy and computational scalability comparable to state-of-the-art methods such as RNN, and ease of interpretability comparable to traditional models.}, - urldate = {2025-02-21}, - booktitle = {Advances in {Neural} {Information} {Processing} {Systems}}, + shorttitle = {{RETAIN}}, + abstract = {Accuracy and interpretability are two dominant features of successful predictive models. Typically, a choice must be made in favor of complex black box models such as recurrent neural networks ({RNN}) for accuracy versus less accurate but more interpretable traditional models such as logistic regression. This tradeoff poses challenges in medicine where both accuracy and interpretability are important. We addressed this challenge by developing the {REverse} Time {AttentIoN} model ({RETAIN}) for application to Electronic Health Records ({EHR}) data. {RETAIN} achieves high accuracy while remaining clinically interpretable and is based on a two-level neural attention model that detects influential past visits and significant clinical variables within those visits (e.g. key diagnoses). {RETAIN} mimics physician practice by attending the {EHR} data in a reverse time order so that recent clinical visits are likely to receive higher attention. {RETAIN} was tested on a large health system {EHR} dataset with 14 million visits completed by 263K patients over an 8 year period and demonstrated predictive accuracy and computational scalability comparable to state-of-the-art methods such as {RNN}, and ease of interpretability comparable to traditional models.}, + booktitle = {Advances in Neural Information Processing Systems}, publisher = {Curran Associates, Inc.}, author = {Choi, Edward and Bahadori, Mohammad Taha and Sun, Jimeng and Kulas, Joshua and Schuetz, Andy and Stewart, Walter}, - year = {2016}, + urldate = {2025-02-21}, + date = {2016}, file = {Full Text PDF:/home/alex/Zotero/storage/XQLMYHUU/Choi et al. - 2016 - RETAIN An Interpretable Predictive Model for Healthcare using Reverse Time Attention Mechanism.pdf:application/pdf}, } @misc{serrano_is_2019, - title = {Is {Attention} {Interpretable}?}, + title = {Is Attention Interpretable?}, url = {http://arxiv.org/abs/1906.03731}, doi = {10.48550/arXiv.1906.03731}, - abstract = {Attention mechanisms have recently boosted performance on a range of NLP tasks. Because attention layers explicitly weight input components' representations, it is also often assumed that attention can be used to identify information that models found important (e.g., specific contextualized word tokens). We test whether that assumption holds by manipulating attention weights in already-trained text classification models and analyzing the resulting differences in their predictions. While we observe some ways in which higher attention weights correlate with greater impact on model predictions, we also find many ways in which this does not hold, i.e., where gradient-based rankings of attention weights better predict their effects than their magnitudes. We conclude that while attention noisily predicts input components' overall importance to a model, it is by no means a fail-safe indicator.}, - urldate = {2025-02-21}, - publisher = {arXiv}, + abstract = {Attention mechanisms have recently boosted performance on a range of {NLP} tasks. Because attention layers explicitly weight input components' representations, it is also often assumed that attention can be used to identify information that models found important (e.g., specific contextualized word tokens). We test whether that assumption holds by manipulating attention weights in already-trained text classification models and analyzing the resulting differences in their predictions. While we observe some ways in which higher attention weights correlate with greater impact on model predictions, we also find many ways in which this does not hold, i.e., where gradient-based rankings of attention weights better predict their effects than their magnitudes. We conclude that while attention noisily predicts input components' overall importance to a model, it is by no means a fail-safe indicator.}, + number = {{arXiv}:1906.03731}, + publisher = {{arXiv}}, author = {Serrano, Sofia and Smith, Noah A.}, - month = jun, - year = {2019}, - note = {arXiv:1906.03731 [cs]}, + urldate = {2025-02-21}, + date = {2019-06-09}, + eprinttype = {arxiv}, + eprint = {1906.03731 [cs]}, keywords = {Computer Science - Computation and Language}, - annote = {Comment: To appear at ACL 2019}, file = {Preprint PDF:/home/alex/Zotero/storage/DFZ28RG8/Serrano and Smith - 2019 - Is Attention Interpretable.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/63B9XS2Y/1906.html:text/html}, } @article{lyzwinski_innovative_2024, - title = {Innovative {Approaches} to {Menstruation} and {Fertility} {Tracking} {Using} {Wearable} {Reproductive} {Health} {Technology}: {Systematic} {Review}}, + title = {Innovative Approaches to Menstruation and Fertility Tracking Using Wearable Reproductive Health Technology: Systematic Review}, volume = {26}, - shorttitle = {Innovative {Approaches} to {Menstruation} and {Fertility} {Tracking} {Using} {Wearable} {Reproductive} {Health} {Technology}}, url = {https://www.jmir.org/2024/1/e45139}, doi = {10.2196/45139}, + shorttitle = {Innovative Approaches to Menstruation and Fertility Tracking Using Wearable Reproductive Health Technology}, abstract = {Background: Emerging digital health technology has moved into the reproductive health market for female individuals. In the past, mobile health apps have been used to monitor the menstrual cycle using manual entry. New technological trends involve the use of wearable devices to track fertility by assessing physiological changes such as temperature, heart rate, and respiratory rate. Objective: The primary aims of this study are to review the types of wearables that have been developed and evaluated for menstrual cycle tracking and to examine whether they may detect changes in the menstrual cycle in female individuals. Another aim is to review whether these devices are effective for tracking various stages in the menstrual cycle including ovulation and menstruation. Finally, the secondary aim is to assess whether the studies have validated their findings by reporting accuracy and sensitivity. -Methods: A review of PubMed or MEDLINE was undertaken to evaluate wearable devices for their effectiveness in predicting fertility and differentiating between the different stages of the menstrual cycle. +Methods: A review of {PubMed} or {MEDLINE} was undertaken to evaluate wearable devices for their effectiveness in predicting fertility and differentiating between the different stages of the menstrual cycle. Results: Fertility cycle–tracking wearables include devices that can be worn on the wrists, on the fingers, intravaginally, and inside the ear. Wearable devices hold promise for predicting different stages of the menstrual cycle including the fertile window and may be used by female individuals as part of their reproductive health. Most devices had high accuracy for detecting fertility and were able to differentiate between the luteal phase (early and late), fertile window, and menstruation by assessing changes in heart rate, heart rate variability, temperature, and respiratory rate. Conclusions: More research is needed to evaluate consumer perspectives on reproductive technology for monitoring fertility, and ethical issues around the privacy of digital data need to be addressed. Additionally, there is also a need for more studies to validate and confirm this research, given its scarcity, especially in relation to changes in respiratory rate as a proxy for reproductive cycle staging.}, - language = {EN}, + pages = {e45139}, number = {1}, - urldate = {2025-02-21}, - journal = {Journal of Medical Internet Research}, + journaltitle = {Journal of Medical Internet Research}, author = {Lyzwinski, Lynnette and Elgendi, Mohamed and Menon, Carlo}, - month = feb, - year = {2024}, + urldate = {2025-02-21}, + date = {2024-02-15}, note = {Company: Journal of Medical Internet Research Distributor: Journal of Medical Internet Research Institution: Journal of Medical Internet Research Label: Journal of Medical Internet Research -Publisher: JMIR Publications Inc., Toronto, Canada}, - pages = {e45139}, +Publisher: {JMIR} Publications Inc., Toronto, Canada}, file = {Full Text:/home/alex/Zotero/storage/IP23WZLE/Lyzwinski et al. - 2024 - Innovative Approaches to Menstruation and Fertility Tracking Using Wearable Reproductive Health Tech.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/TJN3WV3I/e45139.html:text/html}, } -@misc{noauthor_zyklus-apps_nodate, - title = {Zyklus-{Apps} zur {Verhütung} – sicher oder {Gesellschaftsspiel}? - {ProQuest}}, - shorttitle = {Zyklus-{Apps} zur {Verhütung} – sicher oder {Gesellschaftsspiel}?}, +@online{noauthor_zyklus-apps_nodate, + title = {Zyklus-Apps zur Verhütung – sicher oder Gesellschaftsspiel? - {ProQuest}}, url = {https://www.proquest.com/openview/739071fff0941b30f3a5d33b56259c60/1?pq-origsite=gscholar&cbl=6629261}, - abstract = {Explore millions of resources from scholarly journals, books, newspapers, videos and more, on the ProQuest Platform.}, - language = {en}, + shorttitle = {Zyklus-Apps zur Verhütung – sicher oder Gesellschaftsspiel?}, + abstract = {Explore millions of resources from scholarly journals, books, newspapers, videos and more, on the {ProQuest} Platform.}, urldate = {2025-02-21}, + langid = {english}, file = {Snapshot:/home/alex/Zotero/storage/QHYUJU9G/1.html:text/html}, } @article{goeckenjan_continuous_2020, - title = {Continuous {Body} {Temperature} {Monitoring} to {Improve} the {Diagnosis} of {Female} {Infertility}}, + title = {Continuous Body Temperature Monitoring to Improve the Diagnosis of Female Infertility}, volume = {80}, - copyright = {Georg Thieme Verlag KG Stuttgart · New York}, + rights = {Georg Thieme Verlag {KG} Stuttgart · New York}, issn = {0016-5751}, url = {https://www.thieme-connect.com/products/ejournals/html/10.1055/a-1191-7888}, doi = {10.1055/a-1191-7888}, @@ -692,7 +355,7 @@ Publisher: JMIR Publications Inc., Toronto, Canada}, Material and Methods This prospective interventional study was performed in a reproductive medicine department of a university hospital. The menstrual cycles of 51 women with infertility were monitored and analysed using three different strategies: sonographic and hormonal assessment (standard approach), continuous core body temperature measurement and analysis - using the algorithm of OvulaRing, and lowest daily body temperature measurement monitored with a vaginal biosensor and analysed based on the body temperature curves used in natural family + using the algorithm of {OvulaRing}, and lowest daily body temperature measurement monitored with a vaginal biosensor and analysed based on the body temperature curves used in natural family planning. Results Statistically significant differences were found in the temperature curves of women with luteal phase deficiency and polycystic ovary syndrome compared to women with normal @@ -700,18 +363,17 @@ Results Statistically significant differences were found in the temperature curv Conclusions Continuous body temperature monitoring with a vaginal biosensor can improve the standard diagnostic procedures used to determine ovulatory dysfunction, especially if dysfunction is due to luteal phase deficiency and polycystic ovary syndrome. Analysis of the lowest daily body temperature combined with the basal body temperature measurements used in - fertility awareness methods may be equieffective to continuous body temperature measurements with OvulaRing. The results of this study show that a revised diagnostic approach using fewer + fertility awareness methods may be equieffective to continuous body temperature measurements with {OvulaRing}. The results of this study show that a revised diagnostic approach using fewer hormonal assessments combined with continuous body temperature monitoring can reduce the number of appointments in an infertility clinic as well as the costs.}, - language = {en}, - urldate = {2025-02-21}, - journal = {Geburtshilfe und Frauenheilkunde}, + pages = {702--712}, + journaltitle = {Geburtshilfe und Frauenheilkunde}, author = {Goeckenjan, Maren and Schiwek, Esther and Wimberger, Pauline}, - month = jul, - year = {2020}, - note = {Publisher: Georg Thieme Verlag KG}, + urldate = {2025-02-21}, + date = {2020-07-14}, + langid = {english}, + note = {Publisher: Georg Thieme Verlag {KG}}, keywords = {infertility, Key words fertility awareness, luteal phase deficiency, polycystic ovary syndrome, vaginal biosensor}, - pages = {702--712}, file = {Full Text PDF:/home/alex/Zotero/storage/QKPIJD23/Goeckenjan et al. - 2020 - Continuous Body Temperature Monitoring to Improve the Diagnosis of Female Infertility.pdf:application/pdf}, } @@ -721,56 +383,53 @@ Conclusions Continuous body temperature monitoring with a vaginal biosensor can issn = {0951-3590}, url = {https://doi.org/10.1080/09513590.2017.1390737}, doi = {10.1080/09513590.2017.1390737}, - abstract = {Fertility awareness-based (FAB) methods represent a term that includes all family planning methods that are based on the identification of the fertile window. They are based on the woman’s observation of physiological signs of the fertile and infertile phases of the menstrual cycle. The first approach consists basically in symptothermal methods accompanied by cervical mucus measurements and clinical menstrual cycling data recording. The second most often used methods are the urinary measurement of E3G and luteinizing hormone (LH) with a personalized computer system. Hence these systems lack the efficacy of the continuous circadian and circamensual measurement of the core body temperature. Only this approach enables the accurate detection of the ovulation during the fertile window. A new medical device called OvulaRing has been developed to fill this gap. In the present study, the system and its first clinical results are presented. OvulaRing is a medical device used just like a tampon. The device is a vaginal ring of evatane that contains an integrated biosensor. This sensor measures continuously every 5 min the core body temperature throughout the entire cycle. This device allows a circadian and circamensual intravaginal exact measurement. With this system, 288 measurements are created per day. The system can detect retrospectively and predict prospectively the fertile window of the users. One hundred and fifty eight women aged between 18 and 45 years used this medical device in an open non-randomized clinical study for 15 months. A total of 470 cycles could be recorded and were able for analysis. By the same time in a subgroup of patients, hormonal assessments of LH, follicle-stimulating hormone, estradiol and progesterone as well as vaginal ultrasound were performed in parallel between the 9th and the 36th day of the cycle. The validation error due to software errors was 0.89\% for the retrospective analysis; that means that the accuracy for the detection of the ovulation was 99.11\%. Accuracy of 88.8\% for a window of 3 days before ovulation, the day of ovulation and the 3 days after ovulation was achieved for the prospective analysis. In the subgroup of woman with recorded pregnancies, it could be shown that after 3.79 months of use (median) pregnancies were observed. In 67.72\% in up to 3 months, in 16.36\% between 3 and 6 months of use, in 7.27\% between 7 and 9 months, in 5.45\% between 10 and 12 months and in 1.82\% between 13 and 15 months of use of the system. With this new web-based system, a precise determination of the fertile window even in women with ultralong cycles ({\textgreater}35 days) could be detected independently of their personal live circumstances. Exact determination of the fertile window is herewith possible so that OvulaRing represents an evolution in the FAB method for the cycle diagnosis of women with regular, irregular or anovulatory menstrual cycles.}, + abstract = {Fertility awareness-based ({FAB}) methods represent a term that includes all family planning methods that are based on the identification of the fertile window. They are based on the woman’s observation of physiological signs of the fertile and infertile phases of the menstrual cycle. The first approach consists basically in symptothermal methods accompanied by cervical mucus measurements and clinical menstrual cycling data recording. The second most often used methods are the urinary measurement of E3G and luteinizing hormone ({LH}) with a personalized computer system. Hence these systems lack the efficacy of the continuous circadian and circamensual measurement of the core body temperature. Only this approach enables the accurate detection of the ovulation during the fertile window. A new medical device called {OvulaRing} has been developed to fill this gap. In the present study, the system and its first clinical results are presented. {OvulaRing} is a medical device used just like a tampon. The device is a vaginal ring of evatane that contains an integrated biosensor. This sensor measures continuously every 5 min the core body temperature throughout the entire cycle. This device allows a circadian and circamensual intravaginal exact measurement. With this system, 288 measurements are created per day. The system can detect retrospectively and predict prospectively the fertile window of the users. One hundred and fifty eight women aged between 18 and 45 years used this medical device in an open non-randomized clinical study for 15 months. A total of 470 cycles could be recorded and were able for analysis. By the same time in a subgroup of patients, hormonal assessments of {LH}, follicle-stimulating hormone, estradiol and progesterone as well as vaginal ultrasound were performed in parallel between the 9th and the 36th day of the cycle. The validation error due to software errors was 0.89\% for the retrospective analysis; that means that the accuracy for the detection of the ovulation was 99.11\%. Accuracy of 88.8\% for a window of 3 days before ovulation, the day of ovulation and the 3 days after ovulation was achieved for the prospective analysis. In the subgroup of woman with recorded pregnancies, it could be shown that after 3.79 months of use (median) pregnancies were observed. In 67.72\% in up to 3 months, in 16.36\% between 3 and 6 months of use, in 7.27\% between 7 and 9 months, in 5.45\% between 10 and 12 months and in 1.82\% between 13 and 15 months of use of the system. With this new web-based system, a precise determination of the fertile window even in women with ultralong cycles ({\textgreater}35 days) could be detected independently of their personal live circumstances. Exact determination of the fertile window is herewith possible so that {OvulaRing} represents an evolution in the {FAB} method for the cycle diagnosis of women with regular, irregular or anovulatory menstrual cycles.}, + pages = {256--260}, number = {3}, - urldate = {2025-02-21}, - journal = {Gynecological Endocrinology}, + journaltitle = {Gynecological Endocrinology}, author = {Regidor, Pedro-Antonio and Kaczmarczyk, Marta and Schiweck, Esther and Goeckenjan-Festag, Maren and Alexander, Henry}, - month = mar, - year = {2018}, + urldate = {2025-02-21}, + date = {2018-03-04}, pmid = {29082805}, note = {Publisher: Taylor \& Francis \_eprint: https://doi.org/10.1080/09513590.2017.1390737}, - keywords = {Infertility, central nervous system, circadian rhythm, circamensual rhythm, core body temperature, fertile window, vagina}, - pages = {256--260}, + keywords = {central nervous system, circadian rhythm, circamensual rhythm, core body temperature, fertile window, Infertility, vagina}, file = {Full Text PDF:/home/alex/Zotero/storage/ITD68HTW/Regidor et al. - 2018 - Identification and prediction of the fertile window with a new web-based medical device using a vagi.pdf:application/pdf}, } @article{alexander_fertilitatsmonitoring_2014, - title = {Fertilitätsmonitoring mit vaginalem {Biosensor} ({OvulaRing}©)}, + title = {Fertilitätsmonitoring mit vaginalem Biosensor ({OvulaRing}©)}, volume = {74}, issn = {0016-5751}, url = {https://www.thieme-connect.com/products/ejournals/abstract/10.1055/s-0034-1388603}, doi = {10.1055/s-0034-1388603}, abstract = {Thieme E-Books \& E-Journals}, - language = {de}, - urldate = {2025-02-21}, - journal = {Geburtshilfe und Frauenheilkunde}, - author = {Alexander, H. and Kaczmarczyk, M. and Pretzsch, G. and Kersken, T. and Puschmann, D. and Schiwek, E. and Goeckenjan, M.}, - month = sep, - year = {2014}, - keywords = {60. Kongress der Deutschen Gesellschaft für Gynäkologie und Geburtshilfe}, pages = {FV\_08\_05}, + journaltitle = {Geburtshilfe und Frauenheilkunde}, + author = {Alexander, H. and Kaczmarczyk, M. and Pretzsch, G. and Kersken, T. and Puschmann, D. and Schiwek, E. and Goeckenjan, M.}, + urldate = {2025-02-21}, + date = {2014-09-05}, + langid = {german}, + keywords = {60. Kongress der Deutschen Gesellschaft für Gynäkologie und Geburtshilfe}, file = {Snapshot:/home/alex/Zotero/storage/HPL6XYJW/s-0034-1388603.html:text/html}, } @inproceedings{regidor_identifizierung_2018, - title = {Identifizierung und {Vorhersage} des fertilen {Fensters} des weiblichen {Zyklus} mit einem neuen web basierten {Medizinprodukt} ({OvulaRing}®).}, + title = {Identifizierung und Vorhersage des fertilen Fensters des weiblichen Zyklus mit einem neuen web basierten Medizinprodukt ({OvulaRing}®).}, volume = {78}, - copyright = {Georg Thieme Verlag KG Stuttgart · New York}, + rights = {Georg Thieme Verlag {KG} Stuttgart · New York}, url = {https://www.thieme-connect.com/products/ejournals/html/10.1055/s-0038-1671278}, doi = {10.1055/s-0038-1671278}, abstract = {Thieme E-Books \& E-Journals}, - language = {de}, - urldate = {2025-02-21}, - booktitle = {Geburtshilfe und {Frauenheilkunde}}, - publisher = {Georg Thieme Verlag KG}, - author = {Regidor, P. A. and Alexander, H.}, - month = sep, - year = {2018}, - note = {ISSN: 0016-5751}, - keywords = {Präsidentin der DGGG e.V.: Prof. Dr. Birgit Seelbach-Göbel{\textless}/conf-president{\textgreater}{\textless}/conference{\textgreater}}, pages = {P 23}, + booktitle = {Geburtshilfe und Frauenheilkunde}, + publisher = {Georg Thieme Verlag {KG}}, + author = {Regidor, P. A. and Alexander, H.}, + urldate = {2025-02-21}, + date = {2018-09-20}, + langid = {german}, + note = {{ISSN}: 0016-5751}, + keywords = {Präsidentin der {DGGG} e.V.: Prof. Dr. Birgit Seelbach-Göbel{\textless}/conf-president{\textgreater}{\textless}/conference{\textgreater}}, file = {Snapshot:/home/alex/Zotero/storage/DW6578ZM/s-0038-1671278.html:text/html}, } @@ -778,18 +437,17 @@ Conclusions Continuous body temperature monitoring with a vaginal biosensor can title = {The prediction of ovulation: a comparison of the basal body temperature graph, cervical mucus score, and real-time pelvic ultrasonography}, volume = {43}, issn = {0015-0282}, - shorttitle = {The prediction of ovulation}, url = {https://www.sciencedirect.com/science/article/pii/S0015028216484360}, doi = {10.1016/S0015-0282(16)48436-0}, - abstract = {Ninety-five menstrual cycles were studied in 20 women undergoing donor artificial insemination (AID). In 49 cycles basal body temperature (BBT) change…}, - language = {en-US}, - number = {3}, - urldate = {2025-02-20}, - journal = {Fertility and Sterility}, - month = mar, - year = {1985}, - note = {Publisher: Elsevier}, + shorttitle = {The prediction of ovulation}, + abstract = {Ninety-five menstrual cycles were studied in 20 women undergoing donor artificial insemination ({AID}). In 49 cycles basal body temperature ({BBT}) change…}, pages = {385--388}, + number = {3}, + journaltitle = {Fertility and Sterility}, + urldate = {2025-02-20}, + date = {1985-03-01}, + langid = {american}, + note = {Publisher: Elsevier}, file = {Snapshot:/home/alex/Zotero/storage/FYI9GPUM/S0015028216484360.html:text/html}, } @@ -800,134 +458,128 @@ Conclusions Continuous body temperature monitoring with a vaginal biosensor can url = {https://www.sciencedirect.com/science/article/abs/pii/S0002937804018770}, doi = {10.1016/j.ajog.2004.11.006}, abstract = {The purpose of this study was to evaluate changes in cervicovaginal fluid characteristics to identify ovulation.Several ovulation indicators were stud…}, - language = {en-US}, - number = {1}, - urldate = {2025-02-20}, - journal = {American Journal of Obstetrics and Gynecology}, - month = jul, - year = {2005}, - note = {Publisher: Mosby}, pages = {71--75}, + number = {1}, + journaltitle = {American Journal of Obstetrics and Gynecology}, + urldate = {2025-02-20}, + date = {2005-07-01}, + langid = {american}, + note = {Publisher: Mosby}, file = {Snapshot:/home/alex/Zotero/storage/2JIAY9KL/S0002937804018770.html:text/html}, } @article{sato_novel_2024, - title = {Novel {Methodology} for {Identifying} the {Occurrence} of {Ovulation} by {Estimating} {Core} {Body} {Temperature} {During} {Sleeping}: {Validity} and {Effectiveness} {Study}}, + title = {Novel Methodology for Identifying the Occurrence of Ovulation by Estimating Core Body Temperature During Sleeping: Validity and Effectiveness Study}, volume = {8}, - copyright = {Unless stated otherwise, all articles are open-access distributed under the terms of the Creative Commons Attribution License (http://creativecommons.org/licenses/by/2.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work ("first published in the Journal of Medical Internet Research...") is properly cited with original URL and bibliographic citation information. The complete bibliographic information, a link to the original publication on http://www.jmir.org/, as well as this copyright and license information must be included.}, - shorttitle = {Novel {Methodology} for {Identifying} the {Occurrence} of {Ovulation} by {Estimating} {Core} {Body} {Temperature} {During} {Sleeping}}, + rights = {Unless stated otherwise, all articles are open-access distributed under the terms of the Creative Commons Attribution License (http://creativecommons.org/licenses/by/2.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work ("first published in the Journal of Medical Internet Research...") is properly cited with original {URL} and bibliographic citation information. The complete bibliographic information, a link to the original publication on http://www.jmir.org/, as well as this copyright and license information must be included.}, url = {https://formative.jmir.org/2024/1/e55834}, doi = {10.2196/55834}, - abstract = {Background: Body temperature is the most-used noninvasive biomarker to determine menstrual cycle and ovulation. However, issues related to its low accuracy are still under discussion. Objective: This study aimed to improve the accuracy of identifying the presence or absence of ovulation within a menstrual cycle. We investigated whether core body temperature (CBT) estimation can improve the accuracy of temperature biphasic shift discrimination in the menstrual cycle. The study consisted of 2 parts: experiment 1 assessed the validity of the CBT estimation method, while experiment 2 focused on the effectiveness of the method in discriminating biphasic temperature shifts. Methods: In experiment 1, healthy women aged between 18 and 40 years had their true CBT measured using an ingestible thermometer and their CBT estimated from skin temperature and ambient temperature measured during sleep in both the follicular and luteal phases of their menstrual cycles. This study analyzed the differences between these 2 measurements, the variations in temperature between the 2 phases, and the repeated measures correlation between the true and estimated CBT. Experiment 2 followed a similar methodology, but focused on evaluating the diagnostic accuracy of these 2 temperature measurement approaches (estimated CBT and traditional oral basal body temperature [BBT]) for identifying ovulatory cycles. This was performed using urine luteinizing hormone (LH) as the reference standard. Menstrual cycles were categorized based on the results of the LH tests, and a temperature shift was identified using a specific criterion called the “three-over-six rule.” This rule and the nested design of the study facilitated the assessment of diagnostic measures, such as sensitivity and specificity. Results: The main findings showed that CBT estimated from skin temperature and ambient temperature during sleep was consistently lower than directly measured CBT in both the follicular and luteal phases of the menstrual cycle. Despite this, the pattern of temperature variation between these phases was comparable for both the estimated and true CBT measurements, suggesting that the estimated CBT accurately reflected the cyclical variations in the true CBT. Significantly, the CBT estimation method showed higher sensitivity and specificity for detecting the occurrence of ovulation than traditional oral BBT measurements, highlighting its potential as an effective tool for reproductive health monitoring. The current method for estimating the CBT provides a practical and noninvasive method for monitoring CBT, which is essential for identifying biphasic shifts in the BBT throughout the menstrual cycle. Conclusions: This study demonstrated that the estimated CBT derived from skin temperature and ambient temperature during sleep accurately captures variations in true CBT and is more accurate in determining the presence or absence of ovulation than traditional oral BBT measurements. This method holds promise for improving reproductive health monitoring and understanding of menstrual cycle dynamics.}, - language = {EN}, - number = {1}, - urldate = {2025-02-20}, - journal = {JMIR Formative Research}, - author = {Sato, Daisuke and Ikarashi, Koyuki and Nakajima, Fumiko and Fujimoto, Tomomi}, - month = jul, - year = {2024}, - note = {Company: JMIR Formative Research -Distributor: JMIR Formative Research -Institution: JMIR Formative Research -Label: JMIR Formative Research -Publisher: JMIR Publications Inc., Toronto, Canada}, + shorttitle = {Novel Methodology for Identifying the Occurrence of Ovulation by Estimating Core Body Temperature During Sleeping}, + abstract = {Background: Body temperature is the most-used noninvasive biomarker to determine menstrual cycle and ovulation. However, issues related to its low accuracy are still under discussion. Objective: This study aimed to improve the accuracy of identifying the presence or absence of ovulation within a menstrual cycle. We investigated whether core body temperature ({CBT}) estimation can improve the accuracy of temperature biphasic shift discrimination in the menstrual cycle. The study consisted of 2 parts: experiment 1 assessed the validity of the {CBT} estimation method, while experiment 2 focused on the effectiveness of the method in discriminating biphasic temperature shifts. Methods: In experiment 1, healthy women aged between 18 and 40 years had their true {CBT} measured using an ingestible thermometer and their {CBT} estimated from skin temperature and ambient temperature measured during sleep in both the follicular and luteal phases of their menstrual cycles. This study analyzed the differences between these 2 measurements, the variations in temperature between the 2 phases, and the repeated measures correlation between the true and estimated {CBT}. Experiment 2 followed a similar methodology, but focused on evaluating the diagnostic accuracy of these 2 temperature measurement approaches (estimated {CBT} and traditional oral basal body temperature [{BBT}]) for identifying ovulatory cycles. This was performed using urine luteinizing hormone ({LH}) as the reference standard. Menstrual cycles were categorized based on the results of the {LH} tests, and a temperature shift was identified using a specific criterion called the “three-over-six rule.” This rule and the nested design of the study facilitated the assessment of diagnostic measures, such as sensitivity and specificity. Results: The main findings showed that {CBT} estimated from skin temperature and ambient temperature during sleep was consistently lower than directly measured {CBT} in both the follicular and luteal phases of the menstrual cycle. Despite this, the pattern of temperature variation between these phases was comparable for both the estimated and true {CBT} measurements, suggesting that the estimated {CBT} accurately reflected the cyclical variations in the true {CBT}. Significantly, the {CBT} estimation method showed higher sensitivity and specificity for detecting the occurrence of ovulation than traditional oral {BBT} measurements, highlighting its potential as an effective tool for reproductive health monitoring. The current method for estimating the {CBT} provides a practical and noninvasive method for monitoring {CBT}, which is essential for identifying biphasic shifts in the {BBT} throughout the menstrual cycle. Conclusions: This study demonstrated that the estimated {CBT} derived from skin temperature and ambient temperature during sleep accurately captures variations in true {CBT} and is more accurate in determining the presence or absence of ovulation than traditional oral {BBT} measurements. This method holds promise for improving reproductive health monitoring and understanding of menstrual cycle dynamics.}, pages = {e55834}, + number = {1}, + journaltitle = {{JMIR} Formative Research}, + author = {Sato, Daisuke and Ikarashi, Koyuki and Nakajima, Fumiko and Fujimoto, Tomomi}, + urldate = {2025-02-20}, + date = {2024-07-05}, + note = {Company: {JMIR} Formative Research +Distributor: {JMIR} Formative Research +Institution: {JMIR} Formative Research +Label: {JMIR} Formative Research +Publisher: {JMIR} Publications Inc., Toronto, Canada}, file = {PubMed Central Full Text PDF:/home/alex/Zotero/storage/NPBA84BU/Sato et al. - 2024 - Novel Methodology for Identifying the Occurrence of Ovulation by Estimating Core Body Temperature Du.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/BAJUVPFE/e55834.html:text/html}, } @article{royston_identifying_1991-1, title = {Identifying the fertile phase of the human menstrual cycle}, volume = {10}, - copyright = {Copyright © 1991 John Wiley \& Sons, Ltd.}, + rights = {Copyright © 1991 John Wiley \& Sons, Ltd.}, issn = {1097-0258}, url = {https://onlinelibrary.wiley.com/doi/abs/10.1002/sim.4780100207}, doi = {10.1002/sim.4780100207}, - abstract = {The identification of the human fertile phase as the time during which a woman or a couple may conceive is elusive. The fertile time depends on many factors in each individual menstrual cycle and may be said to be more of a statistical than a physiological entity. This paper reviews the application of statistical methods to three areas related to conception and the fertile phase. The first is the prediction and detection of ovulation from serial measurements, such as hormones, basal body temperature and cervical mucus, throughout the menstrual cycle. Typically, such variables increase from some baseline level to a peak around ovulation (the most fertile time), then subside to low levels in the postovulatory phase. The statistical challenge is to detect the rise (signalling the onset of potential fertility) and subsequent fall. Analytic methods considered include thresholds, Bayesian change-point models and particularly the cumulative sum (cusum) technique which is both simple to apply and understand, and effective. The second area comprises appropriate methods of analysing and interpreting data from clinical studies of the fertile phase, especially in so-alled natural family planning (NFP) where it is usual for women to observe several indices of potential fertility. Such studies usually try to establish the temporal relationships between markers of the fertile phase and examine the success of different combinations of markers in delineating the fertile time in comparison with a standard ‘defined’ phase, for example, the interval from three days before to two days after the peak of luteinizing hormone. The third area is the assessment of the probability of conception on certain days of the cycle, which is vital to the understanding of the fertile phase and its application to NFP. Direct estimation of such probabilities is impractical; instead, resort must be made to estimation by maximum likelihood of the parameters of specially constructed models. Suitable models are described. Finally, the need for a new prospective study of the probability of conception in relation to the markers of the fertile phase used in the symptothermal method of NFP is discussed.}, - language = {en}, - number = {2}, - urldate = {2025-02-20}, - journal = {Statistics in Medicine}, - author = {Royston, Patrick}, - year = {1991}, - note = {\_eprint: https://onlinelibrary.wiley.com/doi/pdf/10.1002/sim.4780100207}, + abstract = {The identification of the human fertile phase as the time during which a woman or a couple may conceive is elusive. The fertile time depends on many factors in each individual menstrual cycle and may be said to be more of a statistical than a physiological entity. This paper reviews the application of statistical methods to three areas related to conception and the fertile phase. The first is the prediction and detection of ovulation from serial measurements, such as hormones, basal body temperature and cervical mucus, throughout the menstrual cycle. Typically, such variables increase from some baseline level to a peak around ovulation (the most fertile time), then subside to low levels in the postovulatory phase. The statistical challenge is to detect the rise (signalling the onset of potential fertility) and subsequent fall. Analytic methods considered include thresholds, Bayesian change-point models and particularly the cumulative sum (cusum) technique which is both simple to apply and understand, and effective. The second area comprises appropriate methods of analysing and interpreting data from clinical studies of the fertile phase, especially in so-alled natural family planning ({NFP}) where it is usual for women to observe several indices of potential fertility. Such studies usually try to establish the temporal relationships between markers of the fertile phase and examine the success of different combinations of markers in delineating the fertile time in comparison with a standard ‘defined’ phase, for example, the interval from three days before to two days after the peak of luteinizing hormone. The third area is the assessment of the probability of conception on certain days of the cycle, which is vital to the understanding of the fertile phase and its application to {NFP}. Direct estimation of such probabilities is impractical; instead, resort must be made to estimation by maximum likelihood of the parameters of specially constructed models. Suitable models are described. Finally, the need for a new prospective study of the probability of conception in relation to the markers of the fertile phase used in the symptothermal method of {NFP} is discussed.}, pages = {221--240}, + number = {2}, + journaltitle = {Statistics in Medicine}, + author = {Royston, Patrick}, + urldate = {2025-02-20}, + date = {1991}, + langid = {english}, + note = {\_eprint: https://onlinelibrary.wiley.com/doi/pdf/10.1002/sim.4780100207}, file = {Snapshot:/home/alex/Zotero/storage/9XL95LUJ/sim.html:text/html}, } @article{su_detection_2017, title = {Detection of ovulation, a review of currently available methods}, volume = {2}, - copyright = {© 2017 The Authors. Bioengineering \& Translational Medicine is published by Wiley Periodicals, Inc. on behalf of The American Institute of Chemical Engineers}, + rights = {© 2017 The Authors. Bioengineering \& Translational Medicine is published by Wiley Periodicals, Inc. on behalf of The American Institute of Chemical Engineers}, issn = {2380-6761}, url = {https://onlinelibrary.wiley.com/doi/abs/10.1002/btm2.10058}, doi = {10.1002/btm2.10058}, abstract = {The ability to identify the precise time of ovulation is important for women who want to plan conception or practice contraception. Here, we review the current literature on various methods for detecting ovulation including a review of point-of-care device technology. We incorporate an examination of methods to detect ovulation that have been developed and practiced for decades and analyze the indications and limitations of each—transvaginal ultrasonography, urinary luteinizing hormone detection, serum progesterone and urinary pregnanediol 3-glucuronide detection, urinary follicular stimulating hormone detection, basal body temperature monitoring, and cervical mucus and salivary ferning analysis. Some point-of-care ovulation detection devices have been developed and commercialized based on these methods, however previous research was limited by small sample size and an inconsistent standard reference to true ovulation.}, - language = {en}, - number = {3}, - urldate = {2025-02-20}, - journal = {Bioengineering \& Translational Medicine}, - author = {Su, Hsiu-Wei and Yi, Yu-Chiao and Wei, Ting-Yen and Chang, Ting-Chang and Cheng, Chao-Min}, - year = {2017}, - note = {\_eprint: https://onlinelibrary.wiley.com/doi/pdf/10.1002/btm2.10058}, - keywords = {ovulation detection, family planning, fertility window}, pages = {238--246}, + number = {3}, + journaltitle = {Bioengineering \& Translational Medicine}, + author = {Su, Hsiu-Wei and Yi, Yu-Chiao and Wei, Ting-Yen and Chang, Ting-Chang and Cheng, Chao-Min}, + urldate = {2025-02-20}, + date = {2017}, + langid = {english}, + note = {\_eprint: https://onlinelibrary.wiley.com/doi/pdf/10.1002/btm2.10058}, + keywords = {family planning, fertility window, ovulation detection}, file = {Full Text PDF:/home/alex/Zotero/storage/ZEACCGE5/Su et al. - 2017 - Detection of ovulation, a review of currently available methods.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/RDDQD8EA/btm2.html:text/html}, } @article{noauthor_basal_1981, - title = {Basal {Body} {Temperature}: {Unreliable} {Method} of {Ovulation} {Detection}}, + title = {Basal Body Temperature: Unreliable Method of Ovulation Detection}, volume = {36}, issn = {0015-0282}, - shorttitle = {Basal {Body} {Temperature}}, url = {https://www.sciencedirect.com/science/article/pii/S0015028216459169}, doi = {10.1016/S0015-0282(16)45916-9}, - abstract = {Basal body temperature (BBT) charts from menstrual cycles of 98 women were evaluated by six experienced physicians. The time of ovulation as estimated…}, - language = {en-US}, - number = {6}, - urldate = {2025-02-20}, - journal = {Fertility and Sterility}, - month = dec, - year = {1981}, - note = {Publisher: Elsevier}, + shorttitle = {Basal Body Temperature}, + abstract = {Basal body temperature ({BBT}) charts from menstrual cycles of 98 women were evaluated by six experienced physicians. The time of ovulation as estimated…}, pages = {729--733}, + number = {6}, + journaltitle = {Fertility and Sterility}, + urldate = {2025-02-20}, + date = {1981-12-01}, + langid = {american}, + note = {Publisher: Elsevier}, file = {Snapshot:/home/alex/Zotero/storage/AWZ4LX8D/S0015028216459169.html:text/html}, } @article{luo_detection_2020, - title = {Detection and {Prediction} of {Ovulation} {From} {Body} {Temperature} {Measured} by an {In}-{Ear} {Wearable} {Thermometer}}, + title = {Detection and Prediction of Ovulation From Body Temperature Measured by an In-Ear Wearable Thermometer}, volume = {67}, issn = {1558-2531}, url = {https://ieeexplore.ieee.org/abstract/document/8715448}, doi = {10.1109/TBME.2019.2916823}, - abstract = {Objective: We present a non-invasive wearable device for fertility monitoring and propose an effective and flexible statistical learning algorithm to detect and predict ovulation using data captured by this device. Methods: The system consists of an earpiece, which measures the ear canal temperature every 5 min during night sleep hours, and a base station that transmits data to a smartphone application for analysis. We establish a data-cleaning protocol for data preprocessing and then fit a Hidden Markov Model (HMM) with two hidden states of high and low temperature to identify the more probable state of each time point via the predicted probabilities. Finally, a post-processing procedure is developed to incorporate biorhythm information to form a time-course biphasic profile for each subject. Results: The performance of the proposed algorithms applied to data collected by the device are compared with traditional methods in terms of match rate with self-reported ovulation days confirmed with an ovulation test kit. Empirical study results from a group of 34 users yielded significant improvements over the traditional methods in terms of detection accuracy (with sensitivity 92.31\%) and prediction power (23.07-31.55\% higher). Conclusion: We demonstrated the feasibility for reliable ovulation detection and prediction with high-frequency temperature data collected by a non-invasive wearable device. Significance: Traditional fertility monitoring methods are often either inaccurate or inconvenient. The wearable device and learning algorithm presented in this paper provide a user friendly and reliable platform for tracking ovulation, which may have a broad impact on both fertility research and real-world family planning.}, - number = {2}, - urldate = {2025-02-20}, - journal = {IEEE Transactions on Biomedical Engineering}, - author = {Luo, Lan and She, Xichen and Cao, Jiexuan and Zhang, Yunlong and Li, Yijiang and Song, Peter X. K.}, - month = feb, - year = {2020}, - note = {Conference Name: IEEE Transactions on Biomedical Engineering}, - keywords = {Basal body temperature, Biomedical monitoring, Hidden Markov Model (HMM), Hidden Markov models, ovulation, prediction, Prediction algorithms, Temperature distribution, Temperature measurement, Temperature sensors, tracking data, wearable}, + abstract = {Objective: We present a non-invasive wearable device for fertility monitoring and propose an effective and flexible statistical learning algorithm to detect and predict ovulation using data captured by this device. Methods: The system consists of an earpiece, which measures the ear canal temperature every 5 min during night sleep hours, and a base station that transmits data to a smartphone application for analysis. We establish a data-cleaning protocol for data preprocessing and then fit a Hidden Markov Model ({HMM}) with two hidden states of high and low temperature to identify the more probable state of each time point via the predicted probabilities. Finally, a post-processing procedure is developed to incorporate biorhythm information to form a time-course biphasic profile for each subject. Results: The performance of the proposed algorithms applied to data collected by the device are compared with traditional methods in terms of match rate with self-reported ovulation days confirmed with an ovulation test kit. Empirical study results from a group of 34 users yielded significant improvements over the traditional methods in terms of detection accuracy (with sensitivity 92.31\%) and prediction power (23.07-31.55\% higher). Conclusion: We demonstrated the feasibility for reliable ovulation detection and prediction with high-frequency temperature data collected by a non-invasive wearable device. Significance: Traditional fertility monitoring methods are often either inaccurate or inconvenient. The wearable device and learning algorithm presented in this paper provide a user friendly and reliable platform for tracking ovulation, which may have a broad impact on both fertility research and real-world family planning.}, pages = {512--522}, + number = {2}, + journaltitle = {{IEEE} Transactions on Biomedical Engineering}, + author = {Luo, Lan and She, Xichen and Cao, Jiexuan and Zhang, Yunlong and Li, Yijiang and Song, Peter X. K.}, + urldate = {2025-02-20}, + date = {2020-02}, + note = {Conference Name: {IEEE} Transactions on Biomedical Engineering}, + keywords = {Basal body temperature, Biomedical monitoring, Hidden Markov Model ({HMM}), Hidden Markov models, ovulation, prediction, Prediction algorithms, Temperature distribution, Temperature measurement, Temperature sensors, tracking data, wearable}, file = {IEEE Xplore Abstract Record:/home/alex/Zotero/storage/W8SQ4ASJ/8715448.html:text/html}, } @article{owen_physiological_2013, - title = {Physiological {Signs} of {Ovulation} and {Fertility} {Readily} {Observable} by {Women}}, + title = {Physiological Signs of Ovulation and Fertility Readily Observable by Women}, volume = {80}, issn = {0024-3639}, url = {https://doi.org/10.1179/0024363912Z.0000000005}, doi = {10.1179/0024363912Z.0000000005}, - abstract = {IntroductionConfirmation of ovulation can be difficult in clinical practice, as gold standard methods including serial transvaginal ultrasonography, serum luteinizing hormone (LH) measurements, or laparoscopic follicular observation are impractical. Numerous surrogate markers have been proposed and evaluated in relation to these gold standards that have more practical clinical applications.PurposeTo review the evidence on physiological signs of ovulation timing and fertility in order to determine valid markers that can be easily identified by women.MethodsA literature review of primary resources in Ovid Medline was undertaken to identify studies examining physiological signs as they relate to gold standard assessment of ovulation. Studies examining the efficacy/effectiveness of different types of natural family planning were excluded.ResultsThe most commonly encountered physiological signs were urine LH, cervical mucus, and basal body temperature (BBT). Urine LH as assessed by home monitoring systems indicated ovulation 91 percent of the time during the 2 days of peak fertility on the monitor and 97 percent during the 2 peak days plus 1. Cervical mucus peak characteristics were identified 78 percent of the time ±1 day, and 91 percent of the time ±2 days of LH surge indicating ovulation. Further research supports the importance of cervical mucus in overall fertility, as conception rates were more closely related to mucus quality than to timing of intercourse related to ovulation. As a lone indicator of ovulation, BBT is at best a retrospective marker, and functions best in conjunction with other signs of ovulation. Additionally, salivary ferning, salivary and vaginal fluid electrical potential, finger–finger electrical potential, and differential skin temperature were postulated as possible indicators, but were not found to be temporally related to ovulation. The research on differential skin temperature is promising, but minimal thus far in number, and has not been evaluated as an adjunct to BBT as yet.ConclusionHome urinary LH monitors are becoming more widely available and less expensive giving women the potential to assess the ovulatory status of their cycle in real time. Cervical mucus observation is an effective and cost-efficient method, but requires some teaching to increase the confidence of users. In conjunction, LH monitors and cervical mucus can give the best indication of fertility and ovulation timing.}, - language = {en}, - number = {1}, - urldate = {2025-02-20}, - journal = {The Linacre Quarterly}, - author = {Owen, Martin}, - month = jan, - year = {2013}, - note = {Publisher: SAGE Publications Inc}, + abstract = {{IntroductionConfirmation} of ovulation can be difficult in clinical practice, as gold standard methods including serial transvaginal ultrasonography, serum luteinizing hormone ({LH}) measurements, or laparoscopic follicular observation are impractical. Numerous surrogate markers have been proposed and evaluated in relation to these gold standards that have more practical clinical applications.{PurposeTo} review the evidence on physiological signs of ovulation timing and fertility in order to determine valid markers that can be easily identified by women.{MethodsA} literature review of primary resources in Ovid Medline was undertaken to identify studies examining physiological signs as they relate to gold standard assessment of ovulation. Studies examining the efficacy/effectiveness of different types of natural family planning were excluded.{ResultsThe} most commonly encountered physiological signs were urine {LH}, cervical mucus, and basal body temperature ({BBT}). Urine {LH} as assessed by home monitoring systems indicated ovulation 91 percent of the time during the 2 days of peak fertility on the monitor and 97 percent during the 2 peak days plus 1. Cervical mucus peak characteristics were identified 78 percent of the time ±1 day, and 91 percent of the time ±2 days of {LH} surge indicating ovulation. Further research supports the importance of cervical mucus in overall fertility, as conception rates were more closely related to mucus quality than to timing of intercourse related to ovulation. As a lone indicator of ovulation, {BBT} is at best a retrospective marker, and functions best in conjunction with other signs of ovulation. Additionally, salivary ferning, salivary and vaginal fluid electrical potential, finger–finger electrical potential, and differential skin temperature were postulated as possible indicators, but were not found to be temporally related to ovulation. The research on differential skin temperature is promising, but minimal thus far in number, and has not been evaluated as an adjunct to {BBT} as yet.{ConclusionHome} urinary {LH} monitors are becoming more widely available and less expensive giving women the potential to assess the ovulatory status of their cycle in real time. Cervical mucus observation is an effective and cost-efficient method, but requires some teaching to increase the confidence of users. In conjunction, {LH} monitors and cervical mucus can give the best indication of fertility and ovulation timing.}, pages = {17--23}, + number = {1}, + journaltitle = {Linacre Q}, + author = {Owen, Martin}, + urldate = {2025-02-20}, + date = {2013-01-01}, + langid = {english}, + note = {Publisher: {SAGE} Publications Inc}, file = {Full Text:/home/alex/Zotero/storage/IBVUICCU/Owen - 2013 - Physiological Signs of Ovulation and Fertility Readily Observable by Women.pdf:application/pdf}, } @@ -937,43 +589,29 @@ Publisher: JMIR Publications Inc., Toronto, Canada}, issn = {2399-3529}, url = {https://doi.org/10.1093/hropen/hoaa011}, doi = {10.1093/hropen/hoaa011}, - abstract = {What variations underlie the menstrual cycle length and ovulation day of women trying to conceive?Big data from a connected ovulation test revealed the extent of variation in menstrual cycle length and ovulation day in women trying to conceive.Timing intercourse to coincide with the fertile period of a woman maximises the chances of conception. The day of ovulation varies on an inter- and intra-individual level.A total of 32 595 women who had purchased a connected ovulation test system contributed 75 981 cycles for analysis. Day of ovulation was determined from the fertility test results. The connected home ovulation test system enables users to identify their fertile phase. The app benefits users by enabling them to understand their personal fertility information. During each menstrual cycle, users input their perceived cycle length into an accessory application, and data on hormone levels from the tests are uploaded to the application and stored in an anonymised cloud database. This study compared users’ perceived cycle characteristics with actual cycle characteristics. The perceived and actual cycle length information was analysed to provide population ranges.This study analysed data from the at-home use of a commercially available connected home ovulation test by women across the USA and UK.Overall, 25.3\% of users selected a 28-day cycle as their perceived cycle length; however, only 12.4\% of users actually had a 28-day cycle. Most women (87\%) had actual menstrual cycle lengths between 23 and 35 days, with a normal distribution centred on day 28, and over half of the users (52\%) had cycles that varied by 5 days or more. There was a 10-day spread of observed ovulation days for a 28-day cycle, with the most common day of ovulation being Day 15. Similar variation was observed for all cycle lengths examined. For users who conducted a test on every day requested by the app, a luteinising hormone (LH) surge was detected in 97.9\% of cycles.Data were from a self-selected population of women who were prepared to purchase a commercially available product to aid conception and so may not fully represent the wider population. No corresponding demographic data were collected with the cycle information.Using big data has provided more personalised insights into women’s fertility; this could enable women trying to conceive to better time intercourse, increasing the likelihood of conception.The study was funded by SPD Development Company Ltd (Bedford, UK), a fully owned subsidiary of SPD Swiss Precision Diagnostics GmbH (Geneva, Switzerland). I.S., B.G. and S.J. are employees of the SPD Development Company Ltd.}, - number = {2}, - urldate = {2025-02-20}, - journal = {Human Reproduction Open}, - author = {Soumpasis, I and Grace, B and Johnson, S}, - month = feb, - year = {2020}, + abstract = {What variations underlie the menstrual cycle length and ovulation day of women trying to conceive?Big data from a connected ovulation test revealed the extent of variation in menstrual cycle length and ovulation day in women trying to conceive.Timing intercourse to coincide with the fertile period of a woman maximises the chances of conception. The day of ovulation varies on an inter- and intra-individual level.A total of 32 595 women who had purchased a connected ovulation test system contributed 75 981 cycles for analysis. Day of ovulation was determined from the fertility test results. The connected home ovulation test system enables users to identify their fertile phase. The app benefits users by enabling them to understand their personal fertility information. During each menstrual cycle, users input their perceived cycle length into an accessory application, and data on hormone levels from the tests are uploaded to the application and stored in an anonymised cloud database. This study compared users’ perceived cycle characteristics with actual cycle characteristics. The perceived and actual cycle length information was analysed to provide population ranges.This study analysed data from the at-home use of a commercially available connected home ovulation test by women across the {USA} and {UK}.Overall, 25.3\% of users selected a 28-day cycle as their perceived cycle length; however, only 12.4\% of users actually had a 28-day cycle. Most women (87\%) had actual menstrual cycle lengths between 23 and 35 days, with a normal distribution centred on day 28, and over half of the users (52\%) had cycles that varied by 5 days or more. There was a 10-day spread of observed ovulation days for a 28-day cycle, with the most common day of ovulation being Day 15. Similar variation was observed for all cycle lengths examined. For users who conducted a test on every day requested by the app, a luteinising hormone ({LH}) surge was detected in 97.9\% of cycles.Data were from a self-selected population of women who were prepared to purchase a commercially available product to aid conception and so may not fully represent the wider population. No corresponding demographic data were collected with the cycle information.Using big data has provided more personalised insights into women’s fertility; this could enable women trying to conceive to better time intercourse, increasing the likelihood of conception.The study was funded by {SPD} Development Company Ltd (Bedford, {UK}), a fully owned subsidiary of {SPD} Swiss Precision Diagnostics {GmbH} (Geneva, Switzerland). I.S., B.G. and S.J. are employees of the {SPD} Development Company Ltd.}, pages = {hoaa011}, - annote = { - -not really interesting, as they only look at cycle length - - -they use “perceived” cycle length, which is hard to defend, when there are intermediate bleedings etc. - - - - - -}, + number = {2}, + journaltitle = {Human Reproduction Open}, + author = {Soumpasis, I and Grace, B and Johnson, S}, + urldate = {2025-02-20}, + date = {2020-02-01}, file = {Full Text PDF:/home/alex/Zotero/storage/PS9UC298/Soumpasis et al. - 2020 - Real-life insights on menstrual cycles and ovulation using big data.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/LTFATQIC/5820371.html:text/html}, } @article{brewis_demographic_2005, - title = {Demographic {Evidence} {That} {Human} {Ovulation} {Is} {Undetectable} ({At} {Least} in {Pair} {Bonds})}, + title = {Demographic Evidence That Human Ovulation Is Undetectable (At Least in Pair Bonds)}, volume = {46}, issn = {0011-3204}, url = {https://www.journals.uchicago.edu/doi/abs/10.1086/430016}, doi = {10.1086/430016}, - number = {3}, - urldate = {2025-02-20}, - journal = {Current Anthropology}, - author = {Brewis, Alexandra and Meyer, Mary}, - month = jun, - year = {2005}, - note = {Publisher: The University of Chicago Press}, pages = {465--471}, + number = {3}, + journaltitle = {Current Anthropology}, + author = {Brewis, Alexandra and Meyer, Mary}, + urldate = {2025-02-20}, + date = {2005-06}, + note = {Publisher: The University of Chicago Press}, } @article{noauthor_monitoring_1987, @@ -983,14 +621,13 @@ they use “perceived” cycle length, which is hard to defend, when there are i url = {https://www.sciencedirect.com/science/article/pii/S0015028216500028}, doi = {10.1016/S0015-0282(16)50002-8}, abstract = {This study was designed to evaluate the accuracy of various methods in predicting and detecting ovulation in 14 spontaneous and 17 clomiphene citrate …}, - language = {en-US}, - number = {2}, - urldate = {2025-02-20}, - journal = {Fertility and Sterility}, - month = feb, - year = {1987}, - note = {Publisher: Elsevier}, pages = {259--264}, + number = {2}, + journaltitle = {Fertility and Sterility}, + urldate = {2025-02-20}, + date = {1987-02-01}, + langid = {american}, + note = {Publisher: Elsevier}, file = {Snapshot:/home/alex/Zotero/storage/M5P9EZ67/S0015028216500028.html:text/html}, } @@ -1001,14 +638,13 @@ they use “perceived” cycle length, which is hard to defend, when there are i url = {https://www.sciencedirect.com/science/article/pii/S0022030216306725}, doi = {10.3168/jds.2016-11247}, abstract = {The objective of this study was to determine the relative importance and contribution of several physiological factors as predictors of pregnancy risk…}, - language = {en-US}, - number = {12}, - urldate = {2025-02-20}, - journal = {Journal of Dairy Science}, - month = dec, - year = {2016}, - note = {Publisher: Elsevier}, pages = {10077--10092}, + number = {12}, + journaltitle = {Journal of Dairy Science}, + urldate = {2025-02-20}, + date = {2016-12-01}, + langid = {american}, + note = {Publisher: Elsevier}, file = {Snapshot:/home/alex/Zotero/storage/EBK9WJP4/S0022030216306725.html:text/html}, } @@ -1018,32 +654,31 @@ they use “perceived” cycle length, which is hard to defend, when there are i issn = {1573-7195}, url = {https://doi.org/10.1007/BF01849284}, doi = {10.1007/BF01849284}, - abstract = {Simple and reliable methods have been sought for both predicting and confirming ovulation. Application of these methods could include management of infertile couples to aid in conception and for increasing the reliability of natural family planning (NFP) as a method of birth control. With the advent of specific hormone assays, serial measurements of estrogens, progesterone (and metabolites), and luteinizing hormone have been the gold standard of monitoring ovarian function in women, However, newer and simpler methodologies have been described and are currently either in use or being tested. These include the measurement of basal body temperature (BBT), the evaluation of the volume, consistency and electro-conductivity of cervicovaginal fluid, salivary steroid content and cellular enzymatic activity, the use of enzyme-linked immunosorbent assays applied to solid-phase formats, and the investigation of new hormonal molecules as markers of reproductive state and function. These new technologies are described herein and their potential for monitoring ovarian function is discussed.}, - language = {en}, - number = {4}, - urldate = {2025-02-20}, - journal = {Advances in Contraception}, - author = {Albertson, B. D. and Zinaman, M. J.}, - month = dec, - year = {1987}, - keywords = {Birth Control, Estrogen, Luteinizing Hormone, Progesterone, Serial Measurement}, + abstract = {Simple and reliable methods have been sought for both predicting and confirming ovulation. Application of these methods could include management of infertile couples to aid in conception and for increasing the reliability of natural family planning ({NFP}) as a method of birth control. With the advent of specific hormone assays, serial measurements of estrogens, progesterone (and metabolites), and luteinizing hormone have been the gold standard of monitoring ovarian function in women, However, newer and simpler methodologies have been described and are currently either in use or being tested. These include the measurement of basal body temperature ({BBT}), the evaluation of the volume, consistency and electro-conductivity of cervicovaginal fluid, salivary steroid content and cellular enzymatic activity, the use of enzyme-linked immunosorbent assays applied to solid-phase formats, and the investigation of new hormonal molecules as markers of reproductive state and function. These new technologies are described herein and their potential for monitoring ovarian function is discussed.}, pages = {263--290}, + number = {4}, + journaltitle = {Adv Contracept}, + author = {Albertson, B. D. and Zinaman, M. J.}, + urldate = {2025-02-20}, + date = {1987-12-01}, + langid = {english}, + keywords = {Birth Control, Estrogen, Luteinizing Hormone, Progesterone, Serial Measurement}, } @inproceedings{serrano_is_2019-1, - address = {Florence, Italy}, - title = {Is {Attention} {Interpretable}?}, + location = {Florence, Italy}, + title = {Is Attention Interpretable?}, url = {https://aclanthology.org/P19-1282/}, doi = {10.18653/v1/P19-1282}, - abstract = {Attention mechanisms have recently boosted performance on a range of NLP tasks. Because attention layers explicitly weight input components' representations, it is also often assumed that attention can be used to identify information that models found important (e.g., specific contextualized word tokens). We test whether that assumption holds by manipulating attention weights in already-trained text classification models and analyzing the resulting differences in their predictions. While we observe some ways in which higher attention weights correlate with greater impact on model predictions, we also find many ways in which this does not hold, i.e., where gradient-based rankings of attention weights better predict their effects than their magnitudes. We conclude that while attention noisily predicts input components' overall importance to a model, it is by no means a fail-safe indicator.}, - urldate = {2025-02-19}, - booktitle = {Proceedings of the 57th {Annual} {Meeting} of the {Association} for {Computational} {Linguistics}}, + abstract = {Attention mechanisms have recently boosted performance on a range of {NLP} tasks. Because attention layers explicitly weight input components' representations, it is also often assumed that attention can be used to identify information that models found important (e.g., specific contextualized word tokens). We test whether that assumption holds by manipulating attention weights in already-trained text classification models and analyzing the resulting differences in their predictions. While we observe some ways in which higher attention weights correlate with greater impact on model predictions, we also find many ways in which this does not hold, i.e., where gradient-based rankings of attention weights better predict their effects than their magnitudes. We conclude that while attention noisily predicts input components' overall importance to a model, it is by no means a fail-safe indicator.}, + eventtitle = {{ACL} 2019}, + pages = {2931--2951}, + booktitle = {Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics}, publisher = {Association for Computational Linguistics}, author = {Serrano, Sofia and Smith, Noah A.}, editor = {Korhonen, Anna and Traum, David and Màrquez, Lluís}, - month = jul, - year = {2019}, - pages = {2931--2951}, + urldate = {2025-02-19}, + date = {2019-07}, file = {Full Text PDF:/home/alex/Zotero/storage/I6J6YP3C/Serrano and Smith - 2019 - Is Attention Interpretable.pdf:application/pdf}, } @@ -1054,282 +689,591 @@ they use “perceived” cycle length, which is hard to defend, when there are i url = {https://www.sciencedirect.com/science/article/abs/pii/S0360835223006915}, doi = {10.1016/j.cie.2023.109667}, abstract = {Recently, transformer-based models have exhibited great performance in multi-horizon time series forecasting tasks. However, the core module of these …}, - language = {en-US}, - urldate = {2025-02-19}, - journal = {Computers \& Industrial Engineering}, - month = nov, - year = {2023}, - note = {Publisher: Pergamon}, pages = {109667}, + journaltitle = {Computers \& Industrial Engineering}, + urldate = {2025-02-19}, + date = {2023-11-01}, + langid = {american}, + note = {Publisher: Pergamon}, file = {Snapshot:/home/alex/Zotero/storage/TJ634T5V/S0360835223006915.html:text/html}, } @article{hu_pattern-oriented_2025, - title = {Pattern-oriented {Attention} {Mechanism} for {Multivariate} {Time} {Series} {Forecasting}}, + title = {Pattern-oriented Attention Mechanism for Multivariate Time Series Forecasting}, volume = {19}, issn = {1556-4681}, url = {https://doi.org/10.1145/3712606}, doi = {10.1145/3712606}, - abstract = {Multivariate time series forecasting is applied in many domains, such as finance, transportation, and industry. The main challenge of precise forecasting lies in accurately capturing latent dependencies. Recent studies develop various frameworks to reduce computational complexity or to enhance the learning of intricate relationships, while lacking interpretability and generality. In this article, we aim to elucidate the capture of dependencies as the recognition of patterns. We believe that patterns can be formally described from two aspects: the shapes of segments that frequently repeat and the corresponding forms of repetitions. Drawing upon this idea, we design a multivariate time series forecasting model named PRformer,1 which incorporates a pattern-oriented attention mechanism and a pattern-based projector. The attention mechanism can perceive different forms of repetitions by embedded with various similarity evaluation metrics between segments, and filter out noise from segments to extract potential patterns with a statistical-driven weighting scheme. The pattern-based projector is employed to form the forecasting results by deriving the representative patterns from the set of potential ones. By incorporating explicit definitions of patterns, PRformer is interpretable and general to various time series scenarios. Experimental results on seven datasets demonstrate that PRformer outperforms six state-of-the-art models by about 10.7\% in forecasting accuracy.}, - number = {2}, - urldate = {2025-02-19}, - journal = {ACM Trans. Knowl. Discov. Data}, - author = {Hu, Hanwen and Han, Zhangchi and Qian, Shiyou and Yang, Dingyu and Cao, Jian and Xue, Guangtao}, - month = feb, - year = {2025}, + abstract = {Multivariate time series forecasting is applied in many domains, such as finance, transportation, and industry. The main challenge of precise forecasting lies in accurately capturing latent dependencies. Recent studies develop various frameworks to reduce computational complexity or to enhance the learning of intricate relationships, while lacking interpretability and generality. In this article, we aim to elucidate the capture of dependencies as the recognition of patterns. We believe that patterns can be formally described from two aspects: the shapes of segments that frequently repeat and the corresponding forms of repetitions. Drawing upon this idea, we design a multivariate time series forecasting model named {PRformer},1 which incorporates a pattern-oriented attention mechanism and a pattern-based projector. The attention mechanism can perceive different forms of repetitions by embedded with various similarity evaluation metrics between segments, and filter out noise from segments to extract potential patterns with a statistical-driven weighting scheme. The pattern-based projector is employed to form the forecasting results by deriving the representative patterns from the set of potential ones. By incorporating explicit definitions of patterns, {PRformer} is interpretable and general to various time series scenarios. Experimental results on seven datasets demonstrate that {PRformer} outperforms six state-of-the-art models by about 10.7\% in forecasting accuracy.}, pages = {38:1--38:26}, -} - -@misc{helbling_conceptattention_2025, - title = {{ConceptAttention}: {Diffusion} {Transformers} {Learn} {Highly} {Interpretable} {Features}}, - shorttitle = {{ConceptAttention}}, - url = {http://arxiv.org/abs/2502.04320}, - doi = {10.48550/arXiv.2502.04320}, - abstract = {Do the rich representations of multi-modal diffusion transformers (DiTs) exhibit unique properties that enhance their interpretability? We introduce ConceptAttention, a novel method that leverages the expressive power of DiT attention layers to generate high-quality saliency maps that precisely locate textual concepts within images. Without requiring additional training, ConceptAttention repurposes the parameters of DiT attention layers to produce highly contextualized concept embeddings, contributing the major discovery that performing linear projections in the output space of DiT attention layers yields significantly sharper saliency maps compared to commonly used cross-attention mechanisms. Remarkably, ConceptAttention even achieves state-of-the-art performance on zero-shot image segmentation benchmarks, outperforming 11 other zero-shot interpretability methods on the ImageNet-Segmentation dataset and on a single-class subset of PascalVOC. Our work contributes the first evidence that the representations of multi-modal DiT models like Flux are highly transferable to vision tasks like segmentation, even outperforming multi-modal foundation models like CLIP.}, - urldate = {2025-02-25}, - publisher = {arXiv}, - author = {Helbling, Alec and Meral, Tuna Han Salih and Hoover, Ben and Yanardag, Pinar and Chau, Duen Horng}, - month = feb, - year = {2025}, - note = {arXiv:2502.04320 [cs]}, - keywords = {Computer Science - Machine Learning, Computer Science - Computer Vision and Pattern Recognition}, - file = {Preprint PDF:/home/alex/Zotero/storage/AEEDM4ZW/Helbling et al. - 2025 - ConceptAttention Diffusion Transformers Learn Highly Interpretable Features.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/SAHGAINH/2502.html:text/html}, -} - -@misc{chefer_transformer_2021, - title = {Transformer {Interpretability} {Beyond} {Attention} {Visualization}}, - url = {http://arxiv.org/abs/2012.09838}, - doi = {10.48550/arXiv.2012.09838}, - abstract = {Self-attention techniques, and specifically Transformers, are dominating the field of text processing and are becoming increasingly popular in computer vision classification tasks. In order to visualize the parts of the image that led to a certain classification, existing methods either rely on the obtained attention maps or employ heuristic propagation along the attention graph. In this work, we propose a novel way to compute relevancy for Transformer networks. The method assigns local relevance based on the Deep Taylor Decomposition principle and then propagates these relevancy scores through the layers. This propagation involves attention layers and skip connections, which challenge existing methods. Our solution is based on a specific formulation that is shown to maintain the total relevancy across layers. We benchmark our method on very recent visual Transformer networks, as well as on a text classification problem, and demonstrate a clear advantage over the existing explainability methods.}, - urldate = {2025-02-25}, - publisher = {arXiv}, - author = {Chefer, Hila and Gur, Shir and Wolf, Lior}, - month = apr, - year = {2021}, - note = {arXiv:2012.09838 [cs]}, - keywords = {Computer Science - Computer Vision and Pattern Recognition}, - file = {Preprint PDF:/home/alex/Zotero/storage/3FRISAP7/Chefer et al. - 2021 - Transformer Interpretability Beyond Attention Visualization.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/QUPBJPC9/2012.html:text/html}, -} - -@misc{sprang_enforcing_2024, - title = {Enforcing {Interpretability} in {Time} {Series} {Transformers}: {A} {Concept} {Bottleneck} {Framework}}, - shorttitle = {Enforcing {Interpretability} in {Time} {Series} {Transformers}}, - url = {http://arxiv.org/abs/2410.06070}, - doi = {10.48550/arXiv.2410.06070}, - abstract = {There has been a recent push of research on Transformer-based models for long-term time series forecasting, even though they are inherently difficult to interpret and explain. While there is a large body of work on interpretability methods for various domains and architectures, the interpretability of Transformer-based forecasting models remains largely unexplored. To address this gap, we develop a framework based on Concept Bottleneck Models to enforce interpretability of time series Transformers. We modify the training objective to encourage a model to develop representations similar to predefined interpretable concepts. In our experiments, we enforce similarity using Centered Kernel Alignment, and the predefined concepts include time features and an interpretable, autoregressive surrogate model (AR). We apply the framework to the Autoformer model, and present an in-depth analysis for a variety of benchmark tasks. We find that the model performance remains mostly unaffected, while the model shows much improved interpretability. Additionally, interpretable concepts become local, which makes the trained model easily intervenable. As a proof of concept, we demonstrate a successful intervention in the scenario of a time shift in the data, which eliminates the need to retrain.}, - urldate = {2025-02-25}, - publisher = {arXiv}, - author = {Sprang, Angela van and Acar, Erman and Zuidema, Willem}, - month = oct, - year = {2024}, - note = {arXiv:2410.06070 [cs]}, - keywords = {Computer Science - Machine Learning}, - file = {Preprint PDF:/home/alex/Zotero/storage/HVUXXRXJ/Sprang et al. - 2024 - Enforcing Interpretability in Time Series Transformers A Concept Bottleneck Framework.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/6QS78DZP/2410.html:text/html}, -} - -@article{yuan_dcfa-itimenet_2024, - title = {{DCFA}-{iTimeNet}: {Dynamic} cross-fusion attention network for interpretable time series prediction}, - volume = {55}, - issn = {1573-7497}, - shorttitle = {{DCFA}-{iTimeNet}}, - url = {https://doi.org/10.1007/s10489-024-05973-2}, - doi = {10.1007/s10489-024-05973-2}, - abstract = {Although time series prediction research among engineering and technology has made breakthrough progress in performance, challenges remain in modeling complex dynamic interactions between variables and interpretability. To address these two problems, a novel two-stage strategy framework called DCFA-iTimeNet is introduced. In the first stage, this paper innovatively proposes a dynamic cross-fusion attention mechanism (DCFA) . This module facilitates the model to exchange information between different patches of the time series, thereby capturing the complex interactions between variables across time. In the second stage, we exploit a decomposition-based linear explainable Bidirectional Gated Recurrent Unit (DeLEBiGRU), which consists mainly of standard BiGRU and tensorized BiGRU. It is proposed to analyze each variable’s historical long-term, instantaneous, and future impacts. Such design is crucial for understanding how each variable impacts the overall prediction over time. Extensive experimental results demonstrate that the proposed model can effectively model and interpret complex dynamic relationships of multivariate time series and understand the model’s decision-making process. Moreover, the performance outperforms the state-of-the-art methods.}, - language = {en}, number = {2}, - urldate = {2025-02-25}, - journal = {Applied Intelligence}, - author = {Yuan, Jianjun and Wu, Fujun and Zhao, Luoming and Pan, Dongbo and Yu, Xinyue}, - month = dec, - year = {2024}, - keywords = {Artificial Intelligence, Dynamic cross-fusion attention, Dynamic interaction, Interpretability, Time series prediction}, - pages = {86}, - file = {Full Text PDF:/home/alex/Zotero/storage/JEXYN7BN/Yuan et al. - 2024 - DCFA-iTimeNet Dynamic cross-fusion attention network for interpretable time series prediction.pdf:application/pdf}, + journaltitle = {{ACM} Trans. Knowl. Discov. Data}, + author = {Hu, Hanwen and Han, Zhangchi and Qian, Shiyou and Yang, Dingyu and Cao, Jian and Xue, Guangtao}, + urldate = {2025-02-19}, + date = {2025-02-06}, } -@inproceedings{guo_exploring_2019, - title = {Exploring interpretable {LSTM} neural networks over multi-variable data}, - url = {https://proceedings.mlr.press/v97/guo19b.html}, - abstract = {For recurrent neural networks trained on time series with target and exogenous variables, in addition to accurate prediction, it is also desired to provide interpretable insights into the data. In this paper, we explore the structure of LSTM recurrent neural networks to learn variable-wise hidden states, with the aim to capture different dynamics in multi-variable time series and distinguish the contribution of variables to the prediction. With these variable-wise hidden states, a mixture attention mechanism is proposed to model the generative process of the target. Then we develop associated training methods to jointly learn network parameters, variable and temporal importance w.r.t the prediction of the target variable. Extensive experiments on real datasets demonstrate enhanced prediction performance by capturing the dynamics of different variables. Meanwhile, we evaluate the interpretation results both qualitatively and quantitatively. It exhibits the prospect as an end-to-end framework for both forecasting and knowledge extraction over multi-variable data.}, - language = {en}, - urldate = {2025-02-25}, - booktitle = {Proceedings of the 36th {International} {Conference} on {Machine} {Learning}}, - publisher = {PMLR}, - author = {Guo, Tian and Lin, Tao and Antulov-Fantulin, Nino}, - month = may, - year = {2019}, - note = {ISSN: 2640-3498}, - pages = {2494--2504}, - file = {Full Text PDF:/home/alex/Zotero/storage/VV3I2T4E/Guo et al. - 2019 - Exploring interpretable LSTM neural networks over multi-variable data.pdf:application/pdf;Supplementary PDF:/home/alex/Zotero/storage/57IK29PA/Guo et al. - 2019 - Exploring interpretable LSTM neural networks over multi-variable data.pdf:application/pdf}, -} - -@incollection{iliadis_temporal_2023, - address = {Cham}, - title = {Temporal {Attention} {Signatures} for {Interpretable} {Time}-{Series} {Prediction}}, - volume = {14259}, - isbn = {978-3-031-44222-3 978-3-031-44223-0}, - url = {https://link.springer.com/10.1007/978-3-031-44223-0_22}, - abstract = {Deep neural networks have become a staple in time-series prediction due to their remarkable accuracy. However, their internal workings often remain elusive. Significant advancements have been made in the interpretability of these networks, with attention mechanisms and feature maps being notably effective for image classification by highlighting the crucial data points. While human observers can readily confirm the significance of features in image classification, the interpretability of time-series data and its modeling remains challenging. To address this, we put forth an innovative approach that unifies temporal attention and visualization as a blend of recurrent neural networks, self-attention, and general attention. This synergy results in the generation of temporal attention signatures, akin to image attention heat maps. Temporal attention not only enhances prediction accuracy beyond that of recurrent networks alone but also demonstrates that varying label classes yield distinct attention signatures. This observation indicates that neural networks focus on different sections of time-series sequences contingent on the prediction target. We conclude with a discussion on the practical implications of this novel approach, including its applicability to model interpretation, sequence length selection, and model validation. This leads to more accurate, robust, and interpretable models, instilling greater confidence in their results.}, - language = {en}, - urldate = {2025-02-25}, - booktitle = {Artificial {Neural} {Networks} and {Machine} {Learning} – {ICANN} 2023}, - publisher = {Springer Nature Switzerland}, - author = {Katrompas, Alexander and Metsis, Vangelis}, - editor = {Iliadis, Lazaros and Papaleonidas, Antonios and Angelov, Plamen and Jayne, Chrisina}, - year = {2023}, - doi = {10.1007/978-3-031-44223-0_22}, - note = {Series Title: Lecture Notes in Computer Science}, - pages = {268--280}, - file = {PDF:/home/alex/Zotero/storage/LV7IVKZK/Katrompas and Metsis - 2023 - Temporal Attention Signatures for Interpretable Time-Series Prediction.pdf:application/pdf}, -} - -@inproceedings{schwenke_constructing_2021, - title = {Constructing {Global} {Coherence} {Representations}: {Identifying} {Interpretability} and {Coherences} of {Transformer} {Attention} in {Time} {Series} {Data}}, - shorttitle = {Constructing {Global} {Coherence} {Representations}}, - url = {https://ieeexplore.ieee.org/document/9564126/?arnumber=9564126}, - doi = {10.1109/DSAA53316.2021.9564126}, - abstract = {Transformer models have shown significant advances recently based on the general concept of Attention — to focus on specifically important and relevant parts of the input data. However, methods for enhancing their interpretability and explainability are still lacking. This is the problem which we tackle in this paper, to make Multi-Headed Attention more interpretable and explainable for time series classification. We present a method for constructing global coherence representations from Multi-Headed Attention of Transformer architectures. Accordingly, we present abstraction and interpretation methods, leading to intuitive visualizations of the respective attention patterns. We evaluate our proposed approach and the presented methods on several datasets demonstrating their efficacy.}, - urldate = {2025-02-25}, - booktitle = {2021 {IEEE} 8th {International} {Conference} on {Data} {Science} and {Advanced} {Analytics} ({DSAA})}, - author = {Schwenke, Leonid and Atzmueller, Martin}, - month = oct, - year = {2021}, - keywords = {Interpretability, Attention, Coherence, Comprehensibility, Conferences, Data science, Data visualization, Deep Learning, Explainability, Global Class Representation, Scalability, Time series analysis, Time Series Classification, Transformer, Transformers, Visualization}, - pages = {1--12}, - file = {Full Text PDF:/home/alex/Zotero/storage/VRGIDFY8/Schwenke and Atzmueller - 2021 - Constructing Global Coherence Representations Identifying Interpretability and Coherences of Transf.pdf:application/pdf;IEEE Xplore Abstract Record:/home/alex/Zotero/storage/Z45R3XWF/9564126.html:text/html}, -} - -@article{schwenke_show_nodate, - title = {Show {Me} {What} {You}’re {Looking} {For}: {Visualizing} {Abstracted} {Transformer} {Attention} for {Enhancing} {Their} {Local} {Interpretability} on {Time} {Series} {Data}}, - abstract = {While Transformers have shown their advantages considering their learning performance, their lack of explainability and interpretability is still a major problem. This specifically relates to the processing of time series, as a specific form of complex data. In this paper, we propose an approach for visualizing abstracted information in order to enable computational sensemaking and local interpretability on the respective Transformer model. Our results demonstrate the efficacy of the proposed abstraction method and visualization, utilizing both synthetic and real world data for evaluation.}, - language = {en}, - author = {Schwenke, Leonid and Atzmueller, Martin}, - file = {PDF:/home/alex/Zotero/storage/SLVVAAXA/Schwenke and Atzmueller - Show Me What You’re Looking For Visualizing Abstracted Transformer Attention for Enhancing Their Lo.pdf:application/pdf}, -} - -@article{wu_interpretable_2022, - title = {Interpretable wind speed prediction with multivariate time series and temporal fusion transformers}, - volume = {252}, - issn = {03605442}, - url = {https://linkinghub.elsevier.com/retrieve/pii/S0360544222008933}, - doi = {10.1016/j.energy.2022.123990}, - abstract = {Wind power has been utilized well in power systems, so steady and successful wind speed forecasting is crucial to security management power grid market economy. To date, most researchers have often discounted the interpretability of prediction models, leading to obscure forecasts. This study puts forward a unique forecasting methodology that incorporates notable decomposition techniques, multifactor interpretable forecasting models, and optimization algorithms. In the proposed model, variational mode decomposition is employed to break down the raw wind speed sequence into a set of intrinsic mode functions. Adaptive differential evolution is then used for optimizing several parameters of temporal fusion transformers (TFT) to achieve satisfactory forecasting performance. TFT is a new attention-based deep learning model that puts together high-performance multi-horizon prediction and interpretable insights into temporal dynamics. Empirical studies using eight real-world 1-h wind speed data sets in Albert, Canada, and Five Points, USA demonstrate that the system using the proposed model outperforms those employing other comparable models in nearly all performance metrics. Examples of TFT's interpretable outputs are the importance ranking of the decomposed wind speed sub-sequences and meteorological data and attention analysis of different step lengths. The findings signify substantial progress for wind speed prediction and aid policymakers.}, - language = {en}, - urldate = {2025-02-25}, - journal = {Energy}, - author = {Wu, Binrong and Wang, Lin and Zeng, Yu-Rong}, - month = aug, - year = {2022}, - pages = {123990}, - file = {PDF:/home/alex/Zotero/storage/MHAFY926/Wu et al. - 2022 - Interpretable wind speed prediction with multivariate time series and temporal fusion transformers.pdf:application/pdf}, -} - -@inproceedings{habib_n-beats_2025, - address = {Cham}, - title = {N-{BEATS} \& {Temporal} {Fusion} {Transformer} {Based} {Surface} {Temperature} {Prediction} and {Forecasting} for {Realizing} {Global} {Warming} {Trends}}, - isbn = {978-3-031-75167-7}, - doi = {10.1007/978-3-031-75167-7_3}, - abstract = {At the pinnacle of civilization, where the impacts of climate change have been increasingly felt, weather prediction plays a critical role in mitigating the potential disasters that may arise. Moreover, with the gradual change on climate, surface temperature of the earth is increasing. This increasing rate of the surface temperature causing global warming which is a matter of intimidation. To leave off this global warming, weather forecasting can be used as an arsenal. Selecting the appropriate tools and models for weather prediction is a crucial step in ensuring accurate forecasts. In this research paper, the focus was on studying the versatility of three specific architectures for weather prediction: LSTM, Temporal Fusion Transformer, and N-BEATS. To assess these architectures’ performance, we conducted a number of experiments. With the lowest Mean Absolute Error (MAE) and Root Mean Square Error (RMSE) of the three, NBEATS stood out. This shows that when compared to the other models, the N-BEATS architecture had greater prediction accuracy. It's vital to remember, too, that the trials also showed that the Temporal Fusion Transformer and LSTM performed well. The only distinction was that these models required larger sizes in terms of parameters and computational complexity to achieve their performance levels. Consequently, considering both performance and model size, the researchers determined that N-BEATS was the most optimal and versatile architecture for weather prediction. Its ability to achieve excellent results with a smaller model size makes it a favorable choice for practical applications.}, - language = {en}, - booktitle = {Artificial {Intelligence} and {Speech} {Technology}}, - publisher = {Springer Nature Switzerland}, - author = {Habib, Adria Binte and Ashraf, Faisal Bin and Hossain, Muhammad Iqbal and Alam, Golam Rabiul}, - editor = {Dev, Amita and Sharma, Arun and Agrawal, S. S. and Rani, Ritu}, - year = {2025}, - keywords = {LSTM, N-BEATS, Temporal Fusion Transformer, Time Series Analysis, Weather Prediction}, - pages = {30--41}, - annote = { - -shows that n beats has better performance on smaller datasets - - -we have a relatively large dataset, thus tft will be focused on - - -}, -} - -@article{papacharalampous_predictability_2018, - title = {Predictability of monthly temperature and precipitation using automatic time series forecasting methods}, - volume = {66}, - issn = {1895-7455}, - url = {https://doi.org/10.1007/s11600-018-0120-7}, - doi = {10.1007/s11600-018-0120-7}, - abstract = {We investigate the predictability of monthly temperature and precipitation by applying automatic univariate time series forecasting methods to a sample of 985 40-year-long monthly temperature and 1552 40-year-long monthly precipitation time series. The methods include a naïve one based on the monthly values of the last year, as well as the random walk (with drift), AutoRegressive Fractionally Integrated Moving Average (ARFIMA), exponential smoothing state-space model with Box–Cox transformation, ARMA errors, Trend and Seasonal components (BATS), simple exponential smoothing, Theta and Prophet methods. Prophet is a recently introduced model inspired by the nature of time series forecasted at Facebook and has not been applied to hydrometeorological time series before, while the use of random walk, BATS, simple exponential smoothing and Theta is rare in hydrology. The methods are tested in performing multi-step ahead forecasts for the last 48 months of the data. We further investigate how different choices of handling the seasonality and non-normality affect the performance of the models. The results indicate that: (a) all the examined methods apart from the naïve and random walk ones are accurate enough to be used in long-term applications; (b) monthly temperature and precipitation can be forecasted to a level of accuracy which can barely be improved using other methods; (c) the externally applied classical seasonal decomposition results mostly in better forecasts compared to the automatic seasonal decomposition used by the BATS and Prophet methods; and (d) Prophet is competitive, especially when it is combined with externally applied classical seasonal decomposition.}, - language = {en}, - number = {4}, - urldate = {2025-03-04}, - journal = {Acta Geophysica}, - author = {Papacharalampous, Georgia and Tyralis, Hristos and Koutsoyiannis, Demetris}, - month = aug, - year = {2018}, - keywords = {ARFIMA, Multi-step ahead forecasting, Precipitation forecasting, Prophet, Temperature forecasting, Time series forecasting}, - pages = {807--831}, -} - -@article{dunson_day-specific_1999, - title = {Day-specific probabilities of clinical pregnancy based on two studies with imperfect measures of ovulation}, +@article{braude_machine_2024, + title = {Machine learning for predicting elective fertility preservation outcomes}, volume = {14}, - issn = {1460-2350, 0268-1161}, - url = {https://academic.oup.com/humrep/article-lookup/doi/10.1093/humrep/14.7.1835}, - doi = {10.1093/humrep/14.7.1835}, - language = {en}, - number = {7}, - urldate = {2025-03-05}, - journal = {Human Reproduction}, - author = {Dunson, D.B. and Baird, D.D. and Wilcox, A.J. and Weinberg, C.R.}, - month = jul, - year = {1999}, - pages = {1835--1839}, - file = {PDF:/home/alex/Zotero/storage/8DE3ZLPJ/Dunson et al. - 1999 - Day-specific probabilities of clinical pregnancy based on two studies with imperfect measures of ovu.pdf:application/pdf}, + rights = {2024 The Author(s)}, + issn = {2045-2322}, + url = {https://www.nature.com/articles/s41598-024-60671-w}, + doi = {10.1038/s41598-024-60671-w}, + abstract = {This retrospective study applied machine-learning models to predict treatment outcomes of women undergoing elective fertility preservation. Two-hundred-fifty women who underwent elective fertility preservation at a tertiary center, 2019–2022 were included. Primary outcome was the number of metaphase {II} oocytes retrieved. Outcome class was based on oocyte count ({OC}): Low (≤ 8), Medium (9–15) or High (≥ 16). Machine-learning models and statistical regression were used to predict outcome class, first based on pre-treatment parameters, and then using post-treatment data from ovulation-triggering day. {OC} was 136 Low, 80 Medium, and 34 High. Random Forest Classifier ({RFC}) was the most accurate model (pre-treatment receiver operating characteristic ({ROC}) area under the curve ({AUC}) was 77\%, and post-treatment {ROC} {AUC} was 87\%), followed by {XGBoost} Classifier (pre-treatment {ROC} {AUC} 74\%, post-treatment {ROC} {AUC} 86\%). The most important pre-treatment parameters for {RFC} were basal {FSH} (22.6\%), basal {LH} (19.1\%), {AFC} (18.2\%), and basal estradiol (15.6\%). Post-treatment parameters were estradiol levels on trigger-day (17.7\%), basal {FSH} (11\%), basal {LH} (9\%), and {AFC} (8\%). Machine-learning models trained with clinical data appear to predict fertility preservation treatment outcomes with relatively high accuracy.}, + pages = {10158}, + number = {1}, + journaltitle = {Sci Rep}, + author = {Braude, Itai and Haikin Herzberger, Einat and Semo, Mor and Soifer, Kim and Goren Gepstein, Nitzan and Wiser, Amir and Miller, Netanella}, + urldate = {2025-02-11}, + date = {2024-05-02}, + langid = {english}, + note = {Publisher: Nature Publishing Group}, + keywords = {Computational models, Outcomes research}, + file = {Full Text PDF:/home/alex/Zotero/storage/URDGBHLV/Braude et al. - 2024 - Machine learning for predicting elective fertility preservation outcomes.pdf:application/pdf}, } -@misc{wikimedia_commons_basic_2019, - title = {Basic {Female} {Reproductive} {System}}, - url = {https://en.wikipedia.org/wiki/File:Basic_Female_Reproductive_System_(English).svg}, - author = {Wikimedia Commons}, - year = {2019}, - file = {background_female_reproductive_organs:/home/alex/Zotero/storage/5MAJ4CG5/background_female_reproductive_organs.png:image/png}, +@article{fanton_interpretable_2022, + title = {An interpretable machine learning model for predicting the optimal day of trigger during ovarian stimulation}, + volume = {118}, + issn = {0015-0282, 1556-5653}, + url = {https://www.fertstert.org/article/S0015-0282%2822%2900244-8/fulltext}, + doi = {10.1016/j.fertnstert.2022.04.003}, + pages = {101--108}, + number = {1}, + journaltitle = {Fertility and Sterility}, + author = {Fanton, Michael and Nutting, Veronica and Solano, Funmi and Maeder-York, Paxton and Hariton, Eduardo and Barash, Oleksii and Weckstein, Louis and Sakkas, Denny and Copperman, Alan B. and Loewke, Kevin}, + urldate = {2025-02-11}, + date = {2022-07-01}, + note = {Publisher: Elsevier}, + keywords = {Artificial intelligence, in vitro fertilization, machine learning, ovarian stimulation, trigger}, + file = {Full Text PDF:/home/alex/Zotero/storage/UNX7ASLP/Fanton et al. - 2022 - An interpretable machine learning model for predicting the optimal day of trigger during ovarian sti.pdf:application/pdf}, } -@article{silberstein_physiology_2000, - title = {Physiology of the {Menstrual} {Cycle}}, +@inproceedings{azaria_semi-supervised_2019, + title = {Semi-Supervised Ovulation Detection Based on Multiple Properties}, + url = {https://ieeexplore.ieee.org/document/8995235}, + doi = {10.1109/ICTAI.2019.00039}, + abstract = {Despite being a well-researched problem, ovulation detection in human female remains a difficult task. Most current methods for ovulation detection rely on measurements of a single property (e.g. morning body temperature) or at most on two properties (e.g. both salivary and vaginal electrical resistance). In this paper we present a machine learning based method for detecting the day in which ovulation occurs. Our method considered measurements of five different properties. We crawled a data-set from the web and showed that our method outperforms current state-of-the-art methods for ovulation detection. Our method performs well also when considering measurements of fewer properties. We show that our method's performance can be further improved by using unlabeled data, that is, mensuration cycles without a know ovulation date. Our resulted machine learning model can be very useful for women trying to conceive that have trouble in recognizing their ovulation period, especially when some measurements are missing.}, + eventtitle = {2019 {IEEE} 31st International Conference on Tools with Artificial Intelligence ({ICTAI})}, + pages = {222--228}, + booktitle = {2019 {IEEE} 31st International Conference on Tools with Artificial Intelligence ({ICTAI})}, + author = {Azaria, Amos and Azaria, Seagal}, + urldate = {2025-02-11}, + date = {2019-11}, + note = {{ISSN}: 2375-0197}, + keywords = {ovulation detection, semi supervised learning}, + file = {IEEE Xplore Abstract Record:/home/alex/Zotero/storage/JSAEWRGB/8995235.html:text/html;PDF:/home/alex/Zotero/storage/P74SEG9S/Azaria and Azaria - 2019 - Semi-Supervised Ovulation Detection Based on Multiple Properties.pdf:application/pdf}, +} + +@article{luz_p-656_2023, + title = {P-656 Machine learning algorithm automatically manages and accurately predicts ovulation in natural frozen-thawed embryo transfer cycles.}, + volume = {38}, + issn = {0268-1161}, + url = {https://doi.org/10.1093/humrep/dead093.983}, + doi = {10.1093/humrep/dead093.983}, + abstract = {Can an Artificial Intelligence ({AI}) algorithm automatically manage frozen-thawed embryo transfer ({NC}-{FET}) treatment cycles and give an accurate prediction of ovulation day.An {AI} algorithm automatically managed and predicted the ovulation of {NC}-{FET} treatment cycles with 94.8\% accuracy using an average of 3.01 test days.Today the preferred method for frozen embryo transfer is natural cycle based on ovulation detection. Currently, there is no software capable of managing the treatment cycle automatically and identifying the time of ovulation to support doctor decisions. The aim of this study is to develop a physician support {AI} software for determining ovulation time reliably with high accuracy.2083 {NC}-{FET} cycles from September 2018 to June 2021 were used to develop the ovulation detection and treatment management algorithms.Each cycle had data from at least 2 visits including: hormonal levels (Estrogen/Progesterone/{LH}) and follicle sizes.The dataset was divided into a train set and two test sets. In the 1st test set ovulation was determined by experts’ opinions and the 2nd test set included cycles in which follicle rupture was documented in consecutive ultrasounds.Two algorithms were developed, an ovulation prediction algorithm based on an {NGBoost} model and a treatment management algorithm that used the model to determine if and when to call for a new test or declare the ovulation day.Both algorithms were jointly tuned to reach the highest success rate, defined as providing the correct day of ovulation using the available cycle data, with as few test days as possible.On the first test set, which consisted of 176 cycles in which ovulation was determined through the majority decision of 2 independent experts and the attending physician, the treatment management algorithm required on average 3.01 tests to reach a prediction and successfully predicted the ovulation day in 94.8\% of cycles.In the second test set, which consisted of 29 cycles in which ovulation was determined through the follicular rupture in two consecutive ultrasounds, only the ovulation prediction model was tested. To ensure that the model provides a reliable answer and does not rely solely on the follicular disappearances, examined cycles were tested twice: Once using the ovulation day without the day prior to it, and again using only the day prior to ovulation without the ovulation day itself. The algorithm accurately predicted ovulation in 28 out of 29 instances (96.6\%) using the day of ovulation and in 28 out of 29 instances (96.6\%) using the day before ovulation.The main drawback is this being a retrospective study: while the algorithm was trained to maximize accuracy when it selects the test days, the dataset test days were selected by the attending physicians. Statistical methods were used to overcome this, however further prospective trials are needed to validate the results.This is the first {AI} algorithm designed to automatically manage {NC}-{FET} {IVF} treatment cycles and predict ovulation. The high accuracy and low average tests count might improve treatment outcomes, reduce the patients’ life disruption, and allow physicians to spend less time monitoring their patients’ treatments.not applicable}, + pages = {dead093.983}, + issue = {Supplement\_1}, + journaltitle = {Human Reproduction}, + author = {Luz, A and Hourvitz, R and Reuvenny, S and Youngster, M and Baum, M and Hourvitz, A and Maman, E}, + urldate = {2025-02-11}, + date = {2023-06-01}, + file = {Full Text PDF:/home/alex/Zotero/storage/C8YM765Y/Luz et al. - 2023 - P-656 Machine learning algorithm automatically manages and accurately predicts ovulation in natural.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/WANIZX3H/7202977.html:text/html}, +} + +@article{lin_transformer_2023, + title = {Transformer neural network to predict and interpret pregnancy loss from activity data in Holstein dairy cows}, + volume = {205}, + issn = {0168-1699}, + url = {https://www.sciencedirect.com/science/article/pii/S0168169923000261}, + doi = {10.1016/j.compag.2023.107638}, + abstract = {Predicting/detecting pregnancy loss of dairy cows offers the opportunity to shorten the time interval between artificial inseminations. Although several methods of pregnancy detection are being practiced, models with accurate, timely and interpretable detection of pregnancy are still lacking. This study proposed a transformer neural network to predict the probability of pregnancy loss based on continuous activity data, which were collected from activity-monitoring tags attached to 185 Holstein cows from a commercial dairy farm in Cayuga County, {NY}, {USA}. Our best model achieved an average accuracy of 0.87, F1 score of 0.87, recall of 0.87 and specificity of 0.90 using 14-day time-series activity windows (90\% overlap) using 5-fold cross-validation, outperforming commonly used classic statistical learning and deep learning models for time-series data. The results indicated that our predictive model gave high probabilities of correctly detecting pregnancy loss prior to the increased activities and veterinary confirmation by transrectal ultrasound. In addition, our model interpretation aligned with the changes in the temporal activity levels, revealing that drastic fluctuations in time-series activity data contributed heavily to the final prediction. To the best of our knowledge, this is the first work on developing transformer models for the prediction of pregnancy loss in dairy cows. In addition to facilitating the development of future precision management on modern farms, our work potentiates an increase in the reproductive efficiency and profitability of dairy farms.}, + pages = {107638}, + journaltitle = {Computers and Electronics in Agriculture}, + author = {Lin, Dan and Kenéz, Ákos and {McArt}, Jessica A. A. and Li, Jun}, + urldate = {2025-02-11}, + date = {2023-02-01}, + keywords = {Dairy cow, Precision livestock farming, Pregnancy loss prediction, Time-series activity}, +} + +@article{shkodzik_innovative_2024, + title = {Innovative Approaches to Digital Health in Ovulation Detection: A Review of Current Methods and Emerging Technologies}, + volume = {42}, + issn = {1526-4564}, + doi = {10.1055/s-0044-1793829}, + shorttitle = {Innovative Approaches to Digital Health in Ovulation Detection}, + abstract = {Ovulation is a vital sign, as significant as body temperature, heart rate, respiratory rate, and blood pressure, in assessing overall health and identifying potential health issues. Ovulation is a key event of the menstrual cycle that provides insights into the hormonal and reproductive health aspects. Affected by the orchestra of hormones, namely thyroid, prolactin, and androgens, disruptions in ovulation can indicate endocrinological conditions and lead to gynecological problems, such as heavy menstrual bleeding, irregular periods, amenorrhea, dysmenorrhea, and difficulties in getting pregnant. Monitoring ovulation and detecting disruptions can aid in the early detection of health issues, extending beyond reproductive health concerns. It can help identify underlying causes of symptoms like excessive fatigue and abnormal hair growth. The integration of digital health technologies, such as mobile apps using machine learning algorithms, wearables tracking temperature, heart rate, breath rate, and sleep patterns, and devices measuring reproductive hormones in urine or saliva samples, offers a wealth of opportunities in family planning, early health issue diagnosis, treatment adjustment, and tracking menstrual cycles during assisted reproductive techniques. These advancements provide a comprehensive approach to health monitoring, addressing both reproductive and overall health concerns.}, + pages = {81--89}, + number = {2}, + journaltitle = {Semin Reprod Med}, + author = {Shkodzik, Katerina}, + date = {2024-06}, + pmid = {39572028}, + keywords = {Digital Health, Female, Humans, Mobile Applications, Ovulation, Ovulation Detection, Telemedicine, Wearable Electronic Devices}, +} + +@article{pratikno_pdf_2024, + title = {({PDF}) A novel women's ovulation prediction through salivary ferning using the box counting and deep learning}, + url = {https://www.researchgate.net/publication/379467622_A_novel_women's_ovulation_prediction_through_salivary_ferning_using_the_box_counting_and_deep_learning}, + doi = {10.11591/eei.v13i2.5847}, + abstract = {{PDF} {\textbar} There are several methods to predict a woman's ovulation time, including using a calendar system, basal body temperature, ovulation prediction... {\textbar} Find, read and cite all the research you need on {ResearchGate}}, + journaltitle = {{ResearchGate}}, + author = {Pratikno and Ibrahim and Jusak}, + urldate = {2025-02-11}, + date = {2024-12-09}, + langid = {english}, + file = {Full Text:/home/alex/Zotero/storage/MVSBPZQG/2024 - (PDF) A novel women's ovulation prediction through salivary ferning using the box counting and deep.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/8FYLC8YZ/379467622_A_novel_women's_ovulation_prediction_through_salivary_ferning_using_the_box_counting_.html:text/html}, +} + +@article{luz_improved_2024, + title = {Improved clinical pregnancy rates in natural frozen-thawed embryo transfer cycles with machine learning ovulation prediction: insights from a retrospective cohort study}, + volume = {14}, + rights = {2024 The Author(s)}, + issn = {2045-2322}, + url = {https://www.nature.com/articles/s41598-024-80356-8}, + doi = {10.1038/s41598-024-80356-8}, + shorttitle = {Improved clinical pregnancy rates in natural frozen-thawed embryo transfer cycles with machine learning ovulation prediction}, + abstract = {This study aims to develop physician support software for determining ovulation time and assess its impact on pregnancy outcomes in natural cycle frozen embryo transfers ({NC}-{FET}). To develop, assess, and validate an ovulation prediction model, three datasets were used: {REI} Ovulation Determination dataset (500 cycles) split into training (309), validation (90), and test (101) sets; the Documented Ovulation dataset (101 cycles) with confirmed ovulation (documented follicular rupture and {LH} surge); and the Clinical Pregnancy Rates dataset (515 {NC}-{FET} cycles), categorized into “Matched” and “Mismatched” based on alignment with the model’s ovulation determination. Pregnancy outcomes were compared between the groups. The ovulation prediction model exhibited 93.85\% and 92.89\% matching rates with the {REI} Ovulation Determination and Documented Ovulation datasets, respectively. In the Clinical Pregnancy Rates dataset, the Matched group (282 cycles) showed significantly higher clinical pregnancy rates than the Mismatched group (34.6\% vs. 25.9\%, p = 0.04) and similar results for patients under 37 (41.1\% vs. 30.7\%, p = 0.04). Logistic regression indicated lower pregnancy rates in Mismatched cases (odds ratio 0.67 for the general population, 0.63 for patients under 37). In conclusion, we introduce a highly accurate {AI} ovulation prediction model. Treatment cycles aligning with the model’s recommendations had significantly increased clinical pregnancy rates.}, + pages = {29451}, + number = {1}, + journaltitle = {Sci Rep}, + author = {Luz, Almog and Hourvitz, Ariel and Moran, Eden and Itzhak, Nevo and Reuvenny, Shachar and Hourvitz, Rohi and Youngster, Michal and Baum, Micha and Maman, Ettie}, + urldate = {2025-02-11}, + date = {2024-11-27}, + langid = {english}, + note = {Publisher: Nature Publishing Group}, + keywords = {Infertility, Outcomes research}, + file = {Full Text PDF:/home/alex/Zotero/storage/BSHNTIFD/Luz et al. - 2024 - Improved clinical pregnancy rates in natural frozen-thawed embryo transfer cycles with machine learn.pdf:application/pdf}, +} + +@article{yu_tracking_2022, + title = {Tracking of menstrual cycles and prediction of the fertile window via measurements of basal body temperature and heart rate as well as machine-learning algorithms}, volume = {20}, - copyright = {https://journals.sagepub.com/page/policies/text-and-data-mining-license}, - issn = {0333-1024, 1468-2982}, - url = {https://journals.sagepub.com/doi/10.1046/j.1468-2982.2000.00034.x}, - doi = {10.1046/j.1468-2982.2000.00034.x}, - abstract = {The normal female life cycle is associated with a number of hormonal milestones: menarche, pregnancy, contraceptive use, menopause, and the use of replacement sex hormones. All these events and interventions alter the levels and cycling of sex hormones and may cause a change in the prevalence or intensity of headache. The menstrual cycle is the result of a carefully orchestrated sequence of interactions among the hypothalamus, pituitary, ovary, and endometrium, with the sex hormones acting as modulators and effectors at each level. Oestrogen and progestins have potent effects on central serotonergic and opioid neurons, modulating both neuronal activity and receptor density. The primary trigger of menstrual migraine appears to be the withdrawal of oestrogen rather than the maintenance of sustained high or low oestrogen levels. However, changes in the sustained oestrogen levels with pregnancy (increased) and menopause (decreased) appear to affect headaches. Headaches that occur with premenstrual syndrome appear to be centrally generated, involving the inherent rhythm of CNS neurons, including perhaps the serotonergic pain-modulating systems.}, - language = {en}, - number = {3}, - urldate = {2025-03-10}, - journal = {Cephalalgia}, - author = {Silberstein, S D and Merriam, G R}, - month = apr, - year = {2000}, - pages = {148--154}, - file = {Full Text:/home/alex/Zotero/storage/EL585H2P/Silberstein and Merriam - 2000 - Physiology of the Menstrual Cycle.pdf:application/pdf}, + issn = {1477-7827}, + url = {https://doi.org/10.1186/s12958-022-00993-4}, + doi = {10.1186/s12958-022-00993-4}, + abstract = {Fertility awareness and menses prediction are important for improving fecundability and health management. Previous studies have used physiological parameters, such as basal body temperature ({BBT}) and heart rate ({HR}), to predict the fertile window and menses. However, their accuracy is far from satisfactory. Additionally, few researchers have examined irregular menstruators. Thus, we aimed to develop fertile window and menstruation prediction algorithms for both regular and irregular menstruators.}, + pages = {118}, + number = {1}, + journaltitle = {Reproductive Biology and Endocrinology}, + author = {Yu, Jia-Le and Su, Yun-Fei and Zhang, Chen and Jin, Li and Lin, Xian-Hua and Chen, Lu-Ting and Huang, He-Feng and Wu, Yan-Ting}, + urldate = {2025-02-11}, + date = {2022-08-13}, + keywords = {Basal body temperature, Fertile window, Heart rate, Machine learning, Menstrual cycle, Wearable device}, + file = {Full Text PDF:/home/alex/Zotero/storage/7Z8P97UF/Yu et al. - 2022 - Tracking of menstrual cycles and prediction of the fertile window via measurements of basal body tem.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/SMRBT6EG/s12958-022-00993-4.html:text/html}, } -@misc{pedroso_menstrual_2022, - title = {The {Menstrual} {Cycle}}, - url = {https://kindbody.com/the-menstrual-cycle/}, - abstract = {Fertility, gynecology, and wellness services in modern, tech-enabled clinics. Best-in-class care, accessible pricing, and a seamless patient experience.}, - urldate = {2025-03-10}, - journal = {Kindbody}, - author = {Pedroso, Dr Jasmine}, - month = jun, - year = {2022}, - file = {Snapshot:/home/alex/Zotero/storage/738JEKW7/the-menstrual-cycle.html:text/html}, +@article{luo_prediction_2025, + title = {Prediction of the fertile window and menstrual cycles with a wearable device via machine-learning algorithms}, + issn = {1472-6483}, + url = {https://www.sciencedirect.com/science/article/pii/S1472648325000021}, + doi = {10.1016/j.rbmo.2025.104795}, + abstract = {Research question +We aimed to develop fertile window and menstruation prediction algorithms through machine learning based on women's physiological parameters data collected by Huawei Band 6 pro from both regular and irregular menstruators. +Design +This was a prospective observational cohort study conducted at Obstetrics and Gynecology Hospital of Fudan University. Participants were recruited from November 2021 to September 2022. Each participant wore Huawei Band 6 pro to record wrist skin temperature ({WST}), heart rate ({HR}), heart rate variability, and respiratory rate. Algorithms were developed to predict the fertile window and menstrual cycle based on {WST} and {HR}. +Results +We included data from 270 and 84 qualified cycles with confirmed ovulations from 136 regular and 47 irregular menstruators. For regular menstruators, the prediction algorithm based on {WST} and {HR} for the fertile window had an accuracy of 85.47\%, a sensitivity of 70.07\%, a specificity of 89.77\%, and {AUC} of 0.869. The algorithms for menstrual first day labelling and onset within 3 days gained an accuracy of 83.6\% and 75.0\%. For irregular menstruators, the accuracy, sensitivity, specificity and {AUC} were 79.85\%, 42.79\%, 87.28\%, and 0.763 respectively, for fertile window prediction. The accuracy of menses labelling and prediction were 61.2\%, and 50.8\% respectively. +Conclusions +Based on {WST} and {HR} data from the wearable device, the algorithms demonstrated reliable performance in predicting the fertile window and menstruation day among regular menstruators. These algorithms also showed potential for assisting irregular menstruators in managing their cycles and planning for conception.}, + pages = {104795}, + journaltitle = {Reproductive {BioMedicine} Online}, + author = {Luo, Chuan and Su, Yun-Fei and Ren, Yun-Yun and Zhang, Qin and Li, Ran and Zhang, Qi and Li, Cheng and Hao, Yan-Hui and Zhang, An-Qi and Zhang, Hao and Huang, He-Feng and Wu, Yan-Ting}, + urldate = {2025-02-11}, + date = {2025-01-07}, + keywords = {Fertile window, Machine learning, Menstrual cycle, Natural cycle, Non-invasive wearable device, Wrist skin temperature}, + file = {PDF:/home/alex/Zotero/storage/YIV6T8MS/Luo et al. - 2025 - Prediction of the fertile window and menstrual cycles with a wearable device via machine-learning al.pdf:application/pdf;ScienceDirect Snapshot:/home/alex/Zotero/storage/3IJMUH82/S1472648325000021.html:text/html}, +} + +@article{maman_prediction_2023, + title = {Prediction of ovulation: new insight into an old challenge}, + volume = {13}, + issn = {2045-2322}, + url = {https://www.ncbi.nlm.nih.gov/pmc/articles/PMC10651856/}, + doi = {10.1038/s41598-023-47241-2}, + shorttitle = {Prediction of ovulation}, + abstract = {Ultrasound monitoring and hormonal blood testing are considered by many as an accurate method to predict ovulation time. However, uniform and validated algorithms for predicting ovulation have yet to be defined. Daily hormonal tests and transvaginal ultrasounds were recorded to develop an algorithm for ovulation prediction. The rupture of the leading ovarian follicle was a marker for ovulation day. The model was validated retrospectively on natural cycles frozen embryo transfer cycles with documented ovulation. Circulating levels of {LH} or its relative variation failed, by themselves, to reliably predict ovulation. Any decrease in estrogen was 100\% associated with ovulation emergence the same day or the next day. Progesterone levels {\textgreater} 2 nmol/L had low specificity to predict ovulation the next day (62.7\%), yet its sensitivity was high (91.5\%). A model for ovulation prediction, combining the three hormone levels and ultrasound was created with an accuracy of 95\% to 100\% depending on the combination of the hormone levels. Model validation showed correct ovulation prediction in 97\% of these cycles. We present an accurate ovulation prediction algorithm. The algorithm is simple and user-friendly so both reproductive endocrinologists and general practitioners can use it to benefit their patients.}, + pages = {20003}, + journaltitle = {Sci Rep}, + author = {Maman, Ettie and Adashi, Eli Y. and Baum, Micha and Hourvitz, Ariel}, + urldate = {2025-02-11}, + date = {2023-11-15}, + pmid = {37968377}, + pmcid = {PMC10651856}, + file = {PubMed Central Full Text PDF:/home/alex/Zotero/storage/MTHDPZ5B/Maman et al. - 2023 - Prediction of ovulation new insight into an old challenge.pdf:application/pdf}, +} + +@article{masuda_machine_2025, + title = {Machine learning model for menstrual cycle phase classification and ovulation day detection based on sleeping heart rate under free-living conditions}, + volume = {187}, + issn = {0010-4825}, + url = {https://www.sciencedirect.com/science/article/pii/S0010482525000551}, + doi = {10.1016/j.compbiomed.2025.109705}, + abstract = {The accurate classification of menstrual cycle phases and detection of ovulation is critical for women's health management, particularly in addressing infertility, alleviating premenstrual syndrome, and preventing hormone-related disorders. However, traditional basal body temperature ({BBT}) measurement methods are susceptible to disruptions in sleep timing and environmental conditions, limiting practical application. This study is aimed to overcome these limitations by introducing a novel feature, heart rate at the circadian rhythm nadir ({minHR}), for classifying menstrual cycle phases and predicting ovulation. A machine learning model was developed using {XGBoost}, and data were collected under free-living conditions from 40 healthy women (18–34 years) over a maximum of three menstrual cycles. Three feature combinations— “day,” “day + {minHR},” and “day + {BBT}”—were evaluated, and model performance was assessed using nested leave-one-group-out cross-validation. The feature “day” represents the number of days elapsed since the onset of menstruation. Participants were stratified into groups depending on high variability and low variability in sleep timing. Results demonstrated that adding {minHR} significantly improved luteal phase classification and ovulation day detection performance compared to “day” only. Furthermore, in participants with high variability in sleep timing, the {minHR}-based model outperformed the {BBT}-based model, significantly improving luteal phase recall and reducing ovulation day detection absolute errors by 2 d (p {\textless} 0.05). These findings highlight the robustness and practicality of the {minHR}-based model for menstrual cycle tracking, particularly in individuals with high variability in sleep timing. The proposed model holds great promise for personalized health management and large-scale epidemiological research.}, + pages = {109705}, + journaltitle = {Computers in Biology and Medicine}, + author = {Masuda, Hazuki and Okada, Shima and Shiozawa, Naruhiro and Sakaue, Yusuke and Manno, Masanobu and Makikawa, Masaaki and Isaka, Tadao}, + urldate = {2025-02-11}, + date = {2025-03-01}, + keywords = {Heart rate, Machine learning, Circadian rhythm, Menstrual cycle tracking, Ovulation day detection, Wearable sensor, {XGBoost}}, + file = {ScienceDirect Snapshot:/home/alex/Zotero/storage/PC6FSQIA/S0010482525000551.html:text/html}, +} + +@online{noauthor_keras_nodate, + title = {Keras: Deep Learning for humans}, + url = {https://keras.io/}, + urldate = {2024-10-23}, + file = {Keras\: Deep Learning for humans:/home/alex/Zotero/storage/MS4QLPRC/keras.io.html:text/html}, +} + +@article{coninck_dianne_2018, + title = {{DIANNE}: a modular framework for designing, training and deploying deep neural networks on heterogeneous distributed infrastructure}, + volume = {141}, + issn = {01641212}, + url = {https://linkinghub.elsevier.com/retrieve/pii/S0164121218300487}, + doi = {10.1016/j.jss.2018.03.032}, + shorttitle = {{DIANNE}}, + pages = {52--65}, + journaltitle = {Journal of Systems and Software}, + author = {Coninck, Elias De and Bohez, Steven and Leroux, Sam and Verbelen, Tim and Vankeirsbilck, Bert and Simoens, Pieter and Dhoedt, Bart}, + urldate = {2024-10-23}, + date = {2018-07}, + langid = {english}, + file = {Full Text:/home/alex/Zotero/storage/SFH656QN/Coninck et al. - 2018 - DIANNE a modular framework for designing, training and deploying deep neural networks on heterogene.pdf:application/pdf}, +} + +@misc{shi_time-moe_2024, + title = {Time-{MoE}: Billion-Scale Time Series Foundation Models with Mixture of Experts}, + url = {http://arxiv.org/abs/2409.16040}, + shorttitle = {Time-{MoE}}, + abstract = {Deep learning for time series forecasting has seen significant advancements over the past decades. However, despite the success of large-scale pre-training in language and vision domains, pre-trained time series models remain limited in scale and operate at a high cost, hindering the development of larger capable forecasting models in real-world applications. In response, we introduce Time-{MoE}, a scalable and unified architecture designed to pre-train larger, more capable forecasting foundation models while reducing inference costs. By leveraging a sparse mixture-of-experts ({MoE}) design, Time-{MoE} enhances computational efficiency by activating only a subset of networks for each prediction, reducing computational load while maintaining high model capacity. This allows Time-{MoE} to scale effectively without a corresponding increase in inference costs. Time-{MoE} comprises a family of decoder-only transformer models that operate in an auto-regressive manner and support flexible forecasting horizons with varying input context lengths. We pre-trained these models on our newly introduced large-scale data Time-300B, which spans over 9 domains and encompassing over 300 billion time points. For the first time, we scaled a time series foundation model up to 2.4 billion parameters, achieving significantly improved forecasting precision. Our results validate the applicability of scaling laws for training tokens and model size in the context of time series forecasting. Compared to dense models with the same number of activated parameters or equivalent computation budgets, our models consistently outperform them by large margin. These advancements position Time-{MoE} as a state-of-the-art solution for tackling real-world time series forecasting challenges with superior capability, efficiency, and flexibility.}, + number = {{arXiv}:2409.16040}, + publisher = {{arXiv}}, + author = {Shi, Xiaoming and Wang, Shiyu and Nie, Yuqi and Li, Dianqi and Ye, Zhou and Wen, Qingsong and Jin, Ming}, + urldate = {2024-10-16}, + date = {2024-10-02}, + eprinttype = {arxiv}, + eprint = {2409.16040}, + keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, + file = {Preprint PDF:/home/alex/Zotero/storage/49N63CMZ/Shi et al. - 2024 - Time-MoE Billion-Scale Time Series Foundation Models with Mixture of Experts.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/HPHN7WNJ/2409.html:text/html}, +} + +@online{noauthor_decoder-only_nodate, + title = {A decoder-only foundation model for time-series forecasting}, + url = {http://research.google/blog/a-decoder-only-foundation-model-for-time-series-forecasting/}, + abstract = {Posted by Rajat Sen and Yichen Zhou, Google Research Time-series forecasting is ubiquitous in various domains, such as retail, finance, manufacturi...}, + urldate = {2024-10-16}, + langid = {english}, + file = {Snapshot:/home/alex/Zotero/storage/JV9JIF73/a-decoder-only-foundation-model-for-time-series-forecasting.html:text/html}, +} + +@misc{goswami_moment_2024, + title = {{MOMENT}: A Family of Open Time-series Foundation Models}, + url = {http://arxiv.org/abs/2402.03885}, + shorttitle = {{MOMENT}}, + abstract = {We introduce {MOMENT}, a family of open-source foundation models for general-purpose time series analysis. Pre-training large models on time series data is challenging due to (1) the absence of a large and cohesive public time series repository, and (2) diverse time series characteristics which make multi-dataset training onerous. Additionally, (3) experimental benchmarks to evaluate these models, especially in scenarios with limited resources, time, and supervision, are still in their nascent stages. To address these challenges, we compile a large and diverse collection of public time series, called the Time series Pile, and systematically tackle time series-specific challenges to unlock large-scale multi-dataset pre-training. Finally, we build on recent work to design a benchmark to evaluate time series foundation models on diverse tasks and datasets in limited supervision settings. Experiments on this benchmark demonstrate the effectiveness of our pre-trained models with minimal data and task-specific fine-tuning. Finally, we present several interesting empirical observations about large pre-trained time series models. Pre-trained models ({AutonLab}/{MOMENT}-1-large) and Time Series Pile ({AutonLab}/Timeseries-{PILE}) are available on Huggingface.}, + number = {{arXiv}:2402.03885}, + publisher = {{arXiv}}, + author = {Goswami, Mononito and Szafer, Konrad and Choudhry, Arjun and Cai, Yifu and Li, Shuo and Dubrawski, Artur}, + urldate = {2024-10-16}, + date = {2024-10-10}, + eprinttype = {arxiv}, + eprint = {2402.03885}, + keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, + file = {Preprint PDF:/home/alex/Zotero/storage/QF6E6J8W/Goswami et al. - 2024 - MOMENT A Family of Open Time-series Foundation Models.pdf:application/pdf}, +} + +@misc{liang_foundation_2024, + title = {Foundation Models for Time Series Analysis: A Tutorial and Survey}, + url = {http://arxiv.org/abs/2403.14735}, + shorttitle = {Foundation Models for Time Series Analysis}, + abstract = {Time series analysis stands as a focal point within the data mining community, serving as a cornerstone for extracting valuable insights crucial to a myriad of real-world applications. Recent advances in Foundation Models ({FMs}) have fundamentally reshaped the paradigm of model design for time series analysis, boosting various downstream tasks in practice. These innovative approaches often leverage pre-trained or fine-tuned {FMs} to harness generalized knowledge tailored for time series analysis. This survey aims to furnish a comprehensive and up-to-date overview of {FMs} for time series analysis. While prior surveys have predominantly focused on either application or pipeline aspects of {FMs} in time series analysis, they have often lacked an in-depth understanding of the underlying mechanisms that elucidate why and how {FMs} benefit time series analysis. To address this gap, our survey adopts a methodology-centric classification, delineating various pivotal elements of time-series {FMs}, including model architectures, pre-training techniques, adaptation methods, and data modalities. Overall, this survey serves to consolidate the latest advancements in {FMs} pertinent to time series analysis, accentuating their theoretical underpinnings, recent strides in development, and avenues for future exploration.}, + number = {{arXiv}:2403.14735}, + publisher = {{arXiv}}, + author = {Liang, Yuxuan and Wen, Haomin and Nie, Yuqi and Jiang, Yushan and Jin, Ming and Song, Dongjin and Pan, Shirui and Wen, Qingsong}, + urldate = {2024-10-16}, + date = {2024-06-18}, + eprinttype = {arxiv}, + eprint = {2403.14735}, + keywords = {Computer Science - Machine Learning}, + file = {Preprint PDF:/home/alex/Zotero/storage/YEA28FMJ/Liang et al. - 2024 - Foundation Models for Time Series Analysis A Tutorial and Survey.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/9SGH3AIN/2403.html:text/html}, +} + +@online{noauthor_pytorch-forecasting_2024, + title = {Pytorch-Forecasting}, + url = {https://pytorch-forecasting.readthedocs.io/en/stable/}, + urldate = {2024-10-15}, + date = {2024}, +} + +@misc{taylor_forecasting_2017, + title = {Forecasting at scale}, + rights = {http://creativecommons.org/licenses/by/4.0/}, + url = {https://peerj.com/preprints/3190v2}, + doi = {10.7287/peerj.preprints.3190v2}, + abstract = {Forecasting is a common data science task that helps organizations with capacity planning, goal setting, and anomaly detection. Despite its importance, there are serious challenges associated with producing reliable and high quality forecasts –especially when there are a variety of time series and analysts with expertise in time series modeling are relatively rare. To address these challenges, we describe a practical approach to forecasting “at scale” that combines configurable models with analyst-in-the-loop performance analysis. We propose a modular regression model with interpretable parameters that can be intuitively adjusted by analysts with domain knowledge about the time series. We describe performance analyses to compare and evaluate forecasting procedures, and automatically flag forecasts for manual review and adjustment. Tools that help analysts to use their expertise most effectively enable reliable, practical forecasting of business time series.}, + publisher = {{PeerJ} Preprints}, + author = {Taylor, Sean J and Letham, Benjamin}, + urldate = {2024-10-14}, + date = {2017-09-27}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/GK5AIG2V/Taylor and Letham - 2017 - Forecasting at scale.pdf:application/pdf}, +} + +@collection{hutter_machine_2021, + location = {Cham}, + title = {Machine Learning and Knowledge Discovery in Databases: European Conference, {ECML} {PKDD} 2020, Ghent, Belgium, September 14–18, 2020, Proceedings, Part {III}}, + volume = {12459}, + rights = {https://www.springernature.com/gp/researchers/text-and-data-mining}, + isbn = {978-3-030-67663-6 978-3-030-67664-3}, + url = {https://link.springer.com/10.1007/978-3-030-67664-3}, + series = {Lecture Notes in Computer Science}, + shorttitle = {Machine Learning and Knowledge Discovery in Databases}, + publisher = {Springer International Publishing}, + editor = {Hutter, Frank and Kersting, Kristian and Lijffijt, Jefrey and Valera, Isabel}, + urldate = {2024-10-14}, + date = {2021}, + langid = {english}, + doi = {10.1007/978-3-030-67664-3}, + file = {Submitted Version:/home/alex/Zotero/storage/JMVJMLJ5/Hutter et al. - 2021 - Machine Learning and Knowledge Discovery in Databases European Conference, ECML PKDD 2020, Ghent, B.pdf:application/pdf}, +} + +@incollection{hutter_general_2021, + location = {Cham}, + title = {A General Machine Learning Framework for Survival Analysis}, + volume = {12459}, + isbn = {978-3-030-67663-6 978-3-030-67664-3}, + url = {https://link.springer.com/10.1007/978-3-030-67664-3_10}, + pages = {158--173}, + booktitle = {Machine Learning and Knowledge Discovery in Databases}, + publisher = {Springer International Publishing}, + author = {Bender, Andreas and Rügamer, David and Scheipl, Fabian and Bischl, Bernd}, + editor = {Hutter, Frank and Kersting, Kristian and Lijffijt, Jefrey and Valera, Isabel}, + urldate = {2024-10-14}, + date = {2021}, + langid = {english}, + doi = {10.1007/978-3-030-67664-3_10}, + note = {Series Title: Lecture Notes in Computer Science}, + file = {Submitted Version:/home/alex/Zotero/storage/WQIHZ7IP/Bender et al. - 2021 - A General Machine Learning Framework for Survival Analysis.pdf:application/pdf}, +} + +@online{lightningai_pytorch_2024, + title = {{PyTorch} Lightning}, + url = {https://www.pytorchlightning.ai}, + author = {lightning.ai}, + urldate = {2024-10-14}, + date = {2024}, +} + +@misc{alexandrov_gluonts_2019, + title = {{GluonTS}: Probabilistic Time Series Models in Python}, + url = {http://arxiv.org/abs/1906.05264}, + shorttitle = {{GluonTS}}, + abstract = {We introduce Gluon Time Series ({GluonTS}, available at https://gluon-ts.mxnet.io), a library for deep-learning-based time series modeling. {GluonTS} simplifies the development of and experimentation with time series models for common tasks such as forecasting or anomaly detection. It provides all necessary components and tools that scientists need for quickly building new models, for efficiently running and analyzing experiments and for evaluating model accuracy.}, + number = {{arXiv}:1906.05264}, + publisher = {{arXiv}}, + author = {Alexandrov, Alexander and Benidis, Konstantinos and Bohlke-Schneider, Michael and Flunkert, Valentin and Gasthaus, Jan and Januschowski, Tim and Maddix, Danielle C. and Rangapuram, Syama and Salinas, David and Schulz, Jasper and Stella, Lorenzo and Türkmen, Ali Caner and Wang, Yuyang}, + urldate = {2024-10-14}, + date = {2019-06-14}, + eprinttype = {arxiv}, + eprint = {1906.05264}, + keywords = {Computer Science - Machine Learning, Statistics - Machine Learning}, + file = {Preprint PDF:/home/alex/Zotero/storage/JP9K74A8/Alexandrov et al. - 2019 - GluonTS Probabilistic Time Series Models in Python.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/RJYSBT29/1906.html:text/html}, +} + +@misc{cho_learning_2014, + title = {Learning Phrase Representations using {RNN} Encoder-Decoder for Statistical Machine Translation}, + url = {http://arxiv.org/abs/1406.1078}, + abstract = {In this paper, we propose a novel neural network model called {RNN} Encoder-Decoder that consists of two recurrent neural networks ({RNN}). One {RNN} encodes a sequence of symbols into a fixed-length vector representation, and the other decodes the representation into another sequence of symbols. The encoder and decoder of the proposed model are jointly trained to maximize the conditional probability of a target sequence given a source sequence. The performance of a statistical machine translation system is empirically found to improve by using the conditional probabilities of phrase pairs computed by the {RNN} Encoder-Decoder as an additional feature in the existing log-linear model. Qualitatively, we show that the proposed model learns a semantically and syntactically meaningful representation of linguistic phrases.}, + number = {{arXiv}:1406.1078}, + publisher = {{arXiv}}, + author = {Cho, Kyunghyun and Merrienboer, Bart van and Gulcehre, Caglar and Bahdanau, Dzmitry and Bougares, Fethi and Schwenk, Holger and Bengio, Yoshua}, + urldate = {2024-10-10}, + date = {2014-09-03}, + eprinttype = {arxiv}, + eprint = {1406.1078}, + keywords = {Computer Science - Machine Learning, Statistics - Machine Learning, Computer Science - Computation and Language, Computer Science - Neural and Evolutionary Computing}, + file = {Preprint PDF:/home/alex/Zotero/storage/E8WMK2IN/Cho et al. - 2014 - Learning Phrase Representations using RNN Encoder-Decoder for Statistical Machine Translation.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/6PTCL8LW/1406.html:text/html}, +} + +@article{hochreiter_long_1997, + title = {Long Short-Term Memory}, + volume = {9}, + issn = {0899-7667, 1530-888X}, + url = {https://direct.mit.edu/neco/article/9/8/1735-1780/6109}, + doi = {10.1162/neco.1997.9.8.1735}, + abstract = {Learningtostoreinformationoverextendedtimeintervalsviarecurrentbackpropagation takesaverylongtime,mostlyduetoinsu cient,decayingerrorbackow.Webrieyreview Hochreiter's1991analysisofthisproblem,thenaddressitbyintroducinganovel,e cient, gradient-basedmethodcalled{\textbackslash}{LongShort}-{TermMemory}"({LSTM}).Truncatingthegradient wherethisdoesnotdoharm,{LSTMcanlearntobridgeminimaltimelagsinexcessof}1000 discretetimestepsbyenforcingconstanterrorowthrough{\textbackslash}constanterrorcarrousels"within specialunits.Multiplicativegateunitslearntoopenandcloseaccesstotheconstanterror ow.{LSTMislocalinspaceandtime};itscomputationalcomplexitypertimestepandweight {isO}(1).Ourexperimentswitharticialdatainvolvelocal,distributed,real-valued,andnoisy patternrepresentations.{IncomparisonswithRTRL},{BPTT},{RecurrentCascade}-Correlation, Elmannets,{andNeuralSequenceChunking},{LSTMleadstomanymoresuccessfulruns},and learnsmuchfaster.{LSTMalsosolvescomplex},articiallongtimelagtasksthathavenever beensolvedbypreviousrecurrentnetworkalgorithms.}, + pages = {1735--1780}, + number = {8}, + journaltitle = {Neural Computation}, + author = {Hochreiter, Sepp and Schmidhuber, Jürgen}, + urldate = {2024-10-10}, + date = {1997-11-01}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/CZSV2ASE/Hochreiter and Schmidhuber - 1997 - Long Short-Term Memory.pdf:application/pdf}, +} + +@misc{lim_temporal_2020, + title = {Temporal Fusion Transformers for Interpretable Multi-horizon Time Series Forecasting}, + url = {http://arxiv.org/abs/1912.09363}, + abstract = {Multi-horizon forecasting problems often contain a complex mix of inputs -- including static (i.e. time-invariant) covariates, known future inputs, and other exogenous time series that are only observed historically -- without any prior information on how they interact with the target. While several deep learning models have been proposed for multi-step prediction, they typically comprise black-box models which do not account for the full range of inputs present in common scenarios. In this paper, we introduce the Temporal Fusion Transformer ({TFT}) -- a novel attention-based architecture which combines high-performance multi-horizon forecasting with interpretable insights into temporal dynamics. To learn temporal relationships at different scales, the {TFT} utilizes recurrent layers for local processing and interpretable self-attention layers for learning long-term dependencies. The {TFT} also uses specialized components for the judicious selection of relevant features and a series of gating layers to suppress unnecessary components, enabling high performance in a wide range of regimes. On a variety of real-world datasets, we demonstrate significant performance improvements over existing benchmarks, and showcase three practical interpretability use-cases of {TFT}.}, + number = {{arXiv}:1912.09363}, + publisher = {{arXiv}}, + author = {Lim, Bryan and Arik, Sercan O. and Loeff, Nicolas and Pfister, Tomas}, + urldate = {2024-10-10}, + date = {2020-09-27}, + eprinttype = {arxiv}, + eprint = {1912.09363}, + keywords = {Computer Science - Machine Learning, Statistics - Machine Learning}, + file = {Preprint PDF:/home/alex/Zotero/storage/2R2H34KB/Lim et al. - 2020 - Temporal Fusion Transformers for Interpretable Multi-horizon Time Series Forecasting.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/ETDAYW36/1912.html:text/html}, +} + +@misc{nie_time_2023, + title = {A Time Series is Worth 64 Words: Long-term Forecasting with Transformers}, + url = {http://arxiv.org/abs/2211.14730}, + shorttitle = {A Time Series is Worth 64 Words}, + abstract = {We propose an efficient design of Transformer-based models for multivariate time series forecasting and self-supervised representation learning. It is based on two key components: (i) segmentation of time series into subseries-level patches which are served as input tokens to Transformer; (ii) channel-independence where each channel contains a single univariate time series that shares the same embedding and Transformer weights across all the series. Patching design naturally has three-fold benefit: local semantic information is retained in the embedding; computation and memory usage of the attention maps are quadratically reduced given the same look-back window; and the model can attend longer history. Our channel-independent patch time series Transformer ({PatchTST}) can improve the long-term forecasting accuracy significantly when compared with that of {SOTA} Transformer-based models. We also apply our model to self-supervised pre-training tasks and attain excellent fine-tuning performance, which outperforms supervised training on large datasets. Transferring of masked pre-trained representation on one dataset to others also produces {SOTA} forecasting accuracy. Code is available at: https://github.com/yuqinie98/{PatchTST}.}, + number = {{arXiv}:2211.14730}, + publisher = {{arXiv}}, + author = {Nie, Yuqi and Nguyen, Nam H. and Sinthong, Phanwadee and Kalagnanam, Jayant}, + urldate = {2024-10-10}, + date = {2023-03-05}, + eprinttype = {arxiv}, + eprint = {2211.14730}, + keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, + file = {Preprint PDF:/home/alex/Zotero/storage/DG4ZJCWV/Nie et al. - 2023 - A Time Series is Worth 64 Words Long-term Forecasting with Transformers.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/H6XGVBY6/2211.html:text/html}, +} + +@misc{shao_exploring_2023, + title = {Exploring Progress in Multivariate Time Series Forecasting: Comprehensive Benchmarking and Heterogeneity Analysis}, + url = {http://arxiv.org/abs/2310.06119}, + shorttitle = {Exploring Progress in Multivariate Time Series Forecasting}, + abstract = {Multivariate Time Series ({MTS}) widely exists in real-word complex systems, such as traffic and energy systems, making their forecasting crucial for understanding and influencing these systems. Recently, deep learning-based approaches have gained much popularity for effectively modeling temporal and spatial dependencies in {MTS}, specifically in Long-term Time Series Forecasting ({LTSF}) and Spatial-Temporal Forecasting ({STF}). However, the fair benchmarking issue and the choice of technical approaches have been hotly debated in related work. Such controversies significantly hinder our understanding of progress in this field. Thus, this paper aims to address these controversies to present insights into advancements achieved. To resolve benchmarking issues, we introduce {BasicTS}, a benchmark designed for fair comparisons in {MTS} forecasting. {BasicTS} establishes a unified training pipeline and reasonable evaluation settings, enabling an unbiased evaluation of over 30 popular {MTS} forecasting models on more than 18 datasets. Furthermore, we highlight the heterogeneity among {MTS} datasets and classify them based on temporal and spatial characteristics. We further prove that neglecting heterogeneity is the primary reason for generating controversies in technical approaches. Moreover, based on the proposed {BasicTS} and rich heterogeneous {MTS} datasets, we conduct an exhaustive and reproducible performance and efficiency comparison of popular models, providing insights for researchers in selecting and designing {MTS} forecasting models.}, + number = {{arXiv}:2310.06119}, + publisher = {{arXiv}}, + author = {Shao, Zezhi and Wang, Fei and Xu, Yongjun and Wei, Wei and Yu, Chengqing and Zhang, Zhao and Yao, Di and Jin, Guangyin and Cao, Xin and Cong, Gao and Jensen, Christian S. and Cheng, Xueqi}, + urldate = {2024-10-10}, + date = {2023-10-09}, + eprinttype = {arxiv}, + eprint = {2310.06119}, + keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, + file = {Preprint PDF:/home/alex/Zotero/storage/7EFZ5IT6/Shao et al. - 2023 - Exploring Progress in Multivariate Time Series Forecasting Comprehensive Benchmarking and Heterogen.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/W6RNWBLM/2310.html:text/html}, +} + +@article{zhang_crossformer_2023, + title = {{CROSSFORMER}: {TRANSFORMER} {UTILIZING} {CROSS}- {DIMENSION} {DEPENDENCY} {FOR} {MULTIVARIATE} {TIME} {SERIES} {FORECASTING}}, + abstract = {Recently many deep models have been proposed for multivariate time series ({MTS}) forecasting. In particular, Transformer-based models have shown great potential because they can capture long-term dependency. However, existing Transformerbased models mainly focus on modeling the temporal dependency (cross-time dependency) yet often omit the dependency among different variables (crossdimension dependency), which is critical for {MTS} forecasting. To fill the gap, we propose Crossformer, a Transformer-based model utilizing cross-dimension dependency for {MTS} forecasting. In Crossformer, the input {MTS} is embedded into a 2D vector array through the Dimension-Segment-Wise ({DSW}) embedding to preserve time and dimension information. Then the Two-Stage Attention ({TSA}) layer is proposed to efficiently capture the cross-time and cross-dimension dependency. Utilizing {DSW} embedding and {TSA} layer, Crossformer establishes a Hierarchical Encoder-Decoder ({HED}) to use the information at different scales for the final forecasting. Extensive experimental results on six real-world datasets show the effectiveness of Crossformer against previous state-of-the-arts.}, + author = {Zhang, Yunhao and Yan, Junchi}, + date = {2023}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/NM9CETJS/Zhang and Yan - 2023 - CROSSFORMER TRANSFORMER UTILIZING CROSS- DIMENSION DEPENDENCY FOR MULTIVARIATE TIME SERIES FORECAST.pdf:application/pdf}, +} + +@book{noauthor_research_2021, + title = {Research Funding for Women's Health: A Modeling Study of Societal Impact: Findings for Alzheimer's Disease and Alzheimer's Disease Related Dementia Model}, + url = {https://www.rand.org/pubs/working_papers/WRA708-1.html}, + shorttitle = {Research Funding for Women's Health}, + publisher = {{RAND} Corporation}, + urldate = {2025-03-07}, + date = {2021}, + langid = {english}, + doi = {10.7249/WRA708-1}, + file = {PDF:/home/alex/Zotero/storage/93NC3JH3/2021 - Research Funding for Women's Health A Modeling Study of Societal Impact Findings for Alzheimer's D.pdf:application/pdf}, +} + +@report{mckinsey_health_institute_closing_2024, + title = {Closing the Women’s Health Gap: A \$1 Trillion Opportunity to Improve Lives and Economies}, + shorttitle = {Closing the Women’s Health Gap}, + institution = {{McKinsey} Health Institute, World Economic Forum}, + author = {{McKinsey} Health Institute and World Economic Forum}, + urldate = {2025-03-07}, + date = {2024-01}, + file = {PDF:/home/alex/Zotero/storage/US6TWL7D/closing-the-womens-health-gap-report.pdf:application/pdf}, +} + +@report{mckinsey_health_institute_blueprint_nodate, + title = {Blueprint to Close the Women’s Health Gap: How to Improve Lives and Economies for All}, + shorttitle = {Blueprint to Close the Women’s Health Gap}, + institution = {{McKinsey} Health Institute, World Economic Forum}, + author = {{McKinsey} Health Institute and World Economic Forum}, + urldate = {2025-03-07}, + file = {PDF:/home/alex/Zotero/storage/NF6LAC2S/WEF_Blueprint_to_Close_the_Women’s_Health_Gap_2025.pdf:application/pdf}, +} + +@misc{global_burden_of_disease_collaborative_network_global_2020, + title = {Global Burden of Disease Study 2019 ({GBD} 2019) Disability Weights}, + url = {http://ghdx.healthdata.org/record/ihme-data/gbd-2019-disability-weights}, + doi = {10.6069/1W19-VX76}, + abstract = {"The Global Burden of Disease Study 2019 ({GBD} 2019), coordinated by the Institute for Health Metrics and Evaluation ({IHME}), estimated the burden of diseases, injuries, and risk factors for 204 countries and territories and selected subnational locations. + +Disability weights, which represent the magnitude of health loss associated with specific health outcomes, are used to calculate years lived with disability ({YLD}) for these outcomes in a given population. The weights are measured on a scale from 0 to 1, where 0 equals a state of full health and 1 equals death. This table provides disability weights for the 440 health states (including combined health states) used to estimate nonfatal health outcomes for the {GBD} 2019 study. + +For additional {GBD} results and resources, visit the {GBD} 2019 Data Resources page."}, + publisher = {Institute for Health Metrics and Evaluation ({IHME})}, + author = {{Global Burden of Disease Collaborative Network}}, + urldate = {2025-03-07}, + date = {2020}, +} + +@article{sauer_reproduction_2015, + title = {Reproduction at an advanced maternal age and maternal health}, + volume = {103}, + issn = {00150282}, + url = {https://linkinghub.elsevier.com/retrieve/pii/S0015028215002034}, + doi = {10.1016/j.fertnstert.2015.03.004}, + pages = {1136--1143}, + number = {5}, + journaltitle = {Fertility and Sterility}, + author = {Sauer, Mark V.}, + urldate = {2025-03-07}, + date = {2015-05}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/W6LCEFP6/Sauer - 2015 - Reproduction at an advanced maternal age and maternal health.pdf:application/pdf}, +} + +@incollection{holesh_physiology_2025, + location = {Treasure Island ({FL})}, + title = {Physiology, Ovulation}, + rights = {Copyright © 2025, {StatPearls} Publishing {LLC}.}, + url = {http://www.ncbi.nlm.nih.gov/books/NBK441996/}, + abstract = {Ovulation is a physiologic process defined by the rupture of the dominant follicle of the ovary. This releases an egg into the abdominal cavity. It then is taken up by the fimbriae of the fallopian tube where it has the potential to become fertilized. The ovulation process is regulated by fluxing gonadotropic hormone ({FSH}/{LH}) levels. Ovulation is the third phase within the larger uterine cycle (ie, menstrual cycle). The follicular release follows the Follicular phase (ie, dominant follicle development) and precedes the luteal phase (ie, maintenance of corpus luteum) that progresses to either endometrial shedding or implantation. Follicular release occurs around 14 days prior to menstruation in a cyclic pattern if the hypothalamic-pituitary-ovarian axis function is well regulated.  Structure Genotypic females ({XX}) develop two ovaries that sit adjacent to the uterine horns. Each ovary is anchored to the uterus at the medial pole by the utero-ovarian ligament. The lateral ovarian pole is anchored to the pelvic sidewall by the infundibulopelvic ligament (i.e,. suspensory ligament of the ovary), which carries the ovarian artery and vein. Each ovary contains 1 to 2 million primordial follicles that each contain primary oocytes (ie, eggs) that can supply that female with enough follicles until she reaches her fourth or fifth decades of life. These primordial follicles are arrested in prophase I of meiosis until the onset of puberty. At the onset of pubescence, the gonadotropic hormones began to induce the maturation of the primordial follicle, allowing for the completion of meiosis I, forming a secondary follicle. The secondary follicle begins meiosis {II}, but this phase will not be completed unless that follicle is fertilized. With each ovulatory cycle, the number of follicles decreases, eventually leading to the onset of Menopause or the cessation of ovulatory function. Per each ovulation cycle, the average ovary loses 1,000 follicles to the process of selecting a dominant follicle that will be released. This process accelerates in an age-dependent manner as well. It is also a common thought that the right and left ovaries alternate follicular releases each month. Ovulation is regulated by the fluctuation between the following hormones. Tight regulation and controlled changes between the following hormones are imperative for the development and release of an oocyte into the adnexal uterine structures.   Hormones involved in ovulation include: Gonadotropin-releasing hormone ({GnRH}) is a tropic peptide hormone made and secreted by the hypothalamus. It is a releasing hormone that stimulates the release of {FSH} and {LH} from the anterior pituitary gland through variations in {GnRH} pulse frequency. Low-frequency {GnRH} pulses are responsible for {FSH} secretion, whereas high-frequency pulses are responsible for {LH} secretion. During the Follicular phase of the Uterine cycle, estrogen secretion causes the Granulosa cells to autonomously increase their own production of estrogen, contributing to elevation in estrogen serum levels. This elevation is communicated to the hypothalamus and contributes to the increase in {GnRH} pulse frequency, eventually stimulating the {LH} surge that eventually induces the follicular rupture and release from the corpus luteum and luteinization of the granulosa cells, enabling the synthesis of progesterone in place of estrogen. Finally, the low levels of {LH} following the surge restart the {FSH} production by the slow-pulsation frequency of {GnRH} release. . Gonadotropin hormones are heterodimeric glycoproteins with alpha/beta subunits. The alpha subunit is common to all glycoproteins, including {TSH} (thyroid-stimulating hormone) and {HCG} (human chorionic gonadotropin hormone).  The relationship between {FSH} and {LH} hormones is responsible for the process that induces follicular development, rupture, release, and endometrial reception or shedding. Disruption in the hormonal communication between the gonadotropin-releasing hormones, gonadotropic hormones, and their receptors can lead to anovulation or amenorrhea, leading to various pathologic sequelae as a consequence. Follicle-Stimulating Hormone ({FSH}) is a gonadotropin synthesized and secreted from the anterior pituitary gland in response to slow-frequency pulsatile {GnRH}. {FSH} stimulates the growth and maturation of immature oocytes into mature (Graafian) secondary follicles before ovulation. {FSH} Receptors are G-protein coupled receptors and are found in the Granulosa cells that surround developing ovarian follicles. The granulosa cells initially produce the estrogen needed to maturate the developing dominant follicle. After 2 days of sustained elevation of estrogen levels, the {LH} surge causes luteinization of the granulosa cells into {LH} receptive cells. This transition enables granulosa cells to respond to {LH} levels and produce progesterone. : Estrogen is a steroid hormone that is responsible for the growth and regulation of the female reproductive system and secondary sex characteristics. Estrogen is produced by the granulosa cells of the developing follicle and exerts negative feedback on {LH} production in the early part of the menstrual cycle. However, once estrogen levels reach a critical level as oocytes mature within the ovary in preparation for ovulation, estrogen begins to exert positive feedback on {LH} production, leading to the {LH} surge through its effects on {GnRH} pulse frequency. Estrogen also has many other effects that are important for bone health and cardiovascular health in premenopausal patients, which will be discussed in another article. Luteinizing Hormone ({LH}) is a gonadotropin synthesized and secreted by the anterior pituitary gland in response to high-frequency {GnRH} release. {LH} is responsible for inducing ovulation, preparation for fertilized oocyte uterine implantation, and the ovarian production of progesterone through stimulation of theca cells and luteinized granulosa cells. Prior to the {LH} surge, {LH} interacts with Theca cells that are adjacent to granulosa cells in the ovary. These cells produce androgens, which diffuse into the granulosa cells and convert to estrogen for follicular development. The {LH} surge creates the environment for follicular eruption by increasing the activity of the proteolytic enzymes that weaken the ovarian wall, allowing for the passage of the oocyte. After the oocyte is released, the follicular remnants are theca and luteinized granulosa cells. Their function is now to produce progesterone, which is the hormone responsible for maintaining the uterine environment that can accept a fertilized embryo. Progesterone is a steroid hormone that is responsible for preparing the endometrium for the uterine implantation of the fertilized egg and maintenance of pregnancy. If a fertilized egg implants, the corpus luteum secretes progesterone in early pregnancy until the placenta develops and takes over progesterone production for the remainder of the pregnancy.}, + booktitle = {{StatPearls}}, + publisher = {{StatPearls} Publishing}, + author = {Holesh, Julie E. and Bass, Autumn N. and Lord, Megan}, + urldate = {2025-03-07}, + date = {2025}, + pmid = {28723025}, + file = {Printable HTML:/home/alex/Zotero/storage/D6LTGP64/NBK441996.html:text/html}, +} + +@article{bull_real-world_2019, + title = {Real-world menstrual cycle characteristics of more than 600,000 menstrual cycles}, + volume = {2}, + rights = {2019 The Author(s)}, + issn = {2398-6352}, + url = {https://www.nature.com/articles/s41746-019-0152-7}, + doi = {10.1038/s41746-019-0152-7}, + abstract = {The use of apps that record detailed menstrual cycle data presents a new opportunity to study the menstrual cycle. The aim of this study is to describe menstrual cycle characteristics observed from a large database of cycles collected through an app and investigate associations of menstrual cycle characteristics with cycle length, age and body mass index ({BMI}). Menstrual cycle parameters, including menstruation, basal body temperature ({BBT}) and luteinising hormone ({LH}) tests as well as age and {BMI} were collected anonymously from real-world users of the Natural Cycles app. We analysed 612,613 ovulatory cycles with a mean length of 29.3 days from 124,648 users. The mean follicular phase length was 16.9 days (95\% {CI}: 10–30) and mean luteal phase length was 12.4 days (95\% {CI}: 7–17). Mean cycle length decreased by 0.18 days (95\% {CI}: 0.17–0.18, R2 = 0.99) and mean follicular phase length decreased by 0.19 days (95\% {CI}: 0.19–0.20, R2 = 0.99) per year of age from 25 to 45 years. Mean variation of cycle length per woman was 0.4 days or 14\% higher in women with a {BMI} of over 35 relative to women with a {BMI} of 18.5–25. This analysis details variations in menstrual cycle characteristics that are not widely known yet have significant implications for health and well-being. Clinically, women who wish to plan a pregnancy need to have intercourse on their fertile days. In order to identify the fertile period it is important to track physiological parameters such as basal body temperature and not just cycle length.}, + pages = {1--8}, + number = {1}, + journaltitle = {npj Digit. Med.}, + author = {Bull, Jonathan R. and Rowland, Simon P. and Scherwitzl, Elina Berglund and Scherwitzl, Raoul and Danielsson, Kristina Gemzell and Harper, Joyce}, + urldate = {2025-03-07}, + date = {2019-08-27}, + langid = {english}, + note = {Publisher: Nature Publishing Group}, + keywords = {Preclinical research, Reproductive biology}, + file = {Full Text PDF:/home/alex/Zotero/storage/MUPKFK2K/Bull et al. - 2019 - Real-world menstrual cycle characteristics of more than 600,000 menstrual cycles.pdf:application/pdf}, } @article{munster_length_1992, - title = {Length and variation in the menstrual cycle—a cross‐sectional study from a {Danish} county}, + title = {Length and variation in the menstrual cycle—a cross‐sectional study from a Danish county}, volume = {99}, issn = {1470-0328, 1471-0528}, url = {https://obgyn.onlinelibrary.wiley.com/doi/10.1111/j.1471-0528.1992.tb13762.x}, doi = {10.1111/j.1471-0528.1992.tb13762.x}, - abstract = {ABSTRACT + abstract = {{ABSTRACT} Objective To investigate the current epidemiology of menstrual patters among women of fertile age. @@ -1353,141 +1297,95 @@ we have a relatively large dataset, thus tft will be focused on Conclusion The study confirmed the normally used definitions of polymenorrhoea (cycle length {\textless}21 days) and oligomenorrhoea (cycle length between 36 and 90 days), as these very short or long menstrual cycle lengths were very seldom recorded for a longer period. However, the high frequency in a normal population of large menstrual cycle length variation challenges the view that an intra‐individual variation of {\textgreater}5 days should be regarded as a sign of disease in the woman.}, - language = {en}, - number = {5}, - urldate = {2025-03-07}, - journal = {BJOG: An International Journal of Obstetrics \& Gynaecology}, - author = {Münster, Kirstine and Schmidt, Lone and Helm, Peter}, - month = may, - year = {1992}, pages = {422--429}, + number = {5}, + journaltitle = {{BJOG}}, + author = {Münster, Kirstine and Schmidt, Lone and Helm, Peter}, + urldate = {2025-03-07}, + date = {1992-05}, + langid = {english}, file = {PDF:/home/alex/Zotero/storage/JWE75VL3/Münster et al. - 1992 - Length and variation in the menstrual cycle—a cross‐sectional study from a Danish county.pdf:application/pdf}, } -@article{bull_real-world_2019, - title = {Real-world menstrual cycle characteristics of more than 600,000 menstrual cycles}, - volume = {2}, - copyright = {2019 The Author(s)}, - issn = {2398-6352}, - url = {https://www.nature.com/articles/s41746-019-0152-7}, - doi = {10.1038/s41746-019-0152-7}, - abstract = {The use of apps that record detailed menstrual cycle data presents a new opportunity to study the menstrual cycle. The aim of this study is to describe menstrual cycle characteristics observed from a large database of cycles collected through an app and investigate associations of menstrual cycle characteristics with cycle length, age and body mass index (BMI). Menstrual cycle parameters, including menstruation, basal body temperature (BBT) and luteinising hormone (LH) tests as well as age and BMI were collected anonymously from real-world users of the Natural Cycles app. We analysed 612,613 ovulatory cycles with a mean length of 29.3 days from 124,648 users. The mean follicular phase length was 16.9 days (95\% CI: 10–30) and mean luteal phase length was 12.4 days (95\% CI: 7–17). Mean cycle length decreased by 0.18 days (95\% CI: 0.17–0.18, R2 = 0.99) and mean follicular phase length decreased by 0.19 days (95\% CI: 0.19–0.20, R2 = 0.99) per year of age from 25 to 45 years. Mean variation of cycle length per woman was 0.4 days or 14\% higher in women with a BMI of over 35 relative to women with a BMI of 18.5–25. This analysis details variations in menstrual cycle characteristics that are not widely known yet have significant implications for health and well-being. Clinically, women who wish to plan a pregnancy need to have intercourse on their fertile days. In order to identify the fertile period it is important to track physiological parameters such as basal body temperature and not just cycle length.}, - language = {en}, - number = {1}, - urldate = {2025-03-07}, - journal = {npj Digital Medicine}, - author = {Bull, Jonathan R. and Rowland, Simon P. and Scherwitzl, Elina Berglund and Scherwitzl, Raoul and Danielsson, Kristina Gemzell and Harper, Joyce}, - month = aug, - year = {2019}, - note = {Publisher: Nature Publishing Group}, - keywords = {Preclinical research, Reproductive biology}, - pages = {1--8}, - file = {Full Text PDF:/home/alex/Zotero/storage/MUPKFK2K/Bull et al. - 2019 - Real-world menstrual cycle characteristics of more than 600,000 menstrual cycles.pdf:application/pdf}, +@online{pedroso_menstrual_2022, + title = {The Menstrual Cycle}, + url = {https://kindbody.com/the-menstrual-cycle/}, + abstract = {Fertility, gynecology, and wellness services in modern, tech-enabled clinics. Best-in-class care, accessible pricing, and a seamless patient experience.}, + titleaddon = {Kindbody}, + author = {Pedroso, Dr Jasmine}, + urldate = {2025-03-10}, + date = {2022-06-03}, + file = {Snapshot:/home/alex/Zotero/storage/738JEKW7/the-menstrual-cycle.html:text/html}, } -@incollection{holesh_physiology_2025, - address = {Treasure Island (FL)}, - title = {Physiology, {Ovulation}}, - copyright = {Copyright © 2025, StatPearls Publishing LLC.}, - url = {http://www.ncbi.nlm.nih.gov/books/NBK441996/}, - abstract = {Ovulation is a physiologic process defined by the rupture of the dominant follicle of the ovary. This releases an egg into the abdominal cavity. It then is taken up by the fimbriae of the fallopian tube where it has the potential to become fertilized. The ovulation process is regulated by fluxing gonadotropic hormone (FSH/LH) levels. Ovulation is the third phase within the larger uterine cycle (ie, menstrual cycle). The follicular release follows the Follicular phase (ie, dominant follicle development) and precedes the luteal phase (ie, maintenance of corpus luteum) that progresses to either endometrial shedding or implantation. Follicular release occurs around 14 days prior to menstruation in a cyclic pattern if the hypothalamic-pituitary-ovarian axis function is well regulated.  Structure Genotypic females (XX) develop two ovaries that sit adjacent to the uterine horns. Each ovary is anchored to the uterus at the medial pole by the utero-ovarian ligament. The lateral ovarian pole is anchored to the pelvic sidewall by the infundibulopelvic ligament (i.e,. suspensory ligament of the ovary), which carries the ovarian artery and vein. Each ovary contains 1 to 2 million primordial follicles that each contain primary oocytes (ie, eggs) that can supply that female with enough follicles until she reaches her fourth or fifth decades of life. These primordial follicles are arrested in prophase I of meiosis until the onset of puberty. At the onset of pubescence, the gonadotropic hormones began to induce the maturation of the primordial follicle, allowing for the completion of meiosis I, forming a secondary follicle. The secondary follicle begins meiosis II, but this phase will not be completed unless that follicle is fertilized. With each ovulatory cycle, the number of follicles decreases, eventually leading to the onset of Menopause or the cessation of ovulatory function. Per each ovulation cycle, the average ovary loses 1,000 follicles to the process of selecting a dominant follicle that will be released. This process accelerates in an age-dependent manner as well. It is also a common thought that the right and left ovaries alternate follicular releases each month. Ovulation is regulated by the fluctuation between the following hormones. Tight regulation and controlled changes between the following hormones are imperative for the development and release of an oocyte into the adnexal uterine structures.   Hormones involved in ovulation include: Gonadotropin-releasing hormone (GnRH) is a tropic peptide hormone made and secreted by the hypothalamus. It is a releasing hormone that stimulates the release of FSH and LH from the anterior pituitary gland through variations in GnRH pulse frequency. Low-frequency GnRH pulses are responsible for FSH secretion, whereas high-frequency pulses are responsible for LH secretion. During the Follicular phase of the Uterine cycle, estrogen secretion causes the Granulosa cells to autonomously increase their own production of estrogen, contributing to elevation in estrogen serum levels. This elevation is communicated to the hypothalamus and contributes to the increase in GnRH pulse frequency, eventually stimulating the LH surge that eventually induces the follicular rupture and release from the corpus luteum and luteinization of the granulosa cells, enabling the synthesis of progesterone in place of estrogen. Finally, the low levels of LH following the surge restart the FSH production by the slow-pulsation frequency of GnRH release. . Gonadotropin hormones are heterodimeric glycoproteins with alpha/beta subunits. The alpha subunit is common to all glycoproteins, including TSH (thyroid-stimulating hormone) and HCG (human chorionic gonadotropin hormone).  The relationship between FSH and LH hormones is responsible for the process that induces follicular development, rupture, release, and endometrial reception or shedding. Disruption in the hormonal communication between the gonadotropin-releasing hormones, gonadotropic hormones, and their receptors can lead to anovulation or amenorrhea, leading to various pathologic sequelae as a consequence. Follicle-Stimulating Hormone (FSH) is a gonadotropin synthesized and secreted from the anterior pituitary gland in response to slow-frequency pulsatile GnRH. FSH stimulates the growth and maturation of immature oocytes into mature (Graafian) secondary follicles before ovulation. FSH Receptors are G-protein coupled receptors and are found in the Granulosa cells that surround developing ovarian follicles. The granulosa cells initially produce the estrogen needed to maturate the developing dominant follicle. After 2 days of sustained elevation of estrogen levels, the LH surge causes luteinization of the granulosa cells into LH receptive cells. This transition enables granulosa cells to respond to LH levels and produce progesterone. : Estrogen is a steroid hormone that is responsible for the growth and regulation of the female reproductive system and secondary sex characteristics. Estrogen is produced by the granulosa cells of the developing follicle and exerts negative feedback on LH production in the early part of the menstrual cycle. However, once estrogen levels reach a critical level as oocytes mature within the ovary in preparation for ovulation, estrogen begins to exert positive feedback on LH production, leading to the LH surge through its effects on GnRH pulse frequency. Estrogen also has many other effects that are important for bone health and cardiovascular health in premenopausal patients, which will be discussed in another article. Luteinizing Hormone (LH) is a gonadotropin synthesized and secreted by the anterior pituitary gland in response to high-frequency GnRH release. LH is responsible for inducing ovulation, preparation for fertilized oocyte uterine implantation, and the ovarian production of progesterone through stimulation of theca cells and luteinized granulosa cells. Prior to the LH surge, LH interacts with Theca cells that are adjacent to granulosa cells in the ovary. These cells produce androgens, which diffuse into the granulosa cells and convert to estrogen for follicular development. The LH surge creates the environment for follicular eruption by increasing the activity of the proteolytic enzymes that weaken the ovarian wall, allowing for the passage of the oocyte. After the oocyte is released, the follicular remnants are theca and luteinized granulosa cells. Their function is now to produce progesterone, which is the hormone responsible for maintaining the uterine environment that can accept a fertilized embryo. Progesterone is a steroid hormone that is responsible for preparing the endometrium for the uterine implantation of the fertilized egg and maintenance of pregnancy. If a fertilized egg implants, the corpus luteum secretes progesterone in early pregnancy until the placenta develops and takes over progesterone production for the remainder of the pregnancy.}, - language = {eng}, - urldate = {2025-03-07}, - booktitle = {{StatPearls}}, - publisher = {StatPearls Publishing}, - author = {Holesh, Julie E. and Bass, Autumn N. and Lord, Megan}, - year = {2025}, - pmid = {28723025}, - file = {Printable HTML:/home/alex/Zotero/storage/D6LTGP64/NBK441996.html:text/html}, +@article{silberstein_physiology_2000, + title = {Physiology of the Menstrual Cycle}, + volume = {20}, + rights = {https://journals.sagepub.com/page/policies/text-and-data-mining-license}, + issn = {0333-1024, 1468-2982}, + url = {https://journals.sagepub.com/doi/10.1046/j.1468-2982.2000.00034.x}, + doi = {10.1046/j.1468-2982.2000.00034.x}, + abstract = {The normal female life cycle is associated with a number of hormonal milestones: menarche, pregnancy, contraceptive use, menopause, and the use of replacement sex hormones. All these events and interventions alter the levels and cycling of sex hormones and may cause a change in the prevalence or intensity of headache. The menstrual cycle is the result of a carefully orchestrated sequence of interactions among the hypothalamus, pituitary, ovary, and endometrium, with the sex hormones acting as modulators and effectors at each level. Oestrogen and progestins have potent effects on central serotonergic and opioid neurons, modulating both neuronal activity and receptor density. The primary trigger of menstrual migraine appears to be the withdrawal of oestrogen rather than the maintenance of sustained high or low oestrogen levels. However, changes in the sustained oestrogen levels with pregnancy (increased) and menopause (decreased) appear to affect headaches. Headaches that occur with premenstrual syndrome appear to be centrally generated, involving the inherent rhythm of {CNS} neurons, including perhaps the serotonergic pain-modulating systems.}, + pages = {148--154}, + number = {3}, + journaltitle = {Cephalalgia}, + author = {Silberstein, S D and Merriam, G R}, + urldate = {2025-03-10}, + date = {2000-04}, + langid = {english}, + file = {Full Text:/home/alex/Zotero/storage/EL585H2P/Silberstein and Merriam - 2000 - Physiology of the Menstrual Cycle.pdf:application/pdf}, } -@article{sauer_reproduction_2015, - title = {Reproduction at an advanced maternal age and maternal health}, - volume = {103}, - issn = {00150282}, - url = {https://linkinghub.elsevier.com/retrieve/pii/S0015028215002034}, - doi = {10.1016/j.fertnstert.2015.03.004}, - language = {en}, - number = {5}, - urldate = {2025-03-07}, - journal = {Fertility and Sterility}, - author = {Sauer, Mark V.}, - month = may, - year = {2015}, - pages = {1136--1143}, - file = {PDF:/home/alex/Zotero/storage/W6LCEFP6/Sauer - 2015 - Reproduction at an advanced maternal age and maternal health.pdf:application/pdf}, +@artwork{wikimedia_commons_basic_2019, + title = {Basic Female Reproductive System}, + url = {https://en.wikipedia.org/wiki/File:Basic_Female_Reproductive_System_(English).svg}, + author = {Wikimedia Commons}, + date = {2019}, + file = {background_female_reproductive_organs:/home/alex/Zotero/storage/5MAJ4CG5/background_female_reproductive_organs.png:image/png}, } -@techreport{mckinsey_health_institute_blueprint_nodate, - title = {Blueprint to {Close} the {Women}’s {Health} {Gap}: {How} to {Improve} {Lives} and {Economies} for {All}}, - shorttitle = {Blueprint to {Close} the {Women}’s {Health} {Gap}}, - urldate = {2025-03-07}, - institution = {McKinsey Health Institute, World Economic Forum}, - author = {McKinsey Health Institute and World Economic Forum}, - file = {PDF:/home/alex/Zotero/storage/NF6LAC2S/WEF_Blueprint_to_Close_the_Women’s_Health_Gap_2025.pdf:application/pdf}, +@book{hamilton_time_1994, + location = {Princeton (N.J.)}, + title = {Time series analysis}, + isbn = {978-0-691-04289-3}, + publisher = {Princeton university press}, + author = {Hamilton, James Douglas}, + date = {1994}, + file = {PDF:/home/alex/Zotero/storage/J8GVKKHG/Hamilton - 1994 - Time series analysis.pdf:application/pdf}, } -@techreport{mckinsey_health_institute_closing_2024, - title = {Closing the {Women}’s {Health} {Gap}: {A} \$1 {Trillion} {Opportunity} to {Improve} {Lives} and {Economies}}, - shorttitle = {Closing the {Women}’s {Health} {Gap}}, - urldate = {2025-03-07}, - institution = {McKinsey Health Institute, World Economic Forum}, - author = {McKinsey Health Institute and World Economic Forum}, - month = jan, - year = {2024}, - file = {PDF:/home/alex/Zotero/storage/US6TWL7D/closing-the-womens-health-gap-report.pdf:application/pdf}, +@book{cryer_time_2008, + location = {New York}, + edition = {2nd ed}, + title = {Time series analysis: with applications in R}, + isbn = {978-0-387-75958-6 978-0-387-75959-3}, + series = {Springer texts in statistics}, + shorttitle = {Time series analysis}, + pagetotal = {491}, + publisher = {Springer}, + author = {Cryer, Jonathan D. and Chan, Kung-sik}, + date = {2008}, + langid = {english}, + note = {{OCLC}: ocn191760003}, + keywords = {Data processing, R (Computer program language), Time-series analysis}, + file = {PDF:/home/alex/Zotero/storage/CIYMBUEW/Cryer and Chan - 2008 - Time series analysis with applications in R.pdf:application/pdf}, } -@misc{global_burden_of_disease_collaborative_network_global_2020, - title = {Global {Burden} of {Disease} {Study} 2019 ({GBD} 2019) {Disability} {Weights}}, - url = {http://ghdx.healthdata.org/record/ihme-data/gbd-2019-disability-weights}, - doi = {10.6069/1W19-VX76}, - abstract = {"The Global Burden of Disease Study 2019 (GBD 2019), coordinated by the Institute for Health Metrics and Evaluation (IHME), estimated the burden of diseases, injuries, and risk factors for 204 countries and territories and selected subnational locations. - -Disability weights, which represent the magnitude of health loss associated with specific health outcomes, are used to calculate years lived with disability (YLD) for these outcomes in a given population. The weights are measured on a scale from 0 to 1, where 0 equals a state of full health and 1 equals death. This table provides disability weights for the 440 health states (including combined health states) used to estimate nonfatal health outcomes for the GBD 2019 study. - -For additional GBD results and resources, visit the GBD 2019 Data Resources page."}, - urldate = {2025-03-07}, - publisher = {Institute for Health Metrics and Evaluation (IHME)}, - author = {{Global Burden of Disease Collaborative Network}}, - year = {2020}, -} - -@book{noauthor_research_2021, - title = {Research {Funding} for {Women}'s {Health}: {A} {Modeling} {Study} of {Societal} {Impact}: {Findings} for {Alzheimer}'s {Disease} and {Alzheimer}'s {Disease} {Related} {Dementia} {Model}}, - shorttitle = {Research {Funding} for {Women}'s {Health}}, - url = {https://www.rand.org/pubs/working_papers/WRA708-1.html}, - language = {en}, - urldate = {2025-03-07}, - publisher = {RAND Corporation}, - year = {2021}, - doi = {10.7249/WRA708-1}, - file = {PDF:/home/alex/Zotero/storage/93NC3JH3/2021 - Research Funding for Women's Health A Modeling Study of Societal Impact Findings for Alzheimer's D.pdf:application/pdf}, -} - -@misc{noauthor_create_nodate, - title = {Create baseline model - {ValueError}: too many values to unpack (expected 2) · {Issue} \#230 · sktime/pytorch-forecasting}, - shorttitle = {Create baseline model - {ValueError}}, - url = {https://github.com/sktime/pytorch-forecasting/issues/230}, - abstract = {PyTorch-Forecasting version: 0.7.1 PyTorch version: 1.7.1 Python version: 3.7 Operating System: MAC OS Big Sur: Version 11.1 Expected behavior I executed code actuals = torch.cat([y for x, (y, weig...}, - language = {en}, - urldate = {2025-03-19}, - journal = {GitHub}, -} - -@book{pfannstiel_entrepreneurship_2018, - address = {Wiesbaden}, - title = {Entrepreneurship im {Gesundheitswesen} {II}}, - copyright = {http://www.springer.com/tdm}, - isbn = {978-3-658-14780-8 978-3-658-14781-5}, - url = {http://link.springer.com/10.1007/978-3-658-14781-5}, - language = {de}, +@article{rosenfield_adolescent_2013, + title = {Adolescent Anovulation: Maturational Mechanisms and Implications}, + volume = {98}, + issn = {0021-972X, 1945-7197}, + url = {https://academic.oup.com/jcem/article-lookup/doi/10.1210/jc.2013-1770}, + doi = {10.1210/jc.2013-1770}, + shorttitle = {Adolescent Anovulation}, + pages = {3572--3583}, + number = {9}, + journaltitle = {The Journal of Clinical Endocrinology \& Metabolism}, + author = {Rosenfield, Robert L.}, urldate = {2025-03-17}, - publisher = {Springer Fachmedien Wiesbaden}, - editor = {Pfannstiel, Mario A. and Da-Cruz, Patrick and Rasche, Christoph}, - year = {2018}, - doi = {10.1007/978-3-658-14781-5}, - file = {PDF:/home/alex/Zotero/storage/DTVM5NBP/Pfannstiel et al. - 2018 - Entrepreneurship im Gesundheitswesen II.pdf:application/pdf}, + date = {2013-09}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/BEF7PQT6/Rosenfield - 2013 - Adolescent Anovulation Maturational Mechanisms and Implications.pdf:application/pdf}, } @article{murray_diagnosis_2005, @@ -1496,172 +1394,183 @@ For additional GBD results and resources, visit the GBD 2019 Data Resources page issn = {0820-3946, 1488-2329}, url = {http://www.cmaj.ca/cgi/doi/10.1503/cmaj.050222}, doi = {10.1503/cmaj.050222}, - abstract = {ECTOPIC PREGNANCY IS A LIFE- AND FERTILITY-threatening condition that is commonly seen in Canadian emergency departments. Increases in the availability and use of hormonal markers, coupled with advances in formal and emergency ultrasonography have changed the diagnostic approach to the patient in the emergency department with first-trimester bleeding or pain. Ultrasonography should be the initial investigation for symptomatic women in their first trimester; when the results are indeterminate, the serum β human chorionic gonadotropin (β-hCG) concentration should be measured. Serial measurement of β-hCG and progesterone concentrations may be useful when the diagnosis remains unclear. Advances in surgical and medical therapy for ectopic pregnancy have allowed the proliferation of minimally invasive or noninvasive treatment. Guidelines for laparoscopy and for methotrexate therapy are provided.}, - language = {en}, - number = {8}, - urldate = {2025-03-17}, - journal = {Canadian Medical Association Journal}, - author = {Murray, H.}, - month = oct, - year = {2005}, + abstract = {{ECTOPIC} {PREGNANCY} {IS} A {LIFE}- {AND} {FERTILITY}-threatening condition that is commonly seen in Canadian emergency departments. Increases in the availability and use of hormonal markers, coupled with advances in formal and emergency ultrasonography have changed the diagnostic approach to the patient in the emergency department with first-trimester bleeding or pain. Ultrasonography should be the initial investigation for symptomatic women in their first trimester; when the results are indeterminate, the serum β human chorionic gonadotropin (β-{hCG}) concentration should be measured. Serial measurement of β-{hCG} and progesterone concentrations may be useful when the diagnosis remains unclear. Advances in surgical and medical therapy for ectopic pregnancy have allowed the proliferation of minimally invasive or noninvasive treatment. Guidelines for laparoscopy and for methotrexate therapy are provided.}, pages = {905--912}, + number = {8}, + journaltitle = {Canadian Medical Association Journal}, + author = {Murray, H.}, + urldate = {2025-03-17}, + date = {2005-10-11}, + langid = {english}, file = {PDF:/home/alex/Zotero/storage/Y73KE57K/Murray - 2005 - Diagnosis and treatment of ectopic pregnancy.pdf:application/pdf}, } -@article{rosenfield_adolescent_2013, - title = {Adolescent {Anovulation}: {Maturational} {Mechanisms} and {Implications}}, - volume = {98}, - issn = {0021-972X, 1945-7197}, - shorttitle = {Adolescent {Anovulation}}, - url = {https://academic.oup.com/jcem/article-lookup/doi/10.1210/jc.2013-1770}, - doi = {10.1210/jc.2013-1770}, - language = {en}, - number = {9}, +@collection{pfannstiel_entrepreneurship_2018, + location = {Wiesbaden}, + title = {Entrepreneurship im Gesundheitswesen {II}}, + rights = {http://www.springer.com/tdm}, + isbn = {978-3-658-14780-8 978-3-658-14781-5}, + url = {http://link.springer.com/10.1007/978-3-658-14781-5}, + publisher = {Springer Fachmedien Wiesbaden}, + editor = {Pfannstiel, Mario A. and Da-Cruz, Patrick and Rasche, Christoph}, urldate = {2025-03-17}, - journal = {The Journal of Clinical Endocrinology \& Metabolism}, - author = {Rosenfield, Robert L.}, - month = sep, - year = {2013}, - pages = {3572--3583}, - file = {PDF:/home/alex/Zotero/storage/BEF7PQT6/Rosenfield - 2013 - Adolescent Anovulation Maturational Mechanisms and Implications.pdf:application/pdf}, + date = {2018}, + langid = {german}, + doi = {10.1007/978-3-658-14781-5}, + file = {PDF:/home/alex/Zotero/storage/DTVM5NBP/Pfannstiel et al. - 2018 - Entrepreneurship im Gesundheitswesen II.pdf:application/pdf}, } -@book{cryer_time_2008, - address = {New York}, - edition = {2nd ed}, - series = {Springer texts in statistics}, - title = {Time series analysis: with applications in {R}}, - isbn = {978-0-387-75958-6 978-0-387-75959-3}, - shorttitle = {Time series analysis}, - language = {en}, - publisher = {Springer}, - author = {Cryer, Jonathan D. and Chan, Kung-sik}, - year = {2008}, - note = {OCLC: ocn191760003}, - keywords = {Data processing, R (Computer program language), Time-series analysis}, - file = {PDF:/home/alex/Zotero/storage/CIYMBUEW/Cryer and Chan - 2008 - Time series analysis with applications in R.pdf:application/pdf}, -} - -@book{hamilton_time_1994, - address = {Princeton (N.J.)}, - title = {Time series analysis}, - isbn = {978-0-691-04289-3}, - language = {eng}, - publisher = {Princeton university press}, - author = {Hamilton, James Douglas}, - year = {1994}, - file = {PDF:/home/alex/Zotero/storage/J8GVKKHG/Hamilton - 1994 - Time series analysis.pdf:application/pdf}, -} - -@misc{noauthor_playtikaosstft-torch_2025, - title = {{PlaytikaOSS}/tft-torch}, - copyright = {MIT}, - url = {https://github.com/PlaytikaOSS/tft-torch}, - abstract = {A Python library that implements ״Temporal Fusion Transformers for Interpretable Multi-horizon Time Series Forecasting״}, - urldate = {2025-03-19}, - publisher = {Playtika}, - month = jan, - year = {2025}, - note = {original-date: 2021-11-28T07:08:32Z}, -} - -@misc{sherar_mattsherartemporal_fusion_transform_2025, - title = {mattsherar/{Temporal}\_Fusion\_Transform}, - url = {https://github.com/mattsherar/Temporal_Fusion_Transform}, - abstract = {Pytorch Implementation of Google's TFT}, - urldate = {2025-03-19}, - author = {Sherar, Matthew}, - month = mar, - year = {2025}, - note = {original-date: 2020-01-11T17:54:01Z}, -} - -@misc{noauthor_temporal_nodate, - title = {Temporal {Fusion} {Transformer} ({TFT}) — darts documentation}, +@online{noauthor_temporal_nodate, + title = {Temporal Fusion Transformer ({TFT}) — darts documentation}, url = {https://unit8co.github.io/darts/generated_api/darts.models.forecasting.tft_model.html}, urldate = {2025-03-19}, file = {Temporal Fusion Transformer (TFT) — darts documentation:/home/alex/Zotero/storage/5QNI6WSL/darts.models.forecasting.tft_model.html:text/html}, } -@misc{dauphin_language_2017, - title = {Language {Modeling} with {Gated} {Convolutional} {Networks}}, - url = {http://arxiv.org/abs/1612.08083}, - doi = {10.48550/arXiv.1612.08083}, - abstract = {The pre-dominant approach to language modeling to date is based on recurrent neural networks. Their success on this task is often linked to their ability to capture unbounded context. In this paper we develop a finite context approach through stacked convolutions, which can be more efficient since they allow parallelization over sequential tokens. We propose a novel simplified gating mechanism that outperforms Oord et al (2016) and investigate the impact of key architectural decisions. The proposed approach achieves state-of-the-art on the WikiText-103 benchmark, even though it features long-term dependencies, as well as competitive results on the Google Billion Words benchmark. Our model reduces the latency to score a sentence by an order of magnitude compared to a recurrent baseline. To our knowledge, this is the first time a non-recurrent approach is competitive with strong recurrent models on these large scale language tasks.}, +@software{sherar_mattsherartemporal_fusion_transform_2025, + title = {mattsherar/Temporal\_Fusion\_Transform}, + url = {https://github.com/mattsherar/Temporal_Fusion_Transform}, + abstract = {Pytorch Implementation of Google's {TFT}}, + author = {Sherar, Matthew}, + urldate = {2025-03-19}, + date = {2025-03-07}, + note = {original-date: 2020-01-11T17:54:01Z}, +} + +@software{noauthor_playtikaosstft-torch_2025, + title = {{PlaytikaOSS}/tft-torch}, + rights = {{MIT}}, + url = {https://github.com/PlaytikaOSS/tft-torch}, + abstract = {A Python library that implements ״Temporal Fusion Transformers for Interpretable Multi-horizon Time Series Forecasting״}, + publisher = {Playtika}, + urldate = {2025-03-19}, + date = {2025-01-27}, + note = {original-date: 2021-11-28T07:08:32Z}, +} + +@online{noauthor_create_nodate, + title = {Create baseline model - {ValueError}: too many values to unpack (expected 2) · Issue \#230 · sktime/pytorch-forecasting}, + url = {https://github.com/sktime/pytorch-forecasting/issues/230}, + shorttitle = {Create baseline model - {ValueError}}, + abstract = {{PyTorch}-Forecasting version: 0.7.1 {PyTorch} version: 1.7.1 Python version: 3.7 Operating System: {MAC} {OS} Big Sur: Version 11.1 Expected behavior I executed code actuals = torch.cat([y for x, (y, weig...}, + titleaddon = {{GitHub}}, + urldate = {2025-03-19}, + langid = {english}, +} + +@misc{clevert_fast_2016, + title = {Fast and Accurate Deep Network Learning by Exponential Linear Units ({ELUs})}, + url = {http://arxiv.org/abs/1511.07289}, + doi = {10.48550/arXiv.1511.07289}, + abstract = {We introduce the "exponential linear unit" ({ELU}) which speeds up learning in deep neural networks and leads to higher classification accuracies. Like rectified linear units ({ReLUs}), leaky {ReLUs} ({LReLUs}) and parametrized {ReLUs} ({PReLUs}), {ELUs} alleviate the vanishing gradient problem via the identity for positive values. However, {ELUs} have improved learning characteristics compared to the units with other activation functions. In contrast to {ReLUs}, {ELUs} have negative values which allows them to push mean unit activations closer to zero like batch normalization but with lower computational complexity. Mean shifts toward zero speed up learning by bringing the normal gradient closer to the unit natural gradient because of a reduced bias shift effect. While {LReLUs} and {PReLUs} have negative values, too, they do not ensure a noise-robust deactivation state. {ELUs} saturate to a negative value with smaller inputs and thereby decrease the forward propagated variation and information. Therefore, {ELUs} code the degree of presence of particular phenomena in the input, while they do not quantitatively model the degree of their absence. In experiments, {ELUs} lead not only to faster learning, but also to significantly better generalization performance than {ReLUs} and {LReLUs} on networks with more than 5 layers. On {CIFAR}-100 {ELUs} networks significantly outperform {ReLU} networks with batch normalization while batch normalization does not improve {ELU} networks. {ELU} networks are among the top 10 reported {CIFAR}-10 results and yield the best published result on {CIFAR}-100, without resorting to multi-view evaluation or model averaging. On {ImageNet}, {ELU} networks considerably speed up learning compared to a {ReLU} network with the same architecture, obtaining less than 10\% classification error for a single crop, single model network.}, + number = {{arXiv}:1511.07289}, + publisher = {{arXiv}}, + author = {Clevert, Djork-Arné and Unterthiner, Thomas and Hochreiter, Sepp}, urldate = {2025-03-26}, - publisher = {arXiv}, - author = {Dauphin, Yann N. and Fan, Angela and Auli, Michael and Grangier, David}, - month = sep, - year = {2017}, - note = {arXiv:1612.08083 [cs]}, - keywords = {Computer Science - Computation and Language}, - file = {Full Text PDF:/home/alex/Zotero/storage/4SBUNZ4A/Dauphin et al. - 2017 - Language Modeling with Gated Convolutional Networks.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/TQBL4EZ7/1612.html:text/html}, + date = {2016-02-22}, + eprinttype = {arxiv}, + eprint = {1511.07289 [cs]}, + keywords = {Computer Science - Machine Learning}, + file = {Full Text PDF:/home/alex/Zotero/storage/PU3ZGP4G/Clevert et al. - 2016 - Fast and Accurate Deep Network Learning by Exponential Linear Units (ELUs).pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/3MAW2IWE/1511.html:text/html}, } @misc{ba_layer_2016, - title = {Layer {Normalization}}, + title = {Layer Normalization}, url = {http://arxiv.org/abs/1607.06450}, doi = {10.48550/arXiv.1607.06450}, abstract = {Training state-of-the-art, deep neural networks is computationally expensive. One way to reduce the training time is to normalize the activities of the neurons. A recently introduced technique called batch normalization uses the distribution of the summed input to a neuron over a mini-batch of training cases to compute a mean and variance which are then used to normalize the summed input to that neuron on each training case. This significantly reduces the training time in feed-forward neural networks. However, the effect of batch normalization is dependent on the mini-batch size and it is not obvious how to apply it to recurrent neural networks. In this paper, we transpose batch normalization into layer normalization by computing the mean and variance used for normalization from all of the summed inputs to the neurons in a layer on a single training case. Like batch normalization, we also give each neuron its own adaptive bias and gain which are applied after the normalization but before the non-linearity. Unlike batch normalization, layer normalization performs exactly the same computation at training and test times. It is also straightforward to apply to recurrent neural networks by computing the normalization statistics separately at each time step. Layer normalization is very effective at stabilizing the hidden state dynamics in recurrent networks. Empirically, we show that layer normalization can substantially reduce the training time compared with previously published techniques.}, - urldate = {2025-03-26}, - publisher = {arXiv}, + number = {{arXiv}:1607.06450}, + publisher = {{arXiv}}, author = {Ba, Jimmy Lei and Kiros, Jamie Ryan and Hinton, Geoffrey E.}, - month = jul, - year = {2016}, - note = {arXiv:1607.06450 [stat]}, + urldate = {2025-03-26}, + date = {2016-07-21}, + eprinttype = {arxiv}, + eprint = {1607.06450 [stat]}, keywords = {Computer Science - Machine Learning, Statistics - Machine Learning}, file = {Full Text PDF:/home/alex/Zotero/storage/MJWRDPWE/Ba et al. - 2016 - Layer Normalization.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/F9WSU957/1607.html:text/html}, } -@misc{clevert_fast_2016, - title = {Fast and {Accurate} {Deep} {Network} {Learning} by {Exponential} {Linear} {Units} ({ELUs})}, - url = {http://arxiv.org/abs/1511.07289}, - doi = {10.48550/arXiv.1511.07289}, - abstract = {We introduce the "exponential linear unit" (ELU) which speeds up learning in deep neural networks and leads to higher classification accuracies. Like rectified linear units (ReLUs), leaky ReLUs (LReLUs) and parametrized ReLUs (PReLUs), ELUs alleviate the vanishing gradient problem via the identity for positive values. However, ELUs have improved learning characteristics compared to the units with other activation functions. In contrast to ReLUs, ELUs have negative values which allows them to push mean unit activations closer to zero like batch normalization but with lower computational complexity. Mean shifts toward zero speed up learning by bringing the normal gradient closer to the unit natural gradient because of a reduced bias shift effect. While LReLUs and PReLUs have negative values, too, they do not ensure a noise-robust deactivation state. ELUs saturate to a negative value with smaller inputs and thereby decrease the forward propagated variation and information. Therefore, ELUs code the degree of presence of particular phenomena in the input, while they do not quantitatively model the degree of their absence. In experiments, ELUs lead not only to faster learning, but also to significantly better generalization performance than ReLUs and LReLUs on networks with more than 5 layers. On CIFAR-100 ELUs networks significantly outperform ReLU networks with batch normalization while batch normalization does not improve ELU networks. ELU networks are among the top 10 reported CIFAR-10 results and yield the best published result on CIFAR-100, without resorting to multi-view evaluation or model averaging. On ImageNet, ELU networks considerably speed up learning compared to a ReLU network with the same architecture, obtaining less than 10\% classification error for a single crop, single model network.}, +@misc{dauphin_language_2017, + title = {Language Modeling with Gated Convolutional Networks}, + url = {http://arxiv.org/abs/1612.08083}, + doi = {10.48550/arXiv.1612.08083}, + abstract = {The pre-dominant approach to language modeling to date is based on recurrent neural networks. Their success on this task is often linked to their ability to capture unbounded context. In this paper we develop a finite context approach through stacked convolutions, which can be more efficient since they allow parallelization over sequential tokens. We propose a novel simplified gating mechanism that outperforms Oord et al (2016) and investigate the impact of key architectural decisions. The proposed approach achieves state-of-the-art on the {WikiText}-103 benchmark, even though it features long-term dependencies, as well as competitive results on the Google Billion Words benchmark. Our model reduces the latency to score a sentence by an order of magnitude compared to a recurrent baseline. To our knowledge, this is the first time a non-recurrent approach is competitive with strong recurrent models on these large scale language tasks.}, + number = {{arXiv}:1612.08083}, + publisher = {{arXiv}}, + author = {Dauphin, Yann N. and Fan, Angela and Auli, Michael and Grangier, David}, urldate = {2025-03-26}, - publisher = {arXiv}, - author = {Clevert, Djork-Arné and Unterthiner, Thomas and Hochreiter, Sepp}, - month = feb, - year = {2016}, - note = {arXiv:1511.07289 [cs]}, - keywords = {Computer Science - Machine Learning}, - annote = {Comment: Published as a conference paper at ICLR 2016}, - file = {Full Text PDF:/home/alex/Zotero/storage/PU3ZGP4G/Clevert et al. - 2016 - Fast and Accurate Deep Network Learning by Exponential Linear Units (ELUs).pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/3MAW2IWE/1511.html:text/html}, + date = {2017-09-08}, + eprinttype = {arxiv}, + eprint = {1612.08083 [cs]}, + keywords = {Computer Science - Computation and Language}, + file = {Full Text PDF:/home/alex/Zotero/storage/4SBUNZ4A/Dauphin et al. - 2017 - Language Modeling with Gated Convolutional Networks.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/TQBL4EZ7/1612.html:text/html}, +} + +@misc{saluja_towards_2021, + title = {Towards a Rigorous Evaluation of Explainability for Multivariate Time Series}, + url = {http://arxiv.org/abs/2104.04075}, + doi = {10.48550/arXiv.2104.04075}, + abstract = {Machine learning-based systems are rapidly gaining popularity and in-line with that there has been a huge research surge in the field of explainability to ensure that machine learning models are reliable, fair, and can be held liable for their decision-making process. Explainable Artificial Intelligence ({XAI}) methods are typically deployed to debug black-box machine learning models but in comparison to tabular, text, and image data, explainability in time series is still relatively unexplored. The aim of this study was to achieve and evaluate model agnostic explainability in a time series forecasting problem. This work focused on proving a solution for a digital consultancy company aiming to find a data-driven approach in order to understand the effect of their sales related activities on the sales deals closed. The solution involved framing the problem as a time series forecasting problem to predict the sales deals and the explainability was achieved using two novel model agnostic explainability techniques, Local explainable model-agnostic explanations ({LIME}) and Shapley additive explanations ({SHAP}) which were evaluated using human evaluation of explainability. The results clearly indicate that the explanations produced by {LIME} and {SHAP} greatly helped lay humans in understanding the predictions made by the machine learning model. The presented work can easily be extended to any time}, + number = {{arXiv}:2104.04075}, + publisher = {{arXiv}}, + author = {Saluja, Rohit and Malhi, Avleen and Knapič, Samanta and Främling, Kary and Cavdar, Cicek}, + urldate = {2025-05-06}, + date = {2021-04-06}, + eprinttype = {arxiv}, + eprint = {2104.04075 [cs]}, + keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, + file = {Preprint PDF:/home/alex/Zotero/storage/6XB73Y2F/Saluja et al. - 2021 - Towards a Rigorous Evaluation of Explainability for Multivariate Time Series.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/CDCGX8EZ/2104.html:text/html}, +} + +@inproceedings{hsieh_explainable_2021, + location = {Virtual Event Israel}, + title = {Explainable Multivariate Time Series Classification: A Deep Neural Network Which Learns to Attend to Important Variables As Well As Time Intervals}, + isbn = {978-1-4503-8297-7}, + url = {https://dl.acm.org/doi/10.1145/3437963.3441815}, + doi = {10.1145/3437963.3441815}, + shorttitle = {Explainable Multivariate Time Series Classification}, + eventtitle = {{WSDM} '21: The Fourteenth {ACM} International Conference on Web Search and Data Mining}, + pages = {607--615}, + booktitle = {Proceedings of the 14th {ACM} International Conference on Web Search and Data Mining}, + publisher = {{ACM}}, + author = {Hsieh, Tsung-Yu and Wang, Suhang and Sun, Yiwei and Honavar, Vasant}, + urldate = {2025-05-06}, + date = {2021-03-08}, + langid = {english}, +} + +@article{leon-lopez_anomaly_2022, + title = {Anomaly Detection and Classification in Multispectral Time Series Based on Hidden Markov Models}, + volume = {60}, + issn = {1558-0644}, + url = {https://ieeexplore.ieee.org/abstract/document/9509347}, + doi = {10.1109/TGRS.2021.3101127}, + abstract = {Monitoring agriculture from satellite remote sensing data, such as multispectral images, has become a powerful tool since it has demonstrated a great potential for providing timely and accurate knowledge of crops. Detecting anomalies in time series of multispectral remote sensing images for crop monitoring is generally performed using a large sample of historical data at a pixel level. Conversely, this article presents a framework for anomaly detection ({AD}), localization, and classification that exploits the temporal information contained in a given season at a parcel level to detect and localize outliers using hidden Markov models ({HMMs}). Specifically, the {AD} part is based on the learning of {HMM} parameters associated with unlabeled normal data that are used in a second step to detect abnormal crop parcels referred to as anomalies. The learned {HMM} can also be used in time segments to temporally localize the anomalies affecting the crop parcels. The detected and localized anomalies are finally classified using a supervised classifier, e.g., based on support vector machines. The proposed framework is applicable to images partially covered by clouds and can handle a set of crop parcels acquired in the same season bypassing problems due to crop rotations. Numerical experiments are conducted on synthetic and real data, where the real data correspond to vegetation indices extracted from several multitemporal Sentinel-2 images of rapeseed crops. The proposed approach is compared to standard {AD} methods yielding better detection rates with the advantage of allowing anomalies to be localized and characterized.}, + pages = {1--11}, + journaltitle = {{IEEE} Transactions on Geoscience and Remote Sensing}, + author = {León-López, Kareth M. and Mouret, Florian and Arguello, Henry and Tourneret, Jean-Yves}, + urldate = {2025-05-06}, + date = {2022}, + keywords = {Time series analysis, Hidden Markov models, Agricultural monitoring, Agriculture, anomaly classification, Anomaly detection, anomaly detection ({AD}), Feature extraction, hidden Markov models ({HMMs}), Monitoring, remote sensing, time series, Vegetation mapping}, + file = {Snapshot:/home/alex/Zotero/storage/HIWT8MDH/9509347.html:text/html;Submitted Version:/home/alex/Zotero/storage/S4K2XXHN/León-López et al. - 2022 - Anomaly Detection and Classification in Multispectral Time Series Based on Hidden Markov Models.pdf:application/pdf}, } @article{wang_systematic_2022, - title = {A {Systematic} {Review} of {Time} {Series} {Classification} {Techniques} {Used} in {Biomedical} {Applications}}, + title = {A Systematic Review of Time Series Classification Techniques Used in Biomedical Applications}, volume = {22}, + rights = {https://creativecommons.org/licenses/by/4.0/}, issn = {1424-8220}, - url = {https://www.ncbi.nlm.nih.gov/pmc/articles/PMC9611376/}, + url = {https://www.mdpi.com/1424-8220/22/20/8016}, doi = {10.3390/s22208016}, - abstract = {Background: Digital clinical measures collected via various digital sensing technologies such as smartphones, smartwatches, wearables, and ingestible and implantable sensors are increasingly used by individuals and clinicians to capture the health outcomes or behavioral and physiological characteristics of individuals. Time series classification (TSC) is very commonly used for modeling digital clinical measures. While deep learning models for TSC are very common and powerful, there exist some fundamental challenges. This review presents the non-deep learning models that are commonly used for time series classification in biomedical applications that can achieve high performance. Objective: We performed a systematic review to characterize the techniques that are used in time series classification of digital clinical measures throughout all the stages of data processing and model building. Methods: We conducted a literature search on PubMed, as well as the Institute of Electrical and Electronics Engineers (IEEE), Web of Science, and SCOPUS databases using a range of search terms to retrieve peer-reviewed articles that report on the academic research about digital clinical measures from a five-year period between June 2016 and June 2021. We identified and categorized the research studies based on the types of classification algorithms and sensor input types. Results: We found 452 papers in total from four different databases: PubMed, IEEE, Web of Science Database, and SCOPUS. After removing duplicates and irrelevant papers, 135 articles remained for detailed review and data extraction. Among these, engineered features using time series methods that were subsequently fed into widely used machine learning classifiers were the most commonly used technique, and also most frequently achieved the best performance metrics (77 out of 135 articles). Statistical modeling (24 out of 135 articles) algorithms were the second most common and also the second-best classification technique. Conclusions: In this review paper, summaries of the time series classification models and interpretation methods for biomedical applications are summarized and categorized. While high time series classification performance has been achieved in digital clinical, physiological, or biomedical measures, no standard benchmark datasets, modeling methods, or reporting methodology exist. There is no single widely used method for time series model development or feature interpretation, however many different methods have proven successful.}, - number = {20}, - urldate = {2025-05-06}, - journal = {Sensors (Basel, Switzerland)}, - author = {Wang, Will Ke and Chen, Ina and Hershkovich, Leeor and Yang, Jiamu and Shetty, Ayush and Singh, Geetika and Jiang, Yihang and Kotla, Aditya and Shang, Jason Zisheng and Yerrabelli, Rushil and Roghanizad, Ali R. and Shandhi, Md Mobashir Hasan and Dunn, Jessilyn}, - month = oct, - year = {2022}, - pmid = {36298367}, - pmcid = {PMC9611376}, + abstract = {Background: Digital clinical measures collected via various digital sensing technologies such as smartphones, smartwatches, wearables, and ingestible and implantable sensors are increasingly used by individuals and clinicians to capture the health outcomes or behavioral and physiological characteristics of individuals. Time series classification ({TSC}) is very commonly used for modeling digital clinical measures. While deep learning models for {TSC} are very common and powerful, there exist some fundamental challenges. This review presents the non-deep learning models that are commonly used for time series classification in biomedical applications that can achieve high performance. Objective: We performed a systematic review to characterize the techniques that are used in time series classification of digital clinical measures throughout all the stages of data processing and model building. Methods: We conducted a literature search on {PubMed}, as well as the Institute of Electrical and Electronics Engineers ({IEEE}), Web of Science, and {SCOPUS} databases using a range of search terms to retrieve peer-reviewed articles that report on the academic research about digital clinical measures from a five-year period between June 2016 and June 2021. We identified and categorized the research studies based on the types of classification algorithms and sensor input types. Results: We found 452 papers in total from four different databases: {PubMed}, {IEEE}, Web of Science Database, and {SCOPUS}. After removing duplicates and irrelevant papers, 135 articles remained for detailed review and data extraction. Among these, engineered features using time series methods that were subsequently fed into widely used machine learning classifiers were the most commonly used technique, and also most frequently achieved the best performance metrics (77 out of 135 articles). Statistical modeling (24 out of 135 articles) algorithms were the second most common and also the second-best classification technique. Conclusions: In this review paper, summaries of the time series classification models and interpretation methods for biomedical applications are summarized and categorized. While high time series classification performance has been achieved in digital clinical, physiological, or biomedical measures, no standard benchmark datasets, modeling methods, or reporting methodology exist. There is no single widely used method for time series model development or feature interpretation, however many different methods have proven successful.}, pages = {8016}, - file = {Full Text PDF:/home/alex/Zotero/storage/2E8EIVER/Wang et al. - 2022 - A Systematic Review of Time Series Classification Techniques Used in Biomedical Applications.pdf:application/pdf}, -} - -@article{gharehbaghi_deep_2018, - title = {A {Deep} {Machine} {Learning} {Method} for {Classifying} {Cyclic} {Time} {Series} of {Biological} {Signals} {Using} {Time}-{Growing} {Neural} {Network}}, - volume = {29}, - copyright = {https://ieeexplore.ieee.org/Xplorehelp/downloads/license-information/IEEE.html}, - issn = {2162-237X, 2162-2388}, - url = {https://ieeexplore.ieee.org/document/8066455/}, - doi = {10.1109/TNNLS.2017.2754294}, - number = {9}, + number = {20}, + journaltitle = {Sensors}, + author = {Wang, Will Ke and Chen, Ina and Hershkovich, Leeor and Yang, Jiamu and Shetty, Ayush and Singh, Geetika and Jiang, Yihang and Kotla, Aditya and Shang, Jason Zisheng and Yerrabelli, Rushil and Roghanizad, Ali R. and Shandhi, Md Mobashir Hasan and Dunn, Jessilyn}, urldate = {2025-05-06}, - journal = {IEEE Transactions on Neural Networks and Learning Systems}, - author = {Gharehbaghi, Arash and Linden, Maria}, - month = sep, - year = {2018}, - pages = {4102--4115}, + date = {2022-10-20}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/H5LLUB5K/Wang et al. - 2022 - A Systematic Review of Time Series Classification Techniques Used in Biomedical Applications.pdf:application/pdf}, } @article{masini_machine_2023, @@ -1671,275 +1580,263 @@ For additional GBD results and resources, visit the GBD 2019 Data Resources page url = {https://onlinelibrary.wiley.com/doi/10.1111/joes.12429}, doi = {10.1111/joes.12429}, abstract = {Abstract - In this paper, we survey the most recent advances in supervised machine learning (ML) and high‐dimensional models for time‐series forecasting. We consider both linear and nonlinear alternatives. Among the linear methods, we pay special attention to penalized regressions and ensemble of models. The nonlinear methods considered in the paper include shallow and deep neural networks, in their feedforward and recurrent versions, and tree‐based methods, such as random forests and boosted trees. We also consider ensemble and hybrid models by combining ingredients from different alternatives. Tests for superior predictive ability are briefly reviewed. Finally, we discuss application of ML in economics and finance and provide an illustration with high‐frequency financial data.}, - language = {en}, - number = {1}, - urldate = {2025-05-06}, - journal = {Journal of Economic Surveys}, - author = {Masini, Ricardo P. and Medeiros, Marcelo C. and Mendes, Eduardo F.}, - month = feb, - year = {2023}, + In this paper, we survey the most recent advances in supervised machine learning ({ML}) and high‐dimensional models for time‐series forecasting. We consider both linear and nonlinear alternatives. Among the linear methods, we pay special attention to penalized regressions and ensemble of models. The nonlinear methods considered in the paper include shallow and deep neural networks, in their feedforward and recurrent versions, and tree‐based methods, such as random forests and boosted trees. We also consider ensemble and hybrid models by combining ingredients from different alternatives. Tests for superior predictive ability are briefly reviewed. Finally, we discuss application of {ML} in economics and finance and provide an illustration with high‐frequency financial data.}, pages = {76--111}, + number = {1}, + journaltitle = {Journal of Economic Surveys}, + author = {Masini, Ricardo P. and Medeiros, Marcelo C. and Mendes, Eduardo F.}, + urldate = {2025-05-06}, + date = {2023-02}, + langid = {english}, file = {Submitted Version:/home/alex/Zotero/storage/4TZJJUSU/Masini et al. - 2023 - Machine learning advances for time series forecasting.pdf:application/pdf}, } +@article{gharehbaghi_deep_2018, + title = {A Deep Machine Learning Method for Classifying Cyclic Time Series of Biological Signals Using Time-Growing Neural Network}, + volume = {29}, + rights = {https://ieeexplore.ieee.org/Xplorehelp/downloads/license-information/{IEEE}.html}, + issn = {2162-237X, 2162-2388}, + url = {https://ieeexplore.ieee.org/document/8066455/}, + doi = {10.1109/TNNLS.2017.2754294}, + pages = {4102--4115}, + number = {9}, + journaltitle = {{IEEE} Trans. Neural Netw. Learning Syst.}, + author = {Gharehbaghi, Arash and Linden, Maria}, + urldate = {2025-05-06}, + date = {2018-09}, +} + @article{wang_systematic_2022-1, - title = {A {Systematic} {Review} of {Time} {Series} {Classification} {Techniques} {Used} in {Biomedical} {Applications}}, + title = {A Systematic Review of Time Series Classification Techniques Used in Biomedical Applications}, volume = {22}, - copyright = {https://creativecommons.org/licenses/by/4.0/}, issn = {1424-8220}, - url = {https://www.mdpi.com/1424-8220/22/20/8016}, + url = {https://www.ncbi.nlm.nih.gov/pmc/articles/PMC9611376/}, doi = {10.3390/s22208016}, - abstract = {Background: Digital clinical measures collected via various digital sensing technologies such as smartphones, smartwatches, wearables, and ingestible and implantable sensors are increasingly used by individuals and clinicians to capture the health outcomes or behavioral and physiological characteristics of individuals. Time series classification (TSC) is very commonly used for modeling digital clinical measures. While deep learning models for TSC are very common and powerful, there exist some fundamental challenges. This review presents the non-deep learning models that are commonly used for time series classification in biomedical applications that can achieve high performance. Objective: We performed a systematic review to characterize the techniques that are used in time series classification of digital clinical measures throughout all the stages of data processing and model building. Methods: We conducted a literature search on PubMed, as well as the Institute of Electrical and Electronics Engineers (IEEE), Web of Science, and SCOPUS databases using a range of search terms to retrieve peer-reviewed articles that report on the academic research about digital clinical measures from a five-year period between June 2016 and June 2021. We identified and categorized the research studies based on the types of classification algorithms and sensor input types. Results: We found 452 papers in total from four different databases: PubMed, IEEE, Web of Science Database, and SCOPUS. After removing duplicates and irrelevant papers, 135 articles remained for detailed review and data extraction. Among these, engineered features using time series methods that were subsequently fed into widely used machine learning classifiers were the most commonly used technique, and also most frequently achieved the best performance metrics (77 out of 135 articles). Statistical modeling (24 out of 135 articles) algorithms were the second most common and also the second-best classification technique. Conclusions: In this review paper, summaries of the time series classification models and interpretation methods for biomedical applications are summarized and categorized. While high time series classification performance has been achieved in digital clinical, physiological, or biomedical measures, no standard benchmark datasets, modeling methods, or reporting methodology exist. There is no single widely used method for time series model development or feature interpretation, however many different methods have proven successful.}, - language = {en}, - number = {20}, - urldate = {2025-05-06}, - journal = {Sensors}, - author = {Wang, Will Ke and Chen, Ina and Hershkovich, Leeor and Yang, Jiamu and Shetty, Ayush and Singh, Geetika and Jiang, Yihang and Kotla, Aditya and Shang, Jason Zisheng and Yerrabelli, Rushil and Roghanizad, Ali R. and Shandhi, Md Mobashir Hasan and Dunn, Jessilyn}, - month = oct, - year = {2022}, + abstract = {Background: Digital clinical measures collected via various digital sensing technologies such as smartphones, smartwatches, wearables, and ingestible and implantable sensors are increasingly used by individuals and clinicians to capture the health outcomes or behavioral and physiological characteristics of individuals. Time series classification ({TSC}) is very commonly used for modeling digital clinical measures. While deep learning models for {TSC} are very common and powerful, there exist some fundamental challenges. This review presents the non-deep learning models that are commonly used for time series classification in biomedical applications that can achieve high performance. Objective: We performed a systematic review to characterize the techniques that are used in time series classification of digital clinical measures throughout all the stages of data processing and model building. Methods: We conducted a literature search on {PubMed}, as well as the Institute of Electrical and Electronics Engineers ({IEEE}), Web of Science, and {SCOPUS} databases using a range of search terms to retrieve peer-reviewed articles that report on the academic research about digital clinical measures from a five-year period between June 2016 and June 2021. We identified and categorized the research studies based on the types of classification algorithms and sensor input types. Results: We found 452 papers in total from four different databases: {PubMed}, {IEEE}, Web of Science Database, and {SCOPUS}. After removing duplicates and irrelevant papers, 135 articles remained for detailed review and data extraction. Among these, engineered features using time series methods that were subsequently fed into widely used machine learning classifiers were the most commonly used technique, and also most frequently achieved the best performance metrics (77 out of 135 articles). Statistical modeling (24 out of 135 articles) algorithms were the second most common and also the second-best classification technique. Conclusions: In this review paper, summaries of the time series classification models and interpretation methods for biomedical applications are summarized and categorized. While high time series classification performance has been achieved in digital clinical, physiological, or biomedical measures, no standard benchmark datasets, modeling methods, or reporting methodology exist. There is no single widely used method for time series model development or feature interpretation, however many different methods have proven successful.}, pages = {8016}, - file = {PDF:/home/alex/Zotero/storage/H5LLUB5K/Wang et al. - 2022 - A Systematic Review of Time Series Classification Techniques Used in Biomedical Applications.pdf:application/pdf}, -} - -@article{leon-lopez_anomaly_2022, - title = {Anomaly {Detection} and {Classification} in {Multispectral} {Time} {Series} {Based} on {Hidden} {Markov} {Models}}, - volume = {60}, - issn = {1558-0644}, - url = {https://ieeexplore.ieee.org/abstract/document/9509347}, - doi = {10.1109/TGRS.2021.3101127}, - abstract = {Monitoring agriculture from satellite remote sensing data, such as multispectral images, has become a powerful tool since it has demonstrated a great potential for providing timely and accurate knowledge of crops. Detecting anomalies in time series of multispectral remote sensing images for crop monitoring is generally performed using a large sample of historical data at a pixel level. Conversely, this article presents a framework for anomaly detection (AD), localization, and classification that exploits the temporal information contained in a given season at a parcel level to detect and localize outliers using hidden Markov models (HMMs). Specifically, the AD part is based on the learning of HMM parameters associated with unlabeled normal data that are used in a second step to detect abnormal crop parcels referred to as anomalies. The learned HMM can also be used in time segments to temporally localize the anomalies affecting the crop parcels. The detected and localized anomalies are finally classified using a supervised classifier, e.g., based on support vector machines. The proposed framework is applicable to images partially covered by clouds and can handle a set of crop parcels acquired in the same season bypassing problems due to crop rotations. Numerical experiments are conducted on synthetic and real data, where the real data correspond to vegetation indices extracted from several multitemporal Sentinel-2 images of rapeseed crops. The proposed approach is compared to standard AD methods yielding better detection rates with the advantage of allowing anomalies to be localized and characterized.}, + number = {20}, + journaltitle = {Sensors (Basel)}, + author = {Wang, Will Ke and Chen, Ina and Hershkovich, Leeor and Yang, Jiamu and Shetty, Ayush and Singh, Geetika and Jiang, Yihang and Kotla, Aditya and Shang, Jason Zisheng and Yerrabelli, Rushil and Roghanizad, Ali R. and Shandhi, Md Mobashir Hasan and Dunn, Jessilyn}, urldate = {2025-05-06}, - journal = {IEEE Transactions on Geoscience and Remote Sensing}, - author = {León-López, Kareth M. and Mouret, Florian and Arguello, Henry and Tourneret, Jean-Yves}, - year = {2022}, - keywords = {Hidden Markov models, Time series analysis, Agricultural monitoring, Agriculture, anomaly classification, Anomaly detection, anomaly detection (AD), Feature extraction, hidden Markov models (HMMs), Monitoring, remote sensing, time series, Vegetation mapping}, - pages = {1--11}, - file = {Snapshot:/home/alex/Zotero/storage/HIWT8MDH/9509347.html:text/html;Submitted Version:/home/alex/Zotero/storage/S4K2XXHN/León-López et al. - 2022 - Anomaly Detection and Classification in Multispectral Time Series Based on Hidden Markov Models.pdf:application/pdf}, -} - -@inproceedings{hsieh_explainable_2021, - address = {Virtual Event Israel}, - title = {Explainable {Multivariate} {Time} {Series} {Classification}: {A} {Deep} {Neural} {Network} {Which} {Learns} to {Attend} to {Important} {Variables} {As} {Well} {As} {Time} {Intervals}}, - isbn = {978-1-4503-8297-7}, - shorttitle = {Explainable {Multivariate} {Time} {Series} {Classification}}, - url = {https://dl.acm.org/doi/10.1145/3437963.3441815}, - doi = {10.1145/3437963.3441815}, - language = {en}, - urldate = {2025-05-06}, - booktitle = {Proceedings of the 14th {ACM} {International} {Conference} on {Web} {Search} and {Data} {Mining}}, - publisher = {ACM}, - author = {Hsieh, Tsung-Yu and Wang, Suhang and Sun, Yiwei and Honavar, Vasant}, - month = mar, - year = {2021}, - pages = {607--615}, -} - -@misc{saluja_towards_2021, - title = {Towards a {Rigorous} {Evaluation} of {Explainability} for {Multivariate} {Time} {Series}}, - url = {http://arxiv.org/abs/2104.04075}, - doi = {10.48550/arXiv.2104.04075}, - abstract = {Machine learning-based systems are rapidly gaining popularity and in-line with that there has been a huge research surge in the field of explainability to ensure that machine learning models are reliable, fair, and can be held liable for their decision-making process. Explainable Artificial Intelligence (XAI) methods are typically deployed to debug black-box machine learning models but in comparison to tabular, text, and image data, explainability in time series is still relatively unexplored. The aim of this study was to achieve and evaluate model agnostic explainability in a time series forecasting problem. This work focused on proving a solution for a digital consultancy company aiming to find a data-driven approach in order to understand the effect of their sales related activities on the sales deals closed. The solution involved framing the problem as a time series forecasting problem to predict the sales deals and the explainability was achieved using two novel model agnostic explainability techniques, Local explainable model-agnostic explanations (LIME) and Shapley additive explanations (SHAP) which were evaluated using human evaluation of explainability. The results clearly indicate that the explanations produced by LIME and SHAP greatly helped lay humans in understanding the predictions made by the machine learning model. The presented work can easily be extended to any time}, - urldate = {2025-05-06}, - publisher = {arXiv}, - author = {Saluja, Rohit and Malhi, Avleen and Knapič, Samanta and Främling, Kary and Cavdar, Cicek}, - month = apr, - year = {2021}, - note = {arXiv:2104.04075 [cs]}, - keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, - annote = {Comment: Journal}, - file = {Preprint PDF:/home/alex/Zotero/storage/6XB73Y2F/Saluja et al. - 2021 - Towards a Rigorous Evaluation of Explainability for Multivariate Time Series.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/CDCGX8EZ/2104.html:text/html}, -} - -@misc{wang_timexer_2024, - title = {{TimeXer}: {Empowering} {Transformers} for {Time} {Series} {Forecasting} with {Exogenous} {Variables}}, - shorttitle = {{TimeXer}}, - url = {http://arxiv.org/abs/2402.19072}, - doi = {10.48550/arXiv.2402.19072}, - abstract = {Deep models have demonstrated remarkable performance in time series forecasting. However, due to the partially-observed nature of real-world applications, solely focusing on the target of interest, so-called endogenous variables, is usually insufficient to guarantee accurate forecasting. Notably, a system is often recorded into multiple variables, where the exogenous variables can provide valuable external information for endogenous variables. Thus, unlike well-established multivariate or univariate forecasting paradigms that either treat all the variables equally or ignore exogenous information, this paper focuses on a more practical setting: time series forecasting with exogenous variables. We propose a novel approach, TimeXer, to ingest external information to enhance the forecasting of endogenous variables. With deftly designed embedding layers, TimeXer empowers the canonical Transformer with the ability to reconcile endogenous and exogenous information, where patch-wise self-attention and variate-wise cross-attention are used simultaneously. Moreover, global endogenous tokens are learned to effectively bridge the causal information underlying exogenous series into endogenous temporal patches. Experimentally, TimeXer achieves consistent state-of-the-art performance on twelve real-world forecasting benchmarks and exhibits notable generality and scalability. Code is available at this repository: https://github.com/thuml/TimeXer.}, - urldate = {2025-05-09}, - publisher = {arXiv}, - author = {Wang, Yuxuan and Wu, Haixu and Dong, Jiaxiang and Qin, Guo and Zhang, Haoran and Liu, Yong and Qiu, Yunzhong and Wang, Jianmin and Long, Mingsheng}, - month = nov, - year = {2024}, - note = {arXiv:2402.19072 [cs]}, - keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, - file = {Full Text PDF:/home/alex/Zotero/storage/76BQWVIW/Wang et al. - 2024 - TimeXer Empowering Transformers for Time Series Forecasting with Exogenous Variables.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/JL64E9YL/2402.html:text/html}, -} - -@misc{zeng_are_2022, - title = {Are {Transformers} {Effective} for {Time} {Series} {Forecasting}?}, - url = {http://arxiv.org/abs/2205.13504}, - doi = {10.48550/arXiv.2205.13504}, - abstract = {Recently, there has been a surge of Transformer-based solutions for the long-term time series forecasting (LTSF) task. Despite the growing performance over the past few years, we question the validity of this line of research in this work. Specifically, Transformers is arguably the most successful solution to extract the semantic correlations among the elements in a long sequence. However, in time series modeling, we are to extract the temporal relations in an ordered set of continuous points. While employing positional encoding and using tokens to embed sub-series in Transformers facilitate preserving some ordering information, the nature of the {\textbackslash}emph\{permutation-invariant\} self-attention mechanism inevitably results in temporal information loss. To validate our claim, we introduce a set of embarrassingly simple one-layer linear models named LTSF-Linear for comparison. Experimental results on nine real-life datasets show that LTSF-Linear surprisingly outperforms existing sophisticated Transformer-based LTSF models in all cases, and often by a large margin. Moreover, we conduct comprehensive empirical studies to explore the impacts of various design elements of LTSF models on their temporal relation extraction capability. We hope this surprising finding opens up new research directions for the LTSF task. We also advocate revisiting the validity of Transformer-based solutions for other time series analysis tasks (e.g., anomaly detection) in the future. Code is available at: {\textbackslash}url\{https://github.com/cure-lab/LTSF-Linear\}.}, - urldate = {2025-05-09}, - publisher = {arXiv}, - author = {Zeng, Ailing and Chen, Muxi and Zhang, Lei and Xu, Qiang}, - month = aug, - year = {2022}, - note = {arXiv:2205.13504 [cs]}, - keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, - annote = {Comment: Code is available at https://github.com/cure-lab/LTSF-Linear}, - file = {Full Text PDF:/home/alex/Zotero/storage/V9E95F7E/Zeng et al. - 2022 - Are Transformers Effective for Time Series Forecasting.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/HNVIZG98/2205.html:text/html}, -} - -@article{barrera-animas_rainfall_2022, - title = {Rainfall prediction: {A} comparative analysis of modern machine learning algorithms for time-series forecasting}, - volume = {7}, - issn = {26668270}, - shorttitle = {Rainfall prediction}, - url = {https://linkinghub.elsevier.com/retrieve/pii/S266682702100102X}, - doi = {10.1016/j.mlwa.2021.100204}, - abstract = {Rainfall forecasting has gained utmost research relevance in recent times due to its complexities and persistent applications such as flood forecasting and monitoring of pollutant concentration levels, among others. Existing models use complex statistical models that are often too costly, both computationally and budgetary, or are not applied to downstream applications. Therefore, approaches that use Machine Learning algorithms in conjunction with time-series data are being explored as an alternative to overcome these drawbacks. To this end, this study presents a comparative analysis using simplified rainfall estimation models based on conventional Machine Learning algorithms and Deep Learning architectures that are efficient for these downstream applications. Models based on LSTM, Stacked-LSTM, Bidirectional-LSTM Networks, XGBoost, and an ensemble of Gradient Boosting Regressor, Linear Support Vector Regression, and an Extra-trees Regressor were compared in the task of forecasting hourly rainfall volumes using time-series data. Climate data from 2000 to 2020 from five major cities in the United Kingdom were used. The evaluation metrics of Loss, Root Mean Squared Error, Mean Absolute Error, and Root Mean Squared Logarithmic Error were used to evaluate the models’ performance. Results show that a Bidirectional-LSTM Network can be used as a rainfall forecast model with comparable performance to Stacked-LSTM Networks. Among all the models tested, the StackedLSTM Network with two hidden layers and the Bidirectional-LSTM Network performed best. This suggests that models based on LSTM-Networks with fewer hidden layers perform better for this approach; denoting its ability to be applied as an approach for budget-wise rainfall forecast applications.}, - language = {en}, - urldate = {2025-05-08}, - journal = {Machine Learning with Applications}, - author = {Barrera-Animas, Ari Yair and Oyedele, Lukumon O. and Bilal, Muhammad and Akinosho, Taofeek Dolapo and Delgado, Juan Manuel Davila and Akanbi, Lukman Adewale}, - month = mar, - year = {2022}, - pages = {100204}, - file = {PDF:/home/alex/Zotero/storage/FSXMH5N9/Barrera-Animas et al. - 2022 - Rainfall prediction A comparative analysis of modern machine learning algorithms for time-series fo.pdf:application/pdf}, + date = {2022-10-20}, + pmid = {36298367}, + pmcid = {PMC9611376}, + file = {Full Text PDF:/home/alex/Zotero/storage/2E8EIVER/Wang et al. - 2022 - A Systematic Review of Time Series Classification Techniques Used in Biomedical Applications.pdf:application/pdf}, } @article{ahmed_empirical_2010, - title = {An {Empirical} {Comparison} of {Machine} {Learning} {Models} for {Time} {Series} {Forecasting}}, + title = {An Empirical Comparison of Machine Learning Models for Time Series Forecasting}, volume = {29}, issn = {0747-4938, 1532-4168}, url = {http://www.tandfonline.com/doi/abs/10.1080/07474938.2010.481556}, doi = {10.1080/07474938.2010.481556}, - language = {en}, - number = {5-6}, - urldate = {2025-05-08}, - journal = {Econometric Reviews}, - author = {Ahmed, Nesreen K. and Atiya, Amir F. and Gayar, Neamat El and El-Shishiny, Hisham}, - month = aug, - year = {2010}, pages = {594--621}, + number = {5}, + journaltitle = {Econometric Reviews}, + author = {Ahmed, Nesreen K. and Atiya, Amir F. and Gayar, Neamat El and El-Shishiny, Hisham}, + urldate = {2025-05-08}, + date = {2010-08-30}, + langid = {english}, } -@article{garcia_prediction_1981, - title = {Prediction of the {Time} of {Ovulation}*}, - volume = {36}, - issn = {0015-0282}, - url = {https://www.sciencedirect.com/science/article/pii/S0015028216457304}, - doi = {10.1016/S0015-0282(16)45730-4}, - abstract = {Prediction of ovulation was established by correlation of clinical parameters, follicular development by ultrasound, and estradiol, progesterone, and luteinizing hormone (LH) determination in 71 menstrual cycles. Laparoscopic follicular aspiration was accomplished in 41 of those cycles. A 28-hour interval from the ascending limb of the LH seems to be the “ideal time” for retrieval of a preovulatory oocyte. The variability in the amount of LH to which the follicle is exposed during the LH surge seems to indicate that there is a relatively low specific value necessary for ovulation. Ovulation occurs approximately 10 ± 5 hours from the LH peak. Progesterone occurs in relation to the LH surge and is helpful for the retrospective analysis of the menstrual cycle.}, - number = {3}, - urldate = {2025-05-27}, - journal = {Fertility and Sterility}, - author = {Garcia, Jairo E. and Seegar Jones, Georgeanna and Wright, George L.}, - month = sep, - year = {1981}, - pages = {308--315}, - file = {PDF:/home/alex/Zotero/storage/DB8PW3QR/Garcia et al. - 1981 - Prediction of the Time of Ovulation.pdf:application/pdf;ScienceDirect Snapshot:/home/alex/Zotero/storage/QLFLZFDJ/S0015028216457304.html:text/html}, +@article{barrera-animas_rainfall_2022, + title = {Rainfall prediction: A comparative analysis of modern machine learning algorithms for time-series forecasting}, + volume = {7}, + issn = {26668270}, + url = {https://linkinghub.elsevier.com/retrieve/pii/S266682702100102X}, + doi = {10.1016/j.mlwa.2021.100204}, + shorttitle = {Rainfall prediction}, + abstract = {Rainfall forecasting has gained utmost research relevance in recent times due to its complexities and persistent applications such as flood forecasting and monitoring of pollutant concentration levels, among others. Existing models use complex statistical models that are often too costly, both computationally and budgetary, or are not applied to downstream applications. Therefore, approaches that use Machine Learning algorithms in conjunction with time-series data are being explored as an alternative to overcome these drawbacks. To this end, this study presents a comparative analysis using simplified rainfall estimation models based on conventional Machine Learning algorithms and Deep Learning architectures that are efficient for these downstream applications. Models based on {LSTM}, Stacked-{LSTM}, Bidirectional-{LSTM} Networks, {XGBoost}, and an ensemble of Gradient Boosting Regressor, Linear Support Vector Regression, and an Extra-trees Regressor were compared in the task of forecasting hourly rainfall volumes using time-series data. Climate data from 2000 to 2020 from five major cities in the United Kingdom were used. The evaluation metrics of Loss, Root Mean Squared Error, Mean Absolute Error, and Root Mean Squared Logarithmic Error were used to evaluate the models’ performance. Results show that a Bidirectional-{LSTM} Network can be used as a rainfall forecast model with comparable performance to Stacked-{LSTM} Networks. Among all the models tested, the {StackedLSTM} Network with two hidden layers and the Bidirectional-{LSTM} Network performed best. This suggests that models based on {LSTM}-Networks with fewer hidden layers perform better for this approach; denoting its ability to be applied as an approach for budget-wise rainfall forecast applications.}, + pages = {100204}, + journaltitle = {Machine Learning with Applications}, + author = {Barrera-Animas, Ari Yair and Oyedele, Lukumon O. and Bilal, Muhammad and Akinosho, Taofeek Dolapo and Delgado, Juan Manuel Davila and Akanbi, Lukman Adewale}, + urldate = {2025-05-08}, + date = {2022-03}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/FSXMH5N9/Barrera-Animas et al. - 2022 - Rainfall prediction A comparative analysis of modern machine learning algorithms for time-series fo.pdf:application/pdf}, } -@article{b_s_novel_2022, - title = {Novel {Technique} for {Confirmation} of the {Day} of {Ovulation} and {Prediction} of {Ovulation} in {Subsequent} {Cycles} {Using} a {Skin}-{Worn} {Sensor} in a {Population} {With} {Ovulatory} {Dysfunction}: {A} {Side}-by-{Side} {Comparison} {With} {Existing} {Basal} {Body} {Temperature} {Algorithm} and {Vaginal} {Core} {Body} {Temperature} {Algorithm}}, - volume = {10}, - issn = {2296-4185}, - shorttitle = {Novel {Technique} for {Confirmation} of the {Day} of {Ovulation} and {Prediction} of {Ovulation} in {Subsequent} {Cycles} {Using} a {Skin}-{Worn} {Sensor} in a {Population} {With} {Ovulatory} {Dysfunction}}, - url = {https://www.frontiersin.org/articles/10.3389/fbioe.2022.807139/full}, - doi = {10.3389/fbioe.2022.807139}, - abstract = {Objective: Determine the accuracy of a novel technique for confirmation of the day of ovulation and prediction of ovulation in subsequent cycles for the purpose of conception using a skin-worn sensor in a population with ovulatory dysfunction. -Methods: A total of 80 participants recorded consecutive overnight temperatures using a skin-worn sensor at the same time as a commercially available vaginal sensor for a total of 205 reproductive cycles. The vaginal sensor and its associated algorithm were used to determine the day of ovulation, and the ovulation results obtained using the skin-worn sensor and its associated algorithm were assessed for comparative accuracy alongside a number of other statistical techniques, with a further assessment of the same skin-derived data by means of the “three over six” rule. A number of parameters were used to divide the data into separate comparative groups, and further secondary statistical analyses were performed. -Results: The skin-worn sensor and its associated algorithm (together labeled “SWS”) were 66\% accurate for determining the day of ovulation (±1 day) or the absence of ovulation and 90\% accurate for determining the fertile window (ovulation day ±3 days) in the total study population in comparison to the results obtained from the vaginal sensor and its associated algorithm (together labeled “VS”). -Conclusion: SWS is a useful tool for confirming the fertile window and absence of ovulation (anovulation) in a population with ovulatory dysfunction, both known and Edited by:}, - language = {en}, - urldate = {2025-05-27}, - journal = {Frontiers in Bioengineering and Biotechnology}, - author = {B. S., Hurst and K., Davies and R. C., Milnes and T. G., Knowles and A., Pirrie}, - month = mar, - year = {2022}, - pages = {807139}, - file = {PDF:/home/alex/Zotero/storage/U8UPIT4Q/B. S. et al. - 2022 - Novel Technique for Confirmation of the Day of Ovulation and Prediction of Ovulation in Subsequent C.pdf:application/pdf}, +@misc{zeng_are_2022, + title = {Are Transformers Effective for Time Series Forecasting?}, + url = {http://arxiv.org/abs/2205.13504}, + doi = {10.48550/arXiv.2205.13504}, + abstract = {Recently, there has been a surge of Transformer-based solutions for the long-term time series forecasting ({LTSF}) task. Despite the growing performance over the past few years, we question the validity of this line of research in this work. Specifically, Transformers is arguably the most successful solution to extract the semantic correlations among the elements in a long sequence. However, in time series modeling, we are to extract the temporal relations in an ordered set of continuous points. While employing positional encoding and using tokens to embed sub-series in Transformers facilitate preserving some ordering information, the nature of the {\textbackslash}emph\{permutation-invariant\} self-attention mechanism inevitably results in temporal information loss. To validate our claim, we introduce a set of embarrassingly simple one-layer linear models named {LTSF}-Linear for comparison. Experimental results on nine real-life datasets show that {LTSF}-Linear surprisingly outperforms existing sophisticated Transformer-based {LTSF} models in all cases, and often by a large margin. Moreover, we conduct comprehensive empirical studies to explore the impacts of various design elements of {LTSF} models on their temporal relation extraction capability. We hope this surprising finding opens up new research directions for the {LTSF} task. We also advocate revisiting the validity of Transformer-based solutions for other time series analysis tasks (e.g., anomaly detection) in the future. Code is available at: {\textbackslash}url\{https://github.com/cure-lab/{LTSF}-Linear\}.}, + number = {{arXiv}:2205.13504}, + publisher = {{arXiv}}, + author = {Zeng, Ailing and Chen, Muxi and Zhang, Lei and Xu, Qiang}, + urldate = {2025-05-09}, + date = {2022-08-17}, + eprinttype = {arxiv}, + eprint = {2205.13504 [cs]}, + keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, + file = {Full Text PDF:/home/alex/Zotero/storage/V9E95F7E/Zeng et al. - 2022 - Are Transformers Effective for Time Series Forecasting.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/HNVIZG98/2205.html:text/html}, } -@article{salles_softed_2024, - title = {{SoftED}: {Metrics} for soft evaluation of time series event detection}, - volume = {198}, - issn = {03608352}, - shorttitle = {{SoftED}}, - url = {https://linkinghub.elsevier.com/retrieve/pii/S0360835224008507}, - doi = {10.1016/j.cie.2024.110728}, - abstract = {Time series event detectors are evaluated mainly by standard classification metrics, focusing solely on detection accuracy. However, inaccuracy in detecting an event can often result from its preceding or delayed effects reflected in neighboring detections. These detections are valuable to trigger necessary actions or help mitigate unwelcome consequences. In this context, current metrics are insufficient and inadequate for the context of event detection. There is a demand for metrics that incorporate both the concept of time and temporal tolerance for neighboring detections. Inspired by fuzzy sets, this paper introduces SoftED metrics, a new set designed for soft evaluating event detectors. They enable the evaluation of the detection accuracy and the degree to which their detections represent events. A new general protocol inspired by competency questions is also introduced to evaluate temporal tolerant metrics for event detection. The SoftED metrics can improve event detection evaluations by associating events and their representative detections, incorporating temporal tolerance in over 36\% of the overall detector evaluations compared to the usual classification metrics. Following the proposed evaluation protocol, SoftED metrics were evaluated by domain specialists who indicated their contribution to detection evaluation and method selection.}, - language = {en}, - urldate = {2025-06-03}, - journal = {Computers \& Industrial Engineering}, - author = {Salles, Rebecca and Lima, Janio and Reis, Michel and Coutinho, Rafaelli and Pacitti, Esther and Masseglia, Florent and Akbarinia, Reza and Chen, Chao and Garibaldi, Jonathan and Porto, Fabio and Ogasawara, Eduardo}, - month = dec, - year = {2024}, - pages = {110728}, - file = {PDF:/home/alex/Zotero/storage/6RYA5HIQ/Salles et al. - 2024 - SoftED Metrics for soft evaluation of time series event detection.pdf:application/pdf}, +@misc{wang_timexer_2024, + title = {{TimeXer}: Empowering Transformers for Time Series Forecasting with Exogenous Variables}, + url = {http://arxiv.org/abs/2402.19072}, + doi = {10.48550/arXiv.2402.19072}, + shorttitle = {{TimeXer}}, + abstract = {Deep models have demonstrated remarkable performance in time series forecasting. However, due to the partially-observed nature of real-world applications, solely focusing on the target of interest, so-called endogenous variables, is usually insufficient to guarantee accurate forecasting. Notably, a system is often recorded into multiple variables, where the exogenous variables can provide valuable external information for endogenous variables. Thus, unlike well-established multivariate or univariate forecasting paradigms that either treat all the variables equally or ignore exogenous information, this paper focuses on a more practical setting: time series forecasting with exogenous variables. We propose a novel approach, {TimeXer}, to ingest external information to enhance the forecasting of endogenous variables. With deftly designed embedding layers, {TimeXer} empowers the canonical Transformer with the ability to reconcile endogenous and exogenous information, where patch-wise self-attention and variate-wise cross-attention are used simultaneously. Moreover, global endogenous tokens are learned to effectively bridge the causal information underlying exogenous series into endogenous temporal patches. Experimentally, {TimeXer} achieves consistent state-of-the-art performance on twelve real-world forecasting benchmarks and exhibits notable generality and scalability. Code is available at this repository: https://github.com/thuml/{TimeXer}.}, + number = {{arXiv}:2402.19072}, + publisher = {{arXiv}}, + author = {Wang, Yuxuan and Wu, Haixu and Dong, Jiaxiang and Qin, Guo and Zhang, Haoran and Liu, Yong and Qiu, Yunzhong and Wang, Jianmin and Long, Mingsheng}, + urldate = {2025-05-09}, + date = {2024-11-11}, + eprinttype = {arxiv}, + eprint = {2402.19072 [cs]}, + keywords = {Computer Science - Machine Learning, Computer Science - Artificial Intelligence}, + file = {Full Text PDF:/home/alex/Zotero/storage/76BQWVIW/Wang et al. - 2024 - TimeXer Empowering Transformers for Time Series Forecasting with Exogenous Variables.pdf:application/pdf;Snapshot:/home/alex/Zotero/storage/JL64E9YL/2402.html:text/html}, } -@article{hochreiter_long_1997-1, - title = {Long {Short}-{Term} {Memory}}, - volume = {9}, - issn = {0899-7667, 1530-888X}, - url = {https://direct.mit.edu/neco/article/9/8/1735-1780/6109}, - doi = {10.1162/neco.1997.9.8.1735}, - abstract = {Learning to store information over extended time intervals by recurrent backpropagation takes a very long time, mostly because of insufficient, decaying error backflow. We briefly review Hochreiter's (1991) analysis of this problem, then address it by introducing a novel, efficient, gradient based method called long short-term memory (LSTM). Truncating the gradient where this does not do harm, LSTM can learn to bridge minimal time lags in excess of 1000 discrete-time steps by enforcing constant error flow through constant error carousels within special units. Multiplicative gate units learn to open and close access to the constant error flow. LSTM is local in space and time; its computational complexity per time step and weight is O. 1. Our experiments with artificial data involve local, distributed, real-valued, and noisy pattern representations. In comparisons with real-time recurrent learning, back propagation through time, recurrent cascade correlation, Elman nets, and neural sequence chunking, LSTM leads to many more successful runs, and learns much faster. LSTM also solves complex, artificial long-time-lag tasks that have never been solved by previous recurrent network algorithms.}, - language = {en}, +@article{twenge_declines_2017, + title = {Declines in Sexual Frequency among American Adults, 1989–2014}, + volume = {46}, + issn = {0004-0002, 1573-2800}, + url = {http://link.springer.com/10.1007/s10508-017-0953-1}, + doi = {10.1007/s10508-017-0953-1}, + pages = {2389--2401}, number = {8}, - urldate = {2025-06-11}, - journal = {Neural Computation}, - author = {Hochreiter, Sepp and Schmidhuber, Jürgen}, - month = nov, - year = {1997}, - pages = {1735--1780}, - file = {PDF:/home/alex/Zotero/storage/5IE5G9KY/Hochreiter and Schmidhuber - 1997 - Long Short-Term Memory.pdf:application/pdf}, + journaltitle = {Arch Sex Behav}, + author = {Twenge, Jean M. and Sherman, Ryne A. and Wells, Brooke E.}, + urldate = {2025-06-18}, + date = {2017-11}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/7FLDCF3U/Twenge et al. - 2017 - Declines in Sexual Frequency among American Adults, 1989–2014.pdf:application/pdf}, } -@book{medsker_recurrent_1999, - title = {Recurrent {Neural} {Networks}: {Design} and {Applications}}, - isbn = {978-1-4200-4917-6}, - shorttitle = {Recurrent {Neural} {Networks}}, - abstract = {With existent uses ranging from motion detection to music synthesis to financial forecasting, recurrent neural networks have generated widespread attention. The tremendous interest in these networks drives Recurrent Neural Networks: Design and Applications, a summary of the design, applications, current research, and challenges of this subfield of artificial neural networks.This overview incorporates every aspect of recurrent neural networks. It outlines the wide variety of complex learning techniques and associated research projects. Each chapter addresses architectures, from fully connected to partially connected, including recurrent multilayer feedforward. It presents problems involving trajectories, control systems, and robotics, as well as RNN use in chaotic systems. The authors also share their expert knowledge of ideas for alternate designs and advances in theoretical aspects.The dynamical behavior of recurrent neural networks is useful for solving problems in science, engineering, and business. This approach will yield huge advances in the coming years. Recurrent Neural Networks illuminates the opportunities and provides you with a broad view of the current events in this rich field.}, - language = {en}, - publisher = {CRC Press}, - author = {Medsker, Larry and Jain, Lakhmi C.}, - month = dec, - year = {1999}, - note = {Google-Books-ID: ME1SAkN0PyMC}, - keywords = {Computers / Computer Engineering, Computers / General, Computers / Software Development \& Engineering / Systems Analysis \& Design, Technology \& Engineering / Electronics / General}, +@artwork{chevalier_english_2018, + title = {English: Schematic of the Long-Short Term Memory cell, a component of recurrent neural networks}, + url = {https://commons.wikimedia.org/wiki/File:LSTM_Cell.svg}, + shorttitle = {English}, + author = {Chevalier, Guillaume}, + urldate = {2025-06-11}, + date = {2018-05-16}, + file = {Wikimedia Snapshot:/home/alex/Zotero/storage/NMYA4ZA3/FileLSTM_Cell.html:text/html}, +} + +@artwork{fdeloche_english_2017, + title = {English: A diagram for a one-unit recurrent neural network ({RNN}). From bottom to top : input state, hidden state, output state. U, V, W are the weights of the network. Compressed diagram on the left and the unfold version of it on the right.}, + url = {https://commons.wikimedia.org/wiki/File:Recurrent_neural_network_unfold.svg}, + shorttitle = {English}, + author = {{fdeloche}}, + urldate = {2025-06-11}, + date = {2017-06-19}, + file = {Wikimedia Snapshot:/home/alex/Zotero/storage/RU2XSRZN/FileRecurrent_neural_network_unfold.html:text/html}, } @article{hochreiter_vanishing_1998, - title = {The {Vanishing} {Gradient} {Problem} {During} {Learning} {Recurrent} {Neural} {Nets} and {Problem} {Solutions}}, + title = {The Vanishing Gradient Problem During Learning Recurrent Neural Nets and Problem Solutions}, volume = {06}, issn = {0218-4885, 1793-6411}, url = {https://www.worldscientific.com/doi/abs/10.1142/S0218488598000094}, doi = {10.1142/S0218488598000094}, abstract = {Recurrent nets are in principle capable to store past inputs to produce the currently desired output. Because of this property recurrent nets are used in time series prediction and process control. Practical applications involve temporal dependencies spanning many time steps, e.g. between relevant inputs and desired outputs. In this case, however, gradient based learning methods take too much time. The extremely increased learning time arises because the error vanishes as it gets propagated back. In this article the de-caying error flow is theoretically analyzed. Then methods trying to overcome vanishing gradients are briefly discussed. Finally, experiments comparing conventional algorithms and alternative methods are presented. With advanced methods long time lag problems can be solved in reasonable time.}, - language = {en}, - number = {02}, - urldate = {2025-06-11}, - journal = {International Journal of Uncertainty, Fuzziness and Knowledge-Based Systems}, - author = {Hochreiter, Sepp}, - month = apr, - year = {1998}, pages = {107--116}, + number = {2}, + journaltitle = {Int. J. Unc. Fuzz. Knowl. Based Syst.}, + author = {Hochreiter, Sepp}, + urldate = {2025-06-11}, + date = {1998-04}, + langid = {english}, } -@misc{fdeloche_english_2017, - title = {English: {A} diagram for a one-unit recurrent neural network ({RNN}). {From} bottom to top : input state, hidden state, output state. {U}, {V}, {W} are the weights of the network. {Compressed} diagram on the left and the unfold version of it on the right.}, - shorttitle = {English}, - url = {https://commons.wikimedia.org/wiki/File:Recurrent_neural_network_unfold.svg}, - urldate = {2025-06-11}, - author = {{fdeloche}}, - month = jun, - year = {2017}, - file = {Wikimedia Snapshot:/home/alex/Zotero/storage/RU2XSRZN/FileRecurrent_neural_network_unfold.html:text/html}, +@book{medsker_recurrent_1999, + title = {Recurrent Neural Networks: Design and Applications}, + isbn = {978-1-4200-4917-6}, + shorttitle = {Recurrent Neural Networks}, + abstract = {With existent uses ranging from motion detection to music synthesis to financial forecasting, recurrent neural networks have generated widespread attention. The tremendous interest in these networks drives Recurrent Neural Networks: Design and Applications, a summary of the design, applications, current research, and challenges of this subfield of artificial neural networks.This overview incorporates every aspect of recurrent neural networks. It outlines the wide variety of complex learning techniques and associated research projects. Each chapter addresses architectures, from fully connected to partially connected, including recurrent multilayer feedforward. It presents problems involving trajectories, control systems, and robotics, as well as {RNN} use in chaotic systems. The authors also share their expert knowledge of ideas for alternate designs and advances in theoretical aspects.The dynamical behavior of recurrent neural networks is useful for solving problems in science, engineering, and business. This approach will yield huge advances in the coming years. Recurrent Neural Networks illuminates the opportunities and provides you with a broad view of the current events in this rich field.}, + pagetotal = {414}, + publisher = {{CRC} Press}, + author = {Medsker, Larry and Jain, Lakhmi C.}, + date = {1999-12-20}, + langid = {english}, + note = {Google-Books-{ID}: {ME}1SAkN0PyMC}, + keywords = {Computers / Computer Engineering, Computers / General, Computers / Software Development \& Engineering / Systems Analysis \& Design, Technology \& Engineering / Electronics / General}, } -@misc{chevalier_english_2018, - title = {English: {Schematic} of the {Long}-{Short} {Term} {Memory} cell, a component of recurrent neural networks}, - shorttitle = {English}, - url = {https://commons.wikimedia.org/wiki/File:LSTM_Cell.svg}, +@article{hochreiter_long_1997-1, + title = {Long Short-Term Memory}, + volume = {9}, + issn = {0899-7667, 1530-888X}, + url = {https://direct.mit.edu/neco/article/9/8/1735-1780/6109}, + doi = {10.1162/neco.1997.9.8.1735}, + abstract = {Learning to store information over extended time intervals by recurrent backpropagation takes a very long time, mostly because of insufficient, decaying error backflow. We briefly review Hochreiter's (1991) analysis of this problem, then address it by introducing a novel, efficient, gradient based method called long short-term memory ({LSTM}). Truncating the gradient where this does not do harm, {LSTM} can learn to bridge minimal time lags in excess of 1000 discrete-time steps by enforcing constant error flow through constant error carousels within special units. Multiplicative gate units learn to open and close access to the constant error flow. {LSTM} is local in space and time; its computational complexity per time step and weight is O. 1. Our experiments with artificial data involve local, distributed, real-valued, and noisy pattern representations. In comparisons with real-time recurrent learning, back propagation through time, recurrent cascade correlation, Elman nets, and neural sequence chunking, {LSTM} leads to many more successful runs, and learns much faster. {LSTM} also solves complex, artificial long-time-lag tasks that have never been solved by previous recurrent network algorithms.}, + pages = {1735--1780}, + number = {8}, + journaltitle = {Neural Computation}, + author = {Hochreiter, Sepp and Schmidhuber, Jürgen}, urldate = {2025-06-11}, - author = {Chevalier, Guillaume}, - month = may, - year = {2018}, - file = {Wikimedia Snapshot:/home/alex/Zotero/storage/NMYA4ZA3/FileLSTM_Cell.html:text/html}, + date = {1997-11-01}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/5IE5G9KY/Hochreiter and Schmidhuber - 1997 - Long Short-Term Memory.pdf:application/pdf}, +} + +@article{salles_softed_2024, + title = {{SoftED}: Metrics for soft evaluation of time series event detection}, + volume = {198}, + issn = {03608352}, + url = {https://linkinghub.elsevier.com/retrieve/pii/S0360835224008507}, + doi = {10.1016/j.cie.2024.110728}, + shorttitle = {{SoftED}}, + abstract = {Time series event detectors are evaluated mainly by standard classification metrics, focusing solely on detection accuracy. However, inaccuracy in detecting an event can often result from its preceding or delayed effects reflected in neighboring detections. These detections are valuable to trigger necessary actions or help mitigate unwelcome consequences. In this context, current metrics are insufficient and inadequate for the context of event detection. There is a demand for metrics that incorporate both the concept of time and temporal tolerance for neighboring detections. Inspired by fuzzy sets, this paper introduces {SoftED} metrics, a new set designed for soft evaluating event detectors. They enable the evaluation of the detection accuracy and the degree to which their detections represent events. A new general protocol inspired by competency questions is also introduced to evaluate temporal tolerant metrics for event detection. The {SoftED} metrics can improve event detection evaluations by associating events and their representative detections, incorporating temporal tolerance in over 36\% of the overall detector evaluations compared to the usual classification metrics. Following the proposed evaluation protocol, {SoftED} metrics were evaluated by domain specialists who indicated their contribution to detection evaluation and method selection.}, + pages = {110728}, + journaltitle = {Computers \& Industrial Engineering}, + author = {Salles, Rebecca and Lima, Janio and Reis, Michel and Coutinho, Rafaelli and Pacitti, Esther and Masseglia, Florent and Akbarinia, Reza and Chen, Chao and Garibaldi, Jonathan and Porto, Fabio and Ogasawara, Eduardo}, + urldate = {2025-06-03}, + date = {2024-12}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/6RYA5HIQ/Salles et al. - 2024 - SoftED Metrics for soft evaluation of time series event detection.pdf:application/pdf}, +} + +@article{b_s_novel_2022, + title = {Novel Technique for Confirmation of the Day of Ovulation and Prediction of Ovulation in Subsequent Cycles Using a Skin-Worn Sensor in a Population With Ovulatory Dysfunction: A Side-by-Side Comparison With Existing Basal Body Temperature Algorithm and Vaginal Core Body Temperature Algorithm}, + volume = {10}, + issn = {2296-4185}, + url = {https://www.frontiersin.org/articles/10.3389/fbioe.2022.807139/full}, + doi = {10.3389/fbioe.2022.807139}, + shorttitle = {Novel Technique for Confirmation of the Day of Ovulation and Prediction of Ovulation in Subsequent Cycles Using a Skin-Worn Sensor in a Population With Ovulatory Dysfunction}, + abstract = {Objective: Determine the accuracy of a novel technique for confirmation of the day of ovulation and prediction of ovulation in subsequent cycles for the purpose of conception using a skin-worn sensor in a population with ovulatory dysfunction. +Methods: A total of 80 participants recorded consecutive overnight temperatures using a skin-worn sensor at the same time as a commercially available vaginal sensor for a total of 205 reproductive cycles. The vaginal sensor and its associated algorithm were used to determine the day of ovulation, and the ovulation results obtained using the skin-worn sensor and its associated algorithm were assessed for comparative accuracy alongside a number of other statistical techniques, with a further assessment of the same skin-derived data by means of the “three over six” rule. A number of parameters were used to divide the data into separate comparative groups, and further secondary statistical analyses were performed. +Results: The skin-worn sensor and its associated algorithm (together labeled “{SWS}”) were 66\% accurate for determining the day of ovulation (±1 day) or the absence of ovulation and 90\% accurate for determining the fertile window (ovulation day ±3 days) in the total study population in comparison to the results obtained from the vaginal sensor and its associated algorithm (together labeled “{VS}”). +Conclusion: {SWS} is a useful tool for confirming the fertile window and absence of ovulation (anovulation) in a population with ovulatory dysfunction, both known and Edited by:}, + pages = {807139}, + journaltitle = {Front. Bioeng. Biotechnol.}, + author = {B. S., Hurst and K., Davies and R. C., Milnes and T. G., Knowles and A., Pirrie}, + urldate = {2025-05-27}, + date = {2022-03-04}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/U8UPIT4Q/B. S. et al. - 2022 - Novel Technique for Confirmation of the Day of Ovulation and Prediction of Ovulation in Subsequent C.pdf:application/pdf}, +} + +@article{garcia_prediction_1981, + title = {Prediction of the Time of Ovulation*}, + volume = {36}, + issn = {0015-0282}, + url = {https://www.sciencedirect.com/science/article/pii/S0015028216457304}, + doi = {10.1016/S0015-0282(16)45730-4}, + abstract = {Prediction of ovulation was established by correlation of clinical parameters, follicular development by ultrasound, and estradiol, progesterone, and luteinizing hormone ({LH}) determination in 71 menstrual cycles. Laparoscopic follicular aspiration was accomplished in 41 of those cycles. A 28-hour interval from the ascending limb of the {LH} seems to be the “ideal time” for retrieval of a preovulatory oocyte. The variability in the amount of {LH} to which the follicle is exposed during the {LH} surge seems to indicate that there is a relatively low specific value necessary for ovulation. Ovulation occurs approximately 10 ± 5 hours from the {LH} peak. Progesterone occurs in relation to the {LH} surge and is helpful for the retrospective analysis of the menstrual cycle.}, + pages = {308--315}, + number = {3}, + journaltitle = {Fertility and Sterility}, + author = {Garcia, Jairo E. and Seegar Jones, Georgeanna and Wright, George L.}, + urldate = {2025-05-27}, + date = {1981-09-01}, + file = {PDF:/home/alex/Zotero/storage/DB8PW3QR/Garcia et al. - 1981 - Prediction of the Time of Ovulation.pdf:application/pdf;ScienceDirect Snapshot:/home/alex/Zotero/storage/QLFLZFDJ/S0015028216457304.html:text/html}, +} + +@article{vaswani_attention_nodate, + title = {Attention Is All You Need}, + abstract = {The dominant sequence transduction models are based on complex recurrent or convolutional neural networks that include an encoder and a decoder. The best performing models also connect the encoder and decoder through an attention mechanism. We propose a new simple network architecture, the Transformer, based solely on attention mechanisms, dispensing with recurrence and convolutions entirely. Experiments on two machine translation tasks show these models to be superior in quality while being more parallelizable and requiring significantly less time to train. Our model achieves 28.4 {BLEU} on the {WMT} 2014 Englishto-German translation task, improving over the existing best results, including ensembles, by over 2 {BLEU}. On the {WMT} 2014 English-to-French translation task, our model establishes a new single-model state-of-the-art {BLEU} score of 41.8 after training for 3.5 days on eight {GPUs}, a small fraction of the training costs of the best models from the literature. We show that the Transformer generalizes well to other tasks by applying it successfully to English constituency parsing both with large and limited training data.}, + author = {Vaswani, Ashish and Shazeer, Noam and Parmar, Niki and Uszkoreit, Jakob and Jones, Llion and Gomez, Aidan N and Kaiser, Łukasz and Polosukhin, Illia}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/TES5P5PX/Vaswani et al. - Attention Is All You Need.pdf:application/pdf}, +} + +@article{wu_deep_nodate, + title = {Deep Transformer Models for Time Series Forecasting:The Influenza Prevalence Case}, + abstract = {In this paper, we present a new approach to time series forecasting. Time series data are prevalent in many scientific and engineering disciplines. Time series forecasting is a crucial task in modeling time series data, and is an important area of machine learning. In this work we developed a novel method that employs Transformer-based machine learning models to forecast time series data. This approach works by leveraging selfattention mechanisms to learn complex patterns and dynamics from time series data. Moreover, it is a generic framework and can be applied to univariate and multivariate time series data, as well as time series embeddings. Using influenzalike illness ({ILI}) forecasting as a case study, we show that the forecasting results produced by our approach are favorably comparable to the stateof-the-art.}, + author = {Wu, Neo and Green, Bradley and Ben, Xue and O'Banion, Shawn}, + langid = {english}, + file = {PDF:/home/alex/Zotero/storage/GHT5UMNX/Wu et al. - Deep Transformer Models for Time Series ForecastingThe Influenza Prevalence Case.pdf:application/pdf}, } diff --git a/thesis/resources/figures/background_convolution_example.png b/thesis/resources/figures/background_convolution_example.png new file mode 100644 index 0000000..51f9ccd Binary files /dev/null and b/thesis/resources/figures/background_convolution_example.png differ diff --git a/thesis/resources/figures/background_transformer_architecture.png b/thesis/resources/figures/background_transformer_architecture.png new file mode 100644 index 0000000..e52f4ad Binary files /dev/null and b/thesis/resources/figures/background_transformer_architecture.png differ diff --git a/thesis/sections/background.tex b/thesis/sections/background.tex index 1012ac3..f9b52bd 100644 --- a/thesis/sections/background.tex +++ b/thesis/sections/background.tex @@ -182,23 +182,6 @@ The input gate updates the cell state with new information derived from the curr Finally, the output gate controls how much of the updated cell state contributes to the hidden state \(h_t\), which is passed on to the next time step or used for prediction. -Mathematically, the core LSTM operations are given by: - -\[ - \begin{aligned} - f_t &= \sigma(W_f [h_{t-1}, x_t] + b_f) \\ - i_t &= \sigma(W_i [h_{t-1}, x_t] + b_i) \\ - \tilde{c}_t &= \tanh(W_c [h_{t-1}, x_t] + b_c) \\ - c_t &= f_t \odot c_{t-1} + i_t \odot \tilde{c}_t \\ - o_t &= \sigma(W_o [h_{t-1}, x_t] + b_o) \\ - h_t &= o_t \odot \tanh(c_t) - \end{aligned} -\] - -Here, \( \odot \) denotes element-wise multiplication, and \( \sigma \) is the sigmoid function. -These equations allow for more stable training and long-range temporal modeling. -\\ - LSTMs are widely used in biomedical applications due to their capacity to handle sequences of variable length and complexity. In the context of ovulation prediction, where hormonal patterns exhibit periodicity but also irregularity, LSTMs are well-suited to learn relevant time-dependent signals from sequential physiological measurements. @@ -213,4 +196,117 @@ recurrence in favor of attention mechanisms. \subsubsection{Transformer Models}\label{subsubsec:transformer_models} +Transformer models are a class of neural architectures that use \emph{self-attention} +to model dependencies in sequential data without relying on recurrence~\cite{vaswani_attention_2017}. +Unlike recurrent neural networks (RNNs), Transformers process input sequences in parallel, +allowing them to model relationships between any pair of input tokens or timesteps directly. +This mitigates the limitations of recurrent models, such as long-term memory constraints +and vanishing gradients. + +Originally introduced for machine translation, Transformers have proven broadly applicable to +various sequence modeling tasks due to their flexibility, scalability, and strong performance +on complex temporal patterns. + +At the core of the Transformer is the attention mechanism, which enables the model to compute +context-aware representations by weighing the importance of different input positions for each output. +This is achieved through \emph{scaled dot-product attention}, where queries, keys, and values are +linearly projected from the input and used to compute attention scores. + +\begin{figure} + \centering + \includegraphics[width=0.4\textwidth]{background_transformer_architecture} + \caption{The Transformer - architecture for an encoder-decoder model~\cite{vaswani_attention_2017}} + \label{fig:background_transformer_architecture} +\end{figure} + + +Figure~\ref{fig:background_transformer_architecture} illustrates the original encoder-decoder model introduced by~\citeyear{vaswani_attention_2017}. + +The Transformer architecture consists of two components: an \emph{Encoder} and a \emph{Decoder}. + +\paragraph{Encoder:} + +The encoder is responsible for encoding the input into a contextualized representation. +In the case of machine translation, this input would be a sentence in the source language. + + +The input tokens are first mapped to dense continuous vector representations (embeddings). +Since the attention mechanism permutation-invariant---that is, it does not inherently encode the order of tokens in the sequence--- +\emph{positional encodings} are added to the token embeddings to provide information about the token positions in the sequence. + +Without positional encoding, repeated tokens such as `The` would be indistinguishable +to the model regardless of their location, even if they play different syntactic or semantic roles. +Positional encodings, often based on sinusoidal functions, inject a unique position-dependent signal +into each token, enabling the model to distinguish between identical tokens in different positions. + +Inside each encoder block, \emph{Multi-Head-Attention} is applied to the inputs. +Multi-Head-Attention extends the regular attention mechanism, by adding multiple attention heads that focus on different parts of the embeddings. +Each head does the scaled dot-product attention independently on its slice of the data. +The outputs of all heads are then concatenated and combined via a linear projection. + +The outputs of the attention mechanism are then processed in a feed forward network to allow for a non-linear projection. +Residual connections for both the attention and the feed forward allow for better gradient flow and model stability. + +\paragraph{Decoder:} + +In a sequence-to-sequence Transformer, the decoder generates the output sequence autoregressively, +using the contextualized representation produced by the encoder. + +At inference time, generation begins with a special \emph{start-of-sequence} token. +Like the encoder, the decoder embeds its inputs and augments them with positional encodings +to retain information about token order. + +To ensure that the model does not access future tokens during training, +a \emph{look-ahead mask} is applied within the decoder’s self-attention mechanism. +This masking ensures that each position can only attend to earlier positions in the sequence, +preventing information leakage. +This component is referred to as \emph{masked multi-head self-attention}. + +Following the masked self-attention, the decoder incorporates information from the encoder +via a \emph{cross-attention} layer. +Here, the decoder’s hidden states act as queries, while the encoder’s outputs serve as keys and values. +This allows the decoder to condition its predictions on the entire encoded input sequence. + +The output of the cross-attention layer is passed through a position-wise feed-forward network +and further normalization and residual connections, analogous to the encoder blocks. + +For tasks such as machine translation, the final decoder outputs are linearly projected +to the target vocabulary size, and a softmax function is applied to produce a probability distribution +over possible next tokens. + +During inference, tokens are sampled sequentially from this distribution and fed back into the decoder for the next prediction step. +This process continues until a special \emph{end-of-sequence} token is generated, indicating that the model has completed the output sequence. + +Depending on the use case and data complexity, multiple encoder and decoder layers can be stacked +to increase model capacity and improve predictive performance. + +While originally developed for machine translation, the Transformer architecture has since been applied +successfully to a range of tasks, including time-series forecasting and biomedical data analysis(\cite{wu_deep_nodate,zeng_are_2022}). +Its ability to model long-range dependencies without recurrence makes it particularly suited for biomedical time-series, +where signals may be irregular, noisy, or span varying temporal scales. + +\subsubsection{Convolutional Layers as Temporal Feature Extractors} +For high-resolution time-series data, the input dimensionality can become large, +especially in models like Transformers that process the entire sequence in parallel. +This can lead to increased memory consumption and slower training. +To mitigate this and retain as much information as possible, convolutional layers can be used +to reduce the sequence length while preserving important local patterns. + +In this context, one-dimensional convolutions act as learnable filters that slide over the input sequence to extract temporal features. +Each filter is parameterized to respond to specific local structures in the data, such as peaks, slopes, or short-term motifs. +By adjusting the \emph{stride}—the step size of the convolution—the model can control the degree of downsampling, +effectively reducing the number of time steps passed to subsequent layers. + +Additional dimensionality reduction can be achieved using pooling operations, such as \emph{max pooling}, which retains only the maximum value within a given window. +These techniques reduce the computational load while maintaining salient information for downstream processing. + +Figure~\ref{fig:background_convolution_example} illustrates a simple one-dimensional convolution applied to a sequence using a filter of size 3. +The stride determines how far the filter moves at each step, affecting both the resolution and length of the resulting feature map. + +\begin{figure} + \centering + \includegraphics[width=0.6\textwidth]{background_convolution_example} + \caption{Example of a simple 1-D convolution on an input sequence.} + \label{fig:background_convolution_example} +\end{figure}