<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIRx Med</journal-id><journal-id journal-id-type="publisher-id">xmed</journal-id><journal-id journal-id-type="index">34</journal-id><journal-title>JMIRx Med</journal-title><abbrev-journal-title>JMIRx Med</abbrev-journal-title><issn pub-type="epub">2563-6316</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v7i1e76506</article-id><article-id pub-id-type="doi">10.2196/76506</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>AI-Driven and Automated Systems for Continuous Oxygen Saturation Monitoring in Long-Term Oxygen Therapy: Systematic Review</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Kadariya</surname><given-names>Suman</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Niraula</surname><given-names>Prajita</given-names></name><degrees>BCS, MSCS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Poudel</surname><given-names>Bishal</given-names></name><degrees>MBBS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kadariya</surname><given-names>Sujan</given-names></name><degrees>MBBS</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Conway Regional Medical Center</institution><addr-line>Conway</addr-line><addr-line>AR</addr-line><country>United States</country></aff><aff id="aff2"><institution>Independent Researcher</institution><addr-line>Redcross Rd</addr-line><addr-line>Kathmandu</addr-line><country>Nepal</country></aff><aff id="aff3"><institution>KIST Medical College</institution><addr-line>Kathmandu</addr-line><country>Nepal</country></aff><aff id="aff4"><institution>Kathmandu University School of Medical Sciences</institution><addr-line>Dhulikhel</addr-line><country>Nepal</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Schwartz</surname><given-names>Amy</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Lee</surname><given-names>Juhee</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Hoang</surname><given-names>Nhung H</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Prajita Niraula, BCS, MSCS, Independent Researcher, Redcross Rd, Kathmandu, 44600, Nepal, 977 9818787009; <email>prajita56@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>7</day><month>10</month><year>2026</year></pub-date><volume>7</volume><elocation-id>e76506</elocation-id><history><date date-type="received"><day>27</day><month>04</month><year>2025</year></date><date date-type="rev-recd"><day>30</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>04</day><month>09</month><year>2026</year></date></history><copyright-statement>&#x00A9; Suman Kadariya, Prajita Niraula, Bishal Poudel, Sujan Kadariya. Originally published in JMIRx Med (<ext-link ext-link-type="uri" xlink:href="https://med.jmirx.org">https://med.jmirx.org</ext-link>), 7.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIRx Med, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://med.jmirx.org/">https://med.jmirx.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://xmed.jmir.org/2026/1/e76506"/><related-article related-article-type="companion" ext-link-type="doi" xlink:href="10.1101/2025.04.20.25326131v1" xlink:title="Preprint (medRxiv)" xlink:type="simple">https://www.medrxiv.org/content/10.1101/2025.04.20.25326131v1</related-article><related-article related-article-type="companion" ext-link-type="doi" xlink:href="10.2196/76506" xlink:title="Preprint (JMIR Preprint)" xlink:type="simple">http://preprints.jmir.org/preprint/76506</related-article><related-article related-article-type="companion" ext-link-type="doi" xlink:href="10.2196/111437" xlink:title="Peer-Review Report by Junhee Lee (Reviewer AA)" xlink:type="simple">https://med.jmirx.org/2026/1/e111437</related-article><related-article related-article-type="companion" ext-link-type="doi" xlink:href="10.2196/111439" xlink:title="Peer-Review Report by Nhung H Hoang (Reviewer CR)" xlink:type="simple">https://med.jmirx.org/2026/1/e111439</related-article><related-article related-article-type="companion" ext-link-type="doi" xlink:href="10.2196/111441" xlink:title="Authors' Response to Peer-Review Reports" xlink:type="simple">https://xmed.jmir.org/editor/submissionEditing/111441</related-article><abstract><sec><title>Background</title><p>Long-term oxygen therapy (LTOT) improves outcomes in selected patients with severe chronic hypoxemia, but conventional LTOT uses fixed oxygen flow prescriptions that may not reflect changing needs during activity, sleep, or exacerbations. AI and automated oxygen systems may support continuous peripheral capillary oxygen saturation (SpO<sub>2</sub>) monitoring, signal-quality assessment, and adaptive oxygen titration. Evidence comparing AI-driven and non-AI automated approaches across performance, clinical readiness, LTOT applicability, and equity remains limited.</p></sec><sec><title>Objective</title><p>This systematic review synthesized peer-reviewed evidence on AI-driven and automated systems for continuous SpO<sub>2</sub> monitoring or oxygen titration relevant to adult LTOT, focusing on accuracy, motion robustness, clinical performance, demographic equity, and readiness for home or ambulatory deployment.</p></sec><sec sec-type="methods"><title>Methods</title><p>This review followed PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 and PRISMA-S (Preferred Reporting Items for Systematic Reviews and Meta-Analyses&#x2013;Search). PubMed, IEEE Xplore, Springer Link, ACM Digital Library, and supplementary MDPI publisher-level searches were searched for English-language peer-reviewed studies published during 2000 to January 2025. Eligible studies evaluated AI-driven or automated systems for continuous SpO<sub>2</sub> monitoring or oxygen titration in adult LTOT-relevant populations and addressed motion artifact, low-perfusion signal management, skin tone bias, or prolonged signal stability. Two reviewers independently screened and extracted data; disagreements were resolved with a third reviewer. Risk of bias was assessed using an adapted Risk of Bias in Non-randomized Studies of Interventions (ROBINS-I) framework. Heterogeneous designs and outcomes precluded meta-analysis, so findings were synthesized narratively following Popay et al.</p></sec><sec sec-type="results"><title>Results</title><p>Of 928 records (926 from databases or platforms and 2 from manual reference screening), 912 remained after deduplication, 61 full texts were assessed, and 8 studies were included. Five studies evaluated AI-based systems and 3 evaluated automated non-AI oxygen delivery. AI models reported SpO<sub>2</sub> estimation mean absolute error as low as 0.57% and root mean square error as low as 0.69%, but most were simulated, retrospective, or non&#x2013;chronic obstructive pulmonary disease (COPD) specific. Only Cabanas et al reported skin tone&#x2013;stratified bias analysis. Automated systems showed stronger clinical deployment evidence: O<sub>2</sub>matic maintained the target SpO<sub>2</sub> 85.1% of the time versus 46.6% with manual titration, while Cirio and Nava reported mean SpO<sub>2</sub> of 95% versus 93% manually. Risk of bias was moderate to serious, mainly due to participant selection, limited demographic reporting, and algorithmic transparency.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>AI-driven and automated LTOT-relevant systems address complementary gaps. AI approaches show promise for signal interpretation and personalized prediction, whereas rule-based automated systems have stronger near-term clinical evidence for oxygen titration. Evidence is limited by small study numbers, heterogeneous outcomes, limited COPD or home LTOT validation, and sparse equity reporting. Future work should prioritize longitudinal validation in diverse LTOT populations, prespecified equity outcomes, failure-mode reporting, and hybrid architectures combining AI signal intelligence with safety-critical automated control.</p></sec></abstract><kwd-group><kwd>long-term oxygen therapy</kwd><kwd>artificial intelligence</kwd><kwd>machine learning</kwd><kwd>oxygen saturation monitoring</kwd><kwd>pulse oximetry</kwd><kwd>closed-loop systems</kwd><kwd>automated oxygen titration</kwd><kwd>motion artifacts</kwd><kwd>chronic obstructive pulmonary disease</kwd><kwd>systematic review</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Chronic respiratory diseases, such as chronic obstructive pulmonary disease (COPD), represent a significant and growing global health burden. In 2019, COPD alone accounted for more than 3.2 million fatalities worldwide, and estimates indicate an increasing prevalence and mortality trend through 2050 [<xref ref-type="bibr" rid="ref1">1</xref>]. Long-term oxygen therapy (LTOT) remains critical in the management of patients with severe chronic hypoxemia and has been demonstrated to improve survival, quality of life, and exercise tolerance in selected patients [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>Despite its established clinical efficacy, conventional LTOT is routinely prescribed in a static and reactive modality, with oxygen flow rates determined during clinic appointments according to protocols such as the 6-minute walk test (6MWT). These fixed dosing regimens do not adequately address dynamic variations in patients&#x2019; oxygen requirements across daily activity, sleep, and acute physiological changes [<xref ref-type="bibr" rid="ref4">4</xref>]. As a result, patients remain at risk of under-oxygenation, with the development of hypoxemia, or over-oxygenation, with the possibility of hypercapnia and oxidative injury.</p><p>The history of LTOT technology illustrates a persistent gap between technological ambition and real-world applicability. The landmark Medical Research Council (MRC) [<xref ref-type="bibr" rid="ref3">3</xref>] and Nocturnal Oxygen Therapy Trial (NOTT) [<xref ref-type="bibr" rid="ref2">2</xref>] trials, published in 1981 and 1980, respectively, established that supplemental oxygen delivered at fixed flow rates (typically 1&#x2010;4 L/min) improved survival in patients with chronic hypoxic cor pulmonale. The decades following these trials saw the introduction of demand-flow oxygen systems, which conserved oxygen supply by delivering pulses only during inhalation rather than continuously. While demand-flow devices reduced oxygen consumption, they continued to operate on static, preprogrammed thresholds and could not adapt to real-time changes in patient physiology. The clinical introduction of wearable pulse oximetry in the 1990s enabled continuous peripheral capillary oxygen saturation (SpO<sub>2</sub>) monitoring outside hospital settings but required clinician presence for interpretation and did not translate into automated titration. None of these successive generations of technology could adapt to real-time physiological variability, account for motion artifact contamination of photoplethysmographic (PPG) signals, or accommodate the well-documented demographic variation in measurement accuracy that has since been characterized in the literature [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Against this backdrop, a range of AI approaches have been applied to SpO<sub>2</sub> signal processing and oxygen delivery. Gaussian process regression models have demonstrated mean absolute errors (MAEs) below 1% in SpO<sub>2</sub> estimation from PPG data, with the ability to propagate uncertainty estimates alongside predictions [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Deep neural network architectures, including convolutional neural networks (CNNs), have been applied to both SpO<sub>2</sub> prediction and motion artifact classification, offering the capacity to learn complex nonlinear mappings from raw waveform data [<xref ref-type="bibr" rid="ref9">9</xref>]. Edge-AI frameworks, where inference runs locally on low-power embedded devices rather than in the cloud, have been proposed for predictive oxygen dosing that integrates historical physiological and behavioral trends [<xref ref-type="bibr" rid="ref10">10</xref>]. Reference signal-less machine learning classifiers for PPG signal quality assessment have demonstrated the ability to flag and exclude corrupted signal segments in real time without additional hardware [<xref ref-type="bibr" rid="ref11">11</xref>]. Despite these advances, no AI system reviewed to date has achieved sustained clinical validation in real-world, home-based LTOT populations, and the majority have been tested only in simulated or retrospective contexts.</p><p>In parallel, rule-based automated oxygen delivery systems, including the O<sub>2</sub>matic closed-loop device [<xref ref-type="bibr" rid="ref12">12</xref>], the intelligent portable oxygen concentrator (iPOC) [<xref ref-type="bibr" rid="ref13">13</xref>], and the automated titration device evaluated by Cirio and Nava [<xref ref-type="bibr" rid="ref14">14</xref>], have undergone clinical testing in hospital and ambulatory settings. These systems dynamically adjust oxygen flow in response to continuous SpO<sub>2</sub> readings within predefined thresholds, demonstrating meaningful improvements in time-in-target saturation compared to manual titration. However, these systems lack any learning component, cannot adapt to individual physiological patterns, and do not incorporate signal quality verification or demographic bias mitigation. The gap in evidence is therefore not merely technological: no systematic review has directly and explicitly compared AI-driven and non-AI automated systems across shared technical and clinical evaluation dimensions, with real-world LTOT deployment readiness and equity as primary analytical lenses.</p><p>This review addresses that gap. The following 3 research questions (RQs) guide the synthesis:</p><list list-type="bullet"><list-item><p>RQ1: What is the technical performance, including SpO<sub>2</sub> estimation accuracy, signal quality under motion, and demographic measurement robustness, of AI-driven systems for continuous SpO<sub>2</sub> monitoring in adults with relevance to LTOT, as reported in peer-reviewed literature from 2000 to January 2025?</p></list-item><list-item><p>RQ2: What is the clinical performance, including SpO<sub>2</sub> maintenance within target ranges, usability, and patient outcomes, of automated non-AI oxygen delivery systems evaluated in LTOT-relevant clinical populations?</p></list-item><list-item><p>RQ3: To what extent do AI-driven and automated SpO<sub>2</sub> systems address equity-relevant challenges, specifically skin tone measurement bias and demographic representativeness, and demonstrate readiness for real-world home-based LTOT deployment?</p></list-item></list></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Eligibility Criteria</title><p>This systematic review adhered to PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 reporting guidelines [<xref ref-type="bibr" rid="ref15">15</xref>] (<xref ref-type="supplementary-material" rid="app3">Checklist 1</xref>). The review was not prospectively registered, and no public protocol was prepared. Studies were eligible if they evaluated AI-driven or automated systems for continuous SpO<sub>2</sub> monitoring or oxygen titration relevant to LTOT in adult populations and addressed at least one of the following LTOT-critical technical challenges: (1) motion-induced signal artifact correction, a prerequisite for ambulatory SpO<sub>2</sub> monitoring given the frequency of patient movement in home-based LTOT; (2) low-perfusion or weak signal management, essential for reliable readings in patients with peripheral vascular compromise common in COPD; (3) skin tone measurement bias mitigation, given documented inaccuracies of pulse oximetry in individuals with darker skin pigmentation [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]; or (4) prolonged signal stability (&#x003E;24 h), required for the continuous monitoring that defines LTOT. Technical validation outcomes such as MAE (&#x003C;2%) or root mean square error (RMSE; &#x003C;3%) were prioritized, as were demographic bias analyses. These criteria were designed not as arbitrary technical filters but as clinically motivated prerequisites: any SpO<sub>2</sub> monitoring system intended for long-term ambulatory home use must demonstrably address each of these challenges to be considered viable for the LTOT context.</p><p>Studies were excluded if they were limited to acute or nonchronic conditions (eg, surgical hypoxia, asthma exacerbations), described interventions not relevant to LTOT (eg, nonautomated pulse rate estimation), were non&#x2013;peer-reviewed (preprints, theses, conference abstracts), were inaccessible due to paywall restrictions, were published in languages other than English, or lacked explicit relevance to LTOT monitoring (eg, AI applied to electrocardiogram [ECG] analysis).</p></sec><sec id="s2-2"><title>Information Sources and Search Strategy</title><p>A systematic literature search was conducted in January 2025 across PubMed, IEEE Xplore, Springer Link, ACM Digital Library, and supplementary MDPI publisher-level searches. The search covered publications from January 1, 2000, to January 2025. The search strategy combined controlled vocabulary and free-text terms using Boolean operators to maximize sensitivity. Search terms were organized around three domains: (1) clinical context terms addressing RQ1 and RQ2, including &#x201C;long-term oxygen therapy,&#x201D; &#x201C;LTOT,&#x201D; &#x201C;oxygen concentrator,&#x201D; &#x201C;oxygen titration,&#x201D; and &#x201C;SpO<sub>2</sub> monitoring&#x201D;; (2) technology terms addressing RQ1 and RQ2, including &#x201C;artificial intelligence,&#x201D; &#x201C;machine learning,&#x201D; &#x201C;automated oxygen delivery,&#x201D; &#x201C;closed-loop control,&#x201D; and &#x201C;predictive modeling&#x201D;; and (3) signal quality terms addressing RQ1 and RQ3, including &#x201C;motion artifact correction,&#x201D; &#x201C;photoplethysmography,&#x201D; &#x201C;PPG,&#x201D; &#x201C;skin tone bias,&#x201D; and &#x201C;signal quality.&#x201D; Complete search strings with all Boolean operators, field tags, date ranges, language filters, and records retrieved per source are documented in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, structured in accordance with the PRISMA-S (Preferred Reporting Items for Systematic Reviews and Meta-Analyses&#x2013;Search) reporting guidelines [<xref ref-type="bibr" rid="ref16">16</xref>] (<xref ref-type="supplementary-material" rid="app4">Checklist 2</xref></p><p>MDPI and Springer Link were searched as supplementary publisher or platform sources, not as formal bibliographic databases, to improve coverage of open-access engineering and biomedical sensor literature. Due to platform limitations, MDPI was searched using discrete keyword combinations; each string is listed separately in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Because publisher-platform searching can introduce coverage bias, these sources were treated as supplementary and this limitation is acknowledged in the <italic>Discussion</italic> section.</p></sec><sec id="s2-3"><title>Study Selection Process</title><p>Titles and abstracts were screened independently by 2 reviewers (PN and SK) against the predefined eligibility criteria. Full-text articles meeting the criteria at the abstract stage were retrieved and independently assessed. Disagreements at both stages were resolved by consensus discussion with a third reviewer (BP). The study selection process is documented in accordance with the PRISMA 2020 guidance [<xref ref-type="bibr" rid="ref15">15</xref>].</p></sec><sec id="s2-4"><title>Data Extraction</title><p>Data were extracted independently by 2 reviewers (PN and SK) using a prespecified extraction form developed a priori. Extracted items included the following: study design and setting, population characteristics (sample size, clinical or simulated cohort, patient diagnoses, age range where reported, and sex where reported), technology type and algorithm or device description, primary outcomes and reported performance metrics (MAE, RMSE, time-in-target SpO<sub>2</sub>, mean SpO<sub>2</sub>, functional outcomes), signal quality and bias-related findings, LTOT applicability indicators (deployment context, real-world, or simulated validation), and risk of bias domain ratings. Disagreements on extracted values were resolved through discussion with the third reviewer (BP). The extraction form template is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-5"><title>Data Synthesis</title><p>Due to substantial heterogeneity in study designs, populations, and outcome metrics, AI studies reported MAE and RMSE from signal estimation experiments, automated system studies reported percentage time-in-target SpO<sub>2</sub> from clinical trials, 1 study reported only system architecture feasibility, and quantitative meta-analysis was not feasible. The findings are therefore presented through structured qualitative synthesis following the guidance of Popay et al [<xref ref-type="bibr" rid="ref17">17</xref>] for narrative synthesis in systematic reviews. A convergent integrated approach was applied: the findings were organized by the 3 RQs, with studies grouped by technology type and compared across common evaluation dimensions (accuracy, motion robustness, equity, real-world applicability, and clinical usability). This approach enables comparative insight across methodologically heterogeneous studies while preserving transparency about the evidence base. For evidence synthesis, studies were classified as AI-based when they used machine learning, deep learning, or statistical learning for SpO<sub>2</sub> estimation, signal-quality assessment, or predictive dosing; studies were classified as automated non-AI systems when they used deterministic rule-based or threshold-driven control without a learning component.</p></sec><sec id="s2-6"><title>Risk of Bias Assessment</title><p>Risk of bias was assessed for all 8 included studies using a framework adapted from the Risk of Bias in Non-randomized Studies of Interventions (ROBINS-I). Seven domains were evaluated for each study: (1) bias due to confounding: whether unmeasured variables such as demographics, comorbidities, or dataset composition could explain reported performance; (2) bias in participant selection: whether sampling frames were representative of real-world LTOT populations, including those with varied skin tones, activity levels, and comorbidities; (3) bias in the classification of interventions: whether the technology type and intervention mechanism were transparently described; (4) bias due to deviations from intended interventions: whether performance was assessed under realistic conditions or only under controlled or ideal-case scenarios; (5) bias due to missing data: whether data loss during monitoring (eg, motion-induced dropout, sensor disconnection) was reported and addressed; (6) bias in measurement of outcomes: whether performance metrics were validated against reference standards (eg, Bland-Altman analysis against co-oximetry); and (7) bias in selection of reported results: whether failure modes, edge cases, and suboptimal performance scenarios were reported alongside peak results. Domain ratings (low, moderate, serious, critical, no information) are visualized using the robvis R package (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). For AI-based studies, particular attention was given to algorithmic transparency and reproducibility as forms of bias not traditionally captured in clinical risk of bias tools.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Literature Search Results</title><p>The initial search yielded 926 records across database and supplementary publisher-platform searching, with an additional 2 articles identified through manual reference screening. Following deduplication, 912 unique records remained for title and abstract screening. Of these, 851 were excluded at the abstract stage for clearly not meeting eligibility criteria. Sixty-one full-text articles were retrieved and assessed for eligibility. Fifty-three were excluded at the full-text stage for the following reasons: focused solely on acute or surgical care settings with no LTOT applicability (n=18), lacked an AI or automation component and described passive monitoring only (n=14), not accessible in full text due to paywall restrictions (n=9), did not report SpO<sub>2</sub> as a primary outcome (n=7); were in non&#x2013;peer-reviewed format (conference abstract, thesis, or preprint) (n=3), and published in non-English language (n=2). These exclusion categories map directly to the eligibility criteria described above. A final total of 8 studies were included in the qualitative synthesis. The study selection process is illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 flow diagram of the systematic search and study selection process. Database or platform sources included PubMed, IEEE Xplore, Springer Link, ACM Digital Library, and supplementary MDPI publisher-level searches (N=926 total), with 2 additional records identified through manual reference screening. Source-specific records reported in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> were PubMed (n=111), IEEE Xplore (n=23), Springer Link (n=50), ACM Digital Library (n=731), and MDPI searches (n=11). After 16 duplicates were removed, 912 records were screened at title and abstract stage. Sixty-one reports were sought for retrieval; 9 reports could not be retrieved in full text, 52 full-text reports were assessed for eligibility, and 44 full-text reports were excluded with reasons. Eight studies were included in the qualitative synthesis: 5 AI-based and 3 automated non-AI systems.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="xmed_v7i1e76506_fig01.png"/></fig></sec><sec id="s3-2"><title>Characteristics of the Included Studies</title><p>This systematic review included 8 studies published between 2011 and 2024, spanning 5 countries (Spain, Denmark, Bangladesh or Qatar, Italy, and Colombia). Of the 8 studies, 5 evaluated AI-based SpO<sub>2</sub> monitoring systems and 3 evaluated automated non-AI oxygen delivery systems. Study designs included randomized crossover trials (n=1), pilot crossover studies (n=2), pilot usability studies (n=1), systematic and narrative reviews with model synthesis (n=2), technical validation studies (n=1), and prototype feasibility studies (n=1). Sample sizes ranged from 5 subjects in a conceptual pilot framework [<xref ref-type="bibr" rid="ref10">10</xref>] to 19 patients in a crossover randomized controlled trial [<xref ref-type="bibr" rid="ref12">12</xref>], with one study conducting a retrospective review of model performance across a curated multi-sensor dataset [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Population characteristics varied substantially between AI-based and automated system studies. The 3 automated system studies enrolled patients with COPD or chronic respiratory failure [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>], with mean patient ages in the range of 60&#x2010;75 years where reported. In contrast, the 5 AI-based studies predominantly used healthy volunteers, retrospective PPG datasets, or simulated signal environments, without enrollment of COPD-specific populations. This represents a significant gap: AI technical validation has largely not been conducted in the populations for whom LTOT is primarily indicated. Demographic reporting was limited across all studies; only Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] reported stratified analysis by skin tone, and none of the remaining 7 studies reported performance stratified by race, ethnicity, sex, or comorbidity profile (<xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics of included studies and key findings.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study</td><td align="left" valign="bottom">Country</td><td align="left" valign="bottom">Study design</td><td align="left" valign="bottom">Population and setting</td><td align="left" valign="bottom">Technology type</td><td align="left" valign="bottom">Primary outcomes and key metrics</td><td align="left" valign="bottom">Real-time O<sub>2</sub> titration</td><td align="left" valign="bottom">LTOT<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> context</td><td align="left" valign="bottom">Validation type</td><td align="left" valign="bottom">Overall risk of bias</td></tr></thead><tbody><tr><td align="left" valign="top">Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>], 2024</td><td align="left" valign="top">Spain</td><td align="left" valign="top">Systematic review + bias analysis</td><td align="left" valign="top">Multisensor dataset; skin tones varied; no COPD<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>-specific cohort</td><td align="left" valign="top">AI (GPR, DNN<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>)</td><td align="left" valign="top">MAE<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>: 0.57%; RMSE<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup>: 0.69%; skin tone-stratified bias reported</td><td align="left" valign="top">No (simulation only)</td><td align="left" valign="top">Conceptual or simulated</td><td align="left" valign="top">Systematic review + bias analysis</td><td align="left" valign="top">Low-moderate</td></tr><tr><td align="left" valign="top">Shuzan et al [<xref ref-type="bibr" rid="ref7">7</xref>], 2023</td><td align="left" valign="top">Bangladesh or Qatar</td><td align="left" valign="top">Technical validation</td><td align="left" valign="top">Healthy volunteers; simulated PPG<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup> dataset</td><td align="left" valign="top">AI (GPR, SVR<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup>)</td><td align="left" valign="top">MAE: 0.57%; RMSE: 0.98% for SpO<sub>2</sub><sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup>; also estimated respiration rate</td><td align="left" valign="top">No (simulated only)</td><td align="left" valign="top">Simulated</td><td align="left" valign="top">Technical validation</td><td align="left" valign="top">Moderate-serious</td></tr><tr><td align="left" valign="top">Arg&#x00FC;ello-Prada and Castillo Garc&#x00ED;a [<xref ref-type="bibr" rid="ref11">11</xref>], 2024</td><td align="left" valign="top">Colombia</td><td align="left" valign="top">Narrative review + model synthesis</td><td align="left" valign="top">Not applicable (review of published ML<sup><xref ref-type="table-fn" rid="table1fn9">i</xref></sup> models)</td><td align="left" valign="top">AI (ML motion detection classifiers)</td><td align="left" valign="top">Motion artifact detection performance; signal quality index analysis</td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table1fn10">j</xref></sup></td><td align="left" valign="top">General framework</td><td align="left" valign="top">Narrative review</td><td align="left" valign="top">Moderate-serious</td></tr><tr><td align="left" valign="top">Pascual-Salda&#x00F1;a et al [<xref ref-type="bibr" rid="ref10">10</xref>], 2024</td><td align="left" valign="top">Spain</td><td align="left" valign="top">Pilot framework</td><td align="left" valign="top">Five participants (diagnosis not reported); home ambulatory</td><td align="left" valign="top">AI (edge-AI, predictive dosing)</td><td align="left" valign="top">System architecture feasibility; predictive oxygen dosing concept</td><td align="left" valign="top">Yes: predictive dosing via local AI</td><td align="left" valign="top">Home LTOT (personalized)</td><td align="left" valign="top">Pilot framework</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">Pascual et al [<xref ref-type="bibr" rid="ref9">9</xref>], 2023</td><td align="left" valign="top">Spain</td><td align="left" valign="top">Prototype concept</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">AI (prototype NN<sup><xref ref-type="table-fn" rid="table1fn11">k</xref></sup> models)</td><td align="left" valign="top">Neural network architecture comparison; SpO<sub>2</sub> prediction feasibility</td><td align="left" valign="top">No</td><td align="left" valign="top">Conceptual or simulated</td><td align="left" valign="top">Prototype concept</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Sanchez-Morillo et al [<xref ref-type="bibr" rid="ref13">13</xref>], 2020</td><td align="left" valign="top">Spain</td><td align="left" valign="top">Pilot usability study</td><td align="left" valign="top">COPD and chronic respiratory failure patients; ambulatory home setting</td><td align="left" valign="top">Automation (iPOC<sup><xref ref-type="table-fn" rid="table1fn12">l</xref></sup> system)</td><td align="left" valign="top">Oxygenation stability during activity; patient satisfaction</td><td align="left" valign="top">Yes: physical activity-responsive dosing</td><td align="left" valign="top">Ambulatory or home LTOT</td><td align="left" valign="top">Pilot usability study</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">Hansen et al [<xref ref-type="bibr" rid="ref12">12</xref>], 2018</td><td align="left" valign="top">Denmark</td><td align="left" valign="top">Crossover RCT<sup><xref ref-type="table-fn" rid="table1fn13">m</xref></sup> (n=19)</td><td align="left" valign="top">COPD inpatients with acute exacerbation; hospital setting</td><td align="left" valign="top">Automation (O<sub>2</sub>matic closed-loop)</td><td align="left" valign="top">Time-in-target SpO<sub>2</sub>: 85.1% (auto) vs 46.6% (manual); reduced hypoxemia time</td><td align="left" valign="top">Yes: closed-loop titration</td><td align="left" valign="top">Hospital (acute COPD)</td><td align="left" valign="top">Crossover RCT</td><td align="left" valign="top">Low-moderate</td></tr><tr><td align="left" valign="top">Cirio and Nava [<xref ref-type="bibr" rid="ref14">14</xref>], 2011</td><td align="left" valign="top">Italy</td><td align="left" valign="top">Pilot crossover study (n=18)</td><td align="left" valign="top">LTOT patients during supervised exercise; hospital or rehabilitation setting</td><td align="left" valign="top">Automation (O<sub>2</sub> regulator device)</td><td align="left" valign="top">Mean SpO<sub>2</sub>: 95% (auto) vs 93% (manual); time below target: 171 s vs 340 s</td><td align="left" valign="top">Yes: automated titration during exercise</td><td align="left" valign="top">Exercise + home LTOT transition</td><td align="left" valign="top">Pilot crossover study</td><td align="left" valign="top">Serious</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>LTOT: long-term oxygen therapy.<bold> </bold></p></fn><fn id="table1fn2"><p><sup>b</sup>COPD: chronic obstructive pulmonary disease.</p></fn><fn id="table1fn3"><p><sup>c</sup>DNN: deep neural network.</p></fn><fn id="table1fn4"><p><sup>d</sup>MAE: mean absolute error.</p></fn><fn id="table1fn5"><p><sup>e</sup>RMSE: root mean square error.</p></fn><fn id="table1fn6"><p><sup>f</sup>PPG: photoplethysmography.</p></fn><fn id="table1fn7"><p><sup>g</sup>SVR: support vector regression.</p></fn><fn id="table1fn8"><p><sup>h</sup>SpO<sub>2</sub>: peripheral capillary oxygen saturation.</p></fn><fn id="table1fn9"><p><sup>i</sup>ML: machine learning.</p></fn><fn id="table1fn10"><p><sup>j</sup>N/A: not available.</p></fn><fn id="table1fn11"><p><sup>k</sup>NN: neural network.</p></fn><fn id="table1fn12"><p><sup>l</sup>iPOC: intelligent portable oxygen concentrator.</p></fn><fn id="table1fn13"><p><sup>m</sup>RCT: randomized controlled trial.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>AI Versus Non-AI Classification</title><p>Included studies were classified as either AI-based or non-AI automated based on the mechanism of the primary intervention. Studies were classified as AI-based if they used machine learning, deep learning, or statistical learning models, including Gaussian process regression, support vector regression, convolutional neural networks, or feedforward neural networks, for SpO<sub>2</sub> signal estimation, quality assessment, or predictive oxygen dosing. Studies were classified as non-AI automated if they used rule-based or threshold-driven closed-loop control systems that respond deterministically to real-time SpO<sub>2</sub> readings without a learning component. This classification is clinically and technically meaningful: AI systems can adapt to individual patterns over time, while rule-based automated systems cannot. These distinct mechanisms produce different capability profiles and different evidence gaps, which is the central comparative insight this review seeks to establish.</p></sec><sec id="s3-4"><title>RQ1: Technical Performance of AI-Based SpO<sub>2</sub> Monitoring Systems</title><p>The 5 AI-based studies reported performance using a range of metrics, reflecting the heterogeneity of their technical approaches. The 2 studies reporting direct SpO<sub>2</sub> estimation accuracy achieved comparable and clinically significant results: Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] reported an MAE of 0.57% and an RMSE of 0.69% using a Gaussian process regression model validated across multiple skin tones and sensor types, with Bland-Altman analysis confirming clinical-grade agreement with reference measurements. Shuzan et al [<xref ref-type="bibr" rid="ref7">7</xref>] reported an MAE of 0.57% and an RMSE of 0.98% using machine learning models (Gaussian process regression, support vector regression) estimating SpO<sub>2</sub> and respiratory rate from PPG signals, with performance assessed under controlled laboratory conditions using healthy volunteers. Both studies achieved accuracy below the commonly cited 2% MAE clinical threshold for pulse oximeter validation.</p><p>The remaining 3 AI-based studies did not report direct SpO<sub>2</sub> accuracy metrics. Arg&#x00FC;ello-Prada and Castillo Garc&#x00ED;a [<xref ref-type="bibr" rid="ref11">11</xref>] reviewed signal-quality-aware models capable of classifying PPG segments as artifact-contaminated or clean in real time without requiring a reference channel, a capability directly relevant to ambulatory LTOT where user movement is frequent and sensor contact variable. Pascual-Salda&#x00F1;a et al [<xref ref-type="bibr" rid="ref10">10</xref>] reported system architecture feasibility for a predictive edge-AI dosing framework, not SpO<sub>2</sub> accuracy per se, demonstrating viability in a 5-participant pilot but without quantified performance metrics. Pascual et al [<xref ref-type="bibr" rid="ref9">9</xref>] compared lightweight neural network architectures for SpO<sub>2</sub> prediction but lacked complete model validation or standardized error reporting, limiting the interpretation of the results.</p><p>Regarding motion artifact robustness specifically, of the 5 AI-based studies, only Arg&#x00FC;ello-Prada and Castillo Garc&#x00ED;a [<xref ref-type="bibr" rid="ref11">11</xref>] directly addressed motion artifact as a primary outcome, reviewing reference signal-less classification models that flagged corrupted segments with reported accuracy. Shuzan et al [<xref ref-type="bibr" rid="ref7">7</xref>] addressed motion robustness indirectly through feature engineering and model optimization but did not isolate motion artifact performance as a distinct outcome. Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] included signal noise considerations in their bias analysis but did not report motion-specific performance metrics. The remaining 2 AI studies [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>] did not address motion artifact handling. None of the 3 non-AI automated systems incorporated any preprocessing for motion artifact correction, operating under the assumption of continuous, valid SpO<sub>2</sub> input [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>].</p></sec><sec id="s3-5"><title>RQ2: Clinical Performance of Automated Oxygen Delivery Systems</title><p>The 3 automated system studies were evaluated using clinical efficacy metrics centered on SpO<sub>2</sub> target maintenance. Hansen et al [<xref ref-type="bibr" rid="ref12">12</xref>] conducted a randomized crossover trial (n=19) comparing the O<sub>2</sub>matic closed-loop system to manual oxygen titration in patients hospitalized with acute COPD exacerbation. O<sub>2</sub>matic maintained patients within the prescribed SpO<sub>2</sub> target range 85.1% of the time versus 46.6% under manual titration, a clinically meaningful difference accompanied by significantly reduced time spent in hypoxemia. The system was described by nursing staff as safe and requiring substantially less manual intervention, suggesting a favorable clinician usability profile. Cirio and Nava [<xref ref-type="bibr" rid="ref14">14</xref>] evaluated an automated oxygen titration device during supervised exercise in patients on LTOT (n=18), reporting mean SpO<sub>2</sub> of 95% (automated) versus 93% (manual) and time below the target SpO<sub>2</sub> threshold of 171 seconds versus 340 seconds under manual control. Respiratory therapist intervention time was reduced with the automated system, supporting feasibility in an outpatient exercise and rehabilitation context. Sanchez-Morillo et al [<xref ref-type="bibr" rid="ref13">13</xref>] evaluated the iPOC system in an ambulatory home LTOT setting in patients with COPD and chronic respiratory failure, demonstrating improved oxygenation stability during physical activity and positive patient satisfaction ratings compared to conventional oxygen concentrators. Performance metrics were primarily functional rather than algorithmic.</p><p>Regarding usability across all included studies, no study applied a validated usability assessment instrument (eg, the System Usability Scale), and only 2 studies [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>] explicitly reported clinician or patient perception of system burden and reliability. No AI-based study reported usability outcomes of any kind, consistent with their predominantly preclinical or simulation-stage nature. The absence of formal human factors evaluation across both categories of technology represents a gap in the evidence base for clinical translation.</p></sec><sec id="s3-6"><title>RQ3: Equity and Real-World LTOT Applicability</title><p>Of the 8 included studies, one explicitly addressed skin tone measurement bias: Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] performed stratified analysis across skin pigmentation groups and identified measurable differences in SpO<sub>2</sub> prediction error as a function of skin tone, representing the only equity-directed finding in this review. The remaining 7 studies did not report any demographic stratification by skin tone, race, ethnicity, age, or comorbidity profile. Three of the 5 AI-based studies, Shuzan et al [<xref ref-type="bibr" rid="ref7">7</xref>], Arg&#x00FC;ello-Prada and Castillo Garc&#x00ED;a [<xref ref-type="bibr" rid="ref11">11</xref>], and Pascual et al [<xref ref-type="bibr" rid="ref9">9</xref>], relied on curated or publicly available PPG datasets; none of these datasets included demographic stratification. This finding is particularly consequential given well-documented evidence that pulse oximeters systematically overestimate SpO<sub>2</sub> in individuals with darker skin pigmentation [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>], raising the risk that AI models trained on demographically homogeneous data will propagate and amplify existing measurement inequities.</p><p>Regarding real-world LTOT deployment readiness: 2 studies demonstrated or conceptualized real-world home applicability: Sanchez-Morillo et al [<xref ref-type="bibr" rid="ref13">13</xref>] in an ambulatory home LTOT context and Pascual-Salda&#x00F1;a et al [<xref ref-type="bibr" rid="ref10">10</xref>] in a conceptual edge-AI framework for home deployment. The O<sub>2</sub>matic system [<xref ref-type="bibr" rid="ref12">12</xref>] and the automated titration device by Cirio and Nava [<xref ref-type="bibr" rid="ref14">14</xref>] were both assessed in hospital or supervised exercise environments, limiting their direct applicability to unsupervised home LTOT without further adaptation. The 3 technically strongest AI studies [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref11">11</xref>] were conducted entirely in simulated or retrospective contexts with no home LTOT deployment component. No study provided longitudinal data on system performance over the timescales relevant to LTOT, which is defined by supplemental oxygen use of 15 or more hours per day.</p></sec><sec id="s3-7"><title>Risk of Bias Assessment Results</title><sec id="s3-7-1"><title>Bias Due to Confounding</title><p>Of the 8 included studies, 3 AI-based studies [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref11">11</xref>] were rated at moderate risk of confounding bias, having used publicly available PPG datasets (eg, PhysioNet) that lacked full clinical context or demographic diversity. Only Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] performed subgroup analysis by skin tone, identifying measurable differences in prediction error by pigmentation, reducing confounding risk in that domain. The automated system studies [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>] were rated at moderate-to-serious confounding risk, as none stratified outcomes by comorbidities, disease severity, or socioeconomic factors.</p></sec><sec id="s3-7-2"><title>Bias in Participant Selection</title><p>Shuzan et al [<xref ref-type="bibr" rid="ref7">7</xref>] and Arg&#x00FC;ello-Prada and Castillo Garc&#x00ED;a [<xref ref-type="bibr" rid="ref11">11</xref>] trained or reviewed models on precurated, high-quality PPG signals from controlled environments, raising serious risk of selection bias: these datasets overrepresent idealized sensor conditions and exclude the noisy, variable signals characteristic of real-world home-based LTOT use. Automated system studies [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>] were tested in small, specific clinical populations, limiting generalizability to the broader, demographically diverse LTOT population. Pascual et al [<xref ref-type="bibr" rid="ref9">9</xref>] was rated at serious selection bias risk due to prototype-only testing with no real-world population sampling.</p></sec><sec id="s3-7-3"><title>Bias Due to Missing Data</title><p>Risk was generally low to moderate. Most AI studies reported complete-case analysis but did not address data loss due to motion, sensor disconnection, or prolonged monitoring interruptions. Only the iPOC system [<xref ref-type="bibr" rid="ref13">13</xref>] explicitly described dropout detection and reinitialization protocols. No AI study conducted sensitivity analyses for missing data mechanisms (eg, distinguishing MCAR from MAR missingness), limiting the robustness of performance claims under real-world signal conditions.</p></sec><sec id="s3-7-4"><title>Bias in Measurement of Outcomes</title><p>Of the AI-based studies, only Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] performed Bland-Altman analysis, enabling the assessment of systematic bias and limits of agreement relative to reference measurements, the clinical standard for oximeter validation. The remaining 4 AI studies reported MAE or RMSE without reference standard validation, and Pascual et al [<xref ref-type="bibr" rid="ref9">9</xref>] and Pascual-Salda&#x00F1;a et al [<xref ref-type="bibr" rid="ref10">10</xref>] reported no quantitative SpO<sub>2</sub> performance metrics at all. Among automated system studies, none validated performance against arterial blood gas (ABG) reference measurements; performance was assessed via SpO<sub>2</sub>-based metrics despite the acknowledged limitations of oximetry as its own reference.</p></sec><sec id="s3-7-5"><title>Bias Due to Algorithmic Transparency and Reproducibility</title><p>Transparency varied substantially. Pascual-Salda&#x00F1;a et al [<xref ref-type="bibr" rid="ref10">10</xref>] described model architectures and training methodology in sufficient detail to support replication. Pascual et al [<xref ref-type="bibr" rid="ref9">9</xref>] did not provide source code, training parameters, or hyperparameter details, constituting serious reproducibility risk. Arg&#x00FC;ello-Prada and Castillo Garc&#x00ED;a [<xref ref-type="bibr" rid="ref11">11</xref>] conducted a review of published models, inheriting the variable transparency of those source studies. Only Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] reported demographic stratification, external validation across sensor platforms, and sufficient methodological detail to enable full reproducibility assessment.</p></sec><sec id="s3-7-6"><title>Bias in Selection of Reported Results</title><p>Of the 8 included studies, only Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] and Hansen et al [<xref ref-type="bibr" rid="ref12">12</xref>] discussed performance under suboptimal or adverse conditions. Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] reported bias analysis across varying skin tone groups, including cases of elevated prediction error. Hansen et al [<xref ref-type="bibr" rid="ref12">12</xref>] reported adverse events and time spent outside the target range alongside efficacy results. The remaining 6 studies did not report failure modes, edge-case performance, or conditions under which the system underperformed, limiting the understanding of system robustness. The overall risk of bias ranged from low-moderate (Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>], Hansen et al [<xref ref-type="bibr" rid="ref12">12</xref>]) to serious (Pascual et al [<xref ref-type="bibr" rid="ref9">9</xref>], Cirio and Nava [<xref ref-type="bibr" rid="ref14">14</xref>]), with the highest risks concentrated in participant selection, algorithmic transparency, and outcome measurement. Future studies should adopt CONSORT-AI (Consolidated Standards of Reporting Trials&#x2013;Artificial Intelligence) [<xref ref-type="bibr" rid="ref18">18</xref>] and DECIDE-AI (Developmental and Exploratory Clinical Investigations of Decision Support Systems Driven by Artificial Intelligence) [<xref ref-type="bibr" rid="ref19">19</xref>] reporting guidelines, report failure modes transparently, and pursue model reproducibility through open-source practices (<xref ref-type="table" rid="table2">Table 2</xref>).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Risk of Bias Assessment (Risk of Bias in Non-randomized Studies of Interventions [ROBINS-I] domains).</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study</td><td align="left" valign="bottom">Confounding</td><td align="left" valign="bottom">Selection</td><td align="left" valign="bottom">Classification of interventions</td><td align="left" valign="bottom">Deviations from interventions</td><td align="left" valign="bottom">Missing data</td><td align="left" valign="bottom">Outcome measurement</td><td align="left" valign="bottom">Reporting bias</td><td align="left" valign="bottom">Overall risk</td></tr></thead><tbody><tr><td align="left" valign="top">Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>], 2024</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low-moderate</td></tr><tr><td align="left" valign="top">Shuzan et al [<xref ref-type="bibr" rid="ref7">7</xref>], 2023</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Low</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate-serious</td></tr><tr><td align="left" valign="top">Arg&#x00FC;ello-Prada and Castillo [<xref ref-type="bibr" rid="ref11">11</xref>], 2024</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate-serious</td></tr><tr><td align="left" valign="top">Pascual-Salda&#x00F1;a et al [<xref ref-type="bibr" rid="ref10">10</xref>], 2024</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Low</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">Pascual et al [<xref ref-type="bibr" rid="ref9">9</xref>], 2023</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Sanchez-Morillo et al [<xref ref-type="bibr" rid="ref13">13</xref>], 2020</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">Hansen et al [<xref ref-type="bibr" rid="ref12">12</xref>], 2018</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Low-moderate</td></tr><tr><td align="left" valign="top">Cirio and Nava [<xref ref-type="bibr" rid="ref14">14</xref>], 2011</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td></tr></tbody></table></table-wrap></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This systematic review synthesized findings from 8 peer-reviewed studies examining AI-driven and automated systems for continuous SpO<sub>2</sub> monitoring relevant to LTOT. Across the 3 research questions, a consistent pattern emerged: AI-based systems demonstrated superior technical performance under controlled conditions, particularly in SpO<sub>2</sub> estimation accuracy and signal quality management, while non-AI automated systems demonstrated superior clinical readiness and deployment evidence in real-world patient populations.</p><p>Addressing RQ1, AI models achieved SpO<sub>2</sub> estimation accuracy below the 2% MAE clinical threshold in the 2 studies that reported this metric [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>], with Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] additionally providing Bland-Altman validation and skin tone&#x2013;stratified bias analysis. Motion artifact handling was directly addressed only by Arg&#x00FC;ello-Prada and Castillo Garc&#x00ED;a [<xref ref-type="bibr" rid="ref11">11</xref>], whose review of reference signal-less classifiers supports real-time signal quality assessment as a technically viable approach. These capabilities address specific limitations of conventional pulse oximetry in ambulatory settings. However, none of the AI-based systems reviewed have undergone real-world validation in patients with COPD or other diagnoses for which LTOT is indicated, and none have been tested over the timescales or under the environmental variability characteristic of home-based LTOT.</p><p>Addressing RQ2, the automated system evidence is more mature clinically but narrower technically. Hansen et al [<xref ref-type="bibr" rid="ref12">12</xref>] provided the strongest evidence, demonstrating a 38.5-percentage-point improvement in time-in-target SpO<sub>2</sub> over manual titration in a randomized crossover design. Sanchez-Morillo et al [<xref ref-type="bibr" rid="ref13">13</xref>] and Cirio and Nava [<xref ref-type="bibr" rid="ref14">14</xref>] provided supportive pilot evidence in ambulatory and exercise settings, respectively. These systems respond to current SpO<sub>2</sub> readings within predefined thresholds; they cannot learn from patient history, anticipate physiological change, or adapt to signal artifact. This reactive limitation is precisely where AI-based approaches offer theoretical advantages, suggesting that hybrid systems combining closed-loop automation with AI signal intelligence and predictive modeling represent the most promising development direction.</p><p>Addressing RQ3, the equity evidence is sparse to the point of being a critical gap. Only Cabanas et al [<xref ref-type="bibr" rid="ref8">8</xref>] performed skin tone&#x2013;stratified analysis among all 8 included studies, and no study reported performance stratified by race, ethnicity, age, or comorbidity, despite the well-documented racial disparities in pulse oximetry accuracy [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>] and the disproportionate burden of chronic respiratory disease in underrepresented populations. AI models trained predominantly on demographically homogeneous datasets risk encoding and amplifying existing measurement inequities when deployed in diverse LTOT populations. This finding aligns with and extends broader concerns in the literature about algorithmic fairness in clinical AI [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>The maturity gap between AI and automated systems warrants explicit acknowledgment. Several AI models, including those by Pascual et al [<xref ref-type="bibr" rid="ref9">9</xref>] and Pascual-Salda&#x00F1;a et al [<xref ref-type="bibr" rid="ref10">10</xref>], remain at proof-of-concept or early prototyping stages, while O<sub>2</sub>matic has undergone clinical trial evaluation. This does not reflect a deficiency of AI as a technological approach, but rather the earlier stage of clinical translation for AI-based SpO<sub>2</sub> systems compared to rule-based closed-loop devices.</p></sec><sec id="s4-2"><title>Strengths and Limitations</title><p>This review&#x2019;s principal strengths are its explicit focus on the technical prerequisites for real-world LTOT deployment, including motion robustness, skin tone bias, and signal stability, and its comparative framing of AI-driven versus non-AI automated systems as complementary rather than competing approaches. The use of structured ROBINS-I adapted assessment, PRISMA 2020 reporting, and RQ-aligned synthesis provides a reproducible and transparent methodological foundation.</p><p>Several limitations must be acknowledged. First, the total number of included studies (n=8) is small, reflecting the novelty of the field and the scarcity of clinical trials for AI-enhanced LTOT systems; conclusions should be interpreted with this constraint in mind.</p><p>Second, restriction to English-language open-access literature may have introduced language and publication bias, potentially excluding relevant studies published in other languages or behind paywalls. Among the 9 reports that could not be retrieved in full text, article-level metadata were available for 5 records. These records appeared to address related areas including machine learning classification of COPD using pulse oximetry, smartphone-based SpO<sub>2</sub> estimation, deep learning estimation of SpO<sub>2</sub> from PPG signals, machine learning assessment of pulse oximeter response time, and prototype automated oxygen regulation using SpO<sub>2</sub> feedback. This suggests that the inaccessible literature may have contained additional early-stage engineering, mobile-health, and device-development studies relevant to AI-enabled or automated oxygen monitoring. However, because the full texts could not be assessed, these records were not included in data extraction, risk-of-bias assessment, or synthesis. Their omission may mean that this review underrepresents low-cost prototypes, conference-proceeding technologies, and technical validation studies, but available metadata do not suggest that they would resolve the central evidence gaps identified in this review: limited real-world LTOT validation, sparse demographic equity reporting, and lack of longitudinal home-based evaluation.</p><p>Third, outcome heterogeneity prevented meta-analysis; the findings must be interpreted at the level of individual study design and setting. Fourth, this review did not systematically address regulatory approval pathways, cost-effectiveness, or implementation science considerations, all of which are necessary for translation into routine clinical practice. Fifth, this review was not prospectively registered in PROSPERO; future updates should consider prospective registration to enhance protocol transparency.</p></sec><sec id="s4-3"><title>Implications for Clinical Practice and Future Research</title><p>The convergent finding of this review, namely that AI systems and automated systems occupy complementary capability niches, has a direct implication for the clinical development agenda: hybrid architectures that pair closed-loop automated oxygen titration with AI-based signal quality assurance and predictive dosing represent the most promising near-term direction. For such systems to be clinically deployable, real-world longitudinal validation is needed in ambulatory LTOT populations that include patients across the demographic spectrum of LTOT eligibility, with prespecified equity outcomes and failure mode reporting. Edge-computing approaches [<xref ref-type="bibr" rid="ref10">10</xref>] that enable local AI inference without cloud dependency are particularly relevant for home-based and resource-limited settings. Standardized performance reporting, including demographic stratification, motion-specific accuracy metrics, and validation against ABG or co-oximetry reference standards, should be adopted across the field to facilitate cross-study comparability and regulatory assessment.</p></sec><sec id="s4-4"><title>Conclusions</title><p>This systematic review identified 8 peer-reviewed studies addressing AI-driven and automated systems for continuous SpO<sub>2</sub> monitoring in LTOT. AI-based systems demonstrated strong technical performance in SpO<sub>2</sub> estimation accuracy [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>] and signal quality management under motion [<xref ref-type="bibr" rid="ref11">11</xref>] and offer a promising pathway to personalized, predictive oxygen therapy. Automated non-AI systems demonstrated clinical readiness, with O<sub>2</sub>matic producing a clinically meaningful improvement of 38.5 percentage points in time-in-target SpO<sub>2</sub> compared to manual titration [<xref ref-type="bibr" rid="ref12">12</xref>]. The central finding of this review is that these 2 technology types address different capability gaps and are best understood as complementary rather than competing approaches. Significant gaps persist across both: limited real-world validation in LTOT populations, near-total absence of demographic equity data, inconsistent outcome reporting, and inadequate algorithmic transparency. Closing these gaps through longitudinal clinical trials in diverse LTOT populations, adoption of CONSORT-AI and DECIDE-AI reporting standards, and development of hybrid AI-automated architectures will determine whether these technologies fulfill their promise of safer, more equitable, and more responsive long-term oxygen therapy.</p></sec></sec></body><back><ack><p>AI-assisted tools, including Claude and OpenAI/Codex, were used during revision for language editing, formatting checks, and organization of reviewer responses. The authors independently reviewed and verified all content, references, data interpretation, and final wording and take full responsibility for the manuscript.</p><p>PN is not currently affiliated with any institution and is an independent researcher.</p></ack><notes><sec><title>Funding</title><p>The authors received no financial support for the research, authorship, or publication of this article.</p></sec><sec><title>Data Availability</title><p>This systematic review is based entirely on previously published, publicly available studies. No primary data were collected or generated. The prespecified data extraction form and complete database search strings are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The risk of bias analysis dataset and R code are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. The completed PRISMA-S checklist is provided in <xref ref-type="supplementary-material" rid="app4">Checklist 2</xref>. The completed PRISMA 2020 checklist is provided in <xref ref-type="supplementary-material" rid="app3">Checklist 1</xref>.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: PN, Suman Kadariya</p><p>Data curation: PN</p><p>Formal analysis: PN</p><p>Investigation: PN, Suman Kadariya, BP</p><p>Methodology: PN, Suman Kadariya</p><p>Supervision: Sujan Kadariya</p><p>Validation: Sujan Kadariya, BP</p><p>Writing &#x2013; original draft: PN</p><p>Writing &#x2013; review and editing: PN, Suman Kadariya, BP, Sujan Kadariya</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">6MWT</term><def><p>6-minute walk test</p></def></def-item><def-item><term id="abb2">ABG</term><def><p>arterial blood gas</p></def></def-item><def-item><term id="abb3">CNN</term><def><p>convolutional neural network</p></def></def-item><def-item><term id="abb4">CONSORT-AI</term><def><p>Consolidated Standards of Reporting Trials&#x2013;Artificial Intelligence</p></def></def-item><def-item><term id="abb5">COPD</term><def><p>chronic obstructive pulmonary disease</p></def></def-item><def-item><term id="abb6">DECIDE-AI</term><def><p>Developmental and Exploratory Clinical Investigations of Decision Support Systems Driven by Artificial Intelligence</p></def></def-item><def-item><term id="abb7">ECG</term><def><p>electrocardiogram</p></def></def-item><def-item><term id="abb8">iPOC</term><def><p>intelligent portable oxygen concentrator</p></def></def-item><def-item><term id="abb9">LTOT</term><def><p>long-term oxygen therapy</p></def></def-item><def-item><term id="abb10">MAE</term><def><p>mean absolute error</p></def></def-item><def-item><term id="abb11">MRC</term><def><p>Medical Research Council</p></def></def-item><def-item><term id="abb12">NOTT</term><def><p>Nocturnal Oxygen Therapy Trial</p></def></def-item><def-item><term id="abb13">PPG</term><def><p>photoplethysmography</p></def></def-item><def-item><term id="abb14">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb15">PRISMA-S</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses&#x2013;Search</p></def></def-item><def-item><term id="abb16">RMSE</term><def><p>root mean square error</p></def></def-item><def-item><term id="abb17">ROBINS-I</term><def><p>Risk of Bias in Non-randomized Studies of Interventions</p></def></def-item><def-item><term id="abb18">RQ</term><def><p>research question</p></def></def-item><def-item><term id="abb19">SpO<sub>2</sub></term><def><p>peripheral capillary oxygen saturation</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>Global health estimates: life expectancy and leading causes of death and disability</article-title><source>World Health Organization</source><year>2020</year><access-date>2026-09-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/data/gho/data/themes/mortality-and-global-health-estimates">https://www.who.int/data/gho/data/themes/mortality-and-global-health-estimates</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>Nocturnal Oxygen Therapy Trial Group</collab></person-group><article-title>Continuous or nocturnal oxygen therapy in hypoxemic chronic obstructive lung disease: a clinical trial</article-title><source>Ann Intern Med</source><year>1980</year><month>09</month><volume>93</volume><issue>3</issue><fpage>391</fpage><lpage>398</lpage><pub-id pub-id-type="doi">10.7326/0003-4819-93-3-391</pub-id><pub-id pub-id-type="medline">6776858</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><article-title>Long term domiciliary oxygen therapy in chronic hypoxic cor pulmonale complicating chronic bronchitis and emphysema</article-title><source>Lancet</source><year>1981</year><month>03</month><day>28</day><volume>1</volume><issue>8222</issue><fpage>681</fpage><lpage>686</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(81)91970-X</pub-id><pub-id pub-id-type="medline">6110912</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="report"><article-title>Global strategy for the diagnosis, management, and prevention of chronic obstructive pulmonary disease</article-title><year>2024</year><access-date>2026-09-15</access-date><publisher-name>Global Initiative for Chronic Obstructive Lung Disease (GOLD)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://goldcopd.org/wp-content/uploads/2024/02/GOLD-2024_v1.2-11Jan24_WMV.pdf">https://goldcopd.org/wp-content/uploads/2024/02/GOLD-2024_v1.2-11Jan24_WMV.pdf</ext-link></comment></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sjoding</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Dickson</surname><given-names>RP</given-names> </name><name name-style="western"><surname>Iwashyna</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Gay</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Valley</surname><given-names>TS</given-names> </name></person-group><article-title>Racial bias in pulse oximetry measurement</article-title><source>N Engl J Med</source><year>2020</year><month>12</month><day>17</day><volume>383</volume><issue>25</issue><fpage>2477</fpage><lpage>2478</lpage><pub-id pub-id-type="doi">10.1056/NEJMc2029240</pub-id><pub-id pub-id-type="medline">33326721</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fawzy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>TD</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Racial and ethnic discrepancy in pulse oximetry and delayed identification of treatment eligibility among patients with COVID-19</article-title><source>JAMA Intern Med</source><year>2022</year><month>07</month><day>1</day><volume>182</volume><issue>7</issue><fpage>730</fpage><lpage>738</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2022.1906</pub-id><pub-id pub-id-type="medline">35639368</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shuzan</surname><given-names>MNI</given-names> </name><name name-style="western"><surname>Chowdhury</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Chowdhury</surname><given-names>MEH</given-names> </name><etal/></person-group><article-title>Machine learning-based respiration rate and blood oxygen saturation estimation using photoplethysmogram signals</article-title><source>Bioengineering (Basel)</source><year>2023</year><month>01</month><day>28</day><volume>10</volume><issue>2</issue><fpage>167</fpage><pub-id pub-id-type="doi">10.3390/bioengineering10020167</pub-id><pub-id pub-id-type="medline">36829661</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cabanas</surname><given-names>AM</given-names> </name><name name-style="western"><surname>S&#x00E1;ez</surname><given-names>N</given-names> </name><name name-style="western"><surname>Collao-Caiconte</surname><given-names>PO</given-names> </name><etal/></person-group><article-title>Evaluating AI methods for pulse oximetry: performance, clinical accuracy, and comprehensive bias analysis</article-title><source>Bioengineering (Basel)</source><year>2024</year><month>10</month><day>24</day><volume>11</volume><issue>11</issue><fpage>1061</fpage><pub-id pub-id-type="doi">10.3390/bioengineering11111061</pub-id><pub-id pub-id-type="medline">39593722</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pascual</surname><given-names>H</given-names> </name><name name-style="western"><surname>Masip-Bruin</surname><given-names>X</given-names> </name><name name-style="western"><surname>Alonso</surname><given-names>A</given-names> </name><name name-style="western"><surname>Blanco</surname><given-names>I</given-names> </name></person-group><article-title>Analyzing distinct neural network models for oxygen saturation prediction towards a personalized COPD management</article-title><source>IEEE Int Conf e-Sci</source><year>2023</year><fpage>1</fpage><lpage>8</lpage><pub-id pub-id-type="doi">10.1109/e-Science58273.2023.10254844</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pascual-Salda&#x00F1;a</surname><given-names>H</given-names> </name><name name-style="western"><surname>Masip-Bruin</surname><given-names>X</given-names> </name><name name-style="western"><surname>Asensio</surname><given-names>A</given-names> </name><name name-style="western"><surname>Alonso</surname><given-names>A</given-names> </name><name name-style="western"><surname>Blanco</surname><given-names>I</given-names> </name></person-group><article-title>Innovative predictive approach towards a personalized oxygen dosing system</article-title><source>Sensors (Basel)</source><year>2024</year><month>01</month><day>24</day><volume>24</volume><issue>3</issue><fpage>764</fpage><pub-id pub-id-type="doi">10.3390/s24030764</pub-id><pub-id pub-id-type="medline">38339481</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arg&#x00FC;ello-Prada</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Castillo Garc&#x00ED;a</surname><given-names>JF</given-names> </name></person-group><article-title>Machine learning applied to reference signal-less detection of motion artifacts in photoplethysmographic signals: a review</article-title><source>Sensors (Basel)</source><year>2024</year><month>11</month><day>9</day><volume>24</volume><issue>22</issue><fpage>7193</fpage><pub-id pub-id-type="doi">10.3390/s24227193</pub-id><pub-id pub-id-type="medline">39598970</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hansen</surname><given-names>EF</given-names> </name><name name-style="western"><surname>Hove</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Bech</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Jensen</surname><given-names>JUS</given-names> </name><name name-style="western"><surname>Kallemose</surname><given-names>T</given-names> </name><name name-style="western"><surname>Vestbo</surname><given-names>J</given-names> </name></person-group><article-title>Automated oxygen control with O2matic<sup>&#x00AE;</sup> during admission with exacerbation of COPD</article-title><source>Int J Chron Obstruct Pulmon Dis</source><year>2018</year><volume>13</volume><fpage>3997</fpage><lpage>4003</lpage><pub-id pub-id-type="doi">10.2147/COPD.S183762</pub-id><pub-id pub-id-type="medline">30587955</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sanchez-Morillo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Mu&#x00F1;oz-Zara</surname><given-names>P</given-names> </name><name name-style="western"><surname>Lara-Do&#x00F1;a</surname><given-names>A</given-names> </name><name name-style="western"><surname>Leon-Jimenez</surname><given-names>A</given-names> </name></person-group><article-title>Automated home oxygen delivery for patients with COPD and respiratory failure: a new approach</article-title><source>Sensors (Basel)</source><year>2020</year><month>02</month><day>20</day><volume>20</volume><issue>4</issue><fpage>1178</fpage><pub-id pub-id-type="doi">10.3390/s20041178</pub-id><pub-id pub-id-type="medline">32093418</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cirio</surname><given-names>S</given-names> </name><name name-style="western"><surname>Nava</surname><given-names>S</given-names> </name></person-group><article-title>Pilot study of a new device to titrate oxygen flow in hypoxic patients on long-term oxygen therapy</article-title><source>Respir Care</source><year>2011</year><month>04</month><volume>56</volume><issue>4</issue><fpage>429</fpage><lpage>434</lpage><pub-id pub-id-type="doi">10.4187/respcare.00983</pub-id><pub-id pub-id-type="medline">21255511</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>McKenzie</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Bossuyt</surname><given-names>PM</given-names> </name><etal/></person-group><article-title>The PRISMA 2020 statement: an updated guideline for reporting systematic reviews</article-title><source>BMJ</source><year>2021</year><month>03</month><day>29</day><volume>372</volume><fpage>n71</fpage><pub-id pub-id-type="doi">10.1136/bmj.n71</pub-id><pub-id pub-id-type="medline">33782057</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rethlefsen</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Kirtley</surname><given-names>S</given-names> </name><name name-style="western"><surname>Waffenschmidt</surname><given-names>S</given-names> </name><etal/></person-group><article-title>PRISMA-S: an extension to the PRISMA Statement for reporting literature searches in systematic reviews</article-title><source>Syst Rev</source><year>2021</year><month>01</month><day>26</day><volume>10</volume><issue>1</issue><fpage>39</fpage><pub-id pub-id-type="doi">10.1186/s13643-020-01542-z</pub-id><pub-id pub-id-type="medline">33499930</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Popay</surname><given-names>J</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sowden</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Guidance on the conduct of narrative synthesis in systematic reviews: a product from the ESRC methods programme</article-title><year>2006</year><access-date>2026-09-15</access-date><publisher-name>Lancaster University</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://database.inahta.org/article/2684?">https://database.inahta.org/article/2684?</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cruz Rivera</surname><given-names>S</given-names> </name><name name-style="western"><surname>Moher</surname><given-names>D</given-names> </name><name name-style="western"><surname>Calvert</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Denniston</surname><given-names>AK</given-names> </name><collab>SPIRIT-AI and CONSORT-AI Working Group</collab></person-group><article-title>Reporting guidelines for clinical trial reports for interventions involving artificial intelligence: the CONSORT-AI extension</article-title><source>Nat Med</source><year>2020</year><month>09</month><volume>26</volume><issue>9</issue><fpage>1364</fpage><lpage>1374</lpage><pub-id pub-id-type="doi">10.1038/s41591-020-1034-x</pub-id><pub-id pub-id-type="medline">32908283</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vasey</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nagendran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Campbell</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Reporting guideline for the early stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI</article-title><source>BMJ</source><year>2022</year><month>05</month><day>18</day><volume>377</volume><fpage>e070904</fpage><pub-id pub-id-type="doi">10.1136/bmj-2022-070904</pub-id><pub-id pub-id-type="medline">35584845</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="report"><article-title>Artificial intelligence/machine learning (AI/ML)-based software as a medical device (SaMD) action plan</article-title><year>2021</year><month>01</month><access-date>2026-09-15</access-date><publisher-name>U.S. Food and Drug Administration (FDA)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.fda.gov/media/145022/download">https://www.fda.gov/media/145022/download</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Inclusion and exclusion criteria, search strategies, and data extraction form.</p><media xlink:href="xmed_v7i1e76506_app1.docx" xlink:title="DOCX File, 14 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Risk of bias materials.</p><media xlink:href="xmed_v7i1e76506_app2.docx" xlink:title="DOCX File, 236 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 1</label><p>PRISMA 2020 checklist.</p><media xlink:href="xmed_v7i1e76506_app3.pdf" xlink:title="PDF File, 71 KB"/></supplementary-material><supplementary-material id="app4"><label>Checklist 2</label><p>PRISMA-S checklist.</p><media xlink:href="xmed_v7i1e76506_app4.docx" xlink:title="DOCX File, 3677 KB"/></supplementary-material></app-group></back></article>