<?xml version='1.0' encoding='UTF-8'?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.2 20190208//EN" "JATS-journalpublishing1.dtd"[]>
<article xml:lang="en" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" dtd-version="1.2" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">IJPDS</journal-id>
<journal-title-group>
<journal-title>International Journal of Population Data Science</journal-title>
<abbrev-journal-title>IJPDS</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2399-4908</issn>
<publisher>
<publisher-name>Swansea University</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.23889/ijpds.v11i1.3201</article-id>
<article-id pub-id-type="publisher-id">11:1:3201</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Population Data Science</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>The need for fully-effective federated analytics of data sources for clinical trials</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author"><name><surname>Fabiane</surname><given-names initials="SM">Stella Maris</given-names></name><xref ref-type="aff" rid="affil-1"><sup>1</sup></xref><xref ref-type="aff" rid="affil-10">&#x2020;</xref></contrib>
<contrib contrib-type="author"><name><surname>Love</surname><given-names initials="SB">Sharon B.</given-names></name><xref ref-type="aff" rid="affil-2"><sup>2</sup></xref><xref ref-type="corresp" rid="correspondingAurthor">*</xref><xref ref-type="aff" rid="affil-10">&#x2020;</xref></contrib>
<contrib contrib-type="author"><name><surname>Fisher</surname><given-names initials="D">David</given-names></name><xref ref-type="aff" rid="affil-3"><sup>3</sup></xref></contrib>
<contrib contrib-type="author"><name><surname>White</surname><given-names initials="IR">Ian R.</given-names></name><xref ref-type="aff" rid="affil-4"><sup>4</sup></xref></contrib>
<contrib contrib-type="author"><name><surname>Tierney</surname><given-names initials="JF">Jayne F.</given-names></name><xref ref-type="aff" rid="affil-1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author"><name><surname>Dampney</surname><given-names initials="C">Catherine</given-names></name><xref ref-type="aff" rid="affil-5"><sup>5</sup></xref></contrib>
<contrib contrib-type="author"><name><surname>Asselbergs</surname><given-names initials="FW">Folkert W.</given-names></name><xref ref-type="aff" rid="affil-6"><sup>6</sup></xref></contrib>
<contrib contrib-type="author"><name><surname>Quinlan</surname><given-names initials="PR">Philip R.</given-names></name><xref ref-type="aff" rid="affil-7"><sup>7</sup></xref></contrib>
<contrib contrib-type="author"><name><surname>Murray</surname><given-names initials="ML">Macey L.</given-names></name><xref ref-type="aff" rid="affil-8"><sup>8</sup></xref><xref ref-type="aff" rid="affil-11">&#x2021;</xref></contrib>
<contrib contrib-type="author"><name><surname>Sydes</surname><given-names initials="MR">Matthew R.</given-names></name><xref ref-type="aff" rid="affil-9"><sup>9</sup></xref><xref ref-type="aff" rid="affil-11">&#x2021;</xref></contrib>
<aff id="affil-1"><label>1</label><institution>UCL Innovative Clinical Trials Unit, Institute of Clinical Trials and Methodology, UCL, London, WC1V 6LJ, UK; ORCID: 0000-0001-5957-7275</institution></aff>
<aff id="affil-2"><label>2</label><institution>UCL Innovative Clinical Trials Unit, Institute of Clinical Trials and Methodology, UCL, London, WC1V 6LJ, UK; ORCID: 0000-0002-6695-5390</institution></aff>
<aff id="affil-3"><label>3</label><institution>UCL Innovative Clinical Trials Unit, Institute of Clinical Trials and Methodology, UCL, London, WC1V 6LJ, UK; ORCID: 0000-0002-2512-2296</institution></aff>
<aff id="affil-4"><label>4</label><institution>UCL Innovative Clinical Trials Unit, Institute of Clinical Trials and Methodology, UCL, London, WC1V 6LJ, UK; ORCID: 0000-0002-6718-7661</institution></aff>
<aff id="affil-5"><label>5</label><institution>Kent, Medway and Sussex Secure Data Environment, Kent, Medway and Sussex Secure Data Environment - NHS England Digital;ORCID: 0009-0004-1807-3758</institution></aff>
<aff id="affil-6"><label>6</label><institution>Institute of Health Informatics, University College London, London, NW1 2DA, UK; The National Institute for Health Research University College London Hospitals Biomedical Research Centre, University College London, London, NW1 2PG, UK; Department of Cardiology, Amsterdam Cardiovascular Sciences, Amsterdam University Medical Center, University of Amsterdam, 1105 AZ Amsterdam, The Netherlands; ORCID: 0000-0002-1692-8669</institution></aff>
<aff id="affil-7"><label>7</label><institution>School of Medicine, NIHR Nottingham Biomedical Research Centre, University of Nottingham, NG7 2UH, UK; ORCID: 0000-0002-3012-6646</institution></aff>
<aff id="affil-8"><label>8</label><institution>UCL Innovative Clinical Trials Unit, Institute of Clinical Trials and Methodology, UCL, London, WC1V 6LJ, UK; Transforming Data for Trials programme, Health Data Research UK (HDR UK), London, NW1 2BE, UK; ORCID: 0000-0001-6418-0854</institution></aff>
<aff id="affil-9"><label>9</label><institution>UCL Innovative Clinical Trials Unit, Institute of Clinical Trials and Methodology, UCL, London, WC1V 6LJ, UK; Data for R&amp;D, “Data Access &amp; Partnerships, Technology Data &amp; Digital”, NHS England, London, SE1 8UG, UK; ORCID: 0000-0002-9323-1371</institution></aff>
<aff id="affil-10"><label>&#x2020;</label><institution>These authors contributed equally</institution></aff>
<aff id="affil-11"><label>&#x2021;</label><institution>These authors contributed equally</institution></aff>
</contrib-group>
<author-notes>
<corresp id="correspondingAurthor"><label>*</label>Corresponding author: Sharon B. Love, <email>s.love@ucl.ac.uk</email></corresp>
<fn fn-type="conflict">
<label>Conflicts of interests</label>
<p><bold>SMF</bold> has no conflicts of interest</p>
<p><bold>SBL</bold> has no conflicts of interest</p>
<p><bold>DF</bold>: has no conflicts of interest</p>
<p><bold>IRW</bold> has no conflicts of interest</p>
<p><bold>JT</bold>: has no conflicts of interest</p>
<p><bold>CD</bold> has no conflicts of interest</p>
<p><bold>PQ</bold> has no conflicts of interest</p>
<p><bold>MLM</bold> declares: Salary paid by UCL from grants received from HDR UK (HDR-9005, HDRUK TF2022.14, HDRUK2023.0025);</p>
<p>Dr Murray held an honorary position as a Business and Operational Delivery Manager for the NHS DigiTrials Programme of NHS Digital between Dec-2020 and Dec-2022; Independent member of Trial Steering Committee for TIPTOE Trial (MulTI-domain Self-management in Older People wiTh OstEoarthritis and Multi-Morbidities); and ASPIRING (Antiplatelet Secondary Prevention International Randomised study after INtracerebral haemorrhaGe); both with academic sponsors and none paid. Research Co-Director, Transforming Data for Clinical Trials infrastructure programme for Health Data Research UK; Supervisor for PhD student on the MRC-NIHR Trials Methodology Research Partnership (TMRP) Doctoral Training Programme; TMRP partner representative for HDR UK on the Executive Committee.</p>
<p><bold>MRS</bold> declares: Previous employment on an MRC grant; payment for Educational video on clinical trial statistics (content of MRS choosing) from Eisai and Eli-Lily; speaker fees for lectures on clinical trial statistics (content of MRS choosing and no discussion of particular drugs) including travel to lecture venue from Janssen; Attending Strategic Review Board from Health Research Board, Ireland; Attendance at ICTMC 2024 &amp; 2025 from Trials Research Methodology Network, Ireland; Faculty for CReDO 2024, 2025 &amp; 2026 from National Cancer Grid, India. Independent member of many Independent Data Monitoring Committees but all for academic sponsors and none paid.</p>
</fn>
</author-notes>
<pub-date date-type="pub" publication-format="electronic"><day>27</day><month>08</month><year>2026</year></pub-date>
<pub-date date-type="collection" publication-format="electronic"><year>2026</year></pub-date>
<volume>6</volume>
<issue>1</issue>
<elocation-id>3201</elocation-id>
<permissions>
<license specific-use="CC BY 4.0" xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a Creative Commons Attribution-NonCommercial-NoDerivatives 4.0 International License.</license-p>
</license>
</permissions>
<self-uri xlink:href="https://ijpds.org/article/view/3201">This article is available from the IJPDS website at: https://ijpds.org/article/view/3201</self-uri>
<abstract>
<title>Abstract</title>
<sec>
<title>Introduction</title>
<p>Clinical trials are usually analysed in a single environment allowing for flexible analysis including adjustment or stratification by subgroup: ‘one-stage’ analysis of individual-level data. Health systems datasets, often distributed across geography and providers, can streamline clinical trials. Data are increasingly accessible in secure data environments (SDEs). Future trial analyses may involve working across multiple SDEs. Row-level data and identifiable data often cannot leave, requiring a ‘two-stage approach’, where summary data from each SDE are meta-analysed.</p>
</sec>
<sec>
<title>Objective</title>
<p>To quantify the potential loss of precision and concomitant increases in required sample sizes, and to make recommendations for trial design and conduct, if clinical trial data are split across silos (e.g. SDEs).</p>
</sec>
<sec>
<title>Methods</title>
<p>Simulations used data from clinical trials in breast cancer, tuberculosis and prostate cancer with time-to-event, binary and continuous outcome measures. Silos were mimicked by 1000 random partitions into 2, 4, 10 and 25 equal silos and 4 unequal silos proportionate to the UK nations. Data were analysed as if the data could be pooled ignoring silo, pooled accounting for silo (one-stage) or not pooled (two-stage). Estimates and standard errors were presented graphically.</p>
</sec>
<sec>
<title>Results</title>
<p>For all three outcome measure types, standard errors increased while point estimates spread out as more silos were introduced. Small biases occurred for binary and time-to-event outcomes. This did not always appreciably reduce efficiency. However, in one example with time-to-event data and the largest number of silos, a near-doubling of sample size would have been required to pre-emptively offset the loss of efficiency.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>Any need to use a two-stage analysis approach has a negative effect compared to doing a one-stage analysis. Technical and data governance solutions to support one-stage analyses are recommended.</p>
</sec>
</abstract>
<kwd-group>
<kwd>Federated analytics</kwd>
<kwd>Health systems data</kwd>
<kwd>HSD</kwd>
<kwd>Clinical Trial</kwd>
<kwd>RCT</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="introduction">
<title>Introduction</title>
<p>Clinical trial datasets are usually collated and analysed in a single environment, allowing for flexible analysis including adjustment or stratification by subgroup. However, external factors are culminating in data being held in different environments, which could affect future design and analyses of trials.</p>
<p>Governance of sensitive data has evolved in response to digital and technological advances, with increasing emphasis on data security and protection, especially in health research. The Five Safes framework (safe data, projects, people, settings, and outputs) pioneered by the UK Office for National Statistics is widely adopted by data controllers in secure data environments (SDEs), also known as trusted research environments or data safe havens [<xref ref-type="bibr" rid="ref-1">1</xref>&#x2013;<xref ref-type="bibr" rid="ref-4">4</xref>]. Following the Life Sciences Vision (2021) [<xref ref-type="bibr" rid="ref-5">5</xref>, <xref ref-type="bibr" rid="ref-6">6</xref>] and Goldacre Review (2022) [<xref ref-type="bibr" rid="ref-7">7</xref>], there is now an accelerated move in the UK towards providing researchers access to health data safely via SDEs rather than distributing data to the researcher’s single location for a study. In SDEs, sets of defined data are securely provided by data controllers for appropriate analysis by validated ‘safe researchers’, which safeguard patient privacy and datasecurity.</p>
<p>Judicious use of high-quality data from health systems datasets (HSD) will streamline future clinical trials, facilitating planning, accrual and data collection [<xref ref-type="bibr" rid="ref-8">8</xref>&#x2013;<xref ref-type="bibr" rid="ref-10">10</xref>]. Information about relevant individuals may be held across separate data locations, for example, multiple HSD of hospital admissions, primary care records and disease registries. Currently, clinical trial sponsors can collate the necessary HSD (‘bespoke extracts’) about their trial participants for detailed analysis of outcomes in their single environment [<xref ref-type="bibr" rid="ref-4">4</xref>, <xref ref-type="bibr" rid="ref-11">11</xref>, <xref ref-type="bibr" rid="ref-12">12</xref>]. In the near future in England, the default route to access these data is likely to be in SDEs, potentially based on geographical regions. SDEs operate with disclosure control processes in place, and individual level data (even if anonymised) often cannot easily egress an SDE. Therefore, a trial sponsor would likely be faced with the technical and governance challenges of performing HSD-based analyses in each environment separately and meta-analysing those results.</p>
<p>We sought to assess the impact on trial analyses if data are analysed in (or as if they are in) the same place. Where data can be pooled, the trial analysis becomes analogous to a ‘one-stage’ meta-analysis of individual participant data (IPD) [<xref ref-type="bibr" rid="ref-13">13</xref>], occasionally known as a ‘mega-analysis’ [<xref ref-type="bibr" rid="ref-14">14</xref>]. Any adjustment or stratification for clinical site or other grouping is straightforward and allows for a wide range of analysis possibilities. Where data cannot be pooled, we assumed isolated analyses would need to be done within each environment or silo, and the relevant summary data synthesised in a second analysis stage. This is analogous to a ‘two-stage’ meta-analysis approach, where the second stage is performed on aggregated data [<xref ref-type="bibr" rid="ref-15">15</xref>]. Hence, clinical trialists can learn from meta-analysts who are used to working across different clinical trials and, for various reasons, cannot always access IPD [<xref ref-type="bibr" rid="ref-16">16</xref>]. Meta-analysts usually prefer pooled IPD approaches over aggregated data for the depth and reliability of the analyses available, as well as simplicity [<xref ref-type="bibr" rid="ref-17">17</xref>].</p>
<p>We anticipated a two-stage meta-analytic approach may have an inflationary impact on the number of people required to participate in a study. Impact may be exacerbated by noncollapsibility, defined as “a failure of the measure when taken on a group to equal a simple average of the measure when taken on the group’s members or subgroups [<xref ref-type="bibr" rid="ref-18">18</xref>].” Theory suggests that siloing (splitting data) in noncollapsible settings would be more problematic (i.e. moving from a marginal estimand to one which conditions on silo) [<xref ref-type="bibr" rid="ref-19">19</xref>] and should lead to both larger magnitude of effect size <italic>and</italic> larger standard error, but similar significance levels[<xref ref-type="bibr" rid="ref-20">20</xref>].</p>
<p>The aims were to: explore any loss of precision if two-stage approaches were required; determine how to mitigate this upfront during sample size calculations; and elucidate longer term solutions.</p>
</sec>
<sec id="methods">
<title>Methods</title>
<p>We needed several different siloed datasets with varying outcomes. As datasets that were already siloed across separate SDEs were not readily available, we conducted simulated analyses in large, existing datasets from three clinical data sources. Silos were mimicked with datasets partitioned as if they had been split across siloed environments.</p>
<sec id="data-sources">
<title>Data sources</title>
<p>We used relevant, limited, pseudonymised participant data from three clinical trials that took very different approaches. No HSD, participant identifiers or sensitive characteristics were requested. Where relevant, data access requests were approved following review from each trial’s data access committee. No further ethics committee approval was required. The German breast cancer study is a standard observational dataset for exploring time-to-event analyses provided by StataCorp [<xref ref-type="bibr" rid="ref-21">21</xref>]. The dataset includes recurrence-free survival time for 686 women with primary node-positive breast cancer, with 8 predictor variables. We used recurrence-free survival without the censoring variable as a continuous outcome, and with censoring as a time-to-event outcome. TB-IPD is a pooled, pre-harmonised dataset comprising IPD for randomised controlled trials of treatments for tuberculosis (TB) [<xref ref-type="bibr" rid="ref-22">22</xref>]. We used the 2021 dataset containing 17,507 participants from eight studies. The datasets included a binary outcome measure of cured or not, where ‘not’ includes all other disease stages. For the purposes of this analysis, the data are treated as if they come from one very large clinical trial. STAMPEDE is a multi-arm multi-stage platform trial for newly-diagnosed advanced prostate cancer; data were obtained from one comparison of 1176 participants who were allocated 2:1 to standard-of-care or standard-of-care plus docetaxel [<xref ref-type="bibr" rid="ref-23">23</xref>]. We used prostate-specific antigen (PSA) levels (ng/mL) collected 6 months after randomisation as a continuous outcome, and overall survival as a time-to-event outcome. The primary outcome measures in the original analysis were overall survival in STAMPEDE and recurrence-free survival time in the German breast cancer dataset. The results of our illustrative analyses should not be used to inform clinical practice.</p>
</sec>
<sec id="mimicking-silos">
<title>Mimicking silos</title>
<p>Each dataset effectively comprised IPD in a single analysis environment. To mimic siloing, we partitioned each dataset following three approaches. First, the participants were partitioned randomly across 2, 4, 10, or 25 equally sized silos. Second, the participants were partitioned randomly across unequal silos of size approximately 84%, 8%, 5% and 3%, proportional to the populations of England, Scotland, Wales and Northern Ireland, the four nations of the UK. This scenario was to explore the implications on UK-wide trials if approaches to data access do not permit interacting with the data in one place. Finally, we randomly partitioned the participants into nine silos of unequal sizes of 16%, 16%, 13%, 11%, 11%, 10%, 10%, 9% and 5%, approximately proportional to the populations of nine English regions defined by Office for National Statistics (ONS). [<xref ref-type="bibr" rid="ref-24">24</xref>] In each scenario, the random partitioning was repeated 1,000 times.</p>
</sec>
<sec id="statistical-analysis">
<title>Statistical analysis</title>
<p>Our ‘reference’ analysis was performed by fitting the analysis model to the full data set, without adjustment for silo. This represents the ideal situation of a single analysis environment where the source of siloing could be ignored, and is typical of how clinical trial sponsors have historically interacted with data. A variation on this analysis additionally adjusted for silo as an unordered categorical variable in the model: we refer to this as ‘one-stage’, where silo is equivalent to study in a one-stage IPD meta-analysis. To mimic instances where data cannot be pooled across silos, we performed a ‘two-stage’ analysis by separately fitting the analysis model to the data available in each silo, extracting the estimated treatment effect and standard error. We combined these estimates using a common-effect meta-analysis [<xref ref-type="bibr" rid="ref-25">25</xref>] to provide the overall estimate. Where the treatment effect represented an odds ratio or hazard ratio, the regression coefficients (i.e. the log odds ratio or log hazard ratio) were used for combining across trials. We did not allow for heterogeneity between silos since primary clinical trial analyses do not usually allow for treatment effect heterogeneity across geographical settings, and since silos will typically be subject to the same protocol.</p>
<p>For the German breast cancer study, we illustrated a collapsible analysis by fitting a linear regression of recurrence-free survival time on hormonal therapy, adjusted for the other baseline covariates, ignoring censoring. We illustrated a noncollapsible (time-to-event) analysis with a Cox regression on hormonal therapy, unadjusted and adjusted for age and menopausal status. In each case, the estimand of interest was the coefficient of hormonal therapy, representing the mean difference in survival times assuming no censoring and the log hazard ratio for recurrence or death. For TB-IPD, the analysis of interest was a logistic regression of presence or absence of cure on culture status, age, sex, body mass index, HIV status and smear status at baseline, plus time to culture conversion. The estimand of interest was the coefficient of culture status. We refer to culture status as ‘treatment’ to be consistent with the other data sets. Our covariate of interest was non-randomised in both these datasets, so the coefficients only have causal interpretation under a no unmeasured confounders assumption: however, our comparisons are between analyses adjusting for the <italic>same</italic> confounders, so any residual confounding does not affect the comparisons.</p>
<p>For STAMPEDE a linear regression analysis was done with the outcome being log transformed prostate-specific antigen (PSA) at 6 months post-randomisation. The covariates were treatment, log PSA at randomisation, age dichotomised at 70 years, and presence of metastasis, with 714 observations available. The estimand of interest was the coefficient of docetaxel treatment. We also analysed a second outcome measure, time from randomisation to death or last follow-up. Covariate adjustment is more unstable in smaller datasets, so we explored this analysis both with and without covariate adjustment. Hence, two Cox regressions on treatment arm were done: one adjusting for nodal stage, planned radiotherapy, age dichotomised at 70 years, WHO performance status at randomisation, metastatic status, regular NSAID use at baseline, and trial-specific time period; the other without adjustments.</p>
<p>We examined whether there were systematic differences in standard errors between analysis approaches, broken down by outcome measure type, and whether there was evidence of ‘bias’ (i.e. systematic difference) in effect size between analysis approaches, also broken down by outcome measure.</p>
<p>We show the results graphically in <xref ref-type="fig" rid="fig3">Figures 1</xref>–<xref ref-type="fig" rid="fig7">7</xref>. For the reference analysis there are no silos. For ease of visualisation, results of 100 random partitions (10% randomly selected from 1000 simulations) are displayed. The axes do not include zero. Monochrome versions of these graphs are given in the supplementary file.</p>
<fig id="fig1">
<caption><title>Individual participant data meta-analysis (one-stage) and aggregated data meta-analysis (two-stage) results for the continuous outcome, recurrence-free survival time ignoring censoring on hormonal therapy adjusted for age and menopausal status, from the German breast cancer study, using linear regression</title></caption>
<graphic xlink:href="ijpds-06-3201-g001.tif"/>
</fig>
<fig id="fig2">
<caption><title>Individual participant data meta-analysis (one-stage) and aggregated data meta-analysis (two-stage) results for the continuous outcome, prostate-specific antigen level at 6 months post-randomisation, adjusted for log(PSA) at randomisation and age (dichotomised at 70 years) from STAMPEDE, using linear regression</title></caption>
<graphic xlink:href="ijpds-06-3201-g002.tif"/>
</fig>
<fig id="fig3">
<caption><title>Individual participant data meta-analysis (one-stage) and aggregated data meta-analysis (two-stage) results for the binary outcome, presence or absence of cure from TB-IPD adjusted for culture status, age, sex, body mass index, HIV status and smear status at baseline, plus time to culture conversion, using logistic regression</title></caption>
<graphic xlink:href="ijpds-06-3201-g003.tif"/>
</fig>
<fig id="fig4">
<caption><title>Individual participant data meta-analysis (one-stage) and aggregated data meta-analysis (two-stage) results for the time-to-event outcome of recurrence-free survival time on hormonal therapy from the German breast cancer study, using Cox regression adjusted for age and menopausal status</title></caption>
<graphic xlink:href="ijpds-06-3201-g004.tif"/>
</fig>
<fig id="fig5">
<caption><title>Individual participant data meta-analysis (one-stage) and aggregated data meta-analysis (two-stage) results for the time-to-event outcome of recurrence-free survival time on hormonal therapy from the German breast cancer study, using unadjusted Cox regression</title></caption>
<graphic xlink:href="ijpds-06-3201-g005.tif"/>
</fig>
<fig id="fig6">
<caption><title>Individual participant data meta-analysis (one-stage) and aggregated data meta-analysis (two-stage) results for the time-to-event outcome of overall survival from STAMPEDE, using Cox regression adjusted for nodal stage, planned radiotherapy, age dichotomised at 70 years, WHO performance status at randomisation, metastatic status, regular NSAID use at baseline, and trial-specific time period</title></caption>
<graphic xlink:href="ijpds-06-3201-g006.tif"/>
</fig>
<fig id="fig7">
<caption><title>Individual participant data meta-analysis (one-stage) and aggregated data meta-analysis (two-stage) results for the time-to-event outcome of overall survival from STAMPEDE, using unadjusted Cox regression</title></caption>
<graphic xlink:href="ijpds-06-3201-g007.tif"/>
</fig>
</sec>
<sec id="sample-size-requirements">
<title>Sample size requirements</title>
<p>We used the effect sizes and their standard errors to estimate the increase in sample size that would be required for any clinical trial that used a one-stage or two-stage analysis (using silos) compared to a reference analysis (without silos). Following Julious (2004), [<xref ref-type="bibr" rid="ref-26">26</xref>] we tested the general null hypothesis f(<italic>μ</italic>) = 0 with two-sided Type-I error <italic>α</italic> using a test statistic <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mi>S</mml:mi><mml:mo>∼</mml:mo><mml:mi>N</mml:mi><mml:mo stretchy="false" form="prefix">(</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false" form="prefix">(</mml:mo><mml:mi>μ</mml:mi><mml:mo stretchy="false" form="postfix">)</mml:mo><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mo stretchy="false" form="prefix">(</mml:mo><mml:mi>S</mml:mi><mml:mo stretchy="false" form="postfix">)</mml:mo><mml:mo stretchy="false" form="postfix">)</mml:mo></mml:mrow></mml:math></inline-formula>. The sample size requirement <italic>N</italic> to achieve Type-II error <italic>β</italic> under the alternative f(<italic>μ</italic>) = d is given by solving <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mi>V</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mo stretchy="false" form="prefix">(</mml:mo><mml:mi>S</mml:mi><mml:mo stretchy="false" form="postfix">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:msup><mml:mi>d</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:msup><mml:mrow><mml:mo stretchy="false" form="prefix">(</mml:mo><mml:msub><mml:mi>Z</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>β</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>Z</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>α</mml:mi><mml:mi>/</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false" form="postfix">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mfrac></mml:mrow></mml:math></inline-formula>, where <italic>N</italic> is implicit in <italic>Var</italic>(<italic>S</italic>).</p>
<p>We have two tests <italic>S</italic><sub>1</sub> and <italic>S</italic><sub>2</sub> conducted on the same data set, where <italic>S</italic><sub>1</sub> is the reference analysis and <italic>S</italic><sub>2</sub> is either the one-stage or the two-stage analysis. We assume <italic>Var</italic>(<italic>S</italic><sub>1</sub>) and <italic>Var</italic>(<italic>S</italic><sub>2</sub>) are both inversely proportional to sample size <italic>N</italic>. The target effects for the two tests are <italic>d</italic><sub>1</sub> and <italic>d</italic><sub>2</sub>, which differ for noncollapsible effect measures. Then the ratio of required sample sizes for test 2 compared with test 1 is given by</p>
<disp-formula><alternatives><tex-math>\frac{{Var(S_2)}/{d^2_2}}{{Var(S_1)}/{d^2_1}}.</tex-math><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" display="block"><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:mi>V</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mo stretchy="false" form="prefix">(</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy="false" form="postfix">)</mml:mo></mml:mrow><mml:mi>/</mml:mi><mml:msubsup><mml:mi>d</mml:mi><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:mrow><mml:mrow><mml:mrow><mml:mi>V</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mo stretchy="false" form="prefix">(</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo stretchy="false" form="postfix">)</mml:mo></mml:mrow><mml:mi>/</mml:mi><mml:msubsup><mml:mi>d</mml:mi><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:mrow></mml:mfrac><mml:mi>.</mml:mi></mml:mrow></mml:math></alternatives></disp-formula>
<p>Intuitively, this is the squared ratio of expected Z-statistics. For the reference analysis, we take <italic>Var</italic>(<italic>S</italic><sub>1</sub>) as the squared standard error of <italic>S</italic><sub>1</sub> and <italic>d</italic><sub>1</sub> as the value of <italic>S</italic><sub>1</sub>. For the one-stage and two-stage analyses, which vary across repetitions, we take <italic>Var</italic>(<italic>S</italic><sub>2</sub>) as the average squared standard error of <italic>Var</italic>(<italic>S</italic><sub>2</sub>) and <italic>d</italic><sub>2</sub> as the unweighted average value of <italic>Var</italic>(<italic>S</italic><sub>2</sub>), taken across the 1,000 random partitions. Monte Carlo errors were computed using an extension of the methods used by the package simsum [<xref ref-type="bibr" rid="ref-27">27</xref>].</p>
<p>The analysis was carried out using Stata version 19.5 [<xref ref-type="bibr" rid="ref-28">28</xref>]. The code is given in the Supplementary Material.</p>
</sec>
</sec>
<sec id="results">
<title>Results</title>
<p>Each analysis of the three trials is shown below as a figure with tabulated summary information in the supplement, presented in outcome measure order. In each instance, the point estimate and standard error (SE) is shown for the reference analysis (intersection of the reference lines) and then overlaid for the one-stage (orange circle) and two-stage (purple circle) meta-analysis approaches. To simplify visualisation, results of 100 partitions (10% randomly selected from the 1,000 simulations) are displayed.</p>
<p>For the continuous outcome measure using linear regression in the breast cancer data, the one-stage analysis found the coefficient of hormonal therapy to be 180.6 days, with a standard error (SE) of 53.0 days. <xref ref-type="fig" rid="fig1">Figure 1</xref> and Supplementary Table 1 show results of the one-stage analyses adjusted for silo and the two-stage analyses (purple circles), across different patterns of silos. For the two-stage analyses, the SEs generally increased while the estimates spread out as the number of silos increased.</p>
<p>Using the STAMPEDE prostate cancer dataset for the continuous outcome measure of log PSA value at 6 months, the estimate under the reference analysis was -0.25 and SE 0.07. The pattern of distribution of the estimates and SEs was similar to <xref ref-type="fig" rid="fig1">Figure 1</xref> (<xref ref-type="fig" rid="fig2">Figure 2</xref> and Supplementary Table 2).</p>
<p>For the binary outcome measure in the TB-IPD dataset, analysed with logistic regression, the reference analysis gave an estimated log OR of 0.45 with a SE of 0.10. As the number of silos increased for two-stage analysis, as well as the spread of the estimates and the level of the SEs increasing, the mean of the estimates also increased away from the null (see <xref ref-type="fig" rid="fig3">Figure 3</xref> and Supplementary Table 3).</p>
<p>For the time-to-event analysis using the breast cancer data, the reference analysis gave a log hazard ratio of -0.38 with SE 0.13 (-0.36 and 0.13 respectively for the unadjusted analysis). As the silos were introduced and increased in number, the SEs increased (<xref ref-type="fig" rid="fig4">Figures 4</xref> and <xref ref-type="fig" rid="fig5">5</xref> and Supplementary Tables 4 and 5); the same effect was observed as for the logistic regression. Changes in point estimates and standard errors were similar to those seen in <xref ref-type="fig" rid="fig3">Figure 3</xref>. However, in <xref ref-type="fig" rid="fig4">Figures 4</xref> and <xref alt="5" rid="fig5">5</xref> the positive shifts in point estimates represents a shift towards the null, whereas in Figure <xref alt="3" rid="fig3">3</xref> the shift was away from the null. The spread of values is larger for the two-stage analyses compared to the corresponding one-stage ones with the impact most clearly observed with 25 silos.</p>
<p>For the STAMPEDE time-to-event analyses, the same pattern of analyses was followed for the time-to-event Cox regressions using breast cancer data. The reference adjusted analysis had its estimate as HR=0.78, with a standard error of 0.07. The mean point estimate does not change in <xref ref-type="fig" rid="fig6">Figure 6</xref> and Supplementary Table 6, unlike <xref ref-type="fig" rid="fig3">Figure 3</xref> where it goes away from the null and <xref ref-type="fig" rid="fig4">Figures 4</xref>, <xref alt="5" rid="fig5">5</xref> and <xref alt="7" rid="fig7">7</xref> (and Supplementary Table 7) where it goes towards the null.</p>
<p><xref ref-type="table" rid="table-1">Table 1</xref> and <xref ref-type="fig" rid="fig8">Figure 8</xref> present the indicative modifications to sample size that would have been required to address the observations in the previous examples. Accounting for silo in a one-stage approach, or in a two-stage approach with continuous or binary outcome, would generally require minimal change in sample size. However, a two-stage approach may require substantial inflation under Cox PH regression especially if, as is standard, adjusting for baseline characteristics. With ten silos, increases of 24% and 9% would have been required in the breast cancer and prostate cancer examples, and this was much higher with more silos. Each would require a longer trial period and/or faster accrual rates.</p>
</sec>
<sec id="discussion">
<title>Discussion</title>
<p>For all three outcome measure types, when more silos were used, standard errors increased while point estimates spread out, indicating loss of efficiency. The loss was greater if the participants were distributed across more silos and less evenly. This reduction of efficiency could be offset by inflating the target sample size.</p>
<p>We have shown that any analysis using a two-stage meta-analysis approach due to partitioning of data can be much less efficient compared to being able to do a one-stage approach where all the data contribute simultaneously to analytic models. The spread of estimated treatment effects is larger for the two-stage analyses compared to the corresponding one-stage analysis, with the impact most clearly observed with 25 silos. This problem is more pronounced for more complex outcome measures such as time-to-event(survival).</p>
<p>This loss of efficiency induced by a two-stage approach could be nominally offset by inflating the target sample size of the trial. The inflation factor was &lt;10% for continuous or binary outcomes and for time-to-event outcomes with fewer than 9 silos, but reached 7-24% for time-to-event outcomes with 9 or 10 silos and 35-91% for time-to-event outcomes with 25 silos. We explored a variety of silo sizes, and we anticipated that small silos may, by chance, lead to more extreme estimates of the treatment effect. The nature of the problem was greater if the participants were split more widely and less evenly. This would have substantial implications for many clinical trials. Many trials pre-emptively inflate their sample size in anticipation of issues with follow-up [<xref ref-type="bibr" rid="ref-29">29</xref>]. The inflationary factor required because of a two-stage analytic approach would be separate to this and would be multiplicative rather than additive. For example, a phase III trial requiring 500 people (to observe 200 events) might be inflated by 10% to 550 to account for anticipated loss to follow-up and a further 20% to 660 to account for a two-stage analysis. Therefore, such required inflation in sample size may offset the streamlining gained by using HSD.</p>
<p>In binary outcome measures, two-stage analysis biased the estimated treatment effect away from the null (<xref ref-type="fig" rid="fig3">Figure 3</xref>), probably due to noncollapsibility of the odds ratio. In time-to-event data, two-stage analysis could instead bias the estimated treatment effect towards the null (<xref ref-type="fig" rid="fig4">Figures 4</xref>, <xref alt="5" rid="fig5">5</xref> and <xref alt="7" rid="fig7">7</xref>), even though noncollapsibility of the hazard ratio would suggest a bias away from the null. We believe this bias arises from failure of the Normal approximation when the silo-level data become sparse (few events). [<xref ref-type="bibr" rid="ref-30">30</xref>] These biases need to be taken into account when comparing methods, but the relative biases seen in <xref ref-type="fig" rid="fig3">Figures 3</xref>–<xref ref-type="fig" rid="fig7">7</xref> were of limited clinical importance, being between 10% away from the null and 8% towards the null, except with 25 silos in <xref ref-type="fig" rid="fig4">Figures 4</xref>, <xref ref-type="fig" rid="fig5">5</xref> and <xref ref-type="fig" rid="fig7">7</xref> where they were 13-23% towards the null (Supplementary figure 1).</p>
<p>Use of a two-stage meta-analysis approach also complicates the statistical analysis. Subgroup analyses, whether pre-specified or exploratory, may be impacted because data within a subgroup within a silo will be particularly sparse. Data separated by silos also raises challenges in forming appropriate imputation models to handle missing data [<xref ref-type="bibr" rid="ref-31">31</xref>].</p>
<p>If the data for participants are held in separate locations, some duplication of effort in information governance is likely, regardless of whether the data can be analysed together or only separately. For data held in multiple environments, resolution will be needed for archiving according to the relevant legislation and reconstructing datasets for secondary or tertiary data reuse projects. Better data federation is associated with closer alignment, harmonised governance and improved trust [<xref ref-type="bibr" rid="ref-32">32</xref>].</p>
<table-wrap id="table-1">
<label>Table 1</label><caption><title>Sample size inflation required to offset loss of efficiency for one-stage and two-stage individual participant data analyses across silos, compared to analysing all data together without siloing</title></caption>
<table frame="hsides" rules="groups">
<col width="10%"/>
<col width="10%"/>
<col width="10%"/>
<col width="10%"/>
<col width="10%"/>
<col width="10%"/>
<col width="10%"/>
<col width="02%"/>
<col width="10%"/>
<col width="10%"/>
<tbody>
<tr>
<td rowspan="2" align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="bottom"><bold>Data</bold></td>
<td rowspan="2" align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="bottom"><bold>Outcome model</bold></td>
<td rowspan="2" align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="bottom"><bold>Approach</bold></td>
<td colspan="4" align="center" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="bottom"><bold>Equal silos</bold></td>
<td rowspan="2" align="center" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="bottom"></td>
<td colspan="2" align="center" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="bottom"><bold>Unequal silos</bold></td>
</tr>
<tr>
<td style="border-bottom: solid 1pt;" align="center" valign="bottom"><bold>2</bold></td>
<td style="border-bottom: solid 1pt;" align="center" valign="bottom"><bold>4</bold></td>
<td style="border-bottom: solid 1pt;" align="center" valign="bottom"><bold>10</bold></td>
<td style="border-bottom: solid 1pt;" align="center" valign="bottom"><bold>25</bold></td>
<td style="border-bottom: solid 1pt;" align="center" valign="bottom"><bold>4</bold></td>
<td style="border-bottom: solid 1pt;" align="center" valign="bottom"><bold>9</bold></td>
</tr>
<tr>
<td colspan="10" align="left" valign="top"><bold>Continuous</bold></td>
</tr>
<tr>
<td rowspan="2" align="left" valign="top">Breast Cancer</td>
<td rowspan="2" align="left" valign="top">Linear regression</td>
<td align="left" valign="top">1-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">1%</td>
<td align="center" valign="top">3%</td>
<td rowspan="16" align="left" valign="top"></td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">2%</td>
</tr>
<tr>
<td align="left" valign="top">2-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">1%</td>
<td align="center" valign="top">-1%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
</tr>
<tr>
<td rowspan="2" align="left" valign="top">STAMPEDE</td>
<td rowspan="2" align="left" valign="top">Linear regression</td>
<td align="left" valign="top">1-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">1%</td>
<td align="center" valign="top">4%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">1%</td>
</tr>
<tr>
<td align="left" valign="top">2-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">-1%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">1%</td>
<td align="center" valign="top">0%</td>
</tr>
<tr>
<td colspan="7" align="left" valign="top"><bold>Binary</bold></td>
<td colspan="2" align="left" valign="top"></td>
</tr>
<tr>
<td rowspan="2" align="left" valign="top">TB-IPD</td>
<td rowspan="2" align="left" valign="top">Logistic regression</td>
<td align="left" valign="top">1-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
</tr>
<tr>
<td align="left" valign="top">2-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">-1%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">-1%</td>
</tr>
<tr>
<td colspan="7" align="left" valign="top"><bold>Time-to-event</bold></td>
<td colspan="2" align="left" valign="top"></td>
</tr>
<tr>
<td rowspan="4" align="left" valign="top">Breast Cancer</td>
<td rowspan="2" align="left" valign="top">Cox PH adjusted</td>
<td align="left" valign="top">1-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">-1%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
</tr>
<tr>
<td align="left" valign="top">2-stage</td>
<td align="center" valign="top">3%</td>
<td align="center" valign="top">8%</td>
<td align="center" valign="top">24%</td>
<td align="center" valign="top">91%*</td>
<td align="center" valign="top">8%</td>
<td align="center" valign="top">21%</td>
</tr>
<tr>
<td rowspan="2" align="left" valign="top">Cox PH unadjusted</td>
<td align="left" valign="top">1-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">-1%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
</tr>
<tr>
<td align="left" valign="top">2-stage</td>
<td align="center" valign="top">3%</td>
<td align="center" valign="top">8%</td>
<td align="center" valign="top">24%</td>
<td align="center" valign="top">91%*</td>
<td align="center" valign="top">8%</td>
<td align="center" valign="top">22%</td>
</tr>
<tr>
<td rowspan="4" align="left" valign="top">STAMPEDE</td>
<td rowspan="2" align="left" valign="top">Cox PH adjusted</td>
<td align="left" valign="top">1-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
</tr>
<tr>
<td align="left" valign="top">2-stage</td>
<td align="center" valign="top">1%</td>
<td align="center" valign="top">1%</td>
<td align="center" valign="top">9%</td>
<td align="center" valign="top">35%*</td>
<td align="center" valign="top">3%</td>
<td align="center" valign="top">7%</td>
</tr>
<tr>
<td rowspan="2" align="left" valign="top">Cox PH unadjusted</td>
<td align="left" valign="top">1-stage</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">-1%</td>
<td align="center" valign="top">0%</td>
<td align="center" valign="top">0%</td>
</tr>
<tr>
<td align="left" valign="top">2-stage</td>
<td align="center" valign="top">1%</td>
<td align="center" valign="top">2%</td>
<td align="center" valign="top">11%</td>
<td align="center" valign="top">42%*</td>
<td align="center" valign="top">4%</td>
<td align="center" valign="top">10%</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Monte Carlo standard errors are 6-11% for results marked * and at most 4% for all other results. Cox PH -- Cox proportional hazards regression model.</p>
</table-wrap-foot>
</table-wrap>
<fig id="fig8">
<caption><title>Sample size inflation for two-stage individual participant data analyses across silos, compared to analysing all data together without siloing</title></caption>
<graphic xlink:href="ijpds-06-3201-g008.tif"/>
</fig>
<p>For all of these reasons, the conclusion reached for clinical trials is the same as that for meta-analysis: it is strongly preferable to interact with the data in a one-stage trial approach, where all the data are either actually in, or can be treated as if they are in, one place and included in the analysis model: preventing the problem is better than curingit [<xref ref-type="bibr" rid="ref-33">33</xref>].</p>
<p>The move towards ‘data access’ rather than ‘data sharing’ aims to restrict access to sensitive HSD [<xref ref-type="bibr" rid="ref-34">34</xref>]. SDEs are a safe and appropriate way to access information, particularly for projects where the participants have not given explicit consent, as in many retrospective real-world data projects [<xref ref-type="bibr" rid="ref-35">35</xref>]. For most clinical trials, there is an opportunity to obtain explicit consent which can include consent for the relevant, minimised data to be made safely available for analysis in the sponsor’s secure environment. Consistent forms of wording are required so data providers can egress datasets directly to trial sponsors for analysis [<xref ref-type="bibr" rid="ref-36">36</xref>].</p>
<p>One limitation of our work is that we did not access real HSD located in SDEs. Instead, entire clinical trial datasets were accessed and partitioned randomly to mimic having come from separate SDEs. However, this is also a strength: by creating many partitions of each dataset we were able to learn more about the broad pattern of what might happen, and to estimate a distribution of power loss. Previous authors have compared one-stage and two-stage approaches by simulating multiple trials [<xref ref-type="bibr" rid="ref-37">37</xref>], and by examining multiple different estimated quantities across the same four large trials [<xref ref-type="bibr" rid="ref-38">38</xref>]. However, their main objectives were to compare analysis approaches rather than to investigate the issue of siloing itself, and they did not examine time-to-event outcomes in which we have identified additional issues. Future research could explore a larger, more varied selection of real-world datasets, or a non-random partitioning mechanism.</p>
<p>The exact balance of participants across locations and SDEs for any particular trial may not be easily estimated up front. However, use of data in SDEs for trial planning, as well as delivery-through-data, should give a much better estimate of the actual numbers of participants that might be recruited and feed into sample size adjustment calculators.</p>
<p>The loss of efficiency due to siloing is not unique to clinical trials, but it is a particular issue for trials and other forms of prospective research predicated on <italic>a priori</italic> planning for a fixed number of participants to be recruited. For clinical trials with suitable participant consent, datasets could still be egressed from the relevant data providers, which may be multiple and shared in a safe and secure way to a safe and secure location from the trial sponsor, where they can interact with the data in one model. Alternatively, a full federated approach should be sought whereby the relevant, minimised datasets are pooled in one SDE where the trial sponsor’s representatives can interact with them in analysis. If pooling copies of datasets in one location is not possible, perhaps for reasons of information governance or file size, it will be preferable if technology were widely available to allow analysts to interact with the data as if they were in one place for one model, even though the datasets are actually in separate places. Some work is being started along these lines. Pilot studies run by TRE-FX and TELEPORT, supported by DARE UK, have developed approaches and technical solutions to federated analytics of genomic and health data [<xref ref-type="bibr" rid="ref-39">39</xref>, <xref ref-type="bibr" rid="ref-40">40</xref>, <xref ref-type="bibr" rid="ref-41">41</xref>]. Appropriate transformation of data assets into common data models may facilitate federated analytics; limitations in the mapping processes leading to data exclusion and loss of information have been reported [<xref ref-type="bibr" rid="ref-42">42</xref>, <xref ref-type="bibr" rid="ref-43">43</xref>]. There are emerging technical approaches available for federated analytics although these have potential for statistical inefficiency. This work helps to demonstrate some benefits and risks of this mechanism and can help frame the utility of the technical innovation to support data that is held across multiple SDEs. There is still more to do to ensure full interaction with data, particularly with respect to clinical trials.</p>
<p>One response is for sponsors to continue with trial-specific data collection, an approach which is well understood, traditional, but duplicative of effort. Another is for development of new approaches to analysis that might mitigate the challenge of two-stage analysis across silos. A final response is for data providers to understand the implications of this work in order that they may sooner provide resolution towards one-stage analysis equivalent to pooling data in one place. In the NHS Research SDE Network, a network of geographically-defined SDEs with complete coverage in England and one national secure data environment, work towards full federation is ongoing. Multinational solutions are required. Many UK-wide trials will recruit from across regions covered by different SDEs; other countries have similar administrative divisions; and many clinical trials are multinational, particularly industry-led trials. There are many other blockers to using HSD for delivery-through-data of clinical trials. [<xref ref-type="bibr" rid="ref-8">8</xref>, <xref ref-type="bibr" rid="ref-44">44</xref>] Trial sponsors may persist with tradition until all of these blockers have been addressed. Therefore, work is required rapidly to resolve each of these issues to make progress in streamlining the delivery of clinical trials.</p>
</sec>
<sec id="conclusion">
<title>Conclusion</title>
<p>There is a strong need to streamline clinical trials with better use of health systems data. However, there can be a penalty to be paid for any clinical trial that needs to use a two-stage aggregate data meta-analysis approach, which cannot easily be offset without increasing sample size as a minimum. Further work is required to enable one-stage analysis.</p>
</sec>
<sec id="acknowledgements">
<title>Acknowledgements</title>
<p>The salary of SBL was supported by UKRI (UK Research and Innovation; MRC (Medical Research Council) [grant number MC_UU_00004/08]. The salaries of IRW and DF were supported by UKRI [grant number MC_UU_00004/09]. The salaries of SF, SBL, MLM, MRS were supported by HDRUK: HDRUK2023.0025. MRS is employed by NHS England and contributed to this paper as part of the Data for Research and Development. This research was Powered by NHS Data.</p>
<p>We thank Andy Payne and Will Browne for critical comments during the drafting process.</p>
<p>We thank the investigators of, and all participants in, the STAMPEDE trial, German Breast Cancer study and each of the trials in the TB-IPD dataset for making data available for this project</p>
<p>This work uses data obtained through the WHO TB treatment individual patient data platform (TB-IPD), a collaborative initiative led by the World Health Organization (WHO), Global TB Programme and University College London (UCL).</p>
</sec>
<sec id="contributions">
<title>Contributions</title>
<p>Using CREDIT taxonomy</p>
<table-wrap>
<table frame="hsides" rules="groups">
<tbody>
<tr>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>Area</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>Description</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>SMF</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>SBL</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>DF</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>IRW</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>JFT</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>CD</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>AWF</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>PQ</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>MLM</bold></td>
<td align="left" style="border-top: solid 1pt; border-bottom: solid 1pt;" valign="middle"><bold>MRS</bold></td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Conceptualisation</bold>
</td>
<td align="left" valign="middle">Ideas; formulation or evolution of overarching research goals and aims.</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Data Curation</bold>
</td>
<td align="left" valign="middle">Management activities to annotate (produce metadata), scrub data and maintain research data (including software code, where it is necessary for interpreting the data itself) for initial use and later reuse.</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Formal Analysis</bold>
</td>
<td align="left" valign="middle">Application of statistical, mathematical, computational, or other formal techniques to analyse or synthesise study data.</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
</tr>
<tr>
<td align="left" valign="middle"><bold>Funding Acquisition</bold>
</td>
<td align="left" valign="middle">Acquisition of the financial support for the project leading to this publication.</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Investigation</bold>
</td>
<td align="left" valign="middle">Conducting a research and investigation process, specifically performing the experiments, or data/evidence collection.</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Methodology</bold>
</td>
<td align="left" valign="middle">Development or design of methodology; creation of models.</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Project Administration</bold>
</td>
<td align="left" valign="middle">Management and coordination responsibility for the research activity planning and execution.</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Resources</bold>
</td>
<td align="left" valign="middle">Provision of study materials, reagents, materials, patients, laboratory samples, animals, instrumentation, computing resources, or other analysis tools.</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
</tr>
<tr>
<td align="left" valign="middle"><bold>Software</bold>
</td>
<td align="left" valign="middle">Programming, software development; designing computer programs; implementation of the computer code and supporting algorithms; testing of existing code components.</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Supervision</bold>
</td>
<td align="left" valign="middle">Oversight and leadership responsibility for the research activity planning and execution, including mentorship external to the core team.</td>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Validation</bold>
</td>
<td align="left" valign="middle">Verification, whether as a part of the activity or separate, of the overall replication/reproducibility of results/experiments and other research outputs.</td>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
</tr>
<tr>
<td align="left" valign="middle"><bold>Visualisation</bold>
</td>
<td align="left" valign="middle">Preparation, creation and/or presentation of the published work, specifically visualisation/data presentation.</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Writing – Original Draft Preparation</bold>
</td>
<td align="left" valign="middle">Creation and/or presentation of the published work, specifically writing the initial draft (including substantive translation).</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
</tr>
<tr>
<td align="left" valign="middle"><bold>Writing – Review &amp; Editing</bold>
</td>
<td align="left" valign="middle">Preparation, creation and/or presentation of the published work by those from the original research group, specifically critical review, commentary or revision – including pre- or post-publication stages.</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
<td align="center" valign="middle">X</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</body>
<back>
<sec id="ethics">
<title>Ethics</title>
<p>The German breast cancer data are freely available from StataCorp [<xref ref-type="bibr" rid="ref-21">21</xref>].</p>
<p>TB-IPD is a pooled, pre-harmonised dataset from a publicly available repository [<xref ref-type="bibr" rid="ref-22">22</xref>].</p>
<p>STAMPEDE is trial data with individual patient consent for use in research [<xref ref-type="bibr" rid="ref-23">23</xref>].</p>
</sec>
<sec id="data-availability-statement">
<title>Data availability statement</title>
<p>The German breast cancer data are freely available from StataCorp [<xref ref-type="bibr" rid="ref-21">21</xref>] at <ext-link ext-link-type="uri" xlink:href="https://www.stata-press.com/data/r16/brcancer.dta">https://www.stata-press.com/data/r16/</ext-link><ext-link ext-link-type="uri" xlink:href="https://www.stata-press.com/data/r16/brcancer.dta">brcancer.dta</ext-link>.</p>
<p>TB-IPD is available from the Data Access Committee of TB-IPD: <ext-link ext-link-type="uri" xlink:href="https://www.ucl.ac.uk/population-health-sciences/global-health/research/research-projects/tb-ipd-platform">https://www.ucl.ac.uk/population-health-sciences/global-health/research/research-projects/tb-ipd-platform</ext-link>.</p>
<p>The STAMPEDE trial was sponsored by UCL and coordinated by the MRC Clinical Trials Unit at UCL, full details are available at the trial website: <ext-link ext-link-type="uri" xlink:href="http://www.stampedetrial.org">www.stampedetrial.org</ext-link>. [<xref ref-type="bibr" rid="ref-23">23</xref>] STAMPEDE data can be requested from <ext-link ext-link-type="uri" xlink:href="https://www.mrcctu.ucl.ac.uk/our-research/other-research-policy/data-sharing/">https://www.mrcctu.ucl.ac.uk/our-research/other-research-policy/data-sharing/</ext-link>.</p>
</sec>
<sec id="ai-disclosure-statement">
<title>AI disclosure statement</title>
<p>The authors declare that no generative AI tools were used in the preparation of this manuscript.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref-1"><label>1</label><mixed-citation publication-type="website"><string-name><surname>Desai</surname> <given-names>T</given-names></string-name>, <string-name><surname>Ritchie</surname> <given-names>F</given-names></string-name>, <string-name><surname>Welpton</surname> <given-names>R</given-names></string-name>. <article-title>Five Safes: designing data access for research</article-title>. <source>Working papers in Economics</source>. <year>2016</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://uwe-repository.worktribe.com/output/914745">https://uwe-repository.worktribe.com/output/914745</ext-link> [accessed 22 May 2025].</mixed-citation></ref>
<ref id="ref-2"><label>2</label><mixed-citation publication-type="website"><collab>Health Data Research UK</collab>. <article-title>Trusted Research Environments</article-title>. <source>Health Data Research UK</source>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.hdruk.ac.uk/access-to-health-data/trusted-research-environments/">https://www.hdruk.ac.uk/access-to-health-data/trusted-research-environments/</ext-link> [accessed 22 May 2025].</mixed-citation></ref>
<ref id="ref-3"><label>3</label><mixed-citation publication-type="journal"><collab>UK Health Data Research Alliance, NHSX</collab>. <article-title>Building Trusted Research Environments - Principles and Best Practices; Towards TRE ecosystems</article-title>. <source>Zenodo</source>. <year>2021</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.5767586">https://doi.org/10.5281/zenodo.5767586</ext-link> [accessed 22 May 2025].</mixed-citation></ref>
<ref id="ref-4"><label>4</label><mixed-citation publication-type="journal"><string-name><surname>Sudlow</surname> <given-names>C</given-names></string-name>. <article-title>Uniting the UK’s Health Data: A Huge Opportunity for Society</article-title>. <source>Zenodo</source>. <year>2024</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.13353747">https://doi.org/10.5281/zenodo.13353747</ext-link> [accessed 08 November 2024].</mixed-citation></ref>
<ref id="ref-5"><label>5</label><mixed-citation publication-type="website"><collab>UK Government</collab>. <article-title>Life Sciences Vision</article-title>. <year>2021</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://assets.publishing.service.gov.uk/media/612763b4e90e0705437230c3/life-sciences-vision-2021.pdf">https://assets.publishing.service.gov.uk/media/612763b4e90e0705437230c3/life-sciences-vision-2021.pdf</ext-link> [accessed 22 May 2025].</mixed-citation></ref>
<ref id="ref-6"><label>6</label><mixed-citation publication-type="website"><collab>NHS Digital</collab>. <article-title>Secure Data Environment: Department for Health and Social Care access to the NHS England Secure Data Environment</article-title>. <year>2024</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://digital.nhs.uk/data-and-information/looking-after-information/data-security-and-information-governance/secure-data-environment-department-for-health-and-social-care-access-to-the-nhs-england-secure-data-environment">https://digital.nhs.uk/data-and-information/looking-after-information/data-security-and-information-governance/secure-data-environment-department-for-health-and-social-care-access-to-the-nhs-england-secure-data-environment</ext-link> [accessed 14 July 2025].</mixed-citation></ref>
<ref id="ref-7"><label>7</label><mixed-citation publication-type="website"><string-name><surname>Goldacre</surname> <given-names>B</given-names></string-name>, <string-name><surname>Morley</surname> <given-names>J</given-names></string-name>. <article-title>Better, Broader, Safer: Using Health Data for Research and Analysis: Department of Health and Social Care</article-title>; <year>2022</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.gov.uk/government/publications/better-broader-safer-using-health-data-for-research-and-analysis">https://www.gov.uk/government/publications/better-broader-safer-using-health-data-for-research-and-analysis</ext-link> [accessed 20 January 2025].</mixed-citation></ref>
<ref id="ref-8"><label>8</label><mixed-citation publication-type="journal"><string-name><surname>Sydes</surname> <given-names>MR</given-names></string-name>, <string-name><surname>Barbachano</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Bowman</surname> <given-names>L</given-names></string-name>, <etal>et al</etal>. <article-title>Realising the full potential of data-enabled trials in the UK: a call for action</article-title>. <source>BMJ Open</source> <year>2021</year>;<volume>11</volume>(<issue>6</issue>):<fpage>e043906</fpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1136/bmjopen-2020-043906">https://10.1136/bmjopen-2020-043906</ext-link></mixed-citation></ref>
<ref id="ref-9"><label>9</label><mixed-citation publication-type="journal"><string-name><surname>Lensen</surname> <given-names>S</given-names></string-name>, <string-name><surname>Macnair</surname> <given-names>A</given-names></string-name>, <string-name><surname>Love</surname> <given-names>SB</given-names></string-name>, <etal>et al</etal>. <article-title>Access to routinely collected health data for clinical trials - review of successful data requests to UK registries</article-title>. <source>Trials</source> <year>2020</year>;<volume>21</volume>(<issue>1</issue>):<fpage>398</fpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1186/s13063-020-04329-8">https://10.1186/s13063-020-04329-8</ext-link></mixed-citation></ref>
<ref id="ref-10"><label>10</label><mixed-citation publication-type="journal"><collab>RECOVERY Collaborative Group</collab>, <string-name><surname>Horby</surname> <given-names>P</given-names></string-name>, <string-name><surname>Lim</surname> <given-names>WS</given-names></string-name>, <etal>et al</etal>. <article-title>Dexamethasone in Hospitalized Patients with Covid-19</article-title>. <source>N Engl J Med</source> <year>2021</year>;<volume>384</volume>(<issue>8</issue>):<fpage>693</fpage>-<lpage>704</lpage>. <ext-link ext-link-type="uri" xlink:href="https:// 10.1056/NEJMoa2021436">https:// 10.1056/NEJMoa2021436</ext-link></mixed-citation></ref>
<ref id="ref-11"><label>11</label><mixed-citation publication-type="journal"><string-name><surname>Love</surname> <given-names>SB</given-names></string-name>, <string-name><surname>Kilanowski</surname> <given-names>A</given-names></string-name>, <string-name><surname>Yorke-Edwards</surname> <given-names>V</given-names></string-name>, <etal>et al</etal>. <article-title>Use of routinely collected health data in randomised clinical trials: comparison of trial-specific death data in the BOSS trial with NHS Digital data</article-title>. <source>Trials</source> <year>2021</year>;<volume>22</volume>(<issue>1</issue>):<fpage>654</fpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1186/s13063-021-05613-x">https://10.1186/s13063-021-05613-x</ext-link></mixed-citation></ref>
<ref id="ref-12"><label>12</label><mixed-citation publication-type="journal"><string-name><surname>Macnair</surname> <given-names>A</given-names></string-name>, <string-name><surname>Nankivell</surname> <given-names>M</given-names></string-name>, <string-name><surname>Murray</surname> <given-names>ML</given-names></string-name>, <etal>et al</etal>. <article-title>Healthcare systems data in the context of clinical trials - A comparison of cardiovascular data from a clinical trial dataset with routinely collected data</article-title>. <source>Contemp Clin Trials</source> <year>2023</year>;<volume>128</volume>:<fpage>107162</fpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1016/j.cct.2023.107162">https://10.1016/j.cct.2023.107162</ext-link></mixed-citation></ref>
<ref id="ref-13"><label>13</label><mixed-citation publication-type="journal"><string-name><surname>Burke</surname> <given-names>DL</given-names></string-name>, <string-name><surname>Ensor</surname> <given-names>J</given-names></string-name>, <string-name><surname>Riley</surname> <given-names>RD</given-names></string-name>. <article-title>Meta-analysis using individual participant data: one-stage and two-stage approaches, and why they may differ</article-title>. <source>Stat Med</source> <year>2017</year>;<volume>36</volume>(<issue>5</issue>):<fpage>855</fpage>-<lpage>875</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1002/sim.7141">https://10.1002/sim.7141</ext-link></mixed-citation></ref>
<ref id="ref-14"><label>14</label><mixed-citation publication-type="journal"><string-name><surname>Eisenhauer</surname> <given-names>JG</given-names></string-name>. <article-title>Meta-analysis and mega-analysis: A simple introduction</article-title>. <source>Teaching Statistics</source> <year>2021</year>;<volume>43</volume>(<issue>1</issue>):<fpage>21</fpage>-<lpage>27</lpage>. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1111/test.12242.15">https://doi.org/10.1111/test.12242.15</ext-link></mixed-citation></ref>
<ref id="ref-15"><label>15</label><mixed-citation publication-type="journal"><string-name><surname>Deeks</surname> <given-names>JJ</given-names></string-name>, <string-name><surname>Higgins</surname> <given-names>JPT</given-names></string-name>, <string-name><surname>Altman</surname> <given-names>DG</given-names></string-name>, <string-name><surname>Mckenzie</surname> <given-names>JE</given-names></string-name>, <string-name><surname>Veroniki</surname> <given-names>AA</given-names></string-name>. <article-title>Chapter 10: Analysing data and undertaking meta-analyses</article-title>. In: <string-name><surname>Higgins</surname> <given-names>JPT</given-names></string-name>, <string-name><surname>Thomas</surname> <given-names>J</given-names></string-name>, <string-name><surname>Chandler</surname> <given-names>J</given-names></string-name>, <string-name><surname>Cumpston</surname> <given-names>M</given-names></string-name>, <string-name><surname>Li</surname> <given-names>T</given-names></string-name>, <string-name><surname>Page</surname> <given-names>MJ</given-names></string-name>, <string-name><surname>Welch</surname> <given-names>VA</given-names></string-name>, editors. <source>Cochrane Handbook for Systematic Reviews of Interventions version 65 (updated August 2024): Cochrane</source>; <year>2024</year>.</mixed-citation></ref>
<ref id="ref-16"><label>16</label><mixed-citation publication-type="journal"><string-name><surname>Tierney</surname> <given-names>JF</given-names></string-name>, <string-name><surname>Stewart</surname> <given-names>LA</given-names></string-name>, <string-name><surname>Ghersi</surname> <given-names>D</given-names></string-name>, <string-name><surname>Burdett</surname> <given-names>S</given-names></string-name>, <string-name><surname>Sydes</surname> <given-names>MR</given-names></string-name>. <article-title>Practical methods for incorporating summary time-to-event data into meta-analysis</article-title>. <source>Trials</source> <year>2007</year>;<volume>8</volume>:<fpage>16</fpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1186/1745-6215-8-16">https://10.1186/1745-6215-8-16</ext-link></mixed-citation></ref>
<ref id="ref-17"><label>17</label><mixed-citation publication-type="journal"><string-name><surname>Riley</surname> <given-names>RD</given-names></string-name>, <string-name><surname>Ensor</surname> <given-names>J</given-names></string-name>, <string-name><surname>Hattle</surname> <given-names>M</given-names></string-name>, <string-name><surname>Papadimitropoulou</surname> <given-names>K</given-names></string-name>, <string-name><surname>Morris</surname> <given-names>TP</given-names></string-name>. <article-title>Two-stage or not two-stage? That is the question for IPD meta-analysis projects</article-title>. <source>Res Synth Methods</source> <year>2023</year>;<volume>14</volume>(<issue>6</issue>):<fpage>903</fpage>-<lpage>910</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1002/jrsm.1661">https://10.1002/jrsm.1661</ext-link></mixed-citation></ref>
<ref id="ref-18"><label>18</label><mixed-citation publication-type="journal"><string-name><surname>Greenland</surname> <given-names>S</given-names></string-name>. <article-title>Noncollapsibility, confounding, and sparse-data bias. Part 1: The oddities of odds</article-title>. <source>J Clin Epidemiol</source> <year>2021</year>;<volume>138</volume>:<fpage>178</fpage>-<lpage>181</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1016/j.jclinepi.2021.06.007">https://10.1016/j.jclinepi.2021.06.007</ext-link></mixed-citation></ref>
<ref id="ref-19"><label>19</label><mixed-citation publication-type="journal"><string-name><surname>Greenland</surname> <given-names>S</given-names></string-name>. <article-title>Noncollapsibility, confounding, and sparse-data bias. Part 2: What should researchers make of persistent controversies about the odds ratio?</article-title> <source>J Clin Epidemiol</source> <year>2021</year>;<volume>139</volume>:<fpage>264</fpage>-<lpage>268</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1016/j.jclinepi.2021.06.004">https://10.1016/j.jclinepi.2021.06.004</ext-link></mixed-citation></ref>
<ref id="ref-20"><label>20</label><mixed-citation publication-type="journal"><string-name><surname>Gail</surname> <given-names>MH</given-names></string-name>, <string-name><surname>Wieand</surname> <given-names>S</given-names></string-name>, <string-name><surname>Piantadosi</surname> <given-names>S</given-names></string-name>. <article-title>Biased estimates of treatment effect in randomized experiments with nonlinear regressions and omitted covariates</article-title>. <source>Biometrika</source> <year>1984</year>;<volume>71</volume>(<issue>3</issue>):<fpage>431</fpage>-<lpage>444</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1093/biomet/71.3.431">https://10.1093/biomet/71.3.431</ext-link></mixed-citation></ref>
<ref id="ref-21"><label>21</label><mixed-citation publication-type="journal"><string-name><surname>Schumacher</surname> <given-names>M</given-names></string-name>, <string-name><surname>Bastert</surname> <given-names>G</given-names></string-name>, <string-name><surname>Bojar</surname> <given-names>H</given-names></string-name>, <etal>et al</etal>. <article-title>Randomized 2 x 2 trial evaluating hormonal treatment and the duration of chemotherapy in node-positive breast cancer patients. German Breast Cancer Study Group</article-title>. <source>J Clin Oncol</source> <year>1994</year>;<volume>12</volume>(<issue>10</issue>):<fpage>2086</fpage>-<lpage>2093</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1200/JCO.1994.12.10.2086">https://10.1200/JCO.1994.12.10.2086</ext-link></mixed-citation></ref>
<ref id="ref-22"><label>22</label><mixed-citation publication-type="journal"><string-name><surname>Goodall</surname> <given-names>RL</given-names></string-name>, <string-name><surname>Fabiane</surname> <given-names>SM</given-names></string-name>, <string-name><surname>Hakiman</surname> <given-names>A</given-names></string-name>, <etal>et al</etal>. <article-title>A publicly accessible global data repository - the WHO TB-IPD platform</article-title>. <source>IJTLD Open</source> <year>2024</year>;<volume>1</volume>(<issue>4</issue>):<fpage>151</fpage>-<lpage>153</lpage>. <ext-link ext-link-type="uri" xlink:href="https:// 10.5588/ijtldopen.24.0131">https:// 10.5588/ijtldopen.24.0131</ext-link></mixed-citation></ref>
<ref id="ref-23"><label>23</label><mixed-citation publication-type="journal"><string-name><surname>James</surname> <given-names>ND</given-names></string-name>, <string-name><surname>Sydes</surname> <given-names>MR</given-names></string-name>, <string-name><surname>Clarke</surname> <given-names>NW</given-names></string-name>, <etal>et al</etal>. <article-title>Addition of docetaxel, zoledronic acid, or both to first-line long-term hormone therapy in prostate cancer (STAMPEDE): survival results from an adaptive, multiarm, multistage, platform randomised controlled trial</article-title>. <source>Lancet</source> <year>2016</year>;<volume>387</volume>(<issue>10024</issue>):<fpage>1163</fpage>-<lpage>1177</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1016/S0140-6736(15)01037-5">https://10.1016/S0140-6736(15)01037-5</ext-link></mixed-citation></ref>
<ref id="ref-24"><label>24</label><mixed-citation publication-type="website"><collab>Office for National Statistics</collab>. <article-title>Census 2021</article-title>. <source>England</source>. <year>2022</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.ons.gov.uk/methodology/geography/ukgeographies/administrativegeography/england">https://www.ons.gov.uk/methodology/geography/ukgeographies/administrativegeography/england</ext-link> [accessed 14 July 2025].</mixed-citation></ref>
<ref id="ref-25"><label>25</label><mixed-citation publication-type="book"><string-name><surname>Deeks</surname> <given-names>JJ</given-names></string-name>, <string-name><surname>Riley</surname> <given-names>R</given-names></string-name>, <string-name><surname>Higgins</surname> <given-names>JPT</given-names></string-name>. <chapter-title>Chapter 9: Combining results using meta-analysis</chapter-title>. In: <string-name><surname>Egger</surname> <given-names>M</given-names></string-name>, <string-name><surname>Higgins</surname> <given-names>JPT</given-names></string-name>, <string-name><surname>Smith</surname> <given-names>GD</given-names></string-name>, editors. <source>Systematic reviews in health research: Meta-analysis in context</source>. <edition>3rd ed</edition>. <publisher-loc>Hoboken, NJ</publisher-loc>: <publisher-name>John Wiley &amp; Sons</publisher-name>; <year>2022</year>.</mixed-citation></ref>
<ref id="ref-26"><label>26</label><mixed-citation publication-type="journal"><string-name><surname>Julious</surname> <given-names>SA</given-names></string-name>. <article-title>Sample sizes for clinical trials with normal data</article-title>. <source>Stat Med</source> <year>2004</year>;<volume>23</volume>(<issue>12</issue>):<fpage>1921</fpage>-<lpage>1986</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1002/sim.1783">https://10.1002/sim.1783</ext-link></mixed-citation></ref>
<ref id="ref-27"><label>27</label><mixed-citation publication-type="journal"><string-name><surname>White</surname> <given-names>IR</given-names></string-name>. <article-title>simsum: Analyses of simulation studies including Monte Carlo error</article-title>. <source>Stata Journal</source> <year>2010</year>;<volume>10</volume>(<issue>3</issue>):<fpage>369</fpage>-<lpage>385</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.stata-journal.com/article.html?article=st0200">https://www.stata-journal.com/article.html?article=st0200</ext-link> [accessed 28 January 2026].</mixed-citation></ref>
<ref id="ref-28"><label>28</label><mixed-citation publication-type="book"><collab>StataCorp</collab>. <year>2025</year>. <chapter-title>Stata Statistical Software: Release 19</chapter-title>. <publisher-loc>College Station, TX</publisher-loc>: <publisher-name>StataCorp LLC</publisher-name>.</mixed-citation></ref>
<ref id="ref-29"><label>29</label><mixed-citation publication-type="website"><article-title>PeRSEVERE: Principles for handling end-of-participation events in clinical trials research</article-title>. <source>University of Leeds</source>. Available from: <ext-link ext-link-type="uri" xlink:href="https://persevereprinciples.org/">https://persevereprinciples.org/</ext-link> [accessed 16 July 2025].</mixed-citation></ref>
<ref id="ref-30"><label>30</label><mixed-citation publication-type="journal"><string-name><surname>Jackson</surname> <given-names>D</given-names></string-name>, <string-name><surname>White</surname> <given-names>IR</given-names></string-name>. <article-title>When should meta-analysis avoid making hidden normality assumptions?</article-title> <source>Biometrical Journal</source> <year>2018</year>;<volume>60</volume>:<fpage>1040</fpage>-<lpage>1058</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1002/bimj.201800071">https://10.1002/bimj.201800071</ext-link></mixed-citation></ref>
<ref id="ref-31"><label>31</label><mixed-citation publication-type="journal"><string-name><surname>Silverwood</surname> <given-names>RJ</given-names></string-name>, <string-name><surname>Baranyi</surname> <given-names>G</given-names></string-name>, <string-name><surname>Calderwood</surname> <given-names>L</given-names></string-name>, <etal>et al</etal>. <article-title>Adjusting for confounding in population administrative data when confounders are only measured in a linked cohort</article-title>. <source>SocArXiv</source> <year>2025</year>;10.31235/osf.io/7ec6b_v1. <ext-link ext-link-type="uri" xlink:href="https://10.31235/osf.io/7ec6b_v1">https://10.31235/osf.io/7ec6b_v1</ext-link></mixed-citation></ref>
<ref id="ref-32"><label>32</label><mixed-citation publication-type="website"><collab>World Economic Forum</collab>. <article-title>Federated Data Systems: Balancing Innovation and Trust in the Use of Sensitive Data</article-title>. <year>2019</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www3.weforum.org/docs/WEF_Federated_Data_Systems_2019.pdf">https://www3.weforum.org/docs/WEF_Federated_Data_Systems_2019.pdf</ext-link>. [accessed 23 January 2026].</mixed-citation></ref>
<ref id="ref-33"><label>33</label><mixed-citation publication-type="website"><article-title>Unlocking NHS data for research - how to improve the regional Secure Data Environment network</article-title>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.abpi.org.uk/publications/unlocking-nhs-data-for-research-how-to-improve-the-regional-secure-data-environment-network/">https://www.abpi.org.uk/publications/unlocking-nhs-data-for-research-how-to-improve-the-regional-secure-data-environment-network/</ext-link>. [accessed 16 July 2025].</mixed-citation></ref>
<ref id="ref-34"><label>34</label><mixed-citation publication-type="website"><article-title>Department of Health and Social Care</article-title>. <source>Policy paper: Data access policy update</source>. <year>2023</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.gov.uk/government/publications/data-access-policy-update/data-access-policy-update">https://www.gov.uk/government/publications/data-access-policy-update/data-access-policy-update</ext-link> [accessed 14 July 2025].</mixed-citation></ref>
<ref id="ref-35"><label>35</label><mixed-citation publication-type="journal"><string-name><surname>Avraam</surname> <given-names>D</given-names></string-name>, <string-name><surname>Wilson</surname> <given-names>RC</given-names></string-name>, <string-name><surname>Aguirre Chan</surname> <given-names>N</given-names></string-name>, <etal>et al</etal>. <article-title>DataSHIELD: mitigating disclosure risk in a multi-site federated analysis platform</article-title>. <source>Bioinform Adv</source> <year>2025</year>;<volume>5</volume>(<issue>1</issue>):<fpage>vbaf046</fpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1093/bioadv/vbaf046">https://10.1093/bioadv/vbaf046</ext-link></mixed-citation></ref>
<ref id="ref-36"><label>36</label><mixed-citation publication-type="journal"><string-name><surname>Murray</surname> <given-names>ML</given-names></string-name>, <string-name><surname>Lugg-Widger</surname> <given-names>F</given-names></string-name>, <string-name><surname>Jobson</surname> <given-names>S</given-names></string-name>, <collab>the UK Health Data Research Alliance</collab>. <article-title>Data access and consent to use health systems data in clinical trials: Green paper for consultation</article-title>. <source>October 2025. Zenodo</source>. <year>2025</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.17406642">https://doi.org/10.5281/zenodo.17406642</ext-link> [accessed 21 October 2025].</mixed-citation></ref>
<ref id="ref-37"><label>37</label><mixed-citation publication-type="journal"><string-name><surname>Lin</surname> <given-names>DY</given-names></string-name>, <string-name><surname>Zeng</surname> <given-names>D</given-names></string-name>. <article-title>On the relative efficiency of using summary statistics versus individual-level data in meta-analysis</article-title>. <source>Biometrika</source> <year>2010</year>;<volume>97</volume>(<issue>2</issue>):<fpage>321</fpage>-<lpage>332</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1093/biomet/asq006">https://10.1093/biomet/asq006</ext-link></mixed-citation></ref>
<ref id="ref-38"><label>38</label><mixed-citation publication-type="journal"><string-name><surname>Sung</surname> <given-names>YJ</given-names></string-name>, <string-name><surname>Schwander</surname> <given-names>K</given-names></string-name>, <string-name><surname>Arnett</surname> <given-names>DK</given-names></string-name>, <etal>et al</etal>. <article-title>An empirical comparison of meta-analysis and mega-analysis of individual participant data for identifying gene-environment interactions</article-title>. <source>Genet Epidemiol</source> <year>2014</year>;<volume>38</volume>(<issue>4</issue>):<fpage>369</fpage>-<lpage>378</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1002/gepi.21800">https://10.1002/gepi.21800</ext-link></mixed-citation></ref>
<ref id="ref-39"><label>39</label><mixed-citation publication-type="journal"><string-name><surname>Giles</surname> <given-names>T</given-names></string-name>, <string-name><surname>Soiland-Reyes</surname> <given-names>S</given-names></string-name>, <string-name><surname>Couldridge</surname> <given-names>J</given-names></string-name>, <etal>et al</etal>. <article-title>TRE-FX: Delivering a federated network of trusted research environments to enable safe data analytics</article-title>. <source>Zenodo</source>. <year>2023</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.10055353">https://doi.org/10.5281/zenodo.10055353</ext-link> [accessed 20 January 2025].</mixed-citation></ref>
<ref id="ref-40"><label>40</label><mixed-citation publication-type="journal"><string-name><surname>Orton</surname> <given-names>C</given-names></string-name>, <string-name><surname>Thompson</surname> <given-names>S</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>A</given-names></string-name>, <etal>et al</etal>. <article-title>TELEPORT: Connecting researchers to big data at light speed</article-title>. <source>Zenodo</source>. <year>2023</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.10055357">https://doi.org/10.5281/zenodo.10055357</ext-link> [accessed 20 January 2025].</mixed-citation></ref>
<ref id="ref-41"><label>41</label><mixed-citation publication-type="journal"><collab>DARE UK (Data and Analytics Research Environments UK)</collab>. <article-title>UK Sensitive Data Research Infrastructure: A Landscape Review</article-title>. <source>Zenodo</source>. <year>2023</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.10082544">https://doi.org/10.5281/zenodo.10082544</ext-link> [accessed 20 January 2025].</mixed-citation></ref>
<ref id="ref-42"><label>42</label><mixed-citation publication-type="journal"><string-name><surname>Kent</surname> <given-names>S</given-names></string-name>, <string-name><surname>Burn</surname> <given-names>E</given-names></string-name>, <string-name><surname>Dawoud</surname> <given-names>D</given-names></string-name>, <etal>et al</etal>. <article-title>Common Problems, Common Data Model Solutions: Evidence Generation for Health Technology Assessment</article-title>. <source>Pharmacoeconomics</source> <year>2021</year>;<volume>39</volume>(<issue>3</issue>):<fpage>275</fpage>-<lpage>285</lpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1007/s40273-020-00981-9">https://10.1007/s40273-020-00981-9</ext-link></mixed-citation></ref>
<ref id="ref-43"><label>43</label><mixed-citation publication-type="journal"><string-name><surname>Biedermann</surname> <given-names>P</given-names></string-name>, <string-name><surname>Ong</surname> <given-names>R</given-names></string-name>, <string-name><surname>Davydov</surname> <given-names>A</given-names></string-name>, <etal>et al</etal>. <article-title>Standardizing registry data to the OMOP Common Data Model: experience from three pulmonary hypertension databases</article-title>. <source>BMC Med Res Methodol</source> <year>2021</year>;<volume>21</volume>(<issue>1</issue>):<fpage>238</fpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1186/s12874-021-01434-3">https://10.1186/s12874-021-01434-3</ext-link></mixed-citation></ref>
<ref id="ref-44"><label>44</label><mixed-citation publication-type="journal"><string-name><surname>Williams</surname> <given-names>ADN</given-names></string-name>, <string-name><surname>Davies</surname> <given-names>G</given-names></string-name>, <string-name><surname>Farrin</surname> <given-names>AJ</given-names></string-name>, <etal>et al</etal>. <article-title>A DELPHI study priority setting the remaining challenges for the use of routinely collected data in trials: COMORANT-UK</article-title>. <source>Trials</source> <year>2023</year>;<volume>24</volume>(<issue>1</issue>):<fpage>243</fpage>. <ext-link ext-link-type="uri" xlink:href="https://10.1186/s13063-023-07251-x">https://10.1186/s13063-023-07251-x</ext-link></mixed-citation></ref>
</ref-list>
<glossary>
<title>Abbreviations</title>
<array>
<tbody>
<tr>
<td>HIV:</td>
<td>Human Immunodeficiency Virus</td>
</tr>
<tr>
<td>HSD:</td>
<td>Health Systems Dataset</td>
</tr>
<tr>
<td>IPD:</td>
<td>Individual Participant Data</td>
</tr>
<tr>
<td>NSAID:</td>
<td>Non-Steroidal Anti-Inflammatory Drug</td>
</tr>
<tr>
<td>ONS:</td>
<td>Office for National Statistics</td>
</tr>
<tr>
<td>PSA:</td>
<td>Prostate-Specific Antigen</td>
</tr>
<tr>
<td>SDE:</td>
<td>Secure Data Environments</td>
</tr>
<tr>
<td>SE:</td>
<td>Standard Error</td>
</tr>
<tr>
<td>TB:</td>
<td>Tuberculosis</td>
</tr>
<tr>
<td>Var:</td>
<td>Variance</td>
</tr>
<tr>
<td>WHO:</td>
<td>World Health Organization</td>
</tr>
</tbody>
</array>
</glossary>
</back>
</article>
