array(40) {
["request_overridden_res"]=>
string(1) "3"
["project_status"]=>
string(30) "approved_pending_dua_signature"
["project_assoc_trials"]=>
array(5) {
[0]=>
object(WP_Post)#5798 (24) {
["ID"]=>
int(1259)
["post_author"]=>
string(4) "1363"
["post_date"]=>
string(19) "2014-10-20 16:13:00"
["post_date_gmt"]=>
string(19) "2014-10-20 16:13:00"
["post_content"]=>
string(0) ""
["post_title"]=>
string(300) "NCT01106625 - A Randomized, Double-Blind, Placebo-Controlled, 3-Arm, Parallel-Group, Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of Canagliflozin in the Treatment of Subjects With Type 2 Diabetes Mellitus With Inadequate Glycemic Control on Metformin and Pioglitazone Therapy"
["post_excerpt"]=>
string(0) ""
["post_status"]=>
string(7) "publish"
["comment_status"]=>
string(6) "closed"
["ping_status"]=>
string(6) "closed"
["post_password"]=>
string(0) ""
["post_name"]=>
string(190) "nct01106625-a-randomized-double-blind-placebo-controlled-3-arm-parallel-group-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-canagliflozin-in-the-treatment-of-subjects"
["to_ping"]=>
string(0) ""
["pinged"]=>
string(0) ""
["post_modified"]=>
string(19) "2025-10-24 15:45:13"
["post_modified_gmt"]=>
string(19) "2025-10-24 19:45:13"
["post_content_filtered"]=>
string(0) ""
["post_parent"]=>
int(0)
["guid"]=>
string(239) "https://dev-yoda.pantheonsite.io/clinical-trial/nct01106625-a-randomized-double-blind-placebo-controlled-3-arm-parallel-group-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-canagliflozin-in-the-treatment-of-subjects/"
["menu_order"]=>
int(0)
["post_type"]=>
string(14) "clinical_trial"
["post_mime_type"]=>
string(0) ""
["comment_count"]=>
string(1) "0"
["filter"]=>
string(3) "raw"
}
[1]=>
object(WP_Post)#5794 (24) {
["ID"]=>
int(1280)
["post_author"]=>
string(2) "20"
["post_date"]=>
string(19) "2014-10-20 16:22:00"
["post_date_gmt"]=>
string(19) "2014-10-20 16:22:00"
["post_content"]=>
string(0) ""
["post_title"]=>
string(296) "NCT01137812 - A Randomized, Double-Blind, Active-Controlled, Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of Canagliflozin Versus Sitagliptin in the Treatment of Subjects With Type 2 Diabetes Mellitus With Inadequate Glycemic Control on Metformin and Sulphonylurea Therapy"
["post_excerpt"]=>
string(0) ""
["post_status"]=>
string(7) "publish"
["comment_status"]=>
string(6) "closed"
["ping_status"]=>
string(6) "closed"
["post_password"]=>
string(0) ""
["post_name"]=>
string(192) "nct01137812-a-randomized-double-blind-active-controlled-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-canagliflozin-versus-sitagliptin-in-the-treatment-of-subjects-with"
["to_ping"]=>
string(0) ""
["pinged"]=>
string(0) ""
["post_modified"]=>
string(19) "2025-10-24 15:54:56"
["post_modified_gmt"]=>
string(19) "2025-10-24 19:54:56"
["post_content_filtered"]=>
string(0) ""
["post_parent"]=>
int(0)
["guid"]=>
string(241) "https://dev-yoda.pantheonsite.io/clinical-trial/nct01137812-a-randomized-double-blind-active-controlled-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-canagliflozin-versus-sitagliptin-in-the-treatment-of-subjects-with/"
["menu_order"]=>
int(0)
["post_type"]=>
string(14) "clinical_trial"
["post_mime_type"]=>
string(0) ""
["comment_count"]=>
string(1) "0"
["filter"]=>
string(3) "raw"
}
[2]=>
object(WP_Post)#5795 (24) {
["ID"]=>
int(1274)
["post_author"]=>
string(4) "1363"
["post_date"]=>
string(19) "2014-10-20 16:20:00"
["post_date_gmt"]=>
string(19) "2014-10-20 16:20:00"
["post_content"]=>
string(0) ""
["post_title"]=>
string(302) "NCT01106651 - A Randomized, Double-Blind, Placebo-Controlled, Parallel-Group, Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of Canagliflozin Compared With Placebo in the Treatment of Older Subjects With Type 2 Diabetes Mellitus Inadequately Controlled on Glucose Lowering Therapy"
["post_excerpt"]=>
string(0) ""
["post_status"]=>
string(7) "publish"
["comment_status"]=>
string(6) "closed"
["ping_status"]=>
string(6) "closed"
["post_password"]=>
string(0) ""
["post_name"]=>
string(192) "nct01106651-a-randomized-double-blind-placebo-controlled-parallel-group-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-canagliflozin-compared-with-placebo-in-the-treatme"
["to_ping"]=>
string(0) ""
["pinged"]=>
string(0) ""
["post_modified"]=>
string(19) "2025-10-24 15:52:51"
["post_modified_gmt"]=>
string(19) "2025-10-24 19:52:51"
["post_content_filtered"]=>
string(0) ""
["post_parent"]=>
int(0)
["guid"]=>
string(241) "https://dev-yoda.pantheonsite.io/clinical-trial/nct01106651-a-randomized-double-blind-placebo-controlled-parallel-group-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-canagliflozin-compared-with-placebo-in-the-treatme/"
["menu_order"]=>
int(0)
["post_type"]=>
string(14) "clinical_trial"
["post_mime_type"]=>
string(0) ""
["comment_count"]=>
string(1) "0"
["filter"]=>
string(3) "raw"
}
[3]=>
object(WP_Post)#5797 (24) {
["ID"]=>
int(1268)
["post_author"]=>
string(4) "1363"
["post_date"]=>
string(19) "2014-10-20 16:17:00"
["post_date_gmt"]=>
string(19) "2014-10-20 16:17:00"
["post_content"]=>
string(0) ""
["post_title"]=>
string(298) "NCT01106677 - A Randomized, Double-Blind, Placebo and Active-Controlled, 4-Arm, Parallel Group, Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of Canagliflozin in the Treatment of Subjects With Type 2 Diabetes Mellitus With Inadequate Glycemic Control on Metformin Monotherapy"
["post_excerpt"]=>
string(0) ""
["post_status"]=>
string(7) "publish"
["comment_status"]=>
string(6) "closed"
["ping_status"]=>
string(6) "closed"
["post_password"]=>
string(0) ""
["post_name"]=>
string(191) "nct01106677-a-randomized-double-blind-placebo-and-active-controlled-4-arm-parallel-group-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-canagliflozin-in-the-treatment-o"
["to_ping"]=>
string(0) ""
["pinged"]=>
string(0) ""
["post_modified"]=>
string(19) "2025-10-24 15:50:03"
["post_modified_gmt"]=>
string(19) "2025-10-24 19:50:03"
["post_content_filtered"]=>
string(0) ""
["post_parent"]=>
int(0)
["guid"]=>
string(240) "https://dev-yoda.pantheonsite.io/clinical-trial/nct01106677-a-randomized-double-blind-placebo-and-active-controlled-4-arm-parallel-group-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-canagliflozin-in-the-treatment-o/"
["menu_order"]=>
int(0)
["post_type"]=>
string(14) "clinical_trial"
["post_mime_type"]=>
string(0) ""
["comment_count"]=>
string(1) "0"
["filter"]=>
string(3) "raw"
}
[4]=>
object(WP_Post)#5796 (24) {
["ID"]=>
int(1271)
["post_author"]=>
string(4) "1363"
["post_date"]=>
string(19) "2014-10-20 16:18:00"
["post_date_gmt"]=>
string(19) "2014-10-20 16:18:00"
["post_content"]=>
string(0) ""
["post_title"]=>
string(302) "NCT00968812 - A Randomized, Double-Blind, 3-Arm Parallel-Group, 2-Year (104-Week), Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of JNJ-28431754 Compared With Glimepiride in the Treatment of Subjects With Type 2 Diabetes Mellitus Not Optimally Controlled on Metformin Monotherapy"
["post_excerpt"]=>
string(0) ""
["post_status"]=>
string(7) "publish"
["comment_status"]=>
string(6) "closed"
["ping_status"]=>
string(6) "closed"
["post_password"]=>
string(0) ""
["post_name"]=>
string(190) "nct00968812-a-randomized-double-blind-3-arm-parallel-group-2-year-104-week-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-jnj-28431754-compared-with-glimepiride-in-the"
["to_ping"]=>
string(0) ""
["pinged"]=>
string(0) ""
["post_modified"]=>
string(19) "2025-10-24 15:51:08"
["post_modified_gmt"]=>
string(19) "2025-10-24 19:51:08"
["post_content_filtered"]=>
string(0) ""
["post_parent"]=>
int(0)
["guid"]=>
string(239) "https://dev-yoda.pantheonsite.io/clinical-trial/nct00968812-a-randomized-double-blind-3-arm-parallel-group-2-year-104-week-multicenter-study-to-evaluate-the-efficacy-safety-and-tolerability-of-jnj-28431754-compared-with-glimepiride-in-the/"
["menu_order"]=>
int(0)
["post_type"]=>
string(14) "clinical_trial"
["post_mime_type"]=>
string(0) ""
["comment_count"]=>
string(1) "0"
["filter"]=>
string(3) "raw"
}
}
["project_title"]=>
string(69) "Replication of external control arm methods: PS, G-comp, DDML in T2DM"
["project_narrative_summary"]=>
string(749) "External control arms are comparison groups built from external data when a clinical trial lacks a traditional control group. A 2022 study by Loiseau and colleagues compared three methods—propensity score matching, G‑computation, and doubly debiased machine learning—for building such arms, using five Type 2 diabetes trials from the YODA Project. Our project will independently repeat that study to check if its findings are reliable. Reproducibility is essential for trustworthy science. If we confirm the results, researchers and regulators will have more confidence when using these methods in future drug development and approval decisions. All our analysis code will be made publicly available so others can verify and build on our work."
["project_learn_source"]=>
string(6) "pubmed"
["principal_investigator"]=>
array(7) {
["first_name"]=>
string(5) "Bokai"
["last_name"]=>
string(4) "Chen"
["degree"]=>
string(3) "MSc"
["primary_affiliation"]=>
string(28) "Shanghai Maritime University"
["email"]=>
string(22) "bokaichen362@gmail.com"
["state_or_province"]=>
string(8) "Shanghai"
["country"]=>
string(5) "China"
}
["project_key_personnel"]=>
bool(false)
["project_ext_grants"]=>
array(2) {
["value"]=>
string(2) "no"
["label"]=>
string(68) "No external grants or funds are being used to support this research."
}
["project_date_type"]=>
string(18) "full_crs_supp_docs"
["property_scientific_abstract"]=>
string(1294) "Background: External control arms (ECAs) use external data to create control groups for single-arm trials. Patient characteristics must be aligned, often via propensity scores. Alternative methods include G‑computation and doubly debiased machine learning (DDML), but their evaluation in ECA analysis is limited.
Objective: To independently replicate Loiseau et al. (2022) comparing propensity score, G‑computation, and DDML for ECA construction using five Type 2 diabetes trials.
Study Design: Methodological replication using participant‑level data from five randomized trials.
Participants: Patients from NCT01106625, NCT01137812, NCT01106651, NCT01106677, NCT00968812.
Primary and Secondary Outcomes: Primary: estimated average treatment effect (ATE) between treatment arm and external control built by each method. Secondary: bias, variance, mean squared error, and 95% coverage probability of ATE estimates across configurations.
Statistical Analysis: We will apply (1) propensity score matching and weighting; (2) G‑computation; (3) DDML. We will vary the target trial and external control source as in the original cross‑validation design. Results benchmarked against internal controls. Analyses in R."
["project_brief_bg"]=>
string(3014) "The use of external control arms in clinical trials has gained substantial attention as a strategy to reduce trial costs, accelerate drug development, and provide comparative evidence when randomized controlled trials are infeasible or unethical. A growing body of methodological research has examined various statistical approaches for constructing external controls from historical trial data or real-world data sources. These approaches broadly fall into two categories: propensity score-based methods that align patient characteristics through weighting or matching, and outcome prediction-based methods such as G-computation and Doubly Debiased Machine Learning (DDML) that infer control outcomes using machine learning models.
Loiseau et al. (2022) published a comprehensive evaluation comparing these three major classes of methods using data from five Type 2 diabetes trials made available through the YODA Project. Their findings have important implications for the design of future clinical trials and regulatory submissions that rely on external control data. The study concluded that methods based on outcome prediction models can reduce estimation error and increase statistical power compared to propensity score approaches. Specifically, DDML demonstrated the smallest bias followed by G-computation, while G-computation minimized mean squared error and provided the narrowest confidence intervals.
The significance of this proposed replication study is threefold. First, independent replication is a fundamental principle of the scientific method and is essential for establishing the credibility of methodological recommendations. Without confirmation through replication, even well-conducted studies may produce findings that do not generalize to independent settings. Second, confirming these results will provide stronger evidence for researchers and regulators considering the adoption of these methods in drug development programs and regulatory decision-making. Third, should any discrepancies arise between our replication results and the original findings, this will provide valuable insights into the conditions under which these methods perform optimally, thereby enhancing the generalizability of the methodological knowledge.
The findings of this replication study will directly contribute to the growing body of knowledge on reproducible research practices in clinical trial methodology. By making our analysis code publicly available, we will further promote transparency and enable other researchers to build upon this work. This project aligns with the YODA Project's mission to promote open science and data sharing to advance public health.
References:
Loiseau, N., Trichelair, P., He, M., et al. (2022). External control arm analysis: an evaluation of propensity score approaches, G-computation, and doubly debiased machine learning. BMC Medical Research Methodology, 22, 335. PMID: 36577946; PMCID: PMC9795588."
["project_specific_aims"]=>
string(1679) "The overarching goal of this replication study is to independently reproduce the analytical results reported in Loiseau et al. (2022) and to assess the robustness of their conclusions regarding three methods for external control arm construction.
The specific aims are as follows:
To independently reproduce the point estimates, standard errors, and 95% confidence intervals for treatment effects obtained from each of the three methods—propensity score matching/weighting, G-computation, and doubly debiased machine learning (DDML)—across all pairwise trial combinations used in the original study.
To verify the key performance metrics reported in the original study, including bias, empirical variance, mean squared error, and 95% coverage probability of the treatment effect estimates, under the exact cross-validation design (each trial serving in turn as the single-arm target and another as the external control source).
To test the primary hypothesis that DDML yields lower bias and G-computation yields lower mean squared error compared to propensity score methods, consistent with the original study's conclusions.
To document any discrepancies between our replication results and the original findings, and where discrepancies exist, to explore potential sources—such as differences in software versions, implementation details, or data processing—to inform best practices for future replication efforts.
All replication code will be made publicly available to ensure full transparency and to enable other researchers to extend or challenge our findings.
"
["project_study_design"]=>
array(2) {
["value"]=>
string(8) "meth_res"
["label"]=>
string(23) "Methodological research"
}
["project_purposes"]=>
array(2) {
[0]=>
array(2) {
["value"]=>
string(76) "confirm_or_validate previously_conducted_research_on_treatment_effectiveness"
["label"]=>
string(76) "Confirm or validate previously conducted research on treatment effectiveness"
}
[1]=>
array(2) {
["value"]=>
string(34) "research_on_clinical_trial_methods"
["label"]=>
string(34) "Research on clinical trial methods"
}
}
["project_research_methods"]=>
string(1511) "Data Sources:
This study will utilize participant-level data from five randomized clinical trials in patients with Type 2 diabetes mellitus, which are available through the YODA Project. The five trials are identified by the following NCT numbers: NCT01106625, NCT01137812, NCT01106651, NCT01106677, and NCT00968812.
Inclusion and Exclusion Criteria:
To faithfully replicate the original study by Loiseau et al. (2022), we will apply the same inclusion and exclusion criteria as those used in each of the five individual trials. No additional exclusion criteria will be applied beyond those specified in the original trial protocols.
The specific eligibility criteria for each trial are documented in the respective trial protocols and summary documents available through the YODA Project. In general, these trials enrolled adult patients with Type 2 diabetes mellitus who met the diagnostic criteria for the condition, with varying requirements regarding glycemic control (e.g., HbA1c levels), prior or current antidiabetic treatments, and other clinical parameters. Detailed inclusion/exclusion criteria for each trial will be obtained from the YODA Project data repository upon data access.
No data from sources outside the YODA Project will be used in this replication study. All analyses will be conducted using individual participant-level data (IPD) from the five trials listed above. No aggregate-level meta-analysis is planned."
["project_main_outcome_measure"]=>
string(1907) "Primary Outcome Measure:
The primary outcome is the estimated Average Treatment Effect (ATE) comparing the treatment arm to the external control arm constructed using each of the three methodological approaches. The treatment effect is measured as the difference in means of the continuous outcome variable between the treatment group and the external control group.
Secondary Outcome Measures:
The secondary outcomes include the following performance metrics for each method:
Bias: The difference between the external control-based treatment effect estimate and the true treatment effect estimated using the original internal control arm as the reference standard.
Variance: The empirical variance of the treatment effect estimates across different data configurations.
Mean Squared Error (MSE): The mean squared error of the treatment effect estimates.
95% Coverage Probability: The proportion of 95% confidence intervals that contain the true treatment effect.
Statistical Power: The probability of correctly rejecting the null hypothesis of no treatment effect.
Type I Error Rate: The probability of incorrectly rejecting the null hypothesis when the true treatment effect is zero.
Classification and Definition:
The treatment effect will be defined as the difference in means of the outcome variable between the treatment arm and the control arm. For the null replication experiments, the expected target ATE is 0. For the positive replication experiments, an artificial positive treatment effect is added to each patient outcome in the treatment arm. All outcome measures will be reported with 95% confidence intervals. No changes to the primary or secondary outcome measures are planned from those reported in the original publication."
["project_main_predictor_indep"]=>
string(1007) "Primary Predictor/Independent Variable:
The primary independent variable is the treatment assignment, defined as a binary variable indicating whether a patient belongs to the treatment arm (coded as 1) or to the external control arm (coded as 0).
Classification and Definition:
Treatment assignment is determined by the original trial design and the artificial observational experiment construction. In each observational experiment, one trial serves as the "target" single-arm trial, and the treatment arm from this trial constitutes the treatment group. Another trial from the pool of five serves as the source of the external control arm, and the control arm from this trial constitutes the external control group. This binary variable will be used as the primary predictor in all analyses to estimate the treatment effect.
No changes to the definition of the primary independent variable are planned from that used in the original publication."
["project_other_variables_interest"]=>
string(1805) "Other Variables of Interest:
The following covariates will be used in the analysis for propensity score estimation, outcome modeling, and risk adjustment:
1. Baseline Demographic and Clinical Covariates:
The specific covariates available in each of the five YODA trials will be used for adjustment. Based on the nature of Type 2 diabetes trials, these are expected to include, but are not limited to:
Age (continuous, in years)
Sex (binary, male/female)
Body Mass Index (BMI, continuous)
HbA1c levels (continuous, percentage or mmol/mol)
Fasting plasma glucose (continuous)
Duration of diabetes (continuous, in years)
Previous or current antidiabetic treatments (categorical)
Baseline renal function (e.g., eGFR, continuous)
Blood pressure (continuous)
Lipid profile (continuous)
2. Trial/Source Indicators:
Trial identifier (categorical, indicating which of the five trials the patient belongs to)
Arm indicator (categorical, treatment vs. control within the original trial)
Classification and Definition:
All covariates will be used in their original scale as provided in the YODA dataset. Continuous variables will not be categorized unless specified otherwise in the original analysis. Covariates will be standardized if required by the machine learning algorithms used in the doubly debiased machine learning approach. The specific set of covariates used for adjustment will follow the original publication's approach, and detailed variable definitions will be obtained from the trial protocols available through the YODA Project."
["project_stat_analysis_plan"]=>
string(4092) "Statistical Analysis Plan:
This replication study will closely follow the analytical approach described in Loiseau et al. (2022). The analysis will be conducted in R and will proceed in the following steps:
1. Descriptive Analysis:
For each of the five trials, we will compute descriptive statistics for all baseline covariates, stratified by treatment arm. Continuous variables will be summarized using means, standard deviations, medians, and interquartile ranges. Categorical variables will be summarized using frequencies and percentages. Standardized mean differences (SMD) will be calculated to compare covariate balance between the treatment arm and the external control arm before and after adjustment.
2. Observational Experiment Construction:
Following the original study's replication procedure, we will systematically construct observational experiments from the pool of five trials. In each experiment, one trial serves as the "target" single-arm trial (with its control arm removed), and another trial serves as the source of the external control arm. The treatment arm from the target trial is compared against the control arm from the source trial. This creates a total of 20 pairwise combinations (5 × 4), as in the original study.
3. Implementation of Three Methods:
We will implement the three methodological approaches as described in the original publication:
a) Propensity Score Approaches:
Propensity Score Matching (PSM): We will estimate propensity scores using logistic regression with treatment assignment as the outcome and baseline covariates as predictors. Patients will be matched using nearest-neighbor matching without replacement, with a caliper of 0.2 times the standard deviation of the logit of the propensity score.
Inverse Probability of Treatment Weighting (IPTW): We will estimate weights as the inverse of the propensity score for treated patients and the inverse of (1 - propensity score) for control patients. Stabilized weights may be used to reduce variability.
b) G-computation:
We will fit a parametric regression model for the outcome conditional on treatment assignment and baseline covariates. Using the fitted model, we will predict potential outcomes for each patient under both treatment and control scenarios. The treatment effect is estimated as the average difference between these predicted potential outcomes.
c) Doubly Debiased Machine Learning (DDML):
We will implement DDML using cross-fitting to reduce overfitting bias. This approach combines machine learning-based outcome modeling with propensity score estimation to achieve doubly robust and debiased treatment effect estimates. The specific machine learning algorithms (e.g., random forest, gradient boosting, or Lasso) will be selected to match those used in the original study.
4. Performance Evaluation:
For each method and each observational experiment, we will compute:
Point estimates and 95% confidence intervals of the treatment effect
Bias relative to the true treatment effect (estimated using the original internal control arm)
Empirical variance across experiments
Mean squared error (MSE)
95% coverage probability
Type I error rate and statistical power
5. Hypothesis Testing:
For each method, we will test the null hypothesis of no treatment effect using Wald-type tests based on the estimated treatment effects and their standard errors.
6. Sensitivity Analyses:
We will conduct sensitivity analyses to assess the robustness of our findings to different specifications of the propensity score model, different machine learning algorithms, and different matching calipers.
All analysis code will be made publicly available to ensure full reproducibility."
["project_software_used"]=>
array(3) {
[0]=>
array(2) {
["value"]=>
string(6) "python"
["label"]=>
string(6) "Python"
}
[1]=>
array(2) {
["value"]=>
string(1) "r"
["label"]=>
string(1) "R"
}
[2]=>
array(2) {
["value"]=>
string(7) "rstudio"
["label"]=>
string(7) "RStudio"
}
}
["project_timeline"]=>
string(1063) "Project Timeline:
The following timeline is proposed for this replication study:
Milestone Estimated Date
Data request submission August 2026
Data access approval (estimated) October 2026
Data acquisition and preparation October 2026
Descriptive analysis and covariate balance assessment November 2026
Implementation of propensity score methods November 2026
Implementation of G-computation December 2026
Implementation of DDML December 2026
Performance evaluation and results compilation January 2027
Sensitivity analyses January 2027
Manuscript drafting February 2027
Internal review and revisions March 2027
First manuscript submission April 2027
Results reported to YODA Project April 2027
Total estimated duration: Approximately 6 months from data access to manuscript submission.
Please note that the YODA Project Data Use Agreement allows for 12 months of data access, with the possibility of extension if needed."
["project_dissemination_plan"]=>
string(1282) "Dissemination Plan:
Target Products:
1. Primary Manuscript: A replication results manuscript will be submitted to BMC Medical Research Methodology, Statistics in Medicine, Journal of Clinical Epidemiology, or Clinical Trials. We will compare our findings with the original results, discuss discrepancies, and provide insights into the reproducibility of ECA methodology research.
2. Replication Code: All analysis code will be publicly released on GitHub with clear documentation for reproduction.
3. Preprint: A preprint will be deposited on medRxiv or arXiv prior to journal submission.
Target Audiences:
- Academic researchers: methodologists, biostatisticians, and clinical trialists working on ECAs and causal inference.
- Regulatory agencies: FDA, EMA, and other bodies using external control data for regulatory decisions.
- Pharmaceutical industry: drug development teams considering ECAs in trial design.
- YODA Project community: researchers using YODA data who may benefit from replication methods.
YODA Project Reporting:
Results will be reported to YODA upon manuscript submission, including a summary of findings and a link to the published manuscript when available."
["project_bibliography"]=>
string(1448) "参考:
-
Loiseau N, Trichelair P, He M, et al. External control arm analysis: an evaluation of propensity score approaches, G-computation, and doubly debiased machine learning. BMC Med Res Methodol. 2022;22(1):335. doi:10.1186/s12874-022-01799-z. PMID: 36577946; PMCID: PMC9795588.
-
Rosenbaum PR, Rubin DB. The central role of the propensity score in observational studies for causal effects. Biometrika. 1983;70(1):41-55. doi:10.1093/biomet/70.1.41.
-
Robins JM, Hernán MA, Brumback B. Marginal structural models and causal inference in epidemiology. Epidemiology. 2000;11(5):550-560. doi:10.1097/00001648-200009000-00011.
-
Chernozhukov V, Chetverikov D, Demirer M, et al. Double/debiased machine learning for treatment and structural parameters. Econom J. 2018;21(1):C1-C68. doi:10.1111/ectj.12097.
-
Snowden JM, Rose S, Mortimer KM. Implementation of G-computation on a simulated data set: demonstration of a causal inference technique. Am J Epidemiol. 2011;173(7):731-738. doi:10.1093/aje/kwq472.
"
["project_suppl_material"]=>
bool(false)
["project_coi"]=>
array(1) {
[0]=>
array(1) {
["file_coi"]=>
array(21) {
["ID"]=>
int(19839)
["id"]=>
int(19839)
["title"]=>
string(40) "SV_57KskaKADT3U9Aq-R_9zkYocTYebl3QdY.pdf"
["filename"]=>
string(40) "SV_57KskaKADT3U9Aq-R_9zkYocTYebl3QdY.pdf"
["filesize"]=>
int(37181)
["url"]=>
string(89) "https://yoda.yale.edu/wp-content/uploads/2026/08/SV_57KskaKADT3U9Aq-R_9zkYocTYebl3QdY.pdf"
["link"]=>
string(86) "https://yoda.yale.edu/data-request/2026-0748/sv_57kskakadt3u9aq-r_9zkyoctyebl3qdy-pdf/"
["alt"]=>
string(0) ""
["author"]=>
string(4) "2214"
["description"]=>
string(0) ""
["caption"]=>
string(0) ""
["name"]=>
string(40) "sv_57kskakadt3u9aq-r_9zkyoctyebl3qdy-pdf"
["status"]=>
string(7) "inherit"
["uploaded_to"]=>
int(19834)
["date"]=>
string(19) "2026-08-13 16:36:51"
["modified"]=>
string(19) "2026-08-13 16:36:53"
["menu_order"]=>
int(0)
["mime_type"]=>
string(15) "application/pdf"
["type"]=>
string(11) "application"
["subtype"]=>
string(3) "pdf"
["icon"]=>
string(62) "https://yoda.yale.edu/wp/wp-includes/images/media/document.png"
}
}
}
["data_use_agreement_training"]=>
bool(true)
["human_research_protection_training"]=>
bool(true)
["certification"]=>
bool(true)
["search_order"]=>
string(1) "0"
["project_send_email_updates"]=>
bool(false)
["project_publ_available"]=>
bool(true)
["project_year_access"]=>
string(0) ""
["project_rep_publ"]=>
bool(false)
["project_assoc_data"]=>
array(0) {
}
["project_due_dil_assessment"]=>
bool(false)
["project_title_link"]=>
array(21) {
["ID"]=>
int(19765)
["id"]=>
int(19765)
["title"]=>
string(28) "Data Request Approved Notice"
["filename"]=>
string(32) "Data-Request-Approved-Notice.pdf"
["filesize"]=>
int(195663)
["url"]=>
string(81) "https://yoda.yale.edu/wp-content/uploads/2026/07/Data-Request-Approved-Notice.pdf"
["link"]=>
string(77) "https://yoda.yale.edu/data-request/2026-0640/data-request-approved-notice-72/"
["alt"]=>
string(0) ""
["author"]=>
string(4) "1885"
["description"]=>
string(0) ""
["caption"]=>
string(0) ""
["name"]=>
string(31) "data-request-approved-notice-72"
["status"]=>
string(7) "inherit"
["uploaded_to"]=>
int(19689)
["date"]=>
string(19) "2026-08-06 16:54:58"
["modified"]=>
string(19) "2026-08-06 16:54:58"
["menu_order"]=>
int(0)
["mime_type"]=>
string(15) "application/pdf"
["type"]=>
string(11) "application"
["subtype"]=>
string(3) "pdf"
["icon"]=>
string(62) "https://yoda.yale.edu/wp/wp-includes/images/media/document.png"
}
["project_review_link"]=>
bool(false)
["project_highlight_button"]=>
string(0) ""
["request_data_partner"]=>
string(0) ""
}
data partner
array(1) {
[0]=>
string(0) ""
}
pi country
array(0) {
}
pi affil
array(0) {
}
products
array(0) {
}
num of trials
array(1) {
[0]=>
string(1) "0"
}
res
array(1) {
[0]=>
string(1) "3"
}
General Information
How did you learn about the YODA Project?:
PubMed
Conflict of Interest
Request Clinical Trials
Associated Trial(s):
- NCT01106625 - A Randomized, Double-Blind, Placebo-Controlled, 3-Arm, Parallel-Group, Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of Canagliflozin in the Treatment of Subjects With Type 2 Diabetes Mellitus With Inadequate Glycemic Control on Metformin and Pioglitazone Therapy
- NCT01137812 - A Randomized, Double-Blind, Active-Controlled, Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of Canagliflozin Versus Sitagliptin in the Treatment of Subjects With Type 2 Diabetes Mellitus With Inadequate Glycemic Control on Metformin and Sulphonylurea Therapy
- NCT01106651 - A Randomized, Double-Blind, Placebo-Controlled, Parallel-Group, Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of Canagliflozin Compared With Placebo in the Treatment of Older Subjects With Type 2 Diabetes Mellitus Inadequately Controlled on Glucose Lowering Therapy
- NCT01106677 - A Randomized, Double-Blind, Placebo and Active-Controlled, 4-Arm, Parallel Group, Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of Canagliflozin in the Treatment of Subjects With Type 2 Diabetes Mellitus With Inadequate Glycemic Control on Metformin Monotherapy
- NCT00968812 - A Randomized, Double-Blind, 3-Arm Parallel-Group, 2-Year (104-Week), Multicenter Study to Evaluate the Efficacy, Safety, and Tolerability of JNJ-28431754 Compared With Glimepiride in the Treatment of Subjects With Type 2 Diabetes Mellitus Not Optimally Controlled on Metformin Monotherapy
What type of data are you looking for?:
Individual Participant-Level Data, which includes Full CSR and all supporting documentation
Request Clinical Trials
Data Request Status
Status:
Approved Pending DUA Signature
Research Proposal
Project Title:
Replication of external control arm methods: PS, G-comp, DDML in T2DM
Scientific Abstract:
Background: External control arms (ECAs) use external data to create control groups for single-arm trials. Patient characteristics must be aligned, often via propensity scores. Alternative methods include G‑computation and doubly debiased machine learning (DDML), but their evaluation in ECA analysis is limited.
Objective: To independently replicate Loiseau et al. (2022) comparing propensity score, G‑computation, and DDML for ECA construction using five Type 2 diabetes trials.
Study Design: Methodological replication using participant‑level data from five randomized trials.
Participants: Patients from NCT01106625, NCT01137812, NCT01106651, NCT01106677, NCT00968812.
Primary and Secondary Outcomes: Primary: estimated average treatment effect (ATE) between treatment arm and external control built by each method. Secondary: bias, variance, mean squared error, and 95% coverage probability of ATE estimates across configurations.
Statistical Analysis: We will apply (1) propensity score matching and weighting; (2) G‑computation; (3) DDML. We will vary the target trial and external control source as in the original cross‑validation design. Results benchmarked against internal controls. Analyses in R.
Brief Project Background and Statement of Project Significance:
The use of external control arms in clinical trials has gained substantial attention as a strategy to reduce trial costs, accelerate drug development, and provide comparative evidence when randomized controlled trials are infeasible or unethical. A growing body of methodological research has examined various statistical approaches for constructing external controls from historical trial data or real-world data sources. These approaches broadly fall into two categories: propensity score-based methods that align patient characteristics through weighting or matching, and outcome prediction-based methods such as G-computation and Doubly Debiased Machine Learning (DDML) that infer control outcomes using machine learning models.
Loiseau et al. (2022) published a comprehensive evaluation comparing these three major classes of methods using data from five Type 2 diabetes trials made available through the YODA Project. Their findings have important implications for the design of future clinical trials and regulatory submissions that rely on external control data. The study concluded that methods based on outcome prediction models can reduce estimation error and increase statistical power compared to propensity score approaches. Specifically, DDML demonstrated the smallest bias followed by G-computation, while G-computation minimized mean squared error and provided the narrowest confidence intervals.
The significance of this proposed replication study is threefold. First, independent replication is a fundamental principle of the scientific method and is essential for establishing the credibility of methodological recommendations. Without confirmation through replication, even well-conducted studies may produce findings that do not generalize to independent settings. Second, confirming these results will provide stronger evidence for researchers and regulators considering the adoption of these methods in drug development programs and regulatory decision-making. Third, should any discrepancies arise between our replication results and the original findings, this will provide valuable insights into the conditions under which these methods perform optimally, thereby enhancing the generalizability of the methodological knowledge.
The findings of this replication study will directly contribute to the growing body of knowledge on reproducible research practices in clinical trial methodology. By making our analysis code publicly available, we will further promote transparency and enable other researchers to build upon this work. This project aligns with the YODA Project's mission to promote open science and data sharing to advance public health.
References:
Loiseau, N., Trichelair, P., He, M., et al. (2022). External control arm analysis: an evaluation of propensity score approaches, G-computation, and doubly debiased machine learning. BMC Medical Research Methodology, 22, 335. PMID: 36577946; PMCID: PMC9795588.
Specific Aims of the Project:
The overarching goal of this replication study is to independently reproduce the analytical results reported in Loiseau et al. (2022) and to assess the robustness of their conclusions regarding three methods for external control arm construction.
The specific aims are as follows:
To independently reproduce the point estimates, standard errors, and 95% confidence intervals for treatment effects obtained from each of the three methods--propensity score matching/weighting, G-computation, and doubly debiased machine learning (DDML)--across all pairwise trial combinations used in the original study.
To verify the key performance metrics reported in the original study, including bias, empirical variance, mean squared error, and 95% coverage probability of the treatment effect estimates, under the exact cross-validation design (each trial serving in turn as the single-arm target and another as the external control source).
To test the primary hypothesis that DDML yields lower bias and G-computation yields lower mean squared error compared to propensity score methods, consistent with the original study's conclusions.
To document any discrepancies between our replication results and the original findings, and where discrepancies exist, to explore potential sources--such as differences in software versions, implementation details, or data processing--to inform best practices for future replication efforts.
All replication code will be made publicly available to ensure full transparency and to enable other researchers to extend or challenge our findings.
Study Design:
Methodological research
What is the purpose of the analysis being proposed? Please select all that apply.:
Confirm or validate previously conducted research on treatment effectiveness
Research on clinical trial methods
Software Used:
Python, R, RStudio
Data Source and Inclusion/Exclusion Criteria to be used to define the patient sample for your study:
Data Sources:
This study will utilize participant-level data from five randomized clinical trials in patients with Type 2 diabetes mellitus, which are available through the YODA Project. The five trials are identified by the following NCT numbers: NCT01106625, NCT01137812, NCT01106651, NCT01106677, and NCT00968812.
Inclusion and Exclusion Criteria:
To faithfully replicate the original study by Loiseau et al. (2022), we will apply the same inclusion and exclusion criteria as those used in each of the five individual trials. No additional exclusion criteria will be applied beyond those specified in the original trial protocols.
The specific eligibility criteria for each trial are documented in the respective trial protocols and summary documents available through the YODA Project. In general, these trials enrolled adult patients with Type 2 diabetes mellitus who met the diagnostic criteria for the condition, with varying requirements regarding glycemic control (e.g., HbA1c levels), prior or current antidiabetic treatments, and other clinical parameters. Detailed inclusion/exclusion criteria for each trial will be obtained from the YODA Project data repository upon data access.
No data from sources outside the YODA Project will be used in this replication study. All analyses will be conducted using individual participant-level data (IPD) from the five trials listed above. No aggregate-level meta-analysis is planned.
Primary and Secondary Outcome Measure(s) and how they will be categorized/defined for your study:
Primary Outcome Measure:
The primary outcome is the estimated Average Treatment Effect (ATE) comparing the treatment arm to the external control arm constructed using each of the three methodological approaches. The treatment effect is measured as the difference in means of the continuous outcome variable between the treatment group and the external control group.
Secondary Outcome Measures:
The secondary outcomes include the following performance metrics for each method:
Bias: The difference between the external control-based treatment effect estimate and the true treatment effect estimated using the original internal control arm as the reference standard.
Variance: The empirical variance of the treatment effect estimates across different data configurations.
Mean Squared Error (MSE): The mean squared error of the treatment effect estimates.
95% Coverage Probability: The proportion of 95% confidence intervals that contain the true treatment effect.
Statistical Power: The probability of correctly rejecting the null hypothesis of no treatment effect.
Type I Error Rate: The probability of incorrectly rejecting the null hypothesis when the true treatment effect is zero.
Classification and Definition:
The treatment effect will be defined as the difference in means of the outcome variable between the treatment arm and the control arm. For the null replication experiments, the expected target ATE is 0. For the positive replication experiments, an artificial positive treatment effect is added to each patient outcome in the treatment arm. All outcome measures will be reported with 95% confidence intervals. No changes to the primary or secondary outcome measures are planned from those reported in the original publication.
Main Predictor/Independent Variable and how it will be categorized/defined for your study:
Primary Predictor/Independent Variable:
The primary independent variable is the treatment assignment, defined as a binary variable indicating whether a patient belongs to the treatment arm (coded as 1) or to the external control arm (coded as 0).
Classification and Definition:
Treatment assignment is determined by the original trial design and the artificial observational experiment construction. In each observational experiment, one trial serves as the "target" single-arm trial, and the treatment arm from this trial constitutes the treatment group. Another trial from the pool of five serves as the source of the external control arm, and the control arm from this trial constitutes the external control group. This binary variable will be used as the primary predictor in all analyses to estimate the treatment effect.
No changes to the definition of the primary independent variable are planned from that used in the original publication.
Other Variables of Interest that will be used in your analysis and how they will be categorized/defined for your study:
Other Variables of Interest:
The following covariates will be used in the analysis for propensity score estimation, outcome modeling, and risk adjustment:
1. Baseline Demographic and Clinical Covariates:
The specific covariates available in each of the five YODA trials will be used for adjustment. Based on the nature of Type 2 diabetes trials, these are expected to include, but are not limited to:
Age (continuous, in years)
Sex (binary, male/female)
Body Mass Index (BMI, continuous)
HbA1c levels (continuous, percentage or mmol/mol)
Fasting plasma glucose (continuous)
Duration of diabetes (continuous, in years)
Previous or current antidiabetic treatments (categorical)
Baseline renal function (e.g., eGFR, continuous)
Blood pressure (continuous)
Lipid profile (continuous)
2. Trial/Source Indicators:
Trial identifier (categorical, indicating which of the five trials the patient belongs to)
Arm indicator (categorical, treatment vs. control within the original trial)
Classification and Definition:
All covariates will be used in their original scale as provided in the YODA dataset. Continuous variables will not be categorized unless specified otherwise in the original analysis. Covariates will be standardized if required by the machine learning algorithms used in the doubly debiased machine learning approach. The specific set of covariates used for adjustment will follow the original publication's approach, and detailed variable definitions will be obtained from the trial protocols available through the YODA Project.
Statistical Analysis Plan:
Statistical Analysis Plan:
This replication study will closely follow the analytical approach described in Loiseau et al. (2022). The analysis will be conducted in R and will proceed in the following steps:
1. Descriptive Analysis:
For each of the five trials, we will compute descriptive statistics for all baseline covariates, stratified by treatment arm. Continuous variables will be summarized using means, standard deviations, medians, and interquartile ranges. Categorical variables will be summarized using frequencies and percentages. Standardized mean differences (SMD) will be calculated to compare covariate balance between the treatment arm and the external control arm before and after adjustment.
2. Observational Experiment Construction:
Following the original study's replication procedure, we will systematically construct observational experiments from the pool of five trials. In each experiment, one trial serves as the "target" single-arm trial (with its control arm removed), and another trial serves as the source of the external control arm. The treatment arm from the target trial is compared against the control arm from the source trial. This creates a total of 20 pairwise combinations (5 x 4), as in the original study.
3. Implementation of Three Methods:
We will implement the three methodological approaches as described in the original publication:
a) Propensity Score Approaches:
Propensity Score Matching (PSM): We will estimate propensity scores using logistic regression with treatment assignment as the outcome and baseline covariates as predictors. Patients will be matched using nearest-neighbor matching without replacement, with a caliper of 0.2 times the standard deviation of the logit of the propensity score.
Inverse Probability of Treatment Weighting (IPTW): We will estimate weights as the inverse of the propensity score for treated patients and the inverse of (1 - propensity score) for control patients. Stabilized weights may be used to reduce variability.
b) G-computation:
We will fit a parametric regression model for the outcome conditional on treatment assignment and baseline covariates. Using the fitted model, we will predict potential outcomes for each patient under both treatment and control scenarios. The treatment effect is estimated as the average difference between these predicted potential outcomes.
c) Doubly Debiased Machine Learning (DDML):
We will implement DDML using cross-fitting to reduce overfitting bias. This approach combines machine learning-based outcome modeling with propensity score estimation to achieve doubly robust and debiased treatment effect estimates. The specific machine learning algorithms (e.g., random forest, gradient boosting, or Lasso) will be selected to match those used in the original study.
4. Performance Evaluation:
For each method and each observational experiment, we will compute:
Point estimates and 95% confidence intervals of the treatment effect
Bias relative to the true treatment effect (estimated using the original internal control arm)
Empirical variance across experiments
Mean squared error (MSE)
95% coverage probability
Type I error rate and statistical power
5. Hypothesis Testing:
For each method, we will test the null hypothesis of no treatment effect using Wald-type tests based on the estimated treatment effects and their standard errors.
6. Sensitivity Analyses:
We will conduct sensitivity analyses to assess the robustness of our findings to different specifications of the propensity score model, different machine learning algorithms, and different matching calipers.
All analysis code will be made publicly available to ensure full reproducibility.
Narrative Summary:
External control arms are comparison groups built from external data when a clinical trial lacks a traditional control group. A 2022 study by Loiseau and colleagues compared three methods--propensity score matching, G‑computation, and doubly debiased machine learning--for building such arms, using five Type 2 diabetes trials from the YODA Project. Our project will independently repeat that study to check if its findings are reliable. Reproducibility is essential for trustworthy science. If we confirm the results, researchers and regulators will have more confidence when using these methods in future drug development and approval decisions. All our analysis code will be made publicly available so others can verify and build on our work.
Project Timeline:
Project Timeline:
The following timeline is proposed for this replication study:
Milestone Estimated Date
Data request submission August 2026
Data access approval (estimated) October 2026
Data acquisition and preparation October 2026
Descriptive analysis and covariate balance assessment November 2026
Implementation of propensity score methods November 2026
Implementation of G-computation December 2026
Implementation of DDML December 2026
Performance evaluation and results compilation January 2027
Sensitivity analyses January 2027
Manuscript drafting February 2027
Internal review and revisions March 2027
First manuscript submission April 2027
Results reported to YODA Project April 2027
Total estimated duration: Approximately 6 months from data access to manuscript submission.
Please note that the YODA Project Data Use Agreement allows for 12 months of data access, with the possibility of extension if needed.
Dissemination Plan:
Dissemination Plan:
Target Products:
1. Primary Manuscript: A replication results manuscript will be submitted to BMC Medical Research Methodology, Statistics in Medicine, Journal of Clinical Epidemiology, or Clinical Trials. We will compare our findings with the original results, discuss discrepancies, and provide insights into the reproducibility of ECA methodology research.
2. Replication Code: All analysis code will be publicly released on GitHub with clear documentation for reproduction.
3. Preprint: A preprint will be deposited on medRxiv or arXiv prior to journal submission.
Target Audiences:
- Academic researchers: methodologists, biostatisticians, and clinical trialists working on ECAs and causal inference.
- Regulatory agencies: FDA, EMA, and other bodies using external control data for regulatory decisions.
- Pharmaceutical industry: drug development teams considering ECAs in trial design.
- YODA Project community: researchers using YODA data who may benefit from replication methods.
YODA Project Reporting:
Results will be reported to YODA upon manuscript submission, including a summary of findings and a link to the published manuscript when available.
Bibliography:
参考:
-
Loiseau N, Trichelair P, He M, et al. External control arm analysis: an evaluation of propensity score approaches, G-computation, and doubly debiased machine learning. BMC Med Res Methodol. 2022;22(1):335. doi:10.1186/s12874-022-01799-z. PMID: 36577946; PMCID: PMC9795588.
-
Rosenbaum PR, Rubin DB. The central role of the propensity score in observational studies for causal effects. Biometrika. 1983;70(1):41-55. doi:10.1093/biomet/70.1.41.
-
Robins JM, Hernán MA, Brumback B. Marginal structural models and causal inference in epidemiology. Epidemiology. 2000;11(5):550-560. doi:10.1097/00001648-200009000-00011.
-
Chernozhukov V, Chetverikov D, Demirer M, et al. Double/debiased machine learning for treatment and structural parameters. Econom J. 2018;21(1):C1-C68. doi:10.1111/ectj.12097.
-
Snowden JM, Rose S, Mortimer KM. Implementation of G-computation on a simulated data set: demonstration of a causal inference technique. Am J Epidemiol. 2011;173(7):731-738. doi:10.1093/aje/kwq472.