Pure-MoonBit port of doubleml-for-py: Double / Debiased Machine Learning for partial linear, IRM, IV, DID, SSM, APO(S), PQ, QTE, LPQ, CVaR, RDD, BLP, and policy-tree models
riantr/moonbit_doubleML does not have a README file
pub suberror BootstrapMethodError {
UnknownMethod(String)
}pub suberror BracketSignError {
UpperSignFailed
}pub suberror CalibrationFittingError {
IncompleteCVPartition
}pub suberror ClusterDataError {
MissingUnit(Int)
}pub suberror DIDDataError {
NonBinaryTreatment(Int)
}pub suberror EmptyArrayErrorpub suberror InvalidCalibrationError {
UnknownMethod(String)
}pub suberror PSConfigError {
InconsistentCVCalibration
}pub suberror VarEstClusterError {
JTooSmall(Double, Double, Int)
}type DidCsDatatype DidMultiDatapub struct DoubleMLAPO {
data : DoubleMLData
treatment_level : Double
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
ml_g : LearnerDispatch
ml_m : LearnerDispatch
g_hat : Array[Double]
m_hat : Array[Double]
coef : Double
se : Double
fitted : Bool
} derive(Debug)fn DoubleMLAPO::fit(self : DoubleMLAPO, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLAPOfn DoubleMLAPO::new(data : DoubleMLData, treatment_level? : Double, n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLAPOpub struct DoubleMLAPOS {
data : DoubleMLData
treatment_levels : Array[Double]
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
ml_g : LearnerDispatch
ml_m : LearnerDispatch
coefs : Array[Double]
ses : Array[Double]
fitted : Bool
} derive(Debug)fn DoubleMLAPOS::causal_contrast(self : DoubleMLAPOS, reference_levels : Array[Double]) -> Array[Array[Double]]fn DoubleMLAPOS::fit(self : DoubleMLAPOS, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLAPOSfn DoubleMLAPOS::new(data : DoubleMLData, treatment_levels : Array[Double], n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLAPOSfn DoubleMLBLP::confint_joint(self : DoubleMLBLP, contrast : Matrix, level? : Double) -> Array[(Double, Double)] Y = expit(D * theta_0 + r_0(X)), Y in {0, 1}pub struct DoubleMLCVAR {
data : DoubleMLData
treatment : Double
quantile : Double
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
normalize_ipw : Bool
ml_g : LearnerDispatch
ml_m : LearnerDispatch
g_hat : Array[Double]
m_hat : Array[Double]
coef : Double
se : Double
fitted : Bool
} derive(Debug)fn DoubleMLCVAR::fit(self : DoubleMLCVAR, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLCVARfn DoubleMLCVAR::new(data : DoubleMLData, treatment? : Double, quantile? : Double, n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, normalize_ipw? : Bool, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLCVARpub struct DoubleMLDID {
data : DoubleMLDIDData
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
ps_processor : PSProcessor
score : String
in_sample_normalization : Bool
ml_g : LearnerDispatch
ml_m : LearnerDispatch
strata : Array[Int]
g0_hat : Array[Double]
g1_hat : Array[Double]
m_hat : Array[Double]
coef : Double
se : Double
psi_a : Array[Double]
psi_b : Array[Double]
fitted : Bool
} derive(Debug) Y_post = g_0(0, X) + D * theta + U_post, E[U | D, X] = 0
Y_pre = g_0(0, X) + U_pre, E[U_pre | X] = 0 dY = Y_post - Y_pre = g_1(1, X) - g_0(0, X) + D * theta + (U_post - U_pre) g0(X) = E[Y | D = 0, X] (trained on D = 0)
g1(X) = E[Y | D = 1, X] (trained on D = 1)
m(X) = P(D = 1 | X) (trained on all obs; observational
only, clipped to [eps, 1 - eps]) resid_d0 = Y - g0
p_hat = mean(D)
weight_psi_a = D / p_hat
weight_resid = (D - m) / (p_hat * (1 - m))
psi_b = (D - m)/(p_hat (1 - m)) * (Y - g0) [the g1 term cancels in ATT]
psi_a = -D / p_hat
psi(theta) = theta * psi_a + psi_b theta_hat = -mean(psi_b) / mean(psi_a)
J = mean(psi_a)
gamma = mean(psi(theta_hat)^2)
sigma2 = gamma / (J^2 * n)
se = sqrt(sigma2).fn DoubleMLDID::fit(self : DoubleMLDID, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLDIDfn DoubleMLDID::new(data : DoubleMLDIDData, n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, ps_processor? : PSProcessor, score? : String, in_sample_normalization? : Bool, strata? : Array[Int], ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLDIDpub struct DoubleMLDIDBinary {
data : DoubleMLDIDBinaryData
g_value : Int
t_value_pre : Int
t_value_eval : Int
control_group : String
anticipation_periods : Int
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
ps_processor : PSProcessor
score : String
in_sample_normalization : Bool
eval_idx : Array[Int]
inner : DoubleMLDID
fitted : Bool
} derive(Debug)fn DoubleMLDIDBinary::fit(self : DoubleMLDIDBinary, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLDIDBinaryfn DoubleMLDIDBinary::new(data : DoubleMLDIDBinaryData, g_value : Int, t_value_pre : Int, t_value_eval : Int, control_group? : String, anticipation_periods? : Int, n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, ps_processor? : PSProcessor, score? : String, in_sample_normalization? : Bool) -> DoubleMLDIDBinaryfn DoubleMLDIDBinaryData::new(x : Matrix, y : Array[Double], d : Array[Double], t : Array[Int], g : Array[Int], id : Array[Int]) -> DoubleMLDIDBinaryDatapub struct DoubleMLDIDCS {
data : DoubleMLDIDCSData
control_group : String
anticipation_periods : Int
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
ps_processor : PSProcessor
in_sample_normalization : Bool
coef_matrix : Array[Double]
se_matrix : Array[Double]
psi_matrix : Array[Double]
n_groups : Int
n_periods : Int
fitted : Bool
} derive(Debug)fn DoubleMLDIDCS::fit(self : DoubleMLDIDCS, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLDIDCSfn DoubleMLDIDCS::new(data : DoubleMLDIDCSData, control_group? : String, anticipation_periods? : Int, n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, ps_processor? : PSProcessor, in_sample_normalization? : Bool) -> DoubleMLDIDCSpub struct DoubleMLDIDCSBinary {
data : DoubleMLDIDCSData
g_value : Int
t_value_pre : Int
t_value_eval : Int
control_group : String
anticipation_periods : Int
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
ps_processor : PSProcessor
score : String
in_sample_normalization : Bool
coef : Double
se : Double
psi_a : Array[Double]
psi_b : Array[Double]
g_d0_t0 : Array[Double]
g_d0_t1 : Array[Double]
g_d1_t0 : Array[Double]
g_d1_t1 : Array[Double]
m_hat : Array[Double]
n_obs_subset : Int
n_g_subset : Int
n_c_subset : Int
fitted : Bool
} derive(Debug)fn DoubleMLDIDCSBinary::fit(self : DoubleMLDIDCSBinary, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLDIDCSBinaryfn DoubleMLDIDCSBinary::new(data : DoubleMLDIDCSData, g_value : Int, t_value_pre : Int, t_value_eval : Int, control_group? : String, anticipation_periods? : Int, n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, ps_processor? : PSProcessor, score? : String, in_sample_normalization? : Bool) -> DoubleMLDIDCSBinaryfn DoubleMLDIDCSData::new(x : Matrix, y : Array[Double], d : Array[Double], t : Array[Int], id : Array[Int], g : Array[Int]) -> DoubleMLDIDCSDatapub struct DoubleMLDIDCrossSection {
data : DoubleMLDIDCrossSectionData
n_folds : Int
n_rep : Int
seed : Int
score : String
in_sample_normalization : Bool
propensity_clip : Double
ps_processor : PSProcessor
ml_g : LearnerDispatch
ml_m : LearnerDispatch
coef : Double
se : Double
psi_a : Array[Double]
psi_b : Array[Double]
g_d0_t0 : Array[Double]
g_d0_t1 : Array[Double]
g_d1_t0 : Array[Double]
g_d1_t1 : Array[Double]
m_hat : Array[Double]
fitted : Bool
boot_t_stat : Array[Double]
boot_method : String
n_rep_boot : Int
boot_seed : Int
} derive(Debug)fn DoubleMLDIDCrossSection::bootstrap(self : DoubleMLDIDCrossSection, method_name? : String, n_rep_boot? : Int, seed? : Int) -> DoubleMLDIDCrossSectionfn DoubleMLDIDCrossSection::confint(self : DoubleMLDIDCrossSection, joint? : Bool, level? : Double) -> (Double, Double)fn DoubleMLDIDCrossSection::fit(self : DoubleMLDIDCrossSection, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLDIDCrossSectionfn DoubleMLDIDCrossSection::new(data : DoubleMLDIDCrossSectionData, n_folds? : Int, n_rep? : Int, seed? : Int, score? : String, in_sample_normalization? : Bool, propensity_clip? : Double, ps_processor? : PSProcessor, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLDIDCrossSectionfn DoubleMLDIDCrossSectionData::new(x : Matrix, y : Array[Double], d : Array[Double], t : Array[Int], name? : String) -> DoubleMLDIDCrossSectionDatafn DoubleMLDIDData::new(x : Matrix, y : Array[Double], d : Array[Double]) -> DoubleMLDIDData raise DIDDataErrorpub struct DoubleMLDIDMulti {
data : DoubleMLDIDCSData
gt_combinations : Array[(Int, Int, Int)]
control_group : String
anticipation_periods : Int
n_folds : Int
n_rep : Int
seed : Int
ps_processor : PSProcessor
in_sample_normalization : Bool
ml_g : LearnerDispatch
ml_m : LearnerDispatch
group_sizes : Array[Int]
inner : DoubleMLDIDCS
boot_t_stat : Array[Double]
boot_method : String
n_rep_boot : Int
boot_seed : Int
fitted : Bool
} derive(Debug)fn DoubleMLDIDMulti::bootstrap(self : DoubleMLDIDMulti, method_name? : String, n_rep_boot? : Int, seed? : Int) -> DoubleMLDIDMultifn DoubleMLDIDMulti::confint(self : DoubleMLDIDMulti, joint? : Bool, level? : Double) -> Array[(Double, Double)]fn DoubleMLDIDMulti::fit(self : DoubleMLDIDMulti, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLDIDMultifn DoubleMLDIDMulti::new(data : DoubleMLDIDCSData, gt_combinations? : Array[(Int, Int, Int)], gt_combinations_keyword? : String, control_group? : String, anticipation_periods? : Int, n_folds? : Int, n_rep? : Int, seed? : Int, ps_processor? : PSProcessor, in_sample_normalization? : Bool, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLDIDMultifn DoubleMLData::new(x : Matrix, y : Array[Double], d : Array[Double], cluster_vars? : Array[Int], z? : Array[Double]) -> DoubleMLDatapub struct DoubleMLIIVM {
data : DoubleMLIIVMData
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
ml_g : LearnerDispatch
ml_m : LearnerDispatch
ml_r : LearnerDispatch
g0_hat : Array[Double]
g1_hat : Array[Double]
m_hat : Array[Double]
r0_hat : Array[Double]
r1_hat : Array[Double]
coef : Double
se : Double
fitted : Bool
} derive(Debug) Y = theta * D + g_0(D, X) + U, E[U | D, X] = 0
D = m_0(X, Z) + V, E[V | X, Z] = 0 g0(X) = E[Y | Z = 0, X] (trained only on Z = 0)
g1(X) = E[Y | Z = 1, X] (trained only on Z = 1)
m(X) = E[Z | X] (trained on all obs, then
clipped to [eps, 1 - eps])
r0(X) = E[D | Z = 0, X] (trained only on Z = 0)
r1(X) = E[D | Z = 1, X] (trained only on Z = 1) u_hat0 = Y - g0, u_hat1 = Y - g1
w_hat0 = D - r0, w_hat1 = D - r1 psi_b = (g1 - g0) + Z u_hat1 / m - (1 - Z) u_hat0 / (1 - m)
psi_a = -(r1 - r0) - Z w_hat1 / m + (1 - Z) w_hat0 / (1 - m)
psi(theta) = theta * psi_a + psi_b theta_hat = -mean(psi_b) / mean(psi_a)
J = mean(psi_a)
gamma = mean(psi(theta_hat)^2)
sigma2 = gamma / (J^2 * n)
se = sqrt(sigma2).fn DoubleMLIIVM::fit(self : DoubleMLIIVM, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch, ml_r? : LearnerDispatch, max_attempts? : Int) -> DoubleMLIIVMfn DoubleMLIIVM::new(data : DoubleMLIIVMData, n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch, ml_r? : LearnerDispatch) -> DoubleMLIIVMfn DoubleMLIIVMData::new(x : Matrix, y : Array[Double], d : Array[Double], z : Array[Double], cluster_vars? : Array[Int]) -> DoubleMLIIVMDatapub struct DoubleMLIRM {
data : DoubleMLData
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
ml_g : LearnerDispatch
ml_m : LearnerDispatch
g0_hat : Array[Double]
g1_hat : Array[Double]
m_hat : Array[Double]
m_raw : Array[Double]
coef : Double
se : Double
fitted : Bool
} derive(Debug) Y = g_0(D, X) + U, E[U | D, X] = 0
D = m_0(X) + V, E[V | X] = 0 g0(X) = E[Y | D=0, X]
g1(X) = E[Y | D=1, X]
m(X) = P(D=1 | X) (the propensity score)
u0 = Y - g0(X)
u1 = Y - g1(X)
psi_b = (g1 - g0) + (D u1 / m - (1 - D) u0 / (1 - m))
psi_a = -1
psi(theta) = theta * psi_a + psi_b theta_hat = -mean(psi_b) / mean(psi_a) = mean(psi_b) J = mean(psi_a) = -1
gamma = mean(psi(theta_hat)^2)
sigma2 = gamma / (J^2 * n)
se = sqrt(sigma2)fn DoubleMLIRM::fit(self : DoubleMLIRM, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch, max_attempts? : Int) -> DoubleMLIRMfn DoubleMLIRM::new(data : DoubleMLData, n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLIRMpub struct DoubleMLLPLR {
data : DoubleMLBinaryData
score : String
n_folds : Int
n_folds_inner : Int
n_rep : Int
seed : Int
ml_g : LearnerDispatch
ml_m : LearnerDispatch
coef_ : Double
se_ : Double
r_hat : Array[Double]
m_hat : Array[Double]
a_hat : Array[Double]
fitted : Bool
} derive(Debug)fn DoubleMLLPLR::fit(self : DoubleMLLPLR, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLLPLRfn DoubleMLLPLR::new(data : DoubleMLBinaryData, score? : String, n_folds? : Int, n_folds_inner? : Int, n_rep? : Int, seed? : Int, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLLPLRpub struct DoubleMLLPQ {
data : DoubleMLLPQData
treatment : Double
quantile : Double
n_folds : Int
seed : Int
propensity_clip : Double
coef : Double
se : Double
fitted : Bool
predictions_g0 : Array[Double]
predictions_g1 : Array[Double]
predictions_m : Array[Double]
} derive(Debug)fn DoubleMLLPQ::new(data : DoubleMLLPQData, treatment? : Double, quantile? : Double, n_folds? : Int, seed? : Int, propensity_clip? : Double) -> DoubleMLLPQfn DoubleMLLPQData::new(x : Matrix, y : Array[Double], d : Array[Double], z : Array[Double]) -> DoubleMLLPQDatapub struct DoubleMLPLIV {
data : DoubleMLPLIVData
n_folds : Int
n_rep : Int
seed : Int
learner : LearnerDispatch
l_hat : Array[Double]
r_hat : Array[Double]
m_hat : Array[Double]
coef : Double
se : Double
fitted : Bool
} derive(Debug) Y = D * theta_0 + g_0(X) + zeta, E[zeta | D, X] = 0
D = m_0(X) + V, E[V | X] = 0
Z = ell_0(X) + xi, E[xi | X] = 0,
Cov(Z, V) != 0 (relevance) l_hat = E_hat[Y | X]
r_hat = E_hat[D | X]
m_hat = E_hat[Z | X]
u_hat = Y - l_hat
w_hat = D - r_hat
v_hat = Z - m_hat
psi_a = -w_hat * v_hat
psi_b = v_hat * u_hat
psi(theta) = theta * psi_a + psi_b theta_hat = -mean(psi_b) / mean(psi_a)
= mean(v_hat * u_hat) / mean(w_hat * v_hat)
J = mean(psi_a)
gamma = mean(psi(theta_hat)^2)
sigma2 = gamma / (J^2 * n)
se = sqrt(sigma2).fn DoubleMLPLIV::fit(self : DoubleMLPLIV, learner? : LearnerDispatch, max_attempts? : Int) -> DoubleMLPLIVfn DoubleMLPLIV::new(data : DoubleMLPLIVData, n_folds? : Int, n_rep? : Int, seed? : Int, learner? : LearnerDispatch) -> DoubleMLPLIVfn DoubleMLPLIVData::new(x : Matrix, y : Array[Double], d : Array[Double], z : Array[Double], cluster_vars? : Array[Int]) -> DoubleMLPLIVDatapub struct DoubleMLPLPR {
panel : DoubleMLPanelData
approach : String
score : String
n_folds : Int
n_rep : Int
seed : Int
learner : LearnerDispatch
coef_ : Double
se_ : Double
l_hat : Array[Double]
m_hat : Array[Double]
fitted : Bool
} derive(Debug)fn DoubleMLPLPR::fit(self : DoubleMLPLPR, learner? : LearnerDispatch, max_attempts? : Int) -> DoubleMLPLPRfn DoubleMLPLPR::new(panel : DoubleMLPanelData, approach? : String, score? : String, n_folds? : Int, n_rep? : Int, seed? : Int, learner? : LearnerDispatch) -> DoubleMLPLPRpub struct DoubleMLPLR {
data : DoubleMLData
learner_l : LearnerDispatch
learner_m : LearnerDispatch
n_folds : Int
n_rep : Int
seed : Int
l_hat : Array[Double]
m_hat : Array[Double]
coef : Double
se : Double
fitted : Bool
tune_result : TuneResult?
} derive(Debug) Y = D * theta_0 + g_0(X) + zeta, E[zeta | D, X] = 0
D = m_0(X) + V, E[V | X] = 0 psi_a(theta) = -(D - m_hat)^2,
psi_b(theta) = (D - m_hat) * (Y - l_hat),
psi(theta) = theta * psi_a + psi_b theta_hat = -mean(psi_b) / mean(psi_a)
= mean((D - m_hat)(Y - l_hat)) / mean((D - m_hat)^2). J = mean(psi_a) # expected derivative of psi w.r.t. theta
gamma = mean(psi(theta_hat)^2)
sigma2 = gamma / (J^2 * n)
se = sqrt(sigma2).fn DoubleMLPLR::fit(self : DoubleMLPLR, learner_l? : LearnerDispatch, learner_m? : LearnerDispatch, max_attempts? : Int, score? : String, tune_result? : TuneResult?) -> DoubleMLPLRfn DoubleMLPLR::new(data : DoubleMLData, learner_l? : LearnerDispatch, learner_m? : LearnerDispatch, n_folds? : Int, n_rep? : Int, seed? : Int) -> DoubleMLPLRfn DoubleMLPLR::tune(self : DoubleMLPLR, param_set~ : Array[TuneParam], scoring_method? : String, n_folds_tune? : Int, seed? : Int, score? : String) -> DoubleMLPLRpub struct DoubleMLPQ {
data : DoubleMLData
treatment : Double
quantile : Double
n_folds : Int
seed : Int
propensity_clip : Double
ml_l : LearnerDispatch
ml_m : LearnerDispatch
coef : Double
se : Double
fitted : Bool
} derive(Debug)fn DoubleMLPQ::fit(self : DoubleMLPQ, ml_l? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLPQfn DoubleMLPQ::new(data : DoubleMLData, treatment? : Double, quantile? : Double, n_folds? : Int, seed? : Int, propensity_clip? : Double, ml_l? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLPQ Y_it = D_it * theta_0 + g_0(X_it) + alpha_i + zeta_itfn DoubleMLPanelData::new(x : Matrix, y : Array[Double], d : Array[Double], t : Array[Int], id : Array[Int]) -> DoubleMLPanelDatapub struct DoubleMLPolicyTree {
features : Matrix
orth_signal : Array[Double]
depth : Int
root : PolicyTreeNode
split_feature : Int
split_value : Double
left_treatment : Int
right_treatment : Int
fitted : Bool
} derive(Debug)fn DoubleMLPolicyTree::new(features : Matrix, orth_signal : Array[Double], depth? : Int) -> DoubleMLPolicyTreepub struct DoubleMLQTE {
data : DoubleMLData
quantiles : Array[Double]
n_folds : Int
seed : Int
propensity_clip : Double
ml_l : LearnerDispatch
ml_m : LearnerDispatch
coefs : Array[Double]
ses : Array[Double]
} derive(Debug)fn DoubleMLQTE::fit(self : DoubleMLQTE, ml_l? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLQTEfn DoubleMLQTE::new(data : DoubleMLData, quantiles? : Array[Double], n_folds? : Int, seed? : Int, propensity_clip? : Double, ml_l? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLQTEpub struct DoubleMLRDD {
data : DoubleMLRDDData
cutoff : Double
bandwidth : Double
fuzzy : Bool
cov_type : String
ml_g : LearnerDispatch
coef : Double
se : Double
n_local : Int
fitted : Bool
} derive(Debug)fn DoubleMLRDD::new(data : DoubleMLRDDData, cutoff? : Double, bandwidth? : Double, fuzzy? : Bool, cov_type? : String, ml_g? : LearnerDispatch) -> DoubleMLRDDfn DoubleMLRDDData::new(x : Matrix, y : Array[Double], d : Array[Double], score : Array[Double]) -> DoubleMLRDDDatapub struct DoubleMLSSM {
data : DoubleMLSSMData
n_folds : Int
n_rep : Int
seed : Int
propensity_clip : Double
ml_g : LearnerDispatch
ml_m : LearnerDispatch
pi_hat : Array[Double]
m_hat : Array[Double]
g_d1 : Array[Double]
g_d0 : Array[Double]
coef : Double
se : Double
fitted : Bool
} derive(Debug) Y = theta * D + X @ beta * D + U, E[U | D, X] = 0
S = 1{D + gamma Z + X @ beta + V > 0}, E[V | X, D] = 0 g_d1(X) = E[Y | D = 1, S = 1, X] (trained on D=1 ∧ S=1,
features = X only —
Bug #1 fix: previously
`pi_hat` was appended as
an extra feature, which
leaked the test-fold pi
into the training fold)
g_d0(X) = E[Y | D = 0, S = 1, X] (trained on D=0 ∧ S=1,
features = X only)
m(X) = P(D = 1 | X) (trained on all obs,
clipped to [eps, 1 - eps])
pi(X, D) = P(S = 1 | D, X) (trained on (X, D),
clipped to [eps, 1 - eps]) psi_a = -1
psi_b1 = (D == 1) * S * (Y - g_d1) / (m * pi) + g_d1
psi_b0 = (D == 0) * S * (Y - g_d0) / ((1 - m) * pi) + g_d0
psi_b = psi_b1 - psi_b0 theta_hat = -mean(psi_b) / mean(psi_a) = mean(psi_b)
J = mean(psi_a) = -1
gamma = mean(psi(theta_hat)^2)
sigma2 = gamma / (J^2 * n)
se = sqrt(sigma2).fn DoubleMLSSM::fit(self : DoubleMLSSM, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch, ml_pi? : LearnerDispatch) -> DoubleMLSSMfn DoubleMLSSM::new(data : DoubleMLSSMData, n_folds? : Int, n_rep? : Int, seed? : Int, propensity_clip? : Double, ml_g? : LearnerDispatch, ml_m? : LearnerDispatch) -> DoubleMLSSMfn DoubleMLSSMData::new(x : Matrix, y : Array[Double], d : Array[Double], s : Array[Double]) -> DoubleMLSSMDatafn GainStatsSource::from_blp_cv_repeated(blp : DoubleMLBLP, n_folds? : Int, n_repeats? : Int, seed? : Int) -> GainStatsSourcefn GainStatsSource::new(var_y_residuals : Array[Double], nu2 : Array[Double], all_coef : Array[Double], n_rep : Int, var_y : Double) -> GainStatsSourcetype IivmDatatype IrmConfoundedDatatype IrmDatatype IrmDiscreteDatatype IrmHeterogeneousDatapub enum LearnerDispatch {
LinearRegression(LinearRegression)
Constant(ConstantLearner)
Noop(NoopLearner)
RandomForest(RFLearner)
GradientBoosting(GBLearner)
} derive(Debug)fn LinearRegression::fit(self : LinearRegression, x : Matrix, y : Array[Double]) -> LinearRegressionfn LinearRegression::fit_weighted(self : LinearRegression, x : Matrix, y : Array[Double], w : Array[Double]) -> LinearRegressionfn LinearRegression::sandwich_se(self : LinearRegression, x : Matrix, y : Array[Double]) -> Array[Double]fn LinearRegression::sandwich_se_weighted(self : LinearRegression, x : Matrix, y : Array[Double], w : Array[Double]) -> Array[Double] cov(beta_hat) = (X'WX)^{-1} (X' diag(w · e^2) X) (X'WX)^{-1}fn LogisticRegression::fit(self : LogisticRegression, x : Matrix, y : Array[Double], max_iter? : Int, tol? : Double, ridge? : Double) -> LogisticRegressionfn LogisticRegression::predict_class(self : LogisticRegression, x : Matrix, threshold? : Double) -> Array[Double]type LplrDatafn PSProcessor::adjust_ps(self : PSProcessor, ps : Array[Double], treatment : Array[Double], cv? : Array[(Array[Int], Array[Int])]?) -> Array[Double]pub(all) struct PSProcessorConfig {
clipping_threshold : Double
extreme_threshold : Double
calibration_method : String
cv_calibration : Bool
} derive(Debug)fn PSProcessorConfig::new(clipping_threshold? : Double, extreme_threshold? : Double, calibration_method? : String, cv_calibration? : Bool) -> PSProcessorConfig raise PSConfigErrortype PlivClusterDatatype PlivDatatype PlprDatatype PlrCcddhnr2018type PlrConfoundedDatapub enum PolicyTreeNode {
Leaf(Int)
Split(Int, Double, PolicyTreeNode, PolicyTreeNode)
} derive(Debug)type RddSimpleDatatype SsmData theta_hat = median(theta_1, ..., theta_R)
ub_r = theta_r + 1.96 * se_r for each rep r
ub_hat = median(ub_1, ..., ub_R)
se_hat = (ub_hat - theta_hat) / 1.96 theta_hat = median([c]) = c
ub_hat = median([c + 1.96 * s]) = c + 1.96 * s
se_hat = (c + 1.96 * s - c) / 1.96 = s#callsite(autofill(loc))
fn check(condition : Bool, loc~ : SourceLoc) -> Unit raise PreconditionErrorfn draw_bootstrap_weights(method_name : String, n_rep_boot : Int, n_obs : Int, seed : Int) -> Array[Double] raise BootstrapMethodError theta = - sum_k w_k * sum_{i in k} psi_b
/ sum_k w_k * sum_{i in k} psi_a, w_k = 1/|I_k|.fn expit(x : Double) -> Doublefn fit_predict_one_dispatch(learner : LearnerDispatch, x_train : Matrix, y_train : Array[Double], x_test : Matrix) -> Array[Double]fn g_cross_fit_calls() -> Int f_hat(x) = (1 / (n * h * sqrt(2*pi))) * sum_i exp(-(x - y_i)^2 / (2 * h^2)) f_hat(x) = (1 / (h * sqrt(2*pi))) * sum_i w[i] * exp(-(x - y[i])^2 / (2*h^2)) d/dtheta mean(psi_ipw) = (1/n) * sum_i w_i * delta(y_i - theta)
≈ (1/n) * f_hat_weighted(theta) s = 0.0
for x in arr: s = s + x sum = 0.0
c = 0.0 // compensation for low-order bits lost in the next add
for x in arr:
y = x - c // align x with the running sum's precision
t = sum + y // primary add
c = (t - sum) - y // low-order bits of `t` that did not fit
sum = t
return sumfn logit(p : Double, eps? : Double) -> Doublefn lpq_score_ipw(data : DoubleMLLPQData, treated : Array[Double], m : Array[Double], comp : Double, theta : Double, q : Double, sign : Double) -> Array[Double]fn make_confounded_plr_data(n_obs : Int, dim_x : Int, theta : Double, seed : Int) -> PlrConfoundedDatafn make_irm_confounded_data(n_obs : Int, dim_x : Int, theta : Double, seed : Int) -> IrmConfoundedDatafn make_irm_discrete_treatments(n_obs : Int, dim_x : Int, theta : Double, seed : Int) -> IrmDiscreteDatafn make_pliv_multiway_cluster(n_obs : Int, dim_x : Int, theta : Double, seed : Int) -> PlivClusterDatafn norm_cdf(x : Double) -> Doublefn norm_ppf(p : Double) -> Doublefn norm_sf(x : Double) -> Double#callsite(autofill(loc))
fn require(condition : Bool, loc~ : SourceLoc) -> Unit raise PreconditionErrorfn reset_g_cross_fit_count() -> Unitfn solve_pq(data : DoubleMLData, treatment : Double, q : Double, n_folds : Int, seed : Int, clip : Double) -> (Double, Array[Double], Double) raise BracketSignError theta_hat = -mean(psi_b) / mean(psi_a)
J = mean(psi_a)
gamma = mean(psi(theta_hat)^2) where psi(theta) = theta * psi_a + psi_b
sigma2 = gamma / (J^2 * n)
se = sqrt(sigma2) gamma += S_g^2 / |I_k|, S_g = sum of psi over unit g,
J += (sum of psi_deriv over the fold's rows) / |I_k|,
sigma2 = (gamma / npc) / (N_units * (J / npc)^2).Install
Download zipPure-MoonBit port of doubleml-for-py: Double / Debiased Machine Learning for partial linear, IRM, IV, DID, SSM, APO(S), PQ, QTE, LPQ, CVaR, RDD, BLP, and policy-tree models