In [1]:
import os
import numpy as np
import polars as pl
from survey_kit.imputation.srmi import SRMI
from survey_kit import logger, config
In [2]:
# Two "does this look right" diagnostics, both answering a different
# question from convergence: not "did the chain settle down" but "do the
# imputed values themselves look plausible"
n_rows = 3_000
rng = np.random.default_rng(20260913)
x1 = rng.normal(size=n_rows)
y = 2.0 * x1 + rng.normal(scale=1.0, size=n_rows)
df = pl.DataFrame(dict(row_id=range(n_rows), x1=x1, y=y))
missing = rng.random(n_rows) < 0.25
df = df.with_columns(
pl.when(pl.Series(missing)).then(None).otherwise(pl.col("y")).alias("y")
)
srmi = SRMI.simple_model(
df=df,
index="row_id",
replication=SRMI.Replication(n_implicates=3, n_iterations=3),
parallel=SRMI.Parallel(enabled=False),
bootstrap=SRMI.Bootstrap(enabled=True),
storage=SRMI.Storage(
path_model=f"{config.path_temp_files}/tutorial_diagnostics_quality_propensity",
force_start=True,
),
)
srmi.run()
path_docs_diagnostics = os.path.join(
config.code_root, "..", "..", "docs", "tutorials", "srmi", "diagnostics"
)
os.makedirs(path_docs_diagnostics, exist_ok=True)
auto_detect: 'y' -> class=continuous, modeltype=LightGBM, predictors=['x1']
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi
Variable selection before SRMI run, if necessary
y: Method.No
Hyperparameter tuning before SRMI run, if necessary
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/1.srmi.implicate
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/2.srmi.implicate
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/3.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 3945136551}
Iterations: 100
Model: y=f(x1, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.02717 ┆ 0 ┆ 0.008136 │ └─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 2269.0 ┆ 2269.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.060955 ┆ 0.080526 ┆ 0.027821 │ │ std ┆ 2.192342 ┆ 1.947098 ┆ 2.017499 │ │ min ┆ -7.371072 ┆ -5.171975 ┆ -5.171975 │ │ 25% ┆ -1.347467 ┆ -1.115578 ┆ -1.218412 │ │ 50% ┆ 0.033669 ┆ 0.049477 ┆ -0.000123 │ │ 75% ┆ 1.533057 ┆ 1.34598 ┆ 1.363979 │ │ max ┆ 6.852525 ┆ 5.211096 ┆ 5.211096 │ └────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 2269.0 ┆ 2269.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.060955 ┆ 0.080526 ┆ 0.027821 │ │ std ┆ 2.192342 ┆ 1.947098 ┆ 2.017499 │ │ min ┆ -7.371072 ┆ -5.171975 ┆ -5.171975 │ │ 25% ┆ -1.347467 ┆ -1.115578 ┆ -1.218412 │ │ 50% ┆ 0.033669 ┆ 0.049477 ┆ -0.000123 │ │ 75% ┆ 1.533057 ┆ 1.34598 ┆ 1.363979 │ │ max ┆ 6.852525 ┆ 5.211096 ┆ 5.211096 │ └────────────┴───────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['y']
Most common matches:
shape: (5, 2) ┌────────┬─────────┐ │ row_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞════════╪═════════╡ │ 685 ┆ 4 │ │ 155 ┆ 3 │ │ 256 ┆ 3 │ │ 658 ┆ 3 │ │ 761 ┆ 3 │ └────────┴─────────┘
Post-imputation statistics for ['y']
Where: None
Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬──────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪══════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ y ┆ ┆ 3000 ┆ 3000 ┆ 0.04806 ┆ 2.214 ┆ 0.04806 ┆ 2.214 ┆ -2.774 ┆ -1.384 ┆ 0.02135 ┆ 1.542 ┆ 2.854 ┆ -7.371 ┆ 6.853 │ │ y ┆ 0 ┆ 2269 ┆ 2269 ┆ 0.06095 ┆ 2.192 ┆ 0.06095 ┆ 2.192 ┆ -2.729 ┆ -1.347 ┆ 0.03367 ┆ 1.533 ┆ 2.867 ┆ -7.371 ┆ 6.853 │ │ y ┆ 1 ┆ 731 ┆ 731 ┆ 0.008031 ┆ 2.28 ┆ 0.008031 ┆ 2.28 ┆ -3.09 ┆ -1.449 ┆ -0.01272 ┆ 1.59 ┆ 2.834 ┆ -6.733 ┆ 6.051 │ └──────────┴─────────┴──────┴──────────────┴──────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/1.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 2447140450}
Iterations: 100
Model: y=f(x1, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.02253 ┆ 0 ┆ 0.008136 │ └─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.048059 ┆ 0.029264 ┆ 0.007197 │ │ std ┆ 2.213719 ┆ 1.980332 ┆ 2.029843 │ │ min ┆ -7.371072 ┆ -5.45787 ┆ -5.45787 │ │ 25% ┆ -1.380858 ┆ -1.082619 ┆ -1.214433 │ │ 50% ┆ 0.026495 ┆ 0.062378 ┆ 0.067595 │ │ 75% ┆ 1.542204 ┆ 1.518401 ┆ 1.537569 │ │ max ┆ 6.852525 ┆ 5.23418 ┆ 5.23418 │ └────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.048059 ┆ 0.029264 ┆ 0.007197 │ │ std ┆ 2.213719 ┆ 1.980332 ┆ 2.029843 │ │ min ┆ -7.371072 ┆ -5.45787 ┆ -5.45787 │ │ 25% ┆ -1.380858 ┆ -1.082619 ┆ -1.214433 │ │ 50% ┆ 0.026495 ┆ 0.062378 ┆ 0.067595 │ │ 75% ┆ 1.542204 ┆ 1.518401 ┆ 1.537569 │ │ max ┆ 6.852525 ┆ 5.23418 ┆ 5.23418 │ └────────────┴───────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['y']
Most common matches:
shape: (5, 2) ┌────────┬─────────┐ │ row_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞════════╪═════════╡ │ 429 ┆ 3 │ │ 479 ┆ 3 │ │ 625 ┆ 3 │ │ 814 ┆ 3 │ │ 824 ┆ 3 │ └────────┴─────────┘
Post-imputation statistics for ['y']
Where: None
Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ y ┆ ┆ 3731 ┆ 3731 ┆ 0.04731 ┆ 2.22 ┆ 0.04731 ┆ 2.22 ┆ -2.789 ┆ -1.367 ┆ 0.0135 ┆ 1.559 ┆ 2.854 ┆ -7.371 ┆ 6.853 │ │ y ┆ 0 ┆ 3000 ┆ 3000 ┆ 0.04806 ┆ 2.214 ┆ 0.04806 ┆ 2.214 ┆ -2.788 ┆ -1.384 ┆ 0.02135 ┆ 1.542 ┆ 2.854 ┆ -7.371 ┆ 6.853 │ │ y ┆ 1 ┆ 731 ┆ 731 ┆ 0.04425 ┆ 2.247 ┆ 0.04425 ┆ 2.247 ┆ -2.881 ┆ -1.347 ┆ -0.0747 ┆ 1.598 ┆ 2.846 ┆ -6.216 ┆ 6.252 │ └──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/1.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 3250830704}
Iterations: 100
Model: y=f(x1, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.02253 ┆ 0 ┆ 0.008136 │ └─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.056883 ┆ 0.059056 ┆ 0.035921 │ │ std ┆ 2.205489 ┆ 1.986718 ┆ 2.041139 │ │ min ┆ -7.371072 ┆ -5.574627 ┆ -5.574627 │ │ 25% ┆ -1.347315 ┆ -1.192955 ┆ -1.246748 │ │ 50% ┆ 0.017392 ┆ 0.012496 ┆ 0.018039 │ │ 75% ┆ 1.538951 ┆ 1.395735 ┆ 1.432209 │ │ max ┆ 6.852525 ┆ 5.239808 ┆ 5.239808 │ └────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.056883 ┆ 0.059056 ┆ 0.035921 │ │ std ┆ 2.205489 ┆ 1.986718 ┆ 2.041139 │ │ min ┆ -7.371072 ┆ -5.574627 ┆ -5.574627 │ │ 25% ┆ -1.347315 ┆ -1.192955 ┆ -1.246748 │ │ 50% ┆ 0.017392 ┆ 0.012496 ┆ 0.018039 │ │ 75% ┆ 1.538951 ┆ 1.395735 ┆ 1.432209 │ │ max ┆ 6.852525 ┆ 5.239808 ┆ 5.239808 │ └────────────┴───────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['y']
Most common matches:
shape: (5, 2) ┌────────┬─────────┐ │ row_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞════════╪═════════╡ │ 1499 ┆ 4 │ │ 2935 ┆ 4 │ │ 2211 ┆ 3 │ │ 2651 ┆ 3 │ │ 2684 ┆ 3 │ └────────┴─────────┘
Post-imputation statistics for ['y']
Where: None
Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ y ┆ ┆ 3731 ┆ 3731 ┆ 0.06234 ┆ 2.225 ┆ 0.06234 ┆ 2.225 ┆ -2.789 ┆ -1.353 ┆ 0.03282 ┆ 1.587 ┆ 2.893 ┆ -7.371 ┆ 6.853 │ │ y ┆ 0 ┆ 3000 ┆ 3000 ┆ 0.05688 ┆ 2.205 ┆ 0.05688 ┆ 2.205 ┆ -2.767 ┆ -1.347 ┆ 0.01585 ┆ 1.539 ┆ 2.854 ┆ -7.371 ┆ 6.853 │ │ y ┆ 1 ┆ 731 ┆ 731 ┆ 0.08474 ┆ 2.303 ┆ 0.08474 ┆ 2.303 ┆ -2.987 ┆ -1.376 ┆ 0.04257 ┆ 1.709 ┆ 2.968 ┆ -6.398 ┆ 6.051 │ └──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
y
Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/1.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 3097978218}
Iterations: 100
Model: y=f(x1, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.02717 ┆ 0 ┆ 0.008136 │ └─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 2269.0 ┆ 2269.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.060955 ┆ 0.0629 ┆ 0.010809 │ │ std ┆ 2.192342 ┆ 1.964057 ┆ 2.026492 │ │ min ┆ -7.371072 ┆ -5.265192 ┆ -5.265192 │ │ 25% ┆ -1.347467 ┆ -1.134032 ┆ -1.322259 │ │ 50% ┆ 0.033669 ┆ 0.007652 ┆ -0.002395 │ │ 75% ┆ 1.533057 ┆ 1.307768 ┆ 1.307768 │ │ max ┆ 6.852525 ┆ 5.118538 ┆ 5.118538 │ └────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 2269.0 ┆ 2269.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.060955 ┆ 0.0629 ┆ 0.010809 │ │ std ┆ 2.192342 ┆ 1.964057 ┆ 2.026492 │ │ min ┆ -7.371072 ┆ -5.265192 ┆ -5.265192 │ │ 25% ┆ -1.347467 ┆ -1.134032 ┆ -1.322259 │ │ 50% ┆ 0.033669 ┆ 0.007652 ┆ -0.002395 │ │ 75% ┆ 1.533057 ┆ 1.307768 ┆ 1.307768 │ │ max ┆ 6.852525 ┆ 5.118538 ┆ 5.118538 │ └────────────┴───────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['y']
Most common matches:
shape: (5, 2) ┌────────┬─────────┐ │ row_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞════════╪═════════╡ │ 1133 ┆ 4 │ │ 1324 ┆ 4 │ │ 303 ┆ 3 │ │ 358 ┆ 3 │ │ 572 ┆ 3 │ └────────┴─────────┘
Post-imputation statistics for ['y']
Where: None
Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ y ┆ ┆ 3000 ┆ 3000 ┆ 0.06472 ┆ 2.204 ┆ 0.06472 ┆ 2.204 ┆ -2.756 ┆ -1.353 ┆ 0.04303 ┆ 1.536 ┆ 2.893 ┆ -7.371 ┆ 6.853 │ │ y ┆ 0 ┆ 2269 ┆ 2269 ┆ 0.06095 ┆ 2.192 ┆ 0.06095 ┆ 2.192 ┆ -2.729 ┆ -1.347 ┆ 0.03367 ┆ 1.533 ┆ 2.867 ┆ -7.371 ┆ 6.853 │ │ y ┆ 1 ┆ 731 ┆ 731 ┆ 0.07642 ┆ 2.242 ┆ 0.07642 ┆ 2.242 ┆ -2.791 ┆ -1.395 ┆ 0.1133 ┆ 1.575 ┆ 2.927 ┆ -7.371 ┆ 6.853 │ └──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/2.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 1780209875}
Iterations: 100
Model: y=f(x1, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.02253 ┆ 0 ┆ 0.008136 │ └─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.064724 ┆ 0.04197 ┆ 0.02746 │ │ std ┆ 2.204137 ┆ 1.992147 ┆ 2.043637 │ │ min ┆ -7.371072 ┆ -5.594726 ┆ -5.594726 │ │ 25% ┆ -1.352496 ┆ -1.171525 ┆ -1.203535 │ │ 50% ┆ 0.044082 ┆ 0.112798 ┆ 0.12708 │ │ 75% ┆ 1.535787 ┆ 1.389405 ┆ 1.446623 │ │ max ┆ 6.852525 ┆ 5.364161 ┆ 5.364161 │ └────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.064724 ┆ 0.04197 ┆ 0.02746 │ │ std ┆ 2.204137 ┆ 1.992147 ┆ 2.043637 │ │ min ┆ -7.371072 ┆ -5.594726 ┆ -5.594726 │ │ 25% ┆ -1.352496 ┆ -1.171525 ┆ -1.203535 │ │ 50% ┆ 0.044082 ┆ 0.112798 ┆ 0.12708 │ │ 75% ┆ 1.535787 ┆ 1.389405 ┆ 1.446623 │ │ max ┆ 6.852525 ┆ 5.364161 ┆ 5.364161 │ └────────────┴───────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['y']
Most common matches:
shape: (5, 2) ┌────────┬─────────┐ │ row_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞════════╪═════════╡ │ 223 ┆ 4 │ │ 628 ┆ 3 │ │ 650 ┆ 3 │ │ 1998 ┆ 3 │ │ 135 ┆ 2 │ └────────┴─────────┘
Post-imputation statistics for ['y']
Where: None
Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬──────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪══════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ y ┆ ┆ 3731 ┆ 3731 ┆ 0.05322 ┆ 2.216 ┆ 0.05322 ┆ 2.216 ┆ -2.762 ┆ -1.384 ┆ 0.04408 ┆ 1.536 ┆ 2.895 ┆ -7.371 ┆ 6.853 │ │ y ┆ 0 ┆ 3000 ┆ 3000 ┆ 0.06472 ┆ 2.204 ┆ 0.06472 ┆ 2.204 ┆ -2.756 ┆ -1.353 ┆ 0.04303 ┆ 1.536 ┆ 2.893 ┆ -7.371 ┆ 6.853 │ │ y ┆ 1 ┆ 731 ┆ 731 ┆ 0.005995 ┆ 2.263 ┆ 0.005995 ┆ 2.263 ┆ -2.809 ┆ -1.505 ┆ 0.0483 ┆ 1.499 ┆ 2.905 ┆ -7.371 ┆ 6.853 │ └──────────┴─────────┴──────┴──────────────┴──────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/2.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 1483604190}
Iterations: 100
Model: y=f(x1, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.02253 ┆ 0 ┆ 0.008136 │ └─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.047563 ┆ 0.031695 ┆ 0.010274 │ │ std ┆ 2.209537 ┆ 1.988333 ┆ 2.046537 │ │ min ┆ -7.371072 ┆ -5.227993 ┆ -5.227993 │ │ 25% ┆ -1.376245 ┆ -1.104187 ┆ -1.163529 │ │ 50% ┆ 0.040889 ┆ 0.101236 ┆ 0.077845 │ │ 75% ┆ 1.519058 ┆ 1.372004 ┆ 1.397818 │ │ max ┆ 6.852525 ┆ 5.375761 ┆ 5.375761 │ └────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.047563 ┆ 0.031695 ┆ 0.010274 │ │ std ┆ 2.209537 ┆ 1.988333 ┆ 2.046537 │ │ min ┆ -7.371072 ┆ -5.227993 ┆ -5.227993 │ │ 25% ┆ -1.376245 ┆ -1.104187 ┆ -1.163529 │ │ 50% ┆ 0.040889 ┆ 0.101236 ┆ 0.077845 │ │ 75% ┆ 1.519058 ┆ 1.372004 ┆ 1.397818 │ │ max ┆ 6.852525 ┆ 5.375761 ┆ 5.375761 │ └────────────┴───────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['y']
Most common matches:
shape: (5, 2) ┌────────┬─────────┐ │ row_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞════════╪═════════╡ │ 59 ┆ 3 │ │ 387 ┆ 3 │ │ 707 ┆ 3 │ │ 1096 ┆ 3 │ │ 2623 ┆ 3 │ └────────┴─────────┘
Post-imputation statistics for ['y']
Where: None
Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬──────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪══════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ y ┆ ┆ 3731 ┆ 3731 ┆ 0.02813 ┆ 2.218 ┆ 0.02813 ┆ 2.218 ┆ -2.747 ┆ -1.387 ┆ 0.009221 ┆ 1.55 ┆ 2.854 ┆ -7.371 ┆ 6.853 │ │ y ┆ 0 ┆ 3000 ┆ 3000 ┆ 0.04756 ┆ 2.21 ┆ 0.04756 ┆ 2.21 ┆ -2.749 ┆ -1.381 ┆ 0.04089 ┆ 1.519 ┆ 2.879 ┆ -7.371 ┆ 6.853 │ │ y ┆ 1 ┆ 731 ┆ 731 ┆ -0.05163 ┆ 2.252 ┆ -0.05163 ┆ 2.252 ┆ -2.747 ┆ -1.47 ┆ -0.1398 ┆ 1.624 ┆ 2.775 ┆ -7.371 ┆ 6.051 │ └──────────┴─────────┴──────┴──────────────┴──────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
y
Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/2.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 1124109848}
Iterations: 100
Model: y=f(x1, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.02717 ┆ 0 ┆ 0.008136 │ └─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 2269.0 ┆ 2269.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.060955 ┆ 0.07728 ┆ 0.041624 │ │ std ┆ 2.192342 ┆ 1.991974 ┆ 2.062989 │ │ min ┆ -7.371072 ┆ -5.127213 ┆ -5.127213 │ │ 25% ┆ -1.347467 ┆ -1.174168 ┆ -1.226518 │ │ 50% ┆ 0.033669 ┆ 0.064593 ┆ 0.037433 │ │ 75% ┆ 1.533057 ┆ 1.481218 ┆ 1.485517 │ │ max ┆ 6.852525 ┆ 4.979518 ┆ 4.979518 │ └────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 2269.0 ┆ 2269.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.060955 ┆ 0.07728 ┆ 0.041624 │ │ std ┆ 2.192342 ┆ 1.991974 ┆ 2.062989 │ │ min ┆ -7.371072 ┆ -5.127213 ┆ -5.127213 │ │ 25% ┆ -1.347467 ┆ -1.174168 ┆ -1.226518 │ │ 50% ┆ 0.033669 ┆ 0.064593 ┆ 0.037433 │ │ 75% ┆ 1.533057 ┆ 1.481218 ┆ 1.485517 │ │ max ┆ 6.852525 ┆ 4.979518 ┆ 4.979518 │ └────────────┴───────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['y']
Most common matches:
shape: (5, 2) ┌────────┬─────────┐ │ row_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞════════╪═════════╡ │ 830 ┆ 4 │ │ 1205 ┆ 4 │ │ 526 ┆ 3 │ │ 942 ┆ 3 │ │ 1114 ┆ 3 │ └────────┴─────────┘
Post-imputation statistics for ['y']
Where: None
Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ y ┆ ┆ 3000 ┆ 3000 ┆ 0.04893 ┆ 2.228 ┆ 0.04893 ┆ 2.228 ┆ -2.789 ┆ -1.362 ┆ 0.0265 ┆ 1.55 ┆ 2.909 ┆ -7.371 ┆ 6.853 │ │ y ┆ 0 ┆ 2269 ┆ 2269 ┆ 0.06095 ┆ 2.192 ┆ 0.06095 ┆ 2.192 ┆ -2.729 ┆ -1.347 ┆ 0.03367 ┆ 1.533 ┆ 2.867 ┆ -7.371 ┆ 6.853 │ │ y ┆ 1 ┆ 731 ┆ 731 ┆ 0.0116 ┆ 2.336 ┆ 0.0116 ┆ 2.336 ┆ -3.077 ┆ -1.489 ┆ -0.005746 ┆ 1.624 ┆ 3.025 ┆ -7.371 ┆ 6.051 │ └──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/3.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 1316246661}
Iterations: 100
Model: y=f(x1, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.02253 ┆ 0 ┆ 0.008136 │ └─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.048929 ┆ 0.039477 ┆ 0.021623 │ │ std ┆ 2.227881 ┆ 2.012686 ┆ 2.068261 │ │ min ┆ -7.371072 ┆ -5.544168 ┆ -5.544168 │ │ 25% ┆ -1.362146 ┆ -1.244078 ┆ -1.377165 │ │ 50% ┆ 0.027032 ┆ -0.011173 ┆ -0.045916 │ │ 75% ┆ 1.550215 ┆ 1.470339 ┆ 1.575968 │ │ max ┆ 6.852525 ┆ 5.024937 ┆ 5.024937 │ └────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.048929 ┆ 0.039477 ┆ 0.021623 │ │ std ┆ 2.227881 ┆ 2.012686 ┆ 2.068261 │ │ min ┆ -7.371072 ┆ -5.544168 ┆ -5.544168 │ │ 25% ┆ -1.362146 ┆ -1.244078 ┆ -1.377165 │ │ 50% ┆ 0.027032 ┆ -0.011173 ┆ -0.045916 │ │ 75% ┆ 1.550215 ┆ 1.470339 ┆ 1.575968 │ │ max ┆ 6.852525 ┆ 5.024937 ┆ 5.024937 │ └────────────┴───────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['y']
Most common matches:
shape: (5, 2) ┌────────┬─────────┐ │ row_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞════════╪═════════╡ │ 288 ┆ 3 │ │ 438 ┆ 3 │ │ 477 ┆ 3 │ │ 658 ┆ 3 │ │ 945 ┆ 3 │ └────────┴─────────┘
Post-imputation statistics for ['y']
Where: None
Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ y ┆ ┆ 3731 ┆ 3731 ┆ 0.0418 ┆ 2.232 ┆ 0.0418 ┆ 2.232 ┆ -2.852 ┆ -1.381 ┆ 0.0135 ┆ 1.587 ┆ 2.904 ┆ -7.371 ┆ 6.853 │ │ y ┆ 0 ┆ 3000 ┆ 3000 ┆ 0.04893 ┆ 2.228 ┆ 0.04893 ┆ 2.228 ┆ -2.791 ┆ -1.362 ┆ 0.0265 ┆ 1.55 ┆ 2.909 ┆ -7.371 ┆ 6.853 │ │ y ┆ 1 ┆ 731 ┆ 731 ┆ 0.01253 ┆ 2.25 ┆ 0.01253 ┆ 2.25 ┆ -3.09 ┆ -1.473 ┆ -0.06233 ┆ 1.722 ┆ 2.893 ┆ -6.216 ┆ 6.104 │ └──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/3.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 4063083334}
Iterations: 100
Model: y=f(x1, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.02253 ┆ 0 ┆ 0.008136 │ └─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.049155 ┆ 0.080256 ┆ 0.058195 │ │ std ┆ 2.20628 ┆ 1.985625 ┆ 2.038018 │ │ min ┆ -7.371072 ┆ -5.070231 ┆ -5.070231 │ │ 25% ┆ -1.362146 ┆ -1.188507 ┆ -1.202682 │ │ 50% ┆ 0.017392 ┆ 0.057376 ┆ 0.034958 │ │ 75% ┆ 1.574584 ┆ 1.545585 ┆ 1.678352 │ │ max ┆ 6.852525 ┆ 5.251679 ┆ 5.251679 │ └────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬───────────┬──────────────┬────────────────┐ │ statistic ┆ y ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪═══════════╪══════════════╪════════════════╡ │ count ┆ 3000.0 ┆ 3000.0 ┆ 731.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.049155 ┆ 0.080256 ┆ 0.058195 │ │ std ┆ 2.20628 ┆ 1.985625 ┆ 2.038018 │ │ min ┆ -7.371072 ┆ -5.070231 ┆ -5.070231 │ │ 25% ┆ -1.362146 ┆ -1.188507 ┆ -1.202682 │ │ 50% ┆ 0.017392 ┆ 0.057376 ┆ 0.034958 │ │ 75% ┆ 1.574584 ┆ 1.545585 ┆ 1.678352 │ │ max ┆ 6.852525 ┆ 5.251679 ┆ 5.251679 │ └────────────┴───────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['y']
Most common matches:
shape: (5, 2) ┌────────┬─────────┐ │ row_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞════════╪═════════╡ │ 826 ┆ 3 │ │ 1387 ┆ 3 │ │ 2520 ┆ 3 │ │ 2607 ┆ 3 │ │ 2854 ┆ 3 │ └────────┴─────────┘
Post-imputation statistics for ['y']
Where: None
Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ y ┆ ┆ 3731 ┆ 3731 ┆ 0.04947 ┆ 2.202 ┆ 0.04947 ┆ 2.202 ┆ -2.77 ┆ -1.346 ┆ -0.0008821 ┆ 1.585 ┆ 2.867 ┆ -7.371 ┆ 6.853 │ │ y ┆ 0 ┆ 3000 ┆ 3000 ┆ 0.04916 ┆ 2.206 ┆ 0.04916 ┆ 2.206 ┆ -2.789 ┆ -1.362 ┆ 0.01585 ┆ 1.575 ┆ 2.878 ┆ -7.371 ┆ 6.853 │ │ y ┆ 1 ┆ 731 ┆ 731 ┆ 0.05078 ┆ 2.186 ┆ 0.05078 ┆ 2.186 ┆ -2.717 ┆ -1.283 ┆ -0.04402 ┆ 1.59 ┆ 2.846 ┆ -6.733 ┆ 6.775 │ └──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
y
Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/3.srmi.implicate
In [3]:
logger.info(
"plot_imputation_quality() compares observed vs. imputed values directly - a "
"density plot showing the whole shape of the distribution for each"
)
fig_quality = srmi.plot_imputation_quality(
kind="density",
path=os.path.join(path_docs_diagnostics, "quality_density.html"),
)
plot_imputation_quality() compares observed vs. imputed values directly - a density plot showing the whole shape of the distribution for each
In [4]:
logger.info(
"This marginal comparison has a real limitation: under MAR, the missing rows "
"can legitimately have a different marginal distribution than the observed "
"ones - that's the whole point of a conditional imputation model, not "
"mean-filling. plot_propensity() checks the same question CONDITIONAL on how "
"similar a row's covariates are to a typically-missing row (its predicted "
"response propensity)"
)
fig_propensity = srmi.plot_propensity(
path=os.path.join(path_docs_diagnostics, "propensity_density.html")
)
This marginal comparison has a real limitation: under MAR, the missing rows can legitimately have a different marginal distribution than the observed ones - that's the whole point of a conditional imputation model, not mean-filling. plot_propensity() checks the same question CONDITIONAL on how similar a row's covariates are to a typically-missing row (its predicted response propensity)
Running lightgbm model with parameters: {'objective': 'binary', 'metric': 'binary_logloss', 'num_leaves': 15, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 114751464}
Iterations: 100
Model: ___imp_missing_y_1=f(x1)
Categorical features: []
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 4125624144}
Iterations: 50
Model: ___y=f(___propensity)
Categorical features: []
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 523860852}
Iterations: 50
Model: ___y=f(___propensity)
Categorical features: []
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 3993437831}
Iterations: 50
Model: ___y=f(___propensity)
Categorical features: []
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 2291387283}
Iterations: 50
Model: ___y=f(___propensity)
Categorical features: []
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 1612214049}
Iterations: 50
Model: ___y=f(___propensity)
Categorical features: []
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 2894653257}
Iterations: 50
Model: ___y=f(___propensity)
Categorical features: []