In [1]:
import os
import numpy as np
import polars as pl

from survey_kit.imputation.srmi import SRMI
from survey_kit import logger, config
In [2]:
# Two "does this look right" diagnostics, both answering a different
# question from convergence: not "did the chain settle down" but "do the
# imputed values themselves look plausible"

n_rows = 3_000
rng = np.random.default_rng(20260913)

x1 = rng.normal(size=n_rows)
y = 2.0 * x1 + rng.normal(scale=1.0, size=n_rows)

df = pl.DataFrame(dict(row_id=range(n_rows), x1=x1, y=y))
missing = rng.random(n_rows) < 0.25
df = df.with_columns(
    pl.when(pl.Series(missing)).then(None).otherwise(pl.col("y")).alias("y")
)

srmi = SRMI.simple_model(
    df=df,
    index="row_id",
    replication=SRMI.Replication(n_implicates=3, n_iterations=3),
    parallel=SRMI.Parallel(enabled=False),
    bootstrap=SRMI.Bootstrap(enabled=True),
    storage=SRMI.Storage(
        path_model=f"{config.path_temp_files}/tutorial_diagnostics_quality_propensity",
        force_start=True,
    ),
)
srmi.run()

path_docs_diagnostics = os.path.join(
    config.code_root, "..", "..", "docs", "tutorials", "srmi", "diagnostics"
)
os.makedirs(path_docs_diagnostics, exist_ok=True)
auto_detect: 'y' -> class=continuous, modeltype=LightGBM, predictors=['x1']
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi
Variable selection before SRMI run, if necessary
     y: Method.No
Hyperparameter tuning before SRMI run, if necessary
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/1.srmi.implicate
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/2.srmi.implicate
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/3.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 3945136551}
     Iterations:                        100
Model:     y=f(x1, bbweight__1)
Categorical features: []

┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐
│ Feature ┆ Gain ┆ Frequency ┆           Model ┆   Model ┆          Impute ┆   Impute │
│         ┆      ┆           ┆ share (missing) ┆    mean ┆ share (missing) ┆     mean │
╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡
│      x1 ┆  1.0 ┆       1.0 ┆               0 ┆ 0.02717 ┆               0 ┆ 0.008136 │
└─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 2269.0    ┆ 2269.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.060955  ┆ 0.080526     ┆ 0.027821       │
│ std        ┆ 2.192342  ┆ 1.947098     ┆ 2.017499       │
│ min        ┆ -7.371072 ┆ -5.171975    ┆ -5.171975      │
│ 25%        ┆ -1.347467 ┆ -1.115578    ┆ -1.218412      │
│ 50%        ┆ 0.033669  ┆ 0.049477     ┆ -0.000123      │
│ 75%        ┆ 1.533057  ┆ 1.34598      ┆ 1.363979       │
│ max        ┆ 6.852525  ┆ 5.211096     ┆ 5.211096       │
└────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 2269.0    ┆ 2269.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.060955  ┆ 0.080526     ┆ 0.027821       │
│ std        ┆ 2.192342  ┆ 1.947098     ┆ 2.017499       │
│ min        ┆ -7.371072 ┆ -5.171975    ┆ -5.171975      │
│ 25%        ┆ -1.347467 ┆ -1.115578    ┆ -1.218412      │
│ 50%        ┆ 0.033669  ┆ 0.049477     ┆ -0.000123      │
│ 75%        ┆ 1.533057  ┆ 1.34598      ┆ 1.363979       │
│ max        ┆ 6.852525  ┆ 5.211096     ┆ 5.211096       │
└────────────┴───────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['y']
     Most common matches: 
shape: (5, 2)
┌────────┬─────────┐
│ row_id ┆ nDonors │
│ ---    ┆ ---     │
│ i16    ┆ i8      │
╞════════╪═════════╡
│ 685    ┆ 4       │
│ 155    ┆ 3       │
│ 256    ┆ 3       │
│ 658    ┆ 3       │
│ 761    ┆ 3       │
└────────┴─────────┘


Post-imputation statistics for ['y']
    Where:          None
    Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬──────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆     mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪══════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│        y ┆         ┆ 3000 ┆         3000 ┆  0.04806 ┆ 2.214 ┆      0.04806 ┆       2.214 ┆      -2.774 ┆      -1.384 ┆     0.02135 ┆       1.542 ┆       2.854 ┆      -7.371 ┆       6.853 │
│        y ┆       0 ┆ 2269 ┆         2269 ┆  0.06095 ┆ 2.192 ┆      0.06095 ┆       2.192 ┆      -2.729 ┆      -1.347 ┆     0.03367 ┆       1.533 ┆       2.867 ┆      -7.371 ┆       6.853 │
│        y ┆       1 ┆  731 ┆          731 ┆ 0.008031 ┆  2.28 ┆     0.008031 ┆        2.28 ┆       -3.09 ┆      -1.449 ┆    -0.01272 ┆        1.59 ┆       2.834 ┆      -6.733 ┆       6.051 │
└──────────┴─────────┴──────┴──────────────┴──────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/1.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 2447140450}
     Iterations:                        100
Model:     y=f(x1, bbweight__1)
Categorical features: []

┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐
│ Feature ┆ Gain ┆ Frequency ┆           Model ┆   Model ┆          Impute ┆   Impute │
│         ┆      ┆           ┆ share (missing) ┆    mean ┆ share (missing) ┆     mean │
╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡
│      x1 ┆  1.0 ┆       1.0 ┆               0 ┆ 0.02253 ┆               0 ┆ 0.008136 │
└─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.048059  ┆ 0.029264     ┆ 0.007197       │
│ std        ┆ 2.213719  ┆ 1.980332     ┆ 2.029843       │
│ min        ┆ -7.371072 ┆ -5.45787     ┆ -5.45787       │
│ 25%        ┆ -1.380858 ┆ -1.082619    ┆ -1.214433      │
│ 50%        ┆ 0.026495  ┆ 0.062378     ┆ 0.067595       │
│ 75%        ┆ 1.542204  ┆ 1.518401     ┆ 1.537569       │
│ max        ┆ 6.852525  ┆ 5.23418      ┆ 5.23418        │
└────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.048059  ┆ 0.029264     ┆ 0.007197       │
│ std        ┆ 2.213719  ┆ 1.980332     ┆ 2.029843       │
│ min        ┆ -7.371072 ┆ -5.45787     ┆ -5.45787       │
│ 25%        ┆ -1.380858 ┆ -1.082619    ┆ -1.214433      │
│ 50%        ┆ 0.026495  ┆ 0.062378     ┆ 0.067595       │
│ 75%        ┆ 1.542204  ┆ 1.518401     ┆ 1.537569       │
│ max        ┆ 6.852525  ┆ 5.23418      ┆ 5.23418        │
└────────────┴───────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['y']
     Most common matches: 
shape: (5, 2)
┌────────┬─────────┐
│ row_id ┆ nDonors │
│ ---    ┆ ---     │
│ i16    ┆ i8      │
╞════════╪═════════╡
│ 429    ┆ 3       │
│ 479    ┆ 3       │
│ 625    ┆ 3       │
│ 814    ┆ 3       │
│ 824    ┆ 3       │
└────────┴─────────┘


Post-imputation statistics for ['y']
    Where:          None
    Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆    mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│        y ┆         ┆ 3731 ┆         3731 ┆ 0.04731 ┆  2.22 ┆      0.04731 ┆        2.22 ┆      -2.789 ┆      -1.367 ┆      0.0135 ┆       1.559 ┆       2.854 ┆      -7.371 ┆       6.853 │
│        y ┆       0 ┆ 3000 ┆         3000 ┆ 0.04806 ┆ 2.214 ┆      0.04806 ┆       2.214 ┆      -2.788 ┆      -1.384 ┆     0.02135 ┆       1.542 ┆       2.854 ┆      -7.371 ┆       6.853 │
│        y ┆       1 ┆  731 ┆          731 ┆ 0.04425 ┆ 2.247 ┆      0.04425 ┆       2.247 ┆      -2.881 ┆      -1.347 ┆     -0.0747 ┆       1.598 ┆       2.846 ┆      -6.216 ┆       6.252 │
└──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/1.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 3250830704}
     Iterations:                        100
Model:     y=f(x1, bbweight__1)
Categorical features: []

┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐
│ Feature ┆ Gain ┆ Frequency ┆           Model ┆   Model ┆          Impute ┆   Impute │
│         ┆      ┆           ┆ share (missing) ┆    mean ┆ share (missing) ┆     mean │
╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡
│      x1 ┆  1.0 ┆       1.0 ┆               0 ┆ 0.02253 ┆               0 ┆ 0.008136 │
└─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.056883  ┆ 0.059056     ┆ 0.035921       │
│ std        ┆ 2.205489  ┆ 1.986718     ┆ 2.041139       │
│ min        ┆ -7.371072 ┆ -5.574627    ┆ -5.574627      │
│ 25%        ┆ -1.347315 ┆ -1.192955    ┆ -1.246748      │
│ 50%        ┆ 0.017392  ┆ 0.012496     ┆ 0.018039       │
│ 75%        ┆ 1.538951  ┆ 1.395735     ┆ 1.432209       │
│ max        ┆ 6.852525  ┆ 5.239808     ┆ 5.239808       │
└────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.056883  ┆ 0.059056     ┆ 0.035921       │
│ std        ┆ 2.205489  ┆ 1.986718     ┆ 2.041139       │
│ min        ┆ -7.371072 ┆ -5.574627    ┆ -5.574627      │
│ 25%        ┆ -1.347315 ┆ -1.192955    ┆ -1.246748      │
│ 50%        ┆ 0.017392  ┆ 0.012496     ┆ 0.018039       │
│ 75%        ┆ 1.538951  ┆ 1.395735     ┆ 1.432209       │
│ max        ┆ 6.852525  ┆ 5.239808     ┆ 5.239808       │
└────────────┴───────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['y']
     Most common matches: 
shape: (5, 2)
┌────────┬─────────┐
│ row_id ┆ nDonors │
│ ---    ┆ ---     │
│ i16    ┆ i8      │
╞════════╪═════════╡
│ 1499   ┆ 4       │
│ 2935   ┆ 4       │
│ 2211   ┆ 3       │
│ 2651   ┆ 3       │
│ 2684   ┆ 3       │
└────────┴─────────┘


Post-imputation statistics for ['y']
    Where:          None
    Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆    mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│        y ┆         ┆ 3731 ┆         3731 ┆ 0.06234 ┆ 2.225 ┆      0.06234 ┆       2.225 ┆      -2.789 ┆      -1.353 ┆     0.03282 ┆       1.587 ┆       2.893 ┆      -7.371 ┆       6.853 │
│        y ┆       0 ┆ 3000 ┆         3000 ┆ 0.05688 ┆ 2.205 ┆      0.05688 ┆       2.205 ┆      -2.767 ┆      -1.347 ┆     0.01585 ┆       1.539 ┆       2.854 ┆      -7.371 ┆       6.853 │
│        y ┆       1 ┆  731 ┆          731 ┆ 0.08474 ┆ 2.303 ┆      0.08474 ┆       2.303 ┆      -2.987 ┆      -1.376 ┆     0.04257 ┆       1.709 ┆       2.968 ┆      -6.398 ┆       6.051 │
└──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘





y

Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/1.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 3097978218}
     Iterations:                        100
Model:     y=f(x1, bbweight__1)
Categorical features: []

┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐
│ Feature ┆ Gain ┆ Frequency ┆           Model ┆   Model ┆          Impute ┆   Impute │
│         ┆      ┆           ┆ share (missing) ┆    mean ┆ share (missing) ┆     mean │
╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡
│      x1 ┆  1.0 ┆       1.0 ┆               0 ┆ 0.02717 ┆               0 ┆ 0.008136 │
└─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 2269.0    ┆ 2269.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.060955  ┆ 0.0629       ┆ 0.010809       │
│ std        ┆ 2.192342  ┆ 1.964057     ┆ 2.026492       │
│ min        ┆ -7.371072 ┆ -5.265192    ┆ -5.265192      │
│ 25%        ┆ -1.347467 ┆ -1.134032    ┆ -1.322259      │
│ 50%        ┆ 0.033669  ┆ 0.007652     ┆ -0.002395      │
│ 75%        ┆ 1.533057  ┆ 1.307768     ┆ 1.307768       │
│ max        ┆ 6.852525  ┆ 5.118538     ┆ 5.118538       │
└────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 2269.0    ┆ 2269.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.060955  ┆ 0.0629       ┆ 0.010809       │
│ std        ┆ 2.192342  ┆ 1.964057     ┆ 2.026492       │
│ min        ┆ -7.371072 ┆ -5.265192    ┆ -5.265192      │
│ 25%        ┆ -1.347467 ┆ -1.134032    ┆ -1.322259      │
│ 50%        ┆ 0.033669  ┆ 0.007652     ┆ -0.002395      │
│ 75%        ┆ 1.533057  ┆ 1.307768     ┆ 1.307768       │
│ max        ┆ 6.852525  ┆ 5.118538     ┆ 5.118538       │
└────────────┴───────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['y']
     Most common matches: 
shape: (5, 2)
┌────────┬─────────┐
│ row_id ┆ nDonors │
│ ---    ┆ ---     │
│ i16    ┆ i8      │
╞════════╪═════════╡
│ 1133   ┆ 4       │
│ 1324   ┆ 4       │
│ 303    ┆ 3       │
│ 358    ┆ 3       │
│ 572    ┆ 3       │
└────────┴─────────┘


Post-imputation statistics for ['y']
    Where:          None
    Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆    mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│        y ┆         ┆ 3000 ┆         3000 ┆ 0.06472 ┆ 2.204 ┆      0.06472 ┆       2.204 ┆      -2.756 ┆      -1.353 ┆     0.04303 ┆       1.536 ┆       2.893 ┆      -7.371 ┆       6.853 │
│        y ┆       0 ┆ 2269 ┆         2269 ┆ 0.06095 ┆ 2.192 ┆      0.06095 ┆       2.192 ┆      -2.729 ┆      -1.347 ┆     0.03367 ┆       1.533 ┆       2.867 ┆      -7.371 ┆       6.853 │
│        y ┆       1 ┆  731 ┆          731 ┆ 0.07642 ┆ 2.242 ┆      0.07642 ┆       2.242 ┆      -2.791 ┆      -1.395 ┆      0.1133 ┆       1.575 ┆       2.927 ┆      -7.371 ┆       6.853 │
└──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/2.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 1780209875}
     Iterations:                        100
Model:     y=f(x1, bbweight__1)
Categorical features: []

┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐
│ Feature ┆ Gain ┆ Frequency ┆           Model ┆   Model ┆          Impute ┆   Impute │
│         ┆      ┆           ┆ share (missing) ┆    mean ┆ share (missing) ┆     mean │
╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡
│      x1 ┆  1.0 ┆       1.0 ┆               0 ┆ 0.02253 ┆               0 ┆ 0.008136 │
└─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.064724  ┆ 0.04197      ┆ 0.02746        │
│ std        ┆ 2.204137  ┆ 1.992147     ┆ 2.043637       │
│ min        ┆ -7.371072 ┆ -5.594726    ┆ -5.594726      │
│ 25%        ┆ -1.352496 ┆ -1.171525    ┆ -1.203535      │
│ 50%        ┆ 0.044082  ┆ 0.112798     ┆ 0.12708        │
│ 75%        ┆ 1.535787  ┆ 1.389405     ┆ 1.446623       │
│ max        ┆ 6.852525  ┆ 5.364161     ┆ 5.364161       │
└────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.064724  ┆ 0.04197      ┆ 0.02746        │
│ std        ┆ 2.204137  ┆ 1.992147     ┆ 2.043637       │
│ min        ┆ -7.371072 ┆ -5.594726    ┆ -5.594726      │
│ 25%        ┆ -1.352496 ┆ -1.171525    ┆ -1.203535      │
│ 50%        ┆ 0.044082  ┆ 0.112798     ┆ 0.12708        │
│ 75%        ┆ 1.535787  ┆ 1.389405     ┆ 1.446623       │
│ max        ┆ 6.852525  ┆ 5.364161     ┆ 5.364161       │
└────────────┴───────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['y']
     Most common matches: 
shape: (5, 2)
┌────────┬─────────┐
│ row_id ┆ nDonors │
│ ---    ┆ ---     │
│ i16    ┆ i8      │
╞════════╪═════════╡
│ 223    ┆ 4       │
│ 628    ┆ 3       │
│ 650    ┆ 3       │
│ 1998   ┆ 3       │
│ 135    ┆ 2       │
└────────┴─────────┘


Post-imputation statistics for ['y']
    Where:          None
    Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬──────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆     mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪══════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│        y ┆         ┆ 3731 ┆         3731 ┆  0.05322 ┆ 2.216 ┆      0.05322 ┆       2.216 ┆      -2.762 ┆      -1.384 ┆     0.04408 ┆       1.536 ┆       2.895 ┆      -7.371 ┆       6.853 │
│        y ┆       0 ┆ 3000 ┆         3000 ┆  0.06472 ┆ 2.204 ┆      0.06472 ┆       2.204 ┆      -2.756 ┆      -1.353 ┆     0.04303 ┆       1.536 ┆       2.893 ┆      -7.371 ┆       6.853 │
│        y ┆       1 ┆  731 ┆          731 ┆ 0.005995 ┆ 2.263 ┆     0.005995 ┆       2.263 ┆      -2.809 ┆      -1.505 ┆      0.0483 ┆       1.499 ┆       2.905 ┆      -7.371 ┆       6.853 │
└──────────┴─────────┴──────┴──────────────┴──────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/2.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 1483604190}
     Iterations:                        100
Model:     y=f(x1, bbweight__1)
Categorical features: []

┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐
│ Feature ┆ Gain ┆ Frequency ┆           Model ┆   Model ┆          Impute ┆   Impute │
│         ┆      ┆           ┆ share (missing) ┆    mean ┆ share (missing) ┆     mean │
╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡
│      x1 ┆  1.0 ┆       1.0 ┆               0 ┆ 0.02253 ┆               0 ┆ 0.008136 │
└─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.047563  ┆ 0.031695     ┆ 0.010274       │
│ std        ┆ 2.209537  ┆ 1.988333     ┆ 2.046537       │
│ min        ┆ -7.371072 ┆ -5.227993    ┆ -5.227993      │
│ 25%        ┆ -1.376245 ┆ -1.104187    ┆ -1.163529      │
│ 50%        ┆ 0.040889  ┆ 0.101236     ┆ 0.077845       │
│ 75%        ┆ 1.519058  ┆ 1.372004     ┆ 1.397818       │
│ max        ┆ 6.852525  ┆ 5.375761     ┆ 5.375761       │
└────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.047563  ┆ 0.031695     ┆ 0.010274       │
│ std        ┆ 2.209537  ┆ 1.988333     ┆ 2.046537       │
│ min        ┆ -7.371072 ┆ -5.227993    ┆ -5.227993      │
│ 25%        ┆ -1.376245 ┆ -1.104187    ┆ -1.163529      │
│ 50%        ┆ 0.040889  ┆ 0.101236     ┆ 0.077845       │
│ 75%        ┆ 1.519058  ┆ 1.372004     ┆ 1.397818       │
│ max        ┆ 6.852525  ┆ 5.375761     ┆ 5.375761       │
└────────────┴───────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['y']
     Most common matches: 
shape: (5, 2)
┌────────┬─────────┐
│ row_id ┆ nDonors │
│ ---    ┆ ---     │
│ i16    ┆ i8      │
╞════════╪═════════╡
│ 59     ┆ 3       │
│ 387    ┆ 3       │
│ 707    ┆ 3       │
│ 1096   ┆ 3       │
│ 2623   ┆ 3       │
└────────┴─────────┘


Post-imputation statistics for ['y']
    Where:          None
    Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬──────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆     mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪══════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│        y ┆         ┆ 3731 ┆         3731 ┆  0.02813 ┆ 2.218 ┆      0.02813 ┆       2.218 ┆      -2.747 ┆      -1.387 ┆    0.009221 ┆        1.55 ┆       2.854 ┆      -7.371 ┆       6.853 │
│        y ┆       0 ┆ 3000 ┆         3000 ┆  0.04756 ┆  2.21 ┆      0.04756 ┆        2.21 ┆      -2.749 ┆      -1.381 ┆     0.04089 ┆       1.519 ┆       2.879 ┆      -7.371 ┆       6.853 │
│        y ┆       1 ┆  731 ┆          731 ┆ -0.05163 ┆ 2.252 ┆     -0.05163 ┆       2.252 ┆      -2.747 ┆       -1.47 ┆     -0.1398 ┆       1.624 ┆       2.775 ┆      -7.371 ┆       6.051 │
└──────────┴─────────┴──────┴──────────────┴──────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘





y

Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/2.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 1124109848}
     Iterations:                        100
Model:     y=f(x1, bbweight__1)
Categorical features: []

┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐
│ Feature ┆ Gain ┆ Frequency ┆           Model ┆   Model ┆          Impute ┆   Impute │
│         ┆      ┆           ┆ share (missing) ┆    mean ┆ share (missing) ┆     mean │
╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡
│      x1 ┆  1.0 ┆       1.0 ┆               0 ┆ 0.02717 ┆               0 ┆ 0.008136 │
└─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 2269.0    ┆ 2269.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.060955  ┆ 0.07728      ┆ 0.041624       │
│ std        ┆ 2.192342  ┆ 1.991974     ┆ 2.062989       │
│ min        ┆ -7.371072 ┆ -5.127213    ┆ -5.127213      │
│ 25%        ┆ -1.347467 ┆ -1.174168    ┆ -1.226518      │
│ 50%        ┆ 0.033669  ┆ 0.064593     ┆ 0.037433       │
│ 75%        ┆ 1.533057  ┆ 1.481218     ┆ 1.485517       │
│ max        ┆ 6.852525  ┆ 4.979518     ┆ 4.979518       │
└────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 2269.0    ┆ 2269.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.060955  ┆ 0.07728      ┆ 0.041624       │
│ std        ┆ 2.192342  ┆ 1.991974     ┆ 2.062989       │
│ min        ┆ -7.371072 ┆ -5.127213    ┆ -5.127213      │
│ 25%        ┆ -1.347467 ┆ -1.174168    ┆ -1.226518      │
│ 50%        ┆ 0.033669  ┆ 0.064593     ┆ 0.037433       │
│ 75%        ┆ 1.533057  ┆ 1.481218     ┆ 1.485517       │
│ max        ┆ 6.852525  ┆ 4.979518     ┆ 4.979518       │
└────────────┴───────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['y']
     Most common matches: 
shape: (5, 2)
┌────────┬─────────┐
│ row_id ┆ nDonors │
│ ---    ┆ ---     │
│ i16    ┆ i8      │
╞════════╪═════════╡
│ 830    ┆ 4       │
│ 1205   ┆ 4       │
│ 526    ┆ 3       │
│ 942    ┆ 3       │
│ 1114   ┆ 3       │
└────────┴─────────┘


Post-imputation statistics for ['y']
    Where:          None
    Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆    mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│        y ┆         ┆ 3000 ┆         3000 ┆ 0.04893 ┆ 2.228 ┆      0.04893 ┆       2.228 ┆      -2.789 ┆      -1.362 ┆      0.0265 ┆        1.55 ┆       2.909 ┆      -7.371 ┆       6.853 │
│        y ┆       0 ┆ 2269 ┆         2269 ┆ 0.06095 ┆ 2.192 ┆      0.06095 ┆       2.192 ┆      -2.729 ┆      -1.347 ┆     0.03367 ┆       1.533 ┆       2.867 ┆      -7.371 ┆       6.853 │
│        y ┆       1 ┆  731 ┆          731 ┆  0.0116 ┆ 2.336 ┆       0.0116 ┆       2.336 ┆      -3.077 ┆      -1.489 ┆   -0.005746 ┆       1.624 ┆       3.025 ┆      -7.371 ┆       6.051 │
└──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/3.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 1316246661}
     Iterations:                        100
Model:     y=f(x1, bbweight__1)
Categorical features: []

┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐
│ Feature ┆ Gain ┆ Frequency ┆           Model ┆   Model ┆          Impute ┆   Impute │
│         ┆      ┆           ┆ share (missing) ┆    mean ┆ share (missing) ┆     mean │
╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡
│      x1 ┆  1.0 ┆       1.0 ┆               0 ┆ 0.02253 ┆               0 ┆ 0.008136 │
└─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.048929  ┆ 0.039477     ┆ 0.021623       │
│ std        ┆ 2.227881  ┆ 2.012686     ┆ 2.068261       │
│ min        ┆ -7.371072 ┆ -5.544168    ┆ -5.544168      │
│ 25%        ┆ -1.362146 ┆ -1.244078    ┆ -1.377165      │
│ 50%        ┆ 0.027032  ┆ -0.011173    ┆ -0.045916      │
│ 75%        ┆ 1.550215  ┆ 1.470339     ┆ 1.575968       │
│ max        ┆ 6.852525  ┆ 5.024937     ┆ 5.024937       │
└────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.048929  ┆ 0.039477     ┆ 0.021623       │
│ std        ┆ 2.227881  ┆ 2.012686     ┆ 2.068261       │
│ min        ┆ -7.371072 ┆ -5.544168    ┆ -5.544168      │
│ 25%        ┆ -1.362146 ┆ -1.244078    ┆ -1.377165      │
│ 50%        ┆ 0.027032  ┆ -0.011173    ┆ -0.045916      │
│ 75%        ┆ 1.550215  ┆ 1.470339     ┆ 1.575968       │
│ max        ┆ 6.852525  ┆ 5.024937     ┆ 5.024937       │
└────────────┴───────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['y']
     Most common matches: 
shape: (5, 2)
┌────────┬─────────┐
│ row_id ┆ nDonors │
│ ---    ┆ ---     │
│ i16    ┆ i8      │
╞════════╪═════════╡
│ 288    ┆ 3       │
│ 438    ┆ 3       │
│ 477    ┆ 3       │
│ 658    ┆ 3       │
│ 945    ┆ 3       │
└────────┴─────────┘


Post-imputation statistics for ['y']
    Where:          None
    Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆    mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│        y ┆         ┆ 3731 ┆         3731 ┆  0.0418 ┆ 2.232 ┆       0.0418 ┆       2.232 ┆      -2.852 ┆      -1.381 ┆      0.0135 ┆       1.587 ┆       2.904 ┆      -7.371 ┆       6.853 │
│        y ┆       0 ┆ 3000 ┆         3000 ┆ 0.04893 ┆ 2.228 ┆      0.04893 ┆       2.228 ┆      -2.791 ┆      -1.362 ┆      0.0265 ┆        1.55 ┆       2.909 ┆      -7.371 ┆       6.853 │
│        y ┆       1 ┆  731 ┆          731 ┆ 0.01253 ┆  2.25 ┆      0.01253 ┆        2.25 ┆       -3.09 ┆      -1.473 ┆    -0.06233 ┆       1.722 ┆       2.893 ┆      -6.216 ┆       6.104 │
└──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/3.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 4063083334}
     Iterations:                        100
Model:     y=f(x1, bbweight__1)
Categorical features: []

┌─────────┬──────┬───────────┬─────────────────┬─────────┬─────────────────┬──────────┐
│ Feature ┆ Gain ┆ Frequency ┆           Model ┆   Model ┆          Impute ┆   Impute │
│         ┆      ┆           ┆ share (missing) ┆    mean ┆ share (missing) ┆     mean │
╞═════════╪══════╪═══════════╪═════════════════╪═════════╪═════════════════╪══════════╡
│      x1 ┆  1.0 ┆       1.0 ┆               0 ┆ 0.02253 ┆               0 ┆ 0.008136 │
└─────────┴──────┴───────────┴─────────────────┴─────────┴─────────────────┴──────────┘
Predictions
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.049155  ┆ 0.080256     ┆ 0.058195       │
│ std        ┆ 2.20628   ┆ 1.985625     ┆ 2.038018       │
│ min        ┆ -7.371072 ┆ -5.070231    ┆ -5.070231      │
│ 25%        ┆ -1.362146 ┆ -1.188507    ┆ -1.202682      │
│ 50%        ┆ 0.017392  ┆ 0.057376     ┆ 0.034958       │
│ 75%        ┆ 1.574584  ┆ 1.545585     ┆ 1.678352       │
│ max        ┆ 6.852525  ┆ 5.251679     ┆ 5.251679       │
└────────────┴───────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬───────────┬──────────────┬────────────────┐
│ statistic  ┆ y         ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---       ┆ ---          ┆ ---            │
│ str        ┆ f64       ┆ f64          ┆ f64            │
╞════════════╪═══════════╪══════════════╪════════════════╡
│ count      ┆ 3000.0    ┆ 3000.0       ┆ 731.0          │
│ null_count ┆ 0.0       ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.049155  ┆ 0.080256     ┆ 0.058195       │
│ std        ┆ 2.20628   ┆ 1.985625     ┆ 2.038018       │
│ min        ┆ -7.371072 ┆ -5.070231    ┆ -5.070231      │
│ 25%        ┆ -1.362146 ┆ -1.188507    ┆ -1.202682      │
│ 50%        ┆ 0.017392  ┆ 0.057376     ┆ 0.034958       │
│ 75%        ┆ 1.574584  ┆ 1.545585     ┆ 1.678352       │
│ max        ┆ 6.852525  ┆ 5.251679     ┆ 5.251679       │
└────────────┴───────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['y'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['y']
     Most common matches: 
shape: (5, 2)
┌────────┬─────────┐
│ row_id ┆ nDonors │
│ ---    ┆ ---     │
│ i16    ┆ i8      │
╞════════╪═════════╡
│ 826    ┆ 3       │
│ 1387   ┆ 3       │
│ 2520   ┆ 3       │
│ 2607   ┆ 3       │
│ 2854   ┆ 3       │
└────────┴─────────┘


Post-imputation statistics for ['y']
    Where:          None
    Where (impute): col(___imp_missing_y_1)
┌──────────┬─────────┬──────┬──────────────┬─────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆    mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪═════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│        y ┆         ┆ 3731 ┆         3731 ┆ 0.04947 ┆ 2.202 ┆      0.04947 ┆       2.202 ┆       -2.77 ┆      -1.346 ┆  -0.0008821 ┆       1.585 ┆       2.867 ┆      -7.371 ┆       6.853 │
│        y ┆       0 ┆ 3000 ┆         3000 ┆ 0.04916 ┆ 2.206 ┆      0.04916 ┆       2.206 ┆      -2.789 ┆      -1.362 ┆     0.01585 ┆       1.575 ┆       2.878 ┆      -7.371 ┆       6.853 │
│        y ┆       1 ┆  731 ┆          731 ┆ 0.05078 ┆ 2.186 ┆      0.05078 ┆       2.186 ┆      -2.717 ┆      -1.283 ┆    -0.04402 ┆        1.59 ┆       2.846 ┆      -6.733 ┆       6.775 │
└──────────┴─────────┴──────┴──────────────┴─────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘





y

Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_diagnostics_quality_propensity.srmi/3.srmi.implicate
In [3]:
logger.info(
    "plot_imputation_quality() compares observed vs. imputed values directly - a "
    "density plot showing the whole shape of the distribution for each"
)
fig_quality = srmi.plot_imputation_quality(
    kind="density",
    path=os.path.join(path_docs_diagnostics, "quality_density.html"),
)
plot_imputation_quality() compares observed vs. imputed values directly - a density plot showing the whole shape of the distribution for each
In [4]:
logger.info(
    "This marginal comparison has a real limitation: under MAR, the missing rows "
    "can legitimately have a different marginal distribution than the observed "
    "ones - that's the whole point of a conditional imputation model, not "
    "mean-filling. plot_propensity() checks the same question CONDITIONAL on how "
    "similar a row's covariates are to a typically-missing row (its predicted "
    "response propensity)"
)
fig_propensity = srmi.plot_propensity(
    path=os.path.join(path_docs_diagnostics, "propensity_density.html")
)
This marginal comparison has a real limitation: under MAR, the missing rows can legitimately have a different marginal distribution than the observed ones - that's the whole point of a conditional imputation model, not mean-filling. plot_propensity() checks the same question CONDITIONAL on how similar a row's covariates are to a typically-missing row (its predicted response propensity)
Running lightgbm model with parameters: {'objective': 'binary', 'metric': 'binary_logloss', 'num_leaves': 15, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 114751464}
     Iterations:                        100
Model:     ___imp_missing_y_1=f(x1)
Categorical features: []

Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 4125624144}
     Iterations:                        50
Model:     ___y=f(___propensity)
Categorical features: []

Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 523860852}
     Iterations:                        50
Model:     ___y=f(___propensity)
Categorical features: []

Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 3993437831}
     Iterations:                        50
Model:     ___y=f(___propensity)
Categorical features: []

Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 2291387283}
     Iterations:                        50
Model:     ___y=f(___propensity)
Categorical features: []

Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 1612214049}
     Iterations:                        50
Model:     ___y=f(___propensity)
Categorical features: []

Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'num_leaves': 8, 'learning_rate': 0.05, 'min_data_in_leaf': 30, 'verbose': -1, 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'seed': 2894653257}
     Iterations:                        50
Model:     ___y=f(___propensity)
Categorical features: []