In [1]:
import numpy as np
import narwhals as nw
import polars as pl
from survey_kit.imputation.srmi import SRMI
from survey_kit.utilities.dataframe import summary
from survey_kit import logger, config
In [2]:
# A semicontinuous ("two-part"/hurdle) variable: most people have $0 of
# self-employment income (they don't have any), and everyone else has some
# genuinely continuous positive amount. Modeling that as one plain
# continuous variable fights itself - the "is it zero" and "how much, given
# it's not zero" questions are really two different models.
n_rows = 4_000
rng = np.random.default_rng(20260913)
x1 = rng.normal(size=n_rows)
has_self_employment_income = (rng.normal(size=n_rows) + 0.4 * x1 > 0.8).astype(int)
self_employment_income = np.where(
has_self_employment_income == 1,
15_000 + 6_000 * x1 + rng.normal(scale=4_000, size=n_rows),
0.0,
)
df = pl.DataFrame(
dict(
person_id=range(n_rows),
x1=x1,
has_self_employment_income=has_self_employment_income,
self_employment_income=self_employment_income,
)
)
# Only the dollar amount has missingness here - the yes/no flag is fully
# observed, a common real-world pattern (people usually answer "do
# you have this income source" even when they skip the amount)
missing_amount = rng.random(n_rows) < 0.2
df = df.with_columns(
pl.when(pl.Series(missing_amount))
.then(None)
.otherwise(pl.col("self_employment_income"))
.alias("self_employment_income")
)
In [3]:
logger.info(
"yn_pairs={value_var: yn_var} routes self_employment_income through a proper "
"two-part model: has_self_employment_income gates it, and only the "
"has_self_employment_income==True population gets a real continuous model fit"
)
srmi = SRMI.simple_model(
df=df,
index="person_id",
yn_pairs={"self_employment_income": "has_self_employment_income"},
replication=SRMI.Replication(n_implicates=2, n_iterations=2),
parallel=SRMI.Parallel(enabled=False),
bootstrap=SRMI.Bootstrap(enabled=True),
storage=SRMI.Storage(
path_model=f"{config.path_temp_files}/tutorial_simple_model_semicontinuous",
force_start=True,
),
)
logger.info(f"Built variables: {[v.impute_var for v in srmi.variables]}")
yn_pairs={value_var: yn_var} routes self_employment_income through a proper two-part model: has_self_employment_income gates it, and only the has_self_employment_income==True population gets a real continuous model fit
auto_detect: 'self_employment_income' (yn='has_self_employment_income') -> class=continuous, modeltype=LightGBM, predictors=['x1', 'has_self_employment_income']
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_simple_model_semicontinuous.srmi
Built variables: ['self_employment_income']
In [4]:
logger.info("Run it")
srmi.run()
Run it
Variable selection before SRMI run, if necessary
self_employment_income: Method.No
Hyperparameter tuning before SRMI run, if necessary
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_simple_model_semicontinuous.srmi/1.srmi.implicate
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_simple_model_semicontinuous.srmi/2.srmi.implicate
Calling _two_part_value_consistency
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 2584501089}
Iterations: 100
Model: self_employment_income=f(x1, has_self_employment_income, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.4713 ┆ 0 ┆ 0.5489 │ └─────────┴──────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬────────────────────────┬──────────────┬────────────────┐ │ statistic ┆ self_employment_income ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════════════════╪══════════════╪════════════════╡ │ count ┆ 735.0 ┆ 735.0 ┆ 203.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 17583.916243 ┆ 17564.854296 ┆ 17861.492964 │ │ std ┆ 6890.67421 ┆ 5747.55897 ┆ 5549.527357 │ │ min ┆ -3538.079948 ┆ 4617.411496 ┆ 4617.411496 │ │ 25% ┆ 12926.583631 ┆ 13689.160544 ┆ 14093.755837 │ │ 50% ┆ 17617.58387 ┆ 17411.091987 ┆ 17282.234638 │ │ 75% ┆ 22098.299793 ┆ 21092.092037 ┆ 21938.52967 │ │ max ┆ 39902.742597 ┆ 32118.146326 ┆ 32118.146326 │ └────────────┴────────────────────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬────────────────────────┬──────────────┬────────────────┐ │ statistic ┆ self_employment_income ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════════════════╪══════════════╪════════════════╡ │ count ┆ 735.0 ┆ 735.0 ┆ 203.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 17583.916243 ┆ 17564.854296 ┆ 17861.492964 │ │ std ┆ 6890.67421 ┆ 5747.55897 ┆ 5549.527357 │ │ min ┆ -3538.079948 ┆ 4617.411496 ┆ 4617.411496 │ │ 25% ┆ 12926.583631 ┆ 13689.160544 ┆ 14093.755837 │ │ 50% ┆ 17617.58387 ┆ 17411.091987 ┆ 17282.234638 │ │ 75% ┆ 22098.299793 ┆ 21092.092037 ┆ 21938.52967 │ │ max ┆ 39902.742597 ┆ 32118.146326 ┆ 32118.146326 │ └────────────┴────────────────────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['self_employment_income'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['self_employment_income']
Most common matches:
shape: (5, 2) ┌───────────┬─────────┐ │ person_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════════╪═════════╡ │ 1097 ┆ 3 │ │ 29 ┆ 2 │ │ 444 ┆ 2 │ │ 883 ┆ 2 │ │ 985 ┆ 2 │ └───────────┴─────────┘
Post-imputation statistics for ['self_employment_income']
Where: col(has_self_employment_income)
Where (impute): col(___imp_missing_self_employment_income_1)
┌────────────────────────┬─────────┬─────┬──────────────┬─────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞════════════════════════╪═════════╪═════╪══════════════╪═════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ self_employment_income ┆ ┆ 938 ┆ 938 ┆ 17590.0 ┆ 6887.0 ┆ 17590.0 ┆ 6887.0 ┆ 8721.0 ┆ 12970.0 ┆ 17620.0 ┆ 22240.0 ┆ 26250.0 ┆ -3538.0 ┆ 39900.0 │ │ self_employment_income ┆ 0 ┆ 735 ┆ 735 ┆ 17580.0 ┆ 6891.0 ┆ 17580.0 ┆ 6891.0 ┆ 8640.0 ┆ 12890.0 ┆ 17620.0 ┆ 22100.0 ┆ 26210.0 ┆ -3538.0 ┆ 39900.0 │ │ self_employment_income ┆ 1 ┆ 203 ┆ 203 ┆ 17620.0 ┆ 6892.0 ┆ 17620.0 ┆ 6892.0 ┆ 8814.0 ┆ 13050.0 ┆ 17790.0 ┆ 22590.0 ┆ 26480.0 ┆ -3538.0 ┆ 36180.0 │ └────────────────────────┴─────────┴─────┴──────────────┴─────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_simple_model_semicontinuous.srmi/1.srmi.implicate
Calling _two_part_value_consistency
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 4259448543}
Iterations: 100
Model: self_employment_income=f(x1, has_self_employment_income, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.4881 ┆ 0 ┆ 0.5489 │ └─────────┴──────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬────────────────────────┬──────────────┬────────────────┐ │ statistic ┆ self_employment_income ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════════════════╪══════════════╪════════════════╡ │ count ┆ 938.0 ┆ 938.0 ┆ 203.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 17591.395722 ┆ 17602.228933 ┆ 17930.892608 │ │ std ┆ 6887.312788 ┆ 5917.783504 ┆ 5874.102761 │ │ min ┆ -3538.079948 ┆ 2579.704692 ┆ 2579.704692 │ │ 25% ┆ 12966.611002 ┆ 13924.797817 ┆ 14700.451058 │ │ 50% ┆ 17628.772305 ┆ 17882.123036 ┆ 17758.539496 │ │ 75% ┆ 22237.95632 ┆ 21293.625806 ┆ 22599.317857 │ │ max ┆ 39902.742597 ┆ 31671.3653 ┆ 31671.3653 │ └────────────┴────────────────────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬────────────────────────┬──────────────┬────────────────┐ │ statistic ┆ self_employment_income ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════════════════╪══════════════╪════════════════╡ │ count ┆ 938.0 ┆ 938.0 ┆ 203.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 17591.395722 ┆ 17602.228933 ┆ 17930.892608 │ │ std ┆ 6887.312788 ┆ 5917.783504 ┆ 5874.102761 │ │ min ┆ -3538.079948 ┆ 2579.704692 ┆ 2579.704692 │ │ 25% ┆ 12966.611002 ┆ 13924.797817 ┆ 14700.451058 │ │ 50% ┆ 17628.772305 ┆ 17882.123036 ┆ 17758.539496 │ │ 75% ┆ 22237.95632 ┆ 21293.625806 ┆ 22599.317857 │ │ max ┆ 39902.742597 ┆ 31671.3653 ┆ 31671.3653 │ └────────────┴────────────────────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['self_employment_income'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['self_employment_income']
Most common matches:
shape: (5, 2) ┌───────────┬─────────┐ │ person_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════════╪═════════╡ │ 268 ┆ 3 │ │ 1294 ┆ 3 │ │ 44 ┆ 2 │ │ 74 ┆ 2 │ │ 473 ┆ 2 │ └───────────┴─────────┘
Post-imputation statistics for ['self_employment_income']
Where: col(has_self_employment_income)
Where (impute): col(___imp_missing_self_employment_income_1)
┌────────────────────────┬─────────┬──────┬──────────────┬─────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞════════════════════════╪═════════╪══════╪══════════════╪═════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ self_employment_income ┆ ┆ 1141 ┆ 1141 ┆ 17610.0 ┆ 6904.0 ┆ 17610.0 ┆ 6904.0 ┆ 8721.0 ┆ 12850.0 ┆ 17820.0 ┆ 22280.0 ┆ 26230.0 ┆ -3538.0 ┆ 39900.0 │ │ self_employment_income ┆ 0 ┆ 938 ┆ 938 ┆ 17590.0 ┆ 6887.0 ┆ 17590.0 ┆ 6887.0 ┆ 8721.0 ┆ 12970.0 ┆ 17620.0 ┆ 22240.0 ┆ 26250.0 ┆ -3538.0 ┆ 39900.0 │ │ self_employment_income ┆ 1 ┆ 203 ┆ 203 ┆ 17710.0 ┆ 6997.0 ┆ 17710.0 ┆ 6997.0 ┆ 8640.0 ┆ 12180.0 ┆ 18280.0 ┆ 22580.0 ┆ 26230.0 ┆ 2144.0 ┆ 39000.0 │ └────────────────────────┴─────────┴──────┴──────────────┴─────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
self_employment_income
Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_simple_model_semicontinuous.srmi/1.srmi.implicate
Calling _two_part_value_consistency
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 448163623}
Iterations: 100
Model: self_employment_income=f(x1, has_self_employment_income, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.4713 ┆ 0 ┆ 0.5489 │ └─────────┴──────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬────────────────────────┬──────────────┬────────────────┐ │ statistic ┆ self_employment_income ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════════════════╪══════════════╪════════════════╡ │ count ┆ 735.0 ┆ 735.0 ┆ 203.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 17583.916243 ┆ 17661.080534 ┆ 18041.661252 │ │ std ┆ 6890.67421 ┆ 5768.345769 ┆ 5825.629148 │ │ min ┆ -3538.079948 ┆ 4358.843485 ┆ 4358.843485 │ │ 25% ┆ 12926.583631 ┆ 13734.337262 ┆ 14341.516503 │ │ 50% ┆ 17617.58387 ┆ 17519.822572 ┆ 16947.637552 │ │ 75% ┆ 22098.299793 ┆ 21223.869252 ┆ 23666.718323 │ │ max ┆ 39902.742597 ┆ 30556.951896 ┆ 30556.951896 │ └────────────┴────────────────────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬────────────────────────┬──────────────┬────────────────┐ │ statistic ┆ self_employment_income ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════════════════╪══════════════╪════════════════╡ │ count ┆ 735.0 ┆ 735.0 ┆ 203.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 17583.916243 ┆ 17661.080534 ┆ 18041.661252 │ │ std ┆ 6890.67421 ┆ 5768.345769 ┆ 5825.629148 │ │ min ┆ -3538.079948 ┆ 4358.843485 ┆ 4358.843485 │ │ 25% ┆ 12926.583631 ┆ 13734.337262 ┆ 14341.516503 │ │ 50% ┆ 17617.58387 ┆ 17519.822572 ┆ 16947.637552 │ │ 75% ┆ 22098.299793 ┆ 21223.869252 ┆ 23666.718323 │ │ max ┆ 39902.742597 ┆ 30556.951896 ┆ 30556.951896 │ └────────────┴────────────────────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['self_employment_income'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['self_employment_income']
Most common matches:
shape: (5, 2) ┌───────────┬─────────┐ │ person_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════════╪═════════╡ │ 373 ┆ 3 │ │ 1882 ┆ 3 │ │ 2084 ┆ 3 │ │ 2098 ┆ 3 │ │ 3711 ┆ 3 │ └───────────┴─────────┘
Post-imputation statistics for ['self_employment_income']
Where: col(has_self_employment_income)
Where (impute): col(___imp_missing_self_employment_income_1)
┌────────────────────────┬─────────┬─────┬──────────────┬─────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞════════════════════════╪═════════╪═════╪══════════════╪═════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ self_employment_income ┆ ┆ 938 ┆ 938 ┆ 17710.0 ┆ 6845.0 ┆ 17710.0 ┆ 6845.0 ┆ 8814.0 ┆ 13020.0 ┆ 17710.0 ┆ 22380.0 ┆ 26250.0 ┆ -3538.0 ┆ 39900.0 │ │ self_employment_income ┆ 0 ┆ 735 ┆ 735 ┆ 17580.0 ┆ 6891.0 ┆ 17580.0 ┆ 6891.0 ┆ 8640.0 ┆ 12890.0 ┆ 17620.0 ┆ 22100.0 ┆ 26210.0 ┆ -3538.0 ┆ 39900.0 │ │ self_employment_income ┆ 1 ┆ 203 ┆ 203 ┆ 18170.0 ┆ 6676.0 ┆ 18170.0 ┆ 6676.0 ┆ 9287.0 ┆ 13370.0 ┆ 17990.0 ┆ 23050.0 ┆ 26600.0 ┆ 2431.0 ┆ 34050.0 │ └────────────────────────┴─────────┴─────┴──────────────┴─────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_simple_model_semicontinuous.srmi/2.srmi.implicate
Calling _two_part_value_consistency
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'metric': 'rmse', 'boosting': 'gbdt', 'min_data_per_group': 25, 'num_threads': 1, 'verbose': -1, 'seed': 645139403}
Iterations: 100
Model: self_employment_income=f(x1, has_self_employment_income, bbweight__1)
Categorical features: []
┌─────────┬──────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════╪══════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ x1 ┆ 1.0 ┆ 1.0 ┆ 0 ┆ 0.4881 ┆ 0 ┆ 0.5489 │ └─────────┴──────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬────────────────────────┬──────────────┬────────────────┐ │ statistic ┆ self_employment_income ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════════════════╪══════════════╪════════════════╡ │ count ┆ 938.0 ┆ 938.0 ┆ 203.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 17711.216597 ┆ 17541.123202 ┆ 17899.055143 │ │ std ┆ 6845.4778 ┆ 5913.013268 ┆ 5942.404683 │ │ min ┆ -3538.079948 ┆ 4250.136055 ┆ 4250.136055 │ │ 25% ┆ 13024.471763 ┆ 13983.238833 ┆ 14627.182257 │ │ 50% ┆ 17740.179121 ┆ 16902.260281 ┆ 16771.671033 │ │ 75% ┆ 22380.357717 ┆ 22043.203842 ┆ 22491.360247 │ │ max ┆ 39902.742597 ┆ 29802.026643 ┆ 29802.026643 │ └────────────┴────────────────────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬────────────────────────┬──────────────┬────────────────┐ │ statistic ┆ self_employment_income ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════════════════╪══════════════╪════════════════╡ │ count ┆ 938.0 ┆ 938.0 ┆ 203.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 17711.216597 ┆ 17541.123202 ┆ 17899.055143 │ │ std ┆ 6845.4778 ┆ 5913.013268 ┆ 5942.404683 │ │ min ┆ -3538.079948 ┆ 4250.136055 ┆ 4250.136055 │ │ 25% ┆ 13024.471763 ┆ 13983.238833 ┆ 14627.182257 │ │ 50% ┆ 17740.179121 ┆ 16902.260281 ┆ 16771.671033 │ │ 75% ┆ 22380.357717 ┆ 22043.203842 ┆ 22491.360247 │ │ max ┆ 39902.742597 ┆ 29802.026643 ┆ 29802.026643 │ └────────────┴────────────────────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['self_employment_income'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['self_employment_income']
Most common matches:
shape: (5, 2) ┌───────────┬─────────┐ │ person_id ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════════╪═════════╡ │ 2584 ┆ 3 │ │ 148 ┆ 2 │ │ 560 ┆ 2 │ │ 563 ┆ 2 │ │ 587 ┆ 2 │ └───────────┴─────────┘
Post-imputation statistics for ['self_employment_income']
Where: col(has_self_employment_income)
Where (impute): col(___imp_missing_self_employment_income_1)
┌────────────────────────┬─────────┬──────┬──────────────┬─────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞════════════════════════╪═════════╪══════╪══════════════╪═════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ self_employment_income ┆ ┆ 1141 ┆ 1141 ┆ 17730.0 ┆ 7000.0 ┆ 17730.0 ┆ 7000.0 ┆ 8814.0 ┆ 13020.0 ┆ 17770.0 ┆ 22380.0 ┆ 26510.0 ┆ -3538.0 ┆ 39900.0 │ │ self_employment_income ┆ 0 ┆ 938 ┆ 938 ┆ 17710.0 ┆ 6845.0 ┆ 17710.0 ┆ 6845.0 ┆ 8814.0 ┆ 13020.0 ┆ 17710.0 ┆ 22380.0 ┆ 26250.0 ┆ -3538.0 ┆ 39900.0 │ │ self_employment_income ┆ 1 ┆ 203 ┆ 203 ┆ 17810.0 ┆ 7691.0 ┆ 17810.0 ┆ 7691.0 ┆ 8745.0 ┆ 12720.0 ┆ 18020.0 ┆ 22420.0 ┆ 27460.0 ┆ -3538.0 ┆ 37040.0 │ └────────────────────────┴─────────┴──────┴──────────────┴─────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
self_employment_income
Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/tutorial_simple_model_semicontinuous.srmi/2.srmi.implicate
In [5]:
logger.info(
"Every imputed value respects the hurdle: 0 whenever "
"has_self_employment_income is False, a real positive draw otherwise"
)
# has_self_employment_income is an int-coded (0/1) column, not a real
# boolean - use == 0 rather than ~, which does bitwise (not logical)
# negation on a non-boolean column and would match every row
_ = (
srmi.df_implicates.filter(nw.col("has_self_employment_income") == 0)
.select("self_employment_income")
.pipe(summary)
)
Every imputed value respects the hurdle: 0 whenever has_self_employment_income is False, a real positive draw otherwise
┌────────────────────────┬───────┬─────────────┬──────┬─────┬─────┬─────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ max │ ╞════════════════════════╪═══════╪═════════════╪══════╪═════╪═════╪═════╡ │ self_employment_income ┆ 3,062 ┆ 0 ┆ 0 ┆ 0 ┆ 0 ┆ 0 │ └────────────────────────┴───────┴─────────────┴──────┴─────┴─────┴─────┘
┌────────────────────────┬───────┬─────────────┬──────┬─────┬─────┬─────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ max │ ╞════════════════════════╪═══════╪═════════════╪══════╪═════╪═════╪═════╡ │ self_employment_income ┆ 3,062 ┆ 0 ┆ 0 ┆ 0 ┆ 0 ┆ 0 │ └────────────────────────┴───────┴─────────────┴──────┴─────┴─────┴─────┘