In [1]:
import sys
import os
from pathlib import Path

import narwhals as nw
import polars as pl
import polars.selectors as cs

from survey_kit.utilities.random import RandomData
from survey_kit.utilities.dataframe import summary

from survey_kit.imputation.variable import Variable
from survey_kit.imputation.parameters import Parameters
from survey_kit.imputation.srmi import SRMI
from survey_kit.imputation.selection import Selection
import survey_kit.imputation.utilities.lightgbm_wrapper as rep_lgbm
from survey_kit.imputation.utilities.lightgbm_wrapper import Tuner, Objective
from survey_kit.imputation.utilities.tuning import HyperparameterSpace, IntRange, FloatRange

from survey_kit import logger, config
from survey_kit.utilities.dataframe import summary, columns_from_list
In [2]:
# Draw some random data

n_rows = 10_000
impute_share = 0.25


df = (
    RandomData(n_rows=n_rows, seed=32565437)
    .index("index")
    .integer("year", 2016, 2020)
    .integer("month", 1, 12)
    .integer("var2", 0, 10)
    .integer("var3", 0, 50)
    .float("var4", 0, 1)
    .integer("var5", 0, 1)
    .float("unrelated_1", 0, 1)
    .float("unrelated_2", 0, 1)
    .float("unrelated_3", 0, 1)
    .float("unrelated_4", 0, 1)
    .float("unrelated_5", 0, 1)
    .np_distribution("epsilon_gbm1", "normal", scale=5)
    .np_distribution("epsilon_gbm2", "normal", scale=5)
    .np_distribution("epsilon_gbm3", "normal", scale=5)
    .float("missing_gbm1", 0, 1)
    .float("missing_gbm2", 0, 1)
    .float("missing_gbm3", 0, 1)
    .to_df()
)


#   Convenience references to them for creating dependent variables
c_var2 = pl.col("var2")
c_var3 = pl.col("var3")
c_var4 = pl.col("var4")
c_var5 = pl.col("var5")

c_e_gbm1 = pl.col("epsilon_gbm1")
c_e_gbm2 = pl.col("epsilon_gbm2")


#   Convenience references to them for creating dependent variables
c_var2 = pl.col("var2")
c_var3 = pl.col("var3")
c_var4 = pl.col("var4")
c_var5 = pl.col("var5")


logger.info("var_gbm1 is binary and conditional on other variables")
c_gbm1 = ((c_var2 * 2 - c_var3 * 3 * c_var5 + c_e_gbm1) > 0).alias("var_gbm1")

logger.info("var_gbm2 is != 0 only if var_gbm1 == True")
c_gbm2 = (
    pl.when(pl.col("var_gbm1"))
    .then((c_var2 * 1.5 - c_var3 * 1 * c_var4 + c_e_gbm2))
    .otherwise(pl.lit(0))
    .alias("var_gbm2")
)

c_gbm3 = (
    pl.when(pl.col("var_gbm1"))
    .then((c_var2 * 1.5 - c_var3 * 1 * c_var4 + c_e_gbm2))
    .otherwise(pl.lit(0))
    .alias("var_gbm3")
)
#   Create a bunch of variables that are functions of the variables created above
df = (
    df.with_columns(c_gbm1)
    .with_columns(c_gbm2, c_gbm3)
    .drop(columns_from_list(df=df, columns="epsilon*"))
    .with_row_index(name="_row_index_")
)
df_original = df

#   Set variables to missing according to the uniform random variables missing_
clear_missing = []
for prefixi in ["gbm"]:
    for i in range(1, 4):
        vari = f"var_{prefixi}{i}"
        missingi = f"missing_{prefixi}{i}"

        clear_missing.append(
            pl.when(pl.col(missingi) < impute_share)
            .then(pl.lit(None))
            .otherwise(pl.col(vari))
            .alias(vari)
        )
df = df.with_columns(clear_missing).drop(cs.starts_with("missing_"))

#   Make a fully collinear var for testing
df = df.with_columns(pl.col("unrelated_1").alias("repeat_1"))


summary(df)
var_gbm1 is binary and conditional on other variables
var_gbm2 is != 0 only if var_gbm1 == True
┌─────────────┬────────┬─────────────┬────────────┬─────────────┬────────────┬───────────┐
│    Variable ┆      n ┆ n (missing) ┆       mean ┆         std ┆        min ┆       max │
╞═════════════╪════════╪═════════════╪════════════╪═════════════╪════════════╪═══════════╡
│ _row_index_ ┆ 10,000 ┆           0 ┆    4,999.5 ┆ 2,886.89568 ┆        0.0 ┆   9,999.0 │
│       index ┆ 10,000 ┆           0 ┆    4,999.5 ┆ 2,886.89568 ┆        0.0 ┆   9,999.0 │
│        year ┆ 10,000 ┆           0 ┆ 2,017.9851 ┆    1.415937 ┆    2,016.0 ┆   2,020.0 │
│       month ┆ 10,000 ┆           0 ┆     6.5137 ┆    3.432141 ┆        1.0 ┆      12.0 │
│        var2 ┆ 10,000 ┆           0 ┆     4.9782 ┆    3.154508 ┆        0.0 ┆      10.0 │
│        var3 ┆ 10,000 ┆           0 ┆    25.1084 ┆   14.752302 ┆        0.0 ┆      50.0 │
│        var4 ┆ 10,000 ┆           0 ┆   0.505666 ┆    0.287861 ┆   0.000027 ┆  0.999997 │
│ unrelated_1 ┆ 10,000 ┆           0 ┆   0.502449 ┆    0.288359 ┆   0.000119 ┆  0.999997 │
│ unrelated_2 ┆ 10,000 ┆           0 ┆   0.500105 ┆    0.287638 ┆   0.000049 ┆  0.999539 │
│ unrelated_3 ┆ 10,000 ┆           0 ┆   0.499175 ┆     0.28876 ┆   0.000129 ┆   0.99994 │
│ unrelated_4 ┆ 10,000 ┆           0 ┆   0.500655 ┆    0.288698 ┆   0.000133 ┆  0.999972 │
│ unrelated_5 ┆ 10,000 ┆           0 ┆    0.49876 ┆    0.288979 ┆   0.000071 ┆  0.999867 │
│    var_gbm2 ┆ 10,000 ┆       2,464 ┆  -2.393041 ┆   10.384672 ┆ -55.354108 ┆ 26.213084 │
│    var_gbm3 ┆ 10,000 ┆       2,596 ┆  -2.500425 ┆   10.365353 ┆ -55.354108 ┆ 26.213084 │
│    repeat_1 ┆ 10,000 ┆           0 ┆   0.502449 ┆    0.288359 ┆   0.000119 ┆  0.999997 │
│        var5 ┆ 10,000 ┆           0 ┆     0.4999 ┆    0.500025 ┆        0.0 ┆       1.0 │
│    var_gbm1 ┆ 10,000 ┆       2,483 ┆   0.526008 ┆    0.499356 ┆        0.0 ┆       1.0 │
└─────────────┴────────┴─────────────┴────────────┴─────────────┴────────────┴───────────┘
Out[2]:
naive plan: (run LazyFrame.explain(optimized=True) to see the optimized plan)

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

UNION

PLAN 0:

WITH_COLUMNS:

[col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["_row_index_".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("_row_index__mean").alias("mean"), col("_row_index__std").alias("std"), col("_row_index__rawn_missing").alias("n (missing)"), col("_row_index__rawn").alias("n"), col("_row_index__min").alias("min"), col("_row_index__max").alias("max")]

SELECT [col("___index___"), col("_row_index__mean"), col("_row_index__std"), col("_row_index__rawn_missing"), col("_row_index__rawn"), col("_row_index__min"), col("_row_index__max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 1:

WITH_COLUMNS:

[col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["index".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("index_mean").alias("mean"), col("index_std").alias("std"), col("index_rawn_missing").alias("n (missing)"), col("index_rawn").alias("n"), col("index_min").alias("min"), col("index_max").alias("max")]

SELECT [col("___index___"), col("index_mean"), col("index_std"), col("index_rawn_missing"), col("index_rawn"), col("index_min"), col("index_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 2:

WITH_COLUMNS:

[col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["year".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("year_mean").alias("mean"), col("year_std").alias("std"), col("year_rawn_missing").alias("n (missing)"), col("year_rawn").alias("n"), col("year_min").alias("min"), col("year_max").alias("max")]

SELECT [col("___index___"), col("year_mean"), col("year_std"), col("year_rawn_missing"), col("year_rawn"), col("year_min"), col("year_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 3:

WITH_COLUMNS:

[col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["month".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("month_mean").alias("mean"), col("month_std").alias("std"), col("month_rawn_missing").alias("n (missing)"), col("month_rawn").alias("n"), col("month_min").alias("min"), col("month_max").alias("max")]

SELECT [col("___index___"), col("month_mean"), col("month_std"), col("month_rawn_missing"), col("month_rawn"), col("month_min"), col("month_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 4:

WITH_COLUMNS:

[col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["var2".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("var2_mean").alias("mean"), col("var2_std").alias("std"), col("var2_rawn_missing").alias("n (missing)"), col("var2_rawn").alias("n"), col("var2_min").alias("min"), col("var2_max").alias("max")]

SELECT [col("___index___"), col("var2_mean"), col("var2_std"), col("var2_rawn_missing"), col("var2_rawn"), col("var2_min"), col("var2_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 5:

WITH_COLUMNS:

[col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["var3".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("var3_mean").alias("mean"), col("var3_std").alias("std"), col("var3_rawn_missing").alias("n (missing)"), col("var3_rawn").alias("n"), col("var3_min").alias("min"), col("var3_max").alias("max")]

SELECT [col("___index___"), col("var3_mean"), col("var3_std"), col("var3_rawn_missing"), col("var3_rawn"), col("var3_min"), col("var3_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 6:

WITH_COLUMNS:

[col("n (missing)").cast(Int16)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["var4".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("var4_mean").alias("mean"), col("var4_std").alias("std"), col("var4_rawn_missing").alias("n (missing)"), col("var4_rawn").alias("n"), col("var4_min").alias("min"), col("var4_max").alias("max")]

SELECT [col("___index___"), col("var4_mean"), col("var4_std"), col("var4_rawn_missing"), col("var4_rawn"), col("var4_min"), col("var4_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 7:

WITH_COLUMNS:

[col("n (missing)").cast(Int16)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["unrelated_1".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("unrelated_1_mean").alias("mean"), col("unrelated_1_std").alias("std"), col("unrelated_1_rawn_missing").alias("n (missing)"), col("unrelated_1_rawn").alias("n"), col("unrelated_1_min").alias("min"), col("unrelated_1_max").alias("max")]

SELECT [col("___index___"), col("unrelated_1_mean"), col("unrelated_1_std"), col("unrelated_1_rawn_missing"), col("unrelated_1_rawn"), col("unrelated_1_min"), col("unrelated_1_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 8:

WITH_COLUMNS:

[col("n (missing)").cast(Int16)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["unrelated_2".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("unrelated_2_mean").alias("mean"), col("unrelated_2_std").alias("std"), col("unrelated_2_rawn_missing").alias("n (missing)"), col("unrelated_2_rawn").alias("n"), col("unrelated_2_min").alias("min"), col("unrelated_2_max").alias("max")]

SELECT [col("___index___"), col("unrelated_2_mean"), col("unrelated_2_std"), col("unrelated_2_rawn_missing"), col("unrelated_2_rawn"), col("unrelated_2_min"), col("unrelated_2_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 9:

WITH_COLUMNS:

[col("n (missing)").cast(Int16)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["unrelated_3".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("unrelated_3_mean").alias("mean"), col("unrelated_3_std").alias("std"), col("unrelated_3_rawn_missing").alias("n (missing)"), col("unrelated_3_rawn").alias("n"), col("unrelated_3_min").alias("min"), col("unrelated_3_max").alias("max")]

SELECT [col("___index___"), col("unrelated_3_mean"), col("unrelated_3_std"), col("unrelated_3_rawn_missing"), col("unrelated_3_rawn"), col("unrelated_3_min"), col("unrelated_3_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 10:

WITH_COLUMNS:

[col("n (missing)").cast(Int16)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["unrelated_4".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("unrelated_4_mean").alias("mean"), col("unrelated_4_std").alias("std"), col("unrelated_4_rawn_missing").alias("n (missing)"), col("unrelated_4_rawn").alias("n"), col("unrelated_4_min").alias("min"), col("unrelated_4_max").alias("max")]

SELECT [col("___index___"), col("unrelated_4_mean"), col("unrelated_4_std"), col("unrelated_4_rawn_missing"), col("unrelated_4_rawn"), col("unrelated_4_min"), col("unrelated_4_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 11:

WITH_COLUMNS:

[col("n (missing)").cast(Int16)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["unrelated_5".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("unrelated_5_mean").alias("mean"), col("unrelated_5_std").alias("std"), col("unrelated_5_rawn_missing").alias("n (missing)"), col("unrelated_5_rawn").alias("n"), col("unrelated_5_min").alias("min"), col("unrelated_5_max").alias("max")]

SELECT [col("___index___"), col("unrelated_5_mean"), col("unrelated_5_std"), col("unrelated_5_rawn_missing"), col("unrelated_5_rawn"), col("unrelated_5_min"), col("unrelated_5_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 12:

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["var_gbm2".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("var_gbm2_mean").alias("mean"), col("var_gbm2_std").alias("std"), col("var_gbm2_rawn_missing").alias("n (missing)"), col("var_gbm2_rawn").alias("n"), col("var_gbm2_min").alias("min"), col("var_gbm2_max").alias("max")]

SELECT [col("___index___"), col("var_gbm2_mean"), col("var_gbm2_std"), col("var_gbm2_rawn_missing"), col("var_gbm2_rawn"), col("var_gbm2_min"), col("var_gbm2_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 13:

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["var_gbm3".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("var_gbm3_mean").alias("mean"), col("var_gbm3_std").alias("std"), col("var_gbm3_rawn_missing").alias("n (missing)"), col("var_gbm3_rawn").alias("n"), col("var_gbm3_min").alias("min"), col("var_gbm3_max").alias("max")]

SELECT [col("___index___"), col("var_gbm3_mean"), col("var_gbm3_std"), col("var_gbm3_rawn_missing"), col("var_gbm3_rawn"), col("var_gbm3_min"), col("var_gbm3_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 14:

WITH_COLUMNS:

[col("n (missing)").cast(Int16)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["repeat_1".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("repeat_1_mean").alias("mean"), col("repeat_1_std").alias("std"), col("repeat_1_rawn_missing").alias("n (missing)"), col("repeat_1_rawn").alias("n"), col("repeat_1_min").alias("min"), col("repeat_1_max").alias("max")]

SELECT [col("___index___"), col("repeat_1_mean"), col("repeat_1_std"), col("repeat_1_rawn_missing"), col("repeat_1_rawn"), col("repeat_1_min"), col("repeat_1_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 15:

WITH_COLUMNS:

[col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["var5".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("var5_mean").alias("mean"), col("var5_std").alias("std"), col("var5_rawn_missing").alias("n (missing)"), col("var5_rawn").alias("n"), col("var5_min").alias("min"), col("var5_max").alias("max")]

SELECT [col("___index___"), col("var5_mean"), col("var5_std"), col("var5_rawn_missing"), col("var5_rawn"), col("var5_min"), col("var5_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

PLAN 16:

WITH_COLUMNS:

[col("min").strict_cast(Float64), col("max").strict_cast(Float64)]

SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")]

WITH_COLUMNS:

["var_gbm1".alias("Variable")]

SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")]

SELECT [col("___index___"), col("var_gbm1_mean").alias("mean"), col("var_gbm1_std").alias("std"), col("var_gbm1_rawn_missing").alias("n (missing)"), col("var_gbm1_rawn").alias("n"), col("var_gbm1_min").alias("min"), col("var_gbm1_max").alias("max")]

SELECT [col("___index___"), col("var_gbm1_mean"), col("var_gbm1_std"), col("var_gbm1_rawn_missing"), col("var_gbm1_rawn"), col("var_gbm1_min"), col("var_gbm1_max")]

DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS

END UNION
In [3]:
logger.info("Define some dummy functions to run after imputation of 2")


#   Test a simple pre-post function
#       These would get run gets run in each iteration (in each implicate)
#           before (preFunctions) or after (postFunctions) this variable is imputed
#   Notes for these functions:
#       1) No type hints on imported package types (will throw an error)
#           i.e. no df:pl.DataFrame or -> pl.DataFrame
#       2) Must be completely self-contained (i.e. all imports within the function)
#           This has to do with how it gets saved and loaded in async calls
#       3) Effectively, you have to assume it'll be called
#           in an environment with no imports before it
def square_var(df, var_to_square: str, name: str):
    import narwhals as nw

    return (
        nw.from_native(df)
        .with_columns((nw.col(var_to_square) ** 2).alias(name))
        .to_native()
    )


def recalculate_interaction(df, var1: str, var2: str, name: str):
    import narwhals as nw

    return (
        nw.from_native(df)
        .with_columns((nw.col(var1) * nw.col(var2)).alias(name))
        .to_native()
    )
Define some dummy functions to run after imputation of 2
In [4]:
logger.info("Set up hyperparameter tuning")
tuner = Tuner(
    space=HyperparameterSpace(
        num_leaves=IntRange(2, 256),
        max_depth=IntRange(2, 256),
        min_data_in_leaf=IntRange(10, 250),
        num_iterations=IntRange(25, 200),
        bagging_fraction=FloatRange(0.5, 1.0),
        bagging_freq=IntRange(1, 5),
    ),
    objective=Objective.mae,
    n_trials=50,
    path_save_dir=f"{config.data_root}/tuner_outputs",
    overwrite=True,
)


vars_impute = []
Set up hyperparameter tuning
In [5]:
logger.info("Impute the boolean variable (var_gbm1)")
logger.info("   to the default setup for predicted mean matching")
logger.info("   using lightgbm")
logger.info("   (you can pass a formula, but you don't need to)")

logger.info("First, set up the lightgbm parameters")
logger.info("   This says, do hyperparameter tuning first (tune)")
logger.info("   Redo it at each run (the tuner's own overwrite=True)")
logger.info(
    "   And sets the lightgbm parameter defaults (parameters) that the tuning can overwrite"
)
parameters_lgbm1 = Parameters.LightGBM(
    tune=True,
    tuner=tuner,
    parameters={
        "objective": "binary",
        "num_leaves": 32,
        "min_data_in_leaf": 20,
        "num_iterations": 100,
        "test_size": 0.2,
        "boosting": "gbdt",
        "categorical_feature": ["var5"],
        "verbose": -1,  # ,
    },
    error=Parameters.ErrorDraw.pmm,
)


logger.info("Actually define the variable and the model")
v_gbm1 = Variable(
    impute_var="var_gbm1",
    model=["var_*", "var4", "var3", "var5", "unrelated_*", "repeat_*"],
    modeltype=Variable.ModelType.LightGBM,
    parameters=parameters_lgbm1,
)
logger.info("Add the variable to the list to be imputed")
vars_impute.append(v_gbm1)


logger.info("Impute the continuous variable (var_gbm2) ")
logger.info("   conditional on var_gbm1, using narwhals (nw.col('var_gbm1'))")
logger.info("   as well as a post-processing edit to set var_gbm2=0 when var_gbm1==0")
logger.info("   and some other random post-processing")
logger.info("Different parameters for the continuous variable")
parameters_lgbm2 = Parameters.LightGBM(
    tune=True,
    tuner=tuner,
    parameters={
        "objective": "regression",
        "num_leaves": 32,
        "min_data_in_leaf": 20,
        "num_iterations": 100,
        "test_size": 0.2,
        "boosting": "gbdt",
        "categorical_feature": ["var5"],
        "verbose": -1,  # ,
    },
    error=Parameters.ErrorDraw.pmm,
)

v_gbm2 = Variable(
    impute_var="var_gbm2",
    sample=Variable.Sample(
        Where=nw.col("var_gbm1"),
        #   Needed in case var_gbm1 changes between iterations
        Where_predict=(nw.col("var_gbm2") != 0),
    ),
    model=["var_*", "var4", "var3", "var5", "unrelated_*", "repeat_*"],
    modeltype=Variable.ModelType.LightGBM,
    parameters=parameters_lgbm2,
    transforms=Variable.Transforms(
        post=[
            (
                nw.when(nw.col("var_gbm1"))
                .then(nw.col("var_gbm2"))
                .otherwise(nw.lit(0))
                .alias("var_gbm2")
            ),
            Variable.PrePost.Function(
                recalculate_interaction,
                parameters=dict(var1="var_gbm1", var2="var_gbm2", name="var_gbm12"),
            ),
            Variable.PrePost.Function(
                square_var,
                parameters=dict(var_to_square="var_gbm2", name="var_gbm2_sq"),
            ),
        ]
    ),
)

vars_impute.append(v_gbm2)


logger.info("Now do one with the quantile-regression lightgbm")
logger.info("   To do this, pass quantiles and set objective='quantile'")
parameters_lgbm3 = Parameters.LightGBM(
    tune=True,
    tuner=tuner,
    quantiles=[0.25, 0.5, 0.75],
    parameters={
        "objective": "quantile",
        "num_leaves": 32,
        "min_data_in_leaf": 20,
        "num_iterations": 100,
        "test_size": 0.2,
        "boosting": "gbdt",
        "categorical_feature": ["var5"],
        "verbose": -1,  # ,
    },
    error=Parameters.ErrorDraw.pmm,
)

v_gbm3 = Variable(
    impute_var="var_gbm3",
    sample=Variable.Sample(
        Where=nw.col("var_gbm1"),
        #   Needed in case var_gbm1 changes between iterations
        Where_predict=(nw.col("var_gbm3") != 0),
    ),
    model=["var_*", "var4", "var3", "var5", "unrelated_*", "repeat_*"],
    modeltype=Variable.ModelType.LightGBM,
    parameters=parameters_lgbm3,
    transforms=Variable.Transforms(
        post=[
            (
                nw.when(nw.col("var_gbm1"))
                .then(nw.col("var_gbm3"))
                .otherwise(nw.lit(0))
                .alias("var_gbm3")
            )
        ]
    ),
)

vars_impute.append(v_gbm3)
Impute the boolean variable (var_gbm1)
   to the default setup for predicted mean matching
   using lightgbm
   (you can pass a formula, but you don't need to)
First, set up the lightgbm parameters
   This says, do hyperparameter tuning first (tune)
   Redo it at each run (the tuner's own overwrite=True)
   And sets the lightgbm parameter defaults (parameters) that the tuning can overwrite
Actually define the variable and the model
Add the variable to the list to be imputed
Impute the continuous variable (var_gbm2) 
   conditional on var_gbm1, using narwhals (nw.col('var_gbm1'))
   as well as a post-processing edit to set var_gbm2=0 when var_gbm1==0
   and some other random post-processing
Different parameters for the continuous variable
Now do one with the quantile-regression lightgbm
   To do this, pass quantiles and set objective='quantile'
In [6]:
logger.info("Set up the imputation")
srmi = SRMI(
    df=df,
    variables=vars_impute,
    index=["index"],
    replication=SRMI.Replication(n_implicates=2, n_iterations=2),
    parallel=SRMI.Parallel(enabled=False),
    bootstrap=SRMI.Bootstrap(enabled=True),
    defaults=SRMI.Defaults(modeltype=Variable.ModelType.pmm),
    storage=SRMI.Storage(
        path_model=f"{config.path_temp_files}/py_srmi_test_gbm", force_start=True
    ),
)
Set up the imputation
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi
In [7]:
logger.info("Run it")
srmi.run()

logger.info("It's automatically saved and can be loaded with (see path_model above):")
logger.info("path_model = f'{config.path_temp_files}/py_srmi_test_gbm'")
logger.info("srmi = SRMI.load(path_model)")
Run it
Variable selection before SRMI run, if necessary
     var_gbm1: Method.No
     var_gbm2: Method.No
     var_gbm3: Method.No
Hyperparameter tuning before SRMI run, if necessary
Tuner: 50 trials finished
Tuner: best value = 0.02464
Tuner: best params = {'num_leaves': 235, 'max_depth': 172, 'min_data_in_leaf': 28, 'num_iterations': 177, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1}
TUNING COMPLETE
Tuner: 50 trials finished
Tuner: best value = 1.79798
Tuner: best params = {'num_leaves': 192, 'max_depth': 62, 'min_data_in_leaf': 33, 'num_iterations': 86, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1}
TUNING COMPLETE
Tuner: 50 trials finished
Tuner: best value = 2.45658
Tuner: best params = {'num_leaves': 185, 'max_depth': 31, 'min_data_in_leaf': 32, 'num_iterations': 181, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5}
TUNING COMPLETE
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/1.srmi.implicate
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/2.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'binary', 'num_leaves': 235, 'min_data_in_leaf': 28, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 172, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1, 'seed': 1205842559}
     Iterations:                        177
Model:     var_gbm1=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆   Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆        ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│ unrelated_2 ┆ 0.1488 ┆    0.1535 ┆               0 ┆ 0.4995 ┆               0 ┆ 0.5019 │
│ unrelated_4 ┆ 0.1457 ┆    0.1515 ┆               0 ┆ 0.5007 ┆               0 ┆ 0.5004 │
│ unrelated_3 ┆ 0.1455 ┆    0.1494 ┆               0 ┆    0.5 ┆               0 ┆ 0.4968 │
│ unrelated_1 ┆ 0.1452 ┆    0.1509 ┆               0 ┆ 0.5028 ┆               0 ┆ 0.5015 │
│        var4 ┆ 0.1447 ┆    0.1472 ┆               0 ┆ 0.5041 ┆               0 ┆ 0.5103 │
│ unrelated_5 ┆ 0.1379 ┆    0.1452 ┆               0 ┆ 0.4966 ┆               0 ┆ 0.5053 │
│        var3 ┆ 0.1321 ┆    0.1022 ┆               0 ┆  25.13 ┆               0 ┆  25.05 │
└─────────────┴────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4)
┌────────────┬──────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---      ┆ ---          ┆ ---            │
│ str        ┆ f64      ┆ f64          ┆ f64            │
╞════════════╪══════════╪══════════════╪════════════════╡
│ count      ┆ 7517.0   ┆ 7517.0       ┆ 2483.0         │
│ null_count ┆ 0.0      ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.526008 ┆ 0.5278       ┆ 0.53199        │
│ std        ┆ null     ┆ 0.359735     ┆ 0.28087        │
│ min        ┆ 0.0      ┆ 0.004719     ┆ 0.006007       │
│ 25%        ┆ null     ┆ 0.146288     ┆ 0.292873       │
│ 50%        ┆ null     ┆ 0.56745      ┆ 0.531791       │
│ 75%        ┆ null     ┆ 0.898777     ┆ 0.778923       │
│ max        ┆ 1.0      ┆ 0.999449     ┆ 0.999021       │
└────────────┴──────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬──────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---      ┆ ---          ┆ ---            │
│ str        ┆ f64      ┆ f64          ┆ f64            │
╞════════════╪══════════╪══════════════╪════════════════╡
│ count      ┆ 7517.0   ┆ 7517.0       ┆ 2483.0         │
│ null_count ┆ 0.0      ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.526008 ┆ 0.5278       ┆ 0.53199        │
│ std        ┆ null     ┆ 0.359735     ┆ 0.28087        │
│ min        ┆ 0.0      ┆ 0.004719     ┆ 0.006007       │
│ 25%        ┆ null     ┆ 0.146288     ┆ 0.292873       │
│ 50%        ┆ null     ┆ 0.56745      ┆ 0.531791       │
│ 75%        ┆ null     ┆ 0.898777     ┆ 0.778923       │
│ max        ┆ 1.0      ┆ 0.999449     ┆ 0.999021       │
└────────────┴──────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['var_gbm1'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['var_gbm1']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 9145  ┆ 5       │
│ 1850  ┆ 4       │
│ 5596  ┆ 4       │
│ 6997  ┆ 4       │
│ 8017  ┆ 4       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm1']
    Where:          None
    Where (impute): col(___imp_missing_var_gbm1_1)
┌──────────┬─────────┬───────┬──────────────┬────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆     n ┆ n (not null) ┆   mean ┆    std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪═══════╪══════════════╪════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm1 ┆         ┆ 10000 ┆        10000 ┆ 0.5241 ┆ 0.4994 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
│ var_gbm1 ┆       0 ┆  7517 ┆         7517 ┆  0.526 ┆ 0.4994 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
│ var_gbm1 ┆       1 ┆  2483 ┆         2483 ┆ 0.5183 ┆ 0.4998 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
└──────────┴─────────┴───────┴──────────────┴────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 192, 'min_data_in_leaf': 33, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 62, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1, 'seed': 3440548605}
     Iterations:                        86
Model:     var_gbm2=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆    Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆         ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│        var4 ┆  0.3966 ┆    0.1593 ┆               0 ┆ 0.5034 ┆               0 ┆ 0.5018 │
│        var3 ┆  0.3914 ┆    0.1266 ┆               0 ┆  25.47 ┆               0 ┆  25.16 │
│ unrelated_2 ┆ 0.04912 ┆     0.153 ┆               0 ┆ 0.4978 ┆               0 ┆ 0.5057 │
│ unrelated_1 ┆ 0.04313 ┆    0.1452 ┆               0 ┆ 0.5036 ┆               0 ┆ 0.4943 │
│ unrelated_5 ┆ 0.04091 ┆    0.1409 ┆               0 ┆ 0.4976 ┆               0 ┆ 0.4892 │
│ unrelated_3 ┆ 0.04064 ┆    0.1354 ┆               0 ┆ 0.4996 ┆               0 ┆ 0.4972 │
│ unrelated_4 ┆ 0.03821 ┆    0.1396 ┆               0 ┆ 0.4984 ┆               0 ┆ 0.4992 │
└─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4)
┌────────────┬────────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm2   ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64        ┆ f64          ┆ f64            │
╞════════════╪════════════╪══════════════╪════════════════╡
│ count      ┆ 3500.0     ┆ 3500.0       ┆ 1302.0         │
│ null_count ┆ 0.0        ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.680075  ┆ -4.575018    ┆ -4.301531      │
│ std        ┆ 14.165682  ┆ 12.794151    ┆ 12.310863      │
│ min        ┆ -55.354108 ┆ -44.238519   ┆ -44.179112     │
│ 25%        ┆ -13.396664 ┆ -12.795744   ┆ -11.85476      │
│ 50%        ┆ -2.262626  ┆ -0.800443    ┆ -0.748676      │
│ 75%        ┆ 6.134272   ┆ 5.652468     ┆ 5.351061       │
│ max        ┆ 26.213084  ┆ 16.631379    ┆ 15.886879      │
└────────────┴────────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬────────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm2   ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64        ┆ f64          ┆ f64            │
╞════════════╪════════════╪══════════════╪════════════════╡
│ count      ┆ 3500.0     ┆ 3500.0       ┆ 1302.0         │
│ null_count ┆ 0.0        ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.680075  ┆ -4.575018    ┆ -4.301531      │
│ std        ┆ 14.165682  ┆ 12.794151    ┆ 12.310863      │
│ min        ┆ -55.354108 ┆ -44.238519   ┆ -44.179112     │
│ 25%        ┆ -13.396664 ┆ -12.795744   ┆ -11.85476      │
│ 50%        ┆ -2.262626  ┆ -0.800443    ┆ -0.748676      │
│ 75%        ┆ 6.134272   ┆ 5.652468     ┆ 5.351061       │
│ max        ┆ 26.213084  ┆ 16.631379    ┆ 15.886879      │
└────────────┴────────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['var_gbm2'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['var_gbm2']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 3956  ┆ 4       │
│ 861   ┆ 3       │
│ 1026  ┆ 3       │
│ 1804  ┆ 3       │
│ 2104  ┆ 3       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm2']
    Where:          col(var_gbm1)
    Where (impute): col(___imp_missing_var_gbm2_2)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆   mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm2 ┆         ┆ 4802 ┆         4802 ┆ -4.667 ┆ 14.05 ┆       -4.667 ┆       14.05 ┆      -25.36 ┆      -13.21 ┆      -2.284 ┆       5.887 ┆       11.66 ┆      -55.35 ┆       26.21 │
│ var_gbm2 ┆       0 ┆ 3500 ┆         3500 ┆  -4.68 ┆ 14.17 ┆        -4.68 ┆       14.17 ┆      -25.61 ┆       -13.4 ┆      -2.284 ┆       6.134 ┆       11.65 ┆      -55.35 ┆       26.21 │
│ var_gbm2 ┆       1 ┆ 1302 ┆         1302 ┆ -4.633 ┆ 13.75 ┆       -4.633 ┆       13.75 ┆      -24.17 ┆      -12.66 ┆      -2.284 ┆       5.119 ┆       11.67 ┆      -55.35 ┆       26.21 │
└──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm2), lit(value=0, dtype=None)).alias(name=var_gbm2)

Calling recalculate_interaction

Calling square_var
     Imputation using LightGBM
Running LightGBM for q=0.25
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.25, 'seed': 1341191336}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.25 prediction: 0.953


Running LightGBM for q=0.5
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.5, 'seed': 819323121}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.5 prediction: 0.952


Running LightGBM for q=0.75
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.75, 'seed': 739646208}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.75 prediction: 0.953


┌──────────┬─────────┬───────┬─────────────┬────────┬───────┬────────┬─────────┬───────┐
│ Variable ┆  Sample ┆     n ┆ n (missing) ┆   mean ┆   std ┆    q25 ┆     q50 ┆   q75 │
╞══════════╪═════════╪═══════╪═════════════╪════════╪═══════╪════════╪═════════╪═══════╡
│    p0.25 ┆   Model ┆ 3,500 ┆           0 ┆ -6.513 ┆ 13.55 ┆  -14.3 ┆  -3.842 ┆ 4.114 │
│          ┆ Imputed ┆ 1,300 ┆           0 ┆ -5.795 ┆ 12.64 ┆ -12.19 ┆  -2.943 ┆ 3.357 │
│     p0.5 ┆   Model ┆ 3,500 ┆           0 ┆ -4.986 ┆ 13.84 ┆ -13.19 ┆  -2.547 ┆ 5.638 │
│          ┆ Imputed ┆ 1,300 ┆           0 ┆ -3.968 ┆ 12.86 ┆ -11.03 ┆  -0.789 ┆ 4.691 │
│    p0.75 ┆   Model ┆ 3,500 ┆           0 ┆ -3.663 ┆  13.6 ┆ -12.23 ┆ -0.6311 ┆ 6.543 │
│          ┆ Imputed ┆ 1,300 ┆           0 ┆ -2.692 ┆ 12.58 ┆ -9.708 ┆  0.2726 ┆ 5.792 │
└──────────┴─────────┴───────┴─────────────┴────────┴───────┴────────┴─────────┴───────┘
Running LightGBM for the mean for estimating the marginal distribution
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'seed': 1361803809}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆    Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆         ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│    var_gbm2 ┆  0.9053 ┆    0.1431 ┆               0 ┆  -4.86 ┆               0 ┆ -3.797 │
│        var3 ┆ 0.01722 ┆   0.09623 ┆               0 ┆  25.36 ┆               0 ┆  25.57 │
│        var4 ┆ 0.01703 ┆    0.1278 ┆               0 ┆  0.505 ┆               0 ┆ 0.5007 │
│ unrelated_2 ┆ 0.01313 ┆    0.1351 ┆               0 ┆ 0.4968 ┆               0 ┆ 0.5068 │
│ unrelated_4 ┆ 0.01256 ┆    0.1224 ┆               0 ┆ 0.4994 ┆               0 ┆ 0.5004 │
│ unrelated_1 ┆ 0.01198 ┆    0.1244 ┆               0 ┆ 0.4951 ┆               0 ┆  0.515 │
│ unrelated_3 ┆ 0.01155 ┆    0.1238 ┆               0 ┆ 0.4954 ┆               0 ┆ 0.5081 │
│ unrelated_5 ┆ 0.01124 ┆    0.1273 ┆               0 ┆ 0.4903 ┆               0 ┆  0.507 │
└─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 3)
┌────────────┬──────────────┬────────────────┐
│ statistic  ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64          ┆ f64            │
╞════════════╪══════════════╪════════════════╡
│ count      ┆ 3481.0       ┆ 1341.0         │
│ null_count ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.859764    ┆ -4.044629      │
│ std        ┆ 13.754049    ┆ 12.697181      │
│ min        ┆ -47.646554   ┆ -44.116914     │
│ 25%        ┆ -13.648037   ┆ -10.351482     │
│ 50%        ┆ -2.147882    ┆ -1.230679      │
│ 75%        ┆ 5.724548     ┆ 4.786001       │
│ max        ┆ 21.923253    ┆ 20.196504      │
└────────────┴──────────────┴────────────────┘
     Correlation between var_gbm3 and prediction: 0.975
     Finding 10 nearest neighbors on ['___yhat']
     Randomly picking one and donating ['var_gbm3']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 2569  ┆ 4       │
│ 4125  ┆ 4       │
│ 603   ┆ 3       │
│ 796   ┆ 3       │
│ 1313  ┆ 3       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm3']
    Where:          col(var_gbm1)
    Where (impute): col(___imp_missing_var_gbm3_3)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆   mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm3 ┆         ┆ 4822 ┆         4822 ┆ -4.591 ┆ 13.77 ┆       -4.591 ┆       13.77 ┆      -24.66 ┆      -12.84 ┆      -2.351 ┆       5.696 ┆        11.1 ┆      -55.35 ┆       26.21 │
│ var_gbm3 ┆       0 ┆ 3481 ┆         3481 ┆ -4.805 ┆ 14.08 ┆       -4.805 ┆       14.08 ┆      -25.51 ┆      -13.67 ┆      -2.547 ┆       5.958 ┆        11.4 ┆      -55.35 ┆       26.21 │
│ var_gbm3 ┆       1 ┆ 1341 ┆         1341 ┆ -4.036 ┆ 12.94 ┆       -4.036 ┆       12.94 ┆       -23.1 ┆      -10.73 ┆      -1.829 ┆       5.081 ┆       10.45 ┆      -55.35 ┆       22.89 │
└──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm3), lit(value=0, dtype=None)).alias(name=var_gbm3)
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/1.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'binary', 'num_leaves': 235, 'min_data_in_leaf': 28, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 172, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1, 'seed': 3105532915}
     Iterations:                        177
Model:     var_gbm1=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm3, var_gbm2, bbweight__1)
Categorical features: ['var5']

┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆    Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆         ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│    var_gbm3 ┆  0.6473 ┆   0.02289 ┆               0 ┆ -2.214 ┆               0 ┆ -1.751 │
│    var_gbm2 ┆  0.1026 ┆   0.02327 ┆               0 ┆ -2.241 ┆               0 ┆ -1.768 │
│        var4 ┆ 0.03928 ┆    0.1497 ┆               0 ┆ 0.5057 ┆               0 ┆ 0.5103 │
│ unrelated_4 ┆ 0.03772 ┆    0.1398 ┆               0 ┆ 0.5007 ┆               0 ┆ 0.5004 │
│ unrelated_3 ┆ 0.03703 ┆    0.1345 ┆               0 ┆ 0.4992 ┆               0 ┆ 0.4968 │
│ unrelated_2 ┆ 0.03693 ┆    0.1406 ┆               0 ┆ 0.5001 ┆               0 ┆ 0.5019 │
│ unrelated_5 ┆ 0.03676 ┆    0.1371 ┆               0 ┆ 0.4988 ┆               0 ┆ 0.5053 │
│ unrelated_1 ┆  0.0362 ┆    0.1356 ┆               0 ┆ 0.5024 ┆               0 ┆ 0.5015 │
│        var3 ┆ 0.02619 ┆    0.1164 ┆               0 ┆  25.11 ┆               0 ┆  25.05 │
└─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4)
┌────────────┬──────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---      ┆ ---          ┆ ---            │
│ str        ┆ f64      ┆ f64          ┆ f64            │
╞════════════╪══════════╪══════════════╪════════════════╡
│ count      ┆ 10000.0  ┆ 10000.0      ┆ 2483.0         │
│ null_count ┆ 0.0      ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.5241   ┆ 0.517766     ┆ 0.488754       │
│ std        ┆ null     ┆ 0.497261     ┆ 0.493773       │
│ min        ┆ 0.0      ┆ 0.000002     ┆ 0.000006       │
│ 25%        ┆ null     ┆ 0.00054      ┆ 0.000472       │
│ 50%        ┆ null     ┆ 0.98715      ┆ 0.04815        │
│ 75%        ┆ null     ┆ 0.999994     ┆ 0.999987       │
│ max        ┆ 1.0      ┆ 1.0          ┆ 1.0            │
└────────────┴──────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬──────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---      ┆ ---          ┆ ---            │
│ str        ┆ f64      ┆ f64          ┆ f64            │
╞════════════╪══════════╪══════════════╪════════════════╡
│ count      ┆ 10000.0  ┆ 10000.0      ┆ 2483.0         │
│ null_count ┆ 0.0      ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.5241   ┆ 0.517766     ┆ 0.488754       │
│ std        ┆ null     ┆ 0.497261     ┆ 0.493773       │
│ min        ┆ 0.0      ┆ 0.000002     ┆ 0.000006       │
│ 25%        ┆ null     ┆ 0.00054      ┆ 0.000472       │
│ 50%        ┆ null     ┆ 0.98715      ┆ 0.04815        │
│ 75%        ┆ null     ┆ 0.999994     ┆ 0.999987       │
│ max        ┆ 1.0      ┆ 1.0          ┆ 1.0            │
└────────────┴──────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['var_gbm1'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['var_gbm1']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 1701  ┆ 5       │
│ 371   ┆ 4       │
│ 2707  ┆ 4       │
│ 4126  ┆ 4       │
│ 5686  ┆ 4       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm1']
    Where:          None
    Where (impute): col(___imp_missing_var_gbm1_1)
┌──────────┬─────────┬───────┬──────────────┬────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆     n ┆ n (not null) ┆   mean ┆    std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪═══════╪══════════════╪════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm1 ┆         ┆ 12483 ┆        12483 ┆ 0.5199 ┆ 0.4996 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
│ var_gbm1 ┆       0 ┆ 10000 ┆        10000 ┆ 0.5241 ┆ 0.4994 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
│ var_gbm1 ┆       1 ┆  2483 ┆         2483 ┆  0.503 ┆ 0.5001 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
└──────────┴─────────┴───────┴──────────────┴────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 192, 'min_data_in_leaf': 33, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 62, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1, 'seed': 1339333759}
     Iterations:                        86
Model:     var_gbm2=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm3, var_gbm1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆    Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆         ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│    var_gbm3 ┆  0.8896 ┆    0.1664 ┆               0 ┆ -4.537 ┆               0 ┆ -4.097 │
│        var4 ┆ 0.02422 ┆     0.128 ┆               0 ┆  0.503 ┆               0 ┆ 0.5021 │
│        var3 ┆ 0.02086 ┆    0.1065 ┆               0 ┆  25.38 ┆               0 ┆  25.15 │
│ unrelated_5 ┆ 0.01394 ┆    0.1225 ┆               0 ┆ 0.4953 ┆               0 ┆ 0.4894 │
│ unrelated_3 ┆ 0.01354 ┆     0.123 ┆               0 ┆  0.499 ┆               0 ┆ 0.4971 │
│ unrelated_2 ┆ 0.01307 ┆    0.1235 ┆               0 ┆    0.5 ┆               0 ┆ 0.5054 │
│ unrelated_4 ┆ 0.01253 ┆    0.1161 ┆               0 ┆ 0.4986 ┆               0 ┆ 0.4989 │
│ unrelated_1 ┆ 0.01229 ┆     0.114 ┆               0 ┆ 0.5011 ┆               0 ┆ 0.4949 │
└─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4)
┌────────────┬────────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm2   ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64        ┆ f64          ┆ f64            │
╞════════════╪════════════╪══════════════╪════════════════╡
│ count      ┆ 4802.0     ┆ 4802.0       ┆ 1305.0         │
│ null_count ┆ 0.0        ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.667355  ┆ -4.744643    ┆ -4.623086      │
│ std        ┆ 14.052945  ┆ 13.503224    ┆ 12.730957      │
│ min        ┆ -55.354108 ┆ -46.262256   ┆ -44.94709      │
│ 25%        ┆ -13.212003 ┆ -13.139275   ┆ -12.453296     │
│ 50%        ┆ -2.262626  ┆ -2.101538    ┆ -1.936671      │
│ 75%        ┆ 5.887332   ┆ 5.474333     ┆ 5.069693       │
│ max        ┆ 26.213084  ┆ 21.848721    ┆ 19.554553      │
└────────────┴────────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬────────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm2   ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64        ┆ f64          ┆ f64            │
╞════════════╪════════════╪══════════════╪════════════════╡
│ count      ┆ 4802.0     ┆ 4802.0       ┆ 1305.0         │
│ null_count ┆ 0.0        ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.667355  ┆ -4.744643    ┆ -4.623086      │
│ std        ┆ 14.052945  ┆ 13.503224    ┆ 12.730957      │
│ min        ┆ -55.354108 ┆ -46.262256   ┆ -44.94709      │
│ 25%        ┆ -13.212003 ┆ -13.139275   ┆ -12.453296     │
│ 50%        ┆ -2.262626  ┆ -2.101538    ┆ -1.936671      │
│ 75%        ┆ 5.887332   ┆ 5.474333     ┆ 5.069693       │
│ max        ┆ 26.213084  ┆ 21.848721    ┆ 19.554553      │
└────────────┴────────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['var_gbm2'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['var_gbm2']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 363   ┆ 3       │
│ 660   ┆ 3       │
│ 4267  ┆ 3       │
│ 5428  ┆ 3       │
│ 5955  ┆ 3       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm2']
    Where:          col(var_gbm1)
    Where (impute): col(___imp_missing_var_gbm2_2)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆   mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm2 ┆         ┆ 6107 ┆         6107 ┆ -4.677 ┆ 13.92 ┆       -4.677 ┆       13.92 ┆       -25.1 ┆      -13.06 ┆      -2.313 ┆        5.73 ┆       11.52 ┆      -55.35 ┆       26.21 │
│ var_gbm2 ┆       0 ┆ 4802 ┆         4802 ┆ -4.667 ┆ 14.05 ┆       -4.667 ┆       14.05 ┆      -25.36 ┆      -13.21 ┆      -2.284 ┆       5.887 ┆       11.66 ┆      -55.35 ┆       26.21 │
│ var_gbm2 ┆       1 ┆ 1305 ┆         1305 ┆ -4.712 ┆ 13.41 ┆       -4.712 ┆       13.41 ┆      -23.45 ┆       -12.4 ┆      -2.418 ┆       5.186 ┆       10.87 ┆      -55.35 ┆       22.08 │
└──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm2), lit(value=0, dtype=None)).alias(name=var_gbm2)

Calling recalculate_interaction

Calling square_var
     Imputation using LightGBM
Running LightGBM for q=0.25
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.25, 'seed': 1239395270}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.25 prediction: 0.977


Running LightGBM for q=0.5
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.5, 'seed': 1142944889}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.5 prediction: 0.978


Running LightGBM for q=0.75
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.75, 'seed': 3410969552}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.75 prediction: 0.977


┌──────────┬─────────┬───────┬─────────────┬────────┬───────┬────────┬──────────┬───────┐
│ Variable ┆  Sample ┆     n ┆ n (missing) ┆   mean ┆   std ┆    q25 ┆      q50 ┆   q75 │
╞══════════╪═════════╪═══════╪═════════════╪════════╪═══════╪════════╪══════════╪═══════╡
│    p0.25 ┆   Model ┆ 4,800 ┆           0 ┆ -5.782 ┆ 13.44 ┆ -13.22 ┆   -3.437 ┆ 4.586 │
│          ┆ Imputed ┆ 1,300 ┆           0 ┆  -5.42 ┆  12.7 ┆ -11.59 ┆   -3.319 ┆ 4.002 │
│     p0.5 ┆   Model ┆ 4,800 ┆           0 ┆ -4.666 ┆  13.5 ┆ -12.39 ┆   -2.467 ┆ 5.544 │
│          ┆ Imputed ┆ 1,300 ┆           0 ┆ -4.053 ┆ 12.78 ┆ -9.861 ┆   -1.974 ┆ 5.071 │
│    p0.75 ┆   Model ┆ 4,800 ┆           0 ┆ -3.708 ┆ 13.33 ┆ -11.46 ┆   -1.042 ┆ 6.241 │
│          ┆ Imputed ┆ 1,300 ┆           0 ┆ -2.894 ┆ 12.67 ┆ -8.807 ┆ -0.01167 ┆ 5.911 │
└──────────┴─────────┴───────┴─────────────┴────────┴───────┴────────┴──────────┴───────┘
Running LightGBM for the mean for estimating the marginal distribution
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'seed': 2514065398}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬──────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆     Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆          ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪══════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│    var_gbm2 ┆     0.95 ┆     0.155 ┆               0 ┆ -4.593 ┆               0 ┆ -3.794 │
│        var4 ┆ 0.009838 ┆    0.1263 ┆               0 ┆ 0.5038 ┆               0 ┆ 0.5007 │
│ unrelated_5 ┆ 0.007622 ┆    0.1334 ┆               0 ┆  0.495 ┆               0 ┆  0.507 │
│ unrelated_3 ┆ 0.007199 ┆    0.1335 ┆               0 ┆ 0.4989 ┆               0 ┆  0.508 │
│ unrelated_4 ┆  0.00693 ┆    0.1227 ┆               0 ┆ 0.4997 ┆               0 ┆ 0.4998 │
│ unrelated_1 ┆ 0.006253 ┆    0.1212 ┆               0 ┆ 0.5006 ┆               0 ┆ 0.5149 │
│        var3 ┆ 0.006242 ┆   0.08741 ┆               0 ┆  25.42 ┆               0 ┆  25.56 │
│ unrelated_2 ┆ 0.005946 ┆    0.1204 ┆               0 ┆ 0.4996 ┆               0 ┆ 0.5068 │
└─────────────┴──────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 3)
┌────────────┬──────────────┬────────────────┐
│ statistic  ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64          ┆ f64            │
╞════════════╪══════════════╪════════════════╡
│ count      ┆ 4822.0       ┆ 1343.0         │
│ null_count ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.622871    ┆ -4.070382      │
│ std        ┆ 13.540317    ┆ 12.771805      │
│ min        ┆ -47.738334   ┆ -44.873052     │
│ 25%        ┆ -12.460759   ┆ -10.550641     │
│ 50%        ┆ -2.296351    ┆ -1.83174       │
│ 75%        ┆ 5.575911     ┆ 4.995854       │
│ max        ┆ 22.695594    ┆ 21.303228      │
└────────────┴──────────────┴────────────────┘
     Correlation between var_gbm3 and prediction: 0.988
     Finding 10 nearest neighbors on ['___yhat']
     Randomly picking one and donating ['var_gbm3']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 6604  ┆ 4       │
│ 1908  ┆ 3       │
│ 2143  ┆ 3       │
│ 2392  ┆ 3       │
│ 3314  ┆ 3       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm3']
    Where:          col(var_gbm1)
    Where (impute): col(___imp_missing_var_gbm3_3)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆   mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm3 ┆         ┆ 6165 ┆         6165 ┆ -4.497 ┆ 13.59 ┆       -4.497 ┆       13.59 ┆      -24.22 ┆      -12.39 ┆      -2.211 ┆       5.592 ┆       10.97 ┆      -55.35 ┆       26.21 │
│ var_gbm3 ┆       0 ┆ 4822 ┆         4822 ┆ -4.591 ┆ 13.77 ┆       -4.591 ┆       13.77 ┆      -24.66 ┆      -12.84 ┆      -2.351 ┆       5.696 ┆        11.1 ┆      -55.35 ┆       26.21 │
│ var_gbm3 ┆       1 ┆ 1343 ┆         1343 ┆ -4.161 ┆ 12.92 ┆       -4.161 ┆       12.92 ┆      -22.98 ┆      -10.85 ┆      -1.962 ┆       5.173 ┆       10.43 ┆      -45.28 ┆       20.66 │
└──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm3), lit(value=0, dtype=None)).alias(name=var_gbm3)

var_gbm1

var_gbm2

var_gbm3

Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/1.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'binary', 'num_leaves': 235, 'min_data_in_leaf': 28, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 172, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1, 'seed': 2356231858}
     Iterations:                        177
Model:     var_gbm1=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆   Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆        ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│ unrelated_4 ┆ 0.1517 ┆    0.1538 ┆               0 ┆ 0.5007 ┆               0 ┆ 0.5004 │
│ unrelated_5 ┆ 0.1472 ┆    0.1526 ┆               0 ┆ 0.4966 ┆               0 ┆ 0.5053 │
│        var4 ┆ 0.1445 ┆    0.1469 ┆               0 ┆ 0.5041 ┆               0 ┆ 0.5103 │
│ unrelated_3 ┆  0.142 ┆     0.147 ┆               0 ┆    0.5 ┆               0 ┆ 0.4968 │
│        var3 ┆ 0.1391 ┆    0.1076 ┆               0 ┆  25.13 ┆               0 ┆  25.05 │
│ unrelated_2 ┆ 0.1385 ┆    0.1468 ┆               0 ┆ 0.4995 ┆               0 ┆ 0.5019 │
│ unrelated_1 ┆ 0.1371 ┆    0.1454 ┆               0 ┆ 0.5028 ┆               0 ┆ 0.5015 │
└─────────────┴────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4)
┌────────────┬──────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---      ┆ ---          ┆ ---            │
│ str        ┆ f64      ┆ f64          ┆ f64            │
╞════════════╪══════════╪══════════════╪════════════════╡
│ count      ┆ 7517.0   ┆ 7517.0       ┆ 2483.0         │
│ null_count ┆ 0.0      ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.526008 ┆ 0.528109     ┆ 0.521178       │
│ std        ┆ null     ┆ 0.360953     ┆ 0.287021       │
│ min        ┆ 0.0      ┆ 0.007065     ┆ 0.002619       │
│ 25%        ┆ null     ┆ 0.14316      ┆ 0.27449        │
│ 50%        ┆ null     ┆ 0.557957     ┆ 0.511862       │
│ 75%        ┆ null     ┆ 0.900884     ┆ 0.782061       │
│ max        ┆ 1.0      ┆ 0.999079     ┆ 0.998803       │
└────────────┴──────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬──────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---      ┆ ---          ┆ ---            │
│ str        ┆ f64      ┆ f64          ┆ f64            │
╞════════════╪══════════╪══════════════╪════════════════╡
│ count      ┆ 7517.0   ┆ 7517.0       ┆ 2483.0         │
│ null_count ┆ 0.0      ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.526008 ┆ 0.528109     ┆ 0.521178       │
│ std        ┆ null     ┆ 0.360953     ┆ 0.287021       │
│ min        ┆ 0.0      ┆ 0.007065     ┆ 0.002619       │
│ 25%        ┆ null     ┆ 0.14316      ┆ 0.27449        │
│ 50%        ┆ null     ┆ 0.557957     ┆ 0.511862       │
│ 75%        ┆ null     ┆ 0.900884     ┆ 0.782061       │
│ max        ┆ 1.0      ┆ 0.999079     ┆ 0.998803       │
└────────────┴──────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['var_gbm1'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['var_gbm1']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 5420  ┆ 5       │
│ 281   ┆ 4       │
│ 1296  ┆ 4       │
│ 2797  ┆ 4       │
│ 3220  ┆ 4       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm1']
    Where:          None
    Where (impute): col(___imp_missing_var_gbm1_1)
┌──────────┬─────────┬───────┬──────────────┬────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆     n ┆ n (not null) ┆   mean ┆    std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪═══════╪══════════════╪════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm1 ┆         ┆ 10000 ┆        10000 ┆  0.525 ┆ 0.4994 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
│ var_gbm1 ┆       0 ┆  7517 ┆         7517 ┆  0.526 ┆ 0.4994 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
│ var_gbm1 ┆       1 ┆  2483 ┆         2483 ┆ 0.5219 ┆ 0.4996 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
└──────────┴─────────┴───────┴──────────────┴────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 192, 'min_data_in_leaf': 33, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 62, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1, 'seed': 2366585833}
     Iterations:                        86
Model:     var_gbm2=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆    Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆         ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│        var3 ┆   0.423 ┆    0.1267 ┆               0 ┆  25.46 ┆               0 ┆  25.28 │
│        var4 ┆  0.3746 ┆     0.173 ┆               0 ┆  0.505 ┆               0 ┆ 0.5008 │
│ unrelated_4 ┆ 0.04323 ┆    0.1469 ┆               0 ┆ 0.5003 ┆               0 ┆ 0.5004 │
│ unrelated_2 ┆ 0.04208 ┆    0.1443 ┆               0 ┆ 0.4997 ┆               0 ┆ 0.5017 │
│ unrelated_3 ┆ 0.04189 ┆    0.1404 ┆               0 ┆ 0.5006 ┆               0 ┆ 0.4965 │
│ unrelated_1 ┆ 0.03818 ┆    0.1368 ┆               0 ┆ 0.5042 ┆               0 ┆ 0.4963 │
│ unrelated_5 ┆ 0.03698 ┆    0.1319 ┆               0 ┆ 0.4969 ┆               0 ┆ 0.4845 │
└─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4)
┌────────────┬────────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm2   ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64        ┆ f64          ┆ f64            │
╞════════════╪════════════╪══════════════╪════════════════╡
│ count      ┆ 3512.0     ┆ 3512.0       ┆ 1319.0         │
│ null_count ┆ 0.0        ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.734387  ┆ -4.660909    ┆ -4.255957      │
│ std        ┆ 14.24151   ┆ 12.86899     ┆ 12.141072      │
│ min        ┆ -55.354108 ┆ -43.067682   ┆ -40.717165     │
│ 25%        ┆ -13.470111 ┆ -12.574851   ┆ -11.945201     │
│ 50%        ┆ -2.307279  ┆ -1.36764     ┆ -0.974679      │
│ 75%        ┆ 6.158269   ┆ 5.791027     ┆ 5.665508       │
│ max        ┆ 26.213084  ┆ 18.216354    ┆ 15.718111      │
└────────────┴────────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬────────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm2   ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64        ┆ f64          ┆ f64            │
╞════════════╪════════════╪══════════════╪════════════════╡
│ count      ┆ 3512.0     ┆ 3512.0       ┆ 1319.0         │
│ null_count ┆ 0.0        ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.734387  ┆ -4.660909    ┆ -4.255957      │
│ std        ┆ 14.24151   ┆ 12.86899     ┆ 12.141072      │
│ min        ┆ -55.354108 ┆ -43.067682   ┆ -40.717165     │
│ 25%        ┆ -13.470111 ┆ -12.574851   ┆ -11.945201     │
│ 50%        ┆ -2.307279  ┆ -1.36764     ┆ -0.974679      │
│ 75%        ┆ 6.158269   ┆ 5.791027     ┆ 5.665508       │
│ max        ┆ 26.213084  ┆ 18.216354    ┆ 15.718111      │
└────────────┴────────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['var_gbm2'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['var_gbm2']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 788   ┆ 4       │
│ 3592  ┆ 4       │
│ 426   ┆ 3       │
│ 460   ┆ 3       │
│ 666   ┆ 3       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm2']
    Where:          col(var_gbm1)
    Where (impute): col(___imp_missing_var_gbm2_2)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆   mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm2 ┆         ┆ 4831 ┆         4831 ┆ -4.623 ┆ 14.03 ┆       -4.623 ┆       14.03 ┆      -25.54 ┆      -13.02 ┆      -2.212 ┆       6.003 ┆       11.51 ┆      -55.35 ┆       26.21 │
│ var_gbm2 ┆       0 ┆ 3512 ┆         3512 ┆ -4.734 ┆ 14.24 ┆       -4.734 ┆       14.24 ┆      -25.77 ┆      -13.48 ┆      -2.313 ┆       6.158 ┆       11.65 ┆      -55.35 ┆       26.21 │
│ var_gbm2 ┆       1 ┆ 1319 ┆         1319 ┆ -4.326 ┆ 13.45 ┆       -4.326 ┆       13.45 ┆      -24.34 ┆      -12.11 ┆      -2.089 ┆       5.733 ┆        11.1 ┆      -46.86 ┆       22.69 │
└──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm2), lit(value=0, dtype=None)).alias(name=var_gbm2)

Calling recalculate_interaction

Calling square_var
     Imputation using LightGBM
Running LightGBM for q=0.25
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.25, 'seed': 4065382159}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.25 prediction: 0.952


Running LightGBM for q=0.5
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.5, 'seed': 500046557}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.5 prediction: 0.953


Running LightGBM for q=0.75
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.75, 'seed': 3057184033}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.75 prediction: 0.954


┌──────────┬─────────┬───────┬─────────────┬────────┬───────┬────────┬─────────┬───────┐
│ Variable ┆  Sample ┆     n ┆ n (missing) ┆   mean ┆   std ┆    q25 ┆     q50 ┆   q75 │
╞══════════╪═════════╪═══════╪═════════════╪════════╪═══════╪════════╪═════════╪═══════╡
│    p0.25 ┆   Model ┆ 3,500 ┆           0 ┆ -6.375 ┆  13.4 ┆ -14.29 ┆  -4.004 ┆ 4.141 │
│          ┆ Imputed ┆ 1,400 ┆           0 ┆ -5.956 ┆ 12.67 ┆ -12.18 ┆  -3.149 ┆ 2.844 │
│     p0.5 ┆   Model ┆ 3,500 ┆           0 ┆ -4.862 ┆ 13.68 ┆ -13.41 ┆  -2.365 ┆ 5.862 │
│          ┆ Imputed ┆ 1,400 ┆           0 ┆ -4.227 ┆ 12.96 ┆ -10.96 ┆  -0.466 ┆ 4.782 │
│    p0.75 ┆   Model ┆ 3,500 ┆           0 ┆ -3.567 ┆ 13.47 ┆ -11.97 ┆   -0.75 ┆ 6.628 │
│          ┆ Imputed ┆ 1,400 ┆           0 ┆  -2.94 ┆ 12.76 ┆ -9.556 ┆ 0.09399 ┆  5.59 │
└──────────┴─────────┴───────┴─────────────┴────────┴───────┴────────┴─────────┴───────┘
Running LightGBM for the mean for estimating the marginal distribution
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'seed': 2506042098}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆    Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆         ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│    var_gbm2 ┆  0.9113 ┆    0.1415 ┆               0 ┆ -4.738 ┆               0 ┆ -4.018 │
│        var4 ┆ 0.01631 ┆    0.1287 ┆               0 ┆ 0.5049 ┆               0 ┆ 0.5039 │
│        var3 ┆  0.0139 ┆   0.09113 ┆               0 ┆  25.36 ┆               0 ┆  25.53 │
│ unrelated_5 ┆ 0.01354 ┆    0.1419 ┆               0 ┆ 0.4895 ┆               0 ┆ 0.5064 │
│ unrelated_4 ┆ 0.01187 ┆    0.1253 ┆               0 ┆ 0.5022 ┆               0 ┆ 0.4971 │
│ unrelated_1 ┆  0.0115 ┆    0.1207 ┆               0 ┆  0.496 ┆               0 ┆ 0.5188 │
│ unrelated_2 ┆ 0.01148 ┆    0.1308 ┆               0 ┆ 0.4971 ┆               0 ┆  0.511 │
│ unrelated_3 ┆ 0.01014 ┆    0.1199 ┆               0 ┆ 0.4963 ┆               0 ┆ 0.5032 │
└─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 3)
┌────────────┬──────────────┬────────────────┐
│ statistic  ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64          ┆ f64            │
╞════════════╪══════════════╪════════════════╡
│ count      ┆ 3462.0       ┆ 1371.0         │
│ null_count ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.857463    ┆ -4.343819      │
│ std        ┆ 13.740428    ┆ 12.841089      │
│ min        ┆ -50.424402   ┆ -43.909974     │
│ 25%        ┆ -13.585667   ┆ -10.897252     │
│ 50%        ┆ -2.24905     ┆ -1.453721      │
│ 75%        ┆ 5.717693     ┆ 4.504366       │
│ max        ┆ 22.708233    ┆ 19.029265      │
└────────────┴──────────────┴────────────────┘
     Correlation between var_gbm3 and prediction: 0.974
     Finding 10 nearest neighbors on ['___yhat']
     Randomly picking one and donating ['var_gbm3']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 7290  ┆ 6       │
│ 1462  ┆ 4       │
│ 4886  ┆ 4       │
│ 6439  ┆ 4       │
│ 7099  ┆ 4       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm3']
    Where:          col(var_gbm1)
    Where (impute): col(___imp_missing_var_gbm3_3)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆   mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm3 ┆         ┆ 4833 ┆         4833 ┆ -4.696 ┆ 13.89 ┆       -4.696 ┆       13.89 ┆      -25.56 ┆      -12.83 ┆      -2.212 ┆        5.73 ┆        11.1 ┆      -55.35 ┆       26.21 │
│ var_gbm3 ┆       0 ┆ 3462 ┆         3462 ┆ -4.818 ┆ 14.11 ┆       -4.818 ┆       14.11 ┆      -25.54 ┆      -13.62 ┆      -2.525 ┆       5.958 ┆       11.35 ┆      -55.35 ┆       26.21 │
│ var_gbm3 ┆       1 ┆ 1371 ┆         1371 ┆  -4.39 ┆ 13.32 ┆        -4.39 ┆       13.32 ┆       -25.6 ┆      -11.04 ┆      -1.806 ┆       4.992 ┆       10.49 ┆      -48.11 ┆       22.89 │
└──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm3), lit(value=0, dtype=None)).alias(name=var_gbm3)
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/2.srmi.implicate
     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'binary', 'num_leaves': 235, 'min_data_in_leaf': 28, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 172, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1, 'seed': 46688280}
     Iterations:                        177
Model:     var_gbm1=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm3, var_gbm2, bbweight__1)
Categorical features: ['var5']

┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆    Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆         ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│    var_gbm3 ┆  0.6317 ┆   0.02784 ┆               0 ┆  -2.27 ┆               0 ┆ -1.886 │
│    var_gbm2 ┆  0.1294 ┆   0.02369 ┆               0 ┆ -2.233 ┆               0 ┆ -1.868 │
│ unrelated_5 ┆ 0.03763 ┆    0.1409 ┆               0 ┆ 0.4988 ┆               0 ┆ 0.5053 │
│ unrelated_2 ┆ 0.03622 ┆    0.1411 ┆               0 ┆ 0.5001 ┆               0 ┆ 0.5019 │
│ unrelated_3 ┆  0.0356 ┆     0.137 ┆               0 ┆ 0.4992 ┆               0 ┆ 0.4968 │
│ unrelated_4 ┆ 0.03534 ┆    0.1395 ┆               0 ┆ 0.5007 ┆               0 ┆ 0.5004 │
│ unrelated_1 ┆ 0.03398 ┆    0.1327 ┆               0 ┆ 0.5024 ┆               0 ┆ 0.5015 │
│        var4 ┆ 0.03279 ┆     0.141 ┆               0 ┆ 0.5057 ┆               0 ┆ 0.5103 │
│        var3 ┆ 0.02735 ┆    0.1164 ┆               0 ┆  25.11 ┆               0 ┆  25.05 │
└─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4)
┌────────────┬──────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---      ┆ ---          ┆ ---            │
│ str        ┆ f64      ┆ f64          ┆ f64            │
╞════════════╪══════════╪══════════════╪════════════════╡
│ count      ┆ 10000.0  ┆ 10000.0      ┆ 2483.0         │
│ null_count ┆ 0.0      ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.525    ┆ 0.518959     ┆ 0.493751       │
│ std        ┆ null     ┆ 0.497398     ┆ 0.494509       │
│ min        ┆ 0.0      ┆ 0.000002     ┆ 0.000004       │
│ 25%        ┆ null     ┆ 0.000445     ┆ 0.000367       │
│ 50%        ┆ null     ┆ 0.990332     ┆ 0.075319       │
│ 75%        ┆ null     ┆ 0.999995     ┆ 0.999989       │
│ max        ┆ 1.0      ┆ 1.0          ┆ 1.0            │
└────────────┴──────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬──────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---      ┆ ---          ┆ ---            │
│ str        ┆ f64      ┆ f64          ┆ f64            │
╞════════════╪══════════╪══════════════╪════════════════╡
│ count      ┆ 10000.0  ┆ 10000.0      ┆ 2483.0         │
│ null_count ┆ 0.0      ┆ 0.0          ┆ 0.0            │
│ mean       ┆ 0.525    ┆ 0.518959     ┆ 0.493751       │
│ std        ┆ null     ┆ 0.497398     ┆ 0.494509       │
│ min        ┆ 0.0      ┆ 0.000002     ┆ 0.000004       │
│ 25%        ┆ null     ┆ 0.000445     ┆ 0.000367       │
│ 50%        ┆ null     ┆ 0.990332     ┆ 0.075319       │
│ 75%        ┆ null     ┆ 0.999995     ┆ 0.999989       │
│ max        ┆ 1.0      ┆ 1.0          ┆ 1.0            │
└────────────┴──────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['var_gbm1'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['var_gbm1']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 83    ┆ 4       │
│ 707   ┆ 4       │
│ 2758  ┆ 4       │
│ 4176  ┆ 4       │
│ 6017  ┆ 4       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm1']
    Where:          None
    Where (impute): col(___imp_missing_var_gbm1_1)
┌──────────┬─────────┬───────┬──────────────┬────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆     n ┆ n (not null) ┆   mean ┆    std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪═══════╪══════════════╪════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm1 ┆         ┆ 12483 ┆        12483 ┆ 0.5212 ┆ 0.4996 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
│ var_gbm1 ┆       0 ┆ 10000 ┆        10000 ┆  0.525 ┆ 0.4994 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
│ var_gbm1 ┆       1 ┆  2483 ┆         2483 ┆ 0.5058 ┆ 0.5001 ┆            1 ┆           0 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 ┆           1 │
└──────────┴─────────┴───────┴──────────────┴────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




     Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 192, 'min_data_in_leaf': 33, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 62, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1, 'seed': 1319690195}
     Iterations:                        86
Model:     var_gbm2=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm3, var_gbm1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆    Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆         ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│    var_gbm3 ┆  0.8912 ┆    0.1826 ┆               0 ┆ -4.659 ┆               0 ┆  -4.31 │
│        var4 ┆ 0.02116 ┆     0.122 ┆               0 ┆ 0.5039 ┆               0 ┆ 0.5002 │
│        var3 ┆ 0.01953 ┆    0.1012 ┆               0 ┆  25.41 ┆               0 ┆  25.23 │
│ unrelated_2 ┆ 0.01525 ┆    0.1222 ┆               0 ┆ 0.5003 ┆               0 ┆ 0.5018 │
│ unrelated_1 ┆ 0.01378 ┆     0.121 ┆               0 ┆  0.502 ┆               0 ┆  0.497 │
│ unrelated_5 ┆ 0.01332 ┆    0.1182 ┆               0 ┆ 0.4935 ┆               0 ┆ 0.4861 │
│ unrelated_3 ┆ 0.01301 ┆    0.1173 ┆               0 ┆ 0.4995 ┆               0 ┆ 0.4959 │
│ unrelated_4 ┆ 0.01271 ┆    0.1154 ┆               0 ┆ 0.5004 ┆               0 ┆ 0.5018 │
└─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4)
┌────────────┬────────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm2   ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64        ┆ f64          ┆ f64            │
╞════════════╪════════════╪══════════════╪════════════════╡
│ count      ┆ 4831.0     ┆ 4831.0       ┆ 1327.0         │
│ null_count ┆ 0.0        ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.622961  ┆ -4.59241     ┆ -4.34546       │
│ std        ┆ 14.030754  ┆ 13.533478    ┆ 12.56035       │
│ min        ┆ -55.354108 ┆ -46.446028   ┆ -44.655684     │
│ 25%        ┆ -13.017682 ┆ -12.750504   ┆ -11.591296     │
│ 50%        ┆ -2.211656  ┆ -1.934721    ┆ -1.838333      │
│ 75%        ┆ 6.002604   ┆ 5.450316     ┆ 4.795132       │
│ max        ┆ 26.213084  ┆ 20.22214     ┆ 18.977101      │
└────────────┴────────────┴──────────────┴────────────────┘
shape: (9, 4)
┌────────────┬────────────┬──────────────┬────────────────┐
│ statistic  ┆ var_gbm2   ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64        ┆ f64          ┆ f64            │
╞════════════╪════════════╪══════════════╪════════════════╡
│ count      ┆ 4831.0     ┆ 4831.0       ┆ 1327.0         │
│ null_count ┆ 0.0        ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.622961  ┆ -4.59241     ┆ -4.34546       │
│ std        ┆ 14.030754  ┆ 13.533478    ┆ 12.56035       │
│ min        ┆ -55.354108 ┆ -46.446028   ┆ -44.655684     │
│ 25%        ┆ -13.017682 ┆ -12.750504   ┆ -11.591296     │
│ 50%        ┆ -2.211656  ┆ -1.934721    ┆ -1.838333      │
│ 75%        ┆ 6.002604   ┆ 5.450316     ┆ 4.795132       │
│ max        ┆ 26.213084  ┆ 20.22214     ┆ 18.977101      │
└────────────┴────────────┴──────────────┴────────────────┘
     error=pmm: donating observed value(s) ['var_gbm2'] from 10-nearest matched donors
     Finding 10 nearest neighbors on ['___prediction']
     Randomly picking one and donating ['var_gbm2']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 1234  ┆ 3       │
│ 1879  ┆ 3       │
│ 2898  ┆ 3       │
│ 3913  ┆ 3       │
│ 4109  ┆ 3       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm2']
    Where:          col(var_gbm1)
    Where (impute): col(___imp_missing_var_gbm2_2)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆   mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm2 ┆         ┆ 6158 ┆         6158 ┆ -4.586 ┆ 13.83 ┆       -4.586 ┆       13.83 ┆      -25.35 ┆      -12.66 ┆      -2.212 ┆       5.887 ┆       11.23 ┆      -55.35 ┆       26.21 │
│ var_gbm2 ┆       0 ┆ 4831 ┆         4831 ┆ -4.623 ┆ 14.03 ┆       -4.623 ┆       14.03 ┆      -25.54 ┆      -13.02 ┆      -2.212 ┆       6.003 ┆       11.51 ┆      -55.35 ┆       26.21 │
│ var_gbm2 ┆       1 ┆ 1327 ┆         1327 ┆ -4.454 ┆ 13.09 ┆       -4.454 ┆       13.09 ┆       -23.2 ┆       -11.6 ┆      -2.284 ┆        5.47 ┆       10.16 ┆      -46.86 ┆       24.05 │
└──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm2), lit(value=0, dtype=None)).alias(name=var_gbm2)

Calling recalculate_interaction

Calling square_var
     Imputation using LightGBM
Running LightGBM for q=0.25
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.25, 'seed': 280142868}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.25 prediction: 0.980


Running LightGBM for q=0.5
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.5, 'seed': 2518451516}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.5 prediction: 0.980


Running LightGBM for q=0.75
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.75, 'seed': 45808945}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

     Correlation between var_gbm3 and q=0.75 prediction: 0.979


┌──────────┬─────────┬───────┬─────────────┬────────┬───────┬────────┬──────────┬───────┐
│ Variable ┆  Sample ┆     n ┆ n (missing) ┆   mean ┆   std ┆    q25 ┆      q50 ┆   q75 │
╞══════════╪═════════╪═══════╪═════════════╪════════╪═══════╪════════╪══════════╪═══════╡
│    p0.25 ┆   Model ┆ 4,800 ┆           0 ┆ -5.781 ┆ 13.62 ┆  -13.6 ┆   -3.773 ┆ 4.283 │
│          ┆ Imputed ┆ 1,400 ┆           0 ┆ -5.516 ┆ 13.11 ┆ -12.05 ┆   -3.346 ┆ 3.582 │
│     p0.5 ┆   Model ┆ 4,800 ┆           0 ┆ -4.723 ┆ 13.64 ┆ -12.89 ┆   -2.229 ┆ 5.473 │
│          ┆ Imputed ┆ 1,400 ┆           0 ┆ -4.263 ┆ 13.12 ┆  -11.1 ┆   -1.494 ┆ 4.887 │
│    p0.75 ┆   Model ┆ 4,800 ┆           0 ┆  -3.84 ┆ 13.47 ┆ -12.05 ┆   -1.363 ┆ 6.034 │
│          ┆ Imputed ┆ 1,400 ┆           0 ┆ -3.148 ┆ 12.95 ┆  -10.0 ┆ -0.03091 ┆ 5.637 │
└──────────┴─────────┴───────┴─────────────┴────────┴───────┴────────┴──────────┴───────┘
Running LightGBM for the mean for estimating the marginal distribution
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'seed': 2021398447}
     Iterations:                        181
Model:     var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']

┌─────────────┬──────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐
│     Feature ┆     Gain ┆ Frequency ┆           Model ┆  Model ┆          Impute ┆ Impute │
│             ┆          ┆           ┆ share (missing) ┆   mean ┆ share (missing) ┆   mean │
╞═════════════╪══════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡
│    var_gbm2 ┆   0.9525 ┆    0.1532 ┆               0 ┆ -4.581 ┆               0 ┆ -4.037 │
│        var4 ┆ 0.008931 ┆    0.1306 ┆               0 ┆ 0.5046 ┆               0 ┆ 0.5039 │
│        var3 ┆ 0.007075 ┆   0.09046 ┆               0 ┆  25.41 ┆               0 ┆  25.51 │
│ unrelated_5 ┆ 0.006945 ┆    0.1318 ┆               0 ┆ 0.4943 ┆               0 ┆ 0.5066 │
│ unrelated_4 ┆ 0.006943 ┆     0.128 ┆               0 ┆ 0.5007 ┆               0 ┆ 0.4974 │
│ unrelated_2 ┆ 0.006072 ┆     0.119 ┆               0 ┆ 0.5011 ┆               0 ┆ 0.5109 │
│ unrelated_1 ┆ 0.006006 ┆    0.1276 ┆               0 ┆ 0.5025 ┆               0 ┆ 0.5188 │
│ unrelated_3 ┆ 0.005532 ┆    0.1193 ┆               0 ┆ 0.4983 ┆               0 ┆ 0.5035 │
└─────────────┴──────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 3)
┌────────────┬──────────────┬────────────────┐
│ statistic  ┆ Model (yhat) ┆ Imputed (yhat) │
│ ---        ┆ ---          ┆ ---            │
│ str        ┆ f64          ┆ f64            │
╞════════════╪══════════════╪════════════════╡
│ count      ┆ 4833.0       ┆ 1373.0         │
│ null_count ┆ 0.0          ┆ 0.0            │
│ mean       ┆ -4.704598    ┆ -4.294848      │
│ std        ┆ 13.707223    ┆ 13.111833      │
│ min        ┆ -52.817484   ┆ -46.164067     │
│ 25%        ┆ -12.827819   ┆ -11.225195     │
│ 50%        ┆ -2.281483    ┆ -1.711092      │
│ 75%        ┆ 5.61156      ┆ 4.932624       │
│ max        ┆ 24.794587    ┆ 20.266915      │
└────────────┴──────────────┴────────────────┘
     Correlation between var_gbm3 and prediction: 0.989
     Finding 10 nearest neighbors on ['___yhat']
     Randomly picking one and donating ['var_gbm3']
     Most common matches: 
shape: (5, 2)
┌───────┬─────────┐
│ index ┆ nDonors │
│ ---   ┆ ---     │
│ i16   ┆ i8      │
╞═══════╪═════════╡
│ 514   ┆ 3       │
│ 1825  ┆ 3       │
│ 2050  ┆ 3       │
│ 2753  ┆ 3       │
│ 4306  ┆ 3       │
└───────┴─────────┘


Post-imputation statistics for ['var_gbm3']
    Where:          col(var_gbm1)
    Where (impute): col(___imp_missing_var_gbm3_3)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐
│ Variable ┆ Imputed ┆    n ┆ n (not null) ┆   mean ┆   std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │
╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡
│ var_gbm3 ┆         ┆ 6206 ┆         6206 ┆ -4.621 ┆ 13.75 ┆       -4.621 ┆       13.75 ┆      -25.56 ┆      -12.41 ┆      -2.107 ┆       5.621 ┆        10.8 ┆      -55.35 ┆       26.21 │
│ var_gbm3 ┆       0 ┆ 4833 ┆         4833 ┆ -4.696 ┆ 13.89 ┆       -4.696 ┆       13.89 ┆      -25.56 ┆      -12.83 ┆      -2.212 ┆        5.73 ┆        11.1 ┆      -55.35 ┆       26.21 │
│ var_gbm3 ┆       1 ┆ 1373 ┆         1373 ┆ -4.356 ┆ 13.27 ┆       -4.356 ┆       13.27 ┆      -25.54 ┆      -10.99 ┆      -1.516 ┆       5.182 ┆       10.32 ┆      -48.11 ┆       19.99 │
└──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘




Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm3), lit(value=0, dtype=None)).alias(name=var_gbm3)

var_gbm1

var_gbm2

var_gbm3

Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/2.srmi.implicate
It's automatically saved and can be loaded with (see path_model above):
path_model = f'{config.path_temp_files}/py_srmi_test_gbm'
srmi = SRMI.load(path_model)
In [8]:
logger.info("Get the results")
_ = df_list = srmi.df_implicates
Get the results
In [9]:
logger.info("\n\nLook at the original")
_ = summary(df_original, detailed=True, drb_round=True)

logger.info("\n\nLook at the imputes")
_ = df_list.pipe(summary, detailed=True, drb_round=True)

logger.info("\n\nLook at the imputes | var_gbm1 == 0")
_ = df_list.filter(~nw.col("var_gbm1")).pipe(summary, detailed=True, drb_round=True)

logger.info("\n\nLook at the imputes | var_gbm1 == 1")
_ = df_list.filter(nw.col("var_gbm1")).pipe(summary, detailed=True, drb_round=True)

Look at the original
┌──────────────┬────────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐
│     Variable ┆      n ┆ n (missing) ┆    mean ┆     std ┆       min ┆     q25 ┆     q50 ┆     q75 ┆     max │
╞══════════════╪════════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡
│  _row_index_ ┆ 10,000 ┆           0 ┆ 5,000.0 ┆ 2,887.0 ┆       0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │
│        index ┆ 10,000 ┆           0 ┆ 5,000.0 ┆ 2,887.0 ┆       0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │
│         year ┆ 10,000 ┆           0 ┆ 2,018.0 ┆   1.416 ┆   2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │
│        month ┆ 10,000 ┆           0 ┆   6.514 ┆   3.432 ┆       1.0 ┆     4.0 ┆     6.0 ┆     9.0 ┆    12.0 │
│         var2 ┆ 10,000 ┆           0 ┆   4.978 ┆   3.155 ┆       0.0 ┆     2.0 ┆     5.0 ┆     8.0 ┆    10.0 │
│         var3 ┆ 10,000 ┆           0 ┆   25.11 ┆   14.75 ┆       0.0 ┆    12.0 ┆    25.0 ┆    38.0 ┆    50.0 │
│         var4 ┆ 10,000 ┆           0 ┆  0.5057 ┆  0.2879 ┆  0.000027 ┆  0.2557 ┆  0.5084 ┆  0.7543 ┆     1.0 │
│  unrelated_1 ┆ 10,000 ┆           0 ┆  0.5024 ┆  0.2884 ┆ 0.0001191 ┆  0.2527 ┆  0.5036 ┆   0.753 ┆     1.0 │
│  unrelated_2 ┆ 10,000 ┆           0 ┆  0.5001 ┆  0.2876 ┆  0.000049 ┆   0.253 ┆  0.4975 ┆  0.7486 ┆  0.9995 │
│  unrelated_3 ┆ 10,000 ┆           0 ┆  0.4992 ┆  0.2888 ┆  0.000129 ┆  0.2512 ┆   0.496 ┆  0.7505 ┆  0.9999 │
│  unrelated_4 ┆ 10,000 ┆           0 ┆  0.5007 ┆  0.2887 ┆ 0.0001329 ┆  0.2501 ┆  0.5005 ┆  0.7518 ┆     1.0 │
│  unrelated_5 ┆ 10,000 ┆           0 ┆  0.4988 ┆   0.289 ┆  0.000071 ┆  0.2496 ┆  0.4968 ┆  0.7498 ┆  0.9999 │
│ missing_gbm1 ┆ 10,000 ┆           0 ┆  0.5026 ┆   0.289 ┆  0.000006 ┆  0.2516 ┆  0.5074 ┆  0.7523 ┆     1.0 │
│ missing_gbm2 ┆ 10,000 ┆           0 ┆  0.5032 ┆  0.2905 ┆  0.000005 ┆  0.2531 ┆  0.5041 ┆   0.756 ┆     1.0 │
│ missing_gbm3 ┆ 10,000 ┆           0 ┆  0.4939 ┆  0.2907 ┆ 0.0001079 ┆  0.2407 ┆  0.4906 ┆   0.741 ┆  0.9998 │
│     var_gbm2 ┆ 10,000 ┆           0 ┆  -2.402 ┆   10.33 ┆    -55.35 ┆  -3.025 ┆     0.0 ┆     0.0 ┆   26.21 │
│     var_gbm3 ┆ 10,000 ┆           0 ┆  -2.402 ┆   10.33 ┆    -55.35 ┆  -3.025 ┆     0.0 ┆     0.0 ┆   26.21 │
│         var5 ┆ 10,000 ┆           0 ┆  0.4999 ┆     0.5 ┆       0.0 ┆     0.0 ┆     0.0 ┆     1.0 ┆     1.0 │
│     var_gbm1 ┆ 10,000 ┆           0 ┆  0.5229 ┆  0.4995 ┆       0.0 ┆     0.0 ┆     1.0 ┆     1.0 ┆     1.0 │
└──────────────┴────────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘

Look at the imputes
┌─────────────┬────────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐
│    Variable ┆      n ┆ n (missing) ┆    mean ┆     std ┆       min ┆     q25 ┆     q50 ┆     q75 ┆     max │
╞═════════════╪════════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡
│       index ┆ 10,000 ┆           0 ┆ 5,000.0 ┆ 2,887.0 ┆       0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │
│ _row_index_ ┆ 10,000 ┆           0 ┆ 5,000.0 ┆ 2,887.0 ┆       0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │
│        year ┆ 10,000 ┆           0 ┆ 2,018.0 ┆   1.416 ┆   2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │
│       month ┆ 10,000 ┆           0 ┆   6.514 ┆   3.432 ┆       1.0 ┆     4.0 ┆     6.0 ┆     9.0 ┆    12.0 │
│        var2 ┆ 10,000 ┆           0 ┆   4.978 ┆   3.155 ┆       0.0 ┆     2.0 ┆     5.0 ┆     8.0 ┆    10.0 │
│        var3 ┆ 10,000 ┆           0 ┆   25.11 ┆   14.75 ┆       0.0 ┆    12.0 ┆    25.0 ┆    38.0 ┆    50.0 │
│        var4 ┆ 10,000 ┆           0 ┆  0.5057 ┆  0.2879 ┆  0.000027 ┆  0.2557 ┆  0.5084 ┆  0.7543 ┆     1.0 │
│ unrelated_1 ┆ 10,000 ┆           0 ┆  0.5024 ┆  0.2884 ┆ 0.0001191 ┆  0.2527 ┆  0.5036 ┆   0.753 ┆     1.0 │
│ unrelated_2 ┆ 10,000 ┆           0 ┆  0.5001 ┆  0.2876 ┆  0.000049 ┆   0.253 ┆  0.4975 ┆  0.7486 ┆  0.9995 │
│ unrelated_3 ┆ 10,000 ┆           0 ┆  0.4992 ┆  0.2888 ┆  0.000129 ┆  0.2512 ┆   0.496 ┆  0.7505 ┆  0.9999 │
│ unrelated_4 ┆ 10,000 ┆           0 ┆  0.5007 ┆  0.2887 ┆ 0.0001329 ┆  0.2501 ┆  0.5005 ┆  0.7518 ┆     1.0 │
│ unrelated_5 ┆ 10,000 ┆           0 ┆  0.4988 ┆   0.289 ┆  0.000071 ┆  0.2496 ┆  0.4968 ┆  0.7498 ┆  0.9999 │
│    repeat_1 ┆ 10,000 ┆           0 ┆  0.5024 ┆  0.2884 ┆ 0.0001191 ┆  0.2527 ┆  0.5036 ┆   0.753 ┆     1.0 │
│    var_gbm3 ┆ 10,000 ┆           0 ┆  -2.231 ┆   9.836 ┆    -55.35 ┆  -1.777 ┆     0.0 ┆     0.0 ┆   26.21 │
│    var_gbm2 ┆ 10,000 ┆           0 ┆  -2.253 ┆   9.958 ┆    -55.35 ┆  -1.614 ┆     0.0 ┆     0.0 ┆   26.21 │
│   var_gbm12 ┆ 10,000 ┆           0 ┆  -2.253 ┆   9.958 ┆    -55.35 ┆  -1.614 ┆     0.0 ┆     0.0 ┆   26.21 │
│ var_gbm2_sq ┆ 10,000 ┆           0 ┆   104.2 ┆   267.4 ┆       0.0 ┆     0.0 ┆     0.0 ┆   72.04 ┆ 3,064.0 │
│        var5 ┆ 10,000 ┆           0 ┆  0.4999 ┆     0.5 ┆       0.0 ┆     0.0 ┆     0.0 ┆     1.0 ┆     1.0 │
│    var_gbm1 ┆ 10,000 ┆           0 ┆  0.5203 ┆  0.4996 ┆       0.0 ┆     0.0 ┆     1.0 ┆     1.0 ┆     1.0 │
└─────────────┴────────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘
┌─────────────┬────────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐
│    Variable ┆      n ┆ n (missing) ┆    mean ┆     std ┆       min ┆     q25 ┆     q50 ┆     q75 ┆     max │
╞═════════════╪════════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡
│       index ┆ 10,000 ┆           0 ┆ 5,000.0 ┆ 2,887.0 ┆       0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │
│ _row_index_ ┆ 10,000 ┆           0 ┆ 5,000.0 ┆ 2,887.0 ┆       0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │
│        year ┆ 10,000 ┆           0 ┆ 2,018.0 ┆   1.416 ┆   2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │
│       month ┆ 10,000 ┆           0 ┆   6.514 ┆   3.432 ┆       1.0 ┆     4.0 ┆     6.0 ┆     9.0 ┆    12.0 │
│        var2 ┆ 10,000 ┆           0 ┆   4.978 ┆   3.155 ┆       0.0 ┆     2.0 ┆     5.0 ┆     8.0 ┆    10.0 │
│        var3 ┆ 10,000 ┆           0 ┆   25.11 ┆   14.75 ┆       0.0 ┆    12.0 ┆    25.0 ┆    38.0 ┆    50.0 │
│        var4 ┆ 10,000 ┆           0 ┆  0.5057 ┆  0.2879 ┆  0.000027 ┆  0.2557 ┆  0.5084 ┆  0.7543 ┆     1.0 │
│ unrelated_1 ┆ 10,000 ┆           0 ┆  0.5024 ┆  0.2884 ┆ 0.0001191 ┆  0.2527 ┆  0.5036 ┆   0.753 ┆     1.0 │
│ unrelated_2 ┆ 10,000 ┆           0 ┆  0.5001 ┆  0.2876 ┆  0.000049 ┆   0.253 ┆  0.4975 ┆  0.7486 ┆  0.9995 │
│ unrelated_3 ┆ 10,000 ┆           0 ┆  0.4992 ┆  0.2888 ┆  0.000129 ┆  0.2512 ┆   0.496 ┆  0.7505 ┆  0.9999 │
│ unrelated_4 ┆ 10,000 ┆           0 ┆  0.5007 ┆  0.2887 ┆ 0.0001329 ┆  0.2501 ┆  0.5005 ┆  0.7518 ┆     1.0 │
│ unrelated_5 ┆ 10,000 ┆           0 ┆  0.4988 ┆   0.289 ┆  0.000071 ┆  0.2496 ┆  0.4968 ┆  0.7498 ┆  0.9999 │
│    repeat_1 ┆ 10,000 ┆           0 ┆  0.5024 ┆  0.2884 ┆ 0.0001191 ┆  0.2527 ┆  0.5036 ┆   0.753 ┆     1.0 │
│    var_gbm3 ┆ 10,000 ┆           0 ┆  -2.266 ┆   9.927 ┆    -55.35 ┆  -1.584 ┆     0.0 ┆     0.0 ┆   26.21 │
│    var_gbm2 ┆ 10,000 ┆           0 ┆  -2.254 ┆   9.969 ┆    -55.35 ┆   -1.69 ┆     0.0 ┆     0.0 ┆   26.21 │
│   var_gbm12 ┆ 10,000 ┆           0 ┆  -2.254 ┆   9.969 ┆    -55.35 ┆   -1.69 ┆     0.0 ┆     0.0 ┆   26.21 │
│ var_gbm2_sq ┆ 10,000 ┆           0 ┆   104.4 ┆   266.8 ┆       0.0 ┆     0.0 ┆     0.0 ┆   71.95 ┆ 3,064.0 │
│        var5 ┆ 10,000 ┆           0 ┆  0.4999 ┆     0.5 ┆       0.0 ┆     0.0 ┆     0.0 ┆     1.0 ┆     1.0 │
│    var_gbm1 ┆ 10,000 ┆           0 ┆   0.521 ┆  0.4996 ┆       0.0 ┆     0.0 ┆     1.0 ┆     1.0 ┆     1.0 │
└─────────────┴────────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘

Look at the imputes | var_gbm1 == 0
┌─────────────┬───────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐
│    Variable ┆     n ┆ n (missing) ┆    mean ┆     std ┆       min ┆     q25 ┆     q50 ┆     q75 ┆     max │
╞═════════════╪═══════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡
│       index ┆ 4,800 ┆           0 ┆ 5,062.0 ┆ 2,905.0 ┆       1.0 ┆ 2,548.0 ┆ 5,081.0 ┆ 7,645.0 ┆ 9,997.0 │
│ _row_index_ ┆ 4,800 ┆           0 ┆ 5,062.0 ┆ 2,905.0 ┆       1.0 ┆ 2,548.0 ┆ 5,081.0 ┆ 7,645.0 ┆ 9,997.0 │
│        year ┆ 4,800 ┆           0 ┆ 2,018.0 ┆   1.421 ┆   2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │
│       month ┆ 4,800 ┆           0 ┆   6.533 ┆   3.431 ┆       1.0 ┆     4.0 ┆     6.0 ┆     9.0 ┆    12.0 │
│        var2 ┆ 4,800 ┆           0 ┆   4.635 ┆   3.235 ┆       0.0 ┆     2.0 ┆     4.0 ┆     7.0 ┆    10.0 │
│        var3 ┆ 4,800 ┆           0 ┆   24.85 ┆   12.92 ┆       0.0 ┆    14.0 ┆    25.0 ┆    36.0 ┆    50.0 │
│        var4 ┆ 4,800 ┆           0 ┆  0.5082 ┆   0.288 ┆  0.000027 ┆  0.2552 ┆  0.5109 ┆  0.7563 ┆     1.0 │
│ unrelated_1 ┆ 4,800 ┆           0 ┆  0.5054 ┆  0.2869 ┆ 0.0001191 ┆  0.2613 ┆  0.5037 ┆   0.756 ┆     1.0 │
│ unrelated_2 ┆ 4,800 ┆           0 ┆  0.4995 ┆   0.287 ┆  0.000079 ┆  0.2587 ┆  0.4906 ┆  0.7499 ┆  0.9995 │
│ unrelated_3 ┆ 4,800 ┆           0 ┆  0.4993 ┆  0.2897 ┆  0.000129 ┆  0.2495 ┆  0.4895 ┆   0.755 ┆  0.9999 │
│ unrelated_4 ┆ 4,800 ┆           0 ┆  0.5016 ┆  0.2901 ┆ 0.0001414 ┆  0.2501 ┆  0.4996 ┆  0.7558 ┆  0.9999 │
│ unrelated_5 ┆ 4,800 ┆           0 ┆   0.504 ┆  0.2906 ┆  0.000071 ┆  0.2534 ┆  0.5049 ┆  0.7537 ┆  0.9998 │
│    repeat_1 ┆ 4,800 ┆           0 ┆  0.5054 ┆  0.2869 ┆ 0.0001191 ┆  0.2613 ┆  0.5037 ┆   0.756 ┆     1.0 │
│    var_gbm3 ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
│    var_gbm2 ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
│   var_gbm12 ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
│ var_gbm2_sq ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
│        var5 ┆ 4,800 ┆           0 ┆  0.7947 ┆   0.404 ┆       0.0 ┆     1.0 ┆     1.0 ┆     1.0 ┆     1.0 │
│    var_gbm1 ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
└─────────────┴───────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘
┌─────────────┬───────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐
│    Variable ┆     n ┆ n (missing) ┆    mean ┆     std ┆       min ┆     q25 ┆     q50 ┆     q75 ┆     max │
╞═════════════╪═══════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡
│       index ┆ 4,800 ┆           0 ┆ 5,085.0 ┆ 2,899.0 ┆       1.0 ┆ 2,584.0 ┆ 5,107.0 ┆ 7,677.0 ┆ 9,997.0 │
│ _row_index_ ┆ 4,800 ┆           0 ┆ 5,085.0 ┆ 2,899.0 ┆       1.0 ┆ 2,584.0 ┆ 5,107.0 ┆ 7,677.0 ┆ 9,997.0 │
│        year ┆ 4,800 ┆           0 ┆ 2,018.0 ┆   1.422 ┆   2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │
│       month ┆ 4,800 ┆           0 ┆   6.546 ┆   3.428 ┆       1.0 ┆     4.0 ┆     6.0 ┆    10.0 ┆    12.0 │
│        var2 ┆ 4,800 ┆           0 ┆   4.647 ┆    3.23 ┆       0.0 ┆     2.0 ┆     4.0 ┆     7.0 ┆    10.0 │
│        var3 ┆ 4,800 ┆           0 ┆   24.87 ┆   12.92 ┆       0.0 ┆    14.0 ┆    25.0 ┆    36.0 ┆    50.0 │
│        var4 ┆ 4,800 ┆           0 ┆  0.5062 ┆  0.2877 ┆  0.000027 ┆   0.254 ┆  0.5098 ┆   0.752 ┆     1.0 │
│ unrelated_1 ┆ 4,800 ┆           0 ┆  0.5031 ┆  0.2849 ┆ 0.0001191 ┆  0.2625 ┆  0.5044 ┆  0.7479 ┆     1.0 │
│ unrelated_2 ┆ 4,800 ┆           0 ┆  0.4999 ┆  0.2853 ┆  0.000079 ┆  0.2608 ┆  0.4928 ┆  0.7469 ┆  0.9995 │
│ unrelated_3 ┆ 4,800 ┆           0 ┆  0.4988 ┆  0.2894 ┆  0.000129 ┆  0.2508 ┆  0.4891 ┆  0.7511 ┆  0.9999 │
│ unrelated_4 ┆ 4,800 ┆           0 ┆  0.4995 ┆  0.2903 ┆ 0.0001414 ┆  0.2466 ┆  0.4982 ┆  0.7551 ┆  0.9999 │
│ unrelated_5 ┆ 4,800 ┆           0 ┆   0.504 ┆   0.291 ┆  0.000071 ┆  0.2523 ┆  0.5013 ┆   0.756 ┆  0.9999 │
│    repeat_1 ┆ 4,800 ┆           0 ┆  0.5031 ┆  0.2849 ┆ 0.0001191 ┆  0.2625 ┆  0.5044 ┆  0.7479 ┆     1.0 │
│    var_gbm3 ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
│    var_gbm2 ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
│   var_gbm12 ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
│ var_gbm2_sq ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
│        var5 ┆ 4,800 ┆           0 ┆  0.7927 ┆  0.4054 ┆       0.0 ┆     1.0 ┆     1.0 ┆     1.0 ┆     1.0 │
│    var_gbm1 ┆ 4,800 ┆           0 ┆     0.0 ┆     0.0 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     0.0 │
└─────────────┴───────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘

Look at the imputes | var_gbm1 == 1
┌─────────────┬───────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐
│    Variable ┆     n ┆ n (missing) ┆    mean ┆     std ┆       min ┆     q25 ┆     q50 ┆     q75 ┆     max │
╞═════════════╪═══════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡
│       index ┆ 5,200 ┆           0 ┆ 4,942.0 ┆ 2,870.0 ┆       0.0 ┆ 2,459.0 ┆ 4,936.0 ┆ 7,381.0 ┆ 9,999.0 │
│ _row_index_ ┆ 5,200 ┆           0 ┆ 4,942.0 ┆ 2,870.0 ┆       0.0 ┆ 2,459.0 ┆ 4,936.0 ┆ 7,381.0 ┆ 9,999.0 │
│        year ┆ 5,200 ┆           0 ┆ 2,018.0 ┆   1.411 ┆   2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │
│       month ┆ 5,200 ┆           0 ┆   6.496 ┆   3.434 ┆       1.0 ┆     4.0 ┆     6.0 ┆     9.0 ┆    12.0 │
│        var2 ┆ 5,200 ┆           0 ┆   5.295 ┆   3.044 ┆       0.0 ┆     3.0 ┆     5.0 ┆     8.0 ┆    10.0 │
│        var3 ┆ 5,200 ┆           0 ┆   25.35 ┆   16.26 ┆       0.0 ┆    10.0 ┆    26.0 ┆    40.0 ┆    50.0 │
│        var4 ┆ 5,200 ┆           0 ┆  0.5033 ┆  0.2877 ┆ 0.0001044 ┆  0.2562 ┆  0.5073 ┆  0.7531 ┆  0.9999 │
│ unrelated_1 ┆ 5,200 ┆           0 ┆  0.4997 ┆  0.2897 ┆ 0.0002485 ┆  0.2486 ┆  0.5036 ┆  0.7506 ┆  0.9998 │
│ unrelated_2 ┆ 5,200 ┆           0 ┆  0.5007 ┆  0.2883 ┆  0.000049 ┆   0.249 ┆  0.5038 ┆  0.7457 ┆  0.9995 │
│ unrelated_3 ┆ 5,200 ┆           0 ┆  0.4991 ┆  0.2879 ┆  0.000319 ┆  0.2531 ┆  0.4997 ┆   0.747 ┆  0.9999 │
│ unrelated_4 ┆ 5,200 ┆           0 ┆  0.4998 ┆  0.2874 ┆ 0.0001329 ┆  0.2508 ┆  0.5011 ┆  0.7492 ┆     1.0 │
│ unrelated_5 ┆ 5,200 ┆           0 ┆  0.4939 ┆  0.2875 ┆ 0.0001807 ┆  0.2439 ┆  0.4883 ┆  0.7467 ┆  0.9999 │
│    repeat_1 ┆ 5,200 ┆           0 ┆  0.4997 ┆  0.2897 ┆ 0.0002485 ┆  0.2486 ┆  0.5036 ┆  0.7506 ┆  0.9998 │
│    var_gbm3 ┆ 5,200 ┆           0 ┆  -4.289 ┆   13.31 ┆    -55.35 ┆  -11.74 ┆  -1.019 ┆   5.153 ┆   26.21 │
│    var_gbm2 ┆ 5,200 ┆           0 ┆   -4.33 ┆   13.48 ┆    -55.35 ┆  -12.01 ┆ -0.7496 ┆   5.186 ┆   26.21 │
│   var_gbm12 ┆ 5,200 ┆           0 ┆   -4.33 ┆   13.48 ┆    -55.35 ┆  -12.01 ┆ -0.7496 ┆   5.186 ┆   26.21 │
│ var_gbm2_sq ┆ 5,200 ┆           0 ┆   200.3 ┆   343.7 ┆       0.0 ┆    9.33 ┆   63.08 ┆   213.2 ┆ 3,064.0 │
│        var5 ┆ 5,200 ┆           0 ┆  0.2281 ┆  0.4197 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     1.0 │
│    var_gbm1 ┆ 5,200 ┆           0 ┆     1.0 ┆     0.0 ┆       1.0 ┆     1.0 ┆     1.0 ┆     1.0 ┆     1.0 │
└─────────────┴───────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘
┌─────────────┬───────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐
│    Variable ┆     n ┆ n (missing) ┆    mean ┆     std ┆       min ┆     q25 ┆     q50 ┆     q75 ┆     max │
╞═════════════╪═══════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡
│       index ┆ 5,200 ┆           0 ┆ 4,921.0 ┆ 2,874.0 ┆       0.0 ┆ 2,426.0 ┆ 4,884.0 ┆ 7,362.0 ┆ 9,999.0 │
│ _row_index_ ┆ 5,200 ┆           0 ┆ 4,921.0 ┆ 2,874.0 ┆       0.0 ┆ 2,426.0 ┆ 4,884.0 ┆ 7,362.0 ┆ 9,999.0 │
│        year ┆ 5,200 ┆           0 ┆ 2,018.0 ┆    1.41 ┆   2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │
│       month ┆ 5,200 ┆           0 ┆   6.484 ┆   3.436 ┆       1.0 ┆     4.0 ┆     6.0 ┆     9.0 ┆    12.0 │
│        var2 ┆ 5,200 ┆           0 ┆   5.282 ┆   3.052 ┆       0.0 ┆     3.0 ┆     5.0 ┆     8.0 ┆    10.0 │
│        var3 ┆ 5,200 ┆           0 ┆   25.33 ┆   16.26 ┆       0.0 ┆    10.0 ┆    25.0 ┆    40.0 ┆    50.0 │
│        var4 ┆ 5,200 ┆           0 ┆  0.5052 ┆   0.288 ┆ 0.0001044 ┆   0.257 ┆  0.5073 ┆  0.7563 ┆  0.9999 │
│ unrelated_1 ┆ 5,200 ┆           0 ┆  0.5018 ┆  0.2916 ┆ 0.0002485 ┆  0.2478 ┆  0.5008 ┆  0.7572 ┆  0.9998 │
│ unrelated_2 ┆ 5,200 ┆           0 ┆  0.5003 ┆  0.2898 ┆  0.000049 ┆   0.247 ┆  0.5028 ┆  0.7495 ┆  0.9995 │
│ unrelated_3 ┆ 5,200 ┆           0 ┆  0.4995 ┆  0.2882 ┆  0.000319 ┆  0.2514 ┆   0.499 ┆  0.7497 ┆  0.9999 │
│ unrelated_4 ┆ 5,200 ┆           0 ┆  0.5017 ┆  0.2872 ┆ 0.0001329 ┆  0.2551 ┆  0.5036 ┆  0.7501 ┆     1.0 │
│ unrelated_5 ┆ 5,200 ┆           0 ┆  0.4939 ┆  0.2871 ┆ 0.0001807 ┆  0.2452 ┆  0.4903 ┆  0.7452 ┆  0.9994 │
│    repeat_1 ┆ 5,200 ┆           0 ┆  0.5018 ┆  0.2916 ┆ 0.0002485 ┆  0.2478 ┆  0.5008 ┆  0.7572 ┆  0.9998 │
│    var_gbm3 ┆ 5,200 ┆           0 ┆  -4.349 ┆   13.42 ┆    -55.35 ┆  -11.78 ┆ -0.8137 ┆   5.182 ┆   26.21 │
│    var_gbm2 ┆ 5,200 ┆           0 ┆  -4.326 ┆   13.48 ┆    -55.35 ┆  -11.78 ┆ -0.8498 ┆   5.395 ┆   26.21 │
│   var_gbm12 ┆ 5,200 ┆           0 ┆  -4.326 ┆   13.48 ┆    -55.35 ┆  -11.78 ┆ -0.8498 ┆   5.395 ┆   26.21 │
│ var_gbm2_sq ┆ 5,200 ┆           0 ┆   200.5 ┆   342.6 ┆       0.0 ┆   10.42 ┆   63.37 ┆   209.5 ┆ 3,064.0 │
│        var5 ┆ 5,200 ┆           0 ┆  0.2307 ┆  0.4213 ┆       0.0 ┆     0.0 ┆     0.0 ┆     0.0 ┆     1.0 │
│    var_gbm1 ┆ 5,200 ┆           0 ┆     1.0 ┆     0.0 ┆       1.0 ┆     1.0 ┆     1.0 ┆     1.0 ┆     1.0 │
└─────────────┴───────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘