In [1]:
import sys
import os
from pathlib import Path
import narwhals as nw
import polars as pl
import polars.selectors as cs
from survey_kit.utilities.random import RandomData
from survey_kit.utilities.dataframe import summary
from survey_kit.imputation.variable import Variable
from survey_kit.imputation.parameters import Parameters
from survey_kit.imputation.srmi import SRMI
from survey_kit.imputation.selection import Selection
import survey_kit.imputation.utilities.lightgbm_wrapper as rep_lgbm
from survey_kit.imputation.utilities.lightgbm_wrapper import Tuner, Objective
from survey_kit.imputation.utilities.tuning import HyperparameterSpace, IntRange, FloatRange
from survey_kit import logger, config
from survey_kit.utilities.dataframe import summary, columns_from_list
In [2]:
# Draw some random data
n_rows = 10_000
impute_share = 0.25
df = (
RandomData(n_rows=n_rows, seed=32565437)
.index("index")
.integer("year", 2016, 2020)
.integer("month", 1, 12)
.integer("var2", 0, 10)
.integer("var3", 0, 50)
.float("var4", 0, 1)
.integer("var5", 0, 1)
.float("unrelated_1", 0, 1)
.float("unrelated_2", 0, 1)
.float("unrelated_3", 0, 1)
.float("unrelated_4", 0, 1)
.float("unrelated_5", 0, 1)
.np_distribution("epsilon_gbm1", "normal", scale=5)
.np_distribution("epsilon_gbm2", "normal", scale=5)
.np_distribution("epsilon_gbm3", "normal", scale=5)
.float("missing_gbm1", 0, 1)
.float("missing_gbm2", 0, 1)
.float("missing_gbm3", 0, 1)
.to_df()
)
# Convenience references to them for creating dependent variables
c_var2 = pl.col("var2")
c_var3 = pl.col("var3")
c_var4 = pl.col("var4")
c_var5 = pl.col("var5")
c_e_gbm1 = pl.col("epsilon_gbm1")
c_e_gbm2 = pl.col("epsilon_gbm2")
# Convenience references to them for creating dependent variables
c_var2 = pl.col("var2")
c_var3 = pl.col("var3")
c_var4 = pl.col("var4")
c_var5 = pl.col("var5")
logger.info("var_gbm1 is binary and conditional on other variables")
c_gbm1 = ((c_var2 * 2 - c_var3 * 3 * c_var5 + c_e_gbm1) > 0).alias("var_gbm1")
logger.info("var_gbm2 is != 0 only if var_gbm1 == True")
c_gbm2 = (
pl.when(pl.col("var_gbm1"))
.then((c_var2 * 1.5 - c_var3 * 1 * c_var4 + c_e_gbm2))
.otherwise(pl.lit(0))
.alias("var_gbm2")
)
c_gbm3 = (
pl.when(pl.col("var_gbm1"))
.then((c_var2 * 1.5 - c_var3 * 1 * c_var4 + c_e_gbm2))
.otherwise(pl.lit(0))
.alias("var_gbm3")
)
# Create a bunch of variables that are functions of the variables created above
df = (
df.with_columns(c_gbm1)
.with_columns(c_gbm2, c_gbm3)
.drop(columns_from_list(df=df, columns="epsilon*"))
.with_row_index(name="_row_index_")
)
df_original = df
# Set variables to missing according to the uniform random variables missing_
clear_missing = []
for prefixi in ["gbm"]:
for i in range(1, 4):
vari = f"var_{prefixi}{i}"
missingi = f"missing_{prefixi}{i}"
clear_missing.append(
pl.when(pl.col(missingi) < impute_share)
.then(pl.lit(None))
.otherwise(pl.col(vari))
.alias(vari)
)
df = df.with_columns(clear_missing).drop(cs.starts_with("missing_"))
# Make a fully collinear var for testing
df = df.with_columns(pl.col("unrelated_1").alias("repeat_1"))
summary(df)
var_gbm1 is binary and conditional on other variables
var_gbm2 is != 0 only if var_gbm1 == True
┌─────────────┬────────┬─────────────┬────────────┬─────────────┬────────────┬───────────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ max │ ╞═════════════╪════════╪═════════════╪════════════╪═════════════╪════════════╪═══════════╡ │ _row_index_ ┆ 10,000 ┆ 0 ┆ 4,999.5 ┆ 2,886.89568 ┆ 0.0 ┆ 9,999.0 │ │ index ┆ 10,000 ┆ 0 ┆ 4,999.5 ┆ 2,886.89568 ┆ 0.0 ┆ 9,999.0 │ │ year ┆ 10,000 ┆ 0 ┆ 2,017.9851 ┆ 1.415937 ┆ 2,016.0 ┆ 2,020.0 │ │ month ┆ 10,000 ┆ 0 ┆ 6.5137 ┆ 3.432141 ┆ 1.0 ┆ 12.0 │ │ var2 ┆ 10,000 ┆ 0 ┆ 4.9782 ┆ 3.154508 ┆ 0.0 ┆ 10.0 │ │ var3 ┆ 10,000 ┆ 0 ┆ 25.1084 ┆ 14.752302 ┆ 0.0 ┆ 50.0 │ │ var4 ┆ 10,000 ┆ 0 ┆ 0.505666 ┆ 0.287861 ┆ 0.000027 ┆ 0.999997 │ │ unrelated_1 ┆ 10,000 ┆ 0 ┆ 0.502449 ┆ 0.288359 ┆ 0.000119 ┆ 0.999997 │ │ unrelated_2 ┆ 10,000 ┆ 0 ┆ 0.500105 ┆ 0.287638 ┆ 0.000049 ┆ 0.999539 │ │ unrelated_3 ┆ 10,000 ┆ 0 ┆ 0.499175 ┆ 0.28876 ┆ 0.000129 ┆ 0.99994 │ │ unrelated_4 ┆ 10,000 ┆ 0 ┆ 0.500655 ┆ 0.288698 ┆ 0.000133 ┆ 0.999972 │ │ unrelated_5 ┆ 10,000 ┆ 0 ┆ 0.49876 ┆ 0.288979 ┆ 0.000071 ┆ 0.999867 │ │ var_gbm2 ┆ 10,000 ┆ 2,464 ┆ -2.393041 ┆ 10.384672 ┆ -55.354108 ┆ 26.213084 │ │ var_gbm3 ┆ 10,000 ┆ 2,596 ┆ -2.500425 ┆ 10.365353 ┆ -55.354108 ┆ 26.213084 │ │ repeat_1 ┆ 10,000 ┆ 0 ┆ 0.502449 ┆ 0.288359 ┆ 0.000119 ┆ 0.999997 │ │ var5 ┆ 10,000 ┆ 0 ┆ 0.4999 ┆ 0.500025 ┆ 0.0 ┆ 1.0 │ │ var_gbm1 ┆ 10,000 ┆ 2,483 ┆ 0.526008 ┆ 0.499356 ┆ 0.0 ┆ 1.0 │ └─────────────┴────────┴─────────────┴────────────┴─────────────┴────────────┴───────────┘
Out[2]:
naive plan: (run LazyFrame.explain(optimized=True) to see the optimized plan)
SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] UNION PLAN 0: WITH_COLUMNS: [col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["_row_index_".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("_row_index__mean").alias("mean"), col("_row_index__std").alias("std"), col("_row_index__rawn_missing").alias("n (missing)"), col("_row_index__rawn").alias("n"), col("_row_index__min").alias("min"), col("_row_index__max").alias("max")] SELECT [col("___index___"), col("_row_index__mean"), col("_row_index__std"), col("_row_index__rawn_missing"), col("_row_index__rawn"), col("_row_index__min"), col("_row_index__max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 1: WITH_COLUMNS: [col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["index".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("index_mean").alias("mean"), col("index_std").alias("std"), col("index_rawn_missing").alias("n (missing)"), col("index_rawn").alias("n"), col("index_min").alias("min"), col("index_max").alias("max")] SELECT [col("___index___"), col("index_mean"), col("index_std"), col("index_rawn_missing"), col("index_rawn"), col("index_min"), col("index_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 2: WITH_COLUMNS: [col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["year".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("year_mean").alias("mean"), col("year_std").alias("std"), col("year_rawn_missing").alias("n (missing)"), col("year_rawn").alias("n"), col("year_min").alias("min"), col("year_max").alias("max")] SELECT [col("___index___"), col("year_mean"), col("year_std"), col("year_rawn_missing"), col("year_rawn"), col("year_min"), col("year_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 3: WITH_COLUMNS: [col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["month".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("month_mean").alias("mean"), col("month_std").alias("std"), col("month_rawn_missing").alias("n (missing)"), col("month_rawn").alias("n"), col("month_min").alias("min"), col("month_max").alias("max")] SELECT [col("___index___"), col("month_mean"), col("month_std"), col("month_rawn_missing"), col("month_rawn"), col("month_min"), col("month_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 4: WITH_COLUMNS: [col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["var2".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("var2_mean").alias("mean"), col("var2_std").alias("std"), col("var2_rawn_missing").alias("n (missing)"), col("var2_rawn").alias("n"), col("var2_min").alias("min"), col("var2_max").alias("max")] SELECT [col("___index___"), col("var2_mean"), col("var2_std"), col("var2_rawn_missing"), col("var2_rawn"), col("var2_min"), col("var2_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 5: WITH_COLUMNS: [col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["var3".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("var3_mean").alias("mean"), col("var3_std").alias("std"), col("var3_rawn_missing").alias("n (missing)"), col("var3_rawn").alias("n"), col("var3_min").alias("min"), col("var3_max").alias("max")] SELECT [col("___index___"), col("var3_mean"), col("var3_std"), col("var3_rawn_missing"), col("var3_rawn"), col("var3_min"), col("var3_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 6: WITH_COLUMNS: [col("n (missing)").cast(Int16)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["var4".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("var4_mean").alias("mean"), col("var4_std").alias("std"), col("var4_rawn_missing").alias("n (missing)"), col("var4_rawn").alias("n"), col("var4_min").alias("min"), col("var4_max").alias("max")] SELECT [col("___index___"), col("var4_mean"), col("var4_std"), col("var4_rawn_missing"), col("var4_rawn"), col("var4_min"), col("var4_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 7: WITH_COLUMNS: [col("n (missing)").cast(Int16)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["unrelated_1".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("unrelated_1_mean").alias("mean"), col("unrelated_1_std").alias("std"), col("unrelated_1_rawn_missing").alias("n (missing)"), col("unrelated_1_rawn").alias("n"), col("unrelated_1_min").alias("min"), col("unrelated_1_max").alias("max")] SELECT [col("___index___"), col("unrelated_1_mean"), col("unrelated_1_std"), col("unrelated_1_rawn_missing"), col("unrelated_1_rawn"), col("unrelated_1_min"), col("unrelated_1_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 8: WITH_COLUMNS: [col("n (missing)").cast(Int16)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["unrelated_2".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("unrelated_2_mean").alias("mean"), col("unrelated_2_std").alias("std"), col("unrelated_2_rawn_missing").alias("n (missing)"), col("unrelated_2_rawn").alias("n"), col("unrelated_2_min").alias("min"), col("unrelated_2_max").alias("max")] SELECT [col("___index___"), col("unrelated_2_mean"), col("unrelated_2_std"), col("unrelated_2_rawn_missing"), col("unrelated_2_rawn"), col("unrelated_2_min"), col("unrelated_2_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 9: WITH_COLUMNS: [col("n (missing)").cast(Int16)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["unrelated_3".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("unrelated_3_mean").alias("mean"), col("unrelated_3_std").alias("std"), col("unrelated_3_rawn_missing").alias("n (missing)"), col("unrelated_3_rawn").alias("n"), col("unrelated_3_min").alias("min"), col("unrelated_3_max").alias("max")] SELECT [col("___index___"), col("unrelated_3_mean"), col("unrelated_3_std"), col("unrelated_3_rawn_missing"), col("unrelated_3_rawn"), col("unrelated_3_min"), col("unrelated_3_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 10: WITH_COLUMNS: [col("n (missing)").cast(Int16)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["unrelated_4".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("unrelated_4_mean").alias("mean"), col("unrelated_4_std").alias("std"), col("unrelated_4_rawn_missing").alias("n (missing)"), col("unrelated_4_rawn").alias("n"), col("unrelated_4_min").alias("min"), col("unrelated_4_max").alias("max")] SELECT [col("___index___"), col("unrelated_4_mean"), col("unrelated_4_std"), col("unrelated_4_rawn_missing"), col("unrelated_4_rawn"), col("unrelated_4_min"), col("unrelated_4_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 11: WITH_COLUMNS: [col("n (missing)").cast(Int16)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["unrelated_5".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("unrelated_5_mean").alias("mean"), col("unrelated_5_std").alias("std"), col("unrelated_5_rawn_missing").alias("n (missing)"), col("unrelated_5_rawn").alias("n"), col("unrelated_5_min").alias("min"), col("unrelated_5_max").alias("max")] SELECT [col("___index___"), col("unrelated_5_mean"), col("unrelated_5_std"), col("unrelated_5_rawn_missing"), col("unrelated_5_rawn"), col("unrelated_5_min"), col("unrelated_5_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 12: SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["var_gbm2".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("var_gbm2_mean").alias("mean"), col("var_gbm2_std").alias("std"), col("var_gbm2_rawn_missing").alias("n (missing)"), col("var_gbm2_rawn").alias("n"), col("var_gbm2_min").alias("min"), col("var_gbm2_max").alias("max")] SELECT [col("___index___"), col("var_gbm2_mean"), col("var_gbm2_std"), col("var_gbm2_rawn_missing"), col("var_gbm2_rawn"), col("var_gbm2_min"), col("var_gbm2_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 13: SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["var_gbm3".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("var_gbm3_mean").alias("mean"), col("var_gbm3_std").alias("std"), col("var_gbm3_rawn_missing").alias("n (missing)"), col("var_gbm3_rawn").alias("n"), col("var_gbm3_min").alias("min"), col("var_gbm3_max").alias("max")] SELECT [col("___index___"), col("var_gbm3_mean"), col("var_gbm3_std"), col("var_gbm3_rawn_missing"), col("var_gbm3_rawn"), col("var_gbm3_min"), col("var_gbm3_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 14: WITH_COLUMNS: [col("n (missing)").cast(Int16)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["repeat_1".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("repeat_1_mean").alias("mean"), col("repeat_1_std").alias("std"), col("repeat_1_rawn_missing").alias("n (missing)"), col("repeat_1_rawn").alias("n"), col("repeat_1_min").alias("min"), col("repeat_1_max").alias("max")] SELECT [col("___index___"), col("repeat_1_mean"), col("repeat_1_std"), col("repeat_1_rawn_missing"), col("repeat_1_rawn"), col("repeat_1_min"), col("repeat_1_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 15: WITH_COLUMNS: [col("n (missing)").cast(Int16), col("min").strict_cast(Float64), col("max").strict_cast(Float64)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["var5".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("var5_mean").alias("mean"), col("var5_std").alias("std"), col("var5_rawn_missing").alias("n (missing)"), col("var5_rawn").alias("n"), col("var5_min").alias("min"), col("var5_max").alias("max")] SELECT [col("___index___"), col("var5_mean"), col("var5_std"), col("var5_rawn_missing"), col("var5_rawn"), col("var5_min"), col("var5_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS PLAN 16: WITH_COLUMNS: [col("min").strict_cast(Float64), col("max").strict_cast(Float64)] SELECT [col("Variable"), col("n"), col("n (missing)"), col("mean"), col("std"), col("min"), col("max")] WITH_COLUMNS: ["var_gbm1".alias("Variable")] SELECT [col("mean"), col("std"), col("n (missing)"), col("n"), col("min"), col("max")] SELECT [col("___index___"), col("var_gbm1_mean").alias("mean"), col("var_gbm1_std").alias("std"), col("var_gbm1_rawn_missing").alias("n (missing)"), col("var_gbm1_rawn").alias("n"), col("var_gbm1_min").alias("min"), col("var_gbm1_max").alias("max")] SELECT [col("___index___"), col("var_gbm1_mean"), col("var_gbm1_std"), col("var_gbm1_rawn_missing"), col("var_gbm1_rawn"), col("var_gbm1_min"), col("var_gbm1_max")] DF ["___index___", "_row_index__rawn", "_row_index__mean", "_row_index__std", ...]; PROJECT */103 COLUMNS END UNION
In [3]:
logger.info("Define some dummy functions to run after imputation of 2")
# Test a simple pre-post function
# These would get run gets run in each iteration (in each implicate)
# before (preFunctions) or after (postFunctions) this variable is imputed
# Notes for these functions:
# 1) No type hints on imported package types (will throw an error)
# i.e. no df:pl.DataFrame or -> pl.DataFrame
# 2) Must be completely self-contained (i.e. all imports within the function)
# This has to do with how it gets saved and loaded in async calls
# 3) Effectively, you have to assume it'll be called
# in an environment with no imports before it
def square_var(df, var_to_square: str, name: str):
import narwhals as nw
return (
nw.from_native(df)
.with_columns((nw.col(var_to_square) ** 2).alias(name))
.to_native()
)
def recalculate_interaction(df, var1: str, var2: str, name: str):
import narwhals as nw
return (
nw.from_native(df)
.with_columns((nw.col(var1) * nw.col(var2)).alias(name))
.to_native()
)
Define some dummy functions to run after imputation of 2
In [4]:
logger.info("Set up hyperparameter tuning")
tuner = Tuner(
space=HyperparameterSpace(
num_leaves=IntRange(2, 256),
max_depth=IntRange(2, 256),
min_data_in_leaf=IntRange(10, 250),
num_iterations=IntRange(25, 200),
bagging_fraction=FloatRange(0.5, 1.0),
bagging_freq=IntRange(1, 5),
),
objective=Objective.mae,
n_trials=50,
path_save_dir=f"{config.data_root}/tuner_outputs",
overwrite=True,
)
vars_impute = []
Set up hyperparameter tuning
In [5]:
logger.info("Impute the boolean variable (var_gbm1)")
logger.info(" to the default setup for predicted mean matching")
logger.info(" using lightgbm")
logger.info(" (you can pass a formula, but you don't need to)")
logger.info("First, set up the lightgbm parameters")
logger.info(" This says, do hyperparameter tuning first (tune)")
logger.info(" Redo it at each run (the tuner's own overwrite=True)")
logger.info(
" And sets the lightgbm parameter defaults (parameters) that the tuning can overwrite"
)
parameters_lgbm1 = Parameters.LightGBM(
tune=True,
tuner=tuner,
parameters={
"objective": "binary",
"num_leaves": 32,
"min_data_in_leaf": 20,
"num_iterations": 100,
"test_size": 0.2,
"boosting": "gbdt",
"categorical_feature": ["var5"],
"verbose": -1, # ,
},
error=Parameters.ErrorDraw.pmm,
)
logger.info("Actually define the variable and the model")
v_gbm1 = Variable(
impute_var="var_gbm1",
model=["var_*", "var4", "var3", "var5", "unrelated_*", "repeat_*"],
modeltype=Variable.ModelType.LightGBM,
parameters=parameters_lgbm1,
)
logger.info("Add the variable to the list to be imputed")
vars_impute.append(v_gbm1)
logger.info("Impute the continuous variable (var_gbm2) ")
logger.info(" conditional on var_gbm1, using narwhals (nw.col('var_gbm1'))")
logger.info(" as well as a post-processing edit to set var_gbm2=0 when var_gbm1==0")
logger.info(" and some other random post-processing")
logger.info("Different parameters for the continuous variable")
parameters_lgbm2 = Parameters.LightGBM(
tune=True,
tuner=tuner,
parameters={
"objective": "regression",
"num_leaves": 32,
"min_data_in_leaf": 20,
"num_iterations": 100,
"test_size": 0.2,
"boosting": "gbdt",
"categorical_feature": ["var5"],
"verbose": -1, # ,
},
error=Parameters.ErrorDraw.pmm,
)
v_gbm2 = Variable(
impute_var="var_gbm2",
sample=Variable.Sample(
Where=nw.col("var_gbm1"),
# Needed in case var_gbm1 changes between iterations
Where_predict=(nw.col("var_gbm2") != 0),
),
model=["var_*", "var4", "var3", "var5", "unrelated_*", "repeat_*"],
modeltype=Variable.ModelType.LightGBM,
parameters=parameters_lgbm2,
transforms=Variable.Transforms(
post=[
(
nw.when(nw.col("var_gbm1"))
.then(nw.col("var_gbm2"))
.otherwise(nw.lit(0))
.alias("var_gbm2")
),
Variable.PrePost.Function(
recalculate_interaction,
parameters=dict(var1="var_gbm1", var2="var_gbm2", name="var_gbm12"),
),
Variable.PrePost.Function(
square_var,
parameters=dict(var_to_square="var_gbm2", name="var_gbm2_sq"),
),
]
),
)
vars_impute.append(v_gbm2)
logger.info("Now do one with the quantile-regression lightgbm")
logger.info(" To do this, pass quantiles and set objective='quantile'")
parameters_lgbm3 = Parameters.LightGBM(
tune=True,
tuner=tuner,
quantiles=[0.25, 0.5, 0.75],
parameters={
"objective": "quantile",
"num_leaves": 32,
"min_data_in_leaf": 20,
"num_iterations": 100,
"test_size": 0.2,
"boosting": "gbdt",
"categorical_feature": ["var5"],
"verbose": -1, # ,
},
error=Parameters.ErrorDraw.pmm,
)
v_gbm3 = Variable(
impute_var="var_gbm3",
sample=Variable.Sample(
Where=nw.col("var_gbm1"),
# Needed in case var_gbm1 changes between iterations
Where_predict=(nw.col("var_gbm3") != 0),
),
model=["var_*", "var4", "var3", "var5", "unrelated_*", "repeat_*"],
modeltype=Variable.ModelType.LightGBM,
parameters=parameters_lgbm3,
transforms=Variable.Transforms(
post=[
(
nw.when(nw.col("var_gbm1"))
.then(nw.col("var_gbm3"))
.otherwise(nw.lit(0))
.alias("var_gbm3")
)
]
),
)
vars_impute.append(v_gbm3)
Impute the boolean variable (var_gbm1)
to the default setup for predicted mean matching
using lightgbm
(you can pass a formula, but you don't need to)
First, set up the lightgbm parameters
This says, do hyperparameter tuning first (tune)
Redo it at each run (the tuner's own overwrite=True)
And sets the lightgbm parameter defaults (parameters) that the tuning can overwrite
Actually define the variable and the model
Add the variable to the list to be imputed
Impute the continuous variable (var_gbm2)
conditional on var_gbm1, using narwhals (nw.col('var_gbm1'))
as well as a post-processing edit to set var_gbm2=0 when var_gbm1==0
and some other random post-processing
Different parameters for the continuous variable
Now do one with the quantile-regression lightgbm
To do this, pass quantiles and set objective='quantile'
In [6]:
logger.info("Set up the imputation")
srmi = SRMI(
df=df,
variables=vars_impute,
index=["index"],
replication=SRMI.Replication(n_implicates=2, n_iterations=2),
parallel=SRMI.Parallel(enabled=False),
bootstrap=SRMI.Bootstrap(enabled=True),
defaults=SRMI.Defaults(modeltype=Variable.ModelType.pmm),
storage=SRMI.Storage(
path_model=f"{config.path_temp_files}/py_srmi_test_gbm", force_start=True
),
)
Set up the imputation
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi
In [7]:
logger.info("Run it")
srmi.run()
logger.info("It's automatically saved and can be loaded with (see path_model above):")
logger.info("path_model = f'{config.path_temp_files}/py_srmi_test_gbm'")
logger.info("srmi = SRMI.load(path_model)")
Run it
Variable selection before SRMI run, if necessary
var_gbm1: Method.No
var_gbm2: Method.No
var_gbm3: Method.No
Hyperparameter tuning before SRMI run, if necessary
Tuner: 50 trials finished
Tuner: best value = 0.02464
Tuner: best params = {'num_leaves': 235, 'max_depth': 172, 'min_data_in_leaf': 28, 'num_iterations': 177, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1}
TUNING COMPLETE
Tuner: 50 trials finished
Tuner: best value = 1.79798
Tuner: best params = {'num_leaves': 192, 'max_depth': 62, 'min_data_in_leaf': 33, 'num_iterations': 86, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1}
TUNING COMPLETE
Tuner: 50 trials finished
Tuner: best value = 2.45658
Tuner: best params = {'num_leaves': 185, 'max_depth': 31, 'min_data_in_leaf': 32, 'num_iterations': 181, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5}
TUNING COMPLETE
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/1.srmi.implicate
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/2.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'binary', 'num_leaves': 235, 'min_data_in_leaf': 28, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 172, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1, 'seed': 1205842559}
Iterations: 177
Model: var_gbm1=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ unrelated_2 ┆ 0.1488 ┆ 0.1535 ┆ 0 ┆ 0.4995 ┆ 0 ┆ 0.5019 │ │ unrelated_4 ┆ 0.1457 ┆ 0.1515 ┆ 0 ┆ 0.5007 ┆ 0 ┆ 0.5004 │ │ unrelated_3 ┆ 0.1455 ┆ 0.1494 ┆ 0 ┆ 0.5 ┆ 0 ┆ 0.4968 │ │ unrelated_1 ┆ 0.1452 ┆ 0.1509 ┆ 0 ┆ 0.5028 ┆ 0 ┆ 0.5015 │ │ var4 ┆ 0.1447 ┆ 0.1472 ┆ 0 ┆ 0.5041 ┆ 0 ┆ 0.5103 │ │ unrelated_5 ┆ 0.1379 ┆ 0.1452 ┆ 0 ┆ 0.4966 ┆ 0 ┆ 0.5053 │ │ var3 ┆ 0.1321 ┆ 0.1022 ┆ 0 ┆ 25.13 ┆ 0 ┆ 25.05 │ └─────────────┴────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬──────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪══════════╪══════════════╪════════════════╡ │ count ┆ 7517.0 ┆ 7517.0 ┆ 2483.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.526008 ┆ 0.5278 ┆ 0.53199 │ │ std ┆ null ┆ 0.359735 ┆ 0.28087 │ │ min ┆ 0.0 ┆ 0.004719 ┆ 0.006007 │ │ 25% ┆ null ┆ 0.146288 ┆ 0.292873 │ │ 50% ┆ null ┆ 0.56745 ┆ 0.531791 │ │ 75% ┆ null ┆ 0.898777 ┆ 0.778923 │ │ max ┆ 1.0 ┆ 0.999449 ┆ 0.999021 │ └────────────┴──────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬──────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪══════════╪══════════════╪════════════════╡ │ count ┆ 7517.0 ┆ 7517.0 ┆ 2483.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.526008 ┆ 0.5278 ┆ 0.53199 │ │ std ┆ null ┆ 0.359735 ┆ 0.28087 │ │ min ┆ 0.0 ┆ 0.004719 ┆ 0.006007 │ │ 25% ┆ null ┆ 0.146288 ┆ 0.292873 │ │ 50% ┆ null ┆ 0.56745 ┆ 0.531791 │ │ 75% ┆ null ┆ 0.898777 ┆ 0.778923 │ │ max ┆ 1.0 ┆ 0.999449 ┆ 0.999021 │ └────────────┴──────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['var_gbm1'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['var_gbm1']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 9145 ┆ 5 │ │ 1850 ┆ 4 │ │ 5596 ┆ 4 │ │ 6997 ┆ 4 │ │ 8017 ┆ 4 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm1']
Where: None
Where (impute): col(___imp_missing_var_gbm1_1)
┌──────────┬─────────┬───────┬──────────────┬────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪═══════╪══════════════╪════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm1 ┆ ┆ 10000 ┆ 10000 ┆ 0.5241 ┆ 0.4994 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ │ var_gbm1 ┆ 0 ┆ 7517 ┆ 7517 ┆ 0.526 ┆ 0.4994 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ │ var_gbm1 ┆ 1 ┆ 2483 ┆ 2483 ┆ 0.5183 ┆ 0.4998 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ └──────────┴─────────┴───────┴──────────────┴────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 192, 'min_data_in_leaf': 33, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 62, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1, 'seed': 3440548605}
Iterations: 86
Model: var_gbm2=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var4 ┆ 0.3966 ┆ 0.1593 ┆ 0 ┆ 0.5034 ┆ 0 ┆ 0.5018 │ │ var3 ┆ 0.3914 ┆ 0.1266 ┆ 0 ┆ 25.47 ┆ 0 ┆ 25.16 │ │ unrelated_2 ┆ 0.04912 ┆ 0.153 ┆ 0 ┆ 0.4978 ┆ 0 ┆ 0.5057 │ │ unrelated_1 ┆ 0.04313 ┆ 0.1452 ┆ 0 ┆ 0.5036 ┆ 0 ┆ 0.4943 │ │ unrelated_5 ┆ 0.04091 ┆ 0.1409 ┆ 0 ┆ 0.4976 ┆ 0 ┆ 0.4892 │ │ unrelated_3 ┆ 0.04064 ┆ 0.1354 ┆ 0 ┆ 0.4996 ┆ 0 ┆ 0.4972 │ │ unrelated_4 ┆ 0.03821 ┆ 0.1396 ┆ 0 ┆ 0.4984 ┆ 0 ┆ 0.4992 │ └─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬────────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm2 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════╪══════════════╪════════════════╡ │ count ┆ 3500.0 ┆ 3500.0 ┆ 1302.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.680075 ┆ -4.575018 ┆ -4.301531 │ │ std ┆ 14.165682 ┆ 12.794151 ┆ 12.310863 │ │ min ┆ -55.354108 ┆ -44.238519 ┆ -44.179112 │ │ 25% ┆ -13.396664 ┆ -12.795744 ┆ -11.85476 │ │ 50% ┆ -2.262626 ┆ -0.800443 ┆ -0.748676 │ │ 75% ┆ 6.134272 ┆ 5.652468 ┆ 5.351061 │ │ max ┆ 26.213084 ┆ 16.631379 ┆ 15.886879 │ └────────────┴────────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬────────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm2 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════╪══════════════╪════════════════╡ │ count ┆ 3500.0 ┆ 3500.0 ┆ 1302.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.680075 ┆ -4.575018 ┆ -4.301531 │ │ std ┆ 14.165682 ┆ 12.794151 ┆ 12.310863 │ │ min ┆ -55.354108 ┆ -44.238519 ┆ -44.179112 │ │ 25% ┆ -13.396664 ┆ -12.795744 ┆ -11.85476 │ │ 50% ┆ -2.262626 ┆ -0.800443 ┆ -0.748676 │ │ 75% ┆ 6.134272 ┆ 5.652468 ┆ 5.351061 │ │ max ┆ 26.213084 ┆ 16.631379 ┆ 15.886879 │ └────────────┴────────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['var_gbm2'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['var_gbm2']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 3956 ┆ 4 │ │ 861 ┆ 3 │ │ 1026 ┆ 3 │ │ 1804 ┆ 3 │ │ 2104 ┆ 3 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm2']
Where: col(var_gbm1)
Where (impute): col(___imp_missing_var_gbm2_2)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm2 ┆ ┆ 4802 ┆ 4802 ┆ -4.667 ┆ 14.05 ┆ -4.667 ┆ 14.05 ┆ -25.36 ┆ -13.21 ┆ -2.284 ┆ 5.887 ┆ 11.66 ┆ -55.35 ┆ 26.21 │ │ var_gbm2 ┆ 0 ┆ 3500 ┆ 3500 ┆ -4.68 ┆ 14.17 ┆ -4.68 ┆ 14.17 ┆ -25.61 ┆ -13.4 ┆ -2.284 ┆ 6.134 ┆ 11.65 ┆ -55.35 ┆ 26.21 │ │ var_gbm2 ┆ 1 ┆ 1302 ┆ 1302 ┆ -4.633 ┆ 13.75 ┆ -4.633 ┆ 13.75 ┆ -24.17 ┆ -12.66 ┆ -2.284 ┆ 5.119 ┆ 11.67 ┆ -55.35 ┆ 26.21 │ └──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm2), lit(value=0, dtype=None)).alias(name=var_gbm2)
Calling recalculate_interaction
Calling square_var
Imputation using LightGBM
Running LightGBM for q=0.25
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.25, 'seed': 1341191336}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.25 prediction: 0.953
Running LightGBM for q=0.5
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.5, 'seed': 819323121}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.5 prediction: 0.952
Running LightGBM for q=0.75
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.75, 'seed': 739646208}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.75 prediction: 0.953
┌──────────┬─────────┬───────┬─────────────┬────────┬───────┬────────┬─────────┬───────┐ │ Variable ┆ Sample ┆ n ┆ n (missing) ┆ mean ┆ std ┆ q25 ┆ q50 ┆ q75 │ ╞══════════╪═════════╪═══════╪═════════════╪════════╪═══════╪════════╪═════════╪═══════╡ │ p0.25 ┆ Model ┆ 3,500 ┆ 0 ┆ -6.513 ┆ 13.55 ┆ -14.3 ┆ -3.842 ┆ 4.114 │ │ ┆ Imputed ┆ 1,300 ┆ 0 ┆ -5.795 ┆ 12.64 ┆ -12.19 ┆ -2.943 ┆ 3.357 │ │ p0.5 ┆ Model ┆ 3,500 ┆ 0 ┆ -4.986 ┆ 13.84 ┆ -13.19 ┆ -2.547 ┆ 5.638 │ │ ┆ Imputed ┆ 1,300 ┆ 0 ┆ -3.968 ┆ 12.86 ┆ -11.03 ┆ -0.789 ┆ 4.691 │ │ p0.75 ┆ Model ┆ 3,500 ┆ 0 ┆ -3.663 ┆ 13.6 ┆ -12.23 ┆ -0.6311 ┆ 6.543 │ │ ┆ Imputed ┆ 1,300 ┆ 0 ┆ -2.692 ┆ 12.58 ┆ -9.708 ┆ 0.2726 ┆ 5.792 │ └──────────┴─────────┴───────┴─────────────┴────────┴───────┴────────┴─────────┴───────┘
Running LightGBM for the mean for estimating the marginal distribution
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'seed': 1361803809}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var_gbm2 ┆ 0.9053 ┆ 0.1431 ┆ 0 ┆ -4.86 ┆ 0 ┆ -3.797 │ │ var3 ┆ 0.01722 ┆ 0.09623 ┆ 0 ┆ 25.36 ┆ 0 ┆ 25.57 │ │ var4 ┆ 0.01703 ┆ 0.1278 ┆ 0 ┆ 0.505 ┆ 0 ┆ 0.5007 │ │ unrelated_2 ┆ 0.01313 ┆ 0.1351 ┆ 0 ┆ 0.4968 ┆ 0 ┆ 0.5068 │ │ unrelated_4 ┆ 0.01256 ┆ 0.1224 ┆ 0 ┆ 0.4994 ┆ 0 ┆ 0.5004 │ │ unrelated_1 ┆ 0.01198 ┆ 0.1244 ┆ 0 ┆ 0.4951 ┆ 0 ┆ 0.515 │ │ unrelated_3 ┆ 0.01155 ┆ 0.1238 ┆ 0 ┆ 0.4954 ┆ 0 ┆ 0.5081 │ │ unrelated_5 ┆ 0.01124 ┆ 0.1273 ┆ 0 ┆ 0.4903 ┆ 0 ┆ 0.507 │ └─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 3) ┌────────────┬──────────────┬────────────────┐ │ statistic ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 │ ╞════════════╪══════════════╪════════════════╡ │ count ┆ 3481.0 ┆ 1341.0 │ │ null_count ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.859764 ┆ -4.044629 │ │ std ┆ 13.754049 ┆ 12.697181 │ │ min ┆ -47.646554 ┆ -44.116914 │ │ 25% ┆ -13.648037 ┆ -10.351482 │ │ 50% ┆ -2.147882 ┆ -1.230679 │ │ 75% ┆ 5.724548 ┆ 4.786001 │ │ max ┆ 21.923253 ┆ 20.196504 │ └────────────┴──────────────┴────────────────┘
Correlation between var_gbm3 and prediction: 0.975
Finding 10 nearest neighbors on ['___yhat']
Randomly picking one and donating ['var_gbm3']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 2569 ┆ 4 │ │ 4125 ┆ 4 │ │ 603 ┆ 3 │ │ 796 ┆ 3 │ │ 1313 ┆ 3 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm3']
Where: col(var_gbm1)
Where (impute): col(___imp_missing_var_gbm3_3)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm3 ┆ ┆ 4822 ┆ 4822 ┆ -4.591 ┆ 13.77 ┆ -4.591 ┆ 13.77 ┆ -24.66 ┆ -12.84 ┆ -2.351 ┆ 5.696 ┆ 11.1 ┆ -55.35 ┆ 26.21 │ │ var_gbm3 ┆ 0 ┆ 3481 ┆ 3481 ┆ -4.805 ┆ 14.08 ┆ -4.805 ┆ 14.08 ┆ -25.51 ┆ -13.67 ┆ -2.547 ┆ 5.958 ┆ 11.4 ┆ -55.35 ┆ 26.21 │ │ var_gbm3 ┆ 1 ┆ 1341 ┆ 1341 ┆ -4.036 ┆ 12.94 ┆ -4.036 ┆ 12.94 ┆ -23.1 ┆ -10.73 ┆ -1.829 ┆ 5.081 ┆ 10.45 ┆ -55.35 ┆ 22.89 │ └──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm3), lit(value=0, dtype=None)).alias(name=var_gbm3)
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/1.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'binary', 'num_leaves': 235, 'min_data_in_leaf': 28, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 172, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1, 'seed': 3105532915}
Iterations: 177
Model: var_gbm1=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm3, var_gbm2, bbweight__1)
Categorical features: ['var5']
┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var_gbm3 ┆ 0.6473 ┆ 0.02289 ┆ 0 ┆ -2.214 ┆ 0 ┆ -1.751 │ │ var_gbm2 ┆ 0.1026 ┆ 0.02327 ┆ 0 ┆ -2.241 ┆ 0 ┆ -1.768 │ │ var4 ┆ 0.03928 ┆ 0.1497 ┆ 0 ┆ 0.5057 ┆ 0 ┆ 0.5103 │ │ unrelated_4 ┆ 0.03772 ┆ 0.1398 ┆ 0 ┆ 0.5007 ┆ 0 ┆ 0.5004 │ │ unrelated_3 ┆ 0.03703 ┆ 0.1345 ┆ 0 ┆ 0.4992 ┆ 0 ┆ 0.4968 │ │ unrelated_2 ┆ 0.03693 ┆ 0.1406 ┆ 0 ┆ 0.5001 ┆ 0 ┆ 0.5019 │ │ unrelated_5 ┆ 0.03676 ┆ 0.1371 ┆ 0 ┆ 0.4988 ┆ 0 ┆ 0.5053 │ │ unrelated_1 ┆ 0.0362 ┆ 0.1356 ┆ 0 ┆ 0.5024 ┆ 0 ┆ 0.5015 │ │ var3 ┆ 0.02619 ┆ 0.1164 ┆ 0 ┆ 25.11 ┆ 0 ┆ 25.05 │ └─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬──────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪══════════╪══════════════╪════════════════╡ │ count ┆ 10000.0 ┆ 10000.0 ┆ 2483.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.5241 ┆ 0.517766 ┆ 0.488754 │ │ std ┆ null ┆ 0.497261 ┆ 0.493773 │ │ min ┆ 0.0 ┆ 0.000002 ┆ 0.000006 │ │ 25% ┆ null ┆ 0.00054 ┆ 0.000472 │ │ 50% ┆ null ┆ 0.98715 ┆ 0.04815 │ │ 75% ┆ null ┆ 0.999994 ┆ 0.999987 │ │ max ┆ 1.0 ┆ 1.0 ┆ 1.0 │ └────────────┴──────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬──────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪══════════╪══════════════╪════════════════╡ │ count ┆ 10000.0 ┆ 10000.0 ┆ 2483.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.5241 ┆ 0.517766 ┆ 0.488754 │ │ std ┆ null ┆ 0.497261 ┆ 0.493773 │ │ min ┆ 0.0 ┆ 0.000002 ┆ 0.000006 │ │ 25% ┆ null ┆ 0.00054 ┆ 0.000472 │ │ 50% ┆ null ┆ 0.98715 ┆ 0.04815 │ │ 75% ┆ null ┆ 0.999994 ┆ 0.999987 │ │ max ┆ 1.0 ┆ 1.0 ┆ 1.0 │ └────────────┴──────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['var_gbm1'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['var_gbm1']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 1701 ┆ 5 │ │ 371 ┆ 4 │ │ 2707 ┆ 4 │ │ 4126 ┆ 4 │ │ 5686 ┆ 4 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm1']
Where: None
Where (impute): col(___imp_missing_var_gbm1_1)
┌──────────┬─────────┬───────┬──────────────┬────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪═══════╪══════════════╪════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm1 ┆ ┆ 12483 ┆ 12483 ┆ 0.5199 ┆ 0.4996 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ │ var_gbm1 ┆ 0 ┆ 10000 ┆ 10000 ┆ 0.5241 ┆ 0.4994 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ │ var_gbm1 ┆ 1 ┆ 2483 ┆ 2483 ┆ 0.503 ┆ 0.5001 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ └──────────┴─────────┴───────┴──────────────┴────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 192, 'min_data_in_leaf': 33, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 62, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1, 'seed': 1339333759}
Iterations: 86
Model: var_gbm2=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm3, var_gbm1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var_gbm3 ┆ 0.8896 ┆ 0.1664 ┆ 0 ┆ -4.537 ┆ 0 ┆ -4.097 │ │ var4 ┆ 0.02422 ┆ 0.128 ┆ 0 ┆ 0.503 ┆ 0 ┆ 0.5021 │ │ var3 ┆ 0.02086 ┆ 0.1065 ┆ 0 ┆ 25.38 ┆ 0 ┆ 25.15 │ │ unrelated_5 ┆ 0.01394 ┆ 0.1225 ┆ 0 ┆ 0.4953 ┆ 0 ┆ 0.4894 │ │ unrelated_3 ┆ 0.01354 ┆ 0.123 ┆ 0 ┆ 0.499 ┆ 0 ┆ 0.4971 │ │ unrelated_2 ┆ 0.01307 ┆ 0.1235 ┆ 0 ┆ 0.5 ┆ 0 ┆ 0.5054 │ │ unrelated_4 ┆ 0.01253 ┆ 0.1161 ┆ 0 ┆ 0.4986 ┆ 0 ┆ 0.4989 │ │ unrelated_1 ┆ 0.01229 ┆ 0.114 ┆ 0 ┆ 0.5011 ┆ 0 ┆ 0.4949 │ └─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬────────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm2 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════╪══════════════╪════════════════╡ │ count ┆ 4802.0 ┆ 4802.0 ┆ 1305.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.667355 ┆ -4.744643 ┆ -4.623086 │ │ std ┆ 14.052945 ┆ 13.503224 ┆ 12.730957 │ │ min ┆ -55.354108 ┆ -46.262256 ┆ -44.94709 │ │ 25% ┆ -13.212003 ┆ -13.139275 ┆ -12.453296 │ │ 50% ┆ -2.262626 ┆ -2.101538 ┆ -1.936671 │ │ 75% ┆ 5.887332 ┆ 5.474333 ┆ 5.069693 │ │ max ┆ 26.213084 ┆ 21.848721 ┆ 19.554553 │ └────────────┴────────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬────────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm2 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════╪══════════════╪════════════════╡ │ count ┆ 4802.0 ┆ 4802.0 ┆ 1305.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.667355 ┆ -4.744643 ┆ -4.623086 │ │ std ┆ 14.052945 ┆ 13.503224 ┆ 12.730957 │ │ min ┆ -55.354108 ┆ -46.262256 ┆ -44.94709 │ │ 25% ┆ -13.212003 ┆ -13.139275 ┆ -12.453296 │ │ 50% ┆ -2.262626 ┆ -2.101538 ┆ -1.936671 │ │ 75% ┆ 5.887332 ┆ 5.474333 ┆ 5.069693 │ │ max ┆ 26.213084 ┆ 21.848721 ┆ 19.554553 │ └────────────┴────────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['var_gbm2'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['var_gbm2']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 363 ┆ 3 │ │ 660 ┆ 3 │ │ 4267 ┆ 3 │ │ 5428 ┆ 3 │ │ 5955 ┆ 3 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm2']
Where: col(var_gbm1)
Where (impute): col(___imp_missing_var_gbm2_2)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm2 ┆ ┆ 6107 ┆ 6107 ┆ -4.677 ┆ 13.92 ┆ -4.677 ┆ 13.92 ┆ -25.1 ┆ -13.06 ┆ -2.313 ┆ 5.73 ┆ 11.52 ┆ -55.35 ┆ 26.21 │ │ var_gbm2 ┆ 0 ┆ 4802 ┆ 4802 ┆ -4.667 ┆ 14.05 ┆ -4.667 ┆ 14.05 ┆ -25.36 ┆ -13.21 ┆ -2.284 ┆ 5.887 ┆ 11.66 ┆ -55.35 ┆ 26.21 │ │ var_gbm2 ┆ 1 ┆ 1305 ┆ 1305 ┆ -4.712 ┆ 13.41 ┆ -4.712 ┆ 13.41 ┆ -23.45 ┆ -12.4 ┆ -2.418 ┆ 5.186 ┆ 10.87 ┆ -55.35 ┆ 22.08 │ └──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm2), lit(value=0, dtype=None)).alias(name=var_gbm2)
Calling recalculate_interaction
Calling square_var
Imputation using LightGBM
Running LightGBM for q=0.25
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.25, 'seed': 1239395270}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.25 prediction: 0.977
Running LightGBM for q=0.5
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.5, 'seed': 1142944889}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.5 prediction: 0.978
Running LightGBM for q=0.75
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.75, 'seed': 3410969552}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.75 prediction: 0.977
┌──────────┬─────────┬───────┬─────────────┬────────┬───────┬────────┬──────────┬───────┐ │ Variable ┆ Sample ┆ n ┆ n (missing) ┆ mean ┆ std ┆ q25 ┆ q50 ┆ q75 │ ╞══════════╪═════════╪═══════╪═════════════╪════════╪═══════╪════════╪══════════╪═══════╡ │ p0.25 ┆ Model ┆ 4,800 ┆ 0 ┆ -5.782 ┆ 13.44 ┆ -13.22 ┆ -3.437 ┆ 4.586 │ │ ┆ Imputed ┆ 1,300 ┆ 0 ┆ -5.42 ┆ 12.7 ┆ -11.59 ┆ -3.319 ┆ 4.002 │ │ p0.5 ┆ Model ┆ 4,800 ┆ 0 ┆ -4.666 ┆ 13.5 ┆ -12.39 ┆ -2.467 ┆ 5.544 │ │ ┆ Imputed ┆ 1,300 ┆ 0 ┆ -4.053 ┆ 12.78 ┆ -9.861 ┆ -1.974 ┆ 5.071 │ │ p0.75 ┆ Model ┆ 4,800 ┆ 0 ┆ -3.708 ┆ 13.33 ┆ -11.46 ┆ -1.042 ┆ 6.241 │ │ ┆ Imputed ┆ 1,300 ┆ 0 ┆ -2.894 ┆ 12.67 ┆ -8.807 ┆ -0.01167 ┆ 5.911 │ └──────────┴─────────┴───────┴─────────────┴────────┴───────┴────────┴──────────┴───────┘
Running LightGBM for the mean for estimating the marginal distribution
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'seed': 2514065398}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬──────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪══════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var_gbm2 ┆ 0.95 ┆ 0.155 ┆ 0 ┆ -4.593 ┆ 0 ┆ -3.794 │ │ var4 ┆ 0.009838 ┆ 0.1263 ┆ 0 ┆ 0.5038 ┆ 0 ┆ 0.5007 │ │ unrelated_5 ┆ 0.007622 ┆ 0.1334 ┆ 0 ┆ 0.495 ┆ 0 ┆ 0.507 │ │ unrelated_3 ┆ 0.007199 ┆ 0.1335 ┆ 0 ┆ 0.4989 ┆ 0 ┆ 0.508 │ │ unrelated_4 ┆ 0.00693 ┆ 0.1227 ┆ 0 ┆ 0.4997 ┆ 0 ┆ 0.4998 │ │ unrelated_1 ┆ 0.006253 ┆ 0.1212 ┆ 0 ┆ 0.5006 ┆ 0 ┆ 0.5149 │ │ var3 ┆ 0.006242 ┆ 0.08741 ┆ 0 ┆ 25.42 ┆ 0 ┆ 25.56 │ │ unrelated_2 ┆ 0.005946 ┆ 0.1204 ┆ 0 ┆ 0.4996 ┆ 0 ┆ 0.5068 │ └─────────────┴──────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 3) ┌────────────┬──────────────┬────────────────┐ │ statistic ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 │ ╞════════════╪══════════════╪════════════════╡ │ count ┆ 4822.0 ┆ 1343.0 │ │ null_count ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.622871 ┆ -4.070382 │ │ std ┆ 13.540317 ┆ 12.771805 │ │ min ┆ -47.738334 ┆ -44.873052 │ │ 25% ┆ -12.460759 ┆ -10.550641 │ │ 50% ┆ -2.296351 ┆ -1.83174 │ │ 75% ┆ 5.575911 ┆ 4.995854 │ │ max ┆ 22.695594 ┆ 21.303228 │ └────────────┴──────────────┴────────────────┘
Correlation between var_gbm3 and prediction: 0.988
Finding 10 nearest neighbors on ['___yhat']
Randomly picking one and donating ['var_gbm3']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 6604 ┆ 4 │ │ 1908 ┆ 3 │ │ 2143 ┆ 3 │ │ 2392 ┆ 3 │ │ 3314 ┆ 3 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm3']
Where: col(var_gbm1)
Where (impute): col(___imp_missing_var_gbm3_3)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm3 ┆ ┆ 6165 ┆ 6165 ┆ -4.497 ┆ 13.59 ┆ -4.497 ┆ 13.59 ┆ -24.22 ┆ -12.39 ┆ -2.211 ┆ 5.592 ┆ 10.97 ┆ -55.35 ┆ 26.21 │ │ var_gbm3 ┆ 0 ┆ 4822 ┆ 4822 ┆ -4.591 ┆ 13.77 ┆ -4.591 ┆ 13.77 ┆ -24.66 ┆ -12.84 ┆ -2.351 ┆ 5.696 ┆ 11.1 ┆ -55.35 ┆ 26.21 │ │ var_gbm3 ┆ 1 ┆ 1343 ┆ 1343 ┆ -4.161 ┆ 12.92 ┆ -4.161 ┆ 12.92 ┆ -22.98 ┆ -10.85 ┆ -1.962 ┆ 5.173 ┆ 10.43 ┆ -45.28 ┆ 20.66 │ └──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm3), lit(value=0, dtype=None)).alias(name=var_gbm3)
var_gbm1
var_gbm2
var_gbm3
Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/1.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'binary', 'num_leaves': 235, 'min_data_in_leaf': 28, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 172, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1, 'seed': 2356231858}
Iterations: 177
Model: var_gbm1=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ unrelated_4 ┆ 0.1517 ┆ 0.1538 ┆ 0 ┆ 0.5007 ┆ 0 ┆ 0.5004 │ │ unrelated_5 ┆ 0.1472 ┆ 0.1526 ┆ 0 ┆ 0.4966 ┆ 0 ┆ 0.5053 │ │ var4 ┆ 0.1445 ┆ 0.1469 ┆ 0 ┆ 0.5041 ┆ 0 ┆ 0.5103 │ │ unrelated_3 ┆ 0.142 ┆ 0.147 ┆ 0 ┆ 0.5 ┆ 0 ┆ 0.4968 │ │ var3 ┆ 0.1391 ┆ 0.1076 ┆ 0 ┆ 25.13 ┆ 0 ┆ 25.05 │ │ unrelated_2 ┆ 0.1385 ┆ 0.1468 ┆ 0 ┆ 0.4995 ┆ 0 ┆ 0.5019 │ │ unrelated_1 ┆ 0.1371 ┆ 0.1454 ┆ 0 ┆ 0.5028 ┆ 0 ┆ 0.5015 │ └─────────────┴────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬──────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪══════════╪══════════════╪════════════════╡ │ count ┆ 7517.0 ┆ 7517.0 ┆ 2483.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.526008 ┆ 0.528109 ┆ 0.521178 │ │ std ┆ null ┆ 0.360953 ┆ 0.287021 │ │ min ┆ 0.0 ┆ 0.007065 ┆ 0.002619 │ │ 25% ┆ null ┆ 0.14316 ┆ 0.27449 │ │ 50% ┆ null ┆ 0.557957 ┆ 0.511862 │ │ 75% ┆ null ┆ 0.900884 ┆ 0.782061 │ │ max ┆ 1.0 ┆ 0.999079 ┆ 0.998803 │ └────────────┴──────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬──────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪══════════╪══════════════╪════════════════╡ │ count ┆ 7517.0 ┆ 7517.0 ┆ 2483.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.526008 ┆ 0.528109 ┆ 0.521178 │ │ std ┆ null ┆ 0.360953 ┆ 0.287021 │ │ min ┆ 0.0 ┆ 0.007065 ┆ 0.002619 │ │ 25% ┆ null ┆ 0.14316 ┆ 0.27449 │ │ 50% ┆ null ┆ 0.557957 ┆ 0.511862 │ │ 75% ┆ null ┆ 0.900884 ┆ 0.782061 │ │ max ┆ 1.0 ┆ 0.999079 ┆ 0.998803 │ └────────────┴──────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['var_gbm1'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['var_gbm1']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 5420 ┆ 5 │ │ 281 ┆ 4 │ │ 1296 ┆ 4 │ │ 2797 ┆ 4 │ │ 3220 ┆ 4 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm1']
Where: None
Where (impute): col(___imp_missing_var_gbm1_1)
┌──────────┬─────────┬───────┬──────────────┬────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪═══════╪══════════════╪════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm1 ┆ ┆ 10000 ┆ 10000 ┆ 0.525 ┆ 0.4994 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ │ var_gbm1 ┆ 0 ┆ 7517 ┆ 7517 ┆ 0.526 ┆ 0.4994 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ │ var_gbm1 ┆ 1 ┆ 2483 ┆ 2483 ┆ 0.5219 ┆ 0.4996 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ └──────────┴─────────┴───────┴──────────────┴────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 192, 'min_data_in_leaf': 33, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 62, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1, 'seed': 2366585833}
Iterations: 86
Model: var_gbm2=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var3 ┆ 0.423 ┆ 0.1267 ┆ 0 ┆ 25.46 ┆ 0 ┆ 25.28 │ │ var4 ┆ 0.3746 ┆ 0.173 ┆ 0 ┆ 0.505 ┆ 0 ┆ 0.5008 │ │ unrelated_4 ┆ 0.04323 ┆ 0.1469 ┆ 0 ┆ 0.5003 ┆ 0 ┆ 0.5004 │ │ unrelated_2 ┆ 0.04208 ┆ 0.1443 ┆ 0 ┆ 0.4997 ┆ 0 ┆ 0.5017 │ │ unrelated_3 ┆ 0.04189 ┆ 0.1404 ┆ 0 ┆ 0.5006 ┆ 0 ┆ 0.4965 │ │ unrelated_1 ┆ 0.03818 ┆ 0.1368 ┆ 0 ┆ 0.5042 ┆ 0 ┆ 0.4963 │ │ unrelated_5 ┆ 0.03698 ┆ 0.1319 ┆ 0 ┆ 0.4969 ┆ 0 ┆ 0.4845 │ └─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬────────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm2 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════╪══════════════╪════════════════╡ │ count ┆ 3512.0 ┆ 3512.0 ┆ 1319.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.734387 ┆ -4.660909 ┆ -4.255957 │ │ std ┆ 14.24151 ┆ 12.86899 ┆ 12.141072 │ │ min ┆ -55.354108 ┆ -43.067682 ┆ -40.717165 │ │ 25% ┆ -13.470111 ┆ -12.574851 ┆ -11.945201 │ │ 50% ┆ -2.307279 ┆ -1.36764 ┆ -0.974679 │ │ 75% ┆ 6.158269 ┆ 5.791027 ┆ 5.665508 │ │ max ┆ 26.213084 ┆ 18.216354 ┆ 15.718111 │ └────────────┴────────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬────────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm2 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════╪══════════════╪════════════════╡ │ count ┆ 3512.0 ┆ 3512.0 ┆ 1319.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.734387 ┆ -4.660909 ┆ -4.255957 │ │ std ┆ 14.24151 ┆ 12.86899 ┆ 12.141072 │ │ min ┆ -55.354108 ┆ -43.067682 ┆ -40.717165 │ │ 25% ┆ -13.470111 ┆ -12.574851 ┆ -11.945201 │ │ 50% ┆ -2.307279 ┆ -1.36764 ┆ -0.974679 │ │ 75% ┆ 6.158269 ┆ 5.791027 ┆ 5.665508 │ │ max ┆ 26.213084 ┆ 18.216354 ┆ 15.718111 │ └────────────┴────────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['var_gbm2'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['var_gbm2']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 788 ┆ 4 │ │ 3592 ┆ 4 │ │ 426 ┆ 3 │ │ 460 ┆ 3 │ │ 666 ┆ 3 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm2']
Where: col(var_gbm1)
Where (impute): col(___imp_missing_var_gbm2_2)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm2 ┆ ┆ 4831 ┆ 4831 ┆ -4.623 ┆ 14.03 ┆ -4.623 ┆ 14.03 ┆ -25.54 ┆ -13.02 ┆ -2.212 ┆ 6.003 ┆ 11.51 ┆ -55.35 ┆ 26.21 │ │ var_gbm2 ┆ 0 ┆ 3512 ┆ 3512 ┆ -4.734 ┆ 14.24 ┆ -4.734 ┆ 14.24 ┆ -25.77 ┆ -13.48 ┆ -2.313 ┆ 6.158 ┆ 11.65 ┆ -55.35 ┆ 26.21 │ │ var_gbm2 ┆ 1 ┆ 1319 ┆ 1319 ┆ -4.326 ┆ 13.45 ┆ -4.326 ┆ 13.45 ┆ -24.34 ┆ -12.11 ┆ -2.089 ┆ 5.733 ┆ 11.1 ┆ -46.86 ┆ 22.69 │ └──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm2), lit(value=0, dtype=None)).alias(name=var_gbm2)
Calling recalculate_interaction
Calling square_var
Imputation using LightGBM
Running LightGBM for q=0.25
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.25, 'seed': 4065382159}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.25 prediction: 0.952
Running LightGBM for q=0.5
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.5, 'seed': 500046557}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.5 prediction: 0.953
Running LightGBM for q=0.75
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.75, 'seed': 3057184033}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.75 prediction: 0.954
┌──────────┬─────────┬───────┬─────────────┬────────┬───────┬────────┬─────────┬───────┐ │ Variable ┆ Sample ┆ n ┆ n (missing) ┆ mean ┆ std ┆ q25 ┆ q50 ┆ q75 │ ╞══════════╪═════════╪═══════╪═════════════╪════════╪═══════╪════════╪═════════╪═══════╡ │ p0.25 ┆ Model ┆ 3,500 ┆ 0 ┆ -6.375 ┆ 13.4 ┆ -14.29 ┆ -4.004 ┆ 4.141 │ │ ┆ Imputed ┆ 1,400 ┆ 0 ┆ -5.956 ┆ 12.67 ┆ -12.18 ┆ -3.149 ┆ 2.844 │ │ p0.5 ┆ Model ┆ 3,500 ┆ 0 ┆ -4.862 ┆ 13.68 ┆ -13.41 ┆ -2.365 ┆ 5.862 │ │ ┆ Imputed ┆ 1,400 ┆ 0 ┆ -4.227 ┆ 12.96 ┆ -10.96 ┆ -0.466 ┆ 4.782 │ │ p0.75 ┆ Model ┆ 3,500 ┆ 0 ┆ -3.567 ┆ 13.47 ┆ -11.97 ┆ -0.75 ┆ 6.628 │ │ ┆ Imputed ┆ 1,400 ┆ 0 ┆ -2.94 ┆ 12.76 ┆ -9.556 ┆ 0.09399 ┆ 5.59 │ └──────────┴─────────┴───────┴─────────────┴────────┴───────┴────────┴─────────┴───────┘
Running LightGBM for the mean for estimating the marginal distribution
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'seed': 2506042098}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var_gbm2 ┆ 0.9113 ┆ 0.1415 ┆ 0 ┆ -4.738 ┆ 0 ┆ -4.018 │ │ var4 ┆ 0.01631 ┆ 0.1287 ┆ 0 ┆ 0.5049 ┆ 0 ┆ 0.5039 │ │ var3 ┆ 0.0139 ┆ 0.09113 ┆ 0 ┆ 25.36 ┆ 0 ┆ 25.53 │ │ unrelated_5 ┆ 0.01354 ┆ 0.1419 ┆ 0 ┆ 0.4895 ┆ 0 ┆ 0.5064 │ │ unrelated_4 ┆ 0.01187 ┆ 0.1253 ┆ 0 ┆ 0.5022 ┆ 0 ┆ 0.4971 │ │ unrelated_1 ┆ 0.0115 ┆ 0.1207 ┆ 0 ┆ 0.496 ┆ 0 ┆ 0.5188 │ │ unrelated_2 ┆ 0.01148 ┆ 0.1308 ┆ 0 ┆ 0.4971 ┆ 0 ┆ 0.511 │ │ unrelated_3 ┆ 0.01014 ┆ 0.1199 ┆ 0 ┆ 0.4963 ┆ 0 ┆ 0.5032 │ └─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 3) ┌────────────┬──────────────┬────────────────┐ │ statistic ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 │ ╞════════════╪══════════════╪════════════════╡ │ count ┆ 3462.0 ┆ 1371.0 │ │ null_count ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.857463 ┆ -4.343819 │ │ std ┆ 13.740428 ┆ 12.841089 │ │ min ┆ -50.424402 ┆ -43.909974 │ │ 25% ┆ -13.585667 ┆ -10.897252 │ │ 50% ┆ -2.24905 ┆ -1.453721 │ │ 75% ┆ 5.717693 ┆ 4.504366 │ │ max ┆ 22.708233 ┆ 19.029265 │ └────────────┴──────────────┴────────────────┘
Correlation between var_gbm3 and prediction: 0.974
Finding 10 nearest neighbors on ['___yhat']
Randomly picking one and donating ['var_gbm3']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 7290 ┆ 6 │ │ 1462 ┆ 4 │ │ 4886 ┆ 4 │ │ 6439 ┆ 4 │ │ 7099 ┆ 4 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm3']
Where: col(var_gbm1)
Where (impute): col(___imp_missing_var_gbm3_3)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm3 ┆ ┆ 4833 ┆ 4833 ┆ -4.696 ┆ 13.89 ┆ -4.696 ┆ 13.89 ┆ -25.56 ┆ -12.83 ┆ -2.212 ┆ 5.73 ┆ 11.1 ┆ -55.35 ┆ 26.21 │ │ var_gbm3 ┆ 0 ┆ 3462 ┆ 3462 ┆ -4.818 ┆ 14.11 ┆ -4.818 ┆ 14.11 ┆ -25.54 ┆ -13.62 ┆ -2.525 ┆ 5.958 ┆ 11.35 ┆ -55.35 ┆ 26.21 │ │ var_gbm3 ┆ 1 ┆ 1371 ┆ 1371 ┆ -4.39 ┆ 13.32 ┆ -4.39 ┆ 13.32 ┆ -25.6 ┆ -11.04 ┆ -1.806 ┆ 4.992 ┆ 10.49 ┆ -48.11 ┆ 22.89 │ └──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm3), lit(value=0, dtype=None)).alias(name=var_gbm3)
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/2.srmi.implicate
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'binary', 'num_leaves': 235, 'min_data_in_leaf': 28, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 172, 'bagging_fraction': 0.667893687986521, 'bagging_freq': 1, 'seed': 46688280}
Iterations: 177
Model: var_gbm1=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm3, var_gbm2, bbweight__1)
Categorical features: ['var5']
┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var_gbm3 ┆ 0.6317 ┆ 0.02784 ┆ 0 ┆ -2.27 ┆ 0 ┆ -1.886 │ │ var_gbm2 ┆ 0.1294 ┆ 0.02369 ┆ 0 ┆ -2.233 ┆ 0 ┆ -1.868 │ │ unrelated_5 ┆ 0.03763 ┆ 0.1409 ┆ 0 ┆ 0.4988 ┆ 0 ┆ 0.5053 │ │ unrelated_2 ┆ 0.03622 ┆ 0.1411 ┆ 0 ┆ 0.5001 ┆ 0 ┆ 0.5019 │ │ unrelated_3 ┆ 0.0356 ┆ 0.137 ┆ 0 ┆ 0.4992 ┆ 0 ┆ 0.4968 │ │ unrelated_4 ┆ 0.03534 ┆ 0.1395 ┆ 0 ┆ 0.5007 ┆ 0 ┆ 0.5004 │ │ unrelated_1 ┆ 0.03398 ┆ 0.1327 ┆ 0 ┆ 0.5024 ┆ 0 ┆ 0.5015 │ │ var4 ┆ 0.03279 ┆ 0.141 ┆ 0 ┆ 0.5057 ┆ 0 ┆ 0.5103 │ │ var3 ┆ 0.02735 ┆ 0.1164 ┆ 0 ┆ 25.11 ┆ 0 ┆ 25.05 │ └─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬──────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪══════════╪══════════════╪════════════════╡ │ count ┆ 10000.0 ┆ 10000.0 ┆ 2483.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.525 ┆ 0.518959 ┆ 0.493751 │ │ std ┆ null ┆ 0.497398 ┆ 0.494509 │ │ min ┆ 0.0 ┆ 0.000002 ┆ 0.000004 │ │ 25% ┆ null ┆ 0.000445 ┆ 0.000367 │ │ 50% ┆ null ┆ 0.990332 ┆ 0.075319 │ │ 75% ┆ null ┆ 0.999995 ┆ 0.999989 │ │ max ┆ 1.0 ┆ 1.0 ┆ 1.0 │ └────────────┴──────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬──────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm1 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪══════════╪══════════════╪════════════════╡ │ count ┆ 10000.0 ┆ 10000.0 ┆ 2483.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ 0.525 ┆ 0.518959 ┆ 0.493751 │ │ std ┆ null ┆ 0.497398 ┆ 0.494509 │ │ min ┆ 0.0 ┆ 0.000002 ┆ 0.000004 │ │ 25% ┆ null ┆ 0.000445 ┆ 0.000367 │ │ 50% ┆ null ┆ 0.990332 ┆ 0.075319 │ │ 75% ┆ null ┆ 0.999995 ┆ 0.999989 │ │ max ┆ 1.0 ┆ 1.0 ┆ 1.0 │ └────────────┴──────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['var_gbm1'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['var_gbm1']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 83 ┆ 4 │ │ 707 ┆ 4 │ │ 2758 ┆ 4 │ │ 4176 ┆ 4 │ │ 6017 ┆ 4 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm1']
Where: None
Where (impute): col(___imp_missing_var_gbm1_1)
┌──────────┬─────────┬───────┬──────────────┬────────┬────────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪═══════╪══════════════╪════════╪════════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm1 ┆ ┆ 12483 ┆ 12483 ┆ 0.5212 ┆ 0.4996 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ │ var_gbm1 ┆ 0 ┆ 10000 ┆ 10000 ┆ 0.525 ┆ 0.4994 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ │ var_gbm1 ┆ 1 ┆ 2483 ┆ 2483 ┆ 0.5058 ┆ 0.5001 ┆ 1 ┆ 0 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 ┆ 1 │ └──────────┴─────────┴───────┴──────────────┴────────┴────────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Imputation using LightGBM
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 192, 'min_data_in_leaf': 33, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 62, 'bagging_fraction': 0.5535732695351729, 'bagging_freq': 1, 'seed': 1319690195}
Iterations: 86
Model: var_gbm2=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm3, var_gbm1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬─────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪═════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var_gbm3 ┆ 0.8912 ┆ 0.1826 ┆ 0 ┆ -4.659 ┆ 0 ┆ -4.31 │ │ var4 ┆ 0.02116 ┆ 0.122 ┆ 0 ┆ 0.5039 ┆ 0 ┆ 0.5002 │ │ var3 ┆ 0.01953 ┆ 0.1012 ┆ 0 ┆ 25.41 ┆ 0 ┆ 25.23 │ │ unrelated_2 ┆ 0.01525 ┆ 0.1222 ┆ 0 ┆ 0.5003 ┆ 0 ┆ 0.5018 │ │ unrelated_1 ┆ 0.01378 ┆ 0.121 ┆ 0 ┆ 0.502 ┆ 0 ┆ 0.497 │ │ unrelated_5 ┆ 0.01332 ┆ 0.1182 ┆ 0 ┆ 0.4935 ┆ 0 ┆ 0.4861 │ │ unrelated_3 ┆ 0.01301 ┆ 0.1173 ┆ 0 ┆ 0.4995 ┆ 0 ┆ 0.4959 │ │ unrelated_4 ┆ 0.01271 ┆ 0.1154 ┆ 0 ┆ 0.5004 ┆ 0 ┆ 0.5018 │ └─────────────┴─────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 4) ┌────────────┬────────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm2 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════╪══════════════╪════════════════╡ │ count ┆ 4831.0 ┆ 4831.0 ┆ 1327.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.622961 ┆ -4.59241 ┆ -4.34546 │ │ std ┆ 14.030754 ┆ 13.533478 ┆ 12.56035 │ │ min ┆ -55.354108 ┆ -46.446028 ┆ -44.655684 │ │ 25% ┆ -13.017682 ┆ -12.750504 ┆ -11.591296 │ │ 50% ┆ -2.211656 ┆ -1.934721 ┆ -1.838333 │ │ 75% ┆ 6.002604 ┆ 5.450316 ┆ 4.795132 │ │ max ┆ 26.213084 ┆ 20.22214 ┆ 18.977101 │ └────────────┴────────────┴──────────────┴────────────────┘
shape: (9, 4) ┌────────────┬────────────┬──────────────┬────────────────┐ │ statistic ┆ var_gbm2 ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 ┆ f64 │ ╞════════════╪════════════╪══════════════╪════════════════╡ │ count ┆ 4831.0 ┆ 4831.0 ┆ 1327.0 │ │ null_count ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.622961 ┆ -4.59241 ┆ -4.34546 │ │ std ┆ 14.030754 ┆ 13.533478 ┆ 12.56035 │ │ min ┆ -55.354108 ┆ -46.446028 ┆ -44.655684 │ │ 25% ┆ -13.017682 ┆ -12.750504 ┆ -11.591296 │ │ 50% ┆ -2.211656 ┆ -1.934721 ┆ -1.838333 │ │ 75% ┆ 6.002604 ┆ 5.450316 ┆ 4.795132 │ │ max ┆ 26.213084 ┆ 20.22214 ┆ 18.977101 │ └────────────┴────────────┴──────────────┴────────────────┘
error=pmm: donating observed value(s) ['var_gbm2'] from 10-nearest matched donors
Finding 10 nearest neighbors on ['___prediction']
Randomly picking one and donating ['var_gbm2']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 1234 ┆ 3 │ │ 1879 ┆ 3 │ │ 2898 ┆ 3 │ │ 3913 ┆ 3 │ │ 4109 ┆ 3 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm2']
Where: col(var_gbm1)
Where (impute): col(___imp_missing_var_gbm2_2)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm2 ┆ ┆ 6158 ┆ 6158 ┆ -4.586 ┆ 13.83 ┆ -4.586 ┆ 13.83 ┆ -25.35 ┆ -12.66 ┆ -2.212 ┆ 5.887 ┆ 11.23 ┆ -55.35 ┆ 26.21 │ │ var_gbm2 ┆ 0 ┆ 4831 ┆ 4831 ┆ -4.623 ┆ 14.03 ┆ -4.623 ┆ 14.03 ┆ -25.54 ┆ -13.02 ┆ -2.212 ┆ 6.003 ┆ 11.51 ┆ -55.35 ┆ 26.21 │ │ var_gbm2 ┆ 1 ┆ 1327 ┆ 1327 ┆ -4.454 ┆ 13.09 ┆ -4.454 ┆ 13.09 ┆ -23.2 ┆ -11.6 ┆ -2.284 ┆ 5.47 ┆ 10.16 ┆ -46.86 ┆ 24.05 │ └──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm2), lit(value=0, dtype=None)).alias(name=var_gbm2)
Calling recalculate_interaction
Calling square_var
Imputation using LightGBM
Running LightGBM for q=0.25
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.25, 'seed': 280142868}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.25 prediction: 0.980
Running LightGBM for q=0.5
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.5, 'seed': 2518451516}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.5 prediction: 0.980
Running LightGBM for q=0.75
Running lightgbm model with parameters: {'objective': 'quantile', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'alpha': 0.75, 'seed': 45808945}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
Correlation between var_gbm3 and q=0.75 prediction: 0.979
┌──────────┬─────────┬───────┬─────────────┬────────┬───────┬────────┬──────────┬───────┐ │ Variable ┆ Sample ┆ n ┆ n (missing) ┆ mean ┆ std ┆ q25 ┆ q50 ┆ q75 │ ╞══════════╪═════════╪═══════╪═════════════╪════════╪═══════╪════════╪══════════╪═══════╡ │ p0.25 ┆ Model ┆ 4,800 ┆ 0 ┆ -5.781 ┆ 13.62 ┆ -13.6 ┆ -3.773 ┆ 4.283 │ │ ┆ Imputed ┆ 1,400 ┆ 0 ┆ -5.516 ┆ 13.11 ┆ -12.05 ┆ -3.346 ┆ 3.582 │ │ p0.5 ┆ Model ┆ 4,800 ┆ 0 ┆ -4.723 ┆ 13.64 ┆ -12.89 ┆ -2.229 ┆ 5.473 │ │ ┆ Imputed ┆ 1,400 ┆ 0 ┆ -4.263 ┆ 13.12 ┆ -11.1 ┆ -1.494 ┆ 4.887 │ │ p0.75 ┆ Model ┆ 4,800 ┆ 0 ┆ -3.84 ┆ 13.47 ┆ -12.05 ┆ -1.363 ┆ 6.034 │ │ ┆ Imputed ┆ 1,400 ┆ 0 ┆ -3.148 ┆ 12.95 ┆ -10.0 ┆ -0.03091 ┆ 5.637 │ └──────────┴─────────┴───────┴─────────────┴────────┴───────┴────────┴──────────┴───────┘
Running LightGBM for the mean for estimating the marginal distribution
Running lightgbm model with parameters: {'objective': 'regression', 'num_leaves': 185, 'min_data_in_leaf': 32, 'boosting': 'gbdt', 'verbose': -1, 'metric': 'rmse', 'min_data_per_group': 25, 'num_threads': 1, 'max_depth': 31, 'bagging_fraction': 0.9124353738882424, 'bagging_freq': 5, 'seed': 2021398447}
Iterations: 181
Model: var_gbm3=f(var3, var4, var5, unrelated_1, unrelated_2, unrelated_3, unrelated_4, unrelated_5, repeat_1, var_gbm2, var_gbm1, bbweight__1)
Categorical features: ['var5']
┌─────────────┬──────────┬───────────┬─────────────────┬────────┬─────────────────┬────────┐ │ Feature ┆ Gain ┆ Frequency ┆ Model ┆ Model ┆ Impute ┆ Impute │ │ ┆ ┆ ┆ share (missing) ┆ mean ┆ share (missing) ┆ mean │ ╞═════════════╪══════════╪═══════════╪═════════════════╪════════╪═════════════════╪════════╡ │ var_gbm2 ┆ 0.9525 ┆ 0.1532 ┆ 0 ┆ -4.581 ┆ 0 ┆ -4.037 │ │ var4 ┆ 0.008931 ┆ 0.1306 ┆ 0 ┆ 0.5046 ┆ 0 ┆ 0.5039 │ │ var3 ┆ 0.007075 ┆ 0.09046 ┆ 0 ┆ 25.41 ┆ 0 ┆ 25.51 │ │ unrelated_5 ┆ 0.006945 ┆ 0.1318 ┆ 0 ┆ 0.4943 ┆ 0 ┆ 0.5066 │ │ unrelated_4 ┆ 0.006943 ┆ 0.128 ┆ 0 ┆ 0.5007 ┆ 0 ┆ 0.4974 │ │ unrelated_2 ┆ 0.006072 ┆ 0.119 ┆ 0 ┆ 0.5011 ┆ 0 ┆ 0.5109 │ │ unrelated_1 ┆ 0.006006 ┆ 0.1276 ┆ 0 ┆ 0.5025 ┆ 0 ┆ 0.5188 │ │ unrelated_3 ┆ 0.005532 ┆ 0.1193 ┆ 0 ┆ 0.4983 ┆ 0 ┆ 0.5035 │ └─────────────┴──────────┴───────────┴─────────────────┴────────┴─────────────────┴────────┘
Predictions
shape: (9, 3) ┌────────────┬──────────────┬────────────────┐ │ statistic ┆ Model (yhat) ┆ Imputed (yhat) │ │ --- ┆ --- ┆ --- │ │ str ┆ f64 ┆ f64 │ ╞════════════╪══════════════╪════════════════╡ │ count ┆ 4833.0 ┆ 1373.0 │ │ null_count ┆ 0.0 ┆ 0.0 │ │ mean ┆ -4.704598 ┆ -4.294848 │ │ std ┆ 13.707223 ┆ 13.111833 │ │ min ┆ -52.817484 ┆ -46.164067 │ │ 25% ┆ -12.827819 ┆ -11.225195 │ │ 50% ┆ -2.281483 ┆ -1.711092 │ │ 75% ┆ 5.61156 ┆ 4.932624 │ │ max ┆ 24.794587 ┆ 20.266915 │ └────────────┴──────────────┴────────────────┘
Correlation between var_gbm3 and prediction: 0.989
Finding 10 nearest neighbors on ['___yhat']
Randomly picking one and donating ['var_gbm3']
Most common matches:
shape: (5, 2) ┌───────┬─────────┐ │ index ┆ nDonors │ │ --- ┆ --- │ │ i16 ┆ i8 │ ╞═══════╪═════════╡ │ 514 ┆ 3 │ │ 1825 ┆ 3 │ │ 2050 ┆ 3 │ │ 2753 ┆ 3 │ │ 4306 ┆ 3 │ └───────┴─────────┘
Post-imputation statistics for ['var_gbm3']
Where: col(var_gbm1)
Where (impute): col(___imp_missing_var_gbm3_3)
┌──────────┬─────────┬──────┬──────────────┬────────┬───────┬──────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┬─────────────┐ │ Variable ┆ Imputed ┆ n ┆ n (not null) ┆ mean ┆ std ┆ mean (not 0) ┆ std (not 0) ┆ q10 (not 0) ┆ q25 (not 0) ┆ q50 (not 0) ┆ q75 (not 0) ┆ q90 (not 0) ┆ min (not 0) ┆ max (not 0) │ ╞══════════╪═════════╪══════╪══════════════╪════════╪═══════╪══════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╪═════════════╡ │ var_gbm3 ┆ ┆ 6206 ┆ 6206 ┆ -4.621 ┆ 13.75 ┆ -4.621 ┆ 13.75 ┆ -25.56 ┆ -12.41 ┆ -2.107 ┆ 5.621 ┆ 10.8 ┆ -55.35 ┆ 26.21 │ │ var_gbm3 ┆ 0 ┆ 4833 ┆ 4833 ┆ -4.696 ┆ 13.89 ┆ -4.696 ┆ 13.89 ┆ -25.56 ┆ -12.83 ┆ -2.212 ┆ 5.73 ┆ 11.1 ┆ -55.35 ┆ 26.21 │ │ var_gbm3 ┆ 1 ┆ 1373 ┆ 1373 ┆ -4.356 ┆ 13.27 ┆ -4.356 ┆ 13.27 ┆ -25.54 ┆ -10.99 ┆ -1.516 ┆ 5.182 ┆ 10.32 ┆ -48.11 ┆ 19.99 │ └──────────┴─────────┴──────┴──────────────┴────────┴───────┴──────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┴─────────────┘
Updating data according to narwhals expression: when_then(all_horizontal(col(var_gbm1), ignore_nulls=False), col(var_gbm3), lit(value=0, dtype=None)).alias(name=var_gbm3)
var_gbm1
var_gbm2
var_gbm3
Final Estimates by Iteration
Removing existing directory C:\Users\jonro\OneDrive\Documents\Coding\survey_kit\.scratch\temp_files/py_srmi_test_gbm.srmi/2.srmi.implicate
It's automatically saved and can be loaded with (see path_model above):
path_model = f'{config.path_temp_files}/py_srmi_test_gbm'
srmi = SRMI.load(path_model)
In [8]:
logger.info("Get the results")
_ = df_list = srmi.df_implicates
Get the results
In [9]:
logger.info("\n\nLook at the original")
_ = summary(df_original, detailed=True, drb_round=True)
logger.info("\n\nLook at the imputes")
_ = df_list.pipe(summary, detailed=True, drb_round=True)
logger.info("\n\nLook at the imputes | var_gbm1 == 0")
_ = df_list.filter(~nw.col("var_gbm1")).pipe(summary, detailed=True, drb_round=True)
logger.info("\n\nLook at the imputes | var_gbm1 == 1")
_ = df_list.filter(nw.col("var_gbm1")).pipe(summary, detailed=True, drb_round=True)
Look at the original
┌──────────────┬────────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ q25 ┆ q50 ┆ q75 ┆ max │ ╞══════════════╪════════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡ │ _row_index_ ┆ 10,000 ┆ 0 ┆ 5,000.0 ┆ 2,887.0 ┆ 0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │ │ index ┆ 10,000 ┆ 0 ┆ 5,000.0 ┆ 2,887.0 ┆ 0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │ │ year ┆ 10,000 ┆ 0 ┆ 2,018.0 ┆ 1.416 ┆ 2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │ │ month ┆ 10,000 ┆ 0 ┆ 6.514 ┆ 3.432 ┆ 1.0 ┆ 4.0 ┆ 6.0 ┆ 9.0 ┆ 12.0 │ │ var2 ┆ 10,000 ┆ 0 ┆ 4.978 ┆ 3.155 ┆ 0.0 ┆ 2.0 ┆ 5.0 ┆ 8.0 ┆ 10.0 │ │ var3 ┆ 10,000 ┆ 0 ┆ 25.11 ┆ 14.75 ┆ 0.0 ┆ 12.0 ┆ 25.0 ┆ 38.0 ┆ 50.0 │ │ var4 ┆ 10,000 ┆ 0 ┆ 0.5057 ┆ 0.2879 ┆ 0.000027 ┆ 0.2557 ┆ 0.5084 ┆ 0.7543 ┆ 1.0 │ │ unrelated_1 ┆ 10,000 ┆ 0 ┆ 0.5024 ┆ 0.2884 ┆ 0.0001191 ┆ 0.2527 ┆ 0.5036 ┆ 0.753 ┆ 1.0 │ │ unrelated_2 ┆ 10,000 ┆ 0 ┆ 0.5001 ┆ 0.2876 ┆ 0.000049 ┆ 0.253 ┆ 0.4975 ┆ 0.7486 ┆ 0.9995 │ │ unrelated_3 ┆ 10,000 ┆ 0 ┆ 0.4992 ┆ 0.2888 ┆ 0.000129 ┆ 0.2512 ┆ 0.496 ┆ 0.7505 ┆ 0.9999 │ │ unrelated_4 ┆ 10,000 ┆ 0 ┆ 0.5007 ┆ 0.2887 ┆ 0.0001329 ┆ 0.2501 ┆ 0.5005 ┆ 0.7518 ┆ 1.0 │ │ unrelated_5 ┆ 10,000 ┆ 0 ┆ 0.4988 ┆ 0.289 ┆ 0.000071 ┆ 0.2496 ┆ 0.4968 ┆ 0.7498 ┆ 0.9999 │ │ missing_gbm1 ┆ 10,000 ┆ 0 ┆ 0.5026 ┆ 0.289 ┆ 0.000006 ┆ 0.2516 ┆ 0.5074 ┆ 0.7523 ┆ 1.0 │ │ missing_gbm2 ┆ 10,000 ┆ 0 ┆ 0.5032 ┆ 0.2905 ┆ 0.000005 ┆ 0.2531 ┆ 0.5041 ┆ 0.756 ┆ 1.0 │ │ missing_gbm3 ┆ 10,000 ┆ 0 ┆ 0.4939 ┆ 0.2907 ┆ 0.0001079 ┆ 0.2407 ┆ 0.4906 ┆ 0.741 ┆ 0.9998 │ │ var_gbm2 ┆ 10,000 ┆ 0 ┆ -2.402 ┆ 10.33 ┆ -55.35 ┆ -3.025 ┆ 0.0 ┆ 0.0 ┆ 26.21 │ │ var_gbm3 ┆ 10,000 ┆ 0 ┆ -2.402 ┆ 10.33 ┆ -55.35 ┆ -3.025 ┆ 0.0 ┆ 0.0 ┆ 26.21 │ │ var5 ┆ 10,000 ┆ 0 ┆ 0.4999 ┆ 0.5 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 1.0 ┆ 1.0 │ │ var_gbm1 ┆ 10,000 ┆ 0 ┆ 0.5229 ┆ 0.4995 ┆ 0.0 ┆ 0.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 │ └──────────────┴────────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘
Look at the imputes
┌─────────────┬────────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ q25 ┆ q50 ┆ q75 ┆ max │ ╞═════════════╪════════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡ │ index ┆ 10,000 ┆ 0 ┆ 5,000.0 ┆ 2,887.0 ┆ 0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │ │ _row_index_ ┆ 10,000 ┆ 0 ┆ 5,000.0 ┆ 2,887.0 ┆ 0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │ │ year ┆ 10,000 ┆ 0 ┆ 2,018.0 ┆ 1.416 ┆ 2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │ │ month ┆ 10,000 ┆ 0 ┆ 6.514 ┆ 3.432 ┆ 1.0 ┆ 4.0 ┆ 6.0 ┆ 9.0 ┆ 12.0 │ │ var2 ┆ 10,000 ┆ 0 ┆ 4.978 ┆ 3.155 ┆ 0.0 ┆ 2.0 ┆ 5.0 ┆ 8.0 ┆ 10.0 │ │ var3 ┆ 10,000 ┆ 0 ┆ 25.11 ┆ 14.75 ┆ 0.0 ┆ 12.0 ┆ 25.0 ┆ 38.0 ┆ 50.0 │ │ var4 ┆ 10,000 ┆ 0 ┆ 0.5057 ┆ 0.2879 ┆ 0.000027 ┆ 0.2557 ┆ 0.5084 ┆ 0.7543 ┆ 1.0 │ │ unrelated_1 ┆ 10,000 ┆ 0 ┆ 0.5024 ┆ 0.2884 ┆ 0.0001191 ┆ 0.2527 ┆ 0.5036 ┆ 0.753 ┆ 1.0 │ │ unrelated_2 ┆ 10,000 ┆ 0 ┆ 0.5001 ┆ 0.2876 ┆ 0.000049 ┆ 0.253 ┆ 0.4975 ┆ 0.7486 ┆ 0.9995 │ │ unrelated_3 ┆ 10,000 ┆ 0 ┆ 0.4992 ┆ 0.2888 ┆ 0.000129 ┆ 0.2512 ┆ 0.496 ┆ 0.7505 ┆ 0.9999 │ │ unrelated_4 ┆ 10,000 ┆ 0 ┆ 0.5007 ┆ 0.2887 ┆ 0.0001329 ┆ 0.2501 ┆ 0.5005 ┆ 0.7518 ┆ 1.0 │ │ unrelated_5 ┆ 10,000 ┆ 0 ┆ 0.4988 ┆ 0.289 ┆ 0.000071 ┆ 0.2496 ┆ 0.4968 ┆ 0.7498 ┆ 0.9999 │ │ repeat_1 ┆ 10,000 ┆ 0 ┆ 0.5024 ┆ 0.2884 ┆ 0.0001191 ┆ 0.2527 ┆ 0.5036 ┆ 0.753 ┆ 1.0 │ │ var_gbm3 ┆ 10,000 ┆ 0 ┆ -2.231 ┆ 9.836 ┆ -55.35 ┆ -1.777 ┆ 0.0 ┆ 0.0 ┆ 26.21 │ │ var_gbm2 ┆ 10,000 ┆ 0 ┆ -2.253 ┆ 9.958 ┆ -55.35 ┆ -1.614 ┆ 0.0 ┆ 0.0 ┆ 26.21 │ │ var_gbm12 ┆ 10,000 ┆ 0 ┆ -2.253 ┆ 9.958 ┆ -55.35 ┆ -1.614 ┆ 0.0 ┆ 0.0 ┆ 26.21 │ │ var_gbm2_sq ┆ 10,000 ┆ 0 ┆ 104.2 ┆ 267.4 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 72.04 ┆ 3,064.0 │ │ var5 ┆ 10,000 ┆ 0 ┆ 0.4999 ┆ 0.5 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 1.0 ┆ 1.0 │ │ var_gbm1 ┆ 10,000 ┆ 0 ┆ 0.5203 ┆ 0.4996 ┆ 0.0 ┆ 0.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 │ └─────────────┴────────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘
┌─────────────┬────────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ q25 ┆ q50 ┆ q75 ┆ max │ ╞═════════════╪════════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡ │ index ┆ 10,000 ┆ 0 ┆ 5,000.0 ┆ 2,887.0 ┆ 0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │ │ _row_index_ ┆ 10,000 ┆ 0 ┆ 5,000.0 ┆ 2,887.0 ┆ 0.0 ┆ 2,499.0 ┆ 4,999.0 ┆ 7,499.0 ┆ 9,999.0 │ │ year ┆ 10,000 ┆ 0 ┆ 2,018.0 ┆ 1.416 ┆ 2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │ │ month ┆ 10,000 ┆ 0 ┆ 6.514 ┆ 3.432 ┆ 1.0 ┆ 4.0 ┆ 6.0 ┆ 9.0 ┆ 12.0 │ │ var2 ┆ 10,000 ┆ 0 ┆ 4.978 ┆ 3.155 ┆ 0.0 ┆ 2.0 ┆ 5.0 ┆ 8.0 ┆ 10.0 │ │ var3 ┆ 10,000 ┆ 0 ┆ 25.11 ┆ 14.75 ┆ 0.0 ┆ 12.0 ┆ 25.0 ┆ 38.0 ┆ 50.0 │ │ var4 ┆ 10,000 ┆ 0 ┆ 0.5057 ┆ 0.2879 ┆ 0.000027 ┆ 0.2557 ┆ 0.5084 ┆ 0.7543 ┆ 1.0 │ │ unrelated_1 ┆ 10,000 ┆ 0 ┆ 0.5024 ┆ 0.2884 ┆ 0.0001191 ┆ 0.2527 ┆ 0.5036 ┆ 0.753 ┆ 1.0 │ │ unrelated_2 ┆ 10,000 ┆ 0 ┆ 0.5001 ┆ 0.2876 ┆ 0.000049 ┆ 0.253 ┆ 0.4975 ┆ 0.7486 ┆ 0.9995 │ │ unrelated_3 ┆ 10,000 ┆ 0 ┆ 0.4992 ┆ 0.2888 ┆ 0.000129 ┆ 0.2512 ┆ 0.496 ┆ 0.7505 ┆ 0.9999 │ │ unrelated_4 ┆ 10,000 ┆ 0 ┆ 0.5007 ┆ 0.2887 ┆ 0.0001329 ┆ 0.2501 ┆ 0.5005 ┆ 0.7518 ┆ 1.0 │ │ unrelated_5 ┆ 10,000 ┆ 0 ┆ 0.4988 ┆ 0.289 ┆ 0.000071 ┆ 0.2496 ┆ 0.4968 ┆ 0.7498 ┆ 0.9999 │ │ repeat_1 ┆ 10,000 ┆ 0 ┆ 0.5024 ┆ 0.2884 ┆ 0.0001191 ┆ 0.2527 ┆ 0.5036 ┆ 0.753 ┆ 1.0 │ │ var_gbm3 ┆ 10,000 ┆ 0 ┆ -2.266 ┆ 9.927 ┆ -55.35 ┆ -1.584 ┆ 0.0 ┆ 0.0 ┆ 26.21 │ │ var_gbm2 ┆ 10,000 ┆ 0 ┆ -2.254 ┆ 9.969 ┆ -55.35 ┆ -1.69 ┆ 0.0 ┆ 0.0 ┆ 26.21 │ │ var_gbm12 ┆ 10,000 ┆ 0 ┆ -2.254 ┆ 9.969 ┆ -55.35 ┆ -1.69 ┆ 0.0 ┆ 0.0 ┆ 26.21 │ │ var_gbm2_sq ┆ 10,000 ┆ 0 ┆ 104.4 ┆ 266.8 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 71.95 ┆ 3,064.0 │ │ var5 ┆ 10,000 ┆ 0 ┆ 0.4999 ┆ 0.5 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 1.0 ┆ 1.0 │ │ var_gbm1 ┆ 10,000 ┆ 0 ┆ 0.521 ┆ 0.4996 ┆ 0.0 ┆ 0.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 │ └─────────────┴────────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘
Look at the imputes | var_gbm1 == 0
┌─────────────┬───────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ q25 ┆ q50 ┆ q75 ┆ max │ ╞═════════════╪═══════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡ │ index ┆ 4,800 ┆ 0 ┆ 5,062.0 ┆ 2,905.0 ┆ 1.0 ┆ 2,548.0 ┆ 5,081.0 ┆ 7,645.0 ┆ 9,997.0 │ │ _row_index_ ┆ 4,800 ┆ 0 ┆ 5,062.0 ┆ 2,905.0 ┆ 1.0 ┆ 2,548.0 ┆ 5,081.0 ┆ 7,645.0 ┆ 9,997.0 │ │ year ┆ 4,800 ┆ 0 ┆ 2,018.0 ┆ 1.421 ┆ 2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │ │ month ┆ 4,800 ┆ 0 ┆ 6.533 ┆ 3.431 ┆ 1.0 ┆ 4.0 ┆ 6.0 ┆ 9.0 ┆ 12.0 │ │ var2 ┆ 4,800 ┆ 0 ┆ 4.635 ┆ 3.235 ┆ 0.0 ┆ 2.0 ┆ 4.0 ┆ 7.0 ┆ 10.0 │ │ var3 ┆ 4,800 ┆ 0 ┆ 24.85 ┆ 12.92 ┆ 0.0 ┆ 14.0 ┆ 25.0 ┆ 36.0 ┆ 50.0 │ │ var4 ┆ 4,800 ┆ 0 ┆ 0.5082 ┆ 0.288 ┆ 0.000027 ┆ 0.2552 ┆ 0.5109 ┆ 0.7563 ┆ 1.0 │ │ unrelated_1 ┆ 4,800 ┆ 0 ┆ 0.5054 ┆ 0.2869 ┆ 0.0001191 ┆ 0.2613 ┆ 0.5037 ┆ 0.756 ┆ 1.0 │ │ unrelated_2 ┆ 4,800 ┆ 0 ┆ 0.4995 ┆ 0.287 ┆ 0.000079 ┆ 0.2587 ┆ 0.4906 ┆ 0.7499 ┆ 0.9995 │ │ unrelated_3 ┆ 4,800 ┆ 0 ┆ 0.4993 ┆ 0.2897 ┆ 0.000129 ┆ 0.2495 ┆ 0.4895 ┆ 0.755 ┆ 0.9999 │ │ unrelated_4 ┆ 4,800 ┆ 0 ┆ 0.5016 ┆ 0.2901 ┆ 0.0001414 ┆ 0.2501 ┆ 0.4996 ┆ 0.7558 ┆ 0.9999 │ │ unrelated_5 ┆ 4,800 ┆ 0 ┆ 0.504 ┆ 0.2906 ┆ 0.000071 ┆ 0.2534 ┆ 0.5049 ┆ 0.7537 ┆ 0.9998 │ │ repeat_1 ┆ 4,800 ┆ 0 ┆ 0.5054 ┆ 0.2869 ┆ 0.0001191 ┆ 0.2613 ┆ 0.5037 ┆ 0.756 ┆ 1.0 │ │ var_gbm3 ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ var_gbm2 ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ var_gbm12 ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ var_gbm2_sq ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ var5 ┆ 4,800 ┆ 0 ┆ 0.7947 ┆ 0.404 ┆ 0.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 │ │ var_gbm1 ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ └─────────────┴───────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘
┌─────────────┬───────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ q25 ┆ q50 ┆ q75 ┆ max │ ╞═════════════╪═══════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡ │ index ┆ 4,800 ┆ 0 ┆ 5,085.0 ┆ 2,899.0 ┆ 1.0 ┆ 2,584.0 ┆ 5,107.0 ┆ 7,677.0 ┆ 9,997.0 │ │ _row_index_ ┆ 4,800 ┆ 0 ┆ 5,085.0 ┆ 2,899.0 ┆ 1.0 ┆ 2,584.0 ┆ 5,107.0 ┆ 7,677.0 ┆ 9,997.0 │ │ year ┆ 4,800 ┆ 0 ┆ 2,018.0 ┆ 1.422 ┆ 2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │ │ month ┆ 4,800 ┆ 0 ┆ 6.546 ┆ 3.428 ┆ 1.0 ┆ 4.0 ┆ 6.0 ┆ 10.0 ┆ 12.0 │ │ var2 ┆ 4,800 ┆ 0 ┆ 4.647 ┆ 3.23 ┆ 0.0 ┆ 2.0 ┆ 4.0 ┆ 7.0 ┆ 10.0 │ │ var3 ┆ 4,800 ┆ 0 ┆ 24.87 ┆ 12.92 ┆ 0.0 ┆ 14.0 ┆ 25.0 ┆ 36.0 ┆ 50.0 │ │ var4 ┆ 4,800 ┆ 0 ┆ 0.5062 ┆ 0.2877 ┆ 0.000027 ┆ 0.254 ┆ 0.5098 ┆ 0.752 ┆ 1.0 │ │ unrelated_1 ┆ 4,800 ┆ 0 ┆ 0.5031 ┆ 0.2849 ┆ 0.0001191 ┆ 0.2625 ┆ 0.5044 ┆ 0.7479 ┆ 1.0 │ │ unrelated_2 ┆ 4,800 ┆ 0 ┆ 0.4999 ┆ 0.2853 ┆ 0.000079 ┆ 0.2608 ┆ 0.4928 ┆ 0.7469 ┆ 0.9995 │ │ unrelated_3 ┆ 4,800 ┆ 0 ┆ 0.4988 ┆ 0.2894 ┆ 0.000129 ┆ 0.2508 ┆ 0.4891 ┆ 0.7511 ┆ 0.9999 │ │ unrelated_4 ┆ 4,800 ┆ 0 ┆ 0.4995 ┆ 0.2903 ┆ 0.0001414 ┆ 0.2466 ┆ 0.4982 ┆ 0.7551 ┆ 0.9999 │ │ unrelated_5 ┆ 4,800 ┆ 0 ┆ 0.504 ┆ 0.291 ┆ 0.000071 ┆ 0.2523 ┆ 0.5013 ┆ 0.756 ┆ 0.9999 │ │ repeat_1 ┆ 4,800 ┆ 0 ┆ 0.5031 ┆ 0.2849 ┆ 0.0001191 ┆ 0.2625 ┆ 0.5044 ┆ 0.7479 ┆ 1.0 │ │ var_gbm3 ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ var_gbm2 ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ var_gbm12 ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ var_gbm2_sq ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ │ var5 ┆ 4,800 ┆ 0 ┆ 0.7927 ┆ 0.4054 ┆ 0.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 │ │ var_gbm1 ┆ 4,800 ┆ 0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │ └─────────────┴───────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘
Look at the imputes | var_gbm1 == 1
┌─────────────┬───────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ q25 ┆ q50 ┆ q75 ┆ max │ ╞═════════════╪═══════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡ │ index ┆ 5,200 ┆ 0 ┆ 4,942.0 ┆ 2,870.0 ┆ 0.0 ┆ 2,459.0 ┆ 4,936.0 ┆ 7,381.0 ┆ 9,999.0 │ │ _row_index_ ┆ 5,200 ┆ 0 ┆ 4,942.0 ┆ 2,870.0 ┆ 0.0 ┆ 2,459.0 ┆ 4,936.0 ┆ 7,381.0 ┆ 9,999.0 │ │ year ┆ 5,200 ┆ 0 ┆ 2,018.0 ┆ 1.411 ┆ 2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │ │ month ┆ 5,200 ┆ 0 ┆ 6.496 ┆ 3.434 ┆ 1.0 ┆ 4.0 ┆ 6.0 ┆ 9.0 ┆ 12.0 │ │ var2 ┆ 5,200 ┆ 0 ┆ 5.295 ┆ 3.044 ┆ 0.0 ┆ 3.0 ┆ 5.0 ┆ 8.0 ┆ 10.0 │ │ var3 ┆ 5,200 ┆ 0 ┆ 25.35 ┆ 16.26 ┆ 0.0 ┆ 10.0 ┆ 26.0 ┆ 40.0 ┆ 50.0 │ │ var4 ┆ 5,200 ┆ 0 ┆ 0.5033 ┆ 0.2877 ┆ 0.0001044 ┆ 0.2562 ┆ 0.5073 ┆ 0.7531 ┆ 0.9999 │ │ unrelated_1 ┆ 5,200 ┆ 0 ┆ 0.4997 ┆ 0.2897 ┆ 0.0002485 ┆ 0.2486 ┆ 0.5036 ┆ 0.7506 ┆ 0.9998 │ │ unrelated_2 ┆ 5,200 ┆ 0 ┆ 0.5007 ┆ 0.2883 ┆ 0.000049 ┆ 0.249 ┆ 0.5038 ┆ 0.7457 ┆ 0.9995 │ │ unrelated_3 ┆ 5,200 ┆ 0 ┆ 0.4991 ┆ 0.2879 ┆ 0.000319 ┆ 0.2531 ┆ 0.4997 ┆ 0.747 ┆ 0.9999 │ │ unrelated_4 ┆ 5,200 ┆ 0 ┆ 0.4998 ┆ 0.2874 ┆ 0.0001329 ┆ 0.2508 ┆ 0.5011 ┆ 0.7492 ┆ 1.0 │ │ unrelated_5 ┆ 5,200 ┆ 0 ┆ 0.4939 ┆ 0.2875 ┆ 0.0001807 ┆ 0.2439 ┆ 0.4883 ┆ 0.7467 ┆ 0.9999 │ │ repeat_1 ┆ 5,200 ┆ 0 ┆ 0.4997 ┆ 0.2897 ┆ 0.0002485 ┆ 0.2486 ┆ 0.5036 ┆ 0.7506 ┆ 0.9998 │ │ var_gbm3 ┆ 5,200 ┆ 0 ┆ -4.289 ┆ 13.31 ┆ -55.35 ┆ -11.74 ┆ -1.019 ┆ 5.153 ┆ 26.21 │ │ var_gbm2 ┆ 5,200 ┆ 0 ┆ -4.33 ┆ 13.48 ┆ -55.35 ┆ -12.01 ┆ -0.7496 ┆ 5.186 ┆ 26.21 │ │ var_gbm12 ┆ 5,200 ┆ 0 ┆ -4.33 ┆ 13.48 ┆ -55.35 ┆ -12.01 ┆ -0.7496 ┆ 5.186 ┆ 26.21 │ │ var_gbm2_sq ┆ 5,200 ┆ 0 ┆ 200.3 ┆ 343.7 ┆ 0.0 ┆ 9.33 ┆ 63.08 ┆ 213.2 ┆ 3,064.0 │ │ var5 ┆ 5,200 ┆ 0 ┆ 0.2281 ┆ 0.4197 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 1.0 │ │ var_gbm1 ┆ 5,200 ┆ 0 ┆ 1.0 ┆ 0.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 │ └─────────────┴───────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘
┌─────────────┬───────┬─────────────┬─────────┬─────────┬───────────┬─────────┬─────────┬─────────┬─────────┐ │ Variable ┆ n ┆ n (missing) ┆ mean ┆ std ┆ min ┆ q25 ┆ q50 ┆ q75 ┆ max │ ╞═════════════╪═══════╪═════════════╪═════════╪═════════╪═══════════╪═════════╪═════════╪═════════╪═════════╡ │ index ┆ 5,200 ┆ 0 ┆ 4,921.0 ┆ 2,874.0 ┆ 0.0 ┆ 2,426.0 ┆ 4,884.0 ┆ 7,362.0 ┆ 9,999.0 │ │ _row_index_ ┆ 5,200 ┆ 0 ┆ 4,921.0 ┆ 2,874.0 ┆ 0.0 ┆ 2,426.0 ┆ 4,884.0 ┆ 7,362.0 ┆ 9,999.0 │ │ year ┆ 5,200 ┆ 0 ┆ 2,018.0 ┆ 1.41 ┆ 2,016.0 ┆ 2,017.0 ┆ 2,018.0 ┆ 2,019.0 ┆ 2,020.0 │ │ month ┆ 5,200 ┆ 0 ┆ 6.484 ┆ 3.436 ┆ 1.0 ┆ 4.0 ┆ 6.0 ┆ 9.0 ┆ 12.0 │ │ var2 ┆ 5,200 ┆ 0 ┆ 5.282 ┆ 3.052 ┆ 0.0 ┆ 3.0 ┆ 5.0 ┆ 8.0 ┆ 10.0 │ │ var3 ┆ 5,200 ┆ 0 ┆ 25.33 ┆ 16.26 ┆ 0.0 ┆ 10.0 ┆ 25.0 ┆ 40.0 ┆ 50.0 │ │ var4 ┆ 5,200 ┆ 0 ┆ 0.5052 ┆ 0.288 ┆ 0.0001044 ┆ 0.257 ┆ 0.5073 ┆ 0.7563 ┆ 0.9999 │ │ unrelated_1 ┆ 5,200 ┆ 0 ┆ 0.5018 ┆ 0.2916 ┆ 0.0002485 ┆ 0.2478 ┆ 0.5008 ┆ 0.7572 ┆ 0.9998 │ │ unrelated_2 ┆ 5,200 ┆ 0 ┆ 0.5003 ┆ 0.2898 ┆ 0.000049 ┆ 0.247 ┆ 0.5028 ┆ 0.7495 ┆ 0.9995 │ │ unrelated_3 ┆ 5,200 ┆ 0 ┆ 0.4995 ┆ 0.2882 ┆ 0.000319 ┆ 0.2514 ┆ 0.499 ┆ 0.7497 ┆ 0.9999 │ │ unrelated_4 ┆ 5,200 ┆ 0 ┆ 0.5017 ┆ 0.2872 ┆ 0.0001329 ┆ 0.2551 ┆ 0.5036 ┆ 0.7501 ┆ 1.0 │ │ unrelated_5 ┆ 5,200 ┆ 0 ┆ 0.4939 ┆ 0.2871 ┆ 0.0001807 ┆ 0.2452 ┆ 0.4903 ┆ 0.7452 ┆ 0.9994 │ │ repeat_1 ┆ 5,200 ┆ 0 ┆ 0.5018 ┆ 0.2916 ┆ 0.0002485 ┆ 0.2478 ┆ 0.5008 ┆ 0.7572 ┆ 0.9998 │ │ var_gbm3 ┆ 5,200 ┆ 0 ┆ -4.349 ┆ 13.42 ┆ -55.35 ┆ -11.78 ┆ -0.8137 ┆ 5.182 ┆ 26.21 │ │ var_gbm2 ┆ 5,200 ┆ 0 ┆ -4.326 ┆ 13.48 ┆ -55.35 ┆ -11.78 ┆ -0.8498 ┆ 5.395 ┆ 26.21 │ │ var_gbm12 ┆ 5,200 ┆ 0 ┆ -4.326 ┆ 13.48 ┆ -55.35 ┆ -11.78 ┆ -0.8498 ┆ 5.395 ┆ 26.21 │ │ var_gbm2_sq ┆ 5,200 ┆ 0 ┆ 200.5 ┆ 342.6 ┆ 0.0 ┆ 10.42 ┆ 63.37 ┆ 209.5 ┆ 3,064.0 │ │ var5 ┆ 5,200 ┆ 0 ┆ 0.2307 ┆ 0.4213 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 1.0 │ │ var_gbm1 ┆ 5,200 ┆ 0 ┆ 1.0 ┆ 0.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 ┆ 1.0 │ └─────────────┴───────┴─────────────┴─────────┴─────────┴───────────┴─────────┴─────────┴─────────┴─────────┘