{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T14:36:26.401148Z","iopub.execute_input":"2022-07-14T14:36:26.401448Z","iopub.status.idle":"2022-07-14T14:36:26.406087Z","shell.execute_reply.started":"2022-07-14T14:36:26.401410Z","shell.execute_reply":"2022-07-14T14:36:26.404973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import library for automated feature engineering\nimport featuretools as ft\nimport gc\nfrom os.path import join as pjoin\nfrom os import cpu_count\nimport warnings\nwarnings.simplefilter('ignore')","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2022-07-14T14:36:26.420829Z","iopub.execute_input":"2022-07-14T14:36:26.421223Z","iopub.status.idle":"2022-07-14T14:36:26.427626Z","shell.execute_reply.started":"2022-07-14T14:36:26.421163Z","shell.execute_reply":"2022-07-14T14:36:26.426760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define filepaths\ndata_dir = '../input'\n\nfilepaths = {\n    'data_desc': pjoin(data_dir, 'HomeCredit_columns_description.csv'),\n    'app_train': pjoin(data_dir, 'application_train.csv'),\n    'app_test': pjoin(data_dir, 'application_test.csv'),\n    'bureau': pjoin(data_dir, 'bureau.csv'),\n    'bureau_bl': pjoin(data_dir, 'bureau_balance.csv'),\n    'credit_bl': pjoin(data_dir, 'credit_card_balance.csv'),\n    'install_pays': pjoin(data_dir, 'installments_payments.csv'),\n    'pc_balance': pjoin(data_dir, 'POS_CASH_balance.csv'),\n    'app_prev': pjoin(data_dir, 'previous_application.csv'),\n    \n}\n\nfilepaths","metadata":{"_uuid":"bfed6221af6eeb7788c248ea6718d1976892a7d8","execution":{"iopub.status.busy":"2022-07-14T14:36:26.434835Z","iopub.execute_input":"2022-07-14T14:36:26.435198Z","iopub.status.idle":"2022-07-14T14:36:26.450668Z","shell.execute_reply.started":"2022-07-14T14:36:26.435126Z","shell.execute_reply":"2022-07-14T14:36:26.449777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load Main table (Applications)","metadata":{"_uuid":"05497ca29c3e0e215c84e19bbcf663d3048b88c1"}},{"cell_type":"code","source":"# first X rows are taken for faster calculations, substitute this by whole dataset\nnrows = 30000\n\n# load main datasets\ndf_train = pd.read_csv(\n    filepaths['app_train'], \n    low_memory=False, engine='c',\n    nrows=nrows,\n)\ndf_test = pd.read_csv(\n    filepaths['app_test'], \n    low_memory=False, \n    engine='c',\n)\n\n# concat dataframes together, check shapes\nprint(df_train.shape, df_test.shape)\ndf_joint = pd.concat([df_train, df_test])\nprint(df_joint.shape)\n\ndel df_train, df_test\ngc.collect()\n\nprint('memory usage: {:.2f} MB'.format(df_joint.memory_usage().sum() / 2**20))\n\nint_cols = df_joint.select_dtypes(include=[np.int64]).columns\nfloat_cols = df_joint.select_dtypes(include=[np.float64]).columns \n\ndf_joint[int_cols] = df_joint[int_cols].astype(np.int32)\ndf_joint[float_cols] = df_joint[float_cols].astype(np.float32)\n\nprint('memory usage: {:.2f} MB'.format(df_joint.memory_usage().sum() / 2**20))\n\nprint(df_joint.dtypes.value_counts())\n\n# df_joint.set_index('SK_ID_CURR', inplace=True, drop=True)\ntarget_col = 'TARGET'\n\n# check sample\ndf_joint.head()","metadata":{"_uuid":"d2dd988296912d2d22f4cdf0acc873d82d83a712","execution":{"iopub.status.busy":"2022-07-14T14:36:26.455075Z","iopub.execute_input":"2022-07-14T14:36:26.455568Z","iopub.status.idle":"2022-07-14T14:36:31.083746Z","shell.execute_reply.started":"2022-07-14T14:36:26.455524Z","shell.execute_reply":"2022-07-14T14:36:31.082915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load previous applications table","metadata":{"_uuid":"e7b0d63ef1f8708e92adbe5d5b3c52b672b058cb"}},{"cell_type":"code","source":"df_app_prev = pd.read_csv(\n    filepaths['app_prev'], \n    engine='c', \n    low_memory=False,\n    # first X*3 rows are taken for faster calculations, substitute this by whole dataset\n    nrows=nrows*3,\n)\nprint(df_app_prev.shape)\n\n# optimize memory usage\nprint('memory usage: {:.2f} MB'.format(df_app_prev.memory_usage().sum() / 2**20))\n\nint_cols = df_app_prev.select_dtypes(include=[np.int64]).columns\nfloat_cols = df_app_prev.select_dtypes(include=[np.float64]).columns \n\ndf_app_prev[int_cols] = df_app_prev[int_cols].astype(np.int32)\ndf_app_prev[float_cols] = df_app_prev[float_cols].astype(np.float32)\n\nprint('memory usage: {:.2f} MB'.format(df_app_prev.memory_usage().sum() / 2**20))\nprint(df_app_prev.dtypes.value_counts())\n\n# check sample\ndf_app_prev.head()","metadata":{"_uuid":"61b9d6eff0632cbd7cc4c5a1988ee97b8ba51554","execution":{"iopub.status.busy":"2022-07-14T14:36:31.084748Z","iopub.execute_input":"2022-07-14T14:36:31.085068Z","iopub.status.idle":"2022-07-14T14:36:32.296424Z","shell.execute_reply.started":"2022-07-14T14:36:31.085014Z","shell.execute_reply":"2022-07-14T14:36:32.295339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# substitute encoded NaNs by \"real\" np.nan\ndf_app_prev[\n    [c for c in df_app_prev.columns if c.startswith('DAYS_')]\n] = df_app_prev[\n    [c for c in df_app_prev.columns if c.startswith('DAYS_')]\n].replace(365243, np.nan)","metadata":{"_uuid":"97f15cc36ec511dffbb69a4a46b6ec6eb38757ad","execution":{"iopub.status.busy":"2022-07-14T14:36:32.297650Z","iopub.execute_input":"2022-07-14T14:36:32.297917Z","iopub.status.idle":"2022-07-14T14:36:32.318771Z","shell.execute_reply.started":"2022-07-14T14:36:32.297866Z","shell.execute_reply":"2022-07-14T14:36:32.317953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## TABLE JOINING (FEATURE TOOLS)","metadata":{"_uuid":"d001efa888d3cb9a323d8bc2c8b8f8df4ba60a7e"}},{"cell_type":"markdown","source":"An `EntitySet` is a collection of entities and the relationships between them. \n\nThey are useful for preparing raw, structured datasets for feature engineering. \n<br>While many functions in Featuretools take `entities` and `relationships` as separate arguments,\n<br>it is recommended to create an `EntitySet`, so you can more easily manipulate your data as needed.","metadata":{"_uuid":"56f6d3b4aebea7b9cb78e2c9ea6675d218d402bd"}},{"cell_type":"code","source":"# initialize entityset\nes = ft.EntitySet('application_data')\n\n# add entities (application table itself)\nes.entity_from_dataframe(\n    entity_id='apps', # define entity id\n    dataframe=df_joint.drop('TARGET', axis=1), # select underlying data\n    index='SK_ID_CURR', # define unique index column\n    # specify some datatypes manually (if needed)\n    variable_types={\n        f: ft.variable_types.Categorical \n        for f in df_joint.columns if f.startswith('FLAG_')\n    }\n)","metadata":{"_uuid":"6aaa751a959d7892181afa4917d06fddc38a29af","execution":{"iopub.status.busy":"2022-07-14T14:36:32.320018Z","iopub.execute_input":"2022-07-14T14:36:32.320280Z","iopub.status.idle":"2022-07-14T14:36:33.587368Z","shell.execute_reply.started":"2022-07-14T14:36:32.320230Z","shell.execute_reply":"2022-07-14T14:36:33.586535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# trick! substitute relative days by absolute date shift\n# to be used as \"true\" time_index\n# however, prohibit datetime features (like month or year) as they're irrelevant\ntoday = pd.to_datetime('2018-06-11')\ndf_app_prev['DAYS_DECISION'] = today + pd.to_timedelta(df_app_prev['DAYS_DECISION'], unit='d')","metadata":{"_uuid":"e9b6fce7a880d2b6b127762e3fb9ef8ef8e278e2","execution":{"iopub.status.busy":"2022-07-14T14:36:33.588909Z","iopub.execute_input":"2022-07-14T14:36:33.589256Z","iopub.status.idle":"2022-07-14T14:36:33.601884Z","shell.execute_reply.started":"2022-07-14T14:36:33.589192Z","shell.execute_reply":"2022-07-14T14:36:33.600893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add entities (previous applications table)\nes = es.entity_from_dataframe(\n    entity_id = 'prev_apps', \n    dataframe = df_app_prev,\n    index = 'SK_ID_PREV',\n    time_index = 'DAYS_DECISION',\n    variable_types={\n        f: ft.variable_types.Categorical \n        for f in df_app_prev.columns if f.startswith('NFLAG_')\n    }\n)","metadata":{"_uuid":"a864586cb39db479be44e6578b9fad90f6d1ad40","execution":{"iopub.status.busy":"2022-07-14T14:36:33.603615Z","iopub.execute_input":"2022-07-14T14:36:33.603998Z","iopub.status.idle":"2022-07-14T14:36:34.474216Z","shell.execute_reply.started":"2022-07-14T14:36:33.603928Z","shell.execute_reply":"2022-07-14T14:36:34.473316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the call to `entity_from_dataframe`, we specified three important parameters\n\n- The `index` parameter specifies the column that uniquely identifies rows in the dataframe\n- The `time_index` parameter tells Featuretools when the data was \"created\".\n- The `variable_types` parameter indicates that some columns should be interpreted as a Categorical variable, \neven though it just an integer in the underlying data.\n","metadata":{"_uuid":"0c46dfb476ae0cfe30f20bbf6064112421131945"}},{"cell_type":"markdown","source":"**Adding a Relationship**\n\nWith two entities in our entity set, we can add a relationship between them.\n\nWe want to relate these two entities by the columns called “SK_ID_CURR” in each entity. \n<br>Each application has multiple previous applications associated with it, \n<br>so it is called it the parent entity, while the previous applications  entity is known as the child entity. \n<br>When specifying relationships we list the variable in the parent entity first. Note that each ft.Relationship must denote a **one-to-many relationship** rather than a relationship which is one-to-one or many-to-many.","metadata":{"_uuid":"dad074504a276c381adb7c06b45c3fcb8031e75b"}},{"cell_type":"code","source":"# add relationships\nr_app_cur_to_app_prev = ft.Relationship(\n    es['apps']['SK_ID_CURR'],\n    es['prev_apps']['SK_ID_CURR']\n)\n\n# Add the relationship to the entity set\nes = es.add_relationship(r_app_cur_to_app_prev)\n\n# check constructed entity set\nes","metadata":{"_uuid":"f3a27a1a9b2734aa8b69551ed972afc1f3f2dcfa","execution":{"iopub.status.busy":"2022-07-14T14:36:34.475386Z","iopub.execute_input":"2022-07-14T14:36:34.475708Z","iopub.status.idle":"2022-07-14T14:36:37.478573Z","shell.execute_reply.started":"2022-07-14T14:36:34.475640Z","shell.execute_reply":"2022-07-14T14:36:37.477984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check created entities\nes['apps']","metadata":{"_uuid":"283f5ea0ee51302a246ccbd09927156840715f0a","execution":{"iopub.status.busy":"2022-07-14T14:36:37.479548Z","iopub.execute_input":"2022-07-14T14:36:37.479939Z","iopub.status.idle":"2022-07-14T14:36:37.486267Z","shell.execute_reply.started":"2022-07-14T14:36:37.479886Z","shell.execute_reply":"2022-07-14T14:36:37.485580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check created entities\nes['prev_apps']","metadata":{"_uuid":"c4fdc779aa2e449b97396be1c6b77a069ac1498c","execution":{"iopub.status.busy":"2022-07-14T14:36:37.487356Z","iopub.execute_input":"2022-07-14T14:36:37.487587Z","iopub.status.idle":"2022-07-14T14:36:37.497330Z","shell.execute_reply.started":"2022-07-14T14:36:37.487546Z","shell.execute_reply":"2022-07-14T14:36:37.496402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Feature primitives**\n\nFeature primitives are the building blocks of Featuretools. They define individual computations that can be applied to raw datasets to create new features. Because a primitive only constrains the input and output data types, they can be applied across datasets and can stack to create new calculations.\n\n**Why primitives?**\n\nThe space of potential functions that humans use to create a feature is expansive. By breaking common feature engineering calculations down into primitive components, we are able to capture the underlying structure of the features humans create today.\n\nSee [documentation](https://docs.featuretools.com/automated_feature_engineering/primitives.html) for further details","metadata":{"_uuid":"9d8e2341779bae3ef72ebfba2b692b039a381739"}},{"cell_type":"code","source":"# inspect list of all built-in primitives for feature construction\nft.list_primitives()","metadata":{"_uuid":"ddf1febe8211b41f06b1133ebe9e5aee9a5821b6","execution":{"iopub.status.busy":"2022-07-14T14:36:37.498511Z","iopub.execute_input":"2022-07-14T14:36:37.499305Z","iopub.status.idle":"2022-07-14T14:36:37.531025Z","shell.execute_reply.started":"2022-07-14T14:36:37.499243Z","shell.execute_reply":"2022-07-14T14:36:37.530201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Handling time**\n\nWhen performing feature engineering to learn a model to predict the future, \n<br>the value to predict will be associated with a time. \nIn this case, it is paramount to only incorporate data prior to this `“cutoff time”` when calculating the feature values.\n\nFeaturetools is designed to take time into consideration when required. \n<br>By specifying a cutoff time, we can control what portions of the data are used when calculating features.\n\nWe can specify the time for each instance of the `target_entity` to calculate features. \n<br>The timestamp represents the last time data can be used for calculating features. This is specified using a dataframe of cutoff times. \n\nRead more [here](https://docs.featuretools.com/automated_feature_engineering/handling_time.html)","metadata":{"_uuid":"8a64a338e0777f342c0adfd43dc871bd24b3d175"}},{"cell_type":"code","source":"%%time\n# define cut-off times\n# cut-off times are the \"right\" time values to be used for feature calculation without future leaks\n# none in our case\n\ncutoff_times = pd.DataFrame(df_joint.SK_ID_CURR)\ncutoff_times['time'] = today\n\n# add last_time_index\nes.add_last_time_indexes()","metadata":{"_uuid":"b98dfe6a6c0566179494980ccfe7c81b4b7243cc","execution":{"iopub.status.busy":"2022-07-14T14:36:37.532479Z","iopub.execute_input":"2022-07-14T14:36:37.533082Z","iopub.status.idle":"2022-07-14T14:36:57.601489Z","shell.execute_reply.started":"2022-07-14T14:36:37.533014Z","shell.execute_reply":"2022-07-14T14:36:57.600618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Running DFS with training windows**\n\nTraining windows are an extension of cutoff times: starting from the cutoff time and moving backwards through time, only data within that window of time will be used to calculate features. We will use events **within 2 month time window**","metadata":{"_uuid":"1b166c736d4a440729f398716238c8195fdf9982"}},{"cell_type":"code","source":"# see feature set definitions (no actual computations yet)\n# used for faster prototyping\nfeature_defs = ft.dfs(\n    entityset=es, \n    target_entity=\"apps\", \n    features_only=True,\n    agg_primitives=[\n        \"avg_time_between\",\n        \"time_since_last\", \n        \"num_unique\", \n        \"mean\", \n        \"sum\", \n    ],\n    trans_primitives=[\n        \"time_since_previous\",\n        #\"add\",\n    ],\n    max_depth=1,\n    cutoff_time=cutoff_times,\n    training_window=ft.Timedelta(60, \"d\"), # use only last X days in computations\n    max_features=1000,\n    chunk_size=10000,\n    verbose=True,\n)\n\n# check what's been created so far\nfeature_defs","metadata":{"_uuid":"a86f05337b38a153691098956a8e163b8906520f","execution":{"iopub.status.busy":"2022-07-14T14:36:57.602503Z","iopub.execute_input":"2022-07-14T14:36:57.602716Z","iopub.status.idle":"2022-07-14T14:36:57.636481Z","shell.execute_reply.started":"2022-07-14T14:36:57.602684Z","shell.execute_reply":"2022-07-14T14:36:57.635638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate actual features\nfm, feature_defs = ft.dfs(\n    entityset=es, \n    target_entity=\"apps\", \n    #features_only=True,\n    agg_primitives=[\n        \"avg_time_between\",\n        \"time_since_last\", \n        \"num_unique\", \n        \"mean\", \n        \"sum\", \n    ],\n    trans_primitives=[\n        \"time_since_previous\",\n        #\"add\",\n    ],\n    max_depth=1,\n    cutoff_time=cutoff_times,\n    training_window=ft.Timedelta(60, \"d\"),\n    max_features=1000,\n    chunk_size=4000,\n    verbose=True,\n)","metadata":{"_uuid":"3edfebc5bc1db6e43f99930270c15ee6bdc69854","execution":{"iopub.status.busy":"2022-07-14T14:36:57.637593Z","iopub.execute_input":"2022-07-14T14:36:57.637827Z","iopub.status.idle":"2022-07-14T14:37:14.167083Z","shell.execute_reply.started":"2022-07-14T14:36:57.637787Z","shell.execute_reply":"2022-07-14T14:37:14.166399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check sample of extracted features\nfm = fm.drop_duplicates()\nprint(fm.shape)\nfm[50:100]","metadata":{"_uuid":"358c224f0121bd2e56282c26543b4bdb40b1912d","execution":{"iopub.status.busy":"2022-07-14T14:37:14.168194Z","iopub.execute_input":"2022-07-14T14:37:14.168442Z","iopub.status.idle":"2022-07-14T14:37:15.281348Z","shell.execute_reply.started":"2022-07-14T14:37:14.168404Z","shell.execute_reply":"2022-07-14T14:37:15.280584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define validation strategy and run a model atop of generated features\nfrom sklearn.model_selection import StratifiedKFold\nfrom lightgbm import LGBMClassifier\nfrom sklearn.metrics import roc_auc_score\n\nskf = StratifiedKFold(5, random_state=42)\n\nprint(fm.dtypes.value_counts())\n# label-encode categorical variables\nfor c in fm.select_dtypes(include=['object']).columns:\n    fm[c], _ = pd.factorize(fm[c])\n\nprint(fm.dtypes.value_counts())","metadata":{"_uuid":"7c7f07b260b842e7623abfeaa2c4edcabc7d747d","execution":{"iopub.status.busy":"2022-07-14T14:37:15.282483Z","iopub.execute_input":"2022-07-14T14:37:15.282709Z","iopub.status.idle":"2022-07-14T14:37:15.840240Z","shell.execute_reply.started":"2022-07-14T14:37:15.282669Z","shell.execute_reply":"2022-07-14T14:37:15.838612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define train/test datasets\nidx_train = df_joint[~df_joint.TARGET.isnull()].SK_ID_CURR.tolist()\nidx_test = df_joint[df_joint.TARGET.isnull()].SK_ID_CURR.tolist()\n\nfm_train = fm[fm.index.isin(idx_train)]\nfm_test = fm[fm.index.isin(idx_test)]\nfm_train.shape, fm_test.shape","metadata":{"_uuid":"04342a87773444f02b31d078b718a7c49755ee3f","execution":{"iopub.status.busy":"2022-07-14T14:37:15.842790Z","iopub.execute_input":"2022-07-14T14:37:15.843315Z","iopub.status.idle":"2022-07-14T14:37:15.920313Z","shell.execute_reply.started":"2022-07-14T14:37:15.843258Z","shell.execute_reply":"2022-07-14T14:37:15.919523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\n\n# define lightgbm params\nparams_lgb = {\n    'application': 'binary',\n    'boosting': 'gbdt',\n    'learning_rate': 0.03,\n    'num_leaves': 31,\n    'max_depth': 7,\n    'early_stopping_round': 10,\n    \n    'num_iteration': 2500, \n    'colsample_bytree':.95, \n    'subsample':.87, \n\n    'reg_alpha': 0.04, \n    'reg_lambda': 0.07, \n    'min_split_gain': 0.022, \n    'min_child_weight': 5,\n}\n\n# make dataset\ndata_tr = lgb.Dataset(\n    data=fm_train,\n    label=df_joint[:nrows][target_col],\n)\n\n# run cross-validation\ncv_results = lgb.cv(\n    params_lgb, \n    data_tr, \n    metrics=['auc'], \n    folds=skf.split(fm_train, df_joint[:nrows][target_col]), \n    verbose_eval=25,\n)","metadata":{"_uuid":"e73b0d5f94d38d5c21f14bab77f2a33ea2edfad3","execution":{"iopub.status.busy":"2022-07-14T14:37:15.921773Z","iopub.execute_input":"2022-07-14T14:37:15.922137Z","iopub.status.idle":"2022-07-14T14:37:36.831534Z","shell.execute_reply.started":"2022-07-14T14:37:15.922068Z","shell.execute_reply":"2022-07-14T14:37:36.830482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# train model\nparams_lgb['num_iteration'] = int(len(cv_results['auc-mean']) * 5/4)\n\nmodel = lgb.train(\n    params_lgb, \n    data_tr,  \n)","metadata":{"_uuid":"813513a5ff525b8369934406d6b6c80fa5bdfa1b","execution":{"iopub.status.busy":"2022-07-14T14:37:36.833925Z","iopub.execute_input":"2022-07-14T14:37:36.834151Z","iopub.status.idle":"2022-07-14T14:37:42.092386Z","shell.execute_reply.started":"2022-07-14T14:37:36.834117Z","shell.execute_reply":"2022-07-14T14:37:42.091745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict for test\ndf_joint.loc[df_joint.SK_ID_CURR.isin(idx_test), target_col] = model.predict(fm_test)\n\n# sample submission\ndf_joint.loc[df_joint.SK_ID_CURR.isin(idx_test), ['SK_ID_CURR', target_col]].to_csv(\n    'featuretools_example_subm.csv',\n    index=False\n)","metadata":{"_uuid":"f4749acb21d09643469a2026e170f86e41503548","execution":{"iopub.status.busy":"2022-07-14T14:37:42.093769Z","iopub.execute_input":"2022-07-14T14:37:42.094367Z","iopub.status.idle":"2022-07-14T14:37:43.297744Z","shell.execute_reply.started":"2022-07-14T14:37:42.094310Z","shell.execute_reply":"2022-07-14T14:37:43.296955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Hope this small example inspired you to try this approach yourself! \n<br>Have fun - add custom features, add more tables and relationships, gather hands on experience**\n\n**Likes and comments are welcome :)**","metadata":{"_uuid":"4af27dc2721bfa513d8b4cc9f1cb57699f60ff17"}},{"cell_type":"code","source":"","metadata":{"_uuid":"1422b47bd70b7d89bbcbc246b4cb3120ab6c9215","collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}