{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# H&M - Exploration & Baseline\n\n### Data (CSV files)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndf_articles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv', dtype={'article_id': str})\ndf_customers = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/customers.csv')\ndf_train = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', dtype={'article_id': str})\ndf_sample_submission = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv',\n                                   dtype={'article_id': str})","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-11T12:01:30.912880Z","iopub.execute_input":"2022-02-11T12:01:30.913259Z","iopub.status.idle":"2022-02-11T12:03:00.197963Z","shell.execute_reply.started":"2022-02-11T12:01:30.913155Z","shell.execute_reply":"2022-02-11T12:03:00.197165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_articles.info()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:00.199431Z","iopub.execute_input":"2022-02-11T12:03:00.199651Z","iopub.status.idle":"2022-02-11T12:03:00.397420Z","shell.execute_reply.started":"2022-02-11T12:03:00.199626Z","shell.execute_reply":"2022-02-11T12:03:00.396567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customers.info()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:00.398696Z","iopub.execute_input":"2022-02-11T12:03:00.399366Z","iopub.status.idle":"2022-02-11T12:03:01.017343Z","shell.execute_reply.started":"2022-02-11T12:03:00.399329Z","shell.execute_reply":"2022-02-11T12:03:01.016466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:01.018534Z","iopub.execute_input":"2022-02-11T12:03:01.018764Z","iopub.status.idle":"2022-02-11T12:03:01.028273Z","shell.execute_reply.started":"2022-02-11T12:03:01.018736Z","shell.execute_reply":"2022-02-11T12:03:01.027743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_submission.info()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:01.030078Z","iopub.execute_input":"2022-02-11T12:03:01.030509Z","iopub.status.idle":"2022-02-11T12:03:01.340295Z","shell.execute_reply.started":"2022-02-11T12:03:01.030452Z","shell.execute_reply":"2022-02-11T12:03:01.339489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = len(list(set(df_train.customer_id.unique()) - set(df_sample_submission.customer_id.unique())))\nprint('Customers that have bought at least once during training and do not appear in the submission file:', n)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:01.341652Z","iopub.execute_input":"2022-02-11T12:03:01.342037Z","iopub.status.idle":"2022-02-11T12:03:11.132492Z","shell.execute_reply.started":"2022-02-11T12:03:01.341983Z","shell.execute_reply":"2022-02-11T12:03:11.131549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = len(list(set(df_sample_submission.customer_id.unique()) - set(df_train.customer_id.unique())))\nprint('Customers that have not bought during training and appear in the submission file:', n)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:11.133567Z","iopub.execute_input":"2022-02-11T12:03:11.134231Z","iopub.status.idle":"2022-02-11T12:03:20.623569Z","shell.execute_reply.started":"2022-02-11T12:03:11.134200Z","shell.execute_reply":"2022-02-11T12:03:20.622846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All customers from the submission file appear in the train dataset. However, we find some new customers (~0.7%) that do not appear in the training dataset. We will need to deal with a few customers customers with no historical transaction.\n\n\n#### Adding sales information to each article","metadata":{}},{"cell_type":"code","source":"sales_product = df_train.article_id.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:20.624482Z","iopub.execute_input":"2022-02-11T12:03:20.624955Z","iopub.status.idle":"2022-02-11T12:03:28.403477Z","shell.execute_reply.started":"2022-02-11T12:03:20.624918Z","shell.execute_reply":"2022-02-11T12:03:28.402792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_articles = df_articles.merge(sales_product.rename('sales'), left_on='article_id', right_index=True, how='left')","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:28.404525Z","iopub.execute_input":"2022-02-11T12:03:28.404866Z","iopub.status.idle":"2022-02-11T12:03:28.562474Z","shell.execute_reply.started":"2022-02-11T12:03:28.404837Z","shell.execute_reply":"2022-02-11T12:03:28.561820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_articles['sales'] = df_articles['sales'].fillna(value=0)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:28.563368Z","iopub.execute_input":"2022-02-11T12:03:28.563935Z","iopub.status.idle":"2022-02-11T12:03:28.569425Z","shell.execute_reply.started":"2022-02-11T12:03:28.563901Z","shell.execute_reply":"2022-02-11T12:03:28.568720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nsns.set(rc={'figure.figsize':(20.7,12.27)})\n\nsns.violinplot(data=df_articles, x='index_group_name', y='sales')\nplt.title('Sales distribution per index group')","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:28.570482Z","iopub.execute_input":"2022-02-11T12:03:28.571131Z","iopub.status.idle":"2022-02-11T12:03:30.606844Z","shell.execute_reply.started":"2022-02-11T12:03:28.571098Z","shell.execute_reply":"2022-02-11T12:03:30.605986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(data=df_articles, x='index_group_name', y='sales', showfliers=False)\nplt.title('Sales distribution per index group - no outliers')","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:30.608214Z","iopub.execute_input":"2022-02-11T12:03:30.608754Z","iopub.status.idle":"2022-02-11T12:03:31.029260Z","shell.execute_reply.started":"2022-02-11T12:03:30.608705Z","shell.execute_reply":"2022-02-11T12:03:31.028668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Adding article information to transactions data","metadata":{}},{"cell_type":"code","source":"df_train = df_train.merge(on='article_id', \n                          how='left', \n                          right=df_articles[['article_id', 'index_group_name', 'product_group_name']])","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:31.030310Z","iopub.execute_input":"2022-02-11T12:03:31.030705Z","iopub.status.idle":"2022-02-11T12:03:43.222331Z","shell.execute_reply.started":"2022-02-11T12:03:31.030658Z","shell.execute_reply":"2022-02-11T12:03:43.221426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['t_dat'] = pd.to_datetime(df_train['t_dat'])\nprint(f'Perimeter training: {df_train.t_dat.min(), df_train.t_dat.max()}')","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:43.225474Z","iopub.execute_input":"2022-02-11T12:03:43.225798Z","iopub.status.idle":"2022-02-11T12:03:50.371355Z","shell.execute_reply.started":"2022-02-11T12:03:43.225764Z","shell.execute_reply":"2022-02-11T12:03:50.370302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"On the Evaluation section we can read:\n> For each customer_id observed in the training data, you may predict up to 12 labels for the article_id, which is the predicted items a customer will buy in the next 7-day period after the training time period.\n\n\nThis means that we are going to predict the purchases from the customers from 2020-09-23 to 2020-09-30. Can we use seasonality to make more accurate predictions? It looks like an interesting thing to take into account.","metadata":{}},{"cell_type":"code","source":"df_train['month_year'] = df_train['t_dat'].dt.to_period('M').astype(str)   # Adding YYYY-MM","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:03:50.372683Z","iopub.execute_input":"2022-02-11T12:03:50.372954Z","iopub.status.idle":"2022-02-11T12:07:19.957506Z","shell.execute_reply.started":"2022-02-11T12:03:50.372923Z","shell.execute_reply":"2022-02-11T12:07:19.956560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(data=df_train.month_year.value_counts().sort_index())\nplt.title('Monthly sales during training period')","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:07:19.959221Z","iopub.execute_input":"2022-02-11T12:07:19.959543Z","iopub.status.idle":"2022-02-11T12:07:26.263735Z","shell.execute_reply.started":"2022-02-11T12:07:19.959488Z","shell.execute_reply":"2022-02-11T12:07:26.262774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I wonder if the distribution of sales for different categories changes from september to the whole period.","metadata":{}},{"cell_type":"code","source":"df_september = df_train.query('month_year == \"2019-09\" or month_year == \"2019-09\" or month_year == \"2020-09\"')","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:07:26.264940Z","iopub.execute_input":"2022-02-11T12:07:26.265460Z","iopub.status.idle":"2022-02-11T12:07:36.678225Z","shell.execute_reply.started":"2022-02-11T12:07:26.265426Z","shell.execute_reply":"2022-02-11T12:07:36.677319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_september.index_group_name.value_counts(normalize=True).plot(kind='bar', color='skyblue', position=1, \n                                                               width = .25, label='season')\ndf_train.index_group_name.value_counts(normalize=True).plot(kind='bar', color='red', position=0, \n                                                              width = .25, label='train')\nplt.title('Product group sales distribution - Train VS September')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:07:36.679601Z","iopub.execute_input":"2022-02-11T12:07:36.679908Z","iopub.status.idle":"2022-02-11T12:07:41.984395Z","shell.execute_reply.started":"2022-02-11T12:07:36.679868Z","shell.execute_reply":"2022-02-11T12:07:41.983413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_september.product_group_name.value_counts(normalize=True).plot(kind='bar', color='skyblue', position=1, \n                                                               width = .25, label='season')\ndf_train.product_group_name.value_counts(normalize=True).plot(kind='bar', color='red', position=0, \n                                                              width = .25, label='train')\nplt.title('Product group sales distribution - Train VS September')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:07:41.985895Z","iopub.execute_input":"2022-02-11T12:07:41.986287Z","iopub.status.idle":"2022-02-11T12:07:47.526971Z","shell.execute_reply.started":"2022-02-11T12:07:41.986242Z","shell.execute_reply":"2022-02-11T12:07:47.526027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Customer interactions with articles\nAre products purchased more than once by the same customers?","metadata":{}},{"cell_type":"code","source":"df_customer_articles_count = df_train.groupby(['customer_id', 'article_id'])['article_id'].count()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:07:47.528283Z","iopub.execute_input":"2022-02-11T12:07:47.528772Z","iopub.status.idle":"2022-02-11T12:08:24.484528Z","shell.execute_reply.started":"2022-02-11T12:07:47.528742Z","shell.execute_reply":"2022-02-11T12:08:24.483561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.float_format', lambda x: '%.5f' % x)\ndf_customer_articles_count.describe()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:08:24.485824Z","iopub.execute_input":"2022-02-11T12:08:24.486078Z","iopub.status.idle":"2022-02-11T12:08:25.085240Z","shell.execute_reply.started":"2022-02-11T12:08:24.486048Z","shell.execute_reply":"2022-02-11T12:08:25.084404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Wow some article has been bought 570 times by the same customer. The person must love this product :).","metadata":{}},{"cell_type":"code","source":"sns.histplot(df_customer_articles_count[df_customer_articles_count < 5], stat=\"percent\", discrete=True)\nplt.title('Number of times a product is bought by the same customer')","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:08:25.086606Z","iopub.execute_input":"2022-02-11T12:08:25.086854Z","iopub.status.idle":"2022-02-11T12:08:45.258558Z","shell.execute_reply.started":"2022-02-11T12:08:25.086824Z","shell.execute_reply":"2022-02-11T12:08:45.257661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is a fair amoung of repurchases. We are probably interested in recommending products that the customer has already purchased before.","metadata":{}},{"cell_type":"code","source":"df_enriched = df_customer_articles_count[df_customer_articles_count < 5].rename('purchases').reset_index(level=[0, 1])","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:11:41.654481Z","iopub.execute_input":"2022-02-11T12:11:41.655564Z","iopub.status.idle":"2022-02-11T12:11:45.278042Z","shell.execute_reply.started":"2022-02-11T12:11:41.655476Z","shell.execute_reply":"2022-02-11T12:11:45.277117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_enriched = df_enriched.merge(on='article_id', \n                                how='left', \n                                right=df_articles[['article_id', 'index_group_name']])","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:11:45.279975Z","iopub.execute_input":"2022-02-11T12:11:45.280267Z","iopub.status.idle":"2022-02-11T12:11:57.274286Z","shell.execute_reply.started":"2022-02-11T12:11:45.280227Z","shell.execute_reply":"2022-02-11T12:11:57.273432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_enriched = df_enriched.set_index('index_group_name')","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:11:57.275392Z","iopub.execute_input":"2022-02-11T12:11:57.275606Z","iopub.status.idle":"2022-02-11T12:12:03.721142Z","shell.execute_reply.started":"2022-02-11T12:11:57.275581Z","shell.execute_reply":"2022-02-11T12:12:03.720232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_enriched.groupby('index_group_name').purchases\\\n                                      .value_counts(normalize=True)\\\n                                      .plot.bar(title='Number of times a product is bought by the same customer')","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:40:38.861960Z","iopub.execute_input":"2022-02-11T12:40:38.862279Z","iopub.status.idle":"2022-02-11T12:40:45.989993Z","shell.execute_reply.started":"2022-02-11T12:40:38.862244Z","shell.execute_reply":"2022-02-11T12:40:45.989419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It looks like more or less all index groups share the same distribution of repurchases.\n\n\n#### Baseline\n\nLet's build a baseline model to see how it works this competition. We are going to use some of the insights explored.\n\nFirst, let's get the most popular products of September. This is probably not the best solution since there might be products that appeared years before that are no longer popular. Anyway, this is just a baseline.","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:36:06.402483Z","iopub.execute_input":"2022-02-11T12:36:06.402942Z","iopub.status.idle":"2022-02-11T12:36:19.450353Z","shell.execute_reply.started":"2022-02-11T12:36:06.402894Z","shell.execute_reply":"2022-02-11T12:36:19.449396Z"}}},{"cell_type":"code","source":"most_popular_september = df_september['article_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:51:00.107557Z","iopub.execute_input":"2022-02-11T12:51:00.108526Z","iopub.status.idle":"2022-02-11T12:51:00.557675Z","shell.execute_reply.started":"2022-02-11T12:51:00.108485Z","shell.execute_reply":"2022-02-11T12:51:00.556725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's recommend most repeated purchased products for the customer. Then, we will fill the remaining recommendations with the most popular products of September.","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:10:04.271757Z","iopub.execute_input":"2022-02-11T12:10:04.272413Z","iopub.status.idle":"2022-02-11T12:10:04.537764Z","shell.execute_reply.started":"2022-02-11T12:10:04.272340Z","shell.execute_reply":"2022-02-11T12:10:04.536730Z"}}},{"cell_type":"code","source":"def get_recommendation(most_popular, customer_sales_count):\n    if type(customer_sales_count) == pd.core.series.Series:\n        recommendation = [customer_sales_count.article_id]\n    else:\n        recommendation = list(customer_sales_count.sort_values(by='purchases', ascending=False).article_id[0:12])\n    i = 0\n    while (len(recommendation) < 12):\n        recommendation.append(most_popular.index[i])\n        i += 1\n    return recommendation","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:52:43.738394Z","iopub.execute_input":"2022-02-11T12:52:43.738693Z","iopub.status.idle":"2022-02-11T12:52:43.745795Z","shell.execute_reply.started":"2022-02-11T12:52:43.738652Z","shell.execute_reply":"2022-02-11T12:52:43.744782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_articles_count = df_customer_articles_count.rename('purchases').reset_index(level=[1])","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:53:18.804336Z","iopub.execute_input":"2022-02-11T12:53:18.804633Z","iopub.status.idle":"2022-02-11T12:53:23.496641Z","shell.execute_reply.started":"2022-02-11T12:53:18.804604Z","shell.execute_reply":"2022-02-11T12:53:23.495465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"newcomers = list(set(df_sample_submission.customer_id.unique()) - set(df_train.customer_id.unique()))\ndefault_recc = {customer:1 for customer in newcomers}  ","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:54:54.960468Z","iopub.execute_input":"2022-02-11T12:54:54.960778Z","iopub.status.idle":"2022-02-11T12:55:05.147771Z","shell.execute_reply.started":"2022-02-11T12:54:54.960747Z","shell.execute_reply":"2022-02-11T12:55:05.147017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"How fast is our recommendation function?","metadata":{}},{"cell_type":"code","source":"%%timeit\nget_recommendation(most_popular_september,\n                   df_customer_articles_count.loc['0000f1c71aafe5963c3d195cf273f7bfd50bbf17761c9199e53dbb81641becd7'])","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:55:24.311109Z","iopub.execute_input":"2022-02-11T12:55:24.311527Z","iopub.status.idle":"2022-02-11T12:55:29.266795Z","shell.execute_reply.started":"2022-02-11T12:55:24.311489Z","shell.execute_reply":"2022-02-11T12:55:29.265800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fair enough. I will not spend time on optimizing it since it is already acceptable.","metadata":{}},{"cell_type":"code","source":"df_sample_submission_original = df_sample_submission.copy()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:56:25.972047Z","iopub.execute_input":"2022-02-11T12:56:25.972394Z","iopub.status.idle":"2022-02-11T12:56:26.015284Z","shell.execute_reply.started":"2022-02-11T12:56:25.972344Z","shell.execute_reply":"2022-02-11T12:56:26.014441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\ntqdm.pandas()\n\n\ndf_sample_submission['prediction'] = df_sample_submission\\\n                                        .progress_apply(lambda x: get_recommendation(most_popular_september,\n                                                                                     df_customer_articles_count.loc[x.customer_id])\n                                                                  if x.customer_id not in default_recc \n                                                                  else list(most_popular_september.index[0:12]),\n                                                        axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T12:57:35.752946Z","iopub.execute_input":"2022-02-11T12:57:35.753465Z","iopub.status.idle":"2022-02-11T13:11:56.088080Z","shell.execute_reply.started":"2022-02-11T12:57:35.753430Z","shell.execute_reply":"2022-02-11T13:11:56.087138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Preparing data for submission\n\nWe could integrate it into the recommendation function if we want to avoid two loops.","metadata":{}},{"cell_type":"code","source":"def prepare_list_submission(recommendations):\n    recommendations = str(recommendations)\n    REMOVE_CHARS = [\"'\", \",\", \"[\", \"]\"]\n    for char in REMOVE_CHARS:\n        recommendations = recommendations.replace(char, '')\n    return recommendations","metadata":{"execution":{"iopub.status.busy":"2022-02-11T13:12:10.995879Z","iopub.execute_input":"2022-02-11T13:12:10.996725Z","iopub.status.idle":"2022-02-11T13:12:11.002696Z","shell.execute_reply.started":"2022-02-11T13:12:10.996687Z","shell.execute_reply":"2022-02-11T13:12:11.001852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_submission['prediction'] = df_sample_submission.progress_apply(lambda x: prepare_list_submission(x.prediction),\n                                                                         axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T13:12:11.836363Z","iopub.execute_input":"2022-02-11T13:12:11.836662Z","iopub.status.idle":"2022-02-11T13:12:52.165398Z","shell.execute_reply.started":"2022-02-11T13:12:11.836634Z","shell.execute_reply":"2022-02-11T13:12:52.164193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_submission.iloc[0].prediction","metadata":{"execution":{"iopub.status.busy":"2022-02-11T13:12:52.167662Z","iopub.execute_input":"2022-02-11T13:12:52.168019Z","iopub.status.idle":"2022-02-11T13:12:52.175017Z","shell.execute_reply.started":"2022-02-11T13:12:52.167974Z","shell.execute_reply":"2022-02-11T13:12:52.174458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_submission_original.iloc[0].prediction","metadata":{"execution":{"iopub.status.busy":"2022-02-11T13:12:52.176162Z","iopub.execute_input":"2022-02-11T13:12:52.176589Z","iopub.status.idle":"2022-02-11T13:12:52.188806Z","shell.execute_reply.started":"2022-02-11T13:12:52.176557Z","shell.execute_reply":"2022-02-11T13:12:52.188171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks **OK**! Let's save it and submit.","metadata":{}},{"cell_type":"code","source":"df_sample_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T13:12:52.190250Z","iopub.execute_input":"2022-02-11T13:12:52.191015Z","iopub.status.idle":"2022-02-11T13:13:05.363134Z","shell.execute_reply.started":"2022-02-11T13:12:52.190967Z","shell.execute_reply":"2022-02-11T13:13:05.362232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thanks for reading.","metadata":{}}]}