{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#print(os.getcwd())\nINPUT_DATA_DIR='/kaggle/input/h-and-m-personalized-fashion-recommendations'\nos.chdir(str(INPUT_DATA_DIR))\n#print(os.listdir(os.getcwd()))\nfor _ in os.listdir(os.getcwd()):\n    if(_.endswith('.csv')):\n        print(_)\n    else:\n        os.chdir('./images')\n        print(_)\n        #for i in os.listdir(os.getcwd()):\n        count=0\n        #printing out the first five images names of the first ID in the directory.\n        for dirnames in os.listdir(os.getcwd()):\n            os.chdir(os.getcwd()+'/'+dirnames)\n            print(\"-\"+dirnames)\n            for filenames in os.listdir(os.getcwd()):\n                if count<5:\n                    print(\"--\"+filenames)\n                    count=count+1\n                else:\n                    count=0\n                    break\n            break\n                \n        #for dirname, _, filenames in os.walk(os.getcwd()):\n            #print(dirname,_,filenames)\n                    \n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #print(dirname,_,filenames)\n    #for filename in filenames:\n        #print(filename)\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-20T08:50:23.834249Z","iopub.execute_input":"2022-02-20T08:50:23.834573Z","iopub.status.idle":"2022-02-20T08:50:23.851712Z","shell.execute_reply.started":"2022-02-20T08:50:23.834538Z","shell.execute_reply":"2022-02-20T08:50:23.850547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.read_csv(str(INPUT_DATA_DIR)+str('/customers.csv'))","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:23.853940Z","iopub.execute_input":"2022-02-20T08:50:23.854226Z","iopub.status.idle":"2022-02-20T08:50:27.954077Z","shell.execute_reply.started":"2022-02-20T08:50:23.854191Z","shell.execute_reply":"2022-02-20T08:50:27.952929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:27.956585Z","iopub.execute_input":"2022-02-20T08:50:27.956985Z","iopub.status.idle":"2022-02-20T08:50:27.977847Z","shell.execute_reply.started":"2022-02-20T08:50:27.956943Z","shell.execute_reply":"2022-02-20T08:50:27.976632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:27.979761Z","iopub.execute_input":"2022-02-20T08:50:27.980168Z","iopub.status.idle":"2022-02-20T08:50:27.995073Z","shell.execute_reply.started":"2022-02-20T08:50:27.980095Z","shell.execute_reply":"2022-02-20T08:50:27.993908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(list(data.columns))","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:27.998954Z","iopub.execute_input":"2022-02-20T08:50:27.999381Z","iopub.status.idle":"2022-02-20T08:50:28.010828Z","shell.execute_reply.started":"2022-02-20T08:50:27.999329Z","shell.execute_reply":"2022-02-20T08:50:28.010054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Description.\n**-- 'FN' is if a customer gets Fashion News newsletter.**\n\n**-- 'Active' is if the customer is active for communication.**","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_colwidth',None)\ndata.customer_id","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:28.012237Z","iopub.execute_input":"2022-02-20T08:50:28.012525Z","iopub.status.idle":"2022-02-20T08:50:28.032767Z","shell.execute_reply.started":"2022-02-20T08:50:28.012475Z","shell.execute_reply":"2022-02-20T08:50:28.032032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The ID of the customer seems to be encrypted for privacy purposes.**","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_colwidth',None)\ndata.postal_code","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:28.034146Z","iopub.execute_input":"2022-02-20T08:50:28.035186Z","iopub.status.idle":"2022-02-20T08:50:28.055829Z","shell.execute_reply.started":"2022-02-20T08:50:28.035142Z","shell.execute_reply":"2022-02-20T08:50:28.054533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Even the postal code is encrypted for privacy purposes.**","metadata":{}},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:28.058451Z","iopub.execute_input":"2022-02-20T08:50:28.059353Z","iopub.status.idle":"2022-02-20T08:50:28.686407Z","shell.execute_reply.started":"2022-02-20T08:50:28.059267Z","shell.execute_reply":"2022-02-20T08:50:28.685400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#null_data=pd.DataFrame()\ndef per_cal(data,col):\n    per_NA=(data[col].isnull().sum()/data.shape[0])*100\n    per_VAL=(data[col].notnull().sum()/data.shape[0])*100\n    return per_NA,per_VAL\n#per_NA=(data.FN.isnull().sum()/data.shape[0])*100\n#per_VAL=(data.FN.notnull().sum()/data.shape[0])*100\n#dct={'per_NA':per_NA,'per_VAL':per_VAL}\ndct={}\nlst=[]\nfor col in data.columns[1:]:\n    per_NA,per_VAL=per_cal(data,col)\n    lst.append([per_NA,per_VAL])\n    dct[col]=(per_NA,per_VAL)\nnull_data=pd.DataFrame(lst,columns=['per_NA','per_VAL'],index=data.columns[1:])\nnull_data","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:28.687724Z","iopub.execute_input":"2022-02-20T08:50:28.688331Z","iopub.status.idle":"2022-02-20T08:50:29.651682Z","shell.execute_reply.started":"2022-02-20T08:50:28.688290Z","shell.execute_reply":"2022-02-20T08:50:29.650766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"per_active=((data.Active[data.FN.notnull()].isnull().sum())/data.Active.isnull().sum())*100\nper_active_null=((data.Active[data.FN.isnull()].isnull().sum())/data.Active.isnull().sum())*100\nprint(per_active,per_active_null)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:29.653009Z","iopub.execute_input":"2022-02-20T08:50:29.654274Z","iopub.status.idle":"2022-02-20T08:50:29.721946Z","shell.execute_reply.started":"2022-02-20T08:50:29.654214Z","shell.execute_reply":"2022-02-20T08:50:29.720979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Majority of the rows(almost ***98.6*** percent) contains missing values for both **FN** and **Active** columns and only ***1.38*** percent of the total rows contains missing values for Active column only.**Hence missing values for these columns seems to be highly correlated**.","metadata":{}},{"cell_type":"code","source":"data.Active[data.FN.notnull()].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:29.723574Z","iopub.execute_input":"2022-02-20T08:50:29.724599Z","iopub.status.idle":"2022-02-20T08:50:29.757069Z","shell.execute_reply.started":"2022-02-20T08:50:29.724548Z","shell.execute_reply":"2022-02-20T08:50:29.756076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fig,ax=plt.figure(figsize=(10,24))\nplt.hist(data['age'],bins=20)\nplt.xlabel(\"Customer Age\")\n#plt.xticks(data.age[data.age.notnull()].unique())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T08:50:29.758275Z","iopub.execute_input":"2022-02-20T08:50:29.759383Z","iopub.status.idle":"2022-02-20T08:50:30.012397Z","shell.execute_reply.started":"2022-02-20T08:50:29.759341Z","shell.execute_reply":"2022-02-20T08:50:30.011477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Most of the customers fall within the [20-35] and [40-55] age group**\n\n--the high frequency of age group 20-35 seems to be pretty justified since H&M is known for fast fashion which is common in college going students,fashion influencers etc.\n\n--the age group [25-30] has the highest frequency in the histogram plot.This group might represent people who have recently entered the corporate world and have a good chunk of money in their hand and have bare minimum responsibilities since majority of them might be single and might not have their own family.This might be the extravagant class\n\n--Also,the relatively high frequency of 40-55 age group might be because they might be purchasing for their children who are mostly in their teenage years or early twenties and highly active on social media as well.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}