{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eba0b437fc4af4bb0b09463aa234122880961161"},"cell_type":"code","source":"df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b2757e30fb76df9ff7e73dc33f0640cdd17ef1f4"},"cell_type":"code","source":"df.shape # Prints number of rows and columns in dataframe","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7fe2336402a6186ba8b46fbb762832d20e320d4e"},"cell_type":"code","source":"df.head(10) # Prints first 10 rows of the DataFrame","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f26bba735e101b6f6ac9953af2961fea044919ce"},"cell_type":"code","source":"df.tail(10) # Prints last 10 rows of the DataFrame","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e9c1d97fe029ad0b67d43884da9f4b8bb242cad"},"cell_type":"code","source":"df.info() # Index, Datatype and Memory information","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"009fa81f4028e06c4c98dfd92da5fa590c1f1e59"},"cell_type":"code","source":"df.describe() # Summary statistics for numerical columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48879e8ceb5a96937e214fca81a1030077ec7844"},"cell_type":"code","source":"df['first_active_month'].value_counts(dropna = False) # Views unique values and counts","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"152231d1318a905051874d731a89029e0f6c00f6"},"cell_type":"code","source":"df['feature_1'].value_counts(dropna = False) # Views unique values and counts","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a26ecb335c5a6b41477ebb769331e6bad14af8c"},"cell_type":"code","source":"df['feature_2'].value_counts(dropna = False) # Views unique values and counts","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0f247906fc9b0b50cb71ed6de63935cedae4543"},"cell_type":"code","source":"df['feature_3'].value_counts(dropna = False) # Views unique values and counts","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a8df1a54e3fcf2f264bff273cc7135059006aeb"},"cell_type":"code","source":"df['target'].mean() # Returns the mean of target column","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f888416be3cf0e21d8f61894c19cd89e16613a50"},"cell_type":"code","source":"df.corr() # Returns the correlation between columns in a DataFrame","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"825055568f9fa45f9339386e78a6fa0e9712e0cb"},"cell_type":"code","source":"df.count() # Returns the number of non-null values in each DataFrame column","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4f28870aa0a63154dbdd54205e72ce1babd80c45"},"cell_type":"code","source":"df.max() # Returns the highest value in each column","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3db6790e529a82fdd7088f8065a09071ea67a628"},"cell_type":"code","source":"df.min() # Returns the lowest value in each column","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a345ad9ebd0fe9c23a17b445fbcb6dc3d004d32c"},"cell_type":"code","source":"df.median() # Returns the median of each column","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a15c7a1304fcf4d7607ba1cb69ac461af65d94e"},"cell_type":"code","source":"df.std() # Returns the standard deviation of each column","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b155acc5716c31f9d619050885e09d6ac7215888"},"cell_type":"code","source":"df['card_id'] # Returns column with label card_id as Series","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ebbe05f91a849a0f02d015e47738e2c45a3b7be"},"cell_type":"code","source":"df[['first_active_month', 'card_id']] # Returns Columns as a new DataFrame","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e6d7b5f3065272d66b53c4b5f57430a4f17c2fed"},"cell_type":"code","source":"df['card_id'].iloc[0] # Selection by position (selects first element)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"778e97bd29e6199b593722372fe4c584c47724ea"},"cell_type":"code","source":"df['first_active_month'].loc[0] # Selection by index (selects element at index 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"293a979d445d3104e675708063f6d6c820b736d6"},"cell_type":"code","source":"df.iloc[0,:] # First row","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"379afb7153d27a5371583746ce038616bf785a4a"},"cell_type":"code","source":"df.iloc[0,0] # First element of first column","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b563bb6b09e803341516df5f050b2b20b14a423"},"cell_type":"code","source":"df.columns = ['First_Active_Month','Card_ID','Feature_A','Feature_B','Feature_C','Target'] # Renames columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2a99e8b6a383dc0c20d756a0178d8524016d796"},"cell_type":"code","source":"pd.isnull(df) # Checks for null Values, Returns Boolean Array","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a33a9613e2389a9ddc1c974faa278284ef4ea7ea"},"cell_type":"code","source":"pd.notnull(df) # Opposite of s.isnull()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"78ad83dbaaec6acecabd0b20dbb781f68f3f57ec"},"cell_type":"code","source":"df.dropna() # Drops all rows that contain null values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d7d4bcace576d9b9d4e90d5fcb9cea7ed0068157"},"cell_type":"code","source":"df.dropna(axis=1) # Drops all columns that contain null values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"63fd39117f7ebb89b0b04aa4a7d973a9bc6ca514"},"cell_type":"code","source":"df.dropna(axis=1,thresh=5) # Drops all rows have have less than 5 non null values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2bfa0042708bb84c8a14c80b93c1a84719a6c978"},"cell_type":"code","source":"df.fillna(0) # Replaces all null values with 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7113d0d006fda3c02dd1d073bdf20fde3a4b81b"},"cell_type":"code","source":"df['Target'] = df['Target'].fillna(df['Target'].mean())  # Replaces all null values with the mean (mean can be replaced with almost any function from the statistics section)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65c4a55cb10f2f7853155f49433a4536aa00f27a"},"cell_type":"code","source":"df[['Feature_A', 'Feature_B', 'Feature_C']].astype(float) # Converts the datatype of the series to float","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9680a8cccdc1cd2722a0c91724c22aa06693810"},"cell_type":"code","source":"tmp_df = df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"476000a8c095d977b0df4e4ff3f55e9730ad95bc"},"cell_type":"code","source":"tmp_df['Feature_A'].replace(1,'one') # Replaces all values equal to 1 with 'one'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0660e37d588b6d5f0d30cf72f89a640a893e1fd3"},"cell_type":"code","source":"tmp_df.rename(columns={'Target': 'Target_value'}) # Selective renaming","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b92b39afde57e30bfe92d60c6842a20133d4deee"},"cell_type":"code","source":"tmp_df.set_index('Card_ID') # Changes the index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff95f9fd207f399764cdfbd1fae587355bb80a4e"},"cell_type":"code","source":"df[df['Target'] > 0.5] # Rows where the target column is greater than 0.5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a1d6afabc6ba1f43303259c11d736103148bb2d0"},"cell_type":"code","source":"df[(df['Target'] > 0.5) & (df['Target'] < 0.7)] # Rows where 0.5 < target < 0.7","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7eb22fe64174bf5fa28895fa50e49f3028b9601"},"cell_type":"code","source":"tmp_df.sort_values(['Target']) # Sorts values by target in ascending order","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f10b54b278a240063dfa2decd5bbc2d327e1bf03"},"cell_type":"code","source":"tmp_df.sort_values(['Target'],ascending=False) # Sorts values by target in descending order","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d6d1aadfba934ff99bda432a6041fa39930b3bb2"},"cell_type":"code","source":"tmp_df.sort_values(['First_Active_Month','Target'], ascending=[True,False]) # Sorts values by col1 in ascending order then col2 in descending order","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7b10d7348ab9c1e844ec6cb59ecf59e90d9ba91f"},"cell_type":"code","source":"tmp_df.groupby(tmp_df['First_Active_Month']).mean() # Returns a groupby object for values from one column","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"162d58da640e3e6379fdb6d024aa2bca8485d991"},"cell_type":"code","source":"tmp_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f0d3c389400ae7cfd94d08fdcff72d27f910ebe"},"cell_type":"code","source":"tmp_df.pivot_table(index='Card_ID', values= 'Target', aggfunc='mean') # Creates a pivot table that groups by col1 and calculates the mean of col2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7959db7bef717311667a0027ab79dd841823f0c"},"cell_type":"code","source":"tmp_df.groupby('First_Active_Month').agg(np.mean) # Finds the average across all columns for every unique column 1 group","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}