{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":false},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3b3042ef3e812d8a4b3c3625b8d860758bf9c852"},"cell_type":"markdown","source":"**Introduction**\nPCA Dimention reduction methodology, we are applying PCA on Titanic Dataset.\n\n"},{"metadata":{"_uuid":"55d92956ea1edd80ec5cd25beaf7cf490e5691eb"},"cell_type":"markdown","source":"**Steps Performed on Train data**\n1. Reading the Titanic Dataset from csv files.\n2. Removing Unwanted Colums from train Dataset.\n3. Droping Nan colums & replacing the same with mean of  respective column.\n"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"scrolled":true},"cell_type":"code","source":"titanic_df = pd.read_csv(\"../input/train.csv\")\ntitanic_df = titanic_df.drop([\"Name\",\"Ticket\",\"Cabin\"],axis=1)\n# Remove Row which are having NAN values\ntitanic_df.dropna(axis = 1, how ='all', inplace = True)\n# To check which coulumn has NAN \ntitanic_df.isnull().any()\n#Replaced Nan values in Age column with it's Mean\ntitanic_df.ix[:,4]=(titanic_df.ix[:,4]).fillna(titanic_df.mean(axis=0)[3])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ffaa13f5df7c8331468e8b1c9aec4d4072900cf1"},"cell_type":"markdown","source":"Categorical Features Sex , Embarked and Survived has been replaced by the numeric values by using map."},{"metadata":{"trusted":true,"_uuid":"4cf379ef1c08ac5826133b93b472c2603220fb49"},"cell_type":"code","source":"def binarize(df):\n    for col in ['Sex']:\n        df[col] = df[col].map({'female':1, 'male':0})\n    return df\ndef binarize_Embark(df):\n    for col in ['Embarked']:\n        df[col] = df[col].map({'S':0, 'C':1,'Q':2})\n    return df\ndef label_Survived(df):\n    for col in ['Survived']:\n        df[col] = df[col].map({0:'NotSurvived', 1:'Survived'})\n    return df\ntitanic_df = binarize(titanic_df)\ntitanic_df = binarize_Embark(titanic_df)\ntitanic_df = label_Survived(titanic_df)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"964c21350cd68c055479582c6fc09551a0696787"},"cell_type":"markdown","source":"Creating two seprate variables x for features which contains all features which will undergo Matrix PCA methodlogy \nAnd y is the target variable wherein 0 for not survived and 1 for survived"},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"2c8dda1e369ea4c9b30555eba737ea8b68f3c9ac"},"cell_type":"code","source":"x = titanic_df.ix[:,2:8].values\ny = titanic_df.ix[:,1].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f5488f359047a96a4cc910c5ac0d9f3d67786348"},"cell_type":"code","source":"# we are using function StandardScaler() function for standarizing the Matrix.\nfrom sklearn.preprocessing import StandardScaler\nstd = StandardScaler()\nstd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6e1cd221cae92316b8f3ae63ac9955fbda8a724"},"cell_type":"code","source":"std.fit(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"04cd349e5d1dd817f7b9d699099f1454ff49b161"},"cell_type":"code","source":"X_std = std.transform(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d98c549d148b5b00a5625f0d4d3bfcc5ea79e6ec"},"cell_type":"code","source":"# Below is standard mean & variance featurewise\nstd.mean_, std.var_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5be984221b4b9de0faf2451780160cadf5265fd"},"cell_type":"code","source":"# we can directly use below funtion call in order to get our Matrix standarize\n#X_std = StandardScaler().fit_transform(x)\n#X_std","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"06db991251c7ccc15162b375a6347e715dabc621"},"cell_type":"code","source":"mean_vec = np.mean(X_std, axis=0)\n#mean_vec = mean_vec[~pd.isnull(mean_vec)]\n#mean_vec\ncov_mat = (X_std - mean_vec).T.dot((X_std - mean_vec)) / (X_std.shape[0]-1)\nprint('Covariance matrix \\n%s' %cov_mat)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0699d13692d13dbd1971880089cbdaebb686f7f3","scrolled":true},"cell_type":"code","source":"#We can get Covarience matrix by direclty using below Method instead of doing like blove result will be same\nprint('NumPy covariance matrix: \\n%s' %np.cov(X_std.T))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6bb53a90ae6fc9c70300f9be0331232630efa539"},"cell_type":"markdown","source":"**Calculate Eigen Vector & Eigen Values**"},{"metadata":{"trusted":true,"_uuid":"2388c140c7c5bf4890065f09064afd2bdce47ec3"},"cell_type":"code","source":"cov_mat = np.cov(X_std.T)\ncov_mat\n\neig_vals, eig_vecs = np.linalg.eig(cov_mat)\nnp.linalg.eig\nprint('Eigenvectors \\n%s' %eig_vecs)\nprint('\\nEigenvalues \\n%s' %eig_vals)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b97791e806c50b5a2ffeb774543c68dcce70aea3"},"cell_type":"code","source":"u,s,v = np.linalg.svd(X_std.T)\nu","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"784e7752b084405743c003c792961b7053365dcd"},"cell_type":"code","source":"s","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1131c24fbb021900cd6baeb54063a62039771691"},"cell_type":"code","source":"v","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"9099de2aa679a4fab9281488dff19827f6aace7d"},"cell_type":"code","source":"for ev in eig_vecs:\n    np.testing.assert_array_almost_equal(1.0, np.linalg.norm(ev))\nprint('Everything ok!')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"98a1db6190076296a203a7af43df85fe990d86f8"},"cell_type":"code","source":"import plotly.plotly as py\nfrom plotly.graph_objs import *\nimport plotly.tools as tls\n\nfrom plotly import __version__\nfrom plotly.offline import download_plotlyjs, init_notebook_mode, plot, iplot\n%matplotlib inline\ninit_notebook_mode(connected=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f2602ae904142d4cc3da5da9890179e932a5b94a","scrolled":true},"cell_type":"code","source":"# Make a list of (eigenvalue, eigenvector) tuples\neig_pairs = [(np.abs(eig_vals[i]), eig_vecs[:,i]) for i in range(len(eig_vals))]\n#eig_pairs[1][1]\n# Sort the (eigenvalue, eigenvector) tuples from high to low\neig_pairs.sort()\neig_pairs.reverse()\n\n# Visually confirm that the list is correctly sorted by decreasing eigenvalues\nprint('Eigenvalues in descending order:')\nfor i in eig_pairs:\n    print(i[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4cff9bd403e493441ba582fed6cdfe872e7ac382"},"cell_type":"code","source":"tot = sum(eig_vals)\nvar_exp = [(i / tot)*100 for i in sorted(eig_vals, reverse=True)]\ncum_var_exp = np.cumsum(var_exp)\n\ntrace1 = Bar(\n        x=['PC %s' %i for i in range(1,7)],\n        y=var_exp,\n        showlegend=False)\n\ntrace2 = Scatter(\n        x=['PC %s' %i for i in range(1,7)], \n        y=cum_var_exp,\n        name='cumulative explained variance')\n\ndata = Data([trace1,trace2])\n\nlayout=Layout(\n        yaxis=YAxis(title='Explained variance in percent'),\n        title='Explained variance by different principal components')\n\nfig = Figure(data=data, layout=layout)\niplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b6ca8b90241b50aac481a7ce21108c3c2992fdb0"},"cell_type":"markdown","source":"**Graphical Representation**\nAbove Graph built by Numpy to give Graphical view on which features explain most variation in the dataset. As you can see PC1 & PC2 are both together Explain aroung 57% Variations.\n"},{"metadata":{"_uuid":"93ae2e907ec509dcabbb01aa8684ca040681e2a1"},"cell_type":"markdown","source":"Now Prepare a W Matrix which Contains most relevant/variation explanation feature column"},{"metadata":{"trusted":true,"_uuid":"b551cb39ed98d3ba112775359b9535dda37730cd"},"cell_type":"code","source":"matrix_w = np.hstack((eig_pairs[0][1].reshape(6,1), \n                      eig_pairs[1][1].reshape(6,1)))\nprint('Matrix W:\\n', matrix_w)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e49b1022cd7a59b44e57e175caaa78ce14e50f2","scrolled":false},"cell_type":"code","source":"Y = X_std.dot(matrix_w)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"832282bdce68bb915cd7602ac5af448293450fff"},"cell_type":"code","source":"Y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3218d4adc7203a4c3ff8cd436ccbce841f4da870"},"cell_type":"code","source":"X_std.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40f707ef269be4735a89dbcf395819c11705d36d"},"cell_type":"code","source":"traces = []\n\nfor name in ('NotSurvived','Survived'):\n\n    trace = Scatter(\n        x=Y[y==name,0],\n        y=Y[y==name,1],\n        mode='markers',\n        name=name,\n        marker=Marker(\n            size=10,\n            line=Line(\n                color='rgba(217, 217, 217, 0.14)',\n                width=0.5),\n            opacity=0.8))\n    traces.append(trace)\n\n\ndata = Data(traces)\nlayout = Layout(showlegend=True,\n                scene=Scene(xaxis=XAxis(title='PC1'),\n                yaxis=YAxis(title='PC2'),))\n\nfig = Figure(data=data, layout=layout)\niplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"254fd3806f225a47b30e2e4195289bf669fcc984"},"cell_type":"markdown","source":"**Final Outcome Explanation by Graph**\nWith W Matrix now have tried to plot the graph on 2 Dimentional graph, wherein Not Survived And Survived people have mention.\nwe can not see clear segregation of survived & not survival here while applying PCA however it good for learning purpose."},{"metadata":{"_uuid":"5e10fb76b9ec860d7966932410a90c7e409e509f"},"cell_type":"markdown","source":"**PCA applied on Titanic Dataset using scikit learn Library**\nWith the help of scikit learn library we can direcly apply PCA by using sklearnPCA method\nall above steps we can avoid however it is good to know what it does in beckend."},{"metadata":{"trusted":true,"_uuid":"441f1388a04f321fc6e9ddbaa2fcb497359beb51"},"cell_type":"code","source":"from sklearn.decomposition import PCA as sklearnPCA\nsklearn_pca = sklearnPCA(n_components=3)\nY_sklearn = sklearn_pca.fit_transform(X_std)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0beda5a90f9f29a84e8011896594c71e660d97a"},"cell_type":"code","source":"sklearn_pca.explained_variance_ratio_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a7cac48b2f49138bfba5d061cd28f2c818e8e81"},"cell_type":"code","source":"Y_sklearn.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"b0cc1f58fcc77c12463b4125b15bc75a881ec894"},"cell_type":"code","source":"sklearn_pca.transform()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3add7a4e5af60e4bd8c71acc5613e578ead4e9dc","scrolled":true},"cell_type":"code","source":"traces = []\n\nfor name in ('NotSurvived','Survived'):\n\n    trace = Scatter(\n        x=Y_sklearn[y==name,0],\n        y=Y_sklearn[y==name,1],\n        mode='markers',\n        name=name,\n        marker=Marker(\n            size=10,\n            line=Line(\n                color='rgba(217, 217, 217, 0.14)',\n                width=0.5),\n            opacity=0.8))\n    traces.append(trace)\n\n\ndata = Data(traces)\nlayout = Layout(showlegend=True,\n                scene=Scene(xaxis=XAxis(title='PC1'),\n                yaxis=YAxis(title='PC2'),))\n\nfig = Figure(data=data, layout=layout)\niplot(fig)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"346a7b14e54296f017773a63d3a4780a95ec08d8"},"cell_type":"markdown","source":"**K Mean clustering Algorithm**"},{"metadata":{"trusted":true,"_uuid":"7614b9ec797550e715a46ffa5d7e0e4373353136"},"cell_type":"code","source":"%matplotlib inline\n\nfrom sklearn import datasets, cluster\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"021d937c0f585cc5d9334e183531a171065af097"},"cell_type":"code","source":"np.random.seed(2)\nk_means = cluster.KMeans(n_clusters=2)\nk_means.fit(Y) \nlabels = k_means.labels_\nv\n# check how many of the samples were correctly labeled\ncorrect_labels = sum(y_var == labels)\n\nprint(\"Result: %d out of %d samples were correctly labeled.\" % (correct_labels, y_var.size))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3bbe9786c9c1503fe67319720ffd7e606ecb47f2","scrolled":true},"cell_type":"code","source":"# plot the clusters in color\nfig = plt.figure(1, figsize=(8, 8))\nplt.clf()\nax = Axes3D(fig, rect=[0, 0, 1, 1], elev=8, azim=200)\nplt.cla()\n\nax.scatter(Y[:, 0], Y[:, 1], c=labels.astype(np.float))\n\nax.w_xaxis.set_ticklabels([])\nax.w_yaxis.set_ticklabels([])\nax.set_xlabel('NotSurvived')\nax.set_ylabel('Survived')\n\n\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6c047f6929606013d3500219ff3a75532b36ebc1"},"cell_type":"markdown","source":"Below we have tried to plot in 2-D graph as well."},{"metadata":{"trusted":true,"_uuid":"807b824f0b772c93d035cf132bafb5ea0cccb394"},"cell_type":"code","source":"\ntraces = []\n\nfor tempvar in ('NotSurvived','Survived'):\n    trace = Scatter(\n        x=Y[y==tempvar,0],\n        y=Y[y==tempvar,1],\n        mode='markers',\n        name=name,\n        marker=Marker(\n            size=10,\n            line=Line(\n                color='rgba(217, 217, 217, 0.14)',\n                width=0.5),\n            opacity=0.8))\n    traces.append(trace)\ndata = Data(traces)\nlayout = Layout(showlegend=True,\n                scene=Scene(xaxis=XAxis(title='AxisX'),\n                yaxis=YAxis(title='AxisY'),))\n\nfig = Figure(data=data, layout=layout)\niplot(fig)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}