{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.svm import LinearSVC\nfrom matplotlib import pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db9fcc778ba1c4faad787e112c7e17a5ec05c2e1"},"cell_type":"code","source":"df = pd.read_csv('../input/train.csv')\ndf_test = pd.read_csv('../input/test.csv')\nX = df.iloc[:, 1:]\ny = df.iloc[:, 0]\n\nX_test = df_test.iloc[:, 0:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ffcbe78db41020198bb5042954009e824226ff5"},"cell_type":"code","source":"pca = PCA(X.shape[1])\npca.fit(X)\npca.explained_variance_ratio_\n\nplt.plot([i for i in range(X.shape[1])], \n         [np.sum(pca.explained_variance_ratio_[:i+1]) for i in range(X.shape[1])])\nplt.show()\n\n\n\npca = PCA(0.95)\npca.fit(X)\n\npca.n_components_\n\nX_reduction = pca.transform(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5bb4a46ec1f2ea2fe81b6e291aa5033a2e1621e7"},"cell_type":"code","source":"def PolynomialLinearSVC():\n    return Pipeline([\n        (\"std_scaler\", StandardScaler()),\n        (\"LinearSVC\", LinearSVC())\n    ])\n\npolynomialLinearSVC = PolynomialLinearSVC()\npolynomialLinearSVC.fit(X_reduction, y)\n\n\nX_predict = polynomialLinearSVC.predict(X_reduction)\npolynomialLinearSVC.score(X_reduction, y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0942627ec7659f47a9879cfce1d505cecb3690c2"},"cell_type":"code","source":"polynomialLinearSVC.score(X_reduction, y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90d9181701f8495ce7484c0531861d53f7ce6db0"},"cell_type":"code","source":"X_test_reduction = pca.transform(X_test)\n\nX_test_predict = polynomialLinearSVC.predict(X_test_reduction)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1672a2ebde3beb9a4abf9e6dc9df9414c8ee8d32"},"cell_type":"code","source":"X_test_predict.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"685701b88ff49e6abaa27aaa611c7f472e25a5b8"},"cell_type":"code","source":"X_test_predict[: 20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed88afc21c0b28172b7972ad8d6d43ddc4ffae0b"},"cell_type":"code","source":"predictions = pd.Series(X_test_predict,name=\"Label\")\nsubmission = pd.concat([pd.Series(range(1,28001),name = \"ImageId\"),predictions],axis = 1)\nsubmission.to_csv(\"svc_submission.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3087397e3bd4dd62f8c481d251e9b6698660c3c4"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}