{"cells":[{"metadata":{"_uuid":"6ac208c509772d95a6b27e6a9753f074e521441b"},"cell_type":"markdown","source":"# Project Name: San Francisco Crime Classification\n### Objective: Given time and location, predict the category of crime that occurred."},{"metadata":{"_uuid":"0cc2e43e081d23210289151c26649ac53ab9f99a"},"cell_type":"markdown","source":"# Reference:\n * https://www.analyticsvidhya.com/blog/2017/08/introduction-to-multi-label-classification/"},{"metadata":{"trusted":true,"_uuid":"b28cdd8e6dacd7bc08f3bee754ad09f9ebc10393"},"cell_type":"code","source":"# Reference notebooks\n# Good Viz - https://www.kaggle.com/lesibius/crime-scene-exploration-and-model-fit","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e133dc34c70ad8ef0fc626bbf7ee9110e14a72a"},"cell_type":"code","source":"# Import packages\n\n# visualizations\n%matplotlib inline\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.ticker as ticker\n\n# Stats\nfrom scipy import stats as ss\nimport numpy as np\n\n# datetime\nfrom datetime import tzinfo, timedelta, datetime\n\n#Common Model Helpers\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder\nfrom sklearn import feature_selection\nfrom sklearn import model_selection\nfrom sklearn import metrics\n\n# Random Forest\nfrom sklearn.ensemble import RandomForestClassifier","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c4ba209736122a719b08e534d0433dd86dac357"},"cell_type":"code","source":"# Read train and test\ntrain_data = pd.read_csv('../input/train.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c6b5948d151881ad93209787a0354ba76439fb0b"},"cell_type":"code","source":"# shape of the dataset\nprint(train_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9dd6eb2ff1d6d074887f7add61cf11c8b1e4838"},"cell_type":"code","source":"# Structure of the data\ntrain_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b5d2337994c36232a47af39675dba43079659fa6"},"cell_type":"code","source":"# Read a snapshot of the test dataset\n# Description and resolution is not present in the test dataset\n#test_dataset = pd.read_csv('../input/test.csv')\ntest_data = pd.read_csv('../input/test.csv')\ntest_data.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b6514de262538e105d631e7e30af36307c6a9934"},"cell_type":"markdown","source":"> # Exploratory analysis"},{"metadata":{"trusted":true,"_uuid":"4fa3e8821c45226c81575bd0b9485d475112dea3","scrolled":true},"cell_type":"code","source":"# Type of the dataset\ntrain_data.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97b7a502dbe2a15108fb3f5de06a83ac0c458cee"},"cell_type":"code","source":"# Distribution of crime category in the train dataset\nnumber_of_crimes = train_data[\"Category\"].value_counts()\nnumber_of_crimes.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5bf617958009cf6c9d22f755ca4feef5aedddddb"},"cell_type":"code","source":"crime_plot = sns.barplot(x = number_of_crimes.index, y = number_of_crimes)\ncrime_plot.set_xticklabels(number_of_crimes.index,rotation = 90)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aeabeda5c194c967bfebbb9872be6e6694e2f501"},"cell_type":"code","source":"pareto_crime = number_of_crimes/ sum(number_of_crimes)\npareto_crime = pareto_crime.cumsum()\n_pareto_crime_plot = sns.tsplot(data=pareto_crime)\n_pareto_crime_plot.set_xticklabels(pareto_crime.index,rotation=90)\n_pareto_crime_plot.set_xticks(np.arange(len(pareto_crime)))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e6b851fec2f3b1fb9bfc534c32a0ee2e909e832d"},"cell_type":"markdown","source":"As predicted by our beloved friend Vilfredo, about 20% (exactly 9 out of 39, which is 23%) of the categories account for (nearly exactly) 80% of the crimes.\n\nMy take on this: just concentrate on these 9. The remainder 30 categories won't be of any help to get a good score."},{"metadata":{"trusted":true,"_uuid":"4fc3c9288ba6e6f4803351a6ff3ed9af0834c6a5"},"cell_type":"code","source":"Main_Crime_Categories = list(pareto_crime[0:8].index)\nprint(\"The following categories :\")\nprint(Main_Crime_Categories)\nprint(\"make up to {:.2%} of the crimes\".format(pareto_crime[8]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a75d83a588238c02021bbae2f9282ba8c3dbb3a1"},"cell_type":"code","source":"# Unique levels in the dataset \n#train_data[\"Category\"].unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85dbe1bec3442235c91bfe22ce5812910722603a"},"cell_type":"code","source":"# Weekdays \ntrain_data['DayOfWeek'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6575f65c71f093f3ba0d6ae448612d58e6fee96a"},"cell_type":"code","source":"# Relative Time Scale\norigin_date = datetime.strptime('2003-01-01 00:00:00','%Y-%m-%d %H:%M:%S')\n\ndef delta_origin_date(dt):\n    _ = datetime.strptime(dt,'%Y-%m-%d %H:%M:%S') - origin_date\n    return(_.days+(_.seconds/86400))\n\ndelta_origin_date(train_data.loc[1,\"Dates\"])\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8146611084ebc2cc57cf70fb5672aca71a330733"},"cell_type":"code","source":"tmp = train_data.loc[:,[\"Dates\",\"Category\"]]\ntmp[\"RelativeDates\"]=train_data.Dates.map(delta_origin_date)\ntmp.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36a27b60417e2421449902046de4fa11bcff0bfb"},"cell_type":"code","source":"# At this stage, it can be interesting to see how the number of crimes per ~ quarter evolved \n# to see if the RelativeDates variable makes sense. To proceed, I'll cut my variables by \n# buckets of roughly 90 days and plot it as a stacked area plot. \n# This will allow to see at once both the increase/decrease of total crimes and \n# the split of crime categories over time.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5dc1d933b8a07a0ae6aee333b49f63a9b577c11d"},"cell_type":"code","source":"tmp[\"QuarterBucket\"] = tmp.RelativeDates.map(lambda d: int(d/90.0))\ntmp.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87893907e4b923ea29d5ae6c3c38f59f64555931"},"cell_type":"code","source":"pt = pd.pivot_table(tmp,index=\"QuarterBucket\",columns=\"Category\",aggfunc=len,fill_value=0)\npt = pt[\"Dates\"]\npt[Main_Crime_Categories].iloc[:49,:].cumsum(1).plot()\npt.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f6624fa5507141ead5e5ce886a4cd078da37fde"},"cell_type":"code","source":"# There's a lot of noise in this graph. I'll take a 3Q-smoothed average of it to make trends easier to see.\n#pd.rolling_mean(pt[Main_Crime_Categories],3).iloc[2:49,:].plot()\npt.iloc[2:49,0:9].rolling(3).mean().plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d3110ed003dc5d3bb2ec49d00ff12e354b17d2f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a89d5c2c535fcb849a3bd35b687dabdbf4ba7e4c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac063d6adb14020e43c491ca4ba498b47b39725c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"82f7752848e2e8eaa83dba568a2c86040f012945"},"cell_type":"code","source":"# Function to calculate correlation between categorical variables\n# Source: https://towardsdatascience.com/the-search-for-categorical-correlation-a1cf7f1888c9\n\ndef cramers_v(x, y):\n    confusion_matrix = pd.crosstab(x,y)\n    chi2 = ss.chi2_contingency(confusion_matrix)[0]\n    n = confusion_matrix.sum().sum()\n    phi2 = chi2/n\n    r,k = confusion_matrix.shape\n    phi2corr = max(0, phi2-((k-1)*(r-1))/(n-1))\n    rcorr = r-((r-1)**2)/(n-1)\n    kcorr = k-((k-1)**2)/(n-1)\n    return np.sqrt(phi2corr/min((kcorr-1),(rcorr-1)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"1bb1ef8c1d78ac8b596fabe66eb72eb437ea4739"},"cell_type":"code","source":"# Correlation between Category and Description\n# Ignore Description since it is not present in the test dataset\ncramers_v(train_data['Category'],train_data['Descript'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"1048cfefeb51cbf571765817cb05f53a7f31ee11"},"cell_type":"code","source":"# Correlation between Category and Resolution\n# Ignore Resolution since it is not present in the test dataset\ncramers_v(train_data['Category'],train_data['Resolution'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"151fbe6f132780d0b377686d3fc375d94aae4114"},"cell_type":"code","source":"# Corr b/w Category and Weekday\ncramers_v(train_data['Category'],train_data['DayOfWeek'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f5dc7ef337926f9602a36a300dcddb540a7a576a"},"cell_type":"code","source":"# Corr b/w Category and PdDistrict\ncramers_v(train_data['Category'],train_data['PdDistrict'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"95a1f40db099df5ddfdce60294115b5926607788"},"cell_type":"code","source":"# Correlation b/w month and category\ncramers_v(train_data['Category'],train_data['Address'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"531de4629f29453d8cc9bcf115be9d6c12ff169a"},"cell_type":"code","source":"# 10 unique districts\ntrain_data['PdDistrict'].unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"cbf4d49012fe8bbd5d1548545fad315d8b1db51b"},"cell_type":"code","source":"# Unique address in the dataset\nlen(train_data['Address'].unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"fc9dc6b69a2f8026060de2b694f00fb591c1a034"},"cell_type":"code","source":"train_data.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1486d0fa01dd389d5ccd954f79dc671925ce8dce"},"cell_type":"code","source":"# Convert object data type to datetime type\n# https://chrisalbon.com/machine_learning/preprocessing_dates_and_times/break_up_dates_and_times_into_multiple_features/\n\ntrain_data['Dates'] = pd.to_datetime(train_data['Dates'], format='%Y%m%d %H:%M:%S')\ntrain_data.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"706af1dd7d6560c0072d32bac1a56899897e8269"},"cell_type":"code","source":"# Create year and month column out of date\ntrain_data['year'] = train_data['Dates'].dt.year\ntrain_data['month'] = train_data['Dates'].dt.month","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8511592c2be8d4c90a4265984b455324dd90e5b1"},"cell_type":"code","source":"train_data['year'].value_counts(sort=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36acb5b5194e4201a6bc627ad5b5ce922a76ce06"},"cell_type":"code","source":"train_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2f889f96fb749f3b71222f542a5767ecc4e4b213"},"cell_type":"code","source":"train_data.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d2eaaaf89af506f0c5406d5aa0d60338c8e6d561"},"cell_type":"markdown","source":"# Prepare data for model"},{"metadata":{"trusted":true,"_uuid":"7882e633f92b561b809e1cc4e1918a2ea91afe25","scrolled":true},"cell_type":"code","source":"X = train_data.drop(['Category','Dates','Descript','Resolution','Address','X','Y'],axis = 1)\nX = pd.get_dummies(X)\nX.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"608f04a45df43e0ba3ef8775c4dae658b3686077"},"cell_type":"code","source":"y = pd.get_dummies(train_data['Category'])\ny.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"535f31553f52f2accf2c0d992e09501de8df45f9"},"cell_type":"code","source":"# For logistic regression\ny = train_data['Category']\ny.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"81e2799f91ec1310d5b3e7c8ecc61afeee81eeff"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = model_selection.train_test_split(X,y,test_size = 0.2, random_state = 42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7165f944b7d3e603dd2fcebfc6f9cf3f6f5d8b9"},"cell_type":"code","source":"print(X_train.shape)\nprint(X_test.shape)\nprint(y_train.shape)\nprint(y_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc6f986168de5e36e0c67e13c11c7d00505d5df5"},"cell_type":"code","source":"# Import random forest classifier\nrf_model = RandomForestClassifier(n_estimators=10)\nrf_model.fit(X_train,y_train)\n\n# Predict the result\npredictions = rf_model.predict(X_test)\n\n# Accuracy score\nscore = metrics.accuracy_score(y_test,predictions)\nprint(score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c6993503ae8a3363a7a5c4bcc827a7c3b688c41"},"cell_type":"code","source":"# Logistic regresssion multi classifier\nfrom sklearn.linear_model import LogisticRegression\nlogreg = LogisticRegression(solver='lbfgs',multi_class = 'multinomial')\n\nlogreg.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17bdd502bab641e6451fcccc5468ec2e542a6d89"},"cell_type":"code","source":"# Predict the result\npredictions = logreg.predict(X_test)\n\n# Accuracy score\nscore = metrics.accuracy_score(y_test,predictions)\nprint(score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1b2c5e043172ad46cd427d26444944dbb72a9c6"},"cell_type":"code","source":"## Binary Relevance - methods to solve mutli class classifier\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\n# integer encode\nlabel_encoder = LabelEncoder()\ninteger_encoded = label_encoder.fit_transform(y.values)\nprint(\"Label Encoder:\" ,integer_encoded)\n\n# onehot encode\nonehot_encoder = OneHotEncoder(sparse=False,categories='auto')\ninteger_encoded = integer_encoded.reshape(len(integer_encoded), 1)\nonehot_encoded = onehot_encoder.fit_transform(integer_encoded)\nprint(\"OneHot Encoder:\", onehot_encoded)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55a7192a1c4fdfbc2c3863872ac771604747d736"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train,x_test, y_train,y_test = train_test_split(X, onehot_encoded, test_size=0.33, random_state=42)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"2f8da0a8c6ddb4c1eab12f7b3d05f0daa541e704"},"cell_type":"code","source":"# using binary relevance\nfrom skmultilearn.problem_transform import BinaryRelevance\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import RandomForestClassifier\n\n# initialize binary relevance multi-label classifier\n# with a from sklearn.ensemble import RandomForestClassifier bayes base classifier\nclassifier = BinaryRelevance(GaussianNB())\n\n# train\nclassifier.fit(x_train, y_train)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7bbe09b5bc62be03cdd4df9c13c235f7fbda126"},"cell_type":"code","source":"# predict\npredictions = classifier.predict(x_test)\nprint(metrics.accuracy_score(y_test,predictions))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8dba2eb935dadf2011bedc6cd3b3c6ee85072297"},"cell_type":"code","source":"####### Label powerset\n# using Label Powerset\nfrom skmultilearn.problem_transform import LabelPowerset\nfrom sklearn.naive_bayes import GaussianNB\n\n# initialize Label Powerset multi-label classifier\n# with a gaussian naive bayes base classifier\nclassifier = LabelPowerset(GaussianNB())\n\n# train\nclassifier.fit(x_train, y_train)\n\n# predict\npredictions = classifier.predict(x_test)\n\nmetrics.accuracy_score(y_test,predictions)\n# 0.01604447864","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fdb2e846d27487ec7d9462a7a95d5723f2e75699"},"cell_type":"code","source":"# multi-label version of kNN is represented by MLkNN\n\nfrom skmultilearn.adapt import MLkNN\nclassifier = MLkNN(k=5)\n\n# train\nclassifier.fit(x_train, y_train)\n\n# predict\npredictions = classifier.predict(x_test)\n\nmetrics.accuracy_score(y_test,predictions)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf7850239394151170e98035558f938b51df574f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"anaconda-cloud":{},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}