{"cells":[
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "Latent Destination Feature Exploration: This script is generated by me to better understand the destination dataset as per the code of\n @beluga in Latent Destination Features, all done to get hands-on at kaggle    "
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "%matplotlib inline\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style('darkgrid')\nsns.set_context(\"poster\")\nsns.set(color_codes=True)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "destination_features = pd.read_csv(\"../input/destinations.csv\")"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "destination_features.shape"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test_destinations = pd.read_csv(\"../input/test.csv\", usecols=['srch_destination_id'])"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "a,b = np.unique(test_destinations, return_counts=True)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "len(sorted(b))"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "srch_destinations, count = a,b"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "a = range(0,len(a),200)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "count.sum()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "1.0 * np.array(sorted(count)).cumsum()/count.sum()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "fig, ax = plt.subplots(ncols=2, sharex=True)\nax[0].semilogy(sorted(count))\nax[1].plot(1.0 * np.array(sorted(count)).cumsum()/count.sum())\nax[0].set_xticks(range(0, len(srch_destinations), 10000))\nax[1].set_ylabel('Cumulative sum')\nax[0].set_ylabel('Search destination counts in test set (log scale)')"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#frequent_destinations = srch_destinations[count >= 10]\nprint (1. * count[count >= 10].sum() / count.sum())"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "len(count[count>=10])"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "frequent_destinations = srch_destinations[count >= 10]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "frequent_destinations"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "frequent_destination_features = destination_features[destination_features['srch_destination_id'].isin(frequent_destinations)]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "frequent_destination_features.info()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "frequent_destination_features = frequent_destination_features.drop('srch_destination_id', axis=1)\nprint(frequent_destination_features.shape)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "correlations = frequent_destination_features.corr()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "correlations.shape"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "correlations.tail()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "f = plt.figure()\nax = sns.heatmap(correlations)\nax.set_xticks([])\nax.set_yticks([])\nplt.title('Tartan or correlation matrix')\nf.savefig('tartan.png', dpi=600)\nplt.show()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "correlations.values.reshape(correlations.size).shape"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "fig=plt.figure()\nsns.distplot(correlations.values.reshape(correlations.size), bins=50, color='g')\nplt.title('Correlation values')\nplt.show()\nfig.savefig('CorrelationHist')"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "Hierarchical clustering so as to reorder the features"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "g = sns.clustermap(correlations)\ng.ax_heatmap.set_xticks([])\ng.ax_heatmap.set_yticks([])\ng.savefig('clustermap.png', dpi=300)"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "dendogram_col.reordered_ind gives the index of the original columns.\nAnd Selecting a set of features from the beginning and from the end to check their distributions."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "np.array(g.dendrogram_col.reordered_ind)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "p = [8, 102, 120, 127, 74]\nq = [70, 78, 10, 55, 19]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "top_plot = frequent_destination_features[frequent_destination_features.columns[a]].sample(4000)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "end_plot = frequent_destination_features[frequent_destination_features.columns[q]].sample(4000)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "g = sns.PairGrid(top_plot, size=4)\ng.map_upper(plt.scatter, s=6, alpha=0.4)\ng.map_lower(sns.kdeplot, cmap=\"Blues_d\")\ng.map_diag(sns.kdeplot, legend=False, shade=True)\nplt.suptitle('Top features graphically')\ng.savefig('cluster_1.png', dpi=400) "
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "def blue_kde_hack(x, color, **kwargs):\n    sns.kdeplot(x, color='b', **kwargs)\ng = sns.PairGrid(end_plot, size=4)\ng.map_upper(plt.scatter, s=6, alpha=0.4, color='b')\ng.map_lower(sns.kdeplot, cmap=\"Blues_d\")\ng.map_diag(blue_kde_hack, legend=False, shade=True)\nplt.suptitle('5 correlated features from the end of the list')\ng.savefig('cluster_end.png', dpi=400)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "middle = [89, 69, 115, 105, 71]\ndef green_kde_hack(x, color, **kwargs):\n    sns.kdeplot(x, color='g', **kwargs)\nmiddle_plot = frequent_destination_features[frequent_destination_features.columns[middle]].sample(5000)\ng = sns.PairGrid(middle_plot, size=4)\ng.map_upper(plt.scatter, s=6, alpha=0.4, color='g')\ng.map_lower(sns.kdeplot, cmap=\"Greens_d\")\ng.map_diag(green_kde_hack, legend=False, shade=True)\nplt.suptitle('Middle features depicting correlations')\ng.savefig('cluster_middle.png', dpi=400)"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}