{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport plotly.graph_objects as go\nimport plotly.express as px","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-17T15:16:14.310329Z","iopub.execute_input":"2024-10-17T15:16:14.310767Z","iopub.status.idle":"2024-10-17T15:16:14.722814Z","shell.execute_reply.started":"2024-10-17T15:16:14.310725Z","shell.execute_reply":"2024-10-17T15:16:14.721621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_dir = '/kaggle/input/jane-street-real-time-market-data-forecasting/'","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:16:14.725235Z","iopub.execute_input":"2024-10-17T15:16:14.725916Z","iopub.status.idle":"2024-10-17T15:16:14.731711Z","shell.execute_reply.started":"2024-10-17T15:16:14.725858Z","shell.execute_reply":"2024-10-17T15:16:14.730489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = pd.read_csv(f'{input_dir}features.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:16:14.837872Z","iopub.execute_input":"2024-10-17T15:16:14.838958Z","iopub.status.idle":"2024-10-17T15:16:14.860181Z","shell.execute_reply.started":"2024-10-17T15:16:14.838909Z","shell.execute_reply":"2024-10-17T15:16:14.859112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Features\n\nWhat is the features.csv?\n\n* metadata pertaining to the anonymized features\n\nWhat do the tags mean?\n\n* ?\n\nAre some tags more common?\n\n* Yes! Many features have Tag 3...\n\nWhat tags are shared between features or Cooccurance of features?\n\n* Features 21 - 31 are very similar to each other\n\n\n\n","metadata":{}},{"cell_type":"code","source":"features.columns","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:16:15.329157Z","iopub.execute_input":"2024-10-17T15:16:15.330380Z","iopub.status.idle":"2024-10-17T15:16:15.340932Z","shell.execute_reply.started":"2024-10-17T15:16:15.330297Z","shell.execute_reply":"2024-10-17T15:16:15.339728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_col = ['features']\ntag_cols = list(filter(lambda x: 'tag' in x, features.columns))\n\nmean_tags = features[tag_cols].mean()\nsum_tags = features[tag_cols].sum()\n\ntag_summary_df = pd.DataFrame()\ntag_summary_df['tag_id'] = mean_tags.index\ntag_summary_df['percentage'] = mean_tags.values\ntag_summary_df['count'] = sum_tags.values","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:16:15.576744Z","iopub.execute_input":"2024-10-17T15:16:15.577225Z","iopub.status.idle":"2024-10-17T15:16:15.601808Z","shell.execute_reply.started":"2024-10-17T15:16:15.577170Z","shell.execute_reply":"2024-10-17T15:16:15.600437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.bar(tag_summary_df,\n       'tag_id',\n       'count',\n       title='Tag Occurence')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:29:39.976089Z","iopub.execute_input":"2024-10-17T15:29:39.976556Z","iopub.status.idle":"2024-10-17T15:29:40.146282Z","shell.execute_reply.started":"2024-10-17T15:29:39.976513Z","shell.execute_reply":"2024-10-17T15:29:40.144928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cooccurance of features and tags\n\n# How many pairs\npaired = features.merge(features, how='cross', suffixes=('','_compare'))\nfor i in range(0, 17):\n    paired[f'tag_{i}_match'] = paired[f'tag_{i}'] == paired[f'tag_{i}_compare']\n    \ntag_match_cols = list(filter(lambda x: 'match' in x, paired.columns))\nfeature_names = ['feature','feature_compare']\npaired = paired[feature_names + tag_match_cols]\n\npaired['count'] = paired[tag_match_cols].sum(axis=1)\n\npaired = paired[feature_names + ['count']]","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:19:58.664302Z","iopub.execute_input":"2024-10-17T15:19:58.664767Z","iopub.status.idle":"2024-10-17T15:19:58.697618Z","shell.execute_reply.started":"2024-10-17T15:19:58.664711Z","shell.execute_reply":"2024-10-17T15:19:58.696373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nfig = go.Figure(data=go.Heatmap(\n                   z=paired['count'],\n                   x=paired['feature'],\n                   y=paired['feature_compare']))\n\nfig.update_layout(title='Cooccurence Matrix', width=800, height=800)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:28:33.070986Z","iopub.execute_input":"2024-10-17T15:28:33.071464Z","iopub.status.idle":"2024-10-17T15:28:33.131140Z","shell.execute_reply.started":"2024-10-17T15:28:33.071421Z","shell.execute_reply":"2024-10-17T15:28:33.129488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmax_count = paired['count'].max()\npaired.loc[(paired['feature'] != paired['feature_compare']) & (paired['count'] == max_count)]","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:36:21.085277Z","iopub.execute_input":"2024-10-17T15:36:21.085775Z","iopub.status.idle":"2024-10-17T15:36:21.106491Z","shell.execute_reply.started":"2024-10-17T15:36:21.085731Z","shell.execute_reply":"2024-10-17T15:36:21.105029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Responders","metadata":{}},{"cell_type":"code","source":"responders = pd.read_csv(f'{input_dir}responders.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:54:05.861365Z","iopub.execute_input":"2024-10-17T15:54:05.861954Z","iopub.status.idle":"2024-10-17T15:54:05.882174Z","shell.execute_reply.started":"2024-10-17T15:54:05.861887Z","shell.execute_reply":"2024-10-17T15:54:05.880547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"responder_cols = ['responder']\ntag_cols = list(filter(lambda x: 'tag' in x, responders.columns))\n\nmean_tags = responders[tag_cols].mean()\nsum_tags = responders[tag_cols].sum()\n\ntag_summary_df = pd.DataFrame()\ntag_summary_df['tag_id'] = mean_tags.index\ntag_summary_df['percentage'] = mean_tags.values\ntag_summary_df['count'] = sum_tags.values\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:55:15.750326Z","iopub.execute_input":"2024-10-17T15:55:15.750775Z","iopub.status.idle":"2024-10-17T15:55:15.763835Z","shell.execute_reply.started":"2024-10-17T15:55:15.750733Z","shell.execute_reply":"2024-10-17T15:55:15.762419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"responders","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:55:16.291047Z","iopub.execute_input":"2024-10-17T15:55:16.292171Z","iopub.status.idle":"2024-10-17T15:55:16.307620Z","shell.execute_reply.started":"2024-10-17T15:55:16.292110Z","shell.execute_reply":"2024-10-17T15:55:16.306426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tag_summary_df","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:55:17.619468Z","iopub.execute_input":"2024-10-17T15:55:17.619931Z","iopub.status.idle":"2024-10-17T15:55:17.633135Z","shell.execute_reply.started":"2024-10-17T15:55:17.619880Z","shell.execute_reply":"2024-10-17T15:55:17.631672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paired = responders.merge(responders, how='cross', suffixes=('','_compare'))\nfor i in range(0, 5):\n    paired[f'tag_{i}_match'] = paired[f'tag_{i}'] == paired[f'tag_{i}_compare']\n    \ntag_match_cols = list(filter(lambda x: 'match' in x, paired.columns))\nfeature_names = ['responder','responder_compare']\npaired = paired[feature_names + tag_match_cols]\n\npaired['count'] = paired[tag_match_cols].sum(axis=1)\n\npaired = paired[feature_names + ['count']]","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:57:03.826224Z","iopub.execute_input":"2024-10-17T15:57:03.826757Z","iopub.status.idle":"2024-10-17T15:57:03.849036Z","shell.execute_reply.started":"2024-10-17T15:57:03.826709Z","shell.execute_reply":"2024-10-17T15:57:03.847887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(data=go.Heatmap(\n                   z=paired['count'],\n                   x=paired['responder'],\n                   y=paired['responder_compare']))\n\nfig.update_layout(title='Cooccurence Matrix', width=800, height=800)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:57:26.080191Z","iopub.execute_input":"2024-10-17T15:57:26.080692Z","iopub.status.idle":"2024-10-17T15:57:26.097411Z","shell.execute_reply.started":"2024-10-17T15:57:26.080646Z","shell.execute_reply":"2024-10-17T15:57:26.096044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}