{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q tables\n\nimport numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nprint(\"Done\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-17T12:02:18.549763Z","iopub.execute_input":"2022-08-17T12:02:18.551107Z","iopub.status.idle":"2022-08-17T12:02:31.198139Z","shell.execute_reply.started":"2022-08-17T12:02:18.551061Z","shell.execute_reply":"2022-08-17T12:02:31.196339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Constants","metadata":{}},{"cell_type":"code","source":"INPUT_DIR = \"../input/open-problems-multimodal\"\n\nEVALUATION_DIR = os.path.join(INPUT_DIR, \"evaluation_ids.csv\")\nMETADATA_DIR = os.path.join(INPUT_DIR, \"metadata.csv\")\nSUBMISSION_DIR = os.path.join(INPUT_DIR, \"sample_submission.csv\")\n\nMULTIOME_TRAIN_INPUTS = os.path.join(INPUT_DIR,\"train_multi_inputs.h5\")\nMULTIOME_TRAIN_TARGETS = os.path.join(INPUT_DIR,\"train_multi_targets.h5\")\nMULTIOME_TEST_INPUTS = os.path.join(INPUT_DIR,\"test_multi_inputs.h5\")\nCITE_TRAIN_INPUTS = os.path.join(INPUT_DIR,\"train_cite_inputs.h5\")\nCITE_TRAIN_TARGETS = os.path.join(INPUT_DIR,\"train_cite_targets.h5\")\nCITE_TEST_INPUTS = os.path.join(INPUT_DIR,\"test_cite_inputs.h5\")\nSUBMISSION_PATH = os.path.join(INPUT_DIR,\"sample_submission.csv\")\nEVALUATION_IDS = os.path.join(INPUT_DIR,\"evaluation_ids.csv\")\n\nSTART = int(1e4)\nSTOP = START+10000\n\nROW_ID = \"row_id\"\nTARGET = \"target\"\nGENE_ID_INT = \"gene_id_int\"\nGENE_ID = \"gene_id\"\n\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:31.20244Z","iopub.execute_input":"2022-08-17T12:02:31.202866Z","iopub.status.idle":"2022-08-17T12:02:31.216962Z","shell.execute_reply.started":"2022-08-17T12:02:31.20283Z","shell.execute_reply":"2022-08-17T12:02:31.215577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions ","metadata":{}},{"cell_type":"code","source":"def data_description(df):\n    print(\"Data description\")\n    print(f\"Total number of records {df.shape[0]}\")\n    print(f'number of features {df.shape[1]}\\n\\n')\n    columns = df.columns\n    data_type = []\n    \n    # Get the datatype of features\n    for col in df.columns:\n        data_type.append(df[col].dtype)\n        \n    n_uni = df.nunique()\n    # Number of NaN values\n    n_miss = df.isna().sum()\n    \n    names = list(zip(columns, data_type, n_uni, n_miss))\n    variable_desc = pd.DataFrame(names, columns=[\"Name\",\"Type\",\"Unique levels\",\"Missing\"])\n    print(variable_desc)","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:31.219806Z","iopub.execute_input":"2022-08-17T12:02:31.220386Z","iopub.status.idle":"2022-08-17T12:02:31.231317Z","shell.execute_reply.started":"2022-08-17T12:02:31.220314Z","shell.execute_reply":"2022-08-17T12:02:31.230301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def transform_df(df, column, int_col, drop_col):\n    df = pd.DataFrame(df, columns = [column]).reset_index()\n    df[int_col] = df[drop_col].apply(lambda x: int(x.replace(\"-\",\"\").replace(\".\",\"\")[-8:],34)).astype(int)\n    df.drop([drop_col], axis = 1, inplace = True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:31.235443Z","iopub.execute_input":"2022-08-17T12:02:31.235872Z","iopub.status.idle":"2022-08-17T12:02:31.245526Z","shell.execute_reply.started":"2022-08-17T12:02:31.235837Z","shell.execute_reply":"2022-08-17T12:02:31.24402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preparation","metadata":{}},{"cell_type":"code","source":"train_cite_targ = pd.read_hdf(CITE_TRAIN_TARGETS)\nmetadata = pd.read_csv(METADATA_DIR)\n\ntrain_multi_targ = pd.read_hdf(MULTIOME_TRAIN_TARGETS, start=START, stop=STOP)\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:54:40.939934Z","iopub.execute_input":"2022-08-17T12:54:40.94041Z","iopub.status.idle":"2022-08-17T12:54:49.042968Z","shell.execute_reply.started":"2022-08-17T12:54:40.940369Z","shell.execute_reply":"2022-08-17T12:54:49.04152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:55:29.512333Z","iopub.execute_input":"2022-08-17T12:55:29.512854Z","iopub.status.idle":"2022-08-17T12:55:29.526945Z","shell.execute_reply.started":"2022-08-17T12:55:29.512815Z","shell.execute_reply":"2022-08-17T12:55:29.525647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_targ.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:45:00.778535Z","iopub.execute_input":"2022-08-17T12:45:00.779091Z","iopub.status.idle":"2022-08-17T12:45:00.814025Z","shell.execute_reply.started":"2022-08-17T12:45:00.779008Z","shell.execute_reply":"2022-08-17T12:45:00.812584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_description(train_cite_targ)\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:38.656329Z","iopub.execute_input":"2022-08-17T12:02:38.656856Z","iopub.status.idle":"2022-08-17T12:02:39.139721Z","shell.execute_reply.started":"2022-08-17T12:02:38.656807Z","shell.execute_reply":"2022-08-17T12:02:39.138421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_multi_targ.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:39.141326Z","iopub.execute_input":"2022-08-17T12:02:39.142131Z","iopub.status.idle":"2022-08-17T12:02:39.180424Z","shell.execute_reply.started":"2022-08-17T12:02:39.142084Z","shell.execute_reply":"2022-08-17T12:02:39.179118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_description(train_multi_targ)","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:39.182201Z","iopub.execute_input":"2022-08-17T12:02:39.182583Z","iopub.status.idle":"2022-08-17T12:02:46.301685Z","shell.execute_reply.started":"2022-08-17T12:02:39.18255Z","shell.execute_reply":"2022-08-17T12:02:46.300445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multi_gene_id_mean = train_multi_targ.mean()\nmulti_gene_id_mean\n","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:46.303029Z","iopub.execute_input":"2022-08-17T12:02:46.303414Z","iopub.status.idle":"2022-08-17T12:02:46.881801Z","shell.execute_reply.started":"2022-08-17T12:02:46.303379Z","shell.execute_reply":"2022-08-17T12:02:46.880427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cite_gene_id_mean = train_cite_targ.mean()\ncite_gene_id_mean","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:46.88577Z","iopub.execute_input":"2022-08-17T12:02:46.886133Z","iopub.status.idle":"2022-08-17T12:02:46.916478Z","shell.execute_reply.started":"2022-08-17T12:02:46.886102Z","shell.execute_reply":"2022-08-17T12:02:46.915418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cite_gene_id_mean.index","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:46.917739Z","iopub.execute_input":"2022-08-17T12:02:46.918165Z","iopub.status.idle":"2022-08-17T12:02:46.925405Z","shell.execute_reply.started":"2022-08-17T12:02:46.918134Z","shell.execute_reply":"2022-08-17T12:02:46.924416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multi_gene_id_mean.index","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:46.926729Z","iopub.execute_input":"2022-08-17T12:02:46.927075Z","iopub.status.idle":"2022-08-17T12:02:46.937602Z","shell.execute_reply.started":"2022-08-17T12:02:46.927047Z","shell.execute_reply":"2022-08-17T12:02:46.936403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gene_id_mean = list(cite_gene_id_mean.index) + list(multi_gene_id_mean.index)\ngene_id = pd.DataFrame(gene_id_mean, columns = [GENE_ID])\ngene_id.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:46.939335Z","iopub.execute_input":"2022-08-17T12:02:46.939827Z","iopub.status.idle":"2022-08-17T12:02:46.959388Z","shell.execute_reply.started":"2022-08-17T12:02:46.939791Z","shell.execute_reply":"2022-08-17T12:02:46.957913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gene_id[GENE_ID_INT] = gene_id[GENE_ID].apply(lambda x : int(x.replace(\"-\", \"\").replace(\".\",\"\")[-8:], 34)).astype(int)\ngene_id.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:46.9615Z","iopub.execute_input":"2022-08-17T12:02:46.962457Z","iopub.status.idle":"2022-08-17T12:02:47.002658Z","shell.execute_reply.started":"2022-08-17T12:02:46.962407Z","shell.execute_reply":"2022-08-17T12:02:47.001259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_description(gene_id)","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:47.003984Z","iopub.execute_input":"2022-08-17T12:02:47.004575Z","iopub.status.idle":"2022-08-17T12:02:47.024098Z","shell.execute_reply.started":"2022-08-17T12:02:47.004533Z","shell.execute_reply":"2022-08-17T12:02:47.023149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(SUBMISSION_PATH, usecols = [ROW_ID])\ndata_description(submission)","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:02:47.025182Z","iopub.execute_input":"2022-08-17T12:02:47.02597Z","iopub.status.idle":"2022-08-17T12:03:04.508532Z","shell.execute_reply.started":"2022-08-17T12:02:47.025934Z","shell.execute_reply":"2022-08-17T12:03:04.507045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluation = pd.read_csv(EVALUATION_IDS, usecols=[ROW_ID, GENE_ID])\nevaluation[GENE_ID_INT] = evaluation[GENE_ID].apply(lambda x: int(x.replace('-', '').replace('.', '')[-8:],34)).astype(int)\nevaluation.drop([GENE_ID], axis=1, inplace=True)\ndata_description(evaluation)","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:03:04.510716Z","iopub.execute_input":"2022-08-17T12:03:04.511278Z","iopub.status.idle":"2022-08-17T12:04:56.316926Z","shell.execute_reply.started":"2022-08-17T12:03:04.511219Z","shell.execute_reply":"2022-08-17T12:04:56.315148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluation.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:58:26.174544Z","iopub.execute_input":"2022-08-17T12:58:26.174958Z","iopub.status.idle":"2022-08-17T12:58:26.18681Z","shell.execute_reply.started":"2022-08-17T12:58:26.174928Z","shell.execute_reply":"2022-08-17T12:58:26.18572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submission.merge(evaluation, how = \"left\", on = ROW_ID)\ndata_description(submission)","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:04:56.318718Z","iopub.execute_input":"2022-08-17T12:04:56.321237Z","iopub.status.idle":"2022-08-17T12:05:53.604236Z","shell.execute_reply.started":"2022-08-17T12:04:56.321189Z","shell.execute_reply":"2022-08-17T12:05:53.60341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cite_gene_id_mean = transform_df(cite_gene_id_mean, TARGET, GENE_ID_INT, GENE_ID)\ncite_gene_id_mean.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:05:53.605466Z","iopub.execute_input":"2022-08-17T12:05:53.606262Z","iopub.status.idle":"2022-08-17T12:05:53.620949Z","shell.execute_reply.started":"2022-08-17T12:05:53.606228Z","shell.execute_reply":"2022-08-17T12:05:53.619348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multi_gene_id_mean = transform_df(multi_gene_id_mean, TARGET, GENE_ID_INT, GENE_ID)\nmulti_gene_id_mean.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:05:53.622585Z","iopub.execute_input":"2022-08-17T12:05:53.622944Z","iopub.status.idle":"2022-08-17T12:05:53.659305Z","shell.execute_reply.started":"2022-08-17T12:05:53.622914Z","shell.execute_reply":"2022-08-17T12:05:53.658414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"completed_gene_id_mean = pd.concat([cite_gene_id_mean, multi_gene_id_mean])\ndata_description(completed_gene_id_mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:05:53.660612Z","iopub.execute_input":"2022-08-17T12:05:53.661142Z","iopub.status.idle":"2022-08-17T12:05:53.675464Z","shell.execute_reply.started":"2022-08-17T12:05:53.661111Z","shell.execute_reply":"2022-08-17T12:05:53.674551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submission.merge(completed_gene_id_mean, how = \"left\", on = GENE_ID_INT)\ndata_description(submission)","metadata":{"execution":{"iopub.status.busy":"2022-08-17T12:05:53.676775Z","iopub.execute_input":"2022-08-17T12:05:53.677329Z","iopub.status.idle":"2022-08-17T12:06:17.55045Z","shell.execute_reply.started":"2022-08-17T12:05:53.677296Z","shell.execute_reply":"2022-08-17T12:06:17.549115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_submission[[ROW_ID, TARGET]].to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}