{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# IT IS ONLY DRAFT NOW","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nimport gc\n\nimport numpy as np\nimport pandas as pd\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm\n\nif not os.path.exists('/opt/conda/lib/python3.7/site-packages/tables'):\n    !pip install --quiet tables","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# READ METADATA\nmeta = pd.read_csv(\"../input/open-problems-multimodal/metadata.csv\", index_col = 'cell_id')\nmeta = meta[meta.technology == 'citeseq']\nmeta.drop('technology', axis = 1, inplace = True)\n\n#READ FEATURES\nX = pd.read_hdf('../input/open-problems-multimodal/train_cite_inputs.h5')\n\n# READ TARGETS\nY = pd.read_hdf('../input/open-problems-multimodal/train_cite_targets.h5')\n\n# REMOVE 2% OF OULIERS\ncols = list(Y.columns)\nfor i in tqdm(range(140)):\n    col = cols[i]\n    v = Y[col]\n    threshold = 1.0\n    m1 = np.percentile(v, threshold)\n    m2 = np.percentile(v, 100 - threshold)\n    v = np.clip(v, m1, m2)\n    Y[col] = v\n    \n# SHRINK META TO TRAIN SIZE\nmeta = meta[meta.index.isin(Y.index)]\n\n# MERGE TARGETS WITH METADATA\ndf = Y.join(meta)","metadata":{"execution":{"iopub.status.busy":"2022-10-12T11:09:39.304653Z","iopub.execute_input":"2022-10-12T11:09:39.304973Z","iopub.status.idle":"2022-10-12T11:10:39.564482Z","shell.execute_reply.started":"2022-10-12T11:09:39.304945Z","shell.execute_reply":"2022-10-12T11:10:39.562805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corrs = np.zeros([X.shape[1], Y.shape[1]])\ncorrs = pd.DataFrame(corrs)\ncorrs.columns = Y.columns\ncorrs.index = X.columns\ncorrs","metadata":{"execution":{"iopub.status.busy":"2022-10-12T11:10:39.567312Z","iopub.execute_input":"2022-10-12T11:10:39.567850Z","iopub.status.idle":"2022-10-12T11:10:39.640451Z","shell.execute_reply.started":"2022-10-12T11:10:39.567802Z","shell.execute_reply":"2022-10-12T11:10:39.639303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(140)):\n    y = Y[Y.columns[i]].values\n    for j in range(22050):\n        x = X[X.columns[j]].values\n        corrs[corrs.columns[i]][corrs.index[j]] = np.corrcoef(y, x)[1, 0]\ncorrs","metadata":{"execution":{"iopub.status.busy":"2022-10-12T11:10:39.642965Z","iopub.execute_input":"2022-10-12T11:10:39.643338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top100fits = np.zeros([100, Y.shape[1]])\ntop100fits = pd.DataFrame(top100fits)\ntop100fits.columns = Y.columns\n# top100fits","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(140)):\n    df0 = corrs[[corrs.columns[i]]].copy()\n    df0 = df0.sort_values(corrs.columns[i], axis=0, ascending=False)\n#     print(df0[:100].index)\n    top100fits[top100fits.columns[i]] = df0[:100].index","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top100fits","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}