{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\n#!pip3 install torch==1.10.2+cu102 torchvision==0.11.3+cu102 torchaudio===0.10.2+cu102 -f https://download.pytorch.org/whl/cu102/torch_stable.html\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport plotly.express as px\n\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport lightgbm as lgb\nfrom sklearn.model_selection import *\nimport lightgbm as lgb\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom torch.autograd import Variable \nfrom tqdm import tqdm\nfrom torch.utils.data.sampler import WeightedRandomSampler\nimport lightgbm as lgb\nfrom lightgbm import early_stopping, print_evaluation\nimport ubiquant\nenv = ubiquant.make_env()  ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-06T01:09:56.787839Z","iopub.execute_input":"2022-04-06T01:09:56.788892Z","iopub.status.idle":"2022-04-06T01:10:01.134970Z","shell.execute_reply.started":"2022-04-06T01:09:56.788769Z","shell.execute_reply":"2022-04-06T01:10:01.133282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['f_0', 'f_1',\n       'f_2', 'f_3', 'f_4', 'f_5', 'f_6', 'f_7', 'f_8', 'f_9', 'f_10',\n       'f_11', 'f_12', 'f_13', 'f_14', 'f_15', 'f_16', 'f_17', 'f_18',\n       'f_19', 'f_20', 'f_21', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26',\n       'f_27', 'f_28', 'f_29', 'f_30', 'f_31', 'f_32', 'f_33', 'f_34',\n       'f_35', 'f_36', 'f_37', 'f_38', 'f_39', 'f_40', 'f_41', 'f_42',\n       'f_43', 'f_44', 'f_45', 'f_46', 'f_47', 'f_48', 'f_49', 'f_50',\n       'f_51', 'f_52', 'f_53', 'f_54', 'f_55', 'f_56', 'f_57', 'f_58',\n       'f_59', 'f_60', 'f_61', 'f_62', 'f_63', 'f_64', 'f_65', 'f_66',\n       'f_67', 'f_68', 'f_69', 'f_70', 'f_71', 'f_72', 'f_73', 'f_74',\n       'f_75', 'f_76', 'f_77', 'f_78', 'f_79', 'f_80', 'f_81', 'f_82',\n       'f_83', 'f_84', 'f_85', 'f_86', 'f_87', 'f_88', 'f_89', 'f_90',\n       'f_91', 'f_92', 'f_93', 'f_94', 'f_95', 'f_96', 'f_97', 'f_98',\n       'f_99', 'f_100', 'f_101', 'f_102', 'f_103', 'f_104', 'f_105',\n       'f_106', 'f_107', 'f_108', 'f_109', 'f_110', 'f_111', 'f_112',\n       'f_113', 'f_114', 'f_115', 'f_116', 'f_117', 'f_118', 'f_119',\n       'f_120', 'f_121', 'f_122', 'f_123', 'f_124', 'f_125', 'f_126',\n       'f_127', 'f_128', 'f_129', 'f_130', 'f_131', 'f_132', 'f_133',\n       'f_134', 'f_135', 'f_136', 'f_137', 'f_138', 'f_139', 'f_140',\n       'f_141', 'f_142', 'f_143', 'f_144', 'f_145', 'f_146', 'f_147',\n       'f_148', 'f_149', 'f_150', 'f_151', 'f_152', 'f_153', 'f_154',\n       'f_155', 'f_156', 'f_157', 'f_158', 'f_159', 'f_160', 'f_161',\n       'f_162', 'f_163', 'f_164', 'f_165', 'f_166', 'f_167', 'f_168',\n       'f_169', 'f_170', 'f_171', 'f_172', 'f_173', 'f_174', 'f_175',\n       'f_176', 'f_177', 'f_178', 'f_179', 'f_180', 'f_181', 'f_182',\n       'f_183', 'f_184', 'f_185', 'f_186', 'f_187', 'f_188', 'f_189',\n       'f_190', 'f_191', 'f_192', 'f_193', 'f_194', 'f_195', 'f_196',\n       'f_197', 'f_198', 'f_199', 'f_200', 'f_201', 'f_202', 'f_203',\n       'f_204', 'f_205', 'f_206', 'f_207', 'f_208', 'f_209', 'f_210',\n       'f_211', 'f_212', 'f_213', 'f_214', 'f_215', 'f_216', 'f_217',\n       'f_218', 'f_219', 'f_220', 'f_221', 'f_222', 'f_223', 'f_224',\n       'f_225', 'f_226', 'f_227', 'f_228', 'f_229', 'f_230', 'f_231',\n       'f_232', 'f_233', 'f_234', 'f_235', 'f_236', 'f_237', 'f_238',\n       'f_239', 'f_240', 'f_241', 'f_242', 'f_243', 'f_244', 'f_245',\n       'f_246', 'f_247', 'f_248', 'f_249', 'f_250', 'f_251', 'f_252',\n       'f_253', 'f_254', 'f_255', 'f_256', 'f_257', 'f_258', 'f_259',\n       'f_260', 'f_261', 'f_262', 'f_263', 'f_264', 'f_265', 'f_266',\n       'f_267', 'f_268', 'f_269', 'f_270', 'f_271', 'f_272', 'f_273',\n       'f_274', 'f_275', 'f_276', 'f_277', 'f_278', 'f_279', 'f_280',\n       'f_281', 'f_282', 'f_283', 'f_284', 'f_285', 'f_286', 'f_287',\n       'f_288', 'f_289', 'f_290', 'f_291', 'f_292', 'f_293', 'f_294',\n       'f_295', 'f_296', 'f_297', 'f_298', 'f_299',\"Missing\"]\n\nfor i in ['f_170','f_272','f_182','f_124','f_200','f_175','f_102','f_153','f_108','f_8','f_145', 'f_225', 'f_241', 'f_63', 'f_229', 'f_246', 'f_41', 'f_66', 'f_142', 'f_150', 'f_99', 'f_74', 'f_62', 'f_271']:\n    features.remove(i)","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:10:01.137390Z","iopub.execute_input":"2022-04-06T01:10:01.137717Z","iopub.status.idle":"2022-04-06T01:10:01.157156Z","shell.execute_reply.started":"2022-04-06T01:10:01.137673Z","shell.execute_reply":"2022-04-06T01:10:01.155239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import train data\ndtypes = {\n    'row_id': 'str',\n    'time_id': 'uint16',\n    'investment_id': 'uint16',\n    'target': 'float32',\n}\n\nfor i in range(300):\n    dtypes[f'f_{i}'] = 'float32'\n    \ntrainSet = pd.read_csv('../input/ubiquant-market-prediction/train.csv',dtype=dtypes)\n\npath = '../input/ubiquant-market-prediction/supplemental_train.csv'\nsup_train = pd.read_csv(path,usecols=list(dtypes.keys()),dtype=dtypes)\nsup_train.loc[sup_train.time_id<1217,\"time_id\"] = 1217\ntrainSet = pd.concat([trainSet[trainSet[\"time_id\"]>=849],sup_train],axis=0)\ndel sup_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:10:01.159073Z","iopub.execute_input":"2022-04-06T01:10:01.159718Z","iopub.status.idle":"2022-04-06T01:18:52.692404Z","shell.execute_reply.started":"2022-04-06T01:10:01.159590Z","shell.execute_reply":"2022-04-06T01:18:52.691406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainSet[\"TimeIdPrev\"] = trainSet.groupby(\"investment_id\")[\"time_id\"].shift(1)\ntrainSet[\"Missing\"] = 0\ntrainSet.loc[trainSet[\"TimeIdPrev\"]+1 != trainSet[\"time_id\"] ,\"Missing\"] = 1\n\ntrainSet=trainSet[trainSet[\"time_id\"]>=850] #600\n\ntrainSet[\"target\"] = trainSet.groupby(\"time_id\")[\"target\"].transform(lambda x: (x-x.mean())/x.std())","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:18:52.694666Z","iopub.execute_input":"2022-04-06T01:18:52.694958Z","iopub.status.idle":"2022-04-06T01:18:56.302205Z","shell.execute_reply.started":"2022-04-06T01:18:52.694924Z","shell.execute_reply":"2022-04-06T01:18:56.301219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainSet.replace([np.inf, -np.inf], np.nan, inplace=True)\ntrainSet.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:18:56.304991Z","iopub.execute_input":"2022-04-06T01:18:56.305396Z","iopub.status.idle":"2022-04-06T01:19:00.442517Z","shell.execute_reply.started":"2022-04-06T01:18:56.305345Z","shell.execute_reply":"2022-04-06T01:19:00.441530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ndef createLightgbm():\n    targetVar = \"target\"\n    param = {'objective': 'regression', 'learning_rate':.01, 'max_depth':-1, 'force_col_wise':True, 'lambda_l1': 9.616126068156222, 'lambda_l2': 5.468355127462534, 'num_leaves': 102, 'feature_fraction': 0.41497105932904305, 'bagging_fraction': 0.9931866062016639, 'bagging_freq': 4, 'max_depth': -1, 'max_bin': 251, 'min_data_in_leaf': 42}\n    num_round = 1000\n\n    weights = (np.array(trainSet.loc[:,\"time_id\"]-650)) / (trainSet.loc[:,\"time_id\"].max()-650)\n    lightModel = lgb.train(param, lgb.Dataset(trainSet.loc[:,features], label=trainSet.loc[:,targetVar], weight=weights), num_round)\n    return lightModel\nlightModel = createLightgbm()\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:19:00.444342Z","iopub.execute_input":"2022-04-06T01:19:00.444564Z","iopub.status.idle":"2022-04-06T01:19:00.451037Z","shell.execute_reply.started":"2022-04-06T01:19:00.444536Z","shell.execute_reply":"2022-04-06T01:19:00.450250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missingIndex = 1\nindexMap = {}\nfor index, row in trainSet[trainSet[\"time_id\"] == trainSet[\"time_id\"].max()].iterrows():\n    indexMap[row.investment_id] = missingIndex\nmissingIndex+=1","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:19:00.452402Z","iopub.execute_input":"2022-04-06T01:19:00.452647Z","iopub.status.idle":"2022-04-06T01:19:00.468676Z","shell.execute_reply.started":"2022-04-06T01:19:00.452602Z","shell.execute_reply":"2022-04-06T01:19:00.467695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data.sampler import WeightedRandomSampler\ntargetVar = \"target\"\ntrain = torch.as_tensor(trainSet.loc[:,features].to_numpy(), dtype=torch.float32)\ntrainLabel = torch.as_tensor(trainSet.loc[:,targetVar].to_numpy(), dtype=torch.float32)\ntimeIdTrain = torch.tensor(trainSet.loc[:,\"time_id\"].to_numpy().astype(np.float32)).long()\ntrainDataset = TensorDataset(train,trainLabel,timeIdTrain)\nweights = (np.array(trainSet.loc[:,\"time_id\"]-650)) / (trainSet.loc[:,\"time_id\"].max()-650)\nsampler = WeightedRandomSampler(weights, len(weights))\ndataloaderTrain = DataLoader(trainDataset,batch_size=128, sampler=sampler, shuffle=False)\ndataloaderTrain2 = DataLoader(trainDataset,batch_size=1000,sampler=sampler, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:19:00.470459Z","iopub.execute_input":"2022-04-06T01:19:00.470786Z","iopub.status.idle":"2022-04-06T01:19:07.224236Z","shell.execute_reply.started":"2022-04-06T01:19:00.470752Z","shell.execute_reply":"2022-04-06T01:19:07.223026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainSet\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:19:07.225771Z","iopub.execute_input":"2022-04-06T01:19:07.227286Z","iopub.status.idle":"2022-04-06T01:19:07.464267Z","shell.execute_reply.started":"2022-04-06T01:19:07.227229Z","shell.execute_reply":"2022-04-06T01:19:07.463405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numFeat = len(features)-1\nclass MLP(nn.Module):\n    def __init__(self, input_size):\n        super(MLP, self).__init__()\n        self.bn = nn.BatchNorm1d(numFeat)\n        self.Input = nn.Sequential(nn.Dropout(.1),nn.Linear(input_size,1000),nn.BatchNorm1d(1000),nn.Dropout(.5),nn.ReLU(),nn.Linear(1000,512),nn.Dropout(.25),nn.ReLU(),nn.Linear(512,1))\n    def forward(self, data, isTrain=False):\n        x = torch.cat([self.bn(data[:,0:numFeat]),data[:,numFeat:]],1)\n        x = self.Input(x)\n        return x\ncos = nn.CosineSimilarity(dim=0, eps=1e-6)\ndef cosLoss(pred,label):\n    return -cos(pred,label)\ndef variance(preds, labels):\n    x = preds - preds.mean()\n    y = labels - labels.mean()\n    return (-(x*y)).mean()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:19:07.466535Z","iopub.execute_input":"2022-04-06T01:19:07.467714Z","iopub.status.idle":"2022-04-06T01:19:07.483971Z","shell.execute_reply.started":"2022-04-06T01:19:07.467600Z","shell.execute_reply":"2022-04-06T01:19:07.482772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epoch = 21\ndevice=\"cpu\"\ninput_size = len(features)\ncriterion = nn.MSELoss()\ncriterion2 = variance\ncriterion3 = nn.L1Loss()\nlr = .00002\nvarLR = .000006/10\n\nmodel1a = []\nmodel1b = []","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:19:07.485755Z","iopub.execute_input":"2022-04-06T01:19:07.486360Z","iopub.status.idle":"2022-04-06T01:19:07.503547Z","shell.execute_reply.started":"2022-04-06T01:19:07.486313Z","shell.execute_reply":"2022-04-06T01:19:07.501846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MLPModel = MLP(input_size).to(device)\noptimizer = optim.Adam(params = MLPModel.parameters(), lr=.00002)\nfor e in range(epoch):\n\n    for data, labels, time_id in tqdm(dataloaderTrain):\n        optimizer.zero_grad()\n        data = data.to(device)\n        labels = labels.to(device)\n        preds = MLPModel(data, isTrain=True)\n\n        loss = criterion(preds.reshape(-1), labels)\n\n        loss.backward()\n        optimizer.step()\n    if e==10 or e==15 or e==20:\n        model1a.append(MLPModel.state_dict())\n\n    for g in optimizer.param_groups:\n        lr = g[\"lr\"]\n        g[\"lr\"]=varLR\n\n    for g in optimizer.param_groups:\n        g[\"lr\"]=lr*.85\n        \n    if e==10 or e==15 or e==20:\n        model1b.append(MLPModel.state_dict())\nmodel1 = []\nfor checkpoint in model1a:\n    MLPModel = MLP(input_size).to(device)\n    MLPModel.load_state_dict(checkpoint)\n    model1.append(MLPModel)\nfor checkpoint in model1b:\n    MLPModel = MLP(input_size).to(device)\n    MLPModel.load_state_dict(checkpoint)\n    model1.append(MLPModel)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T01:19:07.505797Z","iopub.execute_input":"2022-04-06T01:19:07.506317Z","iopub.status.idle":"2022-04-06T04:18:37.454388Z","shell.execute_reply.started":"2022-04-06T01:19:07.506160Z","shell.execute_reply":"2022-04-06T04:18:37.453492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for model in model1:\n    model.eval()\nfeatures.pop()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T04:18:37.456690Z","iopub.execute_input":"2022-04-06T04:18:37.457180Z","iopub.status.idle":"2022-04-06T04:18:37.466799Z","shell.execute_reply.started":"2022-04-06T04:18:37.457139Z","shell.execute_reply":"2022-04-06T04:18:37.466048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iter_test = env.iter_test()\nfor (test_df, sample_prediction_df) in iter_test:\n    test_df.replace([np.inf, -np.inf], np.nan, inplace=True)\n    test_df.fillna(0, inplace=True)\n    try:\n        missing = np.zeros(test_df.shape[0])\n        for index in range(test_df.shape[0]):\n            row = test_df.iloc[index]\n            lastTimeId = indexMap.get(row.investment_id,-1)\n            diffLastTime = (missingIndex - lastTimeId)-1\n            if (diffLastTime) and (lastTimeId!=-1):\n                missing[index] = 1\n\n            indexMap[row.investment_id] = missingIndex\n\n        data = np.concatenate([test_df[features].to_numpy(),(missing.reshape(-1,1))],1)\n\n        with torch.no_grad():\n\n            preds1 = torch.zeros(data.shape[0])\n            data = torch.as_tensor(data, dtype=torch.float32).to(device)\n            for model in model1:\n                preds1 = preds1 + model(data).cpu()/6\n\n            sample_prediction_df['target'] = preds1.numpy()\n\n            sample_prediction_df.replace([np.inf, -np.inf], np.nan, inplace=True)\n            sample_prediction_df.fillna(0, inplace=True)\n\n            env.predict(sample_prediction_df)\n    except Exception:\n        sample_prediction_df['target'] = np.random.randn(len(sample_prediction_df))\n\n        env.predict(sample_prediction_df)\n    missingIndex+=1","metadata":{"execution":{"iopub.status.busy":"2022-04-06T04:18:37.467878Z","iopub.execute_input":"2022-04-06T04:18:37.468469Z","iopub.status.idle":"2022-04-06T04:18:37.585389Z","shell.execute_reply.started":"2022-04-06T04:18:37.468438Z","shell.execute_reply":"2022-04-06T04:18:37.584786Z"},"trusted":true},"execution_count":null,"outputs":[]}]}