{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Install TALIB\n!cp ../input/talibinstall/ta-lib-0.4.0-src.tar.gzh  ./ta-lib-0.4.0-src.tar.gz\n!tar -xzvf ta-lib-0.4.0-src.tar.gz > null\n!cd ta-lib && ./configure --prefix=/usr > null && make  > null && make install > null\n!cp ../input/talibinstall/TA-Lib-0.4.21.tar.gzh TA-Lib-0.4.21.tar.gz\n!pip install TA-Lib-0.4.21.tar.gz > null\n!pip install ../input/talibinstall/numpy-1.21.4-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl >null","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:23:44.589158Z","iopub.execute_input":"2022-08-14T09:23:44.589764Z","iopub.status.idle":"2022-08-14T09:25:51.564660Z","shell.execute_reply.started":"2022-08-14T09:23:44.589686Z","shell.execute_reply":"2022-08-14T09:25:51.563230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport jpx_tokyo_market_prediction\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# technical analysis library\nimport talib\n\n# LGMB model\nimport lightgbm as lgbm\nfrom sklearn import tree\nfrom random import random\n\n# load pre-trained model\nimport joblib\n\nfrom datetime import datetime,timedelta\n\nfrom sklearn.preprocessing import RobustScaler, MinMaxScaler, LabelEncoder","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-14T09:25:51.567007Z","iopub.execute_input":"2022-08-14T09:25:51.567385Z","iopub.status.idle":"2022-08-14T09:25:51.588732Z","shell.execute_reply.started":"2022-08-14T09:25:51.567341Z","shell.execute_reply":"2022-08-14T09:25:51.587811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Features","metadata":{}},{"cell_type":"code","source":"fin_docs = ['3QFinancialStatements_Consolidated_JP',\n            'FYFinancialStatements_Consolidated_JP',\n            '1QFinancialStatements_Consolidated_JP',\n            '2QFinancialStatements_Consolidated_JP', \n            '2QFinancialStatements_Consolidated_IFRS',\n            '3QFinancialStatements_Consolidated_IFRS',\n            'FYFinancialStatements_Consolidated_IFRS',\n            '1QFinancialStatements_Consolidated_IFRS',\n            'FYFinancialStatements_NonConsolidated_JP',\n            '1QFinancialStatements_NonConsolidated_JP',\n            '2QFinancialStatements_NonConsolidated_JP',\n            '3QFinancialStatements_NonConsolidated_JP',\n            '3QFinancialStatements_Consolidated_US',\n            'FYFinancialStatements_Consolidated_US',\n            '1QFinancialStatements_Consolidated_US',\n            '2QFinancialStatements_Consolidated_US',\n            'OtherPeriodFinancialStatements_Consolidated_JP',\n            '2QFinancialStatements_NonConsolidated_IFRS',\n            'FYFinancialStatements_NonConsolidated_IFRS',\n            '1QFinancialStatements_NonConsolidated_IFRS',\n            '3QFinancialStatements_NonConsolidated_IFRS']\n\nforecast_docs = ['ForecastRevision','ForecastRevision_REIT']\n\nfin_features = ['EarningsPerShare','ForecastEarningsPerShare',\n              'AverageNumberOfShares','EquityToAssetRatio','BookValuePerShare','CurrentPeriodEndDate']\n\nforecast_features = ['ForecastEarningsPerShare','CurrentPeriodEndDate']\n\nfeatures = ['Open', 'High', 'Low', 'Close', 'Volume','weekday', 'month',\n            'year', 'ret_1', 'ret_rel_perc_1', 'ret_5', 'ret_rel_perc_5', 'ret_10',\n            'ret_rel_perc_10', 'ret_21', 'ret_rel_perc_21', 'ret_63',\n            'ret_rel_perc_63','bbl', 'bbu', 'P/E','P/F','P/B', \n            'Section', 'NewMarketSegment','section_ret_diff', 'mkt_ret_diff','beta_ret_diff']\ntarget = [\"Target\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.590678Z","iopub.execute_input":"2022-08-14T09:25:51.590980Z","iopub.status.idle":"2022-08-14T09:25:51.600561Z","shell.execute_reply.started":"2022-08-14T09:25:51.590946Z","shell.execute_reply":"2022-08-14T09:25:51.599488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def treat_stocks(raw_stocks):\n  # filter for the ones in our prediction universe\n  stocks = raw_stocks.query('Universe0 == True').set_index(['SecuritiesCode']).sort_index()\n  stocks = stocks[['Section/Products','NewMarketSegment']]\n\n  stocks.rename(columns = {'Section/Products' : 'Section'}, inplace=True)\n  # initialize LabelEncoder\n  lbl_enc = LabelEncoder()\n  # encode\n  stocks['Section'] = lbl_enc.fit_transform(stocks['Section'])\n  stocks['NewMarketSegment'] = lbl_enc.fit_transform(stocks['NewMarketSegment'])\n\n  return stocks","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.602141Z","iopub.execute_input":"2022-08-14T09:25:51.602429Z","iopub.status.idle":"2022-08-14T09:25:51.618149Z","shell.execute_reply.started":"2022-08-14T09:25:51.602397Z","shell.execute_reply":"2022-08-14T09:25:51.617025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_date(prices):\n  dates = prices.index.get_level_values('Date')\n  prices['weekday'] = dates.weekday\n  prices['month'] = dates.month\n  prices['year'] = dates.year","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.619348Z","iopub.execute_input":"2022-08-14T09:25:51.619632Z","iopub.status.idle":"2022-08-14T09:25:51.633667Z","shell.execute_reply.started":"2022-08-14T09:25:51.619595Z","shell.execute_reply":"2022-08-14T09:25:51.632774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def treat_na(prices):\n  return prices.fillna(method='bfill')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.635407Z","iopub.execute_input":"2022-08-14T09:25:51.635687Z","iopub.status.idle":"2022-08-14T09:25:51.644753Z","shell.execute_reply.started":"2022-08-14T09:25:51.635656Z","shell.execute_reply":"2022-08-14T09:25:51.643607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Boellinger bandas\ndef get_bollinger(x):\n    u, m, l = talib.BBANDS(x)\n    return pd.DataFrame({'u': u, 'm': m, 'l': l})","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.646605Z","iopub.execute_input":"2022-08-14T09:25:51.647583Z","iopub.status.idle":"2022-08-14T09:25:51.657127Z","shell.execute_reply.started":"2022-08-14T09:25:51.647528Z","shell.execute_reply":"2022-08-14T09:25:51.655877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_returns(df,intervals):\n  returns = []\n  by_sym = df.groupby(level='SecuritiesCode').Close\n  for t in intervals:\n      ret = by_sym.pct_change(t)\n      rel_perc = (ret.groupby(level='Date')\n              .apply(lambda x: pd.qcut(x, q=20, labels=False, duplicates='drop')))\n      returns.extend([ret.to_frame(f'ret_{t}'), rel_perc.to_frame(f'ret_rel_perc_{t}')])\n  \n  return pd.concat(returns, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.659250Z","iopub.execute_input":"2022-08-14T09:25:51.659895Z","iopub.status.idle":"2022-08-14T09:25:51.668942Z","shell.execute_reply.started":"2022-08-14T09:25:51.659852Z","shell.execute_reply":"2022-08-14T09:25:51.667917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_ti(df):\n  # percentage price oscillator\n  ppo = df.groupby(level='SecuritiesCode').Close.apply(talib.PPO).to_frame('PPO')\n  # boellinger \n  bbands = df.groupby(level='SecuritiesCode').Close.apply(get_bollinger)\n  # relative strength index\n  rsi = df.groupby(level='SecuritiesCode').Close.apply(talib.RSI).to_frame('RSI')\n  # normalized ATR\n  natr = df.groupby(level='SecuritiesCode', group_keys=False).apply(lambda x: talib.NATR(x.High, x.Low, x.Close)).to_frame('NATR')\n  \n  return  pd.concat([ppo, natr, rsi, bbands], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.671284Z","iopub.execute_input":"2022-08-14T09:25:51.671875Z","iopub.status.idle":"2022-08-14T09:25:51.683440Z","shell.execute_reply.started":"2022-08-14T09:25:51.671827Z","shell.execute_reply":"2022-08-14T09:25:51.682288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_features (raw_prices, raw_financials, stocks, securities, intervals = [1, 5, 10, 21, 63], from_date = None):\n  \n  # treat financials  \n  financials = raw_financials[raw_financials['SecuritiesCode'].isin(securities)]\n  financials.loc[: ,\"DisclosedDate\"] = pd.to_datetime(financials.loc[: ,\"DisclosedDate\"], format=\"%Y-%m-%d\")\n  financials.loc[: ,\"CurrentPeriodEndDate\"] = pd.to_datetime(financials.loc[: ,\"CurrentPeriodEndDate\"], format=\"%Y-%m-%d\")\n  financials.reset_index()\n  financials = financials.set_index(['SecuritiesCode', \"DisclosedDate\"]).sort_index()  \n\n  #treat prices\n  prices = raw_prices\n  prices.loc[: ,\"Date\"] = pd.to_datetime(prices.loc[: ,\"Date\"], format=\"%Y-%m-%d\")\n  prices.reset_index()\n  prices = prices.set_index(['SecuritiesCode', \"Date\"]).sort_index()\n\n  if from_date is not None:\n    prices = prices.loc(axis=0)[:,from_date:]\n    \n  # treat NA's\n  prices = treat_na(prices)\n  # create Returns\n  returns = create_returns(prices,intervals)\n  # technical indicators\n  ti = create_ti(prices)\n  # time factors\n  add_date(prices)\n  \n  train_prices = pd.concat([prices, returns, ti], axis=1)\n  train_prices['bbl'] = train_prices.Close.div(train_prices.l)\n  train_prices['bbu'] = train_prices.u.div(train_prices.Close)\n  train_prices = train_prices.drop(['u', 'm', 'l'], axis=1)\n\n  fin_data = financials[financials['TypeOfDocument'].isin(fin_docs)][fin_features]\n  forecast_data = financials[financials['TypeOfDocument'].isin(forecast_docs)][forecast_features]\n\n  fin_data = fin_data.groupby(['DisclosedDate', 'SecuritiesCode'], group_keys=False).apply(lambda x: x.nlargest(n=1, columns=['CurrentPeriodEndDate']))\n  forecast_data = forecast_data.groupby(['DisclosedDate', 'SecuritiesCode'], group_keys=False).apply(lambda x: x.nlargest(n=1, columns=['CurrentPeriodEndDate']))\n\n  fin_data.loc[:,'EarningsPerShare'] = pd.to_numeric(fin_data.loc[:,'EarningsPerShare'], errors='coerce').dropna()\n  fin_data.loc[:,'ForecastEarningsPerShare'] = pd.to_numeric(fin_data.loc[:,'ForecastEarningsPerShare'], errors='coerce').dropna()\n  fin_data.loc[:,'AverageNumberOfShares'] = pd.to_numeric(fin_data.loc[:,'AverageNumberOfShares'], errors='coerce').dropna()\n  fin_data.loc[:,'EquityToAssetRatio'] = pd.to_numeric(fin_data.loc[:,'EquityToAssetRatio'], errors='coerce').dropna()\n  fin_data.loc[:,'BookValuePerShare'] = pd.to_numeric(fin_data.loc[:,'BookValuePerShare'], errors='coerce').dropna()\n\n  # get the rolling average per security\n  fin_data = fin_data.groupby(\"SecuritiesCode\", group_keys=False).rolling(2, min_periods=2).sum().dropna()\n  fin_data = fin_data.droplevel(0)\n\n  fin_data.index.names = ['SecuritiesCode','Date']\n  prices = prices.join(fin_data, on=['SecuritiesCode','Date'])\n\n  prices.loc[: ,'EarningsPerShare'] = prices.loc[: ,'EarningsPerShare'].fillna(method='ffill').dropna()\n  prices.loc[: ,'ForecastEarningsPerShare'] = prices.loc[: ,'ForecastEarningsPerShare'].fillna(method='ffill').dropna()\n  prices.loc[: ,'AverageNumberOfShares'] = prices.loc[: ,'AverageNumberOfShares'].fillna(method='ffill').dropna()\n  prices.loc[: ,'EquityToAssetRatio'] = prices.loc[: ,'EquityToAssetRatio'].fillna(method='ffill').dropna()\n  prices.loc[: ,'BookValuePerShare'] = prices.loc[: ,'BookValuePerShare'].fillna(method='ffill').dropna()\n\n  prices['P/E'] = prices.Close.div(prices.EarningsPerShare)\n  prices['P/F'] = prices.Close.div(prices.ForecastEarningsPerShare)\n  prices['P/B'] = prices.Close.div(prices.BookValuePerShare)\n\n  train_prices = pd.concat([prices[['P/E','P/F','P/B']], train_prices], axis=1)\n\n  train_prices = train_prices.join(stocks[['Section','NewMarketSegment']])\n\n  train_prices['section_ret_diff'] = train_prices['ret_1'].to_frame() - train_prices[['Section','ret_1']].groupby(['Date','Section']).transform('mean')\n  train_prices['mkt_ret_diff'] = train_prices['ret_1'].to_frame() - train_prices[['NewMarketSegment','ret_1']].groupby(['Date','NewMarketSegment']).transform('mean')\n  train_prices['beta_ret_diff'] = train_prices['ret_1'].to_frame() - train_prices['ret_1'].groupby(['Date']).transform('mean').to_frame()\n\n  return train_prices","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.685323Z","iopub.execute_input":"2022-08-14T09:25:51.685802Z","iopub.status.idle":"2022-08-14T09:25:51.712578Z","shell.execute_reply.started":"2022-08-14T09:25:51.685748Z","shell.execute_reply":"2022-08-14T09:25:51.711745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load pre-trained model\n\n","metadata":{}},{"cell_type":"code","source":"# model parameters generated by https://www.kaggle.com/code/smeitoma/train-demo\nmodel_file = \"../input/lgbmmodel/lgbm_optuna_best_fit.pkl\"\n\n# load, no need to initialize the loaded_rf\nrf = joblib.load(model_file)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.714170Z","iopub.execute_input":"2022-08-14T09:25:51.714798Z","iopub.status.idle":"2022-08-14T09:25:51.750818Z","shell.execute_reply.started":"2022-08-14T09:25:51.714757Z","shell.execute_reply":"2022-08-14T09:25:51.747401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load training files\n\n","metadata":{}},{"cell_type":"code","source":"base_dir= \"../input/jpx-tokyo-stock-exchange-prediction\"\ntrain_dir = f\"{base_dir}/train_files/\"\n\nraw_prices  = pd.read_csv(f\"{train_dir}/stock_prices.csv\")\nraw_financials = pd.read_csv(f\"{train_dir}/financials.csv\")\nraw_stocks = pd.read_csv(f\"{base_dir}/stock_list.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:51.753986Z","iopub.execute_input":"2022-08-14T09:25:51.754549Z","iopub.status.idle":"2022-08-14T09:25:57.025183Z","shell.execute_reply.started":"2022-08-14T09:25:51.754498Z","shell.execute_reply":"2022-08-14T09:25:57.024276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stocks = treat_stocks(raw_stocks)\nsecurities = raw_prices[\"SecuritiesCode\"].unique().tolist()\nintervals = [1, 5, 10, 21, 63]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:57.029068Z","iopub.execute_input":"2022-08-14T09:25:57.029395Z","iopub.status.idle":"2022-08-14T09:25:57.065401Z","shell.execute_reply.started":"2022-08-14T09:25:57.029361Z","shell.execute_reply":"2022-08-14T09:25:57.064096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setup testing environment \nenv = jpx_tokyo_market_prediction.make_env()\niter_test = env.iter_test()\n# setup lookback for rolling features calculation (65d minimum for ret_65)\nlookback = 70","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:57.066739Z","iopub.execute_input":"2022-08-14T09:25:57.068002Z","iopub.status.idle":"2022-08-14T09:25:57.237358Z","shell.execute_reply.started":"2022-08-14T09:25:57.067920Z","shell.execute_reply":"2022-08-14T09:25:57.236042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = 0\n\n# The API will deliver six dataframes in this specific order:\nfor (prices, options, financials, trades, secondary_prices, sample_prediction) in iter_test:\n        \n    # append the new prices to the historical prices\n    raw_prices = raw_prices.append(prices)\n    raw_financials = raw_financials.append(financials)   \n    \n    # set the current day of processing\n    current_date = datetime.strptime(prices[\"Date\"].iloc[0],'%Y-%m-%d')\n    \n    # calculate from_date\n    p_date = current_date - timedelta(days=lookback)\n\n\n    # get augemented test dataset with rolling features\n    data_test = add_features (raw_prices, raw_financials, stocks, securities, intervals = intervals, from_date = p_date)\n    data_test = treat_na(data_test)\n\n    # only take the current date\n    data_test = data_test.loc(axis=0)[:,current_date][features]\n       \n    #Predict target for test\n    data_test.loc[:, \"predict\"]  = rf.predict(data_test)\n\n    #Rank predictions\n    data_test = data_test.sort_values(\"predict\", ascending=False)\n    data_test.loc[:, \"Rank\"] = np.arange(len(data_test))\n    data_test.reset_index(drop=False, inplace=True)\n    feature_map = data_test.set_index('SecuritiesCode')['Rank'].to_dict()\n    sample_prediction['Rank'] = sample_prediction['SecuritiesCode'].map(feature_map)\n\n    # check Rank\n    assert sample_prediction[\"Rank\"].notna().all()\n    assert sample_prediction[\"Rank\"].min() == 0\n    assert sample_prediction[\"Rank\"].max() == len(sample_prediction[\"Rank\"]) - 1\n\n          \n    # register your predictions\n    env.predict(sample_prediction)\n  \n    counter += 1","metadata":{"execution":{"iopub.status.busy":"2022-08-14T09:25:57.238777Z","iopub.status.idle":"2022-08-14T09:25:57.239897Z","shell.execute_reply.started":"2022-08-14T09:25:57.239562Z","shell.execute_reply":"2022-08-14T09:25:57.239597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-04-30T10:01:46.449100Z","iopub.execute_input":"2022-04-30T10:01:46.449481Z","iopub.status.idle":"2022-04-30T10:01:47.257565Z","shell.execute_reply.started":"2022-04-30T10:01:46.449443Z","shell.execute_reply":"2022-04-30T10:01:47.256775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! tail submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-04-30T10:01:50.301968Z","iopub.execute_input":"2022-04-30T10:01:50.302315Z","iopub.status.idle":"2022-04-30T10:01:51.117643Z","shell.execute_reply.started":"2022-04-30T10:01:50.302272Z","shell.execute_reply":"2022-04-30T10:01:51.116788Z"},"trusted":true},"execution_count":null,"outputs":[]}]}