{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T22:25:21.860547Z","iopub.execute_input":"2022-07-05T22:25:21.860904Z","iopub.status.idle":"2022-07-05T22:25:21.925767Z","shell.execute_reply.started":"2022-07-05T22:25:21.860830Z","shell.execute_reply":"2022-07-05T22:25:21.925021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input/packages/ta/","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:25:23.260232Z","iopub.execute_input":"2022-07-05T22:25:23.260709Z","iopub.status.idle":"2022-07-05T22:25:23.569358Z","shell.execute_reply.started":"2022-07-05T22:25:23.260681Z","shell.execute_reply":"2022-07-05T22:25:23.568651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /tmp/pip/cache/\n!cp ../input/packages/ta/numpy-1.22.4-cp39-cp39-win_amd64.whl /tmp/pip/cache/numpy-1.22.4-cp39-cp39-win_amd64.whl\n!cp ../input/packages/ta/pytz-2022.1-py2.py3-none-any.whl /tmp/pip/cache/pytz-2022.1-py2.py3-none-any.whl\n!cp ../input/packages/ta/pandas-1.4.2-cp39-cp39-win_amd64.whl /tmp/pip/cache/pandas-1.4.2-cp39-cp39-win_amd64.whl\n!cp ../input/packages/ta/six-1.16.0-py2.py3-none-any.whl /tmp/pip/cache/six-1.16.0-py2.py3-none-any.whl\n!cp ../input/packages/ta/python_dateutil-2.8.2-py2.py3-none-any.whl /tmp/pip/cache/python_dateutil-2.8.2-py2.py3-none-any.whl\n!cp ../input/packages/ta/ta-0.10.1.xyz /tmp/pip/cache/ta-0.10.1.tar.gz","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:25:24.276342Z","iopub.execute_input":"2022-07-05T22:25:24.277207Z","iopub.status.idle":"2022-07-05T22:25:26.921666Z","shell.execute_reply.started":"2022-07-05T22:25:24.277181Z","shell.execute_reply":"2022-07-05T22:25:26.920566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --no-index --find-links /tmp/pip/cache/ ta","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:25:26.923175Z","iopub.execute_input":"2022-07-05T22:25:26.923529Z","iopub.status.idle":"2022-07-05T22:25:41.654471Z","shell.execute_reply.started":"2022-07-05T22:25:26.923497Z","shell.execute_reply":"2022-07-05T22:25:41.653586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport ta\nfrom pandarallel import pandarallel\n# Initialization\npandarallel.initialize()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:25:41.655644Z","iopub.execute_input":"2022-07-05T22:25:41.655995Z","iopub.status.idle":"2022-07-05T22:25:46.972588Z","shell.execute_reply.started":"2022-07-05T22:25:41.655963Z","shell.execute_reply":"2022-07-05T22:25:46.971693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#read the train file data\nroot = \"../input/jpx-tokyo-stock-exchange-prediction\"\n# root =  './'\nstock_prices_raw_1 = pd.read_csv(f\"{root}/train_files/stock_prices.csv\")\nfinancials_raw_1 = pd.read_csv(f\"{root}/train_files/financials.csv\")\noptions_raw_1 = pd.read_csv(f\"{root}/train_files/options.csv\")\nsecondary_stock_prices_raw_1 = pd.read_csv(f\"{root}/train_files/secondary_stock_prices.csv\")\ntrades_raw_1 = pd.read_csv(f\"{root}/train_files/trades.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:25:46.974803Z","iopub.execute_input":"2022-07-05T22:25:46.975467Z","iopub.status.idle":"2022-07-05T22:26:17.694565Z","shell.execute_reply.started":"2022-07-05T22:25:46.975437Z","shell.execute_reply":"2022-07-05T22:26:17.693700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#read the supplemental file data\nstock_prices_raw_2 = pd.read_csv(f\"{root}/supplemental_files/stock_prices.csv\")\nfinancials_raw_2 = pd.read_csv(f\"{root}/supplemental_files/financials.csv\")\noptions_raw_2 = pd.read_csv(f\"{root}/supplemental_files/options.csv\")\nsecondary_stock_prices_raw_2 = pd.read_csv(f\"{root}/supplemental_files/secondary_stock_prices.csv\")\ntrades_raw_2 = pd.read_csv(f\"{root}/supplemental_files/trades.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:26:17.695613Z","iopub.execute_input":"2022-07-05T22:26:17.696029Z","iopub.status.idle":"2022-07-05T22:26:22.895604Z","shell.execute_reply.started":"2022-07-05T22:26:17.696006Z","shell.execute_reply":"2022-07-05T22:26:22.893401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#concat the data file\nstock_prices_raw=pd.concat([stock_prices_raw_1,stock_prices_raw_2]).reset_index(drop=True)\nstock_prices_fin=pd.concat([stock_prices_raw_1,stock_prices_raw_2]).reset_index(drop=True)\nfinancials_raw=pd.concat([financials_raw_1,financials_raw_2]).reset_index(drop=True)\noptions_raw=pd.concat([options_raw_1,options_raw_2]).reset_index(drop=True)\nsecondary_stock_prices_raw=pd.concat([secondary_stock_prices_raw_1,secondary_stock_prices_raw_2]).reset_index(drop=True)\ntrades_raw=pd.concat([trades_raw_1,trades_raw_2]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:26:22.898071Z","iopub.execute_input":"2022-07-05T22:26:22.898437Z","iopub.status.idle":"2022-07-05T22:26:26.624880Z","shell.execute_reply.started":"2022-07-05T22:26:22.898407Z","shell.execute_reply":"2022-07-05T22:26:26.624221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculate the adjusted price\ndef prep_prices(price):\n    from decimal import ROUND_HALF_UP, Decimal\n    pcols = [\"Open\",\"High\",\"Low\",\"Close\"]\n    price.ExpectedDividend.fillna(0,inplace=True)\n    def qround(x):\n        return float(Decimal(str(x)).quantize(Decimal('0.1'), rounding=ROUND_HALF_UP))\n    \n    def adjust_prices(df):\n        df = df.sort_values(\"Date\", ascending=False)\n        df.loc[:, \"CumAdjust\"] = df[\"AdjustmentFactor\"].cumprod()\n\n        # generate adjusted prices\n        for p in pcols:     \n            df.loc[:, p] = (df[\"CumAdjust\"] * df[p]).apply(qround)\n        df.loc[:, \"Volume\"] = df[\"Volume\"] / df[\"CumAdjust\"]\n        \n        return df\n\n    # generate Adjusted\n    price = price.sort_values([\"SecuritiesCode\", \"Date\"])\n    price = price.groupby(\"SecuritiesCode\").apply(adjust_prices).reset_index(drop=True)\n    price = price.sort_values(\"RowId\")\n    \n#     price.dropna(subset=[\"Open\",\"High\",\"Low\",\"Close\",'Volume','Target'],inplace=True)\n    return price.drop(['CumAdjust'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:26:26.626122Z","iopub.execute_input":"2022-07-05T22:26:26.626537Z","iopub.status.idle":"2022-07-05T22:26:26.637307Z","shell.execute_reply.started":"2022-07-05T22:26:26.626509Z","shell.execute_reply":"2022-07-05T22:26:26.636300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# financials data cleaning\ndef prep_financials(financials):\n    \"\"\"\n    - Date to datetime type.\n    - Fill all NaN by np.nan.\n    - Transform numbers to numerical type and substitude all '－' to np.nan.\n    - Delete ForecastRevision and duplicated rows.\n    - Will not drop Nan.\n    \"\"\"\n    financials.fillna(np.nan,inplace=True)\n    financials.dropna(subset=['DateCode'],inplace=True)\n    financials = financials[~financials.TypeOfDocument.str.contains('ForecastRevision')]\n    financials = financials[~financials.TypeOfDocument.str.contains('NumericalCorrection')]\n    financials = financials.drop_duplicates(subset=['DateCode'],keep='first')  # delete duplicates\n    financials = financials.apply(pd.to_numeric, errors='ignore')\n    financials = financials.replace('－',np.nan)\n\n    # Clean some number-type columns.\n    lis = ['NetSales','OrdinaryProfit','OperatingProfit','EarningsPerShare','TotalAssets','Profit','NumberOfIssuedAndOutstandingSharesAtTheEndOfFiscalYearIncludingTreasuryStock','Equity']\n    financials[lis] = financials[lis].apply(pd.to_numeric,errors='coerce')\n    \n    financials['Date'] = pd.to_datetime(financials['Date'])\n\n    def adjust_to_year(df):\n    \n        df['TypeMark'] = df.apply(lambda x:0 if x['TypeOfCurrentPeriod']=='FY' else 1,axis=1)\n\n        items_to_adjust = ['NetSales','OperatingProfit','OrdinaryProfit','Profit','EarningsPerShare']\n        \n        def adjust_item(df1):\n            df_to_subtract = df1[items_to_adjust].mul(df1.TypeMark,axis=0)\n            df_to_shift = df_to_subtract.shift()\n            df1[items_to_adjust] = df1[items_to_adjust] - df_to_shift\n            return df1\n        df = df.groupby(\"SecuritiesCode\").apply(adjust_item)\n        return df\n    \n    financials = adjust_to_year(financials)\n    return financials\n\n# financials = prep_financials(financials_raw)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:26:26.638631Z","iopub.execute_input":"2022-07-05T22:26:26.638946Z","iopub.status.idle":"2022-07-05T22:26:26.656495Z","shell.execute_reply.started":"2022-07-05T22:26:26.638923Z","shell.execute_reply":"2022-07-05T22:26:26.655633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_features_for_predict(stock_prices, financials):\n    \"\"\"\n    Args:\n        prices (pd.DataFrame)  : RowId renamed as DateCode\n        financials (pd.DataFrame)  : cleaned\n        ...\n    Returns:\n        feature DataFrame (pd.DataFrame): with Date and SecuritiesCode\n        - Will not drop Nan.\n        - Standardize \n    \"\"\"\n    # financial indicators\n    stock_prices = stock_prices.rename(columns = {'RowId':'DateCode'})\n    financials['netsales_growth_rate'] = financials.groupby('SecuritiesCode')['NetSales'].transform(lambda x:(x-x.shift(1))/x.shift(1))\n    pricesfin = pd.merge(stock_prices[['DateCode','Close']],financials,on='DateCode').sort_values(['DateCode'])\n    features = pricesfin[['DateCode']].copy(deep=True)\n    #PE\n    features['PE'] = pricesfin['Close']/pricesfin['EarningsPerShare']\n    #PEcut\n    features['PEcut'] = pricesfin['Close']/(pricesfin['EarningsPerShare']-pricesfin['OrdinaryProfit']/pricesfin['NumberOfIssuedAndOutstandingSharesAtTheEndOfFiscalYearIncludingTreasuryStock'])\n    #PEG\n    features['PEG'] = features['PE']/pricesfin['netsales_growth_rate']\n    #PS\n    features['PS'] = pricesfin['Close']/(pricesfin['OperatingProfit']/pricesfin['NumberOfIssuedAndOutstandingSharesAtTheEndOfFiscalYearIncludingTreasuryStock'])\n    #PB\n    features['PB'] = pricesfin['Close']/(pricesfin['TotalAssets']/pricesfin['NumberOfIssuedAndOutstandingSharesAtTheEndOfFiscalYearIncludingTreasuryStock'])\n    # ROE, ROA, GPM, NPM, Em\n    features[\"ROE\"] = pricesfin.Profit/pricesfin.Equity\n    features[\"ROA\"] = pricesfin.Profit/pricesfin.TotalAssets\n    # features[\"GPM\"] = pricesfin.OrdinaryProfit/pricesfin.NetSales\n    # features[\"NPM\"] = pricesfin.Profit/pricesfin.NetSales\n    features[\"EM\"] = pricesfin.TotalAssets/pricesfin.Equity\n    # RG, NPG, ROEG, ROAG\n    def calculate_for_single_company(df):\n        df['shift1_NetSales'] = df.NetSales.shift()\n        df['shift1_Profit'] = df.Profit.shift()\n        df['shift1_ROE'] = df.ROE.shift()\n        df['shift1_ROA'] = df.ROA.shift()\n        return df\n    pricesfin = pd.merge(features,pricesfin,on='DateCode',how='left').groupby(\"SecuritiesCode\").apply(calculate_for_single_company)\n\n    features['RG'] = pricesfin.NetSales/pricesfin.shift1_NetSales - 1\n    features['NPG'] = pricesfin.Profit/pricesfin.shift1_Profit - 1\n    features['ROEG'] = pricesfin.ROE/pricesfin.shift1_ROE - 1\n    features['ROAG'] = pricesfin.ROA/pricesfin.shift1_ROA - 1\n    \n    features = pd.merge(stock_prices[['DateCode','Date','SecuritiesCode','Volume']], features, on='DateCode', how='left').sort_values(['DateCode'])\n    features = features.groupby('SecuritiesCode').apply(lambda x: x.ffill())\n    features = features.replace([np.inf,-np.inf],np.nan)\n    return features\n\n# features = get_features_for_predict(stock_prices_raw,financials)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:26:26.659590Z","iopub.execute_input":"2022-07-05T22:26:26.659986Z","iopub.status.idle":"2022-07-05T22:26:26.682202Z","shell.execute_reply.started":"2022-07-05T22:26:26.659954Z","shell.execute_reply":"2022-07-05T22:26:26.681547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ta\nfrom ta import add_all_ta_features\nfrom pandarallel import pandarallel\n# Initialization\npandarallel.initialize()\ndef generate_ta_features(stock_prices,istrain):\n    \"\"\"\n    INPUT: original dataset\n    OUTPUT: features with Date and SecuritiesCode\n    \"\"\"\n    # if 'Target' in stock_prices.columns:\n    #     stock_prices = stock_prices.drop('Target',axis=1)\n    \n    # Calculate features\n    # features = stock_prices.groupby('SecuritiesCode').apply(lambda x: add_all_ta_features(x,open=\"Open\",high='High',low='Low',close='Close',volume='Volume'))\n    stock_prices = stock_prices.drop(['RowId', 'AdjustmentFactor', 'ExpectedDividend', 'SupervisionFlag'],axis=1)\n    \n    \n    stock_prices['Close'] = stock_prices['Close'].fillna(method='ffill')\n\n    features = stock_prices[['Date', 'SecuritiesCode','Target', 'Close']].copy()\n\n    if not istrain:\n        features = features.drop('Target',axis=1)\n\n    features['volume_log'] = np.log(stock_prices['Volume'])\n    # def volume_adi(df):\n    #     return ta.volume.acc_dist_index(high=df.High, low=df.Low, close=df.Close, volume=df.Volume, fillna=False)\n    # features['volume_adi'] = stock_prices.groupby('SecuritiesCode').apply(volume_adi).droplevel(0)\n    def volume_adi_group(df): \n        def volume_adi(High, Low, Close, Volume):\n            import ta\n            return ta.volume.acc_dist_index(high=High, low=Low, close=Close, volume=Volume, fillna=False).iloc[-1]\n        return df.Close.rolling(5).apply(lambda x: volume_adi(df.loc[x.index,'High'],df.loc[x.index,'Low'],df.loc[x.index,'Close'],df.loc[x.index,'Volume']))\n    features['volume_adi'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volume_adi_group).droplevel(0)\n    print('volume_adi')\n    \n#     def volume_obv(df):\n#         '-----------------累计要从训练日期开始算-----------------'\n#         return ta.volume.on_balance_volume(close=df.Close, volume=df.Volume, fillna=False)\n#     features['volume_obv'] = stock_prices.groupby('SecuritiesCode').apply(volume_obv).droplevel(0)\n#     def volume_obv_group(df): \n#         def volume_obv(Close,Volume):\n#             import ta\n#             return ta.volume.on_balance_volume(close=Close, volume=Volume, fillna=False).iloc[-1]\n#         return df.Close.rolling(5).apply(lambda x: volume_obv(df.loc[x.index,'Close'],df.loc[x.index,'Volume']))\n#     features['volume_obv'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volume_obv_group).droplevel(0)\n#     print('volume_obv')\n    \n    def volume_cmf(df,window):\n        import ta\n        return ta.volume.chaikin_money_flow(high=df.High, low=df.Low, close=df.Close, volume=df.Volume, window=window, fillna=False)\n    features['volume_cmf'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volume_cmf, 2).droplevel(0)\n#     def volume_fi(df,window):\n#         return ta.volume.force_index(close=df.Close, volume=df.Volume, window=window, fillna=False)\n#     features['volume_fi'] = stock_prices.groupby('SecuritiesCode').apply(volume_fi, 2).droplevel(0)\n    def volume_em(df,window):\n        import ta\n        return ta.volume.ease_of_movement(high=df.High, low=df.Low, volume=df.Volume, window=window, fillna=False)\n    features['volume_em'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volume_em, 2).droplevel(0)\n    def volume_sem(df,window):\n        import ta\n        return ta.volume.sma_ease_of_movement(high=df.High, low=df.Low, volume=df.Volume, window=window, fillna=False)\n    features['volume_sem'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volume_sem, 4).droplevel(0)\n    def volume_vpt(df):\n        import ta\n        return ta.volume.volume_price_trend(close=df.Close, volume=df.Volume, fillna=False)\n    features['volume_vpt'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volume_vpt).droplevel(0)\n    def volume_vwap(df,window):\n        import ta\n        return ta.volume.volume_weighted_average_price(high=df.High, low=df.Low, close=df.Close, volume=df.Volume, window=window, fillna=False)\n    features['volume_vwap'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volume_vwap,5).droplevel(0)\n#     def volume_mfi(df,window):\n#         return ta.volume.money_flow_index(high=df.High,low=df.Low,close=df.Close,volume=df.Volume,window=window,fillna=False)\n#     features['volume_mfi'] = stock_prices.groupby('SecuritiesCode').apply(volume_mfi,2).droplevel(0)\n    # def volume_nvi(df):\n    #     '-----------------累计要从训练日期开始算-----------------'\n    #     return ta.volume.negative_volume_index(close=df.Close, volume=df.Volume, fillna=False)\n    # features['volume_nvi'] = stock_prices.groupby('SecuritiesCode').apply(volume_nvi).droplevel(0)\n#     def  volume_nvi_group(df):\n#         def volume_nvi(close,volume,window=20):\n#             price_change = close.pct_change()\n#             vol_decrease = volume.shift(1)>volume\n#             nvi = 1000\n#             for i in range(1,len(close)):\n#                 if vol_decrease.iloc[i]:\n#                     nvi = nvi * (1.0 + price_change.iloc[i])\n#                 else:\n#                     continue\n#             return nvi\n#         return df.Close.rolling(20).apply(volume_nvi,kwargs = {'volume':df.Volume})\n#     features['volume_nvi'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volume_nvi_group).droplevel(0)\n#     print('volume_nvi')\n\n\n#     def volatility_bbw(df,window,window_dev):\n#         return ta.volatility.bollinger_wband(close=df.Close,window=window,window_dev=window_dev,fillna=False)\n#     features['volatility_bbw'] = stock_prices.groupby('SecuritiesCode').apply(volatility_bbw,5,1.8).droplevel(0)\n#     def volatility_bbp(df,window,window_dev):\n#         return ta.volatility.bollinger_pband(close=df.Close,window=window,window_dev=window_dev,fillna=False)\n#     features['volatility_bbp'] = stock_prices.groupby('SecuritiesCode').apply(volatility_bbp,5,1.8).droplevel(0)\n#     def volatility_bbhi(df,window,window_dev):\n#         return ta.volatility.bollinger_hband_indicator(close=df.Close,window=window,window_dev=window_dev,fillna=False)\n#     features['volatility_bbhi'] = stock_prices.groupby('SecuritiesCode').apply(volatility_bbhi,5,1.8).droplevel(0)\n    def volatility_bbli(df,window,window_dev):\n        import ta\n        return ta.volatility.bollinger_lband_indicator(close=df.Close,window=window,window_dev=window_dev,fillna=False)\n    features['volatility_bbli'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volatility_bbli,5,1.8).droplevel(0)\n#     def volatility_kcw(df,window):\n#         return ta.volatility.keltner_channel_wband(close=df.Close,high=df.High,low=df.Low,window=window,fillna=False)\n#     features['volatility_kcw'] = stock_prices.groupby('SecuritiesCode').apply(volatility_kcw,5).droplevel(0)\n#     def volatility_kcp(df,window):\n#         return ta.volatility.keltner_channel_pband(close=df.Close,high=df.High,low=df.Low,window=window,fillna=False)\n#     features['volatility_kcp'] = stock_prices.groupby('SecuritiesCode').apply(volatility_kcp,5).droplevel(0)\n#     def volatility_kchi(df,window):\n#         return ta.volatility.keltner_channel_hband_indicator(close=df.Close,high=df.High,low=df.Low,window=window,fillna=False)\n#     features['volatility_kchi'] = stock_prices.groupby('SecuritiesCode').apply(volatility_kchi,5).droplevel(0)\n    def volatility_kcli(df,window):\n        import ta\n        return ta.volatility.keltner_channel_lband_indicator(close=df.Close,high=df.High,low=df.Low,window=window,fillna=False)\n    features['volatility_kcli'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volatility_kcli,5).droplevel(0)\n#     def volatility_dcw(df,window):\n#         return ta.volatility.donchian_channel_wband(high=df.High,low=df.Low,close=df.Close,window=window,fillna=False)\n#     features['volatility_dcw'] = stock_prices.groupby('SecuritiesCode').apply(volatility_dcw,2).droplevel(0)\n#     def volatility_dcp(df,window):\n#         return ta.volatility.donchian_channel_pband(high=df.High,low=df.Low,close=df.Close,window=window,fillna=False)\n#     features['volatility_dcp'] = stock_prices.groupby('SecuritiesCode').apply(volatility_dcp,5).droplevel(0)\n    def volatility_atr(df,window):\n        import ta\n        return ta.volatility.average_true_range(close=df.Close,high=df.High,low=df.Low,window=window,fillna=False)\n    features['volatility_atr'] = stock_prices.groupby('SecuritiesCode').parallel_apply(volatility_atr,2).droplevel(0)\n#     def volatility_ui(df,window):\n#         return ta.volatility.ulcer_index(close=df.Close,window=window,fillna=False)\n#     features['volatility_ui'] = stock_prices.groupby('SecuritiesCode').apply(volatility_ui,5).droplevel(0)\n    \n    # def trend_macd_signal(df, window_slow, window_fast, window_sign):\n    #     '数据长度不一样，计算结果不一样'\n    #     return ta.trend.macd_signal(close=df.Close,window_slow=window_slow, window_fast=window_fast, window_sign=window_sign,fillna=False)\n    # features['trend_macd_signal'] = stock_prices.groupby('SecuritiesCode').apply(trend_macd_signal,20,8,4).droplevel(0)\n    # def trend_macd_diff(df,window_slow, window_fast, window_sign):\n    #     '数据长度不一样，计算结果不一样'\n    #     return ta.trend.macd_diff(close=df.Close,window_slow=window_slow, window_fast=window_fast, window_sign=window_sign,fillna=False)\n    # features['trend_macd_diff'] = stock_prices.groupby('SecuritiesCode').apply(trend_macd_diff,20,8,4).droplevel(0)\n    def macd_signal_group(df):\n        def ema(close, window):\n            return close.ewm(span=window,min_periods=window,adjust=False).mean()\n        def macd_signal(close, window_fast=20,window_slow=8,window_sign=4):\n            emafast = ema(close, window_fast)\n            emaslow = ema(close,window_slow)\n            macd = emafast-emaslow\n            signal = ema(macd, window_sign)\n            return signal.iloc[-1]\n        def macd_diff(close, window_fast=20,window_slow=8,window_sign=4):\n            emafast = ema(close, window_fast)\n            emaslow = ema(close,window_slow)\n            macd = emafast-emaslow\n            signal = ema(macd, window_sign)\n            return macd.iloc[-1] - signal.iloc[-1]\n        return df.Close.rolling(23).apply(macd_signal)\n    def macd_diff_group(df):\n        def ema(close, window):\n            return close.ewm(span=window,min_periods=window,adjust=False).mean()\n        def macd_diff(close, window_fast=20,window_slow=8,window_sign=4):\n            emafast = ema(close, window_fast)\n            emaslow = ema(close,window_slow)\n            macd = emafast-emaslow\n            signal = ema(macd, window_sign)\n            return macd.iloc[-1] - signal.iloc[-1]\n        return df.Close.rolling(23).apply(macd_diff)\n#     features['trend_macd_signal'] = stock_prices.groupby('SecuritiesCode').parallel_apply(macd_signal_group).droplevel(0) # window_fast + window_sign - 1 = 23\n#     print('trend_macd_signal')\n    features['trend_macd_diff'] = stock_prices.groupby('SecuritiesCode').parallel_apply(macd_diff_group).droplevel(0) # window_fast + window_sign - 1 = 23\n    print('trend_macd_diff')\n\n#     def trend_vortex_ind_pos(df,window):\n#         return ta.trend.vortex_indicator_pos(high=df.High,low=df.Low,close=df.Close,window=window,fillna=False)\n#     features['trend_vortex_ind_pos'] = stock_prices.groupby('SecuritiesCode').apply(trend_vortex_ind_pos,2).droplevel(0)\n#     def trend_vortex_ind_neg(df,window):\n#         return ta.trend.vortex_indicator_neg(high=df.High,low=df.Low,close=df.Close,window=window,fillna=False)\n#     features['trend_vortex_ind_neg'] = stock_prices.groupby('SecuritiesCode').apply(trend_vortex_ind_neg,2).droplevel(0)\n\n    # def trend_mass_index(df,window_fast=10,window_slow=8):\n    #     return ta.trend.mass_index(high=df.High,low=df.Low,window_fast=window_fast,window_slow=window_slow,fillna=False)\n    # features['trend_mass_index'] = stock_prices.groupby('SecuritiesCode').apply(trend_mass_index,10,8).droplevel(0)\n#     def trend_mass_index_group(df):\n#         def trend_mass_index(High,Low,window_fast=10,window_slow=8):\n#             import ta\n#             return ta.trend.mass_index(high=High,low=Low,window_fast=window_fast,window_slow=window_slow,fillna=False).iloc[-1]\n#         return df.High.rolling(26).apply(lambda x: trend_mass_index(df.loc[x.index,'High'],df.loc[x.index,'Low']))\n#     features['trend_mass_index'] = stock_prices.groupby('SecuritiesCode').parallel_apply(trend_mass_index_group).droplevel(0)\n#     print('trend_mass_index')\n    def trend_dpo(df,window):\n        import ta\n        return ta.trend.dpo(close=df.Close,window=window,fillna=False)\n    features['trend_dpo'] = stock_prices.groupby('SecuritiesCode').parallel_apply(trend_dpo,15).droplevel(0)\n#     def trend_kst_sig(df,roc1,roc2,roc3,roc4,window1,window2,window3,window4,nsig):\n#         return ta.trend.kst_sig(close=df.Close,roc1=roc1,roc2=roc2,roc3=roc3,roc4=roc4,window1=window1,window2=window2,window3=window3,window4=window4,nsig=nsig,fillna=False)\n#     features['trend_kst_sig'] = stock_prices.groupby('SecuritiesCode').apply(trend_kst_sig,3,5,5,20,3,3,5,5,3).droplevel(0)\n#     def trend_stc(df,window_slow,window_fast,cycle,smooth1,smooth2):\n#         return ta.trend.stc(close=df.Close,window_fast=window_fast,window_slow=window_slow,cycle=cycle,smooth1=smooth1,smooth2=smooth2,fillna=False)\n#     features['trend_stc'] = stock_prices.groupby('SecuritiesCode').apply(trend_stc,10,20,5,2,2).droplevel(0)\n\n    # def trend_adx(df,window):\n    #     return ta.trend.adx(high=df.High,low=df.Low,close=df.Close,window=window,fillna=False)\n    # features['trend_adx'] = stock_prices.groupby('SecuritiesCode').apply(trend_adx,3).droplevel(0)\n#     def trend_adx_group(df):\n#         def trend_adx(High,Low,Close,window=3):\n#             import ta\n#             return ta.trend.adx(high=High,low=Low,close=Close,window=window,fillna=False).iloc[-1]\n#         return df.Close.rolling(6).apply(lambda x: trend_adx(df.loc[x.index,'High'],df.loc[x.index,'Low'],df.loc[x.index,'Close']))\n        \n#     features['trend_adx'] = stock_prices.groupby('SecuritiesCode').parallel_apply(trend_adx_group).droplevel(0)\n#     print('trend_adx')\n\n\n#     def trend_adx_pos(df,window):\n#         return ta.trend.adx_pos(high=df.High,low=df.Low,close=df.Close,window=window,fillna=False)\n#     features['trend_adx_pos'] = stock_prices.groupby('SecuritiesCode').apply(trend_adx_pos,3).droplevel(0)\n#     def trend_adx_neg(df,window):\n#         return ta.trend.adx_neg(high=df.High,low=df.Low,close=df.Close,window=window,fillna=False)\n#     features['trend_adx_neg'] = stock_prices.groupby('SecuritiesCode').apply(trend_adx_neg,3).droplevel(0)\n#     def trend_cci(df,window,constant):\n#         return ta.trend.cci(high=df.High, low=df.Low, close=df.Close, window=window, constant=constant, fillna=False)\n#     features['trend_cci'] = stock_prices.groupby('SecuritiesCode').apply(trend_cci,2,0.015).droplevel(0)\n#     def trend_aroon_up(df,window):\n#         return ta.trend.aroon_up(close=df.Close,window=window,fillna=False)\n#     features['trend_aroon_up'] = stock_prices.groupby('SecuritiesCode').apply(trend_aroon_up,2).droplevel(0)\n#     def trend_aroon_down(df,window):\n#         return ta.trend.aroon_down(close=df.Close,window=window,fillna=False)\n#     features['trend_aroon_down'] = stock_prices.groupby('SecuritiesCode').apply(trend_aroon_down,2).droplevel(0)\n    \n#     def momentum_rsi(df,window):\n#         return ta.momentum.rsi(close=df.Close,window=window,fillna=False)\n#     features['momentum_rsi'] = stock_prices.groupby('SecuritiesCode').apply(momentum_rsi,2).droplevel(0)\n\n#     # def momentum_stochrsi(df,window,smooth1,smooth2):\n#     #     return ta.momentum.stochrsi(close=df.Close,window=window,smooth1=smooth1,smooth2=smooth2,fillna=False)\n#     # features['momentum_stochrsi'] = stock_prices.groupby('SecuritiesCode').apply(momentum_stochrsi,20,3,3).droplevel(0)\n#     def momentum_stochrsi_group(df):\n#         def momentum_stochrsi(Close,window=20,smooth1=3,smooth2=3):\n#             import ta\n#             return ta.momentum.stochrsi(close=Close,window=window,smooth1=smooth1,smooth2=smooth2,fillna=False).iloc[-1]\n#         return df.Close.rolling(40).apply(momentum_stochrsi)\n#     features['momentum_stochrsi'] = stock_prices.groupby('SecuritiesCode').parallel_apply(momentum_stochrsi_group).droplevel(0)\n#     print('momentum_stochrsi')\n\n#     def momentum_roc(df,window):\n#         return ta.momentum.roc(close=df.Close, window=window, fillna=False)\n#     features['momentum_tsi'] = stock_prices.groupby('SecuritiesCode').apply(momentum_roc,8).droplevel(0)\n\n#     # def momentum_pvo(df,window_slow=20,window_fast=15,window_sign=3):\n#     #     return ta.momentum.pvo(volume=df.Volume,window_fast=window_fast,window_slow=window_slow,window_sign=window_sign,fillna=False)\n#     # features['momentum_pvo'] = stock_prices.groupby('SecuritiesCode').apply(momentum_pvo,20,15,3).droplevel(0)\n#     def momentum_pvo_group(df):\n#         def momentum_pvo(Volume,window_slow=20,window_fast=15,window_sign=3):\n#             import ta\n#             return ta.momentum.pvo(volume=Volume,window_fast=window_fast,window_slow=window_slow,window_sign=window_sign,fillna=False).iloc[-1]\n#         return df.Volume.rolling(20).apply(momentum_pvo)\n#     features['momentum_pvo'] = stock_prices.groupby('SecuritiesCode').parallel_apply(momentum_pvo_group).droplevel(0)\n#     print('momentum_pvo')\n\n#     def momentum_pvo_hist(df,window_slow,window_fast,window_sign):\n#         return ta.momentum.pvo_hist(volume=df.Volume,window_fast=window_fast,window_slow=window_slow,window_sign=window_sign,fillna=False)\n#     features['momentum_pvo_hist'] = stock_prices.groupby('SecuritiesCode').apply(momentum_pvo_hist,20,15,3).droplevel(0)\n#     def others_dr(df):\n#         return ta.others.daily_return(close=df.Close,fillna=False)\n#     features['momentum_pvo_hist'] = stock_prices.groupby('SecuritiesCode').apply(others_dr).droplevel(0)\n#     def others_dlr(df):\n#         return ta.others.daily_log_return(close=df.Close,fillna=False)\n#     features['momentum_pvo_hist'] = stock_prices.groupby('SecuritiesCode').apply(others_dlr).droplevel(0)\n\n    # get all the column names\n    # col = features.columns\n    # col = col.drop(['Date','SecuritiesCode'])\n\n    # def minmaxscale(df):\n    #     return (df-df.min())/(df.max()-df.min())\n    # features[col] = features.groupby('SecuritiesCode')[col].apply(minmaxscale)\n\n\n    # filling data for nan and inf\n    if istrain:\n        features = features[features.Date > '2017-02-10']\n        # features = features.dropna(thresh=40,axis=0)\n        # features = pd.concat([features[features.Date<'2021-12-06'].dropna(how='any',axis=0),features[features.Date>='2021-12-06']]) # about 2% is nan\n    \n    # features = features.fillna(0)\n    features = features.replace([np.inf, -np.inf], 0)\n\n    return features","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:26:26.683382Z","iopub.execute_input":"2022-07-05T22:26:26.684078Z","iopub.status.idle":"2022-07-05T22:26:26.746299Z","shell.execute_reply.started":"2022-07-05T22:26:26.684045Z","shell.execute_reply":"2022-07-05T22:26:26.745083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stock_prices = prep_prices(stock_prices_raw)\nfinancials = prep_financials(financials_raw)\n\nfeatures_ta_sup = generate_ta_features(stock_prices,True)\nfeatures_fin = get_features_for_predict(stock_prices,financials)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:26:52.210257Z","iopub.execute_input":"2022-07-05T22:26:52.210609Z","iopub.status.idle":"2022-07-05T22:28:21.760485Z","shell.execute_reply.started":"2022-07-05T22:26:52.210585Z","shell.execute_reply":"2022-07-05T22:28:21.759587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stock_prices","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:28:21.761644Z","iopub.execute_input":"2022-07-05T22:28:21.761900Z","iopub.status.idle":"2022-07-05T22:28:21.797100Z","shell.execute_reply.started":"2022-07-05T22:28:21.761878Z","shell.execute_reply":"2022-07-05T22:28:21.796221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_ta = features_ta_sup","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:38.769448Z","iopub.execute_input":"2022-07-05T20:15:38.769971Z","iopub.status.idle":"2022-07-05T20:15:38.775609Z","shell.execute_reply.started":"2022-07-05T20:15:38.769916Z","shell.execute_reply":"2022-07-05T20:15:38.774517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# features_ta = pd.read_csv('../input/featureta/features_ta.csv',index_col=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:28:21.798481Z","iopub.execute_input":"2022-07-05T22:28:21.799666Z","iopub.status.idle":"2022-07-05T22:29:01.568040Z","shell.execute_reply.started":"2022-07-05T22:28:21.799622Z","shell.execute_reply":"2022-07-05T22:29:01.567114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = pd.merge(features_ta,features_fin,on=['Date','SecuritiesCode']).reset_index(drop=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:29:01.569760Z","iopub.execute_input":"2022-07-05T22:29:01.570003Z","iopub.status.idle":"2022-07-05T22:29:04.217922Z","shell.execute_reply.started":"2022-07-05T22:29:01.569982Z","shell.execute_reply":"2022-07-05T22:29:04.217070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features =features.drop(['DateCode'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:29:04.219199Z","iopub.execute_input":"2022-07-05T22:29:04.219609Z","iopub.status.idle":"2022-07-05T22:29:04.460225Z","shell.execute_reply.started":"2022-07-05T22:29:04.219581Z","shell.execute_reply":"2022-07-05T22:29:04.459512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:29:04.461664Z","iopub.execute_input":"2022-07-05T22:29:04.463039Z","iopub.status.idle":"2022-07-05T22:29:04.487151Z","shell.execute_reply.started":"2022-07-05T22:29:04.462999Z","shell.execute_reply":"2022-07-05T22:29:04.486210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:29:04.488457Z","iopub.execute_input":"2022-07-05T22:29:04.488712Z","iopub.status.idle":"2022-07-05T22:29:04.497937Z","shell.execute_reply.started":"2022-07-05T22:29:04.488689Z","shell.execute_reply":"2022-07-05T22:29:04.496968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout\nfrom sklearn.linear_model import LinearRegression\nfrom lightgbm import LGBMRegressor\nimport lightgbm","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:29:04.499541Z","iopub.execute_input":"2022-07-05T22:29:04.499829Z","iopub.status.idle":"2022-07-05T22:29:07.375976Z","shell.execute_reply.started":"2022-07-05T22:29:04.499790Z","shell.execute_reply":"2022-07-05T22:29:07.374942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LGBMRegressor(num_leaves=4,learning_rate=0.04,n_estimators=50)\nmodel_linear = LinearRegression()\n\nX_train=features.loc[:,['Volume', 'PE', 'PEcut', 'PEG', 'PS', 'PB',\n            'RG','Close', 'volume_log',\n            'volume_adi','volume_cmf','volume_em','volume_sem','volume_vpt',\n            'volume_vwap','volatility_bbli', 'volatility_kcli','volatility_atr',  'trend_macd_diff',  'trend_dpo'\n        ]]\ny_train = features['Target']\n\nfeature_copy=features.copy(deep=True)\nfeature_copy=feature_copy.loc[:,['Date','SecuritiesCode','PE', 'PEcut', 'PEG', 'PS', 'ROE', 'ROA', 'Close','Target']].dropna()\nX_train_linear = feature_copy.loc[:,['PE', 'PEcut', 'PEG', 'PS', 'ROE', 'ROA', 'Close']]\ny_train_linear = feature_copy['Target']\n\nmodel.fit(X_train,y_train)\nmodel_linear.fit(X_train_linear,y_train_linear)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:29:07.377038Z","iopub.execute_input":"2022-07-05T22:29:07.377324Z","iopub.status.idle":"2022-07-05T22:29:15.576977Z","shell.execute_reply.started":"2022-07-05T22:29:07.377298Z","shell.execute_reply":"2022-07-05T22:29:15.576160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Time API","metadata":{}},{"cell_type":"code","source":"import jpx_tokyo_market_prediction\nenv = jpx_tokyo_market_prediction.make_env()   # initialize the environment\niter_test = env.iter_test()    # an iterator which loops over the test files\ncounter = 0\n# fetch data day by day\nfor (prices, options, financials, trades, secondary_prices, sample_prediction) in iter_test:\n    current_date = prices[\"Date\"].iloc[0]\n    sample_prediction_date = sample_prediction[\"Date\"].iloc[0]\n    print(f\"current_date: {current_date}, sample_prediction_date: {sample_prediction_date}\")\n    # filter data to reduce culculation cost\n    threshold = (pd.Timestamp(current_date) - pd.offsets.BDay(50)).strftime(\"%Y-%m-%d\")\n    threshold_fin = (pd.Timestamp(current_date) - pd.offsets.BDay(250)).strftime(\"%Y-%m-%d\")\n    print(f\"threshold: {threshold}\")\n\n    stock_prices_raw = stock_prices_raw.loc[(stock_prices_raw[\"Date\"] < current_date) & (stock_prices_raw[\"Date\"] >= threshold)]\n    stock_prices_fin = stock_prices_fin.loc[(stock_prices_fin[\"Date\"] < current_date) & (stock_prices_fin[\"Date\"] >= threshold_fin)]\n    financials_raw = financials_raw.loc[(financials_raw[\"Date\"] < current_date) & (financials_raw[\"Date\"] >= threshold_fin)]\n    options_raw = options_raw.loc[(options_raw[\"Date\"] < current_date) & (options_raw[\"Date\"] >= threshold)]\n    secondary_stock_prices_raw = secondary_stock_prices_raw.loc[(secondary_stock_prices_raw[\"Date\"] < current_date) & (secondary_stock_prices_raw[\"Date\"] >= threshold)]\n    trades_raw = trades_raw.loc[(trades_raw[\"Date\"] < current_date) & (trades_raw[\"Date\"] >= threshold)]\n\n\n    # to generate AdjustedClose, increment price data\n    stock_prices_raw = pd.concat([stock_prices_raw, prices]).reset_index(drop=True)\n    stock_prices_fin = pd.concat([stock_prices_fin, prices]).reset_index(drop=True)\n    financials_raw = pd.concat([financials_raw, financials]).reset_index(drop=True)\n#     options = pd.concat([options_raw, options])\n#     secondary_stock_prices = pd.concat([secondary_stock_prices_raw,secondary_prices])\n#     trades = pd.concat([trades_raw,trades])\n    \n    # generate AdjustedClose and fin data\n    stock_prices = prep_prices(stock_prices_raw)\n    stock_prices_f = prep_prices(stock_prices_fin)\n    financials = prep_financials(financials_raw)\n\n    # get target SecuritiesCodes\n    codes = sorted(prices[\"SecuritiesCode\"].unique())\n\n    # generate feature\n    features_ta = generate_ta_features(stock_prices,False)\n    features_fin = get_features_for_predict(stock_prices_f,financials)\n    feature = pd.merge(features_ta,features_fin,on=['Date','SecuritiesCode'],how='left').reset_index(drop=True)\n    # filter feature for this iteration\n    feature = feature.loc[feature.Date == current_date]\n    feature = feature.drop('DateCode',axis=1)\n    X_test=feature.loc[:,['Volume', 'PE', 'PEcut', 'PEG', 'PS', 'PB',\n            'RG','Close', 'volume_log',\n            'volume_adi','volume_cmf','volume_em','volume_sem','volume_vpt',\n            'volume_vwap','volatility_bbli', 'volatility_kcli','volatility_atr',  'trend_macd_diff',  'trend_dpo'\n        ]]\n    # prediction lgbm\n    feature.loc[:, \"predict_lgbm\"] = model.predict(X_test)\n    df_lgbm=feature.loc[:,['Date','SecuritiesCode','predict_lgbm']]\n    predict_df=pd.merge(sample_prediction,df_lgbm,how='left',on=['Date','SecuritiesCode'])\n    \n    feature_copy=feature.copy(deep=True)\n    feature_copy=feature_copy.loc[:,['Date','SecuritiesCode','PE', 'PEcut', 'PEG', 'PS', 'ROE', 'ROA', 'Close']].dropna()\n    X_test_linear=feature_copy.loc[:,['PE', 'PEcut', 'PEG', 'PS', 'ROE', 'ROA', 'Close']]\n    #prediction linear\n    feature_copy.loc[:, \"predict_linear\"] = model_linear.predict(X_test_linear)\n    df_linear=feature_copy.loc[:,['Date','SecuritiesCode','predict_linear']]\n    \n    predict_df=pd.merge(predict_df,df_linear,how='left',on=['Date','SecuritiesCode'])\n    predict_df['predict_linear'].fillna(np.nanmedian(predict_df['predict_linear']),inplace=True)\n    predict_df['predict_lgbm'].fillna(np.nanmedian(predict_df['predict_lgbm']),inplace=True)\n    predict_df[\"predict\"]=np.nansum([11*predict_df.loc[:, \"predict_linear\"],predict_df.loc[:, \"predict_lgbm\"]],axis=0)/12\n\n    predict_df['Rank'] = predict_df['predict'].rank(ascending=False,method='first').map(lambda x:int(x)-1)\n    # set rank by predict\n    predict_df.sort_values(by=['Date','SecuritiesCode'],inplace=True)\n    sample_prediction=predict_df.loc[:,['Date','SecuritiesCode','Rank']]\n    \n    # check Rank\n    assert sample_prediction[\"Rank\"].notna().all()\n    assert sample_prediction[\"Rank\"].min() == 0\n    assert sample_prediction[\"Rank\"].max() == len(sample_prediction[\"Rank\"]) - 1\n    # register your predictions\n    env.predict(sample_prediction)\n    counter += 1\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:30:00.686595Z","iopub.execute_input":"2022-07-05T22:30:00.687022Z","iopub.status.idle":"2022-07-05T22:35:44.146614Z","shell.execute_reply.started":"2022-07-05T22:30:00.686992Z","shell.execute_reply":"2022-07-05T22:35:44.145476Z"},"trusted":true},"execution_count":null,"outputs":[]}]}