{"cells":[{"metadata":{},"cell_type":"markdown","source":"**27ara1730: .64 eklemeli eski pred (without minmax correction) devam ediyor**\n**onun uzerine prior test ekliyorum...**\n**min max sorunu duzelirse, .64 olmadan, o sekilde burayi da duzeltirsin...\nrandomwalker konusunu ayri olarak degerlendiriyorum... onu ayri submit edecegim.... bugun.....**\n\n## Now first I will submit with with averaging...\n## Then I will submit with median values averaging (of both functions)\n\n# GERI ALDIM HER SEYI .689 VERSIYONUNA SADECE, TEK DICT KURALINI YANI ESKIYI GUNCELLEMEYI EKLEDIM... .689 VERSIYONU YENI BIR DICT FORMATI KULLANIYORDU... BEN SU AN ESKIYI GUNCELLIYORUM...BURADA... 28 ARALIK SAAT 2030"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport lightgbm as lgb\n\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"#read in all the pickled data you need\nimport pickle\n\nwith open('../input/settim/question_success.pickle', 'rb') as handle:\n    question_success = pickle.load(handle)\n    \nwith open('../input/settim/user_success.pickle', 'rb') as handle:\n    user_success = pickle.load(handle)\n    \nwith open('../input/settim/all_parts.pickle', 'rb') as handle:\n    all_parts = pickle.load(handle)\n    \nwith open('../input/settim/all_tags.pickle', 'rb') as handle:\n    all_tags = pickle.load(handle)\n    \nwith open('../input/settim/user_parts.pickle', 'rb') as handle:\n    user_parts = pickle.load(handle)\n    \nwith open('../input/settim/user_tags.pickle', 'rb') as handle:\n    user_tags = pickle.load(handle)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"names=pd.read_csv('../input/settim/userids.csv',usecols=[1])\nnames=names['0']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dtypes={'part':'int8','question_id':'int16'}\nq=pd.read_csv('../input/riiid-test-answer-prediction/questions.csv',usecols=[0,3,4]) #this is the original questions if need to check...\n\nq.loc[10033,'tags']='-1'   #that is for the one question without a tag...\ntags_list = [list(map(int,x.split())) for x in q.tags.values]   #now this makes a list of tags in q,, to be used for taganalysis\nq['tags'] = tags_list  #now tags are nothing but a list of integers\nq=q.rename(columns={'question_id':'content_id'})\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#model dict loading\n\nwith open('../input/partial-models/models_gbm.pickle', 'rb') as handle:  #USER TAG BASARISINI DA STANDART OLARAK ALALIM...\n    modeller=pickle.load(handle)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def riidpredictor(dataframe):              #you need to get rid of the next line in real time, no need to do it...\n    #dataframe['answered_correctly']=dataframe['answered_correctly'].astype(object)\n    #for i in range(len(dataframe)):\n    for row in dataframe.itertuples():\n        #print(row.content_id)\n#         if row.content_type_id!=0:\n#             #print('at index.....',row.Index, '................','.... id is not 0, it is .......',row.content_type_id,)\n#             continue\n\n        tags=q.loc[row.content_id]['tags']  #careful tags is a list....\n        part=q.loc[row.content_id]['part']\n        user=int(row.user_id)\n        results=[]\n        tagarray=[]\n\n        q_success=question_success['mean'][row.content_id]\n        tagarray.append(q_success)\n    #    print('now it starts for row.................')\n    \n        if row.content_id==10033:\n            dataframe.loc[row.Index,'answered_correctly']=q_success\n            continue\n        \n        if user not in names.values:\n            dataframe.loc[row.Index,'answered_correctly']=q_success\n            continue\n        \n\n        initial_success=user_success['mean'][user]\n        tagarray.append(initial_success)\n        \n        genel_success=q_success + (initial_success - .626)\n#         print('calculated general success and adding to results.......', genel_success)\n        results.append(genel_success)\n\n        if (user,part) in user_parts['mean'].keys():\n            uparts=user_parts['mean'][(user,part)] \n#             print('calculated uparts and adding to results.......', uparts)\n\n            results.append(uparts)\n\n            aparts=all_parts['mean'][part]\n            aparts_touser=q_success + (uparts - aparts)              # it is about adding the success different to the base\n#             print('calculated aparts and adding to results.......', aparts_touser)\n\n            results.append(aparts_touser)\n\n#             else:\n#     #             print('could not find that part in the user info...just adding aparts............',aparts)\n#                 aparts=all_parts['mean'][part]\n#                 results.append(aparts)\n\n        if tags[0]!=-1:\n            for tag in tags:\n                if (user,tag) in user_tags['mean'].keys():  #tektek bakmali, cunku ya herhangi bir tag userda yoksa...\n                    utag=user_tags['mean'][(user,tag)]\n#                     print('found tag.....',tag,'...in user info, so adding utag to results.....',utag)\n\n                    results.append(utag)\n                    tagarray.append(utag)\n\n                    atag=all_tags['mean'][tag]\n                    atag_touser=q_success + (utag - atag)              # it is about adding the success different to the base\n#                     print('also revised the q success with utag atag difference........', atag_touser)\n\n                    results.append(atag_touser)\n                else:\n                    atag=all_tags['mean'][tag]\n                    utag=atag + (initial_success - 0.626)\n#                     print('could not find tag......', tag,'   so adding only general tag to results......',atag)\n                    tagarray.append(utag)\n\n        #results_r=[round(x,2) for x in results]    #you will cast the average of this list to answered_correctly\n        #prediction=sum(results)/len(results)  #that is to add the final bit, the overall success\n        median_pred= (q_success + np.median(results)) / 2\n        median_pred=min(max(median_pred,0.01),0.99)\n        \n        tagarray=np.asarray(tagarray)\n        tagarray=np.reshape(tagarray,[1,tagarray.shape[0]])\n        if row.content_id in modeller:\n            prediction=modeller[row.content_id].predict(tagarray)\n        else:\n            prediction=median_pred\n\n        dataframe.loc[row.Index,'answered_correctly']=prediction  #if you want to use the list, dehighlight\n        #the upper line that changed answered_correctly column data type way above!!!\n        \n    return dataframe\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import riiideducation\n\nenv = riiideducation.make_env()\niter_test = env.iter_test()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prior_test_df = None","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    if (prior_test_df is not None): #prior test comes here with only its important columns.. as filtered in way below solvercells1\n        \n        prior_test_df['answered_correctly'] = eval(test_df['prior_group_answers_correct'].iloc[0])\n        del prior_test_df['prior_group_answers_correct']\n        \n        #the merge with q on content_id, that sets parts and tags to the prior_test_df\n        #this adds part and tags columns to prior_test_df\n        prior_test_df=prior_test_df.reset_index().merge(q, how=\"left\",on='content_id')\n        \n        \n        \n        \n        #now you are going to iterate over the priortest dataframe and makeup the new dictionaries for user success: us\n        #question success: question_success, and ut al up all_parts dictionaries that you will define here...\n        for line in prior_test_df.itertuples():\n            #print(row.content_id)\n            if line.content_type_id!=0: #BU KONTROLE DE GEREK KALMADI KI...CUNKU ZATEN PRIORTESTDF ASAGIDA TEST DF ICINDEN UYGUN OLANLAR ALINARAK BAKILDI...\n                #print('at index.....',row.Index, '................','.... id is not 0, it is .......',row.content_type_id,)\n                continue\n                \n            #for question_success:\n            if line.content_id in question_success['mean']: #ayrica in .keys demene gerek yok bu sekilde biraktiginda zaten keys icine bakiyor\n                question_success['mean'][line.content_id]=(question_success['count'][line.content_id] * question_success['mean'][line.content_id] + line.answered_correctly)/(question_success['count'][line.content_id]+1)\n                question_success['count'][line.content_id]+=1\n            else:\n                question_success['mean'][line.content_id]=line.answered_correctly\n                question_success['count'][line.content_id]=1\n\n                \n            #for user_success:\n            if line.user_id in user_success['mean']:\n                user_success['mean'][line.user_id]=(user_success['count'][line.user_id] * user_success['mean'][line.user_id] + line.answered_correctly)/(user_success['count'][line.user_id]+1)\n                user_success['count'][line.user_id]+=1\n            else:\n                user_success['mean'][line.user_id]=line.answered_correctly\n                user_success['count'][line.user_id]=1\n                \n            \n            #for user_parts:\n            if (line.user_id,line.part) in user_parts['mean']:\n                user_parts['mean'][(line.user_id,line.part)]=(user_parts['count'][(line.user_id,line.part)] * user_parts['mean'][(line.user_id,line.part)] + line.answered_correctly)/(user_parts['count'][(line.user_id,line.part)]+1)\n                user_parts['count'][(line.user_id,line.part)]+=1\n            else:\n                user_parts['mean'][(line.user_id,line.part)]=line.answered_correctly\n                user_parts['count'][(line.user_id,line.part)]=1\n        \n            #for all_parts:\n            if line.part in all_parts['mean']:\n                all_parts['mean'][line.part]=(all_parts['count'][line.part] * all_parts['mean'][line.part] + line.answered_correctly)/(all_parts['count'][line.part]+1)\n                all_parts['count'][(line.part)]+=1\n            else:\n                all_parts['mean'][line.part]=line.answered_correctly\n                all_parts['count'][line.part]=1\n                \n                \n            #tags analysis\n            if line.tags[0]!=-1:\n                for tag in line.tags: \n                    \n                    #for user_tags:\n                    if (line.user_id,tag) in user_tags['mean']:\n                        user_tags['mean'][(line.user_id,tag)]=(user_tags['count'][(line.user_id,tag)] * user_tags['mean'][(line.user_id,tag)] + line.answered_correctly)/(user_tags['count'][(line.user_id,tag)]+1)\n                        user_tags['count'][(line.user_id,tag)]+=1\n                    else:\n                        user_tags['mean'][(line.user_id,tag)]=line.answered_correctly\n                        user_tags['count'][(line.user_id,tag)]=1\n                        \n                    #for all_tags:\n                    if tag in all_tags['mean']:\n                        all_tags['mean'][tag]=(all_tags['count'][tag] * all_tags['mean'][tag] + line.answered_correctly)/(all_tags['count'][tag]+1)\n                        all_tags['count'][tag]+=1\n                    else:\n                        all_tags['mean'][tag]=line.answered_correctly\n                        all_tags['count'][tag]=1\n                        \n            #also we need to update names\n            if line.user_id not in names.values:\n                names.loc[names.index.max()+1] =line.user_id\n        \n    \n    #prior_test_df = test_df.copy()  #updated next line so no need to copy first and then revise,, better do both the same time...\n    prior_test_df = test_df[['row_id','user_id','content_id','content_type_id','prior_group_answers_correct']].copy()\n    \n    test_df = test_df[test_df['content_type_id'] == 0].reset_index(drop=True)\n    results=riidpredictor(test_df)\n    env.predict(results[['row_id', 'answered_correctly']])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}