{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\n\nquestions_types = {\n    'question_id': 'int16',\n    'bundle_id': 'int16',\n    'correct_answer': 'int8',\n    'part': 'int8',\n    'tags': 'string'\n}\nquestions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv', dtype = questions_types) # wszystkie dane ładowane do pamięci\nquestions.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_types = {\n    'row_id': 'int64',\n    'timestamp': 'int64',\n    'user_id': 'int32',\n    'content_id': 'int16',\n    'content_type_id': 'int8',\n    'task_container_id': 'int16',\n    'user_answer': 'int8',\n    'answered_correctly': 'int8',\n    'prior_question_elapsed_time': 'float32', \n    'prior_question_had_explanation': 'boolean'\n}        \n\n\ntrain = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv', dtype = train_types) # wszystkie dane ładowane do pamięci\n# train = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv', dtype = train_types, nrows=1000) # pierwsze 1000 wierszy","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"print('W zbiorze istnieje ' + str(questions.shape[0]) + ' pytań')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Przykładowe tagi:')\nprint(questions.loc[0].tags)\nprint(questions.loc[100].tags)\nprint(questions.loc[220].tags)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_no_tags = questions[questions.tags.isna()]\nprint('W zbiorze istnieje ' + str(questions_no_tags.shape[0]) + ' pytanie bez tagów:')\nprint(questions_no_tags);\n# Ma to duże, znaczenie, zależy w przy przetwarzaniu danych sprawdzić czy tagi nie są NA\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_with_tags = questions[questions.tags.isna() == False]\n\ntags = [x.split() for x in questions_with_tags.tags.values]\ntags_set = set([item for elem in tags for item in elem])\nprint('W zbiorze istnieje ' + str(len(tags_set)) + ' różnych tagów')\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Zmiana typu tagów z str na listę:\nquestions.tags = [tags.split() if type(tags) is str else tags for tags in questions.tags.values]\nquestions.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"no_lectures = train[train.answered_correctly != -1] # wykluczenie wykładów\nx = no_lectures.groupby(['content_id']).size()\ncorrect = no_lectures.groupby(['content_id']).agg({'answered_correctly': ['mean']})\ncorrect = pd.DataFrame(correct)\n\n\ndf = no_lectures.groupby(['content_id']).agg({'answered_correctly': ['mean', 'count', 'sum']})\ndf = df['answered_correctly']\nquestions = questions.merge(df, left_on = 'question_id', right_on = 'content_id', how = \"left\")\nquestions = questions.fillna({'mean': '0.0', 'count': '0', 'sum': '0'})\n#questions['sum'] = questions['sum'].astype('int8')\n#questions['count'] = questions['count'].astype('int8')\nquestions = questions.rename(columns={'mean': 'Srednia poprawnosc', 'count': 'Ile razy zadane', 'sum': 'Ilosc poprwnych'})\nquestions.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tags_list = list(tags_set)\ntags_df = pd.DataFrame()\nquestions_with_tags = questions[questions.tags.isna() == False]\nfor tag in tags_list:\n    df = questions_with_tags[questions_with_tags.tags.apply(lambda l: tag in l)]\n    tmp_df = df.agg({'Ile razy zadane': ['sum'], 'Ilosc poprwnych': ['sum']})\n    tmp_df['tag'] = tag\n    tmp_df['Ilosc pytań z tym tagiem'] = len(df)\n    tmp_df = tmp_df.set_index('tag')\n    tags_df = tags_df.append(tmp_df)\n    \ntags_df['Procent poprawności'] = tags_df['Ilosc poprwnych'] / tags_df['Ile razy zadane']\ntags_df = tags_df.sort_values(by='Procent poprawności')\ntags_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tags_df = tags_df.sort_values(by='Procent poprawności')\nhow_much = 5\nhead = tags_df.head(how_much)\ntail = tags_df.tail(how_much)\n\nallin = head.append(tail)\nx = list(allin['Procent poprawności'].index)\ny = list(allin['Procent poprawności'].values)\n\nfig = plt.figure(figsize = (13, 5)) \nplt.bar(x, y, color=['red'] * how_much + ['lightgreen'] * how_much)\n  \nplt.xlabel('Tag') \nplt.ylabel('Procent poprawności') \nplt.show() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tags_df = tags_df.sort_values(by='Ile razy zadane')\nhow_much = 15\nhead = tags_df.head(how_much)\ntail = tags_df.tail(how_much)\n\nallin = head.append(tail)\nx = list(allin['Ile razy zadane'].index)\ny = list(allin['Ile razy zadane'].values)\n\nfig = plt.figure(figsize = (13, 5)) \nplt.bar(x, y, color=['red'] * how_much + ['lightgreen'] * how_much)\n  \nplt.xlabel('Tag') \nplt.ylabel('Ile razy zadane było pytanie z tym tagiem') \nplt.show() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Parts = części testu TOEIC, parts 1-4 słuchanie, parts 5-7 czytanie\n# Na wykresie: różowe - słuchanie, niebieski - czytanie\nfig = plt.figure(figsize=(13,5))\nquestions.groupby(\"part\").count()['question_id'].plot.bar(color=['pink'] * 4 + ['lightblue'] * 3)\nplt.title(\"Ile pytań zostało zadanych w danej części testu TOEIC\")\nplt.xlabel(\"Część testu\")\nplt.xticks(rotation=0)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Na wykresie: różowe - słuchanie, niebieski - czytanie\nparts = questions.groupby('part').agg({'Ile razy zadane': ['sum'], 'Ilosc poprwnych': ['sum']})\nparts['Procent poprawności'] = parts['Ilosc poprwnych'] / parts['Ile razy zadane']\nparts['Procent poprawności']\n\nx = list(parts['Procent poprawności'].index)\ny = list(parts['Procent poprawności'].values)\n\nfig = plt.figure(figsize = (13, 5)) \nplt.bar(x, y, color=['pink'] * 4 + ['lightblue'] * 3)\n  \nplt.xlabel('Część testu') \nplt.ylabel('Procent poprawnych odpowiedzi') \nplt.show() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bundles = questions['bundle_id']\nbundles.nunique()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bundles = questions.groupby('bundle_id', as_index=True).agg({'Ile razy zadane': ['sum'], 'Ilosc poprwnych': ['sum']})\nbundles['Procent poprawności'] = bundles['Ilosc poprwnych'] / bundles['Ile razy zadane']\nbundles = bundles.sort_values(by='Procent poprawności')\n\nhow_much = 4\n\nhead = bundles.head(how_much)\ntail = bundles.tail(how_much)\n\nallin = head.append(tail)\nx = list(allin['Procent poprawności'].index)\ny = list(allin['Procent poprawności'].values)\nfig = plt.figure(figsize = (13, 5)) \nplt.bar(x, y, width=60)\n  \nplt.xlabel('Bundle id') \nplt.ylabel('Procent poprawnych odpowiedzi') \nplt.show() \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bundles = questions.groupby('bundle_id').agg({'question_id': ['count']})\nbundles_by_count = bundles['question_id'].sort_values(by='count')\nprint(bundles_by_count)\n# maksymalnie 5 pytań występuje w danym bundle\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"index = bundles_by_count.index[len(bundles_by_count) - 1]\nquestions[questions.bundle_id == index]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"index = bundles_by_count.index[len(bundles_by_count) - 2]\nquestions[questions.bundle_id == index]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"index = bundles_by_count.index[len(bundles_by_count) - 3]\nquestions[questions.bundle_id == index]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"index = bundles_by_count.index[len(bundles_by_count) - 4]\nquestions[questions.bundle_id == index]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"index = bundles_by_count.index[len(bundles_by_count) - 5]\nquestions[questions.bundle_id == index]","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}