---
title: "Quick and Dirty Analysis of Avito"
author: "Bukun"
output:
  html_document:
    number_sections: true
    toc: true
    fig_width: 10
    code_folding: hide
    fig_height: 4.5
    theme: cosmo
    highlight: tango
---

#Introduction                 

From the Competiton Page 

> Avito, Russia’s largest classified advertisements website, is deeply familiar with this problem. Sellers on their platform sometimes feel frustrated with both too little demand (indicating something is wrong with the product or the product listing) or too much demand (indicating a hot item with a good description was underpriced).         

> In their fourth Kaggle competition, Avito is challenging you to predict demand for an online advertisement based on its full description (title, description, images, etc.), its context (geographically where it was posted, similar ads already posted) and historical demand for similar ads in similar contexts. With this information, Avito can inform sellers on how to best optimize their listing and provide some indication of how much interest they should realistically expect to receive.   


#Preparation{.tabset .tabset-fade .tabset-pills}

       
##Load Libraries

```{r,message=FALSE,warning=FALSE}

library(tidyverse)
library(tidytext)
library(stringr)
library(knitr)




```

##Read the data

```{r,message=FALSE,warning=FALSE}

rm(list=ls())

fillColor = "#FFA07A"
fillColor2 = "#F1C40F"

train = read_csv("../input/train.csv",locale = locale(encoding = stringi::stri_enc_get()))


```

#Glimpse of Data{.tabset .tabset-fade .tabset-pills}

##Train dataset

```{r,message=FALSE,warning=FALSE}

glimpse(train)

```

#Region Analysis{.tabset .tabset-fade .tabset-pills}

##Most Popular Region

The Most Popular Regions are `Krasnodar region` , `Sverdlovsk` , `Rostov` , `Tatarstan` and `Chelyabinsk`            

```{r,message=FALSE,warning=FALSE}

train %>%
  filter(!is.na(region)) %>%
  group_by(region) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(region = reorder(region,Count)) %>%
  head(10) %>%
  
  ggplot(aes(x = region,y = Count) ) +
  geom_bar(stat='identity',colour="white", fill = fillColor) +
  geom_text(aes(x = region, y = 1, label = paste0("(",round(Count)," )",sep="")),
            hjust=0, vjust=.5, size = 4, colour = 'black',
            fontface = 'bold') +
  labs(x = 'region', 
       y = 'Count', 
       title = 'Most popular region') +
  coord_flip() + 
  theme_bw()

```

##Region Data

```{r,message=FALSE,warning=FALSE}

train %>%
  filter(!is.na(region)) %>%
  group_by(region) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(region = reorder(region,Count)) %>%
  head(10) %>%
  kable()

```


##Region and Deal Probablity

```{r,message=FALSE,warning=FALSE}

dataset <- train %>%
  filter(!is.na(region)) %>%
  group_by(region) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(region = reorder(region,Count)) %>%
  head(10)

train %>%
  filter(region %in% dataset$region) %>%
  mutate( region = as.factor(region)) %>%
  ggplot(aes(x = region, y= deal_probability, fill = region)) +
  geom_boxplot() +
  labs(x= 'Region',y = 'Deal Probablity', 
       title = paste("Distribution of", 'Deal Probablity ')) +
  theme_bw() + theme(axis.text.x = element_text(angle = 90, hjust = 1))


```


#Most Popular City{.tabset .tabset-fade .tabset-pills}

The Most Popular Cities are `Krasnodar` , `Ekaterinburg` , `Novosibirsk` , `Rostov-na-Donu` and `Nizhny Novgorod`

```{r,message=FALSE,warning=FALSE}

train %>%
  filter(!is.na(city)) %>%
  group_by(city) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(city = reorder(city,Count)) %>%
  head(10) %>%
  
  ggplot(aes(x = city,y = Count) ) +
  geom_bar(stat='identity',colour="white", fill = fillColor2) +
  geom_text(aes(x = city, y = 1, label = paste0("(",round(Count)," )",sep="")),
            hjust=0, vjust=.5, size = 4, colour = 'black',
            fontface = 'bold') +
  labs(x = 'city', 
       y = 'Count', 
       title = 'Most popular city') +
  coord_flip() + 
  theme_bw()

```


##City Data

```{r,message=FALSE,warning=FALSE}

train %>%
  filter(!is.na(city)) %>%
  group_by(city) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(city = reorder(city,Count)) %>%
  head(10) %>%
  kable()

```



#Most Popular Category{.tabset .tabset-fade .tabset-pills}

The Most Popular Category is `Clothes, shoes, accessories` , `Children's clothing and footwear` , `Goods for children and toys` , `Apartments` , `Phones`               


```{r,message=FALSE,warning=FALSE}

train %>%
  filter(!is.na(category_name)) %>%
  group_by(category_name) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(category_name = reorder(category_name,Count)) %>%
  head(10) %>%
  
  ggplot(aes(x = category_name,y = Count) ) +
  geom_bar(stat='identity',colour="white", fill = fillColor) +
  geom_text(aes(x = category_name, y = 1, label = paste0("(",round(Count)," )",sep="")),
            hjust=0, vjust=.5, size = 4, colour = 'black',
            fontface = 'bold') +
  labs(x = 'Category', 
       y = 'Count', 
       title = 'Most popular Category') +
  coord_flip() + 
  theme_bw()

```

##Most Popular Category data

```{r,message=FALSE,warning=FALSE}

train %>%
  filter(!is.na(category_name)) %>%
  group_by(category_name) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(category_name = reorder(category_name,Count)) %>%
  head(10) %>%
  kable()

```


#Most Popular Parent Category{.tabset .tabset-fade .tabset-pills}

The Most Popular Parent Categories are `Personal things` , `home and cottages` , `Consumer electronics` , `Property` and `Hobbies and Recreation`              


```{r,message=FALSE,warning=FALSE}

train %>%
  filter(!is.na(parent_category_name)) %>%
  group_by(parent_category_name) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(parent_category_name = reorder(parent_category_name,Count)) %>%
  head(10) %>%
  
  ggplot(aes(x = parent_category_name,y = Count) ) +
  geom_bar(stat='identity',colour="white", fill = fillColor2) +
  geom_text(aes(x = parent_category_name, y = 1, label = paste0("(",round(Count)," )",sep="")),
            hjust=0, vjust=.5, size = 4, colour = 'black',
            fontface = 'bold') +
  labs(x = 'Parent Category', 
       y = 'Count', 
       title = 'Most popular Parent Category') +
  coord_flip() + 
  theme_bw()

```

##Most Popular Category data

```{r,message=FALSE,warning=FALSE}

train %>%
  filter(!is.na(parent_category_name)) %>%
  group_by(parent_category_name) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(parent_category_name = reorder(parent_category_name,Count)) %>%
  head(10) %>%
  kable()

```


#Distribution of Deal Probablity

```{r,message=FALSE,warning=FALSE}

train %>%
  ggplot(aes(x = deal_probability)) +
  geom_histogram(bins = 30,fill = fillColor2) +
  labs(x= 'Deal Probablity',y = 'Count', title = paste("Distribution of", ' Deal Probablity ')) +
  theme_bw()

summary(train$deal_probability)

```


#Distribution of Price

```{r,message=FALSE,warning=FALSE}

train %>%
    ggplot(aes(x = price) )+
    scale_x_log10(
      breaks = scales::trans_breaks("log10", function(x) 10^x),
      labels = scales::trans_format("log10", scales::math_format(10^.x))
    ) +
    scale_y_log10(
      breaks = scales::trans_breaks("log10", function(x) 10^x),
      labels = scales::trans_format("log10", scales::math_format(10^.x))
    ) + 
    geom_histogram(fill = fillColor2,bins=50) +
    labs(x = 'Price' ,y = 'Count', title = paste("Distribution of", "Price")) +
    theme_bw()

summary(train$price)

```

#Distribution of User Type

```{r,message=FALSE,warning=FALSE}

train %>%
  filter(!is.na(user_type)) %>%
  group_by(user_type) %>%
  summarise(Count = n()) %>%
  arrange(desc(Count)) %>%
  mutate(user_type = reorder(user_type,Count)) %>%
  head(10) %>%
  
  ggplot(aes(x = user_type,y = Count) ) +
  geom_bar(stat='identity',colour="white", fill = fillColor2) +
  geom_text(aes(x = user_type, y = 1, label = paste0("(",round(Count)," )",sep="")),
            hjust=0, vjust=.5, size = 4, colour = 'black',
            fontface = 'bold') +
  labs(x = 'User Type', 
       y = 'Count', 
       title = 'Most popular User Type') +
  coord_flip() + 
  theme_bw()

```



