## CSSS/POLS 510 - Maximum Likelihood Estimation 
## TA: Minji Jeong - UW, Department of Political Science
## Lab Week 1: Review of R + Intro to RMarkdown and Overleaf

rm(list = ls()) # clean memory


# Press Option + SHIFT + K (Mac) or ALT + SHIFT + K (Windows) to see all the shortcuts 


# Some Useful RStudio Shortcuts (for Windows)

# Press "CTRL + ENTER" to run code (either current line or selection)
# Press "CTRL + SHIFT + C" to comment/ uncomment code
# Press "CTRL + L" to clear the console
# Press "ALT + -" to insert assignment operator ( <- )
# Press "CTRL + SHIFT + M" to insert pipe operator ( |> )


#### R basics:----------------------------------------------------


# You can use R as a calculator:

# Arithmetic Operators
873.4 + 98827.3
2 * 8
9 / 3
2^3

# Relational Operators
10 > 8
7 <= 6
(2 * 5) == 10
1 != 2

# Concatenate elements in a vector
c(1, 2, 3)

# Assign them to a new object for manipulation
x <- c(1, 2, 3)

print(x) # or simply, x

# Use an object as input to a function
class(x)
length(x)
mean(x)

# Operators on vector
x + 1
x == 1

# Re-assign/override objects
x <- x + 1

# What are vectors?
# (Atomic) vectors are the most basic units of data

x <- c(1, 2, 3)
class(x)

y <- c(TRUE, FALSE, FALSE)
class(y)

names <- c("Peter", "Paul", "Mary")
class(names)

# What if we have different class realizations in a vector?

z <- c("Peter",2,FALSE)

class(z) # The simplest class (character) will override the rest

# Create a vector
numbers <- 1:12
print(numbers)

# Store it as a matrix
matrix1 <- matrix(data = numbers, 
                  nrow = 3)

print(matrix1)

# Basic information
class(matrix1)

dim(matrix1) # dimensions

# We can change the row/column names of matrices
rownames(matrix1) 

rownames(matrix1) <- c("row1", "row2", "row3")
matrix1

# Automate any repetitive process
col_names <- paste0("column", 1:4)
col_names

colnames(matrix1) <- col_names
matrix1

# To augment the matrix with new column
column5 <- c(13, 14, 15)
matrix1 <- cbind(matrix1, column5)

matrix1

# To augment the matrix with new row
row4 <- c("a", "b", "c", "d", "e")
matrix1 <- rbind(matrix1, row4)

matrix1


# Matrices can only contain one **homogenous** type of vectors
# Data frames or tibbles can contain **heterogeneous** types of vectors

df1 <- data.frame(
  names = c("Peter", "Paul", "Mary"),
  age = c(14, 15, 16),
  female = c(FALSE, FALSE, TRUE),
  stringsAsFactors = FALSE
)

print(df1)

# Basic information 
class(df1)
dim(df1)
str(df1)

# 1) Subsetting by row/column indices
# For the element in row 1, column 1
df1[1, 1]

# For all elements in row 1, regardless of columns
df1[1, ]

# For all elements in column 1, regardless of rows
df1[, 1]

# 2) Subsetting by variable names
df1$names
df1$age
df1$female

# 3) Subsetting by evaluations
df1[df1$age >= 15, ]
df1[df1$female == TRUE, ]
df1[df1$name %in% c("Peter", "Paul"), ]

# New vector with different data
y <- c(5:10, NA, -3:3)

mean(y)

# It's essential to know how to seek help:
help(mean)
?mean

# Specify appropriate arguments for functions:
mean(y, na.rm = TRUE)

# Substitute NA by a value of your choice:

y[is.na(y)] <- 0 # replacing NA by 0

y[y==0] <- NA    # replace 0 by NA




## Setting the working directories (wt)

# The getwd() function shows the working directory

getwd() # Note that R uses forward slashes instead of backward slashes.

mywd <- "your current directory here"

# note: directory addresses need to be in 
# forward slash "/" not back slash "\"

# The setwd() function tells R where you would like your files to save

setwd(dir=mywd)
dir()


# This is how generally set up the first lines (preamble) of all my script files:

# To set current folder as your wd. (requires "rstudioapi")
# install.packages("rstudioapi")

setwd(dirname(rstudioapi::getActiveDocumentContext()$path))
dir() # Check the file on your directory


# Other practices: creating projects or setting the wd with the mouse.



#### data wrangling:-------------------------------------------------

rm(list = ls()) # clean memory

# Followed by the list of packages that I will use.

#install.packages("tidyverse")

library(tidyverse)


# load data
df <- read_csv(file = "data/gapminder.csv")


# tibble (tbl) is a special class of data frame
class(df)
print(df)

## Chaining code using the pipe operator %>% or |> 

# Basic data wrangling

df %>%
  count(country)

count(df, country) # Equivalent to df %>% count(country)

# with base R:
arrange(count(df, country), n)

# with dplyr:
df %>%
  count(country) %>%
  arrange(n) # Rather than: arrange(count(df, country), n)


# Basic data wrangling: `filter()`

# Extract rows that meet logical criteria

df %>%
  filter(country == "Brazil")


df %>%
  filter(
    country == "Brazil" | country == "Russia (Soviet Union)" | 
      country == "India"  | country == "China"
  ) %>% print(n=50)

# or

df %>%
  filter(
    country %in% c("Brazil","Russia (Soviet Union)",
                   "India","China")
  ) %>% print(n=50)



df %>% select(country, year, gdpPercap)


USAdata <- 
  df %>%
  filter(country == "United States",
         year %in% 1980:2010) %>%
  select(year, gdpPercap)

USAdata # check data


# Basic data wrangling: `summarize()`

USAdata %>%
  summarize(avg_gdpPercap = mean(gdpPercap))


# Basic data wrangling: `group_by()` & `summarize()`

# Create a grouped version of the table with `group_by()`

df %>%
  group_by(country) %>%
  summarize(avg_gdpPercap = mean(gdpPercap)) %>%
  arrange(desc(avg_gdpPercap)) %>% 
  print(n=50)


# Basic data wrangling: `mutate()`

df %>% 
  mutate(
    row = row_number(),
    #  %/% operator is an integer division.
    decade = year %/% 10 * 10
    # It calculates the quotient of a division operation and discards the decimal part
  ) %>%
  select(row, country, year, decade, gdpPercap)

# What if we want to know countries' average GDP per capita over decades? 

df %>%
  mutate(decade = year %/% 10 * 10) %>%
  group_by(country, decade) %>%
  summarize(decAvg_gdp = mean(gdpPercap)) %>% 
  print(n=50)


# Basic data wrangling: long and wide format

df_long <-
  df %>%
  mutate(decade = year %/% 10 * 10) %>%
  group_by(country, decade) %>%
  summarize(decAvg_gdp = mean(gdpPercap))

# In "long" format:
# - each column is a variable
# - each row is an observation

df_long


# In "wide" format:
# - values of a variable spread over multiple columns

# From long to wide:

df_wide <-
  df_long %>%
  pivot_wider(id_cols = "country", 
              names_from = "decade", 
              values_from = "decAvg_gdp")


# From wide to long

df_wide %>% 
  pivot_longer(cols = `1950`:`2000`, 
               names_to = "deacde", 
               values_to = "gdp")

## load new data

pop <- read_csv("data/pop.csv")

names(pop)

# Compare with pop

names(df)


# merge data using "joint"

df2 <-
  df %>% 
  left_join(pop, by=c("country","year")) %>% 
  na.omit()

# show all columns: options(tibble.width = Inf)

## Let's create a new categorical variable based on population

# cutpoints by the quartiles of the population

Qts <- quantile(df2$pop, prob = c(0.25, 0.75), na.rm = TRUE)
print(Qts)


Q1 <- Qts[1]
Q3 <- Qts[2]

df2 <- 
  df2 %>%
  mutate(popCat = case_when(pop < Q1 ~ "low",
                            pop >= Q1 & pop < Q3 ~ "middle",
                            pop > Q3 ~ "high")) 

## create a dichotomous variable of country income level

df2 <-
  df2 %>% 
  mutate(wealth = ifelse(gdpPercap > mean(gdpPercap),1,0),
         wealth = factor(wealth,
                         levels = c(0,1),
                         labels = c("low_income","high_income")))


# Save the df of interest:

write.csv(df, file = "data/new_dat.csv", row.names = F)

save(df, file = "data/new_dat.Rdata")




#### tinytex for pdf files:-----------------------------------------------------------

# Install packages, uncomment the code below 
# (select lines and press "CTRL + SHIFT + C")

# install.packages(c("MASS","stargazer","questionr")) # use c() for vector
# 
# # You must install tinytex to render pdf documents.
# 
# install.packages("tinytex")
# 
# # WATNING: install tiny text only once! comment the code afterwards
# 
# tinytex::install_tinytex() 


# After tinytext is installed, open the .rmd file "RMarkdownSample" and 
# knit pdf to confirm everything is OK


########################################################
