Week 6: Data visualization: Theory and practice, continued

Announcements

  • The homework assignment originally announced as being due October 4th is now due 11th.
  • We will have time in class next week to work on this assignment.

In-class materials

library(tidyverse)
ebird_out <- readRDS(url("https://gitlab.com/gklab/teaching/rr-f26/-/raw/main/weekly-activities/week-05/ebird-UniversityLakes.rds"))

# Your task:
# For each line of code, explain its purpose
# Avoid using the name of the function in your explanation.
# If you are not sure what the function does, view the dataframe before and after that line.

ebird_out |> 
  # Expand the dataset so that each row is one species in one checklist
  unnest(ebird_out) |># View()
  # Get rid of unused columns
  select(year, months, comName, sciName, howMany) |>  #View()
  filter(year == 2024) |> # View()
  summarise(total_seen = sum(howMany),
            n_obsevers = n(),
            .by = c(months, comName)) |> 
  mutate(avg_obs = total_seen/n_obsevers) |>
  ungroup() |> 
  group_by(months) |> 
  mutate(n_species = n()) |>  #View()
  mutate(facet_label = paste("Months: ",months, "No. of species: ", n_species)) |> # View()
  arrange(-avg_obs, .by_group = T) |> #View()
  mutate(rank_obs = row_number()) |> #View()
  filter(rank_obs < 21) |> #View()
  nest() |>  #print()
  mutate(monthly_plot =
           # Functional programming: 
           map(data, \(x)
               x |>
                 mutate(comName = fct_reorder(comName, avg_obs)) |>
                 ggplot(aes(x = avg_obs, y = comName)) +
                 geom_linerange(aes(xmin = 0, xmax = avg_obs)) +
                 geom_point() +
                 labs(title = x$facet_label)
               ))  |> #print()
  pull(monthly_plot) |>
  patchwork::wrap_plots()







# Older version -- we could not organize species by rank within month this way!
ebird_out |> 
  # Expand the dataset so that each row is one species in one checklist
  unnest(ebird_out) |># View()
  # Get rid of unused columns
  select(year, months, comName, sciName, howMany) |>  #View()
  filter(year == 2024) |> # View()
  summarise(total_seen = sum(howMany),
            n_obsevers = n(),
            .by = c(months, comName)) |> 
  mutate(avg_obs = total_seen/n_obsevers) |>
  ungroup() |> 
  group_by(months) |> 
  mutate(n_species = n()) |>  #View()
  mutate(facet_label = paste("Months: ",months, "No. of species: ", n_species)) |> # View()
  arrange(-avg_obs, .by_group = T) |> #View()
  mutate(rank_obs = row_number()) |> #View()
  filter(rank_obs < 21) |> 
  ggplot(aes(x = avg_obs, y = rank_obs)) +
  geom_point() +
  geom_linerange(aes(xmin = 0, xmax = avg_obs)) + # you could also use geom_segment()
  geom_text(aes(x = avg_obs, y = rank_obs, label = comName), hjust = 0) +
  facet_wrap(~ facet_label, scales = "free") +
  scale_x_log10() +
  scale_y_reverse() +
  theme_bw() + 
  theme(axis.text = element_text(color = "black", size = 12, face = "italic"))
  

# Filter the data set to show only the 20 most abundant species in each month
# Arrange the species by number of observations instead of alphabeticaly
# Not just have a point but also a line connecting the point to the Y-axis (used geom_linerange())
# Use a logarithmic scale for X-axis (used scale_x_log10())