library(tidyverse)
ebird_out <- readRDS(url("https://gitlab.com/gklab/teaching/rr-f26/-/raw/main/weekly-activities/week-05/ebird-UniversityLakes.rds"))
# Your task:
# For each line of code, explain its purpose
# Avoid using the name of the function in your explanation.
# If you are not sure what the function does, view the dataframe before and after that line.
ebird_out |>
# Expand the dataset so that each row is one species in one checklist
unnest(ebird_out) |># View()
# Get rid of unused columns
select(year, months, comName, sciName, howMany) |> #View()
filter(year == 2024) |> # View()
summarise(total_seen = sum(howMany),
n_obsevers = n(),
.by = c(months, comName)) |>
mutate(avg_obs = total_seen/n_obsevers) |>
ungroup() |>
group_by(months) |>
mutate(n_species = n()) |> #View()
mutate(facet_label = paste("Months: ",months, "No. of species: ", n_species)) |> # View()
arrange(-avg_obs, .by_group = T) |> #View()
mutate(rank_obs = row_number()) |> #View()
filter(rank_obs < 21) |> #View()
nest() |> #print()
mutate(monthly_plot =
# Functional programming:
map(data, \(x)
x |>
mutate(comName = fct_reorder(comName, avg_obs)) |>
ggplot(aes(x = avg_obs, y = comName)) +
geom_linerange(aes(xmin = 0, xmax = avg_obs)) +
geom_point() +
labs(title = x$facet_label)
)) |> #print()
pull(monthly_plot) |>
patchwork::wrap_plots()
# Older version -- we could not organize species by rank within month this way!
ebird_out |>
# Expand the dataset so that each row is one species in one checklist
unnest(ebird_out) |># View()
# Get rid of unused columns
select(year, months, comName, sciName, howMany) |> #View()
filter(year == 2024) |> # View()
summarise(total_seen = sum(howMany),
n_obsevers = n(),
.by = c(months, comName)) |>
mutate(avg_obs = total_seen/n_obsevers) |>
ungroup() |>
group_by(months) |>
mutate(n_species = n()) |> #View()
mutate(facet_label = paste("Months: ",months, "No. of species: ", n_species)) |> # View()
arrange(-avg_obs, .by_group = T) |> #View()
mutate(rank_obs = row_number()) |> #View()
filter(rank_obs < 21) |>
ggplot(aes(x = avg_obs, y = rank_obs)) +
geom_point() +
geom_linerange(aes(xmin = 0, xmax = avg_obs)) + # you could also use geom_segment()
geom_text(aes(x = avg_obs, y = rank_obs, label = comName), hjust = 0) +
facet_wrap(~ facet_label, scales = "free") +
scale_x_log10() +
scale_y_reverse() +
theme_bw() +
theme(axis.text = element_text(color = "black", size = 12, face = "italic"))
# Filter the data set to show only the 20 most abundant species in each month
# Arrange the species by number of observations instead of alphabeticaly
# Not just have a point but also a line connecting the point to the Y-axis (used geom_linerange())
# Use a logarithmic scale for X-axis (used scale_x_log10())