library(tidyverse)
library(rvest)
library(lubridate)
library(here)Scrape charts
Goals
Here I work out how to scrape the chart, and then test it as a function.
Given a Billboard chart slug name (from it’s URL) and date (default to current), scrape the chart and save the file.
Charts I’m using this on are Billboard Hot 100 and Billboard 200.
A compilation of the resulting files are handled in a different script.
This same concept is used with the Github Action version that scrapes only the current chart:
action_scrape_charts.R.
The “write_rds” step is commented out so files are not overwritten unless purposely changes.
Setup
Set up the list of charts
This comes from the url of the chart, like “hot-100” from https://www.billboard.com/charts/hot-100/.
charts_list <- c("hot-100", "billboard-200")Date options
I set up some flags to get a specific date or the current date. I won’t need this with the final Github Action
Fpulls the current chartThis won’t work anymore because old charts are behind a paywall.Tpulls the chart that is noted in theedition_requestobject
edition_request <- "2026-09-19"
edition_flag <- FScrape a single file
Here I work out how to scrape the Hot 100 table.
main_url <- "https://www.billboard.com/charts/hot-100/"
# get initial scrape
scrape_first <- read_html(main_url)
# pull date from scrape
chart_date <- scrape_first |>
html_element("div#chart-date-picker") |>
html_attr("data-date")
# pull year from scrape
chart_year <- year(chart_date)
# process results into a tibble
scrape_tibble <- scrape_first |>
html_elements("ul.o-chart-results-list-row") |>
html_text2() |>
as_tibble()Each line has some extra columns repeated … LW, PEAK, WEEKS ON CHART. On the first line for the number 1 song, there is also WEEKS AT NO. 1. This was added the chart week of Sept. 19th when the top song tied for the longest. I’m not sure if it will remain a part of the chart, but I’ve had to make changes to deal with it.
Process first row
Here we grab the first row and extract just the first seven “fields” so I remove the duplicated variables. Doing this one separately from the other 99 allows me to keep the “weeks at No. 1” variable.
scrape_toprow <- scrape_tibble |>
slice(1) |>
mutate(
data_cleaned = str_remove_all(value, " NEW\nNEW|LW\n|PEAK\n|WEEKS ON CHART\n|WEEKS AT NO. 1\n|WEEKS\n"),
data_cleaned = str_remove_all(data_cleaned, " RE- ENTRY\n|RE- ENTRY|RE-ENTRY"),
data_cleaned = str_extract(data_cleaned, "^((?:[^\n]*\n){6}[^\n]*)") # sevens
) |>
select(data_cleaned, value) |>
separate(
col = data_cleaned,
sep = "\n",
into = c(
"current_week",
"title",
"performer",
"last_week",
"peak_pos",
"wks_at_no1",
"wks_on_chart"
)
) |>
select(-value) |>
mutate(chart_week = chart_date, .before = current_week)Process other rows
Here we process the other 99 rows as a 6-column table.
scrape_rest <- scrape_tibble |>
slice(2:100) |>
mutate(
data_cleaned = str_remove_all(value, " NEW\nNEW|LW\n|PEAK\n|WEEKS ON CHART\n|WEEKS AT NO. 1\n|WEEKS\n"),
data_cleaned = str_remove_all(data_cleaned, " RE- ENTRY\n|RE- ENTRY|RE-ENTRY"),
data_cleaned = str_extract(data_cleaned, "^((?:[^\n]*\n){5}[^\n]*)") # sevens
) |>
select(data_cleaned, value) |>
separate(
col = data_cleaned,
sep = "\n",
into = c(
"current_week",
"title",
"performer",
"last_week",
"peak_pos",
# "wks_at_no1",
"wks_on_chart"
)
) |>
select(-value) |>
mutate(chart_week = chart_date, .before = current_week)Bind them together
Now we put those two together to for a single 100-row table.
scrape_clean <- scrape_toprow |>
bind_rows(scrape_rest)
scrape_cleanMost rows have NA for wks_at_no1.
Scraper function starts here
Since the Hot 100 and Billboard 200 (albums) are structured the same, I turn the scraper into a function and run it on a list of charts. I only do the two, though I could theoretically capture the other charts like Artist 100. To get the archive, I would have to subscribe.
It is this code that is duplicated in action_scrape_charts.R
for (chart in charts_list) {
# Sets edition flag
edition_current <- ""
if (edition_flag == T) edition_page <- edition_request else edition_page <- edition_current
# make the url
scrape_url <- paste(
"https://www.billboard.com/charts/",
chart,
"/",
edition_page,
sep = ""
)
# get initial scrape
scrape_first <- read_html(scrape_url)
# pull date from scrape
chart_date <- scrape_first |>
html_element("div#chart-date-picker") |>
html_attr("data-date")
# pull year from scrape
chart_year <- year(chart_date)
# process results into a tibble
scrape_tibble <- scrape_first |>
html_elements("ul.o-chart-results-list-row") |>
html_text2() |>
as_tibble()
# process first row
scrape_toprow <- scrape_tibble |>
slice(1) |>
mutate(
data_cleaned = str_remove_all(value, " NEW\nNEW|LW\n|PEAK\n|WEEKS ON CHART\n|WEEKS AT NO. 1\n|WEEKS\n"),
data_cleaned = str_remove_all(data_cleaned, " RE- ENTRY\n|RE- ENTRY|RE-ENTRY"),
data_cleaned = str_extract(data_cleaned, "^((?:[^\n]*\n){6}[^\n]*)") # sevens
) |>
select(data_cleaned, value) |>
separate(
col = data_cleaned,
sep = "\n",
into = c(
"current_week",
"title",
"performer",
"last_week",
"peak_pos",
"wks_at_no1",
"wks_on_chart"
)
) |>
select(-value) |>
mutate(chart_week = chart_date, .before = current_week)
# process rest of rows
scrape_rest <- scrape_tibble |>
slice(2:100) |>
mutate(
data_cleaned = str_remove_all(value, " NEW\nNEW|LW\n|PEAK\n|WEEKS ON CHART\n|WEEKS AT NO. 1\n|WEEKS\n"),
data_cleaned = str_remove_all(data_cleaned, " RE- ENTRY\n|RE- ENTRY|RE-ENTRY"),
data_cleaned = str_extract(data_cleaned, "^((?:[^\n]*\n){5}[^\n]*)") # sevens
) |>
select(data_cleaned, value) |>
separate(
col = data_cleaned,
sep = "\n",
into = c(
"current_week",
"title",
"performer",
"last_week",
"peak_pos",
# "wks_at_no1",
"wks_on_chart"
)
) |>
select(-value) |>
mutate(chart_week = chart_date, .before = current_week)
# bind them
scrape_clean <- scrape_toprow |>
bind_rows(scrape_rest)
# name path to save file
folder_path <- paste("data-scraped/", chart, "/", chart_year, "/", sep = "")
# create the directory if it doesn't exist
if (!dir.exists(here(folder_path))) {dir.create(here(folder_path), recursive = TRUE)}
# write the file
scrape_clean |> write_csv(paste(folder_path, chart_date, ".csv", sep = ""))
}