soundcloud-scrape/scrape.py

343 lines
12 KiB
Python
Executable File

#!/usr/bin/python3
# It is assumed that you have installed the required packages in requirements.txt
# and have also installed ChromeDriver from your package manager.
# If you have installed ChromeDriver separately, change the line below to the
# path of the ChromeDriver executable.
CHROMEDRIVER_PATH = '/usr/bin/chromedriver'
# The number of steps outward from each starting user the script will scrape.
# For example:
# User <-- Follower <-- Follower's Follower
# ^0 ^1 ^2
# User is zero hops away from itself, so it's degree is zero.
# Follower is one hop away from User, so it's degree is one, and so on.
MAX_DEGREE = 0
# The list of starting users. Can be as many or as few as you would like.
# Bear in mind that each additional user will greatly increase run time, depending
# on how high the max degree is set to (see above).
STARTING_USERS = ['thomotron']
##### Import all the things #################################
import time
import networkx as nx
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions
from selenium.webdriver.chrome.options import Options
from bs4 import BeautifulSoup
##### Set up NetworkX #######################################
graph = nx.DiGraph()
print('NetworkX initialised')
##### Set up WebDriver with Chrome ##########################
chrome_options = Options()
# Start headless so we can run on a server overnight
chrome_options.add_argument('--headless')
# Disable image loading, courtesy of rocky qi
# Found at https://stackoverflow.com/a/31581387, accessed on 2018/09/17 at 12:54 UTC
chrome_options.add_experimental_option("prefs", {"profile.managed_default_content_settings.images":2})
driver = webdriver.Chrome(executable_path=CHROMEDRIVER_PATH, chrome_options=chrome_options)
print('WebDriver initialised')
##### Do stuff ##############################################
# User class, holds all the data we will be collecting from user pages
# It's being used here as a more structured alternative to a dictionary
class User:
url_username = None
username = None
track_count = None
like_count = None
following_count = None
follower_count = None
tracks = None
likes = None
followers = None
following = None
processed = False
def __init__(self):
self.tracks = []
self.likes = []
self.followers = []
self.following = []
class Track:
url = None
title = None
artist = None
date = None
tag = None # We'll only use the first tag for now, saves opening each track
plays = None
likes = None
reposts = None
comments = None
# Gets a user's username, track count, following count, follower count, and like count
def Get_Basic_Info(user):
# Get the user's profile page and soup it
print('GET: https://soundcloud.com/' + user.url_username)
driver.get('https://soundcloud.com/' + user.url_username)
soup = BeautifulSoup(driver.page_source, 'lxml')
print(' Basic Info:')
# Grab their username
# Use stripped_strings generator as workaround for users with premium badge
username = next(soup.find(class_='profileHeaderInfo__userName').stripped_strings)
user.username = username
print(' Username: ' + user.username)
# Grab their track count
track_count = soup.find('a', href='/' + user.url_username + '/tracks', class_='infoStats__statLink').div.string
user.track_count = track_count
print(' Tracks: ' + user.track_count)
# Grab their following count
following_count = soup.find('a', href='/' + user.url_username + '/following', class_='infoStats__statLink').div.string
user.following_count = following_count
print(' Following: ' + str(user.following_count))
# Grab their follower count
follower_count = soup.find('a', href='/' + user.url_username + '/followers', class_='infoStats__statLink').div.string
user.follower_count = follower_count
print(' Followers: ' + str(user.follower_count))
# Grab their like count
like_count_elem = soup.find('a', href='/' + user.url_username + '/likes').find(class_='sidebarHeader__actualTitle')
if like_count_elem:
user.like_count = like_count_elem.string.split(' ')[0]
print(' Likes: ' + str(user.like_count))
def Get_Tracks_Info(url, limit = 100):
# Get the page and soup it
print('GET: ' + url)
driver.get(url)
# Loop until we reach the bottom of the page so all tracks are loaded
soup = None
while True:
print('Scrolling...')
driver.execute_script('window.scrollTo(0, document.body.scrollHeight);') # Scroll to the bottom of the page
time.sleep(1) # Wait a second for the page to load
# Soup what we have
soup = BeautifulSoup(driver.page_source, 'lxml')
loading_elem = soup.find(class_='loading')
track_elems = soup.find_all(class_='soundList__item')
# Check if there are more items or we have enough
if not loading_elem:
print('Reached end of page')
break
elif len(track_elems) >= limit:
print('Reached track limit')
break
print(' Tracks:')
tracks = {}
track_elems = soup.find_all('li', class_='soundList__item')[:limit]
for track_elem in track_elems:
# Initialise a track object to hold our data
track = Track()
# Ignore collections
if track_elem.find(class_='sound__trackList'):
continue
# Grab the title and url
title_elem = track_elem.find(class_='soundTitle__title')
if title_elem:
track.title = title_elem.span.string.strip()
track.url = 'https://soundcloud.com' + title_elem['href']
# Check if we have already gotten this track before
if track.url in tracks.keys():
continue
# Grab who uploaded it (usually the artist unless it's a label)
artist_elem = track_elem.find(class_='soundTitle__usernameText')
if artist_elem:
track.artist = artist_elem.string.strip()
# Grab when it was uploaded
date_elem = track_elem.find(class_='soundTitle__uploadTime')
if date_elem:
track.date = date_elem.time['datetime']
# Grab the first tag featured on the list item
tag_elem = track_elem.find(class_='soundTitle__tagContent')
if tag_elem:
track.tag = tag_elem.string
# Grab the play and comment counts (either of these may or may not be present)
play_comment_elems = track_elem.find_all(class_='sc-ministats-item')
if play_comment_elems:
for elem in play_comment_elems:
num, unit = elem['title'].split(' ')
if unit == 'plays':
track.plays = num
elif unit == 'comments':
track.comments = num
# Grab the like count
like_elem = track_elem.find(class_='sc-button-like')
if like_elem:
track.likes = like_elem.string.strip()
# Grab the repost count
repost_elem = track_elem.find(class_='sc-button-repost')
if repost_elem:
track.reposts = repost_elem.string.strip()
# Finally add the track to the dictionary
tracks[track.url] = track
print(' ' + track.artist + ' - ' + track.title)
# Strip off keys and return a track list
return tracks.values()
def Get_Follows(url, limit = 100):
# Get the page
print('GET: ' + url)
driver.get(url)
# Loop until we reach the bottom of the page so all badges are loaded
soup = None
while True:
print('Scrolling...')
driver.execute_script('window.scrollTo(0, document.body.scrollHeight);') # Scroll to the bottom of the page
time.sleep(1) # Wait a second for the page to load
# Soup what we have
soup = BeautifulSoup(driver.page_source, 'lxml')
loading_elem = soup.find(class_='loading')
track_elems = soup.find_all(class_='badgeList__item')
# Check if there are more items or we have enough
if not loading_elem:
print('Reached end of page')
break
elif len(track_elems) >= limit:
print('Reached user limit')
break
print(' Users:')
users = {}
badge_elems = soup.find_all(class_='badgeList__item')[:limit]
for badge_elem in badge_elems:
username_elem = badge_elem.find('a', class_='userBadgeListItem__heading')
if username_elem:
user = User()
user.url_username = username_elem['href'].strip('/')
users[user.url_username] = user
print(' ' + user.url_username)
return users.values()
user_dict = {}
track_dict = {}
users_to_process = []
users_to_process_next = STARTING_USERS
for i in range(MAX_DEGREE + 1):
users_to_process = users_to_process_next
users_to_process_next = []
for user_str in users_to_process:
print('Processing ' + user_str + ' (distance ' + str(i) + ')')
# Skip any users that have already been passed over to avoid infinite loops
if user_str in user_dict.keys():
if user_dict[user_str].processed:
print(' Already processed, skipping')
continue
# Initialise our user object to store our values in
user = User()
user.url_username = user_str
print(' URL Username: ' + user.url_username)
# Get their basic info
Get_Basic_Info(user)
# Get their tracks
for track in Get_Tracks_Info('https://soundcloud.com/' + user.url_username + '/tracks', 10):
if not track.url in track_dict.keys():
track_dict[track.url] = track
user.tracks.append(track.url)
# Get their likes
for like in Get_Tracks_Info('https://soundcloud.com/' + user.url_username + '/likes', 10):
if not like.url in track_dict.keys():
track_dict[like.url] = like
user.likes.append(like.url)
# Get who they follow
for followed in Get_Follows('https://soundcloud.com/' + user.url_username + '/followed', 10):
if not followed.url_username in user_dict.keys():
user_dict[followed.url_username] = followed
user.followed.append(followed.url_username)
users_to_process_next.append(followed.url_username)
# Get who follows them
for follower in Get_Follows('https://soundcloud.com/' + user.url_username + '/followers', 10):
if not follower.url_username in user_dict.keys():
user_dict[follower.url_username] = follower
user.followers.append(follower.url_username)
users_to_process_next.append(follower.url_username)
# Finally add the user to the dictionary and mark them as processed
user.processed = True
user_dict[user.url_username] = user
driver.quit()
print('Done scraping, here\'s what we got')
print('==================================')
for _, user in user_dict.items():
if not user.processed:
continue
print(user.username)
print(' ' + str(user.track_count) + ' tracks')
print(' ' + str(user.like_count) + ' likes')
print(' Following ' + str(user.following_count))
print(' Followed by ' + str(user.follower_count))
print(' Tracks:')
for track in user.tracks:
print(' ' + str(track))
print(' Likes:')
for track in user.likes:
print(' ' + str(track))
print(' Followers:')
for follower in user.followers:
print(' ' + str(follower))
print(' Following:')
for followed in user.following:
print(' ' + str(followed))