soundcloud-scrape/scrape.py
Thomas Wade 0c496ecd71 Add constructor for User class
This was causing issues where appending to any of the lists would update the class variable rather than the instance variable, leading to everyone having the same likes, tracks, etc.
2018-09-25 19:01:07 +09:30

307 lines
10 KiB
Python
Executable File

#!/usr/bin/python3
# It is assumed that you have installed the required packages in requirements.txt
# and have also installed ChromeDriver from your package manager.
# If you have installed ChromeDriver separately, change the line below to the
# path of the ChromeDriver executable.
CHROMEDRIVER_PATH = '/usr/bin/chromedriver'
import time
import networkx as nx
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions
from selenium.webdriver.chrome.options import Options
from bs4 import BeautifulSoup
##### Set up NetworkX #######################################
graph = nx.DiGraph()
print('NetworkX initialised')
##### Set up WebDriver with Chrome ##########################
chrome_options = Options()
# Start headless so we can run on a server overnight
chrome_options.add_argument('--headless')
# Disable image loading, courtesy of rocky qi
# Found at https://stackoverflow.com/a/31581387, accessed on 2018/09/17 at 12:54 UTC
chrome_options.add_experimental_option("prefs", {"profile.managed_default_content_settings.images":2})
driver = webdriver.Chrome(executable_path=CHROMEDRIVER_PATH, chrome_options=chrome_options)
print('WebDriver initialised')
##### Do stuff ##############################################
# User class, holds all the data we will be collecting from user pages
# It's being used here as a more structured alternative to a dictionary
class User:
url_username = None
username = None
track_count = None
like_count = None
following_count = None
follower_count = None
tracks = None
likes = None
followers = None
following = None
def __init__(self):
self.tracks = []
self.likes = []
self.followers = []
self.following = []
class Track:
url = ''
title = None
artist = None
date = None
tag = None # We'll only use the first tag for now, saves opening each track
plays = None
likes = None
reposts = None
comments = None
# Gets a user's username, track count, following count, follower count, and like count
def Get_Basic_Info(user):
# Get the user's profile page and soup it
print('GET: https://soundcloud.com/' + user.url_username)
driver.get('https://soundcloud.com/' + user.url_username)
soup = BeautifulSoup(driver.page_source, 'lxml')
print(' Basic Info:')
# Grab their username
# Use stripped_strings generator as workaround for users with premium badge
username = next(soup.find(class_='profileHeaderInfo__userName').stripped_strings)
user.username = username
print(' Username: ' + user.username)
# Grab their track count
track_count = soup.find('a', href='/' + user.url_username + '/tracks', class_='infoStats__statLink').div.string
user.track_count = track_count
print(' Tracks: ' + user.track_count)
# Grab their following count
following_count = soup.find('a', href='/' + user.url_username + '/following', class_='infoStats__statLink').div.string
user.following_count = following_count
print(' Following: ' + str(user.following_count))
# Grab their follower count
follower_count = soup.find('a', href='/' + user.url_username + '/followers', class_='infoStats__statLink').div.string
user.follower_count = follower_count
print(' Followers: ' + str(user.follower_count))
# Grab their like count
like_count_elem = soup.find('a', href='/' + user.url_username + '/likes').find(class_='sidebarHeader__actualTitle')
if like_count_elem:
user.like_count = like_count_elem.string.split(' ')[0]
print(' Likes: ' + str(user.like_count))
def Get_Tracks_Info(url, limit = 100):
# Get the page and soup it
print('GET: ' + url)
driver.get(url)
# Loop until we reach the bottom of the page so all tracks are loaded
soup = None
while True:
print('Scrolling...')
driver.execute_script('window.scrollTo(0, document.body.scrollHeight);') # Scroll to the bottom of the page
time.sleep(1) # Wait a second for the page to load
# Soup what we have
soup = BeautifulSoup(driver.page_source, 'lxml')
loading_elem = soup.find(class_='loading')
track_elems = soup.find_all(class_='soundList__item')
# Check if there are more items or we have enough
if not loading_elem:
print('Reached end of page')
break
elif len(track_elems) >= limit:
print('Reached track limit')
break
print(' Tracks:')
tracks = {}
track_elems = soup.find_all('li', class_='soundList__item')[:limit]
for track_elem in track_elems:
# Initialise a track object to hold our data
track = Track()
# Ignore collections
if track_elem.find(class_='sound__trackList'):
continue
# Grab the title and url
title_elem = track_elem.find(class_='soundTitle__title')
if title_elem:
track.title = title_elem.span.string.strip()
track.url = 'https://soundcloud.com' + title_elem['href']
# Check if we have already gotten this track before
if track.url in tracks.keys():
continue
# Grab who uploaded it (usually the artist unless it's a label)
artist_elem = track_elem.find(class_='soundTitle__usernameText')
if artist_elem:
track.artist = artist_elem.string.strip()
# Grab when it was uploaded
date_elem = track_elem.find(class_='soundTitle__uploadTime')
if date_elem:
track.date = date_elem.time['datetime']
# Grab the first tag featured on the list item
tag_elem = track_elem.find(class_='soundTitle__tagContent')
if tag_elem:
track.tag = tag_elem.string
# Grab the play and comment counts (either of these may or may not be present)
play_comment_elems = track_elem.find_all(class_='sc-ministats-item')
if play_comment_elems:
for elem in play_comment_elems:
num, unit = elem['title'].split(' ')
if unit == 'plays':
track.plays = num
elif unit == 'comments':
track.comments = num
# Grab the like count
like_elem = track_elem.find(class_='sc-button-like')
if like_elem:
track.likes = like_elem.string.strip()
# Grab the repost count
repost_elem = track_elem.find(class_='sc-button-repost')
if repost_elem:
track.reposts = repost_elem.string.strip()
# Finally add the track to the dictionary
tracks[track.url] = track
print(' ' + track.artist + ' - ' + track.title)
# Strip off keys and return a track list
return tracks.values()
def Get_Follows(url, limit = 100):
# Get the page
print('GET: ' + url)
driver.get(url)
# Loop until we reach the bottom of the page so all badges are loaded
soup = None
while True:
print('Scrolling...')
driver.execute_script('window.scrollTo(0, document.body.scrollHeight);') # Scroll to the bottom of the page
time.sleep(1) # Wait a second for the page to load
# Soup what we have
soup = BeautifulSoup(driver.page_source, 'lxml')
loading_elem = soup.find(class_='loading')
track_elems = soup.find_all(class_='badgeList__item')
# Check if there are more items or we have enough
if not loading_elem:
print('Reached end of page')
break
elif len(track_elems) >= limit:
print('Reached user limit')
break
print(' Users:')
users = []
badge_elems = soup.find_all(class_='badgeList__item')[:limit]
for badge_elem in badge_elems:
username_elem = badge_elem.find('a', class_='userBadgeListItem__heading')
if username_elem:
url_username = username_elem['href'].strip('/')
users.append(url_username)
print(' ' + url_username)
return users
user_dict = {}
users_to_process = []
users_to_process_next = ['thomotron', 'lacheque', 'slynk', 'bossfightswe']
for i in range(2):
users_to_process = users_to_process_next
users_to_process_next = []
for user_str in users_to_process:
print('Processing ' + user_str)
# Skip any users that have already been passed over to avoid infinite loops
if user_str in user_dict.keys():
print(' Already processed, skipping')
continue
# Initialise our user object to store our values in
user = User()
user.url_username = user_str
print(' URL Username: ' + user.url_username)
# Get their basic info
Get_Basic_Info(user)
# Get their tracks
user.tracks = Get_Tracks_Info('https://soundcloud.com/' + user.url_username + '/tracks', 10)
# Get their likes
user.likes = Get_Tracks_Info('https://soundcloud.com/' + user.url_username + '/likes', 10)
# Get who they follow
user.following = Get_Follows('https://soundcloud.com/' + user.url_username + '/following', 10)
users_to_process_next.extend(user.following)
# Get who follows them
user.followers = Get_Follows('https://soundcloud.com/' + user.url_username + '/followers', 10)
users_to_process_next.extend(user.followers)
# Finally add the user to the dictionary
user_dict[user.url_username] = user
driver.quit()
print('Done scraping, here\'s what we got')
print('==================================')
for _, user in user_dict.items():
print(user.username)
print(' ' + str(user.track_count) + ' tracks')
print(' ' + str(user.like_count) + ' likes')
print(' Following ' + str(user.following_count))
print(' Followed by ' + str(user.follower_count))
print(' Tracks:')
for track in user.tracks:
print(' ' + str(track.title))
print(' Likes:')
for track in user.likes:
print(' ' + str(track.title))
print(' Followers:')
for follower in user.followers:
print(' ' + str(follower.url_username))
print(' Following:')
for followed in user.following:
print(' ' + str(followed.url_username))