Rather than have several copies of a user/track object, collect them all in a single dictionary and give the keys to each user object. This will avoid duplication and make generating graphs easier.
328 lines
11 KiB
Python
Executable File
328 lines
11 KiB
Python
Executable File
#!/usr/bin/python3
|
|
|
|
# It is assumed that you have installed the required packages in requirements.txt
|
|
# and have also installed ChromeDriver from your package manager.
|
|
# If you have installed ChromeDriver separately, change the line below to the
|
|
# path of the ChromeDriver executable.
|
|
CHROMEDRIVER_PATH = '/usr/bin/chromedriver'
|
|
|
|
import time
|
|
import networkx as nx
|
|
from selenium import webdriver
|
|
from selenium.webdriver.common.by import By
|
|
from selenium.webdriver.support.ui import WebDriverWait
|
|
from selenium.webdriver.support import expected_conditions
|
|
from selenium.webdriver.chrome.options import Options
|
|
from bs4 import BeautifulSoup
|
|
|
|
##### Set up NetworkX #######################################
|
|
|
|
graph = nx.DiGraph()
|
|
|
|
print('NetworkX initialised')
|
|
|
|
##### Set up WebDriver with Chrome ##########################
|
|
|
|
chrome_options = Options()
|
|
|
|
# Start headless so we can run on a server overnight
|
|
chrome_options.add_argument('--headless')
|
|
|
|
# Disable image loading, courtesy of rocky qi
|
|
# Found at https://stackoverflow.com/a/31581387, accessed on 2018/09/17 at 12:54 UTC
|
|
chrome_options.add_experimental_option("prefs", {"profile.managed_default_content_settings.images":2})
|
|
|
|
driver = webdriver.Chrome(executable_path=CHROMEDRIVER_PATH, chrome_options=chrome_options)
|
|
|
|
print('WebDriver initialised')
|
|
|
|
##### Do stuff ##############################################
|
|
|
|
# User class, holds all the data we will be collecting from user pages
|
|
# It's being used here as a more structured alternative to a dictionary
|
|
class User:
|
|
url_username = None
|
|
username = None
|
|
track_count = None
|
|
like_count = None
|
|
following_count = None
|
|
follower_count = None
|
|
|
|
tracks = None
|
|
likes = None
|
|
|
|
followers = None
|
|
following = None
|
|
|
|
processed = False
|
|
|
|
def __init__(self):
|
|
self.tracks = []
|
|
self.likes = []
|
|
self.followers = []
|
|
self.following = []
|
|
|
|
class Track:
|
|
url = None
|
|
title = None
|
|
artist = None
|
|
date = None
|
|
tag = None # We'll only use the first tag for now, saves opening each track
|
|
|
|
plays = None
|
|
likes = None
|
|
reposts = None
|
|
comments = None
|
|
|
|
# Gets a user's username, track count, following count, follower count, and like count
|
|
def Get_Basic_Info(user):
|
|
# Get the user's profile page and soup it
|
|
print('GET: https://soundcloud.com/' + user.url_username)
|
|
driver.get('https://soundcloud.com/' + user.url_username)
|
|
soup = BeautifulSoup(driver.page_source, 'lxml')
|
|
|
|
print(' Basic Info:')
|
|
|
|
# Grab their username
|
|
# Use stripped_strings generator as workaround for users with premium badge
|
|
username = next(soup.find(class_='profileHeaderInfo__userName').stripped_strings)
|
|
user.username = username
|
|
print(' Username: ' + user.username)
|
|
|
|
# Grab their track count
|
|
track_count = soup.find('a', href='/' + user.url_username + '/tracks', class_='infoStats__statLink').div.string
|
|
user.track_count = track_count
|
|
print(' Tracks: ' + user.track_count)
|
|
|
|
# Grab their following count
|
|
following_count = soup.find('a', href='/' + user.url_username + '/following', class_='infoStats__statLink').div.string
|
|
user.following_count = following_count
|
|
print(' Following: ' + str(user.following_count))
|
|
|
|
# Grab their follower count
|
|
follower_count = soup.find('a', href='/' + user.url_username + '/followers', class_='infoStats__statLink').div.string
|
|
user.follower_count = follower_count
|
|
print(' Followers: ' + str(user.follower_count))
|
|
|
|
# Grab their like count
|
|
like_count_elem = soup.find('a', href='/' + user.url_username + '/likes').find(class_='sidebarHeader__actualTitle')
|
|
if like_count_elem:
|
|
user.like_count = like_count_elem.string.split(' ')[0]
|
|
print(' Likes: ' + str(user.like_count))
|
|
|
|
def Get_Tracks_Info(url, limit = 100):
|
|
# Get the page and soup it
|
|
print('GET: ' + url)
|
|
driver.get(url)
|
|
|
|
# Loop until we reach the bottom of the page so all tracks are loaded
|
|
soup = None
|
|
while True:
|
|
print('Scrolling...')
|
|
driver.execute_script('window.scrollTo(0, document.body.scrollHeight);') # Scroll to the bottom of the page
|
|
time.sleep(1) # Wait a second for the page to load
|
|
|
|
# Soup what we have
|
|
soup = BeautifulSoup(driver.page_source, 'lxml')
|
|
loading_elem = soup.find(class_='loading')
|
|
track_elems = soup.find_all(class_='soundList__item')
|
|
|
|
# Check if there are more items or we have enough
|
|
if not loading_elem:
|
|
print('Reached end of page')
|
|
break
|
|
elif len(track_elems) >= limit:
|
|
print('Reached track limit')
|
|
break
|
|
|
|
print(' Tracks:')
|
|
|
|
tracks = {}
|
|
|
|
track_elems = soup.find_all('li', class_='soundList__item')[:limit]
|
|
for track_elem in track_elems:
|
|
# Initialise a track object to hold our data
|
|
track = Track()
|
|
|
|
# Ignore collections
|
|
if track_elem.find(class_='sound__trackList'):
|
|
continue
|
|
|
|
# Grab the title and url
|
|
title_elem = track_elem.find(class_='soundTitle__title')
|
|
if title_elem:
|
|
track.title = title_elem.span.string.strip()
|
|
track.url = 'https://soundcloud.com' + title_elem['href']
|
|
|
|
# Check if we have already gotten this track before
|
|
if track.url in tracks.keys():
|
|
continue
|
|
|
|
# Grab who uploaded it (usually the artist unless it's a label)
|
|
artist_elem = track_elem.find(class_='soundTitle__usernameText')
|
|
if artist_elem:
|
|
track.artist = artist_elem.string.strip()
|
|
|
|
# Grab when it was uploaded
|
|
date_elem = track_elem.find(class_='soundTitle__uploadTime')
|
|
if date_elem:
|
|
track.date = date_elem.time['datetime']
|
|
|
|
# Grab the first tag featured on the list item
|
|
tag_elem = track_elem.find(class_='soundTitle__tagContent')
|
|
if tag_elem:
|
|
track.tag = tag_elem.string
|
|
|
|
# Grab the play and comment counts (either of these may or may not be present)
|
|
play_comment_elems = track_elem.find_all(class_='sc-ministats-item')
|
|
if play_comment_elems:
|
|
for elem in play_comment_elems:
|
|
num, unit = elem['title'].split(' ')
|
|
if unit == 'plays':
|
|
track.plays = num
|
|
elif unit == 'comments':
|
|
track.comments = num
|
|
|
|
# Grab the like count
|
|
like_elem = track_elem.find(class_='sc-button-like')
|
|
if like_elem:
|
|
track.likes = like_elem.string.strip()
|
|
|
|
# Grab the repost count
|
|
repost_elem = track_elem.find(class_='sc-button-repost')
|
|
if repost_elem:
|
|
track.reposts = repost_elem.string.strip()
|
|
|
|
# Finally add the track to the dictionary
|
|
tracks[track.url] = track
|
|
|
|
print(' ' + track.artist + ' - ' + track.title)
|
|
|
|
# Strip off keys and return a track list
|
|
return tracks.values()
|
|
|
|
def Get_Follows(url, limit = 100):
|
|
# Get the page
|
|
print('GET: ' + url)
|
|
driver.get(url)
|
|
|
|
# Loop until we reach the bottom of the page so all badges are loaded
|
|
soup = None
|
|
while True:
|
|
print('Scrolling...')
|
|
driver.execute_script('window.scrollTo(0, document.body.scrollHeight);') # Scroll to the bottom of the page
|
|
time.sleep(1) # Wait a second for the page to load
|
|
|
|
# Soup what we have
|
|
soup = BeautifulSoup(driver.page_source, 'lxml')
|
|
loading_elem = soup.find(class_='loading')
|
|
track_elems = soup.find_all(class_='badgeList__item')
|
|
|
|
# Check if there are more items or we have enough
|
|
if not loading_elem:
|
|
print('Reached end of page')
|
|
break
|
|
elif len(track_elems) >= limit:
|
|
print('Reached user limit')
|
|
break
|
|
|
|
print(' Users:')
|
|
|
|
users = {}
|
|
|
|
badge_elems = soup.find_all(class_='badgeList__item')[:limit]
|
|
for badge_elem in badge_elems:
|
|
username_elem = badge_elem.find('a', class_='userBadgeListItem__heading')
|
|
if username_elem:
|
|
user = User()
|
|
user.url_username = username_elem['href'].strip('/')
|
|
users[user.url_username] = user
|
|
|
|
print(' ' + user.url_username)
|
|
|
|
return users.values()
|
|
|
|
|
|
user_dict = {}
|
|
track_dict = {}
|
|
|
|
users_to_process = []
|
|
users_to_process_next = ['thomotron', 'lacheque', 'slynk', 'bossfightswe']
|
|
|
|
for i in range(2):
|
|
users_to_process = users_to_process_next
|
|
users_to_process_next = []
|
|
for user_str in users_to_process:
|
|
print('Processing ' + user_str)
|
|
|
|
# Skip any users that have already been passed over to avoid infinite loops
|
|
if user_str in user_dict.keys():
|
|
if user_dict[user_str].processed:
|
|
print(' Already processed, skipping')
|
|
continue
|
|
|
|
# Initialise our user object to store our values in
|
|
user = User()
|
|
user.url_username = user_str
|
|
print(' URL Username: ' + user.url_username)
|
|
|
|
# Get their basic info
|
|
Get_Basic_Info(user)
|
|
|
|
# Get their tracks
|
|
for track in Get_Tracks_Info('https://soundcloud.com/' + user.url_username + '/tracks', 10):
|
|
if not track.url in track_dict.keys():
|
|
track_dict[track.url] = track
|
|
user.tracks.append(track.url)
|
|
|
|
# Get their likes
|
|
for like in Get_Tracks_Info('https://soundcloud.com/' + user.url_username + '/likes', 10):
|
|
if not like.url in track_dict.keys():
|
|
track_dict[like.url] = like
|
|
user.likes.append(like.url)
|
|
|
|
# Get who they follow
|
|
for followed in Get_Follows('https://soundcloud.com/' + user.url_username + '/followed', 10):
|
|
if not followed.url_username in user_dict.keys():
|
|
user_dict[followed.url_username] = followed
|
|
user.followed.append(followed.url_username)
|
|
users_to_process_next.append(followed.url_username)
|
|
|
|
# Get who follows them
|
|
for follower in Get_Follows('https://soundcloud.com/' + user.url_username + '/followers', 10):
|
|
if not follower.url_username in user_dict.keys():
|
|
user_dict[follower.url_username] = follower
|
|
user.followers.append(follower.url_username)
|
|
users_to_process_next.append(follower.url_username)
|
|
|
|
# Finally add the user to the dictionary and mark them as processed
|
|
user.processed = True
|
|
user_dict[user.url_username] = user
|
|
|
|
driver.quit()
|
|
|
|
print('Done scraping, here\'s what we got')
|
|
print('==================================')
|
|
|
|
for _, user in user_dict.items():
|
|
if not user.processed:
|
|
continue
|
|
|
|
print(user.username)
|
|
print(' ' + str(user.track_count) + ' tracks')
|
|
print(' ' + str(user.like_count) + ' likes')
|
|
print(' Following ' + str(user.following_count))
|
|
print(' Followed by ' + str(user.follower_count))
|
|
print(' Tracks:')
|
|
for track in user.tracks:
|
|
print(' ' + str(track))
|
|
print(' Likes:')
|
|
for track in user.likes:
|
|
print(' ' + str(track))
|
|
print(' Followers:')
|
|
for follower in user.followers:
|
|
print(' ' + str(follower))
|
|
print(' Following:')
|
|
for followed in user.following:
|
|
print(' ' + str(followed))
|