Avoids an issue when running the script for the first time where Shelve would freak out and die because the shelf doesn't exist.
411 lines
14 KiB
Python
Executable File
411 lines
14 KiB
Python
Executable File
#!/usr/bin/python3
|
|
|
|
# It is assumed that you have installed the required packages in requirements.txt
|
|
# and have also installed ChromeDriver from your package manager.
|
|
# If you have installed ChromeDriver separately, change the line below to the
|
|
# path of the ChromeDriver executable.
|
|
CHROMEDRIVER_PATH = '/usr/bin/chromedriver'
|
|
|
|
# The number of steps outward from each starting user the script will scrape.
|
|
# For example:
|
|
# User <-- Follower <-- Follower's Follower
|
|
# ^0 ^1 ^2
|
|
# User is zero hops away from itself, so it's degree is zero.
|
|
# Follower is one hop away from User, so it's degree is one, and so on.
|
|
MAX_DEGREE = 0
|
|
|
|
# The list of starting users. Can be as many or as few as you would like.
|
|
# Bear in mind that each additional user will greatly increase run time, depending
|
|
# on how high the max degree is set to (see above).
|
|
STARTING_USERS = ['thomotron']
|
|
|
|
##### Import all the things #################################
|
|
|
|
import time
|
|
import shelve
|
|
import networkx as nx
|
|
from selenium import webdriver
|
|
from selenium.webdriver.common.by import By
|
|
from selenium.webdriver.support.ui import WebDriverWait
|
|
from selenium.webdriver.support import expected_conditions
|
|
from selenium.webdriver.chrome.options import Options
|
|
from bs4 import BeautifulSoup
|
|
|
|
##### Set up NetworkX #######################################
|
|
|
|
graph = nx.DiGraph()
|
|
|
|
print('NetworkX initialised')
|
|
|
|
##### Set up WebDriver with Chrome ##########################
|
|
|
|
chrome_options = Options()
|
|
|
|
# Start headless so we can run on a server overnight
|
|
chrome_options.add_argument('--headless')
|
|
|
|
# Disable image loading, courtesy of rocky qi
|
|
# Found at https://stackoverflow.com/a/31581387, accessed on 2018/09/17 at 12:54 UTC
|
|
chrome_options.add_experimental_option("prefs", {"profile.managed_default_content_settings.images":2})
|
|
|
|
driver = webdriver.Chrome(executable_path=CHROMEDRIVER_PATH, chrome_options=chrome_options)
|
|
|
|
print('WebDriver initialised')
|
|
|
|
##### Define some classes to hold our data ##################
|
|
|
|
# User class, holds all the data we will be collecting from user pages
|
|
# It's being used here as a more structured alternative to a dictionary
|
|
class User:
|
|
url_username = None
|
|
username = None
|
|
track_count = None
|
|
like_count = None
|
|
following_count = None
|
|
follower_count = None
|
|
|
|
tracks = None
|
|
likes = None
|
|
|
|
followers = None
|
|
following = None
|
|
|
|
processed = False
|
|
|
|
def __init__(self):
|
|
self.tracks = []
|
|
self.likes = []
|
|
self.followers = []
|
|
self.following = []
|
|
|
|
class Track:
|
|
url = None
|
|
title = None
|
|
artist = None
|
|
date = None
|
|
tag = None # We'll only use the first tag for now, saves opening each track
|
|
|
|
plays = None
|
|
likes = None
|
|
reposts = None
|
|
comments = None
|
|
|
|
##### Define some methods to do our scraping ################
|
|
|
|
# Gets a user's username, track count, following count, follower count, and like count
|
|
def Get_Basic_Info(user):
|
|
# Get the user's profile page and soup it
|
|
print('GET: https://soundcloud.com/' + user.url_username)
|
|
driver.get('https://soundcloud.com/' + user.url_username)
|
|
soup = BeautifulSoup(driver.page_source, 'lxml')
|
|
|
|
print(' Basic Info:')
|
|
|
|
# Grab their username
|
|
# Use stripped_strings generator as workaround for users with premium badge
|
|
username = next(soup.find(class_='profileHeaderInfo__userName').stripped_strings)
|
|
user.username = username
|
|
print(' Username: ' + user.username)
|
|
|
|
# Grab their track count
|
|
track_count = soup.find('a', href='/' + user.url_username + '/tracks', class_='infoStats__statLink').div.string
|
|
user.track_count = track_count
|
|
print(' Tracks: ' + user.track_count)
|
|
|
|
# Grab their following count
|
|
following_count = soup.find('a', href='/' + user.url_username + '/following', class_='infoStats__statLink').div.string
|
|
user.following_count = following_count
|
|
print(' Following: ' + str(user.following_count))
|
|
|
|
# Grab their follower count
|
|
follower_count = soup.find('a', href='/' + user.url_username + '/followers', class_='infoStats__statLink').div.string
|
|
user.follower_count = follower_count
|
|
print(' Followers: ' + str(user.follower_count))
|
|
|
|
# Grab their like count
|
|
like_count_elem = soup.find('a', href='/' + user.url_username + '/likes', class_='sidebarHeader__actualTitle')
|
|
if like_count_elem:
|
|
user.like_count = like_count_elem.string.split(' ')[0]
|
|
print(' Likes: ' + str(user.like_count))
|
|
|
|
# Gets a list of fully-populated Track objects
|
|
def Get_Tracks_Info(url, limit = 100):
|
|
# Get the page and soup it
|
|
print('GET: ' + url)
|
|
driver.get(url)
|
|
|
|
# Loop until we reach the bottom of the page so all tracks are loaded
|
|
soup = None
|
|
while True:
|
|
print('Scrolling...')
|
|
driver.execute_script('window.scrollTo(0, document.body.scrollHeight);') # Scroll to the bottom of the page
|
|
time.sleep(1) # Wait a second for the page to load
|
|
|
|
# Soup what we have
|
|
soup = BeautifulSoup(driver.page_source, 'lxml')
|
|
loading_elem = soup.find(class_='loading')
|
|
track_elems = soup.find_all(class_='soundList__item')
|
|
|
|
# Check if there are more items or we have enough
|
|
if not loading_elem:
|
|
print('Reached end of page')
|
|
break
|
|
elif len(track_elems) >= limit:
|
|
print('Reached track limit')
|
|
break
|
|
|
|
print(' Tracks:')
|
|
|
|
tracks = {}
|
|
|
|
track_elems = soup.find_all('li', class_='soundList__item')[:limit]
|
|
for track_elem in track_elems:
|
|
# Initialise a track object to hold our data
|
|
track = Track()
|
|
|
|
# Ignore collections
|
|
if track_elem.find(class_='sound__trackList'):
|
|
continue
|
|
|
|
# Grab the title and url
|
|
title_elem = track_elem.find(class_='soundTitle__title')
|
|
if title_elem:
|
|
track.title = title_elem.span.string.strip()
|
|
track.url = 'https://soundcloud.com' + title_elem['href']
|
|
|
|
# Check if we have already gotten this track before
|
|
if track.url in tracks.keys():
|
|
continue
|
|
|
|
# Grab who uploaded it (usually the artist unless it's a label)
|
|
artist_elem = track_elem.find(class_='soundTitle__usernameText')
|
|
if artist_elem:
|
|
track.artist = artist_elem.string.strip()
|
|
|
|
# Grab when it was uploaded
|
|
date_elem = track_elem.find(class_='soundTitle__uploadTime')
|
|
if date_elem:
|
|
track.date = date_elem.time['datetime']
|
|
|
|
# Grab the first tag featured on the list item
|
|
tag_elem = track_elem.find(class_='soundTitle__tagContent')
|
|
if tag_elem:
|
|
track.tag = tag_elem.string
|
|
|
|
# Grab the play and comment counts (either of these may or may not be present)
|
|
play_comment_elems = track_elem.find_all(class_='sc-ministats-item')
|
|
if play_comment_elems:
|
|
for elem in play_comment_elems:
|
|
num, unit = elem['title'].split(' ')
|
|
if unit == 'plays':
|
|
track.plays = num
|
|
elif unit == 'comments':
|
|
track.comments = num
|
|
|
|
# Grab the like count
|
|
like_elem = track_elem.find(class_='sc-button-like')
|
|
if like_elem:
|
|
track.likes = like_elem.string.strip()
|
|
|
|
# Grab the repost count
|
|
repost_elem = track_elem.find(class_='sc-button-repost')
|
|
if repost_elem:
|
|
track.reposts = repost_elem.string.strip()
|
|
|
|
# Finally add the track to the dictionary
|
|
tracks[track.url] = track
|
|
|
|
print(' ' + track.artist + ' - ' + track.title)
|
|
|
|
# Strip off keys and return a track list
|
|
return tracks.values()
|
|
|
|
# Gets a list of User objects from a badge list containing only their url_username
|
|
def Get_Follows(url, limit = 100):
|
|
# Get the page
|
|
print('GET: ' + url)
|
|
driver.get(url)
|
|
|
|
# Loop until we reach the bottom of the page so all badges are loaded
|
|
soup = None
|
|
while True:
|
|
print('Scrolling...')
|
|
driver.execute_script('window.scrollTo(0, document.body.scrollHeight);') # Scroll to the bottom of the page
|
|
time.sleep(1) # Wait a second for the page to load
|
|
|
|
# Soup what we have
|
|
soup = BeautifulSoup(driver.page_source, 'lxml')
|
|
loading_elem = soup.find(class_='loading')
|
|
track_elems = soup.find_all(class_='badgeList__item')
|
|
|
|
# Check if there are more items or we have enough
|
|
if not loading_elem:
|
|
print('Reached end of page')
|
|
break
|
|
elif len(track_elems) >= limit:
|
|
print('Reached user limit')
|
|
break
|
|
|
|
print(' Users:')
|
|
|
|
users = {}
|
|
|
|
badge_elems = soup.find_all(class_='badgeList__item')[:limit]
|
|
for badge_elem in badge_elems:
|
|
username_elem = badge_elem.find('a', class_='userBadgeListItem__heading')
|
|
if username_elem:
|
|
user = User()
|
|
user.url_username = username_elem['href'].strip('/')
|
|
users[user.url_username] = user
|
|
|
|
print(' ' + user.url_username)
|
|
|
|
return users.values()
|
|
|
|
##### Scrape ################################################
|
|
|
|
user_dict = {}
|
|
track_dict = {}
|
|
|
|
with shelve.open('shelf.db') as shelf:
|
|
if 'user_dict' in shelf.keys():
|
|
user_dict = shelf['user_dict']
|
|
if 'track_dict' in shelf.keys():
|
|
track_dict = shelf['track_dict']
|
|
|
|
users_to_process = []
|
|
users_to_process_next = STARTING_USERS
|
|
|
|
for i in range(MAX_DEGREE + 1):
|
|
users_to_process = users_to_process_next
|
|
users_to_process_next = []
|
|
for user_str in users_to_process:
|
|
print('Processing ' + user_str + ' (distance ' + str(i) + ')')
|
|
|
|
# Skip any users that have already been passed over to avoid infinite loops
|
|
if user_str in user_dict.keys():
|
|
if user_dict[user_str].processed:
|
|
print(' Already processed, skipping')
|
|
continue
|
|
|
|
# Initialise our user object to store our values in
|
|
user = User()
|
|
user.url_username = user_str
|
|
print(' URL Username: ' + user.url_username)
|
|
|
|
# Get their basic info
|
|
Get_Basic_Info(user)
|
|
|
|
# Get their tracks
|
|
for track in Get_Tracks_Info('https://soundcloud.com/' + user.url_username + '/tracks', 10):
|
|
if not track.url in track_dict.keys():
|
|
track_dict[track.url] = track
|
|
user.tracks.append(track.url)
|
|
|
|
# Get their likes
|
|
for like in Get_Tracks_Info('https://soundcloud.com/' + user.url_username + '/likes', 10):
|
|
if not like.url in track_dict.keys():
|
|
track_dict[like.url] = like
|
|
user.likes.append(like.url)
|
|
|
|
# Get who they follow
|
|
for followed in Get_Follows('https://soundcloud.com/' + user.url_username + '/followed', 10):
|
|
if not followed.url_username in user_dict.keys():
|
|
user_dict[followed.url_username] = followed
|
|
user.followed.append(followed.url_username)
|
|
users_to_process_next.append(followed.url_username)
|
|
|
|
# Get who follows them
|
|
for follower in Get_Follows('https://soundcloud.com/' + user.url_username + '/followers', 10):
|
|
if not follower.url_username in user_dict.keys():
|
|
user_dict[follower.url_username] = follower
|
|
user.followers.append(follower.url_username)
|
|
users_to_process_next.append(follower.url_username)
|
|
|
|
# Finally add the user to the dictionary and mark them as processed
|
|
user.processed = True
|
|
user_dict[user.url_username] = user
|
|
|
|
# Update the shelf with the new dictionaries
|
|
with shelve.open('shelf.db') as shelf:
|
|
shelf['user_dict'] = user_dict
|
|
shelf['track_dict'] = track_dict
|
|
|
|
driver.quit()
|
|
|
|
##### Process what was scraped ##############################
|
|
|
|
print('Done scraping, here\'s what we got')
|
|
print('==================================')
|
|
|
|
for _, track in track_dict.items():
|
|
graph.add_node( \
|
|
track.url, \
|
|
type='track', \
|
|
label=str(track.title), \
|
|
artist=str(track.artist), \
|
|
title=str(track.title), \
|
|
date=str(track.date), \
|
|
tag=str(track.tag), \
|
|
plays=str(track.plays), \
|
|
likes=str(track.likes), \
|
|
reposts=str(track.reposts), \
|
|
comments=str(track.comments) \
|
|
)
|
|
|
|
for _, user in user_dict.items():
|
|
if not user.processed:
|
|
continue
|
|
|
|
graph.add_node( \
|
|
user.url_username, \
|
|
type='user', \
|
|
label=str(user.username), \
|
|
track_count=str(user.track_count), \
|
|
like_count=str(user.like_count), \
|
|
following_count=str(user.following_count), \
|
|
follower_count=str(user.follower_count) \
|
|
)
|
|
|
|
print(user.username)
|
|
print(' ' + str(user.track_count) + ' tracks')
|
|
print(' ' + str(user.like_count) + ' likes')
|
|
print(' Following ' + str(user.following_count))
|
|
print(' Followed by ' + str(user.follower_count))
|
|
print(' Tracks:')
|
|
for track in user.tracks:
|
|
print(' ' + str(track))
|
|
graph.add_edge( \
|
|
user.url_username, \
|
|
track, \
|
|
label='created'
|
|
)
|
|
print(' Likes:')
|
|
for track in user.likes:
|
|
print(' ' + str(track))
|
|
graph.add_edge( \
|
|
user.url_username, \
|
|
track, \
|
|
label='likes'
|
|
)
|
|
print(' Followers:')
|
|
for follower in user.followers:
|
|
print(' ' + str(follower))
|
|
graph.add_edge( \
|
|
follower, \
|
|
user.url_username, \
|
|
label='follows'
|
|
)
|
|
print(' Following:')
|
|
for followed in user.following:
|
|
print(' ' + str(followed))
|
|
graph.add_edge( \
|
|
user.url_username, \
|
|
follower, \
|
|
label='follows'
|
|
)
|
|
|
|
print('Making graph...')
|
|
nx.write_gexf(graph, 'graph.gexf')
|
|
print('Graph written to \'graph.gexf\'')
|