112 lines
4.0 KiB
Python
Executable File
112 lines
4.0 KiB
Python
Executable File
#!/usr/bin/python3
|
|
|
|
# It is assumed that you have installed the required packages in requirements.txt
|
|
# and have also installed ChromeDriver from your package manager.
|
|
# If you have installed ChromeDriver separately, change the line below to the
|
|
# path of the ChromeDriver executable.
|
|
CHROMEDRIVER_PATH = '/usr/bin/chromedriver'
|
|
|
|
import networkx as nx
|
|
from selenium import webdriver
|
|
from selenium.webdriver.common.by import By
|
|
from selenium.webdriver.support.ui import WebDriverWait
|
|
from selenium.webdriver.support import expected_conditions
|
|
from selenium.webdriver.chrome.options import Options
|
|
from bs4 import BeautifulSoup
|
|
|
|
##### Set up NetworkX #######################################
|
|
|
|
graph = nx.DiGraph()
|
|
|
|
print('NetworkX initialised')
|
|
|
|
##### Set up WebDriver with Chrome ##########################
|
|
|
|
chrome_options = Options()
|
|
|
|
# Start headless so we can run on a server overnight
|
|
chrome_options.add_argument('--headless')
|
|
|
|
# Disable image loading, courtesy of rocky qi
|
|
# Found at https://stackoverflow.com/a/31581387, accessed on 2018/09/17 at 12:54 UTC
|
|
chrome_options.add_experimental_option("prefs", {"profile.managed_default_content_settings.images":2})
|
|
|
|
driver = webdriver.Chrome(executable_path=CHROMEDRIVER_PATH, chrome_options=chrome_options)
|
|
|
|
print('WebDriver initialised')
|
|
|
|
##### Do stuff ##############################################
|
|
|
|
# User class, holds all the data we will be collecting from user pages
|
|
# It's being used here as a more structured alternative to a dictionary
|
|
class User:
|
|
url_username = None
|
|
username = None
|
|
track_count = None
|
|
like_count = None
|
|
following_count = None
|
|
follower_count = None
|
|
|
|
user_dict = {}
|
|
users_to_process = ['thomotron', 'lacheque', 'slynk', 'bossfightswe']
|
|
|
|
for user_str in users_to_process:
|
|
print('Processing ' + user_str)
|
|
|
|
# Skip any users that have already been passed over to avoid infinite loops
|
|
if user_str in user_dict.keys():
|
|
print(' Already processed, skipping')
|
|
continue
|
|
|
|
# Initialise our user object to store our values in
|
|
user = User()
|
|
user.url_username = user_str
|
|
print(' URL Username: ' + user.url_username)
|
|
|
|
# Get the user's profile page and soup it
|
|
print('GET: https://soundcloud.com/' + user.url_username)
|
|
driver.get('https://soundcloud.com/' + user.url_username)
|
|
soup = BeautifulSoup(driver.page_source, 'lxml')
|
|
|
|
# Grab their username
|
|
# Use stripped_strings generator as workaround for users with premium badge
|
|
username = next(soup.find(class_='profileHeaderInfo__userName').stripped_strings)
|
|
user.username = username
|
|
print(' Username: ' + user.username)
|
|
|
|
# Grab their track count
|
|
track_count = soup.find('a', href='/' + user.url_username + '/tracks', class_='infoStats__statLink').div.string
|
|
user.track_count = track_count
|
|
print(' Tracks: ' + user.track_count)
|
|
|
|
# Grab their following count
|
|
following_count = soup.find('a', href='/' + user.url_username + '/following', class_='infoStats__statLink').div.string
|
|
user.following_count = following_count
|
|
print(' Following: ' + str(user.following_count))
|
|
|
|
# Grab their follower count
|
|
follower_count = soup.find('a', href='/' + user.url_username + '/followers', class_='infoStats__statLink').div.string
|
|
user.follower_count = follower_count
|
|
print(' Followers: ' + str(user.follower_count))
|
|
|
|
# Grab their like count
|
|
like_count_elem = soup.find('a', href='/' + user.url_username + '/likes').find(class_='sidebarHeader__actualTitle')
|
|
if like_count_elem:
|
|
user.like_count = like_count_elem.string.split(' ')[0]
|
|
print(' Likes: ' + str(user.like_count))
|
|
|
|
# Finally add the user to the dictionary
|
|
user_dict[user.url_username] = user
|
|
|
|
driver.quit()
|
|
|
|
print('Done scraping, here\'s what we got')
|
|
print('==================================')
|
|
|
|
for _, user in user_dict.items():
|
|
print(user.username)
|
|
print(' ' + str(user.track_count) + ' tracks')
|
|
print(' ' + str(user.like_count) + ' likes')
|
|
print(' Following ' + str(user.following_count))
|
|
print(' Followed by ' + str(user.follower_count))
|