soundcloud-scrape/scrape.py
2018-09-17 23:58:49 +09:30

62 lines
2.1 KiB
Python
Executable File

#!/usr/bin/python3
# It is assumed that you have installed the required packages in requirements.txt
# and have also installed ChromeDriver from your package manager.
# If you have installed ChromeDriver separately, change the line below to the
# path of the ChromeDriver executable.
CHROMEDRIVER_PATH = '/usr/bin/chromedriver'
import networkx as nx
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions
from selenium.webdriver.chrome.options import Options
from bs4 import BeautifulSoup
##### Set up NetworkX #######################################
graph = nx.DiGraph()
print('NetworkX initialised')
##### Set up WebDriver with Chrome ##########################
chrome_options = Options()
# Start headless so we can run on a server overnight
chrome_options.add_argument('--headless')
# Disable image loading, courtesy of rocky qi
# Found at https://stackoverflow.com/a/31581387, accessed on 2018/09/17 at 12:54 UTC
chrome_options.add_experimental_option("prefs", {"profile.managed_default_content_settings.images":2})
driver = webdriver.Chrome(executable_path=CHROMEDRIVER_PATH, chrome_options=chrome_options)
print('WebDriver initialised')
##### Do stuff ##############################################
user_dict = {}
users_to_process = ['thomotron', 'lacheque', 'slynk']
for user in users:
# Skip any users that have already been passed over to avoid infinite loops
if user in user_dict.keys():
continue
driver.get('https://soundcloud.com/' + user)
soup = BeautifulSoup(driver.page_source, 'lxml')
# Use stripped_strings generator as workaround for users with premium badge
username = next(soup.find(class_='profileHeaderInfo__userName').stripped_strings)
print(username)
track_count = int(soup.find('a', href='/' + username.lower() + '/tracks', class_='infoStats__statLink').div.string)
print(track_count)
like_count = int(soup.find('a', href='/' + username.lower() + '/likes').find(class_='sidebarHeader__actualTitle').string.split(' ')[0])
print(like_count)
driver.quit()