Print what's happening while scraping
This commit is contained in:
parent
4f36369d84
commit
e5b2e8861a
13
scrape.py
13
scrape.py
@ -51,15 +51,20 @@ user_dict = {}
|
|||||||
users_to_process = ['thomotron', 'lacheque', 'slynk', 'bossfightswe']
|
users_to_process = ['thomotron', 'lacheque', 'slynk', 'bossfightswe']
|
||||||
|
|
||||||
for user_str in users_to_process:
|
for user_str in users_to_process:
|
||||||
|
print('Processing ' + user_str)
|
||||||
|
|
||||||
# Skip any users that have already been passed over to avoid infinite loops
|
# Skip any users that have already been passed over to avoid infinite loops
|
||||||
if user_str in user_dict.keys():
|
if user_str in user_dict.keys():
|
||||||
|
print(' Already processed, skipping')
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# Initialise our user object to store our values in
|
# Initialise our user object to store our values in
|
||||||
user = User()
|
user = User()
|
||||||
user.url_username = user_str
|
user.url_username = user_str
|
||||||
|
print(' URL Username: ' + user.url_username)
|
||||||
|
|
||||||
# Get the user's profile page and soup it
|
# Get the user's profile page and soup it
|
||||||
|
print('GET: https://soundcloud.com/' + user.url_username)
|
||||||
driver.get('https://soundcloud.com/' + user.url_username)
|
driver.get('https://soundcloud.com/' + user.url_username)
|
||||||
soup = BeautifulSoup(driver.page_source, 'lxml')
|
soup = BeautifulSoup(driver.page_source, 'lxml')
|
||||||
|
|
||||||
@ -67,29 +72,37 @@ for user_str in users_to_process:
|
|||||||
# Use stripped_strings generator as workaround for users with premium badge
|
# Use stripped_strings generator as workaround for users with premium badge
|
||||||
username = next(soup.find(class_='profileHeaderInfo__userName').stripped_strings)
|
username = next(soup.find(class_='profileHeaderInfo__userName').stripped_strings)
|
||||||
user.username = username
|
user.username = username
|
||||||
|
print(' Username: ' + user.username)
|
||||||
|
|
||||||
# Grab their track count
|
# Grab their track count
|
||||||
track_count = soup.find('a', href='/' + user.url_username + '/tracks', class_='infoStats__statLink').div.string
|
track_count = soup.find('a', href='/' + user.url_username + '/tracks', class_='infoStats__statLink').div.string
|
||||||
user.track_count = track_count
|
user.track_count = track_count
|
||||||
|
print(' Tracks: ' + user.track_count)
|
||||||
|
|
||||||
# Grab their following count
|
# Grab their following count
|
||||||
following_count = soup.find('a', href='/' + user.url_username + '/following', class_='infoStats__statLink').div.string
|
following_count = soup.find('a', href='/' + user.url_username + '/following', class_='infoStats__statLink').div.string
|
||||||
user.following_count = following_count
|
user.following_count = following_count
|
||||||
|
print(' Following: ' + str(user.following_count))
|
||||||
|
|
||||||
# Grab their follower count
|
# Grab their follower count
|
||||||
follower_count = soup.find('a', href='/' + user.url_username + '/followers', class_='infoStats__statLink').div.string
|
follower_count = soup.find('a', href='/' + user.url_username + '/followers', class_='infoStats__statLink').div.string
|
||||||
user.follower_count = follower_count
|
user.follower_count = follower_count
|
||||||
|
print(' Followers: ' + str(user.follower_count))
|
||||||
|
|
||||||
# Grab their like count
|
# Grab their like count
|
||||||
like_count_elem = soup.find('a', href='/' + user.url_username + '/likes').find(class_='sidebarHeader__actualTitle')
|
like_count_elem = soup.find('a', href='/' + user.url_username + '/likes').find(class_='sidebarHeader__actualTitle')
|
||||||
if like_count_elem:
|
if like_count_elem:
|
||||||
user.like_count = like_count_elem.string.split(' ')[0]
|
user.like_count = like_count_elem.string.split(' ')[0]
|
||||||
|
print(' Likes: ' + str(user.like_count))
|
||||||
|
|
||||||
# Finally add the user to the dictionary
|
# Finally add the user to the dictionary
|
||||||
user_dict[user.url_username] = user
|
user_dict[user.url_username] = user
|
||||||
|
|
||||||
driver.quit()
|
driver.quit()
|
||||||
|
|
||||||
|
print('Done scraping, here\'s what we got')
|
||||||
|
print('==================================')
|
||||||
|
|
||||||
for _, user in user_dict.items():
|
for _, user in user_dict.items():
|
||||||
print(user.username)
|
print(user.username)
|
||||||
print(' ' + str(user.track_count) + ' tracks')
|
print(' ' + str(user.track_count) + ' tracks')
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user