from bs4 import BeautifulSoup
import requests
import pandas as pd
url = "https://coinmarketcap.com/all/views/all/"
# puting headers to make a crawler looks human like + using request lib
headers = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_9_5) AppleWebKit 537.36 (KHTML, like Gecko) Chrome",
"Accept": "text/html,application/xhtml+xml,application/xml; q=0.9,image/webp,*/*;q=0.8"}
response = requests.get(url, headers=headers)
soup = BeautifulSoup(response.text, "html.parser")
positions = soup.findAll("td", {"class":"cmc-table__cell cmc-table__cell--sticky cmc-table__cell--sortable cmc-table__cell--left cmc-table__cell--sort-by__rank"})
names = soup.findAll("div", {"class":"cmc-table__column-name sc-1kxikfi-0 eTVhdN"})
symbols = soup.findAll("td", {"class":"cmc-table__cell cmc-table__cell--sortable cmc-table__cell--left cmc-table__cell--sort-by__symbol"})
marketcaps = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__market-cap"})
prices = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__price"})
supplies = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__circulating-supply"})
volumes = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__volume-24-h"})
hours = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__percent-change-1-h"})
days = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__percent-change-24-h"})
weeks = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__percent-change-7-d"})
# collecting positions
mypositions = []
for position in positions:
mypositions.append(position.text)
# collecting names in a list
mynames_list = []
for name in names:
mynames_list.append(name.text)
# collecting symblos in a list
mysymbols_list = []
for symbol in symbols:
mysymbols_list.append(symbol.text)
# collecting a list of market caps
mymarketcap_list = []
for marketcap in marketcaps:
mymarketcap_list.append(marketcap.text)
# collecting prices
myprices_list = []
for price in prices:
myprices_list.append(price.text)
# collecting circulating prices
mysupplies_list = []
for supply in supplies:
mysupplies_list.append(supply.text)
# collecting volumes
myvolumes_list = []
for volume in volumes:
myvolumes_list.append(volume.text)
# collecting hours
myhours_list = []
for hour in hours:
myhours_list.append(hour.text)
# collecting days
mydays_list = []
for day in days:
mydays_list.append(day.text)
# collecting weeks
myweeks_list = []
for week in weeks:
myweeks_list.append(week.text)
# creating CSV file
data_frame = pd.DataFrame({
"#": mypositions,
"Name": mynames_list,
"Symbol": mysymbols_list,
"Market Cap": mymarketcap_list,
"Price": myprices_list,
"Circulating supply": mysupplies_list,
"Volume(24H)": myvolumes_list,
"%1H": myhours_list,
"%24H": mydays_list,
"%7": myweeks_list
})
print(data_frame)
data_frame.to_csv("CoinmarketCap.csv", )
"""
you can also get html
data_frame.to_csv("CoinmarketCap.csv",)
you can get pdf
from weasyprint import HTML
HTML(string=html_out).write_pdf("report.pdf")
"""