Charity


SUBMITTED BY: Charity

DATE: May 10, 2020, 10:21 a.m.

FORMAT: Python 3

SIZE: 3.7 kB

HITS: 565

  1. from bs4 import BeautifulSoup
  2. import requests
  3. import pandas as pd
  4. url = "https://coinmarketcap.com/all/views/all/"
  5. # puting headers to make a crawler looks human like + using request lib
  6. headers = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_9_5) AppleWebKit 537.36 (KHTML, like Gecko) Chrome",
  7. "Accept": "text/html,application/xhtml+xml,application/xml; q=0.9,image/webp,*/*;q=0.8"}
  8. response = requests.get(url, headers=headers)
  9. soup = BeautifulSoup(response.text, "html.parser")
  10. positions = soup.findAll("td", {"class":"cmc-table__cell cmc-table__cell--sticky cmc-table__cell--sortable cmc-table__cell--left cmc-table__cell--sort-by__rank"})
  11. names = soup.findAll("div", {"class":"cmc-table__column-name sc-1kxikfi-0 eTVhdN"})
  12. symbols = soup.findAll("td", {"class":"cmc-table__cell cmc-table__cell--sortable cmc-table__cell--left cmc-table__cell--sort-by__symbol"})
  13. marketcaps = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__market-cap"})
  14. prices = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__price"})
  15. supplies = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__circulating-supply"})
  16. volumes = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__volume-24-h"})
  17. hours = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__percent-change-1-h"})
  18. days = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__percent-change-24-h"})
  19. weeks = soup.findAll("td", {"class": "cmc-table__cell cmc-table__cell--sortable cmc-table__cell--right cmc-table__cell--sort-by__percent-change-7-d"})
  20. # collecting positions
  21. mypositions = []
  22. for position in positions:
  23. mypositions.append(position.text)
  24. # collecting names in a list
  25. mynames_list = []
  26. for name in names:
  27. mynames_list.append(name.text)
  28. # collecting symblos in a list
  29. mysymbols_list = []
  30. for symbol in symbols:
  31. mysymbols_list.append(symbol.text)
  32. # collecting a list of market caps
  33. mymarketcap_list = []
  34. for marketcap in marketcaps:
  35. mymarketcap_list.append(marketcap.text)
  36. # collecting prices
  37. myprices_list = []
  38. for price in prices:
  39. myprices_list.append(price.text)
  40. # collecting circulating prices
  41. mysupplies_list = []
  42. for supply in supplies:
  43. mysupplies_list.append(supply.text)
  44. # collecting volumes
  45. myvolumes_list = []
  46. for volume in volumes:
  47. myvolumes_list.append(volume.text)
  48. # collecting hours
  49. myhours_list = []
  50. for hour in hours:
  51. myhours_list.append(hour.text)
  52. # collecting days
  53. mydays_list = []
  54. for day in days:
  55. mydays_list.append(day.text)
  56. # collecting weeks
  57. myweeks_list = []
  58. for week in weeks:
  59. myweeks_list.append(week.text)
  60. # creating CSV file
  61. data_frame = pd.DataFrame({
  62. "#": mypositions,
  63. "Name": mynames_list,
  64. "Symbol": mysymbols_list,
  65. "Market Cap": mymarketcap_list,
  66. "Price": myprices_list,
  67. "Circulating supply": mysupplies_list,
  68. "Volume(24H)": myvolumes_list,
  69. "%1H": myhours_list,
  70. "%24H": mydays_list,
  71. "%7": myweeks_list
  72. })
  73. print(data_frame)
  74. data_frame.to_csv("CoinmarketCap.csv", )
  75. """
  76. you can also get html
  77. data_frame.to_csv("CoinmarketCap.csv",)
  78. you can get pdf
  79. from weasyprint import HTML
  80. HTML(string=html_out).write_pdf("report.pdf")
  81. """

comments powered by Disqus