-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathdata_scraping.py
More file actions
47 lines (39 loc) · 2.79 KB
/
Copy pathdata_scraping.py
File metadata and controls
47 lines (39 loc) · 2.79 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
"""Mini data scraping project: scrape quotes and save to CSV."""
import csv # built-in module for reading/writing CSV files
import requests # library to download web pages (send HTTP requests)
from bs4 import BeautifulSoup # parses HTML so we can search it easily
# URL template; the {} gets replaced with a page number (page 1, 2, 3...)
BASE_URL = "https://quotes.toscrape.com/page/{}/"
def scrape(): # function that gathers all quotes and returns them
rows = [] # empty list to hold each quote as a dictionary
page = 1 # page counter, starts at page 1
while True: # loop forever until we 'break' out of it
# fill {} with current page number, download it, give up after 10 seconds
resp = requests.get(BASE_URL.format(page), timeout=10)
# turn the raw HTML text into a searchable object
soup = BeautifulSoup(resp.text, "html.parser")
quotes = soup.select(".quote") # find every element with class "quote" (a list)
if not quotes: # if the list is empty, there are no more pages
break # exit the while loop
for q in quotes: # loop over each quote block on this page
rows.append({ # build a dictionary and add it to rows
# find the .text element inside q, get its text, trim whitespace
"text": q.select_one(".text").get_text(strip=True),
# same for the author's name
"author": q.select_one(".author").get_text(strip=True),
# collect all .tag elements and join their text into one string
"tags": ", ".join(t.get_text() for t in q.select(".tag")),
})
print(f"Scraped page {page} ({len(quotes)} quotes)") # progress message
page += 1 # move to the next page
return rows # hand back the full list of quotes
def save(rows, filename="quotes.csv"): # write the list of quotes to a CSV file
# open the file for writing; utf-8 handles special characters
with open(filename, "w", newline="", encoding="utf-8") as f:
# writer that converts dictionaries into CSV rows; sets column order
writer = csv.DictWriter(f, fieldnames=["text", "author", "tags"])
writer.writeheader() # write the header row (column titles)
writer.writerows(rows) # write every quote as its own line
print(f"Saved {len(rows)} quotes to {filename}") # confirmation message
if __name__ == "__main__": # runs only when the file is executed directly
save(scrape()) # scrape the quotes, then save them to CSV