-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathgoogle_search.py
More file actions
142 lines (98 loc) · 9.4 KB
/
Copy pathgoogle_search.py
File metadata and controls
142 lines (98 loc) · 9.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
# """
# google_search_practice.py
# A beginner-friendly Playwright (Python) script that opens Google,
# types a search query, submits it, and reads back the result titles.
# Every meaningful line is explained with an inline comment (#).
# """
# # ── Imports ────────────────────────────────────────────────────────────────
# from playwright.sync_api import sync_playwright # sync_playwright = the entry point for the synchronous (non-async) API
# import time # time = standard library, used here only to pause so you can watch
# # ── Main routine ───────────────────────────────────────────────────────────
# def run_search(query: str): # define a function that takes the text we want to search for
# # sync_playwright() starts the Playwright engine; "with" makes sure it shuts down cleanly at the end
# with sync_playwright() as p: # "p" is the Playwright object giving access to browsers
# # Launch a Chromium browser. headless=False means you SEE the window (great for practice/learning)
# browser = p.chromium.launch(headless=False, slow_mo=500) # slow_mo adds a 500ms delay between actions so it's visible
# # A "context" is like a fresh, isolated browser profile (no saved cookies/history)
# context = browser.new_context( # new_context = clean session, avoids leftover login/cookie state
# locale="en-US", # set language so Google's UI text is predictable
# viewport={"width": 1280, "height": 800} # set a fixed window size so layout is consistent
# )
# page = context.new_page() # open a new blank tab inside that context
# # Navigate to Google. wait_until="domcontentloaded" waits until the HTML is parsed
# page.goto("https://www.google.com", wait_until="domcontentloaded") # load the Google homepage
# # Google often shows a cookie-consent popup. Try to accept it, but don't crash if it's absent.
# try: # start a "try" block so a missing button won't stop the script
# # get_by_role finds a button by its accessible role+name; regex-like text match on "Accept"/"I agree"
# page.get_by_role("button", name="Accept all").click(timeout=3000) # click consent if it appears within 3s
# except Exception: # if that button isn't found (no popup / different wording)...
# pass # ...just ignore the error and move on
# # Find the search box. Google's box has the attribute name="q" (a stable selector)
# search_box = page.locator('textarea[name="q"], input[name="q"]') # locator matches either textarea or input named "q"
# search_box.click() # click the box to focus it before typing
# search_box.fill(query) # fill() clears the field and types the whole query at once
# search_box.press("Enter") # press Enter to submit the search (same as hitting the button)
# # Wait for the results section (id="search") to show up before we read anything
# page.wait_for_selector("#search", timeout=15000) # wait up to 15s for the results container to render
# # Collect result title elements. On the results page, titles are <h3> tags inside the results block.
# titles = page.locator("#search h3") # locator matching every <h3> heading inside #search
# count = titles.count() # count() = how many matching elements were found
# print(f"\nFound {count} result titles for: '{query}'\n") # print a summary header to the terminal
# # Loop through up to the first 10 titles and print their text
# for i in range(min(count, 10)): # min(count,10) keeps us from going past what's available
# text = titles.nth(i).inner_text() # nth(i) picks the i-th element; inner_text() grabs its visible text
# print(f"{i + 1}. {text}") # print a numbered line (i+1 so it starts at 1, not 0)
# time.sleep(3) # pause 3 seconds so you can look at the browser before it closes
# context.close() # close the context (tabs + session)
# browser.close() # close the whole browser
# # ── Entry point ────────────────────────────────────────────────────────────
# if __name__ == "__main__": # this block runs only when you execute the file directly
# run_search("playwright python tutorial") # call our function with a sample search query
"""
firefox_search_practice.py
Same Playwright practice script, but running in FIREFOX instead of Chromium.
It searches quotes.toscrape.com (a site made for automation practice, so no CAPTCHA)
and prints the quotes it finds. Every meaningful line has an inline comment (#).
Note: Playwright uses its OWN Firefox build over its OWN protocol.
It does NOT use Selenium/geckodriver "webdriver" — nothing extra to install
beyond `playwright install firefox`.
"""
# ── Imports ────────────────────────────────────────────────────────────────
from playwright.sync_api import sync_playwright # sync_playwright = entry point for the synchronous API
import time # time = used only to pause so you can watch
# ── Main routine ───────────────────────────────────────────────────────────
def run_search(tag: str): # take a "tag" to search for (e.g. "love", "life", "humor")
with sync_playwright() as p: # start Playwright; "p" gives access to the browsers
# THE KEY CHANGE: p.firefox instead of p.chromium — that's all it takes to switch engines
browser = p.firefox.launch( # launch Playwright's Firefox build
headless=False, # headless=False so you SEE the window
slow_mo=500 # 500ms delay between actions so it's easy to follow
)
context = browser.new_context( # fresh, isolated session (no saved cookies/history)
locale="en-US", # predictable UI language
viewport={"width": 1280, "height": 800} # fixed window size for consistent layout
)
page = context.new_page() # open a new blank tab
# Go to the practice site. wait_until="domcontentloaded" waits until the HTML is parsed
page.goto("https://quotes.toscrape.com", wait_until="domcontentloaded") # load the homepage
# This site lets you filter quotes by tag via a URL like /tag/love/ — click that tag if it exists
tag_link = page.locator(f'a.tag:has-text("{tag}")').first # find the first tag link matching our word
if tag_link.count() > 0: # count()>0 means such a tag link is present on the page
tag_link.click() # click it to filter quotes by that tag
else: # otherwise (tag not on the front page)...
page.goto(f"https://quotes.toscrape.com/tag/{tag}/") # ...go straight to that tag's URL
# Wait for at least one quote block to render before reading anything
page.wait_for_selector("div.quote", timeout=15000) # wait up to 15s for quote elements to appear
quotes = page.locator("div.quote") # locator matching every quote container on the page
count = quotes.count() # how many quotes were found
print(f"\nFound {count} quotes for tag: '{tag}'\n") # print a summary header
for i in range(count): # loop over each quote found
text = quotes.nth(i).locator("span.text").inner_text() # the quote text lives in span.text
author = quotes.nth(i).locator("small.author").inner_text() # the author lives in small.author
print(f"{i + 1}. {text} — {author}") # print numbered line: the quote and who said it
time.sleep(3) # pause so you can look at the Firefox window before it closes
context.close() # close the session
browser.close() # close Firefox
# ── Entry point ────────────────────────────────────────────────────────────
if __name__ == "__main__": # runs only when you execute the file directly
run_search("love") # try a sample tag; swap for "life", "humor", "books", etc.