- Introduction
Selenium was original an automated testing tool, but it's widely used in web scraping to handle JavaScript execution that requests libraries can't manage directly. Selenium works by driving a browser, fully simulating browser operations like navigation, input, clicks, and scrolling to obtain the rendered page results. It supports multiple browsers:
from selenium import webdriver
browser = webdriver.Chrome()
browser = webdriver.Firefox()
browser = webdriver.Edge()
browser = webdriver.Safari()
browser = webdriver.PhantomJS()
Official documantation: http://selenium-python.readthedocs.io
- Installation
2.1 With GUI Browsers
To install Selenium with Chrome:
# Install selenium and chromedriver
pip install selenium
# Download chromedriver.exe and place it in Python's scripts directory
# Note: The latest version is 2.38, not 2.9
# Download from: http://npm.taobao.org/mirrors/chromedriver/2.38/
# Or find the latest version at: https://sites.google.com/a/chromium.org/chromedriver/downloads
Verification:
C:\Users\Administrator>python3
Python 3.6.1 (v3.6.1:69c0db5, Mar 21 2017, 18:41:36) [MSC v.1900 64 bit (AMD64)] on win32
Type "help", "copyright", "credits" or "license" for more information.
>>> from selenium import webdriver
>>> driver = webdriver.Chrome() # Opens browser
>>> driver.get('https://www.baidu.com')
>>> driver.page_source
Note: Selenium 3 defaults to Firefox, which requires geckodriver installation.
2.2 Headless Browsers
PhantomJS is no longer maintained. Chrome now supports headless mode since version 59/60, making it a better alternative.
# selenium: 3.12.0
# webdriver: 2.38
# chrome.exe: 65.0.3325.181 (32-bit)
from selenium import webdriver
from selenium.webdriver.chrome.options import Options
chrome_options = Options()
chrome_options.add_argument('window-size=1920x3000')
chrome_options.add_argument('--disable-gpu')
chrome_options.add_argument('--hide-scrollbars')
chrome_options.add_argument('blink-settings=imagesEnabled=false')
chrome_options.add_argument('--headless')
chrome_options.binary_location = r"C:\Program Files (x86)\Google\Chrome\Application\chrome.exe"
driver = webdriver.Chrome(options=chrome_options)
driver.get('https://www.baidu.com')
print('hao123' in driver.page_source)
driver.close()
- Basic Usage
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
browser = webdriver.Chrome()
try:
browser.get('https://www.baidu.com')
input_field = browser.find_element(By.ID, 'kw')
input_field.send_keys('beautiful women')
input_field.send_keys(Keys.ENTER)
wait = WebDriverWait(browser, 10)
wait.until(EC.presence_of_element_located((By.ID, 'content_left')))
print(browser.page_source)
print(browser.current_url)
print(browser.get_cookies())
finally:
browser.close()
- Selectors
4.1 Basic Selectors
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
import time
driver = webdriver.Chrome()
driver.get('https://www.baidu.com')
wait = WebDriverWait(driver, 10)
try:
# All methods
# 1. find_element(By.ID, 'kw')
# 2. find_element(By.LINK_TEXT, 'text')
# 3. find_element(By.PARTIAL_LINK_TEXT, 'partial text')
# 4. find_element(By.TAG_NAME, 'tag')
# 5. find_element(By.CLASS_NAME, 'class')
# 6. find_element(By.NAME, 'name')
# 7. find_element(By.CSS_SELECTOR, 'selector')
# 8. find_element(By.XPATH, 'xpath')
# Note: find_elements_by_xxx returns a list of elements
# Examples
# 1. By ID
print(driver.find_element(By.ID, 'kw'))
# 2. By link text
# login = driver.find_element(By.LINK_TEXT, 'login')
# login.click()
# 3. By partial link text
login = driver.find_elements(By.PARTIAL_LINK_TEXT, 'login')[0]
login.click()
# 4. By tag name
print(driver.find_element(By.TAG_NAME, 'a'))
# 5. By class name
button = wait.until(EC.element_to_be_clickable((By.CLASS_NAME, 'tang-pass-footerBarULogin')))
button.click()
# 6. By name
user_input = wait.until(EC.presence_of_element_located((By.NAME, 'userName')))
pwd_input = wait.until(EC.presence_of_element_located((By.NAME, 'password')))
submit = wait.until(EC.element_to_be_clickable((By.ID, 'TANGRAM__PSP_10__submit')))
user_input.send_keys('18611453110')
pwd_input.send_keys('password123')
submit.click()
# 7. By CSS selector
driver.find_element(By.CSS_SELECTOR, '#kw')
# 8. By XPath
time.sleep(5)
finally:
driver.close()
4.2 XPath
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
import time
driver = webdriver.Chrome()
driver.get('https://doc.scrapy.org/en/latest/_static/selectors-sample1.html')
driver.implicitly_wait(3)
try:
# XPath examples
# // vs /
# driver.find_element(By.XPATH, '//body/a') # // means search from entire document, / means direct child
driver.find_element(By.XPATH, '//body//a') # // means search from entire document, // means any descendant
driver.find_element(By.CSS_SELECTOR, 'body a')
# Get nth element
res1 = driver.find_elements(By.XPATH, '//body//a[1]') # Get first a tag
print(res1[0].text)
# Find by attributes
res1 = driver.find_element(By.XPATH, '//a[5]')
res2 = driver.find_element(By.XPATH, '//a[@href="image5.html"]')
res3 = driver.find_element(By.XPATH, '//a[contains(@href,"image5")]') # Fuzzy match
print('==>', res1.text)
print('==>', res2.text)
print('==>', res3.text)
# Other examples
res1 = driver.find_element(By.XPATH, '/html/body/div/a')
print(res1.text)
res2 = driver.find_element(By.XPATH, '//a[img/@src="image3_thumb.jpg"]') # Find a tag with img child having specific src
print(res2.tag_name, res2.text)
res3 = driver.find_element(By.XPATH, "//input[@name='continue'][@type='button']")
res4 = driver.find_element(By.XPATH, "//*[@name='continue'][@type='button']")
time.sleep(5)
finally:
driver.close()
4.3 Getting Element Attributes
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
browser = webdriver.Chrome()
browser.get('https://www.amazon.com/')
wait = WebDriverWait(browser, 10)
wait.until(EC.presence_of_element_located((By.ID, 'cc-lm-tcgShowImgContainer')))
tag = browser.find_element(By.CSS_SELECTOR, '#cc-lm-tcgShowImgContainer img')
# Get element attributes
print(tag.get_attribute('src'))
# Get element ID, location, name, size
print(tag.id)
print(tag.location)
print(tag.tag_name)
print(tag.size)
browser.close()
- Waiting for Elements to Load
Selenium simulates browser behavior, but browsers take time to parse pages (execute CSS, JS). Some elements may take time to load. There are two waiting strategies:
5.1 Implicit Waiting
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
browser = webdriver.Chrome()
# Implicit wait: Wait up to 10 seconds for all elements
browser.implicitly_wait(10)
browser.get('https://www.baidu.com')
input_field = browser.find_element(By.ID, 'kw')
input_field.send_keys('beautiful women')
input_field.send_keys(Keys.ENTER)
contents = browser.find_element(By.ID, 'content_left')
print(contents)
browser.close()
5.2 Explicit Waiting
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
browser = webdriver.Chrome()
browser.get('https://www.baidu.com')
input_field = browser.find_element(By.ID, 'kw')
input_field.send_keys('beautiful women')
input_field.send_keys(Keys.ENTER)
# Explicit wait: Wait specifically for an element
wait = WebDriverWait(browser, 10)
wait.until(EC.presence_of_element_located((By.ID, 'content_left')))
contents = browser.find_element(By.CSS_SELECTOR, '#content_left')
print(contents)
browser.close()
- Element Interaction
6.1 Clicking and Clearing
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
browser = webdriver.Chrome()
browser.get('https://www.amazon.com/')
wait = WebDriverWait(browser, 10)
search_input = wait.until(EC.presence_of_element_located((By.ID, 'twotabsearchtextbox')))
search_input.send_keys('iphone 8')
search_button = browser.find_element(By.CSS_SELECTOR, '#nav-search > form > div.nav-right > div > input')
search_button.click()
import time
time.sleep(3)
search_input = browser.find_element(By.ID, 'twotabsearchtextbox')
search_input.clear() # Clear input field
search_input.send_keys('iphone7plus')
search_button = browser.find_element(By.CSS_SELECTOR, '#nav-search > form > div.nav-right > div > input')
search_button.click()
6.2 Action Chains
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
import time
driver = webdriver.Chrome()
driver.get('http://www.runoob.com/try/try.php?filename=jqueryui-api-droppable')
wait = WebDriverWait(driver, 3)
try:
driver.switch_to.frame('iframeResult')
source = driver.find_element(By.ID, 'draggable')
target = driver.find_element(By.ID, 'droppable')
# Method 1: Single action chain
actions = ActionChains(driver)
actions.drag_and_drop(source, target)
actions.perform()
# Method 2: Multiple action chains with different offsets
ActionChains(driver).click_and_hold(source).perform()
distance = target.location['x'] - source.location['x']
track = 0
while track < distance:
ActionChains(driver).move_by_offset(xoffset=2, yoffset=0).perform()
track += 2
ActionChains(driver).release().perform()
time.sleep(10)
finally:
driver.close()
6.3 JavaScript Execution
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
try:
browser = webdriver.Chrome()
browser.get('https://www.baidu.com')
browser.execute_script('alert("hello world")') # Execute JavaScript
finally:
browser.close()
6.4 Frame Switching
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
try:
browser = webdriver.Chrome()
browser.get('http://www.runoob.com/try/try.php?filename=jqueryui-api-droppable')
browser.switch_to.frame('iframeResult')
tag1 = browser.find_element(By.ID, 'droppable')
print(tag1)
browser.switch_to.parent_frame()
tag2 = browser.find_element(By.ID, 'textareaCode')
print(tag2)
finally:
browser.close()
- Additional Features
7.1 Browser Navigation
import time
from selenium import webdriver
browser = webdriver.Chrome()
browser.get('https://www.baidu.com')
browser.get('https://www.taobao.com')
browser.get('http://www.sina.com.cn/')
browser.back()
time.sleep(10)
browser.forward()
browser.close()
7.2 Cookies Management
from selenium import webdriver
browser = webdriver.Chrome()
browser.get('https://www.zhihu.com/explore')
print(browser.get_cookies())
browser.add_cookie({'k1': 'xxx', 'k2': 'yyy'})
print(browser.get_cookies())
# browser.delete_all_cookies()
7.3 Tab Management
import time
from selenium import webdriver
browser = webdriver.Chrome()
browser.get('https://www.baidu.com')
browser.execute_script('window.open()')
print(browser.window_handles)
browser.switch_to.window(browser.window_handles[1])
browser.get('https://www.taobao.com')
time.sleep(10)
browser.switch_to.window(browser.window_handles[0])
browser.get('https://www.sina.com.cn')
browser.close()
7.4 Expection Handling
from selenium import webdriver
from selenium.common.exceptions import TimeoutException, NoSuchElementException, NoSuchFrameException
try:
browser = webdriver.Chrome()
browser.get('http://www.runoob.com/try/try.php?filename=jqueryui-api-droppable')
browser.switch_to.frame('iframeResult')
except TimeoutException as e:
print(e)
except NoSuchFrameException as e:
print(e)
finally:
browser.close()
- Project Exercises
8.1 Automated Email Login and Sending
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
browser = webdriver.Chrome()
try:
browser.get('http://mail.163.com/')
wait = WebDriverWait(browser, 5)
frame = wait.until(EC.presence_of_element_located((By.ID, 'x-URS-iframe')))
browser.switch_to.frame(frame)
wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, '.m-container')))
user_input = browser.find_element(By.NAME, 'email')
pwd_input = browser.find_element(By.NAME, 'password')
login_button = browser.find_element(By.ID, 'dologin')
user_input.send_keys('18611453110')
pwd_input.send_keys('password123')
login_button.click()
# If captcha appears, uncomment the following
# import time
# time.sleep(10)
# login_button = browser.find_element(By.ID, 'dologin')
# login_button.click()
wait.until(EC.presence_of_element_located((By.ID, 'dvNavTop')))
compose = browser.find_elements(By.CSS_SELECTOR, '#dvNavTop li')[1]
compose.click()
wait.until(EC.presence_of_element_located((By.CLASS_NAME, 'tH0')))
recipient = browser.find_element(By.CLASS_NAME, 'nui-editableAddr-ipt')
subject = browser.find_element(By.CSS_SELECTOR, '.dG0 .nui-ipt-input')
recipient.send_keys('378533872@qq.com')
subject.send_keys('Important Message')
frame = wait.until(EC.presence_of_element_located((By.CLASS_NAME, 'APP-editor-iframe')))
browser.switch_to.frame(frame)
body = browser.find_element(By.CSS_SELECTOR, 'body')
body.send_keys('This is an automated message from Selenium.')
browser.switch_to.parent_frame()
send_button = browser.find_element(By.CLASS_NAME, 'nui-toolbar-item')
send_button.click()
import time
time.sleep(10000)
except Exception as e:
print(e)
finally:
browser.close()
8.2 Scraping JD.com Product Information
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
import time
def extract_products(driver):
try:
products = driver.find_elements(By.CLASS_NAME, 'gl-item')
for product in products:
detail_url = product.find_element(By.TAG_NAME, 'a').get_attribute('href')
name = product.find_element(By.CSS_SELECTOR, '.p-name em').text.replace('
', '')
price = product.find_element(By.CSS_SELECTOR, '.p-price i').text
reviews = product.find_element(By.CSS_SELECTOR, '.p-commit a').text
message = '''
Product: %s
Link: %s
Price: %s
Reviews: %s
''' % (name, detail_url, price, reviews)
print(message)
next_button = driver.find_element(By.PARTIAL_LINK_TEXT, 'Next Page')
next_button.click()
time.sleep(1)
extract_products(driver)
except Exception:
pass
def scrape_jd(keyword):
driver = webdriver.Chrome()
driver.get('https://www.jd.com/')
driver.implicitly_wait(3)
try:
search_input = driver.find_element(By.ID, 'key')
search_input.send_keys(keyword)
search_input.send_keys(Keys.ENTER)
extract_products(driver)
finally:
driver.close()
if __name__ == '__main__':
scrape_jd('iPhone8手机')
8.3 Homework
- Scrape Amazon iPhone product information
- Scrape Tmall Python book product information
- Scrape JD.com Xiaomi phone product information