1.chrome瀏覽器
# !/usr/bin/python # -*- coding: utf-8 -*- from selenium import webdriver # 進入瀏覽器設置 options = webdriver.ChromeOptions() #谷歌無頭模式 options.add_argument('--headless') options.add_argument('--disable-gpu') # options.add_argument('window-size=1200x600') # 設置中文 options.add_argument('lang=zh_CN.UTF-8') # 更換頭部 options.add_argument('user-agent="Mozilla/5.0 (iPod; U; CPU iPhone OS 2_1 like Mac OS X; ja-jp) AppleWebKit/525.18.1 (KHTML, like Gecko) Version/3.1.1 Mobile/5F137 Safari/525.20"') #設置代理 if proxy: options.add_argument('proxy-server=' + proxy) if user_agent: options.add_argument('user-agent=' + user_agent) browser = webdriver.Chrome(chrome_options=options) url = "https://httpbin.org/get?show_env=1" browser.get(url) browser.quit()
selenium設置chrome–cookie # !/usr/bin/python # -*- coding: utf-8 -*- from selenium import webdriver browser = webdriver.Chrome() url = "https://www.baidu.com/" browser.get(url) # 通過js新打開一個窗口 newwindow='window.open("https://www.baidu.com");' # 刪除原來的cookie browser.delete_all_cookies() # 攜帶cookie打開 browser.add_cookie({'name':'ABC','value':'DEF'}) # 通過js新打開一個窗口 browser.execute_script(newwindow) input("查看效果") browser.quit()
selenium設置chrome-圖片不加載 from selenium import webdriver options = webdriver.ChromeOptions() prefs = { 'profile.default_content_setting_values': { 'images': 2 } } options.add_experimental_option('prefs', prefs) browser = webdriver.Chrome(chrome_options=options) # browser = webdriver.Chrome() url = "http://image.baidu.com/" browser.get(url) input("是否有圖") browser.quit()
2.firefox瀏覽器
import time from selenium.webdriver.common.proxy import* myProxy = '202.202.90.20:8080' proxy = Proxy({ 'proxyType': ProxyType.MANUAL, 'httpProxy': myProxy, 'ftpProxy': myProxy, 'sslProxy': myProxy, 'noProxy': '' }) profile = webdriver.FirefoxProfile()
if proxy: profile = get_firefox_profile_with_proxy_set(profile, proxy) if user_agent: profile.set_preference("general.useragent.override", user_agent) driver=webdriver.Firefox(proxy=proxy,profile=profile) driver.get('https://www.baidu.com') time.sleep(3) driver.quit()
firefox無頭模式 from selenium import webdriver # 創建的新實例驅動 options = webdriver.FirefoxOptions() #火狐無頭模式 options.add_argument('--headless') options.add_argument('--disable-gpu') # options.add_argument('window-size=1200x600') executable_path='./source/geckodriver/geckodriver.exe' driver_path = webdriver.Firefox(firefox_options=options,executable_path=executable_path)
3.phantomjs瀏覽器
設置ip
方法1:
service_args = [ '--proxy=%s' % ip_html, # 代理 IP:prot (eg:192.168.0.28:808) '--proxy-type=http’, # 代理類型:http/https ‘--load-images=no’, # 關閉圖片加載(可選) '--disk-cache=yes’, # 開啟緩存(可選) '--ignore-ssl-errors=true’ # 忽略https錯誤(可選) ] driver = webdriver.PhantomJS(service_args=service_args)
方法2:
browser=webdriver.PhantomJS(PATH_PHANTOMJS) # 利用DesiredCapabilities(代理設置)參數值,重新打開一個sessionId,我看意思就相當於瀏覽器清空緩存后,加上代理重新訪問一次url proxy=webdriver.Proxy() proxy.proxy_type=ProxyType.MANUAL proxy.http_proxy='1.9.171.51:800' # 將代理設置添加到webdriver.DesiredCapabilities.PHANTOMJS中 proxy.add_to_capabilities(webdriver.DesiredCapabilities.PHANTOMJS) browser.start_session(webdriver.DesiredCapabilities.PHANTOMJS) browser.get('http://1212.ip138.com/ic.asp') print('1: ',browser.session_id) print('2: ',browser.page_source) print('3: ',browser.get_cookies())
還原為系統代理:
# 還原為系統代理 proxy=webdriver.Proxy() proxy.proxy_type=ProxyType.DIRECT proxy.add_to_capabilities(webdriver.DesiredCapabilities.PHANTOMJS) browser.start_session(webdriver.DesiredCapabilities.PHANTOMJS) browser.get('http://1212.ip138.com/ic.asp')
import random,requests,json from selenium import webdriver from selenium.webdriver.common.desired_capabilities import DesiredCapabilities from selenium.webdriver.common.proxy import ProxyType #隨機獲取一個ip def proxies(): r = requests.get("http://120.26.166.214:9840/JProxy/update/proxy/scoreproxy") rr = json.loads(r.text) hh = rr['ip'] + ":" + "8907" print(hh) return hh ips =proxies() #設置phantomjs請求頭和代理方法一: #------------------------------------------------------------------------------------- # 設置代理 service_args = [ '--proxy=%s' % ips, # 代理 IP:prot (eg:192.168.0.28:808) '--ssl-protocol=any', #忽略ssl協議 '--load - images = no', # 關閉圖片加載(可選) '--disk-cache=yes', # 開啟緩存(可選) '--ignore-ssl-errors=true' # 忽略https錯誤(可選) ] #設置請求頭 user_agent = ( "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_8_4) " + "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/29.0.1547.57 Safari/537.36" ) dcap = dict(DesiredCapabilities.PHANTOMJS) dcap["phantomjs.page.settings.userAgent"] = user_agent driver = webdriver.PhantomJS(executable_path=r"C:\soft\phantomjs-2.1.1-windows\bin\phantomjs.exe", desired_capabilities=dcap,service_args=service_args) driver.get(url='http://www.baidu.com') page=driver.page_source print(page) #設置phantomjs請求頭和代理方法二: #------------------------------------------------------------------------------------- desired_capabilities = DesiredCapabilities.PHANTOMJS.copy() # 從USER_AGENTS列表中隨機選一個瀏覽器頭,偽裝瀏覽器 desired_capabilities["phantomjs.page.settings.userAgent"] = (random.choice('請求頭池')) # 不載入圖片,爬頁面速度會快很多 desired_capabilities["phantomjs.page.settings.loadImages"] = False # 利用DesiredCapabilities(代理設置)參數值,重新打開一個sessionId,我看意思就相當於瀏覽器清空緩存后,加上代理重新訪問一次url proxy = webdriver.Proxy() proxy.proxy_type = ProxyType.MANUAL proxy.http_proxy = random.choice('ip池') proxy.add_to_capabilities(desired_capabilities) phantomjs_driver = r'C:\phantomjs-2.1.1-windows\bin\phantomjs.exe' # 打開帶配置信息的phantomJS瀏覽器 driver = webdriver.PhantomJS(executable_path=phantomjs_driver,desired_capabilities=desired_capabilities) driver.start_session(desired_capabilities) driver.get(url='http://www.baidu.com') page=driver.page_source print(page) # 隱式等待5秒,可以自己調節 driver.implicitly_wait(5) # 設置10秒頁面超時返回,類似於requests.get()的timeout選項,driver.get()沒有timeout選項 # 以前遇到過driver.get(url)一直不返回,但也不報錯的問題,這時程序會卡住,設置超時選項能解決這個問題。 driver.set_page_load_timeout(20) # 設置10秒腳本超時時間 driver.set_script_timeout(20) #翻頁命令 driver.execute_script('window.scrollTo(0, document.body.scrollHeight)')