1.python爬蟲瀏覽器偽裝
#導入urllib.request模塊import urllib.request#設置請求頭headers=("User-Agent","Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/49.0.2623.221 Safari/537.36 SE 2.X MetaSr 1.0")#創建一個openeropener=urllib.request.build_opener()#將headers添加到opener中opener.addheaders=[headers]#將opener安裝為全局urllib.request.install_opener(opener)#用urlopen打開網頁data=urllib.request.urlopen(url).read().decode('utf-8','ignore')2.設置代理
#定義代理ipproxy_addr="122.241.72.191:808"#設置代理proxy=urllib.request.ProxyHandle({'http':proxy_addr})#創建一個openeropener=urllib.request.build_opener(proxy,urllib.request.HTTPHandle)#將opener安裝為全局urllib.request.install_opener(opener)#用urlopen打開網頁data=urllib.request.urlopen(url).read().decode('utf-8','ignore')3.同時設置用代理和模擬瀏覽器訪問
#定義代理ipproxy_addr="122.241.72.191:808"#創建一個請求req=urllib.request.Request(url)#添加headersreq.add_header("User-Agent","Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko)#設置代理proxy=urllib.request.ProxyHandle("http":proxy_addr)#創建一個openeropener=urllib.request.build_opener(proxy,urllib.request.HTTPHandle)#將opener安裝為全局urllib.request.install_opener(opener)#用urlopen打開網頁data=urllib.request.urlopen(req).read().decode('utf-8','ignore')4.在請求頭中添加多個信息
import urllib.requestpage_headers={"User-Agent":"Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/49.0.2623.221 Safari/537.36 SE 2.X MetaSr 1.0", "Host":"www.baidu.com", "Cookie":"xxxxxxxx" }req=urllib.request.Request(url,headers=page_headers)data=urllib.request.urlopen(req).read().decode('utf-8','ignore')5.添加post請求參數
import urllib.requestimport urllib.parse#設置post參數page_data=urllib.parse.urlencode([ ('pn',page_num), ('kd',keywords) ])#設置headerspage_headers={ 'User-Agent':'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/49.0.2623.221 Safari/537.36 SE 2.X MetaSr 1.0', 'Connection':'keep-alive', 'Host':'www.lagou.com', 'Origin':'https://www.lagou.com', 'Cookie':'JSESSIONID=ABAAABAABEEAAJA8F28C00A88DC4D771796BB5C6FFA2DDA; user_trace_token=20170715131136-d58c1f22f6434e9992fc0b35819a572b', 'Accept':'application/json, text/javascript, */*; q=0.01', 'Content-Type':'application/x-www-form-urlencoded; charset=UTF-8', 'Referer':'https://www.lagou.com/jobs/list_%E6%95%B0%E6%8D%AE%E6%8C%96%E6%8E%98?labelWords=&fromSearch=true&suginput=', 'X-Anit-Forge-Token':'None', 'X-Requested-With':'XMLHttpRequest' }#打開網頁req=urllib.request.Request(url,headers=page_headers)data=urllib.request.urlopen(req,data=page_data.encode('utf-8')).read().decode('utf-8')
新聞熱點
疑難解答