对python抓取需要登录网站数据的方法详解
scrapy.FormRequest
login.py
class LoginSpider(scrapy.Spider): name = 'login_spider' start_urls = ['http://www.login.com'] def parse(self, response): return [ scrapy.FormRequest.from_response( response, # username和password要根据实际页面的表单的name字段进行修改 formdata={'username': 'your_username', 'password': 'your_password'}, callback=self.after_login)] def after_login(self, response): # 登录后的代码 pass
selenium登录获取cookie
get_cookie_by_selenium.py
import pickle import time from selenium import webdriver def get_cookies(): url = 'https://www.test.com' web_driver = webdriver.Chrome() web_driver.get(url) username = web_driver.find_element_by_id('login-email') username.send_keys('username') password = web_driver.find_element_by_id('login-password') password.send_keys('password') login_button = web_driver.find_element_by_id('login-submit') login_button.click() time.sleep(3) cookies = web_driver.get_cookies() web_driver.close() return cookies if __name__ == '__main__': cookies = get_cookies() pickle.dump(cookies, open('cookies.pkl', 'wb'))
获取浏览器cookie(以Ubuntu的Firefox为例)
get_cookie_by_firefox.py
import sqlite3 import pickle def get_cookie_by_firefox(): cookie_path = '/home/name/.mozilla/firefox/bqtvfe08.default/cookies.sqlite' with sqlite3.connect(cookie_path) as conn: sql = 'select name,value from moz_cookies where baseDomain="test.com"' cur = conn.cursor() cookies = [{'name': name, 'value': value} for name, value in cur.execute(sql).fetchall()] return cookies if __name__ == '__main__': cookies = get_cookie_from_firefox() pickle.dump(cookies, open('cookies.pkl', 'wb'))
scrapy使用获取后的cookie
cookies = pickle.load(open('cookies.pkl', 'rb')) yield scrapy.Request(url, cookies=cookies, callback=self.parse)
requests使用获取后的cookie
cookies = pickle.load(open('cookies.pkl', 'rb')) s = requests.Session() for cookie in cookies: s.cookies.set(cookie['name'], cookie['value'])
selenium使用获取后的cookie
from selenium import webdriver cookies = pickle.load(open('cookies.pkl', 'rb')) w = webdriver.Chrome() # 直接添加cookie会报错,下面是一种解决方案,可能有更好的 # -- start -- w.get('http://www.test.com') w.delete_all_cookies() # -- end -- for cookie in cookies: w.add_cookie(cookie)
相关推荐
houmenghu 2020-11-17
kentrl 2020-11-10
逍遥友 2020-10-26
jincheng 2020-09-01
Blueberry 2020-08-15
xclxcl 2020-08-03
zmzmmf 2020-08-03
阳光之吻 2020-08-03
PkJY 2020-07-08
hzyuhz 2020-07-04
89407707 2020-06-27
服务器端攻城师 2020-06-26
阳光岛主 2020-06-25
笨重的蜗牛 2020-06-20
xuanwenchao 2020-06-14
Lophole 2020-06-13
明瞳 2020-06-12
songerxing 2020-06-11