您好,登錄后才能下訂單哦!
scrapy.FormRequest
login.py
class LoginSpider(scrapy.Spider): name = 'login_spider' start_urls = ['http://www.login.com'] def parse(self, response): return [ scrapy.FormRequest.from_response( response, # username和password要根據(jù)實(shí)際頁面的表單的name字段進(jìn)行修改 formdata={'username': 'your_username', 'password': 'your_password'}, callback=self.after_login)] def after_login(self, response): # 登錄后的代碼 pass
selenium登錄獲取cookie
get_cookie_by_selenium.py
import pickle import time from selenium import webdriver def get_cookies(): url = 'https://www.test.com' web_driver = webdriver.Chrome() web_driver.get(url) username = web_driver.find_element_by_id('login-email') username.send_keys('username') password = web_driver.find_element_by_id('login-password') password.send_keys('password') login_button = web_driver.find_element_by_id('login-submit') login_button.click() time.sleep(3) cookies = web_driver.get_cookies() web_driver.close() return cookies if __name__ == '__main__': cookies = get_cookies() pickle.dump(cookies, open('cookies.pkl', 'wb'))
獲取瀏覽器cookie(以Ubuntu的Firefox為例)
get_cookie_by_firefox.py
import sqlite3 import pickle def get_cookie_by_firefox(): cookie_path = '/home/name/.mozilla/firefox/bqtvfe08.default/cookies.sqlite' with sqlite3.connect(cookie_path) as conn: sql = 'select name,value from moz_cookies where baseDomain="test.com"' cur = conn.cursor() cookies = [{'name': name, 'value': value} for name, value in cur.execute(sql).fetchall()] return cookies if __name__ == '__main__': cookies = get_cookie_from_firefox() pickle.dump(cookies, open('cookies.pkl', 'wb'))
scrapy使用獲取后的cookie
cookies = pickle.load(open('cookies.pkl', 'rb')) yield scrapy.Request(url, cookies=cookies, callback=self.parse)
requests使用獲取后的cookie
cookies = pickle.load(open('cookies.pkl', 'rb')) s = requests.Session() for cookie in cookies: s.cookies.set(cookie['name'], cookie['value'])
selenium使用獲取后的cookie
from selenium import webdriver cookies = pickle.load(open('cookies.pkl', 'rb')) w = webdriver.Chrome() # 直接添加cookie會報錯,下面是一種解決方案,可能有更好的 # -- start -- w.get('http://www.test.com') w.delete_all_cookies() # -- end -- for cookie in cookies: w.add_cookie(cookie)
以上這篇對python抓取需要登錄網(wǎng)站數(shù)據(jù)的方法詳解就是小編分享給大家的全部內(nèi)容了,希望能給大家一個參考,也希望大家多多支持億速云。
免責(zé)聲明:本站發(fā)布的內(nèi)容(圖片、視頻和文字)以原創(chuàng)、轉(zhuǎn)載和分享為主,文章觀點(diǎn)不代表本網(wǎng)站立場,如果涉及侵權(quán)請聯(lián)系站長郵箱:is@yisu.com進(jìn)行舉報,并提供相關(guān)證據(jù),一經(jīng)查實(shí),將立刻刪除涉嫌侵權(quán)內(nèi)容。