python爬虫用webdriver.get("url"),返回403 Forbidden

coding=utf-8

import requests
from selenium import webdriver
import time

class JzSpider:

def __init__(self,):

    self.start_url = "http://radar.itjuzi.com//company"
    self.headers={"User-Agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/64.0.3282.140 Safari/537.36",
                  "Accept":"Accept:text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8",
                  "Connection": "keep - alive",
                  "Accept-Encoding":"gzip, deflate, br"}

def parse_url(self,url):
    proxies = {"http": "http://117.127.0.204:8080"}
    response = requests.get(url, headers=self.headers)
    content = response.content.decode("utf-8")
    return content

def save_content_list(self,content):
    with open("Jz.txt", "w", encoding="utf-8") as f:
        f.write(content)
    print("保存成功")

def run(self):
    driver = webdriver.Chrome()
    # 用driver.get()请求这个网址,返回403,是ip被封了?要怎么设置代理ip或者其它解决方法
    driver.get("https://www.itjuzi.com/user/login?flag=radar&redirect=/company")

    driver.find_element_by_id("create_account_email").send_keys("13333331328")
    driver.find_element_by_id("create_account_password").send_keys("lz133333333334")
    time.sleep(8)
    driver.find_element_by_id("login_btn").click()
    html_str = self.parse_url(self.start_url)
    self.save_content_list(html_str)

if name == '__main__':

Jz_spider = JzSpider()
Jz_spider.run()
阅读 8.7k
1 个回答

403 Forbidden 错误,大多是被服务器屏蔽了,拒绝提供返回内容

一般可以通过更换服务器ip、设置代理服务器,去爬取

最好的办法,是通过模拟浏览器人工采集爬取

selenium + xvfb + firefox + proxy ip

下面是我的解决方案,仅供参考,相互学习

from selenium import webdriver
from selenium.webdriver.firefox.firefox_binary import FirefoxBinary
from selenium.webdriver.common.proxy import *
from pyvirtualdisplay import Display
# from xvfbwrapper import Xvfb

import bs4, os
from base64 import b64encode

import sys
reload(sys)
sys.setdefaultencoding('utf8')


## webdriver + firefox (不使用代理,爬取网页)
def spider_url_firefox(url):
    browser = None
    display = None
    try:
        display = Display(visible=0, size=(800, 600))
        display.start()
        browser = webdriver.Firefox()       # 打开 FireFox 浏览器
        browser.get(url)     
        content = browser.page_source
        print("content: " + str(content))
    finally:
        if browser: browser.quit()
        if display: display.stop()


## webdriver + firefox + proxy + whiteip (无密码,或白名单ip授权)
## 米扑代理:https://proxy.mimvp.com
def spider_url_firefox_by_whiteip(url):
    browser = None
    display = None
    
    ## 白名单ip,请见米扑代理会员中心: https://proxy.mimvp.com/usercenter/userinfo.php?p=whiteip
    mimvp_proxy = { 
                    'ip'            : '140.143.62.84',      # ip
                    'port_https'    : 19480,                # http, https
                    'port_socks'    : 19481,                # socks5
                    'username'      : 'mimvp-user',
                    'password'      : 'mimvp-pass'
                  }
    
    try:
        display = Display(visible=0, size=(800, 600))
        display.start()
        
        profile = webdriver.FirefoxProfile()
        
        # add proxy
        profile.set_preference('network.proxy.type', 1)     # ProxyType.MANUAL = 1
        if url.startswith("http://"):
            profile.set_preference('network.proxy.http', mimvp_proxy['ip'])
            profile.set_preference('network.proxy.http_port', mimvp_proxy['port_https'])    # 访问http网站
        elif url.startswith("https://"):
            profile.set_preference('network.proxy.ssl', mimvp_proxy['ip'])
            profile.set_preference('network.proxy.ssl_port', mimvp_proxy['port_https'])     # 访问https网站
        else:
            profile.set_preference('network.proxy.socks', mimvp_proxy['ip'])
            profile.set_preference('network.proxy.socks_port', mimvp_proxy['port_socks'])
            profile.set_preference('network.proxy.ftp', mimvp_proxy['ip'])
            profile.set_preference('network.proxy.ftp_port', mimvp_proxy['port_https'])
            profile.set_preference('network.proxy.no_proxies_on', 'localhost,127.0.0.1')
        
        ## 不存在此用法,不能这么设置用户名密码 (舍弃)
#         profile.set_preference("network.proxy.username", 'mimvp-user')
#         profile.set_preference("network.proxy.password", 'mimvp-pass')
    
        profile.update_preferences()
        
        browser = webdriver.Firefox(profile)       # 打开 FireFox 浏览器
        browser.get(url)     
        content = browser.page_source
        print("content: " + str(content))
    finally:
        if browser: browser.quit()
        if display: display.stop()
撰写回答
你尚未登录,登录后可以
  • 和开发者交流问题的细节
  • 关注并接收问题和回答的更新提醒
  • 参与内容的编辑和改进,让解决方法与时俱进
推荐问题