搜尋

首頁  >  問答  >  主體

爬蟲圖片 - 請教各位:python爬蟲編碼問題,版本3.6,win10 64位元下?

這是報錯訊息:

Traceback (most recent call last):
  File "D:\py\pic_downfrom2255ok.py", line 45, in <module>
    html = getHtml(url_all[i])
  File "D:\py\pic_downfrom2255ok.py", line 32, in getHtml
    html = response.read().decode()
UnicodeDecodeError: 'utf-8' codec can't decode byte 0xb3 in position 184: invalid start byte

改變了很多地方,主要可能是目標網站是gb2312編碼,
這個程式在別的網站是可以正常下載圖片的,換上現在的網站就有問題
還請各位多多指教,問題出在哪裡?試了幾個方法都不行
原始碼如下:

#coding=utf-8
import urllib.request
from urllib.request import urlopen, urlretrieve 
import urllib
import urllib.parse
import re
import os
from bs4 import BeautifulSoup


url_all =[
'http://www.shop2255.com/showpro/2603.html',
'http://www.shop2255.com/showpro/1558.html',
'http://www.shop2255.com/showpro/1564.html',
'http://www.shop2255.com/showpro/2411.html',
'http://www.shop2255.com/showpro/2409.html',
'http://www.shop2255.com/showpro/1561.html',
'http://www.shop2255.com/showpro/2414.html',
'http://www.shop2255.com/showpro/2609.html',
'http://www.shop2255.com/showpro/2413.html',
'http://www.shop2255.com/showpro/2604.html',
'http://www.shop2255.com/showpro/2605.html',
'http://www.shop2255.com/showpro/2606.html',
'http://www.shop2255.com/showpro/2608.html',
'http://www.shop2255.com/showpro/2607.html',
'http://www.shop2255.com/showpro/2610.html']

def getHtml(url):
    response = urlopen(url)
    html = response.read().decode("gbk")
    return html


def getImg(html):
    reg = 'src="(.+?\.jpg)"'
    imgre = re.compile(reg)
    imglist = re.findall(imgre,html)

    return imglist

for i in range(len(url_all)):
    html = getHtml(url_all[i])
    list=getImg(html.decode())
    x = 0
    for imgurl in list:
        print(x)
        file_path = url_all[i]
        (filepath,tempfilename) = os.path.split(file_path)
        (filename,extension) = os.path.splitext(tempfilename)
        
        if not os.path.exists('d:\%s' % filename):
            os.mkdir('d:\%s' % filename)
        # os.mkdir('D:\%s' % filename2)
        
        local=r'D:\%s\%s.jpg' % (filename,imgurl.splite("/")[-1])
        urllib.request.urlretrieve(imgurl,local)
        x+=1
print("done")
伊谢尔伦伊谢尔伦2753 天前977

全部回覆(2)我來回復

  • 天蓬老师

    天蓬老师2017-05-18 10:55:14

    雷雷

    回覆
    0
  • 漂亮男人

    漂亮男人2017-05-18 10:55:14

    首先在你這個程式碼裡面 local=r'D:%s%s.jpg' % (filename,imgurl.splite("/")[-1])split写成了splite.

    還有 urllib.request.urlretrieve(imgurl,local)这个imgurl不是一个合法的
    url,只是一个相对 url, 要改成绝对 url,需要加上 base_url = 'http://www.shop2255.com/'

    還有產生的檔案路徑好像也有問題.

    # -*- coding: utf-8 -*-
    
    import urllib.request
    from urllib.request import urlopen, urlretrieve
    import urllib
    import urllib.parse
    import re
    import os
    from bs4 import BeautifulSoup
    
    base_url = 'http://www.shop2255.com/'
    
    url_all =[
    'http://www.shop2255.com/showpro/2603.html',
    'http://www.shop2255.com/showpro/1558.html',
    'http://www.shop2255.com/showpro/1564.html',
    'http://www.shop2255.com/showpro/2411.html',
    'http://www.shop2255.com/showpro/2409.html',
    'http://www.shop2255.com/showpro/1561.html',
    'http://www.shop2255.com/showpro/2414.html',
    'http://www.shop2255.com/showpro/2609.html',
    'http://www.shop2255.com/showpro/2413.html',
    'http://www.shop2255.com/showpro/2604.html',
    'http://www.shop2255.com/showpro/2605.html',
    'http://www.shop2255.com/showpro/2606.html',
    'http://www.shop2255.com/showpro/2608.html',
    'http://www.shop2255.com/showpro/2607.html',
    'http://www.shop2255.com/showpro/2610.html']
    
    def getHtml(url):
        response = urlopen(url)
        # print(response.read())
        html = response.read().decode("gbk")
        print(html)
        return html
    
    
    def getImg(html):
        reg = 'src="(.+?\.jpg)"'
        imgre = re.compile(reg)
        imglist = re.findall(imgre, html)
        return imglist
    
    for i in range(len(url_all)):
        html = getHtml(url_all[i])
        # 注意: 我这里没有你那个错误,我只需要改这个就行了
        # list = getImg(html.decode())
        list = getImg(html)
        # print(list)
        x = 0
        for imgurl in list:
            print(x)
            file_path = url_all[i]
            (filepath, tempfilename) = os.path.split(file_path)
            (filename, extension) = os.path.splitext(tempfilename)
    
            if not os.path.exists('d:\%s' % filename):
                os.mkdir('d:\%s' % filename)
            # os.mkdir('D:\%s' % filename2)
    
            local = r'D:\%s\%s.jpg' % (filename, imgurl.split("/")[-1])
            try:
                urllib.request.urlretrieve(base_url + imgurl, local)
            except:
                print("can't retrieve the" + base_url + imgurl)
            x += 1
    
    print("done")
    

    回覆
    0
  • 取消回覆