python爬虫爬取美女图片

最新推荐文章于 2024-05-30 14:27:04 发布

ZY5A59

最新推荐文章于 2024-05-30 14:27:04 发布

阅读量8.5k

点赞数 1

分类专栏： python 文章标签：爬虫 python

本文链接：https://blog.csdn.net/u013480667/article/details/44986047

版权

python 专栏收录该内容

7 篇文章 0 订阅

订阅专栏

python 爬虫爬取美女图片

#coding=utf-8

import urllib
import re
import os
import time
import threading

def getHtml(url):
    page = urllib.urlopen(url)
    html = page.read()
    return html


def getImgUrl(html,src):
    srcre = re.compile(src)
    srclist = re.findall(srcre,html)
    return srclist


def getImgPage(html):
    url = r'http://.*\.html'
    urlre = re.compile(url)
    urllist = re.findall(urlre,html)
    return urllist



def downloadImg(url):
    html = getHtml(url)
    src = r'rel=.*\.jpg'
    srclist = getImgUrl(html,src)
    srclist2 = []
    for srcs in srclist:
        temp = srcs.replace("'",'"')
        temp = temp.split('"')
        srclist2.append(temp[1])

    for srcurl in srclist2:
        imgName = srcurl.replace(':','_')
        imgName = imgName.replace('/','_')
        print 'download pic %s .........' % srcurl
        if os.path.isfile('pic/%s' % imgName):
            continue
        urllib.urlretrieve(srcurl,'pic/%s' % imgName)


class MyThread(threading.Thread):
    def __init__(self,urllist):
        threading.Thread.__init__(self)
        self.urllist = urllist

    def run(self):
        for u in self.urllist:
            downloadImg(u)


def main():
    url = 'http://www.6188.net/'
    html = getHtml(url)
    urllist = getImgPage(html)

    urllist2 = []

    length = len(urllist) / 7
    for i in range(1,8):
        temp = urllist[(i-1)*length:i*length]
        urllist2.append(temp)


    for u in urllist2:
        t = MyThread(u)
        t.start()


main()

ZY5A59

关注

1
点赞
踩
4

收藏

觉得还不错? 一键收藏
0
评论
python爬虫爬取美女图片

python 爬虫爬取美女图片#coding=utf-8import urllibimport reimport osimport timeimport threadingdef getHtml(url): page = urllib.urlopen(url) html = page.read() return htmldef getImg
复制链接

扫一扫