python爬虫爬取美女图片

时间：2015-04-11 08:59:23 阅读：322 评论：0 收藏：0 [点我收藏+]

python 爬虫爬取美女图片

#coding=utf-8

import urllib
import re
import os
import time
import threading

def getHtml(url):
    page = urllib.urlopen(url)
    html = page.read()
    return html


def getImgUrl(html,src):
    srcre = re.compile(src)
    srclist = re.findall(srcre,html)
    return srclist


def getImgPage(html):
    url = r'http://.*\.html'
    urlre = re.compile(url)
    urllist = re.findall(urlre,html)
    return urllist



def downloadImg(url):
    html = getHtml(url)
    src = r'rel=.*\.jpg'
    srclist = getImgUrl(html,src)
    srclist2 = []
    for srcs in srclist:
        temp = srcs.replace("'",'"')
        temp = temp.split('"')
        srclist2.append(temp[1])

    for srcurl in srclist2:
        imgName = srcurl.replace(':','_')
        imgName = imgName.replace('/','_')
        print 'download pic %s .........' % srcurl
        if os.path.isfile('pic/%s' % imgName):
            continue
        urllib.urlretrieve(srcurl,'pic/%s' % imgName)


class MyThread(threading.Thread):
    def __init__(self,urllist):
        threading.Thread.__init__(self)
        self.urllist = urllist

    def run(self):
        for u in self.urllist:
            downloadImg(u)


def main():
    url = 'http://www.6188.net/'
    html = getHtml(url)
    urllist = getImgPage(html)

    urllist2 = []

    length = len(urllist) / 7
    for i in range(1,8):
        temp = urllist[(i-1)*length:i*length]
        urllist2.append(temp)


    for u in urllist2:
        t = MyThread(u)
        t.start()


main()

python爬虫爬取美女图片

原文：http://blog.csdn.net/u013480667/article/details/44986047

踩

(0)

评论一句话评论（0）

分享档案

更多>

2021年09月23日 (328)
2021年09月24日 (313)
2021年09月17日 (191)
2021年09月15日 (369)
2021年09月16日 (411)
2021年09月13日 (439)
2021年09月11日 (398)
2021年09月12日 (393)
2021年09月10日 (160)
2021年09月08日 (222)