# -*- coding: utf-8 -*- """ @author: jiangfuqiang """ import re import urllib2 import cookielib import time def startParser(author,page=1): reg = r'<a href="/\w+/article/details/\d+">\s*\t*\n*\s*\t*\s*.*?\t*\n*\t*\s*</a>' cj = cookielib.LWPCookieJar() cookie_support = urllib2.HTTPCookieProcessor(cj) opener = urllib2.build_opener(cookie_support,urllib2.HTTPHandler) urllib2.install_opener(opener) headers = {'User-Agent' : 'Mozilla/5.0 (Windows NT 6.1; WOW64; rv:14.0) Gecko/20100101 Firefox/14.0.1', 'Referer' : ' http://my.csdn.net/my/favorite'} flag = True while flag == True: time.sleep(2) url = "http://blog.csdn.net/%s/article/list/%d"%(author,page) req = urllib2.Request(url,headers=headers) resp = urllib2.urlopen(req) data = resp.read() regex = re.compile(reg,flags=re.MULTILINE) result = regex.findall(data) for rd in result: print rd if len(result) < 20: flag = False page = page + 1 print 'success............page:%d'%page #print result.group() if __name__ == '__main__': startParser('yiyaaixuexi',1)
这篇python抓取收藏的文章链接和标题中有python发送邮件的代码,可以将此程序稍微改动之后将文章链接发送的邮箱以便以后查阅
使用python抓取CSDN关注人的所有发布的文章,布布扣,bubuko.com
时间: 2024-10-11 06:22:32