import re
import os
import requests
count = 0
for i in range(10):
url = f"http://www.xiaohuar.com/list-1-{count}.html"
response = requests.get(url)
data = response.text
result_list = re.findall('src="(.*?)" /></a>',data)
# print(type(result_list))
for result in result_list:
# print(result,type(result))
if not result.startswith('http'): # 取出
res = f"http://www.xiaohuar.com/{result}" # 拼接图片网址
print(res) # 打印拼接好的图片路径
img_response = requests.get(res) # 获取图片
img_name = res.split('/')[-1] # 文件名字
img_data = img_response.content #将图片转化为二进制
BASE_PATH = os.path.dirname(__file__)
img_path = os.path.join(BASE_PATH,'datas',f'{img_name}')
with open(img_path,'ab') as fw:
fw.write(img_data)
fw.flush()
count += 1
print(f'爬取了{count}页')
"""
http://www.xiaohuar.com/hua/
http://www.xiaohuar.com/list-1-1.html
http://www.xiaohuar.com/list-1-0.html
http://www.xiaohuar.com/list-1-1.html
http://www.xiaohuar.com/list-1-2.html
src="/d/file/20190726/small6880259bcb61b80ce246e497a448185c1564117785.jpg"
"""
原文地址:https://www.cnblogs.com/zuihoudebieli/p/11331768.html
时间: 2024-10-13 19:37:33