赤魂煌 发表于 2020-9-17 17:14

python3爬取看着不咳嗽的图片

# UTF-8
# author wangyeuwen

import requests
from bs4 import BeautifulSoup
import os

headers = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/85.0.4183.102 Safari/537.36 Edg/85.0.564.51',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.9',
    'Accept-Encoding': 'gzip, deflate'
}
path = input('输入图片保存路径:')
if not(os.path.exists(path)):
    os.mkdir(path) #创建文件夹爱
    print('路径不存在,已创建路径')

count = 25 #总页数
print('=========初始化成功,开始爬取=========')
for c in range(1,count+1):
    os.chdir(path)# 设置文件夹
    url = 'http://tp.wcctv.top/index.php/page/{0}/'.format(c)
    resphonse1 = requests.get(url, headers=headers)
    html_soup1 = BeautifulSoup(resphonse1.text, 'lxml')# 解析网页
    for i in range(0,6):
      img_dir = html_soup1.select('.item img').get('alt')
      os.chdir(path)# 设置文件夹
      os.mkdir(img_dir)
      os.chdir(path+'/'+img_dir)
      img_url = html_soup1.select('.item a').get('href') #获取当前页面的url
      resphonse2 = requests.get(img_url,headers=headers)
      html_soup2 = BeautifulSoup(resphonse2.text,'lxml')
      all_img = html_soup2.select('.post-item-img')
      for x in range(0,len(all_img)):
            img = requests.get(all_img.get('src'),headers=headers)
            f = open(all_img.get('title') + '.jpg', 'ab')
            f.write(img.content)
            print(all_img.get('title')+'下载成功!')
            f.close()

违心的牛 发表于 2020-9-17 17:14

6666666666

滑翔的小竹鼠 发表于 2020-9-17 17:17

谢谢大牛

menghuanxiyou 发表于 2020-9-17 17:26

好的,非常感谢

ljc 发表于 2020-9-17 17:43

谢谢分享

zb1314520 发表于 2020-9-17 17:59

6666

fashioned 发表于 2020-9-17 18:27

感谢楼主分享

ETS61 发表于 2020-9-17 20:29

感谢楼主分享!大牛有你更精彩!

as310739216 发表于 2020-9-17 21:14

谢谢大佬
页: [1]
查看完整版本: python3爬取看着不咳嗽的图片