您好,登錄后才能下訂單哦!
本篇文章給大家分享的是有關怎么在Python中使用Scrapy爬取網頁內容,小編覺得挺實用的,因此分享給大家學習,希望大家閱讀完這篇文章后可以有所收獲,話不多說,跟著小編一起來看看吧。
爬蟲主程序:
# -*- coding: utf-8 -*- import scrapy from scrapy.http import Request from zjf.FsmzItems import FsmzItem from scrapy.selector import Selector # 圈圈:情感生活 class MySpider(scrapy.Spider): #爬蟲名 name = "MySpider" #設定域名 allowed_domains = ["nvsheng.com"] #爬取地址 start_urls = [] #flag x = 0 #爬取方法 def parse(self, response): item = FsmzItem() sel = Selector(response) item['title'] = sel.xpath('//h2/text()').extract() item['text'] = sel.xpath('//*[@class="content"]/p/text()').extract() item['imags'] = sel.xpath('//div[@id="content"]/p/a/img/@src|//div[@id="content"]/p/img/@src').extract() if MySpider.x == 0: page_list = MySpider.getUrl(self,response) for page_single in page_list: yield Request(page_single) MySpider.x += 1 yield item #init: 動態傳入參數 #命令行傳參寫法: scrapy crawl MySpider -a start_url="http://some_url" def __init__(self,*args,**kwargs): super(MySpider,self).__init__(*args,**kwargs) self.start_urls = [kwargs.get('start_url')] def getUrl(self, response): url_list = [] select = Selector(response) page_list_tmp = select.xpath('//div[@class="viewnewpages"]/a[not(@class="next")]/@href').extract() for page_tmp in page_list_tmp: if page_tmp not in url_list: url_list.append("http://www.nvsheng.com/emotion/px/" + page_tmp) return url_list
PipeLines類
# -*- coding: utf-8 -*- # Define your item pipelines here # # Don't forget to add your pipeline to the ITEM_PIPELINES setting # See: http://doc.scrapy.org/en/latest/topics/item-pipeline.html from zjf import settings import json,os,re,random import urllib.request import requests, json from requests_toolbelt.multipart.encoder import MultipartEncoder class MyPipeline(object): flag = 1 post_title = '' post_text = [] post_text_imageUrl_list = [] cs = [] user_id= '' def __init__(self): MyPipeline.user_id = MyPipeline.getRandomUser('37619,18441390,18441391') #process the data def process_item(self, item, spider): #獲取隨機user_id,模擬發帖 user_id = MyPipeline.user_id #獲取正文text_str_tmp text = item['text'] text_str_tmp = "" for str in text: text_str_tmp = text_str_tmp + str # print(text_str_tmp) #獲取標題 if MyPipeline.flag == 1: title = item['title'] MyPipeline.post_title = MyPipeline.post_title + title[0] #保存并上傳圖片 text_insert_pic = '' text_insert_pic_w = '' text_insert_pic_h = '' for imag_url in item['imags']: img_name = imag_url.replace('/','').replace('.','').replace('|','').replace(':','') pic_dir = settings.IMAGES_STORE + '%s.jpg' %(img_name) urllib.request.urlretrieve(imag_url,pic_dir) #圖片上傳,返回json upload_img_result = MyPipeline.uploadImage(pic_dir,'image/jpeg') #獲取json中保存圖片路徑 text_insert_pic = upload_img_result['result']['image_url'] text_insert_pic_w = upload_img_result['result']['w'] text_insert_pic_h = upload_img_result['result']['h'] #拼接json if MyPipeline.flag == 1: cs_json = {"c":text_str_tmp,"i":"","w":text_insert_pic_w,"h":text_insert_pic_h} else: cs_json = {"c":text_str_tmp,"i":text_insert_pic,"w":text_insert_pic_w,"h":text_insert_pic_h} MyPipeline.cs.append(cs_json) MyPipeline.flag += 1 return item #spider開啟時被調用 def open_spider(self,spider): pass #sipder 關閉時被調用 def close_spider(self,spider): strcs = json.dumps(MyPipeline.cs) jsonData = {"apisign":"99ea3eda4b45549162c4a741d58baa60","user_id":MyPipeline.user_id,"gid":30,"t":MyPipeline.post_title,"cs":strcs} MyPipeline.uploadPost(jsonData) #上傳圖片 def uploadImage(img_path,content_type): "uploadImage functions" #UPLOAD_IMG_URL = "http://api.qa.douguo.net/robot/uploadpostimage" UPLOAD_IMG_URL = "http://api.douguo.net/robot/uploadpostimage" # 傳圖片 #imgPath = 'D:\pics\http___img_nvsheng_com_uploads_allimg_170119_18-1f1191g440_jpg.jpg' m = MultipartEncoder( # fields={'user_id': '192323', # 'images': ('filename', open(imgPath, 'rb'), 'image/JPEG')} fields={'user_id': MyPipeline.user_id, 'apisign':'99ea3eda4b45549162c4a741d58baa60', 'image': ('filename', open(img_path , 'rb'),'image/jpeg')} ) r = requests.post(UPLOAD_IMG_URL,data=m,headers={'Content-Type': m.content_type}) return r.json() def uploadPost(jsonData): CREATE_POST_URL = http://api.douguo.net/robot/uploadimagespost
reqPost = requests.post(CREATE_POST_URL,data=jsonData)
def getRandomUser(userStr): user_list = [] user_chooesd = '' for user_id in str(userStr).split(','): user_list.append(user_id) userId_idx = random.randint(1,len(user_list)) user_chooesd = user_list[userId_idx-1] return user_chooesd
字段保存Items類
# -*- coding: utf-8 -*- # Define here the models for your scraped items # # See documentation in: # http://doc.scrapy.org/en/latest/topics/items.html import scrapy class FsmzItem(scrapy.Item): # define the fields for your item here like: # name = scrapy.Field() title = scrapy.Field() #tutor = scrapy.Field() #strongText = scrapy.Field() text = scrapy.Field() imags = scrapy.Field()
在命令行里鍵入
scrapy crawl MySpider -a start_url=www.aaa.com
以上就是怎么在Python中使用Scrapy爬取網頁內容,小編相信有部分知識點可能是我們日常工作會見到或用到的。希望你能通過這篇文章學到更多知識。更多詳情敬請關注億速云行業資訊頻道。
免責聲明:本站發布的內容(圖片、視頻和文字)以原創、轉載和分享為主,文章觀點不代表本網站立場,如果涉及侵權請聯系站長郵箱:is@yisu.com進行舉報,并提供相關證據,一經查實,將立刻刪除涉嫌侵權內容。