python爬虫亚马逊评论_亚马逊差爬虫+差评提醒+评论抓取

python爬虫亚马逊评论_亚马逊差爬虫+差评提醒+评论抓取

2023年6月27日发(作者:)

python爬⾍亚马逊评论_亚马逊差爬⾍+差评提醒+评论抓取# -*- coding:utf-8 -*-import urllib,urllib2,refrom lxml import etreeimport requestsimport randomimport jsonimport confimport time,datetimeimport os,('./')def randHeader():head_connection = ['Keep-Alive', 'close']head_accept = ['text/html, application/xhtml+xml, */*']head_accept_language = ['zh-CN,fr-FR;q=0.5', 'en-US,en;q=0.8,zh-Hans-CN;q=0.5,zh-Hans;q=0.3']head_user_agent = ['Mozilla/5.0 (Windows NT 6.3; WOW64; Trident/7.0; rv:11.0) like Gecko','Mozilla/5.0 (Windows NT 5.1) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/28.0.1500.95 Safari/537.36','Mozilla/5.0 (Windows NT 6.1; WOW64; Trident/7.0; SLCC2; .NET CLR 2.0.50727; .NET CLR 3.5.30729; .NET CLR3.0.30729; Media Center PC 6.0; .NET4.0C; rv:11.0) like Gecko)','Mozilla/5.0 (Windows; U; Windows NT 5.2) Gecko/2008070208 Firefox/3.0.1','Mozilla/5.0 (Windows; U; Windows NT 5.1) Gecko/20070309 Firefox/2.0.0.3','Mozilla/5.0 (Windows; U; Windows NT 5.1) Gecko/20070803 Firefox/1.5.0.12','Opera/9.27 (Windows NT 5.2; U; zh-cn)','Mozilla/5.0 (Macintosh; PPC Mac OS X; U; en) Opera 8.0','Opera/8.0 (Macintosh; PPC Mac OS X; U; en)','Mozilla/5.0 (Windows; U; Windows NT 5.1; en-US; rv:1.8.1.12) Gecko/20080219 Firefox/2.0.0.12 Navigator/9.0.0.6','Mozilla/4.0 (compatible; MSIE 8.0; Windows NT 6.1; Win64; x64; Trident/4.0)','Mozilla/4.0 (compatible; MSIE 8.0; Windows NT 6.1; Trident/4.0)','Mozilla/5.0 (compatible; MSIE 10.0; Windows NT 6.1; WOW64; Trident/6.0; SLCC2; .NET CLR 2.0.50727; .NET CLR3.5.30729; .NET CLR 3.0.30729; Media Center PC 6.0; InfoPath.2; .NET4.0C; .NET4.0E)','Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Maxthon/4.0.6.2000Chrome/26.0.1410.43 Safari/537.1 ','Mozilla/5.0 (compatible; MSIE 10.0; Windows NT 6.1; WOW64; Trident/6.0; SLCC2; .NET CLR 2.0.50727; .NET CLR3.5.30729; .NET CLR 3.0.30729; Media Center PC 6.0; InfoPath.2; .NET4.0C; .NET4.0E; QQBrowser/7.3.9825.400)','Mozilla/5.0 (Windows NT 6.1; WOW64; rv:21.0) Gecko/20100101 Firefox/21.0 ','Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/21.0.1180.92 Safari/537.1LBBROWSER','Mozilla/5.0 (compatible; MSIE 10.0; Windows NT 6.1; WOW64; Trident/6.0; BIDUBrowser 2.x)','Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/536.11 (KHTML, like Gecko) Chrome/20.0.1132.11 TaoBrowser/3.0Safari/536.11']header = {'Connection': head_connection[0],'Accept': head_accept[0],'Accept-Language': head_accept_language[1],'User-Agent': head_user_agent[nge(0, len(head_user_agent))]}return headerdef getUrl(goods,star):url = "/hz/reviews-render/ajax/reviews/get/ref=cm_cr_arp_d_viewopt_sr?+ goodssortBy=recent&reviewerType=all_reviews&formatType=&mediaType=&filterByStar="+star+"&pageNumber=1&filterByKeyword=&sreturn urldef getText(url):head_dict = randHeader()r = (url,headers = head_dict)content = turn contentif __name__ == '__main__':dt = ().strftime("%Y-%m-%d%H:%M:%S")#url = "/hz/reviews-render/ajax/reviews/get/ref=cm_cr_arp_d_viewopt_sr?+ goodssortBy=recent&reviewerType=all_reviews&formatType=&mediaType=&filterByStar=one_star&pageNumber=1&filterByKeyword=&sgoods_dict = _dictfor goods in goods_():for star in ['all_stars']:url = getUrl(goods, star)page = getText(url)if len(('&&&')) > 5:all_num = int(((('&&&')[1])[2]).xpath('//*[@class="a-size-base"]')[0].(' ')[3].replace(',',''))file_path = star +"/"+ goodscontent = 0num = 0if (file_path):with open(file_path, 'r') as f:content = int(())if all_num > content:num = all_num - contentif num > 0:for i in range(4, 4 + num):star_num = int(float(((('&&&')[i])[2]).xpath('//*[@class="a-icon-alt"]')[0].(' ')[0]))if star_num < 4:send_message = goods_dict[goods] + ':新增加⼀条'+str(star_num)+'星评论'cmd = 'sh my_ "' + send_message + '"'(cmd)with open(file_path, 'w+') as f:(str(all_num))else:with open(file_path, 'w+') as f:(str(all_num))else:with open(file_path, 'w+') as f:(str(all_num))

发布者:admin,转转请注明出处:http://www.yc00.com/news/1687865622a52036.html

相关推荐

发表回复

评论列表(0条)

  • 暂无评论

联系我们

400-800-8888

在线咨询: QQ交谈

邮件:admin@example.com

工作时间:周一至周五,9:30-18:30,节假日休息

关注微信