最美应用爬虫
- 作者: 扯淡的青春26357104
- 来源: 51数据库
- 2022-08-12
import requests
import re
url = "http://www.51sjk.com/Upload/Articles/1/0/320/320416_20220812154201459.com"
r = requests.get('http://www.51sjk.com/Upload/Articles/1/0/320/320416_20220812154201459.com/community/app/hot/?platform=2')
pattern = re.compile(r'<a class="community-app-cover-wrapper" href="(.*?)" target="_blank">')
urlList = pattern.findall(r.content)
def requestsUrl(url):
r = requests.get(url)
title = re.findall(r'"app-title"><h1>(.*?)</h1>',r.content)
#print title
category = re.findall(r'<a class="app-tag" href="/community/app/category/title/.*?/?platform=2">(.*?)</a>',r.content)
#print category
describe = re.findall(r'<div id="article_content">(.*?)<div class="community-image-wrapper">',r.content)
#print type(describe[0])
strdescribe = srtReplace(describe[0])
#print strdescribe
downloadUrl = re.findall(r'<a class="download-button direct hidden" href="(.*?)"',r.content)
#print downloadUrl
return title,category,strdescribe,downloadUrl
def srtReplace(string):
listReplace = ['<p>', '<br>', '<h1>', '<h2>', '<h3>', '<h4>', '<h5>', '<h6>', '<h7>','<strong>','</p>', '<br/>', '</h1>', '</h2>', '</h3>', '</h4>', '</h5>',
'</h6>', '</h7>','</strong>','<b>', '</b>']
for eachListReplace in listReplace:
string = string.replace(str(eachListReplace),'\n')
string = string.replace('\n\n','')
return string
def categornFinal(category):
categoryFinal =''
for eachCategory in category:
categoryFinal = categoryFinal+str(eachCategory)+'-->'
return categoryFinal
def urlReplace(url):
url = url.replace('&', '&')
return url
requestsUrl("http://www.51sjk.com/Upload/Articles/1/0/320/320416_20220812154201459.com/community/app/27369/?platform=2")
for eachUrl in urlList:
eachUrl = url+eachUrl
content = requestsUrl(eachUrl)
categoryFinal =''
title = content[0][0]
category = categornFinal(content[1])
strdescribe = content[2]
downloadUrl = urlReplace(content[3][0])
with open('c:/wqa.txt', 'a+') as fd:
fd.write('title:'+title+'\n'+'category:'+category+'\n'+'strdescribe:'+strdescribe+'\n'+'downloadUrl:'+downloadUrl+'\n\n\n-----------------------------------------------------------------------------------------------------------------------------\n\n\n')
推荐阅读
热点文章
Discord.py(重写)on_member_update 无法正常工作
0
Discord.py 在 vc 中获取用户分钟数
0
discord.py 重写 |为我的命令出错
0
Discord.py rewrite 如何 DM 命令?
0
播放音频时,最后一部分被切断.如何解决这个问题?(discord.py)
0
在消息删除消息 Discord.py
0
如何使 discord.py 机器人私人/直接消息不是作者的人?
0
(Discord.py) 如何获取整个嵌入内容?
0
Discord bot 尽管获得了许可,但不能提及所有人
0
Discord.py discord.NotFound 异常
0
