功能描述
- 目標(biāo):獲取淘寶搜索頁面的信息,提取其中的關(guān)鍵信息章办。
- 理解:
淘寶的搜索接口
翻頁的處理
- 技術(shù)路線:
requests-re
(使用正則表達(dá)式來提取信息)
import requests
import re
def getHTMLText(url):
kv = {'cookie' : 'miid=413241011674319298; cna=bzhfE71XryECATo6NDSjkj6C; hng=CN%7Czh-CN%7CCNY%7C156; thw=cn; tracknick=%5Cu9759%5Cu5019%5Cu7075%5Cu5F52520370; tg=0; enc=yJqBSdMue2TKATKCxQ5nNmbUHL82jBeGpw%2BhqDIRlcL%2FVHFFQQ9sFshL4l8YJoS%2F5g%2F%2BV2gV3NoT9YYn30kbnQ%3D%3D; _m_h5_tk=b1f75d048d6aeed57d814f7a50d66641_1586700106396; _m_h5_tk_enc=0bdf063ab77a2936ccb2e059d6db64c0; t=8e80b386bca784fb223e0f83f73cebda; v=0; cookie2=1847beaf312bd87963ff8272d0b8d0b7; _tb_token_=5e7353571ad1e; alitrackid=www.taobao.com; _samesite_flag_=true; sgcookie=EKJ4L5czU7sbync8UgcsK; unb=2355217709; uc3=lg2=U%2BGCWk%2F75gdr5Q%3D%3D&id2=UUtO%2FVqSVpj5eg%3D%3D&nk2=3Wy5AkoyvJZ%2BX5%2FSa5Y%3D&vt3=F8dBxdGLZSg9H8AFy1U%3D; csg=5b481f1c; lgc=%5Cu9759%5Cu5019%5Cu7075%5Cu5F52520370; cookie17=UUtO%2FVqSVpj5eg%3D%3D; dnk=%5Cu9759%5Cu5019%5Cu7075%5Cu5F52520370; skt=f9711d0519a0fdcd; existShop=MTU4NzE3NzM4OA%3D%3D; uc4=id4=0%40U2l0uKuhTOtMhU1bieJbpIsIPhXN&nk4=0%403%2BcBjITjn0CCuBQVdwk4JLVExOAMNgh%2FfA%3D%3D; _cc_=U%2BGCWk%2F7og%3D%3D; _l_g_=Ug%3D%3D; sg=092; _nk_=%5Cu9759%5Cu5019%5Cu7075%5Cu5F52520370; cookie1=V33DIdASji0W00xDbJVy0fvJaNbu6aAZ2v0LJi537Ss%3D; JSESSIONID=531AFDD871D866503438DC04A71EB67E; lastalitrackid=login.taobao.com; tfstk=cQ_CByvhtvDCt6bE7TNwYOO1swKPawr66k9NOGQhIF1rwAfeBsjmuKvLcgDsWnd1.; uc1=cookie16=VFC%2FuZ9az08KUQ56dCrZDlbNdA%3D%3D&cookie21=URm48syIYB3rzvI4Dim4&cookie15=VT5L2FSpMGV7TQ%3D%3D&existShop=false&pas=0&cookie14=UoTUPc3rdfGOww%3D%3D; mt=ci=45_1; isg=BA8PUHFyjZyKVomRPVnYAHqfnqMZNGNWFeQ1eyEcQn6L8C7yKASepoeq8iDO7DvO; l=eBPjk4lnQH8FVhdQBO5wlJ4FzbbTCIRb8uPr-CjsIIHca1oCtebG9NQcQeZDSdtjgtCXyeKPFFZ7gRB4r34dTxDDBexrCyConxvO.',
'user-agent' : 'Mozilla/5.0'}
try:
r = requests.get(url, headers = kv, timeout=30)
r.raise_for_status()
r.encoding = r.apparent_encoding
return r.text
except:
return ""
#關(guān)鍵代碼區(qū):用正則表達(dá)式實(shí)現(xiàn)所需信息的提取
def parsePage(ilt, html):
try:
#這些正則表達(dá)式都沒有加轉(zhuǎn)義字符‘\’
plt = re.findall(r'"view_price":"[\d.]*"', html) #找到所有的價(jià)格信息
tlt = re.findall(r'"raw_title":".*?"', html)
flt = re.findall(r'"view_fee":"[\d.]*"', html)
slt = re.findall(r'"view_sales":".*?"', html)
#這一步完成趴久,提取出來的信息都是鍵值對形式廷臼?蝗柔??(如:"view_price":"127.00")情龄,所以要做進(jìn)一步的處理迄汛,只取出值
for i in range(len(plt)):
# eval()函數(shù)
price = eval(plt[i].split(':')[1])
title = eval(tlt[i].split(':')[1])
fee = eval(flt[i].split(':')[1])
sales = eval(slt[i].split(':')[1])
#將列表元素([price, fee, sales, title]),存入ilt列表中
ilt.append([price, fee, sales, title])
except:
print("")
#打印爬取到的信息骤视,完成輸出格式的定義
def printGoodsList(ilt):
tplt = "{:4}\t{:8}\t{:8}\t{:10}\t{:16}"
print(tplt.format("序號(hào)", "價(jià)格", "郵費(fèi)", "已購人數(shù)", "商品名稱"))
count = 0
for g in ilt:
count = count + 1
print(tplt.format(count, g[0], g[1], g[2], g[3]))
def main():
goods = '子時(shí)當(dāng)歸同款茶壺' #淘寶檢索關(guān)鍵字
depth = 2 #需求頁數(shù)
start_url = 'https://s.taobao.com/search?q=' + goods
infoList = [] #定義一個(gè)列表鞍爱,保存爬取回來的數(shù)據(jù)
for i in range(depth):
try:
url = start_url + '&s=' + str(44 * i)
html = getHTMLText(url)
parsePage(infoList, html)
except:
continue
printGoodsList(infoList)
main()
- 當(dāng)爬不出數(shù)據(jù),并且程序并無報(bào)錯(cuò)時(shí)专酗,查看一下網(wǎng)頁的提交數(shù)據(jù)(headers)是不是包含cookie和user-agent睹逃。
- 正則表達(dá)式的使用一定要規(guī)范。
- 翻頁的實(shí)現(xiàn)主要是靠:觀察網(wǎng)頁url幾頁之間參數(shù)的不同笼裳,合理假設(shè)大膽判斷唯卖。