本文主要是介绍51job爬虫的一些问题,希望对大家解决编程问题提供一定的参考价值,需要的开发者们随着小编来一起学习吧!
各位大佬,我想爬取51job的求职详细信息,具体要求是进入官网搜索数字化管理,进到搜索结果页
获取里面的职位名称,薪资等信息,再点击进入每个职位的详细页面,获取里面的岗位职责等具体要求,
我写了一个代码,可以爬取第一个搜索页的 职位名称和薪资等信息,但进到详细页的时候 无法获取详细页的岗位职责等具体信息。
报错信息如下:
import requests
import json
import pandas as pd
from bs4 import BeautifulSoup# 创建空列表,用于存储数据
post_list = []# 使用一个列表来收集所有的文字内容
texts = []for page in range(1, 51):# 设置请求urlpost_url = f'https://we.51job.com/api/job/search-pc?api_key=51job×tamp=1696320374&keyword=%E6%95%B0%E5%AD%97%E5%8C%96%E7%AE%A1%E7%90%86&searchType=2&function=&industry=&jobArea=000000&jobArea2=&landmark=&metro=&salary=&workYear=°ree=&companyType=&companySize=&jobType=&issueDate=&sortType=0&pageNum={page}&requestId=&pageSize=20&source=1&accountId=&pageCode=sou%7Csou%7Csoulb&u_atoken=2ca6e1ae-fa70-41cd-9e38-a08a9e95fb76&u_asession=012t7pRyhGHOghJrYjhVbiG_PenymPB204JRWRSfcWGwvAwkrxQ73PCUt4w2x1lHHznKe-6CQQK6f87md9nMUZctsq8AL43dpOnCClYrgFm6o&u_asig=05k1bKQDkEecjNCG4spv3X0zAV1aEpt0GCCadwzkQrqKTfF1rOoApB5y4ZzMCcE-eWFH6J5a6axjxBy5YYl2LbmqejDiWCeGgyk2CvT8POaE6b0Q6BWSrXgqfhVZ4zHNT20-DqeciYycsNAPQsoqAeMww9q1Sbz5kFQVmRJDgEk_2Rjoy8b6Iyz3ADPMhizAKlksmHjM0JOodanL5-M1Qs1esPktKn_g6n51y0G4V_gzb5l1Tkmp4k4Q-65Vosvt9CJ-JJElOzXY6Opa0trnxsxeg4cbkiKv5-ovPTXDpGPGfY94r_LXIIil3Y3aVPRGAe&u_aref=9x2J%2FTRy3QJabllhhmNJLuq7vyM%3D'# 设置请求头headers = {'Accept': 'application/json, text/plain, */*','Accept-Encoding': 'gzip, deflate, br','Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8,en-GB;q=0.7,en-US;q=0.6,zh-TW;q=0.5','Connection': 'keep-alive','Cookie': "nsearch=jobarea%3D%26%7C%26ord_field%3D%26%7C%26recentSearch0%3D%26%7C%26recentSearch1%3D%26%7C%26recentSearch2%3D%26%7C%26recentSearch3%3D%26%7C%26recentSearch4%3D%26%7C%26collapse_expansion%3D; search=jobarea%7E%60%7C%21recentSearch0%7E%60000000%A1%FB%A1%FA000000%A1%FB%A1%FA0000%A1%FB%A1%FA00%A1%FB%A1%FA99%A1%FB%A1%FA%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA9%A1%FB%A1%FA99%A1%FB%A1%FA%A1%FB%A1%FA0%A1%FB%A1%FA%CA%FD%D7%D6%BB%AF%B9%DC%C0%ED%A1%FB%A1%FA2%A1%FB%A1%FA1%7C%21recentSearch1%7E%60030800%A1%FB%A1%FA030831%A1%FB%A1%FA0000%A1%FB%A1%FA00%A1%FB%A1%FA99%A1%FB%A1%FA%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA9%A1%FB%A1%FA99%A1%FB%A1%FA%A1%FB%A1%FA0%A1%FB%A1%FA%CA%FD%BE%DD%B7%D6%CE%F6%A1%FB%A1%FA2%A1%FB%A1%FA1%7C%21; guid=905277a987104111f0c174845049ec13; slife=lowbrowser%3Dnot%26%7C%26; privacy=1696725594; acw_tc=ac11000116967255980824819e00e15bc47e8c5565cf103e1376789d6cd3b6; sensorsdata2015jssdkcross=%7B%22distinct_id%22%3A%22905277a987104111f0c174845049ec13%22%2C%22first_id%22%3A%2218aef5ba1abb18-0093b34e2b677418-78505775-1338645-18aef5ba1ac28c%22%2C%22props%22%3A%7B%22%24latest_traffic_source_type%22%3A%22%E8%87%AA%E7%84%B6%E6%90%9C%E7%B4%A2%E6%B5%81%E9%87%8F%22%2C%22%24latest_search_keyword%22%3A%22%E6%9C%AA%E5%8F%96%E5%88%B0%E5%80%BC%22%2C%22%24latest_referrer%22%3A%22https%3A%2F%2Fcn.bing.com%2F%22%7D%2C%22identities%22%3A%22eyIkaWRlbnRpdHlfY29va2llX2lkIjoiMThhZWY1YmExYWJiMTgtMDA5M2IzNGUyYjY3NzQxOC03ODUwNTc3NS0xMzM4NjQ1LTE4YWVmNWJhMWFjMjhjIiwiJGlkZW50aXR5X2xvZ2luX2lkIjoiOTA1Mjc3YTk4NzEwNDExMWYwYzE3NDg0NTA0OWVjMTMifQ%3D%3D%22%2C%22history_login_id%22%3A%7B%22name%22%3A%22%24identity_login_id%22%2C%22value%22%3A%22905277a987104111f0c174845049ec13%22%7D%2C%22%24device_id%22%3A%2218aef5ba1abb18-0093b34e2b677418-78505775-1338645-18aef5ba1ac28c%22%7D; acw_sc__v2=6521fa61c4101175d8ceecf8bac489ed20327812; JSESSIONID=057B6308069316097F90AD9FD9295C45; ssxmod_itna=Qq+hY5AICDO3GHD8Yi40IuPCTd1xGxDv6mfKdD/YbGDnqD=GFDK40oYOoriAKiNw9m3xAxFmCOifqjFro2f+EDrc4eDHxY=DUgD+YYD4RKGwD0eG+DD4DWDmeHDnxAQDjxGpc2LkX=DEDYpZDit6D7tD5eDjAk6RYeG0DDt7n4G2tC=DYbd4EMbgK7TqFqDMteGXbYaiFkbaOPeZ9CWq4QpRDB=1xBQMCQUdx0PyBMUDZr+W8i4YSEh0++G4egDoQS4vk0Gd8BGjZoeo90Y4Dg+LSE4iDD396iD=; ssxmod_itna2=Qq+hY5AICDO3GHD8Yi40IuPCTd1xGxDv6mfKG9i=9DBwAg47P40q8+=WG2DkxFqG7GeD",'Dnt': '1','From-Domain': '51job_web','Host': 'we.51job.com','Property': "%7B%22partner%22%3A%22%22%2C%22webId%22%3A2%2C%22fromdomain%22%3A%2251job_web%22%2C%22frompageUrl%22%3A%22https%3A%2F%2Fwe.51job.com%2F%22%2C%22pageUrl%22%3A%22https%3A%2F%2Fwe.51job.com%2Fpc%2Fsearch%3Fkeyword%3D%25E6%2595%25B0%25E5%25AD%2597%25E5%258C%2596%25E7%25AE%25A1%25E7%2590%2586%26searchType%3D2%26sortType%3D0%26metro%3D%22%2C%22identityType%22%3A%22%22%2C%22userType%22%3A%22%22%2C%22isLogin%22%3A%22%E5%90%A6%22%2C%22accountid%22%3A%22%22%2C%22keywordType%22%3A%22%22%7D",'Referer': "https://we.51job.com/pc/search?keyword=%E6%95%B0%E5%AD%97%E5%8C%96%E7%AE%A1%E7%90%86&searchType=2&sortType=0&metro=",'Sec-Ch-Ua': '"Microsoft Edge";v="117", "Not;A=Brand";v="8", "Chromium";v="117"','Sec-Ch-Ua-Mobile': '?0','Sec-Ch-Ua-Platform': '"Windows"','Sec-Fetch-Dest': 'empty','Sec-Fetch-Mode': 'cors','Sec-Fetch-Site': 'same-origin','Sec-Gpc': '1','Partner': '','Sign': '0b6e7a413562570a7b9ceae46ce837e6b9956b3584d4916760031ddec36d36cf','User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/117.0.0.0 Safari/537.36 Edg/117.0.2045.47','User-Token': "",'Uuid': 'd37f9ef2faf49c91ca3b8e9137b95e08'}# 发送请求post_res = requests.get(url=post_url, headers=headers)# 获取响应数据post_data = post_res.json()for item in post_data['resultbody']['job']['items']:# 将数据存储到新字典中post_dict = {'职位名称': item['jobName'],'薪资': item['provideSalaryString'],'公司名称': item['fullCompanyName'],'工作能力要求': item['jobTags'],'工作详细要求链接': item['jobHref'],}# 进入详情页抓取职位具体信息detail_url = 'https://jobs.51job.com/' + item['hrefAreaPinYin'] + '/' + item['jobId'] + '.html'head = {'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7','Accept-Encoding': 'gzip, deflate, br','Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8,en-GB;q=0.7,en-US;q=0.6,zh-TW;q=0.5','Cache-Control': 'max-age=0','Connection': 'keep-alive','Cookie': 'nsearch=jobarea%3D%26%7C%26ord_field%3D%26%7C%26recentSearch0%3D%26%7C%26recentSearch1%3D%26%7C%26recentSearch2%3D%26%7C%26recentSearch3%3D%26%7C%26recentSearch4%3D%26%7C%26collapse_expansion%3D; search=jobarea%7E%60%7C%21recentSearch0%7E%60000000%A1%FB%A1%FA000000%A1%FB%A1%FA0000%A1%FB%A1%FA00%A1%FB%A1%FA99%A1%FB%A1%FA%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA9%A1%FB%A1%FA99%A1%FB%A1%FA%A1%FB%A1%FA0%A1%FB%A1%FA%CA%FD%D7%D6%BB%AF%B9%DC%C0%ED%A1%FB%A1%FA2%A1%FB%A1%FA1%7C%21recentSearch1%7E%60030800%A1%FB%A1%FA030831%A1%FB%A1%FA0000%A1%FB%A1%FA00%A1%FB%A1%FA99%A1%FB%A1%FA%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA99%A1%FB%A1%FA9%A1%FB%A1%FA99%A1%FB%A1%FA%A1%FB%A1%FA0%A1%FB%A1%FA%CA%FD%BE%DD%B7%D6%CE%F6%A1%FB%A1%FA2%A1%FB%A1%FA1%7C%21; guid=905277a987104111f0c174845049ec13; slife=lowbrowser%3Dnot%26%7C%26; privacy=1696725594; sensorsdata2015jssdkcross=%7B%22distinct_id%22%3A%22905277a987104111f0c174845049ec13%22%2C%22first_id%22%3A%2218aef5ba1abb18-0093b34e2b677418-78505775-1338645-18aef5ba1ac28c%22%2C%22props%22%3A%7B%22%24latest_traffic_source_type%22%3A%22%E8%87%AA%E7%84%B6%E6%90%9C%E7%B4%A2%E6%B5%81%E9%87%8F%22%2C%22%24latest_search_keyword%22%3A%22%E6%9C%AA%E5%8F%96%E5%88%B0%E5%80%BC%22%2C%22%24latest_referrer%22%3A%22https%3A%2F%2Fcn.bing.com%2F%22%7D%2C%22identities%22%3A%22eyIkaWRlbnRpdHlfY29va2llX2lkIjoiMThhZWY1YmExYWJiMTgtMDA5M2IzNGUyYjY3NzQxOC03ODUwNTc3NS0xMzM4NjQ1LTE4YWVmNWJhMWFjMjhjIiwiJGlkZW50aXR5X2xvZ2luX2lkIjoiOTA1Mjc3YTk4NzEwNDExMWYwYzE3NDg0NTA0OWVjMTMifQ%3D%3D%22%2C%22history_login_id%22%3A%7B%22name%22%3A%22%24identity_login_id%22%2C%22value%22%3A%22905277a987104111f0c174845049ec13%22%7D%2C%22%24device_id%22%3A%2218aef5ba1abb18-0093b34e2b677418-78505775-1338645-18aef5ba1ac28c%22%7D; acw_sc__v2=6521fa61c4101175d8ceecf8bac489ed20327812; acw_tc=ac11000116967256843172336e00dfce5b8080ea2ff7bd0666eebafa0f2811; ssxmod_itna=eqGxuDRD9GDtzqBPGIrCADgADyGDcAniGQQ1rY=D0y0HeGzDAxn40iDtPrN=KhiST3tLe8BRePKFp2aL3sF2oub4LYCo4GLDmKDyber4GG0xBYDQxAYDGDDPDogPD1D3qDkD7h6CMy1qGWDm4sDY5HDQHGe4DFc2IOP4i7DDvQCx07YRKDe2ahchgCK0gDq0KD9hYDsZ2fe0KpjS3xM2oEI3B3Ix0kl40OyZAC74GdXc5z4/ueLG0o7gier=GxqniKTCG4vW4KYm29VADxT02xPAiF7OxxDDax65hx4D; ssxmod_itna2=eqGxuDRD9GDtzqBPGIrCADgADyGDcAniGQQ1rnDnI8KxDsaKDLG0=aH/+rV=N0r2d4uDnRGsOKG7BXOOz1CK2oTQePobtr2hA9vA/i3jA+OnKDUYv8u8ep6HR=8woCM6P2E=fOnmwkIlKDoXKevV0+KqaCKCKeEeNfix2Rd8zMAZ/r=T0qdu7lY8idNS0hQs0jN5gltroc6AR7KsiBfRtkKRQTQDPW9rOLdPWDHCtRjgQ4h5g+g42pIwA1GAlfYeY8d+A9cj=363M9uyYydTUtC+ReE+A1iwfIQyEdXEojbZ45ZWgHsjOIlGQ9mKc6NAWmEgpQhQnAK5fXErwlG=nQxj0jG==sWrwjYQ3PP/wq+q0n6/1irgtvxUZO0Zae2Dw792MWeSmUrILMZrDST7StKIrE060ZqSEr1+e+75RPQXj6IDUGSuBxUe+=2AIoULqUa+URw9Uw8nr4m0Sar3Ff3eWfNItUw9jERAv+DeA4DQKQD08DiQ7iGxB4QbY4dD','Dnt': '1','Host': 'jobs.51job.com','Referer': 'https://jobs.51job.com/dongguan-fgz/150702673.html?s=sou_sou_soulb&t=0_0&req=7bd49afc5d2508dd3008e440ec8c7978','Sec-Ch-Ua': '"Microsoft Edge";v="117", "Not;A=Brand";v="8", "Chromium";v="117"','Sec-Ch-Ua-Mobile': '?0','Sec-Ch-Ua-Platform': '"Windows"','Sec-Fetch-Dest': 'document','Sec-Fetch-Mode': 'navigate','Sec-Fetch-Site': 'same-origin','Sec-Fetch-User': '?1','Sec-Gpc': '1','Upgrade-Insecure-Requests': '1','User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/117.0.0.0 Safari/537.36 Edg/117.0.2045.47'}params = {'s': 'sou_sou_soulb','t': '0_0','req': '7bd49afc5d2508dd3008e440ec8c7978','timestamp__1258': 'n4+xuDRDgAYx0DU2DBMroDkWgY4WT5qD=Y4D','alichlgref': 'https://we.51job.com/'}# 发送请求detail_res = requests.get(url=detail_url, headers=head, params=params)detail_res.encoding = 'gbk'# 解析请求soup = BeautifulSoup(detail_res.text, 'html.parser')div = soup.find('div', class_='bmsg job_msg inbox')post_dict['工作详细要求'] = div.text.strip()# 将字典添加到列表中post_list.append(post_dict)print(f'第{page}页爬取成功!')
print(post_list)
请各位大佬帮我 解疑释惑!!
这篇关于51job爬虫的一些问题的文章就介绍到这儿,希望我们推荐的文章对编程师们有所帮助!