xingye_proc.py
xingye_proc.py — 兴业银行专项爬虫
注册 Playwright 响应拦截器 → 自动翻页 → 捕获招聘 API 的 JSON 响应 → 字段映射 + generate_html 拼伪 HTML 落盘。三层结构详解见 auto_on_response 主篇。
代码
import time
from functools import partial
import hashlib
import os
import json
import requests
from utils import ner_logger
def xingye_proc(spider_com, page, _key, com_info, k, url, _stat):
"""
兴业银行招聘专用爬虫处理入口
监听接口响应 → 自动翻页 → 抓取职位数据
"""
# 根据接口标识判断招聘类型:社招/校招/实习
job_type = "shezhao"
if k.startswith("shezhao"):
job_type = "shezhao"
elif k.startswith("xiaozhao"):
job_type = "xiaozhao"
elif k.startswith("shixi"):
job_type = "shixi"
# 绑定 response 监听函数,传递固定参数
wrapped_handler = partial(response_handler, spider_com, page, _key, com_info, k, url, _stat, job_type)
page.on('response', wrapped_handler)
# 打开职位列表页面
response = page.goto(url, timeout=10000)
time.sleep(10)
# 设置翻页次数:普通模式3页,全量模式100页
_page_count = 3
if 'method' in _stat and _stat['method'] == "cp_full":
_page_count = 100
# 自动翻页采集
for i in range(1, _page_count):
# 定位“下一页”按钮
next_page_button = page.get_by_title("下一页")
# 判断按钮是否可点击
if next_page_button and next_page_button.is_enabled():
ner_logger.info("兴业银行,翻页第%d页" % (i))
# 检查禁用属性
aria_disabled = next_page_button.get_attribute('aria-disabled')
if aria_disabled != 'true':
next_page_button.click()
time.sleep(30)
else:
ner_logger.info("兴业银行,没有下一页了")
break
def response_handler(spider_com, page, _key, com_info, k, url, _stat, job_type, response):
"""
监听接口响应,拦截职位列表接口并解析数据
"""
# 拦截职位列表接口
if response.url.startswith("https://job.cib.com.cn/ersApi/recruitposition/portalPage"):
_data_json = response.json()
# 接口返回成功时解析数据
if _data_json['message'] == '成功':
_data = _data_json['data']
_list = _data['list']
# 遍历职位,逐条生成结构化数据
for _item in _list:
xingye_json(_item, spider_com, page, _key, com_info, k, url, _stat, job_type)
def xingye_json(original_data, spider_com, page, _key, com_info, k, url, _stat, job_type):
"""
将原始接口数据转为标准结构化JSON,并生成HTML文件
"""
ner_logger.info("兴业银行,开始生成json")
# 拼接职位详情页URL
_fullurl = f"https://job.cib.com.cn/portal/#/positionDetails/{original_data['positionId']}"
# 生成临时文件路径(HTML + JSON)
key_tmp_dir = spider_com.get_key_dir(_key)
_hash = hashlib.md5(_fullurl.encode("utf-8")).hexdigest()
tmp_file = os.path.join(key_tmp_dir, f"detail_{_hash}.html")
tmp_json_file = os.path.join(key_tmp_dir, f"detail_{_hash}.json")
# 生成可视化HTML页面
_context_outtext = generate_html(original_data)
with open(tmp_file, "w", encoding="utf-8") as f:
f.write(_context_outtext)
# 结构化字段映射
converted_data = {
"announcement_name": original_data["positionName"],
"publish_time": original_data["publishTime"].split()[0], # 只保留日期
"link": _fullurl,
"hd_dept": original_data["departmentDesc"],
"hd_loc": original_data["positionAddr"],
"hd_job_num": str(original_data["recruitingNum"]) if original_data["recruitingNum"] != -1 else "",
"hd_job_category": "",
"full_url": _fullurl,
"last_url": _fullurl,
"file_path": tmp_file,
"parent_url": "https://job.cib.com.cn",
"channel": "com_91000",
"job_type": job_type
}
# 保存结构化JSON
with open(tmp_json_file, 'w', encoding='utf-8') as f:
json.dump(converted_data, f, ensure_ascii=False, indent=4)
def generate_html(data):
"""
根据接口原始数据生成可视化HTML
"""
htmllist = []
htmllist.append(f"<div> 职位名 {data['positionName']}")
htmllist.append(f"<div> 地点 {data['positionAddr']}")
htmllist.append(f"<div> 发布机构 {data['firstBusinessUnitDesc']}")
htmllist.append(f"<div> 部门 {data['departmentDesc']}")
htmllist.append(f"<div> 发布日期 {data['publishTime']}")
htmllist.append(f"<div> 过期日期 {data['expiryDate']}")
# 招聘类型映射
if data['recruitType'] == 'CR':
htmllist.append(f"<div> 招聘类型: 校招")
elif data['recruitType'] == 'TR':
htmllist.append(f"<div> 招聘类型: 实习")
else:
htmllist.append(f"<div> 招聘类型: 社招")
# 招聘人数:-1 表示若干
if data['recruitingNum'] != -1:
htmllist.append(f"<div> 招聘人数: {data['recruitingNum']}")
else:
htmllist.append(f"<div> 招聘人数: 若干")
htmllist.append(f"<div> 专业要求 \n{data['majorRequirment']}")
htmllist.append(f"<div> 工作职责 \n{data['jobDuty']}")
htmllist.append(f"<div> 任职要求 \n{data['positionRequirment']}")
return "\n".join(htmllist)
💬 评论