kingdee_data_proc_api.py
kingdee_data_proc_api.py — 金蝶 JS 文件解析
职位数据直接发布在静态 jobs.js 里(较少见的方案):正则从 JS 文本提取 jobs = [...] 数组解析成 JSON,再拼伪 HTML 落盘。
代码
import time
import hashlib
import os
import requests
from urllib.parse import urlencode
import sys
sys.path.append('../')
import json
from utils import ner_logger
import re
from playwright.sync_api import sync_playwright
import threading
from concurrent.futures import ThreadPoolExecutor
# 请求头:伪装浏览器访问金蝶招聘页面
headers = {
"Accept": "*/*",
"Accept-Encoding": "gzip, deflate, br, zstd",
"Accept-Language": "zh-CN,zh;q=0.9",
"Cache-Control": "no-cache",
"Connection": "keep-alive",
"Referer": "https://campus.51job.com/kingdee/",
"Sec-Ch-Ua": '"Not;A=Brand";v="99", "Google Chrome";v="139", "Chromium";v="139"',
"Sec-Ch-Ua-Mobile": "?0",
"Sec-Ch-Ua-Platform": '"Windows"',
"Sec-Fetch-Dest": "script",
"Sec-Fetch-Mode": "no-cors",
"Sec-Fetch-Site": "same-origin",
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/139.0.0.0 Safari/537.36"
}
# 获取金蝶招聘职位数据(从JS文件中提取职位数组)
def get_kingdee_job_data(url):
try:
with requests.Session() as s:
resp = s.get(url, headers=headers, timeout=15)
if resp.status_code == 200:
js_content = resp.text
# 正则提取 jobs = [...] 格式的职位数据
jobs_match = re.search(r'(?:var\s+)?jobs\s*=\s*(\[.*?\]);?', js_content, re.DOTALL)
if jobs_match:
jobs_str = jobs_match.group(1)
jobs_data = json.loads(jobs_str)
return True, jobs_data
else:
# 兜底:提取任意数组
array_match = re.search(r'(\[.*\])', js_content, re.DOTALL)
if array_match:
array_str = array_match.group(1)
array_data = json.loads(array_str)
return True, array_data
else:
ner_logger.info("未能从JS文件中提取职位数据")
return False, []
else:
ner_logger.info("请求失败,状态码: %s", resp.status_code)
return False, []
except Exception as e:
ner_logger.info("请求金蝶招聘数据时出错: %s", str(e))
return False, []
# 将金蝶原始数据转为统一标准JSON
def transform_job_json(item, job_type, channel, tmp_file, json_file):
try:
# 字段映射
field_mapping = {
"announcement_name": "jobname", # 职位名称
"publish_time": "", # 发布时间(无)
"hd_dept": "organization", # 部门
"hd_loc": "city1", # 工作地点
"hd_job_num": "", # 招聘人数(无)
"hd_job_category": "type1" # 职位类别
}
# 构造详情页链接
detail_url = item.get("link", "")
if not detail_url:
job_id = item.get("jobid", "")
if job_id:
detail_url = f"https://campus.51job.com/kingdee/job.html?jobid={job_id}"
else:
detail_url = "https://campus.51job.com/kingdee/"
# 固定公共字段
fixed_fields = {
"link": detail_url,
"full_url": detail_url,
"last_url": detail_url,
"file_path": tmp_file,
"parent_url": "https://campus.51job.com/kingdee/",
"channel": channel,
"job_type": job_type
}
target_json = {}
# 映射字段
for target_field, source_field in field_mapping.items():
if source_field:
target_json[target_field] = item.get(source_field, "")
else:
target_json[target_field] = ""
target_json.update(fixed_fields)
# 保存JSON
with open(json_file, 'w', encoding='utf-8') as f:
json.dump(target_json, f, ensure_ascii=False, indent=4)
return True
except Exception as e:
ner_logger.error("转换职位数据时出错: %s", str(e))
return False
# 生成金蝶职位详情HTML页面
def generate_kingdee_job_html(item, tmp_file):
try:
# 提取字段
job_name = item.get("jobname", "")
organization = item.get("organization", "")
city = item.get("city1", "")
job_type = item.get("type1", "")
job_id = item.get("jobid", "")
address = item.get("address", "")
job_description = item.get("info", "").replace("\r\n", "<br>").replace("\n", "<br>")
job_link = item.get("link", "")
mobile_link = item.get("linkM", "")
# 拼接HTML
html_content = f'''
<!DOCTYPE html>
<html lang="zh-CN">
<head>
<title>金蝶招聘 - {job_name}</title>
body {{
font-family: "Microsoft YaHei", Arial, sans-serif;
margin: 0;
padding: 0;
background-color: #f5f5f5;
}}
.container {{
max-width: 1000px;
margin: 0 auto;
padding: 20px;
background-color: #fff;
}}
.header {{
border-bottom: 1px solid #eee;
padding-bottom: 20px;
margin-bottom: 20px;
}}
.job-title {{
font-size: 28px;
font-weight: bold;
color: #333;
margin-bottom: 10px;
}}
.job-meta {{
display: flex;
flex-wrap: wrap;
gap: 15px;
margin: 15px 0;
}}
.meta-item {{
display: flex;
align-items: center;
}}
.meta-label {{
font-weight: bold;
color: #666;
margin-right: 5px;
}}
.section {{
margin: 25px 0;
}}
.section-title {{
font-size: 20px;
color: #333;
border-left: 4px solid #007acc;
padding-left: 10px;
margin-bottom: 15px;
}}
.job-content {{
line-height: 1.8;
color: #555;
}}
.apply-button {{
display: inline-block;
padding: 12px 30px;
background-color: #007acc;
color: white;
text-decoration: none;
border-radius: 4px;
font-weight: bold;
margin-top: 20px;
}}
.apply-button:hover {{
background-color: #005fa3;
}}
.footer {{
margin-top: 30px;
padding-top: 20px;
border-top: 1px solid #eee;
color: #999;
font-size: 14px;
}}
</head>
<body>
<div class="container">
<div class="header">
<div class="job-title">{job_name}</div>
<div class="job-meta">
<div class="meta-item">
<span class="meta-label">所属组织:</span>
<span>{organization}</span>
</div>
<div class="meta-item">
<span class="meta-label">工作地点:</span>
<span>{city}</span>
</div>
<div class="meta-item">
<span class="meta-label">职位类别:</span>
<span>{job_type}</span>
</div>
<div class="meta-item">
<span class="meta-label">职位ID:</span>
<span>{job_id}</span>
</div>
<div class="meta-item">
<span class="meta-label">工作地址:</span>
<span>{address}</span>
</div>
</div>
</div>
<div class="section">
<h2 class="section-title">职位描述</h2>
<div class="job-content">
{job_description if job_description else "暂无职位描述"}
</div>
</div>
<div class="section">
<h2 class="section-title">应聘方式</h2>
<div class="job-content">
<p>请点击以下按钮投递简历:</p>
<a href="{job_link}" class="apply-button" target="_blank">立即申请</a>
<p>或者访问移动端链接:<a href="{mobile_link}" target="_blank">{mobile_link}</a></p>
</div>
</div>
<div class="footer">
<p>版权 © 金蝶国际软件集团有限公司</p>
<p>未经授权不得转载本网站之所有招聘信息及作品</p>
</div>
</div>
</body>
</html>
'''
# 写入文件
with open(tmp_file, "w", encoding="utf-8") as f:
f.write(html_content)
return True
except Exception as e:
ner_logger.error("生成金蝶招聘详情页失败: %s", str(e))
return False
# 金蝶招聘主处理入口
def api_proc_kingdee(spider_com, _key, com_info, k, url, _stat):
ner_logger.info("开始处理金蝶招聘数据, k: %s, url: %s", k, url)
# 默认职位数据JS地址
if not url or url == "":
url = "https://campus.51job.com/kingdee/js/jobs.js"
job_type = "xiaozhao" # 金蝶默认校招
# 临时文件目录
key_tmp_dir = spider_com.get_key_dir(_key)
ner_logger.info("临时目录: %s", key_tmp_dir)
# 获取职位数据
flag, job_data = get_kingdee_job_data(url)
if flag and job_data:
ner_logger.info("金蝶招聘数据获取成功,共 %s 条职位", len(job_data))
# 保存列表JSON
_hash = hashlib.md5(url.encode("utf-8")).hexdigest()
tmp_fname = f'{key_tmp_dir}/index_{_hash}_1.json'
with open(tmp_fname, 'w', encoding='utf-8') as f:
json.dump(job_data, f, ensure_ascii=False, indent=4)
# 逐条处理职位
ner_logger.info("开始处理 %s 条职位数据", len(job_data))
for i, item in enumerate(job_data):
try:
job_id = item.get("jobid", f"job_{i}")
job_name = item.get("jobname", "")
ner_logger.info("正在处理第 %s 条数据, 职位ID: %s, 职位名称: %s", i+1, job_id, job_name)
# 详情页URL
_fullurl = item.get("link", "")
if not _fullurl:
if job_id:
_fullurl = f"https://campus.51job.com/kingdee/job.html?jobid={job_id}"
else:
_fullurl = "https://campus.51job.com/kingdee/"
# 生成文件路径
_hash = hashlib.md5(_fullurl.encode("utf-8")).hexdigest()
tmp_file = os.path.join(key_tmp_dir, f"detail_{job_id}_{_hash}.html")
tmp_json_file = os.path.join(key_tmp_dir, f"detail_{job_id}_{_hash}.json")
# 文件已存在则跳过,仅更新时间
if os.path.exists(tmp_file) and os.path.exists(tmp_json_file):
try:
current_time = time.time()
os.utime(tmp_file, (current_time, current_time))
os.utime(tmp_json_file, (current_time, current_time))
ner_logger.info("文件 %s 的修改时间已更新为当前时间", tmp_json_file)
except Exception as e:
ner_logger.error("更新文件 %s 的修改时间时出错: %s", tmp_json_file, str(e))
continue
# 转换JSON + 生成HTML
if transform_job_json(item, job_type, _key, tmp_file, tmp_json_file):
if generate_kingdee_job_html(item, tmp_file):
ner_logger.info("完成处理第 %s 条数据", i+1)
else:
ner_logger.error("生成第 %s 条数据HTML失败", i+1)
else:
ner_logger.error("转换第 %s 条数据失败", i+1)
time.sleep(1)
except Exception as e:
ner_logger.error("处理第 %s 条数据时出错: %s", i+1, str(e))
continue
ner_logger.info("金蝶招聘数据处理完成")
return True
else:
ner_logger.error("获取金蝶招聘数据失败")
return False
项目分区导航:jd_data_proc_api ⬅️ | 04-kingdee_data_proc_api | ➡️ picc_data_proc_api
💬 评论