--- title: "05-picc_data_proc_api" created: 2026-04-02 tags: - 项目 aliases: - picc_data_proc_api --- # picc_data_proc_api.py ### `picc_data_proc_api.py` — 中国人保直连(最全) 社招/校招/实习三条通道各自独立的取数 → 字段转换(transform)→ HTML 生成函数,体量是本模块最大的一个(超过 800 行)。 ## 代码 ```python # 时间相关操作库,用于延时 import time # 哈希库,用于生成文件唯一标识 import hashlib # 文件操作系统库,用于路径处理、文件存在性判断等 import os # HTTP请求库,用于发送接口请求 import requests # 系统库,用于修改Python导入路径 import sys # 将上级目录加入搜索路径,以便导入工具类 sys.path.append('../') # JSON处理库,用于解析和生成JSON数据 import json # 日志工具,用于记录运行日志和错误信息 from utils import ner_logger # 请求头信息,模拟浏览器访问,防止被接口拦截 headers = { "accept": "application/json, text/javascript, */*; q=0.01", "accept-encoding": "gzip, deflate, br, zstd", "accept-language": "zh-CN,zh;q=0.9", "cache-control": "no-cache", "connection": "keep-alive", "content-type": "application/json; charset=UTF-8", "origin": "https://picc.zhiye.com", "referer": "https://picc.zhiye.com/custom/social?hideAll=true&ky=&c1=&c2=&d=&c=", "sec-ch-ua": '"Not;A=Brand";v="99", "Google Chrome";v="139", "Chromium";v="139"', "sec-ch-ua-mobile": "?0", "sec-ch-ua-platform": '"Windows"', "sec-fetch-dest": "empty", "sec-fetch-mode": "cors", "sec-fetch-site": "same-origin", "user-agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/139.0.0.0 Safari/537.36", "x-requested-with": "XMLHttpRequest" } def get_picc_job_data(url, page_index=0, page_size=10): """ 获取中国人保 社招 招聘数据 参数: url: 请求地址 page_index: 页码 page_size: 每页数量 返回: (flag, data, count) 元组 flag: 是否成功 data: 职位列表数据 count: 总条数 """ try: # 请求体参数:Category=1 代表社招 payload = { "Category": ["1"], "SpecialType": 0, "PageIndex": str(page_index), "PageSize": page_size, "DisplayFields": [ "Id", "HeadCount", "JobAdId", "JobAdName", "Kind", "LocNames", "Org", "PostDate", "EndTime", "Salary", "ClassificationOne", "ClassificationTwo", "Duty", "Require", "Degree", "YearsOfWorking", "Category" ] } # 发送POST请求 with requests.Session() as s: resp = s.post(url, json=payload, headers=headers, timeout=15) if resp.status_code == 200: result = resp.json() if result.get("Code") == 200: data = result.get("Data", []) count = result.get("Count", 0) return True, data, count else: ner_logger.error("中国人保招聘数据接口返回错误码: %s", result.get("Code")) return False, [], 0 else: ner_logger.error("请求中国人保招聘数据失败,状态码: %s", resp.status_code) return False, [], 0 except Exception as e: ner_logger.error("请求中国人保招聘数据时出错: %s", str(e)) return False, [], 0 def transform_job_json(item, job_type, channel, tmp_file, json_file): """ 将原始职位JSON转换为统一格式JSON 参数: item: 源数据字典 job_type: 职位类型(社招/校招/实习) channel: 渠道标识 tmp_file: 生成的HTML路径 json_file: 输出JSON文件路径 """ try: # 统一字段映射关系 field_mapping = { "announcement_name": "JobAdName", # 职位名称 "publish_time": "PostDate", # 发布时间 "hd_dept": "Org", # 部门 "hd_loc": "LocNames", # 工作地点(数组) "hd_job_num": "HeadCount", # 招聘人数 "hd_job_category": "Kind" # 职位类别 } # 根据职位类型设置详情页URL和父页面URL if job_type == "xiaozhao": detail_url = "https://picc.zhiye.com/custom/campus?hideAll=true&ky=&c1=&c2=&d=&c=" parent_url = "https://picc.zhiye.com/custom/campus" elif job_type == "shixi": detail_url = "https://picc.zhiye.com/custom/shixi?hideAll=true" parent_url = "https://picc.zhiye.com/custom/shixi" else: # 默认为社招 detail_url = "https://picc.zhiye.com/custom/social?hideAll=true&ky=&c1=&c2=&d=&c=" parent_url = "https://picc.zhiye.com/custom/social" # 固定字段 fixed_fields = { "link": detail_url, "full_url": detail_url, "last_url": detail_url, "file_path": tmp_file, "parent_url": parent_url, "channel": channel, "job_type": job_type } # 构建目标JSON target_json = {} # 遍历映射,赋值 for target_field, source_field in field_mapping.items(): value = item.get(source_field, "") # 地点数组转字符串 if source_field == "LocNames" and isinstance(value, list): value = ", ".join(value) # 招聘人数转字符串 if source_field == "HeadCount": value = str(value) target_json[target_field] = value # 合并固定字段 target_json.update(fixed_fields) # 写入JSON文件 with open(json_file, 'w', encoding='utf-8') as f: json.dump(target_json, f, ensure_ascii=False, indent=4) return True except Exception as e: ner_logger.error("转换职位数据时出错: %s", str(e)) return False def generate_picc_job_html(item, tmp_file, job_type="shezhao"): """ 生成中国人保职位详情HTML页面 参数: item: 职位原始数据 tmp_file: HTML保存路径 job_type: 职位类型 """ try: # 提取职位关键字段 job_ad_name = item.get("JobAdName", "") org = item.get("Org", "") loc_names = item.get("LocNames", []) # 地点数组转字符串 if isinstance(loc_names, list): loc_names_str = ", ".join(loc_names) else: loc_names_str = str(loc_names) head_count = item.get("HeadCount", "") kind = item.get("Kind", "") post_date = item.get("PostDate", "") end_time = item.get("EndTime", "") salary = item.get("Salary", "") classification_one = item.get("ClassificationOne", "") classification_two = item.get("ClassificationTwo", "") duty = item.get("Duty", "").replace("\r\n", "
").replace("\n", "
") require = item.get("Require", "").replace("\r\n", "
").replace("\n", "
") degree = item.get("Degree", "") years_of_working = item.get("YearsOfWorking", "") category = item.get("Category", "") job_id = item.get("JobAdId", "") # 根据类型设置标题与申请链接 if job_type == "xiaozhao": title_prefix = "中国人保校园招聘" apply_url = "https://picc.zhiye.com/custom/campus?hideAll=true&ky=&c1=&c2=&d=&c=" apply_text = "前往中国人保官方校园招聘页面" elif job_type == "shixi": title_prefix = "中国人保实习招聘" apply_url = "https://picc.zhiye.com/custom/shixi?hideAll=true" apply_text = "前往中国人保官方实习招聘页面" else: # 社招 title_prefix = "中国人保招聘" apply_url = "https://picc.zhiye.com/custom/social?hideAll=true&ky=&c1=&c2=&d=&c=" apply_text = "前往中国人保官方招聘页面" # 拼接HTML内容 html_content = f''' {title_prefix} - {job_ad_name}
{job_ad_name}
所属机构: {org}
工作地点: {loc_names_str}
招聘人数: {head_count}人
职位类别: {kind}
发布日期: {post_date}
截止日期: {end_time}
薪资范围: {salary}
一级分类: {classification_one}
二级分类: {classification_two}
招聘类型: {category}
学历要求: {degree}
工作经验: {years_of_working}

岗位职责

{duty if duty else "暂无岗位职责信息"}

任职要求

{require if require else "暂无任职要求信息"}

应聘方式

请点击以下按钮{apply_text}投递简历:

立即申请
''' # 写入HTML文件 with open(tmp_file, "w", encoding="utf-8") as f: f.write(html_content) return True except Exception as e: ner_logger.error("生成中国人保招聘详情页失败: %s", str(e)) return False def get_picc_campus_job_data(url, page_index=0, page_size=10): """ 获取中国人保 校园招聘 数据 参数: url: 请求地址 page_index: 页码 page_size: 每页数量 返回: (flag, data, count) 元组 """ try: # Category=2 代表校招 payload = { "Category": ["2"], "SpecialType": 0, "PageIndex": str(page_index), "PageSize": page_size, "DisplayFields": [ "Id", "HeadCount", "JobAdId", "JobAdName", "Kind", "LocNames", "Org", "PostDate", "EndTime", "Salary", "ClassificationOne", "ClassificationTwo", "Duty", "Require", "Degree", "YearsOfWorking", "Category" ] } # 复制请求头并修改referer为校招页面 campus_headers = headers.copy() campus_headers["referer"] = "https://picc.zhiye.com/custom/campus?hideAll=true&ky=&c1=&c2=&d=&c=" # 发送请求 with requests.Session() as s: resp = s.post(url, json=payload, headers=campus_headers, timeout=15) if resp.status_code == 200: result = resp.json() if result.get("Code") == 200: data = result.get("Data", []) count = result.get("Count", 0) return True, data, count else: ner_logger.error("中国人保校园招聘数据接口返回错误码: %s", result.get("Code")) return False, [], 0 else: ner_logger.error("请求中国人保校园招聘数据失败,状态码: %s", resp.status_code) return False, [], 0 except Exception as e: ner_logger.error("请求中国人保校园招聘数据时出错: %s", str(e)) return False, [], 0 def transform_campus_job_json(item, job_type, channel, tmp_file, json_file): """ 校招JSON转换(已废弃,统一使用 transform_job_json) """ return transform_job_json(item, job_type, channel, tmp_file, json_file) def generate_picc_campus_job_html(item, tmp_file): """ 校招HTML生成(已废弃,统一使用 generate_picc_job_html) """ return generate_picc_job_html(item, tmp_file, "xiaozhao") def get_picc_internship_job_data(url, page_index=0, page_size=10): """ 获取中国人保 实习招聘 数据 参数: url: 请求地址 page_index: 页码 page_size: 每页数量 返回: (flag, data, count) 元组 """ try: # Category=3 代表实习 payload = { "Category": ["3"], "SpecialType": 0, "PageIndex": str(page_index), "PageSize": page_size, "DisplayFields": [ "Id", "HeadCount", "JobAdId", "JobAdName", "Kind", "LocNames", "Org", "PostDate", "EndTime", "Salary", "ClassificationOne", "ClassificationTwo", "Duty", "Require", "Degree", "YearsOfWorking", "Category" ] } # 修改referer为实习页面 internship_headers = headers.copy() internship_headers["referer"] = "https://picc.zhiye.com/custom/shixi?hideAll=true" # 发送请求 with requests.Session() as s: resp = s.post(url, json=payload, headers=internship_headers, timeout=15) if resp.status_code == 200: result = resp.json() if result.get("Code") == 200: data = result.get("Data", []) count = result.get("Count", 0) return True, data, count else: ner_logger.error("中国人保实习招聘数据接口返回错误码: %s", result.get("Code")) return False, [], 0 else: ner_logger.error("请求中国人保实习招聘数据失败,状态码: %s", resp.status_code) return False, [], 0 except Exception as e: ner_logger.error("请求中国人保实习招聘数据时出错: %s", str(e)) return False, [], 0 def transform_internship_job_json(item, job_type, channel, tmp_file, json_file): """ 实习JSON转换(已废弃,统一使用 transform_job_json) """ return transform_job_json(item, job_type, channel, tmp_file, json_file) def generate_picc_internship_job_html(item, tmp_file): """ 实习HTML生成(已废弃,统一使用 generate_picc_job_html) """ return generate_picc_job_html(item, tmp_file, "shixi") def api_proc_picc(spider_com, _key, com_info, k, url, _stat): """ 人保招聘主入口:自动识别 社招/校招/实习 """ # 根据k值判断类型 if "xiaozhao" in k: return api_proc_picc_campus(spider_com, _key, com_info, k, url, _stat) elif "shixi" in k: return api_proc_picc_internship(spider_com, _key, com_info, k, url, _stat) else: # 社招处理逻辑 ner_logger.info("开始处理中国人保招聘数据, k: %s, url: %s", k, url) # 默认API地址 if not url or url == "": url = "https://picc.zhiye.com/api/Jobad/GetJobAdPageList" job_type = "shezhao" # 获取临时存储目录 key_tmp_dir = spider_com.get_key_dir(_key) ner_logger.info("临时目录: %s", key_tmp_dir) # 获取总条数 flag, _, total_count = get_picc_job_data(url, 0, 10) if not flag: ner_logger.error("获取中国人保招聘数据总数失败") return False # 计算总页数 total_pages = (total_count // 10) + (1 if total_count % 10 > 0 else 0) ner_logger.info("中国人保招聘数据总共有 %s 条,共 %s 页", total_count, total_pages) # 逐页抓取 for page_index in range(total_pages): ner_logger.info("正在处理第 %s 页,共 %s 页", page_index + 1, total_pages) # 获取当前页数据 flag, job_data, _ = get_picc_job_data(url, page_index, 10) if not flag or not job_data: ner_logger.error("获取第 %s 页数据失败", page_index + 1) continue # 保存原始分页JSON _hash = hashlib.md5(url.encode("utf-8")).hexdigest() tmp_fname = f'{key_tmp_dir}/index_{_hash}_{page_index + 1}.json' with open(tmp_fname, 'w', encoding='utf-8') as f: json.dump(job_data, f, ensure_ascii=False, indent=4) # 处理每条职位 ner_logger.info("开始处理第 %s 页的 %s 条职位数据", page_index + 1, len(job_data)) for i, item in enumerate(job_data): try: job_id = item.get("JobAdId", f"job_{page_index}_{i}") job_ad_name = item.get("JobAdName", "") ner_logger.info("正在处理第 %s 页第 %s 条数据, 职位ID: %s, 职位名称: %s", page_index + 1, i+1, job_id, job_ad_name) # 构造详情页链接 _fullurl = "https://picc.zhiye.com/custom/social?hideAll=true&ky=&c1=&c2=&d=&c=" # 生成文件路径 _hash = hashlib.md5(_fullurl.encode("utf-8")).hexdigest() tmp_file = os.path.join(key_tmp_dir, f"detail_{job_id}_{_hash}.html") tmp_json_file = os.path.join(key_tmp_dir, f"detail_{job_id}_{_hash}.json") # 文件已存在则更新时间并跳过 if os.path.exists(tmp_file) and os.path.exists(tmp_json_file): try: current_time = time.time() os.utime(tmp_file, (current_time, current_time)) os.utime(tmp_json_file, (current_time, current_time)) ner_logger.info("文件 %s 的修改时间已更新为当前时间", tmp_json_file) except Exception as e: ner_logger.error("更新文件 %s 的修改时间时出错: %s", tmp_json_file, str(e)) continue # 转换JSON + 生成HTML if transform_job_json(item, job_type, _key, tmp_file, tmp_json_file): if generate_picc_job_html(item, tmp_file): ner_logger.info("完成处理第 %s 页第 %s 条数据", page_index + 1, i+1) else: ner_logger.error("生成第 %s 页第 %s 条数据HTML失败", page_index + 1, i+1) else: ner_logger.error("转换第 %s 页第 %s 条数据失败", page_index + 1, i+1) # 单条延时 time.sleep(1) except Exception as e: ner_logger.error("处理第 %s 页第 %s 条数据时出错: %s", page_index + 1, i+1, str(e)) continue # 每页结束延时 time.sleep(3) ner_logger.info("中国人保招聘数据处理完成") return True def api_proc_picc_campus(spider_com, _key, com_info, k, url, _stat): """ 处理中国人保 校园招聘 数据 """ ner_logger.info("开始处理中国人保校园招聘数据, k: %s, url: %s", k, url) if not url or url == "": url = "https://picc.zhiye.com/api/Jobad/GetJobAdPageList" job_type = "xiaozhao" # 临时目录 key_tmp_dir = spider_com.get_key_dir(_key) ner_logger.info("临时目录: %s", key_tmp_dir) # 获取总数 flag, _, total_count = get_picc_campus_job_data(url, 0, 10) if not flag: ner_logger.error("获取中国人保校园招聘数据总数失败") return False # 总页数 total_pages = (total_count // 10) + (1 if total_count % 10 > 0 else 0) ner_logger.info("中国人保校园招聘数据总共有 %s 条,共 %s 页", total_count, total_pages) # 遍历页面 for page_index in range(total_pages): ner_logger.info("正在处理第 %s 页,共 %s 页", page_index + 1, total_pages) # 获取校招数据 flag, job_data, _ = get_picc_campus_job_data(url, page_index, 10) if not flag or not job_data: ner_logger.error("获取第 %s 页数据失败", page_index + 1) continue # 保存原始数据 _hash = hashlib.md5(url.encode("utf-8")).hexdigest() tmp_fname = f'{key_tmp_dir}/index_{_hash}_{page_index + 1}.json' with open(tmp_fname, 'w', encoding='utf-8') as f: json.dump(job_data, f, ensure_ascii=False, indent=4) # 处理职位 ner_logger.info("开始处理第 %s 页的 %s 条职位数据", page_index + 1, len(job_data)) for i, item in enumerate(job_data): try: job_id = item.get("JobAdId", f"job_{page_index}_{i}") job_ad_name = item.get("JobAdName", "") ner_logger.info("正在处理第 %s 页第 %s 条数据, 职位ID: %s, 职位名称: %s", page_index + 1, i+1, job_id, job_ad_name) # 校招详情页 _fullurl = "https://picc.zhiye.com/custom/campus?hideAll=true&ky=&c1=&c2=&d=&c=" # 文件路径 _hash = hashlib.md5(_fullurl.encode("utf-8")).hexdigest() tmp_file = os.path.join(key_tmp_dir, f"detail_{job_id}_{_hash}.html") tmp_json_file = os.path.join(key_tmp_dir, f"detail_{job_id}_{_hash}.json") # 已存在则跳过 if os.path.exists(tmp_file) and os.path.exists(tmp_json_file): try: current_time = time.time() os.utime(tmp_file, (current_time, current_time)) os.utime(tmp_json_file, (current_time, current_time)) ner_logger.info("文件 %s 的修改时间已更新为当前时间", tmp_json_file) except Exception as e: ner_logger.error("更新文件 %s 的修改时间时出错: %s", tmp_json_file, str(e)) continue # 转换+生成 if transform_campus_job_json(item, job_type, _key, tmp_file, tmp_json_file): if generate_picc_campus_job_html(item, tmp_file): ner_logger.info("完成处理第 %s 页第 %s 条数据", page_index + 1, i+1) else: ner_logger.error("生成第 %s 页第 %s 条数据HTML失败", page_index + 1, i+1) else: ner_logger.error("转换第 %s 页第 %s 条数据失败", page_index + 1, i+1) time.sleep(1) except Exception as e: ner_logger.error("处理第 %s 页第 %s 条数据时出错: %s", page_index + 1, i+1, str(e)) continue time.sleep(3) ner_logger.info("中国人保校园招聘数据处理完成") return True def api_proc_picc_internship(spider_com, _key, com_info, k, url, _stat): """ 处理中国人保 实习招聘 数据 """ ner_logger.info("开始处理中国人保实习招聘数据, k: %s, url: %s", k, url) if not url or url == "": url = "https://picc.zhiye.com/api/Jobad/GetJobAdPageList" job_type = "shixi" # 临时目录 key_tmp_dir = spider_com.get_key_dir(_key) ner_logger.info("临时目录: %s", key_tmp_dir) # 获取总数 flag, _, total_count = get_picc_internship_job_data(url, 0, 10) if not flag: ner_logger.error("获取中国人保实习招聘数据总数失败") return False # 总页数 total_pages = (total_count // 10) + (1 if total_count % 10 > 0 else 0) ner_logger.info("中国人保实习招聘数据总共有 %s 条,共 %s 页", total_count, total_pages) # 遍历页面 for page_index in range(total_pages): ner_logger.info("正在处理第 %s 页,共 %s 页", page_index + 1, total_pages) # 获取实习数据 flag, job_data, _ = get_picc_internship_job_data(url, page_index, 10) if not flag or not job_data: ner_logger.error("获取第 %s 页数据失败", page_index + 1) continue # 保存原始数据 _hash = hashlib.md5(url.encode("utf-8")).hexdigest() tmp_fname = f'{key_tmp_dir}/index_{_hash}_{page_index + 1}.json' with open(tmp_fname, 'w', encoding='utf-8') as f: json.dump(job_data, f, ensure_ascii=False, indent=4) # 处理职位 ner_logger.info("开始处理第 %s 页的 %s 条职位数据", page_index + 1, len(job_data)) for i, item in enumerate(job_data): try: job_id = item.get("JobAdId", f"job_{page_index}_{i}") job_ad_name = item.get("JobAdName", "") ner_logger.info("正在处理第 %s 页第 %s 条数据, 职位ID: %s, 职位名称: %s", page_index + 1, i+1, job_id, job_ad_name) # 实习详情页 _fullurl = "https://picc.zhiye.com/custom/shixi?hideAll=true" # 文件路径 _hash = hashlib.md5(_fullurl.encode("utf-8")).hexdigest() tmp_file = os.path.join(key_tmp_dir, f"detail_{job_id}_{_hash}.html") tmp_json_file = os.path.join(key_tmp_dir, f"detail_{job_id}_{_hash}.json") # 已存在则跳过 if os.path.exists(tmp_file) and os.path.exists(tmp_json_file): try: current_time = time.time() os.utime(tmp_file, (current_time, current_time)) os.utime(tmp_json_file, (current_time, current_time)) ner_logger.info("文件 %s 的修改时间已更新为当前时间", tmp_json_file) except Exception as e: ner_logger.error("更新文件 %s 的修改时间时出错: %s", tmp_json_file, str(e)) continue # 转换+生成 if transform_internship_job_json(item, job_type, _key, tmp_file, tmp_json_file): if generate_picc_internship_job_html(item, tmp_file): ner_logger.info("完成处理第 %s 页第 %s 条数据", page_index + 1, i+1) else: ner_logger.error("生成第 %s 页第 %s 条数据HTML失败", page_index + 1, i+1) else: ner_logger.error("转换第 %s 页第 %s 条数据失败", page_index + 1, i+1) time.sleep(1) except Exception as e: ner_logger.error("处理第 %s 页第 %s 条数据时出错: %s", page_index + 1, i+1, str(e)) continue time.sleep(3) ner_logger.info("中国人保实习招聘数据处理完成") return True ``` --- **项目分区导航**:[[04-kingdee_data_proc_api|kingdee_data_proc_api]] ⬅️ | 05-picc_data_proc_api | ➡️ [[00-api|api]]