ann_model.py

ann_model.py — 核心调度器(主入口)

整个解析系统的大脑,串联所有模块:公告级解析 → 职位列表解析 → Markdown 格式处理三阶段,输出标准 JSON 模型文件。三阶段详解见 parsegpt 主篇

代码

# -*- coding: utf-8 -*-
"""
校招公告解析核心模块
功能:从MD/HTML文本中,通过大模型提取:公司信息、职位列表、应聘方式、二维码、学历、毕业届等结构化数据
输出:标准JSON模型文件,用于后续入库/展示
"""
import json
import sys
sys.path.append('../')

import os
import re
import datetime

# 工具类:日志、MD5、IP、版本、去重、中文判断
from utils import ner_logger, getMD5Str, get_local_ip, QZ_VERISON, deduplicate_strings, all_zh
# HTML/MD工具:文本清洗、MD提取、GPT后修复
from utils_html import clean_text, get_md_content, fix_md_after_gpt
# 简历工具:学历标准化
from utils_resume import fix_diploma
# 时间工具:日期格式化、过期判断
from utils_date import fix_data_format, is_near_month, get_current_time_string

# 大模型Prompt模板(校园公告、职位、文章、表格等)
from parsegpt.template import get_template_campus_wb10 as get_template_campus
from parsegpt.template import get_template_full_html, get_template_table_html
from parsegpt.template import get_template_job_wb as get_template_job
from parsegpt.template import get_template_article as get_template_article
from parsegpt.template import get_template_article_fix as get_template_article_fix

# 大模型API
from api.doubao_api import call_gpt
from api.doubao_api import call_gpt as doubao_call_gpt
from api.qwen_api import call_gpt as qwen_call_gpt
# 云端查重
from api.quanzhi_api import check_cloud
# 职位解析子模块
from parsegpt.ann_model_job import parse_cjob

# ====================== 主函数:公告结构化解析 ======================
def parse_announcement(_mdfile, _htmlfile, _model_file, _info, _full_text, proc_type, sch_info, _expired_file, _stat):
    """
    【核心入口】解析招聘公告,输出结构化JSON
    流程:清洗文本 → 大模型提取公告主体 → 过期/重复/三无校验 → 职位列表提取 → MD优化 → 输出模型
    :return: (是否成功, 信息描述)
    """
    _ann_dict = {}  # 最终输出的模型字典
    _title = _info['announcement_name']
    _text = clean_text(_full_text)  # 清洗纯文本

    # 调优文本(页面指定区块提取的内容,辅助大模型)
    _tuning_md = ""
    _tuning_full_text = ""
    if 'tuning_content' in _info and len(_info['tuning_content']) > 0:
        _tuning_t = _info['tuning_content']
        if len(_tuning_t) > 6000:  # 超长截断,防止大模型超限
            _tuning_t = _tuning_t[:6000]
        _tuning_full_text = "\n".join(_tuning_t)

    # ========== 1. 大模型解析:公告主体(公司、时间、应聘方式等) ==========
    try:
        _all_text = f"{_text}\n{_tuning_full_text}"
        _t_text = get_template_campus(_title, _all_text)  # 构造Prompt
        (_ok_flag, json_str) = call_gpt(_t_text, True)  # 调用大模型

        if not _ok_flag:
            return False, f"通过大模型获取公告信息Error:{_ok_flag}\n{json_str}"

        json_data = json.loads(json_str, strict=False)  # 转JSON

        # 校验:返回字段太少,视为解析失败
        if len(json_data) < 10:
            return False, f"通过大模型获取公告信息项目太少Error:{_ok_flag}\n{json_str}"

        # 校验:公告过期(截止时间超过1个月)
        if 'OnlineEndDate' in json_data and json_data['OnlineEndDate'].strip() != '':
            _datestr = fix_data_format(json_data['OnlineEndDate'].strip())
            if not is_near_month(_datestr, 1):
                return False, f"处理的公告过期:{_datestr}"

        # 处理:多公司公告 → 统一命名为“校招公告”
        _com_name = json_data['ComName'].strip()
        if 'HasMultipleCompanies' in json_data and json_data['HasMultipleCompanies'].strip() == "是":
            if all_zh(_com_name) and len(_com_name) < 25:
                ner_logger.info(f"处理的公告存在多公司的情况,排除掉:{_com_name}")
            else:
                json_data['ComName'] = "校招公告"
                json_data['ComDesc'] = ""
                json_data['ComIndustry'] = ""
        # 公司名过长/非中文 → 视为多公司
        elif not all_zh(_com_name) and len(_com_name) > 40:
            json_data['ComName'] = "校招公告"
            json_data['ComDesc'] = ""
            json_data['ComIndustry'] = ""
            json_data['HasMultipleCompanies'] = "是"

        # 提取二维码信息(图片解析结果)
        _qrdict = get_qrcode_info(_info)
        json_data['ApplyTypeQrcode'] = _qrdict
        json_data['FullText'] = _all_text

        # 填充来源、标题、IP、时间等附加字段
        set_other_info(sch_info, _info, json_data, _text, _htmlfile)

        # ========== 核心过滤:三无公告(无链接、无文本、无邮箱、无二维码、无联系方式) ==========
        if not json_data['ApplyTypeLink'] and not json_data['ApplyTypeText'] and not json_data['ApplyTypeEmail'] and len(json_data['ApplyTypeQrcode']) == 0 and len(json_data['ApplyContacts']) == 0:
            return False, f"处理的公告存在无链接、无应聘文本、无应聘邮箱、无联系方式的的情况"

        # 过滤:无公司 / 无职位标题
        if not json_data['ComName'] or not json_data['JobTitle']:
            return False, f"处理的公告存在无公司、无公告名称的情况"

        # 云端查重:公告已存在
        _ok, msg = check_cloud(json_data['JobLink'], json_data['ComName'], json_data['FullText'], json_data['GraduationYear'])
        if not _ok:
            return False, f"云端检测公告重复:{msg}"

        # 学历标准化(大专/本科/硕士...)
        fix_diploma_data_map(json_data)

        # 存入公告主体
        _ann_dict['ann'] = json_data

    except json.JSONDecodeError as e:
        import traceback
        traceback.print_exc()
        return False, f"通过大模型获取公告信息Error:\n{e}"

    # ========== 白名单学校:单独走职位解析流程 ==========
    _white_list = ['sch_98534', 'sch_00131', 'sch_98531', 'sch_21131', 'sch_98507', 'sch_00114', 'sch_00102']
    if _info['channel'] in _white_list and json_data['AnnouncementType'] == "职位":
        ner_logger.info(f"处理的公告是职位信息:{_title}")
        # 补充服务信息
        json_data['server_ip'] = get_local_ip()
        json_data['qz_version'] = QZ_VERISON
        json_data['process_time'] = get_current_time_string()
        json_data['file_path'] = _htmlfile

        # MD格式化
        common_process_fix(_mdfile, json_data)

        # 调用职位专用解析
        _ok, annjson_data = parse_cjob(_htmlfile, json_data)
        if _ok:
            # 保存模型文件
            with open(_model_file, 'w', encoding='utf-8') as fw:
                json.dump(annjson_data, fw, ensure_ascii=False, indent=4)
                ner_logger.info(f"处理文章生成公告内职位模型文件成功,{_model_file}")
            # 标记成功
            _job_other_file = _expired_file.replace(".json.expired", ".json.job")
            with open(_job_other_file, 'w', encoding='utf-8') as fw:
                fw.write(f"job 110\n")
            return True, f"{_htmlfile}公告内职位解析成功"
        return False, f"{_htmlfile}公告里面的职位解析失败"
    else:
        # 非白名单,清理职位标记文件
        _job_other_file = _expired_file.replace(".json.expired", ".json.job")
        if os.path.exists(_job_other_file):
            os.remove(_job_other_file)

    # ========== 2. 大模型解析:职位列表 ==========
    try:
        _t_text = get_template_job(_title, _text)
        _ok, json_data = get_all_job_info(_t_text)
        if not _ok:
            return False, f"通过大模型获取职位信息列表Error:\n{json_str}"

        # 无职位 → 尝试从调优文本提取
        if len(json_data) == 0 and len(_tuning_full_text) > 10:
            _t_text = get_template_job(_title, _tuning_full_text)
            _ok, json_data = get_all_job_info(_t_text)
            if not _ok:
                return False, f"通过大模型获取职位信息列表,从优化的内容里面Error:\n{json_str}"
            _tuning_md = fix_md_using_gpt_full(_tuning_full_text)

        # 职位列表后处理
        set_other_job_list(json_data)
        _ann_dict['jobs'] = json_data

    except json.JSONDecodeError as e:
        return False, f" 通过大模型获取职位信息列表总的 Error:\n{e}"

    # ========== 3. 附加信息 & MD优化 ==========
    _info['mdfile_path'] = _model_file
    _ann_dict['other'] = _info  # 附加信息

    # 读取原始HTML
    with open(_htmlfile, 'r', encoding='utf-8') as f:
        _html_file_content = f.read()

    # 根据页面类型选择MD生成策略
    if 'is_large_image' in _info and _info['is_large_image'] == 'OK' or 'type_url' in _info and _info['type_url'] == 'wxwz':
        common_process(_mdfile, _ann_dict, _tuning_md)
    elif 'is_external_link' in _info and _info['is_external_link'] == 'OK':
        _ann_dict['mdfile'] = fix_md_using_gpt_full(_text)
        ner_logger.info(f"使用大模型对全文进行markdown提取,{_htmlfile}")
    elif 'text_to_markdown' in sch_info and sch_info['text_to_markdown'] == 'OK':
        _ann_dict['mdfile'] = fix_md_using_gpt_full(_text)
        ner_logger.info(f"使用大模型含table的全文进行markdown提取,{_htmlfile}")
    elif ('has_table' in _info and _info['has_table'] == 'OK' or
          'html_to_markdown' in sch_info and sch_info['html_to_markdown'] == 'OK') and len(_tuning_full_text) == 0:
        if len(_html_file_content) < 20000:
            _ann_dict['mdfile'] = fix_md_using_gpt_table(_html_file_content)
            ner_logger.info(f"使用大模型对含table进行markdown提取,{_htmlfile}")
        else:
            _ann_dict['mdfile'] = fix_md_using_gpt_full(_text)
            ner_logger.info(f"使用大模型含table的全文进行markdown提取1,{_htmlfile}")
    else:
        common_process(_mdfile, _ann_dict, _tuning_md)

    # HTML内容(为空,节省空间)
    _ann_dict['htmlfile'] = ""

    # ========== 4. 输出最终模型文件 ==========
    if len(_ann_dict) >= 5:
        with open(_model_file, 'w', encoding='utf-8') as fw:
            json.dump(_ann_dict, fw, ensure_ascii=False, indent=4)
            ner_logger.info(f"处理文章生成模型文件成功,{_model_file}")
        # 清理临时文件
        os.remove(_mdfile)
        os.remove(_htmlfile)
        return True, f"{_mdfile}解析成功"
    else:
        return False, f"{_mdfile}解析失败"

# ====================== 通用MD处理 ======================
def common_process(_mdfile, _ann_dict, _tuning_md):
    """读取MD + GPT格式化 + 后处理"""
    _ann_dict['mdfile'] = get_md_content(_mdfile, "")
    if len(_ann_dict['mdfile']) < 20000:
        _ann_dict['mdfile'] = fix_md_using_gpt(_ann_dict['mdfile'])
    _ann_dict['mdfile'] = fix_md_after_gpt(_ann_dict['mdfile'], _tuning_md)

def common_process_fix(_mdfile, _ann_dict):
    """精简版MD格式化(白名单学校专用)"""
    _ann_dict['mdfile'] = get_md_content(_mdfile, "")
    _ann_dict['mdfile'] = fix_md_using_gpt_fix(_ann_dict['mdfile'])
    ner_logger.info(f"处理文章生成模型文件成功,{_ann_dict['mdfile']}")

# ====================== 职位列表解析 ======================
def get_all_job_info(_t_text):
    """调用大模型提取职位列表"""
    ner_logger.info(f"开始通过大模型获取职位信息列表 {_t_text}")
    (_ok_flag, json_str) = call_gpt(_t_text, True)
    if not _ok_flag:
        return False, {}
    json_data = json.loads(json_str, strict=False)

    # 兼容Map包裹单层List的情况
    if isinstance(json_data, dict) and len(json_data) == 1:
        for k, v in json_data.items():
            if isinstance(v, list):
                json_data = v
                break

    fix_diploma_data_list(json_data)  # 学历标准化
    json_data = fix_jobname_data_list(json_data)  # 过滤空职位名
    return True, json_data

# ====================== MD大模型优化 ======================
def fix_md_using_gpt(_content):
    """豆包:优化MD格式"""
    if len(_content) > 20000:
        return _content
    chi = re.findall(r'[\u4E00-\u9FFF]', _content)
    if len(chi) < 100:
        return _content
    _t_text = get_template_article(_content)
    ok, json_str = doubao_call_gpt(_t_text, True)
    if ok:
        try:
            json_data = json.loads(json_str, strict=False)
            if 'mdContent' in json_data and isinstance(json_data['mdContent'], str):
                return json_data['mdContent']
            elif 'mdcontent' in json_data and isinstance(json_data['mdcontent'], str):
                return json_data['mdcontent']
        except:
            ner_logger.error(f"fix_md_using_gpt Error:{json_str}")
    return _content

def fix_md_using_gpt_fix(_content):
    """千问:精简格式化(白名单专用)"""
    if len(_content) > 20000:
        return _content
    chi = re.findall(r'[\u4E00-\u9FFF]', _content)
    if len(chi) < 100:
        return _content
    _t_text = get_template_article_fix(_content)
    ok, json_str = qwen_call_gpt(_t_text, True)
    if ok:
        try:
            json_data = json.loads(json_str, strict=False)
            if 'mdContent' in json_data and isinstance(json_data['mdContent'], str):
                return json_data['mdContent']
        except:
            ner_logger.error(f"fix_md_using_gpt Error")
    return _content

def fix_md_using_gpt_full(_content):
    """全文→MD"""
    _t_text = get_template_full_html(_content)
    return fix_md_using_gpt_full_inner(_content, _t_text)

def fix_md_using_gpt_table(_content):
    """表格→MD"""
    _t_text = get_template_table_html(_content)
    return fix_md_using_gpt_full_inner(_content, _t_text)

def fix_md_using_gpt_full_inner(_content, _t_text):
    if len(_content) > 20000:
        return _content
    ok, json_str = doubao_call_gpt(_t_text, True)
    if ok:
        try:
            json_data = json.loads(json_str, strict=False)
            if 'mdContent' in json_data:
                return json_data['mdContent']
        except:
            ner_logger.error(f"fix_md_using_gpt full Error")
    return _content

# ====================== 数据填充与修复 ======================
def set_other_info(sch_info, _info, json_data, _text, _htmlfile):
    """填充来源、标题、MD5、发布时间、微信信息、地区、专业去重等"""
    json_data['FileId'] = getMD5Str(_text)
    json_data['AnnType'] = 'wx_ann' if _info['type_url'] == 'wxwz' else 'sch_ann'
    json_data['JobLink'] = _info['full_url']
    json_data['JobTitle'] = _info['announcement_name']
    json_data['JobDescribe'] = ""
    json_data['JobReq'] = ""

    # 页面提取的公司名覆盖
    if 'hd_company' in _info and _info['hd_company']:
        json_data['HdCompany'] = _info['hd_company']
        if not json_data.get('ComName'):
            json_data['ComName'] = json_data['HdCompany']

    # 公告标题覆盖
    if 'hd_ann' in _info and len(_info['hd_ann']) > 10:
        json_data['JobTitle'] = _info['hd_ann']

    # 发布时间
    if 'publish_time' in _info:
        json_data['PublishTime'] = _info['publish_time']

    # 微信相关信息
    json_data['WeixinId'] = _info.get('wx_id', '')
    json_data['WeixinName'] = _info.get('wx_name', sch_info.get('sch_name', ''))
    json_data['WeixinTitle'] = _info.get('wx_title', '')
    json_data['WeixinPublishTime'] = _info.get('wx_public_time', '')

    # 工作地点 = 公司地点(为空时)
    if json_data.get('WorkPlace') == '' and json_data.get('ComPlace') != '':
        json_data['WorkPlace'] = json_data['ComPlace']

    # 专业/职位名去重
    json_data['MajorRequirement'] = deduplicate_strings(json_data.get('MajorRequirement', ''))
    json_data['JobName'] = deduplicate_strings(json_data.get('JobName', ''))

    # 公告/职位类型智能修正
    if json_data.get('AnnouncementType') == "公告" and len(json_data['JobTitle']) <= 10:
        if any(k in json_data['JobTitle'] for k in ['人员','师','岗']):
            json_data['AnnouncementType'] = '职位'

    # 薪资标准化
    if json_data.get('Salary') in ['面谈', '待遇从优']:
        json_data['Salary'] = '面议'

    # 毕业年份处理
    fix_graduate_year(json_data)

    # 来源链接提取
    if _info['channel'] != 'sch_88888':
        if json_data.get('SourceLink'):
            _info['source_link'] = json_data['SourceLink']
        else:
            _ok, t, l = get_source_link(_htmlfile)
            if _ok:
                _info['source_link'] = l
                _info['source_link_text'] = t.replace("来源于", "")

    # 服务信息
    _info['server_ip'] = get_local_ip()
    _info['qz_version'] = QZ_VERISON
    _info['process_time'] = get_current_time_string()

def set_other_job_list(json_data):
    """职位列表清空长文本,节省性能"""
    for j in json_data:
        j['JobDescribe'] = ""
        j['JobReq'] = ""

# ====================== 二维码提取 ======================
def get_qrcode_info(info):
    """从图片解析结果中提取二维码链接和图片"""
    if 'img_urls' in info['props']:
        for k, m in info['props']['img_urls'].items():
            if m.get('full_qr') == 'Y':
                return {'qr_url': k, 'qr_pic': m.get('qz_img_url', k), 'all_qr_pics': {k: m.get('qz_img_url', k)}}
            if m.get('full_qr') == 'H' and 'inside_qr_link' in m:
                return {'qr_url': m['inside_qr_link'], 'qr_pic': m['inside_qr_pic'], 'all_qr_pics': m['all_qr_pics']}
    return {}

# ====================== 数据标准化 ======================
def fix_diploma_data_list(json_data):
    """列表学历标准化"""
    for x in json_data:
        if 'Degree' in x:
            x['Degree'] = fix_diploma(x['Degree'])

def fix_diploma_data_map(item):
    """单条学历标准化"""
    if 'Degree' in item:
        item['Degree'] = fix_diploma(item['Degree'])

def fix_jobname_data_list(json_data):
    """过滤空职位名"""
    return [j for j in json_data if j.get('JobTitle')]

def fix_graduate_year(json_data):
    """智能提取毕业年份:标题正则→当前时间自动补全"""
    gy = json_data.get('GraduationYear', [])
    if isinstance(gy, str):
        gy = [gy] if gy else []
    json_data['GraduationYear'] = gy

    # 从标题提取 2025/2026...
    if 'JobTitle' in json_data:
        matches = re.findall(r'202[5-9]', json_data['JobTitle'])
        if matches:
            json_data['GraduationYear'] += matches

    # 兜底:当前年(8月后+1)
    if not json_data['GraduationYear']:
        y = datetime.datetime.now().year
        if datetime.datetime.now().month >= 8:
            y += 1
        json_data['GraduationYear'] = [str(y)]

# ====================== 辅助工具 ======================
def get_wx_xqxx(_data):
    """获取微信预约链接"""
    if 'wx_code_file_config' in _data:
        f = _data['wx_code_file_config']
        if os.path.exists(f):
            return True, json.load(open(f)).get('yqym_url', '')
    return False, ""

def get_source_link(_htmlfile):
    """提取来源链接(微信/网络)"""
    with open(_htmlfile, encoding='utf-8') as f:
        html = f.read()
    for href, text in re.findall(r'<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>', html, re.S):
        if 'mp.weixin.qq.com' in href:
            return True, "来源于微信文章", href
    for href, text in re.findall(r'<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>', html, re.S):
        if '来源' in text:
            return True, "来源于网络", href
    return False, "", ""

项目分区导航ann_md ⬅️ | 03-ann_model | ➡️ ann_model_job