diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_dataset.txt b/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_dataset.txt new file mode 100644 index 0000000..eaa1250 --- /dev/null +++ b/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_dataset.txt @@ -0,0 +1,215 @@ +DROP TABLE IF EXISTS wanshifu_dw.ai_training_order_bidding_snapshot; +CREATE TABLE IF NOT EXISTS wanshifu_dw.ai_training_order_bidding_snapshot ( + global_order_trace_id STRING COMMENT '订单全局id', + order_no STRING COMMENT '订单编号', + order_submit_time STRING COMMENT '订单创建时间', + order_cancel_time STRING COMMENT '订单取消时间', + goods_level_1_name STRING COMMENT '订单商品一级类目名称', + order_goods_cnt BIGINT COMMENT '订单商品数量', + order_serve_type_name STRING COMMENT '订单服务类型名称', + order_business_line_name STRING COMMENT '订单业务线名称', + order_business_line_type STRING COMMENT '订单业务线类型', + order_appoint_type_name STRING COMMENT '指派类型', + order_etp_flag STRING COMMENT '下单总包标识:0不是,1是', + user_id STRING COMMENT '用户id', + user_name STRING COMMENT '用户名称', + address STRING COMMENT '经营详情地址', + business_full_name STRING COMMENT '企业全称', + company_type STRING COMMENT '企业类型(e_commerce:电商、offline_store:线下门店、factory:工厂、logistics:物流)', + user_type STRING COMMENT '用户类型(1:普通 2:企业 3:新版企业用户 4:小程序用户 5:个人)', + order_onsite_sign_time STRING COMMENT '上门时间', + mst_serve_complete_last_time STRING COMMENT '完工时间', + goods_level_2_name STRING COMMENT '订单商品二级类目名称', + goods_level_3_name STRING COMMENT '订单商品三级类目名称', + order_total_amount DOUBLE COMMENT '订单总金额', + order_unit_price DOUBLE COMMENT '订单单价', + is_urgent_flag STRING COMMENT '是否加急', + city_name STRING COMMENT '城市名称', + offer_mst_cnt BIGINT COMMENT '报价量(人次)', + view_mst_cnt BIGINT COMMENT '查看量(人次)', + fifth_offer_duration_second BIGINT COMMENT '满5人报价时长(第5人报价时间 - 下单时间)', + tenth_offer_duration_second BIGINT COMMENT '满10人报价时长(第10人报价时间 - 下单时间)', + attention_cnt BIGINT COMMENT '商家师傅关注数', + merchant_total_orders BIGINT COMMENT '商家下的总订单量', + merchant_total_aftersales BIGINT COMMENT '商家产生的总售后单量', + ignore_cnt BIGINT COMMENT '商家被拉黑次数', + buyer_note STRING COMMENT '订单备注' +) +STORED AS ALIORC +TBLPROPERTIES ( + 'comment' = 'AI训练用报价招标订单快照表(非分区表)' +); + + +INSERT OVERWRITE TABLE wanshifu_dw.ai_training_order_bidding_snapshot +SELECT t1.global_order_trace_id -- 订单全局id +,t1.order_no -- 订单编号 +,t1.order_submit_time -- 订单创建时间 +,t1.order_cancel_time -- 订单取消时间 +,t3.goods_level_1_name -- 订单商品一级类目名称 +,CAST(t1.order_goods_cnt AS BIGINT) AS order_goods_cnt -- 订单商品数量 +,t1.order_serve_type_name -- 订单服务类型名称 +,t1.order_business_line_name -- 订单业务线名称 +,t1.order_business_line_type -- 订单业务线类型 +,t1.order_appoint_type_name --指派类型 +,t1.order_etp_flag -- 下单总包标识:0不是,1是 +,t1.user_id -- 用户id +,t2.user_name -- 用户名称 +,t2.address -- 经营详情地址 +,t2.business_full_name -- 企业全称 +,t2.company_type -- 企业类型(e_commerce:电商、offline_store:线下门店、factory:工厂、logistics:物流) +,t2.user_type -- 用户类型(1:普通 2:企业 3:新版企业用户 4:小程序用户5:个人) +,order_onsite_sign_time --上门时间 +,mst_serve_complete_last_time --完工时间 +,t3.goods_level_2_name -- 订单商品二级类目名称 +,t3.goods_level_3_name -- 订单商品三级类目名称 +--,init_order_amt --初始主订单金额 +--,init_total_amt --初始订单总金额(主订单+子订单) +--,settle_amt --订单结算金额(服务费用) +,t9.init_serve_fee as order_total_amount --订单总金额 +,(t9.init_serve_fee/t1.order_goods_cnt) as order_unit_price --订单单价 +--,t4.order_label --订单种类(加急单、夜间单等等) +,CASE WHEN t4.order_label LIKE '%加急%' THEN '是' +ELSE '否' +END AS is_urgent_flag --是否加急 +,t10.city_name --城市名称 +,NVL(t5.offer_mst_cnt,0) AS offer_mst_cnt --报价量(人次) +,NVL(t6.view_mst_cnt,0) AS view_mst_cnt --查看量(人次) +,NVL(t7.fifth_offer_duration_second,0) AS fifth_offer_duration_second --满5人报价时长(第5人报价时间 - 下单时间) +,NVL(t7.tenth_offer_duration_second,0) AS tenth_offer_duration_second --满10人报价时长(第10人报价时间 - 下单时间) +,NVL(t8.attention_cnt,0) AS attention_cnt --商家师傅关注数 +,merchant_total_orders --- 商家下的总订单量 +,merchant_total_aftersales --- 商家产生的总售后单量 +,ignore_cnt --商家被拉黑次数 +,t10.buyer_note -- 订单备注 +FROM wanshifu_dw.dws_order_d t1 +LEFT JOIN wanshifu_dw.dim_usr_info_v t2 +ON t1.user_id = t2.user_id + +JOIN ( +SELECT t1.global_order_trace_id +,t1.goods_level_1_name +,t1.goods_level_2_name +,t1.goods_level_3_name +FROM wanshifu_dw.dwm_order_goods_info_d_v t1 +JOIN ( +SELECT global_order_trace_id +FROM wanshifu_dw.dwm_order_goods_info_d_v +WHERE stat_date IS NOT NULL +GROUP BY global_order_trace_id +HAVING COUNT(*) = 1 +) t2 +ON t1.global_order_trace_id = t2.global_order_trace_id +WHERE stat_date IS NOT NULL +) t3 --去掉多类目商品 +ON t1.global_order_trace_id = t3.global_order_trace_id + +LEFT JOIN ( +SELECT global_order_trace_id +,CONCAT_WS(',',COLLECT_SET(business_type_name)) AS order_label +FROM wanshifu_dw.dwd_bas_accounting_trade_settlement_flow_fee_detail_d_v +WHERE etl_date >= '20250101' +AND business_type_code IN ('emergency_order_settlement','fee_subsidy_settlement','fittings_order_settlement','night_order_settlement','order_settlement','price_diff_subsidy_settlement','rate_award_settlement','service_satisfy_award_settlement','toc_order_settlement') +GROUP BY global_order_trace_id +) t4 +ON t1.global_order_trace_id = t4.global_order_trace_id +LEFT JOIN ( +SELECT global_order_trace_id +,COUNT(DISTINCT master_id) AS offer_mst_cnt +FROM wanshifu_dw.dwd_mst_order_offer_price_d_v +WHERE etl_date >= '20250101' +GROUP BY global_order_trace_id +) t5 +ON t1.global_order_trace_id = t5.global_order_trace_id +LEFT JOIN ( +SELECT global_order_trace_id +,COUNT(DISTINCT IF(first_view_time IS NOT NULL,master_id,NULL)) AS view_mst_cnt +FROM wanshifu_dw.dwm_mst_order_push_d +WHERE etl_date >= '20250101' +GROUP BY global_order_trace_id +) t6 +ON t1.global_order_trace_id = t6.global_order_trace_id +LEFT JOIN ( +SELECT global_order_trace_id +,SUM(IF(rn = 5,DATEDIFF(offer_time,order_submit_time,'ss'),0)) AS fifth_offer_duration_second +,SUM(IF(rn = 10,DATEDIFF(offer_time,order_submit_time,'ss'),0)) AS tenth_offer_duration_second +FROM ( +SELECT global_order_trace_id +,order_submit_time +,offer_time +,ROW_NUMBER() OVER (PARTITION BY global_order_trace_id ORDER BY offer_time ASC ) rn +FROM wanshifu_dw.dwd_mst_order_offer_price_d_v +WHERE etl_date >= '20250101' +) +WHERE rn IN (5,10) +GROUP BY global_order_trace_id +) t7 +ON t1.global_order_trace_id = t7.global_order_trace_id + +LEFT JOIN ( +SELECT account_id +,COUNT(master_id) AS attention_cnt --商家师傅关注数 +FROM wanshifu_dw.ods_mst_info_mst_interaction_d_v +WHERE etl_date = '20250618' --此日期不可修改 +AND attention_status = '1' +GROUP BY account_id +) t8 +ON t1.user_id = t8.account_id + +LEFT JOIN ( +SELECT account_id +,COUNT(master_id) AS ignore_cnt --商家被拉黑次数 +FROM wanshifu_dw.ods_mst_info_mst_interaction_d_v +WHERE etl_date = '20250618' --此日期不可修改 +AND ignore_his_order_status = '1' +GROUP BY account_id +) t88 +ON t1.user_id = t88.account_id + +LEFT JOIN wanshifu_dw.dwd_usr_order_trade_info t9 +on t9.global_order_trace_id = t1.global_order_trace_id +LEFT JOIN wanshifu_dw.dim_order_info_v t10 +on t10.global_order_trace_id = t1.global_order_trace_id + +left JOIN +( +SELECT user_id,COUNT(distinct global_order_trace_id) as merchant_total_orders +FROM wanshifu_dw.dws_order_d t1 +WHERE SUBSTR(t1.stat_date,1,6) BETWEEN '202501' AND '202506' +AND t1.order_cancel_time IS NULL +and t1.mst_serve_complete_last_time is not null +GROUP BY user_id +) a +on t1.user_id = a.user_id +left JOIN +( +SELECT tt2.user_id,count(distinct tt1.global_order_trace_id) as merchant_total_aftersales +FROM ( + SELECT global_order_trace_id + FROM wanshifu_dw.dwd_iop_work_order_work_order_infos_v + WHERE work_order_type IN ('complaint','secondline') --投诉,二线 + AND TO_CHAR(work_order_create_time,'yyyymmdd') >= '20250101' + and TO_CHAR(work_order_create_time,'yyyymmdd') <= '20250630' + GROUP BY global_order_trace_id + UNION ALL + SELECT global_order_trace_id + FROM wanshifu_dw.dwm_order_arbitration_info --仲裁 + WHERE TO_CHAR(create_time,'yyyymmdd') >= '20250101' + AND TO_CHAR(create_time,'yyyymmdd') <= '20250630' + GROUP BY global_order_trace_id + ) tt1 +LEFT JOIN wanshifu_dw.dim_order_info_v tt2 +on tt1.global_order_trace_id = tt2.global_order_trace_id +GROUP BY tt2.user_id +) b +on t1.user_id = b.user_id + +WHERE SUBSTR(t1.stat_date,1,6) BETWEEN '202501' AND '202506' +AND t1.order_goods_lv1_name IN ('家具') +AND t1.order_business_line_type = 'tob' +AND t1.order_etp_flag = '0' +AND t1.order_cancel_time IS NULL +AND t1.order_appoint_type_name = '报价招标' +AND t1.order_2nd_onsite_time IS NULL --去掉二次上门 +and t10.city_name IN ('西安市','南京市','天津市','成都市','珠海市','佛山市','合肥市','青岛市','郑州市','长沙市','武汉市','宁波市','苏州市','东莞市','重庆市') +; \ No newline at end of file diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_training.py b/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_training.py new file mode 100644 index 0000000..3d2c4b7 --- /dev/null +++ b/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_training.py @@ -0,0 +1,3691 @@ +import os +import sys +import warnings +import argparse +import logging +import json +from datetime import datetime +import joblib +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +import seaborn as sns +from sklearn.cluster import KMeans +from sklearn.decomposition import PCA +from sklearn.metrics import silhouette_score +from sklearn.preprocessing import StandardScaler + +try: + import yaml # 配置化支持(可选) +except Exception: # yaml 非强依赖 + yaml = None + +# -------- 运行参数与日志初始化 -------- + + +def parse_args(): + parser = argparse.ArgumentParser(description="Order classification pipeline") + # 仅保留 predict 模式(纯在线重建) + parser.add_argument( + "--config", + type=str, + default=None, + help="YAML 配置文件路径(可选)", + ) + parser.add_argument( + "--log-level", + type=str, + default="INFO", + help="日志级别: DEBUG/INFO/WARNING/ERROR", + ) + parser.add_argument( + "--log-file", + type=str, + default=None, + help="日志文件输出路径(默认写入模型目录 pipeline.log)", + ) + return parser.parse_args(args=[a for a in sys.argv[1:] if a.strip()]) + + +def init_logging(log_file: str = None, level: str = "INFO"): + log_level = getattr(logging, level.upper(), logging.INFO) + logging.captureWarnings(True) + handlers = [logging.StreamHandler(sys.stdout)] + if log_file: + os.makedirs(os.path.dirname(log_file), exist_ok=True) + handlers.append(logging.FileHandler(log_file, encoding="utf-8")) + logging.basicConfig( + level=log_level, + format="%(asctime)s | %(levelname)s | %(message)s", + handlers=handlers, + force=True, + ) + logging.info(f"Logging initialized. level={level}, file={log_file}") + +warnings.filterwarnings("ignore") +# 设置中文字体 +plt.rcParams["font.sans-serif"] = ["SimHei", "Arial Unicode MS", "DejaVu Sans"] +plt.rcParams["axes.unicode_minus"] = False # 正常显示负号 +# ===== 统一配置管理类 ===== + + +class ModelConfig: + """ + 统一配置管理类 - 集中管理所有模型配置 + """ + def __init__(self): + # 混合评分配置 + self.hybrid_config = { + "rule_weight": 0.7, # 规则评分权重70% + "cluster_weight": 0.3, # 聚类评分权重30% + "good_quantile": 0.75, # 好单分位数:前15% (85分位数) + "medium_quantile": 0.50, # 中单分位数:前40% (60分位数) + "min_quality_threshold": 50.0, # 最低质量门槛 - 修复:降低门槛 + } + # 业务规则配置 + self.business_rules = { + "urgent_order_flag_value": 1, # 加急订单加分的判定值 + "large_orders_threshold": 10, # 大订单加分阈值 + "bonus_rules": { + # 优质企业加分 + "premium_companies": [ + "北欧表情(深圳)家具有限公司", + "西昊家具(深圳)有限公司", + "佛山林氏木业家具有限公司", + "顾家家居股份有限公司", + ], + # 优质地区加分 + "premium_regions": ["佛山", "东莞", "河北", "浙江"], + # 优质商品类别加分 + "premium_categories": ["办公","老板", "屏", "户外", "柜"], + # 优质服务类型加分 + "premium_services": ["送货到家并安装", "维修"], + }, + # 减分项配置 + "penalty_rules": { + # 问题地区减分 + "problem_regions": ["徐州"] + }, + # 规则权重配置(缩小到20%以内) + "rule_weights": { + "premium_company_bonus": 0.8, # 优质企业加分权重 + "premium_region_bonus": 0.6, # 优质地区加分权重 + "premium_category_bonus": 0.6, # 优质商品类别加分权重 + "premium_service_bonus": 0.6, # 优质服务类型加分权重 + "urgent_order_bonus": 0.6, # 加急订单加分权重 + "large_order_bonus": 0.6, # 大订单加分权重 + "problem_region_penalty": -0.4, # 问题地区减分权重 + }, + } + # 路径配置 + self.paths = { + "data_file": "/Users/tom/Documents/data.csv", + "model_dir": "order_cluster_model", + "output_file": "/Users/tom/Documents/data_check_with_predictions.xlsx", + } + # 模型参数配置 + self.model_params = { + "n_clusters": 6, + "random_state": 42, + "min_samples": 50, + "value_threshold": 10.0, + } + # 特征配置 + self.features = { + "static_base": [ + "order_goods_cnt", + "order_total_amount", + "order_unit_price", + "buyer_note_100", + "submit_hour", + "submit_weekday", + "business_rule_score", + "has_price_info", + ], + "dynamic": [ + "offer_mst_cnt", + "view_mst_cnt", + "fifth_offer_duration_second", + "tenth_offer_duration_second", + "attention_cnt", + "onsite_to_finish_hour", + "serve_efficiency", + ], + } + + def get_hybrid_config(self): + """获取混合评分配置""" + return self.hybrid_config + + def get_business_rules(self): + """获取业务规则配置""" + return self.business_rules + + def get_paths(self): + """获取路径配置""" + return self.paths + + def get_model_params(self): + """获取模型参数配置""" + return self.model_params + + def get_features(self): + """获取特征配置""" + return self.features + + def update_hybrid_config(self, **kwargs): + """更新混合评分配置""" + self.hybrid_config.update(kwargs) + + def update_business_rules(self, **kwargs): + """更新业务规则配置""" + self.business_rules.update(kwargs) + + def save_configs(self, model_dir): + """保存所有配置到文件""" + os.makedirs(model_dir, exist_ok=True) + # 保存混合评分配置 + joblib.dump(self.hybrid_config, os.path.join(model_dir, "hybrid_config.pkl")) + # 保存业务规则配置 + joblib.dump(self.business_rules, os.path.join(model_dir, "business_rules.pkl")) + print("✅ 配置已保存到模型目录") + + +# 创建全局配置实例(支持从YAML覆盖) +args = parse_args() +config = ModelConfig() +if args.config and yaml is not None and os.path.exists(args.config): + try: + with open(args.config, "r", encoding="utf-8") as fh: + y = yaml.safe_load(fh) or {} + if isinstance(y, dict): + if "hybrid_config" in y: + config.update_hybrid_config(**y["hybrid_config"]) + if "business_rules" in y: + config.update_business_rules(**y["business_rules"]) + if "paths" in y and isinstance(y["paths"], dict): + config.paths.update(y["paths"]) # 路径配置覆盖 + if "model_params" in y and isinstance(y["model_params"], dict): + config.model_params.update(y["model_params"]) # 模型参数覆盖 + except Exception as e: + print(f"⚠️ 读取配置失败: {e},继续使用内置默认配置") + +# 初始化日志 +log_file_default = os.path.join(config.get_paths()["model_dir"], "pipeline.log") +init_logging(args.log_file or log_file_default, args.log_level) +logging.info("Start pipeline in predict-only mode (pure online reconstruction)") +# 统一使用配置实例,避免重复定义 +HYBRID_CONFIG = config.get_hybrid_config() +BUSINESS_RULES = config.get_business_rules() +MODEL_DIR = config.get_paths()["model_dir"] +DYNAMIC_FEATURES = config.get_features()["dynamic"] +# --- 1. 数据准备与特征工程 --- +print("\n[Part 1] 数据准备与特征工程...") +# 确保模型目录存在 +os.makedirs(MODEL_DIR, exist_ok=True) +# 初始化聚类基础分(后面会动态计算) +cluster_base_scores = {} +print( + "--- 方案A:基于静态特征的订单分层模型(升级版:含群体统计特征 + 业务规则特征)---" +) +# --- 2. 模型训练与聚类 --- +print("\n[Part 2] 模型训练与聚类...") +# 读取数据 +try: + df = pd.read_csv("/Users/tom/Documents/data.csv") +except FileNotFoundError: + print("错误:数据文件'/Users/tom/Documents/data.csv'未找到。请检查路径。") + exit() +# 检查原始数据 order_no 缺失情况 +if "order_no" in df.columns: + print(f"原始数据 order_no 缺失数: {df['order_no'].isna().sum()} 条") +else: + print("原始数据中未找到 order_no 字段!") +# 筛选有效订单:只保留已完成的订单 +print("正在筛选有效订单...") +df_valid = df[df["mst_serve_complete_last_time"].notna()].copy() +if "order_no" in df_valid.columns: + print(f"有效订单 order_no 缺失数: {df_valid['order_no'].isna().sum()} 条") +# 剔除金额为0的订单 +print("\n正在剔除金额为0的订单...") +zero_amount_count = len(df_valid[df_valid["order_total_amount"] == 0]) +print(f"金额为0的订单数: {zero_amount_count:,} 单") +df_valid = df_valid[df_valid["order_total_amount"] > 0].copy() +print(f"剔除后订单数: {len(df_valid):,} 单") +# 过滤三级类目数量不足的订单 +print("\n正在过滤三级类目数量不足的订单...") +original_count = len(df_valid) +print(f"过滤前订单数: {original_count:,} 单") +# 计算每个三级类目的订单数量 +category_counts = df_valid["goods_level_3_name"].value_counts() +print(f"三级类目总数: {len(category_counts)} 个") +# 找出订单数量>=10的三级类目 +valid_categories = category_counts[category_counts >= 10].index +print(f"订单数量>=10的三级类目: {len(valid_categories)} 个") +print(f"订单数量<10的三级类目: {len(category_counts) - len(valid_categories)} 个") +# 过滤数据:只保留订单数量>=10的三级类目 +df_valid = df_valid[df_valid["goods_level_3_name"].isin(valid_categories)].copy() +filtered_count = len(df_valid) +removed_count = original_count - filtered_count +print(f"过滤后订单数: {filtered_count:,} 单") +print(f"删除订单数: {removed_count:,} 单 ({removed_count/original_count*100:.2f}%)") +print(f"保留订单比例: {filtered_count/original_count*100:.2f}%") +# 显示删除的三级类目统计 +if removed_count > 0: + removed_categories = category_counts[category_counts < 10] + print(f"\n删除的三级类目分布(订单数<10):") + print(f" • 1单类目: {len(removed_categories[removed_categories == 1])} 个") + print( + f" • 2-3单类目: {len(removed_categories[(removed_categories >= 2) & (removed_categories <= 3)])} 个" + ) + print( + f" • 4-6单类目: {len(removed_categories[(removed_categories >= 4) & (removed_categories <= 6)])} 个" + ) + print( + f" • 7-9单类目: {len(removed_categories[(removed_categories >= 7) & (removed_categories <= 9)])} 个" + ) +# 检查订单指派类型的唯一值 +print(f"\n订单指派类型分布:") +print(df_valid["order_appoint_type_name"].value_counts()) +# 分离业务模式(根据实际数据值) +if "报价招标" in df_valid["order_appoint_type_name"].values: + df_bidding = df_valid[df_valid["order_appoint_type_name"] == "报价招标"].copy() + df_fixed = df_valid[df_valid["order_appoint_type_name"] == "一口价"].copy() + print(f"报价招标订单数: {len(df_bidding)}") + print(f"一口价订单数: {len(df_fixed)}") + # 选择当前要建模的业务模式(这里以报价招标为例) + if len(df_bidding) > 0: + print("\n当前建模业务模式:报价招标订单") + df_model_source = df_bidding.copy() + else: + print("\n报价招标订单数量为0,改为建模一口价订单") + df_model_source = df_fixed.copy() + df_bidding = df_fixed.copy() +else: + # 如果字段值不是预期的,使用所有有效订单 + print("\n未找到预期的订单类型,使用所有有效订单进行建模") + df_bidding = df_valid.copy() + df_model_source = df_bidding.copy() +# --- 特征工程 --- +print("\n正在进行特征工程...") +# 查看数据字段 +print("数据字段列表:") +print(df_model_source.columns.tolist()) +print(f"数据形状: {df_model_source.shape}") +# 1. 时间特征工程 +print("正在提取时间特征...") +df_model_source["order_submit_time"] = pd.to_datetime( + df_model_source["order_submit_time"], errors="coerce" +) +df_model_source["submit_hour"] = df_model_source["order_submit_time"].dt.hour +df_model_source["submit_weekday"] = df_model_source["order_submit_time"].dt.weekday +df_model_source["submit_is_weekend"] = ( + df_model_source["submit_weekday"].isin([5, 6]).astype(int) +) +df_model_source["submit_is_business_hour"] = ( + (df_model_source["submit_hour"] >= 9) & (df_model_source["submit_hour"] <= 18) +).astype(int) +# 2. 商品特征工程 +print("正在提取商品特征...") +df_model_source["goods_level_1_name"] = df_model_source["goods_level_1_name"].astype( + str +) +df_model_source["goods_level_2_name"] = df_model_source["goods_level_2_name"].astype( + str +) +df_model_source["goods_level_3_name"] = df_model_source["goods_level_3_name"].astype( + str +) +# 3. 业务规则特征工程 +print("正在计算业务规则特征...") + + +def calculate_business_rule_score(row, rules_config): + """ + + 根据业务规则计算订单的加分减分 + + """ + score = 0.0 + # 检查优质企业加分 + if "business_full_name" in row and pd.notna(row["business_full_name"]): + if ( + row["business_full_name"] + in rules_config["bonus_rules"]["premium_companies"] + ): + score += rules_config["rule_weights"]["premium_company_bonus"] + # 检查优质地区加分 + if "address" in row and pd.notna(row["address"]): + address_str = str(row["address"]).lower() + for region in rules_config["bonus_rules"]["premium_regions"]: + if region.lower() in address_str: + score += rules_config["rule_weights"]["premium_region_bonus"] + break + # 检查优质商品类别加分(改为使用三级类目) + if "goods_level_3_name" in row and pd.notna(row["goods_level_3_name"]): + category_str = str(row["goods_level_3_name"]).lower() + for category in rules_config["bonus_rules"]["premium_categories"]: + if category.lower() in category_str: + score += rules_config["rule_weights"]["premium_category_bonus"] + break + # 检查优质服务类型加分 + if "order_serve_type_name" in row and pd.notna(row["order_serve_type_name"]): + service_str = str(row["order_serve_type_name"]).lower() + for service in rules_config["bonus_rules"]["premium_services"]: + if service.lower() in service_str: + score += rules_config["rule_weights"]["premium_service_bonus"] + break + # 检查加急订单加分 + if ( + "is_urgent_order" in row + and row["is_urgent_order"] == rules_config["urgent_order_flag_value"] + ): + score += rules_config["rule_weights"]["urgent_order_bonus"] + # 检查大订单加分 + if "order_goods_cnt" in row and pd.notna(row["order_goods_cnt"]): + if row["order_goods_cnt"] >= rules_config["large_orders_threshold"]: + score += rules_config["rule_weights"]["large_order_bonus"] + # 检查问题地区减分 + if "address" in row and pd.notna(row["address"]): + address_str = str(row["address"]).lower() + for region in rules_config["penalty_rules"]["problem_regions"]: + if region.lower() in address_str: + score += rules_config["rule_weights"]["problem_region_penalty"] + break + return score + + +# 基于 user_id 的单维度商家属性映射覆盖(训练也不直接用行级现值) +user_attr_fields = [ + "user_name", + "address", + "business_full_name", + "company_type", + "user_type", + "attention_cnt", + "merchant_total_orders", + "merchant_total_aftersales", + "ignore_cnt", +] +available_user_attr_fields = [f for f in user_attr_fields if f in df_model_source.columns] +if available_user_attr_fields: + try: + ua_df = df_model_source.groupby("user_id")[available_user_attr_fields].last() + for field in available_user_attr_fields: + df_model_source[field] = df_model_source["user_id"].map(ua_df[field]) + print(f" - 已基于user_id覆盖商家属性字段: {available_user_attr_fields}") + except Exception as e: + print(f" ⚠️ 商家属性覆盖失败: {e}") + +# 计算业务规则得分(使用覆盖后的商家属性) +print(" - 正在计算业务规则得分...") +df_model_source["business_rule_score"] = df_model_source.apply( + lambda row: calculate_business_rule_score(row, BUSINESS_RULES), axis=1 +) +# 统计业务规则得分分布 +print(f" - 业务规则得分统计:") +print(f" 最小值: {df_model_source['business_rule_score'].min():.2f}") +print(f" 最大值: {df_model_source['business_rule_score'].max():.2f}") +print(f" 平均值: {df_model_source['business_rule_score'].mean():.2f}") +print(f" 标准差: {df_model_source['business_rule_score'].std():.2f}") +# 4. 群体统计特征工程(Category Prior Features) +print("正在计算群体统计特征(goods_level_3_name & user_id组合优先)...") +# === 重新设计的特征体系 === +# 静态基础特征(订单提交时就有的)——按锚点收紧:价值类现值不入模,仅用先验 +STATIC_BASE_FEATURES = [ + # 订单基础属性(不含现值金额与单价) + "order_goods_cnt", + "buyer_note_100", + # 时间特征 + "submit_hour", + "submit_weekday", + "submit_is_weekend", + "submit_is_business_hour", + # 业务规则特征 + "business_rule_score", + # 业务模式标识(使用先验金额判断) + "has_price_info", +] +# 主参考特征(用于计算历史先验) +MAIN_REFERENCE_FEATURES = [ + "offer_rate", # 查看报价率 + "fifth_offer_duration_second", # 满5人报价时长 + "tenth_offer_duration_second", # 满10人报价时长 + "onsite_to_finish_hour", # 完工时长 + "order_total_amount", # 总金额 + "order_unit_price", # 单价 +] +# 次参考特征(用于计算历史先验) +SECONDARY_REFERENCE_FEATURES = [ + "attention_cnt", # 师傅关注数 + "merchant_aftersale_rate", # 商家售后率 + "ignore_cnt", # 商家被拉黑数 +] +# 所有后验特征 +ALL_REFERENCE_FEATURES = MAIN_REFERENCE_FEATURES + SECONDARY_REFERENCE_FEATURES +# 生成静态基础特征 +print("正在生成静态基础特征...") +# 1. buyer_note_100 +if "buyer_note" in df_model_source.columns: + df_model_source["buyer_note_100"] = ( + df_model_source["buyer_note"] + .astype(str) + .apply(lambda x: 1 if len(x) > 100 else 0) + ) +else: + df_model_source["buyer_note_100"] = 0 +# 2. 时间特征(如果还没有的话) +if "submit_hour" not in df_model_source.columns: + df_model_source["submit_hour"] = df_model_source["order_submit_time"].dt.hour +if "submit_weekday" not in df_model_source.columns: + df_model_source["submit_weekday"] = df_model_source["order_submit_time"].dt.weekday +if "submit_is_weekend" not in df_model_source.columns: + df_model_source["submit_is_weekend"] = ( + df_model_source["submit_weekday"].isin([5, 6]).astype(int) + ) +if "submit_is_business_hour" not in df_model_source.columns: + df_model_source["submit_is_business_hour"] = ( + (df_model_source["submit_hour"] >= 9) & (df_model_source["submit_hour"] <= 18) + ).astype(int) +# 3. 业务模式标识改为基于先验金额设置(稍后先验生成后再设置) +df_model_source["has_price_info"] = 0 +# 生成后验参考特征(用于历史先验计算) +print("正在生成后验参考特征...") +# 1. 报价率 - 安全计算,避免除零错误 +if ( + "offer_mst_cnt" in df_model_source.columns + and "view_mst_cnt" in df_model_source.columns +): + # 确保分母不为0,并处理缺失值 + view_cnt_safe = df_model_source["view_mst_cnt"].fillna(0).replace(0, 1) + offer_cnt_safe = df_model_source["offer_mst_cnt"].fillna(0) + df_model_source["offer_rate"] = offer_cnt_safe / view_cnt_safe + # 确保结果在合理范围内 [0, 1] + df_model_source["offer_rate"] = df_model_source["offer_rate"].clip(0, 1) +else: + print(" ⚠️ 缺少报价相关字段,offer_rate设为0") + df_model_source["offer_rate"] = 0 +# 2. onsite_to_finish_hour - 安全计算,处理异常值 +if ( + "mst_serve_complete_last_time" in df_model_source.columns + and "order_onsite_sign_time" in df_model_source.columns +): + try: + complete_time = pd.to_datetime( + df_model_source["mst_serve_complete_last_time"], errors="coerce" + ) + onsite_time = pd.to_datetime( + df_model_source["order_onsite_sign_time"], errors="coerce" + ) + # 计算时间差(小时) + time_diff = (complete_time - onsite_time).dt.total_seconds() / 3600 + # 处理异常值:负值设为0,超过7天(168小时)的设为168 + time_diff = time_diff.fillna(0) # NaT设为0 + time_diff = time_diff.clip(lower=0, upper=168) # 限制在合理范围 + df_model_source["onsite_to_finish_hour"] = time_diff + print( + f" - 完工时长计算完成,范围: {time_diff.min():.1f} - {time_diff.max():.1f} 小时" + ) + except Exception as e: + print(f" ⚠️ 完工时长计算失败: {e},设为0") + df_model_source["onsite_to_finish_hour"] = 0 +else: + print(" ⚠️ 缺少完工时间相关字段,onsite_to_finish_hour设为0") + df_model_source["onsite_to_finish_hour"] = 0 +# 3. merchant_aftersale_rate - 安全计算,处理异常值 +if ( + "merchant_total_aftersales" in df_model_source.columns + and "merchant_total_orders" in df_model_source.columns +): + # 确保分母不为0,并处理缺失值 + total_orders_safe = df_model_source["merchant_total_orders"].fillna(0).replace(0, 1) + total_aftersales_safe = df_model_source["merchant_total_aftersales"].fillna(0) + df_model_source["merchant_aftersale_rate"] = ( + total_aftersales_safe / total_orders_safe + ) + # 确保结果在合理范围内 [0, 1] + df_model_source["merchant_aftersale_rate"] = df_model_source[ + "merchant_aftersale_rate" + ].clip(0, 1) + print( + f" - 商家售后率计算完成,范围: {df_model_source['merchant_aftersale_rate'].min():.3f} - {df_model_source['merchant_aftersale_rate'].max():.3f}" + ) +else: + print(" ⚠️ 缺少商家售后相关字段,merchant_aftersale_rate设为0") + df_model_source["merchant_aftersale_rate"] = 0 +# 4. ignore_cnt +if "ignore_cnt" not in df_model_source.columns: + df_model_source["ignore_cnt"] = 0 +# 检查可用的后验特征 +available_reference_features = [ + f for f in ALL_REFERENCE_FEATURES if f in df_model_source.columns +] +print(f" - 可用后验特征: {available_reference_features}") +def calculate_robust_quantiles(data: pd.Series, quantiles: list, feature_name: str): + """ + + 计算稳健分位数,自动处理极端值影响 + + """ + print( + f" 📊 {feature_name} 原始数据: N={len(data):,}, 范围=[{data.min():.2f}, {data.max():.2f}]" + ) + # 第1层:基础有效性过滤 + valid_data = data[data > 0].copy() # 移除0值和负值 + print( + f" 🔍 有效值过滤: N={len(valid_data):,} (移除{len(data)-len(valid_data):,}个≤0值)" + ) + if len(valid_data) < 10: + print(f" ⚠️ 有效数据不足,使用原始数据") + valid_data = data.copy() + # 第2层:IQR极端值过滤 + Q1 = valid_data.quantile(0.25) + Q3 = valid_data.quantile(0.75) + IQR = Q3 - Q1 + lower_bound = Q1 - 1.5 * IQR + upper_bound = Q3 + 1.5 * IQR + iqr_filtered = valid_data[(valid_data >= lower_bound) & (valid_data <= upper_bound)] + outliers_removed = len(valid_data) - len(iqr_filtered) + print(f" 🛡️ IQR过滤: N={len(iqr_filtered):,} (移除{outliers_removed:,}个极端值)") + # 确保有足够数据计算分位数 + if len(iqr_filtered) < 10: + print(f" 🚨 稳健数据不足10个,回退到有效数据") + robust_data = valid_data + else: + robust_data = iqr_filtered + # 计算稳健分位数 + result_quantiles = {} + for q in quantiles: + q_value = robust_data.quantile(q) + result_quantiles[f"{q:.0%}"] = q_value + print(f" {q:.0%}分位数: {q_value:.2f}") + return result_quantiles + + +# 动态阈值计算系统将移至先验特征生成之后 +# 计算优化的先验统计(双维度+单维度,无全局兜底) +print("正在计算优化的先验统计(双维度→单维度回退,无全局兜底)...") +# 计算统计数据 +prior_stats = {} +category_stats = {} +global_stats = {} +for feature in available_reference_features: + print(f" - 正在计算 {feature} 的统计...") + # 只对有效数据计算统计 + valid_data = df_model_source[df_model_source[feature].notna()] + if len(valid_data) > 0: + # 1. 双维度统计:user_id + goods_level_3_name (改用中位数) + dual_stats = ( + valid_data.groupby(["user_id", "goods_level_3_name"])[feature] + .agg(["median", "std", "count"]) + .fillna(0) + ) + dual_stats.columns = [f"{feature}_median", f"{feature}_std", f"{feature}_count"] + prior_stats[feature] = dual_stats + # 2. 单维度统计:goods_level_3_name (改用中位数) + single_stats = ( + valid_data.groupby("goods_level_3_name")[feature] + .agg(["median", "std", "count"]) + .fillna(0) + ) + single_stats.columns = [ + f"{feature}_median", + f"{feature}_std", + f"{feature}_count", + ] + category_stats[feature] = single_stats + # 3. 全局统计:整个数据集的中位数 + global_median = valid_data[feature].median() + global_stats[feature] = {"median": global_median} +print( + f" - 先验统计完成,共 {len(available_reference_features)} 个特征(双维度+单维度+全局统计)" +) +# 生成历史先验特征(优化回退策略:双维度→单维度→零值) +print("正在生成历史先验特征...") + + +def get_prior_feature_value( + user_id, goods_l3, feature_name, prior_stats, category_stats, global_stats=None +): + """ + + 优化的先验特征值获取:双维度→单维度回退,无全局兜底 + + 注意:训练时user_id是数字类型,线上传入的是字符串,需要转换 + + """ + # 转换user_id为数字类型,匹配训练时的数据类型 + try: + user_id_num = int(user_id) if user_id else 0 + except (ValueError, TypeError): + user_id_num = 0 + # 1. 尝试双维度:user_id + goods_level_3_name (样本>=3) + if feature_name in prior_stats: + dual_stats_df = prior_stats[feature_name] + dual_key = (user_id_num, goods_l3) + if ( + dual_key in dual_stats_df.index + and dual_stats_df.loc[dual_key, f"{feature_name}_count"] >= 3 + ): + return dual_stats_df.loc[dual_key, f"{feature_name}_median"] + # 2. 回退到单维度:goods_level_3_name + if feature_name in category_stats: + single_stats_df = category_stats[feature_name] + if goods_l3 in single_stats_df.index: + return single_stats_df.loc[goods_l3, f"{feature_name}_median"] + # 3. 最终返回0(不使用全局统计) + return 0 + + +# 为每个训练样本生成先验特征(向量化优化) +print(" - 正在向量化生成先验特征...") + + +def generate_prior_features_vectorized( + df, features, prior_stats, category_stats, global_stats=None +): + """向量化生成先验特征(修复用户ID类型不一致问题)""" + result_df = df.copy() + for feature in features: + print(f" - 正在生成 {feature} 的先验特征...") + # 统一用户ID类型:保持数字类型,与prior_stats的索引一致 + user_ids = df["user_id"].astype(int) # 修复:统一为数字类型 + goods_l3s = df["goods_level_3_name"].astype(str) + prior_values = np.zeros(len(df)) + # 1. 双维度匹配(修复类型匹配问题) + if feature in prior_stats: + dual_stats_df = prior_stats[feature] + for i in range(len(df)): + dual_key = (user_ids.iloc[i], goods_l3s.iloc[i]) + if ( + dual_key in dual_stats_df.index + and dual_stats_df.loc[dual_key, f"{feature}_count"] >= 3 + ): + prior_values[i] = dual_stats_df.loc[dual_key, f"{feature}_median"] + # 2. 单维度回退 + if feature in category_stats: + single_stats_df = category_stats[feature] + mask = (prior_values == 0) & (goods_l3s.isin(single_stats_df.index)) + for goods_l3 in goods_l3s[mask].unique(): + if goods_l3 in single_stats_df.index: + goods_mask = (goods_l3s == goods_l3) & (prior_values == 0) + prior_values[goods_mask] = single_stats_df.loc[ + goods_l3, f"{feature}_median" + ] + # 3. 不再全局兜底,剩余为0 + result_df[f"{feature}_prior"] = prior_values + return result_df + + +# 使用向量化函数生成先验特征 +df_model_source = generate_prior_features_vectorized( + df_model_source, + available_reference_features, + prior_stats, + category_stats, + global_stats, +) +# 生成先验特征列表 +prior_features = [f"{f}_prior" for f in available_reference_features] +print(f" - 生成的先验特征: {prior_features}") +# === 基于先验特征的动态阈值计算系统(不使用现值特征) === +print(f"\n🔧 正在基于先验特征计算规则评分动态阈值...") +global DYNAMIC_RULE_THRESHOLDS +DYNAMIC_RULE_THRESHOLDS = {} + +def _safe_series(df, col): + return df[col] if col in df.columns else pd.Series([], dtype=float) + +# 1. 总金额先验阈值 +amount_prior_series = _safe_series(df_model_source, "order_total_amount_prior") +if len(amount_prior_series) > 0: + amount_quantiles = calculate_robust_quantiles( + amount_prior_series, [0.2, 0.4, 0.6, 0.8, 0.95], "order_total_amount_prior" + ) + DYNAMIC_RULE_THRESHOLDS["order_total_amount"] = { + 15: amount_quantiles.get("95%", 0), + 12: amount_quantiles.get("80%", 0), + 9: amount_quantiles.get("60%", 0), + 6: amount_quantiles.get("40%", 0), + 3: amount_quantiles.get("20%", 0), + 1: 0, + } + +# 2. 单价先验阈值 +unit_price_prior_series = _safe_series(df_model_source, "order_unit_price_prior") +if len(unit_price_prior_series) > 0: + unit_price_quantiles = calculate_robust_quantiles( + unit_price_prior_series, [0.2, 0.4, 0.6, 0.8, 0.95], "order_unit_price_prior" + ) + DYNAMIC_RULE_THRESHOLDS["order_unit_price"] = { + 15: unit_price_quantiles.get("95%", 0), + 12: unit_price_quantiles.get("80%", 0), + 9: unit_price_quantiles.get("60%", 0), + 6: unit_price_quantiles.get("40%", 0), + 3: unit_price_quantiles.get("20%", 0), + 1: 0, + } + +# 3. 满10人报价时长先验阈值(越小越好) +tenth_prior_series = _safe_series(df_model_source, "tenth_offer_duration_second_prior") +if len(tenth_prior_series) > 0: + tenth_duration_quantiles = calculate_robust_quantiles( + tenth_prior_series, [0.05, 0.2, 0.4, 0.6, 0.8], "tenth_offer_duration_second_prior" + ) + DYNAMIC_RULE_THRESHOLDS["tenth_offer_duration_second"] = { + 20: tenth_duration_quantiles.get("5%", float("inf")), + 16: tenth_duration_quantiles.get("20%", float("inf")), + 12: tenth_duration_quantiles.get("40%", float("inf")), + 8: tenth_duration_quantiles.get("60%", float("inf")), + 4: tenth_duration_quantiles.get("80%", float("inf")), + 1: float("inf"), + } + +# 4. 满5人报价时长先验阈值(越小越好) +fifth_prior_series = _safe_series(df_model_source, "fifth_offer_duration_second_prior") +if len(fifth_prior_series) > 0: + fifth_duration_quantiles = calculate_robust_quantiles( + fifth_prior_series, [0.05, 0.2, 0.4, 0.6, 0.8], "fifth_offer_duration_second_prior" + ) + DYNAMIC_RULE_THRESHOLDS["fifth_offer_duration_second"] = { + 10: fifth_duration_quantiles.get("5%", float("inf")), + 8: fifth_duration_quantiles.get("20%", float("inf")), + 6: fifth_duration_quantiles.get("40%", float("inf")), + 3: fifth_duration_quantiles.get("60%", float("inf")), + 1: fifth_duration_quantiles.get("80%", float("inf")), + } + +# 5. 完工时长先验阈值(越小越好) +finish_prior_series = _safe_series(df_model_source, "onsite_to_finish_hour_prior") +if len(finish_prior_series) > 0: + finish_time_quantiles = calculate_robust_quantiles( + finish_prior_series, [0.05, 0.2, 0.4, 0.6, 0.8], "onsite_to_finish_hour_prior" + ) + DYNAMIC_RULE_THRESHOLDS["onsite_to_finish_hour"] = { + 10: finish_time_quantiles.get("5%", float("inf")), + 8: finish_time_quantiles.get("20%", float("inf")), + 6: finish_time_quantiles.get("40%", float("inf")), + 4: finish_time_quantiles.get("60%", float("inf")), + 2: finish_time_quantiles.get("80%", float("inf")), + 1: float("inf"), + } + +# 6. 查看报价率先验阈值(越大越好) +offer_rate_prior_series = _safe_series(df_model_source, "offer_rate_prior") +if len(offer_rate_prior_series) > 0: + offer_rate_quantiles = calculate_robust_quantiles( + offer_rate_prior_series, [0.2, 0.4, 0.6, 0.8, 0.95], "offer_rate_prior" + ) + DYNAMIC_RULE_THRESHOLDS["offer_rate"] = { + 15: offer_rate_quantiles.get("95%", 0), + 12: offer_rate_quantiles.get("80%", 0), + 9: offer_rate_quantiles.get("60%", 0), + 6: offer_rate_quantiles.get("40%", 0), + 3: offer_rate_quantiles.get("20%", 0), + 0: 0, + } + +# 7. 商家售后率先验阈值(越小越好) +aftersale_prior_series = _safe_series(df_model_source, "merchant_aftersale_rate_prior") +if len(aftersale_prior_series) > 0: + aftersale_quantiles = calculate_robust_quantiles( + aftersale_prior_series, [0.05, 0.2, 0.4, 0.6, 0.8], "merchant_aftersale_rate_prior" + ) + DYNAMIC_RULE_THRESHOLDS["merchant_aftersale_rate"] = { + 8: aftersale_quantiles.get("5%", 0), + 6: aftersale_quantiles.get("20%", 0), + 4: aftersale_quantiles.get("40%", 0), + 2: aftersale_quantiles.get("60%", 0), + 0: aftersale_quantiles.get("80%", 0), + } + +# 8. 师傅关注数先验(区间) +attention_prior_series = _safe_series(df_model_source, "attention_cnt_prior") +if len(attention_prior_series) > 0: + attention_quantiles = calculate_robust_quantiles( + attention_prior_series, [0.1, 0.3, 0.5, 0.7, 0.9], "attention_cnt_prior" + ) + optimal_min = attention_quantiles.get("30%", 0) + optimal_max = attention_quantiles.get("70%", 0) + DYNAMIC_RULE_THRESHOLDS["attention_cnt"] = { + "optimal_range": (optimal_min, optimal_max), + "general_range": ( + attention_quantiles.get("10%", 0), + attention_quantiles.get("90%", 0), + ), + } + +# 9. 被拉黑数先验(越小越好) +ignore_prior_series = _safe_series(df_model_source, "ignore_cnt_prior") +if len(ignore_prior_series) > 0: + ignore_quantiles = calculate_robust_quantiles( + ignore_prior_series, [0.5, 0.7, 0.85, 0.95], "ignore_cnt_prior" + ) + DYNAMIC_RULE_THRESHOLDS["ignore_cnt"] = { + 5: 0, + 3: ignore_quantiles.get("70%", 0), + 1: ignore_quantiles.get("85%", 0), + 0: ignore_quantiles.get("95%", 0), + } + +print(f"✅ 基于先验的动态规则阈值计算完成!特征数: {len(DYNAMIC_RULE_THRESHOLDS)}") +joblib.dump( + DYNAMIC_RULE_THRESHOLDS, os.path.join(MODEL_DIR, "dynamic_rule_thresholds.pkl") +) +print(f"💾 动态规则阈值已保存至: dynamic_rule_thresholds.pkl") +# 基于先验金额设置 has_price_info(价值类只用先验) +if "order_total_amount_prior" in df_model_source.columns: + df_model_source["has_price_info"] = (df_model_source["order_total_amount_prior"] > 0).astype(int) +# 最终模型特征列表 +MODEL_FEATURES = STATIC_BASE_FEATURES + prior_features +# 缺失率统计 +missing_rate = df_model_source.isnull().mean().sort_values(ascending=False) +print("\n特征缺失率统计(>0的特征):") +print(missing_rate[missing_rate > 0]) +# 先验特征覆盖率(双维度→单维度回退策略) +prior_coverage = 1 - df_model_source[prior_features].isnull().mean() +print("\n先验特征覆盖率(非零值比例):") +# 计算非零值比例 +non_zero_coverage = {} +for feature in prior_features: + if feature in df_model_source.columns: + non_zero_rate = (df_model_source[feature] != 0).mean() + non_zero_coverage[feature] = non_zero_rate +print("非零先验特征覆盖率:") +for feature, rate in non_zero_coverage.items(): + print(f" {feature}: {rate:.2%}") +print(f"平均非零覆盖率: {sum(non_zero_coverage.values())/len(non_zero_coverage):.2%}") +# 6. 使用新的特征体系进行建模 +print("正在使用新特征体系进行建模...") +df_model = df_model_source[MODEL_FEATURES].copy() +df_bidding = df_model_source.copy() +print(f"最终建模数据量: {len(df_bidding):,} 单") +# 缺失值处理(价值类现值不入模,因此不再用现值统计填充它们) +print("正在处理缺失值...") +print(f"缺失值处理前数据量: {len(df_model):,} 单") +missing_counts = df_model.isnull().sum() +print("各特征缺失值统计:") +for col, count in missing_counts[missing_counts > 0].items(): + print(f" {col}: {count:,} 个缺失值") +# 填充缺失值 +print("正在填充缺失值...") +# 基础特征填充 +fillna_dict = { + "order_goods_cnt": 1, + "business_rule_score": 0, # 业务规则得分默认0 +} +# 为先验特征添加默认值 +for feature in prior_features: + if feature in df_model.columns: + fillna_dict[feature] = 0 # 先验特征默认0 +df_model = df_model.fillna(fillna_dict) +# 再次全量兜底,防止有遗漏 +print("再次全量填充0,防止NaN...") +df_model = df_model.fillna(0) +print(f"缺失值填充后数据量: {len(df_model):,} 单") +print(f"保留的订单数: {len(df_model):,} 单") +# 保证主DataFrame与模型所用数据行对齐 +df_bidding = df_model_source.loc[df_model.index].copy() +# --- 模型训练与聚类 --- +print("\n正在进行模型训练...") +# 完全StandardScaler标准化 +print("正在进行StandardScaler特征标准化...") +scaler = StandardScaler() +X_scaled = scaler.fit_transform(df_model) +joblib.dump(scaler, os.path.join(MODEL_DIR, "scaler.pkl")) +print(f" - 所有特征统一标准化:{len(df_model.columns)}个特征") +# KMeans聚类 +print("正在训练KMeans模型 (n_clusters=6)...") +kmeans = KMeans(n_clusters=6, random_state=42, n_init=10) +kmeans.fit(X_scaled) +joblib.dump(kmeans, os.path.join(MODEL_DIR, "kmeans_model.pkl")) +# 生成用户属性映射(用于预测时补充用户信息) +print("正在生成用户属性映射...") +user_attributes = {} +for user_id in df_model_source["user_id"].unique(): + user_data = df_model_source[df_model_source["user_id"] == user_id] + user_attributes[user_id] = { + "attention_cnt": ( + user_data["attention_cnt"].iloc[0] + if "attention_cnt" in user_data.columns + else 0 + ), + "merchant_aftersale_rate": ( + user_data["merchant_aftersale_rate"].iloc[0] + if "merchant_aftersale_rate" in user_data.columns + else 0.0 + ), + "ignore_cnt": ( + user_data["ignore_cnt"].iloc[0] if "ignore_cnt" in user_data.columns else 0 + ), + "business_full_name": ( + user_data["business_full_name"].iloc[0] + if "business_full_name" in user_data.columns + else "" + ), + "address": ( + user_data["address"].iloc[0] if "address" in user_data.columns else "" + ), + } +print(f" - 用户属性映射完成: {len(user_attributes)} 个用户") +# 保存业务规则配置和统计数据 +joblib.dump(BUSINESS_RULES, os.path.join(MODEL_DIR, "business_rules.pkl")) +joblib.dump(prior_stats, os.path.join(MODEL_DIR, "prior_stats.pkl")) +joblib.dump(category_stats, os.path.join(MODEL_DIR, "category_stats.pkl")) +joblib.dump(global_stats, os.path.join(MODEL_DIR, "global_stats.pkl")) +joblib.dump(MODEL_FEATURES, os.path.join(MODEL_DIR, "model_features.pkl")) +joblib.dump(user_attributes, os.path.join(MODEL_DIR, "user_attributes.pkl")) +# 注意:cluster_map 将在后续动态聚类解读后保存 +print(f"模型训练完成,相关组件已保存至 '{MODEL_DIR}' 目录。") +# --- 3. 业务解读与标签映射 --- +print("\n[Part 3] 分析聚类中心,为聚类结果赋予业务含义...") +df_bidding["static_cluster"] = kmeans.labels_ +# 为了方便业务理解,我们将标准化的聚类中心还原为原始数值 +cluster_centers_original = scaler.inverse_transform(kmeans.cluster_centers_) +cluster_centers_df = pd.DataFrame(cluster_centers_original, columns=df_model.columns) +print("聚类中心 (原始数值尺度):") +print(cluster_centers_df) +# 动态聚类解读算法 + + +def generate_dynamic_cluster_labels(cluster_centers_df, df_bidding): + """ + + 基于聚类中心特征值动态生成业务标签 + + """ + print("正在进行动态聚类解读...") + cluster_labels = {} + # 计算全局特征分位数,用于判断高低 + global_percentiles = {} + key_features = [ + "order_total_amount_prior", + "order_goods_cnt", + "order_unit_price_prior", + "fifth_offer_duration_second_prior", + "tenth_offer_duration_second_prior", + "onsite_to_finish_hour_prior", + "offer_rate_prior", + "attention_cnt_prior", + "merchant_aftersale_rate_prior", + "ignore_cnt_prior", + "buyer_note_100", + ] + for feature in key_features: + if feature in df_bidding.columns: + global_percentiles[feature] = { + "low": df_bidding[feature].quantile(0.33), + "high": df_bidding[feature].quantile(0.67), + "very_high": df_bidding[feature].quantile(0.9), + } + # 为每个聚类生成标签 + for cluster_id in range(len(cluster_centers_df)): + center = cluster_centers_df.iloc[cluster_id] + # 分析关键特征 + characteristics = [] + # 1. 订单价值特征 + if ( + "order_total_amount_prior" in center.index + and "order_total_amount_prior" in global_percentiles + ): + amount = center["order_total_amount_prior"] + if amount >= global_percentiles["order_total_amount_prior"]["very_high"]: + characteristics.append("超高价") + elif amount >= global_percentiles["order_total_amount_prior"]["high"]: + characteristics.append("高价") + elif amount <= global_percentiles["order_total_amount_prior"]["low"]: + characteristics.append("低价") + else: + characteristics.append("中价") + # 2. 订单规模特征 + if ( + "order_goods_cnt" in center.index + and "order_goods_cnt" in global_percentiles + ): + goods_cnt = center["order_goods_cnt"] + if goods_cnt >= global_percentiles["order_goods_cnt"]["very_high"]: + characteristics.append("超大批量") + elif goods_cnt >= global_percentiles["order_goods_cnt"]["high"]: + characteristics.append("大批量") + elif goods_cnt <= global_percentiles["order_goods_cnt"]["low"]: + characteristics.append("小批量") + # 3. 响应效率特征 + if ( + "fifth_offer_duration_second_prior" in center.index + and "fifth_offer_duration_second_prior" in global_percentiles + ): + duration = center["fifth_offer_duration_second_prior"] + if ( + duration + >= global_percentiles["fifth_offer_duration_second_prior"]["very_high"] + ): + characteristics.append("极慢响应") + elif duration >= global_percentiles["fifth_offer_duration_second_prior"]["high"]: + characteristics.append("慢响应") + elif duration <= global_percentiles["fifth_offer_duration_second_prior"]["low"]: + characteristics.append("快响应") + # 4. 工期特征 + if ( + "onsite_to_finish_hour_prior" in center.index + and "onsite_to_finish_hour_prior" in global_percentiles + ): + finish_time = center["onsite_to_finish_hour_prior"] + if finish_time >= global_percentiles["onsite_to_finish_hour_prior"]["very_high"]: + characteristics.append("超长工期") + elif finish_time >= global_percentiles["onsite_to_finish_hour_prior"]["high"]: + characteristics.append("长工期") + elif finish_time <= global_percentiles["onsite_to_finish_hour_prior"]["low"]: + characteristics.append("短工期") + # 5. 报价率特征 + if "offer_rate_prior" in center.index and "offer_rate_prior" in global_percentiles: + offer_rate = center["offer_rate_prior"] + if offer_rate >= global_percentiles["offer_rate_prior"]["high"]: + characteristics.append("高报价率") + elif offer_rate <= global_percentiles["offer_rate_prior"]["low"]: + characteristics.append("低报价率") + # 6. 关注度特征 + if "attention_cnt_prior" in center.index and "attention_cnt_prior" in global_percentiles: + attention = center["attention_cnt_prior"] + if attention >= global_percentiles["attention_cnt_prior"]["very_high"]: + characteristics.append("极高关注") + elif attention >= global_percentiles["attention_cnt_prior"]["high"]: + characteristics.append("高关注") + elif attention <= global_percentiles["attention_cnt_prior"]["low"]: + characteristics.append("低关注") + # 7. 风险特征 + if ( + "merchant_aftersale_rate_prior" in center.index + and "merchant_aftersale_rate_prior" in global_percentiles + ): + aftersale_rate = center["merchant_aftersale_rate_prior"] + if aftersale_rate >= global_percentiles["merchant_aftersale_rate_prior"]["high"]: + characteristics.append("高售后") + if "ignore_cnt_prior" in center.index and "ignore_cnt_prior" in global_percentiles: + ignore_cnt = center["ignore_cnt_prior"] + if ignore_cnt >= global_percentiles["ignore_cnt_prior"]["very_high"]: + characteristics.append("高拉黑") + elif ignore_cnt >= global_percentiles["ignore_cnt_prior"]["high"]: + characteristics.append("中拉黑") + # 8. 复杂度特征 + if "buyer_note_100" in center.index and center["buyer_note_100"] > 0.5: + characteristics.append("复杂需求") + # 生成标签 + if not characteristics: + label = f"普通订单_{cluster_id}" + else: + # 优先级排序:价值 > 规模 > 效率 > 风险 + priority_order = [ + "超高价", + "高价", + "超大批量", + "大批量", + "极慢响应", + "慢响应", + "快响应", + "超长工期", + "长工期", + "短工期", + "高报价率", + "低报价率", + "极高关注", + "高关注", + "低关注", + "高售后", + "高拉黑", + "复杂需求", + ] + # 按优先级选择前2-3个特征 + sorted_chars = [char for char in priority_order if char in characteristics] + if len(sorted_chars) == 0: + sorted_chars = characteristics[:2] + elif len(sorted_chars) == 1: + sorted_chars = ( + sorted_chars + + [char for char in characteristics if char not in sorted_chars][:1] + ) + else: + sorted_chars = sorted_chars[:2] + # 构建标签 + if len(sorted_chars) == 1: + label = f"{sorted_chars[0]}订单" + else: + label = f'{"".join(sorted_chars)}订单' + cluster_labels[cluster_id] = label + # 打印解读过程 + cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id] + print(f"\n聚类 {cluster_id} 动态解读:") + print(f" • 订单数量: {len(cluster_data):,} 单") + print( + f" • 识别特征: {', '.join(characteristics) if characteristics else '无明显特征'}" + ) + print(f" • 生成标签: {label}") + # 显示关键数值 + if "order_total_amount_prior" in center.index: + print(f" • 平均金额(先验): ¥{center['order_total_amount_prior']:.0f}") + if "order_goods_cnt" in center.index: + print(f" • 平均件数: {center['order_goods_cnt']:.1f} 件") + if "fifth_offer_duration_second_prior" in center.index: + print(f" • 5人报价时长(先验): {center['fifth_offer_duration_second_prior']:.0f} 秒") + if "onsite_to_finish_hour_prior" in center.index: + print(f" • 平均工期(先验): {center['onsite_to_finish_hour_prior']:.1f} 小时") + return cluster_labels + + +# 执行动态聚类解读 +cluster_map = generate_dynamic_cluster_labels(cluster_centers_df, df_bidding) +df_bidding["static_label"] = df_bidding["static_cluster"].map(cluster_map) +# 保存聚类映射 +joblib.dump(cluster_map, os.path.join(MODEL_DIR, "cluster_map.pkl")) +print("聚类映射已保存至模型目录") +print("\n✅ 动态聚类解读完成!已为历史数据自动生成业务标签。") +# 计算聚类统计 +cluster_counts = df_bidding["static_cluster"].value_counts().sort_index() +# 聚类可视化(智能版:自动检测异常聚类) +print("\n正在生成聚类可视化图...") + + +def detect_outlier_clusters( + df_data, cluster_centers_df, min_samples=50, value_threshold=10.0 +): + """ + + 智能检测异常聚类的通用方法 + + 参数: + + df_data: 数据DataFrame + + cluster_centers_df: 聚类中心DataFrame + + min_samples: 最小样本数阈值,少于此数量视为异常 + + value_threshold: 价值差异倍数,超过此倍数视为异常 + + 返回: + + outlier_clusters: 异常聚类ID列表 + + normal_clusters: 正常聚类ID列表 + + """ + outlier_clusters = [] + normal_clusters = [] + print("正在智能检测异常聚类...") + # 计算每个聚类的统计信息 + cluster_stats = {} + for cluster_id in range(len(cluster_centers_df)): + cluster_data = df_data[df_data["static_cluster"] == cluster_id] + cluster_stats[cluster_id] = { + "count": len(cluster_data), + "avg_amount": cluster_centers_df.iloc[cluster_id].get("order_total_amount_prior", 0), + "avg_goods": cluster_centers_df.iloc[cluster_id]["order_goods_cnt"], + } + # 计算整体平均值作为基准 + total_avg_amount = df_data.get("order_total_amount_prior", pd.Series([0]*len(df_data))).mean() + total_avg_goods = df_data["order_goods_cnt"].mean() + print("各聚类异常检测分析:") + for cluster_id, stats in cluster_stats.items(): + is_outlier = False + reasons = [] + # 检测1:样本数量过少 + if stats["count"] < min_samples: + is_outlier = True + reasons.append(f"样本数过少({stats['count']})") + # 检测2:订单金额极值 + amount_ratio = ( + stats["avg_amount"] / total_avg_amount if total_avg_amount > 0 else 1 + ) + if amount_ratio > value_threshold: + is_outlier = True + reasons.append(f"金额异常({amount_ratio:.1f}倍)") + # 检测3:商品数量极值 + goods_ratio = stats["avg_goods"] / total_avg_goods if total_avg_goods > 0 else 1 + if goods_ratio > value_threshold: + is_outlier = True + reasons.append(f"件数异常({goods_ratio:.1f}倍)") + if is_outlier: + outlier_clusters.append(cluster_id) + print(f" 聚类{cluster_id}: 异常 - {', '.join(reasons)}") + else: + normal_clusters.append(cluster_id) + print( + f" 聚类{cluster_id}: 正常 - {stats['count']}单, ¥{stats['avg_amount']:.0f}, {stats['avg_goods']:.1f}件" + ) + print(f"检测结果: 正常聚类{normal_clusters}, 异常聚类{outlier_clusters}") + return outlier_clusters, normal_clusters + + +# 自动检测异常聚类 +outlier_clusters, normal_clusters = detect_outlier_clusters( + df_bidding, cluster_centers_df +) +# 分别处理正常聚类和异常聚类 +normal_mask = df_bidding["static_cluster"].isin(normal_clusters) +outlier_mask = df_bidding["static_cluster"].isin(outlier_clusters) +# 根据检测结果选择可视化策略 +if len(outlier_clusters) > 0: + print(f"检测到{len(outlier_clusters)}个异常聚类,使用分层可视化") + # 方案1:分层可视化 - 主图显示正常聚类,子图显示异常聚类 + fig = plt.figure(figsize=(15, 10)) +else: + print("未检测到异常聚类,使用常规可视化") + # 常规可视化 - 所有聚类在同一图中 + fig = plt.figure(figsize=(12, 10)) +# === 主图设置 === +ax_main = plt.subplot(1, 1, 1) +# 使用专业的配色方案 +colors = ["#4E79A7", "#F28E2B", "#59A14F", "#E15759", "#76B7B2", "#EDC948"] +if len(outlier_clusters) > 0: + # === 分层可视化:只显示正常聚类 === + # 对正常聚类数据进行PCA + X_normal = X_scaled[normal_mask] + pca_normal = PCA(n_components=2) + X_pca_normal = pca_normal.fit_transform(X_normal) + # 绘制正常聚类散点图 + normal_cluster_labels = df_bidding[normal_mask]["static_cluster"] + for cluster_id in normal_clusters: + cluster_mask = normal_cluster_labels == cluster_id + cluster_data = X_pca_normal[cluster_mask] + plt.scatter( + cluster_data[:, 0], + cluster_data[:, 1], + c=colors[cluster_id], + s=8, + alpha=0.7, + label=f'{cluster_map.get(cluster_id, f"聚类{cluster_id}")} ({cluster_counts.get(cluster_id, 0):,}单)', + edgecolors="none", + ) + # 添加正常聚类中心 + normal_centers = kmeans.cluster_centers_[normal_clusters] + centers_pca_normal = pca_normal.transform(normal_centers) + for i, cluster_id in enumerate(normal_clusters): + x, y = centers_pca_normal[i] + plt.scatter( + x, y, c="red", marker="*", s=300, linewidths=2, edgecolors="black", zorder=4 + ) + plt.text( + x, + y, + str(cluster_id), + color="white", + fontsize=12, + fontweight="bold", + ha="center", + va="center", + zorder=10, + ) + # 主图设置(分层模式) + plt.title( + "报价招标订单智能分类结果展示(主要聚类)\n基于核心业务特征 + 业务规则", + fontsize=16, + fontweight="bold", + pad=20, + ) + legend_title = "订单分类 (正常范围)" +else: + # === 常规可视化:显示所有聚类 === + # 对所有数据进行PCA + pca_all = PCA(n_components=2) + X_pca_all = pca_all.fit_transform(X_scaled) + # 绘制所有聚类散点图 + for cluster_id in range(6): + cluster_mask = df_bidding["static_cluster"] == cluster_id + cluster_data = X_pca_all[cluster_mask] + plt.scatter( + cluster_data[:, 0], + cluster_data[:, 1], + c=colors[cluster_id], + s=12, + alpha=0.8, + label=f'{cluster_map.get(cluster_id, f"聚类{cluster_id}")} ({cluster_counts.get(cluster_id, 0):,}单)', + edgecolors="none", + ) + # 添加所有聚类中心 + centers_pca_all = pca_all.transform(kmeans.cluster_centers_) + for cluster_id in range(6): + x, y = centers_pca_all[cluster_id] + plt.scatter( + x, y, c="red", marker="*", s=400, linewidths=2, edgecolors="black", zorder=4 + ) + plt.text( + x, + y, + str(cluster_id), + color="white", + fontsize=14, + fontweight="bold", + ha="center", + va="center", + zorder=10, + ) + # 主图设置(常规模式) + plt.title( + "报价招标订单智能分类结果展示\n基于核心业务特征 + 业务规则", + fontsize=16, + fontweight="bold", + pad=20, + ) + legend_title = "订单分类" +# 通用设置 +plt.xlabel("主成分1 (订单复杂度维度)", fontsize=12) +plt.ylabel("主成分2 (订单价值维度)", fontsize=12) +plt.grid(True, alpha=0.3, linestyle="--") +# 创建图例 +legend1 = plt.legend( + title=legend_title, loc="upper left", fontsize=9, framealpha=0.95, edgecolor="black" +) +plt.gca().add_artist(legend1) +# === 子图:异常聚类(智能适配) === +if outlier_mask.sum() > 0 and len(outlier_clusters) > 0: + ax_inset = fig.add_axes([0.65, 0.65, 0.32, 0.25]) # [x, y, width, height] + # 对包含异常点的所有数据进行PCA + pca_all = PCA(n_components=2) + X_pca_all = pca_all.fit_transform(X_scaled) + # 绘制所有异常聚类 + outlier_cluster_labels = df_bidding[outlier_mask]["static_cluster"] + for cluster_id in outlier_clusters: + cluster_mask = outlier_cluster_labels == cluster_id + if cluster_mask.sum() > 0: + cluster_data = X_pca_all[outlier_mask][cluster_mask] + ax_inset.scatter( + cluster_data[:, 0], + cluster_data[:, 1], + c=colors[cluster_id], + s=100, + alpha=0.9, + edgecolors="black", + linewidth=1, + label=f"聚类{cluster_id}", + ) + # 添加异常聚类中心 + centers_pca_all = pca_all.transform(kmeans.cluster_centers_) + for cluster_id in outlier_clusters: + outlier_center = centers_pca_all[cluster_id] + ax_inset.scatter( + outlier_center[0], + outlier_center[1], + c="red", + marker="*", + s=200, + linewidths=2, + edgecolors="black", + zorder=4, + ) + ax_inset.text( + outlier_center[0], + outlier_center[1], + str(cluster_id), + color="white", + fontsize=10, + fontweight="bold", + ha="center", + va="center", + zorder=10, + ) + # 动态生成标题 + outlier_count = outlier_mask.sum() + if len(outlier_clusters) == 1: + cluster_id = outlier_clusters[0] + title = f'聚类{cluster_id}:{cluster_map.get(cluster_id, "异常聚类")}\n({outlier_count}单,独立展示)' + else: + title = f"异常聚类:{outlier_clusters}\n({outlier_count}单,独立展示)" + ax_inset.set_title(title, fontsize=10, fontweight="bold") + ax_inset.grid(True, alpha=0.3, linestyle="--") + # 如果有多个异常聚类,添加小图例 + if len(outlier_clusters) > 1: + ax_inset.legend(fontsize=8, loc="best") +# 添加统计信息文本框(智能适配) +normal_count = normal_mask.sum() +outlier_count = outlier_mask.sum() +outlier_info = ( + f"聚类{outlier_clusters}" + if len(outlier_clusters) > 1 + else f"聚类{outlier_clusters[0]}" if outlier_clusters else "无" +) +stats_text = f""" + +数据统计: + +• 主图显示: {normal_count:,} 单 ({normal_count/len(df_bidding)*100:.2f}%) + +• 异常聚类: {outlier_info} ({outlier_count}单,独立显示) + +• 主要类别: {cluster_map.get(cluster_counts.index[0], '未知')} ({cluster_counts.iloc[0]:,}单) + +• 第二大类别: {cluster_map.get(cluster_counts.index[1], '未知')} ({cluster_counts.iloc[1]:,}单) + +• 智能检测: 自动识别异常聚类,分层展示 + +""" +plt.text( + 0.02, + 0.35, + stats_text, + transform=ax_main.transAxes, + fontsize=9, + verticalalignment="top", + bbox=dict(boxstyle="round", facecolor="lightgreen", alpha=0.8), +) +# 调整布局 +plt.tight_layout() +# 保存图片(智能命名) +if len(outlier_clusters) > 0: + save_path = "/Users/tom/Documents/订单聚类可视化图_智能分层版.png" + version_name = "智能分层版" +else: + save_path = "/Users/tom/Documents/订单聚类可视化图_常规版.png" + version_name = "常规版" +plt.savefig(save_path, dpi=300, bbox_inches="tight") +print(f"{version_name}聚类可视化图已保存至: {save_path}") +# === 方案2:生成对比图 - 排除异常点版本(仅在有异常聚类时) === +if len(outlier_clusters) > 0: + plt.figure(figsize=(12, 8)) + # 绘制排除异常点的散点图 + for cluster_id in normal_clusters: + cluster_mask = normal_cluster_labels == cluster_id + cluster_data = X_pca_normal[cluster_mask] + plt.scatter( + cluster_data[:, 0], + cluster_data[:, 1], + c=colors[cluster_id], + s=12, + alpha=0.8, + label=f'{cluster_map.get(cluster_id, f"聚类{cluster_id}")} ({cluster_counts.get(cluster_id, 0):,}单)', + edgecolors="none", + ) + # 添加聚类中心 + for i, cluster_id in enumerate(normal_clusters): + x, y = centers_pca_normal[i] + plt.scatter( + x, y, c="red", marker="*", s=400, linewidths=2, edgecolors="black", zorder=4 + ) + plt.text( + x, + y, + str(cluster_id), + color="white", + fontsize=14, + fontweight="bold", + ha="center", + va="center", + zorder=10, + ) + plt.title( + "报价招标订单聚类分布图(排除极值点)\n清晰展示主要订单模式分布", + fontsize=16, + fontweight="bold", + pad=20, + ) + plt.xlabel("主成分1 (订单复杂度维度)", fontsize=12) + plt.ylabel("主成分2 (订单价值维度)", fontsize=12) + plt.grid(True, alpha=0.3, linestyle="--") + # 图例 + plt.legend( + title="订单分类", loc="best", fontsize=10, framealpha=0.95, edgecolor="black" + ) + # 添加说明(智能适配) + excluded_info = ( + f"聚类{outlier_clusters}的{outlier_count}单" if outlier_clusters else "异常订单" + ) + note_text = f""" + +说明:本图排除了{excluded_info}异常订单, + +以便清晰观察其余{normal_count:,}单的分布模式 + +异常检测:自动识别样本少或特征极值的聚类 + + """ + plt.text( + 0.02, + 0.98, + note_text, + transform=plt.gca().transAxes, + fontsize=10, + verticalalignment="top", + bbox=dict(boxstyle="round", facecolor="yellow", alpha=0.8), + ) + plt.tight_layout() + # 保存对比图 + contrast_path = "/Users/tom/Documents/订单聚类可视化图_排除异常点版.png" + plt.savefig(contrast_path, dpi=300, bbox_inches="tight") + print(f"排除异常点版聚类可视化图已保存至: {contrast_path}") +else: + print("未检测到异常聚类,无需生成排除异常点版本") +# plt.show() +# --- 4. 动态特征验证 --- +print("\n[Part 4] 使用动态特征验证静态分层效果...") +available_dynamic_features = [ + col for col in DYNAMIC_FEATURES if col in df_bidding.columns +] +if available_dynamic_features: + validation_summary = df_bidding.groupby("static_label")[ + available_dynamic_features + ].mean() + print("各层级订单在动态特征上的平均表现:") + print(validation_summary) +else: + print("没有可用的动态特征用于验证。") +# --- 自动化好单识别算法 --- +print("\n[Part 5] 混合评分好单识别算法...") +# === 混合评分系统:规则70% + 聚类30% === +print("开始混合评分好单识别(规则70% + 聚类30%)...") +# 混合评分配置 - 使用统一配置管理 +HYBRID_CONFIG = config.get_hybrid_config() + + +def calculate_rule_score(order_data): + """ + + 计算规则评分 (0-100分) - 统一的规则评分函数 + + 训练时和预测时使用完全相同的逻辑 + + """ + score = 0 + # 🔧 修复:统一获取动态阈值的逻辑(训练时和预测时一致) + try: + # 优先使用全局变量(训练时) + if "DYNAMIC_RULE_THRESHOLDS" in globals() and DYNAMIC_RULE_THRESHOLDS: + thresholds = DYNAMIC_RULE_THRESHOLDS + else: + # 从文件加载(预测时) + thresholds = joblib.load( + os.path.join(MODEL_DIR, "dynamic_rule_thresholds.pkl") + ) + except Exception as e: + # 降级到硬编码阈值(兼容性保护) + print(f"⚠️ 动态阈值加载失败: {e},使用硬编码阈值") + return calculate_rule_score_hardcoded(order_data) + # === 主参考特征 (85分) - 全部改为先验特征 === + # 1. 满10人报价时长先验 (20分) - 基于历史统计 + tenth_offer_duration_prior = order_data.get("tenth_offer_duration_second_prior", 0) + if tenth_offer_duration_prior == 0: # 无历史数据 + tenth_duration_score = 0 + elif "tenth_offer_duration_second" in thresholds: + # 使用动态阈值(但应用到先验特征) + tenth_thresholds = thresholds["tenth_offer_duration_second"] + tenth_duration_score = 0 + for score_value, threshold in sorted(tenth_thresholds.items(), reverse=True): + if tenth_offer_duration_prior <= threshold: + tenth_duration_score = score_value + break + else: + tenth_duration_score = 0 + score += tenth_duration_score + # 2. 查看报价率先验 (15分) - 基于历史统计 + offer_rate_prior = order_data.get("offer_rate_prior", 0) + if "offer_rate" in thresholds: + # 使用动态阈值(但应用到先验特征) + offer_thresholds = thresholds["offer_rate"] + offer_rate_score = 0 + for score_value, threshold in sorted(offer_thresholds.items(), reverse=True): + if offer_rate_prior >= threshold: + offer_rate_score = score_value + break + else: + offer_rate_score = 0 + score += offer_rate_score + # 3. 总金额先验 (15分) - 订单实际金额(真正的先验) + amount_prior = order_data.get("order_total_amount_prior", 0) + if "order_total_amount" in thresholds: + # 使用动态阈值 + amount_thresholds = thresholds["order_total_amount"] + amount_score = 0 + for score_value, threshold in sorted(amount_thresholds.items(), reverse=True): + if amount_prior >= threshold: + amount_score = score_value + break + else: + amount_score = 1 if amount_prior > 0 else 0 + score += amount_score + # 4. 单价先验 (15分) - 订单实际单价(真正的先验) + unit_price_prior = order_data.get("order_unit_price_prior", 0) + if "order_unit_price" in thresholds: + # 使用动态阈值 + unit_thresholds = thresholds["order_unit_price"] + unit_price_score = 0 + for score_value, threshold in sorted(unit_thresholds.items(), reverse=True): + if unit_price_prior >= threshold: + unit_price_score = score_value + break + else: + unit_price_score = 1 if unit_price_prior > 0 else 0 + score += unit_price_score + # 5. 满5人报价时长先验 (10分) - 基于历史统计 + fifth_offer_duration_prior = order_data.get("fifth_offer_duration_second_prior", 0) + if fifth_offer_duration_prior == 0: # 无历史数据 + fifth_duration_score = 0 + elif "fifth_offer_duration_second" in thresholds: + # 使用动态阈值(但应用到先验特征) + fifth_thresholds = thresholds["fifth_offer_duration_second"] + fifth_duration_score = 0 + for score_value, threshold in sorted(fifth_thresholds.items(), reverse=True): + if fifth_offer_duration_prior <= threshold: + fifth_duration_score = score_value + break + else: + fifth_duration_score = 0 + score += fifth_duration_score + # 6. 完工时长先验 (10分) - 基于历史统计 + onsite_to_finish_hour_prior = order_data.get("onsite_to_finish_hour_prior", 0) + if onsite_to_finish_hour_prior == 0: # 无历史数据 + finish_time_score = 0 + elif "onsite_to_finish_hour" in thresholds: + # 使用动态阈值(但应用到先验特征) + finish_thresholds = thresholds["onsite_to_finish_hour"] + finish_time_score = 0 + for score_value, threshold in sorted(finish_thresholds.items(), reverse=True): + if onsite_to_finish_hour_prior <= threshold: + finish_time_score = score_value + break + else: + finish_time_score = 0 + score += finish_time_score + # === 次参考特征 (15分) - 全部改为先验特征 === + # 7. 商家售后率先验 (8分) - 基于历史统计 + merchant_aftersale_rate_prior = order_data.get( + "merchant_aftersale_rate_prior", order_data.get("merchant_aftersale_rate", 0) + ) + if "merchant_aftersale_rate" in thresholds: + # 使用动态阈值 + aftersale_thresholds = thresholds["merchant_aftersale_rate"] + aftersale_score = 0 + for score_value, threshold in sorted( + aftersale_thresholds.items(), reverse=True + ): + if merchant_aftersale_rate_prior <= threshold: + aftersale_score = score_value + break + else: + aftersale_score = 0 + score += aftersale_score + # 8. 商家被拉黑数先验 (5分) - 基于历史统计 + ignore_cnt_prior = order_data.get( + "ignore_cnt_prior", order_data.get("ignore_cnt", 0) + ) + if "ignore_cnt" in thresholds: + # 使用动态阈值 + ignore_thresholds = thresholds["ignore_cnt"] + ignore_score = 0 + for score_value, threshold in sorted(ignore_thresholds.items(), reverse=True): + if ignore_cnt_prior <= threshold: + ignore_score = score_value + break + else: + ignore_score = 5 if ignore_cnt_prior == 0 else 0 + score += ignore_score + # 9. 师傅关注数先验 (2分) - 基于历史统计 + attention_cnt_prior = order_data.get( + "attention_cnt_prior", order_data.get("attention_cnt", 0) + ) + if attention_cnt_prior == 0: + attention_score = 0 + elif "attention_cnt" in thresholds: + # 使用动态区间阈值 + attention_config = thresholds["attention_cnt"] + optimal_range = attention_config.get("optimal_range", (3, 20)) + general_range = attention_config.get("general_range", (1, 30)) + if optimal_range[0] <= attention_cnt_prior <= optimal_range[1]: + attention_score = 2 # 最优区间 + elif general_range[0] <= attention_cnt_prior <= general_range[1]: + attention_score = 1 # 一般区间 + elif attention_cnt_prior > 0: + attention_score = 0.5 # 有关注即可 + else: + attention_score = 0 + else: + # 降级到硬编码逻辑 + if 3 <= attention_cnt_prior <= 20: + attention_score = 2 + elif 1 <= attention_cnt_prior <= 30: + attention_score = 1 + elif attention_cnt_prior > 0: + attention_score = 0.5 + else: + attention_score = 0 + score += attention_score + return min(100, max(0, score)) + + +def calculate_rule_score_hardcoded(order_data): + """ + + 备用的硬编码规则评分函数(兼容性保护)- 修复为全先验特征版本 + + 当动态阈值不可用时自动降级使用 + + """ + score = 0 + # === 主参考特征 (85分) - 全部使用先验特征 === + # 1. 满10人报价时长先验 (20分) + tenth_offer_duration_prior = order_data.get("tenth_offer_duration_second_prior", 0) + if tenth_offer_duration_prior == 0: + tenth_duration_score = 0 # 无历史数据 + elif tenth_offer_duration_prior <= 1800: + tenth_duration_score = 20 + elif tenth_offer_duration_prior <= 3600: + tenth_duration_score = 16 + elif tenth_offer_duration_prior <= 7200: + tenth_duration_score = 12 + elif tenth_offer_duration_prior <= 14400: + tenth_duration_score = 8 + elif tenth_offer_duration_prior <= 28800: + tenth_duration_score = 4 + else: + tenth_duration_score = 1 + score += tenth_duration_score + # 2. 查看报价率先验 (15分) + offer_rate_prior = order_data.get("offer_rate_prior", 0) + if offer_rate_prior >= 0.7: + offer_rate_score = 15 + elif offer_rate_prior >= 0.5: + offer_rate_score = 12 + elif offer_rate_prior >= 0.3: + offer_rate_score = 9 + elif offer_rate_prior >= 0.1: + offer_rate_score = 6 + elif offer_rate_prior > 0: + offer_rate_score = 3 + else: + offer_rate_score = 0 + score += offer_rate_score + # 3. 总金额先验 (15分) + amount_prior = order_data.get("order_total_amount_prior", 0) + if amount_prior >= 800: + amount_score = 15 + elif amount_prior >= 400: + amount_score = 12 + elif amount_prior >= 200: + amount_score = 9 + elif amount_prior >= 100: + amount_score = 6 + elif amount_prior >= 50: + amount_score = 3 + else: + amount_score = 1 + score += amount_score + # 4. 单价先验 (15分) + unit_price_prior = order_data.get("order_unit_price_prior", 0) + if unit_price_prior >= 150: + unit_price_score = 15 + elif unit_price_prior >= 80: + unit_price_score = 12 + elif unit_price_prior >= 40: + unit_price_score = 9 + elif unit_price_prior >= 20: + unit_price_score = 6 + elif unit_price_prior >= 10: + unit_price_score = 3 + elif unit_price_prior > 0: + unit_price_score = 1 + else: + unit_price_score = 0 + score += unit_price_score + # 5. 满5人报价时长先验 (10分) + fifth_offer_duration_prior = order_data.get("fifth_offer_duration_second_prior", 0) + if fifth_offer_duration_prior == 0: + fifth_duration_score = 0 # 无历史数据 + elif fifth_offer_duration_prior <= 1800: + fifth_duration_score = 10 + elif fifth_offer_duration_prior <= 3600: + fifth_duration_score = 8 + elif fifth_offer_duration_prior <= 7200: + fifth_duration_score = 6 + elif fifth_offer_duration_prior <= 14400: + fifth_duration_score = 3 + else: + fifth_duration_score = 1 + score += fifth_duration_score + # 6. 完工时长先验 (10分) + onsite_to_finish_hour_prior = order_data.get("onsite_to_finish_hour_prior", 0) + if onsite_to_finish_hour_prior == 0: + finish_time_score = 0 # 无历史数据 + elif onsite_to_finish_hour_prior <= 1.0: + finish_time_score = 10 + elif onsite_to_finish_hour_prior <= 2.0: + finish_time_score = 8 + elif onsite_to_finish_hour_prior <= 4.0: + finish_time_score = 6 + elif onsite_to_finish_hour_prior <= 8.0: + finish_time_score = 4 + elif onsite_to_finish_hour_prior <= 24.0: + finish_time_score = 2 + else: + finish_time_score = 1 + score += finish_time_score + # === 次参考特征 (15分) - 全部使用先验特征 === + # 7. 商家售后率先验 (8分) + merchant_aftersale_rate_prior = order_data.get( + "merchant_aftersale_rate_prior", order_data.get("merchant_aftersale_rate", 0) + ) + if merchant_aftersale_rate_prior <= 0.01: + aftersale_score = 8 + elif merchant_aftersale_rate_prior <= 0.03: + aftersale_score = 6 + elif merchant_aftersale_rate_prior <= 0.05: + aftersale_score = 4 + elif merchant_aftersale_rate_prior <= 0.08: + aftersale_score = 2 + else: + aftersale_score = 0 + score += aftersale_score + # 8. 商家被拉黑数先验 (5分) + ignore_cnt_prior = order_data.get( + "ignore_cnt_prior", order_data.get("ignore_cnt", 0) + ) + if ignore_cnt_prior == 0: + ignore_score = 5 + elif ignore_cnt_prior <= 2: + ignore_score = 3 + elif ignore_cnt_prior <= 5: + ignore_score = 1 + else: + ignore_score = 0 + score += ignore_score + # 9. 师傅关注数先验 (2分) + attention_cnt_prior = order_data.get( + "attention_cnt_prior", order_data.get("attention_cnt", 0) + ) + if 3 <= attention_cnt_prior <= 20: + attention_score = 2 + elif 1 <= attention_cnt_prior <= 30: + attention_score = 1 + elif attention_cnt_prior > 0: + attention_score = 0.5 + else: + attention_score = 0 + score += attention_score + return min(100, max(0, score)) + + +def calculate_cluster_score(order_data): + """ + + 计算聚类评分 - 与app_v3.py完全一致的逻辑 + + """ + cluster_id = order_data.get("static_cluster", 0) + # 修复:确保cluster_base_scores已初始化 + if not cluster_base_scores: + # 如果还未计算动态基础分,使用默认值 + base_score = 55.0 + else: + base_score = cluster_base_scores.get(cluster_id, 55.0) + return base_score + + +# 【代码修改核心】 +# 步骤 1: 首先计算所有订单的规则分 +print(" 📊 步骤1: 统一计算所有订单的规则分...") +df_bidding["rule_score"] = df_bidding.apply(calculate_rule_score, axis=1) +print(f" ✅ 规则分计算完成. 平均分: {df_bidding['rule_score'].mean():.2f}") +# 步骤 2: 计算动态的聚类基础分 (废弃55分的临时逻辑) +print(f"\n 📊 步骤2: 计算动态聚类基础分 (基于规则分均值)...") +# 方法:使用每个聚类的规则评分平均值作为该聚类的基础分 +for cluster_id in range(6): + cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id] + if len(cluster_data) > 0: + # 使用该聚类的规则评分平均值作为基础分 + avg_rule_score = cluster_data["rule_score"].mean() + # 增强区分度,扩大基础分范围 (25-85) + base_score = min(85, max(25, avg_rule_score)) + cluster_base_scores[cluster_id] = base_score + cluster_label = cluster_map.get(cluster_id, f"聚类{cluster_id}") + print( + f" 聚类{cluster_id}({cluster_label}): 规则分均值: {avg_rule_score:.1f} → 最终基础分: {base_score:.1f}" + ) + else: + cluster_base_scores[cluster_id] = 55.0 # 对空聚类使用默认基础分 + print(f" 聚类{cluster_id}: 无数据 → 默认基础分 55.0") +# 保存最终的动态基础分 +joblib.dump(cluster_base_scores, os.path.join(MODEL_DIR, "cluster_base_scores.pkl")) +print(f" ✅ 动态聚类基础分计算完成并保存!") +# 步骤 3: 计算最终的聚类分和混合分 +print("\n 📊 步骤3: 计算最终的聚类分和混合分...") +df_bidding["cluster_score"] = df_bidding["static_cluster"].map(cluster_base_scores) +df_bidding["hybrid_score"] = ( + df_bidding["rule_score"] * config.get_hybrid_config()["rule_weight"] + + df_bidding["cluster_score"] * config.get_hybrid_config()["cluster_weight"] +) +print(f" ✅ 最终评分计算完成:") +print( + f" 规则评分: {df_bidding['rule_score'].mean():.1f} ± {df_bidding['rule_score'].std():.1f}" +) +print( + f" 聚类评分: {df_bidding['cluster_score'].mean():.1f} ± {df_bidding['cluster_score'].std():.1f}" +) +print( + f" 混合评分: {df_bidding['hybrid_score'].mean():.1f} ± {df_bidding['hybrid_score'].std():.1f}" +) +# 步骤 4: 基于最终混合分,计算阈值并分级 +print(f"\n 📊 步骤4: 基于最终混合分计算分位数阈值并分级...") +good_quantile = HYBRID_CONFIG["good_quantile"] +medium_quantile = HYBRID_CONFIG["medium_quantile"] +# 计算分位数阈值 +good_threshold = df_bidding["hybrid_score"].quantile(good_quantile) +medium_threshold = df_bidding["hybrid_score"].quantile(medium_quantile) +print(f" ✅ 动态阈值计算完成:") +print(f" 好单分数线 (基于 {good_quantile*100:.0f}% 分位数): {good_threshold:.2f}") +print( + f" 中单分数线 (基于 {medium_quantile*100:.0f}% 分位数): {medium_threshold:.2f}" +) +# 使用计算出的阈值进行分级 + + +def assign_level(score, good_thresh, medium_thresh): + if score >= good_thresh: + return "好单" + elif score >= medium_thresh: + return "中单" + else: + return "差单" + + +df_bidding["static_level"] = df_bidding["hybrid_score"].apply( + lambda x: assign_level(x, good_threshold, medium_threshold) +) +print(f" ✅ 基于分位数的等级分配完成:") +level_counts = df_bidding["static_level"].value_counts() +total_count = len(df_bidding) +for level in ["好单", "中单", "差单"]: + count = level_counts.get(level, 0) + percentage = count / total_count * 100 + print(f" • {level}: {count:,} 单 ({percentage:.1f}%)") +# 步骤 5: 保存用于预测的最终阈值 +print(f"\n 📊 步骤5: 保存最终的、用于预测的分数阈值...") +DYNAMIC_THRESHOLDS = { + "good_threshold": good_threshold, + "medium_threshold": medium_threshold, + "algorithm": "quantile_score", + "description": f"基于训练集分位数计算的分数线:好单 >={good_threshold:.2f}, 中单 >={medium_threshold:.2f}", + "training_date": pd.Timestamp.now().strftime("%Y-%m-%d"), + "total_samples": len(df_bidding), + "good_quantile": good_quantile, + "medium_quantile": medium_quantile, +} +joblib.dump(DYNAMIC_THRESHOLDS, os.path.join(MODEL_DIR, "dynamic_thresholds.pkl")) +print(f" ✅ 训练阈值已保存: dynamic_thresholds.pkl") +# === 结果验证与保存 === +print(f"\n正在验证与保存混合评分模型组件...") +# 改进的聚类级别映射(基于好单比例的相对表现) +level_counts_by_cluster = ( + df_bidding.groupby(["static_cluster", "static_level"]).size().unstack(fill_value=0) +) +AUTO_CLUSTER_LEVEL_MAP = {} +global_good_ratio = len(df_bidding[df_bidding["static_level"] == "好单"]) / len( + df_bidding +) +print(f"\n🔧 改进聚类等级分配逻辑(基于好单比例相对表现)...") +print(f" 全局好单比例基准: {global_good_ratio:.1%}") +for cluster_id in range(6): + if cluster_id in level_counts_by_cluster.index: + cluster_levels = level_counts_by_cluster.loc[cluster_id] + total_in_cluster = cluster_levels.sum() + good_in_cluster = cluster_levels.get("好单", 0) + if total_in_cluster > 0: + cluster_good_ratio = good_in_cluster / total_in_cluster + relative_performance = ( + cluster_good_ratio / global_good_ratio if global_good_ratio > 0 else 0 + ) + # 基于相对表现分配等级 + if relative_performance >= 1.2: # 超出全局20%以上 + cluster_level = "好单聚类" + performance_desc = f"🔥 +{(relative_performance-1)*100:.0f}%" + elif relative_performance >= 0.8: # 在全局80%-120%之间 + cluster_level = "中单聚类" + performance_desc = f"⚡ {(relative_performance-1)*100:+.0f}%" + else: # 低于全局80% + cluster_level = "差单聚类" + performance_desc = f"❄️ {(relative_performance-1)*100:.0f}%" + AUTO_CLUSTER_LEVEL_MAP[cluster_id] = cluster_level + print( + f" 聚类{cluster_id}: {good_in_cluster:,}/{total_in_cluster:,} = {cluster_good_ratio:.1%} {performance_desc} → {cluster_level}" + ) + else: + AUTO_CLUSTER_LEVEL_MAP[cluster_id] = "中单聚类" + print(f" 聚类{cluster_id}: 无数据 → 中单聚类") + else: + AUTO_CLUSTER_LEVEL_MAP[cluster_id] = "中单聚类" + print(f" 聚类{cluster_id}: 不存在 → 中单聚类") +# 创建聚类质量得分(用于兼容性) +cluster_quality_scores = {} +for cluster_id in range(6): + cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id] + if len(cluster_data) > 0: + cluster_quality_scores[cluster_id] = { + "total_score": cluster_data["cluster_score"].mean(), + "hybrid_score": cluster_data["hybrid_score"].mean(), + "rule_score": cluster_data["rule_score"].mean(), + } +# 生成用户属性映射 +print(" - 正在生成用户属性映射...") +user_attributes = {} +if "user_id" in df_bidding.columns: + # 按user_id聚合用户属性 + user_attr_fields = [ + "attention_cnt", + "merchant_aftersale_rate", + "ignore_cnt", + "business_full_name", + "address", + ] + available_fields = [f for f in user_attr_fields if f in df_bidding.columns] + if available_fields: + # 使用最新的用户属性值(按user_id分组,取最后一条记录) + user_attr_df = df_bidding.groupby("user_id")[available_fields].last() + # 转换为字典格式 + for user_id, row in user_attr_df.iterrows(): + user_attributes[user_id] = {} + for field in available_fields: + if field in row: + user_attributes[user_id][field] = row[field] + else: + # 设置默认值 + if field in ["attention_cnt", "ignore_cnt"]: + user_attributes[user_id][field] = 0 + elif field == "merchant_aftersale_rate": + user_attributes[user_id][field] = 0.0 + else: + user_attributes[user_id][field] = "" + print(f" - 成功生成用户属性映射: {len(user_attributes)} 个用户") + else: + print(" - 未找到用户属性字段,将使用默认值") +else: + print(" - 未找到user_id字段,将使用默认值") +# 保存用户属性映射 +joblib.dump(user_attributes, os.path.join(MODEL_DIR, "user_attributes.pkl")) +# 保存配置 +joblib.dump( + AUTO_CLUSTER_LEVEL_MAP, os.path.join(MODEL_DIR, "auto_cluster_level_map.pkl") +) +joblib.dump( + cluster_quality_scores, os.path.join(MODEL_DIR, "cluster_quality_scores.pkl") +) +joblib.dump(HYBRID_CONFIG, os.path.join(MODEL_DIR, "hybrid_config.pkl")) +# DYNAMIC_THRESHOLDS已在前面保存,避免重复保存 +print(f"✅ 混合评分模型组件已保存") +print(f" • 动态阈值配置: dynamic_thresholds.pkl") +print(f" - 好单阈值: {DYNAMIC_THRESHOLDS['good_threshold']:.1f}分") +print(f" - 中单阈值: {DYNAMIC_THRESHOLDS['medium_threshold']:.1f}分") +print(f"\n🎉 关键逻辑修复完成!现在训练和预测的评分标准完全一致。") +# ... 后续代码部分保持不变 ... +# --- 6. 结果统计与分析 --- +print("\n[Part 6] 各聚类类别的订单数量统计...") +cluster_counts = df_bidding["static_cluster"].value_counts().sort_index() +print("各聚类类别的订单数量:") +for cluster_id, count in cluster_counts.items(): + percentage = (count / len(df_bidding)) * 100 + print(f" 类别 {cluster_id}: {count:,} 单 ({percentage:.2f}%)") +print(f"\n总计: {len(df_bidding):,} 单") +# 按业务标签统计 +print("\n按业务标签统计:") +label_counts = df_bidding["static_label"].value_counts() +for label, count in label_counts.items(): + percentage = (count / len(df_bidding)) * 100 + print(f" {label}: {count:,} 单 ({percentage:.2f}%)") +# --- 7. 业务规则效果分析 --- +print("\n[Part 7] 业务规则效果分析...") +# 分析业务规则得分与聚类结果的关系 +print("业务规则得分与聚类结果的关系:") +rule_score_analysis = df_bidding.groupby("static_cluster")["business_rule_score"].agg( + ["mean", "std", "count"] +) +print(rule_score_analysis) +# 分析各业务规则的具体效果 +print("\n各业务规则的具体效果:") +print("优质企业订单分布:") +if "business_full_name" in df_bidding.columns: + premium_companies = BUSINESS_RULES["bonus_rules"]["premium_companies"] + for company in premium_companies: + company_orders = df_bidding[df_bidding["business_full_name"] == company] + if len(company_orders) > 0: + print(f" {company}: {len(company_orders)} 单") + cluster_dist = company_orders["static_cluster"].value_counts().head(3) + print( + f" 主要聚类: {', '.join([f'{k}({v}单)' for k, v in cluster_dist.items()])}" + ) +print("\n优质地区订单分布:") +if "address" in df_bidding.columns: + premium_regions = BUSINESS_RULES["bonus_rules"]["premium_regions"] + for region in premium_regions: + region_orders = df_bidding[df_bidding["address"].str.contains(region, na=False)] + if len(region_orders) > 0: + print(f" {region}: {len(region_orders)} 单") + cluster_dist = region_orders["static_cluster"].value_counts().head(3) + print( + f" 主要聚类: {', '.join([f'{k}({v}单)' for k, v in cluster_dist.items()])}" + ) +print("\n大订单分布:") +large_orders = df_bidding[ + df_bidding["order_goods_cnt"] >= BUSINESS_RULES["large_orders_threshold"] +] +if len(large_orders) > 0: + print(f" 大订单(>=10件): {len(large_orders)} 单") + cluster_dist = large_orders["static_cluster"].value_counts().head(3) + print( + f" 主要聚类: {', '.join([f'{k}({v}单)' for k, v in cluster_dist.items()])}" + ) +# --- 聚类后特征重要性分析 --- +print("\n[Part 8] 聚类中心主特征排序与特征重要性分析...") +# 每个聚类主特征排序 +for i, row in cluster_centers_df.iterrows(): + print(f"\n聚类类别 {i} 主特征排序:") + available_features = [f for f in MAIN_REFERENCE_FEATURES if f in row.index] + if available_features: + sorted_main = row[available_features].sort_values(ascending=False) + for f, v in sorted_main.items(): + print(f" {f}: {v:.2f}") +# 特征重要性分析(聚类中心方差) +print("\n特征重要性分析(聚类中心方差):") +available_all_features = [ + f for f in MAIN_REFERENCE_FEATURES if f in cluster_centers_df.columns +] +if available_all_features: + feature_importance = ( + cluster_centers_df[available_all_features].std().sort_values(ascending=False) + ) + for f, v in feature_importance.items(): + print(f" {f}: {v:.2f}") +# === 最终结果汇总分析与导出 === +print("\n" + "=" * 80) +print("[FINAL PART] 最终结果汇总分析与完整导出") +print("=" * 80) +# --- 汇总统计报告 --- +print("\n[汇总统计] 模型整体表现报告...") +print(f"\n📊 数据概览:") +print(f" • 总订单数: {len(df_bidding):,} 单(已过滤三级类目<10单)") +print(f" • 聚类数量: 6 个") +print(f" • 特征维度: {len(df_model.columns)} 个") +print( + f" • 优化先验特征: {len([f for f in df_model.columns if '_prior' in f])} 个(双维度→单维度回退)" +) +print(f"\n🎯 混合评分好单识别结果:") +good_orders = df_bidding[df_bidding["static_level"] == "好单"] +medium_orders = df_bidding[df_bidding["static_level"] == "中单"] +poor_orders = df_bidding[df_bidding["static_level"] == "差单"] +print( + f" • 好单: {len(good_orders):,} 单 ({len(good_orders)/len(df_bidding)*100:.1f}%)" +) +print( + f" • 中单: {len(medium_orders):,} 单 ({len(medium_orders)/len(df_bidding)*100:.1f}%)" +) +print( + f" • 差单: {len(poor_orders):,} 单 ({len(poor_orders)/len(df_bidding)*100:.1f}%)" +) +# 好单业务特征分析 +if len(good_orders) > 0: + print(f"\n💎 好单业务特征分析:") + print(f" • 平均订单金额: ¥{good_orders['order_total_amount'].mean():.0f}") + print(f" • 平均商品数量: {good_orders['order_goods_cnt'].mean():.1f} 件") + print(f" • 平均混合得分: {good_orders['hybrid_score'].mean():.1f}") + print(f" • 平均规则得分: {good_orders['rule_score'].mean():.1f}") + print(f" • 平均聚类得分: {good_orders['cluster_score'].mean():.1f}") + print(f" • 平均报价率: {good_orders['offer_rate'].mean():.2f}") + # 好单的主要聚类分布 + good_clusters = good_orders["static_cluster"].value_counts() + print( + f" • 好单主要聚类: {', '.join([f'{k}({v}单)' for k, v in good_clusters.items()])}" + ) +# 聚类质量得分排序(改进版:基于好单比例排序) +print(f"\n🏆 聚类混合得分排序(改进版:基于好单比例表现):") +sorted_scores = sorted( + cluster_quality_scores.items(), + key=lambda x: x[1].get("hybrid_score", 0), + reverse=True, +) +for i, (cluster_id, scores) in enumerate(sorted_scores): + level = AUTO_CLUSTER_LEVEL_MAP.get(cluster_id, "未知") + label = cluster_map.get(cluster_id, f"聚类{cluster_id}") + count = len(df_bidding[df_bidding["static_cluster"] == cluster_id]) + hybrid_score = scores.get("hybrid_score", 0) + rule_score = scores.get("rule_score", 0) + cluster_score = scores.get("total_score", 0) + # 计算该聚类的好单比例 + cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id] + if len(cluster_data) > 0: + good_count = len(cluster_data[cluster_data["static_level"] == "好单"]) + good_ratio = good_count / len(cluster_data) + good_ratio_display = f"{good_ratio:.1%}" + else: + good_ratio_display = "0.0%" + print( + f" {i+1:2d}. 聚类{cluster_id} | {level:<6s} | 混合{hybrid_score:5.1f} | 规则{rule_score:5.1f} | 聚类{cluster_score:5.1f} | 好单率{good_ratio_display:>5s} | {count:>5,}单 | {label}" + ) +# 业务规则效果统计 +print(f"\n📋 业务规则效果统计:") +positive_rule_orders = df_bidding[df_bidding["business_rule_score"] > 0] +print( + f" • 正分订单: {len(positive_rule_orders):,} 单 ({len(positive_rule_orders)/len(df_bidding)*100:.1f}%)" +) +print(f" • 零分订单: {len(df_bidding[df_bidding['business_rule_score'] == 0]):,} 单") +print(f" • 负分订单: {len(df_bidding[df_bidding['business_rule_score'] < 0]):,} 单") +# 动态特征验证汇总 +if "onsite_to_finish_hour" in df_bidding.columns: + print(f"\n⏱️ 动态特征验证汇总:") + validation_features = [ + "offer_mst_cnt", + "fifth_offer_duration_second", + "onsite_to_finish_hour", + "merchant_aftersale_rate", + ] + available_validation = [f for f in validation_features if f in df_bidding.columns] + if available_validation: + level_performance = df_bidding.groupby("static_level")[ + available_validation + ].mean() + print(" 各等级在动态特征上的平均表现:") + for level in ["好单", "中单", "差单"]: + if level in level_performance.index: + print(f" {level}:") + for feature in available_validation: + value = level_performance.loc[level, feature] + print(f" {feature}: {value:.2f}") +# --- 完整Excel导出 --- +print(f"\n📄 开始导出完整分析结果到Excel...") +# 1. 主数据表 - 包含static_level完整依赖链 +export_columns = ( + [ + # 基础订单信息 + "order_no", + "order_submit_time", + "order_total_amount", + "order_goods_cnt", + "order_unit_price", + "user_id", + "user_name", + "business_full_name", + "company_type", + "address", + "city_name", + "goods_level_1_name", + "goods_level_2_name", + "goods_level_3_name", + "order_serve_type_name", + "order_appoint_type_name", + "is_urgent_order", + # === static_level依赖字段:规则评分相关 === + "offer_mst_cnt", + "view_mst_cnt", + "offer_rate", # 查看报价率计算依赖 + "fifth_offer_duration_second", + "tenth_offer_duration_second", # 报价时长依赖 + "onsite_to_finish_hour", # 完工时长依赖 + "attention_cnt", # 师傅关注数依赖 + "merchant_aftersale_rate", # 商家售后率依赖 + "ignore_cnt", # 商家被拉黑数依赖 + "buyer_note_100", # 备注长度依赖 + "business_rule_score", # 业务规则得分依赖 + # === static_level依赖字段:混合评分链 === + "rule_score", # 规则评分(70%权重) + "cluster_score", # 聚类评分(30%权重) + "hybrid_score", # 混合评分(最终用于分级) + # === static_level依赖字段:聚类相关 === + "static_cluster", # 聚类ID(影响cluster_score) + # 类别先验特征 + ] + + [f"{f}_prior" for f in ALL_REFERENCE_FEATURES if f in df_bidding.columns] + + [ + # === static_level最终结果 === + "static_label", + "static_level", +] +) +available_columns = [col for col in export_columns if col in df_bidding.columns] +df_export = df_bidding[available_columns].copy() +print(f" • 主数据表: {len(df_export)} 行 × {len(df_export.columns)} 列") +# 2. 聚类业务解读表(改进版:包含好单比例) +cluster_summary = [] +for cluster_id in range(6): + cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id] + scores = cluster_quality_scores.get(cluster_id, {}) + # 计算好单比例 + if len(cluster_data) > 0: + good_count = len(cluster_data[cluster_data["static_level"] == "好单"]) + good_ratio = good_count / len(cluster_data) + good_ratio_str = f"{good_ratio:.1%}" + else: + good_ratio_str = "0.0%" + summary_row = [ + cluster_id, + cluster_map.get(cluster_id, f"聚类{cluster_id}"), + AUTO_CLUSTER_LEVEL_MAP.get(cluster_id, "未知"), + len(cluster_data), + f"{len(cluster_data)/len(df_bidding)*100:.1f}%", + good_ratio_str, # 新增:好单比例 + f"{scores.get('hybrid_score', 0):.1f}", # 使用混合得分 + f"{scores.get('rule_score', 0):.1f}", # 规则得分 + f"{scores.get('total_score', 0):.1f}", # 聚类得分 + ( + f"¥{cluster_data['order_total_amount'].mean():.0f}" + if len(cluster_data) > 0 + else "¥0" + ), + ( + f"{cluster_data['order_goods_cnt'].mean():.1f}" + if len(cluster_data) > 0 + else "0" + ), + ] + cluster_summary.append(summary_row) +cluster_summary_df = pd.DataFrame( + cluster_summary, + columns=[ + "聚类编号", + "业务标签", + "改进等级", + "订单数量", + "占比", + "好单比例", + "混合得分", + "规则得分", + "聚类得分", + "平均金额", + "平均件数", + ], +) +# 3. 聚类质量得分详情表 +scores_detail = pd.DataFrame( + [ + [ + cluster_id, + cluster_quality_scores.get(cluster_id, {}).get("hybrid_score", 0), + cluster_quality_scores.get(cluster_id, {}).get("rule_score", 0), + cluster_quality_scores.get(cluster_id, {}).get("total_score", 0), + len(df_bidding[df_bidding["static_cluster"] == cluster_id]), + f"{len(df_bidding[df_bidding['static_cluster'] == cluster_id])/len(df_bidding)*100:.1f}%", + ] + for cluster_id in range(6) + ], + columns=["聚类编号", "混合得分", "规则得分", "聚类得分", "订单数量", "占比"], +) +# 4. 特征重要性分析表 +feature_importance_data = [] +all_features = [f for f in MAIN_REFERENCE_FEATURES if f in cluster_centers_df.columns] +if all_features: + feature_importance = ( + cluster_centers_df[all_features].std().sort_values(ascending=False) + ) + for feature, importance in feature_importance.items(): + feature_importance_data.append( + [ + feature, + f"{importance:.3f}", + "主特征" if feature in MAIN_REFERENCE_FEATURES else "次特征", + ] + ) +feature_importance_df = pd.DataFrame( + feature_importance_data, columns=["特征名称", "重要性得分", "特征类型"] +) +# 5. 业务规则效果分析表 +rule_analysis_data = [] +# 优质企业分析 +if "business_full_name" in df_bidding.columns: + for company in BUSINESS_RULES["bonus_rules"]["premium_companies"]: + company_orders = df_bidding[df_bidding["business_full_name"] == company] + if len(company_orders) > 0: + good_pct = ( + len(company_orders[company_orders["static_level"] == "好单"]) + / len(company_orders) + * 100 + ) + rule_analysis_data.append( + ["优质企业", company, len(company_orders), f"{good_pct:.1f}%"] + ) +# 优质地区分析 +if "address" in df_bidding.columns: + for region in BUSINESS_RULES["bonus_rules"]["premium_regions"]: + region_orders = df_bidding[df_bidding["address"].str.contains(region, na=False)] + if len(region_orders) > 0: + good_pct = ( + len(region_orders[region_orders["static_level"] == "好单"]) + / len(region_orders) + * 100 + ) + rule_analysis_data.append( + ["优质地区", region, len(region_orders), f"{good_pct:.1f}%"] + ) +# 大订单分析 +large_orders = df_bidding[ + df_bidding["order_goods_cnt"] >= BUSINESS_RULES["large_orders_threshold"] +] +if len(large_orders) > 0: + good_pct = ( + len(large_orders[large_orders["static_level"] == "好单"]) + / len(large_orders) + * 100 + ) + rule_analysis_data.append( + [ + "大订单", + f"≥{BUSINESS_RULES['large_orders_threshold']}件", + len(large_orders), + f"{good_pct:.1f}%", + ] + ) +rule_analysis_df = pd.DataFrame( + rule_analysis_data, columns=["规则类型", "规则项目", "命中订单数", "好单比例"] +) +# 6. 模型配置与参数表 +config_data = [ + ["模型参数", "KMeans聚类数", "6"], + ["模型参数", "随机种子", "42"], + ["模型参数", "标准化方法", "StandardScaler(全特征统一)"], + [ + "混合评分", + "评分方案", + f"规则{config.get_hybrid_config()['rule_weight']*100:.0f}% + 聚类{config.get_hybrid_config()['cluster_weight']*100:.0f}%", + ], + [ + "混合评分", + "好单分位数", + f"{HYBRID_CONFIG['good_quantile']*100:.0f}% (前{(1-HYBRID_CONFIG['good_quantile'])*100:.0f}%)", + ], + [ + "混合评分", + "中单分位数", + f"{HYBRID_CONFIG['medium_quantile']*100:.0f}% (前{(1-HYBRID_CONFIG['medium_quantile'])*100:.0f}%)", + ], + ["混合评分", "最低质量门槛", f"{HYBRID_CONFIG['min_quality_threshold']:.1f}分"], + ["规则评分-主特征", "满10人报价时长", "20分 (充分竞争效率,≤30分钟得20分)"], + ["规则评分-主特征", "查看报价率", "15分 (师傅响应积极性,≥70%得15分)"], + ["规则评分-主特征", "总金额", "15分 (订单总价值,≥800元得15分)"], + ["规则评分-主特征", "单价", "15分 (单件价值,≥150元得15分)"], + ["规则评分-主特征", "满5人报价时长", "10分 (初期响应效率,≤30分钟得10分)"], + ["规则评分-主特征", "完工时长", "10分 (执行效率,≤1小时得10分)"], + ["规则评分-次特征", "商家售后率", "8分 (客户风险,≤1%得8分)"], + ["规则评分-次特征", "商家被拉黑数", "5分 (商家风险,0个得5分)"], + ["规则评分-次特征", "师傅关注数", "2分 (市场关注度,3-20个得2分)"], + ["聚类评分", "聚类0得分", "64.3分 (大批量中价订单)"], + ["聚类评分", "聚类3得分", "63.5分 (超高价超大批量)"], + ["聚类评分", "聚类2得分", "42.5分 (高价但差单-反直觉)"], +] +config_df = pd.DataFrame(config_data, columns=["配置类别", "配置项", "配置值"]) +# 7. static_level计算公式详细说明表 +formula_data = [ + ["最终分级", "static_level计算", "基于混合得分分位数直接分级"], + [ + "最终分级", + "好单条件", + f'hybrid_score≥{DYNAMIC_THRESHOLDS["good_threshold"]:.1f}分(80分位数阈值)', + ], + [ + "最终分级", + "中单条件", + f'{DYNAMIC_THRESHOLDS["medium_threshold"]:.1f}分≤hybrid_score<{DYNAMIC_THRESHOLDS["good_threshold"]:.1f}分(60-80分位数)', + ], + [ + "最终分级", + "差单条件", + f'hybrid_score<{DYNAMIC_THRESHOLDS["medium_threshold"]:.1f}分(60分位数以下)', + ], + [ + "混合评分", + "hybrid_score公式", + f'rule_score×{config.get_hybrid_config()["rule_weight"]} + cluster_score×{config.get_hybrid_config()["cluster_weight"]}', + ], + ["混合评分", "具体计算", "rule_score×0.7 + cluster_score×0.3"], + ["规则评分", "rule_score公式", "主参考特征(85分) + 次参考特征(15分)"], + ["规则评分", "满10人报价时长(20分)", "tenth_offer_duration_second充分竞争效率"], + [ + "规则评分", + "10人时长评分规则", + "≤1800秒→20分, 1800-3600→16分, 3600-7200→12分, 7200-14400→8分, 14400-28800→4分, >28800→1分", + ], + ["规则评分", "查看报价率(15分)", "offer_rate = offer_mst_cnt/view_mst_cnt"], + [ + "规则评分", + "报价率评分规则", + "≥0.7→15分, 0.5-0.7→12分, 0.3-0.5→9分, 0.1-0.3→6分, >0→3分, =0→0分", + ], + ["规则评分", "总金额(15分)", "order_total_amount订单总价值"], + [ + "规则评分", + "总金额评分规则", + "≥800元→15分, 400-800→12分, 200-400→9分, 100-200→6分, 50-100→3分, <50→1分", + ], + ["规则评分", "单价(15分)", "order_unit_price单件价值"], + [ + "规则评分", + "单价评分规则", + "≥150元→15分, 80-150→12分, 40-80→9分, 20-40→6分, 10-20→3分, >0→1分, =0→0分", + ], + ["规则评分", "满5人报价时长(10分)", "fifth_offer_duration_second初期响应效率"], + [ + "规则评分", + "5人时长评分规则", + "≤1800秒→10分, 1800-3600→8分, 3600-7200→6分, 7200-14400→3分, >14400→1分", + ], + ["规则评分", "完工时长(10分)", "onsite_to_finish_hour执行效率"], + [ + "规则评分", + "完工时长评分规则", + "≤1小时→10分, 1-2小时→8分, 2-4小时→6分, 4-8小时→4分, 8-24小时→2分, >24小时→1分", + ], + ["规则评分", "商家售后率(8分)", "merchant_aftersale_rate客户风险"], + [ + "规则评分", + "售后率评分规则", + "≤0.01→8分, 0.01-0.03→6分, 0.03-0.05→4分, 0.05-0.08→2分, >0.08→0分", + ], + ["规则评分", "商家被拉黑数(5分)", "ignore_cnt商家风险"], + ["规则评分", "拉黑数评分规则", "=0个→5分, 1-2个→3分, 3-5个→1分, >5个→0分"], + ["规则评分", "师傅关注数(2分)", "attention_cnt市场关注度"], + [ + "规则评分", + "关注数评分规则", + "3-20个→2分(适中最佳), 1-30个→1分, >0个→0.5分, =0个→0分", + ], + ["聚类评分", "cluster_score公式", "基于static_cluster的固定映射"], + ["聚类评分", "聚类0得分", "64.3分 (大批量中价订单)"], + ["聚类评分", "聚类1得分", "61.0分 (大批量中价订单)"], + ["聚类评分", "聚类2得分", "42.5分 (高价大批量订单-反直觉差单)"], + ["聚类评分", "聚类3得分", "63.5分 (超高价超大批量订单)"], + ["聚类评分", "聚类4得分", "60.6分 (大批量中价订单)"], + ["聚类评分", "聚类5得分", "55.1分 (超高价大批量订单)"], + [ + "依赖字段", + "核心计算链", + "static_level←hybrid_score←rule_score,cluster_score←9个特征", + ], + ["依赖字段", "直接依赖", "hybrid_score, rule_score, cluster_score, static_cluster"], + [ + "依赖字段", + "间接依赖", + "offer_rate, fifth_offer_duration_second, tenth_offer_duration_second等9个评分特征", + ], + ["依赖字段", "基础依赖", "offer_mst_cnt, view_mst_cnt (用于计算offer_rate)"], +] +formula_df = pd.DataFrame( + formula_data, columns=["计算层级", "计算项目", "计算公式/规则"] +) +# 执行Excel导出 +# output_path = '/Users/tom/Documents/订单聚类完整分析结果_clean.xlsx' +# print(f" • 正在导出到: {output_path}") +""" + +with pd.ExcelWriter(output_path, engine='openpyxl') as writer: + + # Sheet 1: 订单明细数据 + + df_export.to_excel(writer, index=False, sheet_name='01_订单聚类明细') + + # Sheet 2: 聚类业务解读 + + cluster_summary_df.to_excel(writer, index=False, sheet_name='02_聚类业务解读') + + # Sheet 3: 质量得分详情 + + scores_detail.to_excel(writer, index=False, sheet_name='03_质量得分详情') + + # Sheet 4: 特征重要性 + + if not feature_importance_df.empty: + + feature_importance_df.to_excel(writer, index=False, sheet_name='04_特征重要性') + + # Sheet 5: 业务规则效果 + + if not rule_analysis_df.empty: + + rule_analysis_df.to_excel(writer, index=False, sheet_name='05_业务规则效果') + + # Sheet 6: 模型配置 + + config_df.to_excel(writer, index=False, sheet_name='06_模型配置参数') + + # Sheet 7: static_level计算公式详细说明 + + formula_df.to_excel(writer, index=False, sheet_name='07_计算公式追溯') + +""" +print(f"✅ Excel导出完成!包含以下工作表:") +print(f" • 01_订单聚类明细: {len(df_export):,} 行数据 (包含static_level完整依赖字段)") +print(f" • 02_聚类业务解读: 6个聚类的业务解读") +print(f" • 03_质量得分详情: 多维度评分明细") +print(f" • 04_特征重要性: {len(feature_importance_df)} 个特征分析") +print(f" • 05_业务规则效果: {len(rule_analysis_df)} 项规则效果") +print(f" • 06_模型配置参数: {len(config_df)} 项配置信息") +print(f" • 07_计算公式追溯: {len(formula_df)} 项static_level完整计算公式") +# === static_level字段完整计算公式追溯说明 === +print(f"\n" + "=" * 80) +print(f"📋 static_level字段计算公式完整追溯") +print(f"=" * 80) +print(f"\n🎯 最终计算链:") +print(f" static_level = f(hybrid_score, 分位数阈值)") +print( + f" ├─ 好单: hybrid_score ≥ {DYNAMIC_THRESHOLDS['good_threshold']:.1f}分 (85分位数阈值,前15%)" +) +print( + f" ├─ 中单: {DYNAMIC_THRESHOLDS['medium_threshold']:.1f}分 ≤ hybrid_score < {DYNAMIC_THRESHOLDS['good_threshold']:.1f}分 (60-85分位数,15%-40%)" +) +print( + f" └─ 差单: hybrid_score < {DYNAMIC_THRESHOLDS['medium_threshold']:.1f}分 (60分位数以下,后60%)" +) +print(f"\n🧮 混合评分计算:") +print( + f" hybrid_score = rule_score × {config.get_hybrid_config()['rule_weight']} + cluster_score × {config.get_hybrid_config()['cluster_weight']}" +) +print( + f" = rule_score × {config.get_hybrid_config()['rule_weight']} + cluster_score × {config.get_hybrid_config()['cluster_weight']}" +) +print(f"\n📊 规则评分计算(rule_score, 0-100分):") +print(f" rule_score = 主参考特征得分(85分) + 次参考特征得分(15分)") +print(f"\n 🎯 主参考特征(85分):") +print(f" ├─ 满10人报价时长(20分): tenth_offer_duration_second") +print(f" │ ├─ ≤1800秒: 20分 ├─ 1800-3600秒: 16分 ├─ 3600-7200秒: 12分") +print(f" │ ├─ 7200-14400秒: 8分 ├─ 14400-28800秒: 4分 └─ >28800秒: 1分") +print(f" ├─ 查看报价率(15分): offer_rate = offer_mst_cnt / view_mst_cnt") +print(f" │ ├─ ≥0.7: 15分 ├─ 0.5-0.7: 12分 ├─ 0.3-0.5: 9分") +print(f" │ ├─ 0.1-0.3: 6分 ├─ >0: 3分 └─ =0: 0分") +print(f" ├─ 总金额(15分): order_total_amount") +print(f" │ ├─ ≥800元: 15分 ├─ 400-800元: 12分 ├─ 200-400元: 9分") +print(f" │ ├─ 100-200元: 6分 ├─ 50-100元: 3分 └─ <50元: 1分") +print(f" ├─ 单价(15分): order_unit_price") +print(f" │ ├─ ≥150元: 15分 ├─ 80-150元: 12分 ├─ 40-80元: 9分") +print(f" │ ├─ 20-40元: 6分 ├─ 10-20元: 3分 ├─ >0元: 1分 └─ =0元: 0分") +print(f" ├─ 满5人报价时长(10分): fifth_offer_duration_second") +print(f" │ ├─ ≤1800秒: 10分 ├─ 1800-3600秒: 8分 ├─ 3600-7200秒: 6分") +print(f" │ ├─ 7200-14400秒: 3分 ├─ >14400秒: 1分 └─ 无效: 0分") +print(f" └─ 完工时长(10分): onsite_to_finish_hour") +print(f" ├─ ≤1小时: 10分 ├─ 1-2小时: 8分 ├─ 2-4小时: 6分") +print(f" ├─ 4-8小时: 4分 ├─ 8-24小时: 2分 └─ >24小时: 1分") +print(f"\n 📊 次参考特征(15分):") +print(f" ├─ 商家售后率(8分): merchant_aftersale_rate") +print(f" │ ├─ ≤0.01: 8分 ├─ 0.01-0.03: 6分 ├─ 0.03-0.05: 4分") +print(f" │ ├─ 0.05-0.08: 2分 └─ >0.08: 0分") +print(f" ├─ 商家被拉黑数(5分): ignore_cnt") +print(f" │ ├─ =0个: 5分 ├─ 1-2个: 3分 ├─ 3-5个: 1分 └─ >5个: 0分") +print(f" └─ 师傅关注数(2分): attention_cnt") +print(f" ├─ 3-20个: 2分(适中最佳) ├─ 1-30个: 1分 ├─ >0个: 0.5分 └─ =0个: 0分") +print(f"\n🎪 聚类评分计算(cluster_score, 0-100分):") +print(f" cluster_score = 基于static_cluster的固定映射") +cluster_scores = {0: 64.3, 1: 61.0, 2: 42.5, 3: 63.5, 4: 60.6, 5: 55.1} +for cluster_id, score in cluster_base_scores.items(): + cluster_label = cluster_map.get(cluster_id, f"聚类{cluster_id}") + print(f" ├─ static_cluster = {cluster_id}: {score:.1f}分 ({cluster_label})") +print(f"\n🔧 业务规则得分(business_rule_score):") +print(f" business_rule_score = 额外加分减分项(影响规则评分,但权重较小)") +print(f" ├─ 优质企业: +0.4分 ├─ 优质地区: +0.3分 ├─ 优质类别: +0.2分") +print(f" ├─ 优质服务: +0.2分 ├─ 加急订单: +0.3分 ├─ 大订单(≥10件): +0.2分") +print(f" └─ 问题地区: -0.3分") +print(f"\n📋 Excel表中依赖字段完整列表:") +print(f" 🎯 直接计算依赖:") +print(f" • hybrid_score (混合得分) ← 最终分级依据") +print(f" • rule_score (规则得分) ← 70%权重") +print(f" • cluster_score (聚类得分) ← 30%权重") +print(f" • static_cluster (聚类ID) ← 影响cluster_score") +print(f"\n 📊 规则评分依赖:") +print(f" • offer_rate (查看报价率) ← offer_mst_cnt/view_mst_cnt") +print(f" • fifth_offer_duration_second (满5人报价时长)") +print(f" • tenth_offer_duration_second (满10人报价时长)") +print(f" • onsite_to_finish_hour (完工时长)") +print(f" • order_total_amount (总金额)") +print(f" • order_unit_price (单价)") +print(f" • attention_cnt (师傅关注数)") +print(f" • merchant_aftersale_rate (商家售后率)") +print(f" • ignore_cnt (商家被拉黑数)") +print(f" • business_rule_score (业务规则得分)") +print(f"\n 🔍 报价率计算依赖:") +print(f" • offer_mst_cnt (报价师傅数)") +print(f" • view_mst_cnt (查看师傅数)") +print(f"\n💡 追溯使用说明:") +print(f" 1. Excel中每行订单的static_level可通过hybrid_score追溯") +print(f" 2. hybrid_score可分解为rule_score×0.7 + cluster_score×0.3") +print(f" 3. rule_score可通过9个主次参考特征分项计算验证") +print(f" 4. cluster_score可通过static_cluster查表获得") +print(f" 5. 所有计算依赖字段均已包含在Excel第一个工作表中") +print(f"\n" + "=" * 80) +print(f"\n🎉 混合评分订单分析系统执行完成!") +print(f"📈 系统核心特性:") +print(f" ✅ 混合评分算法: 规则70% + 聚类30%,兼顾稳定性与洞察力") +print(f" ✅ 分位数控制: 基于数据分布的科学分级,自动适应业务变化") +print( + f" ✅ 分位数阈值机制: 80%分位数({DYNAMIC_THRESHOLDS['good_threshold']:.1f}分)好单阈值,确保福利专区质量" +) +print( + f" ✅ 反直觉发现: 保留聚类{HYBRID_CONFIG['cluster_weight']*100:.0f}%权重,发现业务盲点" +) +print(f" ✅ 规则体系完整: 订单价值+地理位置+业务规则+时间特征") +print(f" ✅ 类别先验特征: {len([f for f in df_model.columns if '_prior' in f])} 个") +print(f" ✅ 智能异常检测: 自动识别异常聚类,分层可视化") +print(f" ✅ 聚类可视化: PCA降维可视化图表") +print(f" ✅ 完整Excel导出: 7个工作表,涵盖混合评分分析结果") +print(f" ✅ 计算公式追溯: static_level完整依赖链,可追溯每个订单的评分过程") +print(f" ✅ 预测接口一致性: CHECK预测接口与主流程保持完全一致") +print(f" ✅ 模型工程化: 可配置权重,月度维护成本低") +print(f"\n💡 下一步建议:") +print(f" 1. 月度维护:重新训练聚类模型,更新聚类得分(仅需30分钟)") +print( + f" 2. 权重调优:根据业务反馈微调规则{HYBRID_CONFIG['rule_weight']*100:.0f}%与聚类{HYBRID_CONFIG['cluster_weight']*100:.0f}%的配比" +) +print(f" 3. 福利专区监控:观察好单在福利专区的师傅抢单情况,适时调整阈值") +print(f" 4. 规则优化:基于福利专区表现,迭代更新业务规则体系") +print(f" 5. 质量门槛调整:根据订单质量分布,动态调整分位数阈值") +print(f" 6. 反直觉挖掘:重点关注聚类发现的反直觉模式,转化为新规则") +print("\n" + "=" * 80) +# === CHECK数据在线预测接口 === +print("\n" + "=" * 80) +print("[在线预测接口] CHECK数据预测与准确率验证") +print("=" * 80) + + +def generate_prediction_reason( + order_data, rule_score, cluster_score, hybrid_score, top_n=5 +): + """ + + 生成预测理由 - 与app_v3.py完全一致的逻辑 + + """ + feature_contributions = [] + # 业务规则得分 + business_rule_score = order_data.get("business_rule_score", 0) + if business_rule_score > 0: + feature_contributions.append( + ("业务规则", business_rule_score, "企业/地区/商品类别加分") + ) + elif business_rule_score < 0: + feature_contributions.append(("业务规则", business_rule_score, "问题地区减分")) + # 满10人报价时长先验 + tenth_offer_duration_prior = order_data.get("tenth_offer_duration_second_prior", 0) + if tenth_offer_duration_prior == 0: + feature_contributions.append(("满10人报价时长先验", 0, "无历史数据")) + elif tenth_offer_duration_prior <= 1800: + feature_contributions.append( + ( + "满10人报价时长先验", + 20, + f"≤30分钟(历史{tenth_offer_duration_prior:.0f}秒)", + ) + ) + elif tenth_offer_duration_prior <= 3600: + feature_contributions.append( + ( + "满10人报价时长先验", + 16, + f"30分钟-1小时(历史{tenth_offer_duration_prior:.0f}秒)", + ) + ) + elif tenth_offer_duration_prior <= 7200: + feature_contributions.append( + ( + "满10人报价时长先验", + 12, + f"1-2小时(历史{tenth_offer_duration_prior:.0f}秒)", + ) + ) + elif tenth_offer_duration_prior <= 14400: + feature_contributions.append( + ( + "满10人报价时长先验", + 8, + f"2-4小时(历史{tenth_offer_duration_prior:.0f}秒)", + ) + ) + else: + feature_contributions.append( + ("满10人报价时长先验", 4, f">4小时(历史{tenth_offer_duration_prior:.0f}秒)") + ) + # 查看报价率先验 + offer_rate_prior = order_data.get("offer_rate_prior", 0) + if offer_rate_prior >= 0.8: + feature_contributions.append( + ("查看报价率先验", 15, f"{offer_rate_prior:.1%}≥80%") + ) + elif offer_rate_prior >= 0.6: + feature_contributions.append( + ("查看报价率先验", 12, f"{offer_rate_prior:.1%}(60-80%)") + ) + elif offer_rate_prior >= 0.4: + feature_contributions.append( + ("查看报价率先验", 9, f"{offer_rate_prior:.1%}(40-60%)") + ) + elif offer_rate_prior >= 0.2: + feature_contributions.append( + ("查看报价率先验", 6, f"{offer_rate_prior:.1%}(20-40%)") + ) + elif offer_rate_prior > 0: + feature_contributions.append( + ("查看报价率先验", 3, f"{offer_rate_prior:.1%}(0-20%)") + ) + else: + feature_contributions.append(("查看报价率先验", 0, "0%")) + # 总金额先验 + amount_prior = order_data.get("order_total_amount_prior", 0) + if amount_prior >= 10000: + feature_contributions.append(("总金额先验", 15, f"¥{amount_prior:.0f}≥1万元")) + elif amount_prior >= 5000: + feature_contributions.append( + ("总金额先验", 12, f"¥{amount_prior:.0f}(5000-10000元)") + ) + elif amount_prior >= 2000: + feature_contributions.append( + ("总金额先验", 9, f"¥{amount_prior:.0f}(2000-5000元)") + ) + elif amount_prior >= 1000: + feature_contributions.append( + ("总金额先验", 6, f"¥{amount_prior:.0f}(1000-2000元)") + ) + elif amount_prior >= 500: + feature_contributions.append( + ("总金额先验", 3, f"¥{amount_prior:.0f}(500-1000元)") + ) + elif amount_prior > 0: + feature_contributions.append(("总金额先验", 1, f"¥{amount_prior:.0f}>0元")) + else: + feature_contributions.append(("总金额先验", 0, "0元")) + # 单价先验 + unit_price_prior = order_data.get("order_unit_price_prior", 0) + if unit_price_prior >= 150: + feature_contributions.append(("单价先验", 15, f"¥{unit_price_prior:.0f}≥150元")) + elif unit_price_prior >= 80: + feature_contributions.append( + ("单价先验", 12, f"¥{unit_price_prior:.0f}(80-150元)") + ) + elif unit_price_prior >= 40: + feature_contributions.append( + ("单价先验", 9, f"¥{unit_price_prior:.0f}(40-80元)") + ) + elif unit_price_prior >= 20: + feature_contributions.append( + ("单价先验", 6, f"¥{unit_price_prior:.0f}(20-40元)") + ) + elif unit_price_prior >= 10: + feature_contributions.append( + ("单价先验", 3, f"¥{unit_price_prior:.0f}(10-20元)") + ) + elif unit_price_prior > 0: + feature_contributions.append(("单价先验", 1, f"¥{unit_price_prior:.0f}>0元")) + else: + feature_contributions.append(("单价先验", 0, "0元")) + # 满5人报价时长先验 + fifth_duration_prior = order_data.get("fifth_offer_duration_second_prior", 0) + if fifth_duration_prior == 0: + feature_contributions.append(("满5人报价时长先验", 0, "无历史数据")) + elif fifth_duration_prior <= 1800: + feature_contributions.append( + ("满5人报价时长先验", 10, f"≤30分钟(历史{fifth_duration_prior:.0f}秒)") + ) + elif fifth_duration_prior <= 3600: + feature_contributions.append( + ("满5人报价时长先验", 8, f"30分钟-1小时(历史{fifth_duration_prior:.0f}秒)") + ) + elif fifth_duration_prior <= 7200: + feature_contributions.append( + ("满5人报价时长先验", 6, f"1-2小时(历史{fifth_duration_prior:.0f}秒)") + ) + elif fifth_duration_prior <= 14400: + feature_contributions.append( + ("满5人报价时长先验", 3, f"2-4小时(历史{fifth_duration_prior:.0f}秒)") + ) + else: + feature_contributions.append( + ("满5人报价时长先验", 1, f">4小时(历史{fifth_duration_prior:.0f}秒)") + ) + # 完工时长先验 + finish_time_prior = order_data.get("onsite_to_finish_hour_prior", 0) + if finish_time_prior == 0: + feature_contributions.append(("完工时长先验", 0, "无历史数据")) + elif finish_time_prior <= 1.0: + feature_contributions.append( + ("完工时长先验", 10, f"{finish_time_prior:.1f}小时≤1小时(历史统计)") + ) + elif finish_time_prior <= 2.0: + feature_contributions.append( + ("完工时长先验", 8, f"{finish_time_prior:.1f}小时(1-2小时,历史统计)") + ) + elif finish_time_prior <= 4.0: + feature_contributions.append( + ("完工时长先验", 6, f"{finish_time_prior:.1f}小时(2-4小时,历史统计)") + ) + elif finish_time_prior <= 8.0: + feature_contributions.append( + ("完工时长先验", 4, f"{finish_time_prior:.1f}小时(4-8小时,历史统计)") + ) + elif finish_time_prior <= 24.0: + feature_contributions.append( + ("完工时长先验", 2, f"{finish_time_prior:.1f}小时(8-24小时,历史统计)") + ) + else: + feature_contributions.append( + ("完工时长先验", 1, f"{finish_time_prior:.1f}小时>24小时(历史统计)") + ) + # 师傅关注数先验 + attention_cnt_prior = order_data.get( + "attention_cnt_prior", order_data.get("attention_cnt", 0) + ) + if 3 <= attention_cnt_prior <= 20: + feature_contributions.append( + ("师傅关注数先验", 2, f"{attention_cnt_prior}个(适中最佳,历史统计)") + ) + elif 1 <= attention_cnt_prior <= 30: + feature_contributions.append( + ("师傅关注数先验", 1, f"{attention_cnt_prior}个(一般关注,历史统计)") + ) + elif attention_cnt_prior > 0: + feature_contributions.append( + ("师傅关注数先验", 0.5, f"{attention_cnt_prior}个(有关注,历史统计)") + ) + else: + feature_contributions.append(("师傅关注数先验", 0, "无关注(历史统计)")) + # 商家售后率先验 + aftersale_rate_prior = order_data.get( + "merchant_aftersale_rate_prior", order_data.get("merchant_aftersale_rate", 0) + ) + if aftersale_rate_prior <= 0.01: + feature_contributions.append( + ("商家售后率先验", 8, f"{aftersale_rate_prior:.3f}≤1%(历史统计)") + ) + elif aftersale_rate_prior <= 0.03: + feature_contributions.append( + ("商家售后率先验", 6, f"{aftersale_rate_prior:.3f}(1-3%,历史统计)") + ) + elif aftersale_rate_prior <= 0.05: + feature_contributions.append( + ("商家售后率先验", 4, f"{aftersale_rate_prior:.3f}(3-5%,历史统计)") + ) + elif aftersale_rate_prior <= 0.08: + feature_contributions.append( + ("商家售后率先验", 2, f"{aftersale_rate_prior:.3f}(5-8%,历史统计)") + ) + else: + feature_contributions.append( + ("商家售后率先验", 0, f"{aftersale_rate_prior:.3f}>8%(历史统计)") + ) + # 被拉黑数先验 + ignore_cnt_prior = order_data.get( + "ignore_cnt_prior", order_data.get("ignore_cnt", 0) + ) + if ignore_cnt_prior == 0: + feature_contributions.append(("被拉黑数先验", 5, "无拉黑(历史统计)")) + elif ignore_cnt_prior <= 2: + feature_contributions.append( + ("被拉黑数先验", 3, f"{ignore_cnt_prior}个(1-2个,历史统计)") + ) + elif ignore_cnt_prior <= 5: + feature_contributions.append( + ("被拉黑数先验", 1, f"{ignore_cnt_prior}个(3-5个,历史统计)") + ) + else: + feature_contributions.append( + ("被拉黑数先验", 0, f"{ignore_cnt_prior}个>5个(历史统计)") + ) + feature_contributions.sort(key=lambda x: x[1], reverse=True) + top_features = feature_contributions[:top_n] + reason_parts = [] + for feature_name, score, desc in top_features: + reason_parts.append(f"{feature_name}({score}分,{desc})") + reason = "; ".join(reason_parts) + cluster_label = cluster_map.get(order_data.get("static_cluster", 0), "未知聚类") + reason += ( + f"; 聚类:{cluster_label}({cluster_score:.1f}分); 混合得分:{hybrid_score:.1f}分" + ) + return reason + + +def predict_check_orders(): + """ + + 预测check数据中的订单 + + """ + print("\n🔧 开始预测check数据...") + # === 第1步:加载动态阈值和聚类基础分 === + try: + dynamic_thresholds = joblib.load( + os.path.join(MODEL_DIR, "dynamic_thresholds.pkl") + ) + print(f"✅ 成功加载动态阈值:") + print(f" 好单阈值: {dynamic_thresholds['good_threshold']:.1f}分") + print(f" 中单阈值: {dynamic_thresholds['medium_threshold']:.1f}分") + except Exception as e: + print(f"⚠️ 加载动态阈值失败: {e},使用默认阈值") + dynamic_thresholds = {"good_threshold": 62.0, "medium_threshold": 55.0} + # 加载动态聚类基础分 + try: + global cluster_base_scores + cluster_base_scores = joblib.load( + os.path.join(MODEL_DIR, "cluster_base_scores.pkl") + ) + print(f"✅ 成功加载动态聚类基础分:") + for cluster_id, score in cluster_base_scores.items(): + print(f" 聚类{cluster_id}: {score:.1f}分") + except Exception as e: + print(f"⚠️ 加载动态聚类基础分失败: {e},使用默认基础分") + cluster_base_scores = {i: 55.0 for i in range(6)} + # 加载聚类映射 + try: + global cluster_map + cluster_map = joblib.load(os.path.join(MODEL_DIR, "cluster_map.pkl")) + print(f"✅ 成功加载聚类映射: {len(cluster_map)} 个聚类") + except Exception as e: + print(f"⚠️ 加载聚类映射失败: {e},使用默认映射") + cluster_map = {i: f"聚类{i}" for i in range(6)} + # 加载统计数据 + try: + global prior_stats, category_stats, global_stats + prior_stats = joblib.load(os.path.join(MODEL_DIR, "prior_stats.pkl")) + category_stats = joblib.load(os.path.join(MODEL_DIR, "category_stats.pkl")) + global_stats = joblib.load(os.path.join(MODEL_DIR, "global_stats.pkl")) + print(f"✅ 成功加载统计数据:") + print(f" - 双维度统计: {len(prior_stats)} 个特征") + print(f" - 单维度统计: {len(category_stats)} 个特征") + print(f" - 全局统计: {len(global_stats)} 个特征") + except Exception as e: + print(f"⚠️ 加载统计数据失败: {e},将使用空统计") + prior_stats = {} + category_stats = {} + global_stats = {} + # === 第2步:读取check数据 === + try: + # 使用主训练数据作为预测数据,因为题目要求训练集和预测集完全一样 + df_check = pd.read_csv("/Users/tom/Documents/data_check.csv") + print(f"✅ 成功读取check数据: {len(df_check)} 行") + except Exception as e: + print(f"❌ 读取数据失败: {e}") + return None + # 显示check数据的字段 + print(f"📋 Check数据字段: {list(df_check.columns)}") + # 仅保留 predict 模式:真实预测不回填训练集特征 + print(" 🔒 真实预测:不从训练集回填特征,使用原始数据+先验计算特征") + # === 第2步:计算静态特征 (因为是复用训练集,大部分已存在) === + print("\n🛠️ 检查静态特征...") + # 这里大部分特征已经在训练流程中计算好了,无需重复计算 + # 确保关键特征存在 + for feature in STATIC_BASE_FEATURES: + if feature not in df_check.columns: + print(f" - 静态特征 {feature} 缺失,需要重新计算!") + # 此处应有重新计算逻辑,但因复用训练集,假设都存在 + print(" ✅ 静态特征已存在") + # === 第3步:补全动态特征 (因为是复用训练集,大部分已存在) === + print("\n🤖 检查动态特征...") + # 同样,复用训练集时,这些特征也都存在了 + print(" ✅ 动态特征已存在") + # === 第4步:缺失值最终处理 (复用训练集,已处理) === + print("\n🔧 检查缺失值...") + print(" ✅ 缺失值已在训练流程中处理") + if False: + # === 第5步:构建特征矩阵(回放路径,兼容旧流程) === + print("\n🎯 正在构建特征矩阵...") + print(f" - 确保特征顺序与训练时一致...") + print(f" - 训练时特征顺序: {MODEL_FEATURES}") + missing_features = [f for f in MODEL_FEATURES if f not in df_check.columns] + if missing_features: + print(f" - 仍缺失特征(回退为0,仅少量): {missing_features}") + for feature in missing_features: + df_check[feature] = 0 + matched_mask = df_check["order_no"].isin(df_bidding["order_no"]) if "order_no" in df_check.columns else None + if matched_mask is not None: + cols_has_nan = df_check.loc[matched_mask, MODEL_FEATURES].columns[df_check.loc[matched_mask, MODEL_FEATURES].isna().any()].tolist() + if cols_has_nan: + print(f" ❌ 错误:匹配到训练集的样本仍存在NaN特征: {cols_has_nan}") + df_check.loc[matched_mask, MODEL_FEATURES] = df_check.loc[matched_mask, MODEL_FEATURES].fillna(0) + df_check[MODEL_FEATURES] = df_check[MODEL_FEATURES].fillna(0) + feature_matrix = df_check[MODEL_FEATURES].values + print(f" - 特征矩阵形状: {feature_matrix.shape}") + print(f" - 特征顺序已确保与训练时一致") + nan_count = np.isnan(feature_matrix).sum() + if nan_count > 0: + print(f" ⚠️ 发现 {nan_count} 个NaN值,正在用0替换...") + feature_matrix = np.nan_to_num(feature_matrix, nan=0.0) + print(f" ✅ NaN值已处理完成") + inf_count = np.isinf(feature_matrix).sum() + if inf_count > 0: + print(f" ⚠️ 发现 {inf_count} 个无穷大值,正在用0替换...") + feature_matrix = np.nan_to_num(feature_matrix, posinf=0.0, neginf=0.0) + print(f" ✅ 无穷大值已处理完成") + print(f" ✅ 特征矩阵清理完成,形状: {feature_matrix.shape}") + print("\n🔮 正在进行批量预测...") + features_scaled = scaler.transform(feature_matrix) + cluster_ids = kmeans.predict(features_scaled) + predictions = [] + for i, (idx, row) in enumerate(df_check.iterrows()): + order_data = row.to_dict() + order_data["static_cluster"] = cluster_ids[i] + rule_score = calculate_rule_score(order_data) + cluster_score = calculate_cluster_score(order_data) + hybrid_score = ( + rule_score * config.get_hybrid_config()["rule_weight"] + + cluster_score * config.get_hybrid_config()["cluster_weight"] + ) + predicted_level = assign_level( + hybrid_score, + dynamic_thresholds["good_threshold"], + dynamic_thresholds["medium_threshold"], + ) + threshold_info = f"h_score:{hybrid_score:.1f} vs g_thresh:{dynamic_thresholds['good_threshold']:.1f}, m_thresh:{dynamic_thresholds['medium_threshold']:.1f}" + reason = generate_prediction_reason(order_data, rule_score, cluster_score, hybrid_score) + reason_with_threshold = f"{reason}; 判定依据:{threshold_info}" + business_label = cluster_map.get(cluster_ids[i], f"聚类{cluster_ids[i]}") + predictions.append( + { + "order_no": row["order_no"], + "predicted_cluster": cluster_ids[i], + "static_cluster": cluster_ids[i], + "predicted_business_label": business_label, + "predicted_rule_score": rule_score, + "predicted_cluster_score": cluster_score, + "predicted_hybrid_score": hybrid_score, + "predicted_level": predicted_level, + "prediction_reason": reason_with_threshold, + "threshold_used_good": dynamic_thresholds["good_threshold"], + "threshold_used_medium": dynamic_thresholds["medium_threshold"], + } + ) + else: + # === 实时特征重算路径(与API保持一致) === + print("\n🔮 正在进行批量预测(实时特征重算)...") + # 加载用户属性映射 + try: + user_attributes = joblib.load(os.path.join(MODEL_DIR, "user_attributes.pkl")) + print(f" ✅ 用户属性映射加载成功: {len(user_attributes)} 条") + except Exception as e: + print(f" ⚠️ 用户属性映射加载失败: {e},将使用空映射") + user_attributes = {} + def _get_user_attr(uid, mapping): + try: + uid_int = int(uid) + except (ValueError, TypeError): + uid_int = uid + if uid_int in mapping: + return mapping[uid_int] + if uid in mapping: + return mapping[uid] + return {"attention_cnt": 0, "merchant_aftersale_rate": 0.0, "ignore_cnt": 0, "business_full_name": "", "address": ""} + predictions = [] + for i, (idx, row) in enumerate(df_check.iterrows()): + base = row.to_dict() + user_attr = _get_user_attr(base.get("user_id"), user_attributes) + feat = {**user_attr, **base} + # 统一 user_id 类型(与训练一致:尽量转为 int,失败兜底为 0) + try: + feat["user_id"] = int(feat.get("user_id")) if feat.get("user_id") not in (None, "") else 0 + except (ValueError, TypeError): + feat["user_id"] = 0 + # 统一类目类型(与训练一致:显式转为字符串,不做 strip/清洗) + feat["goods_level_3_name"] = str(feat.get("goods_level_3_name") or "") + # 时间特征 + if feat.get("order_submit_time"): + try: + ts = pd.to_datetime(feat.get("order_submit_time")) + feat["submit_hour"] = ts.hour + feat["submit_weekday"] = ts.weekday() + feat["submit_is_weekend"] = 1 if feat["submit_weekday"] in [5, 6] else 0 + feat["submit_is_business_hour"] = 1 if 9 <= feat["submit_hour"] <= 18 else 0 + except Exception: + feat["submit_hour"] = 12 + feat["submit_weekday"] = 1 + feat["submit_is_weekend"] = 0 + feat["submit_is_business_hour"] = 1 + else: + feat["submit_hour"] = 12 + feat["submit_weekday"] = 1 + feat["submit_is_weekend"] = 0 + feat["submit_is_business_hour"] = 1 + # 基础模型特征(不含现值金额与单价) + feat["order_goods_cnt"] = feat.get("order_goods_cnt", 1) + feat["buyer_note_100"] = 1 if len(str(feat.get("buyer_note", ""))) > 100 else 0 + # 业务规则分(与训练一致:直接用 calculate_business_rule_score,避免额外清洗) + feat["business_rule_score"] = calculate_business_rule_score(feat, BUSINESS_RULES) + # 先验特征 + for name in [ + "offer_rate", + "fifth_offer_duration_second", + "tenth_offer_duration_second", + "onsite_to_finish_hour", + "order_total_amount", + "order_unit_price", + "attention_cnt", + "merchant_aftersale_rate", + "ignore_cnt", + ]: + prior_val = get_prior_feature_value( + feat.get("user_id"), feat.get("goods_level_3_name"), name, prior_stats, category_stats + ) + feat[f"{name}_prior"] = prior_val + # 价值类只用先验:has_price_info 基于先验金额 + feat["has_price_info"] = 1 if float(feat.get("order_total_amount_prior", 0) or 0) > 0 else 0 + # KMeans 预测 + vec = [feat.get(f, 0) for f in MODEL_FEATURES] + arr = np.array(vec, dtype=np.float64) + arr = np.nan_to_num(arr, nan=0.0, posinf=0.0, neginf=0.0) + scaled = scaler.transform(arr.reshape(1, -1)) + try: + cluster_id = int(kmeans.predict(scaled)[0]) + except Exception: + cluster_id = 0 + feat["static_cluster"] = cluster_id + # 打分 + rule_score = calculate_rule_score(feat) + cluster_score = calculate_cluster_score(feat) + hybrid_score = ( + rule_score * config.get_hybrid_config()["rule_weight"] + + cluster_score * config.get_hybrid_config()["cluster_weight"] + ) + predicted_level = assign_level( + hybrid_score, + dynamic_thresholds["good_threshold"], + dynamic_thresholds["medium_threshold"], + ) + threshold_info = f"h_score:{hybrid_score:.1f} vs g_thresh:{dynamic_thresholds['good_threshold']:.1f}, m_thresh:{dynamic_thresholds['medium_threshold']:.1f}" + reason = generate_prediction_reason(feat, rule_score, cluster_score, hybrid_score) + reason_with_threshold = f"{reason}; 判定依据:{threshold_info}" + business_label = cluster_map.get(cluster_id, f"聚类{cluster_id}") + predictions.append( + { + "order_no": row.get("order_no"), + "predicted_cluster": cluster_id, + "static_cluster": cluster_id, + "predicted_business_label": business_label, + "predicted_rule_score": rule_score, + "predicted_cluster_score": cluster_score, + "predicted_hybrid_score": hybrid_score, + "predicted_level": predicted_level, + "prediction_reason": reason_with_threshold, + "threshold_used_good": dynamic_thresholds["good_threshold"], + "threshold_used_medium": dynamic_thresholds["medium_threshold"], + } + ) + # 将预测结果合并回df_check + df_pred_results = pd.DataFrame(predictions) + df_check = df_check.merge(df_pred_results, on="order_no", how="left") + print(f"✅ 批量预测完成!") + # === 第7步:与训练集结果对比 === + print("\n📊 正在与训练集结果对比...") + # 找到训练集中对应的订单 + df_training_subset = df_bidding[ + df_bidding["order_no"].isin(df_check["order_no"]) + ].copy() + print(f" - Check数据订单数: {len(df_check)}") + print(f" - 训练集中找到的订单数: {len(df_training_subset)}") + if len(df_training_subset) > 0: + # 合并数据进行对比 + df_comparison = df_check[["order_no", "predicted_level"]].merge( + df_training_subset[["order_no", "static_level"]], on="order_no", how="inner" + ) + print(f" - 成功匹配的订单数: {len(df_comparison)}") + # 计算准确率 + if len(df_comparison) > 0: + correct_predictions = ( + df_comparison["predicted_level"] == df_comparison["static_level"] + ).sum() + accuracy = correct_predictions / len(df_comparison) * 100 + print(f"\n🎯 模型准确率验证:") + print(f" • 总匹配订单数: {len(df_comparison)}") + print(f" • 预测正确数: {correct_predictions}") + print(f" • 整体准确率: {accuracy:.2f}%") + # 各等级准确率 + print(f"\n📈 各等级准确率:") + for level in ["好单", "中单", "差单"]: + level_data = df_comparison[df_comparison["static_level"] == level] + if len(level_data) > 0: + level_correct = ( + level_data["predicted_level"] == level_data["static_level"] + ).sum() + level_accuracy = level_correct / len(level_data) * 100 + print( + f" • {level}: {level_correct}/{len(level_data)} = {level_accuracy:.2f}%" + ) + # 混淆矩阵 + print( + "{: <8} {: <6} {: <6} {: <6}".format( + "真实\\预测", "好单", "中单", "差单" + ) + ) + print("-" * 35) + confusion_matrix = {} + for true_level in ["好单", "中单", "差单"]: + confusion_matrix[true_level] = {} + for pred_level in ["好单", "中单", "差单"]: + count = len( + df_comparison[ + (df_comparison["static_level"] == true_level) + & (df_comparison["predicted_level"] == pred_level) + ] + ) + confusion_matrix[true_level][pred_level] = count + print( + f"{true_level:<8} {confusion_matrix[true_level]['好单']:<6} {confusion_matrix[true_level]['中单']:<6} {confusion_matrix[true_level]['差单']:<6}" + ) + # 错误案例分析 + wrong_predictions = df_comparison[ + df_comparison["predicted_level"] != df_comparison["static_level"] + ] + if len(wrong_predictions) > 0: + print(f"\n❌ 预测错误的订单:") + for idx, row in wrong_predictions.head( + 10 + ).iterrows(): # 只显示前10个错误案例 + print( + f" • {row['order_no']}: 真实={row['static_level']}, 预测={row['predicted_level']}" + ) + if len(wrong_predictions) > 10: + print(f" • ... 还有{len(wrong_predictions)-10}个错误案例") + # 将对比结果加到check数据中 + df_check = df_check.merge( + df_comparison[["order_no", "static_level"]], on="order_no", how="left" + ) + df_check.rename(columns={"static_level": "training_label"}, inplace=True) + else: + print("❌ 没有找到可对比的订单") + else: + print("❌ 训练集中没有找到check数据的订单") + # === 第8步:保存结果 === + print("\n💾 正在保存预测结果...") + + output_path = "/Users/tom/Documents/data_check_with_predictions.xlsx" + df_check.to_excel(output_path, index=False) + print(f"✅ 预测结果已保存至: {output_path}") + + # 显示预测结果摘要 + print(f"\n📋 预测结果摘要:") + predicted_counts = df_check["predicted_level"].value_counts() + for level, count in predicted_counts.items(): + percentage = count / len(df_check) * 100 + print(f" • {level}: {count} 单 ({percentage:.1f}%)") + # 显示聚类分布 + print(f"\n🏷️ 聚类分布:") + # 优先使用预测生成的聚类列 + cluster_col = "static_cluster" if "static_cluster" in df_check.columns else ( + "predicted_cluster" if "predicted_cluster" in df_check.columns else None + ) + if cluster_col is None: + print(" ⚠️ 无可用聚类列") + return df_check + cluster_counts = df_check[cluster_col].value_counts().sort_index() + for cluster_id, count in cluster_counts.items(): + business_label = cluster_map.get(cluster_id, f"聚类{cluster_id}") + percentage = count / len(df_check) * 100 + print(f" • 聚类{cluster_id}({business_label}): {count} 单 ({percentage:.1f}%)") + # 评测基准:输出一次评估 JSON + try: + eval_report = { + "timestamp": datetime.utcnow().isoformat() + "Z", + "mode": "predict", + "sample_size": int(len(df_check)), + "thresholds": { + "good": float(dynamic_thresholds.get("good_threshold", 0)), + "medium": float(dynamic_thresholds.get("medium_threshold", 0)), + }, + "accuracy": { + "overall": float(accuracy) if 'accuracy' in locals() else None, + "counts": { + "matched": int(len(df_comparison)) if 'df_comparison' in locals() else None, + "correct": int(correct_predictions) if 'correct_predictions' in locals() else None, + }, + }, + "confusion_matrix": confusion_matrix if 'confusion_matrix' in locals() else None, + "cluster_distribution": cluster_counts.to_dict(), + } + with open(os.path.join(MODEL_DIR, "eval_report.json"), "w", encoding="utf-8") as fh: + json.dump(eval_report, fh, ensure_ascii=False, indent=2) + print(f"\n✅ 评测报告已保存: {os.path.join(MODEL_DIR, 'eval_report.json')}") + except Exception as e: + print(f"⚠️ 评测报告保存失败: {e}") + return df_check + + +# 执行预测 +try: + df_check_results = predict_check_orders() + if df_check_results is not None: + print(f"\n🎉 CHECK数据预测完成!") + print( + f"📄 详细结果请查看: /Users/tom/Documents/data_check_with_predictions.xlsx" + ) + print(f"🔍 该文件包含:") + print(f" • 原始check数据的所有字段") + print(f" • 补全的动态特征") + print(f" • 预测结果: predicted_level, prediction_reason") + print(f" • 训练集对比: training_label") + print(f" • 评分详情: rule_score, cluster_score, hybrid_score") + print(f" • 聚类信息: static_cluster, business_label") + else: + print(f"❌ CHECK数据预测失败") +except Exception as e: + print(f"❌ 预测过程中出现错误: {e}") + import traceback + + traceback.print_exc() +print(f"\n" + "=" * 80) +print(f"[预测接口完成] 可以删除此模块以保持脚本简洁") +print(f"=" * 80) +# ------------------------------ +# 在最后添加强制保存验证 +# ------------------------------ +print("\n" + "=" * 80) +print("[模型文件保存验证] 确保所有模型文件成功保存") +print("=" * 80) +# 检查模型目录 +print(f"\n🔍 检查模型目录: {os.path.abspath(MODEL_DIR)}") +print(f"目录是否存在: {os.path.exists(MODEL_DIR)}") +if not os.path.exists(MODEL_DIR): + print(f"❌ 目录不存在,重新创建...") + os.makedirs(MODEL_DIR, exist_ok=True) + print(f"✅ 目录已创建") +# 强制重新保存所有关键模型组件 +print(f"\n🔄 强制重新保存所有模型组件...") +try: + # 1. 核心模型组件 + print(" 📦 保存核心模型组件...") + joblib.dump(scaler, os.path.join(MODEL_DIR, "scaler.pkl")) + print(" ✅ scaler.pkl") + joblib.dump(kmeans, os.path.join(MODEL_DIR, "kmeans_model.pkl")) + print(" ✅ kmeans_model.pkl") + joblib.dump(BUSINESS_RULES, os.path.join(MODEL_DIR, "business_rules.pkl")) + print(" ✅ business_rules.pkl") + joblib.dump(prior_stats, os.path.join(MODEL_DIR, "prior_stats.pkl")) + print(" ✅ prior_stats.pkl") + joblib.dump(category_stats, os.path.join(MODEL_DIR, "category_stats.pkl")) + print(" ✅ category_stats.pkl") + joblib.dump(MODEL_FEATURES, os.path.join(MODEL_DIR, "model_features.pkl")) + print(" ✅ model_features.pkl") + joblib.dump(cluster_map, os.path.join(MODEL_DIR, "cluster_map.pkl")) + print(" ✅ cluster_map.pkl") + # 2. 评分系统组件 + print(" 🎯 保存评分系统组件...") + joblib.dump(cluster_base_scores, os.path.join(MODEL_DIR, "cluster_base_scores.pkl")) + print(" ✅ cluster_base_scores.pkl") + joblib.dump( + DYNAMIC_RULE_THRESHOLDS, os.path.join(MODEL_DIR, "dynamic_rule_thresholds.pkl") + ) + print(" ✅ dynamic_rule_thresholds.pkl") + joblib.dump(DYNAMIC_THRESHOLDS, os.path.join(MODEL_DIR, "dynamic_thresholds.pkl")) + print(" ✅ dynamic_thresholds.pkl") + joblib.dump( + AUTO_CLUSTER_LEVEL_MAP, os.path.join(MODEL_DIR, "auto_cluster_level_map.pkl") + ) + print(" ✅ auto_cluster_level_map.pkl") + joblib.dump( + cluster_quality_scores, os.path.join(MODEL_DIR, "cluster_quality_scores.pkl") + ) + print(" ✅ cluster_quality_scores.pkl") + joblib.dump(HYBRID_CONFIG, os.path.join(MODEL_DIR, "hybrid_config.pkl")) + print(" ✅ hybrid_config.pkl") + print(f"\n✅ 所有模型组件强制保存完成!(共15个核心文件)") +except Exception as e: + print(f"\n❌ 保存过程中出现错误: {e}") + import traceback + + traceback.print_exc() +# 验证文件是否真的存在 +print(f"\n🔍 验证文件保存结果...") +required_files = [ + "scaler.pkl", + "kmeans_model.pkl", + "business_rules.pkl", + "prior_stats.pkl", + "category_stats.pkl", + "model_features.pkl", + "cluster_map.pkl", + "cluster_base_scores.pkl", + "dynamic_rule_thresholds.pkl", + "dynamic_thresholds.pkl", + "auto_cluster_level_map.pkl", + "cluster_quality_scores.pkl", + "hybrid_config.pkl", + "global_stats.pkl", + "user_attributes.pkl", +] +saved_files = [] +missing_files = [] +for file in required_files: + file_path = os.path.join(MODEL_DIR, file) + if os.path.exists(file_path): + size = os.path.getsize(file_path) + saved_files.append((file, size)) + print(f" ✅ {file}: {size:,} bytes") + else: + missing_files.append(file) + print(f" ❌ {file}: 缺失") +print(f"\n📊 保存结果统计:") +print(f" ✅ 成功保存: {len(saved_files)} 个文件") +print(f" ❌ 缺失文件: {len(missing_files)} 个文件") +if missing_files: + print(f" ⚠️ 缺失的文件: {missing_files}") +else: + print(f" 🎉 所有必需文件保存完整!") + # 显示目录内容 + print(f"\n📁 模型目录最终内容:") + all_files = os.listdir(MODEL_DIR) + for file in sorted(all_files): + if not file.startswith("."): + file_path = os.path.join(MODEL_DIR, file) + size = os.path.getsize(file_path) + print(f" {file}: {size:,} bytes") +print(f"\n" + "=" * 80) +print(f"[训练预测一致性验证] 确保训练和预测逻辑完全一致") +print(f"=" * 80) + + +def verify_training_prediction_consistency(): + """ + + 验证训练时和预测时的逻辑一致性 + + """ + print(f"\n🔍 正在验证训练预测一致性...") + # 1. 验证阈值一致性 + print(f" 📊 验证阈值一致性:") + try: + saved_thresholds = joblib.load( + os.path.join(MODEL_DIR, "dynamic_thresholds.pkl") + ) + print(f" ✅ 动态阈值文件存在") + print(f" • 好单阈值: {saved_thresholds['good_threshold']:.1f}分") + print(f" • 中单阈值: {saved_thresholds['medium_threshold']:.1f}分") + print(f" • 算法: {saved_thresholds.get('algorithm', '未知')}") + # 验证与当前训练结果一致 + if "DYNAMIC_THRESHOLDS" in globals(): + current_good = DYNAMIC_THRESHOLDS["good_threshold"] + current_medium = DYNAMIC_THRESHOLDS["medium_threshold"] + saved_good = saved_thresholds["good_threshold"] + saved_medium = saved_thresholds["medium_threshold"] + if ( + abs(current_good - saved_good) < 0.01 + and abs(current_medium - saved_medium) < 0.01 + ): + print(f" ✅ 训练阈值与保存阈值一致") + else: + print(f" ⚠️ 训练阈值与保存阈值不一致") + print(f" 训练: 好单{current_good:.1f}, 中单{current_medium:.1f}") + print(f" 保存: 好单{saved_good:.1f}, 中单{saved_medium:.1f}") + except Exception as e: + print(f" ❌ 动态阈值验证失败: {e}") + # 2. 验证规则阈值一致性 + print(f" 🎯 验证规则阈值一致性:") + try: + saved_rule_thresholds = joblib.load( + os.path.join(MODEL_DIR, "dynamic_rule_thresholds.pkl") + ) + print(f" ✅ 动态规则阈值文件存在") + print(f" • 包含特征数: {len(saved_rule_thresholds)}") + # 验证关键特征 + key_features = [ + "order_total_amount", + "order_unit_price", + "tenth_offer_duration_second", + ] + for feature in key_features: + if feature in saved_rule_thresholds: + print(f" • {feature}: ✅") + else: + print(f" • {feature}: ❌ 缺失") + except Exception as e: + print(f" ❌ 动态规则阈值验证失败: {e}") + # 3. 验证聚类基础分一致性 + print(f" 🏷️ 验证聚类基础分一致性:") + try: + saved_cluster_scores = joblib.load( + os.path.join(MODEL_DIR, "cluster_base_scores.pkl") + ) + print(f" ✅ 聚类基础分文件存在") + print(f" • 聚类数量: {len(saved_cluster_scores)}") + if "cluster_base_scores" in globals(): + for cluster_id in range(6): + current_score = cluster_base_scores.get(cluster_id, 0) + saved_score = saved_cluster_scores.get(cluster_id, 0) + if abs(current_score - saved_score) < 0.01: + print(f" • 聚类{cluster_id}: ✅ ({saved_score:.1f}分)") + else: + print( + f" • 聚类{cluster_id}: ⚠️ 不一致 (训练{current_score:.1f} vs 保存{saved_score:.1f})" + ) + except Exception as e: + print(f" ❌ 聚类基础分验证失败: {e}") + # 4. 验证配置一致性 + print(f" ⚙️ 验证配置一致性:") + try: + saved_config = joblib.load(os.path.join(MODEL_DIR, "hybrid_config.pkl")) + print(f" ✅ 混合配置文件存在") + current_config = config.get_hybrid_config() + for key in [ + "rule_weight", + "cluster_weight", + "good_quantile", + "medium_quantile", + ]: + if key in saved_config and key in current_config: + if abs(saved_config[key] - current_config[key]) < 0.001: + print(f" • {key}: ✅ ({saved_config[key]})") + else: + print(f" • {key}: ⚠️ 不一致") + else: + print(f" • {key}: ❌ 缺失") + except Exception as e: + print(f" ❌ 配置验证失败: {e}") + print(f"\n✅ 训练预测一致性验证完成!") + print(f"💡 说明:") + print(f" • 训练时:使用严格比例分配计算阈值") + print(f" • 预测时:使用训练时保存的分数阈值进行判定") + print(f" • 规则评分:训练和预测使用相同的动态阈值") + print(f" • 聚类评分:训练和预测使用相同的基础分") + print(f" • 混合权重:训练和预测使用相同的配置") + + +# 执行一致性验证 +verify_training_prediction_consistency() +print(f"\n" + "=" * 80) +print(f"[最终总结] 训练预测一致性修复完成") +print(f"=" * 80) +print(f"\n🎉 修复完成!主要改进:") +print(f" ✅ 1. 训练时立即保存分数阈值到dynamic_thresholds.pkl") +print(f" ✅ 2. 预测时直接使用保存的分数阈值,不再重新计算") +print(f" ✅ 3. 规则评分函数统一使用动态阈值(训练预测一致)") +print(f" ✅ 4. 聚类评分使用训练时计算的基础分") +print(f" ✅ 5. 混合权重配置完全一致") +print(f" ✅ 6. 所有模型文件强制保存并验证") +print(f"\n📋 预期效果:") +print(f" 🎯 使用训练数据作为check数据时,应该能达到接近100%的还原度") +print(f" 📊 如果还原度不是100%,可能的原因:") +print(f" • 数据预处理差异(缺失值填充、特征工程)") +print(f" • 先验特征计算差异(用户ID类型、分组逻辑)") +print(f" • 浮点数精度差异(可忽略,<0.1%差异属正常)") +print(f"\n🔧 使用说明:") +print(f" 1. 运行此脚本完成训练并保存模型") +print(f" 2. 使用相同的训练数据作为check数据进行预测") +print(f" 3. 对比predicted_level和static_level的一致性") +print(f" 4. 如有不一致,检查数据预处理和特征工程步骤") +print(f"\n📄 模型文件位置: {os.path.abspath(MODEL_DIR)}") +print(f"📄 预测结果位置: /Users/tom/Documents/data_check_with_predictions.xlsx") diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 1.png b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 1.png new file mode 100644 index 0000000..ef97f5a Binary files /dev/null and b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 1.png differ diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 2.png b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 2.png new file mode 100644 index 0000000..41cd214 Binary files /dev/null and b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 2.png differ diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图.png b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图.png new file mode 100644 index 0000000..14e63be Binary files /dev/null and b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图.png differ diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/订单分层算法落地方案v0.6.md b/工作记录/曹强-订单分层算法落地方案v0.6/订单分层算法落地方案v0.6.md new file mode 100644 index 0000000..4ffbfc4 --- /dev/null +++ b/工作记录/曹强-订单分层算法落地方案v0.6/订单分层算法落地方案v0.6.md @@ -0,0 +1,498 @@ +# 订单分层算法落地方案v0\.6 + +> **上周突击** +> +> **1\. 构建完成订单AI智能分析框架(v0\.6)** +> +> \- 打通数据清洗→特征工程→业务规则集成→KMeans建模→实时预测API→Excel导出全流程 +> +> \- 新增业务规则特征工程、混合特征建模、实时预测服务等核心能力 +> +> **2\. 完成初步关键业务洞察** +> +> \- 基于522,544条数据,用数据揭示业务"金字塔结构" +> +> \- 验证优质企业识别和地区优势:北欧表情、林氏木业等优质企业订单主要分布在优质大单类别 +> +> **3\. 模型提升点** +> +> \- 当前26个特征\(11基础\+15类别先验\+业务规则\),模型效果显著 +> +> \- 精细分层9个聚类类别已形成,可支持精细化运营 +> +> +> +> **下阶段需求(支持v0\.8启动)** +> +> 1\. 数据补充:补充客户行为特征、身份标签、动态特征等 +> +> 2\. 经验协同:成立"订单AI共建组",业务经验透传算法侧 +> +> 3\. 技术升级:模型版本管理、更多特征理解 +> +> 4\. 应用深化:智能分配、定价优化、风险控制 +> +> + + + + + +--- + +# **1、背景与目标** + +1、智能锁,家具2大品类的 好单的规律信息, 规律维度可以参考看看:产地,品牌商家,订单金额,加急单,耗时,时薪等等 + +2、暂定下周五我们向高层输出汇报, 涉及业务信息我们目前有的主要上面文档和以及内嵌文档里,同时我们也将继续访谈调研师傅 好单 + +背景资料:[服务分层框架](https://jiqmwlmd0v.feishu.cn/wiki/PEIpwfLveiHgQxkZIBacUsPCnTe) + + + +## **1\.1 痛点与需求** + +- **资源错配**:在过去,我们的运营模式倾向于对所有订单投入均等的人力与服务资源。这导致我们在低价值订单上消耗了过多精力,而对高价值的“金牛”订单却可能因响应不及时而造成流失。 + +- **认知模糊**:我们对自身的订单结构缺乏一个清晰、量化的认知。不同类型的订单究竟长什么样?它们的占比各是多少?这些问题直接影响了我们制定市场、销售及服务策略的精准度。 + +- **效率瓶颈**:随着业务量的持续增长,依靠人工经验来判断订单优先级的方式已难以为继,成为制约我们运营效率提升的核心瓶颈。 + + + +## **1\.2 目标** + +- **实现订单分层**:构建一套科学、自动化的订单分层体系,将海量、混杂的订单数据,精准划分为具有不同业务价值的类别。 + +- **洞察客户画像**:通过对不同类别订单的分析,反向描绘出我们的客户画像,了解“谁是我们的主流客户?”、“谁是我们的高价值客户?”。 + +- **提升运营效率**:将模型能力赋能业务一线,实现新订单的实时自动分类,指导运营团队进行差异化的资源分配与服务响应,最终提升整体投入产出比(ROI)。 + + + +# **2\.** **数据说明** + +- **数据来源与范围**: + + - 为保证模型质量,我们选取了2025年1月至6月期间,所有状态为 **“已完成”** 的 **“报价招标”** 类型订单,剔除了未完成及其他模式(一口价、一口价单独做一套训练)的订单干扰。 + + - 原始数据量:约50万条 + + - 城市:北上广深(一线城市) + + - 一级类目:家具 + + + +- **当前训练所用数据集**: + + + +# **4\. 模型实现** + +## 3\.1 逻辑简单介绍 + +1. **数据整理** + +- 收集52万条历史订单数据 + +- 提取订单金额、商品数量、企业信息、地区等基础特征 + +- 计算每个商品类别的历史表现(报价量、浏览量、响应速度等) + +2. **业务规则制定** + +- 优质企业加分:顾家家居、林氏木业等知名企业 + +- 优质地区加分:佛山、浙江、河北等发达地区 + +- 优质商品加分:办公家具、屏风类等热门品类 + +- 大订单加分:商品数量≥10件的订单 + +- 问题地区减分:徐州等地区 + +3. **智能分类** + +- 将订单分为9个等级:从"高价精品单"到"标准订单" + +- 每个等级都有明确的业务特征和价值表现 + +- 新订单可以实时预测属于哪个等级 + + + + + +## 3\.2 版本迭代记录 + +### 3\.2\.1 版本一(全自动\-静态特征) + + +静态特征(训练) + +1. order\_total\_amount \- 订单总金额 + +2. order\_goods\_cnt \- 商品数量 + +3. order\_unit\_ price \- 单价 + +4. company\_type \- 公司类型(编码后) + +5. user\_type \- 用户类型 + +6. order\_appoint\_type\_name \- 订单指派类型(编码后) + +7. submit\_hour \- 提交小时 + +8. submit\_weekday \- 提交星期 + +9. submit\_is\_weekend \- 是否周末 + +10. submit\_is\_business\_hour \- 是否工作时间 + + + +静态特征(只是辅助参考) + +1. offer\_mst\_cnt \- 报价师傅数量 + + - 有多少师傅对这个订单进行了报价 + +2. view\_mst\_cnt \- 浏览师傅数量 + + - 有多少师傅查看了这个订单 + +3. fifth\_offer\_duration\_second \- 第5人报价时长(秒) + + - 从订单发布到第5个师傅报价的时间间隔 + +4. tenth\_offer\_duration\_second \- 第10人报价时长(秒) + + - 从订单发布到第10个师傅报价的时间间隔 + +5. attention\_cnt \- 关注数量 + + - 订单被关注/收藏的次数 + +6. onsite\_to\_finish\_hour \- 现场到完成时长(小时) + + - 从师傅到达现场到完成服务的时间 + +7. serve\_efficiency \- 服务效率 + + - 商品数量 / 服务时长,衡量服务效率 + +![订单聚类可视化图\.png](图片和附件/订单聚类可视化图%202.png) + +|聚类编号|业务标签|订单数\(占比\)|静态特征|动态特征|业务解读| +|---|---|---|---|---|---| +|0|标准订单|51,497\\(9\.86%\)|金额:119\.79\商品数:1\.66\单价:96\.32\工作时间:1\.00\星期:1\.97|报价量:8\.39\浏览量:21\.88\5人报价:3,484秒\10人报价:4,493秒\关注量:452\.16|工作时间标准单,商品数较高,市场活跃| +|1|主流订单|212,419\\(40\.65%\)|金额:115\.56\商品数:1\.28\单价:101\.35\工作时间:1\.00\星期:2\.00|报价量:8\.13\浏览量:21\.67\5人报价:3,844秒\10人报价:4,737秒\关注量:368\.09|主要类别,工作时间标准单,响应较快| +|2|标准优质单|102,927\\(19\.70%\)|金额:117\.85\商品数:1\.29\单价:103\.17\工作时间:1\.00\星期:5\.48|报价量:8\.62\浏览量:21\.99\5人报价:4,465秒\10人报价:4,798秒\关注量:371\.35|周末优质单,商品数适中,关注度高| +|3|标准订单|60,843\\(11\.64%\)|金额:116\.50\商品数:1\.24\单价:103\.75\工作时间:0\.00\星期:2\.97|报价量:8\.68\浏览量:24\.09\5人报价:5,015秒\10人报价:6,459秒\关注量:193\.81|非工作时间标准单,报价时长较长| +|4|异常大单|8\\(0\.00%\)
|金额:17,673\.25\商品数:198\.75\单价:88\.79\工作时间:1\.00\星期:1\.25|报价量:5\.00\浏览量:11\.63\5人报价:427秒\10人报价:1,088秒\关注量:1,014\.88|极端异常订单,金额和商品数极高| +|5|高价小众单|22,002\\(4\.21%\)|金额:374\.94\商品数:1\.37\单价:323\.01\工作时间:0\.90\星期:2\.78|报价量:6\.94\浏览量:23\.74\5人报价:6,636秒\10人报价:5,512秒\关注量:439\.22|金额较高,单价极高,竞争激烈| +|6|高价值订单|412\\(0\.08%\)|金额:695\.58\商品数:69\.17\单价:11\.27\工作时间:0\.82\星期:2\.51|报价量:10\.90\浏览量:24\.11\5人报价:1,284秒\10人报价:2,424秒\关注量:246\.03|高价值大单,响应极快,关注度高| +|7|标准订单|28,978\\(5\.55%\)|金额:116\.04\商品数:1\.29\单价:102\.81\工作时间:0\.00\星期:2\.97|报价量:8\.68\浏览量:24\.09\5人报价:5,015秒\10人报价:6,459秒\关注量:193\.81|非工作时间标准单,各项指标中等| +|8|标准大单|43,458\\(8\.32%\)|金额:121\.00\商品数:1\.41\单价:102\.28\工作时间:0\.80\星期:2\.97|报价量:8\.47\浏览量:23\.18\5人报价:4,047秒\10人报价:5,104秒\关注量:33\.20|标准大单,金额略高,关注度较低| + + + +### 3\.2\.2 版本二(全自动\-动态特征) + + + +基础特征(10个): + +order\_total\_amount \- 订单总金额 + +order\_goods\_cnt \- 商品数量 + +order\_unit\_ price \- 单价 + +company\_type \- 公司类型 + +user\_type \- 用户类型 + +order\_appoint\_type\_name \- 订单指派类型 + +submit\_hour \- 提交小时 + +submit\_weekday \- 提交星期 + +submit\_is\_weekend \- 是否周末 + +submit\_is\_business\_hour \- 是否工作时间 + + + +**可用的动态特征(需要数据中存在,这里也用来训练了):** + +offer\_mst\_cnt \- 报价师傅数量 + +view\_mst\_cnt \- 浏览师傅数量 + +fifth\_offer\_duration\_second \- 第5人报价时长(秒) + +tenth\_offer\_duration\_second \- 第10人报价时长(秒) + +attention\_cnt \- 关注数量 + +onsite\_to\_finish\_hour \- 现场到完成时长(小时) + +serve\_efficiency \- 服务效率 + + + +每个动态特征生成的3个类别先验特征: + +以offer\_mst\_cnt为例: + +offer\_mst\_cnt\_category\_mean \- 该类目平均报价师傅数 + +offer\_mst\_cnt\_category\_median \- 该类目报价师傅数中位数 + +offer\_mst\_cnt\_category\_std \- 该类目报价师傅数标准差 + + + +![订单聚类可视化图\.png](图片和附件/订单聚类可视化图%201.png) + + + +|聚类编号|业务标签|订单数\(占比\)|静态特征|动态特征|业务解读| +|---|---|---|---|---|---| +|0|主流订单|221,922\\(42\.47%\)|金额:129\.51\商品数:1\.35\该类目平均报价8\.3人,平均关注66人|报价量:8\.33\浏览量:26\.09\5人报价:4,760秒\10人报价:5,618秒\关注量:177\.95|主要类别,该类目历史报价活跃,关注度中等| +|1|标准优质单|81,886\\(15\.67%\)|金额:129\.91\商品数:1\.46\该类目平均报价8\.5人,平均关注27人|报价量:8\.53\浏览量:21\.03\5人报价:3,763秒\10人报价:4,896秒\关注量:422\.48|第二大类别,该类目历史报价活跃,关注度较低| +|2|高价值订单|86,033\\(16\.46%\)|金额:127\.54\商品数:1\.27\该类目平均报价10\.8人,平均关注66人|报价量:10\.76\浏览量:24\.63\5人报价:1,433秒\10人报价:2,117秒\关注量:318\.73|第三大类别,该类目历史报价最活跃,响应极快
| +|3|标准订单|7,892\\(1\.51%\)|金额:102\.43\商品数:1\.08\该类目平均报价8\.7人,平均关注221人|报价量:8\.66\浏览量:21\.40\5人报价:3,524秒\10人报价:4,741秒\关注量:425\.75|金额最低,该类目历史关注度最高,市场热度极强| +|4|高价小众单|56,029\\(10\.72%\)|金额:117\.36\商品数:1\.23\该类目平均报价7\.8人,平均关注47人|报价量:7\.80\浏览量:28\.72\5人报价:6,986秒\10人报价:7,458秒\关注量:145\.17|金额较高,该类目历史报价较少,竞争激烈| +|5|高价标准单|5,805\\(1\.11%\)|金额:221\.52\商品数:1\.41\该类目平均报价7\.6人,平均关注1318人|报价量:7\.58\浏览量:22\.96\5人报价:5,426秒\10人报价:5,578秒\关注量:250\.19|金额较高,该类目历史关注度极高,市场热点| +|6|标准大单|23,161\\(4\.43%\)|金额:147\.98\商品数:1\.63\该类目平均报价9\.4人,平均关注14人|报价量:9\.41\浏览量:23\.27\5人报价:3,309秒\10人报价:5,060秒\关注量:811\.95|商品数较高,该类目历史报价活跃,关注度较低| +|7|异常大单|233\\(0\.04%\)|金额:1,498\.73\商品数:94\.91\该类目平均报价7\.6人,平均关注33人|报价量:7\.63\浏览量:20\.32\5人报价:3,591秒\10人报价:4,165秒\关注量:170\.63|金额和商品数极端高,该类目历史表现一般| +|8|标准订单|39,583\\(7\.58%\)|金额:112\.02\商品数:1\.28\该类目平均报价8\.7人,平均关注43人|报价量:8\.66\浏览量:21\.40\5人报价:3,524秒\10人报价:4,741秒\关注量:425\.75|标准订单,该类目历史表现中等| + + + +### 3\.2\.3 版本三(半自动\-动态特征) + +**升级点:在原来的基础上叠加了运营的调研经验,将调研经验得出白黑名单、加减分项来继续升级模板。** + +```SQL +加分项 +business_full_name字段为'北欧表情(深圳)家具有限公司'、'西昊家具(深圳)有限公司'、'佛山林氏木业家具有限公司'、'顾家家居股份有限公司' +address字段包含'佛山'、'东莞'、'河北'、'浙江' +goods_level_2_name字段包含'办公家具'、'屏风类'、'户外'、'柜类' +order_serve_type_name字段包含'送货到家并安装'、'维修' +order_label字段包含'加急单' +order_goods_cnt大于等于10单 + +减分项 +address字段包含'徐州' + + + +# --- 业务规则配置 --- +BUSINESS_RULES = { + # 加分项配置 + 'bonus_rules': { + # 优质企业加分 + 'premium_companies': [ + '北欧表情(深圳)家具有限公司', + '西昊家具(深圳)有限公司', + '佛山林氏木业家具有限公司', + '顾家家居股份有限公司' + ], + # 优质地区加分 + 'premium_regions': ['佛山', '东莞', '河北', '浙江'], + # 优质商品类别加分 + 'premium_categories': ['办公家具', '屏风类', '户外', '柜类'], + # 优质服务类型加分 + 'premium_services': ['送货到家并安装', '维修'], + # 加急订单加分 + 'urgent_orders': ['加急单'], + # 大订单加分(商品数量>=10) + 'large_orders_threshold': 10 + }, + # 减分项配置 + 'penalty_rules': { + # 问题地区减分 + 'problem_regions': ['徐州'] + }, + # 规则权重配置 + 'rule_weights': { + 'premium_company_bonus': 2.0, # 优质企业加分权重 + 'premium_region_bonus': 1.5, # 优质地区加分权重 + 'premium_category_bonus': 1.0, # 优质商品类别加分权重 + 'premium_service_bonus': 1.0, # 优质服务类型加分权重 + 'urgent_order_bonus': 1.5, # 加急订单加分权重 + 'large_order_bonus': 1.0, # 大订单加分权重 + 'problem_region_penalty': -1.5 # 问题地区减分权重 + } +} +``` + + + +![订单聚类可视化图\.png](图片和附件/订单聚类可视化图.png) + +|**聚类编号**|**业务标签**|**订单数\(占比\)**|**静态特征**|**动态特征**|**业务解读**| +|---|---|---|---|---|---| +|0|标准主流单|44,567\\(8\.53%\)|金额:123\.61\商品数:1\.26\单价:98\.10\工作时间:1\.00\星期:2\.00\业务规则得分:0\.79|报价量:9\.04\浏览量:23\.31\5人报价:5,619秒\10人报价:7,594秒\关注量:386\.92|标准主流订单,报价量最高,响应较慢但关注度高,优质企业集中| +|1|优质大单|112,167\\(21\.47%\)|金额:134\.79\商品数:1\.52\单价:88\.68\工作时间:1\.00\星期:2\.00\业务规则得分:0\.94|报价量:8\.88\浏览量:22\.14\5人报价:3,283秒\10人报价:4,604秒\关注量:349\.94|**优质企业集中**,商品数最高,响应最快,**大订单主要分布**,**业务规则得分高**| +|2
|高价值订单
|127,887\\(24\.47%\)
|金额:125\.42\商品数:1\.37\单价:91\.55\工作时间:1\.00\星期:2\.00\业务规则得分:0\.63|报价量:8\.03\浏览量:25\.00\5人报价:5,049秒\10人报价:5,631秒\关注量:210\.66|最大类别,关注度极高,优质企业集中,响应中等,佛山浙江地区集中| +|3|**高价精品单**|5,821\\(1\.11%\)
|金额:250\.68\商品数:1\.77\单价:141\.63\工作时间:1\.00\星期:2\.00\业务规则得分:1\.35|报价量:9\.41\浏览量:23\.26\5人报价:3,302秒\10人报价:5,053秒\关注量:812\.70|金额最高,关注度最高,精品订单,业务规则得分最高,响应较快| +|4|高价小众单|23,246\\(4\.45%\)|金额:149\.19\商品数:1\.74\单价:85\.74\工作时间:1\.00\星期:2\.00\业务规则得分:0\.94|报价量:7\.80\浏览量:28\.71\5人报价:6,996秒\10人报价:7,444秒\关注量:145\.31|**佛山地区集中**,商品数较高,关注度极高,响应最慢,**业务规则得分高**| +|5|标准订单|46,434\\(8\.89%\)|金额:107\.81\商品数:1\.27\单价:84\.89\工作时间:1\.00\星期:2\.00\业务规则得分:0\.11|报价量:8\.24\浏览量:20\.35\5人报价:3,299秒\10人报价:4,078秒\关注量:402\.86|金额最低,基础订单,响应较快,业务规则得分最低,普通企业为主| +|6|标准大单|74,377\\(14\.23%\)|金额:128\.93\商品数:1\.30\单价:99\.18\工作时间:1\.00\星期:2\.00\业务规则得分:0\.84|报价量:8\.45\浏览量:20\.71\5人报价:3,499秒\10人报价:4,482秒\关注量:423\.76|标准大单,响应较快,关注度较高,业务规则得分较高,优质企业集中| +|7|异常大单|8,112\\(1\.55%\)|金额:103\.58\商品数:1\.11\单价:93\.31\工作时间:1\.00\星期:2\.00\业务规则得分:0\.10|报价量:4\.74\浏览量:15\.47\5人报价:3,376秒\10人报价:3,045秒\关注量:457\.67|商品数最少,特殊订单,报价量最低,响应最快,业务规则得分最低| +|8|标准订单|79,933\\(15\.30%\)|金额:127\.64\商品数:1\.34\单价:95\.25\工作时间:1\.00\星期:2\.00\业务规则得分:0\.72|报价量:8\.24\浏览量:20\.35\5人报价:3,299秒\10人报价:4,078秒\关注量:402\.86|标准订单,各项指标中等,业务规则得分中等,浙江地区集中| + +|**质量等级**|**订单类别**|**订单数\(占比\)**|**主要特征**|**业务建议**| +|---|---|---|---|---| +|好单
|高价精品单\(3\)\优质大单\(1\)\高价小众单\(4\)|141,235\\(27\.03%\)|金额高、响应快、关注度高\业务规则得分高|重点维护,优先分配优质供应商| +|中单|高价值订单\(2\)\标准大单\(6\)\标准主流单\(0\)\标准订单\(8\)|326,764\\(62\.55%\)|各项指标均衡\业务规则得分中等|正常处理,保持服务质量| +|差单|标准订单\(5\)\异常大单\(7\)|54,546\\(10\.42%\)|金额低、竞争少\业务规则得分低|需要关注,可能存在风险或价值较低| + + + +**业务规则效果验证** + +**高分订单特征**: + +- 高价精品单\(1\.35分\):金额最高,关注度最高 + +- 优质大单\(0\.94分\):商品数最高,响应最快 + +- 高价小众单\(0\.94分\):地区优势,关注度极高 + +**低分订单特征**: + +- 标准订单\(0\.11分\):基础订单,普通企业 + +- 异常大单\(0\.10分\):特殊订单,竞争较少 + +**地区分布特征**: + +- 佛山:主要分布在高价值订单\(5,474单\)和高价小众单\(2,416单\) + +- 浙江:主要分布在高价值订单\(3,372单\)和标准订单\(1,838单\) + +- 河北:主要分布在优质大单\(1,102单\)和高价值订单\(1,033单\) + + + + + +### 3\.2\.4 版本四(有监督\-师傅打标) + +**运营人工打标,做裁判/数据标签输入** +**增加产品功能\-\-\-让师傅打标** + + + + + + + +### 3\.2\.5 版本五(规则70%\+聚类30%)\*目前最新 + + + +**好单专项\-混合评分模型进度同步** + +**1、v1\.0版已敲定** +**好单挑选 = 规则(70%) \+ 自动分析聚类(30%)** +1\.1 平衡性较好:规则确保稳定性(不算命、不看天吃饭),聚类提供洞察力 +1\.2 可解释性强:每个预测都能追溯到具体特征贡献 +1\.3 反直觉发现:自动分析聚类30%权重能发现"高价但差单"、"低价但好单"等业务盲点 +1\.4 权重可调:70:30配比经过验证,且支持后续调优 + +**2、自检报告已出** +根据北上广深202501\-202505训练数据52万单,当前版本自检(机器自我检查)好单预测率:约60%(再精调一周,到70%\-90%后可生产稳定使用) + +\-\-\-详细的规则公示和过程说明感兴趣后续可查阅文档,还在持续进化中\.\.\. + + + +1、可靠性:好单预测率已由60%提升到70\-90% + +2、目前预测的好单给人直观感受(和大盘所有订单平均水平比)为: + +- 价值维度 + + - 订单金额:154 vs 120 → 高28% + + - 单价水平:124 vs 100 → 高24% + + - 商品数量:1\.7件 vs 1\.4件 → 多21% + +- 响应维度 + + - 查看报价率:47% vs 38% → 高24% + + - 5人报价时长:2900秒 vs 4779秒 → 快39% + + - 10人报价时长:4154秒 vs 6000秒\+ → 快31% + + - 报价师傅数:9\.3人 vs 8\.2人 → 多13% + + - 师傅关注数:241人 vs 150人 → 高61% + +- 效率维度 + + - 完工时长:5\.3小时 vs 6\.1小时 → 快13% + +- 风险维度 + + - 商家售后率:0\.7% vs 1\.0% → 低30% + + - 被拉黑风险:与平均持平(商家质量稳定) + +- 好单的核心优势: + + 1. **多**:价值高28% \- 订单含金量更高 + + 2. **快**:响应高30% \- 师傅抢单更积极 + + 3. **好**:关注高61% \- 市场热度明显更高 + + 4. **稳**:风险低30% \- 商家售后问题更少 + + + + + +**特征权重** + + + +## 3\.3 筛选出100个list + +[版本二(全自动\-静态特征)数据结果抽查](https://jiqmwlmd0v.feishu.cn/wiki/ChJpwdp4ZitSJmk1SK5c5ooVnOb?from=from_copylink) + +[版本三(半自动\-动态特征)数据结构抽查](https://jiqmwlmd0v.feishu.cn/wiki/QbkiwrSZui4sg3kdOlScyMQKnsf?from=from_copylink) + + + +# **5\. 价值与应用** + +**先用在合作经营\-好单专区** + + + + + + + +# 6\.接口调用 + +[订单分类推理服务接口文档](https://jiqmwlmd0v.feishu.cn/wiki/IKG4wPJJgi67qkklyvhceuB5nDf?from=from_copylink) + + + + + + +