diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_dataset.txt b/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_dataset.txt
new file mode 100644
index 0000000..eaa1250
--- /dev/null
+++ b/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_dataset.txt
@@ -0,0 +1,215 @@
+DROP TABLE IF EXISTS wanshifu_dw.ai_training_order_bidding_snapshot;
+CREATE TABLE IF NOT EXISTS wanshifu_dw.ai_training_order_bidding_snapshot (
+ global_order_trace_id STRING COMMENT '订单全局id',
+ order_no STRING COMMENT '订单编号',
+ order_submit_time STRING COMMENT '订单创建时间',
+ order_cancel_time STRING COMMENT '订单取消时间',
+ goods_level_1_name STRING COMMENT '订单商品一级类目名称',
+ order_goods_cnt BIGINT COMMENT '订单商品数量',
+ order_serve_type_name STRING COMMENT '订单服务类型名称',
+ order_business_line_name STRING COMMENT '订单业务线名称',
+ order_business_line_type STRING COMMENT '订单业务线类型',
+ order_appoint_type_name STRING COMMENT '指派类型',
+ order_etp_flag STRING COMMENT '下单总包标识:0不是,1是',
+ user_id STRING COMMENT '用户id',
+ user_name STRING COMMENT '用户名称',
+ address STRING COMMENT '经营详情地址',
+ business_full_name STRING COMMENT '企业全称',
+ company_type STRING COMMENT '企业类型(e_commerce:电商、offline_store:线下门店、factory:工厂、logistics:物流)',
+ user_type STRING COMMENT '用户类型(1:普通 2:企业 3:新版企业用户 4:小程序用户 5:个人)',
+ order_onsite_sign_time STRING COMMENT '上门时间',
+ mst_serve_complete_last_time STRING COMMENT '完工时间',
+ goods_level_2_name STRING COMMENT '订单商品二级类目名称',
+ goods_level_3_name STRING COMMENT '订单商品三级类目名称',
+ order_total_amount DOUBLE COMMENT '订单总金额',
+ order_unit_price DOUBLE COMMENT '订单单价',
+ is_urgent_flag STRING COMMENT '是否加急',
+ city_name STRING COMMENT '城市名称',
+ offer_mst_cnt BIGINT COMMENT '报价量(人次)',
+ view_mst_cnt BIGINT COMMENT '查看量(人次)',
+ fifth_offer_duration_second BIGINT COMMENT '满5人报价时长(第5人报价时间 - 下单时间)',
+ tenth_offer_duration_second BIGINT COMMENT '满10人报价时长(第10人报价时间 - 下单时间)',
+ attention_cnt BIGINT COMMENT '商家师傅关注数',
+ merchant_total_orders BIGINT COMMENT '商家下的总订单量',
+ merchant_total_aftersales BIGINT COMMENT '商家产生的总售后单量',
+ ignore_cnt BIGINT COMMENT '商家被拉黑次数',
+ buyer_note STRING COMMENT '订单备注'
+)
+STORED AS ALIORC
+TBLPROPERTIES (
+ 'comment' = 'AI训练用报价招标订单快照表(非分区表)'
+);
+
+
+INSERT OVERWRITE TABLE wanshifu_dw.ai_training_order_bidding_snapshot
+SELECT t1.global_order_trace_id -- 订单全局id
+,t1.order_no -- 订单编号
+,t1.order_submit_time -- 订单创建时间
+,t1.order_cancel_time -- 订单取消时间
+,t3.goods_level_1_name -- 订单商品一级类目名称
+,CAST(t1.order_goods_cnt AS BIGINT) AS order_goods_cnt -- 订单商品数量
+,t1.order_serve_type_name -- 订单服务类型名称
+,t1.order_business_line_name -- 订单业务线名称
+,t1.order_business_line_type -- 订单业务线类型
+,t1.order_appoint_type_name --指派类型
+,t1.order_etp_flag -- 下单总包标识:0不是,1是
+,t1.user_id -- 用户id
+,t2.user_name -- 用户名称
+,t2.address -- 经营详情地址
+,t2.business_full_name -- 企业全称
+,t2.company_type -- 企业类型(e_commerce:电商、offline_store:线下门店、factory:工厂、logistics:物流)
+,t2.user_type -- 用户类型(1:普通 2:企业 3:新版企业用户 4:小程序用户5:个人)
+,order_onsite_sign_time --上门时间
+,mst_serve_complete_last_time --完工时间
+,t3.goods_level_2_name -- 订单商品二级类目名称
+,t3.goods_level_3_name -- 订单商品三级类目名称
+--,init_order_amt --初始主订单金额
+--,init_total_amt --初始订单总金额(主订单+子订单)
+--,settle_amt --订单结算金额(服务费用)
+,t9.init_serve_fee as order_total_amount --订单总金额
+,(t9.init_serve_fee/t1.order_goods_cnt) as order_unit_price --订单单价
+--,t4.order_label --订单种类(加急单、夜间单等等)
+,CASE WHEN t4.order_label LIKE '%加急%' THEN '是'
+ELSE '否'
+END AS is_urgent_flag --是否加急
+,t10.city_name --城市名称
+,NVL(t5.offer_mst_cnt,0) AS offer_mst_cnt --报价量(人次)
+,NVL(t6.view_mst_cnt,0) AS view_mst_cnt --查看量(人次)
+,NVL(t7.fifth_offer_duration_second,0) AS fifth_offer_duration_second --满5人报价时长(第5人报价时间 - 下单时间)
+,NVL(t7.tenth_offer_duration_second,0) AS tenth_offer_duration_second --满10人报价时长(第10人报价时间 - 下单时间)
+,NVL(t8.attention_cnt,0) AS attention_cnt --商家师傅关注数
+,merchant_total_orders --- 商家下的总订单量
+,merchant_total_aftersales --- 商家产生的总售后单量
+,ignore_cnt --商家被拉黑次数
+,t10.buyer_note -- 订单备注
+FROM wanshifu_dw.dws_order_d t1
+LEFT JOIN wanshifu_dw.dim_usr_info_v t2
+ON t1.user_id = t2.user_id
+
+JOIN (
+SELECT t1.global_order_trace_id
+,t1.goods_level_1_name
+,t1.goods_level_2_name
+,t1.goods_level_3_name
+FROM wanshifu_dw.dwm_order_goods_info_d_v t1
+JOIN (
+SELECT global_order_trace_id
+FROM wanshifu_dw.dwm_order_goods_info_d_v
+WHERE stat_date IS NOT NULL
+GROUP BY global_order_trace_id
+HAVING COUNT(*) = 1
+) t2
+ON t1.global_order_trace_id = t2.global_order_trace_id
+WHERE stat_date IS NOT NULL
+) t3 --去掉多类目商品
+ON t1.global_order_trace_id = t3.global_order_trace_id
+
+LEFT JOIN (
+SELECT global_order_trace_id
+,CONCAT_WS(',',COLLECT_SET(business_type_name)) AS order_label
+FROM wanshifu_dw.dwd_bas_accounting_trade_settlement_flow_fee_detail_d_v
+WHERE etl_date >= '20250101'
+AND business_type_code IN ('emergency_order_settlement','fee_subsidy_settlement','fittings_order_settlement','night_order_settlement','order_settlement','price_diff_subsidy_settlement','rate_award_settlement','service_satisfy_award_settlement','toc_order_settlement')
+GROUP BY global_order_trace_id
+) t4
+ON t1.global_order_trace_id = t4.global_order_trace_id
+LEFT JOIN (
+SELECT global_order_trace_id
+,COUNT(DISTINCT master_id) AS offer_mst_cnt
+FROM wanshifu_dw.dwd_mst_order_offer_price_d_v
+WHERE etl_date >= '20250101'
+GROUP BY global_order_trace_id
+) t5
+ON t1.global_order_trace_id = t5.global_order_trace_id
+LEFT JOIN (
+SELECT global_order_trace_id
+,COUNT(DISTINCT IF(first_view_time IS NOT NULL,master_id,NULL)) AS view_mst_cnt
+FROM wanshifu_dw.dwm_mst_order_push_d
+WHERE etl_date >= '20250101'
+GROUP BY global_order_trace_id
+) t6
+ON t1.global_order_trace_id = t6.global_order_trace_id
+LEFT JOIN (
+SELECT global_order_trace_id
+,SUM(IF(rn = 5,DATEDIFF(offer_time,order_submit_time,'ss'),0)) AS fifth_offer_duration_second
+,SUM(IF(rn = 10,DATEDIFF(offer_time,order_submit_time,'ss'),0)) AS tenth_offer_duration_second
+FROM (
+SELECT global_order_trace_id
+,order_submit_time
+,offer_time
+,ROW_NUMBER() OVER (PARTITION BY global_order_trace_id ORDER BY offer_time ASC ) rn
+FROM wanshifu_dw.dwd_mst_order_offer_price_d_v
+WHERE etl_date >= '20250101'
+)
+WHERE rn IN (5,10)
+GROUP BY global_order_trace_id
+) t7
+ON t1.global_order_trace_id = t7.global_order_trace_id
+
+LEFT JOIN (
+SELECT account_id
+,COUNT(master_id) AS attention_cnt --商家师傅关注数
+FROM wanshifu_dw.ods_mst_info_mst_interaction_d_v
+WHERE etl_date = '20250618' --此日期不可修改
+AND attention_status = '1'
+GROUP BY account_id
+) t8
+ON t1.user_id = t8.account_id
+
+LEFT JOIN (
+SELECT account_id
+,COUNT(master_id) AS ignore_cnt --商家被拉黑次数
+FROM wanshifu_dw.ods_mst_info_mst_interaction_d_v
+WHERE etl_date = '20250618' --此日期不可修改
+AND ignore_his_order_status = '1'
+GROUP BY account_id
+) t88
+ON t1.user_id = t88.account_id
+
+LEFT JOIN wanshifu_dw.dwd_usr_order_trade_info t9
+on t9.global_order_trace_id = t1.global_order_trace_id
+LEFT JOIN wanshifu_dw.dim_order_info_v t10
+on t10.global_order_trace_id = t1.global_order_trace_id
+
+left JOIN
+(
+SELECT user_id,COUNT(distinct global_order_trace_id) as merchant_total_orders
+FROM wanshifu_dw.dws_order_d t1
+WHERE SUBSTR(t1.stat_date,1,6) BETWEEN '202501' AND '202506'
+AND t1.order_cancel_time IS NULL
+and t1.mst_serve_complete_last_time is not null
+GROUP BY user_id
+) a
+on t1.user_id = a.user_id
+left JOIN
+(
+SELECT tt2.user_id,count(distinct tt1.global_order_trace_id) as merchant_total_aftersales
+FROM (
+ SELECT global_order_trace_id
+ FROM wanshifu_dw.dwd_iop_work_order_work_order_infos_v
+ WHERE work_order_type IN ('complaint','secondline') --投诉,二线
+ AND TO_CHAR(work_order_create_time,'yyyymmdd') >= '20250101'
+ and TO_CHAR(work_order_create_time,'yyyymmdd') <= '20250630'
+ GROUP BY global_order_trace_id
+ UNION ALL
+ SELECT global_order_trace_id
+ FROM wanshifu_dw.dwm_order_arbitration_info --仲裁
+ WHERE TO_CHAR(create_time,'yyyymmdd') >= '20250101'
+ AND TO_CHAR(create_time,'yyyymmdd') <= '20250630'
+ GROUP BY global_order_trace_id
+ ) tt1
+LEFT JOIN wanshifu_dw.dim_order_info_v tt2
+on tt1.global_order_trace_id = tt2.global_order_trace_id
+GROUP BY tt2.user_id
+) b
+on t1.user_id = b.user_id
+
+WHERE SUBSTR(t1.stat_date,1,6) BETWEEN '202501' AND '202506'
+AND t1.order_goods_lv1_name IN ('家具')
+AND t1.order_business_line_type = 'tob'
+AND t1.order_etp_flag = '0'
+AND t1.order_cancel_time IS NULL
+AND t1.order_appoint_type_name = '报价招标'
+AND t1.order_2nd_onsite_time IS NULL --去掉二次上门
+and t10.city_name IN ('西安市','南京市','天津市','成都市','珠海市','佛山市','合肥市','青岛市','郑州市','长沙市','武汉市','宁波市','苏州市','东莞市','重庆市')
+;
\ No newline at end of file
diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_training.py b/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_training.py
new file mode 100644
index 0000000..3d2c4b7
--- /dev/null
+++ b/工作记录/曹强-订单分层算法落地方案v0.6/bidding_order_class_model_training.py
@@ -0,0 +1,3691 @@
+import os
+import sys
+import warnings
+import argparse
+import logging
+import json
+from datetime import datetime
+import joblib
+import matplotlib.pyplot as plt
+import numpy as np
+import pandas as pd
+import seaborn as sns
+from sklearn.cluster import KMeans
+from sklearn.decomposition import PCA
+from sklearn.metrics import silhouette_score
+from sklearn.preprocessing import StandardScaler
+
+try:
+ import yaml # 配置化支持(可选)
+except Exception: # yaml 非强依赖
+ yaml = None
+
+# -------- 运行参数与日志初始化 --------
+
+
+def parse_args():
+ parser = argparse.ArgumentParser(description="Order classification pipeline")
+ # 仅保留 predict 模式(纯在线重建)
+ parser.add_argument(
+ "--config",
+ type=str,
+ default=None,
+ help="YAML 配置文件路径(可选)",
+ )
+ parser.add_argument(
+ "--log-level",
+ type=str,
+ default="INFO",
+ help="日志级别: DEBUG/INFO/WARNING/ERROR",
+ )
+ parser.add_argument(
+ "--log-file",
+ type=str,
+ default=None,
+ help="日志文件输出路径(默认写入模型目录 pipeline.log)",
+ )
+ return parser.parse_args(args=[a for a in sys.argv[1:] if a.strip()])
+
+
+def init_logging(log_file: str = None, level: str = "INFO"):
+ log_level = getattr(logging, level.upper(), logging.INFO)
+ logging.captureWarnings(True)
+ handlers = [logging.StreamHandler(sys.stdout)]
+ if log_file:
+ os.makedirs(os.path.dirname(log_file), exist_ok=True)
+ handlers.append(logging.FileHandler(log_file, encoding="utf-8"))
+ logging.basicConfig(
+ level=log_level,
+ format="%(asctime)s | %(levelname)s | %(message)s",
+ handlers=handlers,
+ force=True,
+ )
+ logging.info(f"Logging initialized. level={level}, file={log_file}")
+
+warnings.filterwarnings("ignore")
+# 设置中文字体
+plt.rcParams["font.sans-serif"] = ["SimHei", "Arial Unicode MS", "DejaVu Sans"]
+plt.rcParams["axes.unicode_minus"] = False # 正常显示负号
+# ===== 统一配置管理类 =====
+
+
+class ModelConfig:
+ """
+ 统一配置管理类 - 集中管理所有模型配置
+ """
+ def __init__(self):
+ # 混合评分配置
+ self.hybrid_config = {
+ "rule_weight": 0.7, # 规则评分权重70%
+ "cluster_weight": 0.3, # 聚类评分权重30%
+ "good_quantile": 0.75, # 好单分位数:前15% (85分位数)
+ "medium_quantile": 0.50, # 中单分位数:前40% (60分位数)
+ "min_quality_threshold": 50.0, # 最低质量门槛 - 修复:降低门槛
+ }
+ # 业务规则配置
+ self.business_rules = {
+ "urgent_order_flag_value": 1, # 加急订单加分的判定值
+ "large_orders_threshold": 10, # 大订单加分阈值
+ "bonus_rules": {
+ # 优质企业加分
+ "premium_companies": [
+ "北欧表情(深圳)家具有限公司",
+ "西昊家具(深圳)有限公司",
+ "佛山林氏木业家具有限公司",
+ "顾家家居股份有限公司",
+ ],
+ # 优质地区加分
+ "premium_regions": ["佛山", "东莞", "河北", "浙江"],
+ # 优质商品类别加分
+ "premium_categories": ["办公","老板", "屏", "户外", "柜"],
+ # 优质服务类型加分
+ "premium_services": ["送货到家并安装", "维修"],
+ },
+ # 减分项配置
+ "penalty_rules": {
+ # 问题地区减分
+ "problem_regions": ["徐州"]
+ },
+ # 规则权重配置(缩小到20%以内)
+ "rule_weights": {
+ "premium_company_bonus": 0.8, # 优质企业加分权重
+ "premium_region_bonus": 0.6, # 优质地区加分权重
+ "premium_category_bonus": 0.6, # 优质商品类别加分权重
+ "premium_service_bonus": 0.6, # 优质服务类型加分权重
+ "urgent_order_bonus": 0.6, # 加急订单加分权重
+ "large_order_bonus": 0.6, # 大订单加分权重
+ "problem_region_penalty": -0.4, # 问题地区减分权重
+ },
+ }
+ # 路径配置
+ self.paths = {
+ "data_file": "/Users/tom/Documents/data.csv",
+ "model_dir": "order_cluster_model",
+ "output_file": "/Users/tom/Documents/data_check_with_predictions.xlsx",
+ }
+ # 模型参数配置
+ self.model_params = {
+ "n_clusters": 6,
+ "random_state": 42,
+ "min_samples": 50,
+ "value_threshold": 10.0,
+ }
+ # 特征配置
+ self.features = {
+ "static_base": [
+ "order_goods_cnt",
+ "order_total_amount",
+ "order_unit_price",
+ "buyer_note_100",
+ "submit_hour",
+ "submit_weekday",
+ "business_rule_score",
+ "has_price_info",
+ ],
+ "dynamic": [
+ "offer_mst_cnt",
+ "view_mst_cnt",
+ "fifth_offer_duration_second",
+ "tenth_offer_duration_second",
+ "attention_cnt",
+ "onsite_to_finish_hour",
+ "serve_efficiency",
+ ],
+ }
+
+ def get_hybrid_config(self):
+ """获取混合评分配置"""
+ return self.hybrid_config
+
+ def get_business_rules(self):
+ """获取业务规则配置"""
+ return self.business_rules
+
+ def get_paths(self):
+ """获取路径配置"""
+ return self.paths
+
+ def get_model_params(self):
+ """获取模型参数配置"""
+ return self.model_params
+
+ def get_features(self):
+ """获取特征配置"""
+ return self.features
+
+ def update_hybrid_config(self, **kwargs):
+ """更新混合评分配置"""
+ self.hybrid_config.update(kwargs)
+
+ def update_business_rules(self, **kwargs):
+ """更新业务规则配置"""
+ self.business_rules.update(kwargs)
+
+ def save_configs(self, model_dir):
+ """保存所有配置到文件"""
+ os.makedirs(model_dir, exist_ok=True)
+ # 保存混合评分配置
+ joblib.dump(self.hybrid_config, os.path.join(model_dir, "hybrid_config.pkl"))
+ # 保存业务规则配置
+ joblib.dump(self.business_rules, os.path.join(model_dir, "business_rules.pkl"))
+ print("✅ 配置已保存到模型目录")
+
+
+# 创建全局配置实例(支持从YAML覆盖)
+args = parse_args()
+config = ModelConfig()
+if args.config and yaml is not None and os.path.exists(args.config):
+ try:
+ with open(args.config, "r", encoding="utf-8") as fh:
+ y = yaml.safe_load(fh) or {}
+ if isinstance(y, dict):
+ if "hybrid_config" in y:
+ config.update_hybrid_config(**y["hybrid_config"])
+ if "business_rules" in y:
+ config.update_business_rules(**y["business_rules"])
+ if "paths" in y and isinstance(y["paths"], dict):
+ config.paths.update(y["paths"]) # 路径配置覆盖
+ if "model_params" in y and isinstance(y["model_params"], dict):
+ config.model_params.update(y["model_params"]) # 模型参数覆盖
+ except Exception as e:
+ print(f"⚠️ 读取配置失败: {e},继续使用内置默认配置")
+
+# 初始化日志
+log_file_default = os.path.join(config.get_paths()["model_dir"], "pipeline.log")
+init_logging(args.log_file or log_file_default, args.log_level)
+logging.info("Start pipeline in predict-only mode (pure online reconstruction)")
+# 统一使用配置实例,避免重复定义
+HYBRID_CONFIG = config.get_hybrid_config()
+BUSINESS_RULES = config.get_business_rules()
+MODEL_DIR = config.get_paths()["model_dir"]
+DYNAMIC_FEATURES = config.get_features()["dynamic"]
+# --- 1. 数据准备与特征工程 ---
+print("\n[Part 1] 数据准备与特征工程...")
+# 确保模型目录存在
+os.makedirs(MODEL_DIR, exist_ok=True)
+# 初始化聚类基础分(后面会动态计算)
+cluster_base_scores = {}
+print(
+ "--- 方案A:基于静态特征的订单分层模型(升级版:含群体统计特征 + 业务规则特征)---"
+)
+# --- 2. 模型训练与聚类 ---
+print("\n[Part 2] 模型训练与聚类...")
+# 读取数据
+try:
+ df = pd.read_csv("/Users/tom/Documents/data.csv")
+except FileNotFoundError:
+ print("错误:数据文件'/Users/tom/Documents/data.csv'未找到。请检查路径。")
+ exit()
+# 检查原始数据 order_no 缺失情况
+if "order_no" in df.columns:
+ print(f"原始数据 order_no 缺失数: {df['order_no'].isna().sum()} 条")
+else:
+ print("原始数据中未找到 order_no 字段!")
+# 筛选有效订单:只保留已完成的订单
+print("正在筛选有效订单...")
+df_valid = df[df["mst_serve_complete_last_time"].notna()].copy()
+if "order_no" in df_valid.columns:
+ print(f"有效订单 order_no 缺失数: {df_valid['order_no'].isna().sum()} 条")
+# 剔除金额为0的订单
+print("\n正在剔除金额为0的订单...")
+zero_amount_count = len(df_valid[df_valid["order_total_amount"] == 0])
+print(f"金额为0的订单数: {zero_amount_count:,} 单")
+df_valid = df_valid[df_valid["order_total_amount"] > 0].copy()
+print(f"剔除后订单数: {len(df_valid):,} 单")
+# 过滤三级类目数量不足的订单
+print("\n正在过滤三级类目数量不足的订单...")
+original_count = len(df_valid)
+print(f"过滤前订单数: {original_count:,} 单")
+# 计算每个三级类目的订单数量
+category_counts = df_valid["goods_level_3_name"].value_counts()
+print(f"三级类目总数: {len(category_counts)} 个")
+# 找出订单数量>=10的三级类目
+valid_categories = category_counts[category_counts >= 10].index
+print(f"订单数量>=10的三级类目: {len(valid_categories)} 个")
+print(f"订单数量<10的三级类目: {len(category_counts) - len(valid_categories)} 个")
+# 过滤数据:只保留订单数量>=10的三级类目
+df_valid = df_valid[df_valid["goods_level_3_name"].isin(valid_categories)].copy()
+filtered_count = len(df_valid)
+removed_count = original_count - filtered_count
+print(f"过滤后订单数: {filtered_count:,} 单")
+print(f"删除订单数: {removed_count:,} 单 ({removed_count/original_count*100:.2f}%)")
+print(f"保留订单比例: {filtered_count/original_count*100:.2f}%")
+# 显示删除的三级类目统计
+if removed_count > 0:
+ removed_categories = category_counts[category_counts < 10]
+ print(f"\n删除的三级类目分布(订单数<10):")
+ print(f" • 1单类目: {len(removed_categories[removed_categories == 1])} 个")
+ print(
+ f" • 2-3单类目: {len(removed_categories[(removed_categories >= 2) & (removed_categories <= 3)])} 个"
+ )
+ print(
+ f" • 4-6单类目: {len(removed_categories[(removed_categories >= 4) & (removed_categories <= 6)])} 个"
+ )
+ print(
+ f" • 7-9单类目: {len(removed_categories[(removed_categories >= 7) & (removed_categories <= 9)])} 个"
+ )
+# 检查订单指派类型的唯一值
+print(f"\n订单指派类型分布:")
+print(df_valid["order_appoint_type_name"].value_counts())
+# 分离业务模式(根据实际数据值)
+if "报价招标" in df_valid["order_appoint_type_name"].values:
+ df_bidding = df_valid[df_valid["order_appoint_type_name"] == "报价招标"].copy()
+ df_fixed = df_valid[df_valid["order_appoint_type_name"] == "一口价"].copy()
+ print(f"报价招标订单数: {len(df_bidding)}")
+ print(f"一口价订单数: {len(df_fixed)}")
+ # 选择当前要建模的业务模式(这里以报价招标为例)
+ if len(df_bidding) > 0:
+ print("\n当前建模业务模式:报价招标订单")
+ df_model_source = df_bidding.copy()
+ else:
+ print("\n报价招标订单数量为0,改为建模一口价订单")
+ df_model_source = df_fixed.copy()
+ df_bidding = df_fixed.copy()
+else:
+ # 如果字段值不是预期的,使用所有有效订单
+ print("\n未找到预期的订单类型,使用所有有效订单进行建模")
+ df_bidding = df_valid.copy()
+ df_model_source = df_bidding.copy()
+# --- 特征工程 ---
+print("\n正在进行特征工程...")
+# 查看数据字段
+print("数据字段列表:")
+print(df_model_source.columns.tolist())
+print(f"数据形状: {df_model_source.shape}")
+# 1. 时间特征工程
+print("正在提取时间特征...")
+df_model_source["order_submit_time"] = pd.to_datetime(
+ df_model_source["order_submit_time"], errors="coerce"
+)
+df_model_source["submit_hour"] = df_model_source["order_submit_time"].dt.hour
+df_model_source["submit_weekday"] = df_model_source["order_submit_time"].dt.weekday
+df_model_source["submit_is_weekend"] = (
+ df_model_source["submit_weekday"].isin([5, 6]).astype(int)
+)
+df_model_source["submit_is_business_hour"] = (
+ (df_model_source["submit_hour"] >= 9) & (df_model_source["submit_hour"] <= 18)
+).astype(int)
+# 2. 商品特征工程
+print("正在提取商品特征...")
+df_model_source["goods_level_1_name"] = df_model_source["goods_level_1_name"].astype(
+ str
+)
+df_model_source["goods_level_2_name"] = df_model_source["goods_level_2_name"].astype(
+ str
+)
+df_model_source["goods_level_3_name"] = df_model_source["goods_level_3_name"].astype(
+ str
+)
+# 3. 业务规则特征工程
+print("正在计算业务规则特征...")
+
+
+def calculate_business_rule_score(row, rules_config):
+ """
+
+ 根据业务规则计算订单的加分减分
+
+ """
+ score = 0.0
+ # 检查优质企业加分
+ if "business_full_name" in row and pd.notna(row["business_full_name"]):
+ if (
+ row["business_full_name"]
+ in rules_config["bonus_rules"]["premium_companies"]
+ ):
+ score += rules_config["rule_weights"]["premium_company_bonus"]
+ # 检查优质地区加分
+ if "address" in row and pd.notna(row["address"]):
+ address_str = str(row["address"]).lower()
+ for region in rules_config["bonus_rules"]["premium_regions"]:
+ if region.lower() in address_str:
+ score += rules_config["rule_weights"]["premium_region_bonus"]
+ break
+ # 检查优质商品类别加分(改为使用三级类目)
+ if "goods_level_3_name" in row and pd.notna(row["goods_level_3_name"]):
+ category_str = str(row["goods_level_3_name"]).lower()
+ for category in rules_config["bonus_rules"]["premium_categories"]:
+ if category.lower() in category_str:
+ score += rules_config["rule_weights"]["premium_category_bonus"]
+ break
+ # 检查优质服务类型加分
+ if "order_serve_type_name" in row and pd.notna(row["order_serve_type_name"]):
+ service_str = str(row["order_serve_type_name"]).lower()
+ for service in rules_config["bonus_rules"]["premium_services"]:
+ if service.lower() in service_str:
+ score += rules_config["rule_weights"]["premium_service_bonus"]
+ break
+ # 检查加急订单加分
+ if (
+ "is_urgent_order" in row
+ and row["is_urgent_order"] == rules_config["urgent_order_flag_value"]
+ ):
+ score += rules_config["rule_weights"]["urgent_order_bonus"]
+ # 检查大订单加分
+ if "order_goods_cnt" in row and pd.notna(row["order_goods_cnt"]):
+ if row["order_goods_cnt"] >= rules_config["large_orders_threshold"]:
+ score += rules_config["rule_weights"]["large_order_bonus"]
+ # 检查问题地区减分
+ if "address" in row and pd.notna(row["address"]):
+ address_str = str(row["address"]).lower()
+ for region in rules_config["penalty_rules"]["problem_regions"]:
+ if region.lower() in address_str:
+ score += rules_config["rule_weights"]["problem_region_penalty"]
+ break
+ return score
+
+
+# 基于 user_id 的单维度商家属性映射覆盖(训练也不直接用行级现值)
+user_attr_fields = [
+ "user_name",
+ "address",
+ "business_full_name",
+ "company_type",
+ "user_type",
+ "attention_cnt",
+ "merchant_total_orders",
+ "merchant_total_aftersales",
+ "ignore_cnt",
+]
+available_user_attr_fields = [f for f in user_attr_fields if f in df_model_source.columns]
+if available_user_attr_fields:
+ try:
+ ua_df = df_model_source.groupby("user_id")[available_user_attr_fields].last()
+ for field in available_user_attr_fields:
+ df_model_source[field] = df_model_source["user_id"].map(ua_df[field])
+ print(f" - 已基于user_id覆盖商家属性字段: {available_user_attr_fields}")
+ except Exception as e:
+ print(f" ⚠️ 商家属性覆盖失败: {e}")
+
+# 计算业务规则得分(使用覆盖后的商家属性)
+print(" - 正在计算业务规则得分...")
+df_model_source["business_rule_score"] = df_model_source.apply(
+ lambda row: calculate_business_rule_score(row, BUSINESS_RULES), axis=1
+)
+# 统计业务规则得分分布
+print(f" - 业务规则得分统计:")
+print(f" 最小值: {df_model_source['business_rule_score'].min():.2f}")
+print(f" 最大值: {df_model_source['business_rule_score'].max():.2f}")
+print(f" 平均值: {df_model_source['business_rule_score'].mean():.2f}")
+print(f" 标准差: {df_model_source['business_rule_score'].std():.2f}")
+# 4. 群体统计特征工程(Category Prior Features)
+print("正在计算群体统计特征(goods_level_3_name & user_id组合优先)...")
+# === 重新设计的特征体系 ===
+# 静态基础特征(订单提交时就有的)——按锚点收紧:价值类现值不入模,仅用先验
+STATIC_BASE_FEATURES = [
+ # 订单基础属性(不含现值金额与单价)
+ "order_goods_cnt",
+ "buyer_note_100",
+ # 时间特征
+ "submit_hour",
+ "submit_weekday",
+ "submit_is_weekend",
+ "submit_is_business_hour",
+ # 业务规则特征
+ "business_rule_score",
+ # 业务模式标识(使用先验金额判断)
+ "has_price_info",
+]
+# 主参考特征(用于计算历史先验)
+MAIN_REFERENCE_FEATURES = [
+ "offer_rate", # 查看报价率
+ "fifth_offer_duration_second", # 满5人报价时长
+ "tenth_offer_duration_second", # 满10人报价时长
+ "onsite_to_finish_hour", # 完工时长
+ "order_total_amount", # 总金额
+ "order_unit_price", # 单价
+]
+# 次参考特征(用于计算历史先验)
+SECONDARY_REFERENCE_FEATURES = [
+ "attention_cnt", # 师傅关注数
+ "merchant_aftersale_rate", # 商家售后率
+ "ignore_cnt", # 商家被拉黑数
+]
+# 所有后验特征
+ALL_REFERENCE_FEATURES = MAIN_REFERENCE_FEATURES + SECONDARY_REFERENCE_FEATURES
+# 生成静态基础特征
+print("正在生成静态基础特征...")
+# 1. buyer_note_100
+if "buyer_note" in df_model_source.columns:
+ df_model_source["buyer_note_100"] = (
+ df_model_source["buyer_note"]
+ .astype(str)
+ .apply(lambda x: 1 if len(x) > 100 else 0)
+ )
+else:
+ df_model_source["buyer_note_100"] = 0
+# 2. 时间特征(如果还没有的话)
+if "submit_hour" not in df_model_source.columns:
+ df_model_source["submit_hour"] = df_model_source["order_submit_time"].dt.hour
+if "submit_weekday" not in df_model_source.columns:
+ df_model_source["submit_weekday"] = df_model_source["order_submit_time"].dt.weekday
+if "submit_is_weekend" not in df_model_source.columns:
+ df_model_source["submit_is_weekend"] = (
+ df_model_source["submit_weekday"].isin([5, 6]).astype(int)
+ )
+if "submit_is_business_hour" not in df_model_source.columns:
+ df_model_source["submit_is_business_hour"] = (
+ (df_model_source["submit_hour"] >= 9) & (df_model_source["submit_hour"] <= 18)
+ ).astype(int)
+# 3. 业务模式标识改为基于先验金额设置(稍后先验生成后再设置)
+df_model_source["has_price_info"] = 0
+# 生成后验参考特征(用于历史先验计算)
+print("正在生成后验参考特征...")
+# 1. 报价率 - 安全计算,避免除零错误
+if (
+ "offer_mst_cnt" in df_model_source.columns
+ and "view_mst_cnt" in df_model_source.columns
+):
+ # 确保分母不为0,并处理缺失值
+ view_cnt_safe = df_model_source["view_mst_cnt"].fillna(0).replace(0, 1)
+ offer_cnt_safe = df_model_source["offer_mst_cnt"].fillna(0)
+ df_model_source["offer_rate"] = offer_cnt_safe / view_cnt_safe
+ # 确保结果在合理范围内 [0, 1]
+ df_model_source["offer_rate"] = df_model_source["offer_rate"].clip(0, 1)
+else:
+ print(" ⚠️ 缺少报价相关字段,offer_rate设为0")
+ df_model_source["offer_rate"] = 0
+# 2. onsite_to_finish_hour - 安全计算,处理异常值
+if (
+ "mst_serve_complete_last_time" in df_model_source.columns
+ and "order_onsite_sign_time" in df_model_source.columns
+):
+ try:
+ complete_time = pd.to_datetime(
+ df_model_source["mst_serve_complete_last_time"], errors="coerce"
+ )
+ onsite_time = pd.to_datetime(
+ df_model_source["order_onsite_sign_time"], errors="coerce"
+ )
+ # 计算时间差(小时)
+ time_diff = (complete_time - onsite_time).dt.total_seconds() / 3600
+ # 处理异常值:负值设为0,超过7天(168小时)的设为168
+ time_diff = time_diff.fillna(0) # NaT设为0
+ time_diff = time_diff.clip(lower=0, upper=168) # 限制在合理范围
+ df_model_source["onsite_to_finish_hour"] = time_diff
+ print(
+ f" - 完工时长计算完成,范围: {time_diff.min():.1f} - {time_diff.max():.1f} 小时"
+ )
+ except Exception as e:
+ print(f" ⚠️ 完工时长计算失败: {e},设为0")
+ df_model_source["onsite_to_finish_hour"] = 0
+else:
+ print(" ⚠️ 缺少完工时间相关字段,onsite_to_finish_hour设为0")
+ df_model_source["onsite_to_finish_hour"] = 0
+# 3. merchant_aftersale_rate - 安全计算,处理异常值
+if (
+ "merchant_total_aftersales" in df_model_source.columns
+ and "merchant_total_orders" in df_model_source.columns
+):
+ # 确保分母不为0,并处理缺失值
+ total_orders_safe = df_model_source["merchant_total_orders"].fillna(0).replace(0, 1)
+ total_aftersales_safe = df_model_source["merchant_total_aftersales"].fillna(0)
+ df_model_source["merchant_aftersale_rate"] = (
+ total_aftersales_safe / total_orders_safe
+ )
+ # 确保结果在合理范围内 [0, 1]
+ df_model_source["merchant_aftersale_rate"] = df_model_source[
+ "merchant_aftersale_rate"
+ ].clip(0, 1)
+ print(
+ f" - 商家售后率计算完成,范围: {df_model_source['merchant_aftersale_rate'].min():.3f} - {df_model_source['merchant_aftersale_rate'].max():.3f}"
+ )
+else:
+ print(" ⚠️ 缺少商家售后相关字段,merchant_aftersale_rate设为0")
+ df_model_source["merchant_aftersale_rate"] = 0
+# 4. ignore_cnt
+if "ignore_cnt" not in df_model_source.columns:
+ df_model_source["ignore_cnt"] = 0
+# 检查可用的后验特征
+available_reference_features = [
+ f for f in ALL_REFERENCE_FEATURES if f in df_model_source.columns
+]
+print(f" - 可用后验特征: {available_reference_features}")
+def calculate_robust_quantiles(data: pd.Series, quantiles: list, feature_name: str):
+ """
+
+ 计算稳健分位数,自动处理极端值影响
+
+ """
+ print(
+ f" 📊 {feature_name} 原始数据: N={len(data):,}, 范围=[{data.min():.2f}, {data.max():.2f}]"
+ )
+ # 第1层:基础有效性过滤
+ valid_data = data[data > 0].copy() # 移除0值和负值
+ print(
+ f" 🔍 有效值过滤: N={len(valid_data):,} (移除{len(data)-len(valid_data):,}个≤0值)"
+ )
+ if len(valid_data) < 10:
+ print(f" ⚠️ 有效数据不足,使用原始数据")
+ valid_data = data.copy()
+ # 第2层:IQR极端值过滤
+ Q1 = valid_data.quantile(0.25)
+ Q3 = valid_data.quantile(0.75)
+ IQR = Q3 - Q1
+ lower_bound = Q1 - 1.5 * IQR
+ upper_bound = Q3 + 1.5 * IQR
+ iqr_filtered = valid_data[(valid_data >= lower_bound) & (valid_data <= upper_bound)]
+ outliers_removed = len(valid_data) - len(iqr_filtered)
+ print(f" 🛡️ IQR过滤: N={len(iqr_filtered):,} (移除{outliers_removed:,}个极端值)")
+ # 确保有足够数据计算分位数
+ if len(iqr_filtered) < 10:
+ print(f" 🚨 稳健数据不足10个,回退到有效数据")
+ robust_data = valid_data
+ else:
+ robust_data = iqr_filtered
+ # 计算稳健分位数
+ result_quantiles = {}
+ for q in quantiles:
+ q_value = robust_data.quantile(q)
+ result_quantiles[f"{q:.0%}"] = q_value
+ print(f" {q:.0%}分位数: {q_value:.2f}")
+ return result_quantiles
+
+
+# 动态阈值计算系统将移至先验特征生成之后
+# 计算优化的先验统计(双维度+单维度,无全局兜底)
+print("正在计算优化的先验统计(双维度→单维度回退,无全局兜底)...")
+# 计算统计数据
+prior_stats = {}
+category_stats = {}
+global_stats = {}
+for feature in available_reference_features:
+ print(f" - 正在计算 {feature} 的统计...")
+ # 只对有效数据计算统计
+ valid_data = df_model_source[df_model_source[feature].notna()]
+ if len(valid_data) > 0:
+ # 1. 双维度统计:user_id + goods_level_3_name (改用中位数)
+ dual_stats = (
+ valid_data.groupby(["user_id", "goods_level_3_name"])[feature]
+ .agg(["median", "std", "count"])
+ .fillna(0)
+ )
+ dual_stats.columns = [f"{feature}_median", f"{feature}_std", f"{feature}_count"]
+ prior_stats[feature] = dual_stats
+ # 2. 单维度统计:goods_level_3_name (改用中位数)
+ single_stats = (
+ valid_data.groupby("goods_level_3_name")[feature]
+ .agg(["median", "std", "count"])
+ .fillna(0)
+ )
+ single_stats.columns = [
+ f"{feature}_median",
+ f"{feature}_std",
+ f"{feature}_count",
+ ]
+ category_stats[feature] = single_stats
+ # 3. 全局统计:整个数据集的中位数
+ global_median = valid_data[feature].median()
+ global_stats[feature] = {"median": global_median}
+print(
+ f" - 先验统计完成,共 {len(available_reference_features)} 个特征(双维度+单维度+全局统计)"
+)
+# 生成历史先验特征(优化回退策略:双维度→单维度→零值)
+print("正在生成历史先验特征...")
+
+
+def get_prior_feature_value(
+ user_id, goods_l3, feature_name, prior_stats, category_stats, global_stats=None
+):
+ """
+
+ 优化的先验特征值获取:双维度→单维度回退,无全局兜底
+
+ 注意:训练时user_id是数字类型,线上传入的是字符串,需要转换
+
+ """
+ # 转换user_id为数字类型,匹配训练时的数据类型
+ try:
+ user_id_num = int(user_id) if user_id else 0
+ except (ValueError, TypeError):
+ user_id_num = 0
+ # 1. 尝试双维度:user_id + goods_level_3_name (样本>=3)
+ if feature_name in prior_stats:
+ dual_stats_df = prior_stats[feature_name]
+ dual_key = (user_id_num, goods_l3)
+ if (
+ dual_key in dual_stats_df.index
+ and dual_stats_df.loc[dual_key, f"{feature_name}_count"] >= 3
+ ):
+ return dual_stats_df.loc[dual_key, f"{feature_name}_median"]
+ # 2. 回退到单维度:goods_level_3_name
+ if feature_name in category_stats:
+ single_stats_df = category_stats[feature_name]
+ if goods_l3 in single_stats_df.index:
+ return single_stats_df.loc[goods_l3, f"{feature_name}_median"]
+ # 3. 最终返回0(不使用全局统计)
+ return 0
+
+
+# 为每个训练样本生成先验特征(向量化优化)
+print(" - 正在向量化生成先验特征...")
+
+
+def generate_prior_features_vectorized(
+ df, features, prior_stats, category_stats, global_stats=None
+):
+ """向量化生成先验特征(修复用户ID类型不一致问题)"""
+ result_df = df.copy()
+ for feature in features:
+ print(f" - 正在生成 {feature} 的先验特征...")
+ # 统一用户ID类型:保持数字类型,与prior_stats的索引一致
+ user_ids = df["user_id"].astype(int) # 修复:统一为数字类型
+ goods_l3s = df["goods_level_3_name"].astype(str)
+ prior_values = np.zeros(len(df))
+ # 1. 双维度匹配(修复类型匹配问题)
+ if feature in prior_stats:
+ dual_stats_df = prior_stats[feature]
+ for i in range(len(df)):
+ dual_key = (user_ids.iloc[i], goods_l3s.iloc[i])
+ if (
+ dual_key in dual_stats_df.index
+ and dual_stats_df.loc[dual_key, f"{feature}_count"] >= 3
+ ):
+ prior_values[i] = dual_stats_df.loc[dual_key, f"{feature}_median"]
+ # 2. 单维度回退
+ if feature in category_stats:
+ single_stats_df = category_stats[feature]
+ mask = (prior_values == 0) & (goods_l3s.isin(single_stats_df.index))
+ for goods_l3 in goods_l3s[mask].unique():
+ if goods_l3 in single_stats_df.index:
+ goods_mask = (goods_l3s == goods_l3) & (prior_values == 0)
+ prior_values[goods_mask] = single_stats_df.loc[
+ goods_l3, f"{feature}_median"
+ ]
+ # 3. 不再全局兜底,剩余为0
+ result_df[f"{feature}_prior"] = prior_values
+ return result_df
+
+
+# 使用向量化函数生成先验特征
+df_model_source = generate_prior_features_vectorized(
+ df_model_source,
+ available_reference_features,
+ prior_stats,
+ category_stats,
+ global_stats,
+)
+# 生成先验特征列表
+prior_features = [f"{f}_prior" for f in available_reference_features]
+print(f" - 生成的先验特征: {prior_features}")
+# === 基于先验特征的动态阈值计算系统(不使用现值特征) ===
+print(f"\n🔧 正在基于先验特征计算规则评分动态阈值...")
+global DYNAMIC_RULE_THRESHOLDS
+DYNAMIC_RULE_THRESHOLDS = {}
+
+def _safe_series(df, col):
+ return df[col] if col in df.columns else pd.Series([], dtype=float)
+
+# 1. 总金额先验阈值
+amount_prior_series = _safe_series(df_model_source, "order_total_amount_prior")
+if len(amount_prior_series) > 0:
+ amount_quantiles = calculate_robust_quantiles(
+ amount_prior_series, [0.2, 0.4, 0.6, 0.8, 0.95], "order_total_amount_prior"
+ )
+ DYNAMIC_RULE_THRESHOLDS["order_total_amount"] = {
+ 15: amount_quantiles.get("95%", 0),
+ 12: amount_quantiles.get("80%", 0),
+ 9: amount_quantiles.get("60%", 0),
+ 6: amount_quantiles.get("40%", 0),
+ 3: amount_quantiles.get("20%", 0),
+ 1: 0,
+ }
+
+# 2. 单价先验阈值
+unit_price_prior_series = _safe_series(df_model_source, "order_unit_price_prior")
+if len(unit_price_prior_series) > 0:
+ unit_price_quantiles = calculate_robust_quantiles(
+ unit_price_prior_series, [0.2, 0.4, 0.6, 0.8, 0.95], "order_unit_price_prior"
+ )
+ DYNAMIC_RULE_THRESHOLDS["order_unit_price"] = {
+ 15: unit_price_quantiles.get("95%", 0),
+ 12: unit_price_quantiles.get("80%", 0),
+ 9: unit_price_quantiles.get("60%", 0),
+ 6: unit_price_quantiles.get("40%", 0),
+ 3: unit_price_quantiles.get("20%", 0),
+ 1: 0,
+ }
+
+# 3. 满10人报价时长先验阈值(越小越好)
+tenth_prior_series = _safe_series(df_model_source, "tenth_offer_duration_second_prior")
+if len(tenth_prior_series) > 0:
+ tenth_duration_quantiles = calculate_robust_quantiles(
+ tenth_prior_series, [0.05, 0.2, 0.4, 0.6, 0.8], "tenth_offer_duration_second_prior"
+ )
+ DYNAMIC_RULE_THRESHOLDS["tenth_offer_duration_second"] = {
+ 20: tenth_duration_quantiles.get("5%", float("inf")),
+ 16: tenth_duration_quantiles.get("20%", float("inf")),
+ 12: tenth_duration_quantiles.get("40%", float("inf")),
+ 8: tenth_duration_quantiles.get("60%", float("inf")),
+ 4: tenth_duration_quantiles.get("80%", float("inf")),
+ 1: float("inf"),
+ }
+
+# 4. 满5人报价时长先验阈值(越小越好)
+fifth_prior_series = _safe_series(df_model_source, "fifth_offer_duration_second_prior")
+if len(fifth_prior_series) > 0:
+ fifth_duration_quantiles = calculate_robust_quantiles(
+ fifth_prior_series, [0.05, 0.2, 0.4, 0.6, 0.8], "fifth_offer_duration_second_prior"
+ )
+ DYNAMIC_RULE_THRESHOLDS["fifth_offer_duration_second"] = {
+ 10: fifth_duration_quantiles.get("5%", float("inf")),
+ 8: fifth_duration_quantiles.get("20%", float("inf")),
+ 6: fifth_duration_quantiles.get("40%", float("inf")),
+ 3: fifth_duration_quantiles.get("60%", float("inf")),
+ 1: fifth_duration_quantiles.get("80%", float("inf")),
+ }
+
+# 5. 完工时长先验阈值(越小越好)
+finish_prior_series = _safe_series(df_model_source, "onsite_to_finish_hour_prior")
+if len(finish_prior_series) > 0:
+ finish_time_quantiles = calculate_robust_quantiles(
+ finish_prior_series, [0.05, 0.2, 0.4, 0.6, 0.8], "onsite_to_finish_hour_prior"
+ )
+ DYNAMIC_RULE_THRESHOLDS["onsite_to_finish_hour"] = {
+ 10: finish_time_quantiles.get("5%", float("inf")),
+ 8: finish_time_quantiles.get("20%", float("inf")),
+ 6: finish_time_quantiles.get("40%", float("inf")),
+ 4: finish_time_quantiles.get("60%", float("inf")),
+ 2: finish_time_quantiles.get("80%", float("inf")),
+ 1: float("inf"),
+ }
+
+# 6. 查看报价率先验阈值(越大越好)
+offer_rate_prior_series = _safe_series(df_model_source, "offer_rate_prior")
+if len(offer_rate_prior_series) > 0:
+ offer_rate_quantiles = calculate_robust_quantiles(
+ offer_rate_prior_series, [0.2, 0.4, 0.6, 0.8, 0.95], "offer_rate_prior"
+ )
+ DYNAMIC_RULE_THRESHOLDS["offer_rate"] = {
+ 15: offer_rate_quantiles.get("95%", 0),
+ 12: offer_rate_quantiles.get("80%", 0),
+ 9: offer_rate_quantiles.get("60%", 0),
+ 6: offer_rate_quantiles.get("40%", 0),
+ 3: offer_rate_quantiles.get("20%", 0),
+ 0: 0,
+ }
+
+# 7. 商家售后率先验阈值(越小越好)
+aftersale_prior_series = _safe_series(df_model_source, "merchant_aftersale_rate_prior")
+if len(aftersale_prior_series) > 0:
+ aftersale_quantiles = calculate_robust_quantiles(
+ aftersale_prior_series, [0.05, 0.2, 0.4, 0.6, 0.8], "merchant_aftersale_rate_prior"
+ )
+ DYNAMIC_RULE_THRESHOLDS["merchant_aftersale_rate"] = {
+ 8: aftersale_quantiles.get("5%", 0),
+ 6: aftersale_quantiles.get("20%", 0),
+ 4: aftersale_quantiles.get("40%", 0),
+ 2: aftersale_quantiles.get("60%", 0),
+ 0: aftersale_quantiles.get("80%", 0),
+ }
+
+# 8. 师傅关注数先验(区间)
+attention_prior_series = _safe_series(df_model_source, "attention_cnt_prior")
+if len(attention_prior_series) > 0:
+ attention_quantiles = calculate_robust_quantiles(
+ attention_prior_series, [0.1, 0.3, 0.5, 0.7, 0.9], "attention_cnt_prior"
+ )
+ optimal_min = attention_quantiles.get("30%", 0)
+ optimal_max = attention_quantiles.get("70%", 0)
+ DYNAMIC_RULE_THRESHOLDS["attention_cnt"] = {
+ "optimal_range": (optimal_min, optimal_max),
+ "general_range": (
+ attention_quantiles.get("10%", 0),
+ attention_quantiles.get("90%", 0),
+ ),
+ }
+
+# 9. 被拉黑数先验(越小越好)
+ignore_prior_series = _safe_series(df_model_source, "ignore_cnt_prior")
+if len(ignore_prior_series) > 0:
+ ignore_quantiles = calculate_robust_quantiles(
+ ignore_prior_series, [0.5, 0.7, 0.85, 0.95], "ignore_cnt_prior"
+ )
+ DYNAMIC_RULE_THRESHOLDS["ignore_cnt"] = {
+ 5: 0,
+ 3: ignore_quantiles.get("70%", 0),
+ 1: ignore_quantiles.get("85%", 0),
+ 0: ignore_quantiles.get("95%", 0),
+ }
+
+print(f"✅ 基于先验的动态规则阈值计算完成!特征数: {len(DYNAMIC_RULE_THRESHOLDS)}")
+joblib.dump(
+ DYNAMIC_RULE_THRESHOLDS, os.path.join(MODEL_DIR, "dynamic_rule_thresholds.pkl")
+)
+print(f"💾 动态规则阈值已保存至: dynamic_rule_thresholds.pkl")
+# 基于先验金额设置 has_price_info(价值类只用先验)
+if "order_total_amount_prior" in df_model_source.columns:
+ df_model_source["has_price_info"] = (df_model_source["order_total_amount_prior"] > 0).astype(int)
+# 最终模型特征列表
+MODEL_FEATURES = STATIC_BASE_FEATURES + prior_features
+# 缺失率统计
+missing_rate = df_model_source.isnull().mean().sort_values(ascending=False)
+print("\n特征缺失率统计(>0的特征):")
+print(missing_rate[missing_rate > 0])
+# 先验特征覆盖率(双维度→单维度回退策略)
+prior_coverage = 1 - df_model_source[prior_features].isnull().mean()
+print("\n先验特征覆盖率(非零值比例):")
+# 计算非零值比例
+non_zero_coverage = {}
+for feature in prior_features:
+ if feature in df_model_source.columns:
+ non_zero_rate = (df_model_source[feature] != 0).mean()
+ non_zero_coverage[feature] = non_zero_rate
+print("非零先验特征覆盖率:")
+for feature, rate in non_zero_coverage.items():
+ print(f" {feature}: {rate:.2%}")
+print(f"平均非零覆盖率: {sum(non_zero_coverage.values())/len(non_zero_coverage):.2%}")
+# 6. 使用新的特征体系进行建模
+print("正在使用新特征体系进行建模...")
+df_model = df_model_source[MODEL_FEATURES].copy()
+df_bidding = df_model_source.copy()
+print(f"最终建模数据量: {len(df_bidding):,} 单")
+# 缺失值处理(价值类现值不入模,因此不再用现值统计填充它们)
+print("正在处理缺失值...")
+print(f"缺失值处理前数据量: {len(df_model):,} 单")
+missing_counts = df_model.isnull().sum()
+print("各特征缺失值统计:")
+for col, count in missing_counts[missing_counts > 0].items():
+ print(f" {col}: {count:,} 个缺失值")
+# 填充缺失值
+print("正在填充缺失值...")
+# 基础特征填充
+fillna_dict = {
+ "order_goods_cnt": 1,
+ "business_rule_score": 0, # 业务规则得分默认0
+}
+# 为先验特征添加默认值
+for feature in prior_features:
+ if feature in df_model.columns:
+ fillna_dict[feature] = 0 # 先验特征默认0
+df_model = df_model.fillna(fillna_dict)
+# 再次全量兜底,防止有遗漏
+print("再次全量填充0,防止NaN...")
+df_model = df_model.fillna(0)
+print(f"缺失值填充后数据量: {len(df_model):,} 单")
+print(f"保留的订单数: {len(df_model):,} 单")
+# 保证主DataFrame与模型所用数据行对齐
+df_bidding = df_model_source.loc[df_model.index].copy()
+# --- 模型训练与聚类 ---
+print("\n正在进行模型训练...")
+# 完全StandardScaler标准化
+print("正在进行StandardScaler特征标准化...")
+scaler = StandardScaler()
+X_scaled = scaler.fit_transform(df_model)
+joblib.dump(scaler, os.path.join(MODEL_DIR, "scaler.pkl"))
+print(f" - 所有特征统一标准化:{len(df_model.columns)}个特征")
+# KMeans聚类
+print("正在训练KMeans模型 (n_clusters=6)...")
+kmeans = KMeans(n_clusters=6, random_state=42, n_init=10)
+kmeans.fit(X_scaled)
+joblib.dump(kmeans, os.path.join(MODEL_DIR, "kmeans_model.pkl"))
+# 生成用户属性映射(用于预测时补充用户信息)
+print("正在生成用户属性映射...")
+user_attributes = {}
+for user_id in df_model_source["user_id"].unique():
+ user_data = df_model_source[df_model_source["user_id"] == user_id]
+ user_attributes[user_id] = {
+ "attention_cnt": (
+ user_data["attention_cnt"].iloc[0]
+ if "attention_cnt" in user_data.columns
+ else 0
+ ),
+ "merchant_aftersale_rate": (
+ user_data["merchant_aftersale_rate"].iloc[0]
+ if "merchant_aftersale_rate" in user_data.columns
+ else 0.0
+ ),
+ "ignore_cnt": (
+ user_data["ignore_cnt"].iloc[0] if "ignore_cnt" in user_data.columns else 0
+ ),
+ "business_full_name": (
+ user_data["business_full_name"].iloc[0]
+ if "business_full_name" in user_data.columns
+ else ""
+ ),
+ "address": (
+ user_data["address"].iloc[0] if "address" in user_data.columns else ""
+ ),
+ }
+print(f" - 用户属性映射完成: {len(user_attributes)} 个用户")
+# 保存业务规则配置和统计数据
+joblib.dump(BUSINESS_RULES, os.path.join(MODEL_DIR, "business_rules.pkl"))
+joblib.dump(prior_stats, os.path.join(MODEL_DIR, "prior_stats.pkl"))
+joblib.dump(category_stats, os.path.join(MODEL_DIR, "category_stats.pkl"))
+joblib.dump(global_stats, os.path.join(MODEL_DIR, "global_stats.pkl"))
+joblib.dump(MODEL_FEATURES, os.path.join(MODEL_DIR, "model_features.pkl"))
+joblib.dump(user_attributes, os.path.join(MODEL_DIR, "user_attributes.pkl"))
+# 注意:cluster_map 将在后续动态聚类解读后保存
+print(f"模型训练完成,相关组件已保存至 '{MODEL_DIR}' 目录。")
+# --- 3. 业务解读与标签映射 ---
+print("\n[Part 3] 分析聚类中心,为聚类结果赋予业务含义...")
+df_bidding["static_cluster"] = kmeans.labels_
+# 为了方便业务理解,我们将标准化的聚类中心还原为原始数值
+cluster_centers_original = scaler.inverse_transform(kmeans.cluster_centers_)
+cluster_centers_df = pd.DataFrame(cluster_centers_original, columns=df_model.columns)
+print("聚类中心 (原始数值尺度):")
+print(cluster_centers_df)
+# 动态聚类解读算法
+
+
+def generate_dynamic_cluster_labels(cluster_centers_df, df_bidding):
+ """
+
+ 基于聚类中心特征值动态生成业务标签
+
+ """
+ print("正在进行动态聚类解读...")
+ cluster_labels = {}
+ # 计算全局特征分位数,用于判断高低
+ global_percentiles = {}
+ key_features = [
+ "order_total_amount_prior",
+ "order_goods_cnt",
+ "order_unit_price_prior",
+ "fifth_offer_duration_second_prior",
+ "tenth_offer_duration_second_prior",
+ "onsite_to_finish_hour_prior",
+ "offer_rate_prior",
+ "attention_cnt_prior",
+ "merchant_aftersale_rate_prior",
+ "ignore_cnt_prior",
+ "buyer_note_100",
+ ]
+ for feature in key_features:
+ if feature in df_bidding.columns:
+ global_percentiles[feature] = {
+ "low": df_bidding[feature].quantile(0.33),
+ "high": df_bidding[feature].quantile(0.67),
+ "very_high": df_bidding[feature].quantile(0.9),
+ }
+ # 为每个聚类生成标签
+ for cluster_id in range(len(cluster_centers_df)):
+ center = cluster_centers_df.iloc[cluster_id]
+ # 分析关键特征
+ characteristics = []
+ # 1. 订单价值特征
+ if (
+ "order_total_amount_prior" in center.index
+ and "order_total_amount_prior" in global_percentiles
+ ):
+ amount = center["order_total_amount_prior"]
+ if amount >= global_percentiles["order_total_amount_prior"]["very_high"]:
+ characteristics.append("超高价")
+ elif amount >= global_percentiles["order_total_amount_prior"]["high"]:
+ characteristics.append("高价")
+ elif amount <= global_percentiles["order_total_amount_prior"]["low"]:
+ characteristics.append("低价")
+ else:
+ characteristics.append("中价")
+ # 2. 订单规模特征
+ if (
+ "order_goods_cnt" in center.index
+ and "order_goods_cnt" in global_percentiles
+ ):
+ goods_cnt = center["order_goods_cnt"]
+ if goods_cnt >= global_percentiles["order_goods_cnt"]["very_high"]:
+ characteristics.append("超大批量")
+ elif goods_cnt >= global_percentiles["order_goods_cnt"]["high"]:
+ characteristics.append("大批量")
+ elif goods_cnt <= global_percentiles["order_goods_cnt"]["low"]:
+ characteristics.append("小批量")
+ # 3. 响应效率特征
+ if (
+ "fifth_offer_duration_second_prior" in center.index
+ and "fifth_offer_duration_second_prior" in global_percentiles
+ ):
+ duration = center["fifth_offer_duration_second_prior"]
+ if (
+ duration
+ >= global_percentiles["fifth_offer_duration_second_prior"]["very_high"]
+ ):
+ characteristics.append("极慢响应")
+ elif duration >= global_percentiles["fifth_offer_duration_second_prior"]["high"]:
+ characteristics.append("慢响应")
+ elif duration <= global_percentiles["fifth_offer_duration_second_prior"]["low"]:
+ characteristics.append("快响应")
+ # 4. 工期特征
+ if (
+ "onsite_to_finish_hour_prior" in center.index
+ and "onsite_to_finish_hour_prior" in global_percentiles
+ ):
+ finish_time = center["onsite_to_finish_hour_prior"]
+ if finish_time >= global_percentiles["onsite_to_finish_hour_prior"]["very_high"]:
+ characteristics.append("超长工期")
+ elif finish_time >= global_percentiles["onsite_to_finish_hour_prior"]["high"]:
+ characteristics.append("长工期")
+ elif finish_time <= global_percentiles["onsite_to_finish_hour_prior"]["low"]:
+ characteristics.append("短工期")
+ # 5. 报价率特征
+ if "offer_rate_prior" in center.index and "offer_rate_prior" in global_percentiles:
+ offer_rate = center["offer_rate_prior"]
+ if offer_rate >= global_percentiles["offer_rate_prior"]["high"]:
+ characteristics.append("高报价率")
+ elif offer_rate <= global_percentiles["offer_rate_prior"]["low"]:
+ characteristics.append("低报价率")
+ # 6. 关注度特征
+ if "attention_cnt_prior" in center.index and "attention_cnt_prior" in global_percentiles:
+ attention = center["attention_cnt_prior"]
+ if attention >= global_percentiles["attention_cnt_prior"]["very_high"]:
+ characteristics.append("极高关注")
+ elif attention >= global_percentiles["attention_cnt_prior"]["high"]:
+ characteristics.append("高关注")
+ elif attention <= global_percentiles["attention_cnt_prior"]["low"]:
+ characteristics.append("低关注")
+ # 7. 风险特征
+ if (
+ "merchant_aftersale_rate_prior" in center.index
+ and "merchant_aftersale_rate_prior" in global_percentiles
+ ):
+ aftersale_rate = center["merchant_aftersale_rate_prior"]
+ if aftersale_rate >= global_percentiles["merchant_aftersale_rate_prior"]["high"]:
+ characteristics.append("高售后")
+ if "ignore_cnt_prior" in center.index and "ignore_cnt_prior" in global_percentiles:
+ ignore_cnt = center["ignore_cnt_prior"]
+ if ignore_cnt >= global_percentiles["ignore_cnt_prior"]["very_high"]:
+ characteristics.append("高拉黑")
+ elif ignore_cnt >= global_percentiles["ignore_cnt_prior"]["high"]:
+ characteristics.append("中拉黑")
+ # 8. 复杂度特征
+ if "buyer_note_100" in center.index and center["buyer_note_100"] > 0.5:
+ characteristics.append("复杂需求")
+ # 生成标签
+ if not characteristics:
+ label = f"普通订单_{cluster_id}"
+ else:
+ # 优先级排序:价值 > 规模 > 效率 > 风险
+ priority_order = [
+ "超高价",
+ "高价",
+ "超大批量",
+ "大批量",
+ "极慢响应",
+ "慢响应",
+ "快响应",
+ "超长工期",
+ "长工期",
+ "短工期",
+ "高报价率",
+ "低报价率",
+ "极高关注",
+ "高关注",
+ "低关注",
+ "高售后",
+ "高拉黑",
+ "复杂需求",
+ ]
+ # 按优先级选择前2-3个特征
+ sorted_chars = [char for char in priority_order if char in characteristics]
+ if len(sorted_chars) == 0:
+ sorted_chars = characteristics[:2]
+ elif len(sorted_chars) == 1:
+ sorted_chars = (
+ sorted_chars
+ + [char for char in characteristics if char not in sorted_chars][:1]
+ )
+ else:
+ sorted_chars = sorted_chars[:2]
+ # 构建标签
+ if len(sorted_chars) == 1:
+ label = f"{sorted_chars[0]}订单"
+ else:
+ label = f'{"".join(sorted_chars)}订单'
+ cluster_labels[cluster_id] = label
+ # 打印解读过程
+ cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id]
+ print(f"\n聚类 {cluster_id} 动态解读:")
+ print(f" • 订单数量: {len(cluster_data):,} 单")
+ print(
+ f" • 识别特征: {', '.join(characteristics) if characteristics else '无明显特征'}"
+ )
+ print(f" • 生成标签: {label}")
+ # 显示关键数值
+ if "order_total_amount_prior" in center.index:
+ print(f" • 平均金额(先验): ¥{center['order_total_amount_prior']:.0f}")
+ if "order_goods_cnt" in center.index:
+ print(f" • 平均件数: {center['order_goods_cnt']:.1f} 件")
+ if "fifth_offer_duration_second_prior" in center.index:
+ print(f" • 5人报价时长(先验): {center['fifth_offer_duration_second_prior']:.0f} 秒")
+ if "onsite_to_finish_hour_prior" in center.index:
+ print(f" • 平均工期(先验): {center['onsite_to_finish_hour_prior']:.1f} 小时")
+ return cluster_labels
+
+
+# 执行动态聚类解读
+cluster_map = generate_dynamic_cluster_labels(cluster_centers_df, df_bidding)
+df_bidding["static_label"] = df_bidding["static_cluster"].map(cluster_map)
+# 保存聚类映射
+joblib.dump(cluster_map, os.path.join(MODEL_DIR, "cluster_map.pkl"))
+print("聚类映射已保存至模型目录")
+print("\n✅ 动态聚类解读完成!已为历史数据自动生成业务标签。")
+# 计算聚类统计
+cluster_counts = df_bidding["static_cluster"].value_counts().sort_index()
+# 聚类可视化(智能版:自动检测异常聚类)
+print("\n正在生成聚类可视化图...")
+
+
+def detect_outlier_clusters(
+ df_data, cluster_centers_df, min_samples=50, value_threshold=10.0
+):
+ """
+
+ 智能检测异常聚类的通用方法
+
+ 参数:
+
+ df_data: 数据DataFrame
+
+ cluster_centers_df: 聚类中心DataFrame
+
+ min_samples: 最小样本数阈值,少于此数量视为异常
+
+ value_threshold: 价值差异倍数,超过此倍数视为异常
+
+ 返回:
+
+ outlier_clusters: 异常聚类ID列表
+
+ normal_clusters: 正常聚类ID列表
+
+ """
+ outlier_clusters = []
+ normal_clusters = []
+ print("正在智能检测异常聚类...")
+ # 计算每个聚类的统计信息
+ cluster_stats = {}
+ for cluster_id in range(len(cluster_centers_df)):
+ cluster_data = df_data[df_data["static_cluster"] == cluster_id]
+ cluster_stats[cluster_id] = {
+ "count": len(cluster_data),
+ "avg_amount": cluster_centers_df.iloc[cluster_id].get("order_total_amount_prior", 0),
+ "avg_goods": cluster_centers_df.iloc[cluster_id]["order_goods_cnt"],
+ }
+ # 计算整体平均值作为基准
+ total_avg_amount = df_data.get("order_total_amount_prior", pd.Series([0]*len(df_data))).mean()
+ total_avg_goods = df_data["order_goods_cnt"].mean()
+ print("各聚类异常检测分析:")
+ for cluster_id, stats in cluster_stats.items():
+ is_outlier = False
+ reasons = []
+ # 检测1:样本数量过少
+ if stats["count"] < min_samples:
+ is_outlier = True
+ reasons.append(f"样本数过少({stats['count']})")
+ # 检测2:订单金额极值
+ amount_ratio = (
+ stats["avg_amount"] / total_avg_amount if total_avg_amount > 0 else 1
+ )
+ if amount_ratio > value_threshold:
+ is_outlier = True
+ reasons.append(f"金额异常({amount_ratio:.1f}倍)")
+ # 检测3:商品数量极值
+ goods_ratio = stats["avg_goods"] / total_avg_goods if total_avg_goods > 0 else 1
+ if goods_ratio > value_threshold:
+ is_outlier = True
+ reasons.append(f"件数异常({goods_ratio:.1f}倍)")
+ if is_outlier:
+ outlier_clusters.append(cluster_id)
+ print(f" 聚类{cluster_id}: 异常 - {', '.join(reasons)}")
+ else:
+ normal_clusters.append(cluster_id)
+ print(
+ f" 聚类{cluster_id}: 正常 - {stats['count']}单, ¥{stats['avg_amount']:.0f}, {stats['avg_goods']:.1f}件"
+ )
+ print(f"检测结果: 正常聚类{normal_clusters}, 异常聚类{outlier_clusters}")
+ return outlier_clusters, normal_clusters
+
+
+# 自动检测异常聚类
+outlier_clusters, normal_clusters = detect_outlier_clusters(
+ df_bidding, cluster_centers_df
+)
+# 分别处理正常聚类和异常聚类
+normal_mask = df_bidding["static_cluster"].isin(normal_clusters)
+outlier_mask = df_bidding["static_cluster"].isin(outlier_clusters)
+# 根据检测结果选择可视化策略
+if len(outlier_clusters) > 0:
+ print(f"检测到{len(outlier_clusters)}个异常聚类,使用分层可视化")
+ # 方案1:分层可视化 - 主图显示正常聚类,子图显示异常聚类
+ fig = plt.figure(figsize=(15, 10))
+else:
+ print("未检测到异常聚类,使用常规可视化")
+ # 常规可视化 - 所有聚类在同一图中
+ fig = plt.figure(figsize=(12, 10))
+# === 主图设置 ===
+ax_main = plt.subplot(1, 1, 1)
+# 使用专业的配色方案
+colors = ["#4E79A7", "#F28E2B", "#59A14F", "#E15759", "#76B7B2", "#EDC948"]
+if len(outlier_clusters) > 0:
+ # === 分层可视化:只显示正常聚类 ===
+ # 对正常聚类数据进行PCA
+ X_normal = X_scaled[normal_mask]
+ pca_normal = PCA(n_components=2)
+ X_pca_normal = pca_normal.fit_transform(X_normal)
+ # 绘制正常聚类散点图
+ normal_cluster_labels = df_bidding[normal_mask]["static_cluster"]
+ for cluster_id in normal_clusters:
+ cluster_mask = normal_cluster_labels == cluster_id
+ cluster_data = X_pca_normal[cluster_mask]
+ plt.scatter(
+ cluster_data[:, 0],
+ cluster_data[:, 1],
+ c=colors[cluster_id],
+ s=8,
+ alpha=0.7,
+ label=f'{cluster_map.get(cluster_id, f"聚类{cluster_id}")} ({cluster_counts.get(cluster_id, 0):,}单)',
+ edgecolors="none",
+ )
+ # 添加正常聚类中心
+ normal_centers = kmeans.cluster_centers_[normal_clusters]
+ centers_pca_normal = pca_normal.transform(normal_centers)
+ for i, cluster_id in enumerate(normal_clusters):
+ x, y = centers_pca_normal[i]
+ plt.scatter(
+ x, y, c="red", marker="*", s=300, linewidths=2, edgecolors="black", zorder=4
+ )
+ plt.text(
+ x,
+ y,
+ str(cluster_id),
+ color="white",
+ fontsize=12,
+ fontweight="bold",
+ ha="center",
+ va="center",
+ zorder=10,
+ )
+ # 主图设置(分层模式)
+ plt.title(
+ "报价招标订单智能分类结果展示(主要聚类)\n基于核心业务特征 + 业务规则",
+ fontsize=16,
+ fontweight="bold",
+ pad=20,
+ )
+ legend_title = "订单分类 (正常范围)"
+else:
+ # === 常规可视化:显示所有聚类 ===
+ # 对所有数据进行PCA
+ pca_all = PCA(n_components=2)
+ X_pca_all = pca_all.fit_transform(X_scaled)
+ # 绘制所有聚类散点图
+ for cluster_id in range(6):
+ cluster_mask = df_bidding["static_cluster"] == cluster_id
+ cluster_data = X_pca_all[cluster_mask]
+ plt.scatter(
+ cluster_data[:, 0],
+ cluster_data[:, 1],
+ c=colors[cluster_id],
+ s=12,
+ alpha=0.8,
+ label=f'{cluster_map.get(cluster_id, f"聚类{cluster_id}")} ({cluster_counts.get(cluster_id, 0):,}单)',
+ edgecolors="none",
+ )
+ # 添加所有聚类中心
+ centers_pca_all = pca_all.transform(kmeans.cluster_centers_)
+ for cluster_id in range(6):
+ x, y = centers_pca_all[cluster_id]
+ plt.scatter(
+ x, y, c="red", marker="*", s=400, linewidths=2, edgecolors="black", zorder=4
+ )
+ plt.text(
+ x,
+ y,
+ str(cluster_id),
+ color="white",
+ fontsize=14,
+ fontweight="bold",
+ ha="center",
+ va="center",
+ zorder=10,
+ )
+ # 主图设置(常规模式)
+ plt.title(
+ "报价招标订单智能分类结果展示\n基于核心业务特征 + 业务规则",
+ fontsize=16,
+ fontweight="bold",
+ pad=20,
+ )
+ legend_title = "订单分类"
+# 通用设置
+plt.xlabel("主成分1 (订单复杂度维度)", fontsize=12)
+plt.ylabel("主成分2 (订单价值维度)", fontsize=12)
+plt.grid(True, alpha=0.3, linestyle="--")
+# 创建图例
+legend1 = plt.legend(
+ title=legend_title, loc="upper left", fontsize=9, framealpha=0.95, edgecolor="black"
+)
+plt.gca().add_artist(legend1)
+# === 子图:异常聚类(智能适配) ===
+if outlier_mask.sum() > 0 and len(outlier_clusters) > 0:
+ ax_inset = fig.add_axes([0.65, 0.65, 0.32, 0.25]) # [x, y, width, height]
+ # 对包含异常点的所有数据进行PCA
+ pca_all = PCA(n_components=2)
+ X_pca_all = pca_all.fit_transform(X_scaled)
+ # 绘制所有异常聚类
+ outlier_cluster_labels = df_bidding[outlier_mask]["static_cluster"]
+ for cluster_id in outlier_clusters:
+ cluster_mask = outlier_cluster_labels == cluster_id
+ if cluster_mask.sum() > 0:
+ cluster_data = X_pca_all[outlier_mask][cluster_mask]
+ ax_inset.scatter(
+ cluster_data[:, 0],
+ cluster_data[:, 1],
+ c=colors[cluster_id],
+ s=100,
+ alpha=0.9,
+ edgecolors="black",
+ linewidth=1,
+ label=f"聚类{cluster_id}",
+ )
+ # 添加异常聚类中心
+ centers_pca_all = pca_all.transform(kmeans.cluster_centers_)
+ for cluster_id in outlier_clusters:
+ outlier_center = centers_pca_all[cluster_id]
+ ax_inset.scatter(
+ outlier_center[0],
+ outlier_center[1],
+ c="red",
+ marker="*",
+ s=200,
+ linewidths=2,
+ edgecolors="black",
+ zorder=4,
+ )
+ ax_inset.text(
+ outlier_center[0],
+ outlier_center[1],
+ str(cluster_id),
+ color="white",
+ fontsize=10,
+ fontweight="bold",
+ ha="center",
+ va="center",
+ zorder=10,
+ )
+ # 动态生成标题
+ outlier_count = outlier_mask.sum()
+ if len(outlier_clusters) == 1:
+ cluster_id = outlier_clusters[0]
+ title = f'聚类{cluster_id}:{cluster_map.get(cluster_id, "异常聚类")}\n({outlier_count}单,独立展示)'
+ else:
+ title = f"异常聚类:{outlier_clusters}\n({outlier_count}单,独立展示)"
+ ax_inset.set_title(title, fontsize=10, fontweight="bold")
+ ax_inset.grid(True, alpha=0.3, linestyle="--")
+ # 如果有多个异常聚类,添加小图例
+ if len(outlier_clusters) > 1:
+ ax_inset.legend(fontsize=8, loc="best")
+# 添加统计信息文本框(智能适配)
+normal_count = normal_mask.sum()
+outlier_count = outlier_mask.sum()
+outlier_info = (
+ f"聚类{outlier_clusters}"
+ if len(outlier_clusters) > 1
+ else f"聚类{outlier_clusters[0]}" if outlier_clusters else "无"
+)
+stats_text = f"""
+
+数据统计:
+
+• 主图显示: {normal_count:,} 单 ({normal_count/len(df_bidding)*100:.2f}%)
+
+• 异常聚类: {outlier_info} ({outlier_count}单,独立显示)
+
+• 主要类别: {cluster_map.get(cluster_counts.index[0], '未知')} ({cluster_counts.iloc[0]:,}单)
+
+• 第二大类别: {cluster_map.get(cluster_counts.index[1], '未知')} ({cluster_counts.iloc[1]:,}单)
+
+• 智能检测: 自动识别异常聚类,分层展示
+
+"""
+plt.text(
+ 0.02,
+ 0.35,
+ stats_text,
+ transform=ax_main.transAxes,
+ fontsize=9,
+ verticalalignment="top",
+ bbox=dict(boxstyle="round", facecolor="lightgreen", alpha=0.8),
+)
+# 调整布局
+plt.tight_layout()
+# 保存图片(智能命名)
+if len(outlier_clusters) > 0:
+ save_path = "/Users/tom/Documents/订单聚类可视化图_智能分层版.png"
+ version_name = "智能分层版"
+else:
+ save_path = "/Users/tom/Documents/订单聚类可视化图_常规版.png"
+ version_name = "常规版"
+plt.savefig(save_path, dpi=300, bbox_inches="tight")
+print(f"{version_name}聚类可视化图已保存至: {save_path}")
+# === 方案2:生成对比图 - 排除异常点版本(仅在有异常聚类时) ===
+if len(outlier_clusters) > 0:
+ plt.figure(figsize=(12, 8))
+ # 绘制排除异常点的散点图
+ for cluster_id in normal_clusters:
+ cluster_mask = normal_cluster_labels == cluster_id
+ cluster_data = X_pca_normal[cluster_mask]
+ plt.scatter(
+ cluster_data[:, 0],
+ cluster_data[:, 1],
+ c=colors[cluster_id],
+ s=12,
+ alpha=0.8,
+ label=f'{cluster_map.get(cluster_id, f"聚类{cluster_id}")} ({cluster_counts.get(cluster_id, 0):,}单)',
+ edgecolors="none",
+ )
+ # 添加聚类中心
+ for i, cluster_id in enumerate(normal_clusters):
+ x, y = centers_pca_normal[i]
+ plt.scatter(
+ x, y, c="red", marker="*", s=400, linewidths=2, edgecolors="black", zorder=4
+ )
+ plt.text(
+ x,
+ y,
+ str(cluster_id),
+ color="white",
+ fontsize=14,
+ fontweight="bold",
+ ha="center",
+ va="center",
+ zorder=10,
+ )
+ plt.title(
+ "报价招标订单聚类分布图(排除极值点)\n清晰展示主要订单模式分布",
+ fontsize=16,
+ fontweight="bold",
+ pad=20,
+ )
+ plt.xlabel("主成分1 (订单复杂度维度)", fontsize=12)
+ plt.ylabel("主成分2 (订单价值维度)", fontsize=12)
+ plt.grid(True, alpha=0.3, linestyle="--")
+ # 图例
+ plt.legend(
+ title="订单分类", loc="best", fontsize=10, framealpha=0.95, edgecolor="black"
+ )
+ # 添加说明(智能适配)
+ excluded_info = (
+ f"聚类{outlier_clusters}的{outlier_count}单" if outlier_clusters else "异常订单"
+ )
+ note_text = f"""
+
+说明:本图排除了{excluded_info}异常订单,
+
+以便清晰观察其余{normal_count:,}单的分布模式
+
+异常检测:自动识别样本少或特征极值的聚类
+
+ """
+ plt.text(
+ 0.02,
+ 0.98,
+ note_text,
+ transform=plt.gca().transAxes,
+ fontsize=10,
+ verticalalignment="top",
+ bbox=dict(boxstyle="round", facecolor="yellow", alpha=0.8),
+ )
+ plt.tight_layout()
+ # 保存对比图
+ contrast_path = "/Users/tom/Documents/订单聚类可视化图_排除异常点版.png"
+ plt.savefig(contrast_path, dpi=300, bbox_inches="tight")
+ print(f"排除异常点版聚类可视化图已保存至: {contrast_path}")
+else:
+ print("未检测到异常聚类,无需生成排除异常点版本")
+# plt.show()
+# --- 4. 动态特征验证 ---
+print("\n[Part 4] 使用动态特征验证静态分层效果...")
+available_dynamic_features = [
+ col for col in DYNAMIC_FEATURES if col in df_bidding.columns
+]
+if available_dynamic_features:
+ validation_summary = df_bidding.groupby("static_label")[
+ available_dynamic_features
+ ].mean()
+ print("各层级订单在动态特征上的平均表现:")
+ print(validation_summary)
+else:
+ print("没有可用的动态特征用于验证。")
+# --- 自动化好单识别算法 ---
+print("\n[Part 5] 混合评分好单识别算法...")
+# === 混合评分系统:规则70% + 聚类30% ===
+print("开始混合评分好单识别(规则70% + 聚类30%)...")
+# 混合评分配置 - 使用统一配置管理
+HYBRID_CONFIG = config.get_hybrid_config()
+
+
+def calculate_rule_score(order_data):
+ """
+
+ 计算规则评分 (0-100分) - 统一的规则评分函数
+
+ 训练时和预测时使用完全相同的逻辑
+
+ """
+ score = 0
+ # 🔧 修复:统一获取动态阈值的逻辑(训练时和预测时一致)
+ try:
+ # 优先使用全局变量(训练时)
+ if "DYNAMIC_RULE_THRESHOLDS" in globals() and DYNAMIC_RULE_THRESHOLDS:
+ thresholds = DYNAMIC_RULE_THRESHOLDS
+ else:
+ # 从文件加载(预测时)
+ thresholds = joblib.load(
+ os.path.join(MODEL_DIR, "dynamic_rule_thresholds.pkl")
+ )
+ except Exception as e:
+ # 降级到硬编码阈值(兼容性保护)
+ print(f"⚠️ 动态阈值加载失败: {e},使用硬编码阈值")
+ return calculate_rule_score_hardcoded(order_data)
+ # === 主参考特征 (85分) - 全部改为先验特征 ===
+ # 1. 满10人报价时长先验 (20分) - 基于历史统计
+ tenth_offer_duration_prior = order_data.get("tenth_offer_duration_second_prior", 0)
+ if tenth_offer_duration_prior == 0: # 无历史数据
+ tenth_duration_score = 0
+ elif "tenth_offer_duration_second" in thresholds:
+ # 使用动态阈值(但应用到先验特征)
+ tenth_thresholds = thresholds["tenth_offer_duration_second"]
+ tenth_duration_score = 0
+ for score_value, threshold in sorted(tenth_thresholds.items(), reverse=True):
+ if tenth_offer_duration_prior <= threshold:
+ tenth_duration_score = score_value
+ break
+ else:
+ tenth_duration_score = 0
+ score += tenth_duration_score
+ # 2. 查看报价率先验 (15分) - 基于历史统计
+ offer_rate_prior = order_data.get("offer_rate_prior", 0)
+ if "offer_rate" in thresholds:
+ # 使用动态阈值(但应用到先验特征)
+ offer_thresholds = thresholds["offer_rate"]
+ offer_rate_score = 0
+ for score_value, threshold in sorted(offer_thresholds.items(), reverse=True):
+ if offer_rate_prior >= threshold:
+ offer_rate_score = score_value
+ break
+ else:
+ offer_rate_score = 0
+ score += offer_rate_score
+ # 3. 总金额先验 (15分) - 订单实际金额(真正的先验)
+ amount_prior = order_data.get("order_total_amount_prior", 0)
+ if "order_total_amount" in thresholds:
+ # 使用动态阈值
+ amount_thresholds = thresholds["order_total_amount"]
+ amount_score = 0
+ for score_value, threshold in sorted(amount_thresholds.items(), reverse=True):
+ if amount_prior >= threshold:
+ amount_score = score_value
+ break
+ else:
+ amount_score = 1 if amount_prior > 0 else 0
+ score += amount_score
+ # 4. 单价先验 (15分) - 订单实际单价(真正的先验)
+ unit_price_prior = order_data.get("order_unit_price_prior", 0)
+ if "order_unit_price" in thresholds:
+ # 使用动态阈值
+ unit_thresholds = thresholds["order_unit_price"]
+ unit_price_score = 0
+ for score_value, threshold in sorted(unit_thresholds.items(), reverse=True):
+ if unit_price_prior >= threshold:
+ unit_price_score = score_value
+ break
+ else:
+ unit_price_score = 1 if unit_price_prior > 0 else 0
+ score += unit_price_score
+ # 5. 满5人报价时长先验 (10分) - 基于历史统计
+ fifth_offer_duration_prior = order_data.get("fifth_offer_duration_second_prior", 0)
+ if fifth_offer_duration_prior == 0: # 无历史数据
+ fifth_duration_score = 0
+ elif "fifth_offer_duration_second" in thresholds:
+ # 使用动态阈值(但应用到先验特征)
+ fifth_thresholds = thresholds["fifth_offer_duration_second"]
+ fifth_duration_score = 0
+ for score_value, threshold in sorted(fifth_thresholds.items(), reverse=True):
+ if fifth_offer_duration_prior <= threshold:
+ fifth_duration_score = score_value
+ break
+ else:
+ fifth_duration_score = 0
+ score += fifth_duration_score
+ # 6. 完工时长先验 (10分) - 基于历史统计
+ onsite_to_finish_hour_prior = order_data.get("onsite_to_finish_hour_prior", 0)
+ if onsite_to_finish_hour_prior == 0: # 无历史数据
+ finish_time_score = 0
+ elif "onsite_to_finish_hour" in thresholds:
+ # 使用动态阈值(但应用到先验特征)
+ finish_thresholds = thresholds["onsite_to_finish_hour"]
+ finish_time_score = 0
+ for score_value, threshold in sorted(finish_thresholds.items(), reverse=True):
+ if onsite_to_finish_hour_prior <= threshold:
+ finish_time_score = score_value
+ break
+ else:
+ finish_time_score = 0
+ score += finish_time_score
+ # === 次参考特征 (15分) - 全部改为先验特征 ===
+ # 7. 商家售后率先验 (8分) - 基于历史统计
+ merchant_aftersale_rate_prior = order_data.get(
+ "merchant_aftersale_rate_prior", order_data.get("merchant_aftersale_rate", 0)
+ )
+ if "merchant_aftersale_rate" in thresholds:
+ # 使用动态阈值
+ aftersale_thresholds = thresholds["merchant_aftersale_rate"]
+ aftersale_score = 0
+ for score_value, threshold in sorted(
+ aftersale_thresholds.items(), reverse=True
+ ):
+ if merchant_aftersale_rate_prior <= threshold:
+ aftersale_score = score_value
+ break
+ else:
+ aftersale_score = 0
+ score += aftersale_score
+ # 8. 商家被拉黑数先验 (5分) - 基于历史统计
+ ignore_cnt_prior = order_data.get(
+ "ignore_cnt_prior", order_data.get("ignore_cnt", 0)
+ )
+ if "ignore_cnt" in thresholds:
+ # 使用动态阈值
+ ignore_thresholds = thresholds["ignore_cnt"]
+ ignore_score = 0
+ for score_value, threshold in sorted(ignore_thresholds.items(), reverse=True):
+ if ignore_cnt_prior <= threshold:
+ ignore_score = score_value
+ break
+ else:
+ ignore_score = 5 if ignore_cnt_prior == 0 else 0
+ score += ignore_score
+ # 9. 师傅关注数先验 (2分) - 基于历史统计
+ attention_cnt_prior = order_data.get(
+ "attention_cnt_prior", order_data.get("attention_cnt", 0)
+ )
+ if attention_cnt_prior == 0:
+ attention_score = 0
+ elif "attention_cnt" in thresholds:
+ # 使用动态区间阈值
+ attention_config = thresholds["attention_cnt"]
+ optimal_range = attention_config.get("optimal_range", (3, 20))
+ general_range = attention_config.get("general_range", (1, 30))
+ if optimal_range[0] <= attention_cnt_prior <= optimal_range[1]:
+ attention_score = 2 # 最优区间
+ elif general_range[0] <= attention_cnt_prior <= general_range[1]:
+ attention_score = 1 # 一般区间
+ elif attention_cnt_prior > 0:
+ attention_score = 0.5 # 有关注即可
+ else:
+ attention_score = 0
+ else:
+ # 降级到硬编码逻辑
+ if 3 <= attention_cnt_prior <= 20:
+ attention_score = 2
+ elif 1 <= attention_cnt_prior <= 30:
+ attention_score = 1
+ elif attention_cnt_prior > 0:
+ attention_score = 0.5
+ else:
+ attention_score = 0
+ score += attention_score
+ return min(100, max(0, score))
+
+
+def calculate_rule_score_hardcoded(order_data):
+ """
+
+ 备用的硬编码规则评分函数(兼容性保护)- 修复为全先验特征版本
+
+ 当动态阈值不可用时自动降级使用
+
+ """
+ score = 0
+ # === 主参考特征 (85分) - 全部使用先验特征 ===
+ # 1. 满10人报价时长先验 (20分)
+ tenth_offer_duration_prior = order_data.get("tenth_offer_duration_second_prior", 0)
+ if tenth_offer_duration_prior == 0:
+ tenth_duration_score = 0 # 无历史数据
+ elif tenth_offer_duration_prior <= 1800:
+ tenth_duration_score = 20
+ elif tenth_offer_duration_prior <= 3600:
+ tenth_duration_score = 16
+ elif tenth_offer_duration_prior <= 7200:
+ tenth_duration_score = 12
+ elif tenth_offer_duration_prior <= 14400:
+ tenth_duration_score = 8
+ elif tenth_offer_duration_prior <= 28800:
+ tenth_duration_score = 4
+ else:
+ tenth_duration_score = 1
+ score += tenth_duration_score
+ # 2. 查看报价率先验 (15分)
+ offer_rate_prior = order_data.get("offer_rate_prior", 0)
+ if offer_rate_prior >= 0.7:
+ offer_rate_score = 15
+ elif offer_rate_prior >= 0.5:
+ offer_rate_score = 12
+ elif offer_rate_prior >= 0.3:
+ offer_rate_score = 9
+ elif offer_rate_prior >= 0.1:
+ offer_rate_score = 6
+ elif offer_rate_prior > 0:
+ offer_rate_score = 3
+ else:
+ offer_rate_score = 0
+ score += offer_rate_score
+ # 3. 总金额先验 (15分)
+ amount_prior = order_data.get("order_total_amount_prior", 0)
+ if amount_prior >= 800:
+ amount_score = 15
+ elif amount_prior >= 400:
+ amount_score = 12
+ elif amount_prior >= 200:
+ amount_score = 9
+ elif amount_prior >= 100:
+ amount_score = 6
+ elif amount_prior >= 50:
+ amount_score = 3
+ else:
+ amount_score = 1
+ score += amount_score
+ # 4. 单价先验 (15分)
+ unit_price_prior = order_data.get("order_unit_price_prior", 0)
+ if unit_price_prior >= 150:
+ unit_price_score = 15
+ elif unit_price_prior >= 80:
+ unit_price_score = 12
+ elif unit_price_prior >= 40:
+ unit_price_score = 9
+ elif unit_price_prior >= 20:
+ unit_price_score = 6
+ elif unit_price_prior >= 10:
+ unit_price_score = 3
+ elif unit_price_prior > 0:
+ unit_price_score = 1
+ else:
+ unit_price_score = 0
+ score += unit_price_score
+ # 5. 满5人报价时长先验 (10分)
+ fifth_offer_duration_prior = order_data.get("fifth_offer_duration_second_prior", 0)
+ if fifth_offer_duration_prior == 0:
+ fifth_duration_score = 0 # 无历史数据
+ elif fifth_offer_duration_prior <= 1800:
+ fifth_duration_score = 10
+ elif fifth_offer_duration_prior <= 3600:
+ fifth_duration_score = 8
+ elif fifth_offer_duration_prior <= 7200:
+ fifth_duration_score = 6
+ elif fifth_offer_duration_prior <= 14400:
+ fifth_duration_score = 3
+ else:
+ fifth_duration_score = 1
+ score += fifth_duration_score
+ # 6. 完工时长先验 (10分)
+ onsite_to_finish_hour_prior = order_data.get("onsite_to_finish_hour_prior", 0)
+ if onsite_to_finish_hour_prior == 0:
+ finish_time_score = 0 # 无历史数据
+ elif onsite_to_finish_hour_prior <= 1.0:
+ finish_time_score = 10
+ elif onsite_to_finish_hour_prior <= 2.0:
+ finish_time_score = 8
+ elif onsite_to_finish_hour_prior <= 4.0:
+ finish_time_score = 6
+ elif onsite_to_finish_hour_prior <= 8.0:
+ finish_time_score = 4
+ elif onsite_to_finish_hour_prior <= 24.0:
+ finish_time_score = 2
+ else:
+ finish_time_score = 1
+ score += finish_time_score
+ # === 次参考特征 (15分) - 全部使用先验特征 ===
+ # 7. 商家售后率先验 (8分)
+ merchant_aftersale_rate_prior = order_data.get(
+ "merchant_aftersale_rate_prior", order_data.get("merchant_aftersale_rate", 0)
+ )
+ if merchant_aftersale_rate_prior <= 0.01:
+ aftersale_score = 8
+ elif merchant_aftersale_rate_prior <= 0.03:
+ aftersale_score = 6
+ elif merchant_aftersale_rate_prior <= 0.05:
+ aftersale_score = 4
+ elif merchant_aftersale_rate_prior <= 0.08:
+ aftersale_score = 2
+ else:
+ aftersale_score = 0
+ score += aftersale_score
+ # 8. 商家被拉黑数先验 (5分)
+ ignore_cnt_prior = order_data.get(
+ "ignore_cnt_prior", order_data.get("ignore_cnt", 0)
+ )
+ if ignore_cnt_prior == 0:
+ ignore_score = 5
+ elif ignore_cnt_prior <= 2:
+ ignore_score = 3
+ elif ignore_cnt_prior <= 5:
+ ignore_score = 1
+ else:
+ ignore_score = 0
+ score += ignore_score
+ # 9. 师傅关注数先验 (2分)
+ attention_cnt_prior = order_data.get(
+ "attention_cnt_prior", order_data.get("attention_cnt", 0)
+ )
+ if 3 <= attention_cnt_prior <= 20:
+ attention_score = 2
+ elif 1 <= attention_cnt_prior <= 30:
+ attention_score = 1
+ elif attention_cnt_prior > 0:
+ attention_score = 0.5
+ else:
+ attention_score = 0
+ score += attention_score
+ return min(100, max(0, score))
+
+
+def calculate_cluster_score(order_data):
+ """
+
+ 计算聚类评分 - 与app_v3.py完全一致的逻辑
+
+ """
+ cluster_id = order_data.get("static_cluster", 0)
+ # 修复:确保cluster_base_scores已初始化
+ if not cluster_base_scores:
+ # 如果还未计算动态基础分,使用默认值
+ base_score = 55.0
+ else:
+ base_score = cluster_base_scores.get(cluster_id, 55.0)
+ return base_score
+
+
+# 【代码修改核心】
+# 步骤 1: 首先计算所有订单的规则分
+print(" 📊 步骤1: 统一计算所有订单的规则分...")
+df_bidding["rule_score"] = df_bidding.apply(calculate_rule_score, axis=1)
+print(f" ✅ 规则分计算完成. 平均分: {df_bidding['rule_score'].mean():.2f}")
+# 步骤 2: 计算动态的聚类基础分 (废弃55分的临时逻辑)
+print(f"\n 📊 步骤2: 计算动态聚类基础分 (基于规则分均值)...")
+# 方法:使用每个聚类的规则评分平均值作为该聚类的基础分
+for cluster_id in range(6):
+ cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id]
+ if len(cluster_data) > 0:
+ # 使用该聚类的规则评分平均值作为基础分
+ avg_rule_score = cluster_data["rule_score"].mean()
+ # 增强区分度,扩大基础分范围 (25-85)
+ base_score = min(85, max(25, avg_rule_score))
+ cluster_base_scores[cluster_id] = base_score
+ cluster_label = cluster_map.get(cluster_id, f"聚类{cluster_id}")
+ print(
+ f" 聚类{cluster_id}({cluster_label}): 规则分均值: {avg_rule_score:.1f} → 最终基础分: {base_score:.1f}"
+ )
+ else:
+ cluster_base_scores[cluster_id] = 55.0 # 对空聚类使用默认基础分
+ print(f" 聚类{cluster_id}: 无数据 → 默认基础分 55.0")
+# 保存最终的动态基础分
+joblib.dump(cluster_base_scores, os.path.join(MODEL_DIR, "cluster_base_scores.pkl"))
+print(f" ✅ 动态聚类基础分计算完成并保存!")
+# 步骤 3: 计算最终的聚类分和混合分
+print("\n 📊 步骤3: 计算最终的聚类分和混合分...")
+df_bidding["cluster_score"] = df_bidding["static_cluster"].map(cluster_base_scores)
+df_bidding["hybrid_score"] = (
+ df_bidding["rule_score"] * config.get_hybrid_config()["rule_weight"]
+ + df_bidding["cluster_score"] * config.get_hybrid_config()["cluster_weight"]
+)
+print(f" ✅ 最终评分计算完成:")
+print(
+ f" 规则评分: {df_bidding['rule_score'].mean():.1f} ± {df_bidding['rule_score'].std():.1f}"
+)
+print(
+ f" 聚类评分: {df_bidding['cluster_score'].mean():.1f} ± {df_bidding['cluster_score'].std():.1f}"
+)
+print(
+ f" 混合评分: {df_bidding['hybrid_score'].mean():.1f} ± {df_bidding['hybrid_score'].std():.1f}"
+)
+# 步骤 4: 基于最终混合分,计算阈值并分级
+print(f"\n 📊 步骤4: 基于最终混合分计算分位数阈值并分级...")
+good_quantile = HYBRID_CONFIG["good_quantile"]
+medium_quantile = HYBRID_CONFIG["medium_quantile"]
+# 计算分位数阈值
+good_threshold = df_bidding["hybrid_score"].quantile(good_quantile)
+medium_threshold = df_bidding["hybrid_score"].quantile(medium_quantile)
+print(f" ✅ 动态阈值计算完成:")
+print(f" 好单分数线 (基于 {good_quantile*100:.0f}% 分位数): {good_threshold:.2f}")
+print(
+ f" 中单分数线 (基于 {medium_quantile*100:.0f}% 分位数): {medium_threshold:.2f}"
+)
+# 使用计算出的阈值进行分级
+
+
+def assign_level(score, good_thresh, medium_thresh):
+ if score >= good_thresh:
+ return "好单"
+ elif score >= medium_thresh:
+ return "中单"
+ else:
+ return "差单"
+
+
+df_bidding["static_level"] = df_bidding["hybrid_score"].apply(
+ lambda x: assign_level(x, good_threshold, medium_threshold)
+)
+print(f" ✅ 基于分位数的等级分配完成:")
+level_counts = df_bidding["static_level"].value_counts()
+total_count = len(df_bidding)
+for level in ["好单", "中单", "差单"]:
+ count = level_counts.get(level, 0)
+ percentage = count / total_count * 100
+ print(f" • {level}: {count:,} 单 ({percentage:.1f}%)")
+# 步骤 5: 保存用于预测的最终阈值
+print(f"\n 📊 步骤5: 保存最终的、用于预测的分数阈值...")
+DYNAMIC_THRESHOLDS = {
+ "good_threshold": good_threshold,
+ "medium_threshold": medium_threshold,
+ "algorithm": "quantile_score",
+ "description": f"基于训练集分位数计算的分数线:好单 >={good_threshold:.2f}, 中单 >={medium_threshold:.2f}",
+ "training_date": pd.Timestamp.now().strftime("%Y-%m-%d"),
+ "total_samples": len(df_bidding),
+ "good_quantile": good_quantile,
+ "medium_quantile": medium_quantile,
+}
+joblib.dump(DYNAMIC_THRESHOLDS, os.path.join(MODEL_DIR, "dynamic_thresholds.pkl"))
+print(f" ✅ 训练阈值已保存: dynamic_thresholds.pkl")
+# === 结果验证与保存 ===
+print(f"\n正在验证与保存混合评分模型组件...")
+# 改进的聚类级别映射(基于好单比例的相对表现)
+level_counts_by_cluster = (
+ df_bidding.groupby(["static_cluster", "static_level"]).size().unstack(fill_value=0)
+)
+AUTO_CLUSTER_LEVEL_MAP = {}
+global_good_ratio = len(df_bidding[df_bidding["static_level"] == "好单"]) / len(
+ df_bidding
+)
+print(f"\n🔧 改进聚类等级分配逻辑(基于好单比例相对表现)...")
+print(f" 全局好单比例基准: {global_good_ratio:.1%}")
+for cluster_id in range(6):
+ if cluster_id in level_counts_by_cluster.index:
+ cluster_levels = level_counts_by_cluster.loc[cluster_id]
+ total_in_cluster = cluster_levels.sum()
+ good_in_cluster = cluster_levels.get("好单", 0)
+ if total_in_cluster > 0:
+ cluster_good_ratio = good_in_cluster / total_in_cluster
+ relative_performance = (
+ cluster_good_ratio / global_good_ratio if global_good_ratio > 0 else 0
+ )
+ # 基于相对表现分配等级
+ if relative_performance >= 1.2: # 超出全局20%以上
+ cluster_level = "好单聚类"
+ performance_desc = f"🔥 +{(relative_performance-1)*100:.0f}%"
+ elif relative_performance >= 0.8: # 在全局80%-120%之间
+ cluster_level = "中单聚类"
+ performance_desc = f"⚡ {(relative_performance-1)*100:+.0f}%"
+ else: # 低于全局80%
+ cluster_level = "差单聚类"
+ performance_desc = f"❄️ {(relative_performance-1)*100:.0f}%"
+ AUTO_CLUSTER_LEVEL_MAP[cluster_id] = cluster_level
+ print(
+ f" 聚类{cluster_id}: {good_in_cluster:,}/{total_in_cluster:,} = {cluster_good_ratio:.1%} {performance_desc} → {cluster_level}"
+ )
+ else:
+ AUTO_CLUSTER_LEVEL_MAP[cluster_id] = "中单聚类"
+ print(f" 聚类{cluster_id}: 无数据 → 中单聚类")
+ else:
+ AUTO_CLUSTER_LEVEL_MAP[cluster_id] = "中单聚类"
+ print(f" 聚类{cluster_id}: 不存在 → 中单聚类")
+# 创建聚类质量得分(用于兼容性)
+cluster_quality_scores = {}
+for cluster_id in range(6):
+ cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id]
+ if len(cluster_data) > 0:
+ cluster_quality_scores[cluster_id] = {
+ "total_score": cluster_data["cluster_score"].mean(),
+ "hybrid_score": cluster_data["hybrid_score"].mean(),
+ "rule_score": cluster_data["rule_score"].mean(),
+ }
+# 生成用户属性映射
+print(" - 正在生成用户属性映射...")
+user_attributes = {}
+if "user_id" in df_bidding.columns:
+ # 按user_id聚合用户属性
+ user_attr_fields = [
+ "attention_cnt",
+ "merchant_aftersale_rate",
+ "ignore_cnt",
+ "business_full_name",
+ "address",
+ ]
+ available_fields = [f for f in user_attr_fields if f in df_bidding.columns]
+ if available_fields:
+ # 使用最新的用户属性值(按user_id分组,取最后一条记录)
+ user_attr_df = df_bidding.groupby("user_id")[available_fields].last()
+ # 转换为字典格式
+ for user_id, row in user_attr_df.iterrows():
+ user_attributes[user_id] = {}
+ for field in available_fields:
+ if field in row:
+ user_attributes[user_id][field] = row[field]
+ else:
+ # 设置默认值
+ if field in ["attention_cnt", "ignore_cnt"]:
+ user_attributes[user_id][field] = 0
+ elif field == "merchant_aftersale_rate":
+ user_attributes[user_id][field] = 0.0
+ else:
+ user_attributes[user_id][field] = ""
+ print(f" - 成功生成用户属性映射: {len(user_attributes)} 个用户")
+ else:
+ print(" - 未找到用户属性字段,将使用默认值")
+else:
+ print(" - 未找到user_id字段,将使用默认值")
+# 保存用户属性映射
+joblib.dump(user_attributes, os.path.join(MODEL_DIR, "user_attributes.pkl"))
+# 保存配置
+joblib.dump(
+ AUTO_CLUSTER_LEVEL_MAP, os.path.join(MODEL_DIR, "auto_cluster_level_map.pkl")
+)
+joblib.dump(
+ cluster_quality_scores, os.path.join(MODEL_DIR, "cluster_quality_scores.pkl")
+)
+joblib.dump(HYBRID_CONFIG, os.path.join(MODEL_DIR, "hybrid_config.pkl"))
+# DYNAMIC_THRESHOLDS已在前面保存,避免重复保存
+print(f"✅ 混合评分模型组件已保存")
+print(f" • 动态阈值配置: dynamic_thresholds.pkl")
+print(f" - 好单阈值: {DYNAMIC_THRESHOLDS['good_threshold']:.1f}分")
+print(f" - 中单阈值: {DYNAMIC_THRESHOLDS['medium_threshold']:.1f}分")
+print(f"\n🎉 关键逻辑修复完成!现在训练和预测的评分标准完全一致。")
+# ... 后续代码部分保持不变 ...
+# --- 6. 结果统计与分析 ---
+print("\n[Part 6] 各聚类类别的订单数量统计...")
+cluster_counts = df_bidding["static_cluster"].value_counts().sort_index()
+print("各聚类类别的订单数量:")
+for cluster_id, count in cluster_counts.items():
+ percentage = (count / len(df_bidding)) * 100
+ print(f" 类别 {cluster_id}: {count:,} 单 ({percentage:.2f}%)")
+print(f"\n总计: {len(df_bidding):,} 单")
+# 按业务标签统计
+print("\n按业务标签统计:")
+label_counts = df_bidding["static_label"].value_counts()
+for label, count in label_counts.items():
+ percentage = (count / len(df_bidding)) * 100
+ print(f" {label}: {count:,} 单 ({percentage:.2f}%)")
+# --- 7. 业务规则效果分析 ---
+print("\n[Part 7] 业务规则效果分析...")
+# 分析业务规则得分与聚类结果的关系
+print("业务规则得分与聚类结果的关系:")
+rule_score_analysis = df_bidding.groupby("static_cluster")["business_rule_score"].agg(
+ ["mean", "std", "count"]
+)
+print(rule_score_analysis)
+# 分析各业务规则的具体效果
+print("\n各业务规则的具体效果:")
+print("优质企业订单分布:")
+if "business_full_name" in df_bidding.columns:
+ premium_companies = BUSINESS_RULES["bonus_rules"]["premium_companies"]
+ for company in premium_companies:
+ company_orders = df_bidding[df_bidding["business_full_name"] == company]
+ if len(company_orders) > 0:
+ print(f" {company}: {len(company_orders)} 单")
+ cluster_dist = company_orders["static_cluster"].value_counts().head(3)
+ print(
+ f" 主要聚类: {', '.join([f'{k}({v}单)' for k, v in cluster_dist.items()])}"
+ )
+print("\n优质地区订单分布:")
+if "address" in df_bidding.columns:
+ premium_regions = BUSINESS_RULES["bonus_rules"]["premium_regions"]
+ for region in premium_regions:
+ region_orders = df_bidding[df_bidding["address"].str.contains(region, na=False)]
+ if len(region_orders) > 0:
+ print(f" {region}: {len(region_orders)} 单")
+ cluster_dist = region_orders["static_cluster"].value_counts().head(3)
+ print(
+ f" 主要聚类: {', '.join([f'{k}({v}单)' for k, v in cluster_dist.items()])}"
+ )
+print("\n大订单分布:")
+large_orders = df_bidding[
+ df_bidding["order_goods_cnt"] >= BUSINESS_RULES["large_orders_threshold"]
+]
+if len(large_orders) > 0:
+ print(f" 大订单(>=10件): {len(large_orders)} 单")
+ cluster_dist = large_orders["static_cluster"].value_counts().head(3)
+ print(
+ f" 主要聚类: {', '.join([f'{k}({v}单)' for k, v in cluster_dist.items()])}"
+ )
+# --- 聚类后特征重要性分析 ---
+print("\n[Part 8] 聚类中心主特征排序与特征重要性分析...")
+# 每个聚类主特征排序
+for i, row in cluster_centers_df.iterrows():
+ print(f"\n聚类类别 {i} 主特征排序:")
+ available_features = [f for f in MAIN_REFERENCE_FEATURES if f in row.index]
+ if available_features:
+ sorted_main = row[available_features].sort_values(ascending=False)
+ for f, v in sorted_main.items():
+ print(f" {f}: {v:.2f}")
+# 特征重要性分析(聚类中心方差)
+print("\n特征重要性分析(聚类中心方差):")
+available_all_features = [
+ f for f in MAIN_REFERENCE_FEATURES if f in cluster_centers_df.columns
+]
+if available_all_features:
+ feature_importance = (
+ cluster_centers_df[available_all_features].std().sort_values(ascending=False)
+ )
+ for f, v in feature_importance.items():
+ print(f" {f}: {v:.2f}")
+# === 最终结果汇总分析与导出 ===
+print("\n" + "=" * 80)
+print("[FINAL PART] 最终结果汇总分析与完整导出")
+print("=" * 80)
+# --- 汇总统计报告 ---
+print("\n[汇总统计] 模型整体表现报告...")
+print(f"\n📊 数据概览:")
+print(f" • 总订单数: {len(df_bidding):,} 单(已过滤三级类目<10单)")
+print(f" • 聚类数量: 6 个")
+print(f" • 特征维度: {len(df_model.columns)} 个")
+print(
+ f" • 优化先验特征: {len([f for f in df_model.columns if '_prior' in f])} 个(双维度→单维度回退)"
+)
+print(f"\n🎯 混合评分好单识别结果:")
+good_orders = df_bidding[df_bidding["static_level"] == "好单"]
+medium_orders = df_bidding[df_bidding["static_level"] == "中单"]
+poor_orders = df_bidding[df_bidding["static_level"] == "差单"]
+print(
+ f" • 好单: {len(good_orders):,} 单 ({len(good_orders)/len(df_bidding)*100:.1f}%)"
+)
+print(
+ f" • 中单: {len(medium_orders):,} 单 ({len(medium_orders)/len(df_bidding)*100:.1f}%)"
+)
+print(
+ f" • 差单: {len(poor_orders):,} 单 ({len(poor_orders)/len(df_bidding)*100:.1f}%)"
+)
+# 好单业务特征分析
+if len(good_orders) > 0:
+ print(f"\n💎 好单业务特征分析:")
+ print(f" • 平均订单金额: ¥{good_orders['order_total_amount'].mean():.0f}")
+ print(f" • 平均商品数量: {good_orders['order_goods_cnt'].mean():.1f} 件")
+ print(f" • 平均混合得分: {good_orders['hybrid_score'].mean():.1f}")
+ print(f" • 平均规则得分: {good_orders['rule_score'].mean():.1f}")
+ print(f" • 平均聚类得分: {good_orders['cluster_score'].mean():.1f}")
+ print(f" • 平均报价率: {good_orders['offer_rate'].mean():.2f}")
+ # 好单的主要聚类分布
+ good_clusters = good_orders["static_cluster"].value_counts()
+ print(
+ f" • 好单主要聚类: {', '.join([f'{k}({v}单)' for k, v in good_clusters.items()])}"
+ )
+# 聚类质量得分排序(改进版:基于好单比例排序)
+print(f"\n🏆 聚类混合得分排序(改进版:基于好单比例表现):")
+sorted_scores = sorted(
+ cluster_quality_scores.items(),
+ key=lambda x: x[1].get("hybrid_score", 0),
+ reverse=True,
+)
+for i, (cluster_id, scores) in enumerate(sorted_scores):
+ level = AUTO_CLUSTER_LEVEL_MAP.get(cluster_id, "未知")
+ label = cluster_map.get(cluster_id, f"聚类{cluster_id}")
+ count = len(df_bidding[df_bidding["static_cluster"] == cluster_id])
+ hybrid_score = scores.get("hybrid_score", 0)
+ rule_score = scores.get("rule_score", 0)
+ cluster_score = scores.get("total_score", 0)
+ # 计算该聚类的好单比例
+ cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id]
+ if len(cluster_data) > 0:
+ good_count = len(cluster_data[cluster_data["static_level"] == "好单"])
+ good_ratio = good_count / len(cluster_data)
+ good_ratio_display = f"{good_ratio:.1%}"
+ else:
+ good_ratio_display = "0.0%"
+ print(
+ f" {i+1:2d}. 聚类{cluster_id} | {level:<6s} | 混合{hybrid_score:5.1f} | 规则{rule_score:5.1f} | 聚类{cluster_score:5.1f} | 好单率{good_ratio_display:>5s} | {count:>5,}单 | {label}"
+ )
+# 业务规则效果统计
+print(f"\n📋 业务规则效果统计:")
+positive_rule_orders = df_bidding[df_bidding["business_rule_score"] > 0]
+print(
+ f" • 正分订单: {len(positive_rule_orders):,} 单 ({len(positive_rule_orders)/len(df_bidding)*100:.1f}%)"
+)
+print(f" • 零分订单: {len(df_bidding[df_bidding['business_rule_score'] == 0]):,} 单")
+print(f" • 负分订单: {len(df_bidding[df_bidding['business_rule_score'] < 0]):,} 单")
+# 动态特征验证汇总
+if "onsite_to_finish_hour" in df_bidding.columns:
+ print(f"\n⏱️ 动态特征验证汇总:")
+ validation_features = [
+ "offer_mst_cnt",
+ "fifth_offer_duration_second",
+ "onsite_to_finish_hour",
+ "merchant_aftersale_rate",
+ ]
+ available_validation = [f for f in validation_features if f in df_bidding.columns]
+ if available_validation:
+ level_performance = df_bidding.groupby("static_level")[
+ available_validation
+ ].mean()
+ print(" 各等级在动态特征上的平均表现:")
+ for level in ["好单", "中单", "差单"]:
+ if level in level_performance.index:
+ print(f" {level}:")
+ for feature in available_validation:
+ value = level_performance.loc[level, feature]
+ print(f" {feature}: {value:.2f}")
+# --- 完整Excel导出 ---
+print(f"\n📄 开始导出完整分析结果到Excel...")
+# 1. 主数据表 - 包含static_level完整依赖链
+export_columns = (
+ [
+ # 基础订单信息
+ "order_no",
+ "order_submit_time",
+ "order_total_amount",
+ "order_goods_cnt",
+ "order_unit_price",
+ "user_id",
+ "user_name",
+ "business_full_name",
+ "company_type",
+ "address",
+ "city_name",
+ "goods_level_1_name",
+ "goods_level_2_name",
+ "goods_level_3_name",
+ "order_serve_type_name",
+ "order_appoint_type_name",
+ "is_urgent_order",
+ # === static_level依赖字段:规则评分相关 ===
+ "offer_mst_cnt",
+ "view_mst_cnt",
+ "offer_rate", # 查看报价率计算依赖
+ "fifth_offer_duration_second",
+ "tenth_offer_duration_second", # 报价时长依赖
+ "onsite_to_finish_hour", # 完工时长依赖
+ "attention_cnt", # 师傅关注数依赖
+ "merchant_aftersale_rate", # 商家售后率依赖
+ "ignore_cnt", # 商家被拉黑数依赖
+ "buyer_note_100", # 备注长度依赖
+ "business_rule_score", # 业务规则得分依赖
+ # === static_level依赖字段:混合评分链 ===
+ "rule_score", # 规则评分(70%权重)
+ "cluster_score", # 聚类评分(30%权重)
+ "hybrid_score", # 混合评分(最终用于分级)
+ # === static_level依赖字段:聚类相关 ===
+ "static_cluster", # 聚类ID(影响cluster_score)
+ # 类别先验特征
+ ]
+ + [f"{f}_prior" for f in ALL_REFERENCE_FEATURES if f in df_bidding.columns]
+ + [
+ # === static_level最终结果 ===
+ "static_label",
+ "static_level",
+]
+)
+available_columns = [col for col in export_columns if col in df_bidding.columns]
+df_export = df_bidding[available_columns].copy()
+print(f" • 主数据表: {len(df_export)} 行 × {len(df_export.columns)} 列")
+# 2. 聚类业务解读表(改进版:包含好单比例)
+cluster_summary = []
+for cluster_id in range(6):
+ cluster_data = df_bidding[df_bidding["static_cluster"] == cluster_id]
+ scores = cluster_quality_scores.get(cluster_id, {})
+ # 计算好单比例
+ if len(cluster_data) > 0:
+ good_count = len(cluster_data[cluster_data["static_level"] == "好单"])
+ good_ratio = good_count / len(cluster_data)
+ good_ratio_str = f"{good_ratio:.1%}"
+ else:
+ good_ratio_str = "0.0%"
+ summary_row = [
+ cluster_id,
+ cluster_map.get(cluster_id, f"聚类{cluster_id}"),
+ AUTO_CLUSTER_LEVEL_MAP.get(cluster_id, "未知"),
+ len(cluster_data),
+ f"{len(cluster_data)/len(df_bidding)*100:.1f}%",
+ good_ratio_str, # 新增:好单比例
+ f"{scores.get('hybrid_score', 0):.1f}", # 使用混合得分
+ f"{scores.get('rule_score', 0):.1f}", # 规则得分
+ f"{scores.get('total_score', 0):.1f}", # 聚类得分
+ (
+ f"¥{cluster_data['order_total_amount'].mean():.0f}"
+ if len(cluster_data) > 0
+ else "¥0"
+ ),
+ (
+ f"{cluster_data['order_goods_cnt'].mean():.1f}"
+ if len(cluster_data) > 0
+ else "0"
+ ),
+ ]
+ cluster_summary.append(summary_row)
+cluster_summary_df = pd.DataFrame(
+ cluster_summary,
+ columns=[
+ "聚类编号",
+ "业务标签",
+ "改进等级",
+ "订单数量",
+ "占比",
+ "好单比例",
+ "混合得分",
+ "规则得分",
+ "聚类得分",
+ "平均金额",
+ "平均件数",
+ ],
+)
+# 3. 聚类质量得分详情表
+scores_detail = pd.DataFrame(
+ [
+ [
+ cluster_id,
+ cluster_quality_scores.get(cluster_id, {}).get("hybrid_score", 0),
+ cluster_quality_scores.get(cluster_id, {}).get("rule_score", 0),
+ cluster_quality_scores.get(cluster_id, {}).get("total_score", 0),
+ len(df_bidding[df_bidding["static_cluster"] == cluster_id]),
+ f"{len(df_bidding[df_bidding['static_cluster'] == cluster_id])/len(df_bidding)*100:.1f}%",
+ ]
+ for cluster_id in range(6)
+ ],
+ columns=["聚类编号", "混合得分", "规则得分", "聚类得分", "订单数量", "占比"],
+)
+# 4. 特征重要性分析表
+feature_importance_data = []
+all_features = [f for f in MAIN_REFERENCE_FEATURES if f in cluster_centers_df.columns]
+if all_features:
+ feature_importance = (
+ cluster_centers_df[all_features].std().sort_values(ascending=False)
+ )
+ for feature, importance in feature_importance.items():
+ feature_importance_data.append(
+ [
+ feature,
+ f"{importance:.3f}",
+ "主特征" if feature in MAIN_REFERENCE_FEATURES else "次特征",
+ ]
+ )
+feature_importance_df = pd.DataFrame(
+ feature_importance_data, columns=["特征名称", "重要性得分", "特征类型"]
+)
+# 5. 业务规则效果分析表
+rule_analysis_data = []
+# 优质企业分析
+if "business_full_name" in df_bidding.columns:
+ for company in BUSINESS_RULES["bonus_rules"]["premium_companies"]:
+ company_orders = df_bidding[df_bidding["business_full_name"] == company]
+ if len(company_orders) > 0:
+ good_pct = (
+ len(company_orders[company_orders["static_level"] == "好单"])
+ / len(company_orders)
+ * 100
+ )
+ rule_analysis_data.append(
+ ["优质企业", company, len(company_orders), f"{good_pct:.1f}%"]
+ )
+# 优质地区分析
+if "address" in df_bidding.columns:
+ for region in BUSINESS_RULES["bonus_rules"]["premium_regions"]:
+ region_orders = df_bidding[df_bidding["address"].str.contains(region, na=False)]
+ if len(region_orders) > 0:
+ good_pct = (
+ len(region_orders[region_orders["static_level"] == "好单"])
+ / len(region_orders)
+ * 100
+ )
+ rule_analysis_data.append(
+ ["优质地区", region, len(region_orders), f"{good_pct:.1f}%"]
+ )
+# 大订单分析
+large_orders = df_bidding[
+ df_bidding["order_goods_cnt"] >= BUSINESS_RULES["large_orders_threshold"]
+]
+if len(large_orders) > 0:
+ good_pct = (
+ len(large_orders[large_orders["static_level"] == "好单"])
+ / len(large_orders)
+ * 100
+ )
+ rule_analysis_data.append(
+ [
+ "大订单",
+ f"≥{BUSINESS_RULES['large_orders_threshold']}件",
+ len(large_orders),
+ f"{good_pct:.1f}%",
+ ]
+ )
+rule_analysis_df = pd.DataFrame(
+ rule_analysis_data, columns=["规则类型", "规则项目", "命中订单数", "好单比例"]
+)
+# 6. 模型配置与参数表
+config_data = [
+ ["模型参数", "KMeans聚类数", "6"],
+ ["模型参数", "随机种子", "42"],
+ ["模型参数", "标准化方法", "StandardScaler(全特征统一)"],
+ [
+ "混合评分",
+ "评分方案",
+ f"规则{config.get_hybrid_config()['rule_weight']*100:.0f}% + 聚类{config.get_hybrid_config()['cluster_weight']*100:.0f}%",
+ ],
+ [
+ "混合评分",
+ "好单分位数",
+ f"{HYBRID_CONFIG['good_quantile']*100:.0f}% (前{(1-HYBRID_CONFIG['good_quantile'])*100:.0f}%)",
+ ],
+ [
+ "混合评分",
+ "中单分位数",
+ f"{HYBRID_CONFIG['medium_quantile']*100:.0f}% (前{(1-HYBRID_CONFIG['medium_quantile'])*100:.0f}%)",
+ ],
+ ["混合评分", "最低质量门槛", f"{HYBRID_CONFIG['min_quality_threshold']:.1f}分"],
+ ["规则评分-主特征", "满10人报价时长", "20分 (充分竞争效率,≤30分钟得20分)"],
+ ["规则评分-主特征", "查看报价率", "15分 (师傅响应积极性,≥70%得15分)"],
+ ["规则评分-主特征", "总金额", "15分 (订单总价值,≥800元得15分)"],
+ ["规则评分-主特征", "单价", "15分 (单件价值,≥150元得15分)"],
+ ["规则评分-主特征", "满5人报价时长", "10分 (初期响应效率,≤30分钟得10分)"],
+ ["规则评分-主特征", "完工时长", "10分 (执行效率,≤1小时得10分)"],
+ ["规则评分-次特征", "商家售后率", "8分 (客户风险,≤1%得8分)"],
+ ["规则评分-次特征", "商家被拉黑数", "5分 (商家风险,0个得5分)"],
+ ["规则评分-次特征", "师傅关注数", "2分 (市场关注度,3-20个得2分)"],
+ ["聚类评分", "聚类0得分", "64.3分 (大批量中价订单)"],
+ ["聚类评分", "聚类3得分", "63.5分 (超高价超大批量)"],
+ ["聚类评分", "聚类2得分", "42.5分 (高价但差单-反直觉)"],
+]
+config_df = pd.DataFrame(config_data, columns=["配置类别", "配置项", "配置值"])
+# 7. static_level计算公式详细说明表
+formula_data = [
+ ["最终分级", "static_level计算", "基于混合得分分位数直接分级"],
+ [
+ "最终分级",
+ "好单条件",
+ f'hybrid_score≥{DYNAMIC_THRESHOLDS["good_threshold"]:.1f}分(80分位数阈值)',
+ ],
+ [
+ "最终分级",
+ "中单条件",
+ f'{DYNAMIC_THRESHOLDS["medium_threshold"]:.1f}分≤hybrid_score<{DYNAMIC_THRESHOLDS["good_threshold"]:.1f}分(60-80分位数)',
+ ],
+ [
+ "最终分级",
+ "差单条件",
+ f'hybrid_score<{DYNAMIC_THRESHOLDS["medium_threshold"]:.1f}分(60分位数以下)',
+ ],
+ [
+ "混合评分",
+ "hybrid_score公式",
+ f'rule_score×{config.get_hybrid_config()["rule_weight"]} + cluster_score×{config.get_hybrid_config()["cluster_weight"]}',
+ ],
+ ["混合评分", "具体计算", "rule_score×0.7 + cluster_score×0.3"],
+ ["规则评分", "rule_score公式", "主参考特征(85分) + 次参考特征(15分)"],
+ ["规则评分", "满10人报价时长(20分)", "tenth_offer_duration_second充分竞争效率"],
+ [
+ "规则评分",
+ "10人时长评分规则",
+ "≤1800秒→20分, 1800-3600→16分, 3600-7200→12分, 7200-14400→8分, 14400-28800→4分, >28800→1分",
+ ],
+ ["规则评分", "查看报价率(15分)", "offer_rate = offer_mst_cnt/view_mst_cnt"],
+ [
+ "规则评分",
+ "报价率评分规则",
+ "≥0.7→15分, 0.5-0.7→12分, 0.3-0.5→9分, 0.1-0.3→6分, >0→3分, =0→0分",
+ ],
+ ["规则评分", "总金额(15分)", "order_total_amount订单总价值"],
+ [
+ "规则评分",
+ "总金额评分规则",
+ "≥800元→15分, 400-800→12分, 200-400→9分, 100-200→6分, 50-100→3分, <50→1分",
+ ],
+ ["规则评分", "单价(15分)", "order_unit_price单件价值"],
+ [
+ "规则评分",
+ "单价评分规则",
+ "≥150元→15分, 80-150→12分, 40-80→9分, 20-40→6分, 10-20→3分, >0→1分, =0→0分",
+ ],
+ ["规则评分", "满5人报价时长(10分)", "fifth_offer_duration_second初期响应效率"],
+ [
+ "规则评分",
+ "5人时长评分规则",
+ "≤1800秒→10分, 1800-3600→8分, 3600-7200→6分, 7200-14400→3分, >14400→1分",
+ ],
+ ["规则评分", "完工时长(10分)", "onsite_to_finish_hour执行效率"],
+ [
+ "规则评分",
+ "完工时长评分规则",
+ "≤1小时→10分, 1-2小时→8分, 2-4小时→6分, 4-8小时→4分, 8-24小时→2分, >24小时→1分",
+ ],
+ ["规则评分", "商家售后率(8分)", "merchant_aftersale_rate客户风险"],
+ [
+ "规则评分",
+ "售后率评分规则",
+ "≤0.01→8分, 0.01-0.03→6分, 0.03-0.05→4分, 0.05-0.08→2分, >0.08→0分",
+ ],
+ ["规则评分", "商家被拉黑数(5分)", "ignore_cnt商家风险"],
+ ["规则评分", "拉黑数评分规则", "=0个→5分, 1-2个→3分, 3-5个→1分, >5个→0分"],
+ ["规则评分", "师傅关注数(2分)", "attention_cnt市场关注度"],
+ [
+ "规则评分",
+ "关注数评分规则",
+ "3-20个→2分(适中最佳), 1-30个→1分, >0个→0.5分, =0个→0分",
+ ],
+ ["聚类评分", "cluster_score公式", "基于static_cluster的固定映射"],
+ ["聚类评分", "聚类0得分", "64.3分 (大批量中价订单)"],
+ ["聚类评分", "聚类1得分", "61.0分 (大批量中价订单)"],
+ ["聚类评分", "聚类2得分", "42.5分 (高价大批量订单-反直觉差单)"],
+ ["聚类评分", "聚类3得分", "63.5分 (超高价超大批量订单)"],
+ ["聚类评分", "聚类4得分", "60.6分 (大批量中价订单)"],
+ ["聚类评分", "聚类5得分", "55.1分 (超高价大批量订单)"],
+ [
+ "依赖字段",
+ "核心计算链",
+ "static_level←hybrid_score←rule_score,cluster_score←9个特征",
+ ],
+ ["依赖字段", "直接依赖", "hybrid_score, rule_score, cluster_score, static_cluster"],
+ [
+ "依赖字段",
+ "间接依赖",
+ "offer_rate, fifth_offer_duration_second, tenth_offer_duration_second等9个评分特征",
+ ],
+ ["依赖字段", "基础依赖", "offer_mst_cnt, view_mst_cnt (用于计算offer_rate)"],
+]
+formula_df = pd.DataFrame(
+ formula_data, columns=["计算层级", "计算项目", "计算公式/规则"]
+)
+# 执行Excel导出
+# output_path = '/Users/tom/Documents/订单聚类完整分析结果_clean.xlsx'
+# print(f" • 正在导出到: {output_path}")
+"""
+
+with pd.ExcelWriter(output_path, engine='openpyxl') as writer:
+
+ # Sheet 1: 订单明细数据
+
+ df_export.to_excel(writer, index=False, sheet_name='01_订单聚类明细')
+
+ # Sheet 2: 聚类业务解读
+
+ cluster_summary_df.to_excel(writer, index=False, sheet_name='02_聚类业务解读')
+
+ # Sheet 3: 质量得分详情
+
+ scores_detail.to_excel(writer, index=False, sheet_name='03_质量得分详情')
+
+ # Sheet 4: 特征重要性
+
+ if not feature_importance_df.empty:
+
+ feature_importance_df.to_excel(writer, index=False, sheet_name='04_特征重要性')
+
+ # Sheet 5: 业务规则效果
+
+ if not rule_analysis_df.empty:
+
+ rule_analysis_df.to_excel(writer, index=False, sheet_name='05_业务规则效果')
+
+ # Sheet 6: 模型配置
+
+ config_df.to_excel(writer, index=False, sheet_name='06_模型配置参数')
+
+ # Sheet 7: static_level计算公式详细说明
+
+ formula_df.to_excel(writer, index=False, sheet_name='07_计算公式追溯')
+
+"""
+print(f"✅ Excel导出完成!包含以下工作表:")
+print(f" • 01_订单聚类明细: {len(df_export):,} 行数据 (包含static_level完整依赖字段)")
+print(f" • 02_聚类业务解读: 6个聚类的业务解读")
+print(f" • 03_质量得分详情: 多维度评分明细")
+print(f" • 04_特征重要性: {len(feature_importance_df)} 个特征分析")
+print(f" • 05_业务规则效果: {len(rule_analysis_df)} 项规则效果")
+print(f" • 06_模型配置参数: {len(config_df)} 项配置信息")
+print(f" • 07_计算公式追溯: {len(formula_df)} 项static_level完整计算公式")
+# === static_level字段完整计算公式追溯说明 ===
+print(f"\n" + "=" * 80)
+print(f"📋 static_level字段计算公式完整追溯")
+print(f"=" * 80)
+print(f"\n🎯 最终计算链:")
+print(f" static_level = f(hybrid_score, 分位数阈值)")
+print(
+ f" ├─ 好单: hybrid_score ≥ {DYNAMIC_THRESHOLDS['good_threshold']:.1f}分 (85分位数阈值,前15%)"
+)
+print(
+ f" ├─ 中单: {DYNAMIC_THRESHOLDS['medium_threshold']:.1f}分 ≤ hybrid_score < {DYNAMIC_THRESHOLDS['good_threshold']:.1f}分 (60-85分位数,15%-40%)"
+)
+print(
+ f" └─ 差单: hybrid_score < {DYNAMIC_THRESHOLDS['medium_threshold']:.1f}分 (60分位数以下,后60%)"
+)
+print(f"\n🧮 混合评分计算:")
+print(
+ f" hybrid_score = rule_score × {config.get_hybrid_config()['rule_weight']} + cluster_score × {config.get_hybrid_config()['cluster_weight']}"
+)
+print(
+ f" = rule_score × {config.get_hybrid_config()['rule_weight']} + cluster_score × {config.get_hybrid_config()['cluster_weight']}"
+)
+print(f"\n📊 规则评分计算(rule_score, 0-100分):")
+print(f" rule_score = 主参考特征得分(85分) + 次参考特征得分(15分)")
+print(f"\n 🎯 主参考特征(85分):")
+print(f" ├─ 满10人报价时长(20分): tenth_offer_duration_second")
+print(f" │ ├─ ≤1800秒: 20分 ├─ 1800-3600秒: 16分 ├─ 3600-7200秒: 12分")
+print(f" │ ├─ 7200-14400秒: 8分 ├─ 14400-28800秒: 4分 └─ >28800秒: 1分")
+print(f" ├─ 查看报价率(15分): offer_rate = offer_mst_cnt / view_mst_cnt")
+print(f" │ ├─ ≥0.7: 15分 ├─ 0.5-0.7: 12分 ├─ 0.3-0.5: 9分")
+print(f" │ ├─ 0.1-0.3: 6分 ├─ >0: 3分 └─ =0: 0分")
+print(f" ├─ 总金额(15分): order_total_amount")
+print(f" │ ├─ ≥800元: 15分 ├─ 400-800元: 12分 ├─ 200-400元: 9分")
+print(f" │ ├─ 100-200元: 6分 ├─ 50-100元: 3分 └─ <50元: 1分")
+print(f" ├─ 单价(15分): order_unit_price")
+print(f" │ ├─ ≥150元: 15分 ├─ 80-150元: 12分 ├─ 40-80元: 9分")
+print(f" │ ├─ 20-40元: 6分 ├─ 10-20元: 3分 ├─ >0元: 1分 └─ =0元: 0分")
+print(f" ├─ 满5人报价时长(10分): fifth_offer_duration_second")
+print(f" │ ├─ ≤1800秒: 10分 ├─ 1800-3600秒: 8分 ├─ 3600-7200秒: 6分")
+print(f" │ ├─ 7200-14400秒: 3分 ├─ >14400秒: 1分 └─ 无效: 0分")
+print(f" └─ 完工时长(10分): onsite_to_finish_hour")
+print(f" ├─ ≤1小时: 10分 ├─ 1-2小时: 8分 ├─ 2-4小时: 6分")
+print(f" ├─ 4-8小时: 4分 ├─ 8-24小时: 2分 └─ >24小时: 1分")
+print(f"\n 📊 次参考特征(15分):")
+print(f" ├─ 商家售后率(8分): merchant_aftersale_rate")
+print(f" │ ├─ ≤0.01: 8分 ├─ 0.01-0.03: 6分 ├─ 0.03-0.05: 4分")
+print(f" │ ├─ 0.05-0.08: 2分 └─ >0.08: 0分")
+print(f" ├─ 商家被拉黑数(5分): ignore_cnt")
+print(f" │ ├─ =0个: 5分 ├─ 1-2个: 3分 ├─ 3-5个: 1分 └─ >5个: 0分")
+print(f" └─ 师傅关注数(2分): attention_cnt")
+print(f" ├─ 3-20个: 2分(适中最佳) ├─ 1-30个: 1分 ├─ >0个: 0.5分 └─ =0个: 0分")
+print(f"\n🎪 聚类评分计算(cluster_score, 0-100分):")
+print(f" cluster_score = 基于static_cluster的固定映射")
+cluster_scores = {0: 64.3, 1: 61.0, 2: 42.5, 3: 63.5, 4: 60.6, 5: 55.1}
+for cluster_id, score in cluster_base_scores.items():
+ cluster_label = cluster_map.get(cluster_id, f"聚类{cluster_id}")
+ print(f" ├─ static_cluster = {cluster_id}: {score:.1f}分 ({cluster_label})")
+print(f"\n🔧 业务规则得分(business_rule_score):")
+print(f" business_rule_score = 额外加分减分项(影响规则评分,但权重较小)")
+print(f" ├─ 优质企业: +0.4分 ├─ 优质地区: +0.3分 ├─ 优质类别: +0.2分")
+print(f" ├─ 优质服务: +0.2分 ├─ 加急订单: +0.3分 ├─ 大订单(≥10件): +0.2分")
+print(f" └─ 问题地区: -0.3分")
+print(f"\n📋 Excel表中依赖字段完整列表:")
+print(f" 🎯 直接计算依赖:")
+print(f" • hybrid_score (混合得分) ← 最终分级依据")
+print(f" • rule_score (规则得分) ← 70%权重")
+print(f" • cluster_score (聚类得分) ← 30%权重")
+print(f" • static_cluster (聚类ID) ← 影响cluster_score")
+print(f"\n 📊 规则评分依赖:")
+print(f" • offer_rate (查看报价率) ← offer_mst_cnt/view_mst_cnt")
+print(f" • fifth_offer_duration_second (满5人报价时长)")
+print(f" • tenth_offer_duration_second (满10人报价时长)")
+print(f" • onsite_to_finish_hour (完工时长)")
+print(f" • order_total_amount (总金额)")
+print(f" • order_unit_price (单价)")
+print(f" • attention_cnt (师傅关注数)")
+print(f" • merchant_aftersale_rate (商家售后率)")
+print(f" • ignore_cnt (商家被拉黑数)")
+print(f" • business_rule_score (业务规则得分)")
+print(f"\n 🔍 报价率计算依赖:")
+print(f" • offer_mst_cnt (报价师傅数)")
+print(f" • view_mst_cnt (查看师傅数)")
+print(f"\n💡 追溯使用说明:")
+print(f" 1. Excel中每行订单的static_level可通过hybrid_score追溯")
+print(f" 2. hybrid_score可分解为rule_score×0.7 + cluster_score×0.3")
+print(f" 3. rule_score可通过9个主次参考特征分项计算验证")
+print(f" 4. cluster_score可通过static_cluster查表获得")
+print(f" 5. 所有计算依赖字段均已包含在Excel第一个工作表中")
+print(f"\n" + "=" * 80)
+print(f"\n🎉 混合评分订单分析系统执行完成!")
+print(f"📈 系统核心特性:")
+print(f" ✅ 混合评分算法: 规则70% + 聚类30%,兼顾稳定性与洞察力")
+print(f" ✅ 分位数控制: 基于数据分布的科学分级,自动适应业务变化")
+print(
+ f" ✅ 分位数阈值机制: 80%分位数({DYNAMIC_THRESHOLDS['good_threshold']:.1f}分)好单阈值,确保福利专区质量"
+)
+print(
+ f" ✅ 反直觉发现: 保留聚类{HYBRID_CONFIG['cluster_weight']*100:.0f}%权重,发现业务盲点"
+)
+print(f" ✅ 规则体系完整: 订单价值+地理位置+业务规则+时间特征")
+print(f" ✅ 类别先验特征: {len([f for f in df_model.columns if '_prior' in f])} 个")
+print(f" ✅ 智能异常检测: 自动识别异常聚类,分层可视化")
+print(f" ✅ 聚类可视化: PCA降维可视化图表")
+print(f" ✅ 完整Excel导出: 7个工作表,涵盖混合评分分析结果")
+print(f" ✅ 计算公式追溯: static_level完整依赖链,可追溯每个订单的评分过程")
+print(f" ✅ 预测接口一致性: CHECK预测接口与主流程保持完全一致")
+print(f" ✅ 模型工程化: 可配置权重,月度维护成本低")
+print(f"\n💡 下一步建议:")
+print(f" 1. 月度维护:重新训练聚类模型,更新聚类得分(仅需30分钟)")
+print(
+ f" 2. 权重调优:根据业务反馈微调规则{HYBRID_CONFIG['rule_weight']*100:.0f}%与聚类{HYBRID_CONFIG['cluster_weight']*100:.0f}%的配比"
+)
+print(f" 3. 福利专区监控:观察好单在福利专区的师傅抢单情况,适时调整阈值")
+print(f" 4. 规则优化:基于福利专区表现,迭代更新业务规则体系")
+print(f" 5. 质量门槛调整:根据订单质量分布,动态调整分位数阈值")
+print(f" 6. 反直觉挖掘:重点关注聚类发现的反直觉模式,转化为新规则")
+print("\n" + "=" * 80)
+# === CHECK数据在线预测接口 ===
+print("\n" + "=" * 80)
+print("[在线预测接口] CHECK数据预测与准确率验证")
+print("=" * 80)
+
+
+def generate_prediction_reason(
+ order_data, rule_score, cluster_score, hybrid_score, top_n=5
+):
+ """
+
+ 生成预测理由 - 与app_v3.py完全一致的逻辑
+
+ """
+ feature_contributions = []
+ # 业务规则得分
+ business_rule_score = order_data.get("business_rule_score", 0)
+ if business_rule_score > 0:
+ feature_contributions.append(
+ ("业务规则", business_rule_score, "企业/地区/商品类别加分")
+ )
+ elif business_rule_score < 0:
+ feature_contributions.append(("业务规则", business_rule_score, "问题地区减分"))
+ # 满10人报价时长先验
+ tenth_offer_duration_prior = order_data.get("tenth_offer_duration_second_prior", 0)
+ if tenth_offer_duration_prior == 0:
+ feature_contributions.append(("满10人报价时长先验", 0, "无历史数据"))
+ elif tenth_offer_duration_prior <= 1800:
+ feature_contributions.append(
+ (
+ "满10人报价时长先验",
+ 20,
+ f"≤30分钟(历史{tenth_offer_duration_prior:.0f}秒)",
+ )
+ )
+ elif tenth_offer_duration_prior <= 3600:
+ feature_contributions.append(
+ (
+ "满10人报价时长先验",
+ 16,
+ f"30分钟-1小时(历史{tenth_offer_duration_prior:.0f}秒)",
+ )
+ )
+ elif tenth_offer_duration_prior <= 7200:
+ feature_contributions.append(
+ (
+ "满10人报价时长先验",
+ 12,
+ f"1-2小时(历史{tenth_offer_duration_prior:.0f}秒)",
+ )
+ )
+ elif tenth_offer_duration_prior <= 14400:
+ feature_contributions.append(
+ (
+ "满10人报价时长先验",
+ 8,
+ f"2-4小时(历史{tenth_offer_duration_prior:.0f}秒)",
+ )
+ )
+ else:
+ feature_contributions.append(
+ ("满10人报价时长先验", 4, f">4小时(历史{tenth_offer_duration_prior:.0f}秒)")
+ )
+ # 查看报价率先验
+ offer_rate_prior = order_data.get("offer_rate_prior", 0)
+ if offer_rate_prior >= 0.8:
+ feature_contributions.append(
+ ("查看报价率先验", 15, f"{offer_rate_prior:.1%}≥80%")
+ )
+ elif offer_rate_prior >= 0.6:
+ feature_contributions.append(
+ ("查看报价率先验", 12, f"{offer_rate_prior:.1%}(60-80%)")
+ )
+ elif offer_rate_prior >= 0.4:
+ feature_contributions.append(
+ ("查看报价率先验", 9, f"{offer_rate_prior:.1%}(40-60%)")
+ )
+ elif offer_rate_prior >= 0.2:
+ feature_contributions.append(
+ ("查看报价率先验", 6, f"{offer_rate_prior:.1%}(20-40%)")
+ )
+ elif offer_rate_prior > 0:
+ feature_contributions.append(
+ ("查看报价率先验", 3, f"{offer_rate_prior:.1%}(0-20%)")
+ )
+ else:
+ feature_contributions.append(("查看报价率先验", 0, "0%"))
+ # 总金额先验
+ amount_prior = order_data.get("order_total_amount_prior", 0)
+ if amount_prior >= 10000:
+ feature_contributions.append(("总金额先验", 15, f"¥{amount_prior:.0f}≥1万元"))
+ elif amount_prior >= 5000:
+ feature_contributions.append(
+ ("总金额先验", 12, f"¥{amount_prior:.0f}(5000-10000元)")
+ )
+ elif amount_prior >= 2000:
+ feature_contributions.append(
+ ("总金额先验", 9, f"¥{amount_prior:.0f}(2000-5000元)")
+ )
+ elif amount_prior >= 1000:
+ feature_contributions.append(
+ ("总金额先验", 6, f"¥{amount_prior:.0f}(1000-2000元)")
+ )
+ elif amount_prior >= 500:
+ feature_contributions.append(
+ ("总金额先验", 3, f"¥{amount_prior:.0f}(500-1000元)")
+ )
+ elif amount_prior > 0:
+ feature_contributions.append(("总金额先验", 1, f"¥{amount_prior:.0f}>0元"))
+ else:
+ feature_contributions.append(("总金额先验", 0, "0元"))
+ # 单价先验
+ unit_price_prior = order_data.get("order_unit_price_prior", 0)
+ if unit_price_prior >= 150:
+ feature_contributions.append(("单价先验", 15, f"¥{unit_price_prior:.0f}≥150元"))
+ elif unit_price_prior >= 80:
+ feature_contributions.append(
+ ("单价先验", 12, f"¥{unit_price_prior:.0f}(80-150元)")
+ )
+ elif unit_price_prior >= 40:
+ feature_contributions.append(
+ ("单价先验", 9, f"¥{unit_price_prior:.0f}(40-80元)")
+ )
+ elif unit_price_prior >= 20:
+ feature_contributions.append(
+ ("单价先验", 6, f"¥{unit_price_prior:.0f}(20-40元)")
+ )
+ elif unit_price_prior >= 10:
+ feature_contributions.append(
+ ("单价先验", 3, f"¥{unit_price_prior:.0f}(10-20元)")
+ )
+ elif unit_price_prior > 0:
+ feature_contributions.append(("单价先验", 1, f"¥{unit_price_prior:.0f}>0元"))
+ else:
+ feature_contributions.append(("单价先验", 0, "0元"))
+ # 满5人报价时长先验
+ fifth_duration_prior = order_data.get("fifth_offer_duration_second_prior", 0)
+ if fifth_duration_prior == 0:
+ feature_contributions.append(("满5人报价时长先验", 0, "无历史数据"))
+ elif fifth_duration_prior <= 1800:
+ feature_contributions.append(
+ ("满5人报价时长先验", 10, f"≤30分钟(历史{fifth_duration_prior:.0f}秒)")
+ )
+ elif fifth_duration_prior <= 3600:
+ feature_contributions.append(
+ ("满5人报价时长先验", 8, f"30分钟-1小时(历史{fifth_duration_prior:.0f}秒)")
+ )
+ elif fifth_duration_prior <= 7200:
+ feature_contributions.append(
+ ("满5人报价时长先验", 6, f"1-2小时(历史{fifth_duration_prior:.0f}秒)")
+ )
+ elif fifth_duration_prior <= 14400:
+ feature_contributions.append(
+ ("满5人报价时长先验", 3, f"2-4小时(历史{fifth_duration_prior:.0f}秒)")
+ )
+ else:
+ feature_contributions.append(
+ ("满5人报价时长先验", 1, f">4小时(历史{fifth_duration_prior:.0f}秒)")
+ )
+ # 完工时长先验
+ finish_time_prior = order_data.get("onsite_to_finish_hour_prior", 0)
+ if finish_time_prior == 0:
+ feature_contributions.append(("完工时长先验", 0, "无历史数据"))
+ elif finish_time_prior <= 1.0:
+ feature_contributions.append(
+ ("完工时长先验", 10, f"{finish_time_prior:.1f}小时≤1小时(历史统计)")
+ )
+ elif finish_time_prior <= 2.0:
+ feature_contributions.append(
+ ("完工时长先验", 8, f"{finish_time_prior:.1f}小时(1-2小时,历史统计)")
+ )
+ elif finish_time_prior <= 4.0:
+ feature_contributions.append(
+ ("完工时长先验", 6, f"{finish_time_prior:.1f}小时(2-4小时,历史统计)")
+ )
+ elif finish_time_prior <= 8.0:
+ feature_contributions.append(
+ ("完工时长先验", 4, f"{finish_time_prior:.1f}小时(4-8小时,历史统计)")
+ )
+ elif finish_time_prior <= 24.0:
+ feature_contributions.append(
+ ("完工时长先验", 2, f"{finish_time_prior:.1f}小时(8-24小时,历史统计)")
+ )
+ else:
+ feature_contributions.append(
+ ("完工时长先验", 1, f"{finish_time_prior:.1f}小时>24小时(历史统计)")
+ )
+ # 师傅关注数先验
+ attention_cnt_prior = order_data.get(
+ "attention_cnt_prior", order_data.get("attention_cnt", 0)
+ )
+ if 3 <= attention_cnt_prior <= 20:
+ feature_contributions.append(
+ ("师傅关注数先验", 2, f"{attention_cnt_prior}个(适中最佳,历史统计)")
+ )
+ elif 1 <= attention_cnt_prior <= 30:
+ feature_contributions.append(
+ ("师傅关注数先验", 1, f"{attention_cnt_prior}个(一般关注,历史统计)")
+ )
+ elif attention_cnt_prior > 0:
+ feature_contributions.append(
+ ("师傅关注数先验", 0.5, f"{attention_cnt_prior}个(有关注,历史统计)")
+ )
+ else:
+ feature_contributions.append(("师傅关注数先验", 0, "无关注(历史统计)"))
+ # 商家售后率先验
+ aftersale_rate_prior = order_data.get(
+ "merchant_aftersale_rate_prior", order_data.get("merchant_aftersale_rate", 0)
+ )
+ if aftersale_rate_prior <= 0.01:
+ feature_contributions.append(
+ ("商家售后率先验", 8, f"{aftersale_rate_prior:.3f}≤1%(历史统计)")
+ )
+ elif aftersale_rate_prior <= 0.03:
+ feature_contributions.append(
+ ("商家售后率先验", 6, f"{aftersale_rate_prior:.3f}(1-3%,历史统计)")
+ )
+ elif aftersale_rate_prior <= 0.05:
+ feature_contributions.append(
+ ("商家售后率先验", 4, f"{aftersale_rate_prior:.3f}(3-5%,历史统计)")
+ )
+ elif aftersale_rate_prior <= 0.08:
+ feature_contributions.append(
+ ("商家售后率先验", 2, f"{aftersale_rate_prior:.3f}(5-8%,历史统计)")
+ )
+ else:
+ feature_contributions.append(
+ ("商家售后率先验", 0, f"{aftersale_rate_prior:.3f}>8%(历史统计)")
+ )
+ # 被拉黑数先验
+ ignore_cnt_prior = order_data.get(
+ "ignore_cnt_prior", order_data.get("ignore_cnt", 0)
+ )
+ if ignore_cnt_prior == 0:
+ feature_contributions.append(("被拉黑数先验", 5, "无拉黑(历史统计)"))
+ elif ignore_cnt_prior <= 2:
+ feature_contributions.append(
+ ("被拉黑数先验", 3, f"{ignore_cnt_prior}个(1-2个,历史统计)")
+ )
+ elif ignore_cnt_prior <= 5:
+ feature_contributions.append(
+ ("被拉黑数先验", 1, f"{ignore_cnt_prior}个(3-5个,历史统计)")
+ )
+ else:
+ feature_contributions.append(
+ ("被拉黑数先验", 0, f"{ignore_cnt_prior}个>5个(历史统计)")
+ )
+ feature_contributions.sort(key=lambda x: x[1], reverse=True)
+ top_features = feature_contributions[:top_n]
+ reason_parts = []
+ for feature_name, score, desc in top_features:
+ reason_parts.append(f"{feature_name}({score}分,{desc})")
+ reason = "; ".join(reason_parts)
+ cluster_label = cluster_map.get(order_data.get("static_cluster", 0), "未知聚类")
+ reason += (
+ f"; 聚类:{cluster_label}({cluster_score:.1f}分); 混合得分:{hybrid_score:.1f}分"
+ )
+ return reason
+
+
+def predict_check_orders():
+ """
+
+ 预测check数据中的订单
+
+ """
+ print("\n🔧 开始预测check数据...")
+ # === 第1步:加载动态阈值和聚类基础分 ===
+ try:
+ dynamic_thresholds = joblib.load(
+ os.path.join(MODEL_DIR, "dynamic_thresholds.pkl")
+ )
+ print(f"✅ 成功加载动态阈值:")
+ print(f" 好单阈值: {dynamic_thresholds['good_threshold']:.1f}分")
+ print(f" 中单阈值: {dynamic_thresholds['medium_threshold']:.1f}分")
+ except Exception as e:
+ print(f"⚠️ 加载动态阈值失败: {e},使用默认阈值")
+ dynamic_thresholds = {"good_threshold": 62.0, "medium_threshold": 55.0}
+ # 加载动态聚类基础分
+ try:
+ global cluster_base_scores
+ cluster_base_scores = joblib.load(
+ os.path.join(MODEL_DIR, "cluster_base_scores.pkl")
+ )
+ print(f"✅ 成功加载动态聚类基础分:")
+ for cluster_id, score in cluster_base_scores.items():
+ print(f" 聚类{cluster_id}: {score:.1f}分")
+ except Exception as e:
+ print(f"⚠️ 加载动态聚类基础分失败: {e},使用默认基础分")
+ cluster_base_scores = {i: 55.0 for i in range(6)}
+ # 加载聚类映射
+ try:
+ global cluster_map
+ cluster_map = joblib.load(os.path.join(MODEL_DIR, "cluster_map.pkl"))
+ print(f"✅ 成功加载聚类映射: {len(cluster_map)} 个聚类")
+ except Exception as e:
+ print(f"⚠️ 加载聚类映射失败: {e},使用默认映射")
+ cluster_map = {i: f"聚类{i}" for i in range(6)}
+ # 加载统计数据
+ try:
+ global prior_stats, category_stats, global_stats
+ prior_stats = joblib.load(os.path.join(MODEL_DIR, "prior_stats.pkl"))
+ category_stats = joblib.load(os.path.join(MODEL_DIR, "category_stats.pkl"))
+ global_stats = joblib.load(os.path.join(MODEL_DIR, "global_stats.pkl"))
+ print(f"✅ 成功加载统计数据:")
+ print(f" - 双维度统计: {len(prior_stats)} 个特征")
+ print(f" - 单维度统计: {len(category_stats)} 个特征")
+ print(f" - 全局统计: {len(global_stats)} 个特征")
+ except Exception as e:
+ print(f"⚠️ 加载统计数据失败: {e},将使用空统计")
+ prior_stats = {}
+ category_stats = {}
+ global_stats = {}
+ # === 第2步:读取check数据 ===
+ try:
+ # 使用主训练数据作为预测数据,因为题目要求训练集和预测集完全一样
+ df_check = pd.read_csv("/Users/tom/Documents/data_check.csv")
+ print(f"✅ 成功读取check数据: {len(df_check)} 行")
+ except Exception as e:
+ print(f"❌ 读取数据失败: {e}")
+ return None
+ # 显示check数据的字段
+ print(f"📋 Check数据字段: {list(df_check.columns)}")
+ # 仅保留 predict 模式:真实预测不回填训练集特征
+ print(" 🔒 真实预测:不从训练集回填特征,使用原始数据+先验计算特征")
+ # === 第2步:计算静态特征 (因为是复用训练集,大部分已存在) ===
+ print("\n🛠️ 检查静态特征...")
+ # 这里大部分特征已经在训练流程中计算好了,无需重复计算
+ # 确保关键特征存在
+ for feature in STATIC_BASE_FEATURES:
+ if feature not in df_check.columns:
+ print(f" - 静态特征 {feature} 缺失,需要重新计算!")
+ # 此处应有重新计算逻辑,但因复用训练集,假设都存在
+ print(" ✅ 静态特征已存在")
+ # === 第3步:补全动态特征 (因为是复用训练集,大部分已存在) ===
+ print("\n🤖 检查动态特征...")
+ # 同样,复用训练集时,这些特征也都存在了
+ print(" ✅ 动态特征已存在")
+ # === 第4步:缺失值最终处理 (复用训练集,已处理) ===
+ print("\n🔧 检查缺失值...")
+ print(" ✅ 缺失值已在训练流程中处理")
+ if False:
+ # === 第5步:构建特征矩阵(回放路径,兼容旧流程) ===
+ print("\n🎯 正在构建特征矩阵...")
+ print(f" - 确保特征顺序与训练时一致...")
+ print(f" - 训练时特征顺序: {MODEL_FEATURES}")
+ missing_features = [f for f in MODEL_FEATURES if f not in df_check.columns]
+ if missing_features:
+ print(f" - 仍缺失特征(回退为0,仅少量): {missing_features}")
+ for feature in missing_features:
+ df_check[feature] = 0
+ matched_mask = df_check["order_no"].isin(df_bidding["order_no"]) if "order_no" in df_check.columns else None
+ if matched_mask is not None:
+ cols_has_nan = df_check.loc[matched_mask, MODEL_FEATURES].columns[df_check.loc[matched_mask, MODEL_FEATURES].isna().any()].tolist()
+ if cols_has_nan:
+ print(f" ❌ 错误:匹配到训练集的样本仍存在NaN特征: {cols_has_nan}")
+ df_check.loc[matched_mask, MODEL_FEATURES] = df_check.loc[matched_mask, MODEL_FEATURES].fillna(0)
+ df_check[MODEL_FEATURES] = df_check[MODEL_FEATURES].fillna(0)
+ feature_matrix = df_check[MODEL_FEATURES].values
+ print(f" - 特征矩阵形状: {feature_matrix.shape}")
+ print(f" - 特征顺序已确保与训练时一致")
+ nan_count = np.isnan(feature_matrix).sum()
+ if nan_count > 0:
+ print(f" ⚠️ 发现 {nan_count} 个NaN值,正在用0替换...")
+ feature_matrix = np.nan_to_num(feature_matrix, nan=0.0)
+ print(f" ✅ NaN值已处理完成")
+ inf_count = np.isinf(feature_matrix).sum()
+ if inf_count > 0:
+ print(f" ⚠️ 发现 {inf_count} 个无穷大值,正在用0替换...")
+ feature_matrix = np.nan_to_num(feature_matrix, posinf=0.0, neginf=0.0)
+ print(f" ✅ 无穷大值已处理完成")
+ print(f" ✅ 特征矩阵清理完成,形状: {feature_matrix.shape}")
+ print("\n🔮 正在进行批量预测...")
+ features_scaled = scaler.transform(feature_matrix)
+ cluster_ids = kmeans.predict(features_scaled)
+ predictions = []
+ for i, (idx, row) in enumerate(df_check.iterrows()):
+ order_data = row.to_dict()
+ order_data["static_cluster"] = cluster_ids[i]
+ rule_score = calculate_rule_score(order_data)
+ cluster_score = calculate_cluster_score(order_data)
+ hybrid_score = (
+ rule_score * config.get_hybrid_config()["rule_weight"]
+ + cluster_score * config.get_hybrid_config()["cluster_weight"]
+ )
+ predicted_level = assign_level(
+ hybrid_score,
+ dynamic_thresholds["good_threshold"],
+ dynamic_thresholds["medium_threshold"],
+ )
+ threshold_info = f"h_score:{hybrid_score:.1f} vs g_thresh:{dynamic_thresholds['good_threshold']:.1f}, m_thresh:{dynamic_thresholds['medium_threshold']:.1f}"
+ reason = generate_prediction_reason(order_data, rule_score, cluster_score, hybrid_score)
+ reason_with_threshold = f"{reason}; 判定依据:{threshold_info}"
+ business_label = cluster_map.get(cluster_ids[i], f"聚类{cluster_ids[i]}")
+ predictions.append(
+ {
+ "order_no": row["order_no"],
+ "predicted_cluster": cluster_ids[i],
+ "static_cluster": cluster_ids[i],
+ "predicted_business_label": business_label,
+ "predicted_rule_score": rule_score,
+ "predicted_cluster_score": cluster_score,
+ "predicted_hybrid_score": hybrid_score,
+ "predicted_level": predicted_level,
+ "prediction_reason": reason_with_threshold,
+ "threshold_used_good": dynamic_thresholds["good_threshold"],
+ "threshold_used_medium": dynamic_thresholds["medium_threshold"],
+ }
+ )
+ else:
+ # === 实时特征重算路径(与API保持一致) ===
+ print("\n🔮 正在进行批量预测(实时特征重算)...")
+ # 加载用户属性映射
+ try:
+ user_attributes = joblib.load(os.path.join(MODEL_DIR, "user_attributes.pkl"))
+ print(f" ✅ 用户属性映射加载成功: {len(user_attributes)} 条")
+ except Exception as e:
+ print(f" ⚠️ 用户属性映射加载失败: {e},将使用空映射")
+ user_attributes = {}
+ def _get_user_attr(uid, mapping):
+ try:
+ uid_int = int(uid)
+ except (ValueError, TypeError):
+ uid_int = uid
+ if uid_int in mapping:
+ return mapping[uid_int]
+ if uid in mapping:
+ return mapping[uid]
+ return {"attention_cnt": 0, "merchant_aftersale_rate": 0.0, "ignore_cnt": 0, "business_full_name": "", "address": ""}
+ predictions = []
+ for i, (idx, row) in enumerate(df_check.iterrows()):
+ base = row.to_dict()
+ user_attr = _get_user_attr(base.get("user_id"), user_attributes)
+ feat = {**user_attr, **base}
+ # 统一 user_id 类型(与训练一致:尽量转为 int,失败兜底为 0)
+ try:
+ feat["user_id"] = int(feat.get("user_id")) if feat.get("user_id") not in (None, "") else 0
+ except (ValueError, TypeError):
+ feat["user_id"] = 0
+ # 统一类目类型(与训练一致:显式转为字符串,不做 strip/清洗)
+ feat["goods_level_3_name"] = str(feat.get("goods_level_3_name") or "")
+ # 时间特征
+ if feat.get("order_submit_time"):
+ try:
+ ts = pd.to_datetime(feat.get("order_submit_time"))
+ feat["submit_hour"] = ts.hour
+ feat["submit_weekday"] = ts.weekday()
+ feat["submit_is_weekend"] = 1 if feat["submit_weekday"] in [5, 6] else 0
+ feat["submit_is_business_hour"] = 1 if 9 <= feat["submit_hour"] <= 18 else 0
+ except Exception:
+ feat["submit_hour"] = 12
+ feat["submit_weekday"] = 1
+ feat["submit_is_weekend"] = 0
+ feat["submit_is_business_hour"] = 1
+ else:
+ feat["submit_hour"] = 12
+ feat["submit_weekday"] = 1
+ feat["submit_is_weekend"] = 0
+ feat["submit_is_business_hour"] = 1
+ # 基础模型特征(不含现值金额与单价)
+ feat["order_goods_cnt"] = feat.get("order_goods_cnt", 1)
+ feat["buyer_note_100"] = 1 if len(str(feat.get("buyer_note", ""))) > 100 else 0
+ # 业务规则分(与训练一致:直接用 calculate_business_rule_score,避免额外清洗)
+ feat["business_rule_score"] = calculate_business_rule_score(feat, BUSINESS_RULES)
+ # 先验特征
+ for name in [
+ "offer_rate",
+ "fifth_offer_duration_second",
+ "tenth_offer_duration_second",
+ "onsite_to_finish_hour",
+ "order_total_amount",
+ "order_unit_price",
+ "attention_cnt",
+ "merchant_aftersale_rate",
+ "ignore_cnt",
+ ]:
+ prior_val = get_prior_feature_value(
+ feat.get("user_id"), feat.get("goods_level_3_name"), name, prior_stats, category_stats
+ )
+ feat[f"{name}_prior"] = prior_val
+ # 价值类只用先验:has_price_info 基于先验金额
+ feat["has_price_info"] = 1 if float(feat.get("order_total_amount_prior", 0) or 0) > 0 else 0
+ # KMeans 预测
+ vec = [feat.get(f, 0) for f in MODEL_FEATURES]
+ arr = np.array(vec, dtype=np.float64)
+ arr = np.nan_to_num(arr, nan=0.0, posinf=0.0, neginf=0.0)
+ scaled = scaler.transform(arr.reshape(1, -1))
+ try:
+ cluster_id = int(kmeans.predict(scaled)[0])
+ except Exception:
+ cluster_id = 0
+ feat["static_cluster"] = cluster_id
+ # 打分
+ rule_score = calculate_rule_score(feat)
+ cluster_score = calculate_cluster_score(feat)
+ hybrid_score = (
+ rule_score * config.get_hybrid_config()["rule_weight"]
+ + cluster_score * config.get_hybrid_config()["cluster_weight"]
+ )
+ predicted_level = assign_level(
+ hybrid_score,
+ dynamic_thresholds["good_threshold"],
+ dynamic_thresholds["medium_threshold"],
+ )
+ threshold_info = f"h_score:{hybrid_score:.1f} vs g_thresh:{dynamic_thresholds['good_threshold']:.1f}, m_thresh:{dynamic_thresholds['medium_threshold']:.1f}"
+ reason = generate_prediction_reason(feat, rule_score, cluster_score, hybrid_score)
+ reason_with_threshold = f"{reason}; 判定依据:{threshold_info}"
+ business_label = cluster_map.get(cluster_id, f"聚类{cluster_id}")
+ predictions.append(
+ {
+ "order_no": row.get("order_no"),
+ "predicted_cluster": cluster_id,
+ "static_cluster": cluster_id,
+ "predicted_business_label": business_label,
+ "predicted_rule_score": rule_score,
+ "predicted_cluster_score": cluster_score,
+ "predicted_hybrid_score": hybrid_score,
+ "predicted_level": predicted_level,
+ "prediction_reason": reason_with_threshold,
+ "threshold_used_good": dynamic_thresholds["good_threshold"],
+ "threshold_used_medium": dynamic_thresholds["medium_threshold"],
+ }
+ )
+ # 将预测结果合并回df_check
+ df_pred_results = pd.DataFrame(predictions)
+ df_check = df_check.merge(df_pred_results, on="order_no", how="left")
+ print(f"✅ 批量预测完成!")
+ # === 第7步:与训练集结果对比 ===
+ print("\n📊 正在与训练集结果对比...")
+ # 找到训练集中对应的订单
+ df_training_subset = df_bidding[
+ df_bidding["order_no"].isin(df_check["order_no"])
+ ].copy()
+ print(f" - Check数据订单数: {len(df_check)}")
+ print(f" - 训练集中找到的订单数: {len(df_training_subset)}")
+ if len(df_training_subset) > 0:
+ # 合并数据进行对比
+ df_comparison = df_check[["order_no", "predicted_level"]].merge(
+ df_training_subset[["order_no", "static_level"]], on="order_no", how="inner"
+ )
+ print(f" - 成功匹配的订单数: {len(df_comparison)}")
+ # 计算准确率
+ if len(df_comparison) > 0:
+ correct_predictions = (
+ df_comparison["predicted_level"] == df_comparison["static_level"]
+ ).sum()
+ accuracy = correct_predictions / len(df_comparison) * 100
+ print(f"\n🎯 模型准确率验证:")
+ print(f" • 总匹配订单数: {len(df_comparison)}")
+ print(f" • 预测正确数: {correct_predictions}")
+ print(f" • 整体准确率: {accuracy:.2f}%")
+ # 各等级准确率
+ print(f"\n📈 各等级准确率:")
+ for level in ["好单", "中单", "差单"]:
+ level_data = df_comparison[df_comparison["static_level"] == level]
+ if len(level_data) > 0:
+ level_correct = (
+ level_data["predicted_level"] == level_data["static_level"]
+ ).sum()
+ level_accuracy = level_correct / len(level_data) * 100
+ print(
+ f" • {level}: {level_correct}/{len(level_data)} = {level_accuracy:.2f}%"
+ )
+ # 混淆矩阵
+ print(
+ "{: <8} {: <6} {: <6} {: <6}".format(
+ "真实\\预测", "好单", "中单", "差单"
+ )
+ )
+ print("-" * 35)
+ confusion_matrix = {}
+ for true_level in ["好单", "中单", "差单"]:
+ confusion_matrix[true_level] = {}
+ for pred_level in ["好单", "中单", "差单"]:
+ count = len(
+ df_comparison[
+ (df_comparison["static_level"] == true_level)
+ & (df_comparison["predicted_level"] == pred_level)
+ ]
+ )
+ confusion_matrix[true_level][pred_level] = count
+ print(
+ f"{true_level:<8} {confusion_matrix[true_level]['好单']:<6} {confusion_matrix[true_level]['中单']:<6} {confusion_matrix[true_level]['差单']:<6}"
+ )
+ # 错误案例分析
+ wrong_predictions = df_comparison[
+ df_comparison["predicted_level"] != df_comparison["static_level"]
+ ]
+ if len(wrong_predictions) > 0:
+ print(f"\n❌ 预测错误的订单:")
+ for idx, row in wrong_predictions.head(
+ 10
+ ).iterrows(): # 只显示前10个错误案例
+ print(
+ f" • {row['order_no']}: 真实={row['static_level']}, 预测={row['predicted_level']}"
+ )
+ if len(wrong_predictions) > 10:
+ print(f" • ... 还有{len(wrong_predictions)-10}个错误案例")
+ # 将对比结果加到check数据中
+ df_check = df_check.merge(
+ df_comparison[["order_no", "static_level"]], on="order_no", how="left"
+ )
+ df_check.rename(columns={"static_level": "training_label"}, inplace=True)
+ else:
+ print("❌ 没有找到可对比的订单")
+ else:
+ print("❌ 训练集中没有找到check数据的订单")
+ # === 第8步:保存结果 ===
+ print("\n💾 正在保存预测结果...")
+
+ output_path = "/Users/tom/Documents/data_check_with_predictions.xlsx"
+ df_check.to_excel(output_path, index=False)
+ print(f"✅ 预测结果已保存至: {output_path}")
+
+ # 显示预测结果摘要
+ print(f"\n📋 预测结果摘要:")
+ predicted_counts = df_check["predicted_level"].value_counts()
+ for level, count in predicted_counts.items():
+ percentage = count / len(df_check) * 100
+ print(f" • {level}: {count} 单 ({percentage:.1f}%)")
+ # 显示聚类分布
+ print(f"\n🏷️ 聚类分布:")
+ # 优先使用预测生成的聚类列
+ cluster_col = "static_cluster" if "static_cluster" in df_check.columns else (
+ "predicted_cluster" if "predicted_cluster" in df_check.columns else None
+ )
+ if cluster_col is None:
+ print(" ⚠️ 无可用聚类列")
+ return df_check
+ cluster_counts = df_check[cluster_col].value_counts().sort_index()
+ for cluster_id, count in cluster_counts.items():
+ business_label = cluster_map.get(cluster_id, f"聚类{cluster_id}")
+ percentage = count / len(df_check) * 100
+ print(f" • 聚类{cluster_id}({business_label}): {count} 单 ({percentage:.1f}%)")
+ # 评测基准:输出一次评估 JSON
+ try:
+ eval_report = {
+ "timestamp": datetime.utcnow().isoformat() + "Z",
+ "mode": "predict",
+ "sample_size": int(len(df_check)),
+ "thresholds": {
+ "good": float(dynamic_thresholds.get("good_threshold", 0)),
+ "medium": float(dynamic_thresholds.get("medium_threshold", 0)),
+ },
+ "accuracy": {
+ "overall": float(accuracy) if 'accuracy' in locals() else None,
+ "counts": {
+ "matched": int(len(df_comparison)) if 'df_comparison' in locals() else None,
+ "correct": int(correct_predictions) if 'correct_predictions' in locals() else None,
+ },
+ },
+ "confusion_matrix": confusion_matrix if 'confusion_matrix' in locals() else None,
+ "cluster_distribution": cluster_counts.to_dict(),
+ }
+ with open(os.path.join(MODEL_DIR, "eval_report.json"), "w", encoding="utf-8") as fh:
+ json.dump(eval_report, fh, ensure_ascii=False, indent=2)
+ print(f"\n✅ 评测报告已保存: {os.path.join(MODEL_DIR, 'eval_report.json')}")
+ except Exception as e:
+ print(f"⚠️ 评测报告保存失败: {e}")
+ return df_check
+
+
+# 执行预测
+try:
+ df_check_results = predict_check_orders()
+ if df_check_results is not None:
+ print(f"\n🎉 CHECK数据预测完成!")
+ print(
+ f"📄 详细结果请查看: /Users/tom/Documents/data_check_with_predictions.xlsx"
+ )
+ print(f"🔍 该文件包含:")
+ print(f" • 原始check数据的所有字段")
+ print(f" • 补全的动态特征")
+ print(f" • 预测结果: predicted_level, prediction_reason")
+ print(f" • 训练集对比: training_label")
+ print(f" • 评分详情: rule_score, cluster_score, hybrid_score")
+ print(f" • 聚类信息: static_cluster, business_label")
+ else:
+ print(f"❌ CHECK数据预测失败")
+except Exception as e:
+ print(f"❌ 预测过程中出现错误: {e}")
+ import traceback
+
+ traceback.print_exc()
+print(f"\n" + "=" * 80)
+print(f"[预测接口完成] 可以删除此模块以保持脚本简洁")
+print(f"=" * 80)
+# ------------------------------
+# 在最后添加强制保存验证
+# ------------------------------
+print("\n" + "=" * 80)
+print("[模型文件保存验证] 确保所有模型文件成功保存")
+print("=" * 80)
+# 检查模型目录
+print(f"\n🔍 检查模型目录: {os.path.abspath(MODEL_DIR)}")
+print(f"目录是否存在: {os.path.exists(MODEL_DIR)}")
+if not os.path.exists(MODEL_DIR):
+ print(f"❌ 目录不存在,重新创建...")
+ os.makedirs(MODEL_DIR, exist_ok=True)
+ print(f"✅ 目录已创建")
+# 强制重新保存所有关键模型组件
+print(f"\n🔄 强制重新保存所有模型组件...")
+try:
+ # 1. 核心模型组件
+ print(" 📦 保存核心模型组件...")
+ joblib.dump(scaler, os.path.join(MODEL_DIR, "scaler.pkl"))
+ print(" ✅ scaler.pkl")
+ joblib.dump(kmeans, os.path.join(MODEL_DIR, "kmeans_model.pkl"))
+ print(" ✅ kmeans_model.pkl")
+ joblib.dump(BUSINESS_RULES, os.path.join(MODEL_DIR, "business_rules.pkl"))
+ print(" ✅ business_rules.pkl")
+ joblib.dump(prior_stats, os.path.join(MODEL_DIR, "prior_stats.pkl"))
+ print(" ✅ prior_stats.pkl")
+ joblib.dump(category_stats, os.path.join(MODEL_DIR, "category_stats.pkl"))
+ print(" ✅ category_stats.pkl")
+ joblib.dump(MODEL_FEATURES, os.path.join(MODEL_DIR, "model_features.pkl"))
+ print(" ✅ model_features.pkl")
+ joblib.dump(cluster_map, os.path.join(MODEL_DIR, "cluster_map.pkl"))
+ print(" ✅ cluster_map.pkl")
+ # 2. 评分系统组件
+ print(" 🎯 保存评分系统组件...")
+ joblib.dump(cluster_base_scores, os.path.join(MODEL_DIR, "cluster_base_scores.pkl"))
+ print(" ✅ cluster_base_scores.pkl")
+ joblib.dump(
+ DYNAMIC_RULE_THRESHOLDS, os.path.join(MODEL_DIR, "dynamic_rule_thresholds.pkl")
+ )
+ print(" ✅ dynamic_rule_thresholds.pkl")
+ joblib.dump(DYNAMIC_THRESHOLDS, os.path.join(MODEL_DIR, "dynamic_thresholds.pkl"))
+ print(" ✅ dynamic_thresholds.pkl")
+ joblib.dump(
+ AUTO_CLUSTER_LEVEL_MAP, os.path.join(MODEL_DIR, "auto_cluster_level_map.pkl")
+ )
+ print(" ✅ auto_cluster_level_map.pkl")
+ joblib.dump(
+ cluster_quality_scores, os.path.join(MODEL_DIR, "cluster_quality_scores.pkl")
+ )
+ print(" ✅ cluster_quality_scores.pkl")
+ joblib.dump(HYBRID_CONFIG, os.path.join(MODEL_DIR, "hybrid_config.pkl"))
+ print(" ✅ hybrid_config.pkl")
+ print(f"\n✅ 所有模型组件强制保存完成!(共15个核心文件)")
+except Exception as e:
+ print(f"\n❌ 保存过程中出现错误: {e}")
+ import traceback
+
+ traceback.print_exc()
+# 验证文件是否真的存在
+print(f"\n🔍 验证文件保存结果...")
+required_files = [
+ "scaler.pkl",
+ "kmeans_model.pkl",
+ "business_rules.pkl",
+ "prior_stats.pkl",
+ "category_stats.pkl",
+ "model_features.pkl",
+ "cluster_map.pkl",
+ "cluster_base_scores.pkl",
+ "dynamic_rule_thresholds.pkl",
+ "dynamic_thresholds.pkl",
+ "auto_cluster_level_map.pkl",
+ "cluster_quality_scores.pkl",
+ "hybrid_config.pkl",
+ "global_stats.pkl",
+ "user_attributes.pkl",
+]
+saved_files = []
+missing_files = []
+for file in required_files:
+ file_path = os.path.join(MODEL_DIR, file)
+ if os.path.exists(file_path):
+ size = os.path.getsize(file_path)
+ saved_files.append((file, size))
+ print(f" ✅ {file}: {size:,} bytes")
+ else:
+ missing_files.append(file)
+ print(f" ❌ {file}: 缺失")
+print(f"\n📊 保存结果统计:")
+print(f" ✅ 成功保存: {len(saved_files)} 个文件")
+print(f" ❌ 缺失文件: {len(missing_files)} 个文件")
+if missing_files:
+ print(f" ⚠️ 缺失的文件: {missing_files}")
+else:
+ print(f" 🎉 所有必需文件保存完整!")
+ # 显示目录内容
+ print(f"\n📁 模型目录最终内容:")
+ all_files = os.listdir(MODEL_DIR)
+ for file in sorted(all_files):
+ if not file.startswith("."):
+ file_path = os.path.join(MODEL_DIR, file)
+ size = os.path.getsize(file_path)
+ print(f" {file}: {size:,} bytes")
+print(f"\n" + "=" * 80)
+print(f"[训练预测一致性验证] 确保训练和预测逻辑完全一致")
+print(f"=" * 80)
+
+
+def verify_training_prediction_consistency():
+ """
+
+ 验证训练时和预测时的逻辑一致性
+
+ """
+ print(f"\n🔍 正在验证训练预测一致性...")
+ # 1. 验证阈值一致性
+ print(f" 📊 验证阈值一致性:")
+ try:
+ saved_thresholds = joblib.load(
+ os.path.join(MODEL_DIR, "dynamic_thresholds.pkl")
+ )
+ print(f" ✅ 动态阈值文件存在")
+ print(f" • 好单阈值: {saved_thresholds['good_threshold']:.1f}分")
+ print(f" • 中单阈值: {saved_thresholds['medium_threshold']:.1f}分")
+ print(f" • 算法: {saved_thresholds.get('algorithm', '未知')}")
+ # 验证与当前训练结果一致
+ if "DYNAMIC_THRESHOLDS" in globals():
+ current_good = DYNAMIC_THRESHOLDS["good_threshold"]
+ current_medium = DYNAMIC_THRESHOLDS["medium_threshold"]
+ saved_good = saved_thresholds["good_threshold"]
+ saved_medium = saved_thresholds["medium_threshold"]
+ if (
+ abs(current_good - saved_good) < 0.01
+ and abs(current_medium - saved_medium) < 0.01
+ ):
+ print(f" ✅ 训练阈值与保存阈值一致")
+ else:
+ print(f" ⚠️ 训练阈值与保存阈值不一致")
+ print(f" 训练: 好单{current_good:.1f}, 中单{current_medium:.1f}")
+ print(f" 保存: 好单{saved_good:.1f}, 中单{saved_medium:.1f}")
+ except Exception as e:
+ print(f" ❌ 动态阈值验证失败: {e}")
+ # 2. 验证规则阈值一致性
+ print(f" 🎯 验证规则阈值一致性:")
+ try:
+ saved_rule_thresholds = joblib.load(
+ os.path.join(MODEL_DIR, "dynamic_rule_thresholds.pkl")
+ )
+ print(f" ✅ 动态规则阈值文件存在")
+ print(f" • 包含特征数: {len(saved_rule_thresholds)}")
+ # 验证关键特征
+ key_features = [
+ "order_total_amount",
+ "order_unit_price",
+ "tenth_offer_duration_second",
+ ]
+ for feature in key_features:
+ if feature in saved_rule_thresholds:
+ print(f" • {feature}: ✅")
+ else:
+ print(f" • {feature}: ❌ 缺失")
+ except Exception as e:
+ print(f" ❌ 动态规则阈值验证失败: {e}")
+ # 3. 验证聚类基础分一致性
+ print(f" 🏷️ 验证聚类基础分一致性:")
+ try:
+ saved_cluster_scores = joblib.load(
+ os.path.join(MODEL_DIR, "cluster_base_scores.pkl")
+ )
+ print(f" ✅ 聚类基础分文件存在")
+ print(f" • 聚类数量: {len(saved_cluster_scores)}")
+ if "cluster_base_scores" in globals():
+ for cluster_id in range(6):
+ current_score = cluster_base_scores.get(cluster_id, 0)
+ saved_score = saved_cluster_scores.get(cluster_id, 0)
+ if abs(current_score - saved_score) < 0.01:
+ print(f" • 聚类{cluster_id}: ✅ ({saved_score:.1f}分)")
+ else:
+ print(
+ f" • 聚类{cluster_id}: ⚠️ 不一致 (训练{current_score:.1f} vs 保存{saved_score:.1f})"
+ )
+ except Exception as e:
+ print(f" ❌ 聚类基础分验证失败: {e}")
+ # 4. 验证配置一致性
+ print(f" ⚙️ 验证配置一致性:")
+ try:
+ saved_config = joblib.load(os.path.join(MODEL_DIR, "hybrid_config.pkl"))
+ print(f" ✅ 混合配置文件存在")
+ current_config = config.get_hybrid_config()
+ for key in [
+ "rule_weight",
+ "cluster_weight",
+ "good_quantile",
+ "medium_quantile",
+ ]:
+ if key in saved_config and key in current_config:
+ if abs(saved_config[key] - current_config[key]) < 0.001:
+ print(f" • {key}: ✅ ({saved_config[key]})")
+ else:
+ print(f" • {key}: ⚠️ 不一致")
+ else:
+ print(f" • {key}: ❌ 缺失")
+ except Exception as e:
+ print(f" ❌ 配置验证失败: {e}")
+ print(f"\n✅ 训练预测一致性验证完成!")
+ print(f"💡 说明:")
+ print(f" • 训练时:使用严格比例分配计算阈值")
+ print(f" • 预测时:使用训练时保存的分数阈值进行判定")
+ print(f" • 规则评分:训练和预测使用相同的动态阈值")
+ print(f" • 聚类评分:训练和预测使用相同的基础分")
+ print(f" • 混合权重:训练和预测使用相同的配置")
+
+
+# 执行一致性验证
+verify_training_prediction_consistency()
+print(f"\n" + "=" * 80)
+print(f"[最终总结] 训练预测一致性修复完成")
+print(f"=" * 80)
+print(f"\n🎉 修复完成!主要改进:")
+print(f" ✅ 1. 训练时立即保存分数阈值到dynamic_thresholds.pkl")
+print(f" ✅ 2. 预测时直接使用保存的分数阈值,不再重新计算")
+print(f" ✅ 3. 规则评分函数统一使用动态阈值(训练预测一致)")
+print(f" ✅ 4. 聚类评分使用训练时计算的基础分")
+print(f" ✅ 5. 混合权重配置完全一致")
+print(f" ✅ 6. 所有模型文件强制保存并验证")
+print(f"\n📋 预期效果:")
+print(f" 🎯 使用训练数据作为check数据时,应该能达到接近100%的还原度")
+print(f" 📊 如果还原度不是100%,可能的原因:")
+print(f" • 数据预处理差异(缺失值填充、特征工程)")
+print(f" • 先验特征计算差异(用户ID类型、分组逻辑)")
+print(f" • 浮点数精度差异(可忽略,<0.1%差异属正常)")
+print(f"\n🔧 使用说明:")
+print(f" 1. 运行此脚本完成训练并保存模型")
+print(f" 2. 使用相同的训练数据作为check数据进行预测")
+print(f" 3. 对比predicted_level和static_level的一致性")
+print(f" 4. 如有不一致,检查数据预处理和特征工程步骤")
+print(f"\n📄 模型文件位置: {os.path.abspath(MODEL_DIR)}")
+print(f"📄 预测结果位置: /Users/tom/Documents/data_check_with_predictions.xlsx")
diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 1.png b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 1.png
new file mode 100644
index 0000000..ef97f5a
Binary files /dev/null and b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 1.png differ
diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 2.png b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 2.png
new file mode 100644
index 0000000..41cd214
Binary files /dev/null and b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图 2.png differ
diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图.png b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图.png
new file mode 100644
index 0000000..14e63be
Binary files /dev/null and b/工作记录/曹强-订单分层算法落地方案v0.6/图片和附件/订单聚类可视化图.png differ
diff --git a/工作记录/曹强-订单分层算法落地方案v0.6/订单分层算法落地方案v0.6.md b/工作记录/曹强-订单分层算法落地方案v0.6/订单分层算法落地方案v0.6.md
new file mode 100644
index 0000000..4ffbfc4
--- /dev/null
+++ b/工作记录/曹强-订单分层算法落地方案v0.6/订单分层算法落地方案v0.6.md
@@ -0,0 +1,498 @@
+# 订单分层算法落地方案v0\.6
+
+> **上周突击**
+>
+> **1\. 构建完成订单AI智能分析框架(v0\.6)**
+>
+> \- 打通数据清洗→特征工程→业务规则集成→KMeans建模→实时预测API→Excel导出全流程
+>
+> \- 新增业务规则特征工程、混合特征建模、实时预测服务等核心能力
+>
+> **2\. 完成初步关键业务洞察**
+>
+> \- 基于522,544条数据,用数据揭示业务"金字塔结构"
+>
+> \- 验证优质企业识别和地区优势:北欧表情、林氏木业等优质企业订单主要分布在优质大单类别
+>
+> **3\. 模型提升点**
+>
+> \- 当前26个特征\(11基础\+15类别先验\+业务规则\),模型效果显著
+>
+> \- 精细分层9个聚类类别已形成,可支持精细化运营
+>
+>
+>
+> **下阶段需求(支持v0\.8启动)**
+>
+> 1\. 数据补充:补充客户行为特征、身份标签、动态特征等
+>
+> 2\. 经验协同:成立"订单AI共建组",业务经验透传算法侧
+>
+> 3\. 技术升级:模型版本管理、更多特征理解
+>
+> 4\. 应用深化:智能分配、定价优化、风险控制
+>
+>
+
+
+
+
+
+---
+
+# **1、背景与目标**
+
+1、智能锁,家具2大品类的 好单的规律信息, 规律维度可以参考看看:产地,品牌商家,订单金额,加急单,耗时,时薪等等
+
+2、暂定下周五我们向高层输出汇报, 涉及业务信息我们目前有的主要上面文档和以及内嵌文档里,同时我们也将继续访谈调研师傅 好单
+
+背景资料:[服务分层框架](https://jiqmwlmd0v.feishu.cn/wiki/PEIpwfLveiHgQxkZIBacUsPCnTe)
+
+
+
+## **1\.1 痛点与需求**
+
+- **资源错配**:在过去,我们的运营模式倾向于对所有订单投入均等的人力与服务资源。这导致我们在低价值订单上消耗了过多精力,而对高价值的“金牛”订单却可能因响应不及时而造成流失。
+
+- **认知模糊**:我们对自身的订单结构缺乏一个清晰、量化的认知。不同类型的订单究竟长什么样?它们的占比各是多少?这些问题直接影响了我们制定市场、销售及服务策略的精准度。
+
+- **效率瓶颈**:随着业务量的持续增长,依靠人工经验来判断订单优先级的方式已难以为继,成为制约我们运营效率提升的核心瓶颈。
+
+
+
+## **1\.2 目标**
+
+- **实现订单分层**:构建一套科学、自动化的订单分层体系,将海量、混杂的订单数据,精准划分为具有不同业务价值的类别。
+
+- **洞察客户画像**:通过对不同类别订单的分析,反向描绘出我们的客户画像,了解“谁是我们的主流客户?”、“谁是我们的高价值客户?”。
+
+- **提升运营效率**:将模型能力赋能业务一线,实现新订单的实时自动分类,指导运营团队进行差异化的资源分配与服务响应,最终提升整体投入产出比(ROI)。
+
+
+
+# **2\.** **数据说明**
+
+- **数据来源与范围**:
+
+ - 为保证模型质量,我们选取了2025年1月至6月期间,所有状态为 **“已完成”** 的 **“报价招标”** 类型订单,剔除了未完成及其他模式(一口价、一口价单独做一套训练)的订单干扰。
+
+ - 原始数据量:约50万条
+
+ - 城市:北上广深(一线城市)
+
+ - 一级类目:家具
+
+
+
+- **当前训练所用数据集**:
+
+
+
+# **4\. 模型实现**
+
+## 3\.1 逻辑简单介绍
+
+1. **数据整理**
+
+- 收集52万条历史订单数据
+
+- 提取订单金额、商品数量、企业信息、地区等基础特征
+
+- 计算每个商品类别的历史表现(报价量、浏览量、响应速度等)
+
+2. **业务规则制定**
+
+- 优质企业加分:顾家家居、林氏木业等知名企业
+
+- 优质地区加分:佛山、浙江、河北等发达地区
+
+- 优质商品加分:办公家具、屏风类等热门品类
+
+- 大订单加分:商品数量≥10件的订单
+
+- 问题地区减分:徐州等地区
+
+3. **智能分类**
+
+- 将订单分为9个等级:从"高价精品单"到"标准订单"
+
+- 每个等级都有明确的业务特征和价值表现
+
+- 新订单可以实时预测属于哪个等级
+
+
+
+
+
+## 3\.2 版本迭代记录
+
+### 3\.2\.1 版本一(全自动\-静态特征)
+
+
+静态特征(训练)
+
+1. order\_total\_amount \- 订单总金额
+
+2. order\_goods\_cnt \- 商品数量
+
+3. order\_unit\_ price \- 单价
+
+4. company\_type \- 公司类型(编码后)
+
+5. user\_type \- 用户类型
+
+6. order\_appoint\_type\_name \- 订单指派类型(编码后)
+
+7. submit\_hour \- 提交小时
+
+8. submit\_weekday \- 提交星期
+
+9. submit\_is\_weekend \- 是否周末
+
+10. submit\_is\_business\_hour \- 是否工作时间
+
+
+
+静态特征(只是辅助参考)
+
+1. offer\_mst\_cnt \- 报价师傅数量
+
+ - 有多少师傅对这个订单进行了报价
+
+2. view\_mst\_cnt \- 浏览师傅数量
+
+ - 有多少师傅查看了这个订单
+
+3. fifth\_offer\_duration\_second \- 第5人报价时长(秒)
+
+ - 从订单发布到第5个师傅报价的时间间隔
+
+4. tenth\_offer\_duration\_second \- 第10人报价时长(秒)
+
+ - 从订单发布到第10个师傅报价的时间间隔
+
+5. attention\_cnt \- 关注数量
+
+ - 订单被关注/收藏的次数
+
+6. onsite\_to\_finish\_hour \- 现场到完成时长(小时)
+
+ - 从师傅到达现场到完成服务的时间
+
+7. serve\_efficiency \- 服务效率
+
+ - 商品数量 / 服务时长,衡量服务效率
+
+
+
+|聚类编号|业务标签|订单数\(占比\)|静态特征|动态特征|业务解读|
+|---|---|---|---|---|---|
+|0|标准订单|51,497\
\(9\.86%\)|金额:119\.79\
商品数:1\.66\
单价:96\.32\
工作时间:1\.00\
星期:1\.97|报价量:8\.39\
浏览量:21\.88\
5人报价:3,484秒\
10人报价:4,493秒\
关注量:452\.16|工作时间标准单,商品数较高,市场活跃|
+|1|主流订单|212,419\
\(40\.65%\)|金额:115\.56\
商品数:1\.28\
单价:101\.35\
工作时间:1\.00\
星期:2\.00|报价量:8\.13\
浏览量:21\.67\
5人报价:3,844秒\
10人报价:4,737秒\
关注量:368\.09|主要类别,工作时间标准单,响应较快|
+|2|标准优质单|102,927\
\(19\.70%\)|金额:117\.85\
商品数:1\.29\
单价:103\.17\
工作时间:1\.00\
星期:5\.48|报价量:8\.62\
浏览量:21\.99\
5人报价:4,465秒\
10人报价:4,798秒\
关注量:371\.35|周末优质单,商品数适中,关注度高|
+|3|标准订单|60,843\
\(11\.64%\)|金额:116\.50\
商品数:1\.24\
单价:103\.75\
工作时间:0\.00\
星期:2\.97|报价量:8\.68\
浏览量:24\.09\
5人报价:5,015秒\
10人报价:6,459秒\
关注量:193\.81|非工作时间标准单,报价时长较长|
+|4|异常大单|8\
\(0\.00%\)
|金额:17,673\.25\
商品数:198\.75\
单价:88\.79\
工作时间:1\.00\
星期:1\.25|报价量:5\.00\
浏览量:11\.63\
5人报价:427秒\
10人报价:1,088秒\
关注量:1,014\.88|极端异常订单,金额和商品数极高|
+|5|高价小众单|22,002\
\(4\.21%\)|金额:374\.94\
商品数:1\.37\
单价:323\.01\
工作时间:0\.90\
星期:2\.78|报价量:6\.94\
浏览量:23\.74\
5人报价:6,636秒\
10人报价:5,512秒\
关注量:439\.22|金额较高,单价极高,竞争激烈|
+|6|高价值订单|412\
\(0\.08%\)|金额:695\.58\
商品数:69\.17\
单价:11\.27\
工作时间:0\.82\
星期:2\.51|报价量:10\.90\
浏览量:24\.11\
5人报价:1,284秒\
10人报价:2,424秒\
关注量:246\.03|高价值大单,响应极快,关注度高|
+|7|标准订单|28,978\
\(5\.55%\)|金额:116\.04\
商品数:1\.29\
单价:102\.81\
工作时间:0\.00\
星期:2\.97|报价量:8\.68\
浏览量:24\.09\
5人报价:5,015秒\
10人报价:6,459秒\
关注量:193\.81|非工作时间标准单,各项指标中等|
+|8|标准大单|43,458\
\(8\.32%\)|金额:121\.00\
商品数:1\.41\
单价:102\.28\
工作时间:0\.80\
星期:2\.97|报价量:8\.47\
浏览量:23\.18\
5人报价:4,047秒\
10人报价:5,104秒\
关注量:33\.20|标准大单,金额略高,关注度较低|
+
+
+
+### 3\.2\.2 版本二(全自动\-动态特征)
+
+
+
+基础特征(10个):
+
+order\_total\_amount \- 订单总金额
+
+order\_goods\_cnt \- 商品数量
+
+order\_unit\_ price \- 单价
+
+company\_type \- 公司类型
+
+user\_type \- 用户类型
+
+order\_appoint\_type\_name \- 订单指派类型
+
+submit\_hour \- 提交小时
+
+submit\_weekday \- 提交星期
+
+submit\_is\_weekend \- 是否周末
+
+submit\_is\_business\_hour \- 是否工作时间
+
+
+
+**可用的动态特征(需要数据中存在,这里也用来训练了):**
+
+offer\_mst\_cnt \- 报价师傅数量
+
+view\_mst\_cnt \- 浏览师傅数量
+
+fifth\_offer\_duration\_second \- 第5人报价时长(秒)
+
+tenth\_offer\_duration\_second \- 第10人报价时长(秒)
+
+attention\_cnt \- 关注数量
+
+onsite\_to\_finish\_hour \- 现场到完成时长(小时)
+
+serve\_efficiency \- 服务效率
+
+
+
+每个动态特征生成的3个类别先验特征:
+
+以offer\_mst\_cnt为例:
+
+offer\_mst\_cnt\_category\_mean \- 该类目平均报价师傅数
+
+offer\_mst\_cnt\_category\_median \- 该类目报价师傅数中位数
+
+offer\_mst\_cnt\_category\_std \- 该类目报价师傅数标准差
+
+
+
+
+
+
+
+|聚类编号|业务标签|订单数\(占比\)|静态特征|动态特征|业务解读|
+|---|---|---|---|---|---|
+|0|主流订单|221,922\
\(42\.47%\)|金额:129\.51\
商品数:1\.35\
该类目平均报价8\.3人,平均关注66人|报价量:8\.33\
浏览量:26\.09\
5人报价:4,760秒\
10人报价:5,618秒\
关注量:177\.95|主要类别,该类目历史报价活跃,关注度中等|
+|1|标准优质单|81,886\
\(15\.67%\)|金额:129\.91\
商品数:1\.46\
该类目平均报价8\.5人,平均关注27人|报价量:8\.53\
浏览量:21\.03\
5人报价:3,763秒\
10人报价:4,896秒\
关注量:422\.48|第二大类别,该类目历史报价活跃,关注度较低|
+|2|高价值订单|86,033\
\(16\.46%\)|金额:127\.54\
商品数:1\.27\
该类目平均报价10\.8人,平均关注66人|报价量:10\.76\
浏览量:24\.63\
5人报价:1,433秒\
10人报价:2,117秒\
关注量:318\.73|第三大类别,该类目历史报价最活跃,响应极快
|
+|3|标准订单|7,892\
\(1\.51%\)|金额:102\.43\
商品数:1\.08\
该类目平均报价8\.7人,平均关注221人|报价量:8\.66\
浏览量:21\.40\
5人报价:3,524秒\
10人报价:4,741秒\
关注量:425\.75|金额最低,该类目历史关注度最高,市场热度极强|
+|4|高价小众单|56,029\
\(10\.72%\)|金额:117\.36\
商品数:1\.23\
该类目平均报价7\.8人,平均关注47人|报价量:7\.80\
浏览量:28\.72\
5人报价:6,986秒\
10人报价:7,458秒\
关注量:145\.17|金额较高,该类目历史报价较少,竞争激烈|
+|5|高价标准单|5,805\
\(1\.11%\)|金额:221\.52\
商品数:1\.41\
该类目平均报价7\.6人,平均关注1318人|报价量:7\.58\
浏览量:22\.96\
5人报价:5,426秒\
10人报价:5,578秒\
关注量:250\.19|金额较高,该类目历史关注度极高,市场热点|
+|6|标准大单|23,161\
\(4\.43%\)|金额:147\.98\
商品数:1\.63\
该类目平均报价9\.4人,平均关注14人|报价量:9\.41\
浏览量:23\.27\
5人报价:3,309秒\
10人报价:5,060秒\
关注量:811\.95|商品数较高,该类目历史报价活跃,关注度较低|
+|7|异常大单|233\
\(0\.04%\)|金额:1,498\.73\
商品数:94\.91\
该类目平均报价7\.6人,平均关注33人|报价量:7\.63\
浏览量:20\.32\
5人报价:3,591秒\
10人报价:4,165秒\
关注量:170\.63|金额和商品数极端高,该类目历史表现一般|
+|8|标准订单|39,583\
\(7\.58%\)|金额:112\.02\
商品数:1\.28\
该类目平均报价8\.7人,平均关注43人|报价量:8\.66\
浏览量:21\.40\
5人报价:3,524秒\
10人报价:4,741秒\
关注量:425\.75|标准订单,该类目历史表现中等|
+
+
+
+### 3\.2\.3 版本三(半自动\-动态特征)
+
+**升级点:在原来的基础上叠加了运营的调研经验,将调研经验得出白黑名单、加减分项来继续升级模板。**
+
+```SQL
+加分项
+business_full_name字段为'北欧表情(深圳)家具有限公司'、'西昊家具(深圳)有限公司'、'佛山林氏木业家具有限公司'、'顾家家居股份有限公司'
+address字段包含'佛山'、'东莞'、'河北'、'浙江'
+goods_level_2_name字段包含'办公家具'、'屏风类'、'户外'、'柜类'
+order_serve_type_name字段包含'送货到家并安装'、'维修'
+order_label字段包含'加急单'
+order_goods_cnt大于等于10单
+
+减分项
+address字段包含'徐州'
+
+
+
+# --- 业务规则配置 ---
+BUSINESS_RULES = {
+ # 加分项配置
+ 'bonus_rules': {
+ # 优质企业加分
+ 'premium_companies': [
+ '北欧表情(深圳)家具有限公司',
+ '西昊家具(深圳)有限公司',
+ '佛山林氏木业家具有限公司',
+ '顾家家居股份有限公司'
+ ],
+ # 优质地区加分
+ 'premium_regions': ['佛山', '东莞', '河北', '浙江'],
+ # 优质商品类别加分
+ 'premium_categories': ['办公家具', '屏风类', '户外', '柜类'],
+ # 优质服务类型加分
+ 'premium_services': ['送货到家并安装', '维修'],
+ # 加急订单加分
+ 'urgent_orders': ['加急单'],
+ # 大订单加分(商品数量>=10)
+ 'large_orders_threshold': 10
+ },
+ # 减分项配置
+ 'penalty_rules': {
+ # 问题地区减分
+ 'problem_regions': ['徐州']
+ },
+ # 规则权重配置
+ 'rule_weights': {
+ 'premium_company_bonus': 2.0, # 优质企业加分权重
+ 'premium_region_bonus': 1.5, # 优质地区加分权重
+ 'premium_category_bonus': 1.0, # 优质商品类别加分权重
+ 'premium_service_bonus': 1.0, # 优质服务类型加分权重
+ 'urgent_order_bonus': 1.5, # 加急订单加分权重
+ 'large_order_bonus': 1.0, # 大订单加分权重
+ 'problem_region_penalty': -1.5 # 问题地区减分权重
+ }
+}
+```
+
+
+
+
+
+|**聚类编号**|**业务标签**|**订单数\(占比\)**|**静态特征**|**动态特征**|**业务解读**|
+|---|---|---|---|---|---|
+|0|标准主流单|44,567\
\(8\.53%\)|金额:123\.61\
商品数:1\.26\
单价:98\.10\
工作时间:1\.00\
星期:2\.00\
业务规则得分:0\.79|报价量:9\.04\
浏览量:23\.31\
5人报价:5,619秒\
10人报价:7,594秒\
关注量:386\.92|标准主流订单,报价量最高,响应较慢但关注度高,优质企业集中|
+|1|优质大单|112,167\
\(21\.47%\)|金额:134\.79\
商品数:1\.52\
单价:88\.68\
工作时间:1\.00\
星期:2\.00\
业务规则得分:0\.94|报价量:8\.88\
浏览量:22\.14\
5人报价:3,283秒\
10人报价:4,604秒\
关注量:349\.94|**优质企业集中**,商品数最高,响应最快,**大订单主要分布**,**业务规则得分高**|
+|2
|高价值订单
|127,887\
\(24\.47%\)
|金额:125\.42\
商品数:1\.37\
单价:91\.55\
工作时间:1\.00\
星期:2\.00\
业务规则得分:0\.63|报价量:8\.03\
浏览量:25\.00\
5人报价:5,049秒\
10人报价:5,631秒\
关注量:210\.66|最大类别,关注度极高,优质企业集中,响应中等,佛山浙江地区集中|
+|3|**高价精品单**|5,821\
\(1\.11%\)
|金额:250\.68\
商品数:1\.77\
单价:141\.63\
工作时间:1\.00\
星期:2\.00\
业务规则得分:1\.35|报价量:9\.41\
浏览量:23\.26\
5人报价:3,302秒\
10人报价:5,053秒\
关注量:812\.70|金额最高,关注度最高,精品订单,业务规则得分最高,响应较快|
+|4|高价小众单|23,246\
\(4\.45%\)|金额:149\.19\
商品数:1\.74\
单价:85\.74\
工作时间:1\.00\
星期:2\.00\
业务规则得分:0\.94|报价量:7\.80\
浏览量:28\.71\
5人报价:6,996秒\
10人报价:7,444秒\
关注量:145\.31|**佛山地区集中**,商品数较高,关注度极高,响应最慢,**业务规则得分高**|
+|5|标准订单|46,434\
\(8\.89%\)|金额:107\.81\
商品数:1\.27\
单价:84\.89\
工作时间:1\.00\
星期:2\.00\
业务规则得分:0\.11|报价量:8\.24\
浏览量:20\.35\
5人报价:3,299秒\
10人报价:4,078秒\
关注量:402\.86|金额最低,基础订单,响应较快,业务规则得分最低,普通企业为主|
+|6|标准大单|74,377\
\(14\.23%\)|金额:128\.93\
商品数:1\.30\
单价:99\.18\
工作时间:1\.00\
星期:2\.00\
业务规则得分:0\.84|报价量:8\.45\
浏览量:20\.71\
5人报价:3,499秒\
10人报价:4,482秒\
关注量:423\.76|标准大单,响应较快,关注度较高,业务规则得分较高,优质企业集中|
+|7|异常大单|8,112\
\(1\.55%\)|金额:103\.58\
商品数:1\.11\
单价:93\.31\
工作时间:1\.00\
星期:2\.00\
业务规则得分:0\.10|报价量:4\.74\
浏览量:15\.47\
5人报价:3,376秒\
10人报价:3,045秒\
关注量:457\.67|商品数最少,特殊订单,报价量最低,响应最快,业务规则得分最低|
+|8|标准订单|79,933\
\(15\.30%\)|金额:127\.64\
商品数:1\.34\
单价:95\.25\
工作时间:1\.00\
星期:2\.00\
业务规则得分:0\.72|报价量:8\.24\
浏览量:20\.35\
5人报价:3,299秒\
10人报价:4,078秒\
关注量:402\.86|标准订单,各项指标中等,业务规则得分中等,浙江地区集中|
+
+|**质量等级**|**订单类别**|**订单数\(占比\)**|**主要特征**|**业务建议**|
+|---|---|---|---|---|
+|好单
|高价精品单\(3\)\
优质大单\(1\)\
高价小众单\(4\)|141,235\
\(27\.03%\)|金额高、响应快、关注度高\
业务规则得分高|重点维护,优先分配优质供应商|
+|中单|高价值订单\(2\)\
标准大单\(6\)\
标准主流单\(0\)\
标准订单\(8\)|326,764\
\(62\.55%\)|各项指标均衡\
业务规则得分中等|正常处理,保持服务质量|
+|差单|标准订单\(5\)\
异常大单\(7\)|54,546\
\(10\.42%\)|金额低、竞争少\
业务规则得分低|需要关注,可能存在风险或价值较低|
+
+
+
+**业务规则效果验证**
+
+**高分订单特征**:
+
+- 高价精品单\(1\.35分\):金额最高,关注度最高
+
+- 优质大单\(0\.94分\):商品数最高,响应最快
+
+- 高价小众单\(0\.94分\):地区优势,关注度极高
+
+**低分订单特征**:
+
+- 标准订单\(0\.11分\):基础订单,普通企业
+
+- 异常大单\(0\.10分\):特殊订单,竞争较少
+
+**地区分布特征**:
+
+- 佛山:主要分布在高价值订单\(5,474单\)和高价小众单\(2,416单\)
+
+- 浙江:主要分布在高价值订单\(3,372单\)和标准订单\(1,838单\)
+
+- 河北:主要分布在优质大单\(1,102单\)和高价值订单\(1,033单\)
+
+
+
+
+
+### 3\.2\.4 版本四(有监督\-师傅打标)
+
+**运营人工打标,做裁判/数据标签输入**
+**增加产品功能\-\-\-让师傅打标**
+
+
+
+
+
+
+
+### 3\.2\.5 版本五(规则70%\+聚类30%)\*目前最新
+
+
+
+**好单专项\-混合评分模型进度同步**
+
+**1、v1\.0版已敲定**
+**好单挑选 = 规则(70%) \+ 自动分析聚类(30%)**
+1\.1 平衡性较好:规则确保稳定性(不算命、不看天吃饭),聚类提供洞察力
+1\.2 可解释性强:每个预测都能追溯到具体特征贡献
+1\.3 反直觉发现:自动分析聚类30%权重能发现"高价但差单"、"低价但好单"等业务盲点
+1\.4 权重可调:70:30配比经过验证,且支持后续调优
+
+**2、自检报告已出**
+根据北上广深202501\-202505训练数据52万单,当前版本自检(机器自我检查)好单预测率:约60%(再精调一周,到70%\-90%后可生产稳定使用)
+
+\-\-\-详细的规则公示和过程说明感兴趣后续可查阅文档,还在持续进化中\.\.\.
+
+
+
+1、可靠性:好单预测率已由60%提升到70\-90%
+
+2、目前预测的好单给人直观感受(和大盘所有订单平均水平比)为:
+
+- 价值维度
+
+ - 订单金额:154 vs 120 → 高28%
+
+ - 单价水平:124 vs 100 → 高24%
+
+ - 商品数量:1\.7件 vs 1\.4件 → 多21%
+
+- 响应维度
+
+ - 查看报价率:47% vs 38% → 高24%
+
+ - 5人报价时长:2900秒 vs 4779秒 → 快39%
+
+ - 10人报价时长:4154秒 vs 6000秒\+ → 快31%
+
+ - 报价师傅数:9\.3人 vs 8\.2人 → 多13%
+
+ - 师傅关注数:241人 vs 150人 → 高61%
+
+- 效率维度
+
+ - 完工时长:5\.3小时 vs 6\.1小时 → 快13%
+
+- 风险维度
+
+ - 商家售后率:0\.7% vs 1\.0% → 低30%
+
+ - 被拉黑风险:与平均持平(商家质量稳定)
+
+- 好单的核心优势:
+
+ 1. **多**:价值高28% \- 订单含金量更高
+
+ 2. **快**:响应高30% \- 师傅抢单更积极
+
+ 3. **好**:关注高61% \- 市场热度明显更高
+
+ 4. **稳**:风险低30% \- 商家售后问题更少
+
+
+
+
+
+**特征权重**
+
+
+
+## 3\.3 筛选出100个list
+
+[版本二(全自动\-静态特征)数据结果抽查](https://jiqmwlmd0v.feishu.cn/wiki/ChJpwdp4ZitSJmk1SK5c5ooVnOb?from=from_copylink)
+
+[版本三(半自动\-动态特征)数据结构抽查](https://jiqmwlmd0v.feishu.cn/wiki/QbkiwrSZui4sg3kdOlScyMQKnsf?from=from_copylink)
+
+
+
+# **5\. 价值与应用**
+
+**先用在合作经营\-好单专区**
+
+
+
+
+
+
+
+# 6\.接口调用
+
+[订单分类推理服务接口文档](https://jiqmwlmd0v.feishu.cn/wiki/IKG4wPJJgi67qkklyvhceuB5nDf?from=from_copylink)
+
+
+
+
+
+
+