# -*- coding: utf-8 -*- """evolution/hypothesis_miner.py — 数据归纳假设引擎 v2 从策略最新交易数据 + 面板特征,归纳可描述的优化假设(方向一核心) 数据源:strategy_research trades(code+date)→ 关联 panel_12d 的入场日特征 """ import json import sqlite3 import pandas as pd # 可归纳特征:panel 字段名 + 标签 + 高值是否坏 CANDIDATE_FEATURES = [ ("mkt_adx", "大盘趋势强度ADX", True), ("mkt_rsi", "大盘RSI", True), ("mkt_ret20", "大盘近20日涨幅", False), ("bias60", "个股60日偏离", True), ("rsi", "个股RSI", True), ("vol_ratio", "量比", False), ("sec_ret20", "行业近20日涨幅", False), ] _PANEL_CACHE = {} # 模块级缓存:market -> panel(避免每次重载600万行pkl) def _load_panel(market): """加载面板:A股 panel_12d.pkl,港股 panel_12d_hk.pkl(带缓存)""" global _PANEL_CACHE if market in _PANEL_CACHE: return _PANEL_CACHE[market] path = "/tmp/panel_12d_hk.pkl" if market == "hk" else "/tmp/panel_12d.pkl" p = pd.read_pickle(path) p = p.sort_values(["code", "date"]).reset_index(drop=True) # 建 code+date → 特征映射 p["_key"] = p["code"].astype(str) + "_" + p["date"].astype(str) p = p.drop_duplicates(subset=["_key"]) p = p.set_index("_key") _PANEL_CACHE[market] = p return p def load_trades(version, market, period_tag="2y"): """从 strategy_research 读 trades,关联面板特征""" conn = sqlite3.connect("/home/hmo/MoFin/data/mofin.db", timeout=10) conn.row_factory = sqlite3.Row r = conn.execute( "SELECT results_json FROM strategy_research WHERE version=? AND market=? AND period_tag=? " "ORDER BY id DESC LIMIT 1", (version, market, period_tag)).fetchone() conn.close() if not r: return [] res = json.loads(r["results_json"] or "{}") trades = res.get("trades", []) try: panel = _load_panel(market) except Exception as e: print(f"panel 加载失败: {e}", flush=True) return [] out = [] for t in trades: key = str(t.get("code")) + "_" + str(t.get("entry_date")) row = panel.loc[key] if key in panel.index else None out.append({ "profit_pct": t.get("profit_pct", 0), "win": t.get("profit_pct", 0) > 0, "hold_days": t.get("hold_days", 0), "exit_reason": t.get("exit_reason", ""), "mkt_adx": row["mkt_adx"] if row is not None and pd.notna(row.get("mkt_adx")) else None, "mkt_rsi": row["mkt_rsi"] if row is not None and pd.notna(row.get("mkt_rsi")) else None, "mkt_ret20": row["mkt_ret20"] if row is not None and pd.notna(row.get("mkt_ret20")) else None, "bias60": row["bias60"] if row is not None and pd.notna(row.get("bias60")) else None, "rsi": row["rsi"] if row is not None and pd.notna(row.get("rsi")) else None, "vol_ratio": row["vol_ratio"] if row is not None and pd.notna(row.get("vol_ratio")) else None, "sec_ret20": row["sec_ret20"] if row is not None and pd.notna(row.get("sec_ret20")) else None, }) return out def _percentile(vals, p): if not vals: return None s = sorted(vals) return s[int((len(s) - 1) * p)] def induce_hypotheses(version, market, period_tag="2y", min_trades=20, min_effect=15): """归纳优化假设:找盈利/亏损组的特征差异""" trades = load_trades(version, market, period_tag) if len(trades) < min_trades: return [], trades overall_wr = sum(1 for t in trades if t["win"]) / len(trades) * 100 hypotheses = [] for feat_key, feat_label, high_is_bad in CANDIDATE_FEATURES: vals = [t[feat_key] for t in trades if t.get(feat_key) is not None] if len(vals) < max(5, min_trades * 0.3): continue hi = _percentile(vals, 0.75) lo = _percentile(vals, 0.25) if hi is None or lo is None or hi == lo: continue hi_trades = [t for t in trades if t.get(feat_key) is not None and t[feat_key] >= hi] lo_trades = [t for t in trades if t.get(feat_key) is not None and t[feat_key] <= lo] if len(hi_trades) < 5 or len(lo_trades) < 5: continue hi_wr = sum(1 for t in hi_trades if t["win"]) / len(hi_trades) * 100 lo_wr = sum(1 for t in lo_trades if t["win"]) / len(lo_trades) * 100 if high_is_bad and hi_wr < overall_wr - min_effect: hypotheses.append({ "hypothesis": f"当{feat_label}({feat_key})≥{hi:.1f}时胜率仅{hi_wr:.0f}%(整体{overall_wr:.0f}%),应规避", "feature": feat_key, "direction": "max", "threshold": round(hi, 2), "win_rate_affected": round(hi_wr, 1), "win_rate_clean": round(lo_wr, 1), "overall_wr": round(overall_wr, 1), "effect_pp": round(overall_wr - hi_wr, 1), "evidence": f"高分组{len(hi_trades)}笔胜率{hi_wr:.0f}% vs 低分组{len(lo_trades)}笔胜率{lo_wr:.0f}%", "trades_affected": len(hi_trades), }) elif not high_is_bad and lo_wr < overall_wr - min_effect: hypotheses.append({ "hypothesis": f"当{feat_label}({feat_key})≤{lo:.1f}时胜率仅{lo_wr:.0f}%(整体{overall_wr:.0f}%),应规避", "feature": feat_key, "direction": "min", "threshold": round(lo, 2), "win_rate_affected": round(lo_wr, 1), "win_rate_clean": round(hi_wr, 1), "overall_wr": round(overall_wr, 1), "effect_pp": round(overall_wr - lo_wr, 1), "evidence": f"低分组{len(lo_trades)}笔胜率{lo_wr:.0f}% vs 高分组{len(hi_trades)}笔胜率{hi_wr:.0f}%", "trades_affected": len(lo_trades), }) hypotheses.sort(key=lambda h: -h["effect_pp"]) return hypotheses, trades if __name__ == "__main__": import sys version = sys.argv[1] if len(sys.argv) > 1 else "v_oversold" market = sys.argv[2] if len(sys.argv) > 2 else "a" hs, trades = induce_hypotheses(version, market) print(f"=== {version} [{market}] {len(trades)}笔 ===") for h in hs: print(f" [Δ{h['effect_pp']:.0f}pp] {h['hypothesis']}") print(f" 证据: {h['evidence']}") if not hs: print(" 未归纳出显著假设")