DEB 降级为模型集群中的普通一员,AI 不再照搬 DEB 做预测

此前 DEB 以独立字段 deb.prediction 传给 AI,提示词又要求"必须综合 DEB",
导致 AI 直接输出 DEB 值而非独立判断。改动:
- 移除 AI 输入中的独立 deb 字段,DEB 只作为 model_cluster.sources 中的
  一条记录 (model: "DEB (fusion)"),与其他模型平等
- 提示词改为:以模型集群集中区间为基线,用报文观测信号独立判断上修/下修/维持
- 流式/非流式提示词均强调"不要直接照搬 DEB 的值,差异是正常的"
- 回退路径改用模型集群中位数作为默认预测,DEB 仅作为备选参考
This commit is contained in:
2569718930@qq.com
2026-05-06 16:03:22 +08:00
parent f9d0ea31bc
commit eb42d644a3
4 changed files with 83 additions and 47 deletions
+39 -12
View File
@@ -39,26 +39,37 @@ def _city_ai_model_cluster_note(ai_input: Dict[str, Any], *, locale: str) -> str
if isinstance(item, dict) and _safe_float(item.get("value")) is not None
]
count = len(values)
deb_value = _safe_float((ai_input.get("deb") or {}).get("prediction") if isinstance(ai_input.get("deb"), dict) else None)
deb_value = next(
(_safe_float(item.get("value")) for item in sources if isinstance(item, dict) and "DEB" in str(item.get("model") or "")),
None,
)
non_deb_values = [
_safe_float(item.get("value"))
for item in sources
if isinstance(item, dict) and _safe_float(item.get("value")) is not None and "DEB" not in str(item.get("model") or "")
]
cluster_median = (
sorted(non_deb_values)[len(non_deb_values) // 2]
if non_deb_values
else (values[0] if values else None)
)
if locale == "en-US":
if count <= 0:
return f"No usable model cluster was returned; rely on DEB and {observation_label} only."
return f"No usable model cluster was returned; rely on {observation_label} only."
if count <= 2:
return f"Only {count} model source(s) are available, so model support is thin and should be treated as context."
range_text = f"{min(values):.1f}{unit} to {max(values):.1f}{unit}"
if deb_value is None:
return f"{count} model sources cluster between {range_text}; DEB support cannot be cross-checked."
supporting = sum(1 for value in values if abs(value - deb_value) <= 2.0)
return f"{supporting}/{count} model sources sit within 2{unit} of DEB; model range is {range_text}."
return f"{count} model sources cluster between {range_text}."
return f"{count} model sources cluster between {range_text}; DEB sits at {deb_value:.1f}{unit}."
if count <= 0:
return f"没有可用的多模型集合,只能把 DEB 与{observation_label}作为主要依据。"
return f"没有可用的多模型集合,只能把{observation_label}作为主要依据。"
if count <= 2:
return f"当前只有 {count} 个模型来源,模型支撑偏薄,只能作为辅助上下文。"
range_text = f"{min(values):.1f}{unit} ~ {max(values):.1f}{unit}"
if deb_value is None:
return f"{count} 个模型集中在 {range_text},但无法与 DEB 做一致性校验"
supporting = sum(1 for value in values if abs(value - deb_value) <= 2.0)
return f"{supporting}/{count} 个模型落在 DEB ±2{unit} 内;模型区间为 {range_text}"
return f"{count} 个模型集中在 {range_text}"
return f"{count} 个模型集中在 {range_text}DEB 位于 {deb_value:.1f}{unit}"
def _build_city_ai_fallback(
@@ -71,12 +82,26 @@ def _build_city_ai_fallback(
) -> Dict[str, Any]:
unit = str(ai_input.get("temp_symbol") or "")
cluster = ai_input.get("model_cluster") if isinstance(ai_input.get("model_cluster"), dict) else {}
all_sources = cluster.get("sources") if isinstance(cluster.get("sources"), list) else []
values = [
_safe_float(item.get("value"))
for item in (cluster.get("sources") if isinstance(cluster.get("sources"), list) else [])
for item in all_sources
if isinstance(item, dict) and _safe_float(item.get("value")) is not None
]
deb_value = _safe_float((ai_input.get("deb") or {}).get("prediction") if isinstance(ai_input.get("deb"), dict) else None)
deb_value = next(
(_safe_float(item.get("value")) for item in all_sources if isinstance(item, dict) and "DEB" in str(item.get("model") or "")),
None,
)
non_deb_values = [
_safe_float(item.get("value"))
for item in all_sources
if isinstance(item, dict) and _safe_float(item.get("value")) is not None and "DEB" not in str(item.get("model") or "")
]
cluster_median = (
sorted(non_deb_values)[len(non_deb_values) // 2]
if non_deb_values
else (sorted(values)[len(values) // 2] if values else None)
)
observation_anchor = ai_input.get("observation_anchor") if isinstance(ai_input.get("observation_anchor"), dict) else {}
is_airport_metar = observation_anchor.get("is_airport_metar") is not False
airport_current = ai_input.get("airport_current") if isinstance(ai_input.get("airport_current"), dict) else {}
@@ -100,7 +125,9 @@ def _build_city_ai_fallback(
default=None,
)
observed_high_for_revision = None if observation_stale else observed_high_so_far
predicted = deb_value
predicted = cluster_median
if predicted is None and deb_value is not None:
predicted = deb_value
if predicted is None and values:
predicted = sum(values) / len(values)
if predicted is None:
+35 -25
View File
@@ -24,7 +24,7 @@ def _observation_prompt_context(ai_input: Dict[str, Any]) -> Dict[str, Any]:
return {
"anchor": observation_anchor,
"is_airport_metar": is_airport_metar,
"read_label_zh": str(observation_anchor.get("read_label_zh") or ("机场报文解读" if is_airport_metar else "官方观测解读")),
"read_label_zh": str(observation_anchor.get("read_label_zh") or ("\u673a\u573a\u62a5\u6587\u89e3\u8bfb" if is_airport_metar else "\u5b98\u65b9\u89c2\u6d4b\u89e3\u8bfb")),
"instruction_zh": str(observation_anchor.get("instruction_zh") or ""),
}
@@ -41,24 +41,28 @@ def build_city_ai_request_json(
observation_label_zh = context["read_label_zh"]
observation_instruction = context["instruction_zh"]
system_prompt = (
"你是 PolyWeather 的城市最高温 AI 预测员。你必须直接阅读用户给出的城市 JSON,"
"判断该城市今日最高温路径。不要写套利、交易、BUY YES/NO、价格、edge 或 Kelly。"
f"你的核心输出是:最终最高温点估计、置信区间、置信度、最终判断、{observation_label_zh}、判断依据和风险。"
"必须综合 DEB 最终融合值、全部天气模型预测、实测序列、最新观测、峰值窗口、当地时间、季节背景。"
"\u4f60\u662f PolyWeather \u7684\u57ce\u5e02\u6700\u9ad8\u6e29 AI \u9884\u6d4b\u5458\u3002\u4f60\u5fc5\u987b\u76f4\u63a5\u9605\u8bfb\u7528\u6237\u7ed9\u51fa\u7684\u57ce\u5e02 JSON\uff0c"
"\u72ec\u7acb\u5224\u65ad\u8be5\u57ce\u5e02\u4eca\u65e5\u6700\u9ad8\u6e29\u8def\u5f84\u3002\u4e0d\u8981\u5199\u5957\u5229\u3001\u4ea4\u6613\u3001BUY YES/NO\u3001\u4ef7\u683c\u3001edge \u6216 Kelly\u3002"
f"\u4f60\u7684\u6838\u5fc3\u8f93\u51fa\u662f\uff1a\u6700\u7ec8\u6700\u9ad8\u6e29\u70b9\u4f30\u8ba1\u3001\u7f6e\u4fe1\u533a\u95f4\u3001\u7f6e\u4fe1\u5ea6\u3001\u6700\u7ec8\u5224\u65ad\u3001{observation_label_zh}\u3001\u5224\u65ad\u4f9d\u636e\u548c\u98ce\u9669\u3002"
"\u9884\u6d4b\u65b9\u6cd5\uff1a\u9996\u5148\u67e5\u770b model_cluster.sources \u4e2d\u7684\u5168\u90e8\u6a21\u578b\uff08\u542b DEB\uff09\u7684\u96c6\u4e2d\u533a\u95f4\u548c\u4e2d\u4f4d\u6570\u4f5c\u4e3a\u57fa\u7ebf\uff1b"
"\u7136\u540e\u91cd\u70b9\u9605\u8bfb metar_context \u4e2d\u7684\u6700\u65b0\u89c2\u6d4b/\u62a5\u6587\uff08\u6e29\u5ea6\u8d8b\u52bf\u3001\u98ce\u5411\u98ce\u901f\u3001\u4e91\u91cf\u3001\u80fd\u89c1\u5ea6\u3001\u9732\u70b9\uff09\uff0c"
"\u6839\u636e\u5b9e\u6d4b\u4fe1\u53f7\u72ec\u7acb\u5224\u65ad\u57fa\u7ebf\u5e94\u8be5\u662f\u4e0a\u4fee\u3001\u4e0b\u4fee\u8fd8\u662f\u7ef4\u6301\uff0c\u5e76\u5728 predicted_max \u4e2d\u7ed9\u51fa\u4f60\u7684\u72ec\u7acb\u5224\u65ad\u3002"
"DEB \u53ea\u662f\u6a21\u578b\u96c6\u7fa4\u4e2d\u7684\u4e00\u4e2a\u878d\u5408\u53c2\u8003\uff0c\u4e0d\u5e94\u8be5\u76f4\u63a5\u7167\u642c\u4e3a predicted_max\uff1b"
"\u4f60\u5fc5\u987b\u7efc\u5408\u6240\u6709\u6a21\u578b + \u89c2\u6d4b\u4fe1\u53f7\u540e\u7ed9\u51fa\u81ea\u5df1\u7684\u6570\u5b57\uff0c\u4e0e DEB \u6709\u5dee\u5f02\u662f\u6b63\u5e38\u7684\u3002"
f"{observation_instruction}"
"如果实测温度与 DEB 预测走势出现偏差,要明确说明偏差方向和可能修正。"
"你可以基于城市、时间、季节、站点位置、风向/风速、云、能见度、露点等判断风或天气是否可能影响温度路径,"
"但必须使用“可能”“倾向”“需要确认”等非绝对表达。"
"观测解读必须具体:写清楚最新观测/报文时间、温度、风向风速、云量/天气、能见度或露点中与温度路径相关的因素。"
"涉及风时必须说明该风向对本城市/机场最高温路径倾向增温、降温还是中性,并给出理由;"
"不得只写“风向切换可能冷平流”,必须说明是哪一类风向或哪段风向切换可能带来冷/暖平流。"
"涉及 TAF 或云雨扰动时必须给出报文中的有效时间、BECMG/TEMPO/FM 时间窗或说明“未给出明确时间”;"
"如果没有 TAF 时间依据,不要笼统写“峰值窗口云雨扰动风险”。"
"如果峰值窗口尚未到来,不能过早下最终结论;如果峰值窗口已过或实测已创高,需要更重视最新实测。"
"所有面向用户的自然语言字段必须同时填写简体中文和英文两套内容:"
"_zh 字段写简体中文,_en 字段写英文。前端会按用户界面语言直接切换字段,不能留空。"
"risks 最多 2 条,每条必须包含触发条件或方向来源;reasoning、model_cluster_note 各 1 句,metar_read 用 1-2 句。"
"只返回 JSON object,不要 Markdown。"
"\u5982\u679c\u5b9e\u6d4b\u6e29\u5ea6\u4e0e\u6a21\u578b\u96c6\u7fa4\u8d70\u52bf\u51fa\u73b0\u504f\u5dee\uff0c\u8981\u660e\u786e\u8bf4\u660e\u504f\u5dee\u65b9\u5411\u548c\u53ef\u80fd\u4fee\u6b63\u3002"
"\u4f60\u53ef\u4ee5\u57fa\u4e8e\u57ce\u5e02\u3001\u65f6\u95f4\u3001\u5b63\u8282\u3001\u7ad9\u70b9\u4f4d\u7f6e\u3001\u98ce\u5411/\u98ce\u901f\u3001\u4e91\u3001\u80fd\u89c1\u5ea6\u3001\u9732\u70b9\u7b49\u5224\u65ad\u98ce\u6216\u5929\u6c14\u662f\u5426\u53ef\u80fd\u5f71\u54cd\u6e29\u5ea6\u8def\u5f84\uff0c"
"\u4f46\u5fc5\u987b\u4f7f\u7528\u201c\u53ef\u80fd\u201d\u201c\u503e\u5411\u201d\u201c\u9700\u8981\u786e\u8ba4\u201d\u7b49\u975e\u7edd\u5bf9\u8868\u8fbe\u3002"
"\u89c2\u6d4b\u89e3\u8bfb\u5fc5\u987b\u5177\u4f53\uff1a\u5199\u6e05\u695a\u6700\u65b0\u89c2\u6d4b/\u62a5\u6587\u65f6\u95f4\u3001\u6e29\u5ea6\u3001\u98ce\u5411\u98ce\u901f\u3001\u4e91\u91cf/\u5929\u6c14\u3001\u80fd\u89c1\u5ea6\u6216\u9732\u70b9\u4e2d\u4e0e\u6e29\u5ea6\u8def\u5f84\u76f8\u5173\u7684\u56e0\u7d20\u3002"
"\u6d89\u53ca\u98ce\u65f6\u5fc5\u987b\u8bf4\u660e\u8be5\u98ce\u5411\u5bf9\u672c\u57ce\u5e02/\u673a\u573a\u6700\u9ad8\u6e29\u8def\u5f84\u503e\u5411\u589e\u6e29\u3001\u964d\u6e29\u8fd8\u662f\u4e2d\u6027\uff0c\u5e76\u7ed9\u51fa\u7406\u7531\uff1b"
"\u4e0d\u5f97\u53ea\u5199\u201c\u98ce\u5411\u5207\u6362\u53ef\u80fd\u51b7\u5e73\u6d41\u201d\uff0c\u5fc5\u987b\u8bf4\u660e\u662f\u54ea\u4e00\u7c7b\u98ce\u5411\u6216\u54ea\u6bb5\u98ce\u5411\u5207\u6362\u53ef\u80fd\u5e26\u6765\u51b7/\u6696\u5e73\u6d41\u3002"
"\u6d89\u53ca TAF \u6216\u4e91\u96e8\u6270\u52a8\u65f6\u5fc5\u987b\u7ed9\u51fa\u62a5\u6587\u4e2d\u7684\u6709\u6548\u65f6\u95f4\u3001BECMG/TEMPO/FM \u65f6\u95f4\u7a97\u6216\u8bf4\u660e\u201c\u672a\u7ed9\u51fa\u660e\u786e\u65f6\u95f4\u201d\uff1b"
"\u5982\u679c\u6ca1\u6709 TAF \u65f6\u95f4\u4f9d\u636e\uff0c\u4e0d\u8981\u7b3c\u7edf\u5199\u201c\u5cf0\u503c\u7a97\u53e3\u4e91\u96e8\u6270\u52a8\u98ce\u9669\u201d\u3002"
"\u5982\u679c\u5cf0\u503c\u7a97\u53e3\u5c1a\u672a\u5230\u6765\uff0c\u4e0d\u80fd\u8fc7\u65e9\u4e0b\u6700\u7ec8\u7ed3\u8bba\uff1b\u5982\u679c\u5cf0\u503c\u7a97\u53e3\u5df2\u8fc7\u6216\u5b9e\u6d4b\u5df2\u521b\u9ad8\uff0c\u9700\u8981\u66f4\u91cd\u89c6\u6700\u65b0\u5b9e\u6d4b\u3002"
"\u6240\u6709\u9762\u5411\u7528\u6237\u7684\u81ea\u7136\u8bed\u8a00\u5b57\u6bb5\u5fc5\u987b\u540c\u65f6\u586b\u5199\u7b80\u4f53\u4e2d\u6587\u548c\u82f1\u6587\u4e24\u5957\u5185\u5bb9\uff1a"
"_zh \u5b57\u6bb5\u5199\u7b80\u4f53\u4e2d\u6587\uff0c_en \u5b57\u6bb5\u5199\u82f1\u6587\u3002\u524d\u7aef\u4f1a\u6309\u7528\u6237\u754c\u9762\u8bed\u8a00\u76f4\u63a5\u5207\u6362\u5b57\u6bb5\uff0c\u4e0d\u80fd\u7559\u7a7a\u3002"
"risks \u6700\u591a 2 \u6761\uff0c\u6bcf\u6761\u5fc5\u987b\u5305\u542b\u89e6\u53d1\u6761\u4ef6\u6216\u65b9\u5411\u6765\u6e90\uff1breasoning\u3001model_cluster_note \u5404 1 \u53e5\uff0cmetar_read \u7528 1-2 \u53e5\u3002"
"\u53ea\u8fd4\u56de JSON object\uff0c\u4e0d\u8981 Markdown\u3002"
)
user_payload = {
"locale": normalized_locale,
@@ -74,11 +78,12 @@ def build_city_ai_request_json(
"and why in local city/station context. If mentioning cold/warm advection, name the wind direction or "
"direction shift responsible. If mentioning TAF risk, include the concrete TAF time window or say no "
"explicit timing is available. model_cluster_note must state "
"how many model sources are available, whether they support DEB, and whether the sample is too sparse. "
"how many model sources are available, the cluster range and median, whether the spread is tight or wide, "
"and whether you adjusted above or below the cluster median based on observation signals. "
"Keep the whole JSON compact."
),
"required_json_keys": CITY_AI_REQUIRED_FIELDS,
"json_example": _city_ai_response_example(str(ai_input.get("temp_symbol") or "°C")),
"json_example": _city_ai_response_example(str(ai_input.get("temp_symbol") or "\u00b0C")),
"city_snapshot": ai_input,
}
return {
@@ -203,14 +208,17 @@ def build_city_ai_stream_request(
system_prompt = (
f"你是 PolyWeather 的{role_label}"
"只返回一个紧凑 JSON object,不要 Markdown。"
f"必须基于最新观测/报文判断该城市今日最高温,输出 metar_read_zh、metar_read_en、predicted_max、range_low、range_high、unit、confidence、final_judgment_zh、final_judgment_en、reasoning_zh、reasoning_en 字段,便于前端快速显示{read_label}和最高温预测;"
f"必须基于最新观测/报文独立判断该城市今日最高温,输出 metar_read_zh、metar_read_en、predicted_max、range_low、range_high、unit、confidence、final_judgment_zh、final_judgment_en、reasoning_zh、reasoning_en 字段,便于前端快速显示{read_label}和最高温预测;"
"模型一致性和风险清单由后端规则补齐,不要生成这些字段。"
f"预测方法:先看 model_cluster.sources 中各模型(含 DEB)的集中区间作为基线;"
"然后重点阅读 metar_context 中的报文/观测信号——温度趋势、风向风速、云量、能见度——"
"独立判断 predicted_max 应该落在基线区间的哪个位置(偏上/偏下/中枢),不要直接照搬 DEB 的值。"
f"{source_instruction}"
"观测解读必须具体说明观测/报文时间、温度、风向风速、云量/天气/能见度/露点中与温度路径相关的因素;每个字段最多 1-2 句。"
"涉及风时要说明当前风向对站点最高温路径倾向增温、降温还是中性,并给出理由。"
"如果 observation_anchor.is_airport_metar 为 false,不得使用 METAR、TAF、机场报文等称谓。"
"predicted_max 是根据最新观测和模型趋势给出的今日最高温数值预测(float),range_low/range_high 是预测区间,unit 为温度单位,confidence 为 low/medium/high。"
"final_judgment 用一句话给出今日最高温结论。"
"predicted_max 是你的独立预测float),range_low/range_high 是预测区间,unit 为温度单位,confidence 为 low/medium/high。"
"final_judgment 用一句话给出今日最高温结论。reasoning 必须解释你相对于模型集群基线做了哪种修正及原因。"
"所有 *_zh 字段写简体中文,所有 *_en 字段写英文,不得留空。"
"不要写交易建议、BUY/SELL、Kelly 或套利。"
)
@@ -229,13 +237,15 @@ def build_city_ai_stream_request(
"locale": normalized_locale,
"task": (
"Return JSON keys in this exact order: metar_read_zh, metar_read_en, predicted_max, range_low, range_high, unit, confidence, final_judgment_zh, final_judgment_en, reasoning_zh, reasoning_en. "
"predicted_max must be a float number representing today's predicted daily high temperature based on the latest observation and model trends. "
"predicted_max must be your independent float prediction, based on model cluster baseline adjusted by the latest METAR/observation signals. "
"Do not copy DEB directly — use the full model spread + your own reading of wind, cloud, temperature trend from the bulletin. "
"reasoning must explain what adjustment you made relative to the model cluster and why. "
"Do not return risks or model_cluster_note. Keep it compact. "
"Return exactly one JSON object and no markdown."
),
"required_json_keys": CITY_AI_STREAM_PROVIDER_FIELDS,
"json_example": _city_ai_stream_response_example(
str(ai_input.get("temp_symbol") or "°C")
str(ai_input.get("temp_symbol") or "\u00b0C")
),
"city_snapshot": ai_input,
},
-1
View File
@@ -356,7 +356,6 @@ def _compact_ai_city_group(rows: List[Dict[str, Any]]) -> Dict[str, Any]:
"temp_symbol": first.get("temp_symbol") or first.get("target_unit"),
"current_temp": first.get("current_temp"),
"current_max_so_far": first.get("current_max_so_far"),
"deb_prediction": first.get("deb_prediction"),
"window_phase": first.get("window_phase"),
"remaining_window_minutes": first.get("remaining_window_minutes"),
"peak_window_label": first.get("peak_window_label"),
+9 -9
View File
@@ -354,18 +354,18 @@ def _build_city_ai_prompt(data: Dict[str, Any]) -> Dict[str, Any]:
"icao": risk.get("icao") or airport_current.get("station_code") or airport_primary.get("station_code"),
"distance_km": risk.get("distance_km"),
},
"deb": {
"prediction": ((daily_entry.get("deb") or {}).get("prediction") if isinstance(daily_entry.get("deb"), dict) else None)
or ((data.get("deb") or {}).get("prediction") if isinstance(data.get("deb"), dict) else None),
"weights_info": ((data.get("deb") or {}).get("weights_info") if isinstance(data.get("deb"), dict) else None),
},
"model_cluster": {
"sources": [
{"model": str(name), "value": value}
for name, value in (models or {}).items()
if _safe_float(value) is not None
*([
{"model": "DEB (fusion)", "value": ((daily_entry.get("deb") or {}).get("prediction") if isinstance(daily_entry.get("deb"), dict) else None) or ((data.get("deb") or {}).get("prediction") if isinstance(data.get("deb"), dict) else None)}
] if (((daily_entry.get("deb") or {}).get("prediction") if isinstance(daily_entry.get("deb"), dict) else None) or ((data.get("deb") or {}).get("prediction") if isinstance(data.get("deb"), dict) else None)) is not None else []),
*[
{"model": str(name), "value": value}
for name, value in (models or {}).items()
if _safe_float(value) is not None
],
],
"model_count": len(model_values),
"model_count": len(model_values) + (1 if (((daily_entry.get("deb") or {}).get("prediction") if isinstance(daily_entry.get("deb"), dict) else None) or ((data.get("deb") or {}).get("prediction") if isinstance(data.get("deb"), dict) else None)) is not None else 0),
"min": min(model_values) if model_values else None,
"max": max(model_values) if model_values else None,
"spread": (max(model_values) - min(model_values)) if len(model_values) >= 2 else None,