File size: 19,471 Bytes
6724416 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 | from __future__ import annotations
import re
from collections import defaultdict
from typing import Any
from fastapi import HTTPException
from sidebar_catalog import (
HF_SOURCE_CATEGORIES as _HF_SOURCE_CATEGORIES,
HF_SOURCE_NAMES_ZH as _HF_SOURCE_NAMES_ZH,
OCEAN_CATALOG as _OCEAN_CATALOG,
)
def _project_package_plan(project: str) -> dict[str, Any]:
"""Build a reproducible project data plan from the user's project description.
v3.5.1 keeps the planner deterministic/offline so a recommendation can always be
reproduced, but upgrades it from a flat keyword list to a structured plan:
project intent -> requirements -> priority -> source mapping -> coverage hints ->
alternatives. Live file availability/size is still verified by the estimate/build
endpoints before downloading anything.
"""
text = str(project or "").strip()
q = text.lower()
if len(text) < 4:
raise HTTPException(400, "请至少用一句话描述项目目标。")
ocean_scores: dict[str, int] = defaultdict(int)
fish_scores: dict[str, int] = defaultdict(int)
reasons: dict[str, list[str]] = defaultdict(list)
requirement_tags: dict[str, set[str]] = defaultdict(set)
def add_ocean(key: str, score: int, reason: str, requirement: str = ""):
ocean_scores[key] += score
reasons["ocean:" + key].append(reason)
if requirement:
requirement_tags["ocean:" + key].add(requirement)
def add_fish(name: str, score: int, reason: str, requirement: str = ""):
fish_scores[name] += score
reasons["fish:" + name].append(reason)
if requirement:
requirement_tags["fish:" + name].add(requirement)
# ---- 1) Interpret the project itself -------------------------------------------------
species = []
species_rules = [
(("柔鱼", "茎柔鱼", "鱿鱼", "squid"), "柔鱼/鱿鱼"),
(("金枪鱼", "tuna"), "金枪鱼"),
(("黄鳍", "yellowfin"), "黄鳍金枪鱼"),
(("大眼", "bigeye"), "大眼金枪鱼"),
(("长鳍", "albacore"), "长鳍金枪鱼"),
(("蓝鳍", "bluefin"), "蓝鳍金枪鱼"),
(("鲣", "skipjack"), "鲣"),
]
for terms, label in species_rules:
if any(t in q for t in terms) and label not in species:
species.append(label)
task_rules = [
(("中尺度涡旋", "涡旋", "eddy", "mesoscale eddy"), "物理海洋过程研究"),
(("锋面", "海洋锋", "front"), "物理海洋过程研究"),
(("上升流", "upwelling"), "物理海洋过程研究"),
(("环流", "海流", "洋流", "circulation", "current"), "物理海洋过程研究"),
(("生境", "适生区", "habitat", "species distribution", "sdm", "maxent", "分布预测"), "生境/物种分布预测"),
(("资源评估", "种群评估", "stock assessment", "biomass", "资源量", "补充量"), "资源/种群评估"),
(("cpue", "努力量", "effort", "catch rate"), "CPUE/捕捞努力分析"),
(("渔场", "fishing ground", "捕捞热点", "hotspot"), "渔场/热点分析"),
(("气候", "enso", "气候变化", "climate"), "气候影响分析"),
(("趋势", "变化趋势", "时序", "time series"), "时间序列/趋势分析"),
]
task_type = "综合海洋分析"
for terms, label in task_rules:
if any(t in q for t in terms):
task_type = label
break
# Exact day, month, year and rough ranges. Keep exact date separately because Ocean
# export currently needs a single date.
date = ""
time_range = ""
m = re.search(r"(20\d{2})[-/年](1[0-2]|0?[1-9])[-/月](3[01]|[12]\d|0?[1-9])", text)
if m:
date = f"{int(m.group(1)):04d}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
time_range = date
else:
compact = re.search(r"20\d{6}", text)
if compact:
raw = compact.group(0)
date = f"{raw[:4]}-{raw[4:6]}-{raw[6:8]}"
time_range = date
else:
yr = re.search(r"(20\d{2})\s*(?:年)?\s*(?:[-–—至到~~]\s*(20\d{2})\s*年?)?", text)
mo = re.search(r"(20\d{2})[-/年](1[0-2]|0?[1-9])(?:月)?", text)
if mo:
time_range = f"{int(mo.group(1)):04d}-{int(mo.group(2)):02d}"
elif yr:
time_range = yr.group(1) + ((" 至 " + yr.group(2)) if yr.group(2) else "")
regions = [
(("南海",), [99.0, 124.0, 0.0, 25.0], "南海"),
(("东海",), [118.0, 132.0, 23.0, 34.0], "东海"),
(("黄海",), [117.0, 126.0, 30.0, 41.0], "黄海"),
(("西北太平洋",), [120.0, 180.0, 10.0, 50.0], "西北太平洋"),
(("北太平洋",), [120.0, -100.0, 0.0, 60.0], "北太平洋"),
# “太平洋”跨越日期变更线,不能安全地用一个普通 west<east bbox 表示。
# 先识别语义区域,但要求用户在真正导出前补充子区域或经纬度范围。
(("太平洋", "pacific"), [], "太平洋"),
(("印度洋",), [20.0, 120.0, -50.0, 30.0], "印度洋"),
(("大西洋",), [-80.0, 20.0, -60.0, 60.0], "大西洋"),
(("全球", "global"), [-180.0, 180.0, -90.0, 90.0], "全球"),
]
bbox: list[float] = []
region_name = ""
for terms, b, label in regions:
if any(t in text.lower() for t in terms):
bbox, region_name = b, label
break
# ---- 2) Translate research goals into data requirements -------------------------------
reqs: list[dict[str, Any]] = []
def requirement(name: str, priority: str, why: str, variables: list[str], domain: str):
if any(x["name"] == name for x in reqs):
return
reqs.append({"name": name, "priority": priority, "why": why, "variables": variables, "domain": domain})
habitat_intent = any(t in q for t in ("生境", "适生区", "分布预测", "maxent", "物种分布", "环境驱动", "habitat", "sdm"))
tuna_intent = any(t in q for t in ("tuna", "金枪鱼", "鲣", "黄鳍", "大眼", "长鳍", "蓝鳍"))
squid_intent = any(t in q for t in ("squid", "鱿鱼", "柔鱼", "茎柔鱼", "赤鱿"))
effort_intent = any(t in q for t in ("cpue", "努力量", "捕捞量", "catch", "effort", "渔获", "捕捞记录"))
stock_intent = any(t in q for t in ("资源评估", "种群评估", "stock assessment", "biomass", "资源量", "补充量"))
vessel_intent = any(t in q for t in ("渔船", "ais", "捕捞活动", "船舶活动", "夜光", "viirs"))
eddy_intent = any(t in q for t in ("中尺度涡旋", "涡旋", "eddy", "mesoscale eddy"))
physical_ocean_intent = eddy_intent or any(t in q for t in ("锋面", "上升流", "环流", "海流", "洋流", "front", "upwelling", "circulation"))
fisheries_intent = bool(tuna_intent or squid_intent or effort_intent or stock_intent or vessel_intent or any(t in q for t in ("渔业", "渔场", "捕捞", "渔获", "fishery", "fisheries")))
keyword_ocean = [
(("sst", "海温", "水温", "温度"), "oisst", 9, "海表温度是项目明确提出的环境因子", "SST"),
(("盐度", "salinity"), "cmems_physics", 8, "项目明确需要盐度", "盐度"),
(("流速", "海流", "环流", "uo", "vo", "地转流"), "cmems_physics", 8, "项目明确需要海流/环流", "海流"),
(("叶绿素", "chlorophyll", "chl"), "cmems_bgc", 9, "项目明确需要叶绿素/生产力指标", "叶绿素"),
(("营养盐", "no3", "硝酸盐", "磷酸盐"), "cmems_bgc", 8, "项目明确需要营养盐", "营养盐"),
(("溶解氧", "oxygen", " do "), "cmems_bgc", 9, "项目明确需要溶解氧", "溶解氧"),
(("海色", "遥感叶绿素", "oc-cci", "occci"), "occci", 8, "项目明确需要海色遥感", "海色"),
(("风", "风速", "风场", "气温", "气压", "era5", "气象"), "era5", 7, "项目明确需要大气驱动", "风场/气象"),
(("降水", "蒸发", "辐射", "热通量"), "era5_accum", 7, "项目明确需要累积大气量/通量", "大气通量"),
(("混合层", "mlotst"), "cmems_surface", 8, "项目明确需要混合层结构", "混合层"),
(("海面高度", "ssh", "sla", "adt", "zos"), "cmems_surface", 8, "项目明确需要海面高度/高度异常", "SSH/SLA/ADT"),
(("酸化", "ph", "co2", "碳酸盐", "spco2"), "cmems_carbonate", 8, "项目明确需要碳酸盐系统", "碳酸盐"),
]
for terms, key, score, reason, req in keyword_ocean:
if any(t in q for t in terms):
add_ocean(key, score, reason, req)
requirement(req, "必需", reason, [req], "Ocean")
# Domain defaults are not merely keywords: they encode common analysis dependencies.
# Physical-ocean process tasks must stay Ocean-only unless the user explicitly asks
# for a fisheries/ecology response variable.
if eddy_intent:
requirement("SSH / SLA", "必需", "中尺度涡旋通常首先由海表高度或海表高度异常识别与追踪", ["SSH", "SLA", "zos"], "Ocean")
requirement("ADT / 海表动力高度", "推荐", "用于描述涡旋绝对动力高度结构;具体可用变量以实时产品为准", ["ADT", "zos"], "Ocean")
requirement("地转流 / 海流 U-V", "推荐", "用于判断旋转方向、流速和涡旋动力结构", ["uo", "vo"], "Ocean")
requirement("SST", "推荐", "用于分析暖涡/冷涡对应的海表温度异常", ["SST", "anom"], "Ocean")
requirement("叶绿素 / 海色", "可选", "只有研究涡旋生态或初级生产力响应时才需要", ["CHL", "chlor_a"], "Ocean")
requirement("温盐三维剖面", "可选", "研究涡旋垂向结构、温跃层或水团特征时需要", ["thetao", "so"], "Ocean")
add_ocean("cmems_surface", 10, "中尺度涡旋识别的核心海表高度/上层海洋场", "SSH / SLA")
add_ocean("cmems_physics", 7, "涡旋动力结构和三维温盐流场", "地转流 / 海流 U-V")
add_ocean("oisst", 6, "暖涡/冷涡海表温度异常分析", "SST")
add_ocean("occci", 3, "仅用于涡旋生态响应或生产力研究", "叶绿素 / 海色")
if habitat_intent:
requirement("SST", "必需", "生境模型通常需要基础热环境", ["SST"], "Ocean")
if fisheries_intent or species:
requirement("渔业出现/捕捞记录", "必需", "生境模型需要物种出现或捕捞位置作为响应数据", ["occurrence", "catch", "effort"], "Fisheries")
requirement("生产力/叶绿素", "推荐", "用于描述食物资源与生产力环境", ["CHL", "NPP"], "Ocean")
requirement("上层海洋结构", "推荐", "SSH/混合层可补充水团与锋面信息", ["SSH", "MLD"], "Ocean")
requirement("海流", "推荐", "海流影响洄游、输运和锋面结构", ["uo", "vo"], "Ocean")
add_ocean("oisst", 7, "生境模型基础热环境", "SST")
add_ocean("cmems_bgc", 5, "生境模型生产力与生态环境", "生产力/叶绿素")
add_ocean("cmems_surface", 4, "生境模型上层海洋结构", "上层海洋结构")
add_ocean("cmems_physics", 4, "生境模型海流与物理环境", "海流")
if "气候" in task_type:
requirement("大气驱动", "推荐", "用于解释年际/气候尺度变化", ["wind", "air temperature", "pressure"], "Ocean")
add_ocean("era5", 5, "气候影响分析的大气驱动", "大气驱动")
if tuna_intent:
requirement("金枪鱼区域渔业记录", "必需", "研究对象为金枪鱼,需要区域性渔获/努力量记录", ["catch", "effort", "CPUE"], "Fisheries")
for n in ("WCPFC", "IATTC", "ICCAT", "IOTC", "CCSBT"):
score = 7
if region_name in ("西北太平洋", "北太平洋") and n == "WCPFC": score = 10
if region_name == "印度洋" and n == "IOTC": score = 10
if region_name == "大西洋" and n == "ICCAT": score = 10
add_fish(n, score, "金枪鱼区域渔业数据", "金枪鱼区域渔业记录")
if squid_intent:
requirement("柔鱼/鱿鱼区域渔业记录", "必需", "研究对象为柔鱼/鱿鱼,需要区域性捕捞/出现记录", ["catch", "effort", "CPUE", "occurrence"], "Fisheries")
add_fish("SPRFMO", 9 if region_name in ("西北太平洋", "北太平洋") else 7, "柔鱼/鱿鱼区域渔业数据", "柔鱼/鱿鱼区域渔业记录")
add_fish("NPFC", 10 if region_name in ("西北太平洋", "北太平洋") else 7, "北太平洋柔鱼/鱿鱼区域数据", "柔鱼/鱿鱼区域渔业记录")
if effort_intent:
requirement("Catch / Effort / CPUE", "必需", "项目明确涉及捕捞量、努力量或 CPUE", ["catch", "effort", "CPUE"], "Fisheries")
add_fish("FAO", 5, "捕捞统计基线", "Catch / Effort / CPUE")
add_fish("Sea Around Us", 5, "历史重建捕捞量", "Catch / Effort / CPUE")
if stock_intent:
requirement("种群评估指标", "必需", "资源评估需要生物量/补充量等种群指标", ["biomass", "recruitment", "F"], "Fisheries")
add_fish("RAM Legacy", 10, "资源评估与种群指标", "种群评估指标")
if vessel_intent:
requirement("船舶活动/表观努力", "推荐", "用于刻画捕捞活动空间分布", ["AIS fishing hours", "VIIRS detections"], "Fisheries")
add_fish("GFW", 9, "AIS 表观捕捞活动", "船舶活动/表观努力")
add_fish("VIIRS", 7, "夜光船活动观测", "船舶活动/表观努力")
if not ocean_scores and not fish_scores:
requirement("基础海洋环境", "推荐", "项目描述较宽泛,默认先提供海洋环境背景,不自动引入无关渔业数据", ["SST", "SSH"], "Ocean")
add_ocean("oisst", 5, "通用海洋环境背景", "基础海洋环境")
add_ocean("cmems_surface", 5, "通用海表与上层海洋背景", "基础海洋环境")
if fisheries_intent:
requirement("全球渔业统计", "推荐", "项目明确涉及渔业,提供全球渔业统计基线", ["catch"], "Fisheries")
add_fish("FAO", 4, "全球渔业统计基线", "全球渔业统计")
def priority(score: int) -> str:
if score >= 8: return "必需"
if score >= 5: return "推荐"
return "可选"
alternatives = {
"oisst": ["CMEMS Physics(若需要三维温度/更多物理量)"],
"cmems_bgc": ["OC-CCI(若主要关注遥感叶绿素/海色)"],
"occci": ["CMEMS BGC(若还需要营养盐、溶解氧等)"],
"era5": ["ERA5 Accum(若重点是降水、辐射、热通量累计量)"],
"cmems_surface": ["CMEMS Physics(若需要更完整三维物理场)"],
}
fish_alternatives = {
"WCPFC": ["FAO", "Sea Around Us"], "IATTC": ["FAO", "Sea Around Us"],
"ICCAT": ["FAO", "Sea Around Us"], "IOTC": ["FAO", "Sea Around Us"],
"CCSBT": ["FAO", "Sea Around Us"], "SPRFMO": ["NPFC", "FAO"],
"NPFC": ["SPRFMO", "FAO"], "GFW": ["VIIRS"], "VIIRS": ["GFW"],
"RAM Legacy": ["FAO", "Sea Around Us"],
}
ocean_lookup = {x[0]: x for x in _OCEAN_CATALOG}
ocean = []
for key, score in sorted(ocean_scores.items(), key=lambda x: (-x[1], x[0]))[:7]:
entry = ocean_lookup.get(key)
if not entry: continue
_, name, name_zh, variables = entry
ocean.append({
"database": "Ocean", "key": key, "name": name, "name_zh": name_zh,
"variables": list(variables), "score": score, "priority": priority(score),
"reason": ";".join(dict.fromkeys(reasons["ocean:" + key])),
"requirements": sorted(requirement_tags["ocean:" + key]),
"coverage": {"time": "打包前实时核验", "space": "打包前按所选区域核验", "resolution": "需结合模型尺度确认"},
"alternatives": alternatives.get(key, []),
})
fisheries = []
for name, score in sorted(fish_scores.items(), key=lambda x: (-x[1], x[0]))[:10]:
category = _HF_SOURCE_CATEGORIES.get(name, ("general", "综合渔业数据"))[0]
db = "Tuna-Fisheries-Dataset" if category == "tuna" else "squid_dataset" if category == "squid" else "按实际 Hugging Face 仓库分类"
fisheries.append({
"database": db, "name": name, "name_zh": _HF_SOURCE_NAMES_ZH.get(name, name),
"category": category, "score": score, "priority": priority(score),
"reason": ";".join(dict.fromkeys(reasons["fish:" + name])),
"requirements": sorted(requirement_tags["fish:" + name]),
"coverage": {"time": "打包前核验文件时间覆盖", "space": "依据来源组织/文件范围核验", "resolution": "以实际文件粒度为准"},
"alternatives": fish_alternatives.get(name, []),
})
missing_conditions = []
if not region_name:
missing_conditions.append("海域/空间范围")
elif region_name == "太平洋" and not bbox:
missing_conditions.append("更精确的太平洋子区域/经纬度范围")
if not time_range: missing_conditions.append("时间范围")
if not species and fisheries_intent:
missing_conditions.append("目标物种")
# Explicit variables plus inferred requirements give the user a compact project brief.
requirements = sorted(reqs, key=lambda x: ({"必需": 0, "推荐": 1, "可选": 2}.get(x["priority"], 3), x["name"]))
required_count = sum(1 for x in ocean + fisheries if x["priority"] == "必需")
recommended_count = sum(1 for x in ocean + fisheries if x["priority"] == "推荐")
optional_count = sum(1 for x in ocean + fisheries if x["priority"] == "可选")
return {
"planner_version": "3.5.1",
"project": text,
"project_profile": {
"task_type": task_type,
"research_domain": "物理海洋" if physical_ocean_intent and not fisheries_intent else ("渔业/生态" if fisheries_intent else "海洋综合"),
"research_object": "中尺度涡旋" if eddy_intent else ("、".join(species) if species else ""),
"species": species,
"region": region_name,
"bbox": bbox,
"time_range": time_range,
"exact_date": date,
},
"requirements": requirements,
"ocean": ocean,
"fisheries": fisheries,
"date": date,
"time_range": time_range,
"bbox": bbox,
"region_name": region_name,
"ocean_export_ready": bool(date and bbox),
"missing_conditions": missing_conditions,
"priority_summary": {"required": required_count, "recommended": recommended_count, "optional": optional_count},
"checks": [
"生成前通过实时 inventory 核验 Hugging Face 文件是否存在及大小。",
"Ocean 原始格点导出会按日期与区域请求 Marine API;缺少精确日期时只生成可复现请求清单。",
"时间覆盖、空间覆盖和分辨率属于数据源级条件,最终以实时文件/接口返回为准,不用规则结果冒充真实覆盖。",
],
"note": "推荐按“必需 / 推荐 / 可选”分级;物理海洋任务默认只推荐 Ocean,只有出现物种、捕捞、CPUE、渔场等明确渔业意图时才加入 Fisheries。若主来源不可用,可参考替代来源。",
}
|