momo-1.0 / momo_core.py
Bilal140202
Momo 1.0: decision brain of Babymomo
46808c4
Raw History Blame Contribute Delete
6.65 kB
"""
Momo 1.0 — the decision brain of Babymomo.
Core contract shared by train.py / test.py / app.py / data generation.
All intelligence lives in the model weights: one JSON decision per user turn.
No agents.md, no skills files, no LangChain — just Momo.
"""
from __future__ import annotations
import json
import re
from typing import Any, Dict, List, Optional
# ---------------------------------------------------------------- constants --
MOMO_NAME = "Momo"
MOMO_VERSION = "1.0"
REPO_ID = "momo-1.0"
BASE_MODEL_ID = "Qwen/Qwen2-0.5B-Instruct"
EMBED_MODEL_ID = "BAAI/bge-small-en-v1.5" # optional, memory search only
ACTIONS = (
"STORE_MEMORY",
"UPDATE_MEMORY",
"DELETE_MEMORY",
"MEMORY_ONLY",
"WEB_ONLY",
"HYBRID",
)
REQUIRED_KEYS = (
"action",
"memory_query",
"web_query",
"need_memory_search",
"need_web",
"memory_text",
)
# One compact system prompt. LoRA bakes the behavior into the weights, but we
# keep train/inference prompts IDENTICAL for maximum reliability.
SYSTEM_PROMPT = (
"You are Momo 1.0, the decision brain of the Babymomo app. Read the user's "
"message and reply with ONLY one JSON object, no other text:\n"
'{"action": "STORE_MEMORY|UPDATE_MEMORY|DELETE_MEMORY|MEMORY_ONLY|WEB_ONLY|HYBRID", '
'"memory_query": "", "web_query": "", "need_memory_search": false, "need_web": false, '
'"memory_text": ""}\n'
"Rules:\n"
"- New personal fact shared -> STORE_MEMORY: put a clean third-person fact in memory_text.\n"
"- Correction or changed fact -> UPDATE_MEMORY: memory_query finds the old record, memory_text holds the new fact.\n"
"- Wants something forgotten -> DELETE_MEMORY: memory_query describes what to delete.\n"
"- Question about the user's own life -> MEMORY_ONLY: memory_query searches memory.\n"
"- General/world/external question -> WEB_ONLY: web_query is a short web search.\n"
"- Personal + world mix, or save + search together -> HYBRID: fill memory_query and web_query "
"(or memory_text when something must also be saved).\n"
"- NEVER put personal details (private names, phone numbers, emails, ids) in web_query. Anonymize it.\n"
"- If nothing is needed, leave queries empty and flags false."
)
# ----------------------------------------------------------------- builders --
def build_decision(
action: str,
memory_query: str = "",
web_query: str = "",
need_memory_search: bool = False,
need_web: bool = False,
memory_text: str = "",
) -> Dict[str, Any]:
"""Build a decision dict with a FIXED key order (training consistency)."""
if action not in ACTIONS:
raise ValueError(f"invalid action: {action!r}")
return {
"action": action,
"memory_query": memory_query,
"web_query": web_query,
"need_memory_search": bool(need_memory_search),
"need_web": bool(need_web),
"memory_text": memory_text,
}
def decision_to_json(decision: Dict[str, Any]) -> str:
"""Serialize decision dict -> canonical JSON string for training targets."""
ordered = {k: decision.get(k, "" if k in ("memory_query", "web_query", "memory_text") else False)
for k in REQUIRED_KEYS}
if ordered["action"] not in ACTIONS:
raise ValueError(f"invalid action: {ordered['action']!r}")
ordered["need_memory_search"] = bool(ordered["need_memory_search"])
ordered["need_web"] = bool(ordered["need_web"])
return json.dumps(ordered, ensure_ascii=False)
_JSON_BLOCK_RE = re.compile(r"\{.*\}", re.DOTALL)
def parse_decision(text: str) -> Optional[Dict[str, Any]]:
"""
Parse the first JSON-looking block out of model output.
Returns a validated decision dict, or None if unrecoverable.
Tolerant: coerces bool-like values, trims stray text around the object.
"""
if not text or not text.strip():
return None
m = _JSON_BLOCK_RE.search(text)
if not m:
return None
raw = m.group(0)
try:
obj = json.loads(raw)
except json.JSONDecodeError:
try: # single-quote fallback (model slip)
obj = json.loads(raw.replace("'", '"'))
except json.JSONDecodeError:
return None
if not isinstance(obj, dict) or "action" not in obj:
return None
obj["action"] = str(obj["action"]).upper().strip()
if obj["action"] not in ACTIONS:
return None
for k in REQUIRED_KEYS:
obj.setdefault(k, "" if k in ("memory_query", "web_query", "memory_text") else False)
for k in ("memory_query", "web_query", "memory_text"):
if obj[k] is None:
obj[k] = ""
obj[k] = str(obj[k])
for k in ("need_memory_search", "need_web"):
v = obj[k]
if isinstance(v, str):
v = v.strip().lower() in ("true", "1", "yes")
obj[k] = bool(v)
return obj
def validate_decision(decision: Dict[str, Any]) -> List[str]:
"""Semantic sanity checks used by test.py and the data validator."""
problems: List[str] = []
a = decision.get("action")
mq, wq, mt = decision.get("memory_query", ""), decision.get("web_query", ""), decision.get("memory_text", "")
ms, nw = bool(decision.get("need_memory_search")), bool(decision.get("need_web"))
if a == "STORE_MEMORY" and not mt:
problems.append("STORE_MEMORY requires memory_text")
if a == "UPDATE_MEMORY" and (not mq or not mt):
problems.append("UPDATE_MEMORY requires memory_query + memory_text")
if a == "UPDATE_MEMORY" and not ms:
problems.append("UPDATE_MEMORY must search memory to find the old record")
if a == "DELETE_MEMORY" and not mq:
problems.append("DELETE_MEMORY requires memory_query")
if a == "DELETE_MEMORY" and not ms:
problems.append("DELETE_MEMORY must search memory to find the record")
if a == "MEMORY_ONLY" and ms and not mq:
problems.append("MEMORY_ONLY with search=true requires memory_query")
if a == "WEB_ONLY" and not wq:
problems.append("WEB_ONLY requires web_query")
if a == "HYBRID" and not ((mq or mt) and wq):
problems.append("HYBRID requires (memory_query or memory_text) and web_query")
if ms and not mq and a != "STORE_MEMORY":
problems.append("need_memory_search=true but memory_query empty")
if nw and not wq:
problems.append("need_web=true but web_query empty")
# MEMORY_ONLY with everything empty = pure conversation path (chit-chat) — allowed.
if a != "MEMORY_ONLY" and not ms and not nw and not mt:
problems.append("no retrieval requested and nothing to store (dead decision)")
return problems