lightning / keywords.py
sharktide's picture
Create keywords.py
09e1b3e verified
Raw History Blame
4.01 kB
REASONING_KEYWORDS = [
# explicit reasoning requests
"prove", "demonstrate", "derive", "justify", "verify",
"show that", "walk through", "step by step", "reason through",
"chain of reasoning", "rigorous", "formal proof",
# analysis/comparison
"analyze", "analysis of", "compare and contrast",
"evaluate", "critically assess", "explain why",
"explain how", "what causes", "implications of",
# problem solving
"solve", "solution to", "how would you approach",
"strategy for", "optimize", "algorithm for",
# technical domains
"theorem", "lemma", "corollary",
"complexity analysis", "big o", "time complexity",
"mathematical", "statistical", "probabilistic",
"model the", "simulate",
]
CODE_KEYWORDS = [
"await", "async", "print(", "console.log(",
"code", ".ts", ".js", ".py", ".repy", ".rb",
"gnu", "gcc", "clang", "clang++", "program",
"coding"
]
CREATIVE_KEYWORDS = [
# cinematic cues
"cinematic", "film still", "movie scene",
"epic", "dramatic lighting", "moody lighting",
"volumetric lighting", "depth of field",
"anamorphic lens", "8k", "4k",
# art styles
"concept art", "digital painting",
"fantasy art", "sci-fi", "mythical",
"cyberpunk", "steampunk",
"baroque", "surreal", "abstract",
"oil painting", "watercolor",
# rendering engines
"octane render", "unreal engine",
"ray tracing", "global illumination",
# emotional narrative framing
"emotional portrait", "story scene",
"hero shot", "dramatic pose",
]
STRUCTURED_KEYWORDS = [
"return as json",
"output json",
"json schema",
"format as json",
"structured output",
"extract entities",
"extract fields",
"parse this",
"convert to table",
"create a table",
"categorize into",
"classify",
"label the following",
"taxonomy",
"generate schema",
]
MATH_PATTERNS = [
r"\b∫\b", r"\b∑\b", r"\b∂\b",
r"\bmatrix\b",
r"\blimit\b",
r"\bintegral\b",
r"\bderivative\b",
r"\bdifferential equation\b",
r"\blinear algebra\b",
r"\boptimi[sz]e\b",
r"\bgradient\b",
r"\bbackprop\b",
r"\bproof\b",
r"\btheorem\b",
]
LIGHTWEIGHT_KEYWORDS = [
"hello", "hi", "hey",
"thanks", "thank you",
"define", "definition of",
"what is", "who is",
"quick question",
"short answer",
"brief explanation",
"summarize",
"paraphrase",
"rewrite this",
]
def is_long_context(messages: list) -> bool:
total_chars = sum(len(m.get("content", "")) for m in messages)
return total_chars > 4000
def contains_code(prompt: str) -> bool:
if "```" in prompt:
return True
for kw in CODE_KEYWORDS:
if kw in prompt:
return True
return False
def is_code_heavy(prompt: str, code_present: bool, long_context: bool) -> bool:
"""
Determines whether the coding task is substantial enough
to require a code-optimized or larger model.
"""
if not code_present:
return False
heavy_patterns = [
r"\brefactor\b",
r"\boptimi[sz]e\b",
r"\bdebug\b",
r"\bfix this\b",
r"\barchitecture\b",
r"\bdesign pattern\b",
r"\bscalable\b",
r"\bmicroservice\b",
r"\bmultiple files\b",
r"\bentire project\b",
r"\bcodebase\b",
r"\bperformance\b",
]
for pattern in heavy_patterns:
if re.search(pattern, prompt):
return True
if prompt.count("```") >= 2:
return True
if long_context:
return True
return False
def is_math_heavy(prompt: str) -> bool:
for pattern in MATH_PATTERNS:
if re.search(pattern, prompt):
return True
return False
def is_structured_task(prompt: str) -> bool:
for kw in STRUCTURED_KEYWORDS:
if kw in prompt:
return True
return False
def multiple_questions(prompt: str) -> bool:
return prompt.count("?") >= 3