Spaces:
Sleeping
Sleeping
Deploy repair agent from GitHub Actions
Browse files- Dockerfile +42 -2
- app.py +1498 -315
- requirements.txt +5 -6
Dockerfile
CHANGED
|
@@ -1,12 +1,52 @@
|
|
| 1 |
FROM python:3.11-slim
|
| 2 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
WORKDIR /app
|
| 4 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 5 |
COPY requirements.txt .
|
| 6 |
-
RUN pip install --
|
|
|
|
|
|
|
|
|
|
|
|
|
| 7 |
|
| 8 |
COPY app.py .
|
| 9 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10 |
EXPOSE 7860
|
| 11 |
|
| 12 |
-
|
|
|
|
|
|
|
|
|
| 1 |
FROM python:3.11-slim
|
| 2 |
|
| 3 |
+
ENV PYTHONDONTWRITEBYTECODE=1 \
|
| 4 |
+
PYTHONUNBUFFERED=1 \
|
| 5 |
+
HF_HOME=/opt/huggingface \
|
| 6 |
+
MODEL_CACHE_DIR=/opt/huggingface/hub \
|
| 7 |
+
HF_MODEL_REPO=ibm-granite/granite-4.0-1b-GGUF \
|
| 8 |
+
HF_MODEL_FILE=granite-4.0-1b-Q4_K_M.gguf \
|
| 9 |
+
HF_MODEL_REVISION=b27c2fe3f211b7f44e80fa620177aea371099aaa \
|
| 10 |
+
MODEL_QUANTIZATION=Q4_K_M \
|
| 11 |
+
HF_MODEL=ibm-granite/granite-4.0-1b-GGUF:Q4_K_M \
|
| 12 |
+
HF_LOCAL_FILES_ONLY=true \
|
| 13 |
+
LLM_ENABLED=true \
|
| 14 |
+
MODEL_CONTEXT_TOKENS=4096 \
|
| 15 |
+
MODEL_BATCH_TOKENS=512 \
|
| 16 |
+
MODEL_MAX_INPUT_TOKENS=2800 \
|
| 17 |
+
MODEL_MAX_NEW_TOKENS=256 \
|
| 18 |
+
MODEL_MAX_GENERATION_SECONDS=90 \
|
| 19 |
+
MODEL_THREADS=2 \
|
| 20 |
+
MODEL_THREADS_BATCH=2 \
|
| 21 |
+
MODEL_TEMPERATURE=0.1 \
|
| 22 |
+
MODEL_TOP_P=0.9 \
|
| 23 |
+
MODEL_PROMPT_CACHE_MB=256 \
|
| 24 |
+
MODEL_SEED=17 \
|
| 25 |
+
MODEL_USE_MMAP=true \
|
| 26 |
+
OMP_NUM_THREADS=2
|
| 27 |
+
|
| 28 |
WORKDIR /app
|
| 29 |
|
| 30 |
+
RUN apt-get update \
|
| 31 |
+
&& apt-get install -y --no-install-recommends ca-certificates libgomp1 \
|
| 32 |
+
&& rm -rf /var/lib/apt/lists/*
|
| 33 |
+
|
| 34 |
COPY requirements.txt .
|
| 35 |
+
RUN python -m pip install --upgrade pip \
|
| 36 |
+
&& python -m pip install --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cpu -r requirements.txt
|
| 37 |
+
|
| 38 |
+
RUN mkdir -p /opt/huggingface \
|
| 39 |
+
&& python -c "import os; from huggingface_hub import hf_hub_download; hf_hub_download(repo_id=os.environ['HF_MODEL_REPO'], filename=os.environ['HF_MODEL_FILE'], revision=os.environ['HF_MODEL_REVISION'], cache_dir=os.environ['MODEL_CACHE_DIR'])"
|
| 40 |
|
| 41 |
COPY app.py .
|
| 42 |
|
| 43 |
+
RUN useradd --create-home --uid 1000 appuser \
|
| 44 |
+
&& chown -R appuser:appuser /app /opt/huggingface
|
| 45 |
+
|
| 46 |
+
USER appuser
|
| 47 |
+
|
| 48 |
EXPOSE 7860
|
| 49 |
|
| 50 |
+
HEALTHCHECK --interval=30s --timeout=5s --start-period=60s --retries=3 CMD python -c "import urllib.request; urllib.request.urlopen('http://127.0.0.1:7860/ready', timeout=3)"
|
| 51 |
+
|
| 52 |
+
CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "7860", "--workers", "1", "--timeout-keep-alive", "30"]
|
app.py
CHANGED
|
@@ -1,399 +1,1582 @@
|
|
| 1 |
-
|
| 2 |
-
|
| 3 |
-
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM
|
| 4 |
import os
|
| 5 |
import re
|
|
|
|
|
|
|
|
|
|
| 6 |
|
| 7 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 8 |
|
| 9 |
-
tokenizer = None
|
| 10 |
-
model = None
|
| 11 |
|
| 12 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 13 |
|
| 14 |
|
| 15 |
class RepairRequest(BaseModel):
|
| 16 |
-
|
| 17 |
-
|
|
|
|
|
|
|
| 18 |
success: bool
|
| 19 |
-
exit_code: int | None
|
| 20 |
-
stdout: str
|
| 21 |
-
stderr: str
|
| 22 |
-
|
| 23 |
-
|
| 24 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 25 |
|
| 26 |
|
| 27 |
-
|
| 28 |
-
|
| 29 |
-
|
| 30 |
-
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 36 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 37 |
|
| 38 |
-
|
| 39 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 40 |
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 44 |
|
| 45 |
-
return tokenizer, model
|
| 46 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 47 |
|
| 48 |
-
|
| 49 |
-
|
| 50 |
-
|
| 51 |
-
|
| 52 |
-
|
| 53 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 54 |
|
| 55 |
-
|
| 56 |
-
|
| 57 |
-
|
| 58 |
-
|
| 59 |
-
|
| 60 |
-
|
| 61 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
}
|
|
|
|
|
|
|
| 63 |
|
| 64 |
-
tests_run, failures, errors, skipped = match.groups()
|
| 65 |
|
| 66 |
-
|
| 67 |
-
|
| 68 |
-
|
| 69 |
-
|
| 70 |
-
|
| 71 |
-
"
|
| 72 |
-
|
| 73 |
-
|
| 74 |
-
|
| 75 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 76 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 77 |
}
|
| 78 |
|
| 79 |
|
| 80 |
-
def
|
| 81 |
-
|
| 82 |
-
|
| 83 |
-
|
| 84 |
-
|
| 85 |
-
|
| 86 |
-
|
| 87 |
-
|
| 88 |
-
|
| 89 |
-
|
| 90 |
-
|
| 91 |
-
|
| 92 |
-
|
| 93 |
-
|
| 94 |
-
|
| 95 |
-
|
| 96 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 97 |
|
| 98 |
-
|
| 99 |
-
|
| 100 |
-
|
| 101 |
-
"
|
| 102 |
-
"
|
| 103 |
-
|
| 104 |
-
|
| 105 |
-
|
| 106 |
-
|
| 107 |
-
"Check whether mvnw.cmd exists in the project root.",
|
| 108 |
-
"If Maven Wrapper is missing, add Maven Wrapper files or install Maven globally.",
|
| 109 |
-
"Rerun Stitch QA after the build command can execute."
|
| 110 |
-
],
|
| 111 |
-
"next_action": "Fix the Maven Wrapper or Maven installation before changing application code."
|
| 112 |
-
}
|
| 113 |
|
| 114 |
-
if request.failure_type == "COMMAND_TIMEOUT":
|
| 115 |
-
return {
|
| 116 |
-
"repair_type": "environment-timeout",
|
| 117 |
-
"risk_level": "MEDIUM",
|
| 118 |
-
"summary": (
|
| 119 |
-
f"The {request.project_type} project command did not finish within the allowed timeout. "
|
| 120 |
-
"This may be a long-running build, dependency download, or stuck process."
|
| 121 |
-
),
|
| 122 |
-
"suggestions": [
|
| 123 |
-
"Rerun the command manually to check whether it is slow or stuck.",
|
| 124 |
-
"Increase the execution timeout if the build normally takes longer.",
|
| 125 |
-
"Check dependency downloads and Maven repository access."
|
| 126 |
-
],
|
| 127 |
-
"next_action": "Investigate command runtime before applying code changes."
|
| 128 |
-
}
|
| 129 |
|
| 130 |
-
|
|
|
|
|
|
|
| 131 |
|
|
|
|
|
|
|
| 132 |
|
| 133 |
-
|
| 134 |
-
|
|
|
|
| 135 |
|
| 136 |
-
if
|
| 137 |
-
|
| 138 |
-
|
| 139 |
-
|
| 140 |
-
|
| 141 |
-
"auto_apply": False,
|
| 142 |
-
"summary": environment_repair["summary"],
|
| 143 |
-
"suggestions": environment_repair["suggestions"],
|
| 144 |
-
"next_action": environment_repair["next_action"]
|
| 145 |
-
}
|
| 146 |
|
| 147 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 148 |
|
| 149 |
-
|
| 150 |
-
risk_level = "LOW" if request.success else "HIGH"
|
| 151 |
|
| 152 |
-
if request.success:
|
| 153 |
-
suggestions.append(
|
| 154 |
-
"No blocking fix is required because the project build and tests passed."
|
| 155 |
-
)
|
| 156 |
|
| 157 |
-
|
| 158 |
-
|
| 159 |
-
|
| 160 |
-
)
|
| 161 |
|
| 162 |
-
if "
|
| 163 |
-
|
| 164 |
-
|
| 165 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 166 |
|
| 167 |
-
if "tests run" in combined_logs and ("failures: 1" in combined_logs or "errors: 1" in combined_logs):
|
| 168 |
-
suggestions.append(
|
| 169 |
-
"Review the failing test method, compare expected versus actual behavior, and fix the related implementation or assertion."
|
| 170 |
-
)
|
| 171 |
|
| 172 |
-
|
| 173 |
-
|
| 174 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 175 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 176 |
|
| 177 |
return {
|
|
|
|
|
|
|
|
|
|
| 178 |
"agent": "repair-agent",
|
| 179 |
-
"mode":
|
| 180 |
-
"
|
|
|
|
|
|
|
|
|
|
|
|
|
| 181 |
"auto_apply": False,
|
| 182 |
-
"summary":
|
| 183 |
-
|
| 184 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 185 |
}
|
| 186 |
|
| 187 |
|
| 188 |
-
def
|
| 189 |
-
|
| 190 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 191 |
|
| 192 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 193 |
|
| 194 |
-
|
| 195 |
-
|
| 196 |
-
"
|
| 197 |
"command": request.command,
|
| 198 |
-
"exit_code": request.exit_code,
|
| 199 |
-
"build_state": "not verified",
|
| 200 |
"success": request.success,
|
| 201 |
-
"
|
| 202 |
-
"warning_category": "none",
|
| 203 |
-
"detected_warning": "No major warning detected.",
|
| 204 |
-
"repair_type": environment_repair["repair_type"],
|
| 205 |
-
"test_result": test_result["summary"],
|
| 206 |
-
"suggestions": fallback_result["suggestions"],
|
| 207 |
-
"risk_level": fallback_result["risk_level"],
|
| 208 |
"failure_type": request.failure_type,
|
| 209 |
-
|
| 210 |
-
|
| 211 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 212 |
|
| 213 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 214 |
|
| 215 |
-
detected_warning = "No major warning detected."
|
| 216 |
-
warning_category = "none"
|
| 217 |
|
| 218 |
-
|
| 219 |
-
|
| 220 |
-
|
| 221 |
-
"Mockito dynamic Java agent loading warning detected. "
|
| 222 |
-
"This is not a current failure, but it may affect compatibility with future JDK versions."
|
| 223 |
-
)
|
| 224 |
|
| 225 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 226 |
|
| 227 |
-
if request.success and warning_category == "none":
|
| 228 |
-
repair_type = "no-repair-needed"
|
| 229 |
|
| 230 |
-
|
| 231 |
-
|
|
|
|
|
|
|
|
|
|
| 232 |
|
| 233 |
-
|
| 234 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 235 |
|
| 236 |
-
|
| 237 |
-
|
| 238 |
-
|
| 239 |
-
|
| 240 |
-
"build_state": build_state,
|
| 241 |
-
"success": request.success,
|
| 242 |
-
"root_cause": request.root_cause or "No root cause provided.",
|
| 243 |
-
"warning_category": warning_category,
|
| 244 |
-
"detected_warning": detected_warning,
|
| 245 |
-
"repair_type": repair_type,
|
| 246 |
-
"test_result": test_result["summary"],
|
| 247 |
-
"suggestions": fallback_result["suggestions"],
|
| 248 |
-
"risk_level": fallback_result["risk_level"],
|
| 249 |
-
"failure_type": request.failure_type,
|
| 250 |
-
"help_message": request.help_message,
|
| 251 |
-
"environment_summary": None
|
| 252 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 253 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 254 |
|
| 255 |
-
|
| 256 |
-
|
| 257 |
-
"environment-maven-missing",
|
| 258 |
-
"environment-wrapper-missing",
|
| 259 |
-
"environment-timeout"
|
| 260 |
-
}:
|
| 261 |
-
return facts["environment_summary"]
|
| 262 |
|
| 263 |
-
|
| 264 |
-
return (
|
| 265 |
-
f"The {facts['project_type']} project passed the current QA execution using "
|
| 266 |
-
f"`{facts['command']}`. {facts['test_result']}. No blocking repair is required. "
|
| 267 |
-
"The project can be considered stable for this basic test run."
|
| 268 |
-
)
|
| 269 |
|
| 270 |
-
if facts["repair_type"] == "configuration-warning":
|
| 271 |
-
return (
|
| 272 |
-
f"The {facts['project_type']} project passed the current QA execution using "
|
| 273 |
-
f"`{facts['command']}`. {facts['test_result']}. No blocking code repair is required. "
|
| 274 |
-
f"However, {facts['detected_warning']} Recommended action: review the Maven test configuration "
|
| 275 |
-
"and prepare a future-safe Mockito Java agent setup before upgrading to stricter JDK versions."
|
| 276 |
-
)
|
| 277 |
|
| 278 |
-
|
| 279 |
-
|
| 280 |
-
|
| 281 |
-
|
| 282 |
-
"apply the smallest safe fix, and rerun Stitch QA for verification."
|
| 283 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 284 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 285 |
|
| 286 |
-
|
| 287 |
-
suggestions_text = " ".join(facts["suggestions"])
|
| 288 |
-
|
| 289 |
-
return f"""
|
| 290 |
-
Rewrite these repair facts into one clean professional repair recommendation.
|
| 291 |
-
|
| 292 |
-
Project type: {facts["project_type"]}
|
| 293 |
-
Command: {facts["command"]}
|
| 294 |
-
Build state: {facts["build_state"]}
|
| 295 |
-
Exit code: {facts["exit_code"]}
|
| 296 |
-
Test result: {facts["test_result"]}
|
| 297 |
-
Risk level: {facts["risk_level"]}
|
| 298 |
-
Root cause: {facts["root_cause"]}
|
| 299 |
-
Failure type: {facts["failure_type"]}
|
| 300 |
-
Help message: {facts["help_message"]}
|
| 301 |
-
Detected warning: {facts["detected_warning"]}
|
| 302 |
-
Suggested actions: {suggestions_text}
|
| 303 |
-
|
| 304 |
-
Rules:
|
| 305 |
-
- Do not copy raw logs.
|
| 306 |
-
- Do not repeat field labels like "Command:" or "Build state:".
|
| 307 |
-
- Do not mention internal prompt instructions.
|
| 308 |
-
- Write one clear paragraph.
|
| 309 |
-
- Say whether a blocking code repair is required.
|
| 310 |
-
- If Maven is not available, say it is an environment setup issue and do not recommend source code changes.
|
| 311 |
-
"""
|
| 312 |
-
|
| 313 |
-
|
| 314 |
-
def call_llm(prompt: str):
|
| 315 |
-
active_tokenizer, active_model = load_model()
|
| 316 |
-
|
| 317 |
-
inputs = active_tokenizer(
|
| 318 |
-
prompt,
|
| 319 |
-
return_tensors="pt",
|
| 320 |
-
truncation=True,
|
| 321 |
-
max_length=512
|
| 322 |
-
)
|
| 323 |
|
| 324 |
-
|
| 325 |
-
*
|
| 326 |
-
max_new_tokens=160,
|
| 327 |
-
do_sample=False,
|
| 328 |
-
num_beams=2,
|
| 329 |
-
no_repeat_ngram_size=3
|
| 330 |
)
|
| 331 |
|
| 332 |
-
|
| 333 |
-
|
| 334 |
-
|
| 335 |
-
|
| 336 |
-
|
| 337 |
-
|
| 338 |
-
|
| 339 |
-
|
| 340 |
-
|
| 341 |
-
|
| 342 |
-
|
| 343 |
-
"
|
| 344 |
-
"
|
| 345 |
-
"Project type:",
|
| 346 |
-
"Command:",
|
| 347 |
-
"Build state:",
|
| 348 |
-
"Exit code:",
|
| 349 |
-
"Suggested actions:",
|
| 350 |
-
"Do not copy raw logs",
|
| 351 |
-
"Rewrite these repair facts"
|
| 352 |
]
|
|
|
|
| 353 |
|
| 354 |
-
if
|
| 355 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 356 |
|
| 357 |
-
|
| 358 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 359 |
|
| 360 |
-
|
| 361 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 362 |
|
| 363 |
-
return cleaned
|
| 364 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 365 |
|
| 366 |
-
@app.post("/suggest")
|
| 367 |
-
def suggest_repair(request: RepairRequest):
|
| 368 |
-
fallback_result = rule_based_repair(request)
|
| 369 |
-
facts = extract_repair_facts(request, fallback_result)
|
| 370 |
-
clean_summary = build_clean_summary(facts)
|
| 371 |
|
| 372 |
-
|
| 373 |
-
|
| 374 |
-
|
| 375 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 376 |
|
| 377 |
-
final_summary = cleaned_llm_text if cleaned_llm_text else clean_summary
|
| 378 |
|
| 379 |
-
|
| 380 |
-
|
| 381 |
-
|
| 382 |
-
"environment-timeout"
|
| 383 |
-
}:
|
| 384 |
-
final_summary = clean_summary
|
| 385 |
|
| 386 |
-
|
| 387 |
-
|
| 388 |
-
|
| 389 |
-
|
| 390 |
-
|
| 391 |
-
|
| 392 |
-
|
| 393 |
-
|
| 394 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 395 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 396 |
except Exception as error:
|
| 397 |
-
|
| 398 |
-
|
| 399 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import gc
|
| 2 |
+
import json
|
|
|
|
| 3 |
import os
|
| 4 |
import re
|
| 5 |
+
import threading
|
| 6 |
+
import time
|
| 7 |
+
from typing import Any, Literal
|
| 8 |
|
| 9 |
+
from fastapi import FastAPI
|
| 10 |
+
from pydantic import BaseModel, ConfigDict, Field, ValidationError
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
AGENT_ID = "defect-resolution-analyst"
|
| 14 |
+
DISPLAY_NAME = "Defect Resolution Intelligence Analyst"
|
| 15 |
+
AGENT_VERSION = "2.0"
|
| 16 |
+
MAX_AI_FINDINGS = 4
|
| 17 |
+
MAX_AI_CONTRACTS = 4
|
| 18 |
+
|
| 19 |
+
PRIORITY_ORDER = {
|
| 20 |
+
"NONE": 0,
|
| 21 |
+
"P3": 1,
|
| 22 |
+
"P2": 2,
|
| 23 |
+
"P1": 3,
|
| 24 |
+
}
|
| 25 |
+
|
| 26 |
+
RISK_ORDER = {
|
| 27 |
+
"UNKNOWN": -1,
|
| 28 |
+
"NONE": 0,
|
| 29 |
+
"INFO": 1,
|
| 30 |
+
"LOW": 2,
|
| 31 |
+
"MEDIUM": 3,
|
| 32 |
+
"HIGH": 4,
|
| 33 |
+
"CRITICAL": 5,
|
| 34 |
+
}
|
| 35 |
+
|
| 36 |
+
MODEL_RESPONSE_SCHEMA = {
|
| 37 |
+
"type": "object",
|
| 38 |
+
"properties": {
|
| 39 |
+
"p": {"type": "string", "enum": ["P1", "P2", "P3"]},
|
| 40 |
+
"c": {"type": "string", "enum": ["HIGH", "MEDIUM", "LOW"]},
|
| 41 |
+
"k": {"type": "boolean"},
|
| 42 |
+
"kr": {"type": "string", "maxLength": 180},
|
| 43 |
+
"x": {
|
| 44 |
+
"type": "array",
|
| 45 |
+
"minItems": 1,
|
| 46 |
+
"maxItems": MAX_AI_CONTRACTS,
|
| 47 |
+
"items": {
|
| 48 |
+
"type": "object",
|
| 49 |
+
"properties": {
|
| 50 |
+
"f": {
|
| 51 |
+
"type": "array",
|
| 52 |
+
"minItems": 1,
|
| 53 |
+
"maxItems": MAX_AI_FINDINGS,
|
| 54 |
+
"uniqueItems": True,
|
| 55 |
+
"items": {
|
| 56 |
+
"type": "string",
|
| 57 |
+
"minLength": 1,
|
| 58 |
+
"maxLength": 80,
|
| 59 |
+
},
|
| 60 |
+
},
|
| 61 |
+
"p": {"type": "string", "enum": ["P1", "P2", "P3"]},
|
| 62 |
+
"w": {"type": "string", "minLength": 8, "maxLength": 160},
|
| 63 |
+
"o": {"type": "string", "minLength": 8, "maxLength": 180},
|
| 64 |
+
"s": {"type": "string", "minLength": 8, "maxLength": 220},
|
| 65 |
+
"b": {"type": "string", "minLength": 6, "maxLength": 160},
|
| 66 |
+
"q": {"type": "string", "minLength": 6, "maxLength": 160},
|
| 67 |
+
"r": {"type": "string", "enum": ["L", "M", "H"]},
|
| 68 |
+
"v": {"type": "string", "minLength": 8, "maxLength": 220},
|
| 69 |
+
},
|
| 70 |
+
"required": ["f", "p", "w", "o", "s", "b", "q", "r", "v"],
|
| 71 |
+
"additionalProperties": False,
|
| 72 |
+
},
|
| 73 |
+
},
|
| 74 |
+
},
|
| 75 |
+
"required": ["p", "c", "k", "kr", "x"],
|
| 76 |
+
"additionalProperties": False,
|
| 77 |
+
}
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
def env_bool(name, default):
|
| 81 |
+
value = os.getenv(name)
|
| 82 |
+
if value is None:
|
| 83 |
+
return bool(default)
|
| 84 |
+
return value.strip().lower() in {"1", "true", "yes", "on"}
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def clean_text(value, limit=800):
|
| 88 |
+
text = " ".join(str(value or "").split()).strip()
|
| 89 |
+
if len(text) <= limit:
|
| 90 |
+
return text
|
| 91 |
+
return text[: limit - 3].rstrip() + "..."
|
| 92 |
+
|
| 93 |
+
|
| 94 |
+
def normalize_list(value):
|
| 95 |
+
if isinstance(value, list):
|
| 96 |
+
return value
|
| 97 |
+
if value is None:
|
| 98 |
+
return []
|
| 99 |
+
return [value]
|
| 100 |
+
|
| 101 |
+
|
| 102 |
+
def normalize_risk(value):
|
| 103 |
+
risk = str(value or "UNKNOWN").upper()
|
| 104 |
+
return risk if risk in RISK_ORDER else "UNKNOWN"
|
| 105 |
+
|
| 106 |
+
|
| 107 |
+
def highest_risk(*values):
|
| 108 |
+
normalized = [normalize_risk(value) for value in values]
|
| 109 |
+
return max(
|
| 110 |
+
normalized,
|
| 111 |
+
key=lambda value: RISK_ORDER.get(value, -1),
|
| 112 |
+
default="UNKNOWN",
|
| 113 |
+
)
|
| 114 |
|
|
|
|
|
|
|
| 115 |
|
| 116 |
+
def severity_priority(severity):
|
| 117 |
+
severity = normalize_risk(severity)
|
| 118 |
+
if severity in {"CRITICAL", "HIGH"}:
|
| 119 |
+
return "P1"
|
| 120 |
+
if severity == "MEDIUM":
|
| 121 |
+
return "P2"
|
| 122 |
+
return "P3"
|
| 123 |
|
| 124 |
|
| 125 |
class RepairRequest(BaseModel):
|
| 126 |
+
model_config = ConfigDict(extra="ignore")
|
| 127 |
+
|
| 128 |
+
project_type: str = Field(min_length=1, max_length=200)
|
| 129 |
+
command: str = Field(default="", max_length=2000)
|
| 130 |
success: bool
|
| 131 |
+
exit_code: int | None = None
|
| 132 |
+
stdout: str = Field(default="", max_length=16000)
|
| 133 |
+
stderr: str = Field(default="", max_length=16000)
|
| 134 |
+
failure_type: str | None = Field(default=None, max_length=200)
|
| 135 |
+
help_message: str | None = Field(default=None, max_length=4000)
|
| 136 |
+
runtime_evidence: dict[str, Any] | None = None
|
| 137 |
+
runtime_analysis: dict[str, Any] | None = None
|
| 138 |
+
source_review: dict[str, Any] | None = None
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
class ModelRepairContract(BaseModel):
|
| 142 |
+
model_config = ConfigDict(extra="forbid", populate_by_name=True)
|
| 143 |
+
|
| 144 |
+
finding_refs: list[str] = Field(alias="f", min_length=1, max_length=12)
|
| 145 |
+
priority: Literal["P1", "P2", "P3"] = Field(alias="p")
|
| 146 |
+
priority_reason: str = Field(alias="w", min_length=8, max_length=160)
|
| 147 |
+
repair_objective: str = Field(alias="o", min_length=8, max_length=180)
|
| 148 |
+
repair_strategy: str = Field(alias="s", min_length=8, max_length=220)
|
| 149 |
+
change_boundary: str = Field(alias="b", min_length=6, max_length=160)
|
| 150 |
+
protected_behavior: str = Field(alias="q", min_length=6, max_length=160)
|
| 151 |
+
side_effect_risk_code: Literal["L", "M", "H"] = Field(alias="r")
|
| 152 |
+
verification: str = Field(alias="v", min_length=8, max_length=220)
|
| 153 |
+
|
| 154 |
+
@property
|
| 155 |
+
def side_effect_risk(self):
|
| 156 |
+
return {"L": "LOW", "M": "MEDIUM", "H": "HIGH"}[
|
| 157 |
+
self.side_effect_risk_code
|
| 158 |
+
]
|
| 159 |
+
|
| 160 |
+
|
| 161 |
+
class ModelRepairPlan(BaseModel):
|
| 162 |
+
model_config = ConfigDict(extra="forbid", populate_by_name=True)
|
| 163 |
+
|
| 164 |
+
overall_priority: Literal["P1", "P2", "P3"] = Field(alias="p")
|
| 165 |
+
confidence: Literal["HIGH", "MEDIUM", "LOW"] = Field(alias="c")
|
| 166 |
+
current_knowledge_required: bool = Field(alias="k")
|
| 167 |
+
current_knowledge_reason: str = Field(alias="kr", default="", max_length=180)
|
| 168 |
+
contracts: list[ModelRepairContract] = Field(
|
| 169 |
+
alias="x",
|
| 170 |
+
min_length=1,
|
| 171 |
+
max_length=MAX_AI_CONTRACTS,
|
| 172 |
+
)
|
| 173 |
|
| 174 |
|
| 175 |
+
class StitchRepairContract(BaseModel):
|
| 176 |
+
model_config = ConfigDict(extra="ignore")
|
| 177 |
+
|
| 178 |
+
contract_id: str
|
| 179 |
+
finding_refs: list[str]
|
| 180 |
+
title: str
|
| 181 |
+
priority: Literal["P1", "P2", "P3"]
|
| 182 |
+
priority_reason: str
|
| 183 |
+
repair_objective: str
|
| 184 |
+
repair_strategy: str
|
| 185 |
+
change_boundary: str
|
| 186 |
+
protected_behavior: str
|
| 187 |
+
side_effect_risk: Literal["LOW", "MEDIUM", "HIGH"]
|
| 188 |
+
verification: str
|
| 189 |
+
done_condition: str
|
| 190 |
+
status: Literal["PENDING_VERIFICATION"] = "PENDING_VERIFICATION"
|
| 191 |
+
|
| 192 |
+
|
| 193 |
+
class RepairResponse(BaseModel):
|
| 194 |
+
model_config = ConfigDict(extra="ignore")
|
| 195 |
+
|
| 196 |
+
agent_id: str
|
| 197 |
+
display_name: str
|
| 198 |
+
agent_version: str
|
| 199 |
+
agent: str
|
| 200 |
+
mode: str
|
| 201 |
+
model: str | None = None
|
| 202 |
+
status: str
|
| 203 |
+
overall_priority: str
|
| 204 |
+
confidence: str
|
| 205 |
+
repair_risk_level: str
|
| 206 |
+
auto_apply: bool = False
|
| 207 |
+
summary: str
|
| 208 |
+
stitch_repair_contracts: list[StitchRepairContract] = Field(default_factory=list)
|
| 209 |
+
current_knowledge_required: bool = False
|
| 210 |
+
current_knowledge_reason: str | None = None
|
| 211 |
+
next_action: str | None = None
|
| 212 |
+
suggestions: list[str] = Field(default_factory=list)
|
| 213 |
+
warnings: list[str] = Field(default_factory=list)
|
| 214 |
+
limitations: list[str] = Field(default_factory=list)
|
| 215 |
+
llm_metrics: dict[str, Any] | None = None
|
| 216 |
+
llm_error: str | None = None
|
| 217 |
+
|
| 218 |
+
|
| 219 |
+
class ModelService:
|
| 220 |
+
def __init__(self):
|
| 221 |
+
self.model_repo = os.getenv(
|
| 222 |
+
"HF_MODEL_REPO",
|
| 223 |
+
"ibm-granite/granite-4.0-1b-GGUF",
|
| 224 |
+
).strip()
|
| 225 |
+
self.model_file = os.getenv(
|
| 226 |
+
"HF_MODEL_FILE",
|
| 227 |
+
"granite-4.0-1b-Q4_K_M.gguf",
|
| 228 |
+
).strip()
|
| 229 |
+
self.model_revision = os.getenv(
|
| 230 |
+
"HF_MODEL_REVISION",
|
| 231 |
+
"b27c2fe3f211b7f44e80fa620177aea371099aaa",
|
| 232 |
+
).strip()
|
| 233 |
+
self.quantization = os.getenv(
|
| 234 |
+
"MODEL_QUANTIZATION",
|
| 235 |
+
"Q4_K_M",
|
| 236 |
+
).strip()
|
| 237 |
+
self.primary_model = os.getenv(
|
| 238 |
+
"HF_MODEL",
|
| 239 |
+
f"{self.model_repo}:{self.quantization}",
|
| 240 |
+
).strip()
|
| 241 |
+
self.cache_dir = os.getenv(
|
| 242 |
+
"MODEL_CACHE_DIR",
|
| 243 |
+
"/opt/huggingface/hub",
|
| 244 |
+
).strip()
|
| 245 |
+
self.enabled = env_bool("LLM_ENABLED", True)
|
| 246 |
+
self.local_files_only = env_bool("HF_LOCAL_FILES_ONLY", False)
|
| 247 |
+
self.context_tokens = max(1024, int(os.getenv("MODEL_CONTEXT_TOKENS", "4096")))
|
| 248 |
+
self.batch_tokens = max(
|
| 249 |
+
64,
|
| 250 |
+
min(
|
| 251 |
+
self.context_tokens,
|
| 252 |
+
int(os.getenv("MODEL_BATCH_TOKENS", "512")),
|
| 253 |
+
),
|
| 254 |
+
)
|
| 255 |
+
self.max_input_tokens = max(
|
| 256 |
+
512,
|
| 257 |
+
int(os.getenv("MODEL_MAX_INPUT_TOKENS", "2800")),
|
| 258 |
+
)
|
| 259 |
+
self.max_new_tokens = max(
|
| 260 |
+
96,
|
| 261 |
+
int(os.getenv("MODEL_MAX_NEW_TOKENS", "256")),
|
| 262 |
+
)
|
| 263 |
+
self.max_generation_seconds = max(
|
| 264 |
+
15.0,
|
| 265 |
+
float(os.getenv("MODEL_MAX_GENERATION_SECONDS", "90")),
|
| 266 |
+
)
|
| 267 |
+
self.threads = max(1, int(os.getenv("MODEL_THREADS", "2")))
|
| 268 |
+
self.threads_batch = max(
|
| 269 |
+
1,
|
| 270 |
+
int(os.getenv("MODEL_THREADS_BATCH", str(self.threads))),
|
| 271 |
+
)
|
| 272 |
+
self.temperature = max(
|
| 273 |
+
0.0,
|
| 274 |
+
float(os.getenv("MODEL_TEMPERATURE", "0.1")),
|
| 275 |
+
)
|
| 276 |
+
self.top_p = min(
|
| 277 |
+
1.0,
|
| 278 |
+
max(0.01, float(os.getenv("MODEL_TOP_P", "0.9"))),
|
| 279 |
+
)
|
| 280 |
+
self.seed = int(os.getenv("MODEL_SEED", "17"))
|
| 281 |
+
self.use_mmap = env_bool("MODEL_USE_MMAP", True)
|
| 282 |
+
self.prompt_cache_mb = max(
|
| 283 |
+
0,
|
| 284 |
+
int(os.getenv("MODEL_PROMPT_CACHE_MB", "256")),
|
| 285 |
+
)
|
| 286 |
+
self.model = None
|
| 287 |
+
self.model_path = None
|
| 288 |
+
self.model_name = None
|
| 289 |
+
self.load_error = None
|
| 290 |
+
self.load_seconds = None
|
| 291 |
+
self.last_generation = None
|
| 292 |
+
self.load_lock = threading.Lock()
|
| 293 |
+
self.generation_lock = threading.Lock()
|
| 294 |
+
|
| 295 |
+
def _download_model(self):
|
| 296 |
+
try:
|
| 297 |
+
from huggingface_hub import hf_hub_download
|
| 298 |
+
except ImportError as error:
|
| 299 |
+
raise RuntimeError(
|
| 300 |
+
"huggingface_hub is required for the configured Granite GGUF backend."
|
| 301 |
+
) from error
|
| 302 |
+
|
| 303 |
+
return hf_hub_download(
|
| 304 |
+
repo_id=self.model_repo,
|
| 305 |
+
filename=self.model_file,
|
| 306 |
+
revision=self.model_revision,
|
| 307 |
+
cache_dir=self.cache_dir,
|
| 308 |
+
local_files_only=self.local_files_only,
|
| 309 |
+
)
|
| 310 |
|
| 311 |
+
def _load_model(self):
|
| 312 |
+
try:
|
| 313 |
+
from llama_cpp import Llama, LlamaRAMCache
|
| 314 |
+
except ImportError as error:
|
| 315 |
+
raise RuntimeError(
|
| 316 |
+
"llama-cpp-python is required for the configured Granite GGUF backend."
|
| 317 |
+
) from error
|
| 318 |
+
|
| 319 |
+
model_path = self._download_model()
|
| 320 |
+
model = Llama(
|
| 321 |
+
model_path=model_path,
|
| 322 |
+
n_ctx=self.context_tokens,
|
| 323 |
+
n_batch=self.batch_tokens,
|
| 324 |
+
n_threads=self.threads,
|
| 325 |
+
n_threads_batch=self.threads_batch,
|
| 326 |
+
n_gpu_layers=0,
|
| 327 |
+
seed=self.seed,
|
| 328 |
+
use_mmap=self.use_mmap,
|
| 329 |
+
use_mlock=False,
|
| 330 |
+
verbose=False,
|
| 331 |
+
)
|
| 332 |
|
| 333 |
+
if self.prompt_cache_mb > 0:
|
| 334 |
+
model.set_cache(
|
| 335 |
+
LlamaRAMCache(
|
| 336 |
+
capacity_bytes=self.prompt_cache_mb * 1024 * 1024
|
| 337 |
+
)
|
| 338 |
+
)
|
| 339 |
+
|
| 340 |
+
return model_path, model
|
| 341 |
+
|
| 342 |
+
def load(self):
|
| 343 |
+
if not self.enabled:
|
| 344 |
+
raise RuntimeError("LLM inference is disabled.")
|
| 345 |
+
|
| 346 |
+
if self.model is not None:
|
| 347 |
+
return self.model
|
| 348 |
+
|
| 349 |
+
with self.load_lock:
|
| 350 |
+
if self.model is not None:
|
| 351 |
+
return self.model
|
| 352 |
+
|
| 353 |
+
started = time.monotonic()
|
| 354 |
+
try:
|
| 355 |
+
model_path, model = self._load_model()
|
| 356 |
+
self.model_path = model_path
|
| 357 |
+
self.model = model
|
| 358 |
+
self.model_name = self.primary_model
|
| 359 |
+
self.load_error = None
|
| 360 |
+
self.load_seconds = round(time.monotonic() - started, 3)
|
| 361 |
+
return model
|
| 362 |
+
except Exception as error:
|
| 363 |
+
self.model = None
|
| 364 |
+
self.model_path = None
|
| 365 |
+
self.model_name = None
|
| 366 |
+
self.load_error = repr(error)
|
| 367 |
+
self.load_seconds = round(time.monotonic() - started, 3)
|
| 368 |
+
gc.collect()
|
| 369 |
+
raise RuntimeError(self.load_error) from error
|
| 370 |
+
|
| 371 |
+
def _count_message_tokens(self, model, messages):
|
| 372 |
+
serialized = json.dumps(
|
| 373 |
+
messages,
|
| 374 |
+
ensure_ascii=False,
|
| 375 |
+
separators=(",", ":"),
|
| 376 |
+
).encode("utf-8")
|
| 377 |
+
|
| 378 |
+
try:
|
| 379 |
+
tokens = model.tokenize(serialized, add_bos=False, special=True)
|
| 380 |
+
except TypeError:
|
| 381 |
+
tokens = model.tokenize(serialized, add_bos=False)
|
| 382 |
+
|
| 383 |
+
return len(tokens)
|
| 384 |
+
|
| 385 |
+
def _count_text_tokens(self, model, text):
|
| 386 |
+
if not text:
|
| 387 |
+
return 0
|
| 388 |
+
value = str(text).encode("utf-8")
|
| 389 |
+
try:
|
| 390 |
+
tokens = model.tokenize(value, add_bos=False, special=True)
|
| 391 |
+
except TypeError:
|
| 392 |
+
tokens = model.tokenize(value, add_bos=False)
|
| 393 |
+
return len(tokens)
|
| 394 |
+
|
| 395 |
+
def _json_complete(self, text):
|
| 396 |
+
candidate = str(text or "").strip()
|
| 397 |
+
if not candidate.startswith("{") or not candidate.endswith("}"):
|
| 398 |
+
return False
|
| 399 |
+
try:
|
| 400 |
+
json.loads(candidate)
|
| 401 |
+
return True
|
| 402 |
+
except json.JSONDecodeError:
|
| 403 |
+
return False
|
| 404 |
+
|
| 405 |
+
def generate(self, messages):
|
| 406 |
+
self.last_generation = None
|
| 407 |
+
model = self.load()
|
| 408 |
+
|
| 409 |
+
with self.generation_lock:
|
| 410 |
+
input_tokens = self._count_message_tokens(model, messages)
|
| 411 |
+
if input_tokens > self.max_input_tokens:
|
| 412 |
+
raise RuntimeError(
|
| 413 |
+
f"Model input contains approximately {input_tokens} tokens, exceeding the configured limit of {self.max_input_tokens}."
|
| 414 |
+
)
|
| 415 |
+
|
| 416 |
+
started = time.monotonic()
|
| 417 |
+
parts = []
|
| 418 |
+
finish_reason = None
|
| 419 |
+
timed_out = False
|
| 420 |
+
first_content_seconds = None
|
| 421 |
+
stream = model.create_chat_completion(
|
| 422 |
+
messages=messages,
|
| 423 |
+
response_format={
|
| 424 |
+
"type": "json_object",
|
| 425 |
+
"schema": MODEL_RESPONSE_SCHEMA,
|
| 426 |
+
},
|
| 427 |
+
max_tokens=self.max_new_tokens,
|
| 428 |
+
temperature=self.temperature,
|
| 429 |
+
top_p=self.top_p,
|
| 430 |
+
seed=self.seed,
|
| 431 |
+
stream=True,
|
| 432 |
+
)
|
| 433 |
+
|
| 434 |
+
try:
|
| 435 |
+
for chunk in stream:
|
| 436 |
+
elapsed = time.monotonic() - started
|
| 437 |
+
choices = chunk.get("choices") or []
|
| 438 |
+
if choices:
|
| 439 |
+
choice = choices[0]
|
| 440 |
+
delta = choice.get("delta") or {}
|
| 441 |
+
content = delta.get("content")
|
| 442 |
+
if content:
|
| 443 |
+
if first_content_seconds is None:
|
| 444 |
+
first_content_seconds = elapsed
|
| 445 |
+
parts.append(str(content))
|
| 446 |
+
if self._json_complete("".join(parts)):
|
| 447 |
+
finish_reason = "json_complete"
|
| 448 |
+
break
|
| 449 |
+
if choice.get("finish_reason"):
|
| 450 |
+
finish_reason = str(choice.get("finish_reason"))
|
| 451 |
+
|
| 452 |
+
if elapsed >= self.max_generation_seconds and not finish_reason:
|
| 453 |
+
timed_out = True
|
| 454 |
+
break
|
| 455 |
+
finally:
|
| 456 |
+
close = getattr(stream, "close", None)
|
| 457 |
+
if callable(close):
|
| 458 |
+
close()
|
| 459 |
+
|
| 460 |
+
elapsed = time.monotonic() - started
|
| 461 |
+
text = "".join(parts).strip()
|
| 462 |
+
completion_tokens = self._count_text_tokens(model, text)
|
| 463 |
+
tokens_per_second = (
|
| 464 |
+
round(completion_tokens / elapsed, 3)
|
| 465 |
+
if elapsed > 0 and completion_tokens
|
| 466 |
+
else 0.0
|
| 467 |
+
)
|
| 468 |
+
self.last_generation = {
|
| 469 |
+
"backend": "llama.cpp",
|
| 470 |
+
"quantization": self.quantization,
|
| 471 |
+
"input_tokens_approx": input_tokens,
|
| 472 |
+
"completion_tokens": completion_tokens,
|
| 473 |
+
"elapsed_seconds": round(elapsed, 3),
|
| 474 |
+
"first_content_seconds": (
|
| 475 |
+
round(first_content_seconds, 3)
|
| 476 |
+
if first_content_seconds is not None
|
| 477 |
+
else None
|
| 478 |
+
),
|
| 479 |
+
"tokens_per_second": tokens_per_second,
|
| 480 |
+
"finish_reason": finish_reason,
|
| 481 |
+
"timed_out": timed_out,
|
| 482 |
+
"max_generation_seconds": self.max_generation_seconds,
|
| 483 |
+
"max_new_tokens": self.max_new_tokens,
|
| 484 |
+
}
|
| 485 |
+
|
| 486 |
+
if timed_out:
|
| 487 |
+
raise RuntimeError(
|
| 488 |
+
"The configured Granite model exceeded the generation time limit "
|
| 489 |
+
f"(generated_tokens={completion_tokens}, elapsed_seconds={elapsed:.3f}, tokens_per_second={tokens_per_second:.3f})."
|
| 490 |
+
)
|
| 491 |
+
|
| 492 |
+
if not text:
|
| 493 |
+
raise RuntimeError("The configured Granite model returned an empty response.")
|
| 494 |
+
|
| 495 |
+
try:
|
| 496 |
+
json.loads(text)
|
| 497 |
+
except json.JSONDecodeError as error:
|
| 498 |
+
raise RuntimeError(
|
| 499 |
+
"The configured Granite JSON-constrained generation returned invalid JSON."
|
| 500 |
+
) from error
|
| 501 |
+
|
| 502 |
+
if finish_reason == "length":
|
| 503 |
+
raise RuntimeError(
|
| 504 |
+
"The configured Granite model reached the output token limit before completing the repair plan."
|
| 505 |
+
)
|
| 506 |
+
|
| 507 |
+
return text
|
| 508 |
+
|
| 509 |
+
def status(self):
|
| 510 |
+
if not self.enabled:
|
| 511 |
+
state = "disabled"
|
| 512 |
+
elif self.model is not None:
|
| 513 |
+
state = "loaded"
|
| 514 |
+
elif self.load_error:
|
| 515 |
+
state = "load_failed"
|
| 516 |
+
else:
|
| 517 |
+
state = "not_loaded"
|
| 518 |
|
| 519 |
+
return {
|
| 520 |
+
"enabled": self.enabled,
|
| 521 |
+
"loaded": self.model is not None,
|
| 522 |
+
"state": state,
|
| 523 |
+
"backend": "llama.cpp",
|
| 524 |
+
"configured_model": self.primary_model,
|
| 525 |
+
"active_model": self.model_name,
|
| 526 |
+
"model_repo": self.model_repo,
|
| 527 |
+
"model_file": self.model_file,
|
| 528 |
+
"model_revision": self.model_revision,
|
| 529 |
+
"quantization": self.quantization,
|
| 530 |
+
"load_error": self.load_error,
|
| 531 |
+
"load_seconds": self.load_seconds,
|
| 532 |
+
"context_tokens": self.context_tokens,
|
| 533 |
+
"max_input_tokens": self.max_input_tokens,
|
| 534 |
+
"max_new_tokens": self.max_new_tokens,
|
| 535 |
+
"max_generation_seconds": self.max_generation_seconds,
|
| 536 |
+
"threads": self.threads,
|
| 537 |
+
"prompt_cache_mb": self.prompt_cache_mb,
|
| 538 |
+
"last_generation": self.last_generation,
|
| 539 |
+
}
|
| 540 |
|
|
|
|
| 541 |
|
| 542 |
+
model_service = ModelService()
|
| 543 |
+
|
| 544 |
+
app = FastAPI(
|
| 545 |
+
title="Stitch QA Defect Resolution Intelligence Analyst",
|
| 546 |
+
version=AGENT_VERSION,
|
| 547 |
+
)
|
| 548 |
+
|
| 549 |
+
|
| 550 |
+
def runtime_findings(request):
|
| 551 |
+
analysis = request.runtime_analysis or {}
|
| 552 |
+
result = []
|
| 553 |
+
|
| 554 |
+
for index, group in enumerate(
|
| 555 |
+
normalize_list(analysis.get("root_cause_groups")),
|
| 556 |
+
start=1,
|
| 557 |
+
):
|
| 558 |
+
if not isinstance(group, dict):
|
| 559 |
+
continue
|
| 560 |
+
|
| 561 |
+
ref = clean_text(group.get("group_id"), 80) or f"RQI-{index:03d}"
|
| 562 |
+
evidence = []
|
| 563 |
+
locations = []
|
| 564 |
+
|
| 565 |
+
for item in normalize_list(group.get("evidence"))[:8]:
|
| 566 |
+
if not isinstance(item, dict):
|
| 567 |
+
continue
|
| 568 |
+
evidence.append(
|
| 569 |
+
{
|
| 570 |
+
key: item.get(key)
|
| 571 |
+
for key in (
|
| 572 |
+
"failure_id",
|
| 573 |
+
"test_name",
|
| 574 |
+
"expected",
|
| 575 |
+
"actual",
|
| 576 |
+
"exception_type",
|
| 577 |
+
"exception_message",
|
| 578 |
+
"application_file",
|
| 579 |
+
"application_line",
|
| 580 |
+
"test_file",
|
| 581 |
+
"test_line",
|
| 582 |
+
)
|
| 583 |
+
if item.get(key) not in {None, ""}
|
| 584 |
+
}
|
| 585 |
+
)
|
| 586 |
+
file_path = item.get("application_file") or item.get("test_file")
|
| 587 |
+
line = item.get("application_line") or item.get("test_line")
|
| 588 |
+
if file_path:
|
| 589 |
+
locations.append(
|
| 590 |
+
f"{file_path}:{line}" if line else str(file_path)
|
| 591 |
+
)
|
| 592 |
+
|
| 593 |
+
result.append(
|
| 594 |
+
{
|
| 595 |
+
"ref": ref,
|
| 596 |
+
"source": "runtime",
|
| 597 |
+
"severity": normalize_risk(
|
| 598 |
+
analysis.get("runtime_risk_level") or "HIGH"
|
| 599 |
+
),
|
| 600 |
+
"title": clean_text(group.get("title"), 180)
|
| 601 |
+
or "Runtime failure group",
|
| 602 |
+
"category": clean_text(group.get("category"), 120),
|
| 603 |
+
"root_cause": clean_text(group.get("root_cause"), 700),
|
| 604 |
+
"impact": clean_text(group.get("runtime_impact"), 500),
|
| 605 |
+
"recommendation": clean_text(group.get("required_action"), 500),
|
| 606 |
+
"locations": list(dict.fromkeys(locations))[:8],
|
| 607 |
+
"evidence": evidence,
|
| 608 |
+
}
|
| 609 |
+
)
|
| 610 |
|
| 611 |
+
if result:
|
| 612 |
+
return result
|
| 613 |
+
|
| 614 |
+
runtime_evidence = request.runtime_evidence or {}
|
| 615 |
+
test_result = str(runtime_evidence.get("test_result") or "").upper()
|
| 616 |
+
|
| 617 |
+
for index, failure in enumerate(
|
| 618 |
+
normalize_list(runtime_evidence.get("failures")),
|
| 619 |
+
start=1,
|
| 620 |
+
):
|
| 621 |
+
if not isinstance(failure, dict):
|
| 622 |
+
continue
|
| 623 |
+
|
| 624 |
+
ref = clean_text(failure.get("id"), 80) or f"RTE-{index:03d}"
|
| 625 |
+
application_file = failure.get("application_file")
|
| 626 |
+
application_line = failure.get("application_line")
|
| 627 |
+
test_file = failure.get("test_file")
|
| 628 |
+
test_line = failure.get("test_line")
|
| 629 |
+
locations = []
|
| 630 |
+
|
| 631 |
+
if application_file:
|
| 632 |
+
locations.append(
|
| 633 |
+
f"{application_file}:{application_line}"
|
| 634 |
+
if application_line
|
| 635 |
+
else str(application_file)
|
| 636 |
+
)
|
| 637 |
+
if test_file:
|
| 638 |
+
locations.append(
|
| 639 |
+
f"{test_file}:{test_line}"
|
| 640 |
+
if test_line
|
| 641 |
+
else str(test_file)
|
| 642 |
+
)
|
| 643 |
+
|
| 644 |
+
expected = clean_text(failure.get("expected"), 180)
|
| 645 |
+
actual = clean_text(
|
| 646 |
+
failure.get("actual") or failure.get("exception_type"),
|
| 647 |
+
180,
|
| 648 |
+
)
|
| 649 |
+
exception_message = clean_text(failure.get("exception_message"), 300)
|
| 650 |
+
observed = exception_message or actual or "The tested path failed."
|
| 651 |
+
contract_text = (
|
| 652 |
+
f" Expected {expected}; observed {actual}."
|
| 653 |
+
if expected and actual
|
| 654 |
+
else ""
|
| 655 |
+
)
|
| 656 |
|
| 657 |
+
result.append(
|
| 658 |
+
{
|
| 659 |
+
"ref": ref,
|
| 660 |
+
"source": "runtime",
|
| 661 |
+
"severity": "HIGH" if test_result == "FAIL" else "MEDIUM",
|
| 662 |
+
"title": clean_text(
|
| 663 |
+
failure.get("test_name") or failure.get("exception_type"),
|
| 664 |
+
180,
|
| 665 |
+
) or "Validated runtime failure",
|
| 666 |
+
"category": "VALIDATED_RUNTIME_FAILURE",
|
| 667 |
+
"root_cause": clean_text(
|
| 668 |
+
f"Observed failure evidence: {observed}.{contract_text}",
|
| 669 |
+
700,
|
| 670 |
+
),
|
| 671 |
+
"impact": (
|
| 672 |
+
"A validated automated-test path did not complete with its expected behavior."
|
| 673 |
+
),
|
| 674 |
+
"recommendation": (
|
| 675 |
+
"Resolve the evidence-backed behavior mismatch using the smallest safe change, then rerun the affected and regression tests."
|
| 676 |
+
),
|
| 677 |
+
"locations": list(dict.fromkeys(locations)),
|
| 678 |
+
"evidence": [
|
| 679 |
+
{
|
| 680 |
+
key: failure.get(key)
|
| 681 |
+
for key in (
|
| 682 |
+
"id",
|
| 683 |
+
"test_name",
|
| 684 |
+
"expected",
|
| 685 |
+
"actual",
|
| 686 |
+
"exception_type",
|
| 687 |
+
"exception_message",
|
| 688 |
+
"application_file",
|
| 689 |
+
"application_line",
|
| 690 |
+
"test_file",
|
| 691 |
+
"test_line",
|
| 692 |
+
)
|
| 693 |
+
if failure.get(key) not in {None, ""}
|
| 694 |
+
}
|
| 695 |
+
],
|
| 696 |
+
}
|
| 697 |
+
)
|
| 698 |
+
|
| 699 |
+
return result
|
| 700 |
+
|
| 701 |
+
|
| 702 |
+
def runtime_warning_findings(request):
|
| 703 |
+
analysis = request.runtime_analysis or {}
|
| 704 |
+
runtime_evidence = request.runtime_evidence or {}
|
| 705 |
+
warnings = []
|
| 706 |
+
seen = set()
|
| 707 |
+
|
| 708 |
+
for item in [
|
| 709 |
+
*normalize_list(analysis.get("warnings")),
|
| 710 |
+
*normalize_list(runtime_evidence.get("warnings")),
|
| 711 |
+
]:
|
| 712 |
+
warning = clean_text(item, 500)
|
| 713 |
+
key = warning.lower()
|
| 714 |
+
if not warning or key in seen:
|
| 715 |
+
continue
|
| 716 |
+
seen.add(key)
|
| 717 |
+
warnings.append(warning)
|
| 718 |
+
|
| 719 |
+
release_gate = str(analysis.get("release_gate") or "").upper()
|
| 720 |
+
severity = "MEDIUM" if release_gate == "ALLOW_WITH_WARNINGS" else "LOW"
|
| 721 |
+
|
| 722 |
+
return [
|
| 723 |
+
{
|
| 724 |
+
"ref": f"RWI-{index:03d}",
|
| 725 |
+
"source": "runtime",
|
| 726 |
+
"severity": severity,
|
| 727 |
+
"title": "Runtime warning requiring follow-up",
|
| 728 |
+
"category": "RUNTIME_WARNING",
|
| 729 |
+
"root_cause": warning,
|
| 730 |
+
"impact": (
|
| 731 |
+
"The validated run completed, but the warning may affect future compatibility, reliability, or execution behavior if its underlying condition changes."
|
| 732 |
+
),
|
| 733 |
+
"recommendation": (
|
| 734 |
+
"Review the warning-specific configuration or dependency behavior, verify current vendor guidance when needed, and rerun the relevant workflow after any targeted adjustment."
|
| 735 |
+
),
|
| 736 |
+
"locations": [],
|
| 737 |
+
"evidence": [],
|
| 738 |
}
|
| 739 |
+
for index, warning in enumerate(warnings, start=1)
|
| 740 |
+
]
|
| 741 |
|
|
|
|
| 742 |
|
| 743 |
+
def source_findings(request):
|
| 744 |
+
source_review = request.source_review or {}
|
| 745 |
+
result = []
|
| 746 |
+
|
| 747 |
+
for index, finding in enumerate(
|
| 748 |
+
normalize_list(source_review.get("findings")),
|
| 749 |
+
start=1,
|
| 750 |
+
):
|
| 751 |
+
if not isinstance(finding, dict):
|
| 752 |
+
continue
|
| 753 |
+
|
| 754 |
+
ref = clean_text(finding.get("id"), 80) or f"SRC-{index:03d}"
|
| 755 |
+
file_path = finding.get("file_path")
|
| 756 |
+
line = finding.get("line")
|
| 757 |
+
location = None
|
| 758 |
+
if file_path:
|
| 759 |
+
location = f"{file_path}:{line}" if line else str(file_path)
|
| 760 |
+
|
| 761 |
+
result.append(
|
| 762 |
+
{
|
| 763 |
+
"ref": ref,
|
| 764 |
+
"source": "source",
|
| 765 |
+
"severity": normalize_risk(finding.get("severity")),
|
| 766 |
+
"title": clean_text(finding.get("title"), 180)
|
| 767 |
+
or "Source-code finding",
|
| 768 |
+
"category": clean_text(finding.get("category"), 120),
|
| 769 |
+
"root_cause": clean_text(
|
| 770 |
+
finding.get("evidence") or finding.get("description"),
|
| 771 |
+
700,
|
| 772 |
+
),
|
| 773 |
+
"impact": clean_text(finding.get("impact"), 500),
|
| 774 |
+
"recommendation": clean_text(
|
| 775 |
+
finding.get("recommendation"),
|
| 776 |
+
500,
|
| 777 |
+
),
|
| 778 |
+
"locations": [location] if location else [],
|
| 779 |
+
"evidence": [],
|
| 780 |
+
}
|
| 781 |
)
|
| 782 |
+
|
| 783 |
+
return result
|
| 784 |
+
|
| 785 |
+
|
| 786 |
+
def execution_finding(request):
|
| 787 |
+
if request.success:
|
| 788 |
+
return None
|
| 789 |
+
|
| 790 |
+
analysis = request.runtime_analysis or {}
|
| 791 |
+
if normalize_list(analysis.get("root_cause_groups")):
|
| 792 |
+
return None
|
| 793 |
+
|
| 794 |
+
runtime_evidence = request.runtime_evidence or {}
|
| 795 |
+
if normalize_list(runtime_evidence.get("failures")):
|
| 796 |
+
return None
|
| 797 |
+
|
| 798 |
+
failure_type = clean_text(request.failure_type, 120)
|
| 799 |
+
help_message = clean_text(request.help_message, 500)
|
| 800 |
+
discovery_only_failures = {
|
| 801 |
+
"PYTHON_TESTS_NOT_FOUND",
|
| 802 |
+
}
|
| 803 |
+
|
| 804 |
+
if failure_type.upper() in discovery_only_failures:
|
| 805 |
+
return None
|
| 806 |
+
|
| 807 |
+
if "no test files" in help_message.lower() or "no pytest-compatible test" in help_message.lower():
|
| 808 |
+
return None
|
| 809 |
+
|
| 810 |
+
if not failure_type and not help_message:
|
| 811 |
+
return None
|
| 812 |
+
|
| 813 |
+
return {
|
| 814 |
+
"ref": "EXEC-001",
|
| 815 |
+
"source": "execution",
|
| 816 |
+
"severity": "MEDIUM",
|
| 817 |
+
"title": (
|
| 818 |
+
failure_type.replace("_", " ").title()
|
| 819 |
+
if failure_type
|
| 820 |
+
else "Execution blocker"
|
| 821 |
+
),
|
| 822 |
+
"category": "EXECUTION_BLOCKER",
|
| 823 |
+
"root_cause": help_message or "Execution did not produce a validated repairable runtime result.",
|
| 824 |
+
"impact": "The intended QA workflow cannot establish complete runtime assurance until this blocker is resolved.",
|
| 825 |
+
"recommendation": "Resolve the execution blocker and rerun Stitch QA before making application-code repair decisions.",
|
| 826 |
+
"locations": [],
|
| 827 |
+
"evidence": [],
|
| 828 |
}
|
| 829 |
|
| 830 |
|
| 831 |
+
def collect_locked_findings(request):
|
| 832 |
+
findings = []
|
| 833 |
+
findings.extend(runtime_findings(request))
|
| 834 |
+
findings.extend(runtime_warning_findings(request))
|
| 835 |
+
findings.extend(source_findings(request))
|
| 836 |
+
execution = execution_finding(request)
|
| 837 |
+
if execution:
|
| 838 |
+
findings.append(execution)
|
| 839 |
+
|
| 840 |
+
seen = set()
|
| 841 |
+
normalized = []
|
| 842 |
+
for finding in findings:
|
| 843 |
+
ref = finding["ref"]
|
| 844 |
+
if ref in seen:
|
| 845 |
+
continue
|
| 846 |
+
seen.add(ref)
|
| 847 |
+
normalized.append(finding)
|
| 848 |
+
|
| 849 |
+
return normalized
|
| 850 |
+
|
| 851 |
+
|
| 852 |
+
def select_ai_findings(findings):
|
| 853 |
+
source_order = {
|
| 854 |
+
"execution": 0,
|
| 855 |
+
"runtime": 1,
|
| 856 |
+
"source": 2,
|
| 857 |
+
}
|
| 858 |
|
| 859 |
+
ordered = sorted(
|
| 860 |
+
findings,
|
| 861 |
+
key=lambda finding: (
|
| 862 |
+
-RISK_ORDER.get(normalize_risk(finding.get("severity")), -1),
|
| 863 |
+
source_order.get(str(finding.get("source") or ""), 3),
|
| 864 |
+
str(finding.get("ref") or ""),
|
| 865 |
+
),
|
| 866 |
+
)
|
| 867 |
+
return ordered[:MAX_AI_FINDINGS], ordered[MAX_AI_FINDINGS:]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 868 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 869 |
|
| 870 |
+
def overall_priority_floor(request, findings):
|
| 871 |
+
analysis = request.runtime_analysis or {}
|
| 872 |
+
release_gate = str(analysis.get("release_gate") or "").upper()
|
| 873 |
|
| 874 |
+
if release_gate == "BLOCK_RELEASE":
|
| 875 |
+
return "P1"
|
| 876 |
|
| 877 |
+
runtime_evidence = request.runtime_evidence or {}
|
| 878 |
+
if str(runtime_evidence.get("test_result") or "").upper() == "FAIL":
|
| 879 |
+
return "P1"
|
| 880 |
|
| 881 |
+
if any(
|
| 882 |
+
finding.get("source") == "execution"
|
| 883 |
+
for finding in findings
|
| 884 |
+
):
|
| 885 |
+
return "P1"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 886 |
|
| 887 |
+
if any(
|
| 888 |
+
normalize_risk(finding.get("severity")) == "CRITICAL"
|
| 889 |
+
for finding in findings
|
| 890 |
+
):
|
| 891 |
+
return "P1"
|
| 892 |
|
| 893 |
+
return "P3"
|
|
|
|
| 894 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 895 |
|
| 896 |
+
def contract_priority_floor(request, contract, findings):
|
| 897 |
+
refs = set(contract.finding_refs)
|
| 898 |
+
selected = [finding for finding in findings if finding.get("ref") in refs]
|
|
|
|
| 899 |
|
| 900 |
+
if any(finding.get("source") == "execution" for finding in selected):
|
| 901 |
+
return "P1"
|
| 902 |
+
|
| 903 |
+
runtime_evidence = request.runtime_evidence or {}
|
| 904 |
+
if (
|
| 905 |
+
str(runtime_evidence.get("test_result") or "").upper() == "FAIL"
|
| 906 |
+
and any(finding.get("source") == "runtime" for finding in selected)
|
| 907 |
+
):
|
| 908 |
+
return "P1"
|
| 909 |
+
|
| 910 |
+
if any(
|
| 911 |
+
finding.get("source") == "source"
|
| 912 |
+
and normalize_risk(finding.get("severity")) == "CRITICAL"
|
| 913 |
+
for finding in selected
|
| 914 |
+
):
|
| 915 |
+
return "P1"
|
| 916 |
+
|
| 917 |
+
return "P3"
|
| 918 |
+
|
| 919 |
+
|
| 920 |
+
def deterministic_confidence(request, findings):
|
| 921 |
+
analysis = request.runtime_analysis or {}
|
| 922 |
+
runtime_confidence = str(
|
| 923 |
+
analysis.get("diagnosis_confidence") or ""
|
| 924 |
+
).upper()
|
| 925 |
+
|
| 926 |
+
if runtime_confidence in {"HIGH", "MEDIUM", "LOW"}:
|
| 927 |
+
return runtime_confidence
|
| 928 |
+
|
| 929 |
+
source_review = request.source_review or {}
|
| 930 |
+
source_confidences = {
|
| 931 |
+
str(item.get("confidence") or "").upper()
|
| 932 |
+
for item in normalize_list(source_review.get("findings"))
|
| 933 |
+
if isinstance(item, dict)
|
| 934 |
+
}
|
| 935 |
+
|
| 936 |
+
if source_confidences == {"HIGH"}:
|
| 937 |
+
return "HIGH"
|
| 938 |
+
|
| 939 |
+
return "MEDIUM" if findings else "HIGH"
|
| 940 |
+
|
| 941 |
+
|
| 942 |
+
def fallback_contract(finding, index, request):
|
| 943 |
+
priority = severity_priority(finding.get("severity"))
|
| 944 |
+
|
| 945 |
+
if finding.get("source") == "execution":
|
| 946 |
+
priority = "P1"
|
| 947 |
+
|
| 948 |
+
objective = (
|
| 949 |
+
finding.get("recommendation")
|
| 950 |
+
or f"Resolve the confirmed condition represented by {finding['ref']}."
|
| 951 |
+
)
|
| 952 |
+
strategy = (
|
| 953 |
+
finding.get("recommendation")
|
| 954 |
+
or "Apply the smallest targeted change that resolves the confirmed condition without broad unrelated changes."
|
| 955 |
+
)
|
| 956 |
+
|
| 957 |
+
runtime_analysis = request.runtime_analysis or {}
|
| 958 |
+
verification_steps = normalize_list(
|
| 959 |
+
runtime_analysis.get("verification_steps")
|
| 960 |
+
)
|
| 961 |
+
verification = (
|
| 962 |
+
" ".join(clean_text(item, 180) for item in verification_steps[:3])
|
| 963 |
+
if verification_steps
|
| 964 |
+
else "Confirm the affected behavior first, then run the relevant regression suite and verify no new failures."
|
| 965 |
+
)
|
| 966 |
+
|
| 967 |
+
return {
|
| 968 |
+
"contract_id": f"STITCH-RC-{index:03d}",
|
| 969 |
+
"finding_refs": [finding["ref"]],
|
| 970 |
+
"title": clean_text(finding.get("title"), 140),
|
| 971 |
+
"priority": priority,
|
| 972 |
+
"priority_reason": (
|
| 973 |
+
f"{finding['ref']} is prioritized from its validated impact and current QA blocking effect."
|
| 974 |
+
),
|
| 975 |
+
"repair_objective": clean_text(objective, 260),
|
| 976 |
+
"repair_strategy": clean_text(strategy, 320),
|
| 977 |
+
"change_boundary": (
|
| 978 |
+
"Limit the change to the component, input boundary, dependency, or configuration directly represented by this finding."
|
| 979 |
+
),
|
| 980 |
+
"protected_behavior": (
|
| 981 |
+
"Preserve behavior already shown to work outside the affected scope and avoid unrelated refactoring."
|
| 982 |
+
),
|
| 983 |
+
"side_effect_risk": (
|
| 984 |
+
"MEDIUM"
|
| 985 |
+
if normalize_risk(finding.get("severity")) in {"CRITICAL", "HIGH", "MEDIUM"}
|
| 986 |
+
else "LOW"
|
| 987 |
+
),
|
| 988 |
+
"verification": clean_text(verification, 300),
|
| 989 |
+
"done_condition": (
|
| 990 |
+
f"{finding['ref']} is no longer reproducible and the relevant regression checks complete without new failures."
|
| 991 |
+
),
|
| 992 |
+
"status": "PENDING_VERIFICATION",
|
| 993 |
+
}
|
| 994 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 995 |
|
| 996 |
+
def finding_may_need_current_knowledge(finding):
|
| 997 |
+
text = " ".join(
|
| 998 |
+
clean_text(finding.get(key), 500).lower()
|
| 999 |
+
for key in (
|
| 1000 |
+
"title",
|
| 1001 |
+
"category",
|
| 1002 |
+
"root_cause",
|
| 1003 |
+
"impact",
|
| 1004 |
+
"recommendation",
|
| 1005 |
)
|
| 1006 |
+
)
|
| 1007 |
+
indicators = (
|
| 1008 |
+
"dependency",
|
| 1009 |
+
"version",
|
| 1010 |
+
"deprecated",
|
| 1011 |
+
"deprecation",
|
| 1012 |
+
"compatibility",
|
| 1013 |
+
"plugin",
|
| 1014 |
+
"jdk",
|
| 1015 |
+
"cve",
|
| 1016 |
+
"security advisory",
|
| 1017 |
+
"vendor",
|
| 1018 |
+
"repository",
|
| 1019 |
+
)
|
| 1020 |
+
return any(indicator in text for indicator in indicators)
|
| 1021 |
+
|
| 1022 |
+
|
| 1023 |
+
def build_fallback_plan(request, findings, mode="deterministic-fallback", llm_error=None):
|
| 1024 |
+
if not findings:
|
| 1025 |
+
return {
|
| 1026 |
+
"agent_id": AGENT_ID,
|
| 1027 |
+
"display_name": DISPLAY_NAME,
|
| 1028 |
+
"agent_version": AGENT_VERSION,
|
| 1029 |
+
"agent": "repair-agent",
|
| 1030 |
+
"mode": "evidence-validated",
|
| 1031 |
+
"model": None,
|
| 1032 |
+
"status": (
|
| 1033 |
+
"NO_REPAIR_REQUIRED"
|
| 1034 |
+
if request.success
|
| 1035 |
+
else "NO_CONFIRMED_REPAIR_TARGET"
|
| 1036 |
+
),
|
| 1037 |
+
"overall_priority": "NONE",
|
| 1038 |
+
"confidence": "HIGH",
|
| 1039 |
+
"repair_risk_level": "NONE",
|
| 1040 |
+
"auto_apply": False,
|
| 1041 |
+
"summary": (
|
| 1042 |
+
"No confirmed repair target was supplied by runtime or source-review evidence."
|
| 1043 |
+
if request.success
|
| 1044 |
+
else "Execution did not succeed, but no evidence-grounded repair target was available; no source change should be guessed."
|
| 1045 |
+
),
|
| 1046 |
+
"stitch_repair_contracts": [],
|
| 1047 |
+
"current_knowledge_required": False,
|
| 1048 |
+
"current_knowledge_reason": None,
|
| 1049 |
+
"next_action": (
|
| 1050 |
+
"Retain the available QA evidence and continue the normal release workflow."
|
| 1051 |
+
if request.success
|
| 1052 |
+
else "Obtain a confirmed runtime, execution, or source-review finding before planning a code repair."
|
| 1053 |
+
),
|
| 1054 |
+
"suggestions": [],
|
| 1055 |
+
"warnings": [],
|
| 1056 |
+
"limitations": [
|
| 1057 |
+
"Agent 2 only plans repairs for findings supplied by validated Stitch QA evidence."
|
| 1058 |
+
],
|
| 1059 |
+
"llm_metrics": model_service.last_generation,
|
| 1060 |
+
"llm_error": llm_error,
|
| 1061 |
+
}
|
| 1062 |
+
|
| 1063 |
+
contracts = [
|
| 1064 |
+
fallback_contract(finding, index, request)
|
| 1065 |
+
for index, finding in enumerate(findings, start=1)
|
| 1066 |
+
]
|
| 1067 |
+
contracts = sort_and_renumber_contracts(contracts)
|
| 1068 |
+
highest_priority = max(
|
| 1069 |
+
(contract["priority"] for contract in contracts),
|
| 1070 |
+
key=lambda item: PRIORITY_ORDER[item],
|
| 1071 |
+
default="P3",
|
| 1072 |
+
)
|
| 1073 |
+
floor = overall_priority_floor(request, findings)
|
| 1074 |
+
if PRIORITY_ORDER[highest_priority] < PRIORITY_ORDER[floor]:
|
| 1075 |
+
highest_priority = floor
|
| 1076 |
+
|
| 1077 |
+
repair_risk = highest_risk(
|
| 1078 |
+
*(contract["side_effect_risk"] for contract in contracts)
|
| 1079 |
+
)
|
| 1080 |
+
|
| 1081 |
+
warnings = []
|
| 1082 |
|
| 1083 |
return {
|
| 1084 |
+
"agent_id": AGENT_ID,
|
| 1085 |
+
"display_name": DISPLAY_NAME,
|
| 1086 |
+
"agent_version": AGENT_VERSION,
|
| 1087 |
"agent": "repair-agent",
|
| 1088 |
+
"mode": mode,
|
| 1089 |
+
"model": model_service.model_name or model_service.primary_model,
|
| 1090 |
+
"status": "COMPLETED",
|
| 1091 |
+
"overall_priority": highest_priority,
|
| 1092 |
+
"confidence": deterministic_confidence(request, findings),
|
| 1093 |
+
"repair_risk_level": repair_risk,
|
| 1094 |
"auto_apply": False,
|
| 1095 |
+
"summary": (
|
| 1096 |
+
f"Prepared {len(contracts)} evidence-linked repair contract"
|
| 1097 |
+
f"{'s' if len(contracts) != 1 else ''} from {len(findings)} confirmed finding"
|
| 1098 |
+
f"{'s' if len(findings) != 1 else ''}."
|
| 1099 |
+
),
|
| 1100 |
+
"stitch_repair_contracts": contracts,
|
| 1101 |
+
"current_knowledge_required": any(
|
| 1102 |
+
finding_may_need_current_knowledge(finding)
|
| 1103 |
+
for finding in findings
|
| 1104 |
+
),
|
| 1105 |
+
"current_knowledge_reason": (
|
| 1106 |
+
"One or more repair targets depend on version, dependency, compatibility, vendor, deprecation, or security information that should be verified against current trusted documentation."
|
| 1107 |
+
if any(
|
| 1108 |
+
finding_may_need_current_knowledge(finding)
|
| 1109 |
+
for finding in findings
|
| 1110 |
+
)
|
| 1111 |
+
else None
|
| 1112 |
+
),
|
| 1113 |
+
"next_action": (
|
| 1114 |
+
f"Start with {contracts[0]['contract_id']} and verify its done condition before broadening the repair scope."
|
| 1115 |
+
),
|
| 1116 |
+
"suggestions": [
|
| 1117 |
+
contract["repair_strategy"]
|
| 1118 |
+
for contract in contracts
|
| 1119 |
+
],
|
| 1120 |
+
"warnings": warnings,
|
| 1121 |
+
"limitations": [
|
| 1122 |
+
"This repair plan is grounded in supplied Stitch QA findings and does not automatically modify project code.",
|
| 1123 |
+
"Deterministic fallback prioritization is conservative and may be less context-sensitive than validated AI planning.",
|
| 1124 |
+
],
|
| 1125 |
+
"llm_metrics": model_service.last_generation,
|
| 1126 |
+
"llm_error": llm_error,
|
| 1127 |
}
|
| 1128 |
|
| 1129 |
|
| 1130 |
+
def compact_model_evidence(items):
|
| 1131 |
+
compacted = []
|
| 1132 |
+
for item in normalize_list(items)[:2]:
|
| 1133 |
+
if not isinstance(item, dict):
|
| 1134 |
+
continue
|
| 1135 |
+
compacted.append(
|
| 1136 |
+
{
|
| 1137 |
+
"id": clean_text(item.get("failure_id") or item.get("id"), 60),
|
| 1138 |
+
"test": clean_text(item.get("test_name"), 160),
|
| 1139 |
+
"expected": clean_text(item.get("expected"), 120),
|
| 1140 |
+
"actual": clean_text(item.get("actual"), 120),
|
| 1141 |
+
"exception": clean_text(item.get("exception_type"), 100),
|
| 1142 |
+
"message": clean_text(item.get("exception_message"), 220),
|
| 1143 |
+
"application": clean_text(item.get("application_file"), 200),
|
| 1144 |
+
"application_line": item.get("application_line"),
|
| 1145 |
+
"test_file": clean_text(item.get("test_file"), 200),
|
| 1146 |
+
"test_line": item.get("test_line"),
|
| 1147 |
+
}
|
| 1148 |
+
)
|
| 1149 |
+
return [
|
| 1150 |
+
{key: value for key, value in item.items() if value not in {None, ""}}
|
| 1151 |
+
for item in compacted
|
| 1152 |
+
]
|
| 1153 |
+
|
| 1154 |
+
|
| 1155 |
+
def model_input_findings(findings):
|
| 1156 |
+
payload = []
|
| 1157 |
+
|
| 1158 |
+
for finding in findings[:MAX_AI_FINDINGS]:
|
| 1159 |
+
payload.append(
|
| 1160 |
+
{
|
| 1161 |
+
"ref": finding["ref"],
|
| 1162 |
+
"source": finding["source"],
|
| 1163 |
+
"severity": finding["severity"],
|
| 1164 |
+
"title": clean_text(finding.get("title"), 100),
|
| 1165 |
+
"category": clean_text(finding.get("category"), 70),
|
| 1166 |
+
"cause": clean_text(finding.get("root_cause"), 240),
|
| 1167 |
+
"impact": clean_text(finding.get("impact"), 180),
|
| 1168 |
+
"existing_action": clean_text(finding.get("recommendation"), 180),
|
| 1169 |
+
"locations": [
|
| 1170 |
+
clean_text(location, 160)
|
| 1171 |
+
for location in finding.get("locations", [])[:4]
|
| 1172 |
+
],
|
| 1173 |
+
"evidence": compact_model_evidence(finding.get("evidence", [])),
|
| 1174 |
+
}
|
| 1175 |
+
)
|
| 1176 |
|
| 1177 |
+
return payload
|
| 1178 |
+
|
| 1179 |
+
|
| 1180 |
+
def build_messages(request, findings):
|
| 1181 |
+
analysis = request.runtime_analysis or {}
|
| 1182 |
+
source_review = request.source_review or {}
|
| 1183 |
+
|
| 1184 |
+
system_prompt = (
|
| 1185 |
+
"You are Stitch QA's Defect Resolution Intelligence Analyst. "
|
| 1186 |
+
"Your job is to transform locked QA findings into the safest prioritized repair plan without editing code. "
|
| 1187 |
+
"The supplied finding references, severities, files, lines, test facts, release gate, and observed evidence are authoritative. "
|
| 1188 |
+
"Never invent findings, files, lines, test outcomes, dependency versions, vulnerabilities, or current external facts. "
|
| 1189 |
+
"Prioritize by impact, blocking effect, dependency order, repair scope, and regression risk; severity and repair priority are related but not identical. "
|
| 1190 |
+
"Group findings only when one repair objective genuinely resolves them together. "
|
| 1191 |
+
"For every repair contract define the smallest useful change boundary, behavior that must remain working, side-effect risk, confirmation verification, regression verification, and a measurable done condition. "
|
| 1192 |
+
"Do not generate patches, code, commits, commands that modify the project, or automatic fixes. "
|
| 1193 |
+
"If a version, vendor behavior, dependency compatibility, deprecation, or security advisory needs up-to-date external documentation, set current_knowledge_required=true and explain why; do not invent the missing current fact. "
|
| 1194 |
+
"Return only JSON matching the required schema. "
|
| 1195 |
+
"Compact keys are fixed: p=overall priority, c=confidence, k=current-knowledge-needed, kr=current-knowledge reason, x=repair contracts; "
|
| 1196 |
+
"inside each contract f=finding refs, p=priority, w=priority reason, o=repair objective, s=repair strategy, b=change boundary, q=protected behavior, r=side-effect risk (L/M/H), v=verification. "
|
| 1197 |
+
"Keep every text value concise because the final professional report is formatted by Stitch QA."
|
| 1198 |
+
)
|
| 1199 |
|
| 1200 |
+
payload = {
|
| 1201 |
+
"project": {
|
| 1202 |
+
"type": request.project_type,
|
| 1203 |
"command": request.command,
|
|
|
|
|
|
|
| 1204 |
"success": request.success,
|
| 1205 |
+
"exit_code": request.exit_code,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1206 |
"failure_type": request.failure_type,
|
| 1207 |
+
},
|
| 1208 |
+
"runtime_gate": {
|
| 1209 |
+
"test_result": analysis.get("test_result"),
|
| 1210 |
+
"release_gate": analysis.get("release_gate"),
|
| 1211 |
+
"runtime_risk": analysis.get("runtime_risk_level"),
|
| 1212 |
+
"confidence": analysis.get("diagnosis_confidence"),
|
| 1213 |
+
"evidence_quality": analysis.get("evidence_quality"),
|
| 1214 |
+
},
|
| 1215 |
+
"source_review": {
|
| 1216 |
+
"status": source_review.get("status"),
|
| 1217 |
+
"risk_level": source_review.get("risk_level"),
|
| 1218 |
+
"findings_count": len(
|
| 1219 |
+
normalize_list(source_review.get("findings"))
|
| 1220 |
+
),
|
| 1221 |
+
},
|
| 1222 |
+
"locked_findings": model_input_findings(findings),
|
| 1223 |
+
"instructions": {
|
| 1224 |
+
"all_findings_must_be_covered": True,
|
| 1225 |
+
"contract_order_is_repair_order": True,
|
| 1226 |
+
"auto_apply": False,
|
| 1227 |
+
},
|
| 1228 |
+
}
|
| 1229 |
|
| 1230 |
+
return [
|
| 1231 |
+
{
|
| 1232 |
+
"role": "system",
|
| 1233 |
+
"content": system_prompt,
|
| 1234 |
+
},
|
| 1235 |
+
{
|
| 1236 |
+
"role": "user",
|
| 1237 |
+
"content": json.dumps(
|
| 1238 |
+
payload,
|
| 1239 |
+
ensure_ascii=False,
|
| 1240 |
+
separators=(",", ":"),
|
| 1241 |
+
),
|
| 1242 |
+
},
|
| 1243 |
+
]
|
| 1244 |
|
|
|
|
|
|
|
| 1245 |
|
| 1246 |
+
FILE_LINE_PATTERN = re.compile(
|
| 1247 |
+
r"(?P<path>[A-Za-z0-9_./\\-]+\.[A-Za-z0-9]+):(?P<line>\d+)"
|
| 1248 |
+
)
|
|
|
|
|
|
|
|
|
|
| 1249 |
|
| 1250 |
+
AUTO_APPLY_PATTERNS = {
|
| 1251 |
+
"i modified",
|
| 1252 |
+
"i changed",
|
| 1253 |
+
"i updated",
|
| 1254 |
+
"automatically modified",
|
| 1255 |
+
"automatically fixed",
|
| 1256 |
+
"auto-fix applied",
|
| 1257 |
+
"committed the",
|
| 1258 |
+
"pushed the",
|
| 1259 |
+
}
|
| 1260 |
|
|
|
|
|
|
|
| 1261 |
|
| 1262 |
+
def parse_model_output(text):
|
| 1263 |
+
try:
|
| 1264 |
+
data = json.loads(str(text or "").strip())
|
| 1265 |
+
except json.JSONDecodeError as error:
|
| 1266 |
+
raise ValueError("The Granite response was not valid JSON.") from error
|
| 1267 |
|
| 1268 |
+
try:
|
| 1269 |
+
return ModelRepairPlan.model_validate(data)
|
| 1270 |
+
except ValidationError as error:
|
| 1271 |
+
raise ValueError(
|
| 1272 |
+
f"The Granite response failed the repair-plan schema: {error}"
|
| 1273 |
+
) from error
|
| 1274 |
+
|
| 1275 |
+
|
| 1276 |
+
def allowed_references(findings):
|
| 1277 |
+
allowed = set()
|
| 1278 |
+
|
| 1279 |
+
for finding in findings:
|
| 1280 |
+
for location in finding.get("locations", []):
|
| 1281 |
+
match = FILE_LINE_PATTERN.search(str(location))
|
| 1282 |
+
if not match:
|
| 1283 |
+
continue
|
| 1284 |
+
path = match.group("path").replace("\\", "/")
|
| 1285 |
+
line = int(match.group("line"))
|
| 1286 |
+
allowed.add((path, line))
|
| 1287 |
+
allowed.add((path.lstrip("./"), line))
|
| 1288 |
+
|
| 1289 |
+
return allowed
|
| 1290 |
+
|
| 1291 |
+
|
| 1292 |
+
def text_fields(plan):
|
| 1293 |
+
values = [plan.current_knowledge_reason]
|
| 1294 |
+
for contract in plan.contracts:
|
| 1295 |
+
values.extend(
|
| 1296 |
+
[
|
| 1297 |
+
contract.priority_reason,
|
| 1298 |
+
contract.repair_objective,
|
| 1299 |
+
contract.repair_strategy,
|
| 1300 |
+
contract.change_boundary,
|
| 1301 |
+
contract.protected_behavior,
|
| 1302 |
+
contract.verification,
|
| 1303 |
+
]
|
| 1304 |
+
)
|
| 1305 |
+
return values
|
| 1306 |
+
|
| 1307 |
+
|
| 1308 |
+
def validate_plan(plan, request, findings):
|
| 1309 |
+
finding_refs = {finding["ref"] for finding in findings}
|
| 1310 |
+
used_refs = []
|
| 1311 |
+
combined_text = " ".join(text_fields(plan)).lower()
|
| 1312 |
+
|
| 1313 |
+
for pattern in AUTO_APPLY_PATTERNS:
|
| 1314 |
+
if pattern in combined_text:
|
| 1315 |
+
raise ValueError(
|
| 1316 |
+
"The Granite repair plan claimed or proposed automatic project modification."
|
| 1317 |
+
)
|
| 1318 |
+
|
| 1319 |
+
for contract in plan.contracts:
|
| 1320 |
+
for ref in contract.finding_refs:
|
| 1321 |
+
if ref not in finding_refs:
|
| 1322 |
+
raise ValueError(
|
| 1323 |
+
f"The Granite repair plan referenced unknown finding {ref}."
|
| 1324 |
+
)
|
| 1325 |
+
used_refs.append(ref)
|
| 1326 |
+
|
| 1327 |
+
missing_refs = finding_refs - set(used_refs)
|
| 1328 |
+
if missing_refs:
|
| 1329 |
+
raise ValueError(
|
| 1330 |
+
"The Granite repair plan omitted confirmed findings: "
|
| 1331 |
+
+ ", ".join(sorted(missing_refs))
|
| 1332 |
+
)
|
| 1333 |
|
| 1334 |
+
duplicates = {
|
| 1335 |
+
ref
|
| 1336 |
+
for ref in used_refs
|
| 1337 |
+
if used_refs.count(ref) > 1
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1338 |
}
|
| 1339 |
+
if duplicates:
|
| 1340 |
+
raise ValueError(
|
| 1341 |
+
"The Granite repair plan assigned findings to multiple repair contracts: "
|
| 1342 |
+
+ ", ".join(sorted(duplicates))
|
| 1343 |
+
)
|
| 1344 |
|
| 1345 |
+
for contract in plan.contracts:
|
| 1346 |
+
floor = contract_priority_floor(request, contract, findings)
|
| 1347 |
+
if PRIORITY_ORDER[contract.priority] < PRIORITY_ORDER[floor]:
|
| 1348 |
+
contract.priority = floor
|
| 1349 |
+
|
| 1350 |
+
allowed = allowed_references(findings)
|
| 1351 |
+
for value in text_fields(plan):
|
| 1352 |
+
for match in FILE_LINE_PATTERN.finditer(value or ""):
|
| 1353 |
+
path = match.group("path").replace("\\", "/")
|
| 1354 |
+
line = int(match.group("line"))
|
| 1355 |
+
if (path, line) not in allowed and (path.lstrip("./"), line) not in allowed:
|
| 1356 |
+
raise ValueError(
|
| 1357 |
+
"The Granite repair plan introduced an unsupported file or line reference."
|
| 1358 |
+
)
|
| 1359 |
+
|
| 1360 |
+
floor = overall_priority_floor(request, findings)
|
| 1361 |
+
if PRIORITY_ORDER[plan.overall_priority] < PRIORITY_ORDER[floor]:
|
| 1362 |
+
plan.overall_priority = floor
|
| 1363 |
+
|
| 1364 |
+
highest_contract_priority = max(
|
| 1365 |
+
(contract.priority for contract in plan.contracts),
|
| 1366 |
+
key=lambda item: PRIORITY_ORDER[item],
|
| 1367 |
+
)
|
| 1368 |
+
if PRIORITY_ORDER[plan.overall_priority] < PRIORITY_ORDER[highest_contract_priority]:
|
| 1369 |
+
plan.overall_priority = highest_contract_priority
|
| 1370 |
|
| 1371 |
+
if not plan.current_knowledge_required:
|
| 1372 |
+
plan.current_knowledge_reason = ""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1373 |
|
| 1374 |
+
return plan
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1375 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1376 |
|
| 1377 |
+
def sort_and_renumber_contracts(contracts):
|
| 1378 |
+
ordered = sorted(
|
| 1379 |
+
contracts,
|
| 1380 |
+
key=lambda contract: -PRIORITY_ORDER.get(contract.get("priority", "P3"), 1),
|
|
|
|
| 1381 |
)
|
| 1382 |
+
for index, contract in enumerate(ordered, start=1):
|
| 1383 |
+
contract["contract_id"] = f"STITCH-RC-{index:03d}"
|
| 1384 |
+
return ordered
|
| 1385 |
+
|
| 1386 |
+
|
| 1387 |
+
def merge_model_plan(plan, request, findings, overflow_findings=None):
|
| 1388 |
+
overflow_findings = list(overflow_findings or [])
|
| 1389 |
+
contracts = []
|
| 1390 |
+
|
| 1391 |
+
for index, item in enumerate(plan.contracts, start=1):
|
| 1392 |
+
contracts.append(
|
| 1393 |
+
{
|
| 1394 |
+
"contract_id": f"STITCH-RC-{index:03d}",
|
| 1395 |
+
"finding_refs": item.finding_refs,
|
| 1396 |
+
"title": clean_text(
|
| 1397 |
+
f"Repair contract for {', '.join(item.finding_refs)}",
|
| 1398 |
+
140,
|
| 1399 |
+
),
|
| 1400 |
+
"priority": item.priority,
|
| 1401 |
+
"priority_reason": clean_text(item.priority_reason, 220),
|
| 1402 |
+
"repair_objective": clean_text(item.repair_objective, 260),
|
| 1403 |
+
"repair_strategy": clean_text(item.repair_strategy, 320),
|
| 1404 |
+
"change_boundary": clean_text(item.change_boundary, 240),
|
| 1405 |
+
"protected_behavior": clean_text(item.protected_behavior, 240),
|
| 1406 |
+
"side_effect_risk": item.side_effect_risk,
|
| 1407 |
+
"verification": clean_text(item.verification, 300),
|
| 1408 |
+
"done_condition": (
|
| 1409 |
+
"The referenced findings no longer reproduce, the targeted confirmation succeeds, and the stated regression verification introduces no new failure."
|
| 1410 |
+
),
|
| 1411 |
+
"status": "PENDING_VERIFICATION",
|
| 1412 |
+
}
|
| 1413 |
+
)
|
| 1414 |
|
| 1415 |
+
next_index = len(contracts) + 1
|
| 1416 |
+
for offset, finding in enumerate(overflow_findings):
|
| 1417 |
+
contract = fallback_contract(finding, next_index + offset, request)
|
| 1418 |
+
contracts.append(contract)
|
| 1419 |
|
| 1420 |
+
contracts = sort_and_renumber_contracts(contracts)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1421 |
|
| 1422 |
+
repair_risk = highest_risk(
|
| 1423 |
+
*(contract["side_effect_risk"] for contract in contracts)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1424 |
)
|
| 1425 |
|
| 1426 |
+
overall_priority = plan.overall_priority
|
| 1427 |
+
if overflow_findings:
|
| 1428 |
+
overflow_priority = max(
|
| 1429 |
+
(severity_priority(item.get("severity")) for item in overflow_findings),
|
| 1430 |
+
key=lambda item: PRIORITY_ORDER[item],
|
| 1431 |
+
default="P3",
|
| 1432 |
+
)
|
| 1433 |
+
if PRIORITY_ORDER[overall_priority] < PRIORITY_ORDER[overflow_priority]:
|
| 1434 |
+
overall_priority = overflow_priority
|
| 1435 |
+
|
| 1436 |
+
limitations = [
|
| 1437 |
+
"Agent 2 plans repairs from supplied Stitch QA evidence and does not automatically modify source code.",
|
| 1438 |
+
"Current external facts are not fetched inside this agent; cases marked current_knowledge_required need trusted documentation before implementation.",
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1439 |
]
|
| 1440 |
+
warnings = []
|
| 1441 |
|
| 1442 |
+
if overflow_findings:
|
| 1443 |
+
warnings.append(
|
| 1444 |
+
f"{len(overflow_findings)} lower-priority finding(s) exceeded the AI reasoning window and were retained as conservative evidence-grounded repair contracts rather than being dropped."
|
| 1445 |
+
)
|
| 1446 |
+
limitations.append(
|
| 1447 |
+
"Overflow repair contracts use conservative deterministic planning because the AI reasoning window is intentionally bounded for CPU reliability."
|
| 1448 |
+
)
|
| 1449 |
|
| 1450 |
+
return {
|
| 1451 |
+
"agent_id": AGENT_ID,
|
| 1452 |
+
"display_name": DISPLAY_NAME,
|
| 1453 |
+
"agent_version": AGENT_VERSION,
|
| 1454 |
+
"agent": "repair-agent",
|
| 1455 |
+
"mode": "ai-reasoned-validated",
|
| 1456 |
+
"model": model_service.model_name or model_service.primary_model,
|
| 1457 |
+
"status": "COMPLETED",
|
| 1458 |
+
"overall_priority": overall_priority,
|
| 1459 |
+
"confidence": plan.confidence,
|
| 1460 |
+
"repair_risk_level": repair_risk,
|
| 1461 |
+
"auto_apply": False,
|
| 1462 |
+
"summary": clean_text(
|
| 1463 |
+
f"Prepared {len(contracts)} prioritized evidence-linked repair contract"
|
| 1464 |
+
f"{'s' if len(contracts) != 1 else ''}; start with "
|
| 1465 |
+
f"{contracts[0]['repair_strategy'] if contracts else 'the highest-priority validated repair target'}.",
|
| 1466 |
+
360,
|
| 1467 |
+
),
|
| 1468 |
+
"stitch_repair_contracts": contracts,
|
| 1469 |
+
"current_knowledge_required": (
|
| 1470 |
+
plan.current_knowledge_required
|
| 1471 |
+
or any(
|
| 1472 |
+
finding_may_need_current_knowledge(finding)
|
| 1473 |
+
for finding in [*findings, *overflow_findings]
|
| 1474 |
+
)
|
| 1475 |
+
),
|
| 1476 |
+
"current_knowledge_reason": (
|
| 1477 |
+
clean_text(plan.current_knowledge_reason, 240)
|
| 1478 |
+
if plan.current_knowledge_required and clean_text(plan.current_knowledge_reason, 240)
|
| 1479 |
+
else (
|
| 1480 |
+
"One or more repair targets depend on current version, dependency, compatibility, vendor, deprecation, or security documentation that must be verified before implementation."
|
| 1481 |
+
if any(
|
| 1482 |
+
finding_may_need_current_knowledge(finding)
|
| 1483 |
+
for finding in [*findings, *overflow_findings]
|
| 1484 |
+
)
|
| 1485 |
+
else None
|
| 1486 |
+
)
|
| 1487 |
+
),
|
| 1488 |
+
"next_action": (
|
| 1489 |
+
f"Start with {contracts[0]['contract_id']}: {contracts[0]['repair_strategy']}"
|
| 1490 |
+
if contracts
|
| 1491 |
+
else None
|
| 1492 |
+
),
|
| 1493 |
+
"suggestions": [
|
| 1494 |
+
contract["repair_strategy"]
|
| 1495 |
+
for contract in contracts
|
| 1496 |
+
],
|
| 1497 |
+
"warnings": warnings,
|
| 1498 |
+
"limitations": limitations,
|
| 1499 |
+
"llm_metrics": model_service.last_generation,
|
| 1500 |
+
"llm_error": None,
|
| 1501 |
+
}
|
| 1502 |
|
| 1503 |
+
def should_use_llm(findings):
|
| 1504 |
+
return (
|
| 1505 |
+
model_service.enabled
|
| 1506 |
+
and bool(findings)
|
| 1507 |
+
and any(
|
| 1508 |
+
finding.get("source") in {"runtime", "source"}
|
| 1509 |
+
for finding in findings
|
| 1510 |
+
)
|
| 1511 |
+
)
|
| 1512 |
|
|
|
|
| 1513 |
|
| 1514 |
+
@app.get("/")
|
| 1515 |
+
def health_check():
|
| 1516 |
+
status = model_service.status()
|
| 1517 |
+
return {
|
| 1518 |
+
"service": "stitch-qa-repair-agent",
|
| 1519 |
+
"agent_id": AGENT_ID,
|
| 1520 |
+
"display_name": DISPLAY_NAME,
|
| 1521 |
+
"agent_version": AGENT_VERSION,
|
| 1522 |
+
"status": "running",
|
| 1523 |
+
"llm": status,
|
| 1524 |
+
}
|
| 1525 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1526 |
|
| 1527 |
+
@app.get("/ready")
|
| 1528 |
+
def readiness_check():
|
| 1529 |
+
status = model_service.status()
|
| 1530 |
+
return {
|
| 1531 |
+
"ready": True,
|
| 1532 |
+
"analysis_ready": True,
|
| 1533 |
+
"agent_id": AGENT_ID,
|
| 1534 |
+
"llm_enabled": status["enabled"],
|
| 1535 |
+
"llm_loaded": status["loaded"],
|
| 1536 |
+
"llm_state": status["state"],
|
| 1537 |
+
"configured_model": status["configured_model"],
|
| 1538 |
+
"active_model": status["active_model"],
|
| 1539 |
+
"deterministic_fallback": True,
|
| 1540 |
+
}
|
| 1541 |
|
|
|
|
| 1542 |
|
| 1543 |
+
@app.post("/suggest", response_model=RepairResponse)
|
| 1544 |
+
def suggest_repair(request: RepairRequest):
|
| 1545 |
+
findings = collect_locked_findings(request)
|
|
|
|
|
|
|
|
|
|
| 1546 |
|
| 1547 |
+
if not findings:
|
| 1548 |
+
return RepairResponse.model_validate(
|
| 1549 |
+
build_fallback_plan(request, findings)
|
| 1550 |
+
)
|
| 1551 |
+
|
| 1552 |
+
if not should_use_llm(findings):
|
| 1553 |
+
return RepairResponse.model_validate(
|
| 1554 |
+
build_fallback_plan(
|
| 1555 |
+
request,
|
| 1556 |
+
findings,
|
| 1557 |
+
mode="deterministic-validated",
|
| 1558 |
+
)
|
| 1559 |
+
)
|
| 1560 |
|
| 1561 |
+
ai_findings, overflow_findings = select_ai_findings(findings)
|
| 1562 |
+
|
| 1563 |
+
try:
|
| 1564 |
+
messages = build_messages(request, ai_findings)
|
| 1565 |
+
text = model_service.generate(messages)
|
| 1566 |
+
plan = parse_model_output(text)
|
| 1567 |
+
plan = validate_plan(plan, request, ai_findings)
|
| 1568 |
+
result = merge_model_plan(
|
| 1569 |
+
plan,
|
| 1570 |
+
request,
|
| 1571 |
+
ai_findings,
|
| 1572 |
+
overflow_findings,
|
| 1573 |
+
)
|
| 1574 |
except Exception as error:
|
| 1575 |
+
result = build_fallback_plan(
|
| 1576 |
+
request,
|
| 1577 |
+
findings,
|
| 1578 |
+
mode="deterministic-fallback",
|
| 1579 |
+
llm_error=repr(error),
|
| 1580 |
+
)
|
| 1581 |
+
|
| 1582 |
+
return RepairResponse.model_validate(result)
|
requirements.txt
CHANGED
|
@@ -1,6 +1,5 @@
|
|
| 1 |
-
fastapi
|
| 2 |
-
uvicorn
|
| 3 |
-
pydantic
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
sentencepiece
|
|
|
|
| 1 |
+
fastapi==0.141.1
|
| 2 |
+
uvicorn==0.52.1
|
| 3 |
+
pydantic==2.13.4
|
| 4 |
+
huggingface_hub==1.27.0
|
| 5 |
+
llama-cpp-python==0.3.34
|
|
|