hashan-77 commited on
Commit
39839d0
·
verified ·
1 Parent(s): 8e8c864

Deploy repair agent from GitHub Actions

Browse files
Files changed (3) hide show
  1. Dockerfile +42 -2
  2. app.py +1498 -315
  3. requirements.txt +5 -6
Dockerfile CHANGED
@@ -1,12 +1,52 @@
1
  FROM python:3.11-slim
2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3
  WORKDIR /app
4
 
 
 
 
 
5
  COPY requirements.txt .
6
- RUN pip install --no-cache-dir -r requirements.txt
 
 
 
 
7
 
8
  COPY app.py .
9
 
 
 
 
 
 
10
  EXPOSE 7860
11
 
12
- CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "7860"]
 
 
 
1
  FROM python:3.11-slim
2
 
3
+ ENV PYTHONDONTWRITEBYTECODE=1 \
4
+ PYTHONUNBUFFERED=1 \
5
+ HF_HOME=/opt/huggingface \
6
+ MODEL_CACHE_DIR=/opt/huggingface/hub \
7
+ HF_MODEL_REPO=ibm-granite/granite-4.0-1b-GGUF \
8
+ HF_MODEL_FILE=granite-4.0-1b-Q4_K_M.gguf \
9
+ HF_MODEL_REVISION=b27c2fe3f211b7f44e80fa620177aea371099aaa \
10
+ MODEL_QUANTIZATION=Q4_K_M \
11
+ HF_MODEL=ibm-granite/granite-4.0-1b-GGUF:Q4_K_M \
12
+ HF_LOCAL_FILES_ONLY=true \
13
+ LLM_ENABLED=true \
14
+ MODEL_CONTEXT_TOKENS=4096 \
15
+ MODEL_BATCH_TOKENS=512 \
16
+ MODEL_MAX_INPUT_TOKENS=2800 \
17
+ MODEL_MAX_NEW_TOKENS=256 \
18
+ MODEL_MAX_GENERATION_SECONDS=90 \
19
+ MODEL_THREADS=2 \
20
+ MODEL_THREADS_BATCH=2 \
21
+ MODEL_TEMPERATURE=0.1 \
22
+ MODEL_TOP_P=0.9 \
23
+ MODEL_PROMPT_CACHE_MB=256 \
24
+ MODEL_SEED=17 \
25
+ MODEL_USE_MMAP=true \
26
+ OMP_NUM_THREADS=2
27
+
28
  WORKDIR /app
29
 
30
+ RUN apt-get update \
31
+ && apt-get install -y --no-install-recommends ca-certificates libgomp1 \
32
+ && rm -rf /var/lib/apt/lists/*
33
+
34
  COPY requirements.txt .
35
+ RUN python -m pip install --upgrade pip \
36
+ && python -m pip install --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cpu -r requirements.txt
37
+
38
+ RUN mkdir -p /opt/huggingface \
39
+ && python -c "import os; from huggingface_hub import hf_hub_download; hf_hub_download(repo_id=os.environ['HF_MODEL_REPO'], filename=os.environ['HF_MODEL_FILE'], revision=os.environ['HF_MODEL_REVISION'], cache_dir=os.environ['MODEL_CACHE_DIR'])"
40
 
41
  COPY app.py .
42
 
43
+ RUN useradd --create-home --uid 1000 appuser \
44
+ && chown -R appuser:appuser /app /opt/huggingface
45
+
46
+ USER appuser
47
+
48
  EXPOSE 7860
49
 
50
+ HEALTHCHECK --interval=30s --timeout=5s --start-period=60s --retries=3 CMD python -c "import urllib.request; urllib.request.urlopen('http://127.0.0.1:7860/ready', timeout=3)"
51
+
52
+ CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "7860", "--workers", "1", "--timeout-keep-alive", "30"]
app.py CHANGED
@@ -1,399 +1,1582 @@
1
- from fastapi import FastAPI
2
- from pydantic import BaseModel
3
- from transformers import AutoTokenizer, AutoModelForSeq2SeqLM
4
  import os
5
  import re
 
 
 
6
 
7
- HF_MODEL = os.getenv("HF_MODEL", "google/flan-t5-xl")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8
 
9
- tokenizer = None
10
- model = None
11
 
12
- app = FastAPI(title="Stitch QA Repair Agent")
 
 
 
 
 
 
13
 
14
 
15
  class RepairRequest(BaseModel):
16
- project_type: str
17
- command: str
 
 
18
  success: bool
19
- exit_code: int | None
20
- stdout: str
21
- stderr: str
22
- root_cause: str | None = None
23
- failure_type: str | None = None
24
- help_message: str | None = None
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
25
 
26
 
27
- @app.get("/")
28
- def health_check():
29
- return {
30
- "service": "stitch-qa-repair-agent",
31
- "status": "running",
32
- "llm_enabled": True,
33
- "llm_mode": "local-transformers",
34
- "model": HF_MODEL
35
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
36
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
37
 
38
- def load_model():
39
- global tokenizer, model
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
40
 
41
- if tokenizer is None or model is None:
42
- tokenizer = AutoTokenizer.from_pretrained(HF_MODEL)
43
- model = AutoModelForSeq2SeqLM.from_pretrained(HF_MODEL)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
44
 
45
- return tokenizer, model
46
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
47
 
48
- def extract_test_result(logs: str):
49
- match = re.search(
50
- r"Tests run:\s*(\d+),\s*Failures:\s*(\d+),\s*Errors:\s*(\d+),\s*Skipped:\s*(\d+)",
51
- logs,
52
- re.IGNORECASE
53
- )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
54
 
55
- if not match:
56
- return {
57
- "tests_run": None,
58
- "failures": None,
59
- "errors": None,
60
- "skipped": None,
61
- "summary": "Test result could not be extracted from logs."
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
62
  }
 
 
63
 
64
- tests_run, failures, errors, skipped = match.groups()
65
 
66
- return {
67
- "tests_run": int(tests_run),
68
- "failures": int(failures),
69
- "errors": int(errors),
70
- "skipped": int(skipped),
71
- "summary": (
72
- f"Tests run: {tests_run}, "
73
- f"Failures: {failures}, "
74
- f"Errors: {errors}, "
75
- f"Skipped: {skipped}"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
76
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
77
  }
78
 
79
 
80
- def get_environment_repair(request: RepairRequest):
81
- if request.failure_type == "MAVEN_NOT_AVAILABLE":
82
- return {
83
- "repair_type": "environment-maven-missing",
84
- "risk_level": "LOW",
85
- "summary": (
86
- f"The {request.project_type} project could not be verified because Maven is not installed "
87
- "or not available in PATH. This is an environment setup issue, not a confirmed project code failure. "
88
- "No source code repair should be applied until Maven execution is working."
89
- ),
90
- "suggestions": [
91
- "Install Apache Maven and add the Maven bin directory to the system PATH.",
92
- "Alternatively, add Maven Wrapper files to the project so it can run with mvnw.cmd on Windows.",
93
- "After Maven is available, rerun Stitch QA to verify the actual project build and tests."
94
- ],
95
- "next_action": "Fix the Maven environment first, then rerun Stitch QA."
96
- }
 
 
 
 
 
 
 
 
 
 
97
 
98
- if request.failure_type == "MAVEN_WRAPPER_NOT_AVAILABLE":
99
- return {
100
- "repair_type": "environment-wrapper-missing",
101
- "risk_level": "LOW",
102
- "summary": (
103
- f"The {request.project_type} project could not be verified because the Maven Wrapper command "
104
- "was not available. This is an execution setup issue, not a confirmed application code failure."
105
- ),
106
- "suggestions": [
107
- "Check whether mvnw.cmd exists in the project root.",
108
- "If Maven Wrapper is missing, add Maven Wrapper files or install Maven globally.",
109
- "Rerun Stitch QA after the build command can execute."
110
- ],
111
- "next_action": "Fix the Maven Wrapper or Maven installation before changing application code."
112
- }
113
 
114
- if request.failure_type == "COMMAND_TIMEOUT":
115
- return {
116
- "repair_type": "environment-timeout",
117
- "risk_level": "MEDIUM",
118
- "summary": (
119
- f"The {request.project_type} project command did not finish within the allowed timeout. "
120
- "This may be a long-running build, dependency download, or stuck process."
121
- ),
122
- "suggestions": [
123
- "Rerun the command manually to check whether it is slow or stuck.",
124
- "Increase the execution timeout if the build normally takes longer.",
125
- "Check dependency downloads and Maven repository access."
126
- ],
127
- "next_action": "Investigate command runtime before applying code changes."
128
- }
129
 
130
- return None
 
 
131
 
 
 
132
 
133
- def rule_based_repair(request: RepairRequest):
134
- environment_repair = get_environment_repair(request)
 
135
 
136
- if environment_repair:
137
- return {
138
- "agent": "repair-agent",
139
- "mode": "rule-based",
140
- "risk_level": environment_repair["risk_level"],
141
- "auto_apply": False,
142
- "summary": environment_repair["summary"],
143
- "suggestions": environment_repair["suggestions"],
144
- "next_action": environment_repair["next_action"]
145
- }
146
 
147
- combined_logs = f"{request.stdout}\n{request.stderr}".lower()
 
 
 
 
148
 
149
- suggestions = []
150
- risk_level = "LOW" if request.success else "HIGH"
151
 
152
- if request.success:
153
- suggestions.append(
154
- "No blocking fix is required because the project build and tests passed."
155
- )
156
 
157
- if "mockito" in combined_logs and "dynamic loading of agents" in combined_logs:
158
- suggestions.append(
159
- "Configure Mockito as a Java agent in the Maven test setup to improve future JDK compatibility."
160
- )
161
 
162
- if "compilation failure" in combined_logs:
163
- suggestions.append(
164
- "Review the Java compiler error, identify the affected source file, and apply a targeted syntax or dependency fix."
165
- )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
166
 
167
- if "tests run" in combined_logs and ("failures: 1" in combined_logs or "errors: 1" in combined_logs):
168
- suggestions.append(
169
- "Review the failing test method, compare expected versus actual behavior, and fix the related implementation or assertion."
170
- )
171
 
172
- if not suggestions:
173
- suggestions.append(
174
- "No specific repair suggestion could be generated from the current logs."
 
 
 
 
 
 
175
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
176
 
177
  return {
 
 
 
178
  "agent": "repair-agent",
179
- "mode": "rule-based",
180
- "risk_level": risk_level,
 
 
 
 
181
  "auto_apply": False,
182
- "summary": "Repair suggestions generated successfully.",
183
- "suggestions": suggestions,
184
- "next_action": "Review suggestions manually before applying any code changes."
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
185
  }
186
 
187
 
188
- def extract_repair_facts(request: RepairRequest, fallback_result):
189
- combined_logs = f"{request.stdout}\n{request.stderr}".lower()
190
- test_result = extract_test_result(f"{request.stdout}\n{request.stderr}")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
191
 
192
- environment_repair = get_environment_repair(request)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
193
 
194
- if environment_repair:
195
- return {
196
- "project_type": request.project_type,
197
  "command": request.command,
198
- "exit_code": request.exit_code,
199
- "build_state": "not verified",
200
  "success": request.success,
201
- "root_cause": request.root_cause or request.help_message or "Environment setup issue detected.",
202
- "warning_category": "none",
203
- "detected_warning": "No major warning detected.",
204
- "repair_type": environment_repair["repair_type"],
205
- "test_result": test_result["summary"],
206
- "suggestions": fallback_result["suggestions"],
207
- "risk_level": fallback_result["risk_level"],
208
  "failure_type": request.failure_type,
209
- "help_message": request.help_message,
210
- "environment_summary": environment_repair["summary"]
211
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
212
 
213
- build_state = "passed" if request.success else "failed"
 
 
 
 
 
 
 
 
 
 
 
 
 
214
 
215
- detected_warning = "No major warning detected."
216
- warning_category = "none"
217
 
218
- if "mockito" in combined_logs and "dynamic loading of agents" in combined_logs:
219
- warning_category = "mockito-dynamic-agent"
220
- detected_warning = (
221
- "Mockito dynamic Java agent loading warning detected. "
222
- "This is not a current failure, but it may affect compatibility with future JDK versions."
223
- )
224
 
225
- repair_type = "none"
 
 
 
 
 
 
 
 
 
226
 
227
- if request.success and warning_category == "none":
228
- repair_type = "no-repair-needed"
229
 
230
- if request.success and warning_category == "mockito-dynamic-agent":
231
- repair_type = "configuration-warning"
 
 
 
232
 
233
- if not request.success:
234
- repair_type = "failure-repair-required"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
235
 
236
- return {
237
- "project_type": request.project_type,
238
- "command": request.command,
239
- "exit_code": request.exit_code,
240
- "build_state": build_state,
241
- "success": request.success,
242
- "root_cause": request.root_cause or "No root cause provided.",
243
- "warning_category": warning_category,
244
- "detected_warning": detected_warning,
245
- "repair_type": repair_type,
246
- "test_result": test_result["summary"],
247
- "suggestions": fallback_result["suggestions"],
248
- "risk_level": fallback_result["risk_level"],
249
- "failure_type": request.failure_type,
250
- "help_message": request.help_message,
251
- "environment_summary": None
252
  }
 
 
 
 
 
253
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
254
 
255
- def build_clean_summary(facts):
256
- if facts["repair_type"] in {
257
- "environment-maven-missing",
258
- "environment-wrapper-missing",
259
- "environment-timeout"
260
- }:
261
- return facts["environment_summary"]
262
 
263
- if facts["repair_type"] == "no-repair-needed":
264
- return (
265
- f"The {facts['project_type']} project passed the current QA execution using "
266
- f"`{facts['command']}`. {facts['test_result']}. No blocking repair is required. "
267
- "The project can be considered stable for this basic test run."
268
- )
269
 
270
- if facts["repair_type"] == "configuration-warning":
271
- return (
272
- f"The {facts['project_type']} project passed the current QA execution using "
273
- f"`{facts['command']}`. {facts['test_result']}. No blocking code repair is required. "
274
- f"However, {facts['detected_warning']} Recommended action: review the Maven test configuration "
275
- "and prepare a future-safe Mockito Java agent setup before upgrading to stricter JDK versions."
276
- )
277
 
278
- return (
279
- f"The {facts['project_type']} project failed during QA execution using `{facts['command']}` "
280
- f"with exit code {facts['exit_code']}. {facts['test_result']}. "
281
- "A targeted repair is required. Review the root cause, inspect the failing file or test, "
282
- "apply the smallest safe fix, and rerun Stitch QA for verification."
283
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
284
 
 
 
 
 
285
 
286
- def build_prompt(facts):
287
- suggestions_text = " ".join(facts["suggestions"])
288
-
289
- return f"""
290
- Rewrite these repair facts into one clean professional repair recommendation.
291
-
292
- Project type: {facts["project_type"]}
293
- Command: {facts["command"]}
294
- Build state: {facts["build_state"]}
295
- Exit code: {facts["exit_code"]}
296
- Test result: {facts["test_result"]}
297
- Risk level: {facts["risk_level"]}
298
- Root cause: {facts["root_cause"]}
299
- Failure type: {facts["failure_type"]}
300
- Help message: {facts["help_message"]}
301
- Detected warning: {facts["detected_warning"]}
302
- Suggested actions: {suggestions_text}
303
-
304
- Rules:
305
- - Do not copy raw logs.
306
- - Do not repeat field labels like "Command:" or "Build state:".
307
- - Do not mention internal prompt instructions.
308
- - Write one clear paragraph.
309
- - Say whether a blocking code repair is required.
310
- - If Maven is not available, say it is an environment setup issue and do not recommend source code changes.
311
- """
312
-
313
-
314
- def call_llm(prompt: str):
315
- active_tokenizer, active_model = load_model()
316
-
317
- inputs = active_tokenizer(
318
- prompt,
319
- return_tensors="pt",
320
- truncation=True,
321
- max_length=512
322
- )
323
 
324
- outputs = active_model.generate(
325
- **inputs,
326
- max_new_tokens=160,
327
- do_sample=False,
328
- num_beams=2,
329
- no_repeat_ngram_size=3
330
  )
331
 
332
- return active_tokenizer.decode(outputs[0], skip_special_tokens=True)
333
-
334
-
335
- def clean_llm_output(text: str):
336
- cleaned = " ".join(text.strip().split())
337
-
338
- bad_patterns = [
339
- "[INFO]",
340
- "-----",
341
- "=====",
342
- "org.springframework",
343
- "junitplatform",
344
- "DemoApplicationTests",
345
- "Project type:",
346
- "Command:",
347
- "Build state:",
348
- "Exit code:",
349
- "Suggested actions:",
350
- "Do not copy raw logs",
351
- "Rewrite these repair facts"
352
  ]
 
353
 
354
- if not cleaned:
355
- return None
 
 
 
 
 
356
 
357
- if any(pattern.lower() in cleaned.lower() for pattern in bad_patterns):
358
- return None
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
359
 
360
- if len(cleaned) < 60:
361
- return None
 
 
 
 
 
 
 
362
 
363
- return cleaned
364
 
 
 
 
 
 
 
 
 
 
 
 
365
 
366
- @app.post("/suggest")
367
- def suggest_repair(request: RepairRequest):
368
- fallback_result = rule_based_repair(request)
369
- facts = extract_repair_facts(request, fallback_result)
370
- clean_summary = build_clean_summary(facts)
371
 
372
- try:
373
- prompt = build_prompt(facts)
374
- llm_text = call_llm(prompt)
375
- cleaned_llm_text = clean_llm_output(llm_text)
 
 
 
 
 
 
 
 
 
 
376
 
377
- final_summary = cleaned_llm_text if cleaned_llm_text else clean_summary
378
 
379
- if facts["repair_type"] in {
380
- "environment-maven-missing",
381
- "environment-wrapper-missing",
382
- "environment-timeout"
383
- }:
384
- final_summary = clean_summary
385
 
386
- return {
387
- "agent": "repair-agent",
388
- "mode": "llm",
389
- "risk_level": fallback_result["risk_level"],
390
- "auto_apply": False,
391
- "summary": final_summary,
392
- "suggestions": fallback_result["suggestions"],
393
- "next_action": fallback_result["next_action"]
394
- }
 
 
 
 
395
 
 
 
 
 
 
 
 
 
 
 
 
 
 
396
  except Exception as error:
397
- fallback_result["llm_error"] = repr(error)
398
- fallback_result["summary"] = clean_summary
399
- return fallback_result
 
 
 
 
 
 
1
+ import gc
2
+ import json
 
3
  import os
4
  import re
5
+ import threading
6
+ import time
7
+ from typing import Any, Literal
8
 
9
+ from fastapi import FastAPI
10
+ from pydantic import BaseModel, ConfigDict, Field, ValidationError
11
+
12
+
13
+ AGENT_ID = "defect-resolution-analyst"
14
+ DISPLAY_NAME = "Defect Resolution Intelligence Analyst"
15
+ AGENT_VERSION = "2.0"
16
+ MAX_AI_FINDINGS = 4
17
+ MAX_AI_CONTRACTS = 4
18
+
19
+ PRIORITY_ORDER = {
20
+ "NONE": 0,
21
+ "P3": 1,
22
+ "P2": 2,
23
+ "P1": 3,
24
+ }
25
+
26
+ RISK_ORDER = {
27
+ "UNKNOWN": -1,
28
+ "NONE": 0,
29
+ "INFO": 1,
30
+ "LOW": 2,
31
+ "MEDIUM": 3,
32
+ "HIGH": 4,
33
+ "CRITICAL": 5,
34
+ }
35
+
36
+ MODEL_RESPONSE_SCHEMA = {
37
+ "type": "object",
38
+ "properties": {
39
+ "p": {"type": "string", "enum": ["P1", "P2", "P3"]},
40
+ "c": {"type": "string", "enum": ["HIGH", "MEDIUM", "LOW"]},
41
+ "k": {"type": "boolean"},
42
+ "kr": {"type": "string", "maxLength": 180},
43
+ "x": {
44
+ "type": "array",
45
+ "minItems": 1,
46
+ "maxItems": MAX_AI_CONTRACTS,
47
+ "items": {
48
+ "type": "object",
49
+ "properties": {
50
+ "f": {
51
+ "type": "array",
52
+ "minItems": 1,
53
+ "maxItems": MAX_AI_FINDINGS,
54
+ "uniqueItems": True,
55
+ "items": {
56
+ "type": "string",
57
+ "minLength": 1,
58
+ "maxLength": 80,
59
+ },
60
+ },
61
+ "p": {"type": "string", "enum": ["P1", "P2", "P3"]},
62
+ "w": {"type": "string", "minLength": 8, "maxLength": 160},
63
+ "o": {"type": "string", "minLength": 8, "maxLength": 180},
64
+ "s": {"type": "string", "minLength": 8, "maxLength": 220},
65
+ "b": {"type": "string", "minLength": 6, "maxLength": 160},
66
+ "q": {"type": "string", "minLength": 6, "maxLength": 160},
67
+ "r": {"type": "string", "enum": ["L", "M", "H"]},
68
+ "v": {"type": "string", "minLength": 8, "maxLength": 220},
69
+ },
70
+ "required": ["f", "p", "w", "o", "s", "b", "q", "r", "v"],
71
+ "additionalProperties": False,
72
+ },
73
+ },
74
+ },
75
+ "required": ["p", "c", "k", "kr", "x"],
76
+ "additionalProperties": False,
77
+ }
78
+
79
+
80
+ def env_bool(name, default):
81
+ value = os.getenv(name)
82
+ if value is None:
83
+ return bool(default)
84
+ return value.strip().lower() in {"1", "true", "yes", "on"}
85
+
86
+
87
+ def clean_text(value, limit=800):
88
+ text = " ".join(str(value or "").split()).strip()
89
+ if len(text) <= limit:
90
+ return text
91
+ return text[: limit - 3].rstrip() + "..."
92
+
93
+
94
+ def normalize_list(value):
95
+ if isinstance(value, list):
96
+ return value
97
+ if value is None:
98
+ return []
99
+ return [value]
100
+
101
+
102
+ def normalize_risk(value):
103
+ risk = str(value or "UNKNOWN").upper()
104
+ return risk if risk in RISK_ORDER else "UNKNOWN"
105
+
106
+
107
+ def highest_risk(*values):
108
+ normalized = [normalize_risk(value) for value in values]
109
+ return max(
110
+ normalized,
111
+ key=lambda value: RISK_ORDER.get(value, -1),
112
+ default="UNKNOWN",
113
+ )
114
 
 
 
115
 
116
+ def severity_priority(severity):
117
+ severity = normalize_risk(severity)
118
+ if severity in {"CRITICAL", "HIGH"}:
119
+ return "P1"
120
+ if severity == "MEDIUM":
121
+ return "P2"
122
+ return "P3"
123
 
124
 
125
  class RepairRequest(BaseModel):
126
+ model_config = ConfigDict(extra="ignore")
127
+
128
+ project_type: str = Field(min_length=1, max_length=200)
129
+ command: str = Field(default="", max_length=2000)
130
  success: bool
131
+ exit_code: int | None = None
132
+ stdout: str = Field(default="", max_length=16000)
133
+ stderr: str = Field(default="", max_length=16000)
134
+ failure_type: str | None = Field(default=None, max_length=200)
135
+ help_message: str | None = Field(default=None, max_length=4000)
136
+ runtime_evidence: dict[str, Any] | None = None
137
+ runtime_analysis: dict[str, Any] | None = None
138
+ source_review: dict[str, Any] | None = None
139
+
140
+
141
+ class ModelRepairContract(BaseModel):
142
+ model_config = ConfigDict(extra="forbid", populate_by_name=True)
143
+
144
+ finding_refs: list[str] = Field(alias="f", min_length=1, max_length=12)
145
+ priority: Literal["P1", "P2", "P3"] = Field(alias="p")
146
+ priority_reason: str = Field(alias="w", min_length=8, max_length=160)
147
+ repair_objective: str = Field(alias="o", min_length=8, max_length=180)
148
+ repair_strategy: str = Field(alias="s", min_length=8, max_length=220)
149
+ change_boundary: str = Field(alias="b", min_length=6, max_length=160)
150
+ protected_behavior: str = Field(alias="q", min_length=6, max_length=160)
151
+ side_effect_risk_code: Literal["L", "M", "H"] = Field(alias="r")
152
+ verification: str = Field(alias="v", min_length=8, max_length=220)
153
+
154
+ @property
155
+ def side_effect_risk(self):
156
+ return {"L": "LOW", "M": "MEDIUM", "H": "HIGH"}[
157
+ self.side_effect_risk_code
158
+ ]
159
+
160
+
161
+ class ModelRepairPlan(BaseModel):
162
+ model_config = ConfigDict(extra="forbid", populate_by_name=True)
163
+
164
+ overall_priority: Literal["P1", "P2", "P3"] = Field(alias="p")
165
+ confidence: Literal["HIGH", "MEDIUM", "LOW"] = Field(alias="c")
166
+ current_knowledge_required: bool = Field(alias="k")
167
+ current_knowledge_reason: str = Field(alias="kr", default="", max_length=180)
168
+ contracts: list[ModelRepairContract] = Field(
169
+ alias="x",
170
+ min_length=1,
171
+ max_length=MAX_AI_CONTRACTS,
172
+ )
173
 
174
 
175
+ class StitchRepairContract(BaseModel):
176
+ model_config = ConfigDict(extra="ignore")
177
+
178
+ contract_id: str
179
+ finding_refs: list[str]
180
+ title: str
181
+ priority: Literal["P1", "P2", "P3"]
182
+ priority_reason: str
183
+ repair_objective: str
184
+ repair_strategy: str
185
+ change_boundary: str
186
+ protected_behavior: str
187
+ side_effect_risk: Literal["LOW", "MEDIUM", "HIGH"]
188
+ verification: str
189
+ done_condition: str
190
+ status: Literal["PENDING_VERIFICATION"] = "PENDING_VERIFICATION"
191
+
192
+
193
+ class RepairResponse(BaseModel):
194
+ model_config = ConfigDict(extra="ignore")
195
+
196
+ agent_id: str
197
+ display_name: str
198
+ agent_version: str
199
+ agent: str
200
+ mode: str
201
+ model: str | None = None
202
+ status: str
203
+ overall_priority: str
204
+ confidence: str
205
+ repair_risk_level: str
206
+ auto_apply: bool = False
207
+ summary: str
208
+ stitch_repair_contracts: list[StitchRepairContract] = Field(default_factory=list)
209
+ current_knowledge_required: bool = False
210
+ current_knowledge_reason: str | None = None
211
+ next_action: str | None = None
212
+ suggestions: list[str] = Field(default_factory=list)
213
+ warnings: list[str] = Field(default_factory=list)
214
+ limitations: list[str] = Field(default_factory=list)
215
+ llm_metrics: dict[str, Any] | None = None
216
+ llm_error: str | None = None
217
+
218
+
219
+ class ModelService:
220
+ def __init__(self):
221
+ self.model_repo = os.getenv(
222
+ "HF_MODEL_REPO",
223
+ "ibm-granite/granite-4.0-1b-GGUF",
224
+ ).strip()
225
+ self.model_file = os.getenv(
226
+ "HF_MODEL_FILE",
227
+ "granite-4.0-1b-Q4_K_M.gguf",
228
+ ).strip()
229
+ self.model_revision = os.getenv(
230
+ "HF_MODEL_REVISION",
231
+ "b27c2fe3f211b7f44e80fa620177aea371099aaa",
232
+ ).strip()
233
+ self.quantization = os.getenv(
234
+ "MODEL_QUANTIZATION",
235
+ "Q4_K_M",
236
+ ).strip()
237
+ self.primary_model = os.getenv(
238
+ "HF_MODEL",
239
+ f"{self.model_repo}:{self.quantization}",
240
+ ).strip()
241
+ self.cache_dir = os.getenv(
242
+ "MODEL_CACHE_DIR",
243
+ "/opt/huggingface/hub",
244
+ ).strip()
245
+ self.enabled = env_bool("LLM_ENABLED", True)
246
+ self.local_files_only = env_bool("HF_LOCAL_FILES_ONLY", False)
247
+ self.context_tokens = max(1024, int(os.getenv("MODEL_CONTEXT_TOKENS", "4096")))
248
+ self.batch_tokens = max(
249
+ 64,
250
+ min(
251
+ self.context_tokens,
252
+ int(os.getenv("MODEL_BATCH_TOKENS", "512")),
253
+ ),
254
+ )
255
+ self.max_input_tokens = max(
256
+ 512,
257
+ int(os.getenv("MODEL_MAX_INPUT_TOKENS", "2800")),
258
+ )
259
+ self.max_new_tokens = max(
260
+ 96,
261
+ int(os.getenv("MODEL_MAX_NEW_TOKENS", "256")),
262
+ )
263
+ self.max_generation_seconds = max(
264
+ 15.0,
265
+ float(os.getenv("MODEL_MAX_GENERATION_SECONDS", "90")),
266
+ )
267
+ self.threads = max(1, int(os.getenv("MODEL_THREADS", "2")))
268
+ self.threads_batch = max(
269
+ 1,
270
+ int(os.getenv("MODEL_THREADS_BATCH", str(self.threads))),
271
+ )
272
+ self.temperature = max(
273
+ 0.0,
274
+ float(os.getenv("MODEL_TEMPERATURE", "0.1")),
275
+ )
276
+ self.top_p = min(
277
+ 1.0,
278
+ max(0.01, float(os.getenv("MODEL_TOP_P", "0.9"))),
279
+ )
280
+ self.seed = int(os.getenv("MODEL_SEED", "17"))
281
+ self.use_mmap = env_bool("MODEL_USE_MMAP", True)
282
+ self.prompt_cache_mb = max(
283
+ 0,
284
+ int(os.getenv("MODEL_PROMPT_CACHE_MB", "256")),
285
+ )
286
+ self.model = None
287
+ self.model_path = None
288
+ self.model_name = None
289
+ self.load_error = None
290
+ self.load_seconds = None
291
+ self.last_generation = None
292
+ self.load_lock = threading.Lock()
293
+ self.generation_lock = threading.Lock()
294
+
295
+ def _download_model(self):
296
+ try:
297
+ from huggingface_hub import hf_hub_download
298
+ except ImportError as error:
299
+ raise RuntimeError(
300
+ "huggingface_hub is required for the configured Granite GGUF backend."
301
+ ) from error
302
+
303
+ return hf_hub_download(
304
+ repo_id=self.model_repo,
305
+ filename=self.model_file,
306
+ revision=self.model_revision,
307
+ cache_dir=self.cache_dir,
308
+ local_files_only=self.local_files_only,
309
+ )
310
 
311
+ def _load_model(self):
312
+ try:
313
+ from llama_cpp import Llama, LlamaRAMCache
314
+ except ImportError as error:
315
+ raise RuntimeError(
316
+ "llama-cpp-python is required for the configured Granite GGUF backend."
317
+ ) from error
318
+
319
+ model_path = self._download_model()
320
+ model = Llama(
321
+ model_path=model_path,
322
+ n_ctx=self.context_tokens,
323
+ n_batch=self.batch_tokens,
324
+ n_threads=self.threads,
325
+ n_threads_batch=self.threads_batch,
326
+ n_gpu_layers=0,
327
+ seed=self.seed,
328
+ use_mmap=self.use_mmap,
329
+ use_mlock=False,
330
+ verbose=False,
331
+ )
332
 
333
+ if self.prompt_cache_mb > 0:
334
+ model.set_cache(
335
+ LlamaRAMCache(
336
+ capacity_bytes=self.prompt_cache_mb * 1024 * 1024
337
+ )
338
+ )
339
+
340
+ return model_path, model
341
+
342
+ def load(self):
343
+ if not self.enabled:
344
+ raise RuntimeError("LLM inference is disabled.")
345
+
346
+ if self.model is not None:
347
+ return self.model
348
+
349
+ with self.load_lock:
350
+ if self.model is not None:
351
+ return self.model
352
+
353
+ started = time.monotonic()
354
+ try:
355
+ model_path, model = self._load_model()
356
+ self.model_path = model_path
357
+ self.model = model
358
+ self.model_name = self.primary_model
359
+ self.load_error = None
360
+ self.load_seconds = round(time.monotonic() - started, 3)
361
+ return model
362
+ except Exception as error:
363
+ self.model = None
364
+ self.model_path = None
365
+ self.model_name = None
366
+ self.load_error = repr(error)
367
+ self.load_seconds = round(time.monotonic() - started, 3)
368
+ gc.collect()
369
+ raise RuntimeError(self.load_error) from error
370
+
371
+ def _count_message_tokens(self, model, messages):
372
+ serialized = json.dumps(
373
+ messages,
374
+ ensure_ascii=False,
375
+ separators=(",", ":"),
376
+ ).encode("utf-8")
377
+
378
+ try:
379
+ tokens = model.tokenize(serialized, add_bos=False, special=True)
380
+ except TypeError:
381
+ tokens = model.tokenize(serialized, add_bos=False)
382
+
383
+ return len(tokens)
384
+
385
+ def _count_text_tokens(self, model, text):
386
+ if not text:
387
+ return 0
388
+ value = str(text).encode("utf-8")
389
+ try:
390
+ tokens = model.tokenize(value, add_bos=False, special=True)
391
+ except TypeError:
392
+ tokens = model.tokenize(value, add_bos=False)
393
+ return len(tokens)
394
+
395
+ def _json_complete(self, text):
396
+ candidate = str(text or "").strip()
397
+ if not candidate.startswith("{") or not candidate.endswith("}"):
398
+ return False
399
+ try:
400
+ json.loads(candidate)
401
+ return True
402
+ except json.JSONDecodeError:
403
+ return False
404
+
405
+ def generate(self, messages):
406
+ self.last_generation = None
407
+ model = self.load()
408
+
409
+ with self.generation_lock:
410
+ input_tokens = self._count_message_tokens(model, messages)
411
+ if input_tokens > self.max_input_tokens:
412
+ raise RuntimeError(
413
+ f"Model input contains approximately {input_tokens} tokens, exceeding the configured limit of {self.max_input_tokens}."
414
+ )
415
+
416
+ started = time.monotonic()
417
+ parts = []
418
+ finish_reason = None
419
+ timed_out = False
420
+ first_content_seconds = None
421
+ stream = model.create_chat_completion(
422
+ messages=messages,
423
+ response_format={
424
+ "type": "json_object",
425
+ "schema": MODEL_RESPONSE_SCHEMA,
426
+ },
427
+ max_tokens=self.max_new_tokens,
428
+ temperature=self.temperature,
429
+ top_p=self.top_p,
430
+ seed=self.seed,
431
+ stream=True,
432
+ )
433
+
434
+ try:
435
+ for chunk in stream:
436
+ elapsed = time.monotonic() - started
437
+ choices = chunk.get("choices") or []
438
+ if choices:
439
+ choice = choices[0]
440
+ delta = choice.get("delta") or {}
441
+ content = delta.get("content")
442
+ if content:
443
+ if first_content_seconds is None:
444
+ first_content_seconds = elapsed
445
+ parts.append(str(content))
446
+ if self._json_complete("".join(parts)):
447
+ finish_reason = "json_complete"
448
+ break
449
+ if choice.get("finish_reason"):
450
+ finish_reason = str(choice.get("finish_reason"))
451
+
452
+ if elapsed >= self.max_generation_seconds and not finish_reason:
453
+ timed_out = True
454
+ break
455
+ finally:
456
+ close = getattr(stream, "close", None)
457
+ if callable(close):
458
+ close()
459
+
460
+ elapsed = time.monotonic() - started
461
+ text = "".join(parts).strip()
462
+ completion_tokens = self._count_text_tokens(model, text)
463
+ tokens_per_second = (
464
+ round(completion_tokens / elapsed, 3)
465
+ if elapsed > 0 and completion_tokens
466
+ else 0.0
467
+ )
468
+ self.last_generation = {
469
+ "backend": "llama.cpp",
470
+ "quantization": self.quantization,
471
+ "input_tokens_approx": input_tokens,
472
+ "completion_tokens": completion_tokens,
473
+ "elapsed_seconds": round(elapsed, 3),
474
+ "first_content_seconds": (
475
+ round(first_content_seconds, 3)
476
+ if first_content_seconds is not None
477
+ else None
478
+ ),
479
+ "tokens_per_second": tokens_per_second,
480
+ "finish_reason": finish_reason,
481
+ "timed_out": timed_out,
482
+ "max_generation_seconds": self.max_generation_seconds,
483
+ "max_new_tokens": self.max_new_tokens,
484
+ }
485
+
486
+ if timed_out:
487
+ raise RuntimeError(
488
+ "The configured Granite model exceeded the generation time limit "
489
+ f"(generated_tokens={completion_tokens}, elapsed_seconds={elapsed:.3f}, tokens_per_second={tokens_per_second:.3f})."
490
+ )
491
+
492
+ if not text:
493
+ raise RuntimeError("The configured Granite model returned an empty response.")
494
+
495
+ try:
496
+ json.loads(text)
497
+ except json.JSONDecodeError as error:
498
+ raise RuntimeError(
499
+ "The configured Granite JSON-constrained generation returned invalid JSON."
500
+ ) from error
501
+
502
+ if finish_reason == "length":
503
+ raise RuntimeError(
504
+ "The configured Granite model reached the output token limit before completing the repair plan."
505
+ )
506
+
507
+ return text
508
+
509
+ def status(self):
510
+ if not self.enabled:
511
+ state = "disabled"
512
+ elif self.model is not None:
513
+ state = "loaded"
514
+ elif self.load_error:
515
+ state = "load_failed"
516
+ else:
517
+ state = "not_loaded"
518
 
519
+ return {
520
+ "enabled": self.enabled,
521
+ "loaded": self.model is not None,
522
+ "state": state,
523
+ "backend": "llama.cpp",
524
+ "configured_model": self.primary_model,
525
+ "active_model": self.model_name,
526
+ "model_repo": self.model_repo,
527
+ "model_file": self.model_file,
528
+ "model_revision": self.model_revision,
529
+ "quantization": self.quantization,
530
+ "load_error": self.load_error,
531
+ "load_seconds": self.load_seconds,
532
+ "context_tokens": self.context_tokens,
533
+ "max_input_tokens": self.max_input_tokens,
534
+ "max_new_tokens": self.max_new_tokens,
535
+ "max_generation_seconds": self.max_generation_seconds,
536
+ "threads": self.threads,
537
+ "prompt_cache_mb": self.prompt_cache_mb,
538
+ "last_generation": self.last_generation,
539
+ }
540
 
 
541
 
542
+ model_service = ModelService()
543
+
544
+ app = FastAPI(
545
+ title="Stitch QA Defect Resolution Intelligence Analyst",
546
+ version=AGENT_VERSION,
547
+ )
548
+
549
+
550
+ def runtime_findings(request):
551
+ analysis = request.runtime_analysis or {}
552
+ result = []
553
+
554
+ for index, group in enumerate(
555
+ normalize_list(analysis.get("root_cause_groups")),
556
+ start=1,
557
+ ):
558
+ if not isinstance(group, dict):
559
+ continue
560
+
561
+ ref = clean_text(group.get("group_id"), 80) or f"RQI-{index:03d}"
562
+ evidence = []
563
+ locations = []
564
+
565
+ for item in normalize_list(group.get("evidence"))[:8]:
566
+ if not isinstance(item, dict):
567
+ continue
568
+ evidence.append(
569
+ {
570
+ key: item.get(key)
571
+ for key in (
572
+ "failure_id",
573
+ "test_name",
574
+ "expected",
575
+ "actual",
576
+ "exception_type",
577
+ "exception_message",
578
+ "application_file",
579
+ "application_line",
580
+ "test_file",
581
+ "test_line",
582
+ )
583
+ if item.get(key) not in {None, ""}
584
+ }
585
+ )
586
+ file_path = item.get("application_file") or item.get("test_file")
587
+ line = item.get("application_line") or item.get("test_line")
588
+ if file_path:
589
+ locations.append(
590
+ f"{file_path}:{line}" if line else str(file_path)
591
+ )
592
+
593
+ result.append(
594
+ {
595
+ "ref": ref,
596
+ "source": "runtime",
597
+ "severity": normalize_risk(
598
+ analysis.get("runtime_risk_level") or "HIGH"
599
+ ),
600
+ "title": clean_text(group.get("title"), 180)
601
+ or "Runtime failure group",
602
+ "category": clean_text(group.get("category"), 120),
603
+ "root_cause": clean_text(group.get("root_cause"), 700),
604
+ "impact": clean_text(group.get("runtime_impact"), 500),
605
+ "recommendation": clean_text(group.get("required_action"), 500),
606
+ "locations": list(dict.fromkeys(locations))[:8],
607
+ "evidence": evidence,
608
+ }
609
+ )
610
 
611
+ if result:
612
+ return result
613
+
614
+ runtime_evidence = request.runtime_evidence or {}
615
+ test_result = str(runtime_evidence.get("test_result") or "").upper()
616
+
617
+ for index, failure in enumerate(
618
+ normalize_list(runtime_evidence.get("failures")),
619
+ start=1,
620
+ ):
621
+ if not isinstance(failure, dict):
622
+ continue
623
+
624
+ ref = clean_text(failure.get("id"), 80) or f"RTE-{index:03d}"
625
+ application_file = failure.get("application_file")
626
+ application_line = failure.get("application_line")
627
+ test_file = failure.get("test_file")
628
+ test_line = failure.get("test_line")
629
+ locations = []
630
+
631
+ if application_file:
632
+ locations.append(
633
+ f"{application_file}:{application_line}"
634
+ if application_line
635
+ else str(application_file)
636
+ )
637
+ if test_file:
638
+ locations.append(
639
+ f"{test_file}:{test_line}"
640
+ if test_line
641
+ else str(test_file)
642
+ )
643
+
644
+ expected = clean_text(failure.get("expected"), 180)
645
+ actual = clean_text(
646
+ failure.get("actual") or failure.get("exception_type"),
647
+ 180,
648
+ )
649
+ exception_message = clean_text(failure.get("exception_message"), 300)
650
+ observed = exception_message or actual or "The tested path failed."
651
+ contract_text = (
652
+ f" Expected {expected}; observed {actual}."
653
+ if expected and actual
654
+ else ""
655
+ )
656
 
657
+ result.append(
658
+ {
659
+ "ref": ref,
660
+ "source": "runtime",
661
+ "severity": "HIGH" if test_result == "FAIL" else "MEDIUM",
662
+ "title": clean_text(
663
+ failure.get("test_name") or failure.get("exception_type"),
664
+ 180,
665
+ ) or "Validated runtime failure",
666
+ "category": "VALIDATED_RUNTIME_FAILURE",
667
+ "root_cause": clean_text(
668
+ f"Observed failure evidence: {observed}.{contract_text}",
669
+ 700,
670
+ ),
671
+ "impact": (
672
+ "A validated automated-test path did not complete with its expected behavior."
673
+ ),
674
+ "recommendation": (
675
+ "Resolve the evidence-backed behavior mismatch using the smallest safe change, then rerun the affected and regression tests."
676
+ ),
677
+ "locations": list(dict.fromkeys(locations)),
678
+ "evidence": [
679
+ {
680
+ key: failure.get(key)
681
+ for key in (
682
+ "id",
683
+ "test_name",
684
+ "expected",
685
+ "actual",
686
+ "exception_type",
687
+ "exception_message",
688
+ "application_file",
689
+ "application_line",
690
+ "test_file",
691
+ "test_line",
692
+ )
693
+ if failure.get(key) not in {None, ""}
694
+ }
695
+ ],
696
+ }
697
+ )
698
+
699
+ return result
700
+
701
+
702
+ def runtime_warning_findings(request):
703
+ analysis = request.runtime_analysis or {}
704
+ runtime_evidence = request.runtime_evidence or {}
705
+ warnings = []
706
+ seen = set()
707
+
708
+ for item in [
709
+ *normalize_list(analysis.get("warnings")),
710
+ *normalize_list(runtime_evidence.get("warnings")),
711
+ ]:
712
+ warning = clean_text(item, 500)
713
+ key = warning.lower()
714
+ if not warning or key in seen:
715
+ continue
716
+ seen.add(key)
717
+ warnings.append(warning)
718
+
719
+ release_gate = str(analysis.get("release_gate") or "").upper()
720
+ severity = "MEDIUM" if release_gate == "ALLOW_WITH_WARNINGS" else "LOW"
721
+
722
+ return [
723
+ {
724
+ "ref": f"RWI-{index:03d}",
725
+ "source": "runtime",
726
+ "severity": severity,
727
+ "title": "Runtime warning requiring follow-up",
728
+ "category": "RUNTIME_WARNING",
729
+ "root_cause": warning,
730
+ "impact": (
731
+ "The validated run completed, but the warning may affect future compatibility, reliability, or execution behavior if its underlying condition changes."
732
+ ),
733
+ "recommendation": (
734
+ "Review the warning-specific configuration or dependency behavior, verify current vendor guidance when needed, and rerun the relevant workflow after any targeted adjustment."
735
+ ),
736
+ "locations": [],
737
+ "evidence": [],
738
  }
739
+ for index, warning in enumerate(warnings, start=1)
740
+ ]
741
 
 
742
 
743
+ def source_findings(request):
744
+ source_review = request.source_review or {}
745
+ result = []
746
+
747
+ for index, finding in enumerate(
748
+ normalize_list(source_review.get("findings")),
749
+ start=1,
750
+ ):
751
+ if not isinstance(finding, dict):
752
+ continue
753
+
754
+ ref = clean_text(finding.get("id"), 80) or f"SRC-{index:03d}"
755
+ file_path = finding.get("file_path")
756
+ line = finding.get("line")
757
+ location = None
758
+ if file_path:
759
+ location = f"{file_path}:{line}" if line else str(file_path)
760
+
761
+ result.append(
762
+ {
763
+ "ref": ref,
764
+ "source": "source",
765
+ "severity": normalize_risk(finding.get("severity")),
766
+ "title": clean_text(finding.get("title"), 180)
767
+ or "Source-code finding",
768
+ "category": clean_text(finding.get("category"), 120),
769
+ "root_cause": clean_text(
770
+ finding.get("evidence") or finding.get("description"),
771
+ 700,
772
+ ),
773
+ "impact": clean_text(finding.get("impact"), 500),
774
+ "recommendation": clean_text(
775
+ finding.get("recommendation"),
776
+ 500,
777
+ ),
778
+ "locations": [location] if location else [],
779
+ "evidence": [],
780
+ }
781
  )
782
+
783
+ return result
784
+
785
+
786
+ def execution_finding(request):
787
+ if request.success:
788
+ return None
789
+
790
+ analysis = request.runtime_analysis or {}
791
+ if normalize_list(analysis.get("root_cause_groups")):
792
+ return None
793
+
794
+ runtime_evidence = request.runtime_evidence or {}
795
+ if normalize_list(runtime_evidence.get("failures")):
796
+ return None
797
+
798
+ failure_type = clean_text(request.failure_type, 120)
799
+ help_message = clean_text(request.help_message, 500)
800
+ discovery_only_failures = {
801
+ "PYTHON_TESTS_NOT_FOUND",
802
+ }
803
+
804
+ if failure_type.upper() in discovery_only_failures:
805
+ return None
806
+
807
+ if "no test files" in help_message.lower() or "no pytest-compatible test" in help_message.lower():
808
+ return None
809
+
810
+ if not failure_type and not help_message:
811
+ return None
812
+
813
+ return {
814
+ "ref": "EXEC-001",
815
+ "source": "execution",
816
+ "severity": "MEDIUM",
817
+ "title": (
818
+ failure_type.replace("_", " ").title()
819
+ if failure_type
820
+ else "Execution blocker"
821
+ ),
822
+ "category": "EXECUTION_BLOCKER",
823
+ "root_cause": help_message or "Execution did not produce a validated repairable runtime result.",
824
+ "impact": "The intended QA workflow cannot establish complete runtime assurance until this blocker is resolved.",
825
+ "recommendation": "Resolve the execution blocker and rerun Stitch QA before making application-code repair decisions.",
826
+ "locations": [],
827
+ "evidence": [],
828
  }
829
 
830
 
831
+ def collect_locked_findings(request):
832
+ findings = []
833
+ findings.extend(runtime_findings(request))
834
+ findings.extend(runtime_warning_findings(request))
835
+ findings.extend(source_findings(request))
836
+ execution = execution_finding(request)
837
+ if execution:
838
+ findings.append(execution)
839
+
840
+ seen = set()
841
+ normalized = []
842
+ for finding in findings:
843
+ ref = finding["ref"]
844
+ if ref in seen:
845
+ continue
846
+ seen.add(ref)
847
+ normalized.append(finding)
848
+
849
+ return normalized
850
+
851
+
852
+ def select_ai_findings(findings):
853
+ source_order = {
854
+ "execution": 0,
855
+ "runtime": 1,
856
+ "source": 2,
857
+ }
858
 
859
+ ordered = sorted(
860
+ findings,
861
+ key=lambda finding: (
862
+ -RISK_ORDER.get(normalize_risk(finding.get("severity")), -1),
863
+ source_order.get(str(finding.get("source") or ""), 3),
864
+ str(finding.get("ref") or ""),
865
+ ),
866
+ )
867
+ return ordered[:MAX_AI_FINDINGS], ordered[MAX_AI_FINDINGS:]
 
 
 
 
 
 
868
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
869
 
870
+ def overall_priority_floor(request, findings):
871
+ analysis = request.runtime_analysis or {}
872
+ release_gate = str(analysis.get("release_gate") or "").upper()
873
 
874
+ if release_gate == "BLOCK_RELEASE":
875
+ return "P1"
876
 
877
+ runtime_evidence = request.runtime_evidence or {}
878
+ if str(runtime_evidence.get("test_result") or "").upper() == "FAIL":
879
+ return "P1"
880
 
881
+ if any(
882
+ finding.get("source") == "execution"
883
+ for finding in findings
884
+ ):
885
+ return "P1"
 
 
 
 
 
886
 
887
+ if any(
888
+ normalize_risk(finding.get("severity")) == "CRITICAL"
889
+ for finding in findings
890
+ ):
891
+ return "P1"
892
 
893
+ return "P3"
 
894
 
 
 
 
 
895
 
896
+ def contract_priority_floor(request, contract, findings):
897
+ refs = set(contract.finding_refs)
898
+ selected = [finding for finding in findings if finding.get("ref") in refs]
 
899
 
900
+ if any(finding.get("source") == "execution" for finding in selected):
901
+ return "P1"
902
+
903
+ runtime_evidence = request.runtime_evidence or {}
904
+ if (
905
+ str(runtime_evidence.get("test_result") or "").upper() == "FAIL"
906
+ and any(finding.get("source") == "runtime" for finding in selected)
907
+ ):
908
+ return "P1"
909
+
910
+ if any(
911
+ finding.get("source") == "source"
912
+ and normalize_risk(finding.get("severity")) == "CRITICAL"
913
+ for finding in selected
914
+ ):
915
+ return "P1"
916
+
917
+ return "P3"
918
+
919
+
920
+ def deterministic_confidence(request, findings):
921
+ analysis = request.runtime_analysis or {}
922
+ runtime_confidence = str(
923
+ analysis.get("diagnosis_confidence") or ""
924
+ ).upper()
925
+
926
+ if runtime_confidence in {"HIGH", "MEDIUM", "LOW"}:
927
+ return runtime_confidence
928
+
929
+ source_review = request.source_review or {}
930
+ source_confidences = {
931
+ str(item.get("confidence") or "").upper()
932
+ for item in normalize_list(source_review.get("findings"))
933
+ if isinstance(item, dict)
934
+ }
935
+
936
+ if source_confidences == {"HIGH"}:
937
+ return "HIGH"
938
+
939
+ return "MEDIUM" if findings else "HIGH"
940
+
941
+
942
+ def fallback_contract(finding, index, request):
943
+ priority = severity_priority(finding.get("severity"))
944
+
945
+ if finding.get("source") == "execution":
946
+ priority = "P1"
947
+
948
+ objective = (
949
+ finding.get("recommendation")
950
+ or f"Resolve the confirmed condition represented by {finding['ref']}."
951
+ )
952
+ strategy = (
953
+ finding.get("recommendation")
954
+ or "Apply the smallest targeted change that resolves the confirmed condition without broad unrelated changes."
955
+ )
956
+
957
+ runtime_analysis = request.runtime_analysis or {}
958
+ verification_steps = normalize_list(
959
+ runtime_analysis.get("verification_steps")
960
+ )
961
+ verification = (
962
+ " ".join(clean_text(item, 180) for item in verification_steps[:3])
963
+ if verification_steps
964
+ else "Confirm the affected behavior first, then run the relevant regression suite and verify no new failures."
965
+ )
966
+
967
+ return {
968
+ "contract_id": f"STITCH-RC-{index:03d}",
969
+ "finding_refs": [finding["ref"]],
970
+ "title": clean_text(finding.get("title"), 140),
971
+ "priority": priority,
972
+ "priority_reason": (
973
+ f"{finding['ref']} is prioritized from its validated impact and current QA blocking effect."
974
+ ),
975
+ "repair_objective": clean_text(objective, 260),
976
+ "repair_strategy": clean_text(strategy, 320),
977
+ "change_boundary": (
978
+ "Limit the change to the component, input boundary, dependency, or configuration directly represented by this finding."
979
+ ),
980
+ "protected_behavior": (
981
+ "Preserve behavior already shown to work outside the affected scope and avoid unrelated refactoring."
982
+ ),
983
+ "side_effect_risk": (
984
+ "MEDIUM"
985
+ if normalize_risk(finding.get("severity")) in {"CRITICAL", "HIGH", "MEDIUM"}
986
+ else "LOW"
987
+ ),
988
+ "verification": clean_text(verification, 300),
989
+ "done_condition": (
990
+ f"{finding['ref']} is no longer reproducible and the relevant regression checks complete without new failures."
991
+ ),
992
+ "status": "PENDING_VERIFICATION",
993
+ }
994
 
 
 
 
 
995
 
996
+ def finding_may_need_current_knowledge(finding):
997
+ text = " ".join(
998
+ clean_text(finding.get(key), 500).lower()
999
+ for key in (
1000
+ "title",
1001
+ "category",
1002
+ "root_cause",
1003
+ "impact",
1004
+ "recommendation",
1005
  )
1006
+ )
1007
+ indicators = (
1008
+ "dependency",
1009
+ "version",
1010
+ "deprecated",
1011
+ "deprecation",
1012
+ "compatibility",
1013
+ "plugin",
1014
+ "jdk",
1015
+ "cve",
1016
+ "security advisory",
1017
+ "vendor",
1018
+ "repository",
1019
+ )
1020
+ return any(indicator in text for indicator in indicators)
1021
+
1022
+
1023
+ def build_fallback_plan(request, findings, mode="deterministic-fallback", llm_error=None):
1024
+ if not findings:
1025
+ return {
1026
+ "agent_id": AGENT_ID,
1027
+ "display_name": DISPLAY_NAME,
1028
+ "agent_version": AGENT_VERSION,
1029
+ "agent": "repair-agent",
1030
+ "mode": "evidence-validated",
1031
+ "model": None,
1032
+ "status": (
1033
+ "NO_REPAIR_REQUIRED"
1034
+ if request.success
1035
+ else "NO_CONFIRMED_REPAIR_TARGET"
1036
+ ),
1037
+ "overall_priority": "NONE",
1038
+ "confidence": "HIGH",
1039
+ "repair_risk_level": "NONE",
1040
+ "auto_apply": False,
1041
+ "summary": (
1042
+ "No confirmed repair target was supplied by runtime or source-review evidence."
1043
+ if request.success
1044
+ else "Execution did not succeed, but no evidence-grounded repair target was available; no source change should be guessed."
1045
+ ),
1046
+ "stitch_repair_contracts": [],
1047
+ "current_knowledge_required": False,
1048
+ "current_knowledge_reason": None,
1049
+ "next_action": (
1050
+ "Retain the available QA evidence and continue the normal release workflow."
1051
+ if request.success
1052
+ else "Obtain a confirmed runtime, execution, or source-review finding before planning a code repair."
1053
+ ),
1054
+ "suggestions": [],
1055
+ "warnings": [],
1056
+ "limitations": [
1057
+ "Agent 2 only plans repairs for findings supplied by validated Stitch QA evidence."
1058
+ ],
1059
+ "llm_metrics": model_service.last_generation,
1060
+ "llm_error": llm_error,
1061
+ }
1062
+
1063
+ contracts = [
1064
+ fallback_contract(finding, index, request)
1065
+ for index, finding in enumerate(findings, start=1)
1066
+ ]
1067
+ contracts = sort_and_renumber_contracts(contracts)
1068
+ highest_priority = max(
1069
+ (contract["priority"] for contract in contracts),
1070
+ key=lambda item: PRIORITY_ORDER[item],
1071
+ default="P3",
1072
+ )
1073
+ floor = overall_priority_floor(request, findings)
1074
+ if PRIORITY_ORDER[highest_priority] < PRIORITY_ORDER[floor]:
1075
+ highest_priority = floor
1076
+
1077
+ repair_risk = highest_risk(
1078
+ *(contract["side_effect_risk"] for contract in contracts)
1079
+ )
1080
+
1081
+ warnings = []
1082
 
1083
  return {
1084
+ "agent_id": AGENT_ID,
1085
+ "display_name": DISPLAY_NAME,
1086
+ "agent_version": AGENT_VERSION,
1087
  "agent": "repair-agent",
1088
+ "mode": mode,
1089
+ "model": model_service.model_name or model_service.primary_model,
1090
+ "status": "COMPLETED",
1091
+ "overall_priority": highest_priority,
1092
+ "confidence": deterministic_confidence(request, findings),
1093
+ "repair_risk_level": repair_risk,
1094
  "auto_apply": False,
1095
+ "summary": (
1096
+ f"Prepared {len(contracts)} evidence-linked repair contract"
1097
+ f"{'s' if len(contracts) != 1 else ''} from {len(findings)} confirmed finding"
1098
+ f"{'s' if len(findings) != 1 else ''}."
1099
+ ),
1100
+ "stitch_repair_contracts": contracts,
1101
+ "current_knowledge_required": any(
1102
+ finding_may_need_current_knowledge(finding)
1103
+ for finding in findings
1104
+ ),
1105
+ "current_knowledge_reason": (
1106
+ "One or more repair targets depend on version, dependency, compatibility, vendor, deprecation, or security information that should be verified against current trusted documentation."
1107
+ if any(
1108
+ finding_may_need_current_knowledge(finding)
1109
+ for finding in findings
1110
+ )
1111
+ else None
1112
+ ),
1113
+ "next_action": (
1114
+ f"Start with {contracts[0]['contract_id']} and verify its done condition before broadening the repair scope."
1115
+ ),
1116
+ "suggestions": [
1117
+ contract["repair_strategy"]
1118
+ for contract in contracts
1119
+ ],
1120
+ "warnings": warnings,
1121
+ "limitations": [
1122
+ "This repair plan is grounded in supplied Stitch QA findings and does not automatically modify project code.",
1123
+ "Deterministic fallback prioritization is conservative and may be less context-sensitive than validated AI planning.",
1124
+ ],
1125
+ "llm_metrics": model_service.last_generation,
1126
+ "llm_error": llm_error,
1127
  }
1128
 
1129
 
1130
+ def compact_model_evidence(items):
1131
+ compacted = []
1132
+ for item in normalize_list(items)[:2]:
1133
+ if not isinstance(item, dict):
1134
+ continue
1135
+ compacted.append(
1136
+ {
1137
+ "id": clean_text(item.get("failure_id") or item.get("id"), 60),
1138
+ "test": clean_text(item.get("test_name"), 160),
1139
+ "expected": clean_text(item.get("expected"), 120),
1140
+ "actual": clean_text(item.get("actual"), 120),
1141
+ "exception": clean_text(item.get("exception_type"), 100),
1142
+ "message": clean_text(item.get("exception_message"), 220),
1143
+ "application": clean_text(item.get("application_file"), 200),
1144
+ "application_line": item.get("application_line"),
1145
+ "test_file": clean_text(item.get("test_file"), 200),
1146
+ "test_line": item.get("test_line"),
1147
+ }
1148
+ )
1149
+ return [
1150
+ {key: value for key, value in item.items() if value not in {None, ""}}
1151
+ for item in compacted
1152
+ ]
1153
+
1154
+
1155
+ def model_input_findings(findings):
1156
+ payload = []
1157
+
1158
+ for finding in findings[:MAX_AI_FINDINGS]:
1159
+ payload.append(
1160
+ {
1161
+ "ref": finding["ref"],
1162
+ "source": finding["source"],
1163
+ "severity": finding["severity"],
1164
+ "title": clean_text(finding.get("title"), 100),
1165
+ "category": clean_text(finding.get("category"), 70),
1166
+ "cause": clean_text(finding.get("root_cause"), 240),
1167
+ "impact": clean_text(finding.get("impact"), 180),
1168
+ "existing_action": clean_text(finding.get("recommendation"), 180),
1169
+ "locations": [
1170
+ clean_text(location, 160)
1171
+ for location in finding.get("locations", [])[:4]
1172
+ ],
1173
+ "evidence": compact_model_evidence(finding.get("evidence", [])),
1174
+ }
1175
+ )
1176
 
1177
+ return payload
1178
+
1179
+
1180
+ def build_messages(request, findings):
1181
+ analysis = request.runtime_analysis or {}
1182
+ source_review = request.source_review or {}
1183
+
1184
+ system_prompt = (
1185
+ "You are Stitch QA's Defect Resolution Intelligence Analyst. "
1186
+ "Your job is to transform locked QA findings into the safest prioritized repair plan without editing code. "
1187
+ "The supplied finding references, severities, files, lines, test facts, release gate, and observed evidence are authoritative. "
1188
+ "Never invent findings, files, lines, test outcomes, dependency versions, vulnerabilities, or current external facts. "
1189
+ "Prioritize by impact, blocking effect, dependency order, repair scope, and regression risk; severity and repair priority are related but not identical. "
1190
+ "Group findings only when one repair objective genuinely resolves them together. "
1191
+ "For every repair contract define the smallest useful change boundary, behavior that must remain working, side-effect risk, confirmation verification, regression verification, and a measurable done condition. "
1192
+ "Do not generate patches, code, commits, commands that modify the project, or automatic fixes. "
1193
+ "If a version, vendor behavior, dependency compatibility, deprecation, or security advisory needs up-to-date external documentation, set current_knowledge_required=true and explain why; do not invent the missing current fact. "
1194
+ "Return only JSON matching the required schema. "
1195
+ "Compact keys are fixed: p=overall priority, c=confidence, k=current-knowledge-needed, kr=current-knowledge reason, x=repair contracts; "
1196
+ "inside each contract f=finding refs, p=priority, w=priority reason, o=repair objective, s=repair strategy, b=change boundary, q=protected behavior, r=side-effect risk (L/M/H), v=verification. "
1197
+ "Keep every text value concise because the final professional report is formatted by Stitch QA."
1198
+ )
1199
 
1200
+ payload = {
1201
+ "project": {
1202
+ "type": request.project_type,
1203
  "command": request.command,
 
 
1204
  "success": request.success,
1205
+ "exit_code": request.exit_code,
 
 
 
 
 
 
1206
  "failure_type": request.failure_type,
1207
+ },
1208
+ "runtime_gate": {
1209
+ "test_result": analysis.get("test_result"),
1210
+ "release_gate": analysis.get("release_gate"),
1211
+ "runtime_risk": analysis.get("runtime_risk_level"),
1212
+ "confidence": analysis.get("diagnosis_confidence"),
1213
+ "evidence_quality": analysis.get("evidence_quality"),
1214
+ },
1215
+ "source_review": {
1216
+ "status": source_review.get("status"),
1217
+ "risk_level": source_review.get("risk_level"),
1218
+ "findings_count": len(
1219
+ normalize_list(source_review.get("findings"))
1220
+ ),
1221
+ },
1222
+ "locked_findings": model_input_findings(findings),
1223
+ "instructions": {
1224
+ "all_findings_must_be_covered": True,
1225
+ "contract_order_is_repair_order": True,
1226
+ "auto_apply": False,
1227
+ },
1228
+ }
1229
 
1230
+ return [
1231
+ {
1232
+ "role": "system",
1233
+ "content": system_prompt,
1234
+ },
1235
+ {
1236
+ "role": "user",
1237
+ "content": json.dumps(
1238
+ payload,
1239
+ ensure_ascii=False,
1240
+ separators=(",", ":"),
1241
+ ),
1242
+ },
1243
+ ]
1244
 
 
 
1245
 
1246
+ FILE_LINE_PATTERN = re.compile(
1247
+ r"(?P<path>[A-Za-z0-9_./\\-]+\.[A-Za-z0-9]+):(?P<line>\d+)"
1248
+ )
 
 
 
1249
 
1250
+ AUTO_APPLY_PATTERNS = {
1251
+ "i modified",
1252
+ "i changed",
1253
+ "i updated",
1254
+ "automatically modified",
1255
+ "automatically fixed",
1256
+ "auto-fix applied",
1257
+ "committed the",
1258
+ "pushed the",
1259
+ }
1260
 
 
 
1261
 
1262
+ def parse_model_output(text):
1263
+ try:
1264
+ data = json.loads(str(text or "").strip())
1265
+ except json.JSONDecodeError as error:
1266
+ raise ValueError("The Granite response was not valid JSON.") from error
1267
 
1268
+ try:
1269
+ return ModelRepairPlan.model_validate(data)
1270
+ except ValidationError as error:
1271
+ raise ValueError(
1272
+ f"The Granite response failed the repair-plan schema: {error}"
1273
+ ) from error
1274
+
1275
+
1276
+ def allowed_references(findings):
1277
+ allowed = set()
1278
+
1279
+ for finding in findings:
1280
+ for location in finding.get("locations", []):
1281
+ match = FILE_LINE_PATTERN.search(str(location))
1282
+ if not match:
1283
+ continue
1284
+ path = match.group("path").replace("\\", "/")
1285
+ line = int(match.group("line"))
1286
+ allowed.add((path, line))
1287
+ allowed.add((path.lstrip("./"), line))
1288
+
1289
+ return allowed
1290
+
1291
+
1292
+ def text_fields(plan):
1293
+ values = [plan.current_knowledge_reason]
1294
+ for contract in plan.contracts:
1295
+ values.extend(
1296
+ [
1297
+ contract.priority_reason,
1298
+ contract.repair_objective,
1299
+ contract.repair_strategy,
1300
+ contract.change_boundary,
1301
+ contract.protected_behavior,
1302
+ contract.verification,
1303
+ ]
1304
+ )
1305
+ return values
1306
+
1307
+
1308
+ def validate_plan(plan, request, findings):
1309
+ finding_refs = {finding["ref"] for finding in findings}
1310
+ used_refs = []
1311
+ combined_text = " ".join(text_fields(plan)).lower()
1312
+
1313
+ for pattern in AUTO_APPLY_PATTERNS:
1314
+ if pattern in combined_text:
1315
+ raise ValueError(
1316
+ "The Granite repair plan claimed or proposed automatic project modification."
1317
+ )
1318
+
1319
+ for contract in plan.contracts:
1320
+ for ref in contract.finding_refs:
1321
+ if ref not in finding_refs:
1322
+ raise ValueError(
1323
+ f"The Granite repair plan referenced unknown finding {ref}."
1324
+ )
1325
+ used_refs.append(ref)
1326
+
1327
+ missing_refs = finding_refs - set(used_refs)
1328
+ if missing_refs:
1329
+ raise ValueError(
1330
+ "The Granite repair plan omitted confirmed findings: "
1331
+ + ", ".join(sorted(missing_refs))
1332
+ )
1333
 
1334
+ duplicates = {
1335
+ ref
1336
+ for ref in used_refs
1337
+ if used_refs.count(ref) > 1
 
 
 
 
 
 
 
 
 
 
 
 
1338
  }
1339
+ if duplicates:
1340
+ raise ValueError(
1341
+ "The Granite repair plan assigned findings to multiple repair contracts: "
1342
+ + ", ".join(sorted(duplicates))
1343
+ )
1344
 
1345
+ for contract in plan.contracts:
1346
+ floor = contract_priority_floor(request, contract, findings)
1347
+ if PRIORITY_ORDER[contract.priority] < PRIORITY_ORDER[floor]:
1348
+ contract.priority = floor
1349
+
1350
+ allowed = allowed_references(findings)
1351
+ for value in text_fields(plan):
1352
+ for match in FILE_LINE_PATTERN.finditer(value or ""):
1353
+ path = match.group("path").replace("\\", "/")
1354
+ line = int(match.group("line"))
1355
+ if (path, line) not in allowed and (path.lstrip("./"), line) not in allowed:
1356
+ raise ValueError(
1357
+ "The Granite repair plan introduced an unsupported file or line reference."
1358
+ )
1359
+
1360
+ floor = overall_priority_floor(request, findings)
1361
+ if PRIORITY_ORDER[plan.overall_priority] < PRIORITY_ORDER[floor]:
1362
+ plan.overall_priority = floor
1363
+
1364
+ highest_contract_priority = max(
1365
+ (contract.priority for contract in plan.contracts),
1366
+ key=lambda item: PRIORITY_ORDER[item],
1367
+ )
1368
+ if PRIORITY_ORDER[plan.overall_priority] < PRIORITY_ORDER[highest_contract_priority]:
1369
+ plan.overall_priority = highest_contract_priority
1370
 
1371
+ if not plan.current_knowledge_required:
1372
+ plan.current_knowledge_reason = ""
 
 
 
 
 
1373
 
1374
+ return plan
 
 
 
 
 
1375
 
 
 
 
 
 
 
 
1376
 
1377
+ def sort_and_renumber_contracts(contracts):
1378
+ ordered = sorted(
1379
+ contracts,
1380
+ key=lambda contract: -PRIORITY_ORDER.get(contract.get("priority", "P3"), 1),
 
1381
  )
1382
+ for index, contract in enumerate(ordered, start=1):
1383
+ contract["contract_id"] = f"STITCH-RC-{index:03d}"
1384
+ return ordered
1385
+
1386
+
1387
+ def merge_model_plan(plan, request, findings, overflow_findings=None):
1388
+ overflow_findings = list(overflow_findings or [])
1389
+ contracts = []
1390
+
1391
+ for index, item in enumerate(plan.contracts, start=1):
1392
+ contracts.append(
1393
+ {
1394
+ "contract_id": f"STITCH-RC-{index:03d}",
1395
+ "finding_refs": item.finding_refs,
1396
+ "title": clean_text(
1397
+ f"Repair contract for {', '.join(item.finding_refs)}",
1398
+ 140,
1399
+ ),
1400
+ "priority": item.priority,
1401
+ "priority_reason": clean_text(item.priority_reason, 220),
1402
+ "repair_objective": clean_text(item.repair_objective, 260),
1403
+ "repair_strategy": clean_text(item.repair_strategy, 320),
1404
+ "change_boundary": clean_text(item.change_boundary, 240),
1405
+ "protected_behavior": clean_text(item.protected_behavior, 240),
1406
+ "side_effect_risk": item.side_effect_risk,
1407
+ "verification": clean_text(item.verification, 300),
1408
+ "done_condition": (
1409
+ "The referenced findings no longer reproduce, the targeted confirmation succeeds, and the stated regression verification introduces no new failure."
1410
+ ),
1411
+ "status": "PENDING_VERIFICATION",
1412
+ }
1413
+ )
1414
 
1415
+ next_index = len(contracts) + 1
1416
+ for offset, finding in enumerate(overflow_findings):
1417
+ contract = fallback_contract(finding, next_index + offset, request)
1418
+ contracts.append(contract)
1419
 
1420
+ contracts = sort_and_renumber_contracts(contracts)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1421
 
1422
+ repair_risk = highest_risk(
1423
+ *(contract["side_effect_risk"] for contract in contracts)
 
 
 
 
1424
  )
1425
 
1426
+ overall_priority = plan.overall_priority
1427
+ if overflow_findings:
1428
+ overflow_priority = max(
1429
+ (severity_priority(item.get("severity")) for item in overflow_findings),
1430
+ key=lambda item: PRIORITY_ORDER[item],
1431
+ default="P3",
1432
+ )
1433
+ if PRIORITY_ORDER[overall_priority] < PRIORITY_ORDER[overflow_priority]:
1434
+ overall_priority = overflow_priority
1435
+
1436
+ limitations = [
1437
+ "Agent 2 plans repairs from supplied Stitch QA evidence and does not automatically modify source code.",
1438
+ "Current external facts are not fetched inside this agent; cases marked current_knowledge_required need trusted documentation before implementation.",
 
 
 
 
 
 
 
1439
  ]
1440
+ warnings = []
1441
 
1442
+ if overflow_findings:
1443
+ warnings.append(
1444
+ f"{len(overflow_findings)} lower-priority finding(s) exceeded the AI reasoning window and were retained as conservative evidence-grounded repair contracts rather than being dropped."
1445
+ )
1446
+ limitations.append(
1447
+ "Overflow repair contracts use conservative deterministic planning because the AI reasoning window is intentionally bounded for CPU reliability."
1448
+ )
1449
 
1450
+ return {
1451
+ "agent_id": AGENT_ID,
1452
+ "display_name": DISPLAY_NAME,
1453
+ "agent_version": AGENT_VERSION,
1454
+ "agent": "repair-agent",
1455
+ "mode": "ai-reasoned-validated",
1456
+ "model": model_service.model_name or model_service.primary_model,
1457
+ "status": "COMPLETED",
1458
+ "overall_priority": overall_priority,
1459
+ "confidence": plan.confidence,
1460
+ "repair_risk_level": repair_risk,
1461
+ "auto_apply": False,
1462
+ "summary": clean_text(
1463
+ f"Prepared {len(contracts)} prioritized evidence-linked repair contract"
1464
+ f"{'s' if len(contracts) != 1 else ''}; start with "
1465
+ f"{contracts[0]['repair_strategy'] if contracts else 'the highest-priority validated repair target'}.",
1466
+ 360,
1467
+ ),
1468
+ "stitch_repair_contracts": contracts,
1469
+ "current_knowledge_required": (
1470
+ plan.current_knowledge_required
1471
+ or any(
1472
+ finding_may_need_current_knowledge(finding)
1473
+ for finding in [*findings, *overflow_findings]
1474
+ )
1475
+ ),
1476
+ "current_knowledge_reason": (
1477
+ clean_text(plan.current_knowledge_reason, 240)
1478
+ if plan.current_knowledge_required and clean_text(plan.current_knowledge_reason, 240)
1479
+ else (
1480
+ "One or more repair targets depend on current version, dependency, compatibility, vendor, deprecation, or security documentation that must be verified before implementation."
1481
+ if any(
1482
+ finding_may_need_current_knowledge(finding)
1483
+ for finding in [*findings, *overflow_findings]
1484
+ )
1485
+ else None
1486
+ )
1487
+ ),
1488
+ "next_action": (
1489
+ f"Start with {contracts[0]['contract_id']}: {contracts[0]['repair_strategy']}"
1490
+ if contracts
1491
+ else None
1492
+ ),
1493
+ "suggestions": [
1494
+ contract["repair_strategy"]
1495
+ for contract in contracts
1496
+ ],
1497
+ "warnings": warnings,
1498
+ "limitations": limitations,
1499
+ "llm_metrics": model_service.last_generation,
1500
+ "llm_error": None,
1501
+ }
1502
 
1503
+ def should_use_llm(findings):
1504
+ return (
1505
+ model_service.enabled
1506
+ and bool(findings)
1507
+ and any(
1508
+ finding.get("source") in {"runtime", "source"}
1509
+ for finding in findings
1510
+ )
1511
+ )
1512
 
 
1513
 
1514
+ @app.get("/")
1515
+ def health_check():
1516
+ status = model_service.status()
1517
+ return {
1518
+ "service": "stitch-qa-repair-agent",
1519
+ "agent_id": AGENT_ID,
1520
+ "display_name": DISPLAY_NAME,
1521
+ "agent_version": AGENT_VERSION,
1522
+ "status": "running",
1523
+ "llm": status,
1524
+ }
1525
 
 
 
 
 
 
1526
 
1527
+ @app.get("/ready")
1528
+ def readiness_check():
1529
+ status = model_service.status()
1530
+ return {
1531
+ "ready": True,
1532
+ "analysis_ready": True,
1533
+ "agent_id": AGENT_ID,
1534
+ "llm_enabled": status["enabled"],
1535
+ "llm_loaded": status["loaded"],
1536
+ "llm_state": status["state"],
1537
+ "configured_model": status["configured_model"],
1538
+ "active_model": status["active_model"],
1539
+ "deterministic_fallback": True,
1540
+ }
1541
 
 
1542
 
1543
+ @app.post("/suggest", response_model=RepairResponse)
1544
+ def suggest_repair(request: RepairRequest):
1545
+ findings = collect_locked_findings(request)
 
 
 
1546
 
1547
+ if not findings:
1548
+ return RepairResponse.model_validate(
1549
+ build_fallback_plan(request, findings)
1550
+ )
1551
+
1552
+ if not should_use_llm(findings):
1553
+ return RepairResponse.model_validate(
1554
+ build_fallback_plan(
1555
+ request,
1556
+ findings,
1557
+ mode="deterministic-validated",
1558
+ )
1559
+ )
1560
 
1561
+ ai_findings, overflow_findings = select_ai_findings(findings)
1562
+
1563
+ try:
1564
+ messages = build_messages(request, ai_findings)
1565
+ text = model_service.generate(messages)
1566
+ plan = parse_model_output(text)
1567
+ plan = validate_plan(plan, request, ai_findings)
1568
+ result = merge_model_plan(
1569
+ plan,
1570
+ request,
1571
+ ai_findings,
1572
+ overflow_findings,
1573
+ )
1574
  except Exception as error:
1575
+ result = build_fallback_plan(
1576
+ request,
1577
+ findings,
1578
+ mode="deterministic-fallback",
1579
+ llm_error=repr(error),
1580
+ )
1581
+
1582
+ return RepairResponse.model_validate(result)
requirements.txt CHANGED
@@ -1,6 +1,5 @@
1
- fastapi
2
- uvicorn
3
- pydantic
4
- transformers
5
- torch
6
- sentencepiece
 
1
+ fastapi==0.141.1
2
+ uvicorn==0.52.1
3
+ pydantic==2.13.4
4
+ huggingface_hub==1.27.0
5
+ llama-cpp-python==0.3.34