"""
Adaptive Agent Memory Resilience: Interactive Research Playground
Author: Sumit Das (arXiv Pre-print 2026)
Demonstrates Bayesian Trust Updating, Pessimistic Lower Confidence Bound (LCB) Retrieval,
and Theorem 1 Statistical Quarantine for Autonomous LLM Agents.
Strict compliance: Zero emojis, rigorous mathematical formulas, typed logic.
"""
import math
from pathlib import Path
from typing import Dict, List, Tuple
import gradio as gr
import numpy as np
import pandas as pd
import plotly.graph_objects as go
import scipy.special as sc
import scipy.stats as stats
# --- Mathematical Engine (Conjugate Beta-Bernoulli & Theorem 1) ---
def beta_trust_mean(successes: int, failures: int, alpha_0: float = 3.0, beta_0: float = 1.0) -> float:
alpha_post = alpha_0 + float(successes)
beta_post = beta_0 + float(failures)
return float(alpha_post / (alpha_post + beta_post))
def beta_trust_var(successes: int, failures: int, alpha_0: float = 3.0, beta_0: float = 1.0) -> float:
alpha_post = alpha_0 + float(successes)
beta_post = beta_0 + float(failures)
total = alpha_post + beta_post
return float((alpha_post * beta_post) / ((total ** 2) * (total + 1.0)))
def beta_trust_std(successes: int, failures: int, alpha_0: float = 3.0, beta_0: float = 1.0) -> float:
return float(math.sqrt(beta_trust_var(successes, failures, alpha_0, beta_0)))
def beta_lcb(
successes: int,
failures: int,
lambda_risk: float = 1.0,
alpha_0: float = 3.0,
beta_0: float = 1.0,
) -> float:
mu = beta_trust_mean(successes, failures, alpha_0, beta_0)
sigma = beta_trust_std(successes, failures, alpha_0, beta_0)
score = mu - (lambda_risk * sigma)
return float(max(0.0, min(1.0, score)))
def posterior_probability_reliable(
successes: int,
failures: int,
threshold: float = 0.70,
alpha_0: float = 3.0,
beta_0: float = 1.0,
) -> float:
alpha_post = alpha_0 + float(successes)
beta_post = beta_0 + float(failures)
cdf_at_thresh = float(sc.betainc(alpha_post, beta_post, threshold))
return float(max(0.0, min(1.0, 1.0 - cdf_at_thresh)))
def should_quarantine(
successes: int,
failures: int,
gamma: float = 0.05,
threshold: float = 0.70,
alpha_0: float = 3.0,
beta_0: float = 1.0,
) -> bool:
prob = posterior_probability_reliable(successes, failures, threshold, alpha_0, beta_0)
return prob < gamma
def compute_consecutive_failure_series(
alpha_0: float = 3.0,
beta_0: float = 1.0,
gamma: float = 0.05,
threshold: float = 0.70,
max_steps: int = 6,
) -> pd.DataFrame:
rows = []
for t in range(0, max_steps + 1):
alpha_post = alpha_0
beta_post = beta_0 + float(t)
mu = alpha_post / (alpha_post + beta_post)
std = math.sqrt((alpha_post * beta_post) / (((alpha_post + beta_post) ** 2) * (alpha_post + beta_post + 1.0)))
p_rel = posterior_probability_reliable(0, t, threshold, alpha_0, beta_0)
is_q = p_rel < gamma
status = "QUARANTINED (Threshold Triggered)" if is_q else "ACTIVE (Admissible)"
rows.append({
"Consecutive Failures (t)": t,
"Posterior Beta": f"Beta({alpha_post:.1f}, {beta_post:.1f})",
"Posterior Mean E[theta]": f"{mu:.4f}",
"Uncertainty (sigma)": f"{std:.4f}",
"P(theta > 0.70)": f"{p_rel * 100:.2f}%",
"Quarantine Status": status,
})
return pd.DataFrame(rows)
# --- Visualizers ---
def plot_beta_posterior(
alpha_0: float,
beta_0: float,
successes: int,
failures: int,
lambda_risk: float,
threshold: float,
gamma: float,
) -> Tuple[go.Figure, str, str, str, str, str]:
alpha_post = alpha_0 + float(successes)
beta_post = beta_0 + float(failures)
mu = beta_trust_mean(successes, failures, alpha_0, beta_0)
std = beta_trust_std(successes, failures, alpha_0, beta_0)
lcb_val = beta_lcb(successes, failures, lambda_risk, alpha_0, beta_0)
p_rel = posterior_probability_reliable(successes, failures, threshold, alpha_0, beta_0)
is_quarantined = p_rel < gamma
x = np.linspace(0.001, 0.999, 600)
y = stats.beta.pdf(x, alpha_post, beta_post)
fig = go.Figure()
# Prior curve for visual comparison
y_prior = stats.beta.pdf(x, alpha_0, beta_0)
fig.add_trace(go.Scatter(
x=x,
y=y_prior,
mode="lines",
line=dict(color="rgba(140, 140, 140, 0.6)", width=1.5, dash="dash"),
name=f"Prior Beta({alpha_0:.1f}, {beta_0:.1f})",
))
# Posterior curve
fig.add_trace(go.Scatter(
x=x,
y=y,
mode="lines",
line=dict(color="#2563eb", width=3),
name=f"Posterior Beta({alpha_post:.1f}, {beta_post:.1f})",
))
# Shading for Reliable region (theta >= threshold)
x_rel = x[x >= threshold]
y_rel = y[x >= threshold]
if len(x_rel) > 0:
fig.add_trace(go.Scatter(
x=np.concatenate([[threshold], x_rel, [x_rel[-1]]]),
y=np.concatenate([[0], y_rel, [0]]),
fill="toself",
fillcolor="rgba(34, 197, 94, 0.25)",
line=dict(color="rgba(255,255,255,0)"),
name=f"Admissible Region (P={p_rel*100:.1f}%)",
hoverinfo="skip",
))
# Vertical reference lines
max_y = float(np.max(y)) * 1.05 if np.max(y) > 0 else 5.0
# Operational threshold line
fig.add_vline(
x=threshold,
line=dict(color="#ef4444", width=2, dash="dash"),
annotation_text=f"Admissibility Threshold ({threshold:.2f})",
annotation_position="top left",
)
# Posterior mean line
fig.add_vline(
x=mu,
line=dict(color="#2563eb", width=2),
annotation_text=f"Mean E[theta]={mu:.3f}",
annotation_position="top right",
)
# LCB line
fig.add_vline(
x=lcb_val,
line=dict(color="#8b5cf6", width=2, dash="dot"),
annotation_text=f"LCB(lambda={lambda_risk:.1f})={lcb_val:.3f}",
annotation_position="bottom left",
)
fig.update_layout(
title=dict(
text=f"Posterior Reliability Distribution: Beta({alpha_post:.1f}, {beta_post:.1f}) vs Prior Beta({alpha_0:.1f}, {beta_0:.1f})",
font=dict(size=15),
),
xaxis=dict(title="True Latent Reliability theta in [0, 1]", range=[0.0, 1.0], gridcolor="#e5e7eb"),
yaxis=dict(title="Probability Density f(theta)", range=[0.0, max_y], gridcolor="#e5e7eb"),
template="plotly_white",
margin=dict(l=40, r=40, t=50, b=40),
legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1.0),
)
status_str = "QUARANTINED (Pruned from Retrieval)" if is_quarantined else "ACTIVE (Admissible for Retrieval)"
status_bg = "#fee2e2" if is_quarantined else "#dcfce7"
status_text_color = "#991b1b" if is_quarantined else "#166534"
status_card = (
f"
"
f"Operational Status: {status_str}
"
)
metric_mean = f"{mu:.4f}"
metric_sigma = f"{std:.4f}"
metric_lcb = f"{lcb_val:.4f}"
metric_prob = f"{p_rel * 100:.2f}%"
return fig, status_card, metric_mean, metric_sigma, metric_lcb, metric_prob
# --- Retrieval Simulation Data ---
SAMPLE_MEMORIES = [
{
"id": "MEM-CORRUPT-01",
"domain": "coding",
"lesson": "Use global shared state across async workers without mutex lock for throughput.",
"cosine_sim": 0.94,
"successes": 2,
"failures": 5,
"note": "Corrupted/stale reflection causing race condition deadlocks.",
},
{
"id": "MEM-ROBUST-02",
"domain": "coding",
"lesson": "Implement asyncio.Lock with timeout fallback and circuit breaker isolation.",
"cosine_sim": 0.83,
"successes": 22,
"failures": 1,
"note": "High empirical validation under stress tests.",
},
{
"id": "MEM-UNTESTED-03",
"domain": "coding",
"lesson": "Refactor threading pool to use separate child process queues.",
"cosine_sim": 0.89,
"successes": 0,
"failures": 0,
"note": "Newly distilled reflection with zero empirical execution history.",
},
{
"id": "MEM-MARGINAL-04",
"domain": "coding",
"lesson": "Log execution traceback to local scratch buffer before propagating exceptions.",
"cosine_sim": 0.65,
"successes": 15,
"failures": 2,
"note": "Low direct task relevance, but high historical reliability.",
},
]
def evaluate_retrieval_ranking(lambda_risk: float, alpha_0: float = 3.0, beta_0: float = 1.0) -> pd.DataFrame:
rows = []
for m in SAMPLE_MEMORIES:
ns = m["successes"]
nf = m["failures"]
sim = m["cosine_sim"]
mu = beta_trust_mean(ns, nf, alpha_0, beta_0)
std = beta_trust_std(ns, nf, alpha_0, beta_0)
lcb_score = beta_lcb(ns, nf, lambda_risk, alpha_0, beta_0)
p_rel = posterior_probability_reliable(ns, nf, 0.70, alpha_0, beta_0)
is_q = should_quarantine(ns, nf, 0.05, 0.70, alpha_0, beta_0)
naive_score = sim
composite_score = 0.0 if is_q else sim * lcb_score
rows.append({
"Memory ID": m["id"],
"Cosine Sim": sim,
"Successes (ns)": ns,
"Failures (nf)": nf,
"Posterior Mean E[theta]": round(mu, 3),
"Uncertainty (sigma)": round(std, 3),
"LCB Score": round(lcb_score, 3),
"Naive Retrieval Score": round(naive_score, 3),
"Pessimistic LCB Composite": round(composite_score, 3),
"Quarantine": "QUARANTINED" if is_q else "ACTIVE",
"Strategy Directives": m["lesson"],
})
df = pd.DataFrame(rows)
# Sort primarily by proposed composite score descending
df = df.sort_values(by="Pessimistic LCB Composite", ascending=False).reset_index(drop=True)
df.insert(0, "LCB Rank", [f"#{i+1}" for i in range(len(df))])
return df
# --- Load Real Benchmark Data for Explorer ---
DATA_DIR = Path(__file__).parent / "data"
def load_benchmark_summary() -> Tuple[pd.DataFrame, pd.DataFrame]:
traces_path = DATA_DIR / "telemetry_traces.csv"
mem_path = DATA_DIR / "memory_bank_experiences.csv"
if traces_path.exists():
df_traces = pd.read_csv(traces_path)
summary = (
df_traces.groupby("ablation_condition")
.agg(
Total_Steps=("step_index", "count"),
Accuracy=("task_success", "mean"),
Avg_Reward=("observed_reward", "mean"),
Mean_Similarity=("cosine_similarity", "mean"),
Quarantine_Triggers=("quarantine_triggered", "sum"),
)
.reset_index()
)
summary["Accuracy"] = summary["Accuracy"].apply(lambda v: f"{v * 100:.2f}%")
summary["Avg_Reward"] = summary["Avg_Reward"].apply(lambda v: f"{v:.3f}")
summary["Mean_Similarity"] = summary["Mean_Similarity"].apply(lambda v: f"{v:.3f}")
else:
summary = pd.DataFrame({"Notice": ["telemetry_traces.csv not found locally."]})
if mem_path.exists():
df_mem = pd.read_csv(mem_path)
mem_preview = df_mem[[
"experience_id", "task_domain", "successes_count", "failures_count",
"posterior_mean_trust", "posterior_variance", "pessimistic_lcb_score",
"quarantine_status", "is_adversarial_sample"
]].head(25)
else:
mem_preview = pd.DataFrame({"Notice": ["memory_bank_experiences.csv not found locally."]})
return summary, mem_preview
# --- Gradio Application Layout ---
def build_app() -> gr.Blocks:
theme = gr.themes.Soft(
primary_hue="blue",
neutral_hue="slate",
)
with gr.Blocks(title="Adaptive Agent Memory Resilience Playground") as demo:
gr.Markdown(
"""
# Adaptive Agent Memory Resilience Playground
### Mathematical Trust Dynamics & Pessimistic LCB Memory Retrieval for Autonomous LLM Agents
**Author:** Sumit Das (`@sumitaidev`) | Research Paper Pre-print (2026) | [Benchmark Dataset](https://huggingface.co/datasets/sumitaidev/agent-memory-resilience-benchmark)
This research tool allows AI researchers and engineers to interactively analyze how Bayesian conjugate updating,
epistemic variance quantification, and pessimistic Lower Confidence Bound (LCB) composite retrieval prevent
**Negative Transfer** and **Memory Poisoning** in agent systems (such as LangGraph, AutoGen, and CrewAI).
"""
)
with gr.Tabs():
# TAB 1: Bayesian Reliability & Quarantine Engine
with gr.TabItem("1. Bayesian Reliability & Theorem 1"):
gr.Markdown(
"""
### Conjugate Beta-Bernoulli Posterior Update Engine
Simulates task trials for an individual experiential memory record under weakly-informative prior $\\operatorname{Beta}(\\alpha_0=3.0, \\beta_0=1.0)$.
Theorem 1 states that at operational threshold $\\theta=0.70$ and significance $\\gamma=0.05$, exactly $t^* = 4$ consecutive failures triggers deterministic quarantine.
"""
)
with gr.Row():
with gr.Column(scale=1):
alpha_0_input = gr.Slider(0.5, 10.0, value=3.0, step=0.5, label="Prior Alpha (alpha_0)")
beta_0_input = gr.Slider(0.5, 10.0, value=1.0, step=0.5, label="Prior Beta (beta_0)")
successes_input = gr.Slider(0, 50, value=0, step=1, label="Observed Task Successes (n_s)")
failures_input = gr.Slider(0, 20, value=0, step=1, label="Observed Task Failures (n_f)")
lambda_input = gr.Slider(0.0, 3.0, value=1.0, step=0.2, label="Risk Aversion Parameter (lambda)")
threshold_input = gr.Slider(0.5, 0.9, value=0.70, step=0.05, label="Admissibility Standard (threshold)")
gamma_input = gr.Slider(0.01, 0.20, value=0.05, step=0.01, label="Quarantine Significance Level (gamma)")
status_html = gr.HTML()
with gr.Column(scale=2):
with gr.Row():
mean_kpi = gr.Textbox(label="Posterior Mean E[theta]", interactive=False)
sigma_kpi = gr.Textbox(label="Uncertainty (sigma)", interactive=False)
lcb_kpi = gr.Textbox(label="LCB Score", interactive=False)
prob_kpi = gr.Textbox(label="P(theta > threshold)", interactive=False)
plot_output = gr.Plot(label="Posterior Reliability Distribution")
gr.Markdown("#### Theorem 1 Deterministic Verification: Consecutive Failure Trajectory")
theorem_table = gr.Dataframe(
headers=["Consecutive Failures (t)", "Posterior Beta", "Posterior Mean E[theta]", "Uncertainty (sigma)", "P(theta > 0.70)", "Quarantine Status"],
interactive=False,
)
def update_tab1(a0, b0, ns, nf, lam, thresh, gam):
fig, card, m_val, s_val, lcb_val, p_val = plot_beta_posterior(
a0, b0, int(ns), int(nf), lam, thresh, gam
)
th_df = compute_consecutive_failure_series(a0, b0, gam, thresh)
return fig, card, m_val, s_val, lcb_val, p_val, th_df
inputs_tab1 = [alpha_0_input, beta_0_input, successes_input, failures_input, lambda_input, threshold_input, gamma_input]
outputs_tab1 = [plot_output, status_html, mean_kpi, sigma_kpi, lcb_kpi, prob_kpi, theorem_table]
for comp in inputs_tab1:
comp.change(fn=update_tab1, inputs=inputs_tab1, outputs=outputs_tab1)
demo.load(fn=update_tab1, inputs=inputs_tab1, outputs=outputs_tab1)
# TAB 2: Retrieval Ranking Playground
with gr.TabItem("2. Pessimistic LCB Memory Retrieval"):
gr.Markdown(
"""
### Retrieval Competition: Naive Cosine Similarity vs Pessimistic LCB
Demonstrates how standard vector RAG repeatedly retrieves corrupted memories with high semantic similarity,
causing catastrophic cascade failures. Our composite formulation:
$$\\operatorname{Score}(e; q) = \\operatorname{Sim}(\\mathbf{q}, \\mathbf{v}_e) \\times \\operatorname{LCB}_\\lambda(e)$$
penalizes uncertain reflections and quarantines verified failures.
"""
)
with gr.Row():
retrieval_lambda = gr.Slider(0.0, 3.0, value=1.0, step=0.25, label="Risk Aversion Parameter (lambda)")
refresh_retrieval_btn = gr.Button("Recompute Rankings")
retrieval_table = gr.Dataframe(interactive=False)
gr.Markdown(
"""
**Key Analytical Observations**:
- At $\\lambda = 0.0$ (Risk-neutral/Naive Cosine): `MEM-CORRUPT-01` ranks #1 because its cosine similarity is 0.94.
- At $\\lambda \\ge 1.0$ (Pessimistic LCB): `MEM-CORRUPT-01` is flagged for quarantine, and `MEM-ROBUST-02` (proven with 22 successes) rightfully takes #1 rank.
- `MEM-UNTESTED-03` with 0 executions receives an uncertainty penalty, preventing the agent from blindly over-trusting unvalidated reflections.
"""
)
def update_retrieval(lam):
return evaluate_retrieval_ranking(lam)
retrieval_lambda.change(fn=update_retrieval, inputs=[retrieval_lambda], outputs=[retrieval_table])
refresh_retrieval_btn.click(fn=update_retrieval, inputs=[retrieval_lambda], outputs=[retrieval_table])
demo.load(fn=update_retrieval, inputs=[retrieval_lambda], outputs=[retrieval_table])
# TAB 3: Benchmark Traces & Empirical Results
with gr.TabItem("3. Benchmark Traces & Ablation"):
gr.Markdown(
"""
### Empirical Benchmark Telemetry Summary (1,200 Execution Steps)
Comparative performance across four controlled experimental conditions under adversarial noise injection (steps 60 to 140).
"""
)
summary_df, preview_df = load_benchmark_summary()
gr.Markdown("#### Ablation Conditions Summary")
gr.Dataframe(value=summary_df, interactive=False)
gr.Markdown("#### Experiential Memory Bank Sample (100 Verified Records)")
gr.Dataframe(value=preview_df, interactive=False)
gr.Markdown(
"""
### Citation
```bibtex
@article{das2026adaptive,
title={Adaptive Agent Memory Resilience: Mitigating Negative Transfer and Memory Poisoning via Bayesian Trust Updating and Pessimistic Lower Confidence Bound Retrieval},
author={Das, Sumit},
journal={arXiv preprint},
year={2026}
}
```
"""
)
return demo
if __name__ == "__main__":
app = build_app()
app.launch(theme=gr.themes.Soft(primary_hue="blue", neutral_hue="slate"))