""" Adaptive Agent Memory Resilience: Interactive Research Playground Author: Sumit Das (arXiv Pre-print 2026) Demonstrates Bayesian Trust Updating, Pessimistic Lower Confidence Bound (LCB) Retrieval, and Theorem 1 Statistical Quarantine for Autonomous LLM Agents. Strict compliance: Zero emojis, rigorous mathematical formulas, typed logic. """ import math from pathlib import Path from typing import Dict, List, Tuple import gradio as gr import numpy as np import pandas as pd import plotly.graph_objects as go import scipy.special as sc import scipy.stats as stats # --- Mathematical Engine (Conjugate Beta-Bernoulli & Theorem 1) --- def beta_trust_mean(successes: int, failures: int, alpha_0: float = 3.0, beta_0: float = 1.0) -> float: alpha_post = alpha_0 + float(successes) beta_post = beta_0 + float(failures) return float(alpha_post / (alpha_post + beta_post)) def beta_trust_var(successes: int, failures: int, alpha_0: float = 3.0, beta_0: float = 1.0) -> float: alpha_post = alpha_0 + float(successes) beta_post = beta_0 + float(failures) total = alpha_post + beta_post return float((alpha_post * beta_post) / ((total ** 2) * (total + 1.0))) def beta_trust_std(successes: int, failures: int, alpha_0: float = 3.0, beta_0: float = 1.0) -> float: return float(math.sqrt(beta_trust_var(successes, failures, alpha_0, beta_0))) def beta_lcb( successes: int, failures: int, lambda_risk: float = 1.0, alpha_0: float = 3.0, beta_0: float = 1.0, ) -> float: mu = beta_trust_mean(successes, failures, alpha_0, beta_0) sigma = beta_trust_std(successes, failures, alpha_0, beta_0) score = mu - (lambda_risk * sigma) return float(max(0.0, min(1.0, score))) def posterior_probability_reliable( successes: int, failures: int, threshold: float = 0.70, alpha_0: float = 3.0, beta_0: float = 1.0, ) -> float: alpha_post = alpha_0 + float(successes) beta_post = beta_0 + float(failures) cdf_at_thresh = float(sc.betainc(alpha_post, beta_post, threshold)) return float(max(0.0, min(1.0, 1.0 - cdf_at_thresh))) def should_quarantine( successes: int, failures: int, gamma: float = 0.05, threshold: float = 0.70, alpha_0: float = 3.0, beta_0: float = 1.0, ) -> bool: prob = posterior_probability_reliable(successes, failures, threshold, alpha_0, beta_0) return prob < gamma def compute_consecutive_failure_series( alpha_0: float = 3.0, beta_0: float = 1.0, gamma: float = 0.05, threshold: float = 0.70, max_steps: int = 6, ) -> pd.DataFrame: rows = [] for t in range(0, max_steps + 1): alpha_post = alpha_0 beta_post = beta_0 + float(t) mu = alpha_post / (alpha_post + beta_post) std = math.sqrt((alpha_post * beta_post) / (((alpha_post + beta_post) ** 2) * (alpha_post + beta_post + 1.0))) p_rel = posterior_probability_reliable(0, t, threshold, alpha_0, beta_0) is_q = p_rel < gamma status = "QUARANTINED (Threshold Triggered)" if is_q else "ACTIVE (Admissible)" rows.append({ "Consecutive Failures (t)": t, "Posterior Beta": f"Beta({alpha_post:.1f}, {beta_post:.1f})", "Posterior Mean E[theta]": f"{mu:.4f}", "Uncertainty (sigma)": f"{std:.4f}", "P(theta > 0.70)": f"{p_rel * 100:.2f}%", "Quarantine Status": status, }) return pd.DataFrame(rows) # --- Visualizers --- def plot_beta_posterior( alpha_0: float, beta_0: float, successes: int, failures: int, lambda_risk: float, threshold: float, gamma: float, ) -> Tuple[go.Figure, str, str, str, str, str]: alpha_post = alpha_0 + float(successes) beta_post = beta_0 + float(failures) mu = beta_trust_mean(successes, failures, alpha_0, beta_0) std = beta_trust_std(successes, failures, alpha_0, beta_0) lcb_val = beta_lcb(successes, failures, lambda_risk, alpha_0, beta_0) p_rel = posterior_probability_reliable(successes, failures, threshold, alpha_0, beta_0) is_quarantined = p_rel < gamma x = np.linspace(0.001, 0.999, 600) y = stats.beta.pdf(x, alpha_post, beta_post) fig = go.Figure() # Prior curve for visual comparison y_prior = stats.beta.pdf(x, alpha_0, beta_0) fig.add_trace(go.Scatter( x=x, y=y_prior, mode="lines", line=dict(color="rgba(140, 140, 140, 0.6)", width=1.5, dash="dash"), name=f"Prior Beta({alpha_0:.1f}, {beta_0:.1f})", )) # Posterior curve fig.add_trace(go.Scatter( x=x, y=y, mode="lines", line=dict(color="#2563eb", width=3), name=f"Posterior Beta({alpha_post:.1f}, {beta_post:.1f})", )) # Shading for Reliable region (theta >= threshold) x_rel = x[x >= threshold] y_rel = y[x >= threshold] if len(x_rel) > 0: fig.add_trace(go.Scatter( x=np.concatenate([[threshold], x_rel, [x_rel[-1]]]), y=np.concatenate([[0], y_rel, [0]]), fill="toself", fillcolor="rgba(34, 197, 94, 0.25)", line=dict(color="rgba(255,255,255,0)"), name=f"Admissible Region (P={p_rel*100:.1f}%)", hoverinfo="skip", )) # Vertical reference lines max_y = float(np.max(y)) * 1.05 if np.max(y) > 0 else 5.0 # Operational threshold line fig.add_vline( x=threshold, line=dict(color="#ef4444", width=2, dash="dash"), annotation_text=f"Admissibility Threshold ({threshold:.2f})", annotation_position="top left", ) # Posterior mean line fig.add_vline( x=mu, line=dict(color="#2563eb", width=2), annotation_text=f"Mean E[theta]={mu:.3f}", annotation_position="top right", ) # LCB line fig.add_vline( x=lcb_val, line=dict(color="#8b5cf6", width=2, dash="dot"), annotation_text=f"LCB(lambda={lambda_risk:.1f})={lcb_val:.3f}", annotation_position="bottom left", ) fig.update_layout( title=dict( text=f"Posterior Reliability Distribution: Beta({alpha_post:.1f}, {beta_post:.1f}) vs Prior Beta({alpha_0:.1f}, {beta_0:.1f})", font=dict(size=15), ), xaxis=dict(title="True Latent Reliability theta in [0, 1]", range=[0.0, 1.0], gridcolor="#e5e7eb"), yaxis=dict(title="Probability Density f(theta)", range=[0.0, max_y], gridcolor="#e5e7eb"), template="plotly_white", margin=dict(l=40, r=40, t=50, b=40), legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1.0), ) status_str = "QUARANTINED (Pruned from Retrieval)" if is_quarantined else "ACTIVE (Admissible for Retrieval)" status_bg = "#fee2e2" if is_quarantined else "#dcfce7" status_text_color = "#991b1b" if is_quarantined else "#166534" status_card = ( f"
" f"Operational Status: {status_str}
" ) metric_mean = f"{mu:.4f}" metric_sigma = f"{std:.4f}" metric_lcb = f"{lcb_val:.4f}" metric_prob = f"{p_rel * 100:.2f}%" return fig, status_card, metric_mean, metric_sigma, metric_lcb, metric_prob # --- Retrieval Simulation Data --- SAMPLE_MEMORIES = [ { "id": "MEM-CORRUPT-01", "domain": "coding", "lesson": "Use global shared state across async workers without mutex lock for throughput.", "cosine_sim": 0.94, "successes": 2, "failures": 5, "note": "Corrupted/stale reflection causing race condition deadlocks.", }, { "id": "MEM-ROBUST-02", "domain": "coding", "lesson": "Implement asyncio.Lock with timeout fallback and circuit breaker isolation.", "cosine_sim": 0.83, "successes": 22, "failures": 1, "note": "High empirical validation under stress tests.", }, { "id": "MEM-UNTESTED-03", "domain": "coding", "lesson": "Refactor threading pool to use separate child process queues.", "cosine_sim": 0.89, "successes": 0, "failures": 0, "note": "Newly distilled reflection with zero empirical execution history.", }, { "id": "MEM-MARGINAL-04", "domain": "coding", "lesson": "Log execution traceback to local scratch buffer before propagating exceptions.", "cosine_sim": 0.65, "successes": 15, "failures": 2, "note": "Low direct task relevance, but high historical reliability.", }, ] def evaluate_retrieval_ranking(lambda_risk: float, alpha_0: float = 3.0, beta_0: float = 1.0) -> pd.DataFrame: rows = [] for m in SAMPLE_MEMORIES: ns = m["successes"] nf = m["failures"] sim = m["cosine_sim"] mu = beta_trust_mean(ns, nf, alpha_0, beta_0) std = beta_trust_std(ns, nf, alpha_0, beta_0) lcb_score = beta_lcb(ns, nf, lambda_risk, alpha_0, beta_0) p_rel = posterior_probability_reliable(ns, nf, 0.70, alpha_0, beta_0) is_q = should_quarantine(ns, nf, 0.05, 0.70, alpha_0, beta_0) naive_score = sim composite_score = 0.0 if is_q else sim * lcb_score rows.append({ "Memory ID": m["id"], "Cosine Sim": sim, "Successes (ns)": ns, "Failures (nf)": nf, "Posterior Mean E[theta]": round(mu, 3), "Uncertainty (sigma)": round(std, 3), "LCB Score": round(lcb_score, 3), "Naive Retrieval Score": round(naive_score, 3), "Pessimistic LCB Composite": round(composite_score, 3), "Quarantine": "QUARANTINED" if is_q else "ACTIVE", "Strategy Directives": m["lesson"], }) df = pd.DataFrame(rows) # Sort primarily by proposed composite score descending df = df.sort_values(by="Pessimistic LCB Composite", ascending=False).reset_index(drop=True) df.insert(0, "LCB Rank", [f"#{i+1}" for i in range(len(df))]) return df # --- Load Real Benchmark Data for Explorer --- DATA_DIR = Path(__file__).parent / "data" def load_benchmark_summary() -> Tuple[pd.DataFrame, pd.DataFrame]: traces_path = DATA_DIR / "telemetry_traces.csv" mem_path = DATA_DIR / "memory_bank_experiences.csv" if traces_path.exists(): df_traces = pd.read_csv(traces_path) summary = ( df_traces.groupby("ablation_condition") .agg( Total_Steps=("step_index", "count"), Accuracy=("task_success", "mean"), Avg_Reward=("observed_reward", "mean"), Mean_Similarity=("cosine_similarity", "mean"), Quarantine_Triggers=("quarantine_triggered", "sum"), ) .reset_index() ) summary["Accuracy"] = summary["Accuracy"].apply(lambda v: f"{v * 100:.2f}%") summary["Avg_Reward"] = summary["Avg_Reward"].apply(lambda v: f"{v:.3f}") summary["Mean_Similarity"] = summary["Mean_Similarity"].apply(lambda v: f"{v:.3f}") else: summary = pd.DataFrame({"Notice": ["telemetry_traces.csv not found locally."]}) if mem_path.exists(): df_mem = pd.read_csv(mem_path) mem_preview = df_mem[[ "experience_id", "task_domain", "successes_count", "failures_count", "posterior_mean_trust", "posterior_variance", "pessimistic_lcb_score", "quarantine_status", "is_adversarial_sample" ]].head(25) else: mem_preview = pd.DataFrame({"Notice": ["memory_bank_experiences.csv not found locally."]}) return summary, mem_preview # --- Gradio Application Layout --- def build_app() -> gr.Blocks: theme = gr.themes.Soft( primary_hue="blue", neutral_hue="slate", ) with gr.Blocks(title="Adaptive Agent Memory Resilience Playground") as demo: gr.Markdown( """ # Adaptive Agent Memory Resilience Playground ### Mathematical Trust Dynamics & Pessimistic LCB Memory Retrieval for Autonomous LLM Agents **Author:** Sumit Das (`@sumitaidev`) | Research Paper Pre-print (2026) | [Benchmark Dataset](https://huggingface.co/datasets/sumitaidev/agent-memory-resilience-benchmark) This research tool allows AI researchers and engineers to interactively analyze how Bayesian conjugate updating, epistemic variance quantification, and pessimistic Lower Confidence Bound (LCB) composite retrieval prevent **Negative Transfer** and **Memory Poisoning** in agent systems (such as LangGraph, AutoGen, and CrewAI). """ ) with gr.Tabs(): # TAB 1: Bayesian Reliability & Quarantine Engine with gr.TabItem("1. Bayesian Reliability & Theorem 1"): gr.Markdown( """ ### Conjugate Beta-Bernoulli Posterior Update Engine Simulates task trials for an individual experiential memory record under weakly-informative prior $\\operatorname{Beta}(\\alpha_0=3.0, \\beta_0=1.0)$. Theorem 1 states that at operational threshold $\\theta=0.70$ and significance $\\gamma=0.05$, exactly $t^* = 4$ consecutive failures triggers deterministic quarantine. """ ) with gr.Row(): with gr.Column(scale=1): alpha_0_input = gr.Slider(0.5, 10.0, value=3.0, step=0.5, label="Prior Alpha (alpha_0)") beta_0_input = gr.Slider(0.5, 10.0, value=1.0, step=0.5, label="Prior Beta (beta_0)") successes_input = gr.Slider(0, 50, value=0, step=1, label="Observed Task Successes (n_s)") failures_input = gr.Slider(0, 20, value=0, step=1, label="Observed Task Failures (n_f)") lambda_input = gr.Slider(0.0, 3.0, value=1.0, step=0.2, label="Risk Aversion Parameter (lambda)") threshold_input = gr.Slider(0.5, 0.9, value=0.70, step=0.05, label="Admissibility Standard (threshold)") gamma_input = gr.Slider(0.01, 0.20, value=0.05, step=0.01, label="Quarantine Significance Level (gamma)") status_html = gr.HTML() with gr.Column(scale=2): with gr.Row(): mean_kpi = gr.Textbox(label="Posterior Mean E[theta]", interactive=False) sigma_kpi = gr.Textbox(label="Uncertainty (sigma)", interactive=False) lcb_kpi = gr.Textbox(label="LCB Score", interactive=False) prob_kpi = gr.Textbox(label="P(theta > threshold)", interactive=False) plot_output = gr.Plot(label="Posterior Reliability Distribution") gr.Markdown("#### Theorem 1 Deterministic Verification: Consecutive Failure Trajectory") theorem_table = gr.Dataframe( headers=["Consecutive Failures (t)", "Posterior Beta", "Posterior Mean E[theta]", "Uncertainty (sigma)", "P(theta > 0.70)", "Quarantine Status"], interactive=False, ) def update_tab1(a0, b0, ns, nf, lam, thresh, gam): fig, card, m_val, s_val, lcb_val, p_val = plot_beta_posterior( a0, b0, int(ns), int(nf), lam, thresh, gam ) th_df = compute_consecutive_failure_series(a0, b0, gam, thresh) return fig, card, m_val, s_val, lcb_val, p_val, th_df inputs_tab1 = [alpha_0_input, beta_0_input, successes_input, failures_input, lambda_input, threshold_input, gamma_input] outputs_tab1 = [plot_output, status_html, mean_kpi, sigma_kpi, lcb_kpi, prob_kpi, theorem_table] for comp in inputs_tab1: comp.change(fn=update_tab1, inputs=inputs_tab1, outputs=outputs_tab1) demo.load(fn=update_tab1, inputs=inputs_tab1, outputs=outputs_tab1) # TAB 2: Retrieval Ranking Playground with gr.TabItem("2. Pessimistic LCB Memory Retrieval"): gr.Markdown( """ ### Retrieval Competition: Naive Cosine Similarity vs Pessimistic LCB Demonstrates how standard vector RAG repeatedly retrieves corrupted memories with high semantic similarity, causing catastrophic cascade failures. Our composite formulation: $$\\operatorname{Score}(e; q) = \\operatorname{Sim}(\\mathbf{q}, \\mathbf{v}_e) \\times \\operatorname{LCB}_\\lambda(e)$$ penalizes uncertain reflections and quarantines verified failures. """ ) with gr.Row(): retrieval_lambda = gr.Slider(0.0, 3.0, value=1.0, step=0.25, label="Risk Aversion Parameter (lambda)") refresh_retrieval_btn = gr.Button("Recompute Rankings") retrieval_table = gr.Dataframe(interactive=False) gr.Markdown( """ **Key Analytical Observations**: - At $\\lambda = 0.0$ (Risk-neutral/Naive Cosine): `MEM-CORRUPT-01` ranks #1 because its cosine similarity is 0.94. - At $\\lambda \\ge 1.0$ (Pessimistic LCB): `MEM-CORRUPT-01` is flagged for quarantine, and `MEM-ROBUST-02` (proven with 22 successes) rightfully takes #1 rank. - `MEM-UNTESTED-03` with 0 executions receives an uncertainty penalty, preventing the agent from blindly over-trusting unvalidated reflections. """ ) def update_retrieval(lam): return evaluate_retrieval_ranking(lam) retrieval_lambda.change(fn=update_retrieval, inputs=[retrieval_lambda], outputs=[retrieval_table]) refresh_retrieval_btn.click(fn=update_retrieval, inputs=[retrieval_lambda], outputs=[retrieval_table]) demo.load(fn=update_retrieval, inputs=[retrieval_lambda], outputs=[retrieval_table]) # TAB 3: Benchmark Traces & Empirical Results with gr.TabItem("3. Benchmark Traces & Ablation"): gr.Markdown( """ ### Empirical Benchmark Telemetry Summary (1,200 Execution Steps) Comparative performance across four controlled experimental conditions under adversarial noise injection (steps 60 to 140). """ ) summary_df, preview_df = load_benchmark_summary() gr.Markdown("#### Ablation Conditions Summary") gr.Dataframe(value=summary_df, interactive=False) gr.Markdown("#### Experiential Memory Bank Sample (100 Verified Records)") gr.Dataframe(value=preview_df, interactive=False) gr.Markdown( """ ### Citation ```bibtex @article{das2026adaptive, title={Adaptive Agent Memory Resilience: Mitigating Negative Transfer and Memory Poisoning via Bayesian Trust Updating and Pessimistic Lower Confidence Bound Retrieval}, author={Das, Sumit}, journal={arXiv preprint}, year={2026} } ``` """ ) return demo if __name__ == "__main__": app = build_app() app.launch(theme=gr.themes.Soft(primary_hue="blue", neutral_hue="slate"))