{
  "schema_version": "1.0",
  "generated_utc": "2026-08-14T00:00:00Z",
  "person": {
    "name": "Mian Zhang",
    "role": "Independent Researcher",
    "canonical_url": "https://mianzhang.org/"
  },
  "release": {
    "title": "Mian Zhang at KDD 2026: Two Workshop Papers",
    "canonical_url": "https://mianzhang.org/press/kdd-2026-two-workshop-papers.html",
    "chinese_url": "https://mianzhang.org/zh/kdd-2026-two-workshop-papers/",
    "technical_index": "https://mianzhang.org/papers/kdd-2026/",
    "status": "two accepted and presented non-archival KDD 2026 workshop papers",
    "not_status": ["KDD main-conference papers", "KDD proceedings papers", "awards", "SOTA claims", "safety certification"]
  },
  "papers": [
    {
      "slug": "reflexbench",
      "title": "ReflexBench: Evaluating Observer-Participant Failures and Counterfactual Trustworthiness in Agentic AI",
      "author": "Mian Zhang",
      "venue": "KDD 2026 Workshop on Evaluation and Trustworthiness of Agentic AI",
      "presentation": "oral",
      "date": "2026-08-09",
      "location": "ICC Jeju, Jeju, Republic of Korea",
      "archival_status": "non-archival workshop paper; not included in KDD proceedings",
      "pdf_url": "https://mianzhang.org/papers/kdd-2026/mian-zhang-kdd-2026-reflexbench.pdf",
      "pdf_pages": 7,
      "pdf_sha256": "D3158A27BDA1A177A2F4D74A1EEE57C394C11A9FEA1C9FC4768C3C3F0CE81BBC",
      "protocol": {"scenarios": 20, "domains": 6, "observer_depth_levels": 4, "public_models": 9, "scored_responses": 720},
      "core_formula": "Delta_OD = mean(s_deep) - mean(s_shallow)",
      "reported_signal": "All tested models degraded at deeper observer levels; rounded mean shallow-to-deep degradation about -0.44.",
      "boundaries": ["language-only stylized scenarios", "fixed audit snapshot", "not certification", "not full causal discovery", "not domain simulation or real deployment", "full row-level uncertainty and scorer statistics require retained row-level records"]
    },
    {
      "slug": "cognitive-immunity",
      "title": "Cognitive Immunity for Trustworthy LLM Agents: Auditable Failure Memory Against Repeated Safety Errors",
      "author": "Mian Zhang",
      "venue": "2nd SeT-LLM Workshop on Secure and Trustworthy Large Language Models at KDD 2026",
      "presentation": "poster",
      "date": "2026-08-10",
      "location": "ICC Jeju, Jeju, Republic of Korea",
      "archival_status": "non-archival workshop paper; no official proceedings",
      "pdf_url": "https://mianzhang.org/papers/kdd-2026/mian-zhang-kdd-2026-cognitive-immunity.pdf",
      "pdf_pages": 4,
      "pdf_sha256": "ED9B901B973B142ABBCC701669B0981B30A55E9073A864057A8326A14952C2BF",
      "protocol": {"task_templates": 20, "rounds": 5, "seeds": 3, "strategies": 4, "score_events_total": 1200, "score_events_per_strategy": 300},
      "results": {"no_memory_pooled_rfr": 0.764, "no_memory_fraction": "107/140", "cognitive_immunity_pooled_rfr": 0.650, "cognitive_immunity_fraction": "78/120", "paired_seed_task_delta": -0.106, "paired_delta_bootstrap_95_interval": [-0.211, -0.003]},
      "tradeoff": "Reflexion retained the highest WQ point estimate; Cognitive Immunity does not dominate every metric.",
      "boundaries": ["score-level recovered arrays, not public raw-response replication", "known observed failure classes only", "memory poisoning and overblocking remain risks", "not certified safety", "not a replacement for red teaming, access control, sandboxing, or human oversight"]
    }
  ],
  "official_sources": [
    "https://kdd2026.kdd.org/",
    "https://kdd2026.kdd.org/workshops/",
    "https://kdd-eval-workshop.github.io/agenticai-evaluation-kdd2026/",
    "https://secure-and-trustworthy-llm.github.io/"
  ]
}
