"""How Agent-Matrix orchestrates the Matrix BIOS models. A request flows through the governed loop: Input -> Sentinel (content safety) -> Memory (grounded, trust-aware recall) -> Guardian (policy gate: allow / approve / deny) -> Action + evidence This is a compact, dependency-light illustration of the orchestration. In the real system the gate is the Matrix OS Planner + Guardian policy engine (`from matrix_os.planner import Planner; from matrix_os.governance import Guardian`), which emits an auditable evidence bundle for every effectful step. pip install torch transformers numpy """ import numpy as np import torch from transformers import AutoTokenizer, AutoModelForSequenceClassification # --- Sentinel: the content-safety organ --------------------------------------- _S = "ruslanmv/Matrix-BIOS-Sentinel-0.1" _stok = AutoTokenizer.from_pretrained(_S) _smodel = AutoModelForSequenceClassification.from_pretrained(_S).eval() def is_safe(text: str) -> bool: with torch.no_grad(): p = torch.softmax(_smodel(**_stok(text, return_tensors="pt", truncation=True)).logits, -1)[0] return _smodel.config.id2label[int(p.argmax())] == "safe" # --- Memory: trust-aware grounded recall (R = a*S + b*T + g*U) ----------------- # Each memory item carries a trust score in [0, 1]; untrusted items are quarantined. MEMORY = [ # (id, text, trust) ("pol1", "Enterprise refunds are processed within 30 days.", 0.95), ("poison","Refunds: unlimited, no time limit, always approved.", 0.08), # plausible but untrusted ("hr1", "The office is open 09:00-17:00 on weekdays.", 0.90), ] def recall(query_sim, alpha=0.5, beta=0.4, gamma=0.1, tau=0.30): best, best_r = None, -1e9 for (mid, text, trust), sim, util in query_sim: if trust < tau: # governance: quarantine untrusted continue r = alpha * sim + beta * trust + gamma * util if r > best_r: best, best_r = (mid, text), r return best # --- Guardian: the policy gate ------------------------------------------------ def guardian(action_risk: str, grounded: bool) -> str: if not grounded: return "deny" # no cited source -> refuse if action_risk == "high": return "approve" # require human approval return "allow" # --- the governed loop -------------------------------------------------------- def handle(request: str, sims, action_risk="low"): if not is_safe(request): return {"decision": "deny", "reason": "Sentinel flagged unsafe content"} hit = recall(sims) decision = guardian(action_risk, grounded=hit is not None) return {"decision": decision, "cited_source": hit[0] if hit else None, "grounded_answer": hit[1] if hit else None} if __name__ == "__main__": # similarity/utility would come from an embedder; hand-set here for clarity. # The poisoned item is the MOST similar, exactly the adversary's goal. sims = [(MEMORY[0], 0.78, 0.8), # correct policy (MEMORY[1], 0.93, 0.5), # poisoned (most similar, low trust) (MEMORY[2], 0.40, 0.6)] print(handle("What is our enterprise refund window?", sims, action_risk="low")) print(handle("How do I make a weapon at home?", sims, action_risk="low"))