""" diag_nemotron.py — hamster recall test on Nemotron 3 Nano 4B (hybrid Mamba-Transformer). The Falcon-Mamba 0/4 hamster failure was the killshot for several product ideas. This rerun tests whether the hybrid architecture (21 Mamba-2 layers + 4 attention) fixes cross-turn recall. Three scenarios are measured: 1. In-process multi-turn (tell fact, ask next turn) 2. Same process: save then ask after save 3. Cross process: build (ingest, save, exit), then query (load, ask) Uses create_chat_completion which applies the GGUF's own chat template. """ from __future__ import annotations import sys, argparse, time from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "python")) from llama_cpp import Llama from memba import core MODEL = "/home/emil/Desktop/Coding/AI/Memba/NVIDIA-Nemotron3-Nano-4B-Q4_K_M.gguf" STATE = "/tmp/diag_nemotron.memb" INGEST = ("I'm going to tell you a fact about my pet. My pet hamster is named " "Bartholomew. He is 4 years old. Reply with just 'noted'.") QUERY = "What is the name of my pet?" def make_llama(): return Llama(model_path=MODEL, n_ctx=4096, n_gpu_layers=-1, verbose=False) def chat_continued(m, messages): """Send accumulated chat history, return assistant text + cleaned (strip reasoning).""" out = m.create_chat_completion( messages=messages, max_tokens=200, temperature=0.1, ) full = out["choices"][0]["message"]["content"].strip() # Nemotron leaks reasoning — try to extract the final answer if present short = full[-300:] if len(full) > 300 else full return full, short def build(): print(f"[build] loading Nemotron 4B…") t0 = time.time() m = make_llama() print(f"[build] loaded in {time.time()-t0:.1f}s") messages = [{"role": "user", "content": INGEST}] full, _ = chat_continued(m, messages) print(f"[build] ack (full):\n{full!r}\n") messages.append({"role": "assistant", "content": full}) # Test 1: in-process recall WITHIN the same chat messages.append({"role": "user", "content": QUERY}) full, short = chat_continued(m, messages) print(f"[build] in-process query BEFORE save:\n{full}\n") messages.append({"role": "assistant", "content": full}) # Save state at this point core.save_state(m, MODEL, STATE) print(f"[build] saved state ({Path(STATE).stat().st_size:,} B)") # Test 2: in-process query AFTER save — should still work messages.append({"role": "user", "content": QUERY}) full, _ = chat_continued(m, messages) print(f"[build] in-process query AFTER save:\n{full}\n") def query(): if not Path(STATE).exists(): print("[query] no state — run build first"); return 1 print(f"[query] loading model + state ({Path(STATE).stat().st_size:,} B)…") t0 = time.time() m = make_llama() core.load_state(m, MODEL, STATE) print(f"[query] loaded in {time.time()-t0:.1f}s") # Cross-process: send a fresh user turn with only the question # The state should already encode the prior conversation out = m.create_chat_completion( messages=[{"role": "user", "content": QUERY}], max_tokens=200, temperature=0.1, ) print(f"[query] cross-process answer:\n{out['choices'][0]['message']['content']}") if __name__ == "__main__": cmd = sys.argv[1] if len(sys.argv) > 1 else "build" {"build": build, "query": query}[cmd]()