C++ core (libmemba.so): - include/memba/state.h — C API (state_new/free/save/load/get_size) - src/state.cpp — MEMB file format: magic, version, SHA-256 model_id, CRC-32, opaque llama_state_*_data() blob - src/cli.cpp — minimal demo binary with greedy sampler - CMakeLists.txt + build.sh with llama.cpp submodule, CUDA auto-detect Python SDK (memba): - core.py — file I/O via llama-cpp-python's exposed C functions, unwraps _LlamaContext to access raw context pointer (≥0.3.x) - session.py — high-level Session with auto-save/load, ChatML wrapper for instruct models, raw mode for base models - cli.py — typer-based: chat (REPL), run (one-shot), list, rm, info Examples: - 01_basic_save_load.py, 02_chat_session.py Experiments (throwaway POCs documenting product-direction findings): - recall_poc.py — git log → state → cross-process query - mood_poc.py — batch sentiment trajectory, Mamba vs Transformer - mood_stream_poc.py, mood_batch_poc.py — variants - diag_saveload.py — minimal save/load isolation test - README.md documents the headline finding: save/load is byte-identical, but Falcon-Mamba-7B-Instruct does not retain facts across conversation turns even in-process — limits viable products to single-prompt analysis and persona priming. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
92 lines
3.5 KiB
Python
92 lines
3.5 KiB
Python
"""
|
|
mood_stream_poc.py — streaming sentiment, then save/load across processes.
|
|
|
|
This is the REAL product test:
|
|
1. Open memba session
|
|
2. Feed 15 chat messages ONE AT A TIME as "observations"
|
|
3. Save state, exit process
|
|
4. In a fresh process: load state, query mood
|
|
"""
|
|
from __future__ import annotations
|
|
import sys, argparse, time
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "python"))
|
|
from memba import Session
|
|
|
|
MAMBA = "/home/emil/Desktop/Coding/AI/Memba/falcon-mamba-7B-instruct-Q4_K_M.gguf"
|
|
STATE_DIR = "/tmp/mood_stream_test"
|
|
SESSION = "mood_stream"
|
|
|
|
CHAT_LOG = [
|
|
"Morning team! Coffee in hand, ready to tackle the auth refactor today.",
|
|
"Just pushed PR #234 fixing the token validation bug. Should be a quick merge.",
|
|
"Code review comments came in fast, all good catches. Iterating now.",
|
|
"Basic flow working locally, tests passing. Feeling good about this.",
|
|
"Heading to lunch, hopefully wrap this up by EOD.",
|
|
"Back. CI is failing on something unrelated, looking into it.",
|
|
"OK the 'unrelated' thing is actually related. Auth tests use a stale fixture.",
|
|
"Why does the fixture rebuild take 12 minutes. Every. Single. Time.",
|
|
"Cancelled the run twice now. Going to bypass and run tests locally.",
|
|
"Local passes, CI fails. Classic.",
|
|
"Two hours gone on this fixture issue. Not even what I was supposed to be doing.",
|
|
"Now there's a merge conflict with main because someone restructured migrations.",
|
|
"Whoever shipped those migrations on a Friday afternoon, I will find you.",
|
|
"Closing the laptop. Will fight this tomorrow.",
|
|
"Actually no. One more try before I sleep.",
|
|
]
|
|
|
|
QUESTIONS = [
|
|
"Briefly: what is this person's current emotional state? One sentence.",
|
|
"Has their mood changed during this monitoring session? One sentence.",
|
|
"Roughly when did they start having a hard time?",
|
|
]
|
|
|
|
|
|
def build():
|
|
state_path = Path(STATE_DIR) / f"{SESSION}.memb"
|
|
if state_path.exists():
|
|
state_path.unlink()
|
|
print(f"[build] cleared previous state")
|
|
|
|
s = Session(
|
|
model_path=MAMBA, session_id=SESSION, state_dir=STATE_DIR,
|
|
n_gpu_layers=-1, n_ctx=4096, chat_format="chatml",
|
|
)
|
|
print(f"[build] init state {s.state_size:,} B")
|
|
|
|
# Stream-feed: each message wrapped as if WE'RE TELLING the model
|
|
# "here's a new message you're observing"
|
|
for i, msg in enumerate(CHAT_LOG, 1):
|
|
observation = f"You are silently observing one person's chat messages. New message just arrived:\n[msg {i}] {msg}\nReply with just 'noted'."
|
|
ack = s.chat(observation, max_tokens=4)
|
|
print(f"[build] msg {i:>2}: {msg[:50]:<50} → ack={ack!r}")
|
|
|
|
print(f"[build] state after streaming: {s.state_size:,} B")
|
|
s.save()
|
|
print(f"[build] saved → {state_path}")
|
|
|
|
|
|
def query():
|
|
state_path = Path(STATE_DIR) / f"{SESSION}.memb"
|
|
if not state_path.exists():
|
|
print("[query] no state — run build first"); return 1
|
|
print(f"[query] loading state {state_path.stat().st_size:,} B")
|
|
s = Session(
|
|
model_path=MAMBA, session_id=SESSION, state_dir=STATE_DIR,
|
|
n_gpu_layers=-1, n_ctx=4096, chat_format="chatml",
|
|
)
|
|
for i, q in enumerate(QUESTIONS, 1):
|
|
print(f"\n[Q{i}] {q}")
|
|
print(f"[A{i}] {s.chat(q, max_tokens=120)}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
p = argparse.ArgumentParser()
|
|
p.add_argument("cmd", choices=["build", "query"])
|
|
args = p.parse_args()
|
|
if args.cmd == "build":
|
|
build()
|
|
else:
|
|
query()
|