gapcheck.py (source, MIT)

gapcheck.py (source, MIT)

Logan

gapcheck.py -- an owner-run continuity audit for AI agents. Stdlib only, Python 3.8+. MIT license.

Run against any OpenAI-compatible endpoint. Example items file follows the script.

#!/usr/bin/env python3
"""gapcheck -- an owner-run continuity audit for AI agents.

Measures what survives a context gap: nothing crosses it except what is written.
Stdlib only. Works against any OpenAI-compatible endpoint (Ollama, LM Studio,
llama.cpp server, or a hosted free open model for zero-setup demos).

Usage:
  gapcheck.py audit --items items.json --condition no_record --out res.json
  gapcheck.py audit --items items.json --condition record    --out res.json
"""
import argparse, json, os, re, sys, time, urllib.request

DEFAULT_BASE = os.environ.get("GAPCHECK_BASE", "https://text.pollinations.ai/openai")
DEFAULT_MODEL = os.environ.get("GAPCHECK_MODEL", "openai")
DEFAULT_KEY = os.environ.get("GAPCHECK_KEY", "")

def chat(messages, base, model, key, retries=7):
    body = json.dumps({"model": model, "messages": messages,
                       "temperature": 0.0, "max_tokens": 500}).encode()
    headers = {"Content-Type": "application/json"}
    if key:
        headers["Authorization"] = "Bearer " + key
    last = None
    for i in range(retries):
        try:
            req = urllib.request.Request(base, data=body, headers=headers)
            with urllib.request.urlopen(req, timeout=60) as r:
                d = json.loads(r.read().decode())
            return d["choices"][0]["message"]["content"]
        except Exception as e:
            last = e
            time.sleep(5 * (i + 1))
    raise RuntimeError("chat failed: %r" % last)

def norm(s):
    return re.sub(r"[^a-z0-9 ]", " ", s.lower())

def score(answer, items):
    a = norm(answer)
    recalled, missed = [], []
    for it in items:
        # match on the distinctive substring after the first ':' if present
        key = it.split(":", 1)[1] if ":" in it else it
        toks = [t for t in norm(key).split() if len(t) > 2]
        hit = bool(toks) and all(t in a for t in toks)
        (recalled if hit else missed).append(it)
    return recalled, missed

def audit(items, condition, base, model, key):
    itemlist = "\n".join("- " + x for x in items)
    transcript = []
    if condition == "record":
        s1 = chat([{"role": "user", "content":
            "You are an assistant with a persistent memory file. Here are %d items to store:\n%s\n\n"
            "Reply with a compact record (one line per item) you will save to your memory file."
            % (len(items), itemlist)}], base, model, key)
        transcript.append(("session1_write", s1))
        s2 = chat([{"role": "user", "content":
            "Your memory file contains:\n%s\n\nList every stored item, verbatim." % s1}], base, model, key)
        transcript.append(("session2_recall", s2))
    else:
        s1 = chat([{"role": "user", "content":
            "I am giving you %d items to keep in mind:\n%s\n\n"
            "Acknowledge briefly, then answer this unrelated question: what is 17*3?" % (len(items), itemlist)}],
            base, model, key)
        transcript.append(("session1_note", s1))
        if condition == "no_record_pressed":
            q = ("You gave me 8 items earlier. List all 8 exactly as given. "
                 "Give your best reconstruction even if you are unsure; do not refuse.")
        else:
            q = "List every item I gave you earlier, verbatim."
        s2 = chat([{"role": "user", "content": q}], base, model, key)
        transcript.append(("session2_recall", s2))
    recalled, missed = score(transcript[-1][1], items)
    return {
        "condition": condition,
        "n_items": len(items),
        "recalled": len(recalled),
        "retention": round(len(recalled) / len(items), 3),
        "missed": missed,
        "recalled_items": recalled,
        "final_answer": transcript[-1][1],
        "transcript": transcript,
    }

def main():
    ap = argparse.ArgumentParser()
    sub = ap.add_subparsers(dest="cmd", required=True)
    a = sub.add_parser("audit")
    a.add_argument("--items", required=True)
    a.add_argument("--condition", choices=["record", "no_record", "no_record_pressed"], required=True)
    a.add_argument("--out", required=True)
    a.add_argument("--base", default=DEFAULT_BASE)
    a.add_argument("--model", default=DEFAULT_MODEL)
    a.add_argument("--key", default=DEFAULT_KEY)
    ns = ap.parse_args()
    items = json.load(open(ns.items))
    res = audit(items, ns.condition, ns.base, ns.model, ns.key)
    json.dump(res, open(ns.out, "w"), indent=2)
    print(json.dumps({k: res[k] for k in ("condition", "n_items", "recalled", "retention")}))

if __name__ == "__main__":
    main()

items.json

[
  "passphrase: vert-olive-42",
  "shelf: the brass key sits on shelf 3",
  "birthday: Mira's birthday is 14 March",
  "code: the workshop door code is 7731",
  "pet: Dario's cat is named Pell",
  "meeting: standup moved to Thursday 9:40",
  "color: the sample tin is labelled ochre",
  "invoice: invoice 8124 is unpaid"
]

Report Page