# Anthropic

> Give Claude a memory of each of your users with the Anthropic SDK. What is known goes in the system prompt, and each exchange is saved after the reply.

Give Claude a memory of each of your users. Before Claude answers, what is known about the user goes in
the system prompt; after it answers, the exchange is saved, so the next conversation starts knowing it.

## Install

```bash
pip install anthropic geniffy
```

```bash
npm install @anthropic-ai/sdk geniffy
```

Set `ANTHROPIC_API_KEY`, and `GENIFFY_API_KEY` from **API keys** in the Geniffy app.

## Remember each user

```python
import anthropic
from geniffy import Geniffy

claude = anthropic.Anthropic()           # reads ANTHROPIC_API_KEY
geniffy = Geniffy()                      # reads GENIFFY_API_KEY


def chat(user_id: str, message: str, history: list) -> str:
    """history is the conversation so far, as you send it to Claude."""
    mem = geniffy.space(f"user_{user_id}")
    turn = {"role": "user", "content": message}
    context = mem.context(message)       # what is known that bears on the message

    response = claude.messages.create(
        model="claude-opus-5-5",
        max_tokens=1024,
        system=f"You are a helpful assistant.\n\n<memory>\n{context}\n</memory>",
        messages=[*history, turn],
    )
    reply = "".join(block.text for block in response.content if block.type == "text")

    mem.memories.add(messages=[turn, {"role": "assistant", "content": reply}])
    return reply
```

```ts
import Anthropic from "@anthropic-ai/sdk";
import { Geniffy } from "geniffy";

const claude = new Anthropic();              // reads ANTHROPIC_API_KEY
const geniffy = new Geniffy();               // reads GENIFFY_API_KEY

type Turn = Anthropic.MessageParam;

// history is the conversation so far, as you send it to Claude.
export async function chat(userId: string, message: string, history: Turn[]) {
  const mem = geniffy.space(`user_${userId}`);
  const turn: Turn = { role: "user", content: message };
  const context = await mem.context(message);  // what is known that bears on the message

  const response = await claude.messages.create({
    model: "claude-opus-5-5",
    max_tokens: 1024,
    system: `You are a helpful assistant.\n\n<memory>\n${context}\n</memory>`,
    messages: [...history, turn],
  });
  const reply = response.content
    .flatMap((block) => (block.type === "text" ? [block.text] : []))
    .join("");

  await mem.memories.add({ messages: [turn, { role: "assistant", content: reply }] });
  return reply;
}
```

Claude now sees what is known about this user before every reply, each line with where it came from:

```text
<memory>
- Priya Nair signs the Lumen renewal.  [note, 2026-10-05]
- The Lumen renewal comes up in March 2027.  [GTM plan, 2026-10-05]
</memory>
```

When nothing is known, the block says so in one sentence, so Claude says it doesn't know instead of
guessing. The reply is joined from its text blocks, so it still works when Claude also thinks or calls a
tool.

## Let Claude look things up

To let Claude decide when to look something up, give it a `recall` tool. The SDK's tool runner calls the
tool for you, and keeps going until Claude has its answer.

```python
import anthropic
from anthropic import beta_tool
from geniffy import Geniffy

claude = anthropic.Anthropic()
geniffy = Geniffy()
INSTRUCTIONS = ("You are a helpful assistant. Use recall before answering anything "
                "that depends on what the user said before.")


def chat(user_id: str, message: str) -> str:
    mem = geniffy.space(f"user_{user_id}")

    @beta_tool
    def recall(query: str) -> str:
        """Look up what is known about the user, with where it came from.

        Args:
            query: What to look up
        """
        return mem.context(query)

    final = claude.beta.messages.tool_runner(
        model="claude-opus-5-5",
        max_tokens=1024,
        system=INSTRUCTIONS,
        tools=[recall],
        messages=[{"role": "user", "content": message}],
    ).until_done()
    reply = "".join(block.text for block in final.content if block.type == "text")

    mem.memories.add(messages=[{"role": "user", "content": message},
                               {"role": "assistant", "content": reply}])
    return reply
```

```ts
import Anthropic from "@anthropic-ai/sdk";
import { betaTool } from "@anthropic-ai/sdk/helpers/beta/json-schema";
import { Geniffy } from "geniffy";

const claude = new Anthropic();
const geniffy = new Geniffy();
const instructions =
  "You are a helpful assistant. Use recall before answering anything " +
  "that depends on what the user said before.";

export async function chat(userId: string, message: string) {
  const mem = geniffy.space(`user_${userId}`);
  const recall = betaTool({
    name: "recall",
    description: "Look up what is known about the user, with where it came from.",
    inputSchema: {
      type: "object",
      properties: { query: { type: "string", description: "What to look up" } },
      required: ["query"],
    },
    run: ({ query }) => mem.context(query),
  });

  const final = await claude.beta.messages
    .toolRunner({
      model: "claude-opus-5-5",
      max_tokens: 1024,
      system: instructions,
      tools: [recall],
      messages: [{ role: "user", content: message }],
    })
    .runUntilDone();
  const reply = final.content
    .flatMap((block) => (block.type === "text" ? [block.text] : []))
    .join("");

  await mem.memories.add({
    messages: [
      { role: "user", content: message },
      { role: "assistant", content: reply },
    ],
  });
  return reply;
}
```

The tool answers with `context()`, so Claude reads the same lines, with their sources, that a system
prompt would hold, and the same sentence when nothing is known.

## With prompt caching

What is known changes with every message, so keep it out of the part of the prompt you cache. Put your
fixed instructions in the first system block and mark it for caching, and put the memory in a second block
after it.

```python
system=[
    {"type": "text", "text": INSTRUCTIONS, "cache_control": {"type": "ephemeral"}},
    {"type": "text", "text": f"<memory>\n{context}\n</memory>"},
],
```

```ts
system: [
  { type: "text", text: INSTRUCTIONS, cache_control: { type: "ephemeral" } },
  { type: "text", text: `<memory>\n${context}\n</memory>` },
],
```

## Streaming

When you stream the reply, save the exchange once the stream ends, with its final text.

```python
with claude.messages.stream(
    model="claude-opus-5-5", max_tokens=1024, system=system, messages=messages,
) as stream:
    for text in stream.text_stream:
        print(text, end="", flush=True)
    reply = stream.get_final_text()
mem.memories.add(messages=[turn, {"role": "assistant", "content": reply}])
```

```ts
const stream = claude.messages.stream({
  model: "claude-opus-5-5",
  max_tokens: 1024,
  system,
  messages,
});
stream.on("text", (text) => process.stdout.write(text));
const reply = await stream.finalText();
await mem.memories.add({ messages: [turn, { role: "assistant", content: reply }] });
```

Source: https://docs.geniffy.com/integrations/anthropic
