GeniffyDocs
Changelog Log In Get a key

Google Drive

Keep each of your users' Google Drive in their memory. Your app already holds an access token for each user, from their Google sign-in with the drive.readonly scope; this sync adds their Docs, Slides and Sheets, and their PDF, Word, PowerPoint, Excel and text files, each under its own id with the label channel: google-drive. The first run reads everything. After that, Drive's own list of changes says what to fetch, so a run that finds nothing new downloads nothing. A changed file teaches only what changed, a file that is gone takes what it taught with it, and when the user disconnects Drive, one call forgets everything that came from it.

Install

Terminal
pip install httpx geniffy

Set GENIFFY_API_KEY from API keys in the Geniffy app.

The sync

drive_sync.py
import httpx
from geniffy import BadRequestError, Geniffy, NotFoundError

geniffy = Geniffy()                               # reads GENIFFY_API_KEY
LABELS = {"channel": "google-drive"}
FIELDS = "id, name, mimeType, size, trashed"
XLSX = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
EXPORT = {                                        # Google's own files, as Drive exports them: every tab of a Sheet
    "application/vnd.google-apps.document": ("text/plain", ".txt"),
    "application/vnd.google-apps.presentation": ("text/plain", ".txt"),
    "application/vnd.google-apps.spreadsheet": (XLSX, ".xlsx"),
}
READS = (".pdf", ".docx", ".pptx", ".xlsx", ".txt", ".md", ".csv", ".html")    # and these, as they are


def drive(token: str) -> httpx.Client:
    """One user's Drive, with the access token from their Google sign-in."""
    return httpx.Client(base_url="https://www.googleapis.com/drive/v3", timeout=60,
                        headers={"Authorization": f"Bearer {token}"})


def keep(mem, api: httpx.Client, file: dict) -> bool:
    """Add one file under its Drive id. False when it holds nothing Geniffy can read."""
    if file["mimeType"] in EXPORT:
        kind, ext = EXPORT[file["mimeType"]]
        got, name = api.get(f"/files/{file['id']}/export", params={"mimeType": kind}), file["name"] + ext
    elif file["name"].lower().endswith(READS) and int(file.get("size", 0)) <= 25 * 1024 * 1024:
        got, name = api.get(f"/files/{file['id']}", params={"alt": "media"}), file["name"]
    else:
        return False                              # a folder, an image, a video, or over 25 MB
    try:
        mem.memories.add_file(got.raise_for_status().content, filename=name, title=file["name"],
                              external_id=f"drive:{file['id']}", labels=LABELS)
    except BadRequestError:                       # empty, or a scan with no text in it
        return False
    return True


def forget(mem, file_id: str) -> None:
    try:
        mem.sources.delete(external_id=f"drive:{file_id}")
    except NotFoundError:                         # never added: a folder, an image
        pass


def read_all(mem, api: httpx.Client) -> None:
    """Everything in the user's Drive; and what memory holds from Drive that is no longer there goes."""
    seen, params = set(), {"q": "trashed = false", "pageSize": 100, "fields": f"nextPageToken, files({FIELDS})"}
    while True:
        out = api.get("/files", params=params).raise_for_status().json()
        seen |= {f"drive:{file['id']}" for file in out["files"] if keep(mem, api, file)}
        if "nextPageToken" not in out:
            break
        params["pageToken"] = out["nextPageToken"]
    mem.sources.delete_labelled(LABELS, keep=seen)


def sync(user_id: str, token: str, page_token: str | None = None) -> str:
    """Bring one user's Drive into their memory. Returns the token to pass next time: pass None the
    first time and everything is read; after that, only what changed since the last run is fetched."""
    mem = geniffy.space(f"user_{user_id}")
    with drive(token) as api:
        if page_token is None:
            start = api.get("/changes/startPageToken").raise_for_status().json()["startPageToken"]
            read_all(mem, api)
            return start
        params = {"pageToken": page_token, "pageSize": 100,
                  "fields": f"nextPageToken, newStartPageToken, changes(changeType, fileId, removed, file({FIELDS}))"}
        while True:
            out = api.get("/changes", params=params).raise_for_status().json()
            for change in out["changes"]:
                if change.get("changeType") == "drive":
                    continue                      # a shared drive itself changed, not a file in it
                file = change.get("file")
                if change.get("removed") or not file or file.get("trashed") or not keep(mem, api, file):
                    forget(mem, change["fileId"])
            if "newStartPageToken" in out:
                return out["newStartPageToken"]
            params["pageToken"] = out["nextPageToken"]


def disconnect(user_id: str) -> int:
    """The user disconnected Drive: everything that came from it goes. Returns how many files."""
    return geniffy.space(f"user_{user_id}").sources.delete_labelled(LABELS)

Run it on a schedule. Keep the token each run returns with the user, and pass it to the next run. Google's access tokens last an hour, so refresh the user's before each run, as your Google sign-in library does.

drive.readonly is one of Google's restricted scopes: before your app reads the Drive of people outside your own Google Workspace, Google's verification asks for a security assessment. With drive.file instead, which needs none, the same sync sees the files each user picks for your app, and the ones your app made.

How it behaves

  • Each file is one source, under its Drive id. Changed, only the paragraphs that changed are learned, and what was removed is taken back. See Your own ids.
  • Only what changed is fetched. After the first run, Drive's list of changes names the files to read again, and a run with nothing new downloads nothing.
  • A file that is gone goes from memory too, with what it taught: deleted, in the bin, or no longer shared with the user. Passing None reads everything again and puts memory right, if a token is ever lost.
  • What can't be read is left out: folders, images and video, files over 25 MB, and scans with no text in them. See Files.
  • Recall can keep to Drive, or leave it out: mem.context(question, labels={"channel": "google-drive"}). See Labels.
  • Disconnecting forgets it all in one call, and nothing the user added another way.
Last updated October 6, 2026