Google Drive
Keep each of your users' Google Drive in their memory. Your app already holds an access token for each user,
from their Google sign-in with the drive.readonly scope; this sync adds their Docs, Slides and Sheets, and
their PDF, Word, PowerPoint, Excel and text files, each under its own id with the label
channel: google-drive. The first run reads everything. After that, Drive's own list of changes says what to
fetch, so a run that finds nothing new downloads nothing. A changed file teaches only what changed, a file
that is gone takes what it taught with it, and when the user disconnects Drive, one call forgets everything
that came from it.
Install
pip install httpx geniffyuv add httpx geniffySet GENIFFY_API_KEY from API keys in the Geniffy app.
The sync
import httpx
from geniffy import BadRequestError, Geniffy, NotFoundError
geniffy = Geniffy() # reads GENIFFY_API_KEY
LABELS = {"channel": "google-drive"}
FIELDS = "id, name, mimeType, size, trashed"
XLSX = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
EXPORT = { # Google's own files, as Drive exports them: every tab of a Sheet
"application/vnd.google-apps.document": ("text/plain", ".txt"),
"application/vnd.google-apps.presentation": ("text/plain", ".txt"),
"application/vnd.google-apps.spreadsheet": (XLSX, ".xlsx"),
}
READS = (".pdf", ".docx", ".pptx", ".xlsx", ".txt", ".md", ".csv", ".html") # and these, as they are
def drive(token: str) -> httpx.Client:
"""One user's Drive, with the access token from their Google sign-in."""
return httpx.Client(base_url="https://www.googleapis.com/drive/v3", timeout=60,
headers={"Authorization": f"Bearer {token}"})
def keep(mem, api: httpx.Client, file: dict) -> bool:
"""Add one file under its Drive id. False when it holds nothing Geniffy can read."""
if file["mimeType"] in EXPORT:
kind, ext = EXPORT[file["mimeType"]]
got, name = api.get(f"/files/{file['id']}/export", params={"mimeType": kind}), file["name"] + ext
elif file["name"].lower().endswith(READS) and int(file.get("size", 0)) <= 25 * 1024 * 1024:
got, name = api.get(f"/files/{file['id']}", params={"alt": "media"}), file["name"]
else:
return False # a folder, an image, a video, or over 25 MB
try:
mem.memories.add_file(got.raise_for_status().content, filename=name, title=file["name"],
external_id=f"drive:{file['id']}", labels=LABELS)
except BadRequestError: # empty, or a scan with no text in it
return False
return True
def forget(mem, file_id: str) -> None:
try:
mem.sources.delete(external_id=f"drive:{file_id}")
except NotFoundError: # never added: a folder, an image
pass
def read_all(mem, api: httpx.Client) -> None:
"""Everything in the user's Drive; and what memory holds from Drive that is no longer there goes."""
seen, params = set(), {"q": "trashed = false", "pageSize": 100, "fields": f"nextPageToken, files({FIELDS})"}
while True:
out = api.get("/files", params=params).raise_for_status().json()
seen |= {f"drive:{file['id']}" for file in out["files"] if keep(mem, api, file)}
if "nextPageToken" not in out:
break
params["pageToken"] = out["nextPageToken"]
mem.sources.delete_labelled(LABELS, keep=seen)
def sync(user_id: str, token: str, page_token: str | None = None) -> str:
"""Bring one user's Drive into their memory. Returns the token to pass next time: pass None the
first time and everything is read; after that, only what changed since the last run is fetched."""
mem = geniffy.space(f"user_{user_id}")
with drive(token) as api:
if page_token is None:
start = api.get("/changes/startPageToken").raise_for_status().json()["startPageToken"]
read_all(mem, api)
return start
params = {"pageToken": page_token, "pageSize": 100,
"fields": f"nextPageToken, newStartPageToken, changes(changeType, fileId, removed, file({FIELDS}))"}
while True:
out = api.get("/changes", params=params).raise_for_status().json()
for change in out["changes"]:
if change.get("changeType") == "drive":
continue # a shared drive itself changed, not a file in it
file = change.get("file")
if change.get("removed") or not file or file.get("trashed") or not keep(mem, api, file):
forget(mem, change["fileId"])
if "newStartPageToken" in out:
return out["newStartPageToken"]
params["pageToken"] = out["nextPageToken"]
def disconnect(user_id: str) -> int:
"""The user disconnected Drive: everything that came from it goes. Returns how many files."""
return geniffy.space(f"user_{user_id}").sources.delete_labelled(LABELS)Run it on a schedule. Keep the token each run returns with the user, and pass it to the next run. Google's access tokens last an hour, so refresh the user's before each run, as your Google sign-in library does.
drive.readonly is one of Google's restricted scopes: before your app reads the Drive of people outside your
own Google Workspace, Google's verification asks for a security assessment. With drive.file instead, which
needs none, the same sync sees the files each user picks for your app, and the ones your app made.
How it behaves
- Each file is one source, under its Drive id. Changed, only the paragraphs that changed are learned, and what was removed is taken back. See Your own ids.
- Only what changed is fetched. After the first run, Drive's list of changes names the files to read again, and a run with nothing new downloads nothing.
- A file that is gone goes from memory too, with what it taught: deleted, in the bin, or no longer shared with the user. Passing None reads everything again and puts memory right, if a token is ever lost.
- What can't be read is left out: folders, images and video, files over 25 MB, and scans with no text in them. See Files.
- Recall can keep to Drive, or leave it out:
mem.context(question, labels={"channel": "google-drive"}). See Labels. - Disconnecting forgets it all in one call, and nothing the user added another way.