diff --git a/docs/docs.json b/docs/docs.json
index 27f6b633..b5898c6f 100644
--- a/docs/docs.json
+++ b/docs/docs.json
@@ -104,8 +104,10 @@
"group": "Tutorials",
"pages": [
"v3/guides/discord",
+ "v3/guides/granola",
"v3/guides/telegram",
- "v3/guides/integrations/reachy-mini"
+ "v3/guides/integrations/reachy-mini",
+ "v3/guides/gmail"
]
},
{
@@ -302,13 +304,12 @@
"pages": [
"v2/integrations/crewai",
"v2/integrations/langgraph",
- "v2/integrations/mcp",
- "v2/integrations/n8n"
+ "v2/integrations/mcp"
]
},
{
"group": "Application Interfaces",
- "pages": ["v2/guides/discord", "v2/guides/telegram"]
+ "pages": ["v2/guides/discord", "v2/guides/n8n", "v2/guides/telegram"]
}
]
},
diff --git a/docs/v2/integrations/n8n.mdx b/docs/v2/guides/n8n.mdx
similarity index 100%
rename from docs/v2/integrations/n8n.mdx
rename to docs/v2/guides/n8n.mdx
diff --git a/docs/v3/guides/gmail.mdx b/docs/v3/guides/gmail.mdx
new file mode 100644
index 00000000..a1963652
--- /dev/null
+++ b/docs/v3/guides/gmail.mdx
@@ -0,0 +1,631 @@
+---
+title: "Gmail"
+icon: 'envelope'
+description: "Load Gmail threads into Honcho to give your AI agents memory of email conversations."
+sidebarTitle: 'Gmail'
+---
+
+In this tutorial, we'll walk through how to ingest your Gmail emails into Honcho. By the end, each email thread will be a Honcho session and each participant will be a peer — giving your agents memory of who said what across your email history.
+
+This guide includes a ready-to-run Python script that handles everything: Gmail OAuth, thread fetching, participant extraction, and Honcho ingestion. You can run it as-is or use the full tutorial below to understand each piece as you go.
+
+
+The full script is available on [GitHub](https://github.com/plastic-labs/honcho/tree/main/examples/gmail). This is a developer-focused tutorial — it requires creating a Google Cloud project and OAuth credentials.
+
+
+## TL;DR
+
+If you just want to get your emails into Honcho, here's everything you need.
+
+### 1. Set Up Google Cloud Credentials
+
+Follow Google's official [Gmail API Python Quickstart](https://developers.google.com/gmail/api/quickstart/python) to:
+
+1. Create a Google Cloud project and enable the Gmail API
+2. Configure the OAuth consent screen
+3. Create OAuth credentials (select **Desktop app** as the application type)
+4. Download the credentials JSON into the same directory as the script
+
+The script auto-detects Google's default `client_secret_*.json` filename, so no renaming needed. The script only needs the `gmail.readonly` scope.
+
+### 2. Install Dependencies
+
+
+```bash uv
+uv pip install google-api-python-client google-auth-oauthlib honcho-ai
+```
+
+```bash pip
+pip install google-api-python-client google-auth-oauthlib honcho-ai
+```
+
+
+### 3. Preview with a Dry Run
+
+
+```bash uv
+uv run honcho_gmail.py --dry-run --max-threads 5
+```
+
+```bash python
+python honcho_gmail.py --dry-run --max-threads 5
+```
+
+
+On first run, a browser window opens for OAuth consent. After authorizing, a `token.json` file is created — future runs skip this step.
+
+### 4. Load into Honcho
+
+
+```bash uv
+export HONCHO_API_KEY=your_api_key
+uv run honcho_gmail.py --workspace gmail-inbox --max-threads 20
+```
+
+```bash python
+export HONCHO_API_KEY=your_api_key
+python honcho_gmail.py --workspace gmail-inbox --max-threads 20
+```
+
+
+You can filter threads with Gmail search syntax:
+
+
+```bash uv
+uv run honcho_gmail.py --query "from:alice@example.com"
+uv run honcho_gmail.py --label INBOX
+uv run honcho_gmail.py --query "after:2024/01/01 has:attachment" --max-threads 50
+```
+
+```bash python
+python honcho_gmail.py --query "from:alice@example.com"
+python honcho_gmail.py --label INBOX
+python honcho_gmail.py --query "after:2024/01/01 has:attachment" --max-threads 50
+```
+
+
+That's it — your emails are now queryable in Honcho. Read on if you want to understand how the script works and the design decisions behind it.
+
+---
+
+## Full Tutorial
+
+### How Gmail Maps to Honcho
+
+The core idea is straightforward: each Gmail thread becomes a Honcho session, and each email participant becomes a peer. Here's the full mapping:
+
+| Gmail Concept | Honcho Concept | Details |
+|---------------|----------------|---------|
+| Your Gmail account | Workspace (`gmail`) | One workspace for all email data |
+| Email participant | Peer | Email address as ID for deduplication |
+| Email thread | Session (`gmail-thread-{id}`) | One session per thread, all participants attached |
+| Individual email | Message | Attributed to the sender with original timestamp |
+
+### Email as Peer ID
+
+The script normalizes email addresses into URL-safe peer IDs — `alice@example.com` becomes `alice-example-com`. This means the same person is automatically deduplicated across threads. If Alice emails you in 10 different threads, all of those conversations accumulate under a single peer.
+
+```python
+def peer_id_from_email(email: str) -> str:
+ """Convert email to a valid Honcho peer ID."""
+ return email.replace("@", "-").replace(".", "-")
+```
+
+This also means peers are consistent across data sources. If you import Granola meetings and Gmail threads for the same person, they merge under the same peer ID.
+
+### Extracting Participants
+
+Every email has a sender, recipients, and optionally CC/BCC addresses. The script extracts all of these to build a complete picture of who's involved in each thread:
+
+```python
+for m in msgs:
+ register_peer(m["from"])
+ for addr in parse_address_list(m["to"]):
+ register_peer(addr)
+ for addr in parse_address_list(m["cc"]):
+ register_peer(addr)
+ for addr in parse_address_list(m["bcc"]):
+ register_peer(addr)
+```
+
+Display names are extracted when available (e.g., `Alice Smith ` → name: "Alice Smith"). When only an email is present, the script generates a name from the local part.
+
+### Message Attribution and Timestamps
+
+Each email becomes a message attributed to its sender via `peer.message()`. The original email timestamp is preserved using `created_at`, so Honcho sees the conversation in chronological order — not the order you imported it.
+
+```python
+honcho_msgs.append(peer.message(
+ content,
+ metadata={
+ "gmail_id": m["id"],
+ "subject": m["subject"],
+ "from": m["from"],
+ "to": m["to"],
+ "labels": m["labels"],
+ },
+ created_at=m["timestamp"],
+))
+```
+
+### Multi-Peer Sessions
+
+Each thread's session is linked to all participants using `session.add_peers()`. This means when you query Honcho about a peer, it has context not just from their messages but from the full conversations they participated in.
+
+```python
+session = honcho.session(session_id, metadata={
+ "gmail_thread_id": tid,
+ "subject": subject,
+ "source": "gmail",
+ "message_count": len(msgs),
+})
+session.add_peers(thread_peers)
+```
+
+### Stripping Quoted Replies
+
+Email threads are full of quoted replies — each message repeats everything above it. The script strips these out so only the new content is stored per message, avoiding duplication in Honcho's memory:
+
+```python
+def strip_quoted_replies(text: str) -> str:
+ """Strip quoted reply text, keeping only the new content."""
+ lines = text.split("\n")
+ clean_lines = []
+ for line in lines:
+ stripped = line.strip()
+ if re.match(r"^On .+wrote:\s*$", stripped):
+ break
+ if stripped.startswith(">"):
+ break
+ # ... other reply markers
+ clean_lines.append(line)
+ return "\n".join(clean_lines).rstrip()
+```
+
+### Querying After Import
+
+Once your emails are in Honcho, you can query any peer:
+
+```python
+import os
+from honcho import Honcho
+
+honcho = Honcho(workspace_id="gmail-inbox", api_key=os.environ["HONCHO_API_KEY"])
+
+alice = honcho.peer("alice-example-com")
+print(alice.chat("What has Alice been discussing with me?"))
+print(alice.chat("What action items has Alice mentioned?"))
+```
+
+---
+
+## CLI Reference
+
+```
+usage: honcho_gmail.py [-h] [--workspace WORKSPACE] [--query QUERY]
+ [--label LABEL] [--max-threads N] [--dry-run]
+ [--credentials PATH] [--token PATH]
+
+options:
+ --workspace, -w Honcho workspace ID (default: gmail)
+ --query, -q Gmail search query (e.g., 'from:alice@example.com')
+ --label, -l Gmail label to filter by (e.g., INBOX)
+ --max-threads, -n Max threads to fetch (default: 10)
+ --dry-run Preview without writing to Honcho
+ --credentials, -c Path to OAuth credentials JSON (auto-detects client_secret*.json)
+ --token, -t Path to store access token (default: token.json)
+```
+
+## Troubleshooting
+
+### "No client_secret*.json file found"
+
+Download OAuth credentials from Google Cloud Console and place the `client_secret_*.json` file in the same directory as the script.
+
+### "Access blocked: This app's request is invalid"
+
+Your OAuth consent screen may not be configured correctly. Ensure you've added the `gmail.readonly` scope.
+
+### "Token has been expired or revoked"
+
+Delete `token.json` and run the script again to re-authenticate.
+
+### Rate Limits
+
+The script includes a small delay when creating peers to avoid hitting Honcho's rate limits. For large imports (100+ threads), consider running in batches.
+
+### Unique Messages
+
+Use an AI assistant in your inbox? Want to parse out its messages differently? Feel free to modify and improve the structure of this script to fit your bespoke email setup. This script was written for agents and as such is easy to update with your coding assistant.
+
+## Full Script
+
+
+```python
+#!/usr/bin/env python3
+"""Load Gmail messages into Honcho.
+
+Uses the Gmail API directly (with OAuth) to fetch emails and the Honcho Python SDK to store them.
+Each Gmail thread becomes a Honcho session, each sender becomes a peer.
+
+Prerequisites:
+1. Create a Google Cloud project and enable the Gmail API
+2. Create OAuth 2.0 credentials (Desktop app type)
+3. Download the credentials JSON (client_secret_*.json) into this directory
+4. Install dependencies:
+ pip install google-api-python-client google-auth-oauthlib honcho-ai
+
+On first run, a browser window will open for OAuth consent. After authorizing,
+a 'token.json' file will be created to store your credentials for future runs.
+"""
+
+import argparse
+import base64
+import glob
+import os
+import re
+import time
+from datetime import datetime, timezone
+from email.header import decode_header, make_header
+from email.utils import getaddresses, parseaddr
+
+from google.auth.transport.requests import Request
+from google.oauth2.credentials import Credentials
+from google_auth_oauthlib.flow import InstalledAppFlow
+from googleapiclient.discovery import build
+from googleapiclient.errors import HttpError
+
+SCOPES = ["https://www.googleapis.com/auth/gmail.readonly"]
+PEER_ID_PATTERN = re.compile(r"^[a-zA-Z0-9_-]+$")
+
+
+def find_credentials() -> str:
+ """Find a Google OAuth credentials file in the current directory."""
+ matches = glob.glob("client_secret*.json")
+ if matches:
+ return matches[0]
+ raise FileNotFoundError(
+ "No client_secret*.json file found.\n"
+ "Download OAuth credentials from Google Cloud Console:\n"
+ "1. Go to console.cloud.google.com\n"
+ "2. Create/select a project and enable Gmail API\n"
+ "3. Create OAuth 2.0 credentials (Desktop app)\n"
+ "4. Download the JSON into this directory"
+ )
+
+
+def get_gmail_service(credentials_file: str | None = None, token_file: str = "token.json"):
+ """Authenticate and return a Gmail API service instance."""
+ creds = None
+
+ if os.path.exists(token_file):
+ creds = Credentials.from_authorized_user_file(token_file, SCOPES)
+
+ if not creds or not creds.valid:
+ if creds and creds.expired and creds.refresh_token:
+ print("Refreshing expired credentials...")
+ creds.refresh(Request())
+ else:
+ if credentials_file is None:
+ credentials_file = find_credentials()
+ print(f"Using credentials: {credentials_file}")
+ print("Opening browser for OAuth consent...")
+ flow = InstalledAppFlow.from_client_secrets_file(credentials_file, SCOPES)
+ creds = flow.run_local_server(port=0)
+
+ with open(token_file, "w") as token:
+ token.write(creds.to_json())
+ print(f"Credentials saved to {token_file}")
+
+ return build("gmail", "v1", credentials=creds)
+
+
+def list_threads(service, query: str = None, label_ids: list = None, max_results: int = 10) -> list[dict]:
+ """List Gmail threads with pagination support."""
+ all_threads = []
+ page_token = None
+
+ while len(all_threads) < max_results:
+ try:
+ params = {
+ "userId": "me",
+ "maxResults": min(100, max_results - len(all_threads)),
+ }
+ if query:
+ params["q"] = query
+ if label_ids:
+ params["labelIds"] = label_ids
+ if page_token:
+ params["pageToken"] = page_token
+
+ response = service.users().threads().list(**params).execute()
+ threads = response.get("threads", [])
+ all_threads.extend(threads)
+
+ page_token = response.get("nextPageToken")
+ if not page_token:
+ break
+
+ except HttpError as e:
+ print(f"Error listing threads: {e}")
+ break
+
+ return all_threads[:max_results]
+
+
+def get_thread(service, thread_id: str) -> dict:
+ """Fetch a complete Gmail thread with all messages."""
+ try:
+ return service.users().threads().get(
+ userId="me",
+ id=thread_id,
+ format="full"
+ ).execute()
+ except HttpError as e:
+ print(f"Error fetching thread {thread_id}: {e}")
+ return {}
+
+
+def _decode_header_str(header: str) -> str:
+ """Decode an RFC 2047 encoded header string to plain Unicode."""
+ return str(make_header(decode_header(header)))
+
+
+def extract_email(from_header: str) -> str:
+ """Extract bare email from an RFC 5322 header value."""
+ _, addr = parseaddr(_decode_header_str(from_header))
+ return addr.lower().strip()
+
+
+def extract_name(from_header: str) -> str:
+ """Extract display name from an RFC 5322 header value."""
+ name, _ = parseaddr(_decode_header_str(from_header))
+ return name.strip() or from_header.strip()
+
+
+def decode_body(payload: dict) -> str:
+ """Recursively extract plain text from a Gmail message payload."""
+ if payload.get("mimeType") == "text/plain":
+ data = payload.get("body", {}).get("data", "")
+ if data:
+ return base64.urlsafe_b64decode(data).decode("utf-8", errors="replace")
+
+ parts = payload.get("parts", [])
+ for part in parts:
+ text = decode_body(part)
+ if text:
+ return text
+ return ""
+
+
+def strip_quoted_replies(text: str) -> str:
+ """Strip quoted reply text from an email body, keeping only the new content."""
+ lines = text.split("\n")
+ clean_lines = []
+ for line in lines:
+ stripped = line.strip()
+ if re.match(r"^On .+wrote:\s*$", stripped):
+ break
+ if stripped.startswith("---------- Forwarded message"):
+ break
+ if stripped.startswith(">"):
+ break
+ if re.match(r"^[-_]{10,}$", stripped):
+ break
+ clean_lines.append(line)
+ return "\n".join(clean_lines).rstrip()
+
+
+def parse_address_list(header: str) -> list[str]:
+ """Parse a comma-separated email header into individual addresses."""
+ if not header.strip():
+ return []
+ decoded = _decode_header_str(header)
+ return [
+ f"{name} <{addr}>" if name else addr
+ for name, addr in getaddresses([decoded])
+ if addr
+ ]
+
+
+def peer_id_from_email(email: str) -> str:
+ """Convert email to a valid Honcho peer ID."""
+ peer_id = re.sub(r"[^A-Za-z0-9_-]+", "-", email).strip("-").lower()
+ peer_id = re.sub(r"-{2,}", "-", peer_id)
+ if not peer_id:
+ peer_id = "unknown-peer"
+
+ if not PEER_ID_PATTERN.fullmatch(peer_id):
+ raise ValueError(f"Generated peer ID is invalid: {peer_id!r}")
+ return peer_id
+
+
+def fetch_thread_messages(service, thread_id: str) -> list[dict]:
+ """Fetch all messages in a Gmail thread with full content."""
+ data = get_thread(service, thread_id)
+ messages = []
+
+ for msg in data.get("messages", []):
+ headers = {h["name"]: h["value"] for h in msg.get("payload", {}).get("headers", [])}
+ body = strip_quoted_replies(decode_body(msg.get("payload", {})))
+ ts = int(msg.get("internalDate", "0")) / 1000
+
+ messages.append({
+ "id": msg["id"],
+ "thread_id": msg["threadId"],
+ "from": headers.get("From", ""),
+ "to": headers.get("To", ""),
+ "cc": headers.get("Cc", ""),
+ "bcc": headers.get("Bcc", ""),
+ "subject": headers.get("Subject", ""),
+ "date": headers.get("Date", ""),
+ "timestamp": datetime.fromtimestamp(ts, tz=timezone.utc),
+ "body": body.strip(),
+ "labels": msg.get("labelIds", []),
+ "snippet": msg.get("snippet", ""),
+ })
+
+ return messages
+
+
+def main():
+ parser = argparse.ArgumentParser(description="Load Gmail messages into Honcho")
+ parser.add_argument("--workspace", "-w", default="gmail", help="Honcho workspace ID (default: gmail)")
+ parser.add_argument("--query", "-q", default=None, help="Gmail search query (e.g. 'from:alice@example.com')")
+ parser.add_argument("--label", "-l", default=None, help="Gmail label to filter by (e.g. INBOX)")
+ parser.add_argument("--max-threads", "-n", type=int, default=10, help="Max threads to fetch (default: 10)")
+ parser.add_argument("--dry-run", action="store_true", help="Print what would be loaded without writing to Honcho")
+ parser.add_argument("--credentials", "-c", default=None, help="Path to OAuth credentials JSON (auto-detects client_secret*.json)")
+ parser.add_argument("--token", "-t", default="token.json", help="Path to store/load access token")
+ args = parser.parse_args()
+
+ # Authenticate
+ print("Authenticating with Gmail API...")
+ service = get_gmail_service(args.credentials, args.token)
+ print(" Authenticated successfully!")
+
+ label_ids = [args.label] if args.label else None
+
+ # List threads
+ print(f"\nFetching up to {args.max_threads} threads from Gmail...")
+ threads = list_threads(service, query=args.query, label_ids=label_ids, max_results=args.max_threads)
+ print(f" Found {len(threads)} threads")
+
+ if not threads:
+ print("No threads found. Try adjusting --query or --label.")
+ return
+
+ # Fetch full messages for each thread
+ all_thread_messages = {}
+ seen_peers = {}
+
+ def register_peer(addr: str):
+ email = extract_email(addr)
+ if email and email not in seen_peers:
+ name = extract_name(addr)
+ if name.lower().strip() == email or "@" in name:
+ name = email.split("@")[0].replace(".", " ").title()
+ seen_peers[email] = {
+ "name": name,
+ "peer_id": peer_id_from_email(email),
+ "email": email,
+ }
+
+ for i, t in enumerate(threads):
+ tid = t["id"]
+ print(f" Fetching thread {i+1}/{len(threads)}: {tid}")
+ msgs = fetch_thread_messages(service, tid)
+ all_thread_messages[tid] = msgs
+ for m in msgs:
+ register_peer(m["from"])
+ for addr in parse_address_list(m["to"]):
+ register_peer(addr)
+ for addr in parse_address_list(m["cc"]):
+ register_peer(addr)
+ for addr in parse_address_list(m["bcc"]):
+ register_peer(addr)
+
+ # Summary
+ total_msgs = sum(len(v) for v in all_thread_messages.values())
+ print("\nSummary:")
+ print(f" Threads: {len(all_thread_messages)}")
+ print(f" Messages: {total_msgs}")
+ print(f" Unique participants: {len(seen_peers)}")
+ for email, info in seen_peers.items():
+ print(f" {info['peer_id']} ({info['name']} <{email}>)")
+
+ if args.dry_run:
+ print("\n[DRY RUN] Would create the above in Honcho. Showing first message per thread:")
+ for tid, msgs in all_thread_messages.items():
+ m = msgs[0]
+ body_preview = m["body"][:120].replace("\n", " ") if m["body"] else m["snippet"][:120]
+ print(f" Thread {tid}: {m['subject']}")
+ print(f" {m['from']} @ {m['date']}")
+ print(f" {body_preview}...")
+ return
+
+ # Load into Honcho
+ from honcho import Honcho
+
+ print(f"\nLoading into Honcho workspace '{args.workspace}'...")
+ honcho = Honcho(workspace_id=args.workspace)
+
+ # Create peers
+ peers = {}
+ for i, (email, info) in enumerate(seen_peers.items()):
+ if i > 0 and i % 4 == 0:
+ time.sleep(1)
+ peers[email] = honcho.peer(info["peer_id"], metadata={
+ "email": email,
+ "name": info["name"],
+ "source": "gmail",
+ })
+ print(f" Peer: {info['peer_id']}")
+
+ # Create sessions and messages per thread
+ for tid, msgs in all_thread_messages.items():
+ subject = msgs[0]["subject"] if msgs else "No subject"
+ session_id = f"gmail-thread-{tid}"
+
+ thread_peer_emails = set()
+ for m in msgs:
+ thread_peer_emails.add(extract_email(m["from"]))
+ for addr in parse_address_list(m["to"]):
+ thread_peer_emails.add(extract_email(addr))
+ for addr in parse_address_list(m["cc"]):
+ thread_peer_emails.add(extract_email(addr))
+ for addr in parse_address_list(m["bcc"]):
+ thread_peer_emails.add(extract_email(addr))
+ thread_peers = [peers[e] for e in thread_peer_emails if e in peers]
+
+ session = honcho.session(session_id, metadata={
+ "gmail_thread_id": tid,
+ "subject": subject,
+ "source": "gmail",
+ "message_count": len(msgs),
+ })
+ session.add_peers(thread_peers)
+
+ honcho_msgs = []
+ for m in msgs:
+ email = extract_email(m["from"])
+ peer = peers.get(email)
+ if not peer:
+ continue
+ content = m["body"] if m["body"] else m["snippet"]
+ if not content:
+ continue
+ honcho_msgs.append(peer.message(
+ content,
+ metadata={
+ "gmail_id": m["id"],
+ "subject": m["subject"],
+ "from": m["from"],
+ "to": m["to"],
+ "labels": m["labels"],
+ },
+ created_at=m["timestamp"],
+ ))
+
+ if honcho_msgs:
+ session.add_messages(honcho_msgs)
+ print(f" Session {session_id}: {len(honcho_msgs)} messages — {subject[:60]}")
+
+ print(f"\nDone! Loaded {total_msgs} messages into workspace '{args.workspace}'.")
+
+
+if __name__ == "__main__":
+ main()
+
+```
+
+## Next Steps
+
+
+
+ See how the Granola integration maps to common Honcho patterns.
+
+
+ Source code and example script.
+
+
diff --git a/docs/v3/guides/granola.mdx b/docs/v3/guides/granola.mdx
new file mode 100644
index 00000000..2c850264
--- /dev/null
+++ b/docs/v3/guides/granola.mdx
@@ -0,0 +1,953 @@
+---
+title: "Granola"
+icon: 'microphone'
+description: "Import meeting notes and transcripts from Granola into Honcho"
+sidebarTitle: 'Granola'
+---
+
+In this tutorial, we'll walk through how to import your [Granola](https://granola.ai) meeting data into Honcho. By the end, your meeting participants, transcripts, and summaries will be mapped onto Honcho's peer and session model — giving your agents queryable memory of the people you meet with.
+
+This guide includes a ready-to-run Python script that handles everything: Granola OAuth, meeting fetching, participant detection, and interactive import. You can run it as-is or use the full tutorial below to understand each design decision.
+
+
+The full script is available on [GitHub](https://github.com/plastic-labs/honcho/tree/main/examples/granola).
+
+
+## TL;DR
+
+If you just want to get your meetings into Honcho, here's everything you need.
+
+### 1. Install Dependencies
+
+
+```bash uv
+uv pip install honcho-ai httpx
+```
+
+```bash pip
+pip install honcho-ai httpx
+```
+
+
+### 2. Set Your API Key
+
+```bash
+export HONCHO_API_KEY="your-key-from-app.honcho.dev"
+```
+
+### 3. Run the Script
+
+
+```bash uv
+uv run python honcho_granola.py
+```
+
+```bash python
+python honcho_granola.py
+```
+
+
+The script will:
+1. Open your browser for Granola OAuth authentication
+2. Fetch all meetings and their content
+3. Walk you through each meeting interactively — confirm peers, choose import mode, skip meetings you don't want
+4. Print a summary of what was transferred
+
+That's it — your meetings are now queryable in Honcho. Read on if you want to understand how the script works and the design decisions behind it.
+
+---
+
+## Full Tutorial
+
+### How Granola Maps to Honcho
+
+The core idea is straightforward: each Granola meeting becomes a Honcho session, and each participant becomes a peer. Here's the full mapping:
+
+| Granola Concept | Honcho Concept | Details |
+|-----------------|----------------|---------|
+| Your Granola account | Workspace (`granola`) | One workspace for all meetings |
+| Meeting participant | Peer | Email as ID for deduplication across meetings |
+| Individual meeting | Session (`meeting-{id}`) | One session per meeting |
+| Transcript turns | Messages with attribution | Two-person calls get full speaker attribution |
+| Meeting summary | Message from note creator | Multi-person calls store the summary |
+
+### Email as Peer ID
+
+The script uses email addresses as the basis for peer IDs, normalized to a URL-safe format (e.g., `alice@example.com` becomes `alice-example-com`). This ensures consistent identification across meetings — if you meet someone in 5 different calls, all conversations accumulate under the same peer.
+
+```python
+# These all resolve to the same peer:
+honcho.peer("alice-example-com") # From Meeting A
+honcho.peer("alice-example-com") # From Meeting B
+```
+
+This also means peers are consistent across data sources. If you import both Granola meetings and Gmail threads for the same person, they merge under the same peer ID.
+
+### Auto-Detecting "Me"
+
+Granola marks the note creator in its participant list with `(note creator)`. The script uses this to identify you automatically — no configuration needed.
+
+```
+Participants: You (note creator) from Your Company ,
+ Alice from Acme Corp
+```
+
+### Two-Person Calls: Full Attribution
+
+When exactly one other participant is present *and* the transcript contains `Them:` turns, the script stores the transcript with speaker-attributed messages. Consecutive same-speaker turns are merged before storing, cleaning up the fragmentation that's common in raw transcripts.
+
+```python
+session.add_messages([
+ me.message("What's your timeline for the launch?"),
+ them.message("We're targeting Q2, but it depends on the API integration."),
+])
+```
+
+### Multi-Person Calls: Summary Mode
+
+Granola's transcript uses `Them:` for all non-creator speakers with no disambiguation — in a 4-person call, everyone else is just `Them:`. Rather than guess incorrectly, the script stores Granola's summary as your record of the meeting, with participants in metadata.
+
+```python
+session.add_messages([
+ me.message(
+ f"Meeting: Product Planning\n"
+ f"Date: Mar 5, 2026 2:00 PM\n"
+ f"Participants: Alice from Acme Corp, Bob from Widgets Inc\n\n"
+ f"{meeting_summary}",
+ metadata={
+ "participants": "Alice from Acme Corp, Bob from Widgets Inc",
+ "mode": "summary",
+ "granola_meeting_id": meeting_id,
+ }
+ )
+])
+```
+
+The summary is attributed to you because it's *your* record of what happened. Granola captured your notes from a meeting where those people were present.
+
+### Interactive Confirmation
+
+For each meeting, you choose the import mode: two-person (full attribution), summary, or skip. For multi-person calls that are actually 1:1s (extra participants listed but didn't speak), you can override the detection and select the actual speaker.
+
+### Noisy Transcripts Preserved
+
+Granola's raw transcripts are often fragmented (`Me: Yeah. Them: Yeah. Me: And.`). The script merges consecutive same-speaker turns but otherwise preserves the raw content. Honcho's reasoning extracts signal from noisy data.
+
+### Querying After Import
+
+Once your meetings are in Honcho, you can query any peer:
+
+```python
+import os
+from honcho import Honcho
+
+honcho = Honcho(workspace_id="granola", api_key=os.environ["HONCHO_API_KEY"])
+
+# Peer IDs are normalized from emails: alice@example.com -> alice-example-com
+alice = honcho.peer("alice-example-com")
+print(alice.chat("What is Alice working on?"))
+print(alice.chat("What concerns has Alice raised?"))
+
+me = honcho.peer("you-example-com")
+print(me.chat("What topics do I discuss most frequently?"))
+```
+
+### Combining with Other Sources
+
+Because meetings live in a standard Honcho workspace, you can enrich peer representations with data from other channels:
+
+```python
+# Same workspace, same peer — data accumulates
+alice = honcho.peer("alice-example-com")
+me = honcho.peer("you-example-com")
+
+discord_session = honcho.session("discord-general-2024-03")
+discord_session.add_messages([
+ alice.message("Just shipped the new API version!"),
+ me.message("Congrats! How's the migration guide coming?"),
+])
+
+# Queries now draw from both meeting transcripts AND Discord history
+alice.chat("What has Alice shipped recently?")
+```
+
+---
+
+## Troubleshooting
+
+| Issue | Fix |
+|-------|-----|
+| Granola OAuth fails | Ensure you have a paid Granola plan (MCP requires Pro+). Clear cached token and retry. |
+| Missing transcripts | Free tier has no transcript access. The script falls back to summary content. |
+| 500 errors from Honcho | Check for null bytes or control characters in transcript content. The script sanitizes these automatically. |
+| Rate limiting with many meetings | The script processes sequentially with delays. Honcho ingestion is async — don't poll for immediate results. |
+
+## Full Script
+
+
+```python
+#!/usr/bin/env python3
+"""Load Granola meeting notes into Honcho.
+
+Uses the Granola MCP server (with OAuth) to fetch meetings and the Honcho Python SDK
+to store them. Each meeting becomes a Honcho session. Two-person meetings get full
+speaker attribution; multi-person meetings are stored as summaries.
+
+Prerequisites:
+ pip install honcho-ai httpx
+
+Environment Variables:
+ HONCHO_API_KEY - Your Honcho API key (get from app.honcho.dev/api-keys)
+
+Usage:
+ python honcho_granola.py
+"""
+
+import asyncio
+import base64
+import hashlib
+import json
+import os
+import re
+import secrets
+import sys
+import threading
+import traceback
+import webbrowser
+from dataclasses import dataclass, field
+from datetime import datetime, timezone
+from http.server import HTTPServer, BaseHTTPRequestHandler
+from typing import Any
+from urllib.parse import parse_qs, urlencode, urlparse
+
+import httpx
+
+
+@dataclass
+class Participant:
+ name: str
+ email: str | None = None
+ org: str | None = None
+
+
+@dataclass
+class ParsedParticipants:
+ note_creator: Participant | None = None
+ others: list[Participant] = field(default_factory=list)
+
+
+@dataclass
+class TranscriptTurn:
+ speaker: str
+ text: str
+
+
+# Granola MCP + OAuth endpoints
+GRANOLA_MCP_URL = "https://mcp.granola.ai/mcp"
+AUTH_BASE = "https://mcp-auth.granola.ai"
+OAUTH_REDIRECT_PORT = 8765
+OAUTH_REDIRECT_URI = f"http://localhost:{OAUTH_REDIRECT_PORT}/callback"
+
+# Honcho message size limit (25000 max, leave headroom)
+MAX_MESSAGE_LEN = 24000
+
+
+# ---------------------------------------------------------------------------
+# OAuth callback handler (must be a class for BaseHTTPRequestHandler)
+# ---------------------------------------------------------------------------
+
+class _OAuthCallback(BaseHTTPRequestHandler):
+ auth_result: dict[str, str | None] = {"code": None, "error": None}
+
+ def do_GET(self):
+ params = parse_qs(urlparse(self.path).query)
+ if "code" in params:
+ _OAuthCallback.auth_result["code"] = params["code"][0]
+ self.send_response(200)
+ self.send_header("Content-Type", "text/html")
+ self.end_headers()
+ self.wfile.write(b"Authenticated! You can close this window.
")
+ elif "error" in params:
+ _OAuthCallback.auth_result["error"] = params.get("error_description", params["error"])[0]
+ self.send_response(400)
+ self.send_header("Content-Type", "text/html")
+ self.end_headers()
+ self.wfile.write(f"Error: {_OAuthCallback.auth_result['error']}
".encode())
+ else:
+ self.send_response(404)
+ self.end_headers()
+
+ def log_message(self, fmt, *args):
+ pass
+
+
+# ---------------------------------------------------------------------------
+# Granola OAuth + MCP
+# ---------------------------------------------------------------------------
+
+async def authenticate(http_client: httpx.AsyncClient) -> str:
+ """Perform OAuth (DCR + PKCE) with Granola. Returns access token."""
+ _OAuthCallback.auth_result = {"code": None, "error": None}
+
+ print("\nAuthenticating with Granola...")
+
+ # Register client (DCR)
+ resp = await http_client.post(
+ f"{AUTH_BASE}/oauth2/register",
+ json={
+ "client_name": "Granola to Honcho Transfer",
+ "redirect_uris": [OAUTH_REDIRECT_URI],
+ "grant_types": ["authorization_code"],
+ "response_types": ["code"],
+ "token_endpoint_auth_method": "none",
+ },
+ )
+ if resp.status_code not in (200, 201):
+ raise RuntimeError(f"Client registration failed: {resp.status_code}")
+ client_id = resp.json().get("client_id")
+
+ # PKCE
+ verifier = secrets.token_urlsafe(32)
+ challenge = base64.urlsafe_b64encode(hashlib.sha256(verifier.encode()).digest()).rstrip(b"=").decode()
+
+ # Browser auth
+ auth_url = f"{AUTH_BASE}/oauth2/authorize?" + urlencode({
+ "client_id": client_id,
+ "redirect_uri": OAUTH_REDIRECT_URI,
+ "response_type": "code",
+ "state": "granola-honcho-transfer",
+ "code_challenge": challenge,
+ "code_challenge_method": "S256",
+ })
+
+ server = HTTPServer(("localhost", OAUTH_REDIRECT_PORT), _OAuthCallback)
+ thread = threading.Thread(target=server.handle_request)
+ thread.start()
+
+ print(" Opening browser for authentication...")
+ webbrowser.open(auth_url)
+ thread.join(timeout=120)
+ server.server_close()
+
+ auth_result = _OAuthCallback.auth_result
+ if auth_result["error"]:
+ raise RuntimeError(f"Authentication failed: {auth_result['error']}")
+ if not auth_result["code"]:
+ raise RuntimeError("Authentication timed out")
+
+ # Exchange code for token
+ resp = await http_client.post(
+ f"{AUTH_BASE}/oauth2/token",
+ data={
+ "grant_type": "authorization_code",
+ "code": auth_result["code"],
+ "redirect_uri": OAUTH_REDIRECT_URI,
+ "client_id": client_id,
+ "code_verifier": verifier,
+ },
+ headers={"Content-Type": "application/x-www-form-urlencoded"},
+ )
+ if resp.status_code != 200:
+ raise RuntimeError(f"Token exchange failed: {resp.status_code}")
+
+ print(" Authenticated successfully!")
+ return resp.json()["access_token"]
+
+
+async def call_mcp_tool(
+ http_client: httpx.AsyncClient,
+ access_token: str,
+ tool_name: str,
+ arguments: dict[str, Any] | None = None,
+) -> dict[str, Any]:
+ """Call a Granola MCP tool, handling both JSON and SSE responses."""
+ resp = await http_client.post(
+ GRANOLA_MCP_URL,
+ json={
+ "jsonrpc": "2.0",
+ "id": 1,
+ "method": "tools/call",
+ "params": {"name": tool_name, "arguments": arguments or {}},
+ },
+ headers={
+ "Authorization": f"Bearer {access_token}",
+ "Content-Type": "application/json",
+ "Accept": "application/json, text/event-stream",
+ },
+ )
+ if resp.status_code != 200:
+ raise RuntimeError(f"MCP call failed: {resp.status_code} - {resp.text}")
+
+ # SSE response
+ if "text/event-stream" in resp.headers.get("content-type", ""):
+ result = None
+ for line in resp.text.split("\n"):
+ if line.strip().startswith("data: "):
+ try:
+ parsed = json.loads(line.strip()[6:])
+ if "result" in parsed:
+ result = parsed
+ elif "error" in parsed:
+ raise RuntimeError(f"MCP error: {parsed['error']}")
+ except json.JSONDecodeError:
+ continue
+ if result:
+ final = result.get("result", {})
+ return final if isinstance(final, dict) else {"result": final}
+ raise RuntimeError("No result in SSE response")
+
+ # JSON response
+ result = resp.json()
+ if "error" in result:
+ raise RuntimeError(f"MCP error: {result['error']}")
+ return result.get("result", {})
+
+
+def extract_mcp_text(result: dict[str, Any]) -> str:
+ """Extract text from the first content block of an MCP result.
+
+ Raises ValueError if the response structure is unexpected.
+ """
+ content = result.get("content", [])
+ if not isinstance(content, list) or not content:
+ raise ValueError(f"MCP response missing content array: {list(result.keys())}")
+ first = content[0]
+ if not isinstance(first, dict) or "text" not in first:
+ raise ValueError(f"MCP content block missing 'text' field: {first}")
+ return str(first["text"])
+
+
+# ---------------------------------------------------------------------------
+# Granola data fetching
+# ---------------------------------------------------------------------------
+
+async def list_meetings(
+ http_client: httpx.AsyncClient, access_token: str, limit: int = 100,
+) -> list[dict[str, Any]]:
+ """List meetings from Granola MCP. Parses Granola's XML-like response format."""
+ result = await call_mcp_tool(http_client, access_token, "list_meetings", {"limit": limit})
+ text = extract_mcp_text(result)
+
+ meetings: list[dict[str, Any]] = []
+ for match in re.finditer(r'", match.end())
+ block = text[match.end():block_end] if block_end != -1 else ""
+ p_match = re.search(r"\s*(.*?)\s*", block, re.DOTALL)
+ meetings.append({
+ "id": mid,
+ "title": title,
+ "date": date,
+ "participants": p_match.group(1).strip() if p_match else "",
+ })
+
+ return meetings
+
+
+async def get_meeting_details(
+ http_client: httpx.AsyncClient, access_token: str, meeting_id: str,
+) -> dict[str, Any]:
+ """Get full meeting details including notes."""
+ result = await call_mcp_tool(http_client, access_token, "get_meetings", {"meeting_ids": [meeting_id]})
+ text = extract_mcp_text(result)
+ return {"id": meeting_id, "raw_content": text}
+
+
+async def get_meeting_transcript(
+ http_client: httpx.AsyncClient, access_token: str, meeting_id: str,
+ max_retries: int = 3,
+) -> str | None:
+ """Get transcript for a meeting (paid tiers only).
+
+ Retries on rate limit responses with exponential backoff.
+ """
+ for attempt in range(max_retries):
+ try:
+ result = await call_mcp_tool(http_client, access_token, "get_meeting_transcript", {"meeting_id": meeting_id})
+ text = extract_mcp_text(result)
+ except Exception as e:
+ print(f" Transcript unavailable: {e}")
+ return None
+
+ if not text or "no transcript" in text.lower():
+ return None
+
+ # Granola returns rate limit errors as content text, not HTTP errors
+ if "rate limit" in text.lower():
+ wait = 2 ** attempt * 3 # 3s, 6s, 12s
+ print(f" ⚠ Granola rate limit hit (attempt {attempt + 1}/{max_retries}), waiting {wait}s...")
+ await asyncio.sleep(wait)
+ continue
+
+ return text
+
+ print(f" ⚠ Transcript skipped after {max_retries} rate limit retries")
+ return None
+
+
+async def fetch_all_meetings(
+ http_client: httpx.AsyncClient, access_token: str,
+) -> list[dict[str, Any]]:
+ """Fetch meeting list and enrich each with transcript and details."""
+ print("\nFetching meetings from Granola...")
+ meetings = await list_meetings(http_client, access_token, limit=500)
+ if not meetings:
+ print("No meetings found.")
+ return []
+ print(f" Found {len(meetings)} meetings. Fetching content...\n")
+
+ for i, m in enumerate(meetings, 1):
+ mid = m.get("id")
+ if not mid:
+ continue
+
+ transcript = await get_meeting_transcript(http_client, access_token, mid)
+ if transcript:
+ m["transcript"] = transcript
+
+ try:
+ m.update(await get_meeting_details(http_client, access_token, mid))
+ except Exception as exc:
+ print(f" Failed to fetch details for {mid}: {exc}")
+
+ has_t = "transcript" in m
+ has_s = bool(extract_summary(m))
+ label = "transcript+summary" if has_t and has_s else "transcript only" if has_t else "summary only" if has_s else "basic only"
+ print(f" [{i}/{len(meetings)}] {label}: {m.get('title', 'Untitled')[:45]}")
+ await asyncio.sleep(1.5) # rate limit
+
+ return meetings
+
+
+# ---------------------------------------------------------------------------
+# Parsing helpers
+# ---------------------------------------------------------------------------
+
+def parse_participants(participants_str: str) -> ParsedParticipants:
+ """Parse Granola's participant string into structured participants.
+
+ Warns on unparseable entries instead of silently dropping them.
+ """
+ result = ParsedParticipants()
+ if not participants_str:
+ return result
+
+ # Split on commas, but not inside angle brackets
+ entries, current, depth = [], [], 0
+ for ch in participants_str:
+ if ch == "<":
+ depth += 1
+ elif ch == ">":
+ depth = max(depth - 1, 0)
+ elif ch == "," and depth == 0:
+ entries.append("".join(current))
+ current = []
+ continue
+ current.append(ch)
+ if current:
+ entries.append("".join(current))
+
+ for entry in entries:
+ entry = entry.strip()
+ if not entry:
+ continue
+
+ is_creator = "(note creator)" in entry
+ clean = entry.replace("(note creator)", "").strip()
+
+ email_match = re.search(r"<([^>]+)>", clean)
+ email = email_match.group(1) if email_match else None
+ name = re.sub(r"\s*<[^>]+>", "", clean).strip()
+
+ if not name:
+ print(f" Warning: could not parse participant entry: {entry!r}")
+ continue
+
+ org = None
+ org_match = re.match(r"(.+?)\s+from\s+(.+)", name)
+ if org_match:
+ name, org = org_match.group(1).strip(), org_match.group(2).strip()
+
+ person = Participant(name=name, email=email, org=org)
+ if is_creator:
+ result.note_creator = person
+ else:
+ result.others.append(person)
+
+ return result
+
+
+def parse_transcript_turns(raw: str) -> list[TranscriptTurn]:
+ """Split a Granola transcript into speaker turns."""
+ # Unwrap JSON wrapper if present
+ try:
+ parsed = json.loads(raw)
+ if isinstance(parsed, dict) and "transcript" in parsed:
+ raw = str(parsed["transcript"])
+ except (json.JSONDecodeError, TypeError):
+ pass
+
+ parts = re.split(r"(?:^|\s{2,})(Me|Them):\s*", raw)
+ turns: list[TranscriptTurn] = []
+ i = 1
+ while i < len(parts) - 1:
+ text = parts[i + 1].strip()
+ if text:
+ turns.append(TranscriptTurn(speaker=parts[i], text=text))
+ i += 2
+ return turns
+
+
+def extract_summary(meeting: dict[str, Any]) -> str:
+ """Extract best available summary text from meeting data."""
+ candidates = []
+ for key in ("summary", "notes", "note", "meeting_notes", "description"):
+ val = meeting.get(key)
+ if isinstance(val, str) and val.strip():
+ candidates.append(val.strip())
+
+ raw = meeting.get("raw_content")
+ if isinstance(raw, str) and raw.strip():
+ candidates.append(raw.strip())
+
+ for c in candidates:
+ for tag in ("summary", "notes"):
+ m = re.search(rf"<{tag}>\s*(.*?)\s*{tag}>", c, re.DOTALL)
+ if m:
+ return m.group(1).strip()
+
+ return candidates[0] if candidates else ""
+
+
+def peer_id_from(value: str) -> str:
+ """Normalize a name or email into a Honcho-safe peer ID."""
+ norm = re.sub(r"[^a-z0-9_-]+", "-", value.strip().lower())
+ norm = re.sub(r"-{2,}", "-", norm).strip("-_")
+ return (norm or "peer")[:100]
+
+
+def sanitize(text: str) -> str:
+ """Remove null bytes and control characters."""
+ return re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]", "", text)
+
+
+def parse_date(date_str: str) -> datetime:
+ """Parse Granola's date format into a timezone-aware datetime.
+
+ Raises ValueError if the date string doesn't match any known format.
+ """
+ for fmt in ["%b %d, %Y %I:%M %p", "%b %d, %Y %I:%M:%S %p", "%B %d, %Y %I:%M %p"]:
+ try:
+ return datetime.strptime(date_str, fmt).replace(tzinfo=timezone.utc)
+ except ValueError:
+ continue
+ raise ValueError(f"Unrecognized date format: {date_str!r}")
+
+
+# ---------------------------------------------------------------------------
+# Honcho import helpers
+# ---------------------------------------------------------------------------
+
+def build_messages(
+ peer: Any,
+ content: str,
+ metadata: dict[str, object] | None,
+ created_at: datetime,
+) -> list[Any]:
+ """Build chunked messages for a single peer, attaching metadata to the first chunk."""
+ messages = []
+ content = sanitize(content)
+ for start in range(0, len(content), MAX_MESSAGE_LEN):
+ chunk = content[start:start + MAX_MESSAGE_LEN]
+ msg_meta = metadata if start == 0 else None
+ messages.append(peer.message(chunk, metadata=msg_meta, created_at=created_at))
+ return messages
+
+
+def send_messages(session: Any, messages: list[Any]) -> None:
+ """Send messages to a session in batches of 100."""
+ for batch_start in range(0, len(messages), 100):
+ session.add_messages(messages[batch_start:batch_start + 100])
+
+
+def import_two_person(
+ honcho: Any,
+ session: Any,
+ me_peer_id: str,
+ them_peer_id: str,
+ turns: list[TranscriptTurn],
+ metadata: dict[str, object],
+ created_at: datetime,
+) -> None:
+ """Import a two-person meeting with speaker attribution."""
+ me_peer = honcho.peer(me_peer_id)
+ them_peer = honcho.peer(them_peer_id)
+
+ # Merge consecutive same-speaker turns
+ merged: list[TranscriptTurn] = []
+ for t in turns:
+ if merged and merged[-1].speaker == t.speaker:
+ merged[-1].text += " " + t.text
+ else:
+ merged.append(TranscriptTurn(speaker=t.speaker, text=t.text))
+
+ messages: list[Any] = []
+ for i, t in enumerate(merged):
+ peer = me_peer if t.speaker == "Me" else them_peer
+ msg_meta = metadata if i == 0 else None
+ messages.extend(build_messages(peer, t.text, msg_meta, created_at))
+
+ send_messages(session, messages)
+ print(f" -> Imported as 2-person ({me_peer_id} + {them_peer_id})")
+
+
+def import_summary(
+ honcho: Any,
+ session: Any,
+ me_peer_id: str,
+ meeting: dict[str, Any],
+ metadata: dict[str, object],
+ created_at: datetime,
+) -> None:
+ """Import a meeting as a summary message."""
+ me_peer = honcho.peer(me_peer_id)
+ summary = extract_summary(meeting)
+ if not summary:
+ raw_t = meeting.get("transcript", "")
+ try:
+ parsed = json.loads(raw_t)
+ summary = str(parsed.get("transcript", "")) if isinstance(parsed, dict) else raw_t
+ except (json.JSONDecodeError, TypeError):
+ summary = raw_t
+ summary = summary or "No content available"
+
+ title = meeting.get("title", "Untitled")
+ date = meeting.get("date", "")
+ header = f"Meeting: {title}\nDate: {date}\nParticipants: {meeting.get('participants', '')}\n\n"
+
+ messages = build_messages(me_peer, header + summary, metadata, created_at)
+ send_messages(session, messages)
+ print(" -> Imported as summary")
+
+
+def resolve_them_participant(others: list[Participant]) -> Participant | None:
+ """Ask user to pick which participant is 'Them' from a multi-person meeting."""
+ for j, p in enumerate(others, 1):
+ email_str = f" <{p.email}>" if p.email else ""
+ print(f" {j}. {p.name}{email_str}")
+ idx_str = input(f" Who is 'Them'? [1-{len(others)}]: ").strip()
+ try:
+ return others[int(idx_str) - 1]
+ except (ValueError, IndexError):
+ print(" Invalid selection.")
+ return None
+
+
+def review_meeting(
+ index: int,
+ total: int,
+ meeting: dict[str, Any],
+ participants: ParsedParticipants,
+ turns: list[TranscriptTurn],
+) -> tuple[str, Participant | None]:
+ """Display meeting info and get user's import choice.
+
+ Returns (mode, them_participant) where mode is one of:
+ - "two_person": import with speaker attribution using them_participant
+ - "summary": import as a single summary message
+ - "skip": skip this meeting
+ """
+ title = meeting.get("title", "Untitled")
+ date = meeting.get("date", "")
+ creator = participants.note_creator
+ others = participants.others
+
+ me_turns = sum(1 for t in turns if t.speaker == "Me")
+ them_turns = len(turns) - me_turns
+ total_words = sum(len(t.text.split()) for t in turns)
+
+ print(f"\n{'─' * 60}")
+ print(f" [{index}/{total}] {title}")
+ print(f" Date: {date}")
+ if creator:
+ print(f" You: {creator.name} <{creator.email}>")
+ for j, p in enumerate(others, 1):
+ email_str = f" <{p.email}>" if p.email else ""
+ org_str = f" ({p.org})" if p.org else ""
+ print(f" {j}. {p.name}{email_str}{org_str}")
+
+ has_transcript = bool(meeting.get("transcript"))
+ if turns:
+ print(f" Transcript: {me_turns} Me, {them_turns} Them, ~{total_words} words")
+ if them_turns == 0:
+ print(" ** No 'Them' turns — nobody else spoke **")
+ if total_words < 30:
+ print(" ** Very short — might be empty **")
+ elif has_transcript:
+ raw = meeting["transcript"]
+ print(f" Transcript: present ({len(raw)} chars) but could not parse speaker turns")
+ print(f" Preview: {raw[:200]!r}")
+ else:
+ print(f" Content: {'summary available' if extract_summary(meeting) else 'metadata only'}")
+
+ # Two-person default: exactly one other participant with transcript
+ if len(others) == 1 and them_turns > 0:
+ them_label = others[0].name + (f" <{others[0].email}>" if others[0].email else "")
+ print(f"\n Detected: 2-person call (you + {them_label})")
+ choice = input(" [Enter] 2-person / [s]ummary / [k] skip: ").strip().lower()
+ while choice not in ("", "s", "k"):
+ choice = input(" [Enter] 2-person / [s]ummary / [k] skip: ").strip().lower()
+ if choice == "k":
+ return ("skip", None)
+ if choice == "s":
+ return ("summary", None)
+ return ("two_person", others[0])
+
+ # Multi-person with transcript
+ if len(others) > 1 and them_turns > 0:
+ print(f"\n {len(others)} participants")
+ choice = input(" [Enter] summary / [2] 2-person / [k] skip: ").strip().lower()
+ while choice not in ("", "2", "k"):
+ choice = input(" [Enter] summary / [2] 2-person / [k] skip: ").strip().lower()
+ if choice == "k":
+ return ("skip", None)
+ if choice == "2":
+ them = resolve_them_participant(others)
+ if them is None:
+ return ("summary", None)
+ return ("two_person", them)
+ return ("summary", None)
+
+ # No transcript or no other speakers
+ choice = input(" [Enter] summary / [k] skip: ").strip().lower()
+ while choice not in ("", "k"):
+ choice = input(" [Enter] summary / [k] skip: ").strip().lower()
+ if choice == "k":
+ return ("skip", None)
+ return ("summary", None)
+
+
+# ---------------------------------------------------------------------------
+# Main
+# ---------------------------------------------------------------------------
+
+async def main():
+ print("=" * 60)
+ print(" Granola -> Honcho Meeting Notes Transfer")
+ print("=" * 60)
+
+ if not os.environ.get("HONCHO_API_KEY"):
+ print("\nError: HONCHO_API_KEY not set.")
+ print(" Get your key at: https://app.honcho.dev/api-keys")
+ sys.exit(1)
+
+ async with httpx.AsyncClient(timeout=60.0) as http_client:
+ try:
+ access_token = await authenticate(http_client)
+ meetings = await fetch_all_meetings(http_client, access_token)
+ if not meetings:
+ sys.exit(0)
+
+ from honcho import Honcho
+
+ honcho = Honcho(workspace_id="granola")
+ seen_peers: set[str] = set()
+ results = {"imported": 0, "skipped": 0, "failed": 0}
+
+ print("\n" + "=" * 60)
+ print(" Review each meeting")
+ print("=" * 60)
+
+ for i, m in enumerate(meetings, 1):
+ mid = m.get("id")
+ if not mid:
+ continue
+
+ participants = parse_participants(m.get("participants", ""))
+ turns = parse_transcript_turns(m["transcript"]) if m.get("transcript") else []
+
+ mode, them = review_meeting(i, len(meetings), m, participants, turns)
+
+ if mode == "skip":
+ print(" -> Skipped")
+ results["skipped"] += 1
+ continue
+
+ # Resolve creator peer
+ creator = participants.note_creator
+ me_source = (creator.email or creator.name) if creator else None
+ if not me_source:
+ print(" -> Skipped (no creator identifier)")
+ results["skipped"] += 1
+ continue
+
+ me_peer_id = peer_id_from(me_source)
+ if me_peer_id not in seen_peers:
+ print(f" New peer: {me_source} ({me_peer_id})")
+ seen_peers.add(me_peer_id)
+
+ try:
+ created_at = parse_date(m.get("date", ""))
+ session = honcho.session(f"meeting-{mid}")
+ metadata: dict[str, object] = {
+ "title": m.get("title", "Untitled"),
+ "date": m.get("date", ""),
+ "granola_meeting_id": mid,
+ "mode": mode,
+ }
+
+ if mode == "two_person" and them is not None:
+ them_source = them.email or them.name
+ them_peer_id = peer_id_from(them_source)
+ if them_peer_id not in seen_peers:
+ print(f" New peer: {them_source} ({them_peer_id})")
+ seen_peers.add(them_peer_id)
+ import_two_person(honcho, session, me_peer_id, them_peer_id, turns, metadata, created_at)
+ else:
+ import_summary(honcho, session, me_peer_id, m, metadata, created_at)
+
+ results["imported"] += 1
+
+ except ValueError as e:
+ print(f" -> FAILED: {e}")
+ results["failed"] += 1
+ except Exception as e:
+ print(f" -> FAILED: {e}")
+ traceback.print_exc()
+ results["failed"] += 1
+
+ # Done
+ print("\n" + "=" * 60)
+ print(" Transfer Complete!")
+ print("=" * 60)
+ print(f"\n Imported: {results['imported']}")
+ print(f" Skipped: {results['skipped']}")
+ print(f" Failed: {results['failed']}")
+ print(" Workspace: granola")
+ print(f" Peers: {sorted(seen_peers)}")
+
+ except KeyboardInterrupt:
+ print("\n\nAborted.")
+ sys.exit(0)
+ except Exception as e:
+ print(f"\nTransfer failed: {e}")
+ traceback.print_exc()
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ asyncio.run(main())
+```
+
+
+
+## Next Steps
+
+
+
+ See how the Granola integration maps to common Honcho patterns.
+
+
+ Source code and example script.
+
+
diff --git a/examples/gmail/honcho_gmail.py b/examples/gmail/honcho_gmail.py
new file mode 100644
index 00000000..ec7bb8bd
--- /dev/null
+++ b/examples/gmail/honcho_gmail.py
@@ -0,0 +1,374 @@
+#!/usr/bin/env python3
+"""Load Gmail messages into Honcho.
+
+Uses the Gmail API directly (with OAuth) to fetch emails and the Honcho Python SDK to store them.
+Each Gmail thread becomes a Honcho session, each sender becomes a peer.
+
+Prerequisites:
+1. Create a Google Cloud project and enable the Gmail API
+2. Create OAuth 2.0 credentials (Desktop app type)
+3. Download the credentials JSON (client_secret_*.json) into this directory
+4. Install dependencies:
+ pip install google-api-python-client google-auth-oauthlib honcho-ai
+
+On first run, a browser window will open for OAuth consent. After authorizing,
+a 'token.json' file will be created to store your credentials for future runs.
+"""
+
+import argparse
+import base64
+import glob
+import os
+import re
+import time
+from datetime import datetime, timezone
+from email.header import decode_header, make_header
+from email.utils import getaddresses, parseaddr
+
+from google.auth.transport.requests import Request
+from google.oauth2.credentials import Credentials
+from google_auth_oauthlib.flow import InstalledAppFlow
+from googleapiclient.discovery import build
+from googleapiclient.errors import HttpError
+
+SCOPES = ["https://www.googleapis.com/auth/gmail.readonly"]
+PEER_ID_PATTERN = re.compile(r"^[a-zA-Z0-9_-]+$")
+
+
+def find_credentials() -> str:
+ """Find a Google OAuth credentials file in the current directory."""
+ matches = glob.glob("client_secret*.json")
+ if matches:
+ return matches[0]
+ raise FileNotFoundError(
+ "No client_secret*.json file found.\n"
+ "Download OAuth credentials from Google Cloud Console:\n"
+ "1. Go to console.cloud.google.com\n"
+ "2. Create/select a project and enable Gmail API\n"
+ "3. Create OAuth 2.0 credentials (Desktop app)\n"
+ "4. Download the JSON into this directory"
+ )
+
+
+def get_gmail_service(credentials_file: str | None = None, token_file: str = "token.json"):
+ """Authenticate and return a Gmail API service instance."""
+ creds = None
+
+ if os.path.exists(token_file):
+ creds = Credentials.from_authorized_user_file(token_file, SCOPES)
+
+ if not creds or not creds.valid:
+ if creds and creds.expired and creds.refresh_token:
+ print("Refreshing expired credentials...")
+ creds.refresh(Request())
+ else:
+ if credentials_file is None:
+ credentials_file = find_credentials()
+ print(f"Using credentials: {credentials_file}")
+ print("Opening browser for OAuth consent...")
+ flow = InstalledAppFlow.from_client_secrets_file(credentials_file, SCOPES)
+ creds = flow.run_local_server(port=0)
+
+ with open(token_file, "w") as token:
+ token.write(creds.to_json())
+ print(f"Credentials saved to {token_file}")
+
+ return build("gmail", "v1", credentials=creds)
+
+
+def list_threads(service, query: str = None, label_ids: list = None, max_results: int = 10) -> list[dict]:
+ """List Gmail threads with pagination support."""
+ all_threads = []
+ page_token = None
+
+ while len(all_threads) < max_results:
+ try:
+ params = {
+ "userId": "me",
+ "maxResults": min(100, max_results - len(all_threads)),
+ }
+ if query:
+ params["q"] = query
+ if label_ids:
+ params["labelIds"] = label_ids
+ if page_token:
+ params["pageToken"] = page_token
+
+ response = service.users().threads().list(**params).execute()
+ threads = response.get("threads", [])
+ all_threads.extend(threads)
+
+ page_token = response.get("nextPageToken")
+ if not page_token:
+ break
+
+ except HttpError as e:
+ print(f"Error listing threads: {e}")
+ break
+
+ return all_threads[:max_results]
+
+
+def get_thread(service, thread_id: str) -> dict:
+ """Fetch a complete Gmail thread with all messages."""
+ try:
+ return service.users().threads().get(
+ userId="me",
+ id=thread_id,
+ format="full"
+ ).execute()
+ except HttpError as e:
+ print(f"Error fetching thread {thread_id}: {e}")
+ return {}
+
+
+def _decode_header_str(header: str) -> str:
+ """Decode an RFC 2047 encoded header string to plain Unicode."""
+ return str(make_header(decode_header(header)))
+
+
+def extract_email(from_header: str) -> str:
+ """Extract bare email from an RFC 5322 header value."""
+ _, addr = parseaddr(_decode_header_str(from_header))
+ return addr.lower().strip()
+
+
+def extract_name(from_header: str) -> str:
+ """Extract display name from an RFC 5322 header value."""
+ name, _ = parseaddr(_decode_header_str(from_header))
+ return name.strip() or from_header.strip()
+
+
+def decode_body(payload: dict) -> str:
+ """Recursively extract plain text from a Gmail message payload."""
+ if payload.get("mimeType") == "text/plain":
+ data = payload.get("body", {}).get("data", "")
+ if data:
+ return base64.urlsafe_b64decode(data).decode("utf-8", errors="replace")
+
+ parts = payload.get("parts", [])
+ for part in parts:
+ text = decode_body(part)
+ if text:
+ return text
+ return ""
+
+
+def strip_quoted_replies(text: str) -> str:
+ """Strip quoted reply text from an email body, keeping only the new content."""
+ lines = text.split("\n")
+ clean_lines = []
+ for line in lines:
+ stripped = line.strip()
+ if re.match(r"^On .+wrote:\s*$", stripped):
+ break
+ if stripped.startswith("---------- Forwarded message"):
+ break
+ if stripped.startswith(">"):
+ break
+ if re.match(r"^[-_]{10,}$", stripped):
+ break
+ clean_lines.append(line)
+ return "\n".join(clean_lines).rstrip()
+
+
+def parse_address_list(header: str) -> list[str]:
+ """Parse a comma-separated email header into individual addresses."""
+ if not header.strip():
+ return []
+ decoded = _decode_header_str(header)
+ return [
+ f"{name} <{addr}>" if name else addr
+ for name, addr in getaddresses([decoded])
+ if addr
+ ]
+
+
+def peer_id_from_email(email: str) -> str:
+ """Convert email to a valid Honcho peer ID."""
+ peer_id = re.sub(r"[^A-Za-z0-9_-]+", "-", email).strip("-").lower()
+ peer_id = re.sub(r"-{2,}", "-", peer_id)
+ if not peer_id:
+ peer_id = "unknown-peer"
+
+ if not PEER_ID_PATTERN.fullmatch(peer_id):
+ raise ValueError(f"Generated peer ID is invalid: {peer_id!r}")
+ return peer_id
+
+
+def fetch_thread_messages(service, thread_id: str) -> list[dict]:
+ """Fetch all messages in a Gmail thread with full content."""
+ data = get_thread(service, thread_id)
+ messages = []
+
+ for msg in data.get("messages", []):
+ headers = {h["name"]: h["value"] for h in msg.get("payload", {}).get("headers", [])}
+ body = strip_quoted_replies(decode_body(msg.get("payload", {})))
+ ts = int(msg.get("internalDate", "0")) / 1000
+
+ messages.append({
+ "id": msg["id"],
+ "thread_id": msg["threadId"],
+ "from": headers.get("From", ""),
+ "to": headers.get("To", ""),
+ "cc": headers.get("Cc", ""),
+ "bcc": headers.get("Bcc", ""),
+ "subject": headers.get("Subject", ""),
+ "date": headers.get("Date", ""),
+ "timestamp": datetime.fromtimestamp(ts, tz=timezone.utc),
+ "body": body.strip(),
+ "labels": msg.get("labelIds", []),
+ "snippet": msg.get("snippet", ""),
+ })
+
+ return messages
+
+
+def main():
+ parser = argparse.ArgumentParser(description="Load Gmail messages into Honcho")
+ parser.add_argument("--workspace", "-w", default="gmail", help="Honcho workspace ID (default: gmail)")
+ parser.add_argument("--query", "-q", default=None, help="Gmail search query (e.g. 'from:alice@example.com')")
+ parser.add_argument("--label", "-l", default=None, help="Gmail label to filter by (e.g. INBOX)")
+ parser.add_argument("--max-threads", "-n", type=int, default=10, help="Max threads to fetch (default: 10)")
+ parser.add_argument("--dry-run", action="store_true", help="Print what would be loaded without writing to Honcho")
+ parser.add_argument("--credentials", "-c", default=None, help="Path to OAuth credentials JSON (auto-detects client_secret*.json)")
+ parser.add_argument("--token", "-t", default="token.json", help="Path to store/load access token")
+ args = parser.parse_args()
+
+ # Authenticate
+ print("Authenticating with Gmail API...")
+ service = get_gmail_service(args.credentials, args.token)
+ print(" Authenticated successfully!")
+
+ label_ids = [args.label] if args.label else None
+
+ # List threads
+ print(f"\nFetching up to {args.max_threads} threads from Gmail...")
+ threads = list_threads(service, query=args.query, label_ids=label_ids, max_results=args.max_threads)
+ print(f" Found {len(threads)} threads")
+
+ if not threads:
+ print("No threads found. Try adjusting --query or --label.")
+ return
+
+ # Fetch full messages for each thread
+ all_thread_messages = {}
+ seen_peers = {}
+
+ def register_peer(addr: str):
+ email = extract_email(addr)
+ if email and email not in seen_peers:
+ name = extract_name(addr)
+ if name.lower().strip() == email or "@" in name:
+ name = email.split("@")[0].replace(".", " ").title()
+ seen_peers[email] = {
+ "name": name,
+ "peer_id": peer_id_from_email(email),
+ "email": email,
+ }
+
+ for i, t in enumerate(threads):
+ tid = t["id"]
+ print(f" Fetching thread {i+1}/{len(threads)}: {tid}")
+ msgs = fetch_thread_messages(service, tid)
+ all_thread_messages[tid] = msgs
+ for m in msgs:
+ register_peer(m["from"])
+ for addr in parse_address_list(m["to"]):
+ register_peer(addr)
+ for addr in parse_address_list(m["cc"]):
+ register_peer(addr)
+ for addr in parse_address_list(m["bcc"]):
+ register_peer(addr)
+
+ # Summary
+ total_msgs = sum(len(v) for v in all_thread_messages.values())
+ print("\nSummary:")
+ print(f" Threads: {len(all_thread_messages)}")
+ print(f" Messages: {total_msgs}")
+ print(f" Unique participants: {len(seen_peers)}")
+ for email, info in seen_peers.items():
+ print(f" {info['peer_id']} ({info['name']} <{email}>)")
+
+ if args.dry_run:
+ print("\n[DRY RUN] Would create the above in Honcho. Showing first message per thread:")
+ for tid, msgs in all_thread_messages.items():
+ m = msgs[0]
+ body_preview = m["body"][:120].replace("\n", " ") if m["body"] else m["snippet"][:120]
+ print(f" Thread {tid}: {m['subject']}")
+ print(f" {m['from']} @ {m['date']}")
+ print(f" {body_preview}...")
+ return
+
+ # Load into Honcho
+ from honcho import Honcho
+
+ print(f"\nLoading into Honcho workspace '{args.workspace}'...")
+ honcho = Honcho(workspace_id=args.workspace)
+
+ # Create peers
+ peers = {}
+ for i, (email, info) in enumerate(seen_peers.items()):
+ if i > 0 and i % 4 == 0:
+ time.sleep(1)
+ peers[email] = honcho.peer(info["peer_id"], metadata={
+ "email": email,
+ "name": info["name"],
+ "source": "gmail",
+ })
+ print(f" Peer: {info['peer_id']}")
+
+ # Create sessions and messages per thread
+ for tid, msgs in all_thread_messages.items():
+ subject = msgs[0]["subject"] if msgs else "No subject"
+ session_id = f"gmail-thread-{tid}"
+
+ thread_peer_emails = set()
+ for m in msgs:
+ thread_peer_emails.add(extract_email(m["from"]))
+ for addr in parse_address_list(m["to"]):
+ thread_peer_emails.add(extract_email(addr))
+ for addr in parse_address_list(m["cc"]):
+ thread_peer_emails.add(extract_email(addr))
+ for addr in parse_address_list(m["bcc"]):
+ thread_peer_emails.add(extract_email(addr))
+ thread_peers = [peers[e] for e in thread_peer_emails if e in peers]
+
+ session = honcho.session(session_id, metadata={
+ "gmail_thread_id": tid,
+ "subject": subject,
+ "source": "gmail",
+ "message_count": len(msgs),
+ })
+ session.add_peers(thread_peers)
+
+ honcho_msgs = []
+ for m in msgs:
+ email = extract_email(m["from"])
+ peer = peers.get(email)
+ if not peer:
+ continue
+ content = m["body"] if m["body"] else m["snippet"]
+ if not content:
+ continue
+ honcho_msgs.append(peer.message(
+ content,
+ metadata={
+ "gmail_id": m["id"],
+ "subject": m["subject"],
+ "from": m["from"],
+ "to": m["to"],
+ "labels": m["labels"],
+ },
+ created_at=m["timestamp"],
+ ))
+
+ if honcho_msgs:
+ session.add_messages(honcho_msgs)
+ print(f" Session {session_id}: {len(honcho_msgs)} messages — {subject[:60]}")
+
+ print(f"\nDone! Loaded {total_msgs} messages into workspace '{args.workspace}'.")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/examples/granola/honcho_granola.py b/examples/granola/honcho_granola.py
new file mode 100644
index 00000000..b3881eba
--- /dev/null
+++ b/examples/granola/honcho_granola.py
@@ -0,0 +1,751 @@
+#!/usr/bin/env python3
+"""Load Granola meeting notes into Honcho.
+
+Uses the Granola MCP server (with OAuth) to fetch meetings and the Honcho Python SDK
+to store them. Each meeting becomes a Honcho session. Two-person meetings get full
+speaker attribution; multi-person meetings are stored as summaries.
+
+Prerequisites:
+ pip install honcho-ai httpx
+
+Environment Variables:
+ HONCHO_API_KEY - Your Honcho API key (get from app.honcho.dev/api-keys)
+
+Usage:
+ python honcho_granola.py
+"""
+
+import asyncio
+import base64
+import hashlib
+import json
+import os
+import re
+import secrets
+import sys
+import threading
+import traceback
+import webbrowser
+from dataclasses import dataclass, field
+from datetime import datetime, timezone
+from http.server import HTTPServer, BaseHTTPRequestHandler
+from typing import Any
+from urllib.parse import parse_qs, urlencode, urlparse
+
+import httpx
+
+
+@dataclass
+class Participant:
+ name: str
+ email: str | None = None
+ org: str | None = None
+
+
+@dataclass
+class ParsedParticipants:
+ note_creator: Participant | None = None
+ others: list[Participant] = field(default_factory=list)
+
+
+@dataclass
+class TranscriptTurn:
+ speaker: str
+ text: str
+
+
+# Granola MCP + OAuth endpoints
+GRANOLA_MCP_URL = "https://mcp.granola.ai/mcp"
+AUTH_BASE = "https://mcp-auth.granola.ai"
+OAUTH_REDIRECT_PORT = 8765
+OAUTH_REDIRECT_URI = f"http://localhost:{OAUTH_REDIRECT_PORT}/callback"
+
+# Honcho message size limit (25000 max, leave headroom)
+MAX_MESSAGE_LEN = 24000
+
+
+# ---------------------------------------------------------------------------
+# OAuth callback handler (must be a class for BaseHTTPRequestHandler)
+# ---------------------------------------------------------------------------
+
+class _OAuthCallback(BaseHTTPRequestHandler):
+ auth_result: dict[str, str | None] = {"code": None, "error": None}
+
+ def do_GET(self):
+ params = parse_qs(urlparse(self.path).query)
+ if "code" in params:
+ _OAuthCallback.auth_result["code"] = params["code"][0]
+ self.send_response(200)
+ self.send_header("Content-Type", "text/html")
+ self.end_headers()
+ self.wfile.write(b"Authenticated! You can close this window.
")
+ elif "error" in params:
+ _OAuthCallback.auth_result["error"] = params.get("error_description", params["error"])[0]
+ self.send_response(400)
+ self.send_header("Content-Type", "text/html")
+ self.end_headers()
+ self.wfile.write(f"Error: {_OAuthCallback.auth_result['error']}
".encode())
+ else:
+ self.send_response(404)
+ self.end_headers()
+
+ def log_message(self, fmt, *args):
+ pass
+
+
+# ---------------------------------------------------------------------------
+# Granola OAuth + MCP
+# ---------------------------------------------------------------------------
+
+async def authenticate(http_client: httpx.AsyncClient) -> str:
+ """Perform OAuth (DCR + PKCE) with Granola. Returns access token."""
+ _OAuthCallback.auth_result = {"code": None, "error": None}
+
+ print("\nAuthenticating with Granola...")
+
+ # Register client (DCR)
+ resp = await http_client.post(
+ f"{AUTH_BASE}/oauth2/register",
+ json={
+ "client_name": "Granola to Honcho Transfer",
+ "redirect_uris": [OAUTH_REDIRECT_URI],
+ "grant_types": ["authorization_code"],
+ "response_types": ["code"],
+ "token_endpoint_auth_method": "none",
+ },
+ )
+ if resp.status_code not in (200, 201):
+ raise RuntimeError(f"Client registration failed: {resp.status_code}")
+ client_id = resp.json().get("client_id")
+
+ # PKCE
+ verifier = secrets.token_urlsafe(32)
+ challenge = base64.urlsafe_b64encode(hashlib.sha256(verifier.encode()).digest()).rstrip(b"=").decode()
+
+ # Browser auth
+ auth_url = f"{AUTH_BASE}/oauth2/authorize?" + urlencode({
+ "client_id": client_id,
+ "redirect_uri": OAUTH_REDIRECT_URI,
+ "response_type": "code",
+ "state": "granola-honcho-transfer",
+ "code_challenge": challenge,
+ "code_challenge_method": "S256",
+ })
+
+ server = HTTPServer(("localhost", OAUTH_REDIRECT_PORT), _OAuthCallback)
+ thread = threading.Thread(target=server.handle_request)
+ thread.start()
+
+ print(" Opening browser for authentication...")
+ webbrowser.open(auth_url)
+ thread.join(timeout=120)
+ server.server_close()
+
+ auth_result = _OAuthCallback.auth_result
+ if auth_result["error"]:
+ raise RuntimeError(f"Authentication failed: {auth_result['error']}")
+ if not auth_result["code"]:
+ raise RuntimeError("Authentication timed out")
+
+ # Exchange code for token
+ resp = await http_client.post(
+ f"{AUTH_BASE}/oauth2/token",
+ data={
+ "grant_type": "authorization_code",
+ "code": auth_result["code"],
+ "redirect_uri": OAUTH_REDIRECT_URI,
+ "client_id": client_id,
+ "code_verifier": verifier,
+ },
+ headers={"Content-Type": "application/x-www-form-urlencoded"},
+ )
+ if resp.status_code != 200:
+ raise RuntimeError(f"Token exchange failed: {resp.status_code}")
+
+ print(" Authenticated successfully!")
+ return resp.json()["access_token"]
+
+
+async def call_mcp_tool(
+ http_client: httpx.AsyncClient,
+ access_token: str,
+ tool_name: str,
+ arguments: dict[str, Any] | None = None,
+) -> dict[str, Any]:
+ """Call a Granola MCP tool, handling both JSON and SSE responses."""
+ resp = await http_client.post(
+ GRANOLA_MCP_URL,
+ json={
+ "jsonrpc": "2.0",
+ "id": 1,
+ "method": "tools/call",
+ "params": {"name": tool_name, "arguments": arguments or {}},
+ },
+ headers={
+ "Authorization": f"Bearer {access_token}",
+ "Content-Type": "application/json",
+ "Accept": "application/json, text/event-stream",
+ },
+ )
+ if resp.status_code != 200:
+ raise RuntimeError(f"MCP call failed: {resp.status_code} - {resp.text}")
+
+ # SSE response
+ if "text/event-stream" in resp.headers.get("content-type", ""):
+ result = None
+ for line in resp.text.split("\n"):
+ if line.strip().startswith("data: "):
+ try:
+ parsed = json.loads(line.strip()[6:])
+ if "result" in parsed:
+ result = parsed
+ elif "error" in parsed:
+ raise RuntimeError(f"MCP error: {parsed['error']}")
+ except json.JSONDecodeError:
+ continue
+ if result:
+ final = result.get("result", {})
+ return final if isinstance(final, dict) else {"result": final}
+ raise RuntimeError("No result in SSE response")
+
+ # JSON response
+ result = resp.json()
+ if "error" in result:
+ raise RuntimeError(f"MCP error: {result['error']}")
+ return result.get("result", {})
+
+
+def extract_mcp_text(result: dict[str, Any]) -> str:
+ """Extract text from the first content block of an MCP result.
+
+ Raises ValueError if the response structure is unexpected.
+ """
+ content = result.get("content", [])
+ if not isinstance(content, list) or not content:
+ raise ValueError(f"MCP response missing content array: {list(result.keys())}")
+ first = content[0]
+ if not isinstance(first, dict) or "text" not in first:
+ raise ValueError(f"MCP content block missing 'text' field: {first}")
+ return str(first["text"])
+
+
+# ---------------------------------------------------------------------------
+# Granola data fetching
+# ---------------------------------------------------------------------------
+
+async def list_meetings(
+ http_client: httpx.AsyncClient, access_token: str, limit: int = 100,
+) -> list[dict[str, Any]]:
+ """List meetings from Granola MCP. Parses Granola's XML-like response format."""
+ result = await call_mcp_tool(http_client, access_token, "list_meetings", {"limit": limit})
+ text = extract_mcp_text(result)
+
+ meetings: list[dict[str, Any]] = []
+ for match in re.finditer(r'", match.end())
+ block = text[match.end():block_end] if block_end != -1 else ""
+ p_match = re.search(r"\s*(.*?)\s*", block, re.DOTALL)
+ meetings.append({
+ "id": mid,
+ "title": title,
+ "date": date,
+ "participants": p_match.group(1).strip() if p_match else "",
+ })
+
+ return meetings
+
+
+async def get_meeting_details(
+ http_client: httpx.AsyncClient, access_token: str, meeting_id: str,
+) -> dict[str, Any]:
+ """Get full meeting details including notes."""
+ result = await call_mcp_tool(http_client, access_token, "get_meetings", {"meeting_ids": [meeting_id]})
+ text = extract_mcp_text(result)
+ return {"id": meeting_id, "raw_content": text}
+
+
+async def get_meeting_transcript(
+ http_client: httpx.AsyncClient, access_token: str, meeting_id: str,
+ max_retries: int = 3,
+) -> str | None:
+ """Get transcript for a meeting (paid tiers only).
+
+ Retries on rate limit responses with exponential backoff.
+ """
+ for attempt in range(max_retries):
+ try:
+ result = await call_mcp_tool(http_client, access_token, "get_meeting_transcript", {"meeting_id": meeting_id})
+ text = extract_mcp_text(result)
+ except Exception as e:
+ print(f" Transcript unavailable: {e}")
+ return None
+
+ if not text or "no transcript" in text.lower():
+ return None
+
+ # Granola returns rate limit errors as content text, not HTTP errors
+ if "rate limit" in text.lower():
+ wait = 2 ** attempt * 3 # 3s, 6s, 12s
+ print(f" ⚠ Granola rate limit hit (attempt {attempt + 1}/{max_retries}), waiting {wait}s...")
+ await asyncio.sleep(wait)
+ continue
+
+ return text
+
+ print(f" ⚠ Transcript skipped after {max_retries} rate limit retries")
+ return None
+
+
+async def fetch_all_meetings(
+ http_client: httpx.AsyncClient, access_token: str,
+) -> list[dict[str, Any]]:
+ """Fetch meeting list and enrich each with transcript and details."""
+ print("\nFetching meetings from Granola...")
+ meetings = await list_meetings(http_client, access_token, limit=500)
+ if not meetings:
+ print("No meetings found.")
+ return []
+ print(f" Found {len(meetings)} meetings. Fetching content...\n")
+
+ for i, m in enumerate(meetings, 1):
+ mid = m.get("id")
+ if not mid:
+ continue
+
+ transcript = await get_meeting_transcript(http_client, access_token, mid)
+ if transcript:
+ m["transcript"] = transcript
+
+ try:
+ m.update(await get_meeting_details(http_client, access_token, mid))
+ except Exception as exc:
+ print(f" Failed to fetch details for {mid}: {exc}")
+
+ has_t = "transcript" in m
+ has_s = bool(extract_summary(m))
+ label = "transcript+summary" if has_t and has_s else "transcript only" if has_t else "summary only" if has_s else "basic only"
+ print(f" [{i}/{len(meetings)}] {label}: {m.get('title', 'Untitled')[:45]}")
+ await asyncio.sleep(1.5) # rate limit
+
+ return meetings
+
+
+# ---------------------------------------------------------------------------
+# Parsing helpers
+# ---------------------------------------------------------------------------
+
+def parse_participants(participants_str: str) -> ParsedParticipants:
+ """Parse Granola's participant string into structured participants.
+
+ Warns on unparseable entries instead of silently dropping them.
+ """
+ result = ParsedParticipants()
+ if not participants_str:
+ return result
+
+ # Split on commas, but not inside angle brackets
+ entries, current, depth = [], [], 0
+ for ch in participants_str:
+ if ch == "<":
+ depth += 1
+ elif ch == ">":
+ depth = max(depth - 1, 0)
+ elif ch == "," and depth == 0:
+ entries.append("".join(current))
+ current = []
+ continue
+ current.append(ch)
+ if current:
+ entries.append("".join(current))
+
+ for entry in entries:
+ entry = entry.strip()
+ if not entry:
+ continue
+
+ is_creator = "(note creator)" in entry
+ clean = entry.replace("(note creator)", "").strip()
+
+ email_match = re.search(r"<([^>]+)>", clean)
+ email = email_match.group(1) if email_match else None
+ name = re.sub(r"\s*<[^>]+>", "", clean).strip()
+
+ if not name:
+ print(f" Warning: could not parse participant entry: {entry!r}")
+ continue
+
+ org = None
+ org_match = re.match(r"(.+?)\s+from\s+(.+)", name)
+ if org_match:
+ name, org = org_match.group(1).strip(), org_match.group(2).strip()
+
+ person = Participant(name=name, email=email, org=org)
+ if is_creator:
+ result.note_creator = person
+ else:
+ result.others.append(person)
+
+ return result
+
+
+def parse_transcript_turns(raw: str) -> list[TranscriptTurn]:
+ """Split a Granola transcript into speaker turns."""
+ # Unwrap JSON wrapper if present
+ try:
+ parsed = json.loads(raw)
+ if isinstance(parsed, dict) and "transcript" in parsed:
+ raw = str(parsed["transcript"])
+ except (json.JSONDecodeError, TypeError):
+ pass
+
+ parts = re.split(r"(?:^|\s{2,})(Me|Them):\s*", raw)
+ turns: list[TranscriptTurn] = []
+ i = 1
+ while i < len(parts) - 1:
+ text = parts[i + 1].strip()
+ if text:
+ turns.append(TranscriptTurn(speaker=parts[i], text=text))
+ i += 2
+ return turns
+
+
+def extract_summary(meeting: dict[str, Any]) -> str:
+ """Extract best available summary text from meeting data."""
+ candidates = []
+ for key in ("summary", "notes", "note", "meeting_notes", "description"):
+ val = meeting.get(key)
+ if isinstance(val, str) and val.strip():
+ candidates.append(val.strip())
+
+ raw = meeting.get("raw_content")
+ if isinstance(raw, str) and raw.strip():
+ candidates.append(raw.strip())
+
+ for c in candidates:
+ for tag in ("summary", "notes"):
+ m = re.search(rf"<{tag}>\s*(.*?)\s*{tag}>", c, re.DOTALL)
+ if m:
+ return m.group(1).strip()
+
+ return candidates[0] if candidates else ""
+
+
+def peer_id_from(value: str) -> str:
+ """Normalize a name or email into a Honcho-safe peer ID."""
+ norm = re.sub(r"[^a-z0-9_-]+", "-", value.strip().lower())
+ norm = re.sub(r"-{2,}", "-", norm).strip("-_")
+ return (norm or "peer")[:100]
+
+
+def sanitize(text: str) -> str:
+ """Remove null bytes and control characters."""
+ return re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]", "", text)
+
+
+def parse_date(date_str: str) -> datetime:
+ """Parse Granola's date format into a timezone-aware datetime.
+
+ Raises ValueError if the date string doesn't match any known format.
+ """
+ for fmt in ["%b %d, %Y %I:%M %p", "%b %d, %Y %I:%M:%S %p", "%B %d, %Y %I:%M %p"]:
+ try:
+ return datetime.strptime(date_str, fmt).replace(tzinfo=timezone.utc)
+ except ValueError:
+ continue
+ raise ValueError(f"Unrecognized date format: {date_str!r}")
+
+
+# ---------------------------------------------------------------------------
+# Honcho import helpers
+# ---------------------------------------------------------------------------
+
+def build_messages(
+ peer: Any,
+ content: str,
+ metadata: dict[str, object] | None,
+ created_at: datetime,
+) -> list[Any]:
+ """Build chunked messages for a single peer, attaching metadata to the first chunk."""
+ messages = []
+ content = sanitize(content)
+ for start in range(0, len(content), MAX_MESSAGE_LEN):
+ chunk = content[start:start + MAX_MESSAGE_LEN]
+ msg_meta = metadata if start == 0 else None
+ messages.append(peer.message(chunk, metadata=msg_meta, created_at=created_at))
+ return messages
+
+
+def send_messages(session: Any, messages: list[Any]) -> None:
+ """Send messages to a session in batches of 100."""
+ for batch_start in range(0, len(messages), 100):
+ session.add_messages(messages[batch_start:batch_start + 100])
+
+
+def import_two_person(
+ honcho: Any,
+ session: Any,
+ me_peer_id: str,
+ them_peer_id: str,
+ turns: list[TranscriptTurn],
+ metadata: dict[str, object],
+ created_at: datetime,
+) -> None:
+ """Import a two-person meeting with speaker attribution."""
+ me_peer = honcho.peer(me_peer_id)
+ them_peer = honcho.peer(them_peer_id)
+
+ # Merge consecutive same-speaker turns
+ merged: list[TranscriptTurn] = []
+ for t in turns:
+ if merged and merged[-1].speaker == t.speaker:
+ merged[-1].text += " " + t.text
+ else:
+ merged.append(TranscriptTurn(speaker=t.speaker, text=t.text))
+
+ messages: list[Any] = []
+ for i, t in enumerate(merged):
+ peer = me_peer if t.speaker == "Me" else them_peer
+ msg_meta = metadata if i == 0 else None
+ messages.extend(build_messages(peer, t.text, msg_meta, created_at))
+
+ send_messages(session, messages)
+ print(f" -> Imported as 2-person ({me_peer_id} + {them_peer_id})")
+
+
+def import_summary(
+ honcho: Any,
+ session: Any,
+ me_peer_id: str,
+ meeting: dict[str, Any],
+ metadata: dict[str, object],
+ created_at: datetime,
+) -> None:
+ """Import a meeting as a summary message."""
+ me_peer = honcho.peer(me_peer_id)
+ summary = extract_summary(meeting)
+ if not summary:
+ raw_t = meeting.get("transcript", "")
+ try:
+ parsed = json.loads(raw_t)
+ summary = str(parsed.get("transcript", "")) if isinstance(parsed, dict) else raw_t
+ except (json.JSONDecodeError, TypeError):
+ summary = raw_t
+ summary = summary or "No content available"
+
+ title = meeting.get("title", "Untitled")
+ date = meeting.get("date", "")
+ header = f"Meeting: {title}\nDate: {date}\nParticipants: {meeting.get('participants', '')}\n\n"
+
+ messages = build_messages(me_peer, header + summary, metadata, created_at)
+ send_messages(session, messages)
+ print(" -> Imported as summary")
+
+
+def resolve_them_participant(others: list[Participant]) -> Participant | None:
+ """Ask user to pick which participant is 'Them' from a multi-person meeting."""
+ for j, p in enumerate(others, 1):
+ email_str = f" <{p.email}>" if p.email else ""
+ print(f" {j}. {p.name}{email_str}")
+ idx_str = input(f" Who is 'Them'? [1-{len(others)}]: ").strip()
+ try:
+ return others[int(idx_str) - 1]
+ except (ValueError, IndexError):
+ print(" Invalid selection.")
+ return None
+
+
+def review_meeting(
+ index: int,
+ total: int,
+ meeting: dict[str, Any],
+ participants: ParsedParticipants,
+ turns: list[TranscriptTurn],
+) -> tuple[str, Participant | None]:
+ """Display meeting info and get user's import choice.
+
+ Returns (mode, them_participant) where mode is one of:
+ - "two_person": import with speaker attribution using them_participant
+ - "summary": import as a single summary message
+ - "skip": skip this meeting
+ """
+ title = meeting.get("title", "Untitled")
+ date = meeting.get("date", "")
+ creator = participants.note_creator
+ others = participants.others
+
+ me_turns = sum(1 for t in turns if t.speaker == "Me")
+ them_turns = len(turns) - me_turns
+ total_words = sum(len(t.text.split()) for t in turns)
+
+ print(f"\n{'─' * 60}")
+ print(f" [{index}/{total}] {title}")
+ print(f" Date: {date}")
+ if creator:
+ print(f" You: {creator.name} <{creator.email}>")
+ for j, p in enumerate(others, 1):
+ email_str = f" <{p.email}>" if p.email else ""
+ org_str = f" ({p.org})" if p.org else ""
+ print(f" {j}. {p.name}{email_str}{org_str}")
+
+ has_transcript = bool(meeting.get("transcript"))
+ if turns:
+ print(f" Transcript: {me_turns} Me, {them_turns} Them, ~{total_words} words")
+ if them_turns == 0:
+ print(" ** No 'Them' turns — nobody else spoke **")
+ if total_words < 30:
+ print(" ** Very short — might be empty **")
+ elif has_transcript:
+ raw = meeting["transcript"]
+ print(f" Transcript: present ({len(raw)} chars) but could not parse speaker turns")
+ print(f" Preview: {raw[:200]!r}")
+ else:
+ print(f" Content: {'summary available' if extract_summary(meeting) else 'metadata only'}")
+
+ # Two-person default: exactly one other participant with transcript
+ if len(others) == 1 and them_turns > 0:
+ them_label = others[0].name + (f" <{others[0].email}>" if others[0].email else "")
+ print(f"\n Detected: 2-person call (you + {them_label})")
+ choice = input(" [Enter] 2-person / [s]ummary / [k] skip: ").strip().lower()
+ while choice not in ("", "s", "k"):
+ choice = input(" [Enter] 2-person / [s]ummary / [k] skip: ").strip().lower()
+ if choice == "k":
+ return ("skip", None)
+ if choice == "s":
+ return ("summary", None)
+ return ("two_person", others[0])
+
+ # Multi-person with transcript
+ if len(others) > 1 and them_turns > 0:
+ print(f"\n {len(others)} participants")
+ choice = input(" [Enter] summary / [2] 2-person / [k] skip: ").strip().lower()
+ while choice not in ("", "2", "k"):
+ choice = input(" [Enter] summary / [2] 2-person / [k] skip: ").strip().lower()
+ if choice == "k":
+ return ("skip", None)
+ if choice == "2":
+ them = resolve_them_participant(others)
+ if them is None:
+ return ("summary", None)
+ return ("two_person", them)
+ return ("summary", None)
+
+ # No transcript or no other speakers
+ choice = input(" [Enter] summary / [k] skip: ").strip().lower()
+ while choice not in ("", "k"):
+ choice = input(" [Enter] summary / [k] skip: ").strip().lower()
+ if choice == "k":
+ return ("skip", None)
+ return ("summary", None)
+
+
+# ---------------------------------------------------------------------------
+# Main
+# ---------------------------------------------------------------------------
+
+async def main():
+ print("=" * 60)
+ print(" Granola -> Honcho Meeting Notes Transfer")
+ print("=" * 60)
+
+ if not os.environ.get("HONCHO_API_KEY"):
+ print("\nError: HONCHO_API_KEY not set.")
+ print(" Get your key at: https://app.honcho.dev/api-keys")
+ sys.exit(1)
+
+ async with httpx.AsyncClient(timeout=60.0) as http_client:
+ try:
+ access_token = await authenticate(http_client)
+ meetings = await fetch_all_meetings(http_client, access_token)
+ if not meetings:
+ sys.exit(0)
+
+ from honcho import Honcho
+
+ honcho = Honcho(workspace_id="granola_test")
+ seen_peers: set[str] = set()
+ results = {"imported": 0, "skipped": 0, "failed": 0}
+
+ print("\n" + "=" * 60)
+ print(" Review each meeting")
+ print("=" * 60)
+
+ for i, m in enumerate(meetings, 1):
+ mid = m.get("id")
+ if not mid:
+ continue
+
+ participants = parse_participants(m.get("participants", ""))
+ turns = parse_transcript_turns(m["transcript"]) if m.get("transcript") else []
+
+ mode, them = review_meeting(i, len(meetings), m, participants, turns)
+
+ if mode == "skip":
+ print(" -> Skipped")
+ results["skipped"] += 1
+ continue
+
+ # Resolve creator peer
+ creator = participants.note_creator
+ me_source = (creator.email or creator.name) if creator else None
+ if not me_source:
+ print(" -> Skipped (no creator identifier)")
+ results["skipped"] += 1
+ continue
+
+ me_peer_id = peer_id_from(me_source)
+ if me_peer_id not in seen_peers:
+ print(f" New peer: {me_source} ({me_peer_id})")
+ seen_peers.add(me_peer_id)
+
+ try:
+ created_at = parse_date(m.get("date", ""))
+ session = honcho.session(f"meeting-{mid}")
+ metadata: dict[str, object] = {
+ "title": m.get("title", "Untitled"),
+ "date": m.get("date", ""),
+ "granola_meeting_id": mid,
+ "mode": mode,
+ }
+
+ if mode == "two_person" and them is not None:
+ them_source = them.email or them.name
+ them_peer_id = peer_id_from(them_source)
+ if them_peer_id not in seen_peers:
+ print(f" New peer: {them_source} ({them_peer_id})")
+ seen_peers.add(them_peer_id)
+ import_two_person(honcho, session, me_peer_id, them_peer_id, turns, metadata, created_at)
+ else:
+ import_summary(honcho, session, me_peer_id, m, metadata, created_at)
+
+ results["imported"] += 1
+
+ except ValueError as e:
+ print(f" -> FAILED: {e}")
+ results["failed"] += 1
+ except Exception as e:
+ print(f" -> FAILED: {e}")
+ traceback.print_exc()
+ results["failed"] += 1
+
+ # Done
+ print("\n" + "=" * 60)
+ print(" Transfer Complete!")
+ print("=" * 60)
+ print(f"\n Imported: {results['imported']}")
+ print(f" Skipped: {results['skipped']}")
+ print(f" Failed: {results['failed']}")
+ print(" Workspace: granola")
+ print(f" Peers: {sorted(seen_peers)}")
+
+ except KeyboardInterrupt:
+ print("\n\nAborted.")
+ sys.exit(0)
+ except Exception as e:
+ print(f"\nTransfer failed: {e}")
+ traceback.print_exc()
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ asyncio.run(main())