--- title: "How to backfill historical data into Supermemory" sidebarTitle: "Backfill historical data" description: "Backfill historical documents into Supermemory with date, stable IDs, and the batch ingestion API." icon: "/icons/hugeicons/clock-01.svg" --- Use `POST /ns/{namespace}/document/batch` to backfill exports, emails, messages, or other dated records. Sort the source data oldest to newest, add `date` to every document. ## Backfill in batches Backfill dated content by setting `date` on each document, sorting the source records oldest to newest, and sending them in batches. Each request can contain up to 600 documents. **Endpoint:** [`POST /ns/{namespace}/document/batch`](/api-reference/ingest/batch-add-documents) ```typescript TypeScript import { Supermemory } from "supermemory"; type SourceDocument = { id: string; content: string; createdAt: string; }; const supermemory = new Supermemory({ apiKey: process.env.SUPERMEMORY_API_KEY }); const batchSize = 100; async function backfillHistoricalData(sourceDocuments: SourceDocument[]) { const documents = sourceDocuments .map((document) => ({ content: document.content, id: document.id, date: new Date(document.createdAt).toISOString() })) .sort((a, b) => a.date.localeCompare(b.date)); for (let offset = 0; offset < documents.length; offset += batchSize) { const result = await supermemory.documents.batchAdd("historical_import", { documents: documents.slice(offset, offset + batchSize), }); if (result.failed > 0) { throw new Error(`${result.failed} documents failed to ingest`); } } } ``` ```python Python from datetime import datetime, timezone from supermemory import Supermemory client = Supermemory() batch_size = 100 def to_utc(value: str) -> str: parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) if parsed.tzinfo is None: raise ValueError("created_at must include a timezone") return parsed.astimezone(timezone.utc).isoformat().replace("+00:00", "Z") def backfill_historical_data(source_documents: list[dict[str, str]]) -> None: documents = sorted( [ { "content": document["content"], "id": document["id"], "date": to_utc(document["created_at"]), } for document in source_documents ], key=lambda document: document["date"], ) for offset in range(0, len(documents), batch_size): result = client.documents.batch_add( "historical_import", documents=documents[offset : offset + batch_size], ) if result.failed > 0: raise RuntimeError(f"{result.failed} documents failed to ingest") ``` ```bash curl curl -X POST "https://api.supermemory.ai/ns/historical_import/document/batch" \ -H "Authorization: Bearer $SUPERMEMORY_API_KEY" \ -H "Content-Type: application/json" \ -d '{ "documents": [ { "content": "first message", "id": "msg_001", "date": "2023-01-05T09:00:00Z" }, { "content": "second message", "id": "msg_002", "date": "2023-01-06T14:30:00Z" } ] }' ``` Batch results preserve input order. Inspect `success`, `failed`, and every item in `results`; a batch can contain successful and failed items together. ## Optional: wait for processing to finish **Endpoint:** [`GET /ns/{namespace}/document/{id}`](/api-reference/documents/get-document) The batch endpoint returns after accepting the documents. If a later step depends on the documents being indexed, poll the returned document IDs until `system.status` is `done`. If that later step also needs the extracted memories, send the batch with `dreaming: "instant"`; under the default `dynamic` mode memory extraction is batched and can lag `done` by minutes. ```typescript TypeScript async function waitUntilDone(ids: string[]) { while (true) { const documents = await Promise.all( ids.map((id) => supermemory.documents.get("historical_import", id)) ); if (documents.some((document) => document.system.status === "failed")) { throw new Error("A document failed to process"); } if (documents.every((document) => document.system.status === "done")) { return; } await new Promise((resolve) => setTimeout(resolve, 10_000)); } } ``` ```python Python import time def wait_until_done(ids: list[str]) -> None: while True: documents = [ client.documents.get("historical_import", document_id) for document_id in ids ] if any(document.system.status == "failed" for document in documents): raise RuntimeError("A document failed to process") if all(document.system.status == "done" for document in documents): return time.sleep(10) ``` ```bash curl curl "https://api.supermemory.ai/ns/historical_import/document/msg_001" \ -H "Authorization: Bearer $SUPERMEMORY_API_KEY" # repeat until "system": { "status": "done" } ```