---
title: "How to backfill historical data into Supermemory"
sidebarTitle: "Backfill historical data"
description: "Backfill historical documents into Supermemory with date, stable IDs, and the batch ingestion API."
icon: "/icons/hugeicons/clock-01.svg"
---
Use `POST /ns/{namespace}/document/batch` to backfill exports, emails, messages, or other dated records.
Sort the source data oldest to newest, add `date` to every document.
## Backfill in batches
Backfill dated content by setting `date` on each document, sorting the source records oldest to newest, and sending them in batches. Each request can contain up to 600 documents.
**Endpoint:** [`POST /ns/{namespace}/document/batch`](/api-reference/ingest/batch-add-documents)
```typescript TypeScript
import { Supermemory } from "supermemory";
type SourceDocument = {
id: string;
content: string;
createdAt: string;
};
const supermemory = new Supermemory({ apiKey: process.env.SUPERMEMORY_API_KEY });
const batchSize = 100;
async function backfillHistoricalData(sourceDocuments: SourceDocument[]) {
const documents = sourceDocuments
.map((document) => ({
content: document.content,
id: document.id,
date: new Date(document.createdAt).toISOString()
}))
.sort((a, b) => a.date.localeCompare(b.date));
for (let offset = 0; offset < documents.length; offset += batchSize) {
const result = await supermemory.documents.batchAdd("historical_import", {
documents: documents.slice(offset, offset + batchSize),
});
if (result.failed > 0) {
throw new Error(`${result.failed} documents failed to ingest`);
}
}
}
```
```python Python
from datetime import datetime, timezone
from supermemory import Supermemory
client = Supermemory()
batch_size = 100
def to_utc(value: str) -> str:
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
if parsed.tzinfo is None:
raise ValueError("created_at must include a timezone")
return parsed.astimezone(timezone.utc).isoformat().replace("+00:00", "Z")
def backfill_historical_data(source_documents: list[dict[str, str]]) -> None:
documents = sorted(
[
{
"content": document["content"],
"id": document["id"],
"date": to_utc(document["created_at"]),
}
for document in source_documents
],
key=lambda document: document["date"],
)
for offset in range(0, len(documents), batch_size):
result = client.documents.batch_add(
"historical_import",
documents=documents[offset : offset + batch_size],
)
if result.failed > 0:
raise RuntimeError(f"{result.failed} documents failed to ingest")
```
```bash curl
curl -X POST "https://api.supermemory.ai/ns/historical_import/document/batch" \
-H "Authorization: Bearer $SUPERMEMORY_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"documents": [
{ "content": "first message", "id": "msg_001", "date": "2023-01-05T09:00:00Z" },
{ "content": "second message", "id": "msg_002", "date": "2023-01-06T14:30:00Z" }
]
}'
```
Batch results preserve input order. Inspect `success`, `failed`, and every item in `results`; a batch can contain successful and failed items together.
## Optional: wait for processing to finish
**Endpoint:** [`GET /ns/{namespace}/document/{id}`](/api-reference/documents/get-document)
The batch endpoint returns after accepting the documents. If a later step depends on the documents being indexed, poll the returned document IDs until `system.status` is `done`. If that later step also needs the extracted memories, send the batch with `dreaming: "instant"`; under the default `dynamic` mode memory extraction is batched and can lag `done` by minutes.
```typescript TypeScript
async function waitUntilDone(ids: string[]) {
while (true) {
const documents = await Promise.all(
ids.map((id) => supermemory.documents.get("historical_import", id))
);
if (documents.some((document) => document.system.status === "failed")) {
throw new Error("A document failed to process");
}
if (documents.every((document) => document.system.status === "done")) {
return;
}
await new Promise((resolve) => setTimeout(resolve, 10_000));
}
}
```
```python Python
import time
def wait_until_done(ids: list[str]) -> None:
while True:
documents = [
client.documents.get("historical_import", document_id) for document_id in ids
]
if any(document.system.status == "failed" for document in documents):
raise RuntimeError("A document failed to process")
if all(document.system.status == "done" for document in documents):
return
time.sleep(10)
```
```bash curl
curl "https://api.supermemory.ai/ns/historical_import/document/msg_001" \
-H "Authorization: Bearer $SUPERMEMORY_API_KEY"
# repeat until "system": { "status": "done" }
```