mirror of
https://github.com/supermemoryai/supermemory.git
synced 2026-10-10 03:28:14 +00:00
Rewrites 339 TypeScript calls across 50 pages from the rc.5 `method({ namespace, body })` form to the shipped `method(namespace, { ... })` form, and aligns field names with the live v5 spec: `attach` to `include`, `authUrl` to `authorization`, `lastSync` to `latestRun`, `deletedCount` to `count`, and the paginated `namespaces.list()`.
Renames container tags to namespaces across concepts, connectors, integrations and snippets. The namespace pages keep container tag in the description, search keywords and a rename note so old searches still land, and the v3 reference page points at v5.
The migration guide's SDK table now covers both 5.0.0 SDKs, and the SDK integration page uses the real client options (`baseUrl`, `timeoutInSeconds`, `maxRetries`) and error classes.
158 lines
5 KiB
Text
158 lines
5 KiB
Text
---
|
|
title: "How to backfill historical data into Supermemory"
|
|
sidebarTitle: "Backfill historical data"
|
|
description: "Backfill historical documents into Supermemory with date, stable IDs, and the batch ingestion API."
|
|
icon: "/icons/hugeicons/clock-01.svg"
|
|
---
|
|
|
|
Use `POST /ns/{namespace}/document/batch` to backfill exports, emails, messages, or other dated records.
|
|
|
|
<Warning>
|
|
Sort the source data oldest to newest, add `date` to every document.
|
|
</Warning>
|
|
|
|
## Backfill in batches
|
|
|
|
Backfill dated content by setting `date` on each document, sorting the source records oldest to newest, and sending them in batches. Each request can contain up to 600 documents.
|
|
|
|
**Endpoint:** [`POST /ns/{namespace}/document/batch`](/api-reference/ingest/batch-add-documents)
|
|
|
|
<CodeGroup>
|
|
|
|
```typescript TypeScript
|
|
import { Supermemory } from "supermemory";
|
|
|
|
type SourceDocument = {
|
|
id: string;
|
|
content: string;
|
|
createdAt: string;
|
|
};
|
|
|
|
const supermemory = new Supermemory({ apiKey: process.env.SUPERMEMORY_API_KEY });
|
|
const batchSize = 100;
|
|
|
|
async function backfillHistoricalData(sourceDocuments: SourceDocument[]) {
|
|
const documents = sourceDocuments
|
|
.map((document) => ({
|
|
content: document.content,
|
|
id: document.id,
|
|
date: new Date(document.createdAt).toISOString()
|
|
}))
|
|
.sort((a, b) => a.date.localeCompare(b.date));
|
|
|
|
for (let offset = 0; offset < documents.length; offset += batchSize) {
|
|
const result = await supermemory.documents.batchAdd("historical_import", {
|
|
documents: documents.slice(offset, offset + batchSize),
|
|
});
|
|
|
|
if (result.failed > 0) {
|
|
throw new Error(`${result.failed} documents failed to ingest`);
|
|
}
|
|
}
|
|
}
|
|
```
|
|
|
|
```python Python
|
|
from datetime import datetime, timezone
|
|
from supermemory import Supermemory
|
|
|
|
client = Supermemory()
|
|
batch_size = 100
|
|
|
|
def to_utc(value: str) -> str:
|
|
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
if parsed.tzinfo is None:
|
|
raise ValueError("created_at must include a timezone")
|
|
return parsed.astimezone(timezone.utc).isoformat().replace("+00:00", "Z")
|
|
|
|
def backfill_historical_data(source_documents: list[dict[str, str]]) -> None:
|
|
documents = sorted(
|
|
[
|
|
{
|
|
"content": document["content"],
|
|
"id": document["id"],
|
|
"date": to_utc(document["created_at"]),
|
|
}
|
|
for document in source_documents
|
|
],
|
|
key=lambda document: document["date"],
|
|
)
|
|
|
|
for offset in range(0, len(documents), batch_size):
|
|
result = client.documents.batch_add(
|
|
"historical_import",
|
|
documents=documents[offset : offset + batch_size],
|
|
)
|
|
|
|
if result.failed > 0:
|
|
raise RuntimeError(f"{result.failed} documents failed to ingest")
|
|
```
|
|
|
|
```bash curl
|
|
curl -X POST "https://api.supermemory.ai/ns/historical_import/document/batch" \
|
|
-H "Authorization: Bearer $SUPERMEMORY_API_KEY" \
|
|
-H "Content-Type: application/json" \
|
|
-d '{
|
|
"documents": [
|
|
{ "content": "first message", "id": "msg_001", "date": "2023-01-05T09:00:00Z" },
|
|
{ "content": "second message", "id": "msg_002", "date": "2023-01-06T14:30:00Z" }
|
|
]
|
|
}'
|
|
```
|
|
|
|
</CodeGroup>
|
|
|
|
Batch results preserve input order. Inspect `success`, `failed`, and every item in `results`; a batch can contain successful and failed items together.
|
|
|
|
## Optional: wait for processing to finish
|
|
|
|
**Endpoint:** [`GET /ns/{namespace}/document/{id}`](/api-reference/documents/get-document)
|
|
|
|
The batch endpoint returns after accepting the documents. If a later step depends on the documents being indexed, poll the returned document IDs until `system.status` is `done`. If that later step also needs the extracted memories, send the batch with `dreaming: "instant"`; under the default `dynamic` mode memory extraction is batched and can lag `done` by minutes.
|
|
|
|
<CodeGroup>
|
|
|
|
```typescript TypeScript
|
|
async function waitUntilDone(ids: string[]) {
|
|
while (true) {
|
|
const documents = await Promise.all(
|
|
ids.map((id) => supermemory.documents.get("historical_import", id))
|
|
);
|
|
|
|
if (documents.some((document) => document.system.status === "failed")) {
|
|
throw new Error("A document failed to process");
|
|
}
|
|
|
|
if (documents.every((document) => document.system.status === "done")) {
|
|
return;
|
|
}
|
|
await new Promise((resolve) => setTimeout(resolve, 10_000));
|
|
}
|
|
}
|
|
```
|
|
|
|
```python Python
|
|
import time
|
|
|
|
def wait_until_done(ids: list[str]) -> None:
|
|
while True:
|
|
documents = [
|
|
client.documents.get("historical_import", document_id) for document_id in ids
|
|
]
|
|
|
|
if any(document.system.status == "failed" for document in documents):
|
|
raise RuntimeError("A document failed to process")
|
|
|
|
if all(document.system.status == "done" for document in documents):
|
|
return
|
|
|
|
time.sleep(10)
|
|
```
|
|
|
|
```bash curl
|
|
curl "https://api.supermemory.ai/ns/historical_import/document/msg_001" \
|
|
-H "Authorization: Bearer $SUPERMEMORY_API_KEY"
|
|
# repeat until "system": { "status": "done" }
|
|
```
|
|
|
|
</CodeGroup>
|