mirror of
https://github.com/supermemoryai/supermemory.git
synced 2026-10-08 03:08:21 +00:00
feat: Improve batch processing for vector metadata updates; Support long form content
This commit is contained in:
parent
8f9486b28c
commit
dde7f21c5e
3 changed files with 31 additions and 13 deletions
|
|
@ -155,10 +155,21 @@ export async function batchCreateChunksAndEmbeddings({
|
|||
// the user.
|
||||
if (allIds.keys.length > 0) {
|
||||
const savedVectorIds = allIds.keys.map((key) => key.name);
|
||||
const vectors = await context.env.VECTORIZE_INDEX.getByIds(savedVectorIds);
|
||||
|
||||
const vectors = [];
|
||||
//Search in a batch of 20
|
||||
for (let i = 0; i < savedVectorIds.length; i += 20) {
|
||||
const batch = savedVectorIds.slice(i, i + 20);
|
||||
const batchVectors = await context.env.VECTORIZE_INDEX.getByIds(batch);
|
||||
vectors.push(...batchVectors);
|
||||
}
|
||||
console.log(
|
||||
vectors.map((vector) => {
|
||||
return vector.id;
|
||||
}),
|
||||
);
|
||||
// Now, we'll update all vector metadatas with one more userId and all spaceIds
|
||||
const newVectors = vectors.map((vector) => {
|
||||
console.log(JSON.stringify(vector.metadata));
|
||||
vector.metadata = {
|
||||
...vector.metadata,
|
||||
[`user-${body.user}`]: 1,
|
||||
|
|
@ -172,7 +183,18 @@ export async function batchCreateChunksAndEmbeddings({
|
|||
return vector;
|
||||
});
|
||||
|
||||
await context.env.VECTORIZE_INDEX.upsert(newVectors);
|
||||
// upsert in batch of 20
|
||||
const results = [];
|
||||
for (let i = 0; i < newVectors.length; i += 20) {
|
||||
results.push(newVectors.slice(i, i + 20));
|
||||
console.log(JSON.stringify(newVectors[1].id));
|
||||
}
|
||||
|
||||
await Promise.all(
|
||||
results.map((result) => {
|
||||
return context.env.VECTORIZE_INDEX.upsert(result);
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
|
|
@ -186,6 +208,7 @@ export async function batchCreateChunksAndEmbeddings({
|
|||
url: body.url,
|
||||
[sanitizeKey(`user-${body.user}`)]: 1,
|
||||
};
|
||||
|
||||
const spaceMetadata = body.spaces?.reduce((acc, space) => {
|
||||
acc[`space-${body.user}-${space}`] = 1;
|
||||
return acc;
|
||||
|
|
@ -197,13 +220,15 @@ export async function batchCreateChunksAndEmbeddings({
|
|||
return tweet.chunkedTweet.map((chunk) => {
|
||||
const id = `${uuid}-${i}`;
|
||||
ids.push(id);
|
||||
const { tweetLinks, tweetVids, tweetId } = tweet.metadata;
|
||||
const { tweetLinks, tweetVids, tweetId, tweetImages } =
|
||||
tweet.metadata;
|
||||
return {
|
||||
pageContent: chunk,
|
||||
metadata: {
|
||||
links: tweetLinks,
|
||||
videos: tweetVids,
|
||||
tweetId: tweetId,
|
||||
tweetImages: tweetImages,
|
||||
...commonMetaData,
|
||||
...spaceMetadata,
|
||||
},
|
||||
|
|
|
|||
|
|
@ -92,15 +92,7 @@ app.post("/api/add", zValidator("json", vectorObj), async (c) => {
|
|||
break;
|
||||
}
|
||||
|
||||
console.log(JSON.stringify(chunks));
|
||||
|
||||
if (chunks.chunks.length > 20) {
|
||||
return c.json({
|
||||
status: "error",
|
||||
message:
|
||||
"We are unable to process documents this size just yet, try something smaller",
|
||||
});
|
||||
}
|
||||
await batchCreateChunksAndEmbeddings({
|
||||
store,
|
||||
body,
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ interface Metadata {
|
|||
tweetId: string;
|
||||
tweetLinks: any[];
|
||||
tweetVids: any[];
|
||||
tweetImages: any[];
|
||||
}
|
||||
|
||||
export interface ThreadTweetData {
|
||||
|
|
@ -19,7 +20,6 @@ export interface ThreadTweetData {
|
|||
metadata: Metadata;
|
||||
}
|
||||
|
||||
|
||||
export function chunkThread(threadText: string): TweetChunks {
|
||||
const thread = JSON.parse(threadText);
|
||||
|
||||
|
|
@ -30,6 +30,7 @@ export function chunkThread(threadText: string): TweetChunks {
|
|||
tweetId: tweet.id,
|
||||
tweetLinks: tweet.links,
|
||||
tweetVids: tweet.videos,
|
||||
tweetImages: tweet.images,
|
||||
};
|
||||
|
||||
return { chunkedTweet, metadata };
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue