Press n or j to go to the next uncovered block, b, p or k for the previous block.
| 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 | 28x 28x 28x | import { VectorizeBinding } from '@/core/bindings/VectorizeBinding'
import type { VectorIndex, VectorMatch } from '@bayudwiyansatria/core'
import type { CloudflareEnv } from '@/types/CloudflareEnv'
export type { VectorMatch } from '@bayudwiyansatria/core'
/**
* Vectorize index binding.
*
* Every setting it needs — binding name, `topK`, namespace — arrives resolved
* from the configuration layer, so nothing about them is decided here. The
* binding itself is read from `env` per call, which is what makes one
* module-scope instance safe to share.
*/
const documents = new VectorizeBinding()
/**
* Similarity-search capability, backed by Vectorize.
*
* @remarks
* Embeddings come from `AIService.embed` — this capability stores and searches
* them, and has no opinion about how they were produced. Writes are applied
* asynchronously, so a mutation id comes back rather than the applied state.
*
* @class
*
* @author Bayu Dwiyan Satria
* @version 1.0.0
* @since 1.0.0
*/
export class VectorizeService<E extends CloudflareEnv = CloudflareEnv> implements VectorIndex<E> {
/**
* Stores or replaces a vector.
*
* @param env The Worker environment.
* @param id Id to store the vector under — reuse the source record's id.
* @param vector The embedding.
* @param metadata Metadata returned with future matches.
* @returns The mutation id.
*/
public async index(
env: E,
id: string,
vector: number[],
metadata: Record<string, VectorizeVectorMetadata> = {}
): Promise<string> {
const mutation = await documents.upsert(env, [{ id, values: vector, metadata }])
return mutation.mutationId
}
/**
* Finds the vectors closest to a query embedding.
*
* @param env The Worker environment.
* @param vector The query embedding.
* @param topK How many matches to return. Falls back to the configured value.
* @returns The matches, ordered by similarity.
*/
public async search(env: E, vector: number[], topK?: number): Promise<VectorMatch[]> {
const result = await documents.query(env, vector, {
returnMetadata: 'all',
...(topK ? { topK } : {})
})
return result.matches.map(match => ({
id: match.id,
score: match.score,
metadata: match.metadata as Record<string, unknown>
}))
}
/**
* Finds the vectors closest to one already in the index.
*
* Saves re-embedding a stored record just to search with it.
*
* @param env The Worker environment.
* @param id The stored vector's id.
* @returns The matches, ordered by similarity.
*/
public async similar(env: E, id: string): Promise<VectorMatch[]> {
const result = await documents.queryById(env, id)
return result.matches.map(match => ({ id: match.id, score: match.score }))
}
/**
* Removes vectors by id.
*
* @param env The Worker environment.
* @param ids The vector ids to remove.
* @returns The mutation id.
*/
public async remove(env: E, ids: string[]): Promise<string> {
const mutation = await documents.deleteByIds(env, ids)
return mutation.mutationId
}
/**
* Reports whether the index is bound to this Worker.
*
* @param env The Worker environment.
* @returns `true` when the binding is present.
*/
public isAvailable(env: E): boolean {
return documents.isBound(env)
}
}
|