Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
vector_base.py67 linesDownload Raw Back to vdb
1from __future__ import annotations2 3from abc import ABC, abstractmethod4from typing import Any5 6from core.rag.models.document import Document7 8 9class BaseVector(ABC):10    def __init__(self, collection_name: str):11        self._collection_name = collection_name12 13    @abstractmethod14    def get_type(self) -> str:15        raise NotImplementedError16 17    @abstractmethod18    def create(self, texts: list[Document], embeddings: list[list[float]], **kwargs):19        raise NotImplementedError20 21    @abstractmethod22    def add_texts(self, documents: list[Document], embeddings: list[list[float]], **kwargs):23        raise NotImplementedError24 25    @abstractmethod26    def text_exists(self, id: str) -> bool:27        raise NotImplementedError28 29    @abstractmethod30    def delete_by_ids(self, ids: list[str]) -> None:31        raise NotImplementedError32 33    def get_ids_by_metadata_field(self, key: str, value: str):34        raise NotImplementedError35 36    @abstractmethod37    def delete_by_metadata_field(self, key: str, value: str) -> None:38        raise NotImplementedError39 40    @abstractmethod41    def search_by_vector(self, query_vector: list[float], **kwargs: Any) -> list[Document]:42        raise NotImplementedError43 44    @abstractmethod45    def search_by_full_text(self, query: str, **kwargs: Any) -> list[Document]:46        raise NotImplementedError47 48    @abstractmethod49    def delete(self) -> None:50        raise NotImplementedError51 52    def _filter_duplicate_texts(self, texts: list[Document]) -> list[Document]:53        for text in texts.copy():54            doc_id = text.metadata["doc_id"]55            exists_duplicate_node = self.text_exists(doc_id)56            if exists_duplicate_node:57                texts.remove(text)58 59        return texts60 61    def _get_uuids(self, texts: list[Document]) -> list[str]:62        return [text.metadata["doc_id"] for text in texts]63 64    @property65    def collection_name(self):66        return self._collection_name67