Underground-Digital/Workflow-Engine
0
1from __future__ import annotations2 3from abc import ABC, abstractmethod4from typing import Any5 6from core.rag.models.document import Document7 8 9class BaseVector(ABC):10 def __init__(self, collection_name: str):11 self._collection_name = collection_name12 13 @abstractmethod14 def get_type(self) -> str:15 raise NotImplementedError16 17 @abstractmethod18 def create(self, texts: list[Document], embeddings: list[list[float]], **kwargs):19 raise NotImplementedError20 21 @abstractmethod22 def add_texts(self, documents: list[Document], embeddings: list[list[float]], **kwargs):23 raise NotImplementedError24 25 @abstractmethod26 def text_exists(self, id: str) -> bool:27 raise NotImplementedError28 29 @abstractmethod30 def delete_by_ids(self, ids: list[str]) -> None:31 raise NotImplementedError32 33 def get_ids_by_metadata_field(self, key: str, value: str):34 raise NotImplementedError35 36 @abstractmethod37 def delete_by_metadata_field(self, key: str, value: str) -> None:38 raise NotImplementedError39 40 @abstractmethod41 def search_by_vector(self, query_vector: list[float], **kwargs: Any) -> list[Document]:42 raise NotImplementedError43 44 @abstractmethod45 def search_by_full_text(self, query: str, **kwargs: Any) -> list[Document]:46 raise NotImplementedError47 48 @abstractmethod49 def delete(self) -> None:50 raise NotImplementedError51 52 def _filter_duplicate_texts(self, texts: list[Document]) -> list[Document]:53 for text in texts.copy():54 doc_id = text.metadata["doc_id"]55 exists_duplicate_node = self.text_exists(doc_id)56 if exists_duplicate_node:57 texts.remove(text)58 59 return texts60 61 def _get_uuids(self, texts: list[Document]) -> list[str]:62 return [text.metadata["doc_id"] for text in texts]63 64 @property65 def collection_name(self):66 return self._collection_name67 