Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
bm42.cpython-313.pyc135 linesDownload Raw Back to __pycache__
1�

2��j�:���%SSKrSSKrSSKJr SSKJrJrJrJr SSK	r	SSK3rSSKJ
r
 SSKJr SSKJr SSKJr SSKJr SS	KJrJr SS4KJrJr SSKJrJr \"SS
SSS\"SS9SS/SS9	/r\ \\!S'SS0r"\"RG5VVs0sHupURI5U_M snnr%S\&S\&4Sjr'"SS\\\5r("SS\\5r)gs snnf) �N)�Path)�Any�Iterable�Sequence�Type)�SnowballStemmer)�OnnxProvider)�OnnxOutputContext)�Device)�define_cache_dir)�SparseEmbedding�SparseTextEmbeddingBase)�
OnnxTextModel�TextEmbeddingWorker)�SparseModelDescription�ModelSourcez'Qdrant/bm42-all-minilm-l6-v2-attentionsi:wzYLight sparse embedding model, which assigns an importance score to each token in the textz5apache-2.0g6ףp=7�?z'Qdrant/all_miniLM_L6_v2_with_attentions)�hfz8model.onnx�
stopwords.txtT)	�model�9vocab_size�description�license�10size_in_GB�sources�11model_file�additional_files�requires_idf�supported_bm42_models�english�12model_name�returnc�0�[UR5$�N)�MODEL_TO_LANGUAGE�lower)r s �[D:\code\apps\devtools\python\user_packages\Python313\site-packages\fastembed/sparse/bm42.py�get_language_by_model_namer',s���Z�-�-�/�0�0�c�^�\rSrSrSrS/rSSSS\RSSSS4	S\S\S-S	\	S-S13\14\S-S\S\
\-S
\\	S-S\
S\	S-S\S-S\4U4SjjjrS.SjrS\\\\4S\\\\44SjrS\\\\4S\\\\44Sjr\S\\\\\	4S\\S\\\\44Sj5rS\\\	\4S\\\\\	44SjrS\\\4S\\	\44SjrS\S\S\\4Sjr\S\\4S j5r\S!\ S\\4S"j5r!S/S#\\\-S$\	S%\	S-S\S\\415S&jjr"\S\\S\\	\44S'j5r#S(\\\-S\S\\4S)jr$\S\%\&\4S*j5r'S0S+\\\-S$\	S\S\	4S,jjr(S-r)U=r*$)1�Bm42�0a�16Bm42 is an extension of BM25, which tries to better evaluate importance of tokens in the documents,17by extracting attention weights from the transformer model.18 19Traditional BM25 uses a count of tokens in the document to evaluate the importance of the token,20but this approach doesn't work well with short documents or chunks of text, as almost all tokens21there are unique.22 23BM42 addresses this issue by replacing the token count with the attention weights from the transformer model.24This allows sparse embeddings to work well with short documents, handle rare tokens and leverage traditional NLP25techniques like stemming and stopwords.26 27WARNING: This model is expected to be used with `modifier="idf"` in the sparse vector index of Qdrant.28�attention_6Ng�?Fr �	cache_dir�threads�	providers�alpha�cuda�29device_ids�	lazy_load�	device_id�specific_model_path�kwargsc�@>�[TU]"XU40UD6 X@lX�lUR	U5UlXplX`lSUlU	bX�lO!URbURSUlURU5Ul30[[U55Ul
X�lURURURUR URS9Ul0Ul['5Ul['5Ul['[,R.5Ul['UR1UR"55Ul[5[7UR855UlXPlUR(dUR?5 gg)a�31Args:32    model_name (str): The name of the model to use.33    cache_dir (str, optional): The path to the cache directory.34                               Can be set using the `FASTEMBED_CACHE_PATH` env variable.35                               Defaults to `fastembed_cache` in the system's temp directory.36    threads (int, optional): The number of threads single onnxruntime session can use. Defaults to None.37    providers (Optional[Sequence[OnnxProvider]], optional): The providers to use for onnxruntime.38    alpha (float, optional): Parameter, that defines the importance of the token weight in the document39        versus the importance of the token frequency in the corpus. Defaults to 0.5, based on empirical testing.40        It is recommended to only change this parameter based on training data for a specific dataset.41    cuda (Union[bool, Device], optional): Whether to use cuda for inference. Mutually exclusive with `providers`42        Defaults to Device.AUTO.43    device_ids (Optional[list[int]], optional): The list of device ids to use for data parallel processing in44        workers. Should be used with `cuda` equals to `True`, `Device.AUTO` or `Device.CUDA`, mutually exclusive45        with `providers`. Defaults to None.46    lazy_load (bool, optional): Whether to load the model during class initialization or on demand.47        Should be set to True when using multiple-gpu and parallel encoding. Defaults to False.48    device_id (Optional[int], optional): The device id to use for loading the model in the worker process.49    specific_model_path (Optional[str], optional): The specific path to the onnx model dir if it should be imported from somewhere else50 51Raises:52    ValueError: If the model_name is not in the format <org>/<model> e.g. BAAI/bge-base-en.53Nr)�local_files_onlyr5) �super�__init__r/r3�_select_exposed_session_options�_extra_session_optionsr2r1r4�_get_model_description�model_description�strrr-�_specific_model_path�download_model�_local_files_only�54_model_dir�invert_vocab�set�special_tokens�special_tokens_ids�string�punctuation�_load_stopwords�	stopwordsrr'r �stemmerr0�load_onnx_model)
�selfr r-r.r/r0r1r2r3r4r5r6�	__class__s
            �r&r:�
Bm42.__init__BsT���N	�����B�6�B�"��"��&*�&J�&J�6�&R��#�%���	�&*���� �&�N�
�_�_�
(�!�_�_�Q�/�D�N�!%�!<�!<�Z�!H����-�i�8�9���$7�!��-�-��"�"��N�N�!�3�3� $� 9� 9�	.�55���-/���(+����,/�E����v�1�1�2����T�1�1�$�/�/�B�C���&�'A�$�/�/�'R�S����56��~�~�� � �"�r(r!c57�>�URURURRURUR58URURURS9 URR5R5HupXRU'M [URR55Ul[URR#55Ul[UR'UR55Ulg)N)�	model_dirrr.r/r1r4�extra_session_options)�_load_onnx_modelrCr>rr.r/r1r4r<�	tokenizer�	get_vocab�itemsrDrE�special_token_to_id�keysrF�valuesrGrJrK)rN�token�idxs   r&rM�Bm42.load_onnx_model�s�������o�o��-�-�8�8��L�L��n�n�����n�n�"&�"=�"=�	�	59��.�.�2�2�4�:�:�<�J�E�%*���c�"�=�!�$�":�":�"?�"?�"A�B���"%�d�&>�&>�&E�&E�&G�"H����T�1�1�$�/�/�B�C��r(�tokensc��/nUH7up4X0R;dX0R;aM%URX445 M9 U$r#)rKrI�append)rNr^�resultr[�values     r&�_filter_pair_tokens�Bm42._filter_pair_tokens�s@��(*��"�L�E����&�%�3C�3C�*C���M�M�5�.�)�#��
r(c�z�/nUH2up4URRU5nURXT45 M4 U$r#)rL�	stem_wordr`)rNr^rar[rb�processed_tokens      r&�_stem_pair_tokens�Bm42._stem_pair_tokens�s=��(*��"�L�E�"�l�l�4�4�U�;�O��M�M�?�2�3�#��
r(�weightsc�p^�/nUH,upE[U4SjU55nURXF45 M. U$)Nc3�.># �UH60nTUv� M g7fr#�)�.0r\rjs  �r&�	<genexpr>�*Bm42._aggregate_weights.<locals>.<genexpr>�s����:�T�c�W�S�\�T�s�)�sumr`)�clsr^rjrar[�idxs�61sum_weights  `    r&�_aggregate_weights�Bm42._aggregate_weights�s<���+-��!�K�E��:�T�:�:�J��M�M�5�-�.�"��
r(�62bpe_tokensc��/nSn/nURRRn[U5nUHtupxX�R;aMURU5(aX8US-
nUR
U5 MFU(aUR
X445 /nUnUR
U5 Mv U(aUR
X445 U$)N�)rUr�continuing_subword_prefix�lenrF�63startswithr`)	rNrwra�acc�acc_idxrz�continuing_subword_prefix_lenr\r[s	         r&�_reconstruct_bpe�Bm42._reconstruct_bpe�s���/1������$(�N�N�$8�$8�$R�$R�!�(+�,E�(F�%�$�J�C��+�+�+����� 9�:�:��:�;�<�<�����s�#���M�M�3�.�1� �G������s�#�%���M�M�3�.�)��
r(�vectorc���0nUR5HLup4[[R"U55n[R64"SU-5UR-X%'MN U$)z�65Orders all tokens in the vector by their importance and generates a new score based on the importance order.66So that the scoring doesn't depend on absolute values assigned by the model, but on the relative importance.67��?)rW�abs�mmh3�hash�math�logr0)rNr��68new_vectorr[rb�token_ids      r&�_rescore_vector�Bm42._rescore_vector�sV��(*�69�"�L�L�N�L�E��4�9�9�U�+�,�H�70$(�8�8�C�%�K�#8�D�J�J�#F�J� �
+��r(�outputc+�j^# �URc[S5eURR[5n[R71"URSS2SS2S4SS9UR-n[X45H�upVU4Sj[U55nTRU5nTRU5n	TRU	5n72TRX�5n0nUH#up�[URU
S5U5X�'M% TR!U5n["R$"U5v� M� g7f)Nz7input_ids must be provided for document post-processingr�)�axisc3�J># �UHupUTRU4v� M g7fr#)rD)rnr\r�rNs   �r&ro�1Bm42._post_process_onnx_output.<locals>.<genexpr>�s*����(�%B�M�C��d�'�'��1�2�%B�s� #)�	input_ids�73ValueError�astype�int�np�mean�model_output�attention_mask�zip�	enumerater�rcrhru�max�getr�r
�	from_dict)rNr�r6�token_ids_batch�pooled_attention�document_token_ids�attention_value�document_tokens_with_ids�
reconstructed�filtered�stemmed�weighted�max_token_weightr[�weight�rescoreds`               r&�_post_process_onnx_output�Bm42._post_process_onnx_output�s&�������#��V�W�W� �*�*�1�1�#�6���7�7�6�#6�#6�q�!�Q�w�#?�a�H�6�K`�K`�`��36��3Y�/��(�%.�/A�%B�(�$�74!�1�1�2J�K�M��/�/�
�>�H��,�,�X�6�G��.�.�w�H�H�13��!)�
��*-�.>�.B�.B�5�!�.L�f�*U� �'�"*��+�+�,<�=�H�!�+�+�H�5�5�+4Z�s�D0D3c��[$)z�Lists the supported models.75 76Returns:77    list[SparseModelDescription]: A list of SparseModelDescription objects containing the model information.78)r�rrs r&�_list_supported_models�Bm42._list_supported_modelss79��%�$r(rRc���US-nUR5(d/$[US5nUR5R5sSSS5 $!,(df   g=f)Nr�r)�exists�open�read�80splitlines)rrrR�stopwords_path�fs    r&rJ�Bm42._load_stopwordssJ��"�_�4���$�$�&�&��I�
�.�#�
&�!��6�6�8�&�&�(�'�
&�
&�s�A�81A�	documents�82batch_size�parallelc+�# �URUR[UR5UUUURUR83URURURURURS9Shv�N gN7f)ac84Encode a list of documents into list of embeddings.85We use mean pooling with attention so that the model can handle variable-length inputs.86 87Args:88    documents: Iterator of documents or single document to embed89    batch_size: Batch size for encoding -- higher values will use more memory, but be faster90    parallel:91        If > 1, data-parallel encoding will be used, recommended for offline encoding of large datasets.92        If 0, use all available cores.93        If None, don't use data-parallel processing, use default onnxruntime threading instead.94 95Returns:96    List of embeddings, one per document97)r r-r�r�r�r/r1r2r0r8r5rSN)�_embed_documentsr r?r-r/r1r2r0rBr@r<)rNr�r�r�r6s     r&�embed�98Bm42.embedsw���,�(�(�����$�.�.�)��!���n�n��������*�*�!�3�3� $� 9� 9�"&�"=�"=�)�
99�
	100�
	101�s�BB�B	�Bc�b�0nUH&n[[R"U55nSX$'M( U$)Nr�)r�r�r�)rrr^rar[r�s     r&�
_query_rehash�Bm42._query_rehashBs3��#%���E��4�9�9�U�+�,�H�"�F����
r(�queryc+��# �[U[5(aU/n[US5(a
URcUR	5 UH�nUR102R
U5n[UR5nURU5nURU5nURU5n[R"URSU555v� M� g7f)z�103To emulate BM25 behaviour, we don't need to use smart weights in the query, and104it's enough to just hash the tokens and assign a weight of 1.0 to them.105It is also faster, as we don't need to run the model for the query.106rNc3�*# �UH	upUv� M g7fr#rm)rnr[�_s   r&ro�#Bm42.query_embed.<locals>.<genexpr>]s���>]�U\���u�U\�s�)�107isinstancer?�hasattrrrMrU�encoder�r^r�rcrhr
r�r�)	rNr�r6�text�encodedr�r�r�r�s	         r&�query_embed�Bm42.query_embedJs�����e�S�!�!��G�E��t�W�%�%����);�� � �"��D��n�n�+�+�D�1�G�'0����'@�$� �1�1�2J�K�M��/�/�
�>�H��,�,�X�6�G�!�+�+�D�,>�,>�>]�U\�>]�,]�^�^��s�C"C$c��[$r#)�Bm42TextEmbeddingWorkerr�s r&�_get_worker_class�Bm42._get_worker_class_s��&�&r(�textsc��[US5(a
URcUR5 UR"U4SU0UD6$)Nrr�)r�rrM�_token_count)rNr�r�r6s    r&�token_count�Bm42.token_countcsA���t�W�%�%����);�� � �"�� � ��H�:�H��H�Hr()r<rCr@r0r-r1r4r2rDr3r>r/rIrFrGrLrK)r!N)�N)i)+�__name__�108__module__�__qualname__�__firstlineno__�__doc__�ONNX_OUTPUT_NAMESr�AUTOr?r�rr	�float�bool�listrr:rM�tuplercrh�classmethodrurr��dictr�r109r
r�rr�rrJr�r�r�rrr�r��__static_attributes__�
__classcell__)rOs@r&r*r*0sd���
�'���110!%�"�37��$�k�k�'+�� $�*.�L#��L#���:�L#��t��	L#�111�L�)�D�0�L#��
L#��V�m�L#���I��$�L#��L#���:�L#�!�4�Z�L#��L#�L#�\D�"�$�u�S�#�X��*?��D��s�TW�x��DY����U�3��8�_�(=��$�u�S�RU�X��BW�����%��T�#�Y��/�0��;?��;��	
�e�C��J��	 �����"�5��c��?�3��	
�e�C��c��N�#�	$��:�d�3��:�&6��4��U�112�;K��$ 6�'� 6�36� 6�	�/�	"� 6�D�%�t�,B�'C�%��%��)��)��c��)��)��#�	#113���#��&�#114��#115���*�	#116�117�#118�119�/�	"�
#120�J��8�C�=��T�#�u�*�5E����_��x��}�!4�_��_��Q`�Ha�_�*�'�$�':�?�'K�"L�'��'�=A�I��8�C�=�(�I�69�I�LO�I�	�I�Ir(r*c�.�\rSrSrS\S\S\S\4SjrSrg)	r�ikr r-r6r!c��[SUUS.UD6$)N)r r-rm)r*)rNr r-r6s    r&�init_embedding�&Bm42TextEmbeddingWorker.init_embeddingls$���121�!��122��123�	124r(rmN)	r�r�r�r�r?rr*r�r�rmr(r&r�r�ks$��125��126��127��128�PT�129r(r�)*r�rH�pathlibr�typingrrrrr��numpyr��py_rust_stemmersr�fastembed.commonr	�fastembed.common.onnx_modelr130�fastembed.common.typesr�fastembed.common.utilsr�&fastembed.sparse.sparse_embedding_baser
r�fastembed.text.onnx_text_modelrr�"fastembed.common.model_descriptionrrrr��__annotations__�_MODEL_TO_LANGUAGErWr%r$r?r'r*r�)r �languages00r&�<module>r�s���
��0�0���,�)�9�)�3��N�R��7��o����H�I��)�*��131�7��t�2�3�� .�y���>P�=U�=U�=W��=W�%9�Z�J����� �=W���1321�3�1�3�1�xI�"�M�/�$B�xI�v	133�1�/�B�134��I135s�
C
codekingpro/portable-devtools · Team Ai