
    i/                        d Z ddlmZ ddlZddlmZmZ ddlmZm	Z	m
Z
mZmZmZmZ ddlZddlmZ ddlmZ ddlmZmZ dd	lmZmZ dd
lmZ ddlmZ ddlmZm Z  erddlm!Z!  G d de      Z"y)z
Base RAG Ingestion class.

Provides abstract methods for:
- OCR
- Chunking
- Embedding
- Vector Store operations

Providers can inherit and override methods as needed.
    )annotationsN)ABCabstractmethod)TYPE_CHECKINGAnyDictListOptionalTuplecast)verbose_logger)uuid4)DEFAULT_CHUNK_OVERLAPDEFAULT_CHUNK_SIZE)get_async_httpx_clienthttpxSpecialProvider)extract_text_from_pdf)RecursiveCharacterTextSplitter)RAGIngestOptionsRAGIngestResponse)Routerc                      e Zd ZdZ	 d	 	 	 ddZddZedd       Z	 	 	 d	 	 	 	 	 	 	 ddZ	 	 	 	 	 	 ddZ		 	 	 	 	 	 	 	 ddZ
	 	 	 	 dd	Ze	 	 	 	 	 	 	 	 	 	 	 	 dd
       Z	 	 	 d	 	 	 	 	 	 	 ddZy)BaseRAGIngestiona  
    Base class for RAG ingestion.

    Providers should inherit from this class and override methods as needed.
    For example, OpenAI handles embedding internally when attaching files to
    vector stores, so it overrides the embedding step to be a no-op.
    Nc                   || _         || _        dt                | _        |j	                  d      | _        t        t        t        t        f   |j	                  d      xs ddi      | _
        |j	                  d      | _        t        t        t        t        f   |j	                  d      xs i       | _        |j	                  d      | _        | j                          y )	Ningest_ocrchunking_strategytypeauto	embeddingvector_storename)ingest_optionsrouterr   	ingest_idget
ocr_configr   r   strr   r   embedding_configvector_store_configingest_name_load_credentials_from_config)selfr#   r$   s      u/Users/manta/Documents/Projects/TheRoad-I1/.venv/lib/python3.12/site-packages/litellm/rag/ingestion/base_ingestion.py__init__zBaseRAGIngestion.__init__,   s    
 -"57), ),,U315cN23G7G2
 !/ 2 2; ?37cNN..~>D"4
  *--f5 	**,    c                    ddl m} | j                  j                  d      }|rYt        j
                  rH|j                  |      }|j                         D ]#  \  }}|| j                  vs|| j                  |<   % yyy)z
        Load credentials from litellm_credential_name if provided in vector_store config.

        This allows users to specify a credential name in the vector_store config
        which will be resolved from litellm.credential_list.
        r   )CredentialAccessorlitellm_credential_nameN).litellm.litellm_core_utils.credential_accessorr2   r*   r&   litellmcredential_listget_credential_valuesitems)r-   r2   credential_namecredential_valueskeyvalues         r.   r,   z.BaseRAGIngestion._load_credentials_from_configD   s|     	V22667PQw66 2 H H! 0557
Ud66649D,,S1 8  7?r0   c                :    | j                   j                  dd      S )zGet the vector store provider.custom_llm_provideropenai)r*   r&   )r-   s    r.   r>   z$BaseRAGIngestion.custom_llm_providerW   s     ''++,A8LLr0   c                f  K   |r|\  }}}|||dfS |rt        t        j                        }|j                  |       d{   }|j	                          |j
                  }|j                  d      d   xs d}|j                  j                  dd      }|||dfS |rddd|fS t        d      7 mw)	aG  
        Upload / prepare file for ingestion.

        Args:
            file_data: Tuple of (filename, content_bytes, content_type)
            file_url: URL to fetch file from
            file_id: Existing file ID to use

        Returns:
            Tuple of (filename, file_content, content_type, existing_file_id)
        N)llm_provider/documentzcontent-typezapplication/octet-streamz,Must provide file_data, file_url, or file_id)	r   r   RAGr&   raise_for_statuscontentsplitheaders
ValueError)	r-   	file_datafile_urlfile_idfilenamefile_contentcontent_typehttp_clientresponses	            r.   uploadzBaseRAGIngestion.upload\   s     " 3<0HlL\<==0>R>V>VWK(__X66H%%'#++L~~c*2.<*H#++// :L \<==tW,,GHH 7s   ?B1B/A.B1c                
  K   | j                   r|sy| j                   j                  dd      }|r
d|v rd\  }}nd\  }}t        j                  |      j	                  d      }d| d	| }| j
                  *| j
                  j                  |d
|||i       d{   }n#t        j                  |d
|||i       d{   }t        |d      r.|j                  r"dj                  d |j                  D              S y7 c7 Aw)z
        Perform OCR on file content to extract text.

        Args:
            file_content: Raw file bytes
            content_type: MIME type of the file

        Returns:
            Extracted text or None if OCR not configured/needed
        Nmodelzmistral/mistral-ocr-latestimage)	image_urlrW   )document_urlrX   utf-8zdata:z;base64,r   )rU   rD   pagesz

c              3  N   K   | ]  }t        |d       s|j                    yw)markdownN)hasattrr\   ).0pages     r.   	<genexpr>z'BaseRAGIngestion.ocr.<locals>.<genexpr>   s"      *<$j@Y*<s   %%)r'   r&   base64	b64encodedecoder$   aocrr5   r]   rZ   join)	r-   rO   rP   	ocr_modeldoc_typeurl_keyb64_contentdata_urlocr_responses	            r.   r   zBaseRAGIngestion.ocr   s     lOO''1MN	 G|3 8Hg >Hg &&|4;;GD<.> ;;"!%!1!1 (GX> "2 " L
 ") (GX>" L <)l.@.@;; *6*<*<   !
s%   BDC?#D?D A DDc                   d}|r|}n|r|s	 |j                  d      }|sg S | j                  xs i }|j                  dt              }|j                  dt              }|j                  d	d      }||d
}	|r||	d	<   t        di |	}
|
j                  |      S # t        $ rh |j                  d      r;t        j                  d       t        |      }|s2t        j                  d       g cY S t        j                  d       g cY S Y w xY w)a  
        Split text into chunks using RecursiveCharacterTextSplitter.

        Args:
            text: Text from OCR (if used)
            file_content: Raw file content bytes
            ocr_was_used: Whether OCR was performed

        Returns:
            List of text chunks
        NrY   s   %PDFz(PDF detected, attempting text extractionzkPDF text extraction failed. Install 'pypdf' or 'PyPDF2' for PDF support, or enable OCR with a vision model.z,Binary file detected, skipping text chunking
chunk_sizechunk_overlap
separators)rm   rn    )rc   UnicodeDecodeError
startswithr   debugr   r   r&   r   r   r   
split_text)r-   textrO   ocr_was_usedtext_to_chunksplitter_argsrm   rn   ro   splitter_kwargstext_splitters              r.   chunkzBaseRAGIngestion.chunk   s+   $ (, M, , 3 3G <  I ..4""&&|5GH
%))/;PQ"&&|T:
 %*+
 ,6OL)6II''66C & **73"(()ST$9,$GM(&,,A  "	"(()WXI )s   B AD	-D	D	c                N  K   | j                   r|sy| j                   j                  dd      }| j                  &| j                  j                  ||       d{   }nt	        j                  ||       d{   }|j
                  D cg c]  }|d   	 c}S 7 A7 #c c}w w)z
        Generate embeddings for text chunks.

        Args:
            chunks: List of text chunks

        Returns:
            List of embeddings or None
        NrU   ztext-embedding-3-small)rU   inputr    )r)   r&   r$   
aembeddingr5   data)r-   chunksembedding_modelrR   items        r.   embedzBaseRAGIngestion.embed   s      $$F//33G=UV;;"![[33/QW3XXH$//oVTTH.6mm<md[!m<<	 YT<s6   AB%BB%:B;B%B B%B% B%c                   K   yw)a  
        Store content in vector store.

        This method must be implemented by provider-specific subclasses.

        Args:
            file_content: Raw file bytes
            filename: Name of the file
            content_type: MIME type
            chunks: Text chunks (if chunking was done locally)
            embeddings: Embeddings (if embedding was done locally)

        Returns:
            Tuple of (vector_store_id, file_id)
        Nrp   )r-   rO   rN   rP   r   
embeddingss         r.   storezBaseRAGIngestion.store  s     0 	s   c           
     .  K   | j                  |||       d{   \  }}}}	 | j                  ||       d{   }| j                  ||| j                  du      }	| j	                  |	       d{   }
| j                  ||||	|
       d{   \  }}t        | j                  d|xs d|xs |	      S 7 7 7 J7 .# t        $ rE}t        j                  d
|        t        | j                  dddt        |            cY d}~S d}~ww xY ww)as  
        Execute the full ingestion pipeline.

        Args:
            file_data: Tuple of (filename, content_bytes, content_type)
            file_url: URL to fetch file from
            file_id: Existing file ID to use

        Returns:
            RAGIngestResponse with status and IDs

        Raises:
            ValueError: If no input source is provided
        )rK   rL   rM   N)rO   rP   )ru   rO   rv   )r   )rO   rN   rP   r   r   	completed )idstatusvector_store_idrM   zRAG Pipeline failed: failed)r   r   r   rM   error)rS   r   r{   r'   r   r   r   r%   	Exceptionr   	exceptionr(   )r-   rK   rL   rM   rN   rO   rP   existing_file_idextracted_textr   r   r   result_file_ides                 r.   ingestzBaseRAGIngestion.ingest"  sT    * HL{{ HS H
 B
>,.>)	#'88)) $, $ N ZZ#)!__D8   F  $zzz88J 59JJ)!)% 5? 5 /+O^ %>>" / 52&:*:	 AB
 9/  	$$'<QC%@A$>> "!f 	sr   DB<	DC B>9C 5C 6C C'C ;D>C  C C 	D:DDDDD)N)r#   r   r$   zOptional['Router'])returnNone)r   r(   )NNN)rK    Optional[Tuple[str, bytes, str]]rL   Optional[str]rM   r   r   zCTuple[Optional[str], Optional[bytes], Optional[str], Optional[str]])rO   Optional[bytes]rP   r   r   r   )ru   r   rO   r   rv   boolr   	List[str])r   r   r   Optional[List[List[float]]])rO   r   rN   r   rP   r   r   r   r   r   r   z#Tuple[Optional[str], Optional[str]])rK   r   rL   r   rM   r   r   r   )__name__
__module____qualname____doc__r/   r,   propertyr>   rS   r   r{   r   r   r   r   rp   r0   r.   r   r   #   si    &*-(- #-0:& M M 7;"&!%	#I3#I  #I 	#I
 
M#IJ0%0 $0 
	0d:7:7 &:7 	:7
 
:7x== 
%=2 %   $	
  0 
- 6 7;"&!%	D3D  D 	D
 
Dr0   r   )#r   
__future__r   ra   abcr   r   typingr   r   r   r	   r
   r   r   r5   litellm._loggingr   litellm._uuidr   litellm.constantsr   r   &litellm.llms.custom_httpx.http_handlerr   r   "litellm.rag.ingestion.file_parsersr   litellm.rag.text_splittersr   litellm.types.ragr   r   r   r   rp   r0   r.   <module>r      sO   
 #  # H H H  +  G E E ACs Cr0   