o
    ñ‘OjMO  ã                	   @   s€  d Z ddlZddlZddlZddlZddlZddlZddlZddl	Z	ddl
ZddlZddlZddlmZ ddlmZ ddlmZmZmZmZmZmZmZmZmZmZmZ ddlm Z  ddl!m"Z" ddl#m$Z$m%Z% dd	l&m'Z' dd
l(m)Z) ddl*m+Z+ ddl,m-Z- ddl.m/Z/ ddl0m1Z1 ddl2m3Z3m4Z4 ddl5m6Z6 ddl7m8Z8m9Z9m:Z: er­ddlm;Z; ne<Z;e =e>¡Z?ej@jAjBZCeeDeDf ZEdeDdeeD fdd„ZFG dd„ deGƒZHde"ddfdd„ZIG dd„ deGƒZJdeDde-ddfd d!„ZKdeDde-de"fd"d#„ZLd$eEdeeD fd%d&„ZMd'eDdeDfd(d)„ZNd'eDdeDfd*d+„ZOe	 Pd,e	jQ¡ZRd-eDd.eSdeDfd/d0„ZTdeDdeDfd1d2„ZUd3eeDeeD f d4eDd5eDdee) fd6d7„ZVG d8d9„ d9ƒZWG d:d;„ d;e;ƒZXd<eXdeXfd=d>„ZYeYd?d@dee) fdAdB„ƒZZG dCd@„ d@ƒZ[G dDdE„ dEeƒZ\	dVdFe)dGeeDeGf dHeedI  ddfdJdK„Z]	LdWde"dMeSde[fdNdO„Z^	dVdFe)dee- ded@ fdPdQ„Z_G dRdS„ dSeƒZ`G dTdU„ dUƒZadS )XzO
The main purpose of this module is to expose LinkCollector.collect_sources().
é    N)Ú
HTMLParser)ÚValues)ÚTYPE_CHECKINGÚCallableÚDictÚIterableÚListÚMutableMappingÚ
NamedTupleÚOptionalÚSequenceÚTupleÚUnion)Úrequests)ÚResponse)Ú
RetryErrorÚSSLError)ÚNetworkConnectionError)ÚLink)ÚSearchScope)Ú
PipSession)Úraise_for_status)Úis_archive_file)ÚpairwiseÚredact_auth_from_url)Úvcsé   )ÚCandidatesFromPageÚ
LinkSourceÚbuild_source)ÚProtocolÚurlÚreturnc                 C   s6   t jD ]}|  ¡  |¡r| t|ƒ dv r|  S qdS )zgLook for VCS schemes in the URL.

    Returns the matched VCS scheme, or None if there's no match.
    z+:N)r   ÚschemesÚlowerÚ
startswithÚlen)r!   Úscheme© r(   úc/var/www/html/pharmsmart-cdr-gen/venv/lib/python3.10/site-packages/pip/_internal/index/collector.pyÚ_match_vcs_scheme:   s
   
€r*   c                       s*   e Zd Zdededdf‡ fdd„Z‡  ZS )Ú_NotAPIContentÚcontent_typeÚrequest_descr"   Nc                    s   t ƒ  ||¡ || _|| _d S ©N)ÚsuperÚ__init__r,   r-   )Úselfr,   r-   ©Ú	__class__r(   r)   r0   F   s   
z_NotAPIContent.__init__)Ú__name__Ú
__module__Ú__qualname__Ústrr0   Ú__classcell__r(   r(   r2   r)   r+   E   s    "r+   Úresponsec                 C   s2   | j  dd¡}| ¡ }| d¡rdS t|| jjƒ‚)z°
    Check the Content-Type header to ensure the response contains a Simple
    API Response.

    Raises `_NotAPIContent` if the content type is not a valid content-type.
    úContent-TypeÚUnknown)z	text/htmlz#application/vnd.pypi.simple.v1+htmlú#application/vnd.pypi.simple.v1+jsonN)ÚheadersÚgetr$   r%   r+   ÚrequestÚmethod)r9   r,   Úcontent_type_lr(   r(   r)   Ú_ensure_api_headerL   s   ÿrB   c                   @   s   e Zd ZdS )Ú_NotHTTPN)r4   r5   r6   r(   r(   r(   r)   rC   b   s    rC   Úsessionc                 C   sF   t j | ¡\}}}}}|dvrtƒ ‚|j| dd�}t|ƒ t|ƒ dS )zõ
    Send a HEAD request to the URL, and ensure the response contains a simple
    API Response.

    Raises `_NotHTTP` if the URL is not available for a HEAD request, or
    `_NotAPIContent` if the content type is not a valid content type.
    >   ÚhttpÚhttpsT)Úallow_redirectsN)ÚurllibÚparseÚurlsplitrC   Úheadr   rB   )r!   rD   r'   ÚnetlocÚpathÚqueryÚfragmentÚrespr(   r(   r)   Ú_ensure_api_responsef   s   rQ   c                 C   sx   t t| ƒjƒrt| |d� t dt| ƒ¡ |j| d g d¢¡ddœd�}t	|ƒ t
|ƒ t dt| ƒ|j d	d
¡¡ |S )aY  Access an Simple API response with GET, and return the response.

    This consists of three parts:

    1. If the URL looks suspiciously like an archive, send a HEAD first to
       check the Content-Type is HTML or Simple API, to avoid downloading a
       large file. Raise `_NotHTTP` if the content type cannot be determined, or
       `_NotAPIContent` if it is not HTML or a Simple API.
    2. Actually perform the request. Raise HTTP exceptions on network failures.
    3. Check the Content-Type header to make sure we got a Simple API response,
       and raise `_NotAPIContent` otherwise.
    ©rD   zGetting page %sz, )r<   z*application/vnd.pypi.simple.v1+html; q=0.1ztext/html; q=0.01z	max-age=0)ÚAcceptzCache-Control)r=   zFetched page %s as %sr:   r;   )r   r   ÚfilenamerQ   ÚloggerÚdebugr   r>   Újoinr   rB   r=   )r!   rD   rP   r(   r(   r)   Ú_get_simple_responsex   s&   ÿëþýrX   r=   c                 C   s<   | rd| v rt j ¡ }| d |d< | d¡}|rt|ƒS dS )z=Determine if we have any encoding information in our headers.r:   zcontent-typeÚcharsetN)ÚemailÚmessageÚMessageÚ	get_paramr7   )r=   ÚmrY   r(   r(   r)   Ú_get_encoding_from_headers·   s   

r_   Úpartc                 C   ó   t j t j | ¡¡S )zP
    Clean a "part" of a URL path (i.e. after splitting on "@" characters).
    )rH   rI   ÚquoteÚunquote©r`   r(   r(   r)   Ú_clean_url_path_partÂ   s   re   c                 C   ra   )z•
    Clean the first part of a URL path that corresponds to a local
    filesystem path (i.e. the first part after splitting on "@" characters).
    )rH   r?   Úpathname2urlÚurl2pathnamerd   r(   r(   r)   Ú_clean_file_url_pathÊ   s   
rh   z(@|%2F)rM   Úis_local_pathc                 C   s^   |rt }nt}t | ¡}g }tt |dg¡ƒD ]\}}| ||ƒ¡ | | ¡ ¡ qd 	|¡S )z*
    Clean the path portion of a URL.
    Ú )
rh   re   Ú_reserved_chars_reÚsplitr   Ú	itertoolsÚchainÚappendÚupperrW   )rM   ri   Ú
clean_funcÚpartsÚcleaned_partsÚto_cleanÚreservedr(   r(   r)   Ú_clean_url_pathÛ   s   

rv   c                 C   s6   t j | ¡}|j }t|j|d�}t j |j|d�¡S )z§
    Make sure a link is fully quoted.
    For example, if ' ' occurs in the URL, it will be replaced with "%20",
    and without double-quoting other characters.
    )ri   )rM   )rH   rI   ÚurlparserL   rv   rM   Ú
urlunparseÚ_replace)r!   Úresultri   rM   r(   r(   r)   Ú_clean_linkñ   s   r{   Úelement_attribsÚpage_urlÚbase_urlc                 C   sL   |   d¡}|s	dS ttj ||¡ƒ}|   d¡}|   d¡}t||||d�}|S )zW
    Convert an anchor element's attributes in a simple repository page to a Link.
    ÚhrefNzdata-requires-pythonzdata-yanked)Ú
comes_fromÚrequires_pythonÚyanked_reason)r>   r{   rH   rI   Úurljoinr   )r|   r}   r~   r   r!   Ú	pyrequirer‚   Úlinkr(   r(   r)   Ú_create_link_from_element   s   


ür†   c                   @   s6   e Zd Zddd„Zdedefdd	„Zdefd
d„ZdS )ÚCacheablePageContentÚpageÚIndexContentr"   Nc                 C   s   |j sJ ‚|| _d S r.   )Úcache_link_parsingrˆ   ©r1   rˆ   r(   r(   r)   r0     s   

zCacheablePageContent.__init__Úotherc                 C   s   t |t| ƒƒo| jj|jjkS r.   )Ú
isinstanceÚtyperˆ   r!   )r1   rŒ   r(   r(   r)   Ú__eq__  s   zCacheablePageContent.__eq__c                 C   s   t | jjƒS r.   )Úhashrˆ   r!   ©r1   r(   r(   r)   Ú__hash__"  s   zCacheablePageContent.__hash__)rˆ   r‰   r"   N)	r4   r5   r6   r0   ÚobjectÚboolr�   Úintr’   r(   r(   r(   r)   r‡     s    
r‡   c                   @   s"   e Zd Zdddee fdd„ZdS )Ú
ParseLinksrˆ   r‰   r"   c                 C   s   d S r.   r(   r‹   r(   r(   r)   Ú__call__'  s   zParseLinks.__call__N)r4   r5   r6   r   r   r—   r(   r(   r(   r)   r–   &  s    r–   Úfnc                    sP   t jdd�dtdtt f‡ fdd„ƒ‰t  ˆ ¡dddtt f‡ ‡fd	d
„ƒ}|S )zÚ
    Given a function that parses an Iterable[Link] from an IndexContent, cache the
    function's result (keyed by CacheablePageContent), unless the IndexContent
    `page` has `page.cache_link_parsing == False`.
    N)ÚmaxsizeÚcacheable_pager"   c                    s   t ˆ | jƒƒS r.   )Úlistrˆ   )rš   )r˜   r(   r)   Úwrapper2  s   z*with_cached_index_content.<locals>.wrapperrˆ   r‰   c                    s   | j r	ˆt| ƒƒS tˆ | ƒƒS r.   )rŠ   r‡   r›   )rˆ   ©r˜   rœ   r(   r)   Úwrapper_wrapper6  s   z2with_cached_index_content.<locals>.wrapper_wrapper)Ú	functoolsÚ	lru_cacher‡   r   r   Úwraps)r˜   rž   r(   r�   r)   Úwith_cached_index_content+  s
   
r¢   rˆ   r‰   c              
   c   s  � | j  ¡ }| d¡rQt | j¡}| dg ¡D ]9}| d¡}|du r#q| d¡}|r2t|tƒs2d}n|s6d}t	t
tj | j|¡ƒ| j| d¡|| di ¡d	�V  qt| jƒ}| jpZd
}| | j |¡¡ | j}|jpk|}	|jD ]}
t|
||	d�}|du r}qo|V  qodS )z\
    Parse a Simple API's Index Content, and yield its anchor elements as Link objects.
    r<   Úfilesr!   NÚyankedrj   zrequires-pythonÚhashes)r€   r�   r‚   r¥   zutf-8)r}   r~   )r,   r$   r%   ÚjsonÚloadsÚcontentr>   r�   r7   r   r{   rH   rI   rƒ   r!   ÚHTMLLinkParserÚencodingÚfeedÚdecoder~   Úanchorsr†   )rˆ   rA   ÚdataÚfileÚfile_urlr‚   Úparserrª   r!   r~   Úanchorr…   r(   r(   r)   Úparse_links?  sF   €





û



ýør³   c                   @   sH   e Zd ZdZ	ddededee dededd	fd
d„Zdefdd„Z	d	S )r‰   z5Represents one response (or page), along with its URLTr¨   r,   rª   r!   rŠ   r"   Nc                 C   s"   || _ || _|| _|| _|| _dS )am  
        :param encoding: the encoding to decode the given content.
        :param url: the URL from which the HTML was downloaded.
        :param cache_link_parsing: whether links parsed from this page's url
                                   should be cached. PyPI index urls should
                                   have this set to False, for example.
        N)r¨   r,   rª   r!   rŠ   )r1   r¨   r,   rª   r!   rŠ   r(   r(   r)   r0   q  s
   
zIndexContent.__init__c                 C   s
   t | jƒS r.   )r   r!   r‘   r(   r(   r)   Ú__str__†  s   
zIndexContent.__str__©T)
r4   r5   r6   Ú__doc__Úbytesr7   r   r”   r0   r´   r(   r(   r(   r)   r‰   n  s"    úþýüûú
ùc                       sv   e Zd ZdZdeddf‡ fdd„Zdedeeeee f  ddfd	d
„Z	deeeee f  dee fdd„Z
‡  ZS )r©   zf
    HTMLParser that keeps the first base HREF and a list of all anchor
    elements' attributes.
    r!   r"   Nc                    s$   t ƒ jdd� || _d | _g | _d S )NT)Úconvert_charrefs)r/   r0   r!   r~   r­   )r1   r!   r2   r(   r)   r0   �  s   
zHTMLLinkParser.__init__ÚtagÚattrsc                 C   sR   |dkr| j d u r|  |¡}|d ur|| _ d S d S |dkr'| j t|ƒ¡ d S d S )NÚbaseÚa)r~   Úget_hrefr­   ro   Údict)r1   r¹   rº   r   r(   r(   r)   Úhandle_starttag—  s   

ÿÿzHTMLLinkParser.handle_starttagc                 C   s"   |D ]\}}|dkr|  S qd S )Nr   r(   )r1   rº   ÚnameÚvaluer(   r(   r)   r½   Ÿ  s
   ÿzHTMLLinkParser.get_href)r4   r5   r6   r¶   r7   r0   r   r   r   r¿   r½   r8   r(   r(   r2   r)   r©   Š  s
    &.r©   r…   ÚreasonÚmeth).Nc                 C   s   |d u rt j}|d| |ƒ d S )Nz%Could not fetch URL %s: %s - skipping)rU   rV   )r…   rÂ   rÃ   r(   r(   r)   Ú_handle_get_simple_fail¦  s   rÄ   TrŠ   c                 C   s&   t | jƒ}t| j| jd || j|d�S )Nr:   )rª   r!   rŠ   )r_   r=   r‰   r¨   r!   )r9   rŠ   rª   r(   r(   r)   Ú_make_index_content°  s   
ûrÅ   c           
   
   C   s  |d u rt dƒ‚| j dd¡d }t|ƒ}|r t d|| ¡ d S tj |¡\}}}}}}|dkrPt	j
 tj |¡¡rP| d¡sC|d7 }tj |d¡}t d	|¡ zt||d
�}W n¦ tyh   t d| ¡ Y d S  ty„ } zt d| |j|j¡ W Y d }~d S d }~w ty› } zt| |ƒ W Y d }~d S d }~w ty² } zt| |ƒ W Y d }~d S d }~w tyÔ } zd}	|	t|ƒ7 }	t| |	tjd� W Y d }~d S d }~w tjyï } zt| d|› �ƒ W Y d }~d S d }~w tjyþ   t| dƒ Y d S w t|| j d�S )Nz?_get_html_page() missing 1 required keyword argument: 'session'ú#r   r   zICannot look at %s URL %s because it does not support lookup as web pages.r¯   ú/z
index.htmlz# file: URL is directory, getting %srR   z`Skipping page %s because it looks like an archive, and cannot be checked by a HTTP HEAD request.zºSkipping page %s because the %s request got Content-Type: %s. The only supported Content-Types are application/vnd.pypi.simple.v1+json, application/vnd.pypi.simple.v1+html, and text/htmlz4There was a problem confirming the ssl certificate: )rÃ   zconnection error: z	timed out)rŠ   )!Ú	TypeErrorr!   rl   r*   rU   ÚwarningrH   rI   rw   ÚosrM   Úisdirr?   rg   Úendswithrƒ   rV   rX   rC   r+   r-   r,   r   rÄ   r   r   r7   Úinfor   ÚConnectionErrorÚTimeoutrÅ   rŠ   )
r…   rD   r!   Ú
vcs_schemer'   Ú_rM   rP   ÚexcrÂ   r(   r(   r)   Ú_get_index_content½  sv   ÿý
ýéú€ò€ô€ö€ú€üürÓ   c                   @   s.   e Zd ZU eee  ed< eee  ed< dS )ÚCollectedSourcesÚ
find_linksÚ
index_urlsN)r4   r5   r6   r   r   r   Ú__annotations__r(   r(   r(   r)   rÔ     s   
 rÔ   c                
   @   sŠ   e Zd ZdZdededdfdd„Ze	dded	ed
e	dd fdd„ƒZ
edee fdd„ƒZdedee fdd„Zdededefdd„ZdS )ÚLinkCollectorzµ
    Responsible for collecting Link objects from all configured locations,
    making network requests as needed.

    The class's main method is its collect_sources() method.
    rD   Úsearch_scoper"   Nc                 C   s   || _ || _d S r.   )rÙ   rD   )r1   rD   rÙ   r(   r(   r)   r0     s   
zLinkCollector.__init__FÚoptionsÚsuppress_no_indexc                 C   s`   |j g|j }|jr|st dd dd„ |D ƒ¡¡ g }|jp g }tj||d�}t	||d�}|S )zÆ
        :param session: The Session to use to make requests.
        :param suppress_no_index: Whether to ignore the --no-index option
            when constructing the SearchScope object.
        zIgnoring indexes: %sú,c                 s   s   � | ]}t |ƒV  qd S r.   )r   )Ú.0r!   r(   r(   r)   Ú	<genexpr>'  s   € z'LinkCollector.create.<locals>.<genexpr>©rÕ   rÖ   )rD   rÙ   )
Ú	index_urlÚextra_index_urlsÚno_indexrU   rV   rW   rÕ   r   ÚcreaterØ   )ÚclsrD   rÚ   rÛ   rÖ   rÕ   rÙ   Úlink_collectorr(   r(   r)   rã     s"   
þ
þþzLinkCollector.createc                 C   s   | j jS r.   )rÙ   rÕ   r‘   r(   r(   r)   rÕ   8  s   zLinkCollector.find_linksÚlocationc                 C   s   t || jd�S )z>
        Fetch an HTML page containing package links.
        rR   )rÓ   rD   )r1   ræ   r(   r(   r)   Úfetch_response<  s   zLinkCollector.fetch_responseÚproject_nameÚcandidates_from_pagec                    s¦   t  ‡ ‡fdd„ˆj |¡D ƒ¡ ¡ }t  ‡ ‡fdd„ˆjD ƒ¡ ¡ }t tj	¡rIdd„ t
 ||¡D ƒ}t|ƒ› d|› d�g| }t d |¡¡ tt|ƒt|ƒd	�S )
Nc                 3   ó&   � | ]}t |ˆ ˆjjd d d�V  qdS )F©ré   Úpage_validatorÚ
expand_dirrŠ   N©r   rD   Úis_secure_origin©rÝ   Úloc©ré   r1   r(   r)   rÞ   H  ó   € ùû
ÿz0LinkCollector.collect_sources.<locals>.<genexpr>c                 3   rê   )Trë   Nrî   rð   rò   r(   r)   rÞ   R  ró   c                 S   s*   g | ]}|d ur|j d urd|j › �‘qS )Nz* )r…   )rÝ   Úsr(   r(   r)   Ú
<listcomp>^  s    ýý
ÿz1LinkCollector.collect_sources.<locals>.<listcomp>z' location(s) to search for versions of ú:Ú
rß   )ÚcollectionsÚOrderedDictrÙ   Úget_index_urls_locationsÚvaluesrÕ   rU   ÚisEnabledForÚloggingÚDEBUGrm   rn   r&   rV   rW   rÔ   r›   )r1   rè   ré   Úindex_url_sourcesÚfind_links_sourcesÚlinesr(   rò   r)   Úcollect_sourcesB  s2   
ø	÷
ø	÷
þ
ÿÿýþzLinkCollector.collect_sources)F)r4   r5   r6   r¶   r   r   r0   Úclassmethodr   r”   rã   Úpropertyr   r7   rÕ   r   r   r‰   rç   r   rÔ   r  r(   r(   r(   r)   rØ     s<    þý
üüþýüû þýürØ   r.   rµ   )br¶   rø   Úemail.messagerZ   rŸ   rm   r¦   rý   rÊ   ÚreÚurllib.parserH   Úurllib.requestÚxml.etree.ElementTreeÚxmlÚhtml.parserr   Úoptparser   Útypingr   r   r   r   r   r	   r
   r   r   r   r   Úpip._vendorr   Úpip._vendor.requestsr   Úpip._vendor.requests.exceptionsr   r   Úpip._internal.exceptionsr   Úpip._internal.models.linkr   Ú!pip._internal.models.search_scoper   Úpip._internal.network.sessionr   Úpip._internal.network.utilsr   Úpip._internal.utils.filetypesr   Úpip._internal.utils.miscr   r   Úpip._internal.vcsr   Úsourcesr   r   r   r    r“   Ú	getLoggerr4   rU   ÚetreeÚElementTreeÚElementÚHTMLElementr7   ÚResponseHeadersr*   Ú	Exceptionr+   rB   rC   rQ   rX   r_   re   rh   ÚcompileÚ
IGNORECASErk   r”   rv   r{   r†   r‡   r–   r¢   r³   r‰   r©   rÄ   rÅ   rÓ   rÔ   rØ   r(   r(   r(   r)   Ú<module>   s²    4

?ÿþý
ü.ýÿ
þ
ý
üÿÿÿ
þÿÿÿ
þD