o
    ­—ŽjÂ@  ã                   @  sl  d Z ddlmZ ddlZddlZddlZddlZddlZddl	Z	ddl
Z
ddlZddlmZmZmZmZ ddlmZ ddlmZ ddlmZ ddlmZmZ dd	lmZ dd
lmZ ddlm Z  ddl!m"Z"m#Z#m$Z$m%Z%m&Z&m'Z' ddl(m)Z) ddl*m+Z+ ddl,m-Z- ddl.m/Z/ ddl0m1Z1 ddl2m3Z3 ddl4m5Z5 ddl6m7Z7 ddl8m9Z9m:Z:m;Z; e	 <e=¡Z>ee?e?f Z@dWdd„ZAG dd„ deBƒZCdXd"d#„ZDG d$d%„ d%eBƒZEdYd(d)„ZF	*dZd[d-d.„ZGd\d1d2„ZHG d3d4„ d4ƒZIG d5d6„ d6eƒZJd]d8d9„ZKeKd^d=d>„ƒZLed?d@�G dAd;„ d;ƒƒZMG dBdC„ dCeƒZN	d_d`dJdK„ZO	?dadbdMdN„ZPd*dOœdcdQdR„ZQG dSdT„ dTeƒZRG dUdV„ dVƒZSdS )dzO
The main purpose of this module is to expose LinkCollector.collect_sources().
é    )ÚannotationsN)ÚCallableÚIterableÚMutableMappingÚSequence)Ú	dataclass)Ú
HTMLParser)ÚValues)Ú
NamedTupleÚProtocol)Úcanonicalize_name)ÚResponse)Ú
RetryError)ÚConnectionFailedErrorÚConnectionTimeoutErrorÚNetworkConnectionErrorÚProxyConnectionErrorÚSSLMissingErrorÚSSLVerificationError)ÚLink)ÚSearchScope)Ú
PipSession)Úraise_for_status)Úis_archive_file©Úredact_auth_from_url)Úurl_to_path)Úvcsé   )ÚCandidatesFromPageÚ
LinkSourceÚbuild_sourceÚurlÚstrÚreturnú
str | Nonec                 C  s6   t jD ]}|  ¡  |¡r| t|ƒ dv r|  S qdS )zgLook for VCS schemes in the URL.

    Returns the matched VCS scheme, or None if there's no match.
    z+:N)r   ÚschemesÚlowerÚ
startswithÚlen)r"   Úscheme© r+   úZ/var/www/kodo/Anonymous/send/lib/python3.10/site-packages/pip/_internal/index/collector.pyÚ_match_vcs_scheme4   s
   
€r-   c                      s   e Zd Zd‡ fdd„Z‡  ZS )	Ú_NotAPIContentÚcontent_typer#   Úrequest_descr$   ÚNonec                   s   t ƒ  ||¡ || _|| _d S ©N)ÚsuperÚ__init__r/   r0   )Úselfr/   r0   ©Ú	__class__r+   r,   r4   @   s   
z_NotAPIContent.__init__)r/   r#   r0   r#   r$   r1   )Ú__name__Ú
__module__Ú__qualname__r4   Ú__classcell__r+   r+   r6   r,   r.   ?   s    r.   Úresponser   r1   c                 C  s2   | j  dd¡}| ¡ }| d¡rdS t|| jjƒ‚)z°
    Check the Content-Type header to ensure the response contains a Simple
    API Response.

    Raises `_NotAPIContent` if the content type is not a valid content-type.
    úContent-TypeÚUnknown)z	text/htmlz#application/vnd.pypi.simple.v1+htmlú#application/vnd.pypi.simple.v1+jsonN)ÚheadersÚgetr'   r(   r.   ÚrequestÚmethod)r<   r/   Úcontent_type_lr+   r+   r,   Ú_ensure_api_headerF   s   ÿrE   c                   @  s   e Zd ZdS )Ú_NotHTTPN)r8   r9   r:   r+   r+   r+   r,   rF   \   s    rF   Úsessionr   c                 C  sF   t j | ¡\}}}}}|dvrtƒ ‚|j| dd�}t|ƒ t|ƒ dS )zõ
    Send a HEAD request to the URL, and ensure the response contains a simple
    API Response.

    Raises `_NotHTTP` if the URL is not available for a HEAD request, or
    `_NotAPIContent` if the content type is not a valid content type.
    >   ÚhttpÚhttpsT)Úallow_redirectsN)ÚurllibÚparseÚurlsplitrF   Úheadr   rE   )r"   rG   r*   ÚnetlocÚpathÚqueryÚfragmentÚrespr+   r+   r,   Ú_ensure_api_response`   s   rT   FÚforce_revalidateÚboolc                 C  s�   t t| ƒjƒrt| |d� t dt| ƒ¡ dd g d¢¡i}|r)t d¡ d|d< |j| |d	�}t	|ƒ t
|ƒ t d
t| ƒ|j dd¡¡ |S )aY  Access an Simple API response with GET, and return the response.

    This consists of three parts:

    1. If the URL looks suspiciously like an archive, send a HEAD first to
       check the Content-Type is HTML or Simple API, to avoid downloading a
       large file. Raise `_NotHTTP` if the content type cannot be determined, or
       `_NotAPIContent` if it is not HTML or a Simple API.
    2. Actually perform the request. Raise HTTP exceptions on network failures.
    3. Check the Content-Type header to make sure we got a Simple API response,
       and raise `_NotAPIContent` otherwise.
    )rG   zGetting page %sÚAcceptz, )r?   z*application/vnd.pypi.simple.v1+html; q=0.1ztext/html; q=0.01z"Refreshing package index response.z	max-age=0zCache-Control)r@   zFetched page %s as %sr=   r>   )r   r   ÚfilenamerT   ÚloggerÚdebugr   ÚjoinrA   r   rE   r@   )r"   rG   rU   r@   rS   r+   r+   r,   Ú_get_simple_responser   s&   ÿÿ

ýr\   r@   ÚResponseHeadersc                 C  s<   | rd| v rt j ¡ }| d |d< | d¡}|rt|ƒS dS )z=Determine if we have any encoding information in our headers.r=   zcontent-typeÚcharsetN)ÚemailÚmessageÚMessageÚ	get_paramr#   )r@   Úmr^   r+   r+   r,   Ú_get_encoding_from_headers­   s   

rd   c                   @  s*   e Zd Zddd„Zdd
d„Zddd„ZdS )ÚCacheablePageContentÚpageÚIndexContentr$   r1   c                 C  s   |j sJ ‚|| _d S r2   )Úcache_link_parsingrf   ©r5   rf   r+   r+   r,   r4   ¹   s   

zCacheablePageContent.__init__ÚotherÚobjectrV   c                 C  s   t |t| ƒƒo| jj|jjkS r2   )Ú
isinstanceÚtyperf   r"   )r5   rj   r+   r+   r,   Ú__eq__½   s   zCacheablePageContent.__eq__Úintc                 C  s   t | jjƒS r2   )Úhashrf   r"   ©r5   r+   r+   r,   Ú__hash__À   s   zCacheablePageContent.__hash__N)rf   rg   r$   r1   )rj   rk   r$   rV   )r$   ro   )r8   r9   r:   r4   rn   rr   r+   r+   r+   r,   re   ¸   s    

re   c                   @  s   e Zd Zddd„ZdS )	Ú
ParseLinksrf   rg   r$   úIterable[Link]c                 C  s   d S r2   r+   ri   r+   r+   r,   Ú__call__Å   s    zParseLinks.__call__N©rf   rg   r$   rt   )r8   r9   r:   ru   r+   r+   r+   r,   rs   Ä   s    rs   Úfnc                   s2   t jd‡ fdd„ƒ‰t  ˆ ¡d‡ ‡fd	d
„ƒ}|S )zÚ
    Given a function that parses an Iterable[Link] from an IndexContent, cache the
    function's result (keyed by CacheablePageContent), unless the IndexContent
    `page` has `page.cache_link_parsing == False`.
    Úcacheable_pagere   r$   ú
list[Link]c                   s   t ˆ | jƒƒS r2   )Úlistrf   )rx   )rw   r+   r,   ÚwrapperÏ   s   z*with_cached_index_content.<locals>.wrapperrf   rg   c                   s   | j r	ˆt| ƒƒS tˆ | ƒƒS r2   )rh   re   rz   )rf   ©rw   r{   r+   r,   Úwrapper_wrapperÓ   s   z2with_cached_index_content.<locals>.wrapper_wrapperN)rx   re   r$   ry   )rf   rg   r$   ry   )Ú	functoolsÚcacheÚwraps)rw   r}   r+   r|   r,   Úwith_cached_index_contentÈ   s
   r�   rf   rg   rt   c           
      c  s¼   � | j  ¡ }| d¡r+t | j¡}| dg ¡D ]}t || j	¡}|du r%q|V  qdS t
| j	ƒ}| jp4d}| | j |¡¡ | j	}|jpE|}|jD ]}	tj|	||d�}|du rXqI|V  qIdS )z\
    Parse a Simple API's Index Content, and yield its anchor elements as Link objects.
    r?   ÚfilesNzutf-8)Úpage_urlÚbase_url)r/   r'   r(   ÚjsonÚloadsÚcontentrA   r   Ú	from_jsonr"   ÚHTMLLinkParserÚencodingÚfeedÚdecoder„   ÚanchorsÚfrom_element)
rf   rD   ÚdataÚfileÚlinkÚparserrŠ   r"   r„   Úanchorr+   r+   r,   Úparse_linksÜ   s*   €





ür”   T)Úfrozenc                   @  sH   e Zd ZU dZded< ded< ded< ded< d	Zd
ed< ddd„ZdS )rg   aŒ  Represents one response (or page), along with its URL.

    :param encoding: the encoding to decode the given content.
    :param url: the URL from which the HTML was downloaded.
    :param cache_link_parsing: whether links parsed from this page's url
                               should be cached. PyPI index urls should
                               have this set to False, for example.
    Úbytesr‡   r#   r/   r%   rŠ   r"   TrV   rh   r$   c                 C  s
   t | jƒS r2   )r   r"   rq   r+   r+   r,   Ú__str__
  s   
zIndexContent.__str__N)r$   r#   )r8   r9   r:   Ú__doc__Ú__annotations__rh   r—   r+   r+   r+   r,   rg   ù   s   
 	c                      s6   e Zd ZdZd‡ fdd„Zddd„Zddd„Z‡  ZS )r‰   zf
    HTMLParser that keeps the first base HREF and a list of all anchor
    elements' attributes.
    r"   r#   r$   r1   c                   s$   t ƒ jdd� || _d | _g | _d S )NT)Úconvert_charrefs)r3   r4   r"   r„   r�   )r5   r"   r6   r+   r,   r4     s   
zHTMLLinkParser.__init__ÚtagÚattrsúlist[tuple[str, str | None]]c                 C  sR   |dkr| j d u r|  |¡}|d ur|| _ d S d S |dkr'| j t|ƒ¡ d S d S )NÚbaseÚa)r„   Úget_hrefr�   ÚappendÚdict)r5   r›   rœ   Úhrefr+   r+   r,   Úhandle_starttag  s   

ÿÿzHTMLLinkParser.handle_starttagr%   c                 C  s"   |D ]\}}|dkr|  S qd S )Nr£   r+   )r5   rœ   ÚnameÚvaluer+   r+   r,   r    #  s
   ÿzHTMLLinkParser.get_href)r"   r#   r$   r1   )r›   r#   rœ   r�   r$   r1   )rœ   r�   r$   r%   )r8   r9   r:   r˜   r4   r¤   r    r;   r+   r+   r6   r,   r‰     s
    
r‰   r‘   r   Úreasonústr | ExceptionÚmethúCallable[..., None] | Nonec                 C  s   |d u rt j}|d| |ƒ d S )Nz%Could not fetch URL %s: %s - skipping)rY   rZ   )r‘   r§   r©   r+   r+   r,   Ú_handle_get_simple_fail*  s   r«   rh   c                 C  s&   t | jƒ}t| j| jd || j|d�S )Nr=   )rŠ   r"   rh   )rd   r@   rg   r‡   r"   )r<   rh   rŠ   r+   r+   r,   Ú_make_index_content4  s   
ûr¬   )rU   úIndexContent | Nonec             
   C  s  | j  dd¡d }t|ƒ}|rt d|| ¡ d S | d¡r;tj t	|ƒ¡r;| 
d¡s.|d7 }tj |d¡}t d|¡ z	t|||d	�}W n· tyT   t d
| ¡ Y d S  typ } zt d| |j|j¡ W Y d }~d S d }~w ttfy‰ } zt| |ƒ W Y d }~d S d }~w ttfy« } zd|j› �}t| |tjd� W Y d }~d S d }~w tyÆ } zt| d|j› �ƒ W Y d }~d S d }~w tyá } zt| d|j› �ƒ W Y d }~d S d }~w tyû } zt| t|jƒƒ W Y d }~d S d }~ww t|| j d�S )Nú#r   r   zICannot look at %s URL %s because it does not support lookup as web pages.zfile:ú/z
index.htmlz# file: URL is directory, getting %s©rG   rU   z`Skipping page %s because it looks like an archive, and cannot be checked by a HTTP HEAD request.zºSkipping page %s because the %s request got Content-Type: %s. The only supported Content-Types are application/vnd.pypi.simple.v1+json, application/vnd.pypi.simple.v1+html, and text/htmlz4There was a problem confirming the ssl certificate: )r©   zconnection error: zproxy connection error: )rh   )!r"   Úsplitr-   rY   Úwarningr(   ÚosrP   Úisdirr   ÚendswithrK   rL   ÚurljoinrZ   r\   rF   r.   r0   r/   r   r   r«   r   r   ÚcontextÚinfor   r   r   r#   r¬   rh   )r‘   rG   rU   r"   Ú
vcs_schemerS   Úexcr§   r+   r+   r,   Ú_get_index_contentA  sp   ý

ÿýêú€ó€õ	€ø€ú€ü€ür»   c                   @  s   e Zd ZU ded< ded< dS )ÚCollectedSourceszSequence[LinkSource | None]Ú
find_linksÚ
index_urlsN)r8   r9   r:   r™   r+   r+   r+   r,   r¼   ƒ  s   
 r¼   c                   @  sR   e Zd ZdZd#dd	„Ze	
d$d%dd„ƒZed&dd„ƒZ	d'd(dd„Z	d)d!d"„Z
dS )*ÚLinkCollectorzµ
    Responsible for collecting Link objects from all configured locations,
    making network requests as needed.

    The class's main method is its collect_sources() method.
    rG   r   Úsearch_scoper   r$   r1   c                 C  s   || _ || _d S r2   )rÀ   rG   )r5   rG   rÀ   r+   r+   r,   r4   �  s   
zLinkCollector.__init__FÚoptionsr	   Úsuppress_no_indexrV   c                 C  sd   |j g|j }|jr|st dd dd„ |D ƒ¡¡ g }|jp g }tj|||jd�}t	||d�}|S )zÆ
        :param session: The Session to use to make requests.
        :param suppress_no_index: Whether to ignore the --no-index option
            when constructing the SearchScope object.
        zIgnoring indexes: %sú,c                 s  s   � | ]}t |ƒV  qd S r2   r   )Ú.0r"   r+   r+   r,   Ú	<genexpr>¨  s   € z'LinkCollector.create.<locals>.<genexpr>)r½   r¾   Úno_index)rG   rÀ   )
Ú	index_urlÚextra_index_urlsrÆ   rY   rZ   r[   r½   r   Úcreater¿   )ÚclsrG   rÁ   rÂ   r¾   r½   rÀ   Úlink_collectorr+   r+   r,   rÉ   ˜  s$   
þ
ýþzLinkCollector.createú	list[str]c                 C  s   | j jS r2   )rÀ   r½   rq   r+   r+   r,   r½   º  s   zLinkCollector.find_linksNÚlocationr   Úpackage_namer%   r­   c                 C  s4   | j j}d|v p|duot|ƒ|v }t|| j |d�S )z>
        Fetch an HTML page containing package links.
        z:all:Nr°   )rG   Úrefresh_packager   r»   )r5   rÍ   rÎ   rU   Úshould_force_revalidater+   r+   r,   Úfetch_response¾  s   	
þýzLinkCollector.fetch_responseÚproject_namer#   Úcandidates_from_pager   r¼   c                   sª   t  ‡ ‡‡fdd„ˆj ˆ¡D ƒ¡ ¡ }t  ‡ ‡‡fdd„ˆjD ƒ¡ ¡ }t tj	¡rKdd„ t
 ||¡D ƒ}t|ƒ› dˆ› d�g| }t d |¡¡ tt|ƒt|ƒd	�S )
Nc              	   3  ó(   � | ]}t |ˆ ˆjjd d ˆd�V  qdS )F©rÓ   Úpage_validatorÚ
expand_dirrh   rÒ   N©r!   rG   Úis_secure_origin©rÄ   Úloc©rÓ   rÒ   r5   r+   r,   rÅ   Ø  ó   € 	øú
ÿz0LinkCollector.collect_sources.<locals>.<genexpr>c              	   3  rÔ   )TrÕ   NrØ   rÚ   rÜ   r+   r,   rÅ   ã  rÝ   c                 S  s*   g | ]}|d ur|j d urd|j › �‘qS )Nz* )r‘   )rÄ   Úsr+   r+   r,   Ú
<listcomp>ð  s
    
þz1LinkCollector.collect_sources.<locals>.<listcomp>z' location(s) to search for versions of ú:Ú
)r½   r¾   )ÚcollectionsÚOrderedDictrÀ   Úget_index_urls_locationsÚvaluesr½   rY   ÚisEnabledForÚloggingÚDEBUGÚ	itertoolsÚchainr)   rZ   r[   r¼   rz   )r5   rÒ   rÓ   Úindex_url_sourcesÚfind_links_sourcesÚlinesr+   rÜ   r,   Úcollect_sourcesÒ  s2   
	÷
ö	÷
ö
þ
ÿÿýþzLinkCollector.collect_sources)rG   r   rÀ   r   r$   r1   ©F)rG   r   rÁ   r	   rÂ   rV   r$   r¿   )r$   rÌ   r2   )rÍ   r   rÎ   r%   r$   r­   )rÒ   r#   rÓ   r   r$   r¼   )r8   r9   r:   r˜   r4   ÚclassmethodrÉ   Úpropertyr½   rÑ   rî   r+   r+   r+   r,   r¿   ˆ  s    
ü!ÿr¿   )r"   r#   r$   r%   )r<   r   r$   r1   )r"   r#   rG   r   r$   r1   rï   )r"   r#   rG   r   rU   rV   r$   r   )r@   r]   r$   r%   )rw   rs   r$   rs   rv   r2   )r‘   r   r§   r¨   r©   rª   r$   r1   )T)r<   r   rh   rV   r$   rg   )r‘   r   rG   r   rU   rV   r$   r­   )Tr˜   Ú
__future__r   râ   Úemail.messager_   r~   ré   r…   rç   r³   Úurllib.parserK   Úcollections.abcr   r   r   r   Údataclassesr   Úhtml.parserr   Úoptparser	   Útypingr
   r   Úpip._vendor.packaging.utilsr   Úpip._vendor.requestsr   Úpip._vendor.requests.exceptionsr   Úpip._internal.exceptionsr   r   r   r   r   r   Úpip._internal.models.linkr   Ú!pip._internal.models.search_scoper   Úpip._internal.network.sessionr   Úpip._internal.network.utilsr   Úpip._internal.utils.filetypesr   Úpip._internal.utils.miscr   Úpip._internal.utils.urlsr   Úpip._internal.vcsr   Úsourcesr   r    r!   Ú	getLoggerr8   rY   r#   r]   r-   Ú	Exceptionr.   rE   rF   rT   r\   rd   re   rs   r�   r”   rg   r‰   r«   r¬   r»   r¼   r¿   r+   r+   r+   r,   Ú<module>   sl     



ý
;
ýÿüB