a
    Oc\O                     @   sL  d Z ddlZddlZddlZddlZddlZddlZddlZddl	Z	ddl
ZddlZddlZddlmZ ddlmZ ddlmZmZmZmZmZmZmZmZmZmZmZ ddlm Z  ddl!m"Z" ddl#m$Z$m%Z% dd	l&m'Z' dd
l(m)Z) ddl*m+Z+ ddl,m-Z- ddl.m/Z/ ddl0m1Z1 ddl2m3Z3m4Z4 ddl5m6Z6 ddl7m8Z8m9Z9m:Z: er\ddlm;Z; ne<Z;e=e>Z?ej@jAjBZCeeDeDf ZEeDeeD dddZFG dd deGZHe"ddddZIG dd deGZJeDe-dddd ZKeDe-e"dd!d"ZLeEeeD d#d$d%ZMeDeDd&d'd(ZNeDeDd&d)d*ZOe	Pd+e	jQZReDeSeDd,d-d.ZTeDeDdd/d0ZUeeDeeD f eDeDee) d1d2d3ZVG d4d5 d5ZWG d6d7 d7e;ZXeXeXd8d9d:ZYeYd;ee) d<d=d>ZZG d?d; d;Z[G d@dA dAeZ\dQe)eeDeGf eedB  ddCdDdEZ]dRe"eSe[dGdHdIZ^dSe)ee- ed; dJdKdLZ_G dMdN dNeZ`G dOdP dPZadS )TzO
The main purpose of this module is to expose LinkCollector.collect_sources().
    N)
HTMLParser)Values)TYPE_CHECKINGCallableDictIterableListMutableMapping
NamedTupleOptionalSequenceTupleUnion)requests)Response)
RetryErrorSSLError)NetworkConnectionError)Link)SearchScope)
PipSession)raise_for_status)is_archive_file)pairwiseredact_auth_from_url)vcs   )CandidatesFromPage
LinkSourcebuild_source)Protocolurlreturnc                 C   s6   t jD ]*}|  |r| t| dv r|  S qdS )zgLook for VCS schemes in the URL.

    Returns the matched VCS scheme, or None if there's no match.
    z+:N)r   Zschemeslower
startswithlen)r"   scheme r(   U/var/www/html/megastudio/lib/python3.9/site-packages/pip/_internal/index/collector.py_match_vcs_scheme:   s    

r*   c                       s&   e Zd Zeedd fddZ  ZS )_NotAPIContentN)content_typerequest_descr#   c                    s   t  || || _|| _d S N)super__init__r,   r-   )selfr,   r-   	__class__r(   r)   r0   F   s    z_NotAPIContent.__init__)__name__
__module____qualname__strr0   __classcell__r(   r(   r2   r)   r+   E   s   r+   )responser#   c                 C   s6   | j dd}| }|dr$dS t|| jjdS )z
    Check the Content-Type header to ensure the response contains a Simple
    API Response.

    Raises `_NotAPIContent` if the content type is not a valid content-type.
    Content-TypeUnknown)z	text/htmlz#application/vnd.pypi.simple.v1+html#application/vnd.pypi.simple.v1+jsonN)headersgetr$   r%   r+   requestmethod)r9   r,   content_type_lr(   r(   r)   _ensure_api_headerL   s    rB   c                   @   s   e Zd ZdS )_NotHTTPN)r4   r5   r6   r(   r(   r(   r)   rC   b   s   rC   )r"   sessionr#   c                 C   sF   t j| \}}}}}|dvr$t |j| dd}t| t| dS )z
    Send a HEAD request to the URL, and ensure the response contains a simple
    API Response.

    Raises `_NotHTTP` if the URL is not available for a HEAD request, or
    `_NotAPIContent` if the content type is not a valid content type.
    >   httpshttpT)allow_redirectsN)urllibparseurlsplitrC   headr   rB   )r"   rD   r'   netlocpathqueryfragmentrespr(   r(   r)   _ensure_api_responsef   s    rQ   c                 C   sx   t t| jrt| |d tdt|  |j| dg dddd}t	| t
| tdt| |jd	d
 |S )aY  Access an Simple API response with GET, and return the response.

    This consists of three parts:

    1. If the URL looks suspiciously like an archive, send a HEAD first to
       check the Content-Type is HTML or Simple API, to avoid downloading a
       large file. Raise `_NotHTTP` if the content type cannot be determined, or
       `_NotAPIContent` if it is not HTML or a Simple API.
    2. Actually perform the request. Raise HTTP exceptions on network failures.
    3. Check the Content-Type header to make sure we got a Simple API response,
       and raise `_NotAPIContent` otherwise.
    rD   zGetting page %sz, )r<   z*application/vnd.pypi.simple.v1+html; q=0.1ztext/html; q=0.01z	max-age=0)AcceptzCache-Control)r=   zFetched page %s as %sr:   r;   )r   r   filenamerQ   loggerdebugr   r>   joinr   rB   r=   )r"   rD   rP   r(   r(   r)   _get_simple_responsex   s&    rX   )r=   r#   c                 C   s<   | r8d| v r8t j }| d |d< |d}|r8t|S dS )z=Determine if we have any encoding information in our headers.r:   zcontent-typecharsetN)emailmessageMessage	get_paramr7   )r=   mrY   r(   r(   r)   _get_encoding_from_headers   s    

r_   )partr#   c                 C   s   t jt j| S )zP
    Clean a "part" of a URL path (i.e. after splitting on "@" characters).
    )rH   rI   quoteunquoter`   r(   r(   r)   _clean_url_path_part   s    rd   c                 C   s   t jt j| S )z
    Clean the first part of a URL path that corresponds to a local
    filesystem path (i.e. the first part after splitting on "@" characters).
    )rH   r?   pathname2urlurl2pathnamerc   r(   r(   r)   _clean_file_url_path   s    
rg   z(@|%2F))rM   is_local_pathr#   c                 C   s^   |r
t }nt}t| }g }tt|dgD ]$\}}||| ||  q.d	|S )z*
    Clean the path portion of a URL.
     )
rg   rd   _reserved_chars_resplitr   	itertoolschainappendupperrW   )rM   rh   Z
clean_funcpartsZcleaned_partsZto_cleanreservedr(   r(   r)   _clean_url_path   s    
rr   c                 C   s6   t j| }|j }t|j|d}t j|j|dS )z
    Make sure a link is fully quoted.
    For example, if ' ' occurs in the URL, it will be replaced with "%20",
    and without double-quoting other characters.
    )rh   )rM   )rH   rI   urlparserL   rr   rM   
urlunparse_replace)r"   resultrh   rM   r(   r(   r)   _clean_link   s    rw   )element_attribspage_urlbase_urlr#   c                 C   sL   |  d}|sdS ttj||}|  d}|  d}t||||d}|S )zW
    Convert an anchor element's attributes in a simple repository page to a Link.
    hrefNzdata-requires-pythonzdata-yanked)
comes_fromrequires_pythonyanked_reason)r>   rw   rH   rI   urljoinr   )rx   ry   rz   r{   r"   Z	pyrequirer~   linkr(   r(   r)   _create_link_from_element   s    


r   c                   @   s:   e Zd ZdddddZeedddZed	d
dZdS )CacheablePageContentIndexContentNpager#   c                 C   s   |j s
J || _d S r.   )cache_link_parsingr   r1   r   r(   r(   r)   r0     s    
zCacheablePageContent.__init__)otherr#   c                 C   s   t |t| o| jj|jjkS r.   )
isinstancetyper   r"   )r1   r   r(   r(   r)   __eq__  s    zCacheablePageContent.__eq__r#   c                 C   s   t | jjS r.   )hashr   r"   r1   r(   r(   r)   __hash__"  s    zCacheablePageContent.__hash__)	r4   r5   r6   r0   objectboolr   intr   r(   r(   r(   r)   r     s   r   c                   @   s    e Zd Zdee dddZdS )
ParseLinksr   r   c                 C   s   d S r.   r(   r   r(   r(   r)   __call__'  s    zParseLinks.__call__N)r4   r5   r6   r   r   r   r(   r(   r(   r)   r   &  s   r   )fnr#   c                    sL   t jddttt d fddt  dtt d fdd	}|S )
z
    Given a function that parses an Iterable[Link] from an IndexContent, cache the
    function's result (keyed by CacheablePageContent), unless the IndexContent
    `page` has `page.cache_link_parsing == False`.
    N)maxsize)cacheable_pager#   c                    s   t  | jS r.   )listr   )r   )r   r(   r)   wrapper2  s    z*with_cached_index_content.<locals>.wrapperr   r   c                    s   | j rt| S t | S r.   )r   r   r   )r   r   r   r(   r)   wrapper_wrapper6  s    z2with_cached_index_content.<locals>.wrapper_wrapper)	functools	lru_cacher   r   r   wraps)r   r   r(   r   r)   with_cached_index_content+  s
    
r   r   r   c              
   c   s  | j  }|drt| j}|dg D ]r}|d}|du rDq,|d}|rbt|tsbd}n|sjd}t	t
tj| j|| j|d||di d	V  q,dS t| j}| jpd
}|| j| | j}|jp|}	|jD ]"}
t|
||	d}|du rq|V  qdS )z\
    Parse a Simple API's Index Content, and yield its anchor elements as Link objects.
    r<   filesr"   NZyankedri   zrequires-pythonhashes)r|   r}   r~   r   zutf-8)ry   rz   )r,   r$   r%   jsonloadscontentr>   r   r7   r   rw   rH   rI   r   r"   HTMLLinkParserencodingfeeddecoderz   anchorsr   )r   rA   datafileZfile_urlr~   parserr   r"   rz   anchorr   r(   r(   r)   parse_links?  sD    









r   c                   @   s<   e Zd ZdZd
eeee eeddddZeddd	Z	dS )r   z5Represents one response (or page), along with its URLTN)r   r,   r   r"   r   r#   c                 C   s"   || _ || _|| _|| _|| _dS )am  
        :param encoding: the encoding to decode the given content.
        :param url: the URL from which the HTML was downloaded.
        :param cache_link_parsing: whether links parsed from this page's url
                                   should be cached. PyPI index urls should
                                   have this set to False, for example.
        N)r   r,   r   r"   r   )r1   r   r,   r   r"   r   r(   r(   r)   r0   r  s
    zIndexContent.__init__r   c                 C   s
   t | jS r.   )r   r"   r   r(   r(   r)   __str__  s    zIndexContent.__str__)T)
r4   r5   r6   __doc__bytesr7   r   r   r0   r   r(   r(   r(   r)   r   o  s    c                       sn   e Zd ZdZedd fddZeeeeee f  ddddZ	eeeee f  ee d	d
dZ
  ZS )r   zf
    HTMLParser that keeps the first base HREF and a list of all anchor
    elements' attributes.
    Nr!   c                    s$   t  jdd || _d | _g | _d S )NT)Zconvert_charrefs)r/   r0   r"   rz   r   )r1   r"   r2   r(   r)   r0     s    zHTMLLinkParser.__init__)tagattrsr#   c                 C   sH   |dkr,| j d u r,| |}|d urD|| _ n|dkrD| jt| d S )Nbasea)rz   get_hrefr   rn   dict)r1   r   r   r{   r(   r(   r)   handle_starttag  s    
zHTMLLinkParser.handle_starttag)r   r#   c                 C   s"   |D ]\}}|dkr|  S qd S )Nr{   r(   )r1   r   namevaluer(   r(   r)   r     s    
zHTMLLinkParser.get_href)r4   r5   r6   r   r7   r0   r   r   r   r   r   r8   r(   r(   r2   r)   r     s   "r   ).N)r   reasonmethr#   c                 C   s   |d u rt j}|d| | d S )Nz%Could not fetch URL %s: %s - skipping)rU   rV   )r   r   r   r(   r(   r)   _handle_get_simple_fail  s    r   T)r9   r   r#   c                 C   s&   t | j}t| j| jd || j|dS )Nr:   )r   r"   r   )r_   r=   r   r   r"   )r9   r   r   r(   r(   r)   _make_index_content  s    
r   )r   rD   r#   c           
   
   C   s  |d u rt d| jddd }t|}|r@td||  d S tj|\}}}}}}|dkrt	j
tj|r|ds|d7 }tj|d}td	| zt||d
}W nN ty   td|  Y n> ty } z"td| |j|j W Y d }~nd }~0  ty: } zt| | W Y d }~nd }~0  tyh } zt| | W Y d }~nd }~0  ty } z,d}	|	t|7 }	t| |	tjd W Y d }~nld }~0  tjy } zt| d|  W Y d }~n6d }~0  tjy    t| d Y n0 t|| j dS d S )Nz?_get_html_page() missing 1 required keyword argument: 'session'#r   r   zICannot look at %s URL %s because it does not support lookup as web pages.r   /z
index.htmlz# file: URL is directory, getting %srR   z`Skipping page %s because it looks like an archive, and cannot be checked by a HTTP HEAD request.zSkipping page %s because the %s request got Content-Type: %s. The only supported Content-Types are application/vnd.pypi.simple.v1+json, application/vnd.pypi.simple.v1+html, and text/htmlz4There was a problem confirming the ssl certificate: )r   zconnection error: z	timed out)r   )!	TypeErrorr"   rk   r*   rU   warningrH   rI   rs   osrM   isdirr?   rf   endswithr   rV   rX   rC   r+   r-   r,   r   r   r   r   r7   infor   ConnectionErrorTimeoutr   r   )
r   rD   r"   Z
vcs_schemer'   _rM   rP   excr   r(   r(   r)   _get_index_content  s^    

$$r   c                   @   s.   e Zd ZU eee  ed< eee  ed< dS )CollectedSources
find_links
index_urlsN)r4   r5   r6   r   r   r   __annotations__r(   r(   r(   r)   r     s   
r   c                   @   sx   e Zd ZdZeeddddZedeee	d ddd	Z
eee d
ddZeee dddZeeedddZdS )LinkCollectorz
    Responsible for collecting Link objects from all configured locations,
    making network requests as needed.

    The class's main method is its collect_sources() method.
    N)rD   search_scoper#   c                 C   s   || _ || _d S r.   )r   rD   )r1   rD   r   r(   r(   r)   r0     s    zLinkCollector.__init__F)rD   optionssuppress_no_indexr#   c                 C   s`   |j g|j }|jr8|s8tdddd |D  g }|jp@g }tj||d}t	||d}|S )z
        :param session: The Session to use to make requests.
        :param suppress_no_index: Whether to ignore the --no-index option
            when constructing the SearchScope object.
        zIgnoring indexes: %s,c                 s   s   | ]}t |V  qd S r.   )r   ).0r"   r(   r(   r)   	<genexpr>(      z'LinkCollector.create.<locals>.<genexpr>r   r   )rD   r   )
	index_urlextra_index_urlsno_indexrU   rV   rW   r   r   creater   )clsrD   r   r   r   r   r   link_collectorr(   r(   r)   r     s"    

zLinkCollector.creater   c                 C   s   | j jS r.   )r   r   r   r(   r(   r)   r   9  s    zLinkCollector.find_links)locationr#   c                 C   s   t || jdS )z>
        Fetch an HTML page containing package links.
        rR   )r   rD   )r1   r   r(   r(   r)   fetch_response=  s    zLinkCollector.fetch_response)project_namecandidates_from_pager#   c                    s   t  fddj|D  }t  fddjD  }ttj	rdd t
||D }t| d| dg| }td| tt|t|d	S )
Nc                 3   s$   | ]}t | jjd d dV  qdS )Fr   Zpage_validatorZ
expand_dirr   Nr   rD   Zis_secure_originr   locr   r1   r(   r)   r   I  s   z0LinkCollector.collect_sources.<locals>.<genexpr>c                 3   s$   | ]}t | jjd d dV  qdS )Tr   Nr   r   r   r(   r)   r   S  s   c                 S   s*   g | ]"}|d ur|j d urd|j  qS )Nz* )r   )r   sr(   r(   r)   
<listcomp>_  s   z1LinkCollector.collect_sources.<locals>.<listcomp>z' location(s) to search for versions of :
r   )collectionsOrderedDictr   Zget_index_urls_locationsvaluesr   rU   isEnabledForloggingDEBUGrl   rm   r&   rV   rW   r   r   )r1   r   r   Zindex_url_sourcesZfind_links_sourceslinesr(   r   r)   collect_sourcesC  s*    



zLinkCollector.collect_sources)F)r4   r5   r6   r   r   r   r0   classmethodr   r   r   propertyr   r7   r   r   r   r   r   r   r   r   r(   r(   r(   r)   r     s(   	  r   )N)T)N)br   r   email.messagerZ   r   rl   r   r   r   reurllib.parserH   urllib.requestZxml.etree.ElementTreexmlZhtml.parserr   optparser   typingr   r   r   r   r   r	   r
   r   r   r   r   pip._vendorr   Zpip._vendor.requestsr   Zpip._vendor.requests.exceptionsr   r   pip._internal.exceptionsr   pip._internal.models.linkr   Z!pip._internal.models.search_scoper   pip._internal.network.sessionr   Zpip._internal.network.utilsr   pip._internal.utils.filetypesr   pip._internal.utils.miscr   r   pip._internal.vcsr   sourcesr   r   r   r    r   	getLoggerr4   rU   ZetreeZElementTreeZElementZHTMLElementr7   ZResponseHeadersr*   	Exceptionr+   rB   rC   rQ   rX   r_   rd   rg   compile
IGNORECASErj   r   rr   rw   r   r   r   r   r   r   r   r   r   r   r   r   r(   r(   r(   r)   <module>   s   4

?/ 

  D