o
    üÞhº!  ã                   @   sn  d Z ddlZddlZddlmZ ddlmZ ddlm	Z	 ddl
mZmZmZ zeZW n ey7   eefZY nw zddlmZ W n eyO   ddlmZ Y nw zddlmZ W n eyg   ddlmZ Y nw G d	d
„ d
eƒZzddlmZ W n	 ey�   Y nw G dd„ deƒZeƒ Zdd„ Zddd„Z		ddd„Z		ddd„Zddd„Z ddd„Z!dd„ Z"eƒ Z#dS )z?
An interface to html5lib that mimics the lxml.html interface.
é    N)Ú
HTMLParser)ÚTreeBuilder)Úetree)ÚElementÚXHTML_NAMESPACEÚ_contains_block_level_tag)Úurlopen)Úurlparsec                   @   ó   e Zd ZdZddd„ZdS )r   z*An html5lib HTML parser with lxml as tree.Fc                 K   ó   t j| f|tdœ|¤Ž d S ©N)ÚstrictÚtree)Ú_HTMLParserÚ__init__r   ©Úselfr   Úkwargs© r   úU/var/www/html/premium_crap/venv/lib/python3.10/site-packages/lxml/html/html5parser.pyr      ó   zHTMLParser.__init__N©F©Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r   r   r      ó    r   )ÚXHTMLParserc                   @   r
   )r   z+An html5lib XHTML Parser with lxml as tree.Fc                 K   r   r   )Ú_XHTMLParserr   r   r   r   r   r   r   *   r   zXHTMLParser.__init__Nr   r   r   r   r   r   r   '   r   r   c                 C   s(   |   |¡}|d ur|S |   dt|f ¡S )Nz{%s}%s)Úfindr   )r   ÚtagÚelemr   r   r   Ú	_find_tag0   s   
r#   c                 C   s^   t | tƒs	tdƒ‚|du rt}i }|du rt | tƒrd}|dur$||d< |j| fi |¤Ž ¡ S )zÍ
    Parse a whole document into a string.

    If `guess_charset` is true, or if the input is not Unicode but a
    byte string, the `chardet` library will perform charset guessing
    on the string.
    ústring requiredNTÚ
useChardet)Ú
isinstanceÚ_stringsÚ	TypeErrorÚhtml_parserÚbytesÚparseÚgetroot)ÚhtmlÚguess_charsetÚparserÚoptionsr   r   r   Údocument_fromstring7   s   
r1   Fc                 C   sš   t | tƒs	tdƒ‚|du rt}i }|du rt | tƒrd}|dur$||d< |j| dfi |¤Ž}|rKt |d tƒrK|rK|d  ¡ rHt d|d  ¡‚|d= |S )a`  Parses several HTML elements, returning a list of elements.

    The first item in the list may be a string.  If no_leading_text is true,
    then it will be an error if there is leading text, and it will always be
    a list of only elements.

    If `guess_charset` is true, the `chardet` library will perform charset
    guessing on the string.
    r$   NFr%   Údivr   zThere is leading text: %r)	r&   r'   r(   r)   r*   ÚparseFragmentÚstripr   ÚParserError)r-   Úno_leading_textr.   r/   r0   Úchildrenr   r   r   Úfragments_fromstringO   s$   
ÿr8   c                 C   sÌ   t | tƒs	tdƒ‚t|ƒ}t| ||| d�}|r;t |tƒsd}t|ƒ}|r9t |d tƒr4|d |_|d= | |¡ |S |sBt 	d¡‚t
|ƒdkrMt 	d¡‚|d }|jra|j ¡ rat 	d|j ¡‚d	|_|S )
aÂ  Parses a single HTML element; it is an error if there is more than
    one element, or if anything but whitespace precedes or follows the
    element.

    If 'create_parent' is true (or is a tag name) then a parent node
    will be created to encapsulate the HTML in a single element.  In
    this case, leading or trailing text is allowed.

    If `guess_charset` is true, the `chardet` library will perform charset
    guessing on the string.
    r$   )r.   r/   r6   r2   r   zNo elements foundé   zMultiple elements foundzElement followed by text: %rN)r&   r'   r(   Úboolr8   r   ÚtextÚextendr   r5   ÚlenÚtailr4   )r-   Úcreate_parentr.   r/   Úaccept_leading_textÚelementsÚnew_rootÚresultr   r   r   Úfragment_fromstringq   s4   
þ




rD   c                 C   sÞ   t | tƒs	tdƒ‚t| ||d�}| dd… }t |tƒr!| dd¡}| ¡  ¡ }| d¡s1| d¡r3|S t	|d	ƒ}t
|ƒr>|S t	|d
ƒ}t
|ƒdkra|jrQ|j ¡ sa|d jr]|d j ¡ sa|d S t|ƒrjd|_|S d|_|S )a   Parse the html, returning a single element/document.

    This tries to minimally parse the chunk of text, without knowing if it
    is a fragment or a document.

    'base_url' will set the document's base_url attribute (and the tree's
    docinfo.URL)

    If `guess_charset` is true, or if the input is not Unicode but a
    byte string, the `chardet` library will perform charset guessing
    on the string.
    r$   )r/   r.   Né2   ÚasciiÚreplacez<htmlz	<!doctypeÚheadÚbodyr9   éÿÿÿÿr   r2   Úspan)r&   r'   r(   r1   r*   ÚdecodeÚlstripÚlowerÚ
startswithr#   r=   r;   r4   r>   r   r!   )r-   r.   r/   ÚdocÚstartrH   rI   r   r   r   Ú
fromstring�   s4   
ÿ


ÿÿÿrR   c                 C   s~   |du rt }t| tƒs| }|du rd}nt| ƒr#t| ƒ}|du r"d}nt| dƒ}|du r.d}i }|r6||d< |j|fi |¤ŽS )a*  Parse a filename, URL, or file-like object into an HTML document
    tree.  Note: this returns a tree, not an element.  Use
    ``parse(...).getroot()`` to get the document root.

    If ``guess_charset`` is true, the ``useChardet`` option is passed into
    html5lib to enable character detection.  This option is on by default
    when parsing from URLs, off by default when parsing from file(-like)
    objects (which tend to return Unicode more often than not), and on by
    default when parsing from a file path (which is read in binary mode).
    NFTÚrbr%   )r)   r&   r'   Ú_looks_like_urlr   Úopenr+   )Úfilename_url_or_filer.   r/   Úfpr0   r   r   r   r+   Ó   s&   
€€
r+   c                 C   s<   t | ƒd }|s
dS tjdkr|tjv rt|ƒdkrdS dS )Nr   FÚwin32r9   T)r	   ÚsysÚplatformÚstringÚascii_lettersr=   )ÚstrÚschemer   r   r   rT   ÷   s   

rT   )NN)FNN)$r   rY   r[   Úhtml5libr   r   Ú html5lib.treebuilders.etree_lxmlr   Úlxmlr   Ú	lxml.htmlr   r   r   Ú
basestringr'   Ú	NameErrorr*   r]   Úurllib2r   ÚImportErrorÚurllib.requestr	   Úurllib.parser   r   Úxhtml_parserr#   r1   r8   rD   rR   r+   rT   r)   r   r   r   r   Ú<module>   sT    ÿÿÿÿ

ÿ"
ÿ
,
6$
