Ë
    [Rðiº!  ã                   ó|  — d Z ddlZddlZddlmZ ddlmZ ddlm	Z	 ddl
mZmZmZ 	 eZ	 ddlmZ 	 ddlmZ  G d	„ d
e«      Z	 ddlmZ  G d„ de«      Z e«       Zd„ Zdd„Z	 	 dd„Z	 	 dd„Zdd„Z dd„Z!d„ Z" e«       Z#y# e$ r eefZY Œcw xY w# e$ r	 ddlmZ Y Œmw xY w# e$ r	 ddlmZ Y Œww xY w# e$ r Y Œ^w xY w)z?
An interface to html5lib that mimics the lxml.html interface.
é    N)Ú
HTMLParser)ÚTreeBuilder)Úetree)ÚElementÚXHTML_NAMESPACEÚ_contains_block_level_tag)Úurlopen)Úurlparsec                   ó   — e Zd ZdZdd„Zy)r   z*An html5lib HTML parser with lxml as tree.c                 ó>   — t        j                  | f|t        dœ|¤Ž y ©N)ÚstrictÚtree)Ú_HTMLParserÚ__init__r   ©Úselfr   Úkwargss      úP/var/www/html/Extract/venv/lib/python3.12/site-packages/lxml/html/html5parser.pyr   zHTMLParser.__init__   s   € Ü×Ñ˜TÐM¨&´{ÑMÀfÓMó    N©F©Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   © r   r   r   r      s   „ Ù4ôNr   r   )ÚXHTMLParserc                   ó   — e Zd ZdZdd„Zy)r   z+An html5lib XHTML Parser with lxml as tree.c                 ó>   — t        j                  | f|t        dœ|¤Ž y r   )Ú_XHTMLParserr   r   r   s      r   r   zXHTMLParser.__init__*   s   € Ü×!Ñ! $ÐR¨v¼KÑRÈ6ÓRr   Nr   r   r   r   r   r   r   '   s   „ Ù9ô	Sr   r   c                 ób   — | j                  |«      }|�|S | j                  dt        ›d|›�«      S )Nú{ú})Úfindr   )r   ÚtagÚelems      r   Ú	_find_tagr(   0   s.   € Ø�9‰9�S‹>€DØÐØˆØ�9Š9¦±#Ð6Ó7Ð7r   c                 óÄ   — t        | t        «      st        d«      ‚|€t        }i }|€t        | t        «      rd}|�||d<    |j
                  | fi |¤Žj                  «       S )zÍ
    Parse a whole document into a string.

    If `guess_charset` is true, or if the input is not Unicode but a
    byte string, the `chardet` library will perform charset guessing
    on the string.
    ústring requiredTÚ
useChardet)Ú
isinstanceÚ_stringsÚ	TypeErrorÚhtml_parserÚbytesÚparseÚgetroot)ÚhtmlÚguess_charsetÚparserÚoptionss       r   Údocument_fromstringr7   7   sn   € ô �dœHÔ%ÜÐ)Ó*Ð*à€~Üˆà€GØÐ¤¨D´%Ô!8ð ˆØÐ Ø -ˆ�ÑØˆ6�<‰<˜Ñ( Ñ(×0Ñ0Ó2Ð2r   c                 ó>  — t        | t        «      st        d«      ‚|€t        }i }|€t        | t        «      rd}|�||d<    |j
                  | dfi |¤Ž}|rFt        |d   t        «      r3|r1|d   j                  «       rt        j                  d|d   z  «      ‚|d= |S )a`  Parses several HTML elements, returning a list of elements.

    The first item in the list may be a string.  If no_leading_text is true,
    then it will be an error if there is leading text, and it will always be
    a list of only elements.

    If `guess_charset` is true, the `chardet` library will perform charset
    guessing on the string.
    r*   Fr+   Údivr   zThere is leading text: %r)	r,   r-   r.   r/   r0   ÚparseFragmentÚstripr   ÚParserError)r3   Úno_leading_textr4   r5   r6   Úchildrens         r   Úfragments_fromstringr?   O   s¹   € ô �dœHÔ%ÜÐ)Ó*Ð*à€~Üˆà€GØÐ¤¨D´%Ô!8ð ˆØÐ Ø -ˆ�ÑØ#ˆv×#Ñ# D¨%Ñ;°7Ñ;€HÙ”J˜x¨™{¬HÔ5ÙØ˜‰{× Ñ Ô"Ü×'Ñ'Ð(CØ(0°©ñ)4ó 5ð 5à˜�Ø€Or   c                 ó6  — t        | t        «      st        d«      ‚t        |«      }t	        | ||| ¬«      }|rRt        |t        «      sd}t        |«      }|r1t        |d   t        «      r|d   |_        |d= |j                  |«       |S |st        j                  d«      ‚t        |«      dkD  rt        j                  d«      ‚|d   }|j                  r<|j                  j                  «       r"t        j                  d|j                  z  «      ‚d	|_        |S )
aÂ  Parses a single HTML element; it is an error if there is more than
    one element, or if anything but whitespace precedes or follows the
    element.

    If 'create_parent' is true (or is a tag name) then a parent node
    will be created to encapsulate the HTML in a single element.  In
    this case, leading or trailing text is allowed.

    If `guess_charset` is true, the `chardet` library will perform charset
    guessing on the string.
    r*   )r4   r5   r=   r9   r   zNo elements foundé   zMultiple elements foundzElement followed by text: %rN)r,   r-   r.   Úboolr?   r   ÚtextÚextendr   r<   ÚlenÚtailr;   )r3   Úcreate_parentr4   r5   Úaccept_leading_textÚelementsÚnew_rootÚresults           r   Úfragment_fromstringrL   q   s  € ô �dœHÔ%ÜÐ)Ó*Ð*ä˜}Ó-Ðä#Ø˜M°&Ø/Ð/ô1€Hñ Ü˜-¬Ô2Ø!ˆMÜ˜=Ó)ˆÙÜ˜( 1™+¤xÔ0Ø (¨¡�”Ø˜Q�KØ�O‰O˜HÔ%ØˆáÜ×ÑÐ 3Ó4Ð4Ü
ˆ8ƒ}�qÒÜ×ÑÐ 9Ó:Ð:Ø�a‰[€FØ‡{‚{�v—{‘{×(Ñ(Ô*Ü×ÑÐ >ÀÇÁÑ LÓMÐMØ€F„KØ€Mr   c                 ót  — t        | t        «      st        d«      ‚t        | ||¬«      }| dd }t        |t        «      r|j                  dd«      }|j                  «       j                  «       }|j                  d«      s|j                  d«      r|S t        |d	«      }t        |«      r|S t        |d
«      }t        |«      dk(  rW|j                  r|j                  j                  «       s1|d   j                  r|d   j                  j                  «       s|d   S t        |«      r	d|_        |S d|_        |S )a   Parse the html, returning a single element/document.

    This tries to minimally parse the chunk of text, without knowing if it
    is a fragment or a document.

    'base_url' will set the document's base_url attribute (and the tree's
    docinfo.URL)

    If `guess_charset` is true, or if the input is not Unicode but a
    byte string, the `chardet` library will perform charset guessing
    on the string.
    r*   )r5   r4   Né2   ÚasciiÚreplacez<htmlz	<!doctypeÚheadÚbodyrA   éÿÿÿÿr   r9   Úspan)r,   r-   r.   r7   r0   ÚdecodeÚlstripÚlowerÚ
startswithr(   rE   rC   r;   rF   r   r&   )r3   r4   r5   ÚdocÚstartrQ   rR   s          r   Ú
fromstringr[   �   s  € ô �dœHÔ%ÜÐ)Ó*Ð*Ü
˜d¨6Ø,9ô;€Cð ��"ˆI€EÜ�%œÔð —‘˜W iÓ0ˆà�L‰L‹N× Ñ Ó"€EØ×Ñ˜Ô  E×$4Ñ$4°[Ô$AØˆ
ä�S˜&Ó!€Dô ˆ4„yØˆ
ä�S˜&Ó!€Dô 	ˆD‹	�QŠ §	¢	°·±·±Ô1BØ�b‘—’ d¨2¡h§m¡m×&9Ñ&9Ô&;Ø�A‰wˆô
 ! Ô&ØˆŒð €Kð ˆŒØ€Kr   c                 óÎ   — |€t         }t        | t        «      s| }|€.d}n+t        | «      rt	        | «      }|€d}nt        | d«      }|€d}i }|r||d<    |j                  |fi |¤ŽS )a*  Parse a filename, URL, or file-like object into an HTML document
    tree.  Note: this returns a tree, not an element.  Use
    ``parse(...).getroot()`` to get the document root.

    If ``guess_charset`` is true, the ``useChardet`` option is passed into
    html5lib to enable character detection.  This option is on by default
    when parsing from URLs, off by default when parsing from file(-like)
    objects (which tend to return Unicode more often than not), and on by
    default when parsing from a file path (which is read in binary mode).
    FTÚrbr+   )r/   r,   r-   Ú_looks_like_urlr	   Úopenr1   )Úfilename_url_or_filer4   r5   Úfpr6   s        r   r1   r1   Ó   sŠ   € ð €~ÜˆÜÐ*¬HÔ5Ø!ˆØÐ à!‰MÜ	Ð-Ô	.ÜÐ)Ó*ˆØÐ à ‰MäÐ&¨Ó-ˆØÐ Ø ˆMà€Gñ Ø -ˆ�ÑØˆ6�<‰<˜Ñ&˜gÑ&Ð&r   c                 óŽ   — t        | «      d   }|syt        j                  dk(  r!|t        j                  v rt        |«      dk(  ryy)Nr   FÚwin32rA   T)r
   ÚsysÚplatformÚstringÚascii_lettersrE   )ÚstrÚschemes     r   r^   r^   ÷   sB   € Ü�c‹]˜1Ñ€FÙØÜ
�,‰,˜'Ò
!Ø”f×*Ñ*Ñ*Ü�F“˜qÒ ààr   )NN)FNN)$r   rd   rf   Úhtml5libr   r   Ú html5lib.treebuilders.etree_lxmlr   Úlxmlr   Ú	lxml.htmlr   r   r   Ú
basestringr-   Ú	NameErrorr0   rh   Úurllib2r	   ÚImportErrorÚurllib.requestr
   Úurllib.parser   r!   Úxhtml_parserr(   r7   r?   rL   r[   r1   r^   r/   r   r   r   ú<module>ru      sü   ðñó Û å .Ý 8Ý ß IÑ IðØ€Hð'Ýð&Ý!ô
N�ô Nð!Ý4ôS�lô Sñ “=€Lò8ó3ð0 05Ø48óðD -2Ø37ó)óX3ól!'òH
ñ ‹l�øðk ò Ø�sˆ|‚Hðûð ò 'ß&ð'ûð ò &ß%ð&ûð ò 	Ùð	úsE   ¨B «B ²B" ÁB3 Â	BÂBÂBÂBÂ"B0Â/B0Â3B;Â:B;