Ë
    þ)fK:  ã                   ó¨   — d Z dZdgZddlmZ ddlZddlZddlmZm	Z	m
Z
mZmZ ddlmZmZ ddlmZmZmZmZmZ d	Z G d
„ dee«      Z G d„ de«      Zy)zCUse the HTMLParser library to parse HTML files that aren't too bad.ÚMITÚHTMLParserTreeBuilderé    )Ú
HTMLParserN)ÚCDataÚCommentÚDeclarationÚDoctypeÚProcessingInstruction)ÚEntitySubstitutionÚUnicodeDammit)ÚDetectsXMLParsedAsHTMLÚParserRejectedMarkupÚHTMLÚHTMLTreeBuilderÚSTRICTzhtml.parserc                   ód   — e Zd ZdZdZdZd„ Zd„ Zd„ Zdd„Z	dd„Z
d	„ Zd
„ Zd„ Zd„ Zd„ Zd„ Zd„ Zy)ÚBeautifulSoupHTMLParserz¸A subclass of the Python standard library's HTMLParser class, which
    listens for HTMLParser events and translates them into calls
    to Beautiful Soup's tree construction API.
    ÚignoreÚreplacec                 ó¦   — |j                  d| j                  «      | _        t        j                  | g|¢­i |¤Ž g | _        | j                  «        y)a  Constructor.

        :param on_duplicate_attribute: A strategy for what to do if a
            tag includes the same attribute more than once. Accepted
            values are: REPLACE (replace earlier values with later
            ones, the default), IGNORE (keep the earliest value
            encountered), or a callable. A callable must take three
            arguments: the dictionary of attributes already processed,
            the name of the duplicate attribute, and the most recent value
            encountered.           
        Úon_duplicate_attributeN)ÚpopÚREPLACEr   r   Ú__init__Úalready_closed_empty_elementÚ_initialize_xml_detector)ÚselfÚargsÚkwargss      úT/var/www/html/flask-app/venv/lib/python3.12/site-packages/bs4/builder/_htmlparser.pyr   z BeautifulSoupHTMLParser.__init__.   sN   € ð '-§j¡jØ$ d§l¡ló'
ˆÔ#ô 	×Ñ˜DÐ2 4Ò2¨6Ò2ð -/ˆÔ)à×%Ñ%Õ'ó    c                 ó   — t        |«      ‚)N)r   )r   Úmessages     r    ÚerrorzBeautifulSoupHTMLParser.errorJ   s   € ô # 7Ó+Ð+r!   c                 óN   — | j                  ||d¬«      }| j                  |«       y)zÏHandle an incoming empty-element tag.

        This is only called when the markup looks like <tag/>.

        :param name: Name of the tag.
        :param attrs: Dictionary of the tag's attributes.
        F)Úhandle_empty_elementN)Úhandle_starttagÚhandle_endtag)r   ÚnameÚattrsÚtags       r    Úhandle_startendtagz*BeautifulSoupHTMLParser.handle_startendtagZ   s)   € ð ×"Ñ" 4¨ÀUÐ"ÓKˆØ×Ñ˜4Õ r!   c                 óÔ  — i }|D ]Q  \  }}|€d}||v r=| j                   }|| j                  k(  rn&|d| j                  fv r|||<   n ||||«       n|||<   d}ŒS | j                  «       \  }	}
| j                  j                  |dd||	|
¬«      }|r<|j                  r0|r.| j                  |d¬«       | j                  j                  |«       | j                  €| j                  |«       yy)a3  Handle an opening tag, e.g. '<tag>'

        :param name: Name of the tag.
        :param attrs: Dictionary of the tag's attributes.
        :param handle_empty_element: True if this tag is known to be
            an empty-element tag (i.e. there is not expected to be any
            closing tag).
        NÚ z"")Ú
sourcelineÚ	sourceposF)Úcheck_already_closed)r   ÚIGNOREr   ÚgetposÚsoupr'   Úis_empty_elementr(   r   ÚappendÚ	_root_tagÚ_root_tag_encountered)r   r)   r*   r&   Ú	attr_dictÚkeyÚvalueÚon_dupeÚ	attrvaluer/   r0   r+   s               r    r'   z'BeautifulSoupHTMLParser.handle_starttagi   s  € ð ˆ	Øò 	‰JˆC�ð ˆ}Ø�Ø�iÑð ×5Ñ5�Ø˜dŸk™kÒ)ØØ  t§|¡|Ð 4Ñ4Ø%*�I˜c’Ná˜I s¨EÕ2à!&�	˜#‘Ø‰Ið%	ð( !%§¡£Ñˆ
�IØ�i‰i×'Ñ'Ø�$˜˜i°JØð (ó 
ˆñ �3×'Ò'Ñ,@ð ×Ñ˜t¸%ÐÔ@ð ×-Ñ-×4Ñ4°TÔ:à�>‰>Ð!Ø×&Ñ& tÕ,ð "r!   c                 ó’   — |r*|| j                   v r| j                   j                  |«       y| j                  j                  |«       y)zõHandle a closing tag, e.g. '</tag>'
        
        :param name: A tag name.
        :param check_already_closed: True if this tag is expected to
           be the closing portion of an empty-element tag,
           e.g. '<tag></tag>'.
        N)r   Úremover4   r(   )r   r)   r1   s      r    r(   z%BeautifulSoupHTMLParser.handle_endtag    s<   € ñ   D¨D×,MÑ,MÑ$Mð
 ×-Ñ-×4Ñ4°TÕ:à�I‰I×#Ñ# DÕ)r!   c                 ó:   — | j                   j                  |«       y)z4Handle some textual data that shows up between tags.N)r4   Úhandle_data©r   Údatas     r    rA   z#BeautifulSoupHTMLParser.handle_data²   s   € à�	‰	×Ñ˜dÕ#r!   c                 ó  — |j                  d«      rt        |j                  d«      d«      }n8|j                  d«      rt        |j                  d«      d«      }nt        |«      }d}|dk  r<| j                  j                  dfD ]!  }|sŒ	 t        |g«      j                  |«      }Œ# |s	 t        |«      }|xs d}| j                  |«       y# t        $ r
}Y d}~ŒXd}~ww xY w# t        t        f$ r
}Y d}~ŒBd}~ww xY w)z×Handle a numeric character reference by converting it to the
        corresponding Unicode character and treating it as textual
        data.

        :param name: Character number, possibly in hexadecimal.
        Úxé   ÚXNé   zwindows-1252u   ï¿½)Ú
startswithÚintÚlstripr4   Úoriginal_encodingÚ	bytearrayÚdecodeÚUnicodeDecodeErrorÚchrÚ
ValueErrorÚOverflowErrorrA   )r   r)   Ú	real_namerC   ÚencodingÚes         r    Úhandle_charrefz&BeautifulSoupHTMLParser.handle_charref¶   sù   € ð �?‰?˜3ÔÜ˜DŸK™K¨Ó,¨bÓ1‰IØ�_‰_˜SÔ!Ü˜DŸK™K¨Ó,¨bÓ1‰Iä˜D›	ˆIàˆØ�sŠ?ð "ŸY™Y×8Ñ8¸.ÐIò �ÙØðÜ$ i [Ó1×8Ñ8¸ÓB‘Dð	ñ ðÜ˜9“~�ð Ò2Ð2ˆØ×Ñ˜Õøô *ò Üûðûô
 ¤Ð.ò Üûðús$   ÂCÂ,C% Ã	C"ÃC"Ã%C>Ã9C>c                 óx   — t         j                  j                  |«      }|�|}nd|z  }| j                  |«       y)zÈHandle a named entity reference by converting it to the
        corresponding Unicode character(s) and treating it as textual
        data.

        :param name: Name of the entity reference.
        Nz&%s)r   ÚHTML_ENTITY_TO_CHARACTERÚgetrA   )r   r)   Ú	characterrC   s       r    Úhandle_entityrefz(BeautifulSoupHTMLParser.handle_entityrefÞ   s>   € ô '×?Ñ?×CÑCÀDÓIˆ	ØÐ Ø‰Dð ˜4‘<ˆDØ×Ñ˜Õr!   c                 ó¬   — | j                   j                  «        | j                   j                  |«       | j                   j                  t        «       y)zOHandle an HTML comment.

        :param data: The text of the comment.
        N)r4   ÚendDatarA   r   rB   s     r    Úhandle_commentz&BeautifulSoupHTMLParser.handle_commentñ   s8   € ð
 	�	‰	×ÑÔØ�	‰	×Ñ˜dÔ#Ø�	‰	×Ñœ'Õ"r!   c                 óÈ   — | j                   j                  «        |t        d«      d }| j                   j                  |«       | j                   j                  t        «       y)zYHandle a DOCTYPE declaration.

        :param data: The text of the declaration.
        zDOCTYPE N)r4   r]   ÚlenrA   r	   rB   s     r    Úhandle_declz#BeautifulSoupHTMLParser.handle_declú   sI   € ð
 	�	‰	×ÑÔØ”C˜
“OÐ$Ð%ˆØ�	‰	×Ñ˜dÔ#Ø�	‰	×Ñœ'Õ"r!   c                 ó  — |j                  «       j                  d«      rt        }|t        d«      d }nt        }| j
                  j                  «        | j
                  j                  |«       | j
                  j                  |«       y)z{Handle a declaration of unknown type -- probably a CDATA block.

        :param data: The text of the declaration.
        zCDATA[N)ÚupperrI   r   r`   r   r4   r]   rA   )r   rC   Úclss      r    Úunknown_declz$BeautifulSoupHTMLParser.unknown_decl  sf   € ð
 �:‰:‹<×"Ñ" 8Ô,ÜˆCØœ˜H›˜Ð'‰DäˆCØ�	‰	×ÑÔØ�	‰	×Ñ˜dÔ#Ø�	‰	×Ñ˜#Õr!   c                 óÎ   — | j                   j                  «        | j                   j                  |«       | j                  |«       | j                   j                  t        «       y)z\Handle a processing instruction.

        :param data: The text of the instruction.
        N)r4   r]   rA   Ú_document_might_be_xmlr
   rB   s     r    Ú	handle_piz!BeautifulSoupHTMLParser.handle_pi  sG   € ð
 	�	‰	×ÑÔØ�	‰	×Ñ˜dÔ#Ø×#Ñ# DÔ)Ø�	‰	×ÑÔ/Õ0r!   N)T)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r2   r   r   r$   r,   r'   r(   rA   rV   r[   r^   ra   re   rh   © r!   r    r   r   $   sQ   „ ñð €FØ€Gò(ò8,ò !ó5-ón*ò$$ò&òPò&#ò#òó1r!   r   c                   óP   ‡ — e Zd ZdZdZdZeZeee	gZ
dZdˆ fd„	Z	 	 dd„Zd„ Zˆ xZS )	r   zpA Beautiful soup `TreeBuilder` that uses the `HTMLParser` parser,
    found in the Python standard library.
    FTc                 óÚ   •— t        «       }dD ]  }||v sŒ|j                  |«      }|||<   Œ t        t        | �  di |¤Ž |xs g }|xs i }|j                  |«       d|d<   ||f| _        y)a„  Constructor.

        :param parser_args: Positional arguments to pass into 
            the BeautifulSoupHTMLParser constructor, once it's
            invoked.
        :param parser_kwargs: Keyword arguments to pass into 
            the BeautifulSoupHTMLParser constructor, once it's
            invoked.
        :param kwargs: Keyword arguments for the superclass constructor.
        )r   FÚconvert_charrefsNrm   )Údictr   Úsuperr   r   ÚupdateÚparser_args)r   rt   Úparser_kwargsr   Úextra_parser_kwargsÚargr;   Ú	__class__s          €r    r   zHTMLParserTreeBuilder.__init__*  s‹   ø€ ô #›fÐØ.ò 	1ˆCØ�fŠ}ØŸ
™
 3›�Ø+0Ð# CÒ(ð	1ô 	Ô# TÑ3Ñ=°fÒ=Ø!Ò' RˆØ%Ò+¨ˆØ×ÑÐ0Ô1Ø,1ˆÐ(Ñ)Ø'¨Ð7ˆÕr!   c              #   óÒ   K  — t        |t        «      r	|dddf–— y|g}|g}||g}t        |||d|¬«      }|j                  |j                  |j
                  |j                  f–— y­w)aÜ  Run any preliminary steps necessary to make incoming markup
        acceptable to the parser.

        :param markup: Some markup -- probably a bytestring.
        :param user_specified_encoding: The user asked to try this encoding.
        :param document_declared_encoding: The markup itself claims to be
            in this encoding.
        :param exclude_encodings: The user asked _not_ to try any of
            these encodings.

        :yield: A series of 4-tuples:
         (markup, encoding, declared encoding,
          has undergone character replacement)

         Each 4-tuple represents a strategy for converting the
         document to Unicode and parsing it. Each strategy will be tried 
         in turn.
        NFT)Úknown_definite_encodingsÚuser_encodingsÚis_htmlÚexclude_encodings)Ú
isinstanceÚstrr   ÚmarkuprL   Údeclared_html_encodingÚcontains_replacement_characters)	r   r€   Úuser_specified_encodingÚdocument_declared_encodingr}   rz   r{   Útry_encodingsÚdammits	            r    Úprepare_markupz$HTMLParserTreeBuilder.prepare_markupC  sŠ   è ø€ ô* �fœcÔ"à˜4  uÐ-Ò-Øð %<Ð#<Ð ð 5Ð5ˆà0Ð2LÐMˆÜØØ%=Ø)ØØ/ô
ˆð �}‰}˜f×6Ñ6Ø×,Ñ,Ø×5Ñ5ð7ó 	7ùs   ‚A%A'c                 óä   — | j                   \  }}t        |i |¤Ž}| j                  |_        	 |j                  |«       |j	                  «        g |_        y# t
        $ r}t        |«      ‚d}~ww xY w)z{Run some incoming markup through some parsing process,
        populating the `BeautifulSoup` object in self.soup.
        N)rt   r   r4   ÚfeedÚcloseÚAssertionErrorr   r   )r   r€   r   r   ÚparserrU   s         r    r‰   zHTMLParserTreeBuilder.feedt  sp   € ð ×'Ñ'‰ˆˆfÜ(¨$Ð9°&Ñ9ˆØ—i‘iˆŒð	*Ø�K‰K˜ÔØ�L‰LŒNð /1ˆÕ+øô ò 	*ô ' qÓ)Ð)ûð		*ús   ­!A Á	A/ÁA*Á*A/)NN)NNN)ri   rj   rk   rl   Úis_xmlÚ	picklableÚ
HTMLPARSERÚNAMEr   r   ÚfeaturesÚTRACKS_LINE_NUMBERSr   r‡   r‰   Ú__classcell__)rx   s   @r    r   r     sF   ø„ ñð €FØ€IØ€DØ�d˜FÐ#€Hð Ðõ8ð2 >BØJNó/7öb1r!   )rl   Ú__license__Ú__all__Úhtml.parserr   ÚsysÚwarningsÚbs4.elementr   r   r   r	   r
   Ú
bs4.dammitr   r   Úbs4.builderr   r   r   r   r   r�   r   r   rm   r!   r    ú<module>rœ      sf   ðá Ið €ð ð€õ #ã 
Û ÷õ ÷ 9÷õ ð €
ôv1˜jÐ*@ô v1ôrf1˜Oõ f1r!   